// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import "fmt" // This file implements VEX (AVX/AVX2) instruction encoding. EVEX (AVX-512) // support is a later increment. // // Every encoding choice here is validated two ways in the tests: by // round-trip decoding through golang.org/x/arch's x86 decoder, and by // byte-for-byte comparison against the output of the real Go assembler. // vexForm selects how an instruction's operands map onto the VEX.vvvv, // ModRM.reg and ModRM.rm fields. type vexForm int const ( // vexNDS3 is the three-operand form `OP src2, src1, dst` (Plan 9 order): // ModRM.reg = dst (op2), VEX.vvvv = src1 (op1), ModRM.rm = src2 (op0). vexNDS3 vexForm = iota // vexRM is the two-operand form `OP src, dst` with no vvvv source: // ModRM.reg = dst (op1), ModRM.rm = src (op0), VEX.vvvv unused. vexRM // vexShiftImm is the immediate-shift form `OP $imm, src, dst`: ModRM.reg = // /digit, ModRM.rm = src (op1), VEX.vvvv = dst (op2), imm8 = op0. vexShiftImm // vexImmRM is the immediate form `OP $imm, src, dst` with no vvvv source: // ModRM.reg = dst (op2), ModRM.rm = src (op1), imm8 = op0. VPSHUFD and // VPERMQ use this shape. vexImmRM // vexNDS3Imm is the three-operand plus immediate form `OP $imm, src2, // src1, dst`: ModRM.reg = dst, VEX.vvvv = src1, ModRM.rm = src2, imm8. // VSHUFPD, VPERM2I128 and VINSERTI128 use this shape. vexNDS3Imm // vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg = // ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM // source lives in the reg field, the destination in r/m — the PEXTR-style // layout. VEXTRACTI128 and VEXTRACTF128 use this shape. vexExtract // vexRMRev is the reversed two-operand form `OP src, dst` with the source // in ModRM.reg and the destination in r/m — the layout of the EVEX // narrowing stores (VPMOVDW, VPMOVQD). vexRMRev // vexRMSrcLen is the two-operand conversion form `OP src, dst` whose // vector length follows the source: the packed-double → dword // conversions (VCVTPD2DQ/VCVTTPD2DQ and their X/Y spellings) narrow into // an XMM destination, so the L bit rides with the wider source. The // mnemonic's spelling fixes the length (X = 128, Y = 256), which also // covers a memory source. ModRM.reg = dst, ModRM.rm = src, no vvvv. vexRMSrcLen // vexZero is the no-operand form (VZEROUPPER). vexZero ) // vexSpec describes one VEX instruction's encoding parameters. type vexSpec struct { mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A opcode byte w int // VEX.W (0 for WIG) pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2 opdigit int // ModRM.reg /digit, or -1 when reg is a register form vexForm } // vexTable maps an upper-case mnemonic to its VEX encoding. It is extended // incrementally; every entry is covered by a byte-for-byte ground-truth test // against the Go assembler. var vexTable = map[string]vexSpec{ // VEX.128/256.66.0F.WIG — integer arithmetic / logic / compare. "VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3}, "VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3}, "VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3}, "VPSUBQ": {1, 0xFB, 0, 1, -1, vexNDS3}, "VPXOR": {1, 0xEF, 0, 1, -1, vexNDS3}, "VPOR": {1, 0xEB, 0, 1, -1, vexNDS3}, "VPAND": {1, 0xDB, 0, 1, -1, vexNDS3}, "VPANDN": {1, 0xDF, 0, 1, -1, vexNDS3}, "VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3}, "VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3}, "VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3}, "VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3}, "VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3}, // VEX.256.66.0F38.W0 — dword permute (three-operand NDS form). "VPERMD": {2, 0x36, 0, 1, -1, vexNDS3}, // VEX.128/256.66.0F38.WIG. "VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3}, "VPMULDQ": {2, 0x28, 0, 1, -1, vexNDS3}, "VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3}, "VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3}, // VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic. "VADDPD": {1, 0x58, 0, 1, -1, vexNDS3}, "VMULPD": {1, 0x59, 0, 1, -1, vexNDS3}, "VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3}, "VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3}, "VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3}, "VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3}, // VEX.128/256.0F.WIG — packed single-precision arithmetic. "VADDPS": {1, 0x58, 0, 0, -1, vexNDS3}, "VMULPS": {1, 0x59, 0, 0, -1, vexNDS3}, "VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3}, "VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3}, "VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3}, "VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3}, "VXORPD": {1, 0x57, 0, 1, -1, vexNDS3}, "VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3}, "VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3}, // VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed // opcodes with an F2 pp). "VADDSD": {1, 0x58, 0, 3, -1, vexNDS3}, "VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3}, "VMULSD": {1, 0x59, 0, 3, -1, vexNDS3}, "VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3}, "VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3}, "VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3}, // VEX.128.F3.0F.WIG — scalar single-precision arithmetic (the packed // opcodes with an F3 pp). "VADDSS": {1, 0x58, 0, 2, -1, vexNDS3}, "VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3}, "VMULSS": {1, 0x59, 0, 2, -1, vexNDS3}, "VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3}, "VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3}, "VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3}, // VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form). "VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3}, // VEX.128/256.66.0F38.WIG — sign/zero extend and broadcast (reg=dst, rm=src, // no vvvv). "VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM}, "VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM}, "VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM}, "VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM}, "VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM}, "VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM}, "VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM}, // VEX.128/256.F3.0F.WIG — signed dword to packed double conversion // (reg=dst, rm=src, no vvvv; the length follows the destination). "VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM}, // VEX.128/256.0F.WIG — signed dword to packed single conversion // (reg=dst, rm=src, no vvvv, no mandatory prefix). "VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM}, // VEX.128/256.0F.WIG — packed single to packed double conversion // (reg=dst, rm=src; the destination is the wide operand and sets the // length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but // the Go assembler emits the instruction with pp = 00, and gasm follows // the Go assembler's bytes — its machine code is the oracle, not the // manual. "VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM}, // VEX.128.F2.0F.WIG — duplicate the low double of each 128-bit lane // (reg=dst, rm=src, no vvvv; the length follows the destination). "VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM}, // VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src). "VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM}, "VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD) // VEX.128/256.66.0F.WIG — immediate shifts (opdigit selects the shift). "VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm}, "VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm}, "VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm}, "VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm}, "VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm}, // VEX.128/256.66.0F.WIG — immediate shuffle (reg=dst, rm=src, imm8). "VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM}, // VEX.256.66.0F3A.W1 — qword permute (reg=dst, rm=src, imm8). "VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM}, // VEX.128/256.66.0F.WIG — two-source shuffle (reg=dst, vvvv=src1, rm=src2, // imm8). "VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm}, // VEX.256.66.0F3A.W0 — permute / insert (same shape; VINSERTI128's rm is // the XMM or memory source). "VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm}, "VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm}, // VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8). "VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract}, "VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract}, // VEX.128.0F.W0 — no operands. "VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero}, // VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src). "KTESTW": {1, 0x99, 0, 0, -1, vexRM}, // VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst, // rm=scalar memory; SD is 256-bit only). "VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM}, "VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM}, // VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src). "VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM}, "VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM}, // VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift). "VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm}, "VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm}, "VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm}, // VEX.F2.0F — packed double to packed dword conversions, truncating and // non-truncating. The destination is always XMM; the X/Y spellings fix // the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen. "VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen}, "VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen}, "VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen}, "VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen}, } // vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of // the packed-double → dword conversions) to its fixed vector length: // X = 128 (L = 0), Y = 256 (L = 1). The spelling fixes the length even for // a memory source, matching the Go assembler's ytab. var vexSrcLen = map[string]int{ "VCVTPD2DQX": 0, "VCVTPD2DQY": 1, "VCVTTPD2DQX": 0, "VCVTTPD2DQY": 1, } // vexVarShift maps the shift mnemonics to their variable-count opcode — the // form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8), // an ordinary NDS encoding rather than the /digit immediate form above. var vexVarShift = map[string]byte{ "VPSLLD": 0xF2, "VPSLLQ": 0xF3, "VPSRAD": 0xE2, "VPSRLD": 0xD2, "VPSRLQ": 0xD3, } // vexMoveSpec describes a VEX move, which takes different opcodes (and // sometimes a different VEX.W) per operand direction. The Go assembler // encodes a vector→vector move with the store-form opcode (reg = source, // rm = destination), so regReg defaults to store when zero. type vexMoveSpec struct { mapSel int pp int load byte // r/m → vector: reg=dst, rm=src store byte // vector → r/m: reg=src, rm=dst loadW int storeW int regReg byte // vector → vector opcode; 0 uses store regW int vecOK bool // the non-fixed operand may be a vector register gprOK bool // the non-fixed operand may be a general-purpose register xmmOnly bool // YMM registers are rejected } // vexMoveTable maps an upper-case move mnemonic to its encoding. var vexMoveTable = map[string]vexMoveSpec{ // VEX.128/256.F3.0F.WIG — unaligned integer move. "VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false}, // VEX.128/256.66.0F.WIG — unaligned packed double move. "VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false}, // VEX.128.66.0F.W0 — 32-bit GPR/memory ↔ XMM. "VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true}, // VMOVQ — 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm). "VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true}, // VEX.128.F2.0F.WIG — scalar double move, memory operands only (the // register form takes three operands and is not supported yet). "VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true}, // VEX.128.F3.0F.WIG — scalar single move, memory operands only. "VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true}, // VEX.128/256 — aligned packed moves. "VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false}, "VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false}, } // isVex reports whether the mnemonic is a VEX-encoded instruction we handle. func isVex(mnemUpper string) bool { if _, ok := vexTable[mnemUpper]; ok { return true } _, ok := vexMoveTable[mnemUpper] return ok } // encodeVex encodes a VEX instruction with operands in Plan 9 order. func (e *enc) encodeVex(mnemUpper string, ops []Operand) error { // Vector register indices 16–31 exist only in EVEX encodings; fail // loudly rather than silently truncating the index. for _, op := range ops { if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 { return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx) } } if ms, ok := vexMoveTable[mnemUpper]; ok { return e.encodeVexMove(mnemUpper, ms, ops) } // The shifts come in two shapes under one mnemonic: an immediate count // ($imm, src, dst) and a variable count in an XMM register or memory // (count, src, dst), the latter an ordinary NDS form. if op, ok := vexVarShift[mnemUpper]; ok && len(ops) == 3 { if _, isImm := ops[0].(Imm); !isImm { if !vecOrMem(ops[0]) { return fmt.Errorf("%s: shift count must be an immediate, a vector register or memory", mnemUpper) } return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops) } } spec := vexTable[mnemUpper] switch spec.form { case vexNDS3: return e.encodeVexNDS3(spec, ops) case vexRM: return e.encodeVexRM(spec, ops) case vexShiftImm: return e.encodeVexShiftImm(spec, ops) case vexImmRM: return e.encodeVexImmRM(spec, ops) case vexNDS3Imm: return e.encodeVexNDS3Imm(spec, ops) case vexExtract: return e.encodeVexExtract(spec, ops) case vexRMSrcLen: return e.encodeVexRMSrcLen(mnemUpper, spec, ops) case vexZero: return e.encodeVexZero(mnemUpper, spec, ops) } return fmt.Errorf("unhandled VEX form for %s", mnemUpper) } // encodeVexNDS3 encodes the three-operand NDS form: OP src2, src1, dst. func (e *enc) encodeVexNDS3(spec vexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops)) } src2, src1, dst := ops[0], ops[1], ops[2] dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("VEX destination must be a vector register") } vvvvReg, ok := src1.(Reg) if !ok || !vvvvReg.isVec() { return fmt.Errorf("VEX vvvv operand must be a vector register") } regField := dstReg.idx & 7 rBit := 0 if dstReg.idx >= 8 { rBit = 1 } vvvvBar := 15 - (vvvvReg.idx & 15) return e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2) } // encodeVexRM encodes the two-operand form: OP src, dst (no vvvv source). // ModRM.reg = dst, ModRM.rm = src; the vector length comes from whichever // operand is a vector register (the destination for extends/broadcasts, the // source for the move-mask instructions whose destination is a GPR). func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("VEX two-operand instruction expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok { return fmt.Errorf("VEX destination must be a register") } regField := dstReg.idx & 7 rBit := 0 if dstReg.idx >= 8 { rBit = 1 } // Vector length: from the destination if it is a vector, otherwise from the // source (move-mask instructions have a GPR destination and a vector source). l := 0 if dstReg.isVec() { l = dstReg.vecLenBit() } else if srcReg, ok := src.(Reg); ok && srcReg.isVec() { l = srcReg.vecLenBit() } // An unused vvvv field must be stored as all ones (v̄vvv = 1111); the // hardware raises #UD on any other value. return e.emitVexFields(spec, l, regField, rBit, 15, src) } // encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with // the destination always XMM and the VEX.L bit following the source — fixed // by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when // the source is memory. func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("conversion expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("VEX destination must be a vector register") } ll, ok := vexSrcLen[mnem] if !ok { return fmt.Errorf("no fixed vector length for %s", mnem) } regField := dstReg.idx & 7 rBit := 0 if dstReg.idx >= 8 { rBit = 1 } // An unused vvvv field must be stored as all ones (v̄vvv = 1111). return e.emitVexFields(spec, ll, regField, rBit, 15, src) } // encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst. // The destination is carried in VEX.vvvv, the source in ModRM.rm, and the // shift kind in the ModRM.reg /digit. func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("VEX shift expects 3 operands ($imm, src, dst), got %d", len(ops)) } imm, src, dst := ops[0], ops[1], ops[2] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("shift count must be an immediate") } srcReg, ok := src.(Reg) if !ok || !srcReg.isVec() { return fmt.Errorf("shift source must be a vector register") } dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("shift destination must be a vector register") } vvvvBar := 15 - (dstReg.idx & 15) if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, srcReg); err != nil { return err } immByte, err := imm8(int64(immVal)) if err != nil { return err } e.out = append(e.out, immByte) return nil } // imm8 range-checks an immediate for an 8-bit field. Shuffle controls are // unsigned bit masks, but the negative spelling ($-1 = all bits set) is // accepted, so the accepted span is -128..255. func imm8(v int64) (byte, error) { if v < -128 || v > 255 { return 0, fmt.Errorf("immediate $%d does not fit in 8 bits", v) } return byte(v), nil } // encodeVexImmRM encodes an immediate form with no vvvv source: OP $imm, src, // dst (VPSHUFD, VPERMQ). ModRM.reg = dst, ModRM.rm = src, imm8 appended. func (e *enc) encodeVexImmRM(spec vexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops)) } imm, src, dst := ops[0], ops[1], ops[2] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("shuffle control must be an immediate") } dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("shuffle destination must be a vector register") } // The vector length follows the source when it is a vector register, // otherwise the destination (a memory source carries no length). l := dstReg.vecLenBit() if srcReg, ok := src.(Reg); ok && srcReg.isVec() { l = srcReg.vecLenBit() } regField := dstReg.idx & 7 rBit := 0 if dstReg.idx >= 8 { rBit = 1 } if err := e.emitVexFields(spec, l, regField, rBit, 15, src); err != nil { return err } immByte, err := imm8(int64(immVal)) if err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeVexNDS3Imm encodes the three-operand plus immediate form: OP $imm, // src2, src1, dst (VSHUFPD, VPERM2I128, VINSERTI128). ModRM.reg = dst, // VEX.vvvv = src1, ModRM.rm = src2, imm8 appended. func (e *enc) encodeVexNDS3Imm(spec vexSpec, ops []Operand) error { if len(ops) != 4 { return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops)) } imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("shuffle control must be an immediate") } dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("destination must be a vector register") } vvvvReg, ok := src1.(Reg) if !ok || !vvvvReg.isVec() { return fmt.Errorf("second source must be a vector register") } regField := dstReg.idx & 7 rBit := 0 if dstReg.idx >= 8 { rBit = 1 } vvvvBar := 15 - (vvvvReg.idx & 15) if err := e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2); err != nil { return err } immByte, err := imm8(int64(immVal)) if err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeVexExtract encodes a lane extract: OP $imm, ysrc, xdst // (VEXTRACTI128, VEXTRACTF128). The YMM source occupies ModRM.reg and the // XMM (or memory) destination ModRM.rm; imm8 selects the lane. func (e *enc) encodeVexExtract(spec vexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("extract expects 3 operands ($imm, ysrc, xdst), got %d", len(ops)) } imm, src, dst := ops[0], ops[1], ops[2] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("extract lane must be an immediate") } srcReg, ok := src.(Reg) if !ok || !srcReg.isVec() { return fmt.Errorf("extract source must be a vector register") } regField := srcReg.idx & 7 rBit := 0 if srcReg.idx >= 8 { rBit = 1 } if err := e.emitVexFields(spec, srcReg.vecLenBit(), regField, rBit, 15, dst); err != nil { return err } immByte, err := imm8(int64(immVal)) if err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeVexZero encodes a no-operand instruction (VZEROUPPER). func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error { if len(ops) != 0 { return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops)) } // 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 0. e.out = append(e.out, 0xC5, byte(1<<7|15<<3|spec.pp), spec.opcode) return nil } // encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ, // VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector // move uses the store-form layout (reg = source, rm = destination), matching // the Go assembler. func (e *enc) encodeVexMove(mnem string, ms vexMoveSpec, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("VEX move expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] srcReg, srcIsVec := vecReg(src) dstReg, dstIsVec := vecReg(dst) var reg Reg var rm Operand op, w := ms.store, ms.storeW switch { case srcIsVec && dstIsVec: if !ms.vecOK { return fmt.Errorf("%s does not take two vector registers", mnem) } if ms.xmmOnly && (srcReg.size == 32 || dstReg.size == 32) { return fmt.Errorf("%s operates on XMM registers only", mnem) } if ms.regReg != 0 { op, w = ms.regReg, ms.regW } reg, rm = srcReg, dst // store form: reg = source, rm = destination. case srcIsVec: // vector → memory, or → GPR (VMOVD/VMOVQ only). if !validMoveOther(ms, dst) { return fmt.Errorf("%s: invalid destination operand", mnem) } reg, rm = srcReg, dst case dstIsVec: // memory → vector, or GPR → vector (VMOVD/VMOVQ only). if !validMoveOther(ms, src) { return fmt.Errorf("%s: invalid source operand", mnem) } op, w = ms.load, ms.loadW reg, rm = dstReg, src default: return fmt.Errorf("%s needs a vector register operand", mnem) } if ms.xmmOnly && reg.size == 32 { return fmt.Errorf("%s operates on XMM registers only", mnem) } regField := reg.idx & 7 rBit := 0 if reg.idx >= 8 { rBit = 1 } spec := vexSpec{mapSel: ms.mapSel, opcode: op, w: w, pp: ms.pp, opdigit: -1} return e.emitVexFields(spec, reg.vecLenBit(), regField, rBit, 15, rm) } // vecReg extracts a vector register from an operand. func vecReg(op Operand) (Reg, bool) { r, ok := op.(Reg) return r, ok && r.isVec() } // vecOrMem reports whether op is a vector register or a memory reference. func vecOrMem(op Operand) bool { switch op.(type) { case Mem, sbMem: return true } r, ok := op.(Reg) return ok && r.isVec() } // validMoveOther reports whether the non-vector operand of a move is // acceptable: memory always is, a GPR only for VMOVD/VMOVQ. func validMoveOther(ms vexMoveSpec, op Operand) bool { switch o := op.(type) { case Mem, sbMem: return true case Reg: return ms.gprOK && !o.isVec() } return false } // emitVexFields emits the VEX prefix, opcode, ModR/M, SIB and displacement for // the given precomputed fields. It is shared by every register/rm VEX form; // immediate bytes are appended by the caller. func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Operand) error { if l > 1 { return fmt.Errorf("ZMM operand requires an EVEX instruction") } var modrm, sib int var disp []byte var xBit, bBit int var sb *sbRef switch r := rm.(type) { case Reg: modrm = 0xC0 | regField<<3 | (r.idx & 7) sib = -1 if r.idx >= 8 { bBit = 1 } case Mem: var err error modrm, sib, disp, xBit, bBit, err = memComponents(regField, r) if err != nil { return err } case sbMem: // RIP-relative static-symbol reference; disp32 patched at link time. modrm = regField<<3 | 0x05 sib = -1 disp = le32(0) sb = &sbRef{name: r.name, addend: r.addend} default: return fmt.Errorf("invalid VEX r/m operand") } if spec.mapSel == 1 && xBit == 0 && bBit == 0 && spec.w == 0 { e.out = append(e.out, 0xC5, byte((1-rBit)<<7|vvvvBar<<3|l<<2|spec.pp)) } else { e.out = append(e.out, 0xC4, byte((1-rBit)<<7|(1-xBit)<<6|(1-bBit)<<5|spec.mapSel), byte(spec.w<<7|vvvvBar<<3|l<<2|spec.pp)) } e.out = append(e.out, spec.opcode, byte(modrm)) if sib >= 0 { e.out = append(e.out, byte(sib)) } if sb != nil { e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend}) } e.out = append(e.out, disp...) return nil }