// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import "fmt" // This file implements VEX (AVX/AVX2) instruction encoding. EVEX (AVX-512) // support is a later increment. // // Every encoding choice here is validated two ways in the tests: by // round-trip decoding through golang.org/x/arch's x86 decoder, and by // byte-for-byte comparison against the output of the real Go assembler. // vexForm selects how an instruction's operands map onto the VEX.vvvv, // ModRM.reg and ModRM.rm fields. type vexForm int const ( // vexNDS3 is the three-operand form `OP src2, src1, dst` (Plan 9 order): // ModRM.reg = dst (op2), VEX.vvvv = src1 (op1), ModRM.rm = src2 (op0). vexNDS3 vexForm = iota // vexRM is the two-operand form `OP src, dst` with no vvvv source: // ModRM.reg = dst (op1), ModRM.rm = src (op0), VEX.vvvv unused. vexRM // vexShiftImm is the immediate-shift form `OP $imm, src, dst`: ModRM.reg = // /digit, ModRM.rm = src (op1), VEX.vvvv = dst (op2), imm8 = op0. vexShiftImm // vexImmRM is the immediate form `OP $imm, src, dst` with no vvvv source: // ModRM.reg = dst (op2), ModRM.rm = src (op1), imm8 = op0. VPSHUFD and // VPERMQ use this shape. vexImmRM // vexNDS3Imm is the three-operand plus immediate form `OP $imm, src2, // src1, dst`: ModRM.reg = dst, VEX.vvvv = src1, ModRM.rm = src2, imm8. // VSHUFPD, VPERM2I128 and VINSERTI128 use this shape. vexNDS3Imm // vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg = // ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM // source lives in the reg field, the destination in r/m, the PEXTR-style // layout. VEXTRACTI128 and VEXTRACTF128 use this shape. vexExtract // vexRMRev is the reversed two-operand form `OP src, dst` with the source // in ModRM.reg and the destination in r/m, the layout of the EVEX // narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ. vexRMRev // vexRMSrcLen is the two-operand conversion form `OP src, dst` whose // vector length follows the source: the packed-double → dword // conversions (VCVTPD2DQ/VCVTTPD2DQ and their X/Y spellings) narrow into // an XMM destination, so the L bit rides with the wider source. The // mnemonic's spelling fixes the length (X = 128, Y = 256), which also // covers a memory source. ModRM.reg = dst, ModRM.rm = src, no vvvv. vexRMSrcLen // vexZero is the no-operand form (VZEROUPPER). vexZero // vexZeroAll is the no-operand form that zeroes the full upper state // (VZEROALL, the L = 1 twin of VZEROUPPER). vexZeroAll // vexNDS3GPR is the three-operand NDS form over general-purpose // registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0. vexNDS3GPR // vexImmRMGPR is the immediate form over general-purpose registers // (RORX): reg = dst, rm = src, imm8 = op0, L = 0. vexImmRMGPR // vexRMOpGPR is the two-operand /digit form over general-purpose // registers (BLSI, BLSMSK, BLSR): ModRM.reg = /digit, ModRM.rm = src // (op0), VEX.vvvv = dst (op1), L = 0. vexRMOpGPR // vexCountGPR is the three-operand count form over general-purpose // registers (SHLX, SHRX, SARX, BEXTR, BZHI): the first operand rides // VEX.vvvv and the second is r/m, the opposite pairing of the ANDN // family, with reg = dst (op2), L = 0. vexCountGPR // vexExtractGPR is the lane-extract-to-GPR form `OP $imm, xsrc, GPR/mem // dst`: ModRM.reg = xsrc (op1), ModRM.rm = destination (op2), imm8 = // op0, the VPEXTRB/W/D/Q layout. EVEX only; the destination never // carries a vector length, so the register the L'L field follows is the // XMM source. vexExtractGPR ) // vexSpec describes one VEX instruction's encoding parameters. type vexSpec struct { mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A opcode byte w int // VEX.W (0 for WIG) pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2 opdigit int // ModRM.reg /digit, or -1 when reg is a register form vexForm } // vexTable maps an upper-case mnemonic to its VEX encoding. It is extended // incrementally; every entry is covered by a byte-for-byte ground-truth test // against the Go assembler. var vexTable = map[string]vexSpec{ // VEX.128/256.66.0F.WIG, integer arithmetic / logic / compare. "VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3}, "VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3}, "VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3}, "VPSUBQ": {1, 0xFB, 0, 1, -1, vexNDS3}, "VPXOR": {1, 0xEF, 0, 1, -1, vexNDS3}, "VPOR": {1, 0xEB, 0, 1, -1, vexNDS3}, "VPAND": {1, 0xDB, 0, 1, -1, vexNDS3}, "VPANDN": {1, 0xDF, 0, 1, -1, vexNDS3}, "VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3}, "VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3}, "VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3}, "VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3}, "VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3}, // VEX.256.66.0F38.W0, dword permute (three-operand NDS form). "VPERMD": {2, 0x36, 0, 1, -1, vexNDS3}, // VEX.128/256.66.0F38.WIG. "VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3}, "VPMULDQ": {2, 0x28, 0, 1, -1, vexNDS3}, "VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3}, "VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3}, // VEX.128/256.66.0F.WIG, packed double-precision arithmetic / logic. "VADDPD": {1, 0x58, 0, 1, -1, vexNDS3}, "VMULPD": {1, 0x59, 0, 1, -1, vexNDS3}, "VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3}, "VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3}, "VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3}, "VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3}, // VEX.128/256.0F.WIG, packed single-precision arithmetic. "VADDPS": {1, 0x58, 0, 0, -1, vexNDS3}, "VMULPS": {1, 0x59, 0, 0, -1, vexNDS3}, "VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3}, "VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3}, "VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3}, "VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3}, "VXORPD": {1, 0x57, 0, 1, -1, vexNDS3}, "VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3}, "VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3}, // VEX.128.F2.0F.WIG, scalar double-precision arithmetic (the packed // opcodes with an F2 pp). "VADDSD": {1, 0x58, 0, 3, -1, vexNDS3}, "VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3}, "VMULSD": {1, 0x59, 0, 3, -1, vexNDS3}, "VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3}, "VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3}, "VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3}, // VEX.128.F3.0F.WIG, scalar single-precision arithmetic (the packed // opcodes with an F3 pp). "VADDSS": {1, 0x58, 0, 2, -1, vexNDS3}, "VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3}, "VMULSS": {1, 0x59, 0, 2, -1, vexNDS3}, "VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3}, "VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3}, "VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3}, // VEX.128/256.66.0F38.W1, fused multiply-add (NDS form). "VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3}, // Scalar fused multiply-add (NDS form). The Go assembler carries the // same 66 prefix as the packed forms on every FMA row, and W1 on the // double-precision spellings, so SD shares PD's prefix/W pair and the // scalar width rides on the W bit. "VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3}, "VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3}, // VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src, // no vvvv). "VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM}, "VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM}, "VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM}, "VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM}, "VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM}, "VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM}, "VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM}, "VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM}, "VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM}, "VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM}, "VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM}, "VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM}, "VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM}, "VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM}, "VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM}, // VEX.128/256.F3.0F.WIG, signed dword to packed double conversion // (reg=dst, rm=src, no vvvv; the length follows the destination). "VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM}, // VEX.128/256.0F.WIG, signed dword to packed single conversion // (reg=dst, rm=src, no vvvv, no mandatory prefix). "VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM}, // VEX.128/256.0F.WIG, packed single to packed double conversion // (reg=dst, rm=src; the destination is the wide operand and sets the // length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but // the Go assembler emits the instruction with pp = 00, and gasm follows // the Go assembler's bytes, its machine code is the oracle, not the // manual. "VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM}, // VEX.128.F2.0F.WIG, duplicate the low double of each 128-bit lane // (reg=dst, rm=src, no vvvv; the length follows the destination). "VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM}, // VEX.128/256.66.0F.WIG, move mask to a GPR (reg=gpr dst, rm=vec src). "VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM}, "VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD) // VEX.128/256.66.0F.WIG, immediate shifts (opdigit selects the shift). "VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm}, "VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm}, "VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm}, "VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm}, "VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm}, // VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8). "VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM}, // VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8). "VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM}, // VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2, // imm8). "VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm}, // VEX.256.66.0F3A.W0, permute / insert (same shape; VINSERTI128's rm is // the XMM or memory source). "VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm}, "VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm}, // VEX.256.66.0F3A.W0, lane extract (reg=YMM src, rm=XMM/memory dst, imm8). "VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract}, "VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract}, // VEX.128/256.66.0F3A.W0, half-precision convert back ($imm, src, dst: // reg=src, rm=XMM/memory dst, imm8, the extract layout). "VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract}, // VEX.128.0F.W0, no operands. "VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero}, // VEX.256.0F.W0, zero all vector registers (the L = 1 twin). "VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll}, // VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare. "VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm}, "VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm}, "VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3}, // VEX.128/256.0F.WIG, packed single XOR (NDS form). "VXORPS": {1, 0x57, 0, 0, -1, vexNDS3}, // VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control. "VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm}, "VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm}, // VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling // falls through to the EVEX table. "VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm}, // VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst). "VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm}, // VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8). "VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm}, // BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR). "ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR}, "ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR}, "MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR}, "MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR}, // VEX.NDS.LZ.0F38, the BMI2 three-operand bit ops: BEXTR and BZHI // share the F7/F5 opcodes across W, the variable shifts carry their // direction in the prefix (SHLX 66, SHRX F2, SARX F3) and PDEP/PEXT // in F2/F3. "BEXTRL": {2, 0xF7, 0, 0, -1, vexCountGPR}, "BEXTRQ": {2, 0xF7, 1, 0, -1, vexCountGPR}, "BZHIL": {2, 0xF5, 0, 0, -1, vexCountGPR}, "BZHIQ": {2, 0xF5, 1, 0, -1, vexCountGPR}, "SARXL": {2, 0xF7, 0, 2, -1, vexCountGPR}, "SARXQ": {2, 0xF7, 1, 2, -1, vexCountGPR}, "SHLXL": {2, 0xF7, 0, 1, -1, vexCountGPR}, "SHLXQ": {2, 0xF7, 1, 1, -1, vexCountGPR}, "SHRXL": {2, 0xF7, 0, 3, -1, vexCountGPR}, "SHRXQ": {2, 0xF7, 1, 3, -1, vexCountGPR}, "PDEPL": {2, 0xF5, 0, 3, -1, vexNDS3GPR}, "PDEPQ": {2, 0xF5, 1, 3, -1, vexNDS3GPR}, "PEXTL": {2, 0xF5, 0, 2, -1, vexNDS3GPR}, "PEXTQ": {2, 0xF5, 1, 2, -1, vexNDS3GPR}, // VEX.LZ.0F38.W, the BMI1 unary bit ops (src, dst: ModRM.reg = /digit, // rm = src, vvvv = dst). "BLSIL": {2, 0xF3, 0, 0, 3, vexRMOpGPR}, "BLSIQ": {2, 0xF3, 1, 0, 3, vexRMOpGPR}, "BLSMSKL": {2, 0xF3, 0, 0, 2, vexRMOpGPR}, "BLSMSKQ": {2, 0xF3, 1, 0, 2, vexRMOpGPR}, "BLSRL": {2, 0xF3, 0, 0, 1, vexRMOpGPR}, "BLSRQ": {2, 0xF3, 1, 0, 1, vexRMOpGPR}, "RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR}, "RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR}, // VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src). "KTESTW": {1, 0x99, 0, 0, -1, vexRM}, // VEX.66.0F38.W0, broadcast a single/double to all lanes (reg=dst, // rm=scalar memory; SD is 256-bit only). "VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM}, "VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM}, // VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a // YMM (the encoder rejects an XMM destination, as go tool asm does). "VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM}, // VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg, // memory destination in rm). "VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev}, // VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv). "VPTEST": {2, 0x17, 0, 1, -1, vexRM}, // VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width // source). "VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM}, // VEX.F3.0F.WIG, replicate even/odd singles (reg=dst, rm=src). "VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM}, "VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM}, // VEX.66.0F.WIG, packed double to packed single conversion, the X/Y // spellings: the destination is always XMM and the spelling fixes the // source length (X = 128, Y = 256). "VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen}, "VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen}, // VEX scalar conversions between vector and general-purpose registers. // Vector to GPR (two operands: vec/mem source, GPR destination, vvvv // unused; the length follows the source). "VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM}, "VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM}, "VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM}, "VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM}, "VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM}, "VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM}, "VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM}, "VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM}, // GPR to vector (three operands: GPR/mem source in r/m, the preserved // vector source in vvvv, vector destination in reg). "VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3}, "VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3}, "VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3}, "VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3}, // VEX.128/256.66.0F.WIG, word shifts (opdigit selects the shift). "VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm}, "VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm}, "VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm}, // VEX.F2.0F, packed double to packed dword conversions, truncating and // non-truncating. The destination is always XMM; the X/Y spellings fix // the source length (XMM/YMM), and VEX.L follows it, see vexSrcLen. "VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen}, "VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen}, "VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen}, "VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen}, // --- the VEX forms the avx512enc corpus exercises alongside the EVEX // spellings, read off the toolchain opcode tables --- "VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3}, "VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3}, "VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3}, "VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3}, "VANDNPD": {1, 0x55, 0, 1, -1, vexNDS3}, "VANDPD": {1, 0x54, 0, 1, -1, vexNDS3}, "VCOMISD": {1, 0x2F, 0, 1, -1, vexRM}, "VCVTSD2SS": {1, 0x5A, 0, 3, -1, vexNDS3}, "VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3}, "VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3}, "VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3}, "VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3}, "VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3}, "VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3}, "VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3}, "VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3}, "VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3}, "VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3}, "VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3}, "VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3}, "VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3}, "VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3}, "VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3}, "VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3}, "VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3}, "VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3}, "VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3}, "VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3}, "VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3}, "VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3}, "VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3}, "VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3}, "VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3}, "VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3}, "VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3}, "VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3}, "VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3}, "VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3}, "VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3}, "VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3}, "VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3}, "VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3}, "VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3}, "VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3}, "VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3}, "VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3}, "VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3}, "VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3}, "VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3}, "VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3}, "VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3}, "VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3}, "VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3}, "VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3}, "VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3}, "VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3}, "VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3}, "VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3}, "VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3}, "VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3}, "VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3}, "VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3}, "VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3}, "VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3}, "VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3}, "VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3}, "VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm}, "VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3}, "VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM}, "VMOVNTPD": {1, 0x2B, 0, 1, -1, vexRMRev}, "VORPD": {1, 0x56, 0, 1, -1, vexNDS3}, "VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3}, "VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3}, "VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3}, "VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3}, "VPCMPEQQ": {2, 0x29, 0, 1, -1, vexNDS3}, "VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3}, "VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3}, "VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3}, "VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3}, "VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3}, "VPEXTRB": {3, 0x14, 0, 1, -1, vexExtract}, "VPEXTRD": {3, 0x16, 0, 1, -1, vexExtract}, "VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtract}, "VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm}, "VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm}, "VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3}, "VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3}, "VPMULUDQ": {1, 0xF4, 0, 1, -1, vexNDS3}, "VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3}, "VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3}, "VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3}, "VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3}, "VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3}, "VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3}, "VPUNPCKHQDQ": {1, 0x6D, 0, 1, -1, vexNDS3}, "VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3}, "VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3}, "VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3}, "VSQRTPD": {1, 0x51, 0, 1, -1, vexRM}, "VSQRTSD": {1, 0x51, 0, 3, -1, vexNDS3}, "VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3}, "VUCOMISD": {1, 0x2E, 0, 1, -1, vexRM}, // VEX.0F.WIG, the plain-prefix single/double arithmetic and unpack // spellings (no 66 prefix; WIG, so W = 0). "VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3}, "VANDPS": {1, 0x54, 0, 0, -1, vexNDS3}, "VORPS": {1, 0x56, 0, 0, -1, vexNDS3}, "VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3}, "VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3}, "VSQRTPS": {1, 0x51, 0, 0, -1, vexRM}, "VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev}, // VEX.128.66.0F, the scalar and packed compare forms. "VCOMISS": {1, 0x2F, 0, 1, -1, vexRM}, "VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM}, // VEX.128.0F.F3/F2.W0, the high/low word shuffles ($imm, src, dst). "VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM}, "VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM}, } // vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of // the packed-double → dword conversions) to its fixed vector length: // X = 128 (L = 0), Y = 256 (L = 1). The spelling fixes the length even for // a memory source, matching the Go assembler's ytab. var vexSrcLen = map[string]int{ "VCVTPD2DQX": 0, "VCVTPD2DQY": 1, "VCVTTPD2DQX": 0, "VCVTTPD2DQY": 1, "VCVTPD2PSX": 0, "VCVTPD2PSY": 1, } // vexVarShift maps the shift mnemonics to their variable-count opcode, the // form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8), // an ordinary NDS encoding rather than the /digit immediate form above. var vexVarShift = map[string]byte{ "VPSLLD": 0xF2, "VPSLLQ": 0xF3, "VPSRAD": 0xE2, "VPSRLD": 0xD2, "VPSRLQ": 0xD3, } // vexMoveSpec describes a VEX move, which takes different opcodes (and // sometimes a different VEX.W) per operand direction. The Go assembler // encodes a vector→vector move with the store-form opcode (reg = source, // rm = destination), so regReg defaults to store when zero. type vexMoveSpec struct { mapSel int pp int load byte // r/m → vector: reg=dst, rm=src store byte // vector → r/m: reg=src, rm=dst loadW int storeW int regReg byte // vector → vector opcode; 0 uses store regW int vecOK bool // the non-fixed operand may be a vector register gprOK bool // the non-fixed operand may be a general-purpose register xmmOnly bool // YMM registers are rejected } // vexMoveTable maps an upper-case move mnemonic to its encoding. var vexMoveTable = map[string]vexMoveSpec{ // VEX.128/256.F3.0F.WIG, unaligned integer move. "VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false}, // VEX.128/256.66.0F.WIG, aligned integer move. "VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false}, // VEX.128/256.66.0F.WIG, unaligned packed double move. "VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false}, // VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM. "VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true}, // VMOVQ, 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm). "VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true}, // VEX.128.F2.0F.WIG, scalar double move, memory operands only (the // register form takes three operands and is not supported yet). "VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true}, // VEX.128.F3.0F.WIG, scalar single move, memory operands only. "VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true}, // VEX.128/256, aligned packed moves. "VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false}, "VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false}, } // isVex reports whether the mnemonic is a VEX-encoded instruction we handle. func isVex(mnemUpper string) bool { if _, ok := vexTable[mnemUpper]; ok { return true } _, ok := vexMoveTable[mnemUpper] return ok } // encodeVex encodes a VEX instruction with operands in Plan 9 order. func (e *enc) encodeVex(mnemUpper string, ops []Operand) error { // Vector register indices 16-31 exist only in EVEX encodings; fail // loudly rather than silently truncating the index. for _, op := range ops { if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 { return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx) } } // VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination // only; an XMM destination is rejected exactly as go tool asm does. if mnemUpper == "VBROADCASTI128" { dstReg, ok := ops[len(ops)-1].(Reg) if len(ops) != 2 || !ok || dstReg.size != 32 { return fmt.Errorf("VBROADCASTI128 requires a YMM destination") } } if ms, ok := vexMoveTable[mnemUpper]; ok { return e.encodeVexMove(mnemUpper, ms, ops) } // The shifts come in two shapes under one mnemonic: an immediate count // ($imm, src, dst) and a variable count in an XMM register or memory // (count, src, dst), the latter an ordinary NDS form. if op, ok := vexVarShift[mnemUpper]; ok && len(ops) == 3 { if _, isImm := ops[0].(Imm); !isImm { if !vecOrMem(ops[0]) { return fmt.Errorf("%s: shift count must be an immediate, a vector register or memory", mnemUpper) } return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops) } } spec := vexTable[mnemUpper] switch spec.form { case vexNDS3: return e.encodeVexNDS3(spec, ops) case vexRM: return e.encodeVexRM(spec, ops) case vexShiftImm: return e.encodeVexShiftImm(spec, ops) case vexImmRM: return e.encodeVexImmRM(spec, ops) case vexNDS3Imm: return e.encodeVexNDS3Imm(spec, ops) case vexExtract: return e.encodeVexExtract(spec, ops) case vexRMSrcLen: return e.encodeVexRMSrcLen(mnemUpper, spec, ops) case vexZero: return e.encodeVexZero(mnemUpper, spec, ops) case vexZeroAll: return e.encodeVexZeroAll(mnemUpper, spec, ops) case vexNDS3GPR: return e.encodeVexNDS3GPR(spec, ops) case vexImmRMGPR: return e.encodeVexImmRMGPR(spec, ops) case vexRMOpGPR: return e.encodeVexRMOpGPR(spec, ops) case vexCountGPR: return e.encodeVexCountGPR(spec, ops) case vexRMRev: return e.encodeVexRMRev(spec, ops) } return fmt.Errorf("unhandled VEX form for %s", mnemUpper) } // encodeVexNDS3 encodes the three-operand NDS form: OP src2, src1, dst. func (e *enc) encodeVexNDS3(spec vexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops)) } src2, src1, dst := ops[0], ops[1], ops[2] dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("VEX destination must be a vector register") } vvvvReg, ok := src1.(Reg) if !ok || !vvvvReg.isVec() { return fmt.Errorf("VEX vvvv operand must be a vector register") } regField := dstReg.idx & 7 rBit := 0 if dstReg.idx >= 8 { rBit = 1 } vvvvBar := 15 - (vvvvReg.idx & 15) return e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2) } // encodeVexRM encodes the two-operand form: OP src, dst (no vvvv source). // ModRM.reg = dst, ModRM.rm = src; the vector length comes from whichever // operand is a vector register (the destination for extends/broadcasts, the // source for the move-mask instructions whose destination is a GPR). func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("VEX two-operand instruction expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok { return fmt.Errorf("VEX destination must be a register") } regField := dstReg.idx & 7 rBit := 0 if dstReg.idx >= 8 { rBit = 1 } // Vector length: from the destination if it is a vector, otherwise from the // source (move-mask instructions have a GPR destination and a vector source). l := 0 if dstReg.isVec() { l = dstReg.vecLenBit() } else if srcReg, ok := src.(Reg); ok && srcReg.isVec() { l = srcReg.vecLenBit() } // An unused vvvv field must be stored as all ones (v̄vvv = 1111); the // hardware raises #UD on any other value. return e.emitVexFields(spec, l, regField, rBit, 15, src) } // encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with // the destination always XMM and the VEX.L bit following the source, fixed // by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when // the source is memory. func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("conversion expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("VEX destination must be a vector register") } ll, ok := vexSrcLen[mnem] if !ok { return fmt.Errorf("no fixed vector length for %s", mnem) } regField := dstReg.idx & 7 rBit := 0 if dstReg.idx >= 8 { rBit = 1 } // An unused vvvv field must be stored as all ones (v̄vvv = 1111). return e.emitVexFields(spec, ll, regField, rBit, 15, src) } // encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst. // The destination is carried in VEX.vvvv, the source in ModRM.rm, and the // shift kind in the ModRM.reg /digit. func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("VEX shift expects 3 operands ($imm, src, dst), got %d", len(ops)) } imm, src, dst := ops[0], ops[1], ops[2] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("shift count must be an immediate") } // The count source is a vector register or memory; the VEX length // follows the destination register either way. if !vecOrMem(src) { return fmt.Errorf("shift source must be a vector register or memory") } dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("shift destination must be a vector register") } vvvvBar := 15 - (dstReg.idx & 15) if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, src); err != nil { return err } immByte, err := imm8(int64(immVal)) if err != nil { return err } e.out = append(e.out, immByte) return nil } // imm8 range-checks an immediate for an 8-bit field. Shuffle controls are // unsigned bit masks, but the negative spelling ($-1 = all bits set) is // accepted, so the accepted span is -128..255. func imm8(v int64) (byte, error) { if v < -128 || v > 255 { return 0, fmt.Errorf("immediate $%d does not fit in 8 bits", v) } return byte(v), nil } // encodeVexImmRM encodes an immediate form with no vvvv source: OP $imm, src, // dst (VPSHUFD, VPERMQ). ModRM.reg = dst, ModRM.rm = src, imm8 appended. func (e *enc) encodeVexImmRM(spec vexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops)) } imm, src, dst := ops[0], ops[1], ops[2] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("shuffle control must be an immediate") } dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("shuffle destination must be a vector register") } // The vector length follows the source when it is a vector register, // otherwise the destination (a memory source carries no length). l := dstReg.vecLenBit() if srcReg, ok := src.(Reg); ok && srcReg.isVec() { l = srcReg.vecLenBit() } regField := dstReg.idx & 7 rBit := 0 if dstReg.idx >= 8 { rBit = 1 } if err := e.emitVexFields(spec, l, regField, rBit, 15, src); err != nil { return err } immByte, err := imm8(int64(immVal)) if err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeVexNDS3Imm encodes the three-operand plus immediate form: OP $imm, // src2, src1, dst (VSHUFPD, VPERM2I128, VINSERTI128). ModRM.reg = dst, // VEX.vvvv = src1, ModRM.rm = src2, imm8 appended. func (e *enc) encodeVexNDS3Imm(spec vexSpec, ops []Operand) error { if len(ops) != 4 { return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops)) } imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("shuffle control must be an immediate") } dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("destination must be a vector register") } vvvvReg, ok := src1.(Reg) if !ok || !vvvvReg.isVec() { return fmt.Errorf("second source must be a vector register") } regField := dstReg.idx & 7 rBit := 0 if dstReg.idx >= 8 { rBit = 1 } vvvvBar := 15 - (vvvvReg.idx & 15) if err := e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2); err != nil { return err } immByte, err := imm8(int64(immVal)) if err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeVexExtract encodes a lane extract: OP $imm, ysrc, xdst // (VEXTRACTI128, VEXTRACTF128). The YMM source occupies ModRM.reg and the // XMM (or memory) destination ModRM.rm; imm8 selects the lane. func (e *enc) encodeVexExtract(spec vexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("extract expects 3 operands ($imm, ysrc, xdst), got %d", len(ops)) } imm, src, dst := ops[0], ops[1], ops[2] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("extract lane must be an immediate") } srcReg, ok := src.(Reg) if !ok || !srcReg.isVec() { return fmt.Errorf("extract source must be a vector register") } regField := srcReg.idx & 7 rBit := 0 if srcReg.idx >= 8 { rBit = 1 } if err := e.emitVexFields(spec, srcReg.vecLenBit(), regField, rBit, 15, dst); err != nil { return err } immByte, err := imm8(int64(immVal)) if err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeVexZero encodes a no-operand instruction (VZEROUPPER). func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error { if len(ops) != 0 { return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops)) } // 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 0. e.out = append(e.out, 0xC5, byte(1<<7|15<<3|spec.pp), spec.opcode) return nil } // encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1 // twin of VZEROUPPER. func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error { if len(ops) != 0 { return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops)) } // 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1. e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode) return nil } // encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose // registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1, // rm = src2 and L = 0. func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops)) } src2, src1, dst := ops[0], ops[1], ops[2] dstReg, ok := dst.(Reg) if !ok || dstReg.isVec() { return fmt.Errorf("VEX destination must be a general-purpose register") } vvvvReg, ok := src1.(Reg) if !ok || vvvvReg.isVec() { return fmt.Errorf("VEX vvvv operand must be a general-purpose register") } rBit := 0 if dstReg.idx >= 8 { rBit = 1 } return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2) } // encodeVexImmRMGPR encodes the immediate form over general-purpose // registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0. func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops)) } imm, src, dst := ops[0], ops[1], ops[2] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("shift control must be an immediate") } dstReg, ok := dst.(Reg) if !ok || dstReg.isVec() { return fmt.Errorf("VEX destination must be a general-purpose register") } immByte, err := imm8(int64(immVal)) if err != nil { return err } if err := e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src); err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeVexRMOpGPR encodes the two-operand /digit form over general-purpose // registers (BLSI, BLSMSK, BLSR): OP src, dst with ModRM.reg = /digit, // ModRM.rm = src and VEX.vvvv = dst. func (e *enc) encodeVexRMOpGPR(spec vexSpec, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("instruction expects 2 operands (src, dst), got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok || dstReg.isVec() { return fmt.Errorf("VEX destination must be a general-purpose register") } return e.emitVexFields(spec, 0, spec.opdigit, 0, 15-(dstReg.idx&15), src) } // encodeVexCountGPR encodes the three-operand count form over general-purpose // registers (SHLX, SHRX, SARX, BEXTR, BZHI): OP src, count, dst with // VEX.vvvv = src (op0), ModRM.rm = count (op1), ModRM.reg = dst (op2). func (e *enc) encodeVexCountGPR(spec vexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("VEX count instruction expects 3 operands, got %d", len(ops)) } src, count, dst := ops[0], ops[1], ops[2] dstReg, ok := dst.(Reg) if !ok || dstReg.isVec() { return fmt.Errorf("VEX destination must be a general-purpose register") } countReg, ok := count.(Reg) if !ok || countReg.isVec() { return fmt.Errorf("VEX count operand must be a general-purpose register") } srcReg, ok := src.(Reg) if !ok || srcReg.isVec() { return fmt.Errorf("VEX count source must be a general-purpose register") } rBit := 0 if dstReg.idx >= 8 { rBit = 1 } return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(srcReg.idx&15), count) } // encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the // vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ, // a store with no register-destination form). func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("store expects 2 operands, got %d", len(ops)) } srcReg, ok := ops[0].(Reg) if !ok || !srcReg.isVec() { return fmt.Errorf("store source must be a vector register") } if !memOperand(ops[1]) { return fmt.Errorf("store destination must be memory") } rBit := 0 if srcReg.idx >= 8 { rBit = 1 } return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1]) } // encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ, // VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector // move uses the store-form layout (reg = source, rm = destination), matching // the Go assembler. func (e *enc) encodeVexMove(mnem string, ms vexMoveSpec, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("VEX move expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] srcReg, srcIsVec := vecReg(src) dstReg, dstIsVec := vecReg(dst) var reg Reg var rm Operand op, w := ms.store, ms.storeW switch { case srcIsVec && dstIsVec: if !ms.vecOK { return fmt.Errorf("%s does not take two vector registers", mnem) } if ms.xmmOnly && (srcReg.size == 32 || dstReg.size == 32) { return fmt.Errorf("%s operates on XMM registers only", mnem) } if ms.regReg != 0 { op, w = ms.regReg, ms.regW } reg, rm = srcReg, dst // store form: reg = source, rm = destination. case srcIsVec: // vector → memory, or → GPR (VMOVD/VMOVQ only). if !validMoveOther(ms, dst) { return fmt.Errorf("%s: invalid destination operand", mnem) } reg, rm = srcReg, dst case dstIsVec: // memory → vector, or GPR → vector (VMOVD/VMOVQ only). if !validMoveOther(ms, src) { return fmt.Errorf("%s: invalid source operand", mnem) } op, w = ms.load, ms.loadW reg, rm = dstReg, src default: return fmt.Errorf("%s needs a vector register operand", mnem) } if ms.xmmOnly && reg.size == 32 { return fmt.Errorf("%s operates on XMM registers only", mnem) } regField := reg.idx & 7 rBit := 0 if reg.idx >= 8 { rBit = 1 } spec := vexSpec{mapSel: ms.mapSel, opcode: op, w: w, pp: ms.pp, opdigit: -1} return e.emitVexFields(spec, reg.vecLenBit(), regField, rBit, 15, rm) } // vecReg extracts a vector register from an operand. func vecReg(op Operand) (Reg, bool) { r, ok := op.(Reg) return r, ok && r.isVec() } // vecOrMem reports whether op is a vector register or a memory reference. func vecOrMem(op Operand) bool { switch op.(type) { case Mem, sbMem: return true } r, ok := op.(Reg) return ok && r.isVec() } // validMoveOther reports whether the non-vector operand of a move is // acceptable: memory always is, a GPR only for VMOVD/VMOVQ. func validMoveOther(ms vexMoveSpec, op Operand) bool { switch o := op.(type) { case Mem, sbMem: return true case Reg: return ms.gprOK && !o.isVec() } return false } // emitVexFields emits the VEX prefix, opcode, ModR/M, SIB and displacement for // the given precomputed fields. It is shared by every register/rm VEX form; // immediate bytes are appended by the caller. func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Operand) error { if l > 1 { return fmt.Errorf("ZMM operand requires an EVEX instruction") } var modrm, sib int var disp []byte var xBit, bBit int var sb *sbRef switch r := rm.(type) { case Reg: modrm = 0xC0 | regField<<3 | (r.idx & 7) sib = -1 if r.idx >= 8 { bBit = 1 } case Mem: var err error modrm, sib, disp, xBit, bBit, err = memComponents(regField, r) if err != nil { return err } case sbMem: // RIP-relative static-symbol reference; disp32 patched at link time. modrm = regField<<3 | 0x05 sib = -1 disp = le32(0) sb = &sbRef{name: r.name, addend: r.addend} default: return fmt.Errorf("invalid VEX r/m operand") } if spec.mapSel == 1 && xBit == 0 && bBit == 0 && spec.w == 0 { e.out = append(e.out, 0xC5, byte((1-rBit)<<7|vvvvBar<<3|l<<2|spec.pp)) } else { e.out = append(e.out, 0xC4, byte((1-rBit)<<7|(1-xBit)<<6|(1-bBit)<<5|spec.mapSel), byte(spec.w<<7|vvvvBar<<3|l<<2|spec.pp)) } e.out = append(e.out, spec.opcode, byte(modrm)) if sib >= 0 { e.out = append(e.out, byte(sib)) } if sb != nil { e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend}) } e.out = append(e.out, disp...) return nil }