Files
gasm-sdk/asm/vex.go
T
petrbalvin a69f8cf4a8 fix(asm): match the toolchain's bytes across the corpus sweep
A line-for-line byte comparison of the whole amd64enc.s corpus against
go tool asm surfaced divergences the pass-only accounting never showed:
PEXTRW's GPR form swapped its fields, PUSHW took an imm32 where the
toolchain bounds the immediate to 16 bits, the double shift wrote the
unmasked register number into the reg field, VCOMISS carried a 0x66
prefix, RORX dropped the destination's R bit, and the variable bit
shifts used the manual's per-width opcodes where the toolchain
consolidates each row on one opcode with the W bit.  The VEX forms the
toolchain prefers for plain vector registers (the SSE2/SSSE3/SSE4.1
AVX twins, the compare-with-predicate family, VMOVUPS, VSHUFPS, the
variable shifts) now encode under VEX, with EVEX left to the ZMM,
opmask and index-16+ spellings, and the mnemonics whose rows never
offer the 2-byte prefix force it.  Every line is pinned through the new
corpus parity test (793 lines); the whole corpus file now assembles to
the toolchain's bytes at every commented line (10022 of 10022).

Assisted-by: GLM 5.3 Flash
2026-10-06 23:59:47 +02:00

1493 lines
60 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "fmt"
// This file implements VEX (AVX/AVX2) instruction encoding. EVEX (AVX-512)
// support is a later increment.
//
// Every encoding choice here is validated two ways in the tests: by
// round-trip decoding through golang.org/x/arch's x86 decoder, and by
// byte-for-byte comparison against the output of the real Go assembler.
// vexForm selects how an instruction's operands map onto the VEX.vvvv,
// ModRM.reg and ModRM.rm fields.
type vexForm int
const (
// vexNDS3 is the three-operand form `OP src2, src1, dst` (Plan 9 order):
// ModRM.reg = dst (op2), VEX.vvvv = src1 (op1), ModRM.rm = src2 (op0).
vexNDS3 vexForm = iota
// vexRM is the two-operand form `OP src, dst` with no vvvv source:
// ModRM.reg = dst (op1), ModRM.rm = src (op0), VEX.vvvv unused.
vexRM
// vexShiftImm is the immediate-shift form `OP $imm, src, dst`: ModRM.reg =
// /digit, ModRM.rm = src (op1), VEX.vvvv = dst (op2), imm8 = op0.
vexShiftImm
// vexImmRM is the immediate form `OP $imm, src, dst` with no vvvv source:
// ModRM.reg = dst (op2), ModRM.rm = src (op1), imm8 = op0. VPSHUFD and
// VPERMQ use this shape.
vexImmRM
// vexNDS3Imm is the three-operand plus immediate form `OP $imm, src2,
// src1, dst`: ModRM.reg = dst, VEX.vvvv = src1, ModRM.rm = src2, imm8.
// VSHUFPD, VPERM2I128 and VINSERTI128 use this shape.
vexNDS3Imm
// vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg =
// ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM
// source lives in the reg field, the destination in r/m, the PEXTR-style
// layout. VEXTRACTI128 and VEXTRACTF128 use this shape.
vexExtract
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
// in ModRM.reg and the destination in r/m, the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ.
vexRMRev
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
// vector length follows the source: the packed-double → dword
// conversions (VCVTPD2DQ/VCVTTPD2DQ and their X/Y spellings) narrow into
// an XMM destination, so the L bit rides with the wider source. The
// mnemonic's spelling fixes the length (X = 128, Y = 256), which also
// covers a memory source. ModRM.reg = dst, ModRM.rm = src, no vvvv.
vexRMSrcLen
// vexZero is the no-operand form (VZEROUPPER).
vexZero
// vexZeroAll is the no-operand form that zeroes the full upper state
// (VZEROALL, the L = 1 twin of VZEROUPPER).
vexZeroAll
// vexNDS3GPR is the three-operand NDS form over general-purpose
// registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0.
vexNDS3GPR
// vexImmRMGPR is the immediate form over general-purpose registers
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
vexImmRMGPR
// vexRMOpGPR is the two-operand /digit form over general-purpose
// registers (BLSI, BLSMSK, BLSR): ModRM.reg = /digit, ModRM.rm = src
// (op0), VEX.vvvv = dst (op1), L = 0.
vexRMOpGPR
// vexCountGPR is the three-operand count form over general-purpose
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): the first operand rides
// VEX.vvvv and the second is r/m, the opposite pairing of the ANDN
// family, with reg = dst (op2), L = 0.
vexCountGPR
// vexExtractGPR is the lane-extract-to-GPR form `OP $imm, xsrc, GPR/mem
// dst`: ModRM.reg = xsrc (op1), ModRM.rm = destination (op2), imm8 =
// op0, the VPEXTRB/W/D/Q layout. EVEX only; the destination never
// carries a vector length, so the register the L'L field follows is the
// XMM source.
vexExtractGPR
// vexBlend4 is the four-operand variable blend `OP mask, src2, src1,
// dst` (VPBLENDVB): ModRM.reg = dst (op3), VEX.vvvv = src1 (op2),
// ModRM.rm = src2 (op1) and the mask register in the /is4 byte (op0).
vexBlend4
// vexNDS3Dst is the destination-first NDS form `OP dst, src1, src2`
// (VMASKMOVPS, VPMASKMOVD): ModRM.reg = dst (op0), VEX.vvvv = the mask
// source (op1), ModRM.rm = memory (op2).
vexNDS3Dst
// vexRMOpDigit is the two-operand /digit form over memory (VLDMXCSR,
// VSTMXCSR): ModRM.reg = /digit, ModRM.rm = the memory operand.
vexRMOpDigit
)
// vexSpec describes one VEX instruction's encoding parameters.
type vexSpec struct {
mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A
opcode byte
w int // VEX.W (0 for WIG)
pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2
opdigit int // ModRM.reg /digit, or -1 when reg is a register
form vexForm
vex3 bool // always the 3-byte prefix, as the toolchain emits
}
// vexTable maps an upper-case mnemonic to its VEX encoding. It is extended
// incrementally; every entry is covered by a byte-for-byte ground-truth test
// against the Go assembler.
var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG, integer arithmetic / logic / compare.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, false},
"VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3, false},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, false},
"VPSUBQ": {1, 0xFB, 0, 1, -1, vexNDS3, false},
"VPXOR": {1, 0xEF, 0, 1, -1, vexNDS3, false},
"VPOR": {1, 0xEB, 0, 1, -1, vexNDS3, false},
"VPAND": {1, 0xDB, 0, 1, -1, vexNDS3, false},
"VPANDN": {1, 0xDF, 0, 1, -1, vexNDS3, false},
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, false},
"VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3, false},
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3, false},
"VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3, false},
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, false},
// VEX.256.66.0F38.W0, dword permute (three-operand NDS form).
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, false},
// VEX.128/256.66.0F38.WIG.
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, false},
"VPMULDQ": {2, 0x28, 0, 1, -1, vexNDS3, false},
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3, false},
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3, false},
// VEX.128/256.66.0F.WIG, packed double-precision arithmetic / logic.
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3, false},
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3, false},
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3, false},
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3, false},
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3, false},
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3, false},
// VEX.128/256.0F.WIG, packed single-precision arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3, false},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3, false},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3, false},
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3, false},
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3, false},
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3, false},
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3, false},
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3, false},
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3, false},
// VEX.128.F2.0F.WIG, scalar double-precision arithmetic (the packed
// opcodes with an F2 pp).
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3, false},
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3, false},
"VMULSD": {1, 0x59, 0, 3, -1, vexNDS3, false},
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3, false},
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3, false},
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3, false},
// VEX.128.F3.0F.WIG, scalar single-precision arithmetic (the packed
// opcodes with an F3 pp).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, false},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, false},
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, false},
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3, false},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, false},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, false},
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, false},
// Scalar fused multiply-add (NDS form). The Go assembler carries the
// same 66 prefix as the packed forms on every FMA row, and W1 on the
// double-precision spellings, so SD shares PD's prefix/W pair and the
// scalar width rides on the W bit.
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3, false},
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3, false},
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
// no vvvv).
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM, false},
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, false},
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, false},
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, false},
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM, false},
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, false},
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM, false},
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM, false},
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM, false},
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM, false},
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, false},
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM, false},
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM, false},
"VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM, false},
"VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM, false},
// VEX.128/256.F3.0F.WIG, signed dword to packed double conversion
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, false},
// VEX.128/256.0F.WIG, signed dword to packed single conversion
// (reg=dst, rm=src, no vvvv, no mandatory prefix).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, false},
// VEX.128/256.0F.WIG, packed single to packed double conversion
// (reg=dst, rm=src; the destination is the wide operand and sets the
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
// the Go assembler emits the instruction with pp = 00, and gasm follows
// the Go assembler's bytes, its machine code is the oracle, not the
// manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, false},
// VEX.128.F2.0F.WIG, duplicate the low double of each 128-bit lane
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM, false},
// VEX.128/256.66.0F.WIG, move mask to a GPR (reg=gpr dst, rm=vec src).
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM, false},
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM, false}, // no 66 prefix (that would be VMOVMSKPD)
// VEX.128/256.66.0F.WIG, immediate shifts (opdigit selects the shift).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, false},
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, false},
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm, false},
"VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm, false},
"VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm, false},
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM, false},
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8), and its
// double twin under op 01; the in-lane permutes under 04/05.
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM, false},
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM, false},
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM, false},
"VPERMILPD": {3, 0x05, 0, 1, -1, vexImmRM, false},
// VEX.66.0F3A.W0, the immediate-controlled AVX tail: the rounding
// pair, the AES key assistant and the string compares.
"VROUNDPD": {3, 0x09, 0, 1, -1, vexImmRM, false},
"VROUNDPS": {3, 0x08, 0, 1, -1, vexImmRM, false},
"VAESKEYGENASSIST": {3, 0xDF, 0, 1, -1, vexImmRM, false},
"VPCMPESTRI": {3, 0x61, 0, 1, -1, vexImmRM, false},
"VPCMPESTRM": {3, 0x60, 0, 1, -1, vexImmRM, false},
"VPCMPISTRI": {3, 0x63, 0, 1, -1, vexImmRM, false},
"VPCMPISTRM": {3, 0x62, 0, 1, -1, vexImmRM, false},
// VEX.128.66.0F3A.W0, the scalar lane extract to a GPR or memory
// (reg = the XMM source, r/m = the destination).
"VEXTRACTPS": {3, 0x17, 0, 1, -1, vexExtractGPR, false},
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR, false},
// VEX.128.66.0F3A.W0, the four-operand variable blend with its mask
// register in the /is4 byte.
"VPBLENDVB": {3, 0x4C, 0, 1, -1, vexBlend4, false},
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// imm8).
"VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm, false},
// VEX.256.66.0F3A.W0, permute / insert (same shape; VINSERTI128's rm is
// the XMM or memory source).
"VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm, false},
"VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm, false},
// VEX.256.66.0F3A.W0, lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract, false},
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract, false},
// VEX.128/256.66.0F3A.W0, half-precision convert back ($imm, src, dst:
// reg=src, rm=XMM/memory dst, imm8, the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, false},
// VEX.128.0F.W0, no operands.
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero, false},
// VEX.256.0F.W0, zero all vector registers (the L = 1 twin).
"VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll, false},
// VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare.
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm, false},
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm, false},
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3, false},
// VEX.128/256.0F.WIG, packed single XOR (NDS form).
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3, false},
// VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control.
"VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm, false},
"VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm, false},
// VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling
// falls through to the EVEX table.
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm, false},
// VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst).
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm, false},
// VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8).
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm, false},
// BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR).
"ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR, false},
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR, false},
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR, false},
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR, false},
// VEX.NDS.LZ.0F38, the BMI2 three-operand bit ops: BEXTR and BZHI
// share the F7/F5 opcodes across W, the variable shifts carry their
// direction in the prefix (SHLX 66, SHRX F2, SARX F3) and PDEP/PEXT
// in F2/F3.
"BEXTRL": {2, 0xF7, 0, 0, -1, vexCountGPR, false},
"BEXTRQ": {2, 0xF7, 1, 0, -1, vexCountGPR, false},
"BZHIL": {2, 0xF5, 0, 0, -1, vexCountGPR, false},
"BZHIQ": {2, 0xF5, 1, 0, -1, vexCountGPR, false},
"SARXL": {2, 0xF7, 0, 2, -1, vexCountGPR, false},
"SARXQ": {2, 0xF7, 1, 2, -1, vexCountGPR, false},
"SHLXL": {2, 0xF7, 0, 1, -1, vexCountGPR, false},
"SHLXQ": {2, 0xF7, 1, 1, -1, vexCountGPR, false},
"SHRXL": {2, 0xF7, 0, 3, -1, vexCountGPR, false},
"SHRXQ": {2, 0xF7, 1, 3, -1, vexCountGPR, false},
"PDEPL": {2, 0xF5, 0, 3, -1, vexNDS3GPR, false},
"PDEPQ": {2, 0xF5, 1, 3, -1, vexNDS3GPR, false},
"PEXTL": {2, 0xF5, 0, 2, -1, vexNDS3GPR, false},
"PEXTQ": {2, 0xF5, 1, 2, -1, vexNDS3GPR, false},
// VEX.LZ.0F38.W, the BMI1 unary bit ops (src, dst: ModRM.reg = /digit,
// rm = src, vvvv = dst).
"BLSIL": {2, 0xF3, 0, 0, 3, vexRMOpGPR, false},
"BLSIQ": {2, 0xF3, 1, 0, 3, vexRMOpGPR, false},
"BLSMSKL": {2, 0xF3, 0, 0, 2, vexRMOpGPR, false},
"BLSMSKQ": {2, 0xF3, 1, 0, 2, vexRMOpGPR, false},
"BLSRL": {2, 0xF3, 0, 0, 1, vexRMOpGPR, false},
"BLSRQ": {2, 0xF3, 1, 0, 1, vexRMOpGPR, false},
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR, false},
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR, false},
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM, false},
// VEX.66.0F38.W0, broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; SD is 256-bit only).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM, false},
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM, false},
// VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a
// YMM (the encoder rejects an XMM destination, as go tool asm does).
"VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM, false},
// VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg,
// memory destination in rm).
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev, false},
// VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv).
"VPTEST": {2, 0x17, 0, 1, -1, vexRM, false},
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
// source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, false},
// VEX.F3.0F.WIG, replicate even/odd singles (reg=dst, rm=src).
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM, false},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM, false},
// VEX.66.0F.WIG, packed double to packed single conversion, the X/Y
// spellings: the destination is always XMM and the spelling fixes the
// source length (X = 128, Y = 256).
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen, false},
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen, false},
// VEX scalar conversions between vector and general-purpose registers.
// Vector to GPR (two operands: vec/mem source, GPR destination, vvvv
// unused; the length follows the source).
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM, false},
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM, false},
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM, false},
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM, false},
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM, false},
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM, false},
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM, false},
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM, false},
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
// vector source in vvvv, vector destination in reg).
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3, false},
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3, false},
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3, false},
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3, false},
// VEX.128/256.66.0F.WIG, word shifts (opdigit selects the shift).
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm, false},
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm, false},
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm, false},
// VEX.F2.0F, packed double to packed dword conversions, truncating and
// non-truncating. The destination is always XMM; the X/Y spellings fix
// the source length (XMM/YMM), and VEX.L follows it, see vexSrcLen.
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen, false},
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen, false},
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen, false},
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen, false},
// --- the VEX forms the avx512enc corpus exercises alongside the EVEX
// spellings, read off the toolchain opcode tables ---
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3, false},
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3, false},
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3, false},
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3, false},
"VANDNPD": {1, 0x55, 0, 1, -1, vexNDS3, false},
"VANDPD": {1, 0x54, 0, 1, -1, vexNDS3, false},
"VCOMISD": {1, 0x2F, 0, 1, -1, vexRM, false},
"VCVTSD2SS": {1, 0x5A, 0, 3, -1, vexNDS3, false},
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3, false},
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3, false},
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3, false},
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3, false},
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3, false},
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3, false},
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3, false},
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3, false},
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3, false},
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3, false},
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3, false},
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3, false},
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3, false},
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3, false},
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3, false},
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3, false},
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3, false},
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3, false},
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3, false},
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3, false},
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3, false},
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3, false},
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3, false},
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3, false},
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3, false},
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3, false},
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3, false},
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3, false},
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3, false},
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3, false},
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3, false},
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3, false},
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3, false},
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3, false},
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3, false},
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3, false},
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3, false},
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3, false},
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3, false},
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3, false},
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3, false},
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3, false},
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3, false},
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3, false},
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3, false},
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3, false},
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3, false},
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3, false},
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3, false},
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3, false},
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3, false},
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3, false},
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3, false},
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3, false},
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3, false},
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3, false},
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3, false},
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3, false},
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm, false},
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3, false},
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM, false},
"VMOVNTPD": {1, 0x2B, 0, 1, -1, vexRMRev, false},
"VORPD": {1, 0x56, 0, 1, -1, vexNDS3, false},
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3, false},
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3, false},
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3, false},
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3, false},
"VPCMPEQQ": {2, 0x29, 0, 1, -1, vexNDS3, false},
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3, false},
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3, false},
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3, false},
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3, false},
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3, false},
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtract, false},
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtract, false},
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtract, false},
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm, false},
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm, false},
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3, false},
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3, false},
"VPMULUDQ": {1, 0xF4, 0, 1, -1, vexNDS3, false},
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3, false},
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3, false},
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3, false},
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3, false},
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3, false},
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3, false},
"VPUNPCKHQDQ": {1, 0x6D, 0, 1, -1, vexNDS3, false},
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3, false},
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3, false},
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3, false},
"VSQRTPD": {1, 0x51, 0, 1, -1, vexRM, false},
"VSQRTSD": {1, 0x51, 0, 3, -1, vexNDS3, false},
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3, false},
"VUCOMISD": {1, 0x2E, 0, 1, -1, vexRM, false},
// VEX.0F.WIG, the plain-prefix single/double arithmetic and unpack
// spellings (no 66 prefix; WIG, so W = 0).
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3, false},
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3, false},
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3, false},
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3, false},
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3, false},
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM, false},
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev, false},
// VEX.128.66.0F, the scalar and packed compare forms.
"VCOMISS": {1, 0x2F, 0, 0, -1, vexRM, false},
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM, false},
// VEX.128.0F.F3/F2.W0, the high/low word shuffles ($imm, src, dst).
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM, false},
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM, false},
// --- the SSE2/SSSE3/SSE4.1 AVX twins the corpus exercises under plain
// vector registers, which the toolchain encodes in VEX (EVEX only for
// ZMM, opmask and index-16+ registers) ---
// VEX.NDS.128/256, the compare-with-predicate family ($imm, src2, src1,
// dst): PD/PS carry 66/none, the scalar SD/SS spellings F2/F3.
"VCMPPD": {1, 0xC2, 0, 1, -1, vexNDS3Imm, false},
"VCMPPS": {1, 0xC2, 0, 0, -1, vexNDS3Imm, false},
"VCMPSD": {1, 0xC2, 0, 3, -1, vexNDS3Imm, false},
"VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, false},
// VEX.128/256.66.0F.WIG, the packed integer arithmetic and logic twins
// the main table's EVEX rows shadow.
"VPADDB": {1, 0xFC, 0, 1, -1, vexNDS3, false},
"VPADDW": {1, 0xFD, 0, 1, -1, vexNDS3, false},
"VPSUBB": {1, 0xF8, 0, 1, -1, vexNDS3, false},
"VPSUBW": {1, 0xF9, 0, 1, -1, vexNDS3, false},
"VPMULLW": {1, 0xD5, 0, 1, -1, vexNDS3, false},
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, false},
"VPMADDWD": {1, 0xF5, 0, 1, -1, vexNDS3, false},
"VPMAXSW": {1, 0xEE, 0, 1, -1, vexNDS3, false},
"VPMAXUB": {1, 0xDE, 0, 1, -1, vexNDS3, false},
"VPMINSW": {1, 0xEA, 0, 1, -1, vexNDS3, false},
"VPMINUB": {1, 0xDA, 0, 1, -1, vexNDS3, false},
"VPAVGB": {1, 0xE0, 0, 1, -1, vexNDS3, false},
"VPAVGW": {1, 0xE3, 0, 1, -1, vexNDS3, false},
"VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, false},
"VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, false},
// VEX.128/256.66.0F38.WIG, the SSE4.1 integer twins.
"VPMAXSB": {2, 0x3C, 0, 1, -1, vexNDS3, false},
"VPMAXSD": {2, 0x3D, 0, 1, -1, vexNDS3, false},
"VPMAXUD": {2, 0x3F, 0, 1, -1, vexNDS3, false},
"VPMAXUW": {2, 0x3E, 0, 1, -1, vexNDS3, false},
"VPMINSB": {2, 0x38, 0, 1, -1, vexNDS3, false},
"VPMINSD": {2, 0x39, 0, 1, -1, vexNDS3, false},
"VPMINUD": {2, 0x3B, 0, 1, -1, vexNDS3, false},
"VPMINUW": {2, 0x3A, 0, 1, -1, vexNDS3, false},
// the absolute values are two-operand, reg = destination.
"VPABSB": {2, 0x1C, 0, 1, -1, vexRM, false},
"VPABSW": {2, 0x1D, 0, 1, -1, vexRM, false},
"VPABSD": {2, 0x1E, 0, 1, -1, vexRM, false},
"VPACKUSDW": {2, 0x2B, 0, 1, -1, vexNDS3, false},
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, false},
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM, false},
// VEX.128/256.66.0F38.W0/W1, the variable bit shifts (the count rides
// vvvv or memory).
// The toolchain consolidates each variable shift on one opcode with
// the W bit carrying the width: 45 for the right shifts, 46 for the
// arithmetic and 47 for the left, not the manual's per-width rows.
"VPSLLVD": {2, 0x47, 0, 1, -1, vexNDS3, false},
"VPSLLVQ": {2, 0x47, 1, 1, -1, vexNDS3, false},
"VPSRAVD": {2, 0x46, 0, 1, -1, vexNDS3, false},
"VPSRLVD": {2, 0x45, 0, 1, -1, vexNDS3, false},
"VPSRLVQ": {2, 0x45, 1, 1, -1, vexNDS3, false},
// VEX.128/256.0F and F3.0F, the packed float converts the SSE2 pair
// spells: dword to packed single and its truncating twin.
"VCVTPS2DQ": {1, 0x5B, 0, 1, -1, vexRM, false},
"VCVTTPS2DQ": {1, 0x5B, 0, 2, -1, vexRM, false},
// the low-half register move the corpus exercises.
"VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, false},
// VEX.NDS.128/256, the two-source shuffle with its imm8 control.
"VSHUFPS": {1, 0xC6, 0, 0, -1, vexNDS3Imm, false},
// --- the corpus families from amd64enc.s: the SSE3 horizontal and
// add-subtract pairs, the SSSE3 sign and horizontal integers, the AVX
// reciprocity and test pairs, the masked and non-temporal oddities ---
// VEX.128/256, the horizontal and add-subtract float pairs: PD carries
// 66, PS carries F2 (0F 7C/7D and 0F D0).
"VHADDPD": {1, 0x7C, 0, 1, -1, vexNDS3, false},
"VHADDPS": {1, 0x7C, 0, 3, -1, vexNDS3, false},
"VHSUBPD": {1, 0x7D, 0, 1, -1, vexNDS3, false},
"VHSUBPS": {1, 0x7D, 0, 3, -1, vexNDS3, false},
"VADDSUBPD": {1, 0xD0, 0, 1, -1, vexNDS3, false},
"VADDSUBPS": {1, 0xD0, 0, 3, -1, vexNDS3, false},
// VEX.128/256.0F38.W0, the SSSE3 horizontal integer family and the sign
// controls, all NDS over 66.
"VPHADDW": {2, 0x01, 0, 1, -1, vexNDS3, false},
"VPHADDD": {2, 0x02, 0, 1, -1, vexNDS3, false},
"VPHADDSW": {2, 0x03, 0, 1, -1, vexNDS3, false},
"VPHSUBW": {2, 0x05, 0, 1, -1, vexNDS3, false},
"VPHSUBD": {2, 0x06, 0, 1, -1, vexNDS3, false},
"VPHSUBSW": {2, 0x07, 0, 1, -1, vexNDS3, false},
"VPSIGNB": {2, 0x08, 0, 1, -1, vexNDS3, false},
"VPSIGNW": {2, 0x09, 0, 1, -1, vexNDS3, false},
"VPSIGND": {2, 0x0A, 0, 1, -1, vexNDS3, false},
// VEX.128/256.0F38.W0, the masked loads/stores whose mask source rides
// vvvv, memory in r/m: the Plan 9 order puts the destination first
// (dst, mask, src), its own form below. VMOVHLPS is the plain NDS
// register move under 0F 12, and the test pair and the AES inverse cube
// root are two-operand.
"VMOVHLPS": {1, 0x12, 0, 0, -1, vexNDS3, false},
"VMASKMOVPD": {2, 0x2F, 0, 1, -1, vexNDS3Dst, false},
"VMASKMOVPS": {2, 0x2E, 0, 1, -1, vexNDS3Dst, false},
"VPMASKMOVD": {2, 0x8E, 0, 1, -1, vexNDS3Dst, false},
"VPMASKMOVQ": {2, 0x8E, 1, 1, -1, vexNDS3Dst, false},
"VTESTPD": {2, 0x0F, 0, 1, -1, vexRM, false},
"VTESTPS": {2, 0x0E, 0, 1, -1, vexRM, false},
"VAESIMC": {2, 0xDB, 0, 1, -1, vexRM, false},
"VPHMINPOSUW": {2, 0x41, 0, 1, -1, vexRM, false},
// VEX.128/256.66.0F38.W0, broadcast a 128-bit lane into a YMM.
"VBROADCASTF128": {2, 0x1A, 0, 1, -1, vexRM, false},
// VEX.128/256, the reciprocal square root estimates: PS bare (two-op
// RM), SS F3-prefixed NDS (the scalar preserved source).
"VRCPPS": {1, 0x53, 0, 0, -1, vexRM, false},
"VRSQRTPS": {1, 0x52, 0, 0, -1, vexRM, false},
"VRCPSS": {1, 0x53, 0, 2, -1, vexNDS3, false},
"VRSQRTSS": {1, 0x52, 0, 2, -1, vexNDS3, false},
// VEX.128/256.F2.0F.WIG, the unaligned load with cache hints and the
// duplicated low double, F3 for the singles replica.
"VLDDQU": {1, 0xF0, 0, 3, -1, vexRM, false},
// VEX.128/256.66.0F, the move-mask twin of VMOVMSKPS and the masked
// store over the integer bank.
"VMOVMSKPD": {1, 0x50, 0, 1, -1, vexRM, false},
"VMASKMOVDQU": {1, 0xF7, 0, 1, -1, vexRM, false},
// VEX.128/256.0F3A.W0, the imm8 tail the legacy set carries and its
// variable blends over the is4 byte.
"VBLENDPD": {3, 0x0D, 0, 1, -1, vexNDS3Imm, false},
"VBLENDPS": {3, 0x0C, 0, 1, -1, vexNDS3Imm, false},
"VPBLENDW": {3, 0x0E, 0, 1, -1, vexNDS3Imm, false},
"VDPPD": {3, 0x41, 0, 1, -1, vexNDS3Imm, false},
"VDPPS": {3, 0x40, 0, 1, -1, vexNDS3Imm, false},
"VINSERTPS": {3, 0x21, 0, 1, -1, vexNDS3Imm, false},
"VMPSADBW": {3, 0x42, 0, 1, -1, vexNDS3Imm, false},
"VROUNDSD": {3, 0x0B, 0, 1, -1, vexNDS3Imm, false},
"VROUNDSS": {3, 0x0A, 0, 1, -1, vexNDS3Imm, false},
"VPINSRB": {3, 0x20, 0, 1, -1, vexNDS3Imm, false},
// VEX.128.66.0F3A.W0, the four-operand variable blends over the /is4
// byte (the mask rides is4[7:4], the raw register number times 16).
"VBLENDVPS": {3, 0x4A, 0, 1, -1, vexBlend4, false},
"VBLENDVPD": {3, 0x4B, 0, 1, -1, vexBlend4, false},
// VEX.256.66.0F3A.W0, the lane insert, and the GPR insert the legacy
// set spells: VPINSRW rides plain 0F C4 with 66.
"VINSERTF128": {3, 0x18, 0, 1, -1, vexNDS3Imm, false},
"VPINSRW": {1, 0xC4, 0, 1, -1, vexNDS3Imm, false},
// VEX.128/256.0F, the MXCSR accessors, memory alone, no prefix.
"VLDMXCSR": {1, 0xAE, 0, 0, 2, vexRMOpDigit, false},
"VSTMXCSR": {1, 0xAE, 0, 0, 3, vexRMOpDigit, false},
}
// vex3Only lists the mnemonics whose rows never offer the 2-byte prefix:
// the toolchain emits the 3-byte spelling at every operand shape, and the
// corpus pins it. The table entries for the register-form members carry
// the preference directly.
var vex3Only = map[string]bool{
"VCMPPD": true, "VCMPPS": true, "VCMPSD": true, "VCMPSS": true,
"VCOMISS": true, "VCVTPS2DQ": true, "VCVTTPS2DQ": true,
"VMOVLHPS": true, "VMOVUPS": true,
}
func init() {
for n := range vex3Only {
if s, ok := vexTable[n]; ok {
s.vex3 = true
vexTable[n] = s
}
}
}
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
// the packed-double → dword conversions) to its fixed vector length:
// X = 128 (L = 0), Y = 256 (L = 1). The spelling fixes the length even for
// a memory source, matching the Go assembler's ytab.
var vexSrcLen = map[string]int{
"VCVTPD2DQX": 0,
"VCVTPD2DQY": 1,
"VCVTTPD2DQX": 0,
"VCVTTPD2DQY": 1,
"VCVTPD2PSX": 0,
"VCVTPD2PSY": 1,
}
// vexVarShift maps the shift mnemonics to their variable-count opcode, the
// form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8),
// an ordinary NDS encoding rather than the /digit immediate form above.
var vexVarShift = map[string]byte{
"VPSLLD": 0xF2,
"VPSLLQ": 0xF3,
"VPSLLW": 0xF1,
"VPSRAD": 0xE2,
"VPSRAW": 0xE1,
"VPSRLD": 0xD2,
"VPSRLQ": 0xD3,
"VPSRLW": 0xD1,
}
// vexPermilReg maps the VPERMIL register-control spellings to their 0F38
// NDS opcodes: the control rides vvvv, the immediate form the main table
// carries never enters this path.
var vexPermilReg = map[string]vexSpec{
"VPERMILPS": {2, 0x0C, 0, 1, -1, vexNDS3, false},
"VPERMILPD": {2, 0x0D, 0, 1, -1, vexNDS3, false},
}
// vexMoveSpec describes a VEX move, which takes different opcodes (and
// sometimes a different VEX.W) per operand direction. The Go assembler
// encodes a vector→vector move with the store-form opcode (reg = source,
// rm = destination), so regReg defaults to store when zero.
type vexMoveSpec struct {
mapSel int
pp int
load byte // r/m → vector: reg=dst, rm=src
store byte // vector → r/m: reg=src, rm=dst
loadW int
storeW int
regReg byte // vector → vector opcode; 0 uses store
regW int
vecOK bool // the non-fixed operand may be a vector register
gprOK bool // the non-fixed operand may be a general-purpose register
xmmOnly bool // YMM registers are rejected
}
// vexMoveTable maps an upper-case move mnemonic to its encoding.
var vexMoveTable = map[string]vexMoveSpec{
// VEX.128/256.F3.0F.WIG, unaligned integer move.
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG, aligned integer move.
"VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
"VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true},
// VMOVQ, 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true},
// VEX.128.F2.0F.WIG, scalar double move: two operands move against
// memory, three operands the NDS store-opcode form (see encodeVexMove).
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128.F3.0F.WIG, scalar single move, the same two shapes.
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128/256, the unaligned packed single move the corpus exercises.
"VMOVUPS": {1, 0, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
// VEX.128/256, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
}
// isVex reports whether the mnemonic is a VEX-encoded instruction we handle.
func isVex(mnemUpper string) bool {
if _, ok := vexTable[mnemUpper]; ok {
return true
}
if _, ok := vexMoveTable[mnemUpper]; ok {
return true
}
// The dual-shape moves (VMOVHPD/VMOVLPD and the single-precision twins)
// pick their VEX form by operand count in encodeVex.
switch mnemUpper {
case "VMOVHPD", "VMOVLPD", "VMOVHPS", "VMOVLPS":
return true
}
return false
}
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
// Vector register indices 16-31 exist only in EVEX encodings; fail
// loudly rather than silently truncating the index.
for _, op := range ops {
if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 {
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
}
}
// VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination
// only; an XMM destination is rejected exactly as go tool asm does.
if mnemUpper == "VBROADCASTI128" {
dstReg, ok := ops[len(ops)-1].(Reg)
if len(ops) != 2 || !ok || dstReg.size != 32 {
return fmt.Errorf("VBROADCASTI128 requires a YMM destination")
}
}
if ms, ok := vexMoveTable[mnemUpper]; ok {
return e.encodeVexMove(mnemUpper, ms, ops)
}
// The shifts come in two shapes under one mnemonic: an immediate count
// ($imm, src, dst) and a variable count in an XMM register or memory
// (count, src, dst), the latter an ordinary NDS form.
if op, ok := vexVarShift[mnemUpper]; ok && len(ops) == 3 {
if _, isImm := ops[0].(Imm); !isImm {
if !vecOrMem(ops[0]) {
return fmt.Errorf("%s: shift count must be an immediate, a vector register or memory", mnemUpper)
}
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
}
}
// The VPERMIL register-control form: the control rides vvvv (an NDS
// encoding under 0F38), the immediate form the main table carries.
if spec, ok := vexPermilReg[mnemUpper]; ok && len(ops) == 3 {
if _, isImm := ops[0].(Imm); !isImm {
return e.encodeVexNDS3(spec, ops)
}
}
// The high/low double moves split by operand count: three operands
// load-and-insert (mem, src, dst, an NDS form), two store (xmm, m64,
// the reversed store layout).
if mnemUpper == "VMOVHPD" || mnemUpper == "VMOVLPD" || mnemUpper == "VMOVHPS" || mnemUpper == "VMOVLPS" {
loadOp, storeOp := byte(0x16), byte(0x17)
if mnemUpper == "VMOVLPD" || mnemUpper == "VMOVLPS" {
loadOp, storeOp = 0x12, 0x13
}
pp := 1
if mnemUpper == "VMOVHPS" || mnemUpper == "VMOVLPS" {
pp = 0
}
switch len(ops) {
case 3:
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: loadOp, w: 0, pp: pp, opdigit: -1, form: vexNDS3}, ops)
case 2:
return e.encodeVexRMRev(vexSpec{mapSel: 1, opcode: storeOp, w: 0, pp: pp, opdigit: -1, form: vexRMRev}, ops)
}
return fmt.Errorf("%s expects 2 or 3 operands, got %d", mnemUpper, len(ops))
}
spec := vexTable[mnemUpper]
switch spec.form {
case vexNDS3:
return e.encodeVexNDS3(spec, ops)
case vexRM:
return e.encodeVexRM(spec, ops)
case vexShiftImm:
return e.encodeVexShiftImm(spec, ops)
case vexImmRM:
return e.encodeVexImmRM(spec, ops)
case vexNDS3Imm:
return e.encodeVexNDS3Imm(spec, ops)
case vexExtract:
return e.encodeVexExtract(spec, ops)
case vexExtractGPR:
return e.encodeVexExtractGPR(spec, ops)
case vexBlend4:
return e.encodeVexBlend4(spec, ops)
case vexRMSrcLen:
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero:
return e.encodeVexZero(mnemUpper, spec, ops)
case vexZeroAll:
return e.encodeVexZeroAll(mnemUpper, spec, ops)
case vexNDS3GPR:
return e.encodeVexNDS3GPR(spec, ops)
case vexImmRMGPR:
return e.encodeVexImmRMGPR(spec, ops)
case vexRMOpGPR:
return e.encodeVexRMOpGPR(spec, ops)
case vexCountGPR:
return e.encodeVexCountGPR(spec, ops)
case vexRMRev:
return e.encodeVexRMRev(spec, ops)
case vexNDS3Dst:
return e.encodeVexNDS3Dst(spec, ops)
case vexRMOpDigit:
return e.encodeVexRMOpDigit(mnemUpper, spec, ops)
}
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
}
// encodeVexNDS3Dst encodes the destination-first NDS forms (VMASKMOVPS,
// VPMASKMOVD): ModRM.reg = the vector register, VEX.vvvv = the mask source,
// ModRM.rm = memory. Both directions exist: (dst, mask, mem) stores under
// the table opcode, (mem, mask, dst) loads under its twin two lower (the
// opcode rows pair 2E/2C, 2F/2D and 8E/8C).
func (e *enc) encodeVexNDS3Dst(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
}
opcode := spec.opcode
first, second, third := ops[0], ops[1], ops[2]
if isX86Mem(first) {
first, third = third, first
opcode -= 2
}
dstReg, ok := first.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("VEX destination must be a vector register")
}
vvvvReg, ok := second.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("VEX mask source must be a vector register")
}
if !vecOrMem(third) {
return fmt.Errorf("VEX memory source expected")
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
enc := spec
enc.opcode = opcode
return e.emitVexFields(enc, dstReg.vecLenBit(), regField, rBit, 15-(vvvvReg.idx&15), third)
}
// encodeVexRMOpDigit encodes the MXCSR accessors: the single memory operand
// rides r/m under the fixed /digit, the way the legacy 0F AE pair does.
func (e *enc) encodeVexRMOpDigit(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 memory operand, got %d", mnem, len(ops))
}
if !isX86Mem(ops[0]) {
return fmt.Errorf("%s requires a memory operand", mnem)
}
return e.emitVexFields(spec, 0, spec.opdigit, 0, 15, ops[0])
}
// encodeVexNDS3 encodes the three-operand NDS form: OP src2, src1, dst.
func (e *enc) encodeVexNDS3(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
}
src2, src1, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("VEX destination must be a vector register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("VEX vvvv operand must be a vector register")
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
vvvvBar := 15 - (vvvvReg.idx & 15)
return e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2)
}
// encodeVexRM encodes the two-operand form: OP src, dst (no vvvv source).
// ModRM.reg = dst, ModRM.rm = src; the vector length comes from whichever
// operand is a vector register (the destination for extends/broadcasts, the
// source for the move-mask instructions whose destination is a GPR).
func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("VEX two-operand instruction expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok {
return fmt.Errorf("VEX destination must be a register")
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
// Vector length: from the destination if it is a vector, otherwise from the
// source (move-mask instructions have a GPR destination and a vector source).
l := 0
if dstReg.isVec() {
l = dstReg.vecLenBit()
} else if srcReg, ok := src.(Reg); ok && srcReg.isVec() {
l = srcReg.vecLenBit()
}
// An unused vvvv field must be stored as all ones (v̄vvv = 1111); the
// hardware raises #UD on any other value.
return e.emitVexFields(spec, l, regField, rBit, 15, src)
}
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the VEX.L bit following the source, fixed
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
// the source is memory.
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("VEX destination must be a vector register")
}
ll, ok := vexSrcLen[mnem]
if !ok {
return fmt.Errorf("no fixed vector length for %s", mnem)
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
// An unused vvvv field must be stored as all ones (v̄vvv = 1111).
return e.emitVexFields(spec, ll, regField, rBit, 15, src)
}
// encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst.
// The destination is carried in VEX.vvvv, the source in ModRM.rm, and the
// shift kind in the ModRM.reg /digit.
func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shift count must be an immediate")
}
// The count source is a vector register or memory; the VEX length
// follows the destination register either way.
if !vecOrMem(src) {
return fmt.Errorf("shift source must be a vector register or memory")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("shift destination must be a vector register")
}
vvvvBar := 15 - (dstReg.idx & 15)
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, src); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// imm8 range-checks an immediate for an 8-bit field. Shuffle controls are
// unsigned bit masks, but the negative spelling ($-1 = all bits set) is
// accepted, so the accepted span is -128..255.
func imm8(v int64) (byte, error) {
if v < -128 || v > 255 {
return 0, fmt.Errorf("immediate $%d does not fit in 8 bits", v)
}
return byte(v), nil
}
// encodeVexImmRM encodes an immediate form with no vvvv source: OP $imm, src,
// dst (VPSHUFD, VPERMQ). ModRM.reg = dst, ModRM.rm = src, imm8 appended.
func (e *enc) encodeVexImmRM(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shuffle control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("shuffle destination must be a vector register")
}
// The vector length follows the source when it is a vector register,
// otherwise the destination (a memory source carries no length).
l := dstReg.vecLenBit()
if srcReg, ok := src.(Reg); ok && srcReg.isVec() {
l = srcReg.vecLenBit()
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, l, regField, rBit, 15, src); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexNDS3Imm encodes the three-operand plus immediate form: OP $imm,
// src2, src1, dst (VSHUFPD, VPERM2I128, VINSERTI128). ModRM.reg = dst,
// VEX.vvvv = src1, ModRM.rm = src2, imm8 appended.
func (e *enc) encodeVexNDS3Imm(spec vexSpec, ops []Operand) error {
if len(ops) != 4 {
return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops))
}
imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shuffle control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("destination must be a vector register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("second source must be a vector register")
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
vvvvBar := 15 - (vvvvReg.idx & 15)
if err := e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexExtract encodes a lane extract: OP $imm, ysrc, xdst
// (VEXTRACTI128, VEXTRACTF128). The YMM source occupies ModRM.reg and the
// XMM (or memory) destination ModRM.rm; imm8 selects the lane.
func (e *enc) encodeVexExtract(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, ysrc, xdst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("extract source must be a vector register")
}
regField := srcReg.idx & 7
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, srcReg.vecLenBit(), regField, rBit, 15, dst); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexZero encodes a no-operand instruction (VZEROUPPER).
func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 0 {
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
}
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 0.
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|spec.pp), spec.opcode)
return nil
}
// encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1
// twin of VZEROUPPER.
func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 0 {
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
}
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1.
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode)
return nil
}
// encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose
// registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1,
// rm = src2 and L = 0.
func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
}
src2, src1, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
vvvvReg, ok := src1.(Reg)
if !ok || vvvvReg.isVec() {
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2)
}
// encodeVexImmRMGPR encodes the immediate form over general-purpose
// registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0.
func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shift control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15, src); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexRMOpGPR encodes the two-operand /digit form over general-purpose
// registers (BLSI, BLSMSK, BLSR): OP src, dst with ModRM.reg = /digit,
// ModRM.rm = src and VEX.vvvv = dst.
func (e *enc) encodeVexRMOpGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("instruction expects 2 operands (src, dst), got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
return e.emitVexFields(spec, 0, spec.opdigit, 0, 15-(dstReg.idx&15), src)
}
// encodeVexCountGPR encodes the three-operand count form over general-purpose
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): OP src, count, dst with
// VEX.vvvv = src (op0), ModRM.rm = count (op1, register or memory),
// ModRM.reg = dst (op2).
func (e *enc) encodeVexCountGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX count instruction expects 3 operands, got %d", len(ops))
}
src, count, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
switch count.(type) {
case Reg, Mem, sbMem:
default:
return fmt.Errorf("VEX count operand must be a general-purpose register or memory")
}
if r, ok := count.(Reg); ok && r.isVec() {
return fmt.Errorf("VEX count operand must be a general-purpose register or memory")
}
srcReg, ok := src.(Reg)
if !ok || srcReg.isVec() {
return fmt.Errorf("VEX count source must be a general-purpose register")
}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(srcReg.idx&15), count)
}
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
// a store with no register-destination form).
func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("store expects 2 operands, got %d", len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("store source must be a vector register")
}
if !memOperand(ops[1]) {
return fmt.Errorf("store destination must be memory")
}
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
}
// encodeVexExtractGPR encodes the lane extract to a general-purpose register
// or memory (VEXTRACTPS): OP $imm, xsrc, gpr/mem with the XMM source in
// ModRM.reg and the destination in r/m, L = 0.
func (e *enc) encodeVexExtractGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() || srcReg.size != 16 {
return fmt.Errorf("extract source must be an XMM register")
}
if _, isReg := dst.(Reg); !isReg && !memOperand(dst) {
return fmt.Errorf("extract destination must be a register or memory")
}
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, 0, srcReg.idx&7, rBit, 15, dst); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexBlend4 encodes the four-operand variable blend (VPBLENDVB and the
// VBLENDV pair): OP mask, src2, src1, dst with ModRM.reg = dst, VEX.vvvv =
// src1, r/m = src2 and the mask register in the trailing /is4 byte, whose
// high nibble carries the mask's register number raw.
func (e *enc) encodeVexBlend4(spec vexSpec, ops []Operand) error {
if len(ops) != 4 {
return fmt.Errorf("blend expects 4 operands (mask, src2, src1, dst), got %d", len(ops))
}
mask, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
maskReg, ok := mask.(Reg)
if !ok || !maskReg.isVec() {
return fmt.Errorf("blend mask must be a vector register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("blend second source must be a vector register")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("blend destination must be a vector register")
}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, dstReg.vecLenBit(), dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2); err != nil {
return err
}
// The /is4 byte names the mask register: its number in the high nibble,
// the layout the Go assembler and the hardware agree on for X0-X15.
e.out = append(e.out, byte(maskReg.idx)<<4)
return nil
}
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
// move uses the store-form layout (reg = source, rm = destination), matching
// the Go assembler. The scalar moves (VMOVSD, VMOVSS) also carry a
// three-operand form, which the toolchain encodes with the store opcode:
// reg = the Plan 9 first operand, vvvv = the second, rm = the third.
func (e *enc) encodeVexMove(mnem string, ms vexMoveSpec, ops []Operand) error {
if len(ops) == 3 {
if !ms.xmmOnly {
return fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
op0, ok0 := ops[0].(Reg)
op1, ok1 := ops[1].(Reg)
op2, ok2 := ops[2].(Reg)
if !ok0 || !ok1 || !ok2 || !op0.isVec() || !op1.isVec() || !op2.isVec() {
return fmt.Errorf("%s three-operand form takes three vector registers", mnem)
}
if ms.xmmOnly && (op0.size != 16 || op1.size != 16 || op2.size != 16) {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
rBit := 0
if op0.idx >= 8 {
rBit = 1
}
spec := vexSpec{mapSel: ms.mapSel, opcode: ms.store, w: ms.storeW, pp: ms.pp, opdigit: -1}
return e.emitVexFields(spec, 0, op0.idx&7, rBit, 15-(op1.idx&15), op2)
}
if len(ops) != 2 {
return fmt.Errorf("VEX move expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcIsVec := vecReg(src)
dstReg, dstIsVec := vecReg(dst)
var reg Reg
var rm Operand
op, w := ms.store, ms.storeW
switch {
case srcIsVec && dstIsVec:
if !ms.vecOK {
return fmt.Errorf("%s does not take two vector registers", mnem)
}
if ms.xmmOnly && (srcReg.size == 32 || dstReg.size == 32) {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
if ms.regReg != 0 {
op, w = ms.regReg, ms.regW
}
reg, rm = srcReg, dst // store form: reg = source, rm = destination.
case srcIsVec:
// vector → memory, or → GPR (VMOVD/VMOVQ only).
if !validMoveOther(ms, dst) {
return fmt.Errorf("%s: invalid destination operand", mnem)
}
reg, rm = srcReg, dst
case dstIsVec:
// memory → vector, or GPR → vector (VMOVD/VMOVQ only).
if !validMoveOther(ms, src) {
return fmt.Errorf("%s: invalid source operand", mnem)
}
op, w = ms.load, ms.loadW
reg, rm = dstReg, src
default:
return fmt.Errorf("%s needs a vector register operand", mnem)
}
if ms.xmmOnly && reg.size == 32 {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
regField := reg.idx & 7
rBit := 0
if reg.idx >= 8 {
rBit = 1
}
spec := vexSpec{mapSel: ms.mapSel, opcode: op, w: w, pp: ms.pp, opdigit: -1, vex3: vex3Only[mnem]}
return e.emitVexFields(spec, reg.vecLenBit(), regField, rBit, 15, rm)
}
// vecReg extracts a vector register from an operand.
func vecReg(op Operand) (Reg, bool) {
r, ok := op.(Reg)
return r, ok && r.isVec()
}
// vecOrMem reports whether op is a vector register or a memory reference.
func vecOrMem(op Operand) bool {
switch op.(type) {
case Mem, sbMem:
return true
}
r, ok := op.(Reg)
return ok && r.isVec()
}
// validMoveOther reports whether the non-vector operand of a move is
// acceptable: memory always is, a GPR only for VMOVD/VMOVQ.
func validMoveOther(ms vexMoveSpec, op Operand) bool {
switch o := op.(type) {
case Mem, sbMem:
return true
case Reg:
return ms.gprOK && !o.isVec()
}
return false
}
// emitVexFields emits the VEX prefix, opcode, ModR/M, SIB and displacement for
// the given precomputed fields. It is shared by every register/rm VEX form;
// immediate bytes are appended by the caller.
func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Operand) error {
if l > 1 {
return fmt.Errorf("ZMM operand requires an EVEX instruction")
}
var modrm, sib int
var disp []byte
var xBit, bBit int
var sb *sbRef
switch r := rm.(type) {
case Reg:
modrm = 0xC0 | regField<<3 | (r.idx & 7)
sib = -1
if r.idx >= 8 {
bBit = 1
}
case Mem:
var err error
modrm, sib, disp, xBit, bBit, err = memComponents(regField, r)
if err != nil {
return err
}
case sbMem:
// RIP-relative static-symbol reference; disp32 patched at link time.
modrm = regField<<3 | 0x05
sib = -1
disp = le32(0)
sb = &sbRef{name: r.name, addend: r.addend}
default:
return fmt.Errorf("invalid VEX r/m operand")
}
if !spec.vex3 && spec.mapSel == 1 && xBit == 0 && bBit == 0 && spec.w == 0 {
e.out = append(e.out, 0xC5, byte((1-rBit)<<7|vvvvBar<<3|l<<2|spec.pp))
} else {
e.out = append(e.out, 0xC4,
byte((1-rBit)<<7|(1-xBit)<<6|(1-bBit)<<5|spec.mapSel),
byte(spec.w<<7|vvvvBar<<3|l<<2|spec.pp))
}
e.out = append(e.out, spec.opcode, byte(modrm))
if sib >= 0 {
e.out = append(e.out, byte(sib))
}
if sb != nil {
e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend})
}
e.out = append(e.out, disp...)
return nil
}