Files
gasm-sdk/asm/vex.go
T
2026-09-20 06:44:51 +02:00

888 lines
32 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "fmt"
// This file implements VEX (AVX/AVX2) instruction encoding. EVEX (AVX-512)
// support is a later increment.
//
// Every encoding choice here is validated two ways in the tests: by
// round-trip decoding through golang.org/x/arch's x86 decoder, and by
// byte-for-byte comparison against the output of the real Go assembler.
// vexForm selects how an instruction's operands map onto the VEX.vvvv,
// ModRM.reg and ModRM.rm fields.
type vexForm int
const (
// vexNDS3 is the three-operand form `OP src2, src1, dst` (Plan 9 order):
// ModRM.reg = dst (op2), VEX.vvvv = src1 (op1), ModRM.rm = src2 (op0).
vexNDS3 vexForm = iota
// vexRM is the two-operand form `OP src, dst` with no vvvv source:
// ModRM.reg = dst (op1), ModRM.rm = src (op0), VEX.vvvv unused.
vexRM
// vexShiftImm is the immediate-shift form `OP $imm, src, dst`: ModRM.reg =
// /digit, ModRM.rm = src (op1), VEX.vvvv = dst (op2), imm8 = op0.
vexShiftImm
// vexImmRM is the immediate form `OP $imm, src, dst` with no vvvv source:
// ModRM.reg = dst (op2), ModRM.rm = src (op1), imm8 = op0. VPSHUFD and
// VPERMQ use this shape.
vexImmRM
// vexNDS3Imm is the three-operand plus immediate form `OP $imm, src2,
// src1, dst`: ModRM.reg = dst, VEX.vvvv = src1, ModRM.rm = src2, imm8.
// VSHUFPD, VPERM2I128 and VINSERTI128 use this shape.
vexNDS3Imm
// vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg =
// ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM
// source lives in the reg field, the destination in r/m, the PEXTR-style
// layout. VEXTRACTI128 and VEXTRACTF128 use this shape.
vexExtract
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
// in ModRM.reg and the destination in r/m, the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ.
vexRMRev
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
// vector length follows the source: the packed-double → dword
// conversions (VCVTPD2DQ/VCVTTPD2DQ and their X/Y spellings) narrow into
// an XMM destination, so the L bit rides with the wider source. The
// mnemonic's spelling fixes the length (X = 128, Y = 256), which also
// covers a memory source. ModRM.reg = dst, ModRM.rm = src, no vvvv.
vexRMSrcLen
// vexZero is the no-operand form (VZEROUPPER).
vexZero
// vexZeroAll is the no-operand form that zeroes the full upper state
// (VZEROALL, the L = 1 twin of VZEROUPPER).
vexZeroAll
// vexNDS3GPR is the three-operand NDS form over general-purpose
// registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0.
vexNDS3GPR
// vexImmRMGPR is the immediate form over general-purpose registers
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
vexImmRMGPR
)
// vexSpec describes one VEX instruction's encoding parameters.
type vexSpec struct {
mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A
opcode byte
w int // VEX.W (0 for WIG)
pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2
opdigit int // ModRM.reg /digit, or -1 when reg is a register
form vexForm
}
// vexTable maps an upper-case mnemonic to its VEX encoding. It is extended
// incrementally; every entry is covered by a byte-for-byte ground-truth test
// against the Go assembler.
var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG, integer arithmetic / logic / compare.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3},
"VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3},
"VPSUBQ": {1, 0xFB, 0, 1, -1, vexNDS3},
"VPXOR": {1, 0xEF, 0, 1, -1, vexNDS3},
"VPOR": {1, 0xEB, 0, 1, -1, vexNDS3},
"VPAND": {1, 0xDB, 0, 1, -1, vexNDS3},
"VPANDN": {1, 0xDF, 0, 1, -1, vexNDS3},
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3},
"VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3},
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3},
"VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3},
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3},
// VEX.256.66.0F38.W0, dword permute (three-operand NDS form).
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG.
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3},
"VPMULDQ": {2, 0x28, 0, 1, -1, vexNDS3},
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3},
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F.WIG, packed double-precision arithmetic / logic.
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG, packed single-precision arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3},
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3},
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3},
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
// VEX.128.F2.0F.WIG, scalar double-precision arithmetic (the packed
// opcodes with an F2 pp).
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
"VMULSD": {1, 0x59, 0, 3, -1, vexNDS3},
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
// VEX.128.F3.0F.WIG, scalar single-precision arithmetic (the packed
// opcodes with an F3 pp).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3},
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
// Scalar fused multiply-add (NDS form). The Go assembler carries the
// same 66 prefix as the packed forms on every FMA row, and W1 on the
// double-precision spellings, so SD shares PD's prefix/W pair and the
// scalar width rides on the W bit.
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3},
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
// no vvvv).
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM},
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM},
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM},
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM},
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM},
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM},
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
"VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM},
"VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM},
// VEX.128/256.F3.0F.WIG, signed dword to packed double conversion
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
// VEX.128/256.0F.WIG, signed dword to packed single conversion
// (reg=dst, rm=src, no vvvv, no mandatory prefix).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
// VEX.128/256.0F.WIG, packed single to packed double conversion
// (reg=dst, rm=src; the destination is the wide operand and sets the
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
// the Go assembler emits the instruction with pp = 00, and gasm follows
// the Go assembler's bytes, its machine code is the oracle, not the
// manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
// VEX.128.F2.0F.WIG, duplicate the low double of each 128-bit lane
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
// VEX.128/256.66.0F.WIG, move mask to a GPR (reg=gpr dst, rm=vec src).
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
// VEX.128/256.66.0F.WIG, immediate shifts (opdigit selects the shift).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm},
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm},
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm},
"VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm},
"VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm},
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8).
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// imm8).
"VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm},
// VEX.256.66.0F3A.W0, permute / insert (same shape; VINSERTI128's rm is
// the XMM or memory source).
"VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm},
"VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm},
// VEX.256.66.0F3A.W0, lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
// VEX.128/256.66.0F3A.W0, half-precision convert back ($imm, src, dst:
// reg=src, rm=XMM/memory dst, imm8, the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
// VEX.128.0F.W0, no operands.
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
// VEX.256.0F.W0, zero all vector registers (the L = 1 twin).
"VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll},
// VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare.
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm},
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm},
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG, packed single XOR (NDS form).
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3},
// VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control.
"VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm},
"VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling
// falls through to the EVEX table.
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst).
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8).
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm},
// BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR).
"ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR},
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
// VEX.66.0F38.W0, broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; SD is 256-bit only).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
// VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a
// YMM (the encoder rejects an XMM destination, as go tool asm does).
"VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM},
// VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg,
// memory destination in rm).
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev},
// VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv).
"VPTEST": {2, 0x17, 0, 1, -1, vexRM},
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
// source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
// VEX.F3.0F.WIG, replicate even/odd singles (reg=dst, rm=src).
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
// VEX.66.0F.WIG, packed double to packed single conversion, the X/Y
// spellings: the destination is always XMM and the spelling fixes the
// source length (X = 128, Y = 256).
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
// VEX scalar conversions between vector and general-purpose registers.
// Vector to GPR (two operands: vec/mem source, GPR destination, vvvv
// unused; the length follows the source).
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM},
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM},
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM},
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM},
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM},
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM},
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM},
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM},
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
// vector source in vvvv, vector destination in reg).
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3},
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3},
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
// VEX.128/256.66.0F.WIG, word shifts (opdigit selects the shift).
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
// VEX.F2.0F, packed double to packed dword conversions, truncating and
// non-truncating. The destination is always XMM; the X/Y spellings fix
// the source length (XMM/YMM), and VEX.L follows it, see vexSrcLen.
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
}
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
// the packed-double → dword conversions) to its fixed vector length:
// X = 128 (L = 0), Y = 256 (L = 1). The spelling fixes the length even for
// a memory source, matching the Go assembler's ytab.
var vexSrcLen = map[string]int{
"VCVTPD2DQX": 0,
"VCVTPD2DQY": 1,
"VCVTTPD2DQX": 0,
"VCVTTPD2DQY": 1,
"VCVTPD2PSX": 0,
"VCVTPD2PSY": 1,
}
// vexVarShift maps the shift mnemonics to their variable-count opcode, the
// form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8),
// an ordinary NDS encoding rather than the /digit immediate form above.
var vexVarShift = map[string]byte{
"VPSLLD": 0xF2,
"VPSLLQ": 0xF3,
"VPSRAD": 0xE2,
"VPSRLD": 0xD2,
"VPSRLQ": 0xD3,
}
// vexMoveSpec describes a VEX move, which takes different opcodes (and
// sometimes a different VEX.W) per operand direction. The Go assembler
// encodes a vector→vector move with the store-form opcode (reg = source,
// rm = destination), so regReg defaults to store when zero.
type vexMoveSpec struct {
mapSel int
pp int
load byte // r/m → vector: reg=dst, rm=src
store byte // vector → r/m: reg=src, rm=dst
loadW int
storeW int
regReg byte // vector → vector opcode; 0 uses store
regW int
vecOK bool // the non-fixed operand may be a vector register
gprOK bool // the non-fixed operand may be a general-purpose register
xmmOnly bool // YMM registers are rejected
}
// vexMoveTable maps an upper-case move mnemonic to its encoding.
var vexMoveTable = map[string]vexMoveSpec{
// VEX.128/256.F3.0F.WIG, unaligned integer move.
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG, aligned integer move.
"VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
"VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true},
// VMOVQ, 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true},
// VEX.128.F2.0F.WIG, scalar double move, memory operands only (the
// register form takes three operands and is not supported yet).
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128.F3.0F.WIG, scalar single move, memory operands only.
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128/256, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
}
// isVex reports whether the mnemonic is a VEX-encoded instruction we handle.
func isVex(mnemUpper string) bool {
if _, ok := vexTable[mnemUpper]; ok {
return true
}
_, ok := vexMoveTable[mnemUpper]
return ok
}
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
// Vector register indices 16-31 exist only in EVEX encodings; fail
// loudly rather than silently truncating the index.
for _, op := range ops {
if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 {
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
}
}
// VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination
// only; an XMM destination is rejected exactly as go tool asm does.
if mnemUpper == "VBROADCASTI128" {
dstReg, ok := ops[len(ops)-1].(Reg)
if len(ops) != 2 || !ok || dstReg.size != 32 {
return fmt.Errorf("VBROADCASTI128 requires a YMM destination")
}
}
if ms, ok := vexMoveTable[mnemUpper]; ok {
return e.encodeVexMove(mnemUpper, ms, ops)
}
// The shifts come in two shapes under one mnemonic: an immediate count
// ($imm, src, dst) and a variable count in an XMM register or memory
// (count, src, dst), the latter an ordinary NDS form.
if op, ok := vexVarShift[mnemUpper]; ok && len(ops) == 3 {
if _, isImm := ops[0].(Imm); !isImm {
if !vecOrMem(ops[0]) {
return fmt.Errorf("%s: shift count must be an immediate, a vector register or memory", mnemUpper)
}
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
}
}
spec := vexTable[mnemUpper]
switch spec.form {
case vexNDS3:
return e.encodeVexNDS3(spec, ops)
case vexRM:
return e.encodeVexRM(spec, ops)
case vexShiftImm:
return e.encodeVexShiftImm(spec, ops)
case vexImmRM:
return e.encodeVexImmRM(spec, ops)
case vexNDS3Imm:
return e.encodeVexNDS3Imm(spec, ops)
case vexExtract:
return e.encodeVexExtract(spec, ops)
case vexRMSrcLen:
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero:
return e.encodeVexZero(mnemUpper, spec, ops)
case vexZeroAll:
return e.encodeVexZeroAll(mnemUpper, spec, ops)
case vexNDS3GPR:
return e.encodeVexNDS3GPR(spec, ops)
case vexImmRMGPR:
return e.encodeVexImmRMGPR(spec, ops)
case vexRMRev:
return e.encodeVexRMRev(spec, ops)
}
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
}
// encodeVexNDS3 encodes the three-operand NDS form: OP src2, src1, dst.
func (e *enc) encodeVexNDS3(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
}
src2, src1, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("VEX destination must be a vector register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("VEX vvvv operand must be a vector register")
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
vvvvBar := 15 - (vvvvReg.idx & 15)
return e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2)
}
// encodeVexRM encodes the two-operand form: OP src, dst (no vvvv source).
// ModRM.reg = dst, ModRM.rm = src; the vector length comes from whichever
// operand is a vector register (the destination for extends/broadcasts, the
// source for the move-mask instructions whose destination is a GPR).
func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("VEX two-operand instruction expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok {
return fmt.Errorf("VEX destination must be a register")
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
// Vector length: from the destination if it is a vector, otherwise from the
// source (move-mask instructions have a GPR destination and a vector source).
l := 0
if dstReg.isVec() {
l = dstReg.vecLenBit()
} else if srcReg, ok := src.(Reg); ok && srcReg.isVec() {
l = srcReg.vecLenBit()
}
// An unused vvvv field must be stored as all ones (v̄vvv = 1111); the
// hardware raises #UD on any other value.
return e.emitVexFields(spec, l, regField, rBit, 15, src)
}
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the VEX.L bit following the source, fixed
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
// the source is memory.
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("VEX destination must be a vector register")
}
ll, ok := vexSrcLen[mnem]
if !ok {
return fmt.Errorf("no fixed vector length for %s", mnem)
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
// An unused vvvv field must be stored as all ones (v̄vvv = 1111).
return e.emitVexFields(spec, ll, regField, rBit, 15, src)
}
// encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst.
// The destination is carried in VEX.vvvv, the source in ModRM.rm, and the
// shift kind in the ModRM.reg /digit.
func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shift count must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("shift source must be a vector register")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("shift destination must be a vector register")
}
vvvvBar := 15 - (dstReg.idx & 15)
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, srcReg); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// imm8 range-checks an immediate for an 8-bit field. Shuffle controls are
// unsigned bit masks, but the negative spelling ($-1 = all bits set) is
// accepted, so the accepted span is -128..255.
func imm8(v int64) (byte, error) {
if v < -128 || v > 255 {
return 0, fmt.Errorf("immediate $%d does not fit in 8 bits", v)
}
return byte(v), nil
}
// encodeVexImmRM encodes an immediate form with no vvvv source: OP $imm, src,
// dst (VPSHUFD, VPERMQ). ModRM.reg = dst, ModRM.rm = src, imm8 appended.
func (e *enc) encodeVexImmRM(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shuffle control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("shuffle destination must be a vector register")
}
// The vector length follows the source when it is a vector register,
// otherwise the destination (a memory source carries no length).
l := dstReg.vecLenBit()
if srcReg, ok := src.(Reg); ok && srcReg.isVec() {
l = srcReg.vecLenBit()
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, l, regField, rBit, 15, src); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexNDS3Imm encodes the three-operand plus immediate form: OP $imm,
// src2, src1, dst (VSHUFPD, VPERM2I128, VINSERTI128). ModRM.reg = dst,
// VEX.vvvv = src1, ModRM.rm = src2, imm8 appended.
func (e *enc) encodeVexNDS3Imm(spec vexSpec, ops []Operand) error {
if len(ops) != 4 {
return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops))
}
imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shuffle control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("destination must be a vector register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("second source must be a vector register")
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
vvvvBar := 15 - (vvvvReg.idx & 15)
if err := e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexExtract encodes a lane extract: OP $imm, ysrc, xdst
// (VEXTRACTI128, VEXTRACTF128). The YMM source occupies ModRM.reg and the
// XMM (or memory) destination ModRM.rm; imm8 selects the lane.
func (e *enc) encodeVexExtract(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, ysrc, xdst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("extract source must be a vector register")
}
regField := srcReg.idx & 7
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, srcReg.vecLenBit(), regField, rBit, 15, dst); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexZero encodes a no-operand instruction (VZEROUPPER).
func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 0 {
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
}
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 0.
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|spec.pp), spec.opcode)
return nil
}
// encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1
// twin of VZEROUPPER.
func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 0 {
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
}
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1.
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode)
return nil
}
// encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose
// registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1,
// rm = src2 and L = 0.
func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
}
src2, src1, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
vvvvReg, ok := src1.(Reg)
if !ok || vvvvReg.isVec() {
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
}
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15-(vvvvReg.idx&15), src2)
}
// encodeVexImmRMGPR encodes the immediate form over general-purpose
// registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0.
func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shift control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
// a store with no register-destination form).
func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("store expects 2 operands, got %d", len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("store source must be a vector register")
}
if !memOperand(ops[1]) {
return fmt.Errorf("store destination must be memory")
}
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
}
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
// move uses the store-form layout (reg = source, rm = destination), matching
// the Go assembler.
func (e *enc) encodeVexMove(mnem string, ms vexMoveSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("VEX move expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcIsVec := vecReg(src)
dstReg, dstIsVec := vecReg(dst)
var reg Reg
var rm Operand
op, w := ms.store, ms.storeW
switch {
case srcIsVec && dstIsVec:
if !ms.vecOK {
return fmt.Errorf("%s does not take two vector registers", mnem)
}
if ms.xmmOnly && (srcReg.size == 32 || dstReg.size == 32) {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
if ms.regReg != 0 {
op, w = ms.regReg, ms.regW
}
reg, rm = srcReg, dst // store form: reg = source, rm = destination.
case srcIsVec:
// vector → memory, or → GPR (VMOVD/VMOVQ only).
if !validMoveOther(ms, dst) {
return fmt.Errorf("%s: invalid destination operand", mnem)
}
reg, rm = srcReg, dst
case dstIsVec:
// memory → vector, or GPR → vector (VMOVD/VMOVQ only).
if !validMoveOther(ms, src) {
return fmt.Errorf("%s: invalid source operand", mnem)
}
op, w = ms.load, ms.loadW
reg, rm = dstReg, src
default:
return fmt.Errorf("%s needs a vector register operand", mnem)
}
if ms.xmmOnly && reg.size == 32 {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
regField := reg.idx & 7
rBit := 0
if reg.idx >= 8 {
rBit = 1
}
spec := vexSpec{mapSel: ms.mapSel, opcode: op, w: w, pp: ms.pp, opdigit: -1}
return e.emitVexFields(spec, reg.vecLenBit(), regField, rBit, 15, rm)
}
// vecReg extracts a vector register from an operand.
func vecReg(op Operand) (Reg, bool) {
r, ok := op.(Reg)
return r, ok && r.isVec()
}
// vecOrMem reports whether op is a vector register or a memory reference.
func vecOrMem(op Operand) bool {
switch op.(type) {
case Mem, sbMem:
return true
}
r, ok := op.(Reg)
return ok && r.isVec()
}
// validMoveOther reports whether the non-vector operand of a move is
// acceptable: memory always is, a GPR only for VMOVD/VMOVQ.
func validMoveOther(ms vexMoveSpec, op Operand) bool {
switch o := op.(type) {
case Mem, sbMem:
return true
case Reg:
return ms.gprOK && !o.isVec()
}
return false
}
// emitVexFields emits the VEX prefix, opcode, ModR/M, SIB and displacement for
// the given precomputed fields. It is shared by every register/rm VEX form;
// immediate bytes are appended by the caller.
func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Operand) error {
if l > 1 {
return fmt.Errorf("ZMM operand requires an EVEX instruction")
}
var modrm, sib int
var disp []byte
var xBit, bBit int
var sb *sbRef
switch r := rm.(type) {
case Reg:
modrm = 0xC0 | regField<<3 | (r.idx & 7)
sib = -1
if r.idx >= 8 {
bBit = 1
}
case Mem:
var err error
modrm, sib, disp, xBit, bBit, err = memComponents(regField, r)
if err != nil {
return err
}
case sbMem:
// RIP-relative static-symbol reference; disp32 patched at link time.
modrm = regField<<3 | 0x05
sib = -1
disp = le32(0)
sb = &sbRef{name: r.name, addend: r.addend}
default:
return fmt.Errorf("invalid VEX r/m operand")
}
if spec.mapSel == 1 && xBit == 0 && bBit == 0 && spec.w == 0 {
e.out = append(e.out, 0xC5, byte((1-rBit)<<7|vvvvBar<<3|l<<2|spec.pp))
} else {
e.out = append(e.out, 0xC4,
byte((1-rBit)<<7|(1-xBit)<<6|(1-bBit)<<5|spec.mapSel),
byte(spec.w<<7|vvvvBar<<3|l<<2|spec.pp))
}
e.out = append(e.out, spec.opcode, byte(modrm))
if sib >= 0 {
e.out = append(e.out, byte(sib))
}
if sb != nil {
e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend})
}
e.out = append(e.out, disp...)
return nil
}