The SSE3 horizontal and add-subtract pairs, the SSSE3 sign and horizontal integers, the masked moves in both directions, the reciprocity and test pairs, the AVX imm8 tail (blends, dot products, inserts, rounds, MPSADBW, the string compares), the four-operand variable blends with their /is4 mask byte, the scalar three-operand moves, the MXCSR accessors, the VPERMIL register controls and the variable word shifts, plus the BMI2 count forms over memory. Every encoding is pinned byte for byte against go tool asm through every corpus line the toolchain's own amd64enc.s carries for the families (852 lines); the /is4 byte carries the mask register number in its high nibble, the layout the toolchain emits. Assisted-by: GLM 5.3 Flash
1405 lines
54 KiB
Go
1405 lines
54 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
package asm
|
|
|
|
import "fmt"
|
|
|
|
// This file implements VEX (AVX/AVX2) instruction encoding. EVEX (AVX-512)
|
|
// support is a later increment.
|
|
//
|
|
// Every encoding choice here is validated two ways in the tests: by
|
|
// round-trip decoding through golang.org/x/arch's x86 decoder, and by
|
|
// byte-for-byte comparison against the output of the real Go assembler.
|
|
|
|
// vexForm selects how an instruction's operands map onto the VEX.vvvv,
|
|
// ModRM.reg and ModRM.rm fields.
|
|
type vexForm int
|
|
|
|
const (
|
|
// vexNDS3 is the three-operand form `OP src2, src1, dst` (Plan 9 order):
|
|
// ModRM.reg = dst (op2), VEX.vvvv = src1 (op1), ModRM.rm = src2 (op0).
|
|
vexNDS3 vexForm = iota
|
|
// vexRM is the two-operand form `OP src, dst` with no vvvv source:
|
|
// ModRM.reg = dst (op1), ModRM.rm = src (op0), VEX.vvvv unused.
|
|
vexRM
|
|
// vexShiftImm is the immediate-shift form `OP $imm, src, dst`: ModRM.reg =
|
|
// /digit, ModRM.rm = src (op1), VEX.vvvv = dst (op2), imm8 = op0.
|
|
vexShiftImm
|
|
// vexImmRM is the immediate form `OP $imm, src, dst` with no vvvv source:
|
|
// ModRM.reg = dst (op2), ModRM.rm = src (op1), imm8 = op0. VPSHUFD and
|
|
// VPERMQ use this shape.
|
|
vexImmRM
|
|
// vexNDS3Imm is the three-operand plus immediate form `OP $imm, src2,
|
|
// src1, dst`: ModRM.reg = dst, VEX.vvvv = src1, ModRM.rm = src2, imm8.
|
|
// VSHUFPD, VPERM2I128 and VINSERTI128 use this shape.
|
|
vexNDS3Imm
|
|
// vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg =
|
|
// ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM
|
|
// source lives in the reg field, the destination in r/m, the PEXTR-style
|
|
// layout. VEXTRACTI128 and VEXTRACTF128 use this shape.
|
|
vexExtract
|
|
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
|
// in ModRM.reg and the destination in r/m, the layout of the EVEX
|
|
// narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ.
|
|
vexRMRev
|
|
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
|
// vector length follows the source: the packed-double → dword
|
|
// conversions (VCVTPD2DQ/VCVTTPD2DQ and their X/Y spellings) narrow into
|
|
// an XMM destination, so the L bit rides with the wider source. The
|
|
// mnemonic's spelling fixes the length (X = 128, Y = 256), which also
|
|
// covers a memory source. ModRM.reg = dst, ModRM.rm = src, no vvvv.
|
|
vexRMSrcLen
|
|
// vexZero is the no-operand form (VZEROUPPER).
|
|
vexZero
|
|
// vexZeroAll is the no-operand form that zeroes the full upper state
|
|
// (VZEROALL, the L = 1 twin of VZEROUPPER).
|
|
vexZeroAll
|
|
// vexNDS3GPR is the three-operand NDS form over general-purpose
|
|
// registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0.
|
|
vexNDS3GPR
|
|
// vexImmRMGPR is the immediate form over general-purpose registers
|
|
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
|
|
vexImmRMGPR
|
|
// vexRMOpGPR is the two-operand /digit form over general-purpose
|
|
// registers (BLSI, BLSMSK, BLSR): ModRM.reg = /digit, ModRM.rm = src
|
|
// (op0), VEX.vvvv = dst (op1), L = 0.
|
|
vexRMOpGPR
|
|
// vexCountGPR is the three-operand count form over general-purpose
|
|
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): the first operand rides
|
|
// VEX.vvvv and the second is r/m, the opposite pairing of the ANDN
|
|
// family, with reg = dst (op2), L = 0.
|
|
vexCountGPR
|
|
// vexExtractGPR is the lane-extract-to-GPR form `OP $imm, xsrc, GPR/mem
|
|
// dst`: ModRM.reg = xsrc (op1), ModRM.rm = destination (op2), imm8 =
|
|
// op0, the VPEXTRB/W/D/Q layout. EVEX only; the destination never
|
|
// carries a vector length, so the register the L'L field follows is the
|
|
// XMM source.
|
|
vexExtractGPR
|
|
// vexBlend4 is the four-operand variable blend `OP mask, src2, src1,
|
|
// dst` (VPBLENDVB): ModRM.reg = dst (op3), VEX.vvvv = src1 (op2),
|
|
// ModRM.rm = src2 (op1) and the mask register in the /is4 byte (op0).
|
|
vexBlend4
|
|
// vexNDS3Dst is the destination-first NDS form `OP dst, src1, src2`
|
|
// (VMASKMOVPS, VPMASKMOVD): ModRM.reg = dst (op0), VEX.vvvv = the mask
|
|
// source (op1), ModRM.rm = memory (op2).
|
|
vexNDS3Dst
|
|
// vexRMOpDigit is the two-operand /digit form over memory (VLDMXCSR,
|
|
// VSTMXCSR): ModRM.reg = /digit, ModRM.rm = the memory operand.
|
|
vexRMOpDigit
|
|
)
|
|
|
|
// vexSpec describes one VEX instruction's encoding parameters.
|
|
type vexSpec struct {
|
|
mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A
|
|
opcode byte
|
|
w int // VEX.W (0 for WIG)
|
|
pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2
|
|
opdigit int // ModRM.reg /digit, or -1 when reg is a register
|
|
form vexForm
|
|
}
|
|
|
|
// vexTable maps an upper-case mnemonic to its VEX encoding. It is extended
|
|
// incrementally; every entry is covered by a byte-for-byte ground-truth test
|
|
// against the Go assembler.
|
|
var vexTable = map[string]vexSpec{
|
|
// VEX.128/256.66.0F.WIG, integer arithmetic / logic / compare.
|
|
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3},
|
|
"VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3},
|
|
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3},
|
|
"VPSUBQ": {1, 0xFB, 0, 1, -1, vexNDS3},
|
|
"VPXOR": {1, 0xEF, 0, 1, -1, vexNDS3},
|
|
"VPOR": {1, 0xEB, 0, 1, -1, vexNDS3},
|
|
"VPAND": {1, 0xDB, 0, 1, -1, vexNDS3},
|
|
"VPANDN": {1, 0xDF, 0, 1, -1, vexNDS3},
|
|
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3},
|
|
"VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3},
|
|
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3},
|
|
"VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3},
|
|
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3},
|
|
// VEX.256.66.0F38.W0, dword permute (three-operand NDS form).
|
|
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3},
|
|
// VEX.128/256.66.0F38.WIG.
|
|
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3},
|
|
"VPMULDQ": {2, 0x28, 0, 1, -1, vexNDS3},
|
|
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3},
|
|
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
|
|
|
|
// VEX.128/256.66.0F.WIG, packed double-precision arithmetic / logic.
|
|
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
|
|
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
|
|
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
|
|
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
|
|
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
|
|
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
|
|
// VEX.128/256.0F.WIG, packed single-precision arithmetic.
|
|
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
|
|
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
|
|
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
|
|
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3},
|
|
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3},
|
|
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3},
|
|
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
|
|
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
|
|
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
|
|
// VEX.128.F2.0F.WIG, scalar double-precision arithmetic (the packed
|
|
// opcodes with an F2 pp).
|
|
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
|
|
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
|
|
"VMULSD": {1, 0x59, 0, 3, -1, vexNDS3},
|
|
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
|
|
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
|
|
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
|
|
// VEX.128.F3.0F.WIG, scalar single-precision arithmetic (the packed
|
|
// opcodes with an F3 pp).
|
|
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
|
|
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
|
|
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3},
|
|
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
|
|
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
|
|
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
|
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
|
|
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
|
// Scalar fused multiply-add (NDS form). The Go assembler carries the
|
|
// same 66 prefix as the packed forms on every FMA row, and W1 on the
|
|
// double-precision spellings, so SD shares PD's prefix/W pair and the
|
|
// scalar width rides on the W bit.
|
|
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3},
|
|
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3},
|
|
|
|
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
|
|
// no vvvv).
|
|
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
|
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
|
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM},
|
|
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM},
|
|
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM},
|
|
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
|
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
|
|
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM},
|
|
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM},
|
|
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM},
|
|
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
|
|
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
|
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
|
"VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM},
|
|
"VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM},
|
|
// VEX.128/256.F3.0F.WIG, signed dword to packed double conversion
|
|
// (reg=dst, rm=src, no vvvv; the length follows the destination).
|
|
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
|
|
// VEX.128/256.0F.WIG, signed dword to packed single conversion
|
|
// (reg=dst, rm=src, no vvvv, no mandatory prefix).
|
|
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
|
|
// VEX.128/256.0F.WIG, packed single to packed double conversion
|
|
// (reg=dst, rm=src; the destination is the wide operand and sets the
|
|
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
|
|
// the Go assembler emits the instruction with pp = 00, and gasm follows
|
|
// the Go assembler's bytes, its machine code is the oracle, not the
|
|
// manual.
|
|
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
|
|
// VEX.128.F2.0F.WIG, duplicate the low double of each 128-bit lane
|
|
// (reg=dst, rm=src, no vvvv; the length follows the destination).
|
|
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
|
|
// VEX.128/256.66.0F.WIG, move mask to a GPR (reg=gpr dst, rm=vec src).
|
|
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
|
|
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
|
|
|
|
// VEX.128/256.66.0F.WIG, immediate shifts (opdigit selects the shift).
|
|
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm},
|
|
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm},
|
|
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm},
|
|
"VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm},
|
|
"VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm},
|
|
|
|
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
|
|
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
|
|
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8), and its
|
|
// double twin under op 01; the in-lane permutes under 04/05.
|
|
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
|
|
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM},
|
|
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM},
|
|
"VPERMILPD": {3, 0x05, 0, 1, -1, vexImmRM},
|
|
// VEX.66.0F3A.W0, the immediate-controlled AVX tail: the rounding
|
|
// pair, the AES key assistant and the string compares.
|
|
"VROUNDPD": {3, 0x09, 0, 1, -1, vexImmRM},
|
|
"VROUNDPS": {3, 0x08, 0, 1, -1, vexImmRM},
|
|
"VAESKEYGENASSIST": {3, 0xDF, 0, 1, -1, vexImmRM},
|
|
"VPCMPESTRI": {3, 0x61, 0, 1, -1, vexImmRM},
|
|
"VPCMPESTRM": {3, 0x60, 0, 1, -1, vexImmRM},
|
|
"VPCMPISTRI": {3, 0x63, 0, 1, -1, vexImmRM},
|
|
"VPCMPISTRM": {3, 0x62, 0, 1, -1, vexImmRM},
|
|
// VEX.128.66.0F3A.W0, the scalar lane extract to a GPR or memory
|
|
// (reg = the XMM source, r/m = the destination).
|
|
"VEXTRACTPS": {3, 0x17, 0, 1, -1, vexExtractGPR},
|
|
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR},
|
|
// VEX.128.66.0F3A.W0, the four-operand variable blend with its mask
|
|
// register in the /is4 byte.
|
|
"VPBLENDVB": {3, 0x4C, 0, 1, -1, vexBlend4},
|
|
|
|
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
|
|
// imm8).
|
|
"VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm},
|
|
// VEX.256.66.0F3A.W0, permute / insert (same shape; VINSERTI128's rm is
|
|
// the XMM or memory source).
|
|
"VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm},
|
|
"VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm},
|
|
|
|
// VEX.256.66.0F3A.W0, lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
|
|
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
|
|
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
|
|
// VEX.128/256.66.0F3A.W0, half-precision convert back ($imm, src, dst:
|
|
// reg=src, rm=XMM/memory dst, imm8, the extract layout).
|
|
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
|
|
|
|
// VEX.128.0F.W0, no operands.
|
|
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
|
// VEX.256.0F.W0, zero all vector registers (the L = 1 twin).
|
|
"VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll},
|
|
// VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare.
|
|
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm},
|
|
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm},
|
|
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3},
|
|
// VEX.128/256.0F.WIG, packed single XOR (NDS form).
|
|
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3},
|
|
// VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control.
|
|
"VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm},
|
|
"VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm},
|
|
// VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling
|
|
// falls through to the EVEX table.
|
|
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm},
|
|
// VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst).
|
|
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm},
|
|
// VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8).
|
|
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm},
|
|
// BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR).
|
|
"ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR},
|
|
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
|
|
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
|
|
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
|
|
// VEX.NDS.LZ.0F38, the BMI2 three-operand bit ops: BEXTR and BZHI
|
|
// share the F7/F5 opcodes across W, the variable shifts carry their
|
|
// direction in the prefix (SHLX 66, SHRX F2, SARX F3) and PDEP/PEXT
|
|
// in F2/F3.
|
|
"BEXTRL": {2, 0xF7, 0, 0, -1, vexCountGPR},
|
|
"BEXTRQ": {2, 0xF7, 1, 0, -1, vexCountGPR},
|
|
"BZHIL": {2, 0xF5, 0, 0, -1, vexCountGPR},
|
|
"BZHIQ": {2, 0xF5, 1, 0, -1, vexCountGPR},
|
|
"SARXL": {2, 0xF7, 0, 2, -1, vexCountGPR},
|
|
"SARXQ": {2, 0xF7, 1, 2, -1, vexCountGPR},
|
|
"SHLXL": {2, 0xF7, 0, 1, -1, vexCountGPR},
|
|
"SHLXQ": {2, 0xF7, 1, 1, -1, vexCountGPR},
|
|
"SHRXL": {2, 0xF7, 0, 3, -1, vexCountGPR},
|
|
"SHRXQ": {2, 0xF7, 1, 3, -1, vexCountGPR},
|
|
"PDEPL": {2, 0xF5, 0, 3, -1, vexNDS3GPR},
|
|
"PDEPQ": {2, 0xF5, 1, 3, -1, vexNDS3GPR},
|
|
"PEXTL": {2, 0xF5, 0, 2, -1, vexNDS3GPR},
|
|
"PEXTQ": {2, 0xF5, 1, 2, -1, vexNDS3GPR},
|
|
// VEX.LZ.0F38.W, the BMI1 unary bit ops (src, dst: ModRM.reg = /digit,
|
|
// rm = src, vvvv = dst).
|
|
"BLSIL": {2, 0xF3, 0, 0, 3, vexRMOpGPR},
|
|
"BLSIQ": {2, 0xF3, 1, 0, 3, vexRMOpGPR},
|
|
"BLSMSKL": {2, 0xF3, 0, 0, 2, vexRMOpGPR},
|
|
"BLSMSKQ": {2, 0xF3, 1, 0, 2, vexRMOpGPR},
|
|
"BLSRL": {2, 0xF3, 0, 0, 1, vexRMOpGPR},
|
|
"BLSRQ": {2, 0xF3, 1, 0, 1, vexRMOpGPR},
|
|
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
|
|
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
|
|
|
|
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
|
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
|
|
|
// VEX.66.0F38.W0, broadcast a single/double to all lanes (reg=dst,
|
|
// rm=scalar memory; SD is 256-bit only).
|
|
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
|
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
|
// VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a
|
|
// YMM (the encoder rejects an XMM destination, as go tool asm does).
|
|
"VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM},
|
|
// VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg,
|
|
// memory destination in rm).
|
|
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev},
|
|
// VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv).
|
|
"VPTEST": {2, 0x17, 0, 1, -1, vexRM},
|
|
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
|
|
// source).
|
|
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
|
// VEX.F3.0F.WIG, replicate even/odd singles (reg=dst, rm=src).
|
|
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
|
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
|
// VEX.66.0F.WIG, packed double to packed single conversion, the X/Y
|
|
// spellings: the destination is always XMM and the spelling fixes the
|
|
// source length (X = 128, Y = 256).
|
|
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
|
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
|
|
|
// VEX scalar conversions between vector and general-purpose registers.
|
|
// Vector to GPR (two operands: vec/mem source, GPR destination, vvvv
|
|
// unused; the length follows the source).
|
|
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM},
|
|
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM},
|
|
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM},
|
|
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM},
|
|
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM},
|
|
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM},
|
|
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM},
|
|
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM},
|
|
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
|
// vector source in vvvv, vector destination in reg).
|
|
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3},
|
|
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3},
|
|
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
|
|
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
|
|
|
|
// VEX.128/256.66.0F.WIG, word shifts (opdigit selects the shift).
|
|
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
|
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
|
|
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
|
|
// VEX.F2.0F, packed double to packed dword conversions, truncating and
|
|
// non-truncating. The destination is always XMM; the X/Y spellings fix
|
|
// the source length (XMM/YMM), and VEX.L follows it, see vexSrcLen.
|
|
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
|
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
|
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
|
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
|
|
|
// --- the VEX forms the avx512enc corpus exercises alongside the EVEX
|
|
// spellings, read off the toolchain opcode tables ---
|
|
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3},
|
|
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3},
|
|
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3},
|
|
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3},
|
|
"VANDNPD": {1, 0x55, 0, 1, -1, vexNDS3},
|
|
"VANDPD": {1, 0x54, 0, 1, -1, vexNDS3},
|
|
"VCOMISD": {1, 0x2F, 0, 1, -1, vexRM},
|
|
"VCVTSD2SS": {1, 0x5A, 0, 3, -1, vexNDS3},
|
|
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3},
|
|
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3},
|
|
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3},
|
|
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3},
|
|
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3},
|
|
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3},
|
|
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3},
|
|
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3},
|
|
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3},
|
|
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3},
|
|
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3},
|
|
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3},
|
|
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3},
|
|
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3},
|
|
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3},
|
|
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3},
|
|
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3},
|
|
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3},
|
|
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3},
|
|
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3},
|
|
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3},
|
|
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3},
|
|
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3},
|
|
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3},
|
|
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3},
|
|
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3},
|
|
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3},
|
|
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3},
|
|
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3},
|
|
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3},
|
|
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3},
|
|
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3},
|
|
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3},
|
|
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3},
|
|
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3},
|
|
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3},
|
|
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3},
|
|
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3},
|
|
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3},
|
|
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3},
|
|
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3},
|
|
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3},
|
|
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3},
|
|
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3},
|
|
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3},
|
|
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3},
|
|
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3},
|
|
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3},
|
|
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3},
|
|
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3},
|
|
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3},
|
|
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3},
|
|
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3},
|
|
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3},
|
|
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3},
|
|
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3},
|
|
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3},
|
|
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3},
|
|
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm},
|
|
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3},
|
|
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM},
|
|
"VMOVNTPD": {1, 0x2B, 0, 1, -1, vexRMRev},
|
|
"VORPD": {1, 0x56, 0, 1, -1, vexNDS3},
|
|
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3},
|
|
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3},
|
|
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3},
|
|
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3},
|
|
"VPCMPEQQ": {2, 0x29, 0, 1, -1, vexNDS3},
|
|
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3},
|
|
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3},
|
|
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3},
|
|
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3},
|
|
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3},
|
|
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtract},
|
|
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtract},
|
|
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtract},
|
|
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm},
|
|
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm},
|
|
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3},
|
|
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3},
|
|
"VPMULUDQ": {1, 0xF4, 0, 1, -1, vexNDS3},
|
|
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3},
|
|
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3},
|
|
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3},
|
|
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3},
|
|
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3},
|
|
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3},
|
|
"VPUNPCKHQDQ": {1, 0x6D, 0, 1, -1, vexNDS3},
|
|
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3},
|
|
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3},
|
|
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3},
|
|
"VSQRTPD": {1, 0x51, 0, 1, -1, vexRM},
|
|
"VSQRTSD": {1, 0x51, 0, 3, -1, vexNDS3},
|
|
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3},
|
|
"VUCOMISD": {1, 0x2E, 0, 1, -1, vexRM},
|
|
|
|
// VEX.0F.WIG, the plain-prefix single/double arithmetic and unpack
|
|
// spellings (no 66 prefix; WIG, so W = 0).
|
|
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3},
|
|
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3},
|
|
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3},
|
|
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3},
|
|
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3},
|
|
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM},
|
|
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev},
|
|
// VEX.128.66.0F, the scalar and packed compare forms.
|
|
"VCOMISS": {1, 0x2F, 0, 1, -1, vexRM},
|
|
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM},
|
|
// VEX.128.0F.F3/F2.W0, the high/low word shuffles ($imm, src, dst).
|
|
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM},
|
|
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM},
|
|
|
|
// --- the corpus families from amd64enc.s: the SSE3 horizontal and
|
|
// add-subtract pairs, the SSSE3 sign and horizontal integers, the AVX
|
|
// reciprocity and test pairs, the masked and non-temporal oddities ---
|
|
|
|
// VEX.128/256, the horizontal and add-subtract float pairs: PD carries
|
|
// 66, PS carries F2 (0F 7C/7D and 0F D0).
|
|
"VHADDPD": {1, 0x7C, 0, 1, -1, vexNDS3},
|
|
"VHADDPS": {1, 0x7C, 0, 3, -1, vexNDS3},
|
|
"VHSUBPD": {1, 0x7D, 0, 1, -1, vexNDS3},
|
|
"VHSUBPS": {1, 0x7D, 0, 3, -1, vexNDS3},
|
|
"VADDSUBPD": {1, 0xD0, 0, 1, -1, vexNDS3},
|
|
"VADDSUBPS": {1, 0xD0, 0, 3, -1, vexNDS3},
|
|
// VEX.128/256.0F38.W0, the SSSE3 horizontal integer family and the sign
|
|
// controls, all NDS over 66.
|
|
"VPHADDW": {2, 0x01, 0, 1, -1, vexNDS3},
|
|
"VPHADDD": {2, 0x02, 0, 1, -1, vexNDS3},
|
|
"VPHADDSW": {2, 0x03, 0, 1, -1, vexNDS3},
|
|
"VPHSUBW": {2, 0x05, 0, 1, -1, vexNDS3},
|
|
"VPHSUBD": {2, 0x06, 0, 1, -1, vexNDS3},
|
|
"VPHSUBSW": {2, 0x07, 0, 1, -1, vexNDS3},
|
|
"VPSIGNB": {2, 0x08, 0, 1, -1, vexNDS3},
|
|
"VPSIGNW": {2, 0x09, 0, 1, -1, vexNDS3},
|
|
"VPSIGND": {2, 0x0A, 0, 1, -1, vexNDS3},
|
|
// VEX.128/256.0F38.W0, the masked loads/stores whose mask source rides
|
|
// vvvv, memory in r/m: the Plan 9 order puts the destination first
|
|
// (dst, mask, src), its own form below. VMOVHLPS is the plain NDS
|
|
// register move under 0F 12, and the test pair and the AES inverse cube
|
|
// root are two-operand.
|
|
"VMOVHLPS": {1, 0x12, 0, 0, -1, vexNDS3},
|
|
"VMASKMOVPD": {2, 0x2F, 0, 1, -1, vexNDS3Dst},
|
|
"VMASKMOVPS": {2, 0x2E, 0, 1, -1, vexNDS3Dst},
|
|
"VPMASKMOVD": {2, 0x8E, 0, 1, -1, vexNDS3Dst},
|
|
"VPMASKMOVQ": {2, 0x8E, 1, 1, -1, vexNDS3Dst},
|
|
"VTESTPD": {2, 0x0F, 0, 1, -1, vexRM},
|
|
"VTESTPS": {2, 0x0E, 0, 1, -1, vexRM},
|
|
"VAESIMC": {2, 0xDB, 0, 1, -1, vexRM},
|
|
"VPHMINPOSUW": {2, 0x41, 0, 1, -1, vexRM},
|
|
// VEX.128/256.66.0F38.W0, broadcast a 128-bit lane into a YMM.
|
|
"VBROADCASTF128": {2, 0x1A, 0, 1, -1, vexRM},
|
|
// VEX.128/256, the reciprocal square root estimates: PS bare (two-op
|
|
// RM), SS F3-prefixed NDS (the scalar preserved source).
|
|
"VRCPPS": {1, 0x53, 0, 0, -1, vexRM},
|
|
"VRSQRTPS": {1, 0x52, 0, 0, -1, vexRM},
|
|
"VRCPSS": {1, 0x53, 0, 2, -1, vexNDS3},
|
|
"VRSQRTSS": {1, 0x52, 0, 2, -1, vexNDS3},
|
|
// VEX.128/256.F2.0F.WIG, the unaligned load with cache hints and the
|
|
// duplicated low double, F3 for the singles replica.
|
|
"VLDDQU": {1, 0xF0, 0, 3, -1, vexRM},
|
|
// VEX.128/256.66.0F, the move-mask twin of VMOVMSKPS and the masked
|
|
// store over the integer bank.
|
|
"VMOVMSKPD": {1, 0x50, 0, 1, -1, vexRM},
|
|
"VMASKMOVDQU": {1, 0xF7, 0, 1, -1, vexRM},
|
|
// VEX.128/256.0F3A.W0, the imm8 tail the legacy set carries and its
|
|
// variable blends over the is4 byte.
|
|
"VBLENDPD": {3, 0x0D, 0, 1, -1, vexNDS3Imm},
|
|
"VBLENDPS": {3, 0x0C, 0, 1, -1, vexNDS3Imm},
|
|
"VPBLENDW": {3, 0x0E, 0, 1, -1, vexNDS3Imm},
|
|
"VDPPD": {3, 0x41, 0, 1, -1, vexNDS3Imm},
|
|
"VDPPS": {3, 0x40, 0, 1, -1, vexNDS3Imm},
|
|
"VINSERTPS": {3, 0x21, 0, 1, -1, vexNDS3Imm},
|
|
"VMPSADBW": {3, 0x42, 0, 1, -1, vexNDS3Imm},
|
|
"VROUNDSD": {3, 0x0B, 0, 1, -1, vexNDS3Imm},
|
|
"VROUNDSS": {3, 0x0A, 0, 1, -1, vexNDS3Imm},
|
|
"VPINSRB": {3, 0x20, 0, 1, -1, vexNDS3Imm},
|
|
// VEX.128.66.0F3A.W0, the four-operand variable blends over the /is4
|
|
// byte (the mask rides is4[7:4], the raw register number times 16).
|
|
"VBLENDVPS": {3, 0x4A, 0, 1, -1, vexBlend4},
|
|
"VBLENDVPD": {3, 0x4B, 0, 1, -1, vexBlend4},
|
|
// VEX.256.66.0F3A.W0, the lane insert, and the GPR insert the legacy
|
|
// set spells: VPINSRW rides plain 0F C4 with 66.
|
|
"VINSERTF128": {3, 0x18, 0, 1, -1, vexNDS3Imm},
|
|
"VPINSRW": {1, 0xC4, 0, 1, -1, vexNDS3Imm},
|
|
// VEX.128/256.0F, the MXCSR accessors, memory alone, no prefix.
|
|
"VLDMXCSR": {1, 0xAE, 0, 0, 2, vexRMOpDigit},
|
|
"VSTMXCSR": {1, 0xAE, 0, 0, 3, vexRMOpDigit},
|
|
}
|
|
|
|
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
|
|
// the packed-double → dword conversions) to its fixed vector length:
|
|
// X = 128 (L = 0), Y = 256 (L = 1). The spelling fixes the length even for
|
|
// a memory source, matching the Go assembler's ytab.
|
|
var vexSrcLen = map[string]int{
|
|
"VCVTPD2DQX": 0,
|
|
"VCVTPD2DQY": 1,
|
|
"VCVTTPD2DQX": 0,
|
|
"VCVTTPD2DQY": 1,
|
|
"VCVTPD2PSX": 0,
|
|
"VCVTPD2PSY": 1,
|
|
}
|
|
|
|
// vexVarShift maps the shift mnemonics to their variable-count opcode, the
|
|
// form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8),
|
|
// an ordinary NDS encoding rather than the /digit immediate form above.
|
|
var vexVarShift = map[string]byte{
|
|
"VPSLLD": 0xF2,
|
|
"VPSLLQ": 0xF3,
|
|
"VPSLLW": 0xF1,
|
|
"VPSRAD": 0xE2,
|
|
"VPSRAW": 0xE1,
|
|
"VPSRLD": 0xD2,
|
|
"VPSRLQ": 0xD3,
|
|
"VPSRLW": 0xD1,
|
|
}
|
|
|
|
// vexPermilReg maps the VPERMIL register-control spellings to their 0F38
|
|
// NDS opcodes: the control rides vvvv, the immediate form the main table
|
|
// carries never enters this path.
|
|
var vexPermilReg = map[string]vexSpec{
|
|
"VPERMILPS": {2, 0x0C, 0, 1, -1, vexNDS3},
|
|
"VPERMILPD": {2, 0x0D, 0, 1, -1, vexNDS3},
|
|
}
|
|
|
|
// vexMoveSpec describes a VEX move, which takes different opcodes (and
|
|
// sometimes a different VEX.W) per operand direction. The Go assembler
|
|
// encodes a vector→vector move with the store-form opcode (reg = source,
|
|
// rm = destination), so regReg defaults to store when zero.
|
|
type vexMoveSpec struct {
|
|
mapSel int
|
|
pp int
|
|
load byte // r/m → vector: reg=dst, rm=src
|
|
store byte // vector → r/m: reg=src, rm=dst
|
|
loadW int
|
|
storeW int
|
|
regReg byte // vector → vector opcode; 0 uses store
|
|
regW int
|
|
vecOK bool // the non-fixed operand may be a vector register
|
|
gprOK bool // the non-fixed operand may be a general-purpose register
|
|
xmmOnly bool // YMM registers are rejected
|
|
}
|
|
|
|
// vexMoveTable maps an upper-case move mnemonic to its encoding.
|
|
var vexMoveTable = map[string]vexMoveSpec{
|
|
// VEX.128/256.F3.0F.WIG, unaligned integer move.
|
|
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
|
// VEX.128/256.66.0F.WIG, aligned integer move.
|
|
"VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
|
// VEX.128/256.66.0F.WIG, unaligned packed double move.
|
|
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
|
|
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
|
|
"VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true},
|
|
// VMOVQ, 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
|
|
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true},
|
|
// VEX.128.F2.0F.WIG, scalar double move: two operands move against
|
|
// memory, three operands the NDS store-opcode form (see encodeVexMove).
|
|
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
|
|
// VEX.128.F3.0F.WIG, scalar single move, the same two shapes.
|
|
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
|
|
// VEX.128/256, aligned packed moves.
|
|
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
|
|
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
|
|
}
|
|
|
|
// isVex reports whether the mnemonic is a VEX-encoded instruction we handle.
|
|
func isVex(mnemUpper string) bool {
|
|
if _, ok := vexTable[mnemUpper]; ok {
|
|
return true
|
|
}
|
|
if _, ok := vexMoveTable[mnemUpper]; ok {
|
|
return true
|
|
}
|
|
// The dual-shape moves (VMOVHPD/VMOVLPD and the single-precision twins)
|
|
// pick their VEX form by operand count in encodeVex.
|
|
switch mnemUpper {
|
|
case "VMOVHPD", "VMOVLPD", "VMOVHPS", "VMOVLPS":
|
|
return true
|
|
}
|
|
return false
|
|
}
|
|
|
|
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
|
|
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
|
// Vector register indices 16-31 exist only in EVEX encodings; fail
|
|
// loudly rather than silently truncating the index.
|
|
for _, op := range ops {
|
|
if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 {
|
|
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
|
}
|
|
}
|
|
// VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination
|
|
// only; an XMM destination is rejected exactly as go tool asm does.
|
|
if mnemUpper == "VBROADCASTI128" {
|
|
dstReg, ok := ops[len(ops)-1].(Reg)
|
|
if len(ops) != 2 || !ok || dstReg.size != 32 {
|
|
return fmt.Errorf("VBROADCASTI128 requires a YMM destination")
|
|
}
|
|
}
|
|
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
|
return e.encodeVexMove(mnemUpper, ms, ops)
|
|
}
|
|
// The shifts come in two shapes under one mnemonic: an immediate count
|
|
// ($imm, src, dst) and a variable count in an XMM register or memory
|
|
// (count, src, dst), the latter an ordinary NDS form.
|
|
if op, ok := vexVarShift[mnemUpper]; ok && len(ops) == 3 {
|
|
if _, isImm := ops[0].(Imm); !isImm {
|
|
if !vecOrMem(ops[0]) {
|
|
return fmt.Errorf("%s: shift count must be an immediate, a vector register or memory", mnemUpper)
|
|
}
|
|
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
|
|
}
|
|
}
|
|
// The VPERMIL register-control form: the control rides vvvv (an NDS
|
|
// encoding under 0F38), the immediate form the main table carries.
|
|
if spec, ok := vexPermilReg[mnemUpper]; ok && len(ops) == 3 {
|
|
if _, isImm := ops[0].(Imm); !isImm {
|
|
return e.encodeVexNDS3(spec, ops)
|
|
}
|
|
}
|
|
// The high/low double moves split by operand count: three operands
|
|
// load-and-insert (mem, src, dst, an NDS form), two store (xmm, m64,
|
|
// the reversed store layout).
|
|
if mnemUpper == "VMOVHPD" || mnemUpper == "VMOVLPD" || mnemUpper == "VMOVHPS" || mnemUpper == "VMOVLPS" {
|
|
loadOp, storeOp := byte(0x16), byte(0x17)
|
|
if mnemUpper == "VMOVLPD" || mnemUpper == "VMOVLPS" {
|
|
loadOp, storeOp = 0x12, 0x13
|
|
}
|
|
pp := 1
|
|
if mnemUpper == "VMOVHPS" || mnemUpper == "VMOVLPS" {
|
|
pp = 0
|
|
}
|
|
switch len(ops) {
|
|
case 3:
|
|
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: loadOp, w: 0, pp: pp, opdigit: -1, form: vexNDS3}, ops)
|
|
case 2:
|
|
return e.encodeVexRMRev(vexSpec{mapSel: 1, opcode: storeOp, w: 0, pp: pp, opdigit: -1, form: vexRMRev}, ops)
|
|
}
|
|
return fmt.Errorf("%s expects 2 or 3 operands, got %d", mnemUpper, len(ops))
|
|
}
|
|
spec := vexTable[mnemUpper]
|
|
switch spec.form {
|
|
case vexNDS3:
|
|
return e.encodeVexNDS3(spec, ops)
|
|
case vexRM:
|
|
return e.encodeVexRM(spec, ops)
|
|
case vexShiftImm:
|
|
return e.encodeVexShiftImm(spec, ops)
|
|
case vexImmRM:
|
|
return e.encodeVexImmRM(spec, ops)
|
|
case vexNDS3Imm:
|
|
return e.encodeVexNDS3Imm(spec, ops)
|
|
case vexExtract:
|
|
return e.encodeVexExtract(spec, ops)
|
|
case vexExtractGPR:
|
|
return e.encodeVexExtractGPR(spec, ops)
|
|
case vexBlend4:
|
|
return e.encodeVexBlend4(spec, ops)
|
|
case vexRMSrcLen:
|
|
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
|
case vexZero:
|
|
return e.encodeVexZero(mnemUpper, spec, ops)
|
|
case vexZeroAll:
|
|
return e.encodeVexZeroAll(mnemUpper, spec, ops)
|
|
case vexNDS3GPR:
|
|
return e.encodeVexNDS3GPR(spec, ops)
|
|
case vexImmRMGPR:
|
|
return e.encodeVexImmRMGPR(spec, ops)
|
|
case vexRMOpGPR:
|
|
return e.encodeVexRMOpGPR(spec, ops)
|
|
case vexCountGPR:
|
|
return e.encodeVexCountGPR(spec, ops)
|
|
case vexRMRev:
|
|
return e.encodeVexRMRev(spec, ops)
|
|
case vexNDS3Dst:
|
|
return e.encodeVexNDS3Dst(spec, ops)
|
|
case vexRMOpDigit:
|
|
return e.encodeVexRMOpDigit(mnemUpper, spec, ops)
|
|
}
|
|
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
|
|
}
|
|
|
|
// encodeVexNDS3Dst encodes the destination-first NDS forms (VMASKMOVPS,
|
|
// VPMASKMOVD): ModRM.reg = the vector register, VEX.vvvv = the mask source,
|
|
// ModRM.rm = memory. Both directions exist: (dst, mask, mem) stores under
|
|
// the table opcode, (mem, mask, dst) loads under its twin two lower (the
|
|
// opcode rows pair 2E/2C, 2F/2D and 8E/8C).
|
|
func (e *enc) encodeVexNDS3Dst(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 3 {
|
|
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
|
|
}
|
|
opcode := spec.opcode
|
|
first, second, third := ops[0], ops[1], ops[2]
|
|
if isX86Mem(first) {
|
|
first, third = third, first
|
|
opcode -= 2
|
|
}
|
|
dstReg, ok := first.(Reg)
|
|
if !ok || !dstReg.isVec() {
|
|
return fmt.Errorf("VEX destination must be a vector register")
|
|
}
|
|
vvvvReg, ok := second.(Reg)
|
|
if !ok || !vvvvReg.isVec() {
|
|
return fmt.Errorf("VEX mask source must be a vector register")
|
|
}
|
|
if !vecOrMem(third) {
|
|
return fmt.Errorf("VEX memory source expected")
|
|
}
|
|
regField := dstReg.idx & 7
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
enc := spec
|
|
enc.opcode = opcode
|
|
return e.emitVexFields(enc, dstReg.vecLenBit(), regField, rBit, 15-(vvvvReg.idx&15), third)
|
|
}
|
|
|
|
// encodeVexRMOpDigit encodes the MXCSR accessors: the single memory operand
|
|
// rides r/m under the fixed /digit, the way the legacy 0F AE pair does.
|
|
func (e *enc) encodeVexRMOpDigit(mnem string, spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 1 {
|
|
return fmt.Errorf("%s expects 1 memory operand, got %d", mnem, len(ops))
|
|
}
|
|
if !isX86Mem(ops[0]) {
|
|
return fmt.Errorf("%s requires a memory operand", mnem)
|
|
}
|
|
return e.emitVexFields(spec, 0, spec.opdigit, 0, 15, ops[0])
|
|
}
|
|
|
|
// encodeVexNDS3 encodes the three-operand NDS form: OP src2, src1, dst.
|
|
func (e *enc) encodeVexNDS3(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 3 {
|
|
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
|
|
}
|
|
src2, src1, dst := ops[0], ops[1], ops[2]
|
|
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || !dstReg.isVec() {
|
|
return fmt.Errorf("VEX destination must be a vector register")
|
|
}
|
|
vvvvReg, ok := src1.(Reg)
|
|
if !ok || !vvvvReg.isVec() {
|
|
return fmt.Errorf("VEX vvvv operand must be a vector register")
|
|
}
|
|
|
|
regField := dstReg.idx & 7
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
vvvvBar := 15 - (vvvvReg.idx & 15)
|
|
return e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2)
|
|
}
|
|
|
|
// encodeVexRM encodes the two-operand form: OP src, dst (no vvvv source).
|
|
// ModRM.reg = dst, ModRM.rm = src; the vector length comes from whichever
|
|
// operand is a vector register (the destination for extends/broadcasts, the
|
|
// source for the move-mask instructions whose destination is a GPR).
|
|
func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 2 {
|
|
return fmt.Errorf("VEX two-operand instruction expects 2 operands, got %d", len(ops))
|
|
}
|
|
src, dst := ops[0], ops[1]
|
|
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok {
|
|
return fmt.Errorf("VEX destination must be a register")
|
|
}
|
|
regField := dstReg.idx & 7
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
|
|
// Vector length: from the destination if it is a vector, otherwise from the
|
|
// source (move-mask instructions have a GPR destination and a vector source).
|
|
l := 0
|
|
if dstReg.isVec() {
|
|
l = dstReg.vecLenBit()
|
|
} else if srcReg, ok := src.(Reg); ok && srcReg.isVec() {
|
|
l = srcReg.vecLenBit()
|
|
}
|
|
|
|
// An unused vvvv field must be stored as all ones (v̄vvv = 1111); the
|
|
// hardware raises #UD on any other value.
|
|
return e.emitVexFields(spec, l, regField, rBit, 15, src)
|
|
}
|
|
|
|
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
|
|
// the destination always XMM and the VEX.L bit following the source, fixed
|
|
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
|
|
// the source is memory.
|
|
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 2 {
|
|
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
|
|
}
|
|
src, dst := ops[0], ops[1]
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || !dstReg.isVec() {
|
|
return fmt.Errorf("VEX destination must be a vector register")
|
|
}
|
|
ll, ok := vexSrcLen[mnem]
|
|
if !ok {
|
|
return fmt.Errorf("no fixed vector length for %s", mnem)
|
|
}
|
|
regField := dstReg.idx & 7
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
// An unused vvvv field must be stored as all ones (v̄vvv = 1111).
|
|
return e.emitVexFields(spec, ll, regField, rBit, 15, src)
|
|
}
|
|
|
|
// encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst.
|
|
// The destination is carried in VEX.vvvv, the source in ModRM.rm, and the
|
|
// shift kind in the ModRM.reg /digit.
|
|
func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 3 {
|
|
return fmt.Errorf("VEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
|
|
}
|
|
imm, src, dst := ops[0], ops[1], ops[2]
|
|
immVal, ok := imm.(Imm)
|
|
if !ok {
|
|
return fmt.Errorf("shift count must be an immediate")
|
|
}
|
|
// The count source is a vector register or memory; the VEX length
|
|
// follows the destination register either way.
|
|
if !vecOrMem(src) {
|
|
return fmt.Errorf("shift source must be a vector register or memory")
|
|
}
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || !dstReg.isVec() {
|
|
return fmt.Errorf("shift destination must be a vector register")
|
|
}
|
|
|
|
vvvvBar := 15 - (dstReg.idx & 15)
|
|
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, src); err != nil {
|
|
return err
|
|
}
|
|
immByte, err := imm8(int64(immVal))
|
|
if err != nil {
|
|
return err
|
|
}
|
|
e.out = append(e.out, immByte)
|
|
return nil
|
|
}
|
|
|
|
// imm8 range-checks an immediate for an 8-bit field. Shuffle controls are
|
|
// unsigned bit masks, but the negative spelling ($-1 = all bits set) is
|
|
// accepted, so the accepted span is -128..255.
|
|
func imm8(v int64) (byte, error) {
|
|
if v < -128 || v > 255 {
|
|
return 0, fmt.Errorf("immediate $%d does not fit in 8 bits", v)
|
|
}
|
|
return byte(v), nil
|
|
}
|
|
|
|
// encodeVexImmRM encodes an immediate form with no vvvv source: OP $imm, src,
|
|
// dst (VPSHUFD, VPERMQ). ModRM.reg = dst, ModRM.rm = src, imm8 appended.
|
|
func (e *enc) encodeVexImmRM(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 3 {
|
|
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
|
|
}
|
|
imm, src, dst := ops[0], ops[1], ops[2]
|
|
immVal, ok := imm.(Imm)
|
|
if !ok {
|
|
return fmt.Errorf("shuffle control must be an immediate")
|
|
}
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || !dstReg.isVec() {
|
|
return fmt.Errorf("shuffle destination must be a vector register")
|
|
}
|
|
|
|
// The vector length follows the source when it is a vector register,
|
|
// otherwise the destination (a memory source carries no length).
|
|
l := dstReg.vecLenBit()
|
|
if srcReg, ok := src.(Reg); ok && srcReg.isVec() {
|
|
l = srcReg.vecLenBit()
|
|
}
|
|
regField := dstReg.idx & 7
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
if err := e.emitVexFields(spec, l, regField, rBit, 15, src); err != nil {
|
|
return err
|
|
}
|
|
immByte, err := imm8(int64(immVal))
|
|
if err != nil {
|
|
return err
|
|
}
|
|
e.out = append(e.out, immByte)
|
|
return nil
|
|
}
|
|
|
|
// encodeVexNDS3Imm encodes the three-operand plus immediate form: OP $imm,
|
|
// src2, src1, dst (VSHUFPD, VPERM2I128, VINSERTI128). ModRM.reg = dst,
|
|
// VEX.vvvv = src1, ModRM.rm = src2, imm8 appended.
|
|
func (e *enc) encodeVexNDS3Imm(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 4 {
|
|
return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops))
|
|
}
|
|
imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
|
|
immVal, ok := imm.(Imm)
|
|
if !ok {
|
|
return fmt.Errorf("shuffle control must be an immediate")
|
|
}
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || !dstReg.isVec() {
|
|
return fmt.Errorf("destination must be a vector register")
|
|
}
|
|
vvvvReg, ok := src1.(Reg)
|
|
if !ok || !vvvvReg.isVec() {
|
|
return fmt.Errorf("second source must be a vector register")
|
|
}
|
|
|
|
regField := dstReg.idx & 7
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
vvvvBar := 15 - (vvvvReg.idx & 15)
|
|
if err := e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2); err != nil {
|
|
return err
|
|
}
|
|
immByte, err := imm8(int64(immVal))
|
|
if err != nil {
|
|
return err
|
|
}
|
|
e.out = append(e.out, immByte)
|
|
return nil
|
|
}
|
|
|
|
// encodeVexExtract encodes a lane extract: OP $imm, ysrc, xdst
|
|
// (VEXTRACTI128, VEXTRACTF128). The YMM source occupies ModRM.reg and the
|
|
// XMM (or memory) destination ModRM.rm; imm8 selects the lane.
|
|
func (e *enc) encodeVexExtract(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 3 {
|
|
return fmt.Errorf("extract expects 3 operands ($imm, ysrc, xdst), got %d", len(ops))
|
|
}
|
|
imm, src, dst := ops[0], ops[1], ops[2]
|
|
immVal, ok := imm.(Imm)
|
|
if !ok {
|
|
return fmt.Errorf("extract lane must be an immediate")
|
|
}
|
|
srcReg, ok := src.(Reg)
|
|
if !ok || !srcReg.isVec() {
|
|
return fmt.Errorf("extract source must be a vector register")
|
|
}
|
|
|
|
regField := srcReg.idx & 7
|
|
rBit := 0
|
|
if srcReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
if err := e.emitVexFields(spec, srcReg.vecLenBit(), regField, rBit, 15, dst); err != nil {
|
|
return err
|
|
}
|
|
immByte, err := imm8(int64(immVal))
|
|
if err != nil {
|
|
return err
|
|
}
|
|
e.out = append(e.out, immByte)
|
|
return nil
|
|
}
|
|
|
|
// encodeVexZero encodes a no-operand instruction (VZEROUPPER).
|
|
func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 0 {
|
|
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
|
|
}
|
|
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 0.
|
|
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|spec.pp), spec.opcode)
|
|
return nil
|
|
}
|
|
|
|
// encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1
|
|
// twin of VZEROUPPER.
|
|
func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 0 {
|
|
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
|
|
}
|
|
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1.
|
|
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode)
|
|
return nil
|
|
}
|
|
|
|
// encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose
|
|
// registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1,
|
|
// rm = src2 and L = 0.
|
|
func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 3 {
|
|
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
|
|
}
|
|
src2, src1, dst := ops[0], ops[1], ops[2]
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || dstReg.isVec() {
|
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
|
}
|
|
vvvvReg, ok := src1.(Reg)
|
|
if !ok || vvvvReg.isVec() {
|
|
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
|
|
}
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2)
|
|
}
|
|
|
|
// encodeVexImmRMGPR encodes the immediate form over general-purpose
|
|
// registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0.
|
|
func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 3 {
|
|
return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
|
|
}
|
|
imm, src, dst := ops[0], ops[1], ops[2]
|
|
immVal, ok := imm.(Imm)
|
|
if !ok {
|
|
return fmt.Errorf("shift control must be an immediate")
|
|
}
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || dstReg.isVec() {
|
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
|
}
|
|
immByte, err := imm8(int64(immVal))
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if err := e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src); err != nil {
|
|
return err
|
|
}
|
|
e.out = append(e.out, immByte)
|
|
return nil
|
|
}
|
|
|
|
// encodeVexRMOpGPR encodes the two-operand /digit form over general-purpose
|
|
// registers (BLSI, BLSMSK, BLSR): OP src, dst with ModRM.reg = /digit,
|
|
// ModRM.rm = src and VEX.vvvv = dst.
|
|
func (e *enc) encodeVexRMOpGPR(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 2 {
|
|
return fmt.Errorf("instruction expects 2 operands (src, dst), got %d", len(ops))
|
|
}
|
|
src, dst := ops[0], ops[1]
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || dstReg.isVec() {
|
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
|
}
|
|
return e.emitVexFields(spec, 0, spec.opdigit, 0, 15-(dstReg.idx&15), src)
|
|
}
|
|
|
|
// encodeVexCountGPR encodes the three-operand count form over general-purpose
|
|
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): OP src, count, dst with
|
|
// VEX.vvvv = src (op0), ModRM.rm = count (op1, register or memory),
|
|
// ModRM.reg = dst (op2).
|
|
func (e *enc) encodeVexCountGPR(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 3 {
|
|
return fmt.Errorf("VEX count instruction expects 3 operands, got %d", len(ops))
|
|
}
|
|
src, count, dst := ops[0], ops[1], ops[2]
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || dstReg.isVec() {
|
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
|
}
|
|
switch count.(type) {
|
|
case Reg, Mem, sbMem:
|
|
default:
|
|
return fmt.Errorf("VEX count operand must be a general-purpose register or memory")
|
|
}
|
|
if r, ok := count.(Reg); ok && r.isVec() {
|
|
return fmt.Errorf("VEX count operand must be a general-purpose register or memory")
|
|
}
|
|
srcReg, ok := src.(Reg)
|
|
if !ok || srcReg.isVec() {
|
|
return fmt.Errorf("VEX count source must be a general-purpose register")
|
|
}
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(srcReg.idx&15), count)
|
|
}
|
|
|
|
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
|
|
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
|
|
// a store with no register-destination form).
|
|
func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 2 {
|
|
return fmt.Errorf("store expects 2 operands, got %d", len(ops))
|
|
}
|
|
srcReg, ok := ops[0].(Reg)
|
|
if !ok || !srcReg.isVec() {
|
|
return fmt.Errorf("store source must be a vector register")
|
|
}
|
|
if !memOperand(ops[1]) {
|
|
return fmt.Errorf("store destination must be memory")
|
|
}
|
|
rBit := 0
|
|
if srcReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
|
|
}
|
|
|
|
// encodeVexExtractGPR encodes the lane extract to a general-purpose register
|
|
// or memory (VEXTRACTPS): OP $imm, xsrc, gpr/mem with the XMM source in
|
|
// ModRM.reg and the destination in r/m, L = 0.
|
|
func (e *enc) encodeVexExtractGPR(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 3 {
|
|
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
|
|
}
|
|
imm, src, dst := ops[0], ops[1], ops[2]
|
|
immVal, ok := imm.(Imm)
|
|
if !ok {
|
|
return fmt.Errorf("extract lane must be an immediate")
|
|
}
|
|
srcReg, ok := src.(Reg)
|
|
if !ok || !srcReg.isVec() || srcReg.size != 16 {
|
|
return fmt.Errorf("extract source must be an XMM register")
|
|
}
|
|
if _, isReg := dst.(Reg); !isReg && !memOperand(dst) {
|
|
return fmt.Errorf("extract destination must be a register or memory")
|
|
}
|
|
rBit := 0
|
|
if srcReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
if err := e.emitVexFields(spec, 0, srcReg.idx&7, rBit, 15, dst); err != nil {
|
|
return err
|
|
}
|
|
immByte, err := imm8(int64(immVal))
|
|
if err != nil {
|
|
return err
|
|
}
|
|
e.out = append(e.out, immByte)
|
|
return nil
|
|
}
|
|
|
|
// encodeVexBlend4 encodes the four-operand variable blend (VPBLENDVB and the
|
|
// VBLENDV pair): OP mask, src2, src1, dst with ModRM.reg = dst, VEX.vvvv =
|
|
// src1, r/m = src2 and the mask register in the trailing /is4 byte, whose
|
|
// high nibble carries the mask's register number raw.
|
|
func (e *enc) encodeVexBlend4(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 4 {
|
|
return fmt.Errorf("blend expects 4 operands (mask, src2, src1, dst), got %d", len(ops))
|
|
}
|
|
mask, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
|
|
maskReg, ok := mask.(Reg)
|
|
if !ok || !maskReg.isVec() {
|
|
return fmt.Errorf("blend mask must be a vector register")
|
|
}
|
|
vvvvReg, ok := src1.(Reg)
|
|
if !ok || !vvvvReg.isVec() {
|
|
return fmt.Errorf("blend second source must be a vector register")
|
|
}
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || !dstReg.isVec() {
|
|
return fmt.Errorf("blend destination must be a vector register")
|
|
}
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
if err := e.emitVexFields(spec, dstReg.vecLenBit(), dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2); err != nil {
|
|
return err
|
|
}
|
|
// The /is4 byte names the mask register: its number in the high nibble,
|
|
// the layout the Go assembler and the hardware agree on for X0-X15.
|
|
e.out = append(e.out, byte(maskReg.idx)<<4)
|
|
return nil
|
|
}
|
|
|
|
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
|
|
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
|
|
// move uses the store-form layout (reg = source, rm = destination), matching
|
|
// the Go assembler. The scalar moves (VMOVSD, VMOVSS) also carry a
|
|
// three-operand form, which the toolchain encodes with the store opcode:
|
|
// reg = the Plan 9 first operand, vvvv = the second, rm = the third.
|
|
func (e *enc) encodeVexMove(mnem string, ms vexMoveSpec, ops []Operand) error {
|
|
if len(ops) == 3 {
|
|
if !ms.xmmOnly {
|
|
return fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
op0, ok0 := ops[0].(Reg)
|
|
op1, ok1 := ops[1].(Reg)
|
|
op2, ok2 := ops[2].(Reg)
|
|
if !ok0 || !ok1 || !ok2 || !op0.isVec() || !op1.isVec() || !op2.isVec() {
|
|
return fmt.Errorf("%s three-operand form takes three vector registers", mnem)
|
|
}
|
|
if ms.xmmOnly && (op0.size != 16 || op1.size != 16 || op2.size != 16) {
|
|
return fmt.Errorf("%s operates on XMM registers only", mnem)
|
|
}
|
|
rBit := 0
|
|
if op0.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
spec := vexSpec{mapSel: ms.mapSel, opcode: ms.store, w: ms.storeW, pp: ms.pp, opdigit: -1}
|
|
return e.emitVexFields(spec, 0, op0.idx&7, rBit, 15-(op1.idx&15), op2)
|
|
}
|
|
if len(ops) != 2 {
|
|
return fmt.Errorf("VEX move expects 2 operands, got %d", len(ops))
|
|
}
|
|
src, dst := ops[0], ops[1]
|
|
srcReg, srcIsVec := vecReg(src)
|
|
dstReg, dstIsVec := vecReg(dst)
|
|
|
|
var reg Reg
|
|
var rm Operand
|
|
op, w := ms.store, ms.storeW
|
|
switch {
|
|
case srcIsVec && dstIsVec:
|
|
if !ms.vecOK {
|
|
return fmt.Errorf("%s does not take two vector registers", mnem)
|
|
}
|
|
if ms.xmmOnly && (srcReg.size == 32 || dstReg.size == 32) {
|
|
return fmt.Errorf("%s operates on XMM registers only", mnem)
|
|
}
|
|
if ms.regReg != 0 {
|
|
op, w = ms.regReg, ms.regW
|
|
}
|
|
reg, rm = srcReg, dst // store form: reg = source, rm = destination.
|
|
case srcIsVec:
|
|
// vector → memory, or → GPR (VMOVD/VMOVQ only).
|
|
if !validMoveOther(ms, dst) {
|
|
return fmt.Errorf("%s: invalid destination operand", mnem)
|
|
}
|
|
reg, rm = srcReg, dst
|
|
case dstIsVec:
|
|
// memory → vector, or GPR → vector (VMOVD/VMOVQ only).
|
|
if !validMoveOther(ms, src) {
|
|
return fmt.Errorf("%s: invalid source operand", mnem)
|
|
}
|
|
op, w = ms.load, ms.loadW
|
|
reg, rm = dstReg, src
|
|
default:
|
|
return fmt.Errorf("%s needs a vector register operand", mnem)
|
|
}
|
|
if ms.xmmOnly && reg.size == 32 {
|
|
return fmt.Errorf("%s operates on XMM registers only", mnem)
|
|
}
|
|
|
|
regField := reg.idx & 7
|
|
rBit := 0
|
|
if reg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
spec := vexSpec{mapSel: ms.mapSel, opcode: op, w: w, pp: ms.pp, opdigit: -1}
|
|
return e.emitVexFields(spec, reg.vecLenBit(), regField, rBit, 15, rm)
|
|
}
|
|
|
|
// vecReg extracts a vector register from an operand.
|
|
func vecReg(op Operand) (Reg, bool) {
|
|
r, ok := op.(Reg)
|
|
return r, ok && r.isVec()
|
|
}
|
|
|
|
// vecOrMem reports whether op is a vector register or a memory reference.
|
|
func vecOrMem(op Operand) bool {
|
|
switch op.(type) {
|
|
case Mem, sbMem:
|
|
return true
|
|
}
|
|
r, ok := op.(Reg)
|
|
return ok && r.isVec()
|
|
}
|
|
|
|
// validMoveOther reports whether the non-vector operand of a move is
|
|
// acceptable: memory always is, a GPR only for VMOVD/VMOVQ.
|
|
func validMoveOther(ms vexMoveSpec, op Operand) bool {
|
|
switch o := op.(type) {
|
|
case Mem, sbMem:
|
|
return true
|
|
case Reg:
|
|
return ms.gprOK && !o.isVec()
|
|
}
|
|
return false
|
|
}
|
|
|
|
// emitVexFields emits the VEX prefix, opcode, ModR/M, SIB and displacement for
|
|
// the given precomputed fields. It is shared by every register/rm VEX form;
|
|
// immediate bytes are appended by the caller.
|
|
func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Operand) error {
|
|
if l > 1 {
|
|
return fmt.Errorf("ZMM operand requires an EVEX instruction")
|
|
}
|
|
var modrm, sib int
|
|
var disp []byte
|
|
var xBit, bBit int
|
|
var sb *sbRef
|
|
switch r := rm.(type) {
|
|
case Reg:
|
|
modrm = 0xC0 | regField<<3 | (r.idx & 7)
|
|
sib = -1
|
|
if r.idx >= 8 {
|
|
bBit = 1
|
|
}
|
|
case Mem:
|
|
var err error
|
|
modrm, sib, disp, xBit, bBit, err = memComponents(regField, r)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
case sbMem:
|
|
// RIP-relative static-symbol reference; disp32 patched at link time.
|
|
modrm = regField<<3 | 0x05
|
|
sib = -1
|
|
disp = le32(0)
|
|
sb = &sbRef{name: r.name, addend: r.addend}
|
|
default:
|
|
return fmt.Errorf("invalid VEX r/m operand")
|
|
}
|
|
|
|
if spec.mapSel == 1 && xBit == 0 && bBit == 0 && spec.w == 0 {
|
|
e.out = append(e.out, 0xC5, byte((1-rBit)<<7|vvvvBar<<3|l<<2|spec.pp))
|
|
} else {
|
|
e.out = append(e.out, 0xC4,
|
|
byte((1-rBit)<<7|(1-xBit)<<6|(1-bBit)<<5|spec.mapSel),
|
|
byte(spec.w<<7|vvvvBar<<3|l<<2|spec.pp))
|
|
}
|
|
e.out = append(e.out, spec.opcode, byte(modrm))
|
|
if sib >= 0 {
|
|
e.out = append(e.out, byte(sib))
|
|
}
|
|
if sb != nil {
|
|
e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend})
|
|
}
|
|
e.out = append(e.out, disp...)
|
|
return nil
|
|
}
|