745 lines
26 KiB
Go
745 lines
26 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
package asm
|
|
|
|
import "fmt"
|
|
|
|
// This file implements VEX (AVX/AVX2) instruction encoding. EVEX (AVX-512)
|
|
// support is a later increment.
|
|
//
|
|
// Every encoding choice here is validated two ways in the tests: by
|
|
// round-trip decoding through golang.org/x/arch's x86 decoder, and by
|
|
// byte-for-byte comparison against the output of the real Go assembler.
|
|
|
|
// vexForm selects how an instruction's operands map onto the VEX.vvvv,
|
|
// ModRM.reg and ModRM.rm fields.
|
|
type vexForm int
|
|
|
|
const (
|
|
// vexNDS3 is the three-operand form `OP src2, src1, dst` (Plan 9 order):
|
|
// ModRM.reg = dst (op2), VEX.vvvv = src1 (op1), ModRM.rm = src2 (op0).
|
|
vexNDS3 vexForm = iota
|
|
// vexRM is the two-operand form `OP src, dst` with no vvvv source:
|
|
// ModRM.reg = dst (op1), ModRM.rm = src (op0), VEX.vvvv unused.
|
|
vexRM
|
|
// vexShiftImm is the immediate-shift form `OP $imm, src, dst`: ModRM.reg =
|
|
// /digit, ModRM.rm = src (op1), VEX.vvvv = dst (op2), imm8 = op0.
|
|
vexShiftImm
|
|
// vexImmRM is the immediate form `OP $imm, src, dst` with no vvvv source:
|
|
// ModRM.reg = dst (op2), ModRM.rm = src (op1), imm8 = op0. VPSHUFD and
|
|
// VPERMQ use this shape.
|
|
vexImmRM
|
|
// vexNDS3Imm is the three-operand plus immediate form `OP $imm, src2,
|
|
// src1, dst`: ModRM.reg = dst, VEX.vvvv = src1, ModRM.rm = src2, imm8.
|
|
// VSHUFPD, VPERM2I128 and VINSERTI128 use this shape.
|
|
vexNDS3Imm
|
|
// vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg =
|
|
// ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM
|
|
// source lives in the reg field, the destination in r/m, the PEXTR-style
|
|
// layout. VEXTRACTI128 and VEXTRACTF128 use this shape.
|
|
vexExtract
|
|
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
|
// in ModRM.reg and the destination in r/m, the layout of the EVEX
|
|
// narrowing stores (VPMOVDW, VPMOVQD).
|
|
vexRMRev
|
|
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
|
// vector length follows the source: the packed-double → dword
|
|
// conversions (VCVTPD2DQ/VCVTTPD2DQ and their X/Y spellings) narrow into
|
|
// an XMM destination, so the L bit rides with the wider source. The
|
|
// mnemonic's spelling fixes the length (X = 128, Y = 256), which also
|
|
// covers a memory source. ModRM.reg = dst, ModRM.rm = src, no vvvv.
|
|
vexRMSrcLen
|
|
// vexZero is the no-operand form (VZEROUPPER).
|
|
vexZero
|
|
)
|
|
|
|
// vexSpec describes one VEX instruction's encoding parameters.
|
|
type vexSpec struct {
|
|
mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A
|
|
opcode byte
|
|
w int // VEX.W (0 for WIG)
|
|
pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2
|
|
opdigit int // ModRM.reg /digit, or -1 when reg is a register
|
|
form vexForm
|
|
}
|
|
|
|
// vexTable maps an upper-case mnemonic to its VEX encoding. It is extended
|
|
// incrementally; every entry is covered by a byte-for-byte ground-truth test
|
|
// against the Go assembler.
|
|
var vexTable = map[string]vexSpec{
|
|
// VEX.128/256.66.0F.WIG, integer arithmetic / logic / compare.
|
|
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3},
|
|
"VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3},
|
|
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3},
|
|
"VPSUBQ": {1, 0xFB, 0, 1, -1, vexNDS3},
|
|
"VPXOR": {1, 0xEF, 0, 1, -1, vexNDS3},
|
|
"VPOR": {1, 0xEB, 0, 1, -1, vexNDS3},
|
|
"VPAND": {1, 0xDB, 0, 1, -1, vexNDS3},
|
|
"VPANDN": {1, 0xDF, 0, 1, -1, vexNDS3},
|
|
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3},
|
|
"VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3},
|
|
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3},
|
|
"VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3},
|
|
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3},
|
|
// VEX.256.66.0F38.W0, dword permute (three-operand NDS form).
|
|
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3},
|
|
// VEX.128/256.66.0F38.WIG.
|
|
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3},
|
|
"VPMULDQ": {2, 0x28, 0, 1, -1, vexNDS3},
|
|
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3},
|
|
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
|
|
|
|
// VEX.128/256.66.0F.WIG, packed double-precision arithmetic / logic.
|
|
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
|
|
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
|
|
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
|
|
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
|
|
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
|
|
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
|
|
// VEX.128/256.0F.WIG, packed single-precision arithmetic.
|
|
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
|
|
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
|
|
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
|
|
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3},
|
|
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3},
|
|
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3},
|
|
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
|
|
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
|
|
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
|
|
// VEX.128.F2.0F.WIG, scalar double-precision arithmetic (the packed
|
|
// opcodes with an F2 pp).
|
|
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
|
|
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
|
|
"VMULSD": {1, 0x59, 0, 3, -1, vexNDS3},
|
|
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
|
|
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
|
|
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
|
|
// VEX.128.F3.0F.WIG, scalar single-precision arithmetic (the packed
|
|
// opcodes with an F3 pp).
|
|
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
|
|
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
|
|
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3},
|
|
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
|
|
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
|
|
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
|
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
|
|
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
|
|
|
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
|
|
// no vvvv).
|
|
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
|
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
|
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM},
|
|
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM},
|
|
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM},
|
|
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
|
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
|
|
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM},
|
|
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM},
|
|
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM},
|
|
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
|
|
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
|
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
|
"VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM},
|
|
"VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM},
|
|
// VEX.128/256.F3.0F.WIG, signed dword to packed double conversion
|
|
// (reg=dst, rm=src, no vvvv; the length follows the destination).
|
|
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
|
|
// VEX.128/256.0F.WIG, signed dword to packed single conversion
|
|
// (reg=dst, rm=src, no vvvv, no mandatory prefix).
|
|
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
|
|
// VEX.128/256.0F.WIG, packed single to packed double conversion
|
|
// (reg=dst, rm=src; the destination is the wide operand and sets the
|
|
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
|
|
// the Go assembler emits the instruction with pp = 00, and gasm follows
|
|
// the Go assembler's bytes, its machine code is the oracle, not the
|
|
// manual.
|
|
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
|
|
// VEX.128.F2.0F.WIG, duplicate the low double of each 128-bit lane
|
|
// (reg=dst, rm=src, no vvvv; the length follows the destination).
|
|
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
|
|
// VEX.128/256.66.0F.WIG, move mask to a GPR (reg=gpr dst, rm=vec src).
|
|
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
|
|
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
|
|
|
|
// VEX.128/256.66.0F.WIG, immediate shifts (opdigit selects the shift).
|
|
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm},
|
|
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm},
|
|
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm},
|
|
"VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm},
|
|
"VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm},
|
|
|
|
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
|
|
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
|
|
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8).
|
|
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
|
|
|
|
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
|
|
// imm8).
|
|
"VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm},
|
|
// VEX.256.66.0F3A.W0, permute / insert (same shape; VINSERTI128's rm is
|
|
// the XMM or memory source).
|
|
"VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm},
|
|
"VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm},
|
|
|
|
// VEX.256.66.0F3A.W0, lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
|
|
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
|
|
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
|
|
// VEX.128/256.66.0F3A.W0, half-precision convert back ($imm, src, dst:
|
|
// reg=src, rm=XMM/memory dst, imm8, the extract layout).
|
|
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
|
|
|
|
// VEX.128.0F.W0, no operands.
|
|
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
|
|
|
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
|
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
|
|
|
// VEX.66.0F38.W0, broadcast a single/double to all lanes (reg=dst,
|
|
// rm=scalar memory; SD is 256-bit only).
|
|
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
|
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
|
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
|
|
// source).
|
|
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
|
// VEX.F3.0F.WIG, replicate even/odd singles (reg=dst, rm=src).
|
|
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
|
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
|
// VEX.66.0F.WIG, packed double to packed single conversion, the X/Y
|
|
// spellings: the destination is always XMM and the spelling fixes the
|
|
// source length (X = 128, Y = 256).
|
|
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
|
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
|
|
|
// VEX scalar conversions between vector and general-purpose registers.
|
|
// Vector to GPR (two operands: vec/mem source, GPR destination, vvvv
|
|
// unused; the length follows the source).
|
|
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM},
|
|
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM},
|
|
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM},
|
|
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM},
|
|
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM},
|
|
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM},
|
|
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM},
|
|
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM},
|
|
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
|
// vector source in vvvv, vector destination in reg).
|
|
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3},
|
|
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3},
|
|
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
|
|
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
|
|
|
|
// VEX.128/256.66.0F.WIG, word shifts (opdigit selects the shift).
|
|
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
|
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
|
|
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
|
|
|
|
// VEX.F2.0F, packed double to packed dword conversions, truncating and
|
|
// non-truncating. The destination is always XMM; the X/Y spellings fix
|
|
// the source length (XMM/YMM), and VEX.L follows it, see vexSrcLen.
|
|
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
|
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
|
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
|
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
|
}
|
|
|
|
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
|
|
// the packed-double → dword conversions) to its fixed vector length:
|
|
// X = 128 (L = 0), Y = 256 (L = 1). The spelling fixes the length even for
|
|
// a memory source, matching the Go assembler's ytab.
|
|
var vexSrcLen = map[string]int{
|
|
"VCVTPD2DQX": 0,
|
|
"VCVTPD2DQY": 1,
|
|
"VCVTTPD2DQX": 0,
|
|
"VCVTTPD2DQY": 1,
|
|
"VCVTPD2PSX": 0,
|
|
"VCVTPD2PSY": 1,
|
|
}
|
|
|
|
// vexVarShift maps the shift mnemonics to their variable-count opcode, the
|
|
// form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8),
|
|
// an ordinary NDS encoding rather than the /digit immediate form above.
|
|
var vexVarShift = map[string]byte{
|
|
"VPSLLD": 0xF2,
|
|
"VPSLLQ": 0xF3,
|
|
"VPSRAD": 0xE2,
|
|
"VPSRLD": 0xD2,
|
|
"VPSRLQ": 0xD3,
|
|
}
|
|
|
|
// vexMoveSpec describes a VEX move, which takes different opcodes (and
|
|
// sometimes a different VEX.W) per operand direction. The Go assembler
|
|
// encodes a vector→vector move with the store-form opcode (reg = source,
|
|
// rm = destination), so regReg defaults to store when zero.
|
|
type vexMoveSpec struct {
|
|
mapSel int
|
|
pp int
|
|
load byte // r/m → vector: reg=dst, rm=src
|
|
store byte // vector → r/m: reg=src, rm=dst
|
|
loadW int
|
|
storeW int
|
|
regReg byte // vector → vector opcode; 0 uses store
|
|
regW int
|
|
vecOK bool // the non-fixed operand may be a vector register
|
|
gprOK bool // the non-fixed operand may be a general-purpose register
|
|
xmmOnly bool // YMM registers are rejected
|
|
}
|
|
|
|
// vexMoveTable maps an upper-case move mnemonic to its encoding.
|
|
var vexMoveTable = map[string]vexMoveSpec{
|
|
// VEX.128/256.F3.0F.WIG, unaligned integer move.
|
|
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
|
// VEX.128/256.66.0F.WIG, unaligned packed double move.
|
|
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
|
|
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
|
|
"VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true},
|
|
// VMOVQ, 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
|
|
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true},
|
|
// VEX.128.F2.0F.WIG, scalar double move, memory operands only (the
|
|
// register form takes three operands and is not supported yet).
|
|
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
|
|
// VEX.128.F3.0F.WIG, scalar single move, memory operands only.
|
|
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
|
|
// VEX.128/256, aligned packed moves.
|
|
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
|
|
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
|
|
}
|
|
|
|
// isVex reports whether the mnemonic is a VEX-encoded instruction we handle.
|
|
func isVex(mnemUpper string) bool {
|
|
if _, ok := vexTable[mnemUpper]; ok {
|
|
return true
|
|
}
|
|
_, ok := vexMoveTable[mnemUpper]
|
|
return ok
|
|
}
|
|
|
|
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
|
|
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
|
// Vector register indices 16-31 exist only in EVEX encodings; fail
|
|
// loudly rather than silently truncating the index.
|
|
for _, op := range ops {
|
|
if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 {
|
|
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
|
}
|
|
}
|
|
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
|
return e.encodeVexMove(mnemUpper, ms, ops)
|
|
}
|
|
// The shifts come in two shapes under one mnemonic: an immediate count
|
|
// ($imm, src, dst) and a variable count in an XMM register or memory
|
|
// (count, src, dst), the latter an ordinary NDS form.
|
|
if op, ok := vexVarShift[mnemUpper]; ok && len(ops) == 3 {
|
|
if _, isImm := ops[0].(Imm); !isImm {
|
|
if !vecOrMem(ops[0]) {
|
|
return fmt.Errorf("%s: shift count must be an immediate, a vector register or memory", mnemUpper)
|
|
}
|
|
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
|
|
}
|
|
}
|
|
spec := vexTable[mnemUpper]
|
|
switch spec.form {
|
|
case vexNDS3:
|
|
return e.encodeVexNDS3(spec, ops)
|
|
case vexRM:
|
|
return e.encodeVexRM(spec, ops)
|
|
case vexShiftImm:
|
|
return e.encodeVexShiftImm(spec, ops)
|
|
case vexImmRM:
|
|
return e.encodeVexImmRM(spec, ops)
|
|
case vexNDS3Imm:
|
|
return e.encodeVexNDS3Imm(spec, ops)
|
|
case vexExtract:
|
|
return e.encodeVexExtract(spec, ops)
|
|
case vexRMSrcLen:
|
|
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
|
case vexZero:
|
|
return e.encodeVexZero(mnemUpper, spec, ops)
|
|
}
|
|
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
|
|
}
|
|
|
|
// encodeVexNDS3 encodes the three-operand NDS form: OP src2, src1, dst.
|
|
func (e *enc) encodeVexNDS3(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 3 {
|
|
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
|
|
}
|
|
src2, src1, dst := ops[0], ops[1], ops[2]
|
|
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || !dstReg.isVec() {
|
|
return fmt.Errorf("VEX destination must be a vector register")
|
|
}
|
|
vvvvReg, ok := src1.(Reg)
|
|
if !ok || !vvvvReg.isVec() {
|
|
return fmt.Errorf("VEX vvvv operand must be a vector register")
|
|
}
|
|
|
|
regField := dstReg.idx & 7
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
vvvvBar := 15 - (vvvvReg.idx & 15)
|
|
return e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2)
|
|
}
|
|
|
|
// encodeVexRM encodes the two-operand form: OP src, dst (no vvvv source).
|
|
// ModRM.reg = dst, ModRM.rm = src; the vector length comes from whichever
|
|
// operand is a vector register (the destination for extends/broadcasts, the
|
|
// source for the move-mask instructions whose destination is a GPR).
|
|
func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 2 {
|
|
return fmt.Errorf("VEX two-operand instruction expects 2 operands, got %d", len(ops))
|
|
}
|
|
src, dst := ops[0], ops[1]
|
|
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok {
|
|
return fmt.Errorf("VEX destination must be a register")
|
|
}
|
|
regField := dstReg.idx & 7
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
|
|
// Vector length: from the destination if it is a vector, otherwise from the
|
|
// source (move-mask instructions have a GPR destination and a vector source).
|
|
l := 0
|
|
if dstReg.isVec() {
|
|
l = dstReg.vecLenBit()
|
|
} else if srcReg, ok := src.(Reg); ok && srcReg.isVec() {
|
|
l = srcReg.vecLenBit()
|
|
}
|
|
|
|
// An unused vvvv field must be stored as all ones (v̄vvv = 1111); the
|
|
// hardware raises #UD on any other value.
|
|
return e.emitVexFields(spec, l, regField, rBit, 15, src)
|
|
}
|
|
|
|
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
|
|
// the destination always XMM and the VEX.L bit following the source, fixed
|
|
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
|
|
// the source is memory.
|
|
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 2 {
|
|
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
|
|
}
|
|
src, dst := ops[0], ops[1]
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || !dstReg.isVec() {
|
|
return fmt.Errorf("VEX destination must be a vector register")
|
|
}
|
|
ll, ok := vexSrcLen[mnem]
|
|
if !ok {
|
|
return fmt.Errorf("no fixed vector length for %s", mnem)
|
|
}
|
|
regField := dstReg.idx & 7
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
// An unused vvvv field must be stored as all ones (v̄vvv = 1111).
|
|
return e.emitVexFields(spec, ll, regField, rBit, 15, src)
|
|
}
|
|
|
|
// encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst.
|
|
// The destination is carried in VEX.vvvv, the source in ModRM.rm, and the
|
|
// shift kind in the ModRM.reg /digit.
|
|
func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 3 {
|
|
return fmt.Errorf("VEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
|
|
}
|
|
imm, src, dst := ops[0], ops[1], ops[2]
|
|
immVal, ok := imm.(Imm)
|
|
if !ok {
|
|
return fmt.Errorf("shift count must be an immediate")
|
|
}
|
|
srcReg, ok := src.(Reg)
|
|
if !ok || !srcReg.isVec() {
|
|
return fmt.Errorf("shift source must be a vector register")
|
|
}
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || !dstReg.isVec() {
|
|
return fmt.Errorf("shift destination must be a vector register")
|
|
}
|
|
|
|
vvvvBar := 15 - (dstReg.idx & 15)
|
|
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, srcReg); err != nil {
|
|
return err
|
|
}
|
|
immByte, err := imm8(int64(immVal))
|
|
if err != nil {
|
|
return err
|
|
}
|
|
e.out = append(e.out, immByte)
|
|
return nil
|
|
}
|
|
|
|
// imm8 range-checks an immediate for an 8-bit field. Shuffle controls are
|
|
// unsigned bit masks, but the negative spelling ($-1 = all bits set) is
|
|
// accepted, so the accepted span is -128..255.
|
|
func imm8(v int64) (byte, error) {
|
|
if v < -128 || v > 255 {
|
|
return 0, fmt.Errorf("immediate $%d does not fit in 8 bits", v)
|
|
}
|
|
return byte(v), nil
|
|
}
|
|
|
|
// encodeVexImmRM encodes an immediate form with no vvvv source: OP $imm, src,
|
|
// dst (VPSHUFD, VPERMQ). ModRM.reg = dst, ModRM.rm = src, imm8 appended.
|
|
func (e *enc) encodeVexImmRM(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 3 {
|
|
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
|
|
}
|
|
imm, src, dst := ops[0], ops[1], ops[2]
|
|
immVal, ok := imm.(Imm)
|
|
if !ok {
|
|
return fmt.Errorf("shuffle control must be an immediate")
|
|
}
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || !dstReg.isVec() {
|
|
return fmt.Errorf("shuffle destination must be a vector register")
|
|
}
|
|
|
|
// The vector length follows the source when it is a vector register,
|
|
// otherwise the destination (a memory source carries no length).
|
|
l := dstReg.vecLenBit()
|
|
if srcReg, ok := src.(Reg); ok && srcReg.isVec() {
|
|
l = srcReg.vecLenBit()
|
|
}
|
|
regField := dstReg.idx & 7
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
if err := e.emitVexFields(spec, l, regField, rBit, 15, src); err != nil {
|
|
return err
|
|
}
|
|
immByte, err := imm8(int64(immVal))
|
|
if err != nil {
|
|
return err
|
|
}
|
|
e.out = append(e.out, immByte)
|
|
return nil
|
|
}
|
|
|
|
// encodeVexNDS3Imm encodes the three-operand plus immediate form: OP $imm,
|
|
// src2, src1, dst (VSHUFPD, VPERM2I128, VINSERTI128). ModRM.reg = dst,
|
|
// VEX.vvvv = src1, ModRM.rm = src2, imm8 appended.
|
|
func (e *enc) encodeVexNDS3Imm(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 4 {
|
|
return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops))
|
|
}
|
|
imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
|
|
immVal, ok := imm.(Imm)
|
|
if !ok {
|
|
return fmt.Errorf("shuffle control must be an immediate")
|
|
}
|
|
dstReg, ok := dst.(Reg)
|
|
if !ok || !dstReg.isVec() {
|
|
return fmt.Errorf("destination must be a vector register")
|
|
}
|
|
vvvvReg, ok := src1.(Reg)
|
|
if !ok || !vvvvReg.isVec() {
|
|
return fmt.Errorf("second source must be a vector register")
|
|
}
|
|
|
|
regField := dstReg.idx & 7
|
|
rBit := 0
|
|
if dstReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
vvvvBar := 15 - (vvvvReg.idx & 15)
|
|
if err := e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2); err != nil {
|
|
return err
|
|
}
|
|
immByte, err := imm8(int64(immVal))
|
|
if err != nil {
|
|
return err
|
|
}
|
|
e.out = append(e.out, immByte)
|
|
return nil
|
|
}
|
|
|
|
// encodeVexExtract encodes a lane extract: OP $imm, ysrc, xdst
|
|
// (VEXTRACTI128, VEXTRACTF128). The YMM source occupies ModRM.reg and the
|
|
// XMM (or memory) destination ModRM.rm; imm8 selects the lane.
|
|
func (e *enc) encodeVexExtract(spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 3 {
|
|
return fmt.Errorf("extract expects 3 operands ($imm, ysrc, xdst), got %d", len(ops))
|
|
}
|
|
imm, src, dst := ops[0], ops[1], ops[2]
|
|
immVal, ok := imm.(Imm)
|
|
if !ok {
|
|
return fmt.Errorf("extract lane must be an immediate")
|
|
}
|
|
srcReg, ok := src.(Reg)
|
|
if !ok || !srcReg.isVec() {
|
|
return fmt.Errorf("extract source must be a vector register")
|
|
}
|
|
|
|
regField := srcReg.idx & 7
|
|
rBit := 0
|
|
if srcReg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
if err := e.emitVexFields(spec, srcReg.vecLenBit(), regField, rBit, 15, dst); err != nil {
|
|
return err
|
|
}
|
|
immByte, err := imm8(int64(immVal))
|
|
if err != nil {
|
|
return err
|
|
}
|
|
e.out = append(e.out, immByte)
|
|
return nil
|
|
}
|
|
|
|
// encodeVexZero encodes a no-operand instruction (VZEROUPPER).
|
|
func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
|
|
if len(ops) != 0 {
|
|
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
|
|
}
|
|
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 0.
|
|
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|spec.pp), spec.opcode)
|
|
return nil
|
|
}
|
|
|
|
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
|
|
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
|
|
// move uses the store-form layout (reg = source, rm = destination), matching
|
|
// the Go assembler.
|
|
func (e *enc) encodeVexMove(mnem string, ms vexMoveSpec, ops []Operand) error {
|
|
if len(ops) != 2 {
|
|
return fmt.Errorf("VEX move expects 2 operands, got %d", len(ops))
|
|
}
|
|
src, dst := ops[0], ops[1]
|
|
srcReg, srcIsVec := vecReg(src)
|
|
dstReg, dstIsVec := vecReg(dst)
|
|
|
|
var reg Reg
|
|
var rm Operand
|
|
op, w := ms.store, ms.storeW
|
|
switch {
|
|
case srcIsVec && dstIsVec:
|
|
if !ms.vecOK {
|
|
return fmt.Errorf("%s does not take two vector registers", mnem)
|
|
}
|
|
if ms.xmmOnly && (srcReg.size == 32 || dstReg.size == 32) {
|
|
return fmt.Errorf("%s operates on XMM registers only", mnem)
|
|
}
|
|
if ms.regReg != 0 {
|
|
op, w = ms.regReg, ms.regW
|
|
}
|
|
reg, rm = srcReg, dst // store form: reg = source, rm = destination.
|
|
case srcIsVec:
|
|
// vector → memory, or → GPR (VMOVD/VMOVQ only).
|
|
if !validMoveOther(ms, dst) {
|
|
return fmt.Errorf("%s: invalid destination operand", mnem)
|
|
}
|
|
reg, rm = srcReg, dst
|
|
case dstIsVec:
|
|
// memory → vector, or GPR → vector (VMOVD/VMOVQ only).
|
|
if !validMoveOther(ms, src) {
|
|
return fmt.Errorf("%s: invalid source operand", mnem)
|
|
}
|
|
op, w = ms.load, ms.loadW
|
|
reg, rm = dstReg, src
|
|
default:
|
|
return fmt.Errorf("%s needs a vector register operand", mnem)
|
|
}
|
|
if ms.xmmOnly && reg.size == 32 {
|
|
return fmt.Errorf("%s operates on XMM registers only", mnem)
|
|
}
|
|
|
|
regField := reg.idx & 7
|
|
rBit := 0
|
|
if reg.idx >= 8 {
|
|
rBit = 1
|
|
}
|
|
spec := vexSpec{mapSel: ms.mapSel, opcode: op, w: w, pp: ms.pp, opdigit: -1}
|
|
return e.emitVexFields(spec, reg.vecLenBit(), regField, rBit, 15, rm)
|
|
}
|
|
|
|
// vecReg extracts a vector register from an operand.
|
|
func vecReg(op Operand) (Reg, bool) {
|
|
r, ok := op.(Reg)
|
|
return r, ok && r.isVec()
|
|
}
|
|
|
|
// vecOrMem reports whether op is a vector register or a memory reference.
|
|
func vecOrMem(op Operand) bool {
|
|
switch op.(type) {
|
|
case Mem, sbMem:
|
|
return true
|
|
}
|
|
r, ok := op.(Reg)
|
|
return ok && r.isVec()
|
|
}
|
|
|
|
// validMoveOther reports whether the non-vector operand of a move is
|
|
// acceptable: memory always is, a GPR only for VMOVD/VMOVQ.
|
|
func validMoveOther(ms vexMoveSpec, op Operand) bool {
|
|
switch o := op.(type) {
|
|
case Mem, sbMem:
|
|
return true
|
|
case Reg:
|
|
return ms.gprOK && !o.isVec()
|
|
}
|
|
return false
|
|
}
|
|
|
|
// emitVexFields emits the VEX prefix, opcode, ModR/M, SIB and displacement for
|
|
// the given precomputed fields. It is shared by every register/rm VEX form;
|
|
// immediate bytes are appended by the caller.
|
|
func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Operand) error {
|
|
if l > 1 {
|
|
return fmt.Errorf("ZMM operand requires an EVEX instruction")
|
|
}
|
|
var modrm, sib int
|
|
var disp []byte
|
|
var xBit, bBit int
|
|
var sb *sbRef
|
|
switch r := rm.(type) {
|
|
case Reg:
|
|
modrm = 0xC0 | regField<<3 | (r.idx & 7)
|
|
sib = -1
|
|
if r.idx >= 8 {
|
|
bBit = 1
|
|
}
|
|
case Mem:
|
|
var err error
|
|
modrm, sib, disp, xBit, bBit, err = memComponents(regField, r)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
case sbMem:
|
|
// RIP-relative static-symbol reference; disp32 patched at link time.
|
|
modrm = regField<<3 | 0x05
|
|
sib = -1
|
|
disp = le32(0)
|
|
sb = &sbRef{name: r.name, addend: r.addend}
|
|
default:
|
|
return fmt.Errorf("invalid VEX r/m operand")
|
|
}
|
|
|
|
if spec.mapSel == 1 && xBit == 0 && bBit == 0 && spec.w == 0 {
|
|
e.out = append(e.out, 0xC5, byte((1-rBit)<<7|vvvvBar<<3|l<<2|spec.pp))
|
|
} else {
|
|
e.out = append(e.out, 0xC4,
|
|
byte((1-rBit)<<7|(1-xBit)<<6|(1-bBit)<<5|spec.mapSel),
|
|
byte(spec.w<<7|vvvvBar<<3|l<<2|spec.pp))
|
|
}
|
|
e.out = append(e.out, spec.opcode, byte(modrm))
|
|
if sib >= 0 {
|
|
e.out = append(e.out, byte(sib))
|
|
}
|
|
if sb != nil {
|
|
e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend})
|
|
}
|
|
e.out = append(e.out, disp...)
|
|
return nil
|
|
}
|