2026-07-10 13:20:49 +02:00
|
|
|
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
|
|
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
|
|
|
|
|
|
|
|
package asm
|
|
|
|
|
|
|
|
|
|
|
|
import "fmt"
|
|
|
|
|
|
|
|
|
|
|
|
// This file implements EVEX (AVX-512) instruction encoding: the four-byte
|
|
|
|
|
|
// EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the
|
|
|
|
|
|
// compressed disp8×N displacement, and the operand shapes the go-flac
|
|
|
|
|
|
// AVX-512 kernels use. Masking ({k}) and zeroing ({z}) are not supported —
|
|
|
|
|
|
// the kernels do not use them. K-register operands (mask destinations,
|
|
|
|
|
|
// KMOVW, KTESTW) are.
|
|
|
|
|
|
|
|
|
|
|
|
// evexSpec describes one EVEX instruction's encoding parameters. The form
|
|
|
|
|
|
// field reuses the vexForm shapes, which carry over unchanged.
|
|
|
|
|
|
type evexSpec struct {
|
|
|
|
|
|
mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A
|
|
|
|
|
|
opcode byte
|
|
|
|
|
|
w int
|
|
|
|
|
|
pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2
|
|
|
|
|
|
opdigit int // ModRM.reg /digit, or -1 when reg is a register
|
|
|
|
|
|
form vexForm // vexNDS3, vexRM, vexShiftImm, vexNDS3Imm, vexExtract
|
|
|
|
|
|
n [3]int // disp8×N multiplier per vector length (128/256/512)
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// evexTable maps an upper-case mnemonic to its EVEX encoding. Mnemonics
|
|
|
|
|
|
// that also have a VEX form (VPADDD, VMOVUPD, …) are dispatched here only
|
|
|
|
|
|
// when an operand demands EVEX (a ZMM or K register); EVEX-only mnemonics
|
|
|
|
|
|
// (VPXORD, VALIGND, …) always encode through this table. The N multipliers
|
|
|
|
|
|
// are taken from the Go assembler's opcode tables, which are authoritative
|
|
|
|
|
|
// for byte-for-byte agreement.
|
|
|
|
|
|
var evexTable = map[string]evexSpec{
|
|
|
|
|
|
// EVEX.128/256/512.66.0F — integer arithmetic / logic, NDS form.
|
|
|
|
|
|
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
"VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
"VPSUBQ": {1, 0xFB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
"VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
"VPXORD": {1, 0xEF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
"VPXORQ": {1, 0xEF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
|
|
|
|
|
|
// EVEX.128/256/512.66.0F.W1 — packed double arithmetic.
|
|
|
|
|
|
"VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
"VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
|
|
|
|
|
|
// EVEX.512.66.0F3A — align (NDS + imm8).
|
|
|
|
|
|
"VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
|
|
|
|
|
|
|
|
|
|
|
// EVEX.128/256/512.66.0F — immediate shift (VPSRAD /4).
|
|
|
|
|
|
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
|
|
|
|
|
|
// EVEX.128/256/512.66.0F.W1 — variable shift with an XMM count (VPSRAQ;
|
|
|
|
|
|
// the W bit distinguishes it from VPSRAD's E2 form).
|
|
|
|
|
|
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
|
|
|
|
|
|
// EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst,
|
|
|
|
|
|
// rm=src, no vvvv).
|
|
|
|
|
|
"VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
|
|
|
|
|
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
|
|
|
|
|
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
|
|
|
|
|
// the xmm/ymm/zmm destination lengths).
|
|
|
|
|
|
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
|
|
|
|
|
|
|
|
|
|
|
// EVEX.512.66.0F3A.W1 — lane extract (reg=ZMM source, rm=YMM/memory
|
|
|
|
|
|
// destination, imm8).
|
|
|
|
|
|
"VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
|
|
|
|
|
|
"VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
|
|
|
|
|
|
|
|
|
|
|
|
// EVEX.66.0F38 — more integer NDS forms (W distinguishes D/Q).
|
|
|
|
|
|
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
"VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
|
|
|
|
|
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
|
|
|
|
|
|
|
|
|
|
|
|
// EVEX.66.0F — immediate shift (VPSLLD /6).
|
|
|
|
|
|
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
|
|
|
|
|
|
|
|
|
|
|
|
// EVEX.F3.0F38.W0 — narrowing stores: reg = wide source, rm = narrow
|
|
|
|
|
|
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
|
|
|
|
|
|
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
|
|
|
|
|
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
|
|
|
|
|
|
// depends on the source kind — a GPR source uses opReg, a memory source uses
|
|
|
|
|
|
// opMem with a disp8×N of n.
|
|
|
|
|
|
type evexBcastSpec struct {
|
|
|
|
|
|
mapSel int
|
|
|
|
|
|
opReg byte
|
|
|
|
|
|
opMem byte
|
|
|
|
|
|
w int
|
|
|
|
|
|
n int
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
var evexBcastTable = map[string]evexBcastSpec{
|
|
|
|
|
|
// EVEX.128/256/512.66.0F38 — broadcast a dword/qword to all lanes.
|
|
|
|
|
|
"VPBROADCASTD": {2, 0x7C, 0x58, 0, 4},
|
|
|
|
|
|
"VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8},
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
|
|
|
|
|
|
// move table).
|
|
|
|
|
|
type evexMoveSpec struct {
|
|
|
|
|
|
mapSel int
|
|
|
|
|
|
pp int
|
|
|
|
|
|
load byte // r/m → vector
|
|
|
|
|
|
store byte // vector → r/m
|
|
|
|
|
|
w int
|
|
|
|
|
|
n [3]int
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
|
|
|
|
|
|
var evexMoveTable = map[string]evexMoveSpec{
|
|
|
|
|
|
// EVEX.128/256/512.F3.0F.W0 — unaligned integer move.
|
|
|
|
|
|
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
2026-07-11 17:36:52 +02:00
|
|
|
|
// EVEX.128/256/512.F3.0F.W1 — unaligned qword move.
|
|
|
|
|
|
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
2026-07-10 13:20:49 +02:00
|
|
|
|
// EVEX.128/256/512.66.0F.W1 — unaligned packed double move.
|
|
|
|
|
|
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
|
|
|
|
|
func isEvex(mnemUpper string) bool {
|
|
|
|
|
|
if _, ok := evexTable[mnemUpper]; ok {
|
|
|
|
|
|
return true
|
|
|
|
|
|
}
|
|
|
|
|
|
if _, ok := evexBcastTable[mnemUpper]; ok {
|
|
|
|
|
|
return true
|
|
|
|
|
|
}
|
|
|
|
|
|
_, ok := evexMoveTable[mnemUpper]
|
|
|
|
|
|
return ok
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// evexRequired reports whether the operands force the EVEX encoding of a
|
|
|
|
|
|
// mnemonic that also has a VEX form: ZMM and K registers do, and so do
|
|
|
|
|
|
// register indices 16–31, which only EVEX can represent (X16–Y31 exist
|
|
|
|
|
|
// solely under AVX-512).
|
|
|
|
|
|
func evexRequired(upper string, ops []Operand) bool {
|
|
|
|
|
|
_, inVex := vexTable[upper]
|
|
|
|
|
|
_, inVexMove := vexMoveTable[upper]
|
|
|
|
|
|
if !inVex && !inVexMove {
|
|
|
|
|
|
return true // EVEX-only mnemonic
|
|
|
|
|
|
}
|
|
|
|
|
|
for _, op := range ops {
|
|
|
|
|
|
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
|
|
|
|
|
|
return true
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
return false
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// encodeEvex encodes an EVEX instruction with operands in Plan 9 order.
|
|
|
|
|
|
func (e *enc) encodeEvex(mnemUpper string, ops []Operand) error {
|
|
|
|
|
|
if bs, ok := evexBcastTable[mnemUpper]; ok {
|
|
|
|
|
|
return e.encodeEvexBcast(bs, ops)
|
|
|
|
|
|
}
|
|
|
|
|
|
if ms, ok := evexMoveTable[mnemUpper]; ok {
|
|
|
|
|
|
return e.encodeEvexMove(mnemUpper, ms, ops)
|
|
|
|
|
|
}
|
|
|
|
|
|
spec, ok := evexTable[mnemUpper]
|
|
|
|
|
|
if !ok {
|
|
|
|
|
|
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
|
|
|
|
|
|
}
|
|
|
|
|
|
switch spec.form {
|
|
|
|
|
|
case vexNDS3:
|
|
|
|
|
|
return e.encodeEvexNDS3(spec, ops)
|
|
|
|
|
|
case vexRM:
|
|
|
|
|
|
return e.encodeEvexRM(spec, ops)
|
|
|
|
|
|
case vexRMRev:
|
|
|
|
|
|
return e.encodeEvexRMRev(spec, ops)
|
|
|
|
|
|
case vexShiftImm:
|
|
|
|
|
|
return e.encodeEvexShiftImm(spec, ops)
|
|
|
|
|
|
case vexNDS3Imm:
|
|
|
|
|
|
return e.encodeEvexNDS3Imm(spec, ops)
|
|
|
|
|
|
case vexExtract:
|
|
|
|
|
|
return e.encodeEvexExtract(spec, ops)
|
|
|
|
|
|
}
|
|
|
|
|
|
return fmt.Errorf("unhandled EVEX form for %s", mnemUpper)
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// encodeEvexNDS3 encodes the three-operand NDS form: OP src2, src1, dst. The
|
|
|
|
|
|
// destination may be an opmask register (VPCMPEQD), in which case the vector
|
|
|
|
|
|
// length comes from the sources.
|
|
|
|
|
|
func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand) error {
|
|
|
|
|
|
if len(ops) != 3 {
|
|
|
|
|
|
return fmt.Errorf("EVEX NDS instruction expects 3 operands, got %d", len(ops))
|
|
|
|
|
|
}
|
|
|
|
|
|
src2, src1, dst := ops[0], ops[1], ops[2]
|
|
|
|
|
|
dstReg, ok := dst.(Reg)
|
|
|
|
|
|
if !ok || (!dstReg.isVec() && !dstReg.mask) {
|
|
|
|
|
|
return fmt.Errorf("EVEX destination must be a vector or mask register")
|
|
|
|
|
|
}
|
|
|
|
|
|
vvvvReg, ok := src1.(Reg)
|
|
|
|
|
|
if !ok || !vvvvReg.isVec() {
|
|
|
|
|
|
return fmt.Errorf("EVEX vvvv operand must be a vector register")
|
|
|
|
|
|
}
|
|
|
|
|
|
ll := dstReg.vecLenBit()
|
|
|
|
|
|
if dstReg.mask {
|
|
|
|
|
|
ll = vvvvReg.vecLenBit()
|
|
|
|
|
|
if r, ok := src2.(Reg); ok && r.isVec() {
|
|
|
|
|
|
ll = r.vecLenBit()
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2)
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
|
|
|
|
|
|
// no vvvv), e.g. VCVTQQ2PD.
|
|
|
|
|
|
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand) error {
|
|
|
|
|
|
if len(ops) != 2 {
|
|
|
|
|
|
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
|
|
|
|
|
|
}
|
|
|
|
|
|
src, dst := ops[0], ops[1]
|
|
|
|
|
|
dstReg, ok := dst.(Reg)
|
|
|
|
|
|
if !ok || !dstReg.isVec() {
|
|
|
|
|
|
return fmt.Errorf("EVEX destination must be a vector register")
|
|
|
|
|
|
}
|
|
|
|
|
|
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src)
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// encodeEvexShiftImm encodes an immediate shift: OP $imm, src, dst
|
|
|
|
|
|
// (ModRM.reg = /digit, vvvv = dst, rm = src, imm8), e.g. VPSRAD $31, Z3, Z5.
|
|
|
|
|
|
func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand) error {
|
|
|
|
|
|
if len(ops) != 3 {
|
|
|
|
|
|
return fmt.Errorf("EVEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
|
|
|
|
|
|
}
|
|
|
|
|
|
imm, src, dst := ops[0], ops[1], ops[2]
|
|
|
|
|
|
immVal, ok := imm.(Imm)
|
|
|
|
|
|
if !ok {
|
|
|
|
|
|
return fmt.Errorf("shift count must be an immediate")
|
|
|
|
|
|
}
|
|
|
|
|
|
srcReg, ok := src.(Reg)
|
|
|
|
|
|
if !ok || !srcReg.isVec() {
|
|
|
|
|
|
return fmt.Errorf("shift source must be a vector register")
|
|
|
|
|
|
}
|
|
|
|
|
|
dstReg, ok := dst.(Reg)
|
|
|
|
|
|
if !ok || !dstReg.isVec() {
|
|
|
|
|
|
return fmt.Errorf("shift destination must be a vector register")
|
|
|
|
|
|
}
|
|
|
|
|
|
immByte, err := imm8(int64(immVal))
|
|
|
|
|
|
if err != nil {
|
|
|
|
|
|
return err
|
|
|
|
|
|
}
|
|
|
|
|
|
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg); err != nil {
|
|
|
|
|
|
return err
|
|
|
|
|
|
}
|
|
|
|
|
|
e.out = append(e.out, immByte)
|
|
|
|
|
|
return nil
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// encodeEvexNDS3Imm encodes OP $imm, src2, src1, dst (reg=dst, vvvv=src1,
|
|
|
|
|
|
// rm=src2, imm8), e.g. VALIGND.
|
|
|
|
|
|
func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand) error {
|
|
|
|
|
|
if len(ops) != 4 {
|
|
|
|
|
|
return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops))
|
|
|
|
|
|
}
|
|
|
|
|
|
imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
|
|
|
|
|
|
immVal, ok := imm.(Imm)
|
|
|
|
|
|
if !ok {
|
|
|
|
|
|
return fmt.Errorf("shuffle control must be an immediate")
|
|
|
|
|
|
}
|
|
|
|
|
|
dstReg, ok := dst.(Reg)
|
|
|
|
|
|
if !ok || !dstReg.isVec() {
|
|
|
|
|
|
return fmt.Errorf("destination must be a vector register")
|
|
|
|
|
|
}
|
|
|
|
|
|
vvvvReg, ok := src1.(Reg)
|
|
|
|
|
|
if !ok || !vvvvReg.isVec() {
|
|
|
|
|
|
return fmt.Errorf("second source must be a vector register")
|
|
|
|
|
|
}
|
|
|
|
|
|
immByte, err := imm8(int64(immVal))
|
|
|
|
|
|
if err != nil {
|
|
|
|
|
|
return err
|
|
|
|
|
|
}
|
|
|
|
|
|
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, vvvvReg.idx, src2); err != nil {
|
|
|
|
|
|
return err
|
|
|
|
|
|
}
|
|
|
|
|
|
e.out = append(e.out, immByte)
|
|
|
|
|
|
return nil
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// encodeEvexExtract encodes OP $imm, zsrc, ydst (reg=ZMM source, rm=YMM/memory
|
|
|
|
|
|
// destination, imm8), e.g. VEXTRACTI64X4.
|
|
|
|
|
|
func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand) error {
|
|
|
|
|
|
if len(ops) != 3 {
|
|
|
|
|
|
return fmt.Errorf("extract expects 3 operands ($imm, zsrc, ydst), got %d", len(ops))
|
|
|
|
|
|
}
|
|
|
|
|
|
imm, src, dst := ops[0], ops[1], ops[2]
|
|
|
|
|
|
immVal, ok := imm.(Imm)
|
|
|
|
|
|
if !ok {
|
|
|
|
|
|
return fmt.Errorf("extract lane must be an immediate")
|
|
|
|
|
|
}
|
|
|
|
|
|
srcReg, ok := src.(Reg)
|
|
|
|
|
|
if !ok || !srcReg.isVec() {
|
|
|
|
|
|
return fmt.Errorf("extract source must be a vector register")
|
|
|
|
|
|
}
|
|
|
|
|
|
immByte, err := imm8(int64(immVal))
|
|
|
|
|
|
if err != nil {
|
|
|
|
|
|
return err
|
|
|
|
|
|
}
|
|
|
|
|
|
if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst); err != nil {
|
|
|
|
|
|
return err
|
|
|
|
|
|
}
|
|
|
|
|
|
e.out = append(e.out, immByte)
|
|
|
|
|
|
return nil
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
|
|
|
|
|
|
// the store-form opcode (reg = source, rm = destination), matching the Go
|
|
|
|
|
|
// assembler.
|
|
|
|
|
|
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand) error {
|
|
|
|
|
|
if len(ops) != 2 {
|
|
|
|
|
|
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
|
|
|
|
|
|
}
|
|
|
|
|
|
src, dst := ops[0], ops[1]
|
|
|
|
|
|
srcReg, srcIsVec := vecReg(src)
|
|
|
|
|
|
dstReg, dstIsVec := vecReg(dst)
|
|
|
|
|
|
|
|
|
|
|
|
op := ms.store
|
|
|
|
|
|
var reg Reg
|
|
|
|
|
|
var rm Operand
|
|
|
|
|
|
switch {
|
|
|
|
|
|
case srcIsVec && dstIsVec:
|
|
|
|
|
|
reg, rm = srcReg, dst
|
|
|
|
|
|
case srcIsVec:
|
|
|
|
|
|
if !memOperand(dst) {
|
|
|
|
|
|
return fmt.Errorf("%s: invalid destination operand", mnem)
|
|
|
|
|
|
}
|
|
|
|
|
|
reg, rm = srcReg, dst
|
|
|
|
|
|
case dstIsVec:
|
|
|
|
|
|
if !memOperand(src) {
|
|
|
|
|
|
return fmt.Errorf("%s: invalid source operand", mnem)
|
|
|
|
|
|
}
|
|
|
|
|
|
op = ms.load
|
|
|
|
|
|
reg, rm = dstReg, src
|
|
|
|
|
|
default:
|
|
|
|
|
|
return fmt.Errorf("%s needs a vector register operand", mnem)
|
|
|
|
|
|
}
|
|
|
|
|
|
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
|
|
|
|
|
|
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm)
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// memOperand reports whether op is a memory reference (including a
|
|
|
|
|
|
// static-symbol reference).
|
|
|
|
|
|
func memOperand(op Operand) bool {
|
|
|
|
|
|
switch op.(type) {
|
|
|
|
|
|
case Mem, sbMem:
|
|
|
|
|
|
return true
|
|
|
|
|
|
}
|
|
|
|
|
|
return false
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// encodeEvexRMRev encodes the narrowing-store form: OP src, dst with the wide
|
|
|
|
|
|
// source in the reg field and the narrow destination in r/m (VPMOVDW/QD).
|
|
|
|
|
|
func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand) error {
|
|
|
|
|
|
if len(ops) != 2 {
|
|
|
|
|
|
return fmt.Errorf("EVEX store instruction expects 2 operands, got %d", len(ops))
|
|
|
|
|
|
}
|
|
|
|
|
|
src, dst := ops[0], ops[1]
|
|
|
|
|
|
srcReg, ok := src.(Reg)
|
|
|
|
|
|
if !ok || !srcReg.isVec() {
|
|
|
|
|
|
return fmt.Errorf("EVEX source must be a vector register")
|
|
|
|
|
|
}
|
|
|
|
|
|
return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst)
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// encodeEvexBcast encodes VPBROADCASTD/Q: OP src, dst with the GPR or memory
|
|
|
|
|
|
// source broadcast to every lane of the vector destination.
|
|
|
|
|
|
func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand) error {
|
|
|
|
|
|
if len(ops) != 2 {
|
|
|
|
|
|
return fmt.Errorf("broadcast expects 2 operands, got %d", len(ops))
|
|
|
|
|
|
}
|
|
|
|
|
|
src, dst := ops[0], ops[1]
|
|
|
|
|
|
dstReg, ok := dst.(Reg)
|
|
|
|
|
|
if !ok || !dstReg.isVec() {
|
|
|
|
|
|
return fmt.Errorf("broadcast destination must be a vector register")
|
|
|
|
|
|
}
|
|
|
|
|
|
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
|
|
|
|
|
|
switch src.(type) {
|
|
|
|
|
|
case Mem, sbMem:
|
|
|
|
|
|
spec.opcode = bs.opMem
|
|
|
|
|
|
spec.n = [3]int{bs.n, bs.n, bs.n}
|
|
|
|
|
|
case Reg:
|
|
|
|
|
|
spec.opcode = bs.opReg
|
|
|
|
|
|
default:
|
|
|
|
|
|
return fmt.Errorf("broadcast source must be a register or memory")
|
|
|
|
|
|
}
|
|
|
|
|
|
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src)
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement
|
|
|
|
|
|
// (disp8×N compressed) for the given precomputed fields. regIdx is the
|
|
|
|
|
|
// unextended reg-field register index, or a /digit (0–7); vvvvIdx is the
|
|
|
|
|
|
// vvvv register index, or -1 when unused.
|
|
|
|
|
|
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand) error {
|
|
|
|
|
|
if ll > 2 {
|
|
|
|
|
|
return fmt.Errorf("invalid vector length")
|
|
|
|
|
|
}
|
|
|
|
|
|
// reg-field extension bits (R̄, R'̄), inverted.
|
|
|
|
|
|
rBar, rPrimeBar := 1, 1
|
|
|
|
|
|
if regIdx&8 != 0 {
|
|
|
|
|
|
rBar = 0
|
|
|
|
|
|
}
|
|
|
|
|
|
if regIdx&16 != 0 {
|
|
|
|
|
|
rPrimeBar = 0
|
|
|
|
|
|
}
|
|
|
|
|
|
// vvvv (inverted) and its extension bit V'̄.
|
|
|
|
|
|
vBar, vPrimeBar := 15, 1
|
|
|
|
|
|
if vvvvIdx >= 0 {
|
|
|
|
|
|
vBar = 15 - (vvvvIdx & 15)
|
|
|
|
|
|
if vvvvIdx&16 != 0 {
|
|
|
|
|
|
vPrimeBar = 0
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
var modrm, sib int
|
|
|
|
|
|
var disp []byte
|
|
|
|
|
|
xBar, bBar := 1, 1
|
|
|
|
|
|
var sb *sbRef
|
|
|
|
|
|
switch r := rm.(type) {
|
|
|
|
|
|
case Reg:
|
|
|
|
|
|
// ModRM.mod = 11: rm[3] extends via B̄, rm[4] via X̄.
|
|
|
|
|
|
modrm = 0xC0 | (regIdx&7)<<3 | (r.idx & 7)
|
|
|
|
|
|
sib = -1
|
|
|
|
|
|
if r.idx&8 != 0 {
|
|
|
|
|
|
bBar = 0
|
|
|
|
|
|
}
|
|
|
|
|
|
if r.idx&16 != 0 {
|
|
|
|
|
|
xBar = 0
|
|
|
|
|
|
}
|
|
|
|
|
|
case Mem:
|
|
|
|
|
|
var err error
|
|
|
|
|
|
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
|
|
|
|
|
|
if err != nil {
|
|
|
|
|
|
return err
|
|
|
|
|
|
}
|
|
|
|
|
|
// An indexed memory operand carries index[4] in V'̄ (Go folds it
|
|
|
|
|
|
// together with vvvv[4] into the same bit).
|
|
|
|
|
|
if r.HasIndex && r.Index.idx&16 != 0 {
|
|
|
|
|
|
vPrimeBar = 0
|
|
|
|
|
|
}
|
|
|
|
|
|
case sbMem:
|
|
|
|
|
|
// RIP-relative static-symbol reference; disp32 patched at link time
|
|
|
|
|
|
// (no disp8 scaling for RIP-relative addressing).
|
|
|
|
|
|
modrm = (regIdx&7)<<3 | 0x05
|
|
|
|
|
|
sib = -1
|
|
|
|
|
|
disp = le32(0)
|
|
|
|
|
|
sb = &sbRef{name: r.name, addend: r.addend}
|
|
|
|
|
|
default:
|
|
|
|
|
|
return fmt.Errorf("invalid EVEX r/m operand")
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
p0 := byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4 | spec.mapSel)
|
|
|
|
|
|
p1 := byte(spec.w<<7 | vBar<<3 | 1<<2 | spec.pp)
|
|
|
|
|
|
p2 := byte(ll<<5 | vPrimeBar<<3) // z = 0, b = 0, aaa = 0
|
|
|
|
|
|
e.out = append(e.out, 0x62, p0, p1, p2, spec.opcode, byte(modrm))
|
|
|
|
|
|
if sib >= 0 {
|
|
|
|
|
|
e.out = append(e.out, byte(sib))
|
|
|
|
|
|
}
|
|
|
|
|
|
if sb != nil {
|
|
|
|
|
|
e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend})
|
|
|
|
|
|
}
|
|
|
|
|
|
e.out = append(e.out, disp...)
|
|
|
|
|
|
return nil
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// memComponentsEvex computes the ModR/M byte (with the given reg field), the
|
|
|
|
|
|
// SIB byte (-1 if none), the displacement bytes and the (inverted sense)
|
|
|
|
|
|
// index/base extension bits for an EVEX memory operand. The displacement is
|
|
|
|
|
|
// compressed to disp8×N when it is a multiple of n and the quotient fits a
|
|
|
|
|
|
// signed byte; otherwise a full disp32 is used.
|
|
|
|
|
|
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
|
|
|
|
|
|
sib = -1
|
|
|
|
|
|
xBar, bBar = 1, 1 // inverted bits: 1 = no extension
|
|
|
|
|
|
if !m.HasBase && !m.HasIndex {
|
|
|
|
|
|
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4)
|
|
|
|
|
|
|
|
|
|
|
|
var mod int
|
|
|
|
|
|
switch {
|
|
|
|
|
|
case !m.HasBase:
|
|
|
|
|
|
mod = 0
|
|
|
|
|
|
disp = le32(m.Disp)
|
|
|
|
|
|
case m.Base.idx&7 == 5 && m.Disp == 0:
|
|
|
|
|
|
mod = 1
|
|
|
|
|
|
disp = []byte{0}
|
|
|
|
|
|
case m.Disp == 0:
|
|
|
|
|
|
mod = 0
|
|
|
|
|
|
case n > 0 && m.Disp%int64(n) == 0 && m.Disp/int64(n) >= -128 && m.Disp/int64(n) <= 127:
|
|
|
|
|
|
mod = 1
|
|
|
|
|
|
disp = []byte{byte(int8(m.Disp / int64(n)))}
|
|
|
|
|
|
default:
|
|
|
|
|
|
mod = 2
|
|
|
|
|
|
disp = le32(m.Disp)
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
if needSIB {
|
|
|
|
|
|
idxField := 4 // 100 = no index
|
|
|
|
|
|
if m.HasIndex {
|
|
|
|
|
|
idxField = m.Index.idx & 7
|
|
|
|
|
|
if m.Index.idx&8 != 0 {
|
|
|
|
|
|
xBar = 0
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
baseField := 5 // 101 = no base (with mod=00 → disp32)
|
|
|
|
|
|
if m.HasBase {
|
|
|
|
|
|
baseField = m.Base.idx & 7
|
|
|
|
|
|
if m.Base.idx&8 != 0 {
|
|
|
|
|
|
bBar = 0
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
return mod<<6 | regField<<3 | 0x04, scaleBits(m.Scale)<<6 | idxField<<3 | baseField, disp, xBar, bBar, nil
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
if m.Base.idx&8 != 0 {
|
|
|
|
|
|
bBar = 0
|
|
|
|
|
|
}
|
|
|
|
|
|
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// encodeKmovw encodes KMOVW, whose opcode depends on the operand direction:
|
|
|
|
|
|
// 90 (k/mem → K), 91 (K → mem), 92 (GPR → K), 93 (K → GPR); k → k uses 90.
|
|
|
|
|
|
func (e *enc) encodeKmovw(ops []Operand) error {
|
|
|
|
|
|
if len(ops) != 2 {
|
|
|
|
|
|
return fmt.Errorf("KMOVW expects 2 operands, got %d", len(ops))
|
|
|
|
|
|
}
|
|
|
|
|
|
src, dst := ops[0], ops[1]
|
|
|
|
|
|
srcReg, srcIsReg := src.(Reg)
|
|
|
|
|
|
dstReg, dstIsReg := dst.(Reg)
|
|
|
|
|
|
srcK := srcIsReg && srcReg.mask
|
|
|
|
|
|
dstK := dstIsReg && dstReg.mask
|
|
|
|
|
|
spec := vexSpec{mapSel: 1, w: 0, pp: 0, opdigit: -1}
|
|
|
|
|
|
switch {
|
|
|
|
|
|
case srcK && dstK:
|
|
|
|
|
|
spec.opcode = 0x90 // k ← k: reg = dst, rm = src
|
|
|
|
|
|
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
|
|
|
|
|
case srcK && dstIsReg:
|
|
|
|
|
|
spec.opcode = 0x93 // GPR ← k: reg = dst, rm = src
|
|
|
|
|
|
rBit := 0
|
|
|
|
|
|
if dstReg.idx >= 8 {
|
|
|
|
|
|
rBit = 1
|
|
|
|
|
|
}
|
|
|
|
|
|
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15, src)
|
|
|
|
|
|
case srcK:
|
|
|
|
|
|
if _, ok := dst.(Mem); !ok {
|
|
|
|
|
|
return fmt.Errorf("KMOVW: invalid destination operand")
|
|
|
|
|
|
}
|
|
|
|
|
|
spec.opcode = 0x91 // mem ← k: reg = src, rm = dst
|
|
|
|
|
|
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
|
|
|
|
|
|
case dstK:
|
|
|
|
|
|
spec.opcode = 0x92 // k ← GPR/mem: reg = dst, rm = src
|
|
|
|
|
|
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
|
|
|
|
|
}
|
|
|
|
|
|
return fmt.Errorf("KMOVW requires a K register operand")
|
|
|
|
|
|
}
|