Files
gasm-sdk/asm/evex.go
T

557 lines
18 KiB
Go
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "fmt"
// This file implements EVEX (AVX-512) instruction encoding: the four-byte
// EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the
// compressed disp8×N displacement, and the operand shapes the go-flac
// AVX-512 kernels use. Masking ({k}) and zeroing ({z}) are not supported —
// the kernels do not use them. K-register operands (mask destinations,
// KMOVW, KTESTW) are.
// evexSpec describes one EVEX instruction's encoding parameters. The form
// field reuses the vexForm shapes, which carry over unchanged.
type evexSpec struct {
mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A
opcode byte
w int
pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2
opdigit int // ModRM.reg /digit, or -1 when reg is a register
form vexForm // vexNDS3, vexRM, vexShiftImm, vexNDS3Imm, vexExtract
n [3]int // disp8×N multiplier per vector length (128/256/512)
}
// evexTable maps an upper-case mnemonic to its EVEX encoding. Mnemonics
// that also have a VEX form (VPADDD, VMOVUPD, …) are dispatched here only
// when an operand demands EVEX (a ZMM or K register); EVEX-only mnemonics
// (VPXORD, VALIGND, …) always encode through this table. The N multipliers
// are taken from the Go assembler's opcode tables, which are authoritative
// for byte-for-byte agreement.
var evexTable = map[string]evexSpec{
// EVEX.128/256/512.66.0F — integer arithmetic / logic, NDS form.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBQ": {1, 0xFB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPXORD": {1, 0xEF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPXORQ": {1, 0xEF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — packed double arithmetic.
"VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.512.66.0F3A — align (NDS + imm8).
"VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F — immediate shift (VPSRAD /4).
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — variable shift with an XMM count (VPSRAQ;
// the W bit distinguishes it from VPSRAD's E2 form).
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst,
// rm=src, no vvvv).
"VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
// the xmm/ymm/zmm destination lengths).
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.512.66.0F3A.W1 — lane extract (reg=ZMM source, rm=YMM/memory
// destination, imm8).
"VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
// EVEX.66.0F38 — more integer NDS forms (W distinguishes D/Q).
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
// EVEX.66.0F — immediate shift (VPSLLD /6).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.F3.0F38.W0 — narrowing stores: reg = wide source, rm = narrow
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
}
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
// depends on the source kind — a GPR source uses opReg, a memory source uses
// opMem with a disp8×N of n.
type evexBcastSpec struct {
mapSel int
opReg byte
opMem byte
w int
n int
}
var evexBcastTable = map[string]evexBcastSpec{
// EVEX.128/256/512.66.0F38 — broadcast a dword/qword to all lanes.
"VPBROADCASTD": {2, 0x7C, 0x58, 0, 4},
"VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8},
}
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
// move table).
type evexMoveSpec struct {
mapSel int
pp int
load byte // r/m → vector
store byte // vector → r/m
w int
n [3]int
}
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0 — unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
}
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
func isEvex(mnemUpper string) bool {
if _, ok := evexTable[mnemUpper]; ok {
return true
}
if _, ok := evexBcastTable[mnemUpper]; ok {
return true
}
_, ok := evexMoveTable[mnemUpper]
return ok
}
// evexRequired reports whether the operands force the EVEX encoding of a
// mnemonic that also has a VEX form: ZMM and K registers do, and so do
// register indices 16–31, which only EVEX can represent (X16–Y31 exist
// solely under AVX-512).
func evexRequired(upper string, ops []Operand) bool {
_, inVex := vexTable[upper]
_, inVexMove := vexMoveTable[upper]
if !inVex && !inVexMove {
return true // EVEX-only mnemonic
}
for _, op := range ops {
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
return true
}
}
return false
}
// encodeEvex encodes an EVEX instruction with operands in Plan 9 order.
func (e *enc) encodeEvex(mnemUpper string, ops []Operand) error {
if bs, ok := evexBcastTable[mnemUpper]; ok {
return e.encodeEvexBcast(bs, ops)
}
if ms, ok := evexMoveTable[mnemUpper]; ok {
return e.encodeEvexMove(mnemUpper, ms, ops)
}
spec, ok := evexTable[mnemUpper]
if !ok {
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
}
switch spec.form {
case vexNDS3:
return e.encodeEvexNDS3(spec, ops)
case vexRM:
return e.encodeEvexRM(spec, ops)
case vexRMRev:
return e.encodeEvexRMRev(spec, ops)
case vexShiftImm:
return e.encodeEvexShiftImm(spec, ops)
case vexNDS3Imm:
return e.encodeEvexNDS3Imm(spec, ops)
case vexExtract:
return e.encodeEvexExtract(spec, ops)
}
return fmt.Errorf("unhandled EVEX form for %s", mnemUpper)
}
// encodeEvexNDS3 encodes the three-operand NDS form: OP src2, src1, dst. The
// destination may be an opmask register (VPCMPEQD), in which case the vector
// length comes from the sources.
func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("EVEX NDS instruction expects 3 operands, got %d", len(ops))
}
src2, src1, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || (!dstReg.isVec() && !dstReg.mask) {
return fmt.Errorf("EVEX destination must be a vector or mask register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("EVEX vvvv operand must be a vector register")
}
ll := dstReg.vecLenBit()
if dstReg.mask {
ll = vvvvReg.vecLenBit()
if r, ok := src2.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
}
}
return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2)
}
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
// no vvvv), e.g. VCVTQQ2PD.
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("EVEX destination must be a vector register")
}
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src)
}
// encodeEvexShiftImm encodes an immediate shift: OP $imm, src, dst
// (ModRM.reg = /digit, vvvv = dst, rm = src, imm8), e.g. VPSRAD $31, Z3, Z5.
func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("EVEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shift count must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("shift source must be a vector register")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("shift destination must be a vector register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexNDS3Imm encodes OP $imm, src2, src1, dst (reg=dst, vvvv=src1,
// rm=src2, imm8), e.g. VALIGND.
func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand) error {
if len(ops) != 4 {
return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops))
}
imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shuffle control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("destination must be a vector register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("second source must be a vector register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, vvvvReg.idx, src2); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexExtract encodes OP $imm, zsrc, ydst (reg=ZMM source, rm=YMM/memory
// destination, imm8), e.g. VEXTRACTI64X4.
func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, zsrc, ydst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("extract source must be a vector register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
// the store-form opcode (reg = source, rm = destination), matching the Go
// assembler.
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcIsVec := vecReg(src)
dstReg, dstIsVec := vecReg(dst)
op := ms.store
var reg Reg
var rm Operand
switch {
case srcIsVec && dstIsVec:
reg, rm = srcReg, dst
case srcIsVec:
if !memOperand(dst) {
return fmt.Errorf("%s: invalid destination operand", mnem)
}
reg, rm = srcReg, dst
case dstIsVec:
if !memOperand(src) {
return fmt.Errorf("%s: invalid source operand", mnem)
}
op = ms.load
reg, rm = dstReg, src
default:
return fmt.Errorf("%s needs a vector register operand", mnem)
}
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm)
}
// memOperand reports whether op is a memory reference (including a
// static-symbol reference).
func memOperand(op Operand) bool {
switch op.(type) {
case Mem, sbMem:
return true
}
return false
}
// encodeEvexRMRev encodes the narrowing-store form: OP src, dst with the wide
// source in the reg field and the narrow destination in r/m (VPMOVDW/QD).
func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("EVEX store instruction expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("EVEX source must be a vector register")
}
return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst)
}
// encodeEvexBcast encodes VPBROADCASTD/Q: OP src, dst with the GPR or memory
// source broadcast to every lane of the vector destination.
func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("broadcast expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("broadcast destination must be a vector register")
}
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
switch src.(type) {
case Mem, sbMem:
spec.opcode = bs.opMem
spec.n = [3]int{bs.n, bs.n, bs.n}
case Reg:
spec.opcode = bs.opReg
default:
return fmt.Errorf("broadcast source must be a register or memory")
}
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src)
}
// emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement
// (disp8×N compressed) for the given precomputed fields. regIdx is the
// unextended reg-field register index, or a /digit (0–7); vvvvIdx is the
// vvvv register index, or -1 when unused.
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand) error {
if ll > 2 {
return fmt.Errorf("invalid vector length")
}
// reg-field extension bits (R̄, R'̄), inverted.
rBar, rPrimeBar := 1, 1
if regIdx&8 != 0 {
rBar = 0
}
if regIdx&16 != 0 {
rPrimeBar = 0
}
// vvvv (inverted) and its extension bit V'̄.
vBar, vPrimeBar := 15, 1
if vvvvIdx >= 0 {
vBar = 15 - (vvvvIdx & 15)
if vvvvIdx&16 != 0 {
vPrimeBar = 0
}
}
var modrm, sib int
var disp []byte
xBar, bBar := 1, 1
var sb *sbRef
switch r := rm.(type) {
case Reg:
// ModRM.mod = 11: rm[3] extends via B̄, rm[4] via X̄.
modrm = 0xC0 | (regIdx&7)<<3 | (r.idx & 7)
sib = -1
if r.idx&8 != 0 {
bBar = 0
}
if r.idx&16 != 0 {
xBar = 0
}
case Mem:
var err error
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
if err != nil {
return err
}
// An indexed memory operand carries index[4] in V'̄ (Go folds it
// together with vvvv[4] into the same bit).
if r.HasIndex && r.Index.idx&16 != 0 {
vPrimeBar = 0
}
case sbMem:
// RIP-relative static-symbol reference; disp32 patched at link time
// (no disp8 scaling for RIP-relative addressing).
modrm = (regIdx&7)<<3 | 0x05
sib = -1
disp = le32(0)
sb = &sbRef{name: r.name, addend: r.addend}
default:
return fmt.Errorf("invalid EVEX r/m operand")
}
p0 := byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4 | spec.mapSel)
p1 := byte(spec.w<<7 | vBar<<3 | 1<<2 | spec.pp)
p2 := byte(ll<<5 | vPrimeBar<<3) // z = 0, b = 0, aaa = 0
e.out = append(e.out, 0x62, p0, p1, p2, spec.opcode, byte(modrm))
if sib >= 0 {
e.out = append(e.out, byte(sib))
}
if sb != nil {
e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend})
}
e.out = append(e.out, disp...)
return nil
}
// memComponentsEvex computes the ModR/M byte (with the given reg field), the
// SIB byte (-1 if none), the displacement bytes and the (inverted sense)
// index/base extension bits for an EVEX memory operand. The displacement is
// compressed to disp8×N when it is a multiple of n and the quotient fits a
// signed byte; otherwise a full disp32 is used.
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
sib = -1
xBar, bBar = 1, 1 // inverted bits: 1 = no extension
if !m.HasBase && !m.HasIndex {
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
}
needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4)
var mod int
switch {
case !m.HasBase:
mod = 0
disp = le32(m.Disp)
case m.Base.idx&7 == 5 && m.Disp == 0:
mod = 1
disp = []byte{0}
case m.Disp == 0:
mod = 0
case n > 0 && m.Disp%int64(n) == 0 && m.Disp/int64(n) >= -128 && m.Disp/int64(n) <= 127:
mod = 1
disp = []byte{byte(int8(m.Disp / int64(n)))}
default:
mod = 2
disp = le32(m.Disp)
}
if needSIB {
idxField := 4 // 100 = no index
if m.HasIndex {
idxField = m.Index.idx & 7
if m.Index.idx&8 != 0 {
xBar = 0
}
}
baseField := 5 // 101 = no base (with mod=00 → disp32)
if m.HasBase {
baseField = m.Base.idx & 7
if m.Base.idx&8 != 0 {
bBar = 0
}
}
return mod<<6 | regField<<3 | 0x04, scaleBits(m.Scale)<<6 | idxField<<3 | baseField, disp, xBar, bBar, nil
}
if m.Base.idx&8 != 0 {
bBar = 0
}
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil
}
// encodeKmovw encodes KMOVW, whose opcode depends on the operand direction:
// 90 (k/mem → K), 91 (K → mem), 92 (GPR → K), 93 (K → GPR); k → k uses 90.
func (e *enc) encodeKmovw(ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("KMOVW expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcIsReg := src.(Reg)
dstReg, dstIsReg := dst.(Reg)
srcK := srcIsReg && srcReg.mask
dstK := dstIsReg && dstReg.mask
spec := vexSpec{mapSel: 1, w: 0, pp: 0, opdigit: -1}
switch {
case srcK && dstK:
spec.opcode = 0x90 // k ← k: reg = dst, rm = src
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
case srcK && dstIsReg:
spec.opcode = 0x93 // GPR ← k: reg = dst, rm = src
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15, src)
case srcK:
if _, ok := dst.(Mem); !ok {
return fmt.Errorf("KMOVW: invalid destination operand")
}
spec.opcode = 0x91 // mem ← k: reg = src, rm = dst
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
case dstK:
spec.opcode = 0x92 // k ← GPR/mem: reg = dst, rm = src
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
}
return fmt.Errorf("KMOVW requires a K register operand")
}