// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import "fmt" // This file implements EVEX (AVX-512) instruction encoding: the four-byte // EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the // compressed disp8×N displacement, and the operand shapes the go-flac // AVX-512 kernels use. Masking ({k}) and zeroing ({z}) are not supported — // the kernels do not use them. K-register operands (mask destinations, // KMOVW, KTESTW) are. // evexSpec describes one EVEX instruction's encoding parameters. The form // field reuses the vexForm shapes, which carry over unchanged. type evexSpec struct { mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A opcode byte w int pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2 opdigit int // ModRM.reg /digit, or -1 when reg is a register form vexForm // vexNDS3, vexRM, vexShiftImm, vexNDS3Imm, vexExtract n [3]int // disp8×N multiplier per vector length (128/256/512) } // evexTable maps an upper-case mnemonic to its EVEX encoding. Mnemonics // that also have a VEX form (VPADDD, VMOVUPD, …) are dispatched here only // when an operand demands EVEX (a ZMM or K register); EVEX-only mnemonics // (VPXORD, VALIGND, …) always encode through this table. The N multipliers // are taken from the Go assembler's opcode tables, which are authoritative // for byte-for-byte agreement. var evexTable = map[string]evexSpec{ // EVEX.128/256/512.66.0F — integer arithmetic / logic, NDS form. "VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSUBQ": {1, 0xFB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPXORD": {1, 0xEF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPXORQ": {1, 0xEF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F.W1 — packed double arithmetic. "VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, // EVEX.512.66.0F3A — align (NDS + imm8). "VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F — immediate shift (VPSRAD /4). "VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F.W1 — variable shift with an XMM count (VPSRAQ; // the W bit distinguishes it from VPSRAD's E2 form). "VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, // EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst, // rm=src, no vvvv). "VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory // operand is the narrow source, so disp8×N follows its size (8/16/32 for // the xmm/ymm/zmm destination lengths). "VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, // EVEX.512.66.0F3A.W1 — lane extract (reg=ZMM source, rm=YMM/memory // destination, imm8). "VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}}, "VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}}, // EVEX.66.0F38 — more integer NDS forms (W distinguishes D/Q). "VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}}, // EVEX.66.0F — immediate shift (VPSLLD /6). "VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}}, // EVEX.F3.0F38.W0 — narrowing stores: reg = wide source, rm = narrow // destination (VPMOVDW dword→word, VPMOVQD qword→dword). "VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, "VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, } // evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode // depends on the source kind — a GPR source uses opReg, a memory source uses // opMem with a disp8×N of n. type evexBcastSpec struct { mapSel int opReg byte opMem byte w int n int } var evexBcastTable = map[string]evexBcastSpec{ // EVEX.128/256/512.66.0F38 — broadcast a dword/qword to all lanes. "VPBROADCASTD": {2, 0x7C, 0x58, 0, 4}, "VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8}, } // evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX // move table). type evexMoveSpec struct { mapSel int pp int load byte // r/m → vector store byte // vector → r/m w int n [3]int } // evexMoveTable maps an upper-case EVEX move mnemonic to its encoding. var evexMoveTable = map[string]evexMoveSpec{ // EVEX.128/256/512.F3.0F.W0 — unaligned integer move. "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F.W1 — unaligned packed double move. "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}}, } // isEvex reports whether the mnemonic has an EVEX encoding we handle. func isEvex(mnemUpper string) bool { if _, ok := evexTable[mnemUpper]; ok { return true } if _, ok := evexBcastTable[mnemUpper]; ok { return true } _, ok := evexMoveTable[mnemUpper] return ok } // evexRequired reports whether the operands force the EVEX encoding of a // mnemonic that also has a VEX form: ZMM and K registers do, and so do // register indices 16–31, which only EVEX can represent (X16–Y31 exist // solely under AVX-512). func evexRequired(upper string, ops []Operand) bool { _, inVex := vexTable[upper] _, inVexMove := vexMoveTable[upper] if !inVex && !inVexMove { return true // EVEX-only mnemonic } for _, op := range ops { if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) { return true } } return false } // encodeEvex encodes an EVEX instruction with operands in Plan 9 order. func (e *enc) encodeEvex(mnemUpper string, ops []Operand) error { if bs, ok := evexBcastTable[mnemUpper]; ok { return e.encodeEvexBcast(bs, ops) } if ms, ok := evexMoveTable[mnemUpper]; ok { return e.encodeEvexMove(mnemUpper, ms, ops) } spec, ok := evexTable[mnemUpper] if !ok { return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper) } switch spec.form { case vexNDS3: return e.encodeEvexNDS3(spec, ops) case vexRM: return e.encodeEvexRM(spec, ops) case vexRMRev: return e.encodeEvexRMRev(spec, ops) case vexShiftImm: return e.encodeEvexShiftImm(spec, ops) case vexNDS3Imm: return e.encodeEvexNDS3Imm(spec, ops) case vexExtract: return e.encodeEvexExtract(spec, ops) } return fmt.Errorf("unhandled EVEX form for %s", mnemUpper) } // encodeEvexNDS3 encodes the three-operand NDS form: OP src2, src1, dst. The // destination may be an opmask register (VPCMPEQD), in which case the vector // length comes from the sources. func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("EVEX NDS instruction expects 3 operands, got %d", len(ops)) } src2, src1, dst := ops[0], ops[1], ops[2] dstReg, ok := dst.(Reg) if !ok || (!dstReg.isVec() && !dstReg.mask) { return fmt.Errorf("EVEX destination must be a vector or mask register") } vvvvReg, ok := src1.(Reg) if !ok || !vvvvReg.isVec() { return fmt.Errorf("EVEX vvvv operand must be a vector register") } ll := dstReg.vecLenBit() if dstReg.mask { ll = vvvvReg.vecLenBit() if r, ok := src2.(Reg); ok && r.isVec() { ll = r.vecLenBit() } } return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2) } // encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src, // no vvvv), e.g. VCVTQQ2PD. func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("EVEX destination must be a vector register") } return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src) } // encodeEvexShiftImm encodes an immediate shift: OP $imm, src, dst // (ModRM.reg = /digit, vvvv = dst, rm = src, imm8), e.g. VPSRAD $31, Z3, Z5. func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("EVEX shift expects 3 operands ($imm, src, dst), got %d", len(ops)) } imm, src, dst := ops[0], ops[1], ops[2] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("shift count must be an immediate") } srcReg, ok := src.(Reg) if !ok || !srcReg.isVec() { return fmt.Errorf("shift source must be a vector register") } dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("shift destination must be a vector register") } immByte, err := imm8(int64(immVal)) if err != nil { return err } if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg); err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeEvexNDS3Imm encodes OP $imm, src2, src1, dst (reg=dst, vvvv=src1, // rm=src2, imm8), e.g. VALIGND. func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand) error { if len(ops) != 4 { return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops)) } imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("shuffle control must be an immediate") } dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("destination must be a vector register") } vvvvReg, ok := src1.(Reg) if !ok || !vvvvReg.isVec() { return fmt.Errorf("second source must be a vector register") } immByte, err := imm8(int64(immVal)) if err != nil { return err } if err := e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, vvvvReg.idx, src2); err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeEvexExtract encodes OP $imm, zsrc, ydst (reg=ZMM source, rm=YMM/memory // destination, imm8), e.g. VEXTRACTI64X4. func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("extract expects 3 operands ($imm, zsrc, ydst), got %d", len(ops)) } imm, src, dst := ops[0], ops[1], ops[2] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("extract lane must be an immediate") } srcReg, ok := src.(Reg) if !ok || !srcReg.isVec() { return fmt.Errorf("extract source must be a vector register") } immByte, err := imm8(int64(immVal)) if err != nil { return err } if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst); err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses // the store-form opcode (reg = source, rm = destination), matching the Go // assembler. func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] srcReg, srcIsVec := vecReg(src) dstReg, dstIsVec := vecReg(dst) op := ms.store var reg Reg var rm Operand switch { case srcIsVec && dstIsVec: reg, rm = srcReg, dst case srcIsVec: if !memOperand(dst) { return fmt.Errorf("%s: invalid destination operand", mnem) } reg, rm = srcReg, dst case dstIsVec: if !memOperand(src) { return fmt.Errorf("%s: invalid source operand", mnem) } op = ms.load reg, rm = dstReg, src default: return fmt.Errorf("%s needs a vector register operand", mnem) } spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n} return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm) } // memOperand reports whether op is a memory reference (including a // static-symbol reference). func memOperand(op Operand) bool { switch op.(type) { case Mem, sbMem: return true } return false } // encodeEvexRMRev encodes the narrowing-store form: OP src, dst with the wide // source in the reg field and the narrow destination in r/m (VPMOVDW/QD). func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("EVEX store instruction expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] srcReg, ok := src.(Reg) if !ok || !srcReg.isVec() { return fmt.Errorf("EVEX source must be a vector register") } return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst) } // encodeEvexBcast encodes VPBROADCASTD/Q: OP src, dst with the GPR or memory // source broadcast to every lane of the vector destination. func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("broadcast expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("broadcast destination must be a vector register") } spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1} switch src.(type) { case Mem, sbMem: spec.opcode = bs.opMem spec.n = [3]int{bs.n, bs.n, bs.n} case Reg: spec.opcode = bs.opReg default: return fmt.Errorf("broadcast source must be a register or memory") } return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src) } // emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement // (disp8×N compressed) for the given precomputed fields. regIdx is the // unextended reg-field register index, or a /digit (0–7); vvvvIdx is the // vvvv register index, or -1 when unused. func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand) error { if ll > 2 { return fmt.Errorf("invalid vector length") } // reg-field extension bits (R̄, R'̄), inverted. rBar, rPrimeBar := 1, 1 if regIdx&8 != 0 { rBar = 0 } if regIdx&16 != 0 { rPrimeBar = 0 } // vvvv (inverted) and its extension bit V'̄. vBar, vPrimeBar := 15, 1 if vvvvIdx >= 0 { vBar = 15 - (vvvvIdx & 15) if vvvvIdx&16 != 0 { vPrimeBar = 0 } } var modrm, sib int var disp []byte xBar, bBar := 1, 1 var sb *sbRef switch r := rm.(type) { case Reg: // ModRM.mod = 11: rm[3] extends via B̄, rm[4] via X̄. modrm = 0xC0 | (regIdx&7)<<3 | (r.idx & 7) sib = -1 if r.idx&8 != 0 { bBar = 0 } if r.idx&16 != 0 { xBar = 0 } case Mem: var err error modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll]) if err != nil { return err } // An indexed memory operand carries index[4] in V'̄ (Go folds it // together with vvvv[4] into the same bit). if r.HasIndex && r.Index.idx&16 != 0 { vPrimeBar = 0 } case sbMem: // RIP-relative static-symbol reference; disp32 patched at link time // (no disp8 scaling for RIP-relative addressing). modrm = (regIdx&7)<<3 | 0x05 sib = -1 disp = le32(0) sb = &sbRef{name: r.name, addend: r.addend} default: return fmt.Errorf("invalid EVEX r/m operand") } p0 := byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4 | spec.mapSel) p1 := byte(spec.w<<7 | vBar<<3 | 1<<2 | spec.pp) p2 := byte(ll<<5 | vPrimeBar<<3) // z = 0, b = 0, aaa = 0 e.out = append(e.out, 0x62, p0, p1, p2, spec.opcode, byte(modrm)) if sib >= 0 { e.out = append(e.out, byte(sib)) } if sb != nil { e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend}) } e.out = append(e.out, disp...) return nil } // memComponentsEvex computes the ModR/M byte (with the given reg field), the // SIB byte (-1 if none), the displacement bytes and the (inverted sense) // index/base extension bits for an EVEX memory operand. The displacement is // compressed to disp8×N when it is a multiple of n and the quotient fits a // signed byte; otherwise a full disp32 is used. func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) { sib = -1 xBar, bBar = 1, 1 // inverted bits: 1 = no extension if !m.HasBase && !m.HasIndex { return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative } needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4) var mod int switch { case !m.HasBase: mod = 0 disp = le32(m.Disp) case m.Base.idx&7 == 5 && m.Disp == 0: mod = 1 disp = []byte{0} case m.Disp == 0: mod = 0 case n > 0 && m.Disp%int64(n) == 0 && m.Disp/int64(n) >= -128 && m.Disp/int64(n) <= 127: mod = 1 disp = []byte{byte(int8(m.Disp / int64(n)))} default: mod = 2 disp = le32(m.Disp) } if needSIB { idxField := 4 // 100 = no index if m.HasIndex { idxField = m.Index.idx & 7 if m.Index.idx&8 != 0 { xBar = 0 } } baseField := 5 // 101 = no base (with mod=00 → disp32) if m.HasBase { baseField = m.Base.idx & 7 if m.Base.idx&8 != 0 { bBar = 0 } } return mod<<6 | regField<<3 | 0x04, scaleBits(m.Scale)<<6 | idxField<<3 | baseField, disp, xBar, bBar, nil } if m.Base.idx&8 != 0 { bBar = 0 } return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil } // encodeKmovw encodes KMOVW, whose opcode depends on the operand direction: // 90 (k/mem → K), 91 (K → mem), 92 (GPR → K), 93 (K → GPR); k → k uses 90. func (e *enc) encodeKmovw(ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("KMOVW expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] srcReg, srcIsReg := src.(Reg) dstReg, dstIsReg := dst.(Reg) srcK := srcIsReg && srcReg.mask dstK := dstIsReg && dstReg.mask spec := vexSpec{mapSel: 1, w: 0, pp: 0, opdigit: -1} switch { case srcK && dstK: spec.opcode = 0x90 // k ← k: reg = dst, rm = src return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src) case srcK && dstIsReg: spec.opcode = 0x93 // GPR ← k: reg = dst, rm = src rBit := 0 if dstReg.idx >= 8 { rBit = 1 } return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15, src) case srcK: if _, ok := dst.(Mem); !ok { return fmt.Errorf("KMOVW: invalid destination operand") } spec.opcode = 0x91 // mem ← k: reg = src, rm = dst return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst) case dstK: spec.opcode = 0x92 // k ← GPR/mem: reg = dst, rm = src return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src) } return fmt.Errorf("KMOVW requires a K register operand") }