feat(asm): add the EVEX floating-point and conversion set

Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
2026-07-15 17:13:28 +02:00
parent 0f3146ff2c
commit b914c0e390
7 changed files with 347 additions and 74 deletions
+102 -3
View File
@@ -11,9 +11,11 @@ import (
// This file implements EVEX (AVX-512) instruction encoding: the four-byte
// EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the
// compressed disp8×N displacement, and the operand shapes the go-flac
// AVX-512 kernels use. Masking ({k}) and zeroing ({z}) are not supported —
// the kernels do not use them. K-register operands (mask destinations,
// KMOVW, KTESTW) are.
// AVX-512 kernels use plus the common floating-point and conversion set.
// Masking follows the Go assembler's spelling: an explicit K1–K7 operand
// anywhere among the operands (merging) plus a ".Z" mnemonic suffix for
// zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are
// supported too.
// evexSpec describes one EVEX instruction's encoding parameters. The form
// field reuses the vexForm shapes, which carry over unchanged.
@@ -49,6 +51,31 @@ var evexTable = map[string]evexSpec{
// EVEX.128/256/512.66.0F.W1 — packed double arithmetic.
"VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — packed double unpack.
"VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128.F2.0F.W1 — scalar double arithmetic (the packed opcodes with
// an F2 pp; the EVEX forms exist for masked and zeroing use). The
// memory operand is a single double, so disp8×N = 8.
"VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VSUBSD": {1, 0x5C, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMULSD": {1, 0x59, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VDIVSD": {1, 0x5E, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128.F3.0F.W0 — scalar single arithmetic (disp8×N = 4).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.512.66.0F3A — align (NDS + imm8).
"VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
@@ -62,6 +89,36 @@ var evexTable = map[string]evexSpec{
// EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst,
// rm=src, no vvvv).
"VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1 — duplicate the low double (reg=dst,
// rm=src, no vvvv): a 128-bit destination reads a single double from
// memory (disp8×8), the wider ones read the full operand.
"VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}},
// EVEX.128/256/512.0F.W0 — signed dword to packed single (reg=dst,
// rm=src, no vvvv, no mandatory prefix — as in the VEX form).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0 — packed single to packed double: the
// destination is twice the source width and sets the length; disp8×N
// follows the narrow memory source. No F3 prefix: the Go assembler
// emits this instruction with pp = 00 (Intel's maps would call that
// undefined) and gasm reproduces the Go assembler's bytes — its machine
// code is the oracle, not the manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.128/256/512.F3.0F.W0 — signed dword to packed double (the EVEX
// form of the VEX instruction; the destination sets the length, disp8×N
// follows the narrow memory source).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
// EVEX packed double → dword conversions: the source is the wide
// operand and the mnemonic fixes the length — the bare names are
// 512-bit only (ZMM source, XMM destination), the X/Y spellings are
// EVEX-128/256. Exactly one slot of n is valid; it names the vector
// length (and the disp8×N multiplier) a register or memory source
// encodes.
"VCVTPD2DQ": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTTPD2DQ": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2DQX": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTPD2DQY": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
"VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
// the xmm/ymm/zmm destination lengths).
@@ -292,6 +349,8 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, zeroing bool) error {
return e.encodeEvexNDS3Imm(spec, ops, mask, zeroing)
case vexExtract:
return e.encodeEvexExtract(spec, ops, mask, zeroing)
case vexRMSrcLen:
return e.encodeEvexRMSrcLen(spec, ops, mask, zeroing)
}
return fmt.Errorf("unhandled EVEX form for %s", mnemUpper)
}
@@ -487,6 +546,46 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, zeroing)
}
// encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the length fixed by the mnemonic — the
// single valid slot of spec.n names the vector length (and the disp8×N
// multiplier) a register or memory source encodes.
func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
if len(ops) != 2 {
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("EVEX destination must be a vector register")
}
ll, err := soleLen(spec.n)
if err != nil {
return err
}
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, zeroing)
}
// soleLen returns the vector-length index of the single valid slot of n —
// the length a length-fixed mnemonic (the EVEX conversion spellings) encodes
// regardless of its operands.
func soleLen(n [3]int) (int, error) {
ll := -1
for i, v := range n {
if v == 0 {
continue
}
if ll >= 0 {
return 0, fmt.Errorf("ambiguous vector-length table %v", n)
}
ll = i
}
if ll < 0 {
return 0, fmt.Errorf("empty vector-length table")
}
return ll, nil
}
// memOperand reports whether op is a memory reference (including a
// static-symbol reference).
func memOperand(op Operand) bool {