feat(asm): add the EVEX floating-point and conversion set
Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
+102
-3
@@ -11,9 +11,11 @@ import (
|
||||
// This file implements EVEX (AVX-512) instruction encoding: the four-byte
|
||||
// EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the
|
||||
// compressed disp8×N displacement, and the operand shapes the go-flac
|
||||
// AVX-512 kernels use. Masking ({k}) and zeroing ({z}) are not supported —
|
||||
// the kernels do not use them. K-register operands (mask destinations,
|
||||
// KMOVW, KTESTW) are.
|
||||
// AVX-512 kernels use plus the common floating-point and conversion set.
|
||||
// Masking follows the Go assembler's spelling: an explicit K1–K7 operand
|
||||
// anywhere among the operands (merging) plus a ".Z" mnemonic suffix for
|
||||
// zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are
|
||||
// supported too.
|
||||
|
||||
// evexSpec describes one EVEX instruction's encoding parameters. The form
|
||||
// field reuses the vexForm shapes, which carry over unchanged.
|
||||
@@ -49,6 +51,31 @@ var evexTable = map[string]evexSpec{
|
||||
// EVEX.128/256/512.66.0F.W1 — packed double arithmetic.
|
||||
"VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
// EVEX.128/256/512.66.0F.W1 — packed double unpack.
|
||||
"VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
|
||||
// EVEX.128.F2.0F.W1 — scalar double arithmetic (the packed opcodes with
|
||||
// an F2 pp; the EVEX forms exist for masked and zeroing use). The
|
||||
// memory operand is a single double, so disp8×N = 8.
|
||||
"VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
"VSUBSD": {1, 0x5C, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
"VMULSD": {1, 0x59, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
"VDIVSD": {1, 0x5E, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
"VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
"VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
|
||||
// EVEX.128.F3.0F.W0 — scalar single arithmetic (disp8×N = 4).
|
||||
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
|
||||
// EVEX.512.66.0F3A — align (NDS + imm8).
|
||||
"VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
@@ -62,6 +89,36 @@ var evexTable = map[string]evexSpec{
|
||||
// EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst,
|
||||
// rm=src, no vvvv).
|
||||
"VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||
// EVEX.128/256/512.F2.0F.W1 — duplicate the low double (reg=dst,
|
||||
// rm=src, no vvvv): a 128-bit destination reads a single double from
|
||||
// memory (disp8×8), the wider ones read the full operand.
|
||||
"VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}},
|
||||
// EVEX.128/256/512.0F.W0 — signed dword to packed single (reg=dst,
|
||||
// rm=src, no vvvv, no mandatory prefix — as in the VEX form).
|
||||
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||||
// EVEX.128/256/512.0F.W0 — packed single to packed double: the
|
||||
// destination is twice the source width and sets the length; disp8×N
|
||||
// follows the narrow memory source. No F3 prefix: the Go assembler
|
||||
// emits this instruction with pp = 00 (Intel's maps would call that
|
||||
// undefined) and gasm reproduces the Go assembler's bytes — its machine
|
||||
// code is the oracle, not the manual.
|
||||
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
|
||||
// EVEX.128/256/512.F3.0F.W0 — signed dword to packed double (the EVEX
|
||||
// form of the VEX instruction; the destination sets the length, disp8×N
|
||||
// follows the narrow memory source).
|
||||
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
|
||||
// EVEX packed double → dword conversions: the source is the wide
|
||||
// operand and the mnemonic fixes the length — the bare names are
|
||||
// 512-bit only (ZMM source, XMM destination), the X/Y spellings are
|
||||
// EVEX-128/256. Exactly one slot of n is valid; it names the vector
|
||||
// length (and the disp8×N multiplier) a register or memory source
|
||||
// encodes.
|
||||
"VCVTPD2DQ": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||
"VCVTTPD2DQ": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||
"VCVTPD2DQX": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||
"VCVTPD2DQY": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||
"VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||
"VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
||||
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
||||
// the xmm/ymm/zmm destination lengths).
|
||||
@@ -292,6 +349,8 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, zeroing bool) error {
|
||||
return e.encodeEvexNDS3Imm(spec, ops, mask, zeroing)
|
||||
case vexExtract:
|
||||
return e.encodeEvexExtract(spec, ops, mask, zeroing)
|
||||
case vexRMSrcLen:
|
||||
return e.encodeEvexRMSrcLen(spec, ops, mask, zeroing)
|
||||
}
|
||||
return fmt.Errorf("unhandled EVEX form for %s", mnemUpper)
|
||||
}
|
||||
@@ -487,6 +546,46 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
|
||||
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, zeroing)
|
||||
}
|
||||
|
||||
// encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
|
||||
// the destination always XMM and the length fixed by the mnemonic — the
|
||||
// single valid slot of spec.n names the vector length (and the disp8×N
|
||||
// multiplier) a register or memory source encodes.
|
||||
func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("EVEX destination must be a vector register")
|
||||
}
|
||||
ll, err := soleLen(spec.n)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, zeroing)
|
||||
}
|
||||
|
||||
// soleLen returns the vector-length index of the single valid slot of n —
|
||||
// the length a length-fixed mnemonic (the EVEX conversion spellings) encodes
|
||||
// regardless of its operands.
|
||||
func soleLen(n [3]int) (int, error) {
|
||||
ll := -1
|
||||
for i, v := range n {
|
||||
if v == 0 {
|
||||
continue
|
||||
}
|
||||
if ll >= 0 {
|
||||
return 0, fmt.Errorf("ambiguous vector-length table %v", n)
|
||||
}
|
||||
ll = i
|
||||
}
|
||||
if ll < 0 {
|
||||
return 0, fmt.Errorf("empty vector-length table")
|
||||
}
|
||||
return ll, nil
|
||||
}
|
||||
|
||||
// memOperand reports whether op is a memory reference (including a
|
||||
// static-symbol reference).
|
||||
func memOperand(op Operand) bool {
|
||||
|
||||
Reference in New Issue
Block a user