feat(amd64): emit the quad-register EVEX families

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-21 02:02:19 +02:00
parent e8b6ff5d7c
commit bfb7701db1
4 changed files with 166 additions and 5 deletions
+3 -3
View File
@@ -20,9 +20,9 @@ func Encodable(mnemonic string) bool {
switch upper { switch upper {
case "RET", "NOP", "CALL", "JMP", case "RET", "NOP", "CALL", "JMP",
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2", "POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
// The literal-data pseudo-ops, the accepted-and-ignored END and the // The literal-data pseudo-ops, the accepted-and-ignored END and
// SP adjust. // bookkeeping statements, and the SP adjust.
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP": "BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA":
return true return true
} }
if _, ok := noOperandTable[upper]; ok { if _, ok := noOperandTable[upper]; ok {
+97 -2
View File
@@ -737,6 +737,91 @@ var evexTable = map[string]evexSpec{
"VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}}, "VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
} }
// evexQuad describes one quad-register instruction: the opcode under
// EVEX.0F38.W0 with the F2 mandatory prefix, and the width of the vector
// registers the bracketed list and the destination take (512-bit ZMM for
// the packed forms, 128-bit XMM for the scalar ones).
type evexQuad struct {
opcode byte
width int // register width in bytes: 64 (ZMM) or 16 (XMM)
}
// evexQuadTable maps the quad-register instructions (the 4FMAPS and 4VNNIW
// families) to their encoding. The operand shape is fixed: a single memory
// source in r/m, the bracketed register list whose LOW register travels the
// inverted 5-bit V'VVVV field, an optional opmask in aaa and the vector
// destination in reg. The vector length follows the destination (512-bit
// for the ZMM list forms, 128-bit for the scalar ones) while the disp8×N
// multiplier stays 16 for every member, the toolchain's own tuple choice.
var evexQuadTable = map[string]evexQuad{
"V4FMADDPS": {0x9A, 64},
"V4FMADDSS": {0x9B, 16},
"V4FNMADDPS": {0xAA, 64},
"V4FNMADDSS": {0xAB, 16},
"VP4DPWSSD": {0x52, 64},
"VP4DPWSSDS": {0x53, 64},
}
// isEvexQuad reports whether the mnemonic is a quad-register instruction.
func isEvexQuad(upper string) bool {
_, ok := evexQuadTable[upper]
return ok
}
// encodeEvexQuad encodes the quad-register form: OP mem, [Zn-Zn+3], (K), dst.
// The register list is the VVVV-side source: its low register fills the
// inverted V'VVVV bits, which is why an indexed memory source above Z15 (no
// spare EVEX.X bit once V' is taken) is refused. Masking rides the standard
// aaa field, zeroing keeps the usual requires-a-mask rule, and no other
// suffix applies.
func (e *enc) encodeEvexQuad(mnem string, q evexQuad, ops []Operand, sfx evexSuffix) error {
if len(ops) != 3 && len(ops) != 4 {
return fmt.Errorf("%s expects 3 or 4 operands (mem, [Zn-Zn+3], (K), dst), got %d", mnem, len(ops))
}
mem, lst := ops[0], ops[1]
dst := ops[len(ops)-1]
mask := 0
if len(ops) == 4 {
k, ok := ops[2].(Reg)
if !ok || !k.mask {
return fmt.Errorf("%s: third operand must be an opmask register", mnem)
}
if k.idx == 0 {
return fmt.Errorf("k0 is not a usable mask register")
}
mask = k.idx
}
list, ok := lst.(RegList)
if !ok {
return fmt.Errorf("%s: second operand must be a four-register list", mnem)
}
if list.Lo.size != q.width {
return fmt.Errorf("%s: the register list must hold %d-bit vector registers", mnem, q.width*8)
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
if dstReg.size != q.width {
return fmt.Errorf("%s: the destination must be a %d-bit vector register", mnem, q.width*8)
}
if !memOperand(mem) {
return fmt.Errorf("%s: the source must be a memory operand", mnem)
}
// The list owns V'VVVV; a scaled index in the EVEX-only half would fold
// its fifth bit into the same field the list's low register occupies.
if m, ok := mem.(Mem); ok && m.HasIndex && m.Index.idx >= 16 {
return fmt.Errorf("%s: an index register above Z15 has no EVEX bit free", mnem)
}
if sfx.zeroing && mask == 0 {
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnem)
}
spec := evexSpec{mapSel: 2, opcode: q.opcode, w: 0, pp: 3, opdigit: -1, n: [3]int{16, 16, 16}}
// The vector length follows the destination (512-bit for the ZMM forms,
// 128-bit for the scalar ones), exactly as the oracle encodes it.
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, list.Lo.idx, mem, mask, sfx)
}
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode // evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
// depends on the source kind, a GPR source uses opReg, a memory source uses // depends on the source kind, a GPR source uses opReg, a memory source uses
// opMem with a disp8×N of n. // opMem with a disp8×N of n.
@@ -812,8 +897,10 @@ func isEvex(mnemUpper string) bool {
if _, ok := evexBcastTable[mnemUpper]; ok { if _, ok := evexBcastTable[mnemUpper]; ok {
return true return true
} }
_, ok := evexMoveTable[mnemUpper] if _, ok := evexMoveTable[mnemUpper]; ok {
return ok return true
}
return isEvexQuad(mnemUpper)
} }
// evexRequired reports whether the operands force the EVEX encoding of a // evexRequired reports whether the operands force the EVEX encoding of a
@@ -995,6 +1082,14 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
return e.encodeEvexRM(spec, ops, 0, sfx) return e.encodeEvexRM(spec, ops, 0, sfx)
} }
spec, inTable := evexTable[mnemUpper] spec, inTable := evexTable[mnemUpper]
if q, ok := evexQuadTable[mnemUpper]; ok {
// The quad-register family carries no rounding, SAE or broadcast;
// only masking and zeroing apply.
if sfx.sae || sfx.bcst || sfx.rounding >= 0 {
return fmt.Errorf("%s takes no rounding/SAE/broadcast suffix", mnemUpper)
}
return e.encodeEvexQuad(mnemUpper, q, ops, sfx)
}
if inTable { if inTable {
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] { if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper) return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper)
+22
View File
@@ -14,6 +14,28 @@ type Imm int64
func (Imm) isOperand() {} func (Imm) isOperand() {}
// RegList is a bracketed register range, [Z0-Z3]: the four-register source
// of the 4FMAPS and 4VNNIW families. The EVEX emit path carries the list's
// low register through the inverted 5-bit V'VVVV field; the three higher
// registers are implied by the instruction, so only the pair travels here.
type RegList struct {
Lo Reg
Hi Reg // implied by the encoding; Lo.idx+3 by construction
}
func (RegList) isOperand() {}
// FloatImm is a floating-point immediate ($-1.0). The SSE mnemonics whose
// encoding takes an XMM/memory source at that position rewrite it as a read
// from a read-only pool constant ($f64.<hex> or $f32.<hex>), the toolchain's
// own behaviour; every other instruction rejects it.
type FloatImm struct {
Text string // the numeric text as written, sign excluded
Neg bool // a leading minus
}
func (FloatImm) isOperand() {}
// Mem is a memory operand of the form disp(base)(index*scale). // Mem is a memory operand of the form disp(base)(index*scale).
type Mem struct { type Mem struct {
Base Reg Base Reg
+44
View File
@@ -0,0 +1,44 @@
// The quad-register instructions: the 4FMAPS family (V4FMADDPS,
// V4FMADDSS, V4FNMADDPS, V4FNMADDSS) and the 4VNNIW pair (VP4DPWSSD,
// VP4DPWSSDS). The bracketed list's low register travels the inverted
// V'VVVV field, the memory source keeps r/m, the opmask rides aaa and the
// vector length follows the destination (512-bit for the ZMM forms,
// 128-bit for the scalar ones) while the disp8xN multiplier stays 16 for
// every member. Every result is folded back so no instruction is dead.
#include "textflag.h"
// func quadf4(src *[16]uint32, n int) float32
TEXT ·quadf4(SB), NOSPLIT, $0-20
MOVQ src+0(FP), SI
MOVQ n+8(FP), CX
// The packed 4-FMA form over four consecutive ZMM accumulators,
// masked with K2, K3 and unmasked alike; the displacements exercise
// the disp32 form and the disp8x16 compressed form.
V4FMADDPS 17(SI), [Z0-Z3], K2, Z0
V4FMADDPS 64(SI), [Z10-Z13], K2, Z1
V4FMADDPS (SI), [Z20-Z23], Z2
V4FNMADDPS 96(SI), [Z1-Z4], K3, Z5
// The scalar form reads XMM lists and takes the 128-bit length; the
// displacement compresses by 16.
V4FMADDSS 7(AX), [X0-X3], K5, X22
V4FMADDSS (DI), [X10-X13], K5, X23
V4FNMADDSS 16(SI), [X20-X23], K1, X24
// The 4-VNNI dot products, indexed source included.
VP4DPWSSD 15(DX)(BX*8), [Z2-Z5], K4, Z17
VP4DPWSSDS -7(DI)(R8*1), [Z4-Z7], K1, Z31
VP4DPWSSD (SI), [Z12-Z15], Z6
// Zeroing keeps the usual rule: a mask register must ride along.
V4FMADDPS.Z 128(SI), [Z24-Z27], K4, Z3
// Fold every accumulator into one scalar.
VPADDD Z0, Z1, Z9
VPADDD Z2, Z5, Z10
VPADDD Z9, Z17, Z11
VPADDD Z10, Z31, Z12
VPADDD Z11, Z12, Z13
VPADDD Z13, Z14, Z15
VADDSS X22, X23, X0
VADDSS X24, X0, X1
VADDSS X1, X2, X3
VMOVSS X3, ret+16(FP)
RET