feat(amd64): emit the quad-register EVEX families
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+3
-3
@@ -20,9 +20,9 @@ func Encodable(mnemonic string) bool {
|
||||
switch upper {
|
||||
case "RET", "NOP", "CALL", "JMP",
|
||||
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
|
||||
// The literal-data pseudo-ops, the accepted-and-ignored END and the
|
||||
// SP adjust.
|
||||
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP":
|
||||
// The literal-data pseudo-ops, the accepted-and-ignored END and
|
||||
// bookkeeping statements, and the SP adjust.
|
||||
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA":
|
||||
return true
|
||||
}
|
||||
if _, ok := noOperandTable[upper]; ok {
|
||||
|
||||
+97
-2
@@ -737,6 +737,91 @@ var evexTable = map[string]evexSpec{
|
||||
"VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
}
|
||||
|
||||
// evexQuad describes one quad-register instruction: the opcode under
|
||||
// EVEX.0F38.W0 with the F2 mandatory prefix, and the width of the vector
|
||||
// registers the bracketed list and the destination take (512-bit ZMM for
|
||||
// the packed forms, 128-bit XMM for the scalar ones).
|
||||
type evexQuad struct {
|
||||
opcode byte
|
||||
width int // register width in bytes: 64 (ZMM) or 16 (XMM)
|
||||
}
|
||||
|
||||
// evexQuadTable maps the quad-register instructions (the 4FMAPS and 4VNNIW
|
||||
// families) to their encoding. The operand shape is fixed: a single memory
|
||||
// source in r/m, the bracketed register list whose LOW register travels the
|
||||
// inverted 5-bit V'VVVV field, an optional opmask in aaa and the vector
|
||||
// destination in reg. The vector length follows the destination (512-bit
|
||||
// for the ZMM list forms, 128-bit for the scalar ones) while the disp8×N
|
||||
// multiplier stays 16 for every member, the toolchain's own tuple choice.
|
||||
var evexQuadTable = map[string]evexQuad{
|
||||
"V4FMADDPS": {0x9A, 64},
|
||||
"V4FMADDSS": {0x9B, 16},
|
||||
"V4FNMADDPS": {0xAA, 64},
|
||||
"V4FNMADDSS": {0xAB, 16},
|
||||
"VP4DPWSSD": {0x52, 64},
|
||||
"VP4DPWSSDS": {0x53, 64},
|
||||
}
|
||||
|
||||
// isEvexQuad reports whether the mnemonic is a quad-register instruction.
|
||||
func isEvexQuad(upper string) bool {
|
||||
_, ok := evexQuadTable[upper]
|
||||
return ok
|
||||
}
|
||||
|
||||
// encodeEvexQuad encodes the quad-register form: OP mem, [Zn-Zn+3], (K), dst.
|
||||
// The register list is the VVVV-side source: its low register fills the
|
||||
// inverted V'VVVV bits, which is why an indexed memory source above Z15 (no
|
||||
// spare EVEX.X bit once V' is taken) is refused. Masking rides the standard
|
||||
// aaa field, zeroing keeps the usual requires-a-mask rule, and no other
|
||||
// suffix applies.
|
||||
func (e *enc) encodeEvexQuad(mnem string, q evexQuad, ops []Operand, sfx evexSuffix) error {
|
||||
if len(ops) != 3 && len(ops) != 4 {
|
||||
return fmt.Errorf("%s expects 3 or 4 operands (mem, [Zn-Zn+3], (K), dst), got %d", mnem, len(ops))
|
||||
}
|
||||
mem, lst := ops[0], ops[1]
|
||||
dst := ops[len(ops)-1]
|
||||
mask := 0
|
||||
if len(ops) == 4 {
|
||||
k, ok := ops[2].(Reg)
|
||||
if !ok || !k.mask {
|
||||
return fmt.Errorf("%s: third operand must be an opmask register", mnem)
|
||||
}
|
||||
if k.idx == 0 {
|
||||
return fmt.Errorf("k0 is not a usable mask register")
|
||||
}
|
||||
mask = k.idx
|
||||
}
|
||||
list, ok := lst.(RegList)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s: second operand must be a four-register list", mnem)
|
||||
}
|
||||
if list.Lo.size != q.width {
|
||||
return fmt.Errorf("%s: the register list must hold %d-bit vector registers", mnem, q.width*8)
|
||||
}
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||
}
|
||||
if dstReg.size != q.width {
|
||||
return fmt.Errorf("%s: the destination must be a %d-bit vector register", mnem, q.width*8)
|
||||
}
|
||||
if !memOperand(mem) {
|
||||
return fmt.Errorf("%s: the source must be a memory operand", mnem)
|
||||
}
|
||||
// The list owns V'VVVV; a scaled index in the EVEX-only half would fold
|
||||
// its fifth bit into the same field the list's low register occupies.
|
||||
if m, ok := mem.(Mem); ok && m.HasIndex && m.Index.idx >= 16 {
|
||||
return fmt.Errorf("%s: an index register above Z15 has no EVEX bit free", mnem)
|
||||
}
|
||||
if sfx.zeroing && mask == 0 {
|
||||
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnem)
|
||||
}
|
||||
spec := evexSpec{mapSel: 2, opcode: q.opcode, w: 0, pp: 3, opdigit: -1, n: [3]int{16, 16, 16}}
|
||||
// The vector length follows the destination (512-bit for the ZMM forms,
|
||||
// 128-bit for the scalar ones), exactly as the oracle encodes it.
|
||||
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, list.Lo.idx, mem, mask, sfx)
|
||||
}
|
||||
|
||||
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
|
||||
// depends on the source kind, a GPR source uses opReg, a memory source uses
|
||||
// opMem with a disp8×N of n.
|
||||
@@ -812,8 +897,10 @@ func isEvex(mnemUpper string) bool {
|
||||
if _, ok := evexBcastTable[mnemUpper]; ok {
|
||||
return true
|
||||
}
|
||||
_, ok := evexMoveTable[mnemUpper]
|
||||
return ok
|
||||
if _, ok := evexMoveTable[mnemUpper]; ok {
|
||||
return true
|
||||
}
|
||||
return isEvexQuad(mnemUpper)
|
||||
}
|
||||
|
||||
// evexRequired reports whether the operands force the EVEX encoding of a
|
||||
@@ -995,6 +1082,14 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
||||
return e.encodeEvexRM(spec, ops, 0, sfx)
|
||||
}
|
||||
spec, inTable := evexTable[mnemUpper]
|
||||
if q, ok := evexQuadTable[mnemUpper]; ok {
|
||||
// The quad-register family carries no rounding, SAE or broadcast;
|
||||
// only masking and zeroing apply.
|
||||
if sfx.sae || sfx.bcst || sfx.rounding >= 0 {
|
||||
return fmt.Errorf("%s takes no rounding/SAE/broadcast suffix", mnemUpper)
|
||||
}
|
||||
return e.encodeEvexQuad(mnemUpper, q, ops, sfx)
|
||||
}
|
||||
if inTable {
|
||||
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
||||
return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper)
|
||||
|
||||
@@ -14,6 +14,28 @@ type Imm int64
|
||||
|
||||
func (Imm) isOperand() {}
|
||||
|
||||
// RegList is a bracketed register range, [Z0-Z3]: the four-register source
|
||||
// of the 4FMAPS and 4VNNIW families. The EVEX emit path carries the list's
|
||||
// low register through the inverted 5-bit V'VVVV field; the three higher
|
||||
// registers are implied by the instruction, so only the pair travels here.
|
||||
type RegList struct {
|
||||
Lo Reg
|
||||
Hi Reg // implied by the encoding; Lo.idx+3 by construction
|
||||
}
|
||||
|
||||
func (RegList) isOperand() {}
|
||||
|
||||
// FloatImm is a floating-point immediate ($-1.0). The SSE mnemonics whose
|
||||
// encoding takes an XMM/memory source at that position rewrite it as a read
|
||||
// from a read-only pool constant ($f64.<hex> or $f32.<hex>), the toolchain's
|
||||
// own behaviour; every other instruction rejects it.
|
||||
type FloatImm struct {
|
||||
Text string // the numeric text as written, sign excluded
|
||||
Neg bool // a leading minus
|
||||
}
|
||||
|
||||
func (FloatImm) isOperand() {}
|
||||
|
||||
// Mem is a memory operand of the form disp(base)(index*scale).
|
||||
type Mem struct {
|
||||
Base Reg
|
||||
|
||||
Vendored
+44
@@ -0,0 +1,44 @@
|
||||
// The quad-register instructions: the 4FMAPS family (V4FMADDPS,
|
||||
// V4FMADDSS, V4FNMADDPS, V4FNMADDSS) and the 4VNNIW pair (VP4DPWSSD,
|
||||
// VP4DPWSSDS). The bracketed list's low register travels the inverted
|
||||
// V'VVVV field, the memory source keeps r/m, the opmask rides aaa and the
|
||||
// vector length follows the destination (512-bit for the ZMM forms,
|
||||
// 128-bit for the scalar ones) while the disp8xN multiplier stays 16 for
|
||||
// every member. Every result is folded back so no instruction is dead.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func quadf4(src *[16]uint32, n int) float32
|
||||
TEXT ·quadf4(SB), NOSPLIT, $0-20
|
||||
MOVQ src+0(FP), SI
|
||||
MOVQ n+8(FP), CX
|
||||
// The packed 4-FMA form over four consecutive ZMM accumulators,
|
||||
// masked with K2, K3 and unmasked alike; the displacements exercise
|
||||
// the disp32 form and the disp8x16 compressed form.
|
||||
V4FMADDPS 17(SI), [Z0-Z3], K2, Z0
|
||||
V4FMADDPS 64(SI), [Z10-Z13], K2, Z1
|
||||
V4FMADDPS (SI), [Z20-Z23], Z2
|
||||
V4FNMADDPS 96(SI), [Z1-Z4], K3, Z5
|
||||
// The scalar form reads XMM lists and takes the 128-bit length; the
|
||||
// displacement compresses by 16.
|
||||
V4FMADDSS 7(AX), [X0-X3], K5, X22
|
||||
V4FMADDSS (DI), [X10-X13], K5, X23
|
||||
V4FNMADDSS 16(SI), [X20-X23], K1, X24
|
||||
// The 4-VNNI dot products, indexed source included.
|
||||
VP4DPWSSD 15(DX)(BX*8), [Z2-Z5], K4, Z17
|
||||
VP4DPWSSDS -7(DI)(R8*1), [Z4-Z7], K1, Z31
|
||||
VP4DPWSSD (SI), [Z12-Z15], Z6
|
||||
// Zeroing keeps the usual rule: a mask register must ride along.
|
||||
V4FMADDPS.Z 128(SI), [Z24-Z27], K4, Z3
|
||||
// Fold every accumulator into one scalar.
|
||||
VPADDD Z0, Z1, Z9
|
||||
VPADDD Z2, Z5, Z10
|
||||
VPADDD Z9, Z17, Z11
|
||||
VPADDD Z10, Z31, Z12
|
||||
VPADDD Z11, Z12, Z13
|
||||
VPADDD Z13, Z14, Z15
|
||||
VADDSS X22, X23, X0
|
||||
VADDSS X24, X0, X1
|
||||
VADDSS X1, X2, X3
|
||||
VMOVSS X3, ret+16(FP)
|
||||
RET
|
||||
Reference in New Issue
Block a user