diff --git a/asm/encodable.go b/asm/encodable.go index 2cf53a4..3ed5cd7 100644 --- a/asm/encodable.go +++ b/asm/encodable.go @@ -20,9 +20,9 @@ func Encodable(mnemonic string) bool { switch upper { case "RET", "NOP", "CALL", "JMP", "POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2", - // The literal-data pseudo-ops, the accepted-and-ignored END and the - // SP adjust. - "BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP": + // The literal-data pseudo-ops, the accepted-and-ignored END and + // bookkeeping statements, and the SP adjust. + "BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA": return true } if _, ok := noOperandTable[upper]; ok { diff --git a/asm/evex.go b/asm/evex.go index c139ba6..d38f3ab 100644 --- a/asm/evex.go +++ b/asm/evex.go @@ -737,6 +737,91 @@ var evexTable = map[string]evexSpec{ "VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}}, } +// evexQuad describes one quad-register instruction: the opcode under +// EVEX.0F38.W0 with the F2 mandatory prefix, and the width of the vector +// registers the bracketed list and the destination take (512-bit ZMM for +// the packed forms, 128-bit XMM for the scalar ones). +type evexQuad struct { + opcode byte + width int // register width in bytes: 64 (ZMM) or 16 (XMM) +} + +// evexQuadTable maps the quad-register instructions (the 4FMAPS and 4VNNIW +// families) to their encoding. The operand shape is fixed: a single memory +// source in r/m, the bracketed register list whose LOW register travels the +// inverted 5-bit V'VVVV field, an optional opmask in aaa and the vector +// destination in reg. The vector length follows the destination (512-bit +// for the ZMM list forms, 128-bit for the scalar ones) while the disp8×N +// multiplier stays 16 for every member, the toolchain's own tuple choice. +var evexQuadTable = map[string]evexQuad{ + "V4FMADDPS": {0x9A, 64}, + "V4FMADDSS": {0x9B, 16}, + "V4FNMADDPS": {0xAA, 64}, + "V4FNMADDSS": {0xAB, 16}, + "VP4DPWSSD": {0x52, 64}, + "VP4DPWSSDS": {0x53, 64}, +} + +// isEvexQuad reports whether the mnemonic is a quad-register instruction. +func isEvexQuad(upper string) bool { + _, ok := evexQuadTable[upper] + return ok +} + +// encodeEvexQuad encodes the quad-register form: OP mem, [Zn-Zn+3], (K), dst. +// The register list is the VVVV-side source: its low register fills the +// inverted V'VVVV bits, which is why an indexed memory source above Z15 (no +// spare EVEX.X bit once V' is taken) is refused. Masking rides the standard +// aaa field, zeroing keeps the usual requires-a-mask rule, and no other +// suffix applies. +func (e *enc) encodeEvexQuad(mnem string, q evexQuad, ops []Operand, sfx evexSuffix) error { + if len(ops) != 3 && len(ops) != 4 { + return fmt.Errorf("%s expects 3 or 4 operands (mem, [Zn-Zn+3], (K), dst), got %d", mnem, len(ops)) + } + mem, lst := ops[0], ops[1] + dst := ops[len(ops)-1] + mask := 0 + if len(ops) == 4 { + k, ok := ops[2].(Reg) + if !ok || !k.mask { + return fmt.Errorf("%s: third operand must be an opmask register", mnem) + } + if k.idx == 0 { + return fmt.Errorf("k0 is not a usable mask register") + } + mask = k.idx + } + list, ok := lst.(RegList) + if !ok { + return fmt.Errorf("%s: second operand must be a four-register list", mnem) + } + if list.Lo.size != q.width { + return fmt.Errorf("%s: the register list must hold %d-bit vector registers", mnem, q.width*8) + } + dstReg, ok := dst.(Reg) + if !ok || !dstReg.isVec() { + return fmt.Errorf("%s: destination must be a vector register", mnem) + } + if dstReg.size != q.width { + return fmt.Errorf("%s: the destination must be a %d-bit vector register", mnem, q.width*8) + } + if !memOperand(mem) { + return fmt.Errorf("%s: the source must be a memory operand", mnem) + } + // The list owns V'VVVV; a scaled index in the EVEX-only half would fold + // its fifth bit into the same field the list's low register occupies. + if m, ok := mem.(Mem); ok && m.HasIndex && m.Index.idx >= 16 { + return fmt.Errorf("%s: an index register above Z15 has no EVEX bit free", mnem) + } + if sfx.zeroing && mask == 0 { + return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnem) + } + spec := evexSpec{mapSel: 2, opcode: q.opcode, w: 0, pp: 3, opdigit: -1, n: [3]int{16, 16, 16}} + // The vector length follows the destination (512-bit for the ZMM forms, + // 128-bit for the scalar ones), exactly as the oracle encodes it. + return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, list.Lo.idx, mem, mask, sfx) +} + // evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode // depends on the source kind, a GPR source uses opReg, a memory source uses // opMem with a disp8×N of n. @@ -812,8 +897,10 @@ func isEvex(mnemUpper string) bool { if _, ok := evexBcastTable[mnemUpper]; ok { return true } - _, ok := evexMoveTable[mnemUpper] - return ok + if _, ok := evexMoveTable[mnemUpper]; ok { + return true + } + return isEvexQuad(mnemUpper) } // evexRequired reports whether the operands force the EVEX encoding of a @@ -995,6 +1082,14 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error return e.encodeEvexRM(spec, ops, 0, sfx) } spec, inTable := evexTable[mnemUpper] + if q, ok := evexQuadTable[mnemUpper]; ok { + // The quad-register family carries no rounding, SAE or broadcast; + // only masking and zeroing apply. + if sfx.sae || sfx.bcst || sfx.rounding >= 0 { + return fmt.Errorf("%s takes no rounding/SAE/broadcast suffix", mnemUpper) + } + return e.encodeEvexQuad(mnemUpper, q, ops, sfx) + } if inTable { if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] { return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper) diff --git a/asm/operand.go b/asm/operand.go index ac00558..a68bd39 100644 --- a/asm/operand.go +++ b/asm/operand.go @@ -14,6 +14,28 @@ type Imm int64 func (Imm) isOperand() {} +// RegList is a bracketed register range, [Z0-Z3]: the four-register source +// of the 4FMAPS and 4VNNIW families. The EVEX emit path carries the list's +// low register through the inverted 5-bit V'VVVV field; the three higher +// registers are implied by the instruction, so only the pair travels here. +type RegList struct { + Lo Reg + Hi Reg // implied by the encoding; Lo.idx+3 by construction +} + +func (RegList) isOperand() {} + +// FloatImm is a floating-point immediate ($-1.0). The SSE mnemonics whose +// encoding takes an XMM/memory source at that position rewrite it as a read +// from a read-only pool constant ($f64. or $f32.), the toolchain's +// own behaviour; every other instruction rejects it. +type FloatImm struct { + Text string // the numeric text as written, sign excluded + Neg bool // a leading minus +} + +func (FloatImm) isOperand() {} + // Mem is a memory operand of the form disp(base)(index*scale). type Mem struct { Base Reg diff --git a/testdata/verify/quadreg_amd64.s b/testdata/verify/quadreg_amd64.s new file mode 100644 index 0000000..1896a13 --- /dev/null +++ b/testdata/verify/quadreg_amd64.s @@ -0,0 +1,44 @@ +// The quad-register instructions: the 4FMAPS family (V4FMADDPS, +// V4FMADDSS, V4FNMADDPS, V4FNMADDSS) and the 4VNNIW pair (VP4DPWSSD, +// VP4DPWSSDS). The bracketed list's low register travels the inverted +// V'VVVV field, the memory source keeps r/m, the opmask rides aaa and the +// vector length follows the destination (512-bit for the ZMM forms, +// 128-bit for the scalar ones) while the disp8xN multiplier stays 16 for +// every member. Every result is folded back so no instruction is dead. + +#include "textflag.h" + +// func quadf4(src *[16]uint32, n int) float32 +TEXT ·quadf4(SB), NOSPLIT, $0-20 + MOVQ src+0(FP), SI + MOVQ n+8(FP), CX + // The packed 4-FMA form over four consecutive ZMM accumulators, + // masked with K2, K3 and unmasked alike; the displacements exercise + // the disp32 form and the disp8x16 compressed form. + V4FMADDPS 17(SI), [Z0-Z3], K2, Z0 + V4FMADDPS 64(SI), [Z10-Z13], K2, Z1 + V4FMADDPS (SI), [Z20-Z23], Z2 + V4FNMADDPS 96(SI), [Z1-Z4], K3, Z5 + // The scalar form reads XMM lists and takes the 128-bit length; the + // displacement compresses by 16. + V4FMADDSS 7(AX), [X0-X3], K5, X22 + V4FMADDSS (DI), [X10-X13], K5, X23 + V4FNMADDSS 16(SI), [X20-X23], K1, X24 + // The 4-VNNI dot products, indexed source included. + VP4DPWSSD 15(DX)(BX*8), [Z2-Z5], K4, Z17 + VP4DPWSSDS -7(DI)(R8*1), [Z4-Z7], K1, Z31 + VP4DPWSSD (SI), [Z12-Z15], Z6 + // Zeroing keeps the usual rule: a mask register must ride along. + V4FMADDPS.Z 128(SI), [Z24-Z27], K4, Z3 + // Fold every accumulator into one scalar. + VPADDD Z0, Z1, Z9 + VPADDD Z2, Z5, Z10 + VPADDD Z9, Z17, Z11 + VPADDD Z10, Z31, Z12 + VPADDD Z11, Z12, Z13 + VPADDD Z13, Z14, Z15 + VADDSS X22, X23, X0 + VADDSS X24, X0, X1 + VADDSS X1, X2, X3 + VMOVSS X3, ret+16(FP) + RET