Files
gasm-sdk/testdata/verify/quadreg_amd64.s
T
2026-09-21 02:02:19 +02:00

45 lines
1.8 KiB
ArmAsm

// The quad-register instructions: the 4FMAPS family (V4FMADDPS,
// V4FMADDSS, V4FNMADDPS, V4FNMADDSS) and the 4VNNIW pair (VP4DPWSSD,
// VP4DPWSSDS). The bracketed list's low register travels the inverted
// V'VVVV field, the memory source keeps r/m, the opmask rides aaa and the
// vector length follows the destination (512-bit for the ZMM forms,
// 128-bit for the scalar ones) while the disp8xN multiplier stays 16 for
// every member. Every result is folded back so no instruction is dead.
#include "textflag.h"
// func quadf4(src *[16]uint32, n int) float32
TEXT ·quadf4(SB), NOSPLIT, $0-20
MOVQ src+0(FP), SI
MOVQ n+8(FP), CX
// The packed 4-FMA form over four consecutive ZMM accumulators,
// masked with K2, K3 and unmasked alike; the displacements exercise
// the disp32 form and the disp8x16 compressed form.
V4FMADDPS 17(SI), [Z0-Z3], K2, Z0
V4FMADDPS 64(SI), [Z10-Z13], K2, Z1
V4FMADDPS (SI), [Z20-Z23], Z2
V4FNMADDPS 96(SI), [Z1-Z4], K3, Z5
// The scalar form reads XMM lists and takes the 128-bit length; the
// displacement compresses by 16.
V4FMADDSS 7(AX), [X0-X3], K5, X22
V4FMADDSS (DI), [X10-X13], K5, X23
V4FNMADDSS 16(SI), [X20-X23], K1, X24
// The 4-VNNI dot products, indexed source included.
VP4DPWSSD 15(DX)(BX*8), [Z2-Z5], K4, Z17
VP4DPWSSDS -7(DI)(R8*1), [Z4-Z7], K1, Z31
VP4DPWSSD (SI), [Z12-Z15], Z6
// Zeroing keeps the usual rule: a mask register must ride along.
V4FMADDPS.Z 128(SI), [Z24-Z27], K4, Z3
// Fold every accumulator into one scalar.
VPADDD Z0, Z1, Z9
VPADDD Z2, Z5, Z10
VPADDD Z9, Z17, Z11
VPADDD Z10, Z31, Z12
VPADDD Z11, Z12, Z13
VPADDD Z13, Z14, Z15
VADDSS X22, X23, X0
VADDSS X24, X0, X1
VADDSS X1, X2, X3
VMOVSS X3, ret+16(FP)
RET