45 lines
1.8 KiB
ArmAsm
45 lines
1.8 KiB
ArmAsm
// The quad-register instructions: the 4FMAPS family (V4FMADDPS,
|
|
// V4FMADDSS, V4FNMADDPS, V4FNMADDSS) and the 4VNNIW pair (VP4DPWSSD,
|
|
// VP4DPWSSDS). The bracketed list's low register travels the inverted
|
|
// V'VVVV field, the memory source keeps r/m, the opmask rides aaa and the
|
|
// vector length follows the destination (512-bit for the ZMM forms,
|
|
// 128-bit for the scalar ones) while the disp8xN multiplier stays 16 for
|
|
// every member. Every result is folded back so no instruction is dead.
|
|
|
|
#include "textflag.h"
|
|
|
|
// func quadf4(src *[16]uint32, n int) float32
|
|
TEXT ·quadf4(SB), NOSPLIT, $0-20
|
|
MOVQ src+0(FP), SI
|
|
MOVQ n+8(FP), CX
|
|
// The packed 4-FMA form over four consecutive ZMM accumulators,
|
|
// masked with K2, K3 and unmasked alike; the displacements exercise
|
|
// the disp32 form and the disp8x16 compressed form.
|
|
V4FMADDPS 17(SI), [Z0-Z3], K2, Z0
|
|
V4FMADDPS 64(SI), [Z10-Z13], K2, Z1
|
|
V4FMADDPS (SI), [Z20-Z23], Z2
|
|
V4FNMADDPS 96(SI), [Z1-Z4], K3, Z5
|
|
// The scalar form reads XMM lists and takes the 128-bit length; the
|
|
// displacement compresses by 16.
|
|
V4FMADDSS 7(AX), [X0-X3], K5, X22
|
|
V4FMADDSS (DI), [X10-X13], K5, X23
|
|
V4FNMADDSS 16(SI), [X20-X23], K1, X24
|
|
// The 4-VNNI dot products, indexed source included.
|
|
VP4DPWSSD 15(DX)(BX*8), [Z2-Z5], K4, Z17
|
|
VP4DPWSSDS -7(DI)(R8*1), [Z4-Z7], K1, Z31
|
|
VP4DPWSSD (SI), [Z12-Z15], Z6
|
|
// Zeroing keeps the usual rule: a mask register must ride along.
|
|
V4FMADDPS.Z 128(SI), [Z24-Z27], K4, Z3
|
|
// Fold every accumulator into one scalar.
|
|
VPADDD Z0, Z1, Z9
|
|
VPADDD Z2, Z5, Z10
|
|
VPADDD Z9, Z17, Z11
|
|
VPADDD Z10, Z31, Z12
|
|
VPADDD Z11, Z12, Z13
|
|
VPADDD Z13, Z14, Z15
|
|
VADDSS X22, X23, X0
|
|
VADDSS X24, X0, X1
|
|
VADDSS X1, X2, X3
|
|
VMOVSS X3, ret+16(FP)
|
|
RET
|