Files
2026-09-20 21:17:20 +02:00

192 lines
5.9 KiB
ArmAsm

// The AVX-512 families behind the avx512enc gap: AES round ops, integer
// VNNI and bit algorithms, word shifts and permutes with an immediate or a
// register count, lane broadcasts and extracts, gather and scatter prefetch
// hints, opmask broadcasts, the high/low half moves and the non-temporal
// stores. Every result is folded back so no instruction is dead.
#include "textflag.h"
// func avx512int(p *byte, n int) uint64
TEXT ·avx512int(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
MOVQ n+16(FP), CX
// AES rounds through the EVEX spellings, masks included.
VAESENC Z20, Z21, Z22
VAESENCLAST Z23, Z24, Z25
VAESDEC (SI), Z26, Z27
VAESDECLAST Z28, Z29, Z30
// Integer VNNI and the bit algorithm group.
VPDPBUSD Z1, Z2, K2, Z3
VPDPBUSDS Z4, Z5, K2, Z6
VPDPWSSD Z7, Z8, Z9
VPDPWSSDS Z10, Z11, K2, Z12
VPOPCNTW Z12, K3, Z13
VPOPCNTB Z14, Z15
VGF2P8MULB Z16, Z17, K4, Z18
VGF2P8AFFINEQB $7, Z18, Z19, K5, Z20
// Byte/word arithmetic with saturation and masks.
VPADDSB Z1, Z2, K1, Z3
VPADDUSW Z3, Z4, K1, Z5
VPSUBSW Z5, Z6, K1, Z7
VPSUBUSB Z7, Z8, K1, Z9
VPSADBW Z9, Z10, Z11
VPMULHRSW Z11, Z12, Z13
VPMULHW Z13, Z14, Z15
VPUNPCKLBW Z15, Z16, K2, Z17
VPUNPCKHBW Z17, Z18, K2, Z19
VPUNPCKLWD Z19, Z20, K2, Z21
VPUNPCKHWD Z21, Z22, K2, Z23
VPCMPEQB Z23, Z24, K2, K3
VPCMPGTW Z25, Z26, K2, K3
VPCMPEQQ Z27, Z28, K2
VPMULTISHIFTQB Z29, Z30, K3, Z31
VDBPSADBW $3, Z1, Z2, K3, Z3
MOVQ CX, ret+16(FP)
RET
// func avx512perm(p *byte) uint64
TEXT ·avx512perm(SB), NOSPLIT, $0-16
MOVQ p+0(FP), SI
// Permutations: immediate and register counts, ternary logic.
VALIGNQ $3, Z1, Z2, K1, Z3
VPERMT2B Z3, Z4, K1, Z5
VPERMT2W Z5, Z6, K1, Z7
VPERMT2PS Z7, Z8, K1, Z9
VPERMI2W Z9, Z10, K1, Z11
VPERMI2PS Z11, Z12, K1, Z13
VPERMI2PD Z13, Z14, K1, Z15
VPERMB Z15, Z16, K1, Z17
VPERMW Z17, Z18, K1, Z19
VPERMPS Z19, Z20, Z21
VPERMD Z20, Z21, Z22
VPERMQ $1, Z1, K2, Z2
VPERMQ Z3, Z4, K2, Z5
VPERMPD $1, Z5, K2, Z6
VPERMPD Z7, Z8, K2, Z9
VPERMILPS $5, Z9, K2, Z10
VPERMILPS Z11, Z12, K2, Z13
VPERMILPD $1, Z13, K2, Z14
VPERMILPD Z15, Z16, K2, Z17
VPTERNLOGD $6, Z17, Z18, K2, Z19
VPTERNLOGQ $9, Z19, Z20, K2, Z21
// Lane shuffle and blend families.
VSHUFPD $1, Z1, Z2, K1, Z3
VSHUFPS $2, Z4, Z5, K1, Z6
VBLENDMPD Z7, Z8, K1, Z9
VBLENDMPS Z9, Z10, K1, Z11
VPBLENDMB Z11, Z12, K1, Z13
VPBLENDMW Z13, Z14, K1, Z15
VPBLENDMD Z15, Z16, K1, Z17
VPBLENDMQ Z17, Z18, K1, Z19
// Conflicts and leading zero counts.
VPCONFLICTD Z1, K1, Z2
VPCONFLICTQ Z3, K1, Z4
VPLZCNTD Z5, K1, Z6
VPLZCNTQ Z7, K1, Z8
// Compress and expand, byte and word widths.
VPCOMPRESSB Z1, K1, (SI)
VPCOMPRESSW Z2, K1, (SI)
VPEXPANDB (SI), K1, Z3
VPEXPANDW (SI), K1, Z4
MOVQ SI, ret+8(FP)
RET
// func avx512shift(p *byte) uint64
TEXT ·avx512shift(SB), NOSPLIT, $0-16
MOVQ p+0(FP), SI
// Variable shifts and shuffles with masks.
VPSLLVW Z1, Z2, K1, Z3
VPSRLVW Z3, Z4, K1, Z5
VPSRAVW Z5, Z6, K1, Z7
VPSHLDVW Z7, Z8, K1, Z9
VPSHRDVW Z9, Z10, K1, Z11
VPSHLDVD Z11, Z12, K1, Z13
VPSHLDVQ Z13, Z14, K1, Z15
VPSHRDVD Z15, Z16, K1, Z17
VPSHRDVQ Z17, Z18, K1, Z19
// Immediate shifts, the word/byte-quad widths and masks.
VPSLLW $3, Z1, K2, Z2
VPSRLW $5, Z3, K2, Z4
VPSRAW $7, Z5, K2, Z6
VPSLLDQ $9, Z7, Z8
VPSRLDQ $11, Z9, Z10
// Register-count shifts and their memory-count forms.
VPSLLD X1, Z2, K1, Z3
VPSRLD 16(SI), Z4, K1, Z5
VPSLLQ X6, Z7, K1, Z8
VPSRLQ X9, Z10, K1, Z11
VPSLLW X12, Z13, K1, Z14
VPSRAW X15, Z16, K1, Z17
VPSRAQ $13, Z12, K1, Z13
VPSRAD X14, Z15, K1, Z16
// Lane shuffles in and out.
VPSHLDW $2, Z1, Z2, K1, Z3
VPSHLDQ $4, Z3, Z4, K1, Z5
VPSHRDW $6, Z5, Z6, K1, Z7
VPSHRDQ $8, Z7, Z8, K1, Z9
VPSHUFBITQMB Z9, Z10, K3
VPTESTMB Z11, Z12, K4
VPTESTNMQ Z13, Z14, K5
MOVQ SI, ret+8(FP)
RET
// func avx512float(x float64) float64
TEXT ·avx512float(SB), NOSPLIT, $0-16
// Square roots, compares and the EXP2/RCP28 helpers.
MOVQ x+0(FP), AX
VSQRTPD Z1, K1, Z2
VSQRTPS Z3, K1, Z4
VSQRTSD X1, X2, K1, X3
VSQRTSS X3, X4, X5
VCOMISD X5, X6
VUCOMISS X7, X8
VEXP2PD Z5, K1, Z6
VRCP28PD Z7, K1, Z8
VRCP28SD X9, X8, K1, X10
VRSQRT28PS Z11, K1, Z12
VRSQRT28SS X11, X10, K1, X12
VCVTSD2SS X1, X2, X3
VCVTSS2SD X3, X2, K1, X4
VFMADD132PD Z1, Z2, K1, Z3
VFMADD231SD X1, X2, K1, X3
VFMSUBADD213PS Z3, Z4, K1, Z5
VFNMSUB231PD Z5, Z6, K1, Z7
// Broadcasts and masked moves.
VBROADCASTF32X2 X1, K1, Z2
VBROADCASTI64X2 (SI), K1, Z3
VMOVUPS Z1, K2, Z3
VMOVSD X14, X5, K3, X22
VMOVSS X18, X3, K2, X25
VMOVHPS (SI), X18, X19
VMOVHPS X20, 8(SI)
VMOVLHPS X16, X5, X17
VMOVNTDQ Z7, (SI)
VMOVNTDQA 64(SI), Z8
VMOVNTPD Z9, (SI)
MOVQ SI, ret+8(FP)
RET
// func avx512mask(p *byte) uint64
TEXT ·avx512mask(SB), NOSPLIT, $0-16
MOVQ p+0(FP), SI
// Omask broadcasts and the K register logic.
VPBROADCASTMB2Q K1, Z2
VPBROADCASTMW2D K3, Z4
KUNPCKWD K6, K4, K1
KADDB K2, K3, K5
KORW K1, K2, K7
// Gather and scatter prefetch hints.
VGATHERPF0DPD K5, (SI)(Y29*8)
VSCATTERPF1DPS K2, (SI)(Z28*4)
// Masked gathers ride the EVEX spelling; the data length wins L'L.
VGATHERDPD (SI)(X10*4), K7, Y22
VPSCATTERDQ Y6, K2, (SI)(X4*1)
// Lane extracts to general registers.
VPEXTRB $3, X1, AX
VPEXTRD $1, X2, DI
VPINSRQ $1, SI, X3, X4
VEXTRACTI32X4 $1, Z1, X5
VINSERTI64X2 $1, X6, Z7, K2, Z8
MOVQ SI, ret+8(FP)
RET