feat(amd64): encode the AVX-512 and BMI corpus families
Assisted-by: GLM 5.3 Flash
This commit is contained in:
Vendored
+191
@@ -0,0 +1,191 @@
|
||||
// The AVX-512 families behind the avx512enc gap: AES round ops, integer
|
||||
// VNNI and bit algorithms, word shifts and permutes with an immediate or a
|
||||
// register count, lane broadcasts and extracts, gather and scatter prefetch
|
||||
// hints, opmask broadcasts, the high/low half moves and the non-temporal
|
||||
// stores. Every result is folded back so no instruction is dead.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func avx512int(p *byte, n int) uint64
|
||||
TEXT ·avx512int(SB), NOSPLIT, $0-24
|
||||
MOVQ p+0(FP), SI
|
||||
MOVQ n+16(FP), CX
|
||||
// AES rounds through the EVEX spellings, masks included.
|
||||
VAESENC Z20, Z21, Z22
|
||||
VAESENCLAST Z23, Z24, Z25
|
||||
VAESDEC (SI), Z26, Z27
|
||||
VAESDECLAST Z28, Z29, Z30
|
||||
// Integer VNNI and the bit algorithm group.
|
||||
VPDPBUSD Z1, Z2, K2, Z3
|
||||
VPDPBUSDS Z4, Z5, K2, Z6
|
||||
VPDPWSSD Z7, Z8, Z9
|
||||
VPDPWSSDS Z10, Z11, K2, Z12
|
||||
VPOPCNTW Z12, K3, Z13
|
||||
VPOPCNTB Z14, Z15
|
||||
VGF2P8MULB Z16, Z17, K4, Z18
|
||||
VGF2P8AFFINEQB $7, Z18, Z19, K5, Z20
|
||||
// Byte/word arithmetic with saturation and masks.
|
||||
VPADDSB Z1, Z2, K1, Z3
|
||||
VPADDUSW Z3, Z4, K1, Z5
|
||||
VPSUBSW Z5, Z6, K1, Z7
|
||||
VPSUBUSB Z7, Z8, K1, Z9
|
||||
VPSADBW Z9, Z10, Z11
|
||||
VPMULHRSW Z11, Z12, Z13
|
||||
VPMULHW Z13, Z14, Z15
|
||||
VPUNPCKLBW Z15, Z16, K2, Z17
|
||||
VPUNPCKHBW Z17, Z18, K2, Z19
|
||||
VPUNPCKLWD Z19, Z20, K2, Z21
|
||||
VPUNPCKHWD Z21, Z22, K2, Z23
|
||||
VPCMPEQB Z23, Z24, K2, K3
|
||||
VPCMPGTW Z25, Z26, K2, K3
|
||||
VPCMPEQQ Z27, Z28, K2
|
||||
VPMULTISHIFTQB Z29, Z30, K3, Z31
|
||||
VDBPSADBW $3, Z1, Z2, K3, Z3
|
||||
MOVQ CX, ret+16(FP)
|
||||
RET
|
||||
|
||||
// func avx512perm(p *byte) uint64
|
||||
TEXT ·avx512perm(SB), NOSPLIT, $0-16
|
||||
MOVQ p+0(FP), SI
|
||||
// Permutations: immediate and register counts, ternary logic.
|
||||
VALIGNQ $3, Z1, Z2, K1, Z3
|
||||
VPERMT2B Z3, Z4, K1, Z5
|
||||
VPERMT2W Z5, Z6, K1, Z7
|
||||
VPERMT2PS Z7, Z8, K1, Z9
|
||||
VPERMI2W Z9, Z10, K1, Z11
|
||||
VPERMI2PS Z11, Z12, K1, Z13
|
||||
VPERMI2PD Z13, Z14, K1, Z15
|
||||
VPERMB Z15, Z16, K1, Z17
|
||||
VPERMW Z17, Z18, K1, Z19
|
||||
VPERMPS Z19, Z20, Z21
|
||||
VPERMD Z20, Z21, Z22
|
||||
VPERMQ $1, Z1, K2, Z2
|
||||
VPERMQ Z3, Z4, K2, Z5
|
||||
VPERMPD $1, Z5, K2, Z6
|
||||
VPERMPD Z7, Z8, K2, Z9
|
||||
VPERMILPS $5, Z9, K2, Z10
|
||||
VPERMILPS Z11, Z12, K2, Z13
|
||||
VPERMILPD $1, Z13, K2, Z14
|
||||
VPERMILPD Z15, Z16, K2, Z17
|
||||
VPTERNLOGD $6, Z17, Z18, K2, Z19
|
||||
VPTERNLOGQ $9, Z19, Z20, K2, Z21
|
||||
// Lane shuffle and blend families.
|
||||
VSHUFPD $1, Z1, Z2, K1, Z3
|
||||
VSHUFPS $2, Z4, Z5, K1, Z6
|
||||
VBLENDMPD Z7, Z8, K1, Z9
|
||||
VBLENDMPS Z9, Z10, K1, Z11
|
||||
VPBLENDMB Z11, Z12, K1, Z13
|
||||
VPBLENDMW Z13, Z14, K1, Z15
|
||||
VPBLENDMD Z15, Z16, K1, Z17
|
||||
VPBLENDMQ Z17, Z18, K1, Z19
|
||||
// Conflicts and leading zero counts.
|
||||
VPCONFLICTD Z1, K1, Z2
|
||||
VPCONFLICTQ Z3, K1, Z4
|
||||
VPLZCNTD Z5, K1, Z6
|
||||
VPLZCNTQ Z7, K1, Z8
|
||||
// Compress and expand, byte and word widths.
|
||||
VPCOMPRESSB Z1, K1, (SI)
|
||||
VPCOMPRESSW Z2, K1, (SI)
|
||||
VPEXPANDB (SI), K1, Z3
|
||||
VPEXPANDW (SI), K1, Z4
|
||||
MOVQ SI, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func avx512shift(p *byte) uint64
|
||||
TEXT ·avx512shift(SB), NOSPLIT, $0-16
|
||||
MOVQ p+0(FP), SI
|
||||
// Variable shifts and shuffles with masks.
|
||||
VPSLLVW Z1, Z2, K1, Z3
|
||||
VPSRLVW Z3, Z4, K1, Z5
|
||||
VPSRAVW Z5, Z6, K1, Z7
|
||||
VPSHLDVW Z7, Z8, K1, Z9
|
||||
VPSHRDVW Z9, Z10, K1, Z11
|
||||
VPSHLDVD Z11, Z12, K1, Z13
|
||||
VPSHLDVQ Z13, Z14, K1, Z15
|
||||
VPSHRDVD Z15, Z16, K1, Z17
|
||||
VPSHRDVQ Z17, Z18, K1, Z19
|
||||
// Immediate shifts, the word/byte-quad widths and masks.
|
||||
VPSLLW $3, Z1, K2, Z2
|
||||
VPSRLW $5, Z3, K2, Z4
|
||||
VPSRAW $7, Z5, K2, Z6
|
||||
VPSLLDQ $9, Z7, Z8
|
||||
VPSRLDQ $11, Z9, Z10
|
||||
// Register-count shifts and their memory-count forms.
|
||||
VPSLLD X1, Z2, K1, Z3
|
||||
VPSRLD 16(SI), Z4, K1, Z5
|
||||
VPSLLQ X6, Z7, K1, Z8
|
||||
VPSRLQ X9, Z10, K1, Z11
|
||||
VPSLLW X12, Z13, K1, Z14
|
||||
VPSRAW X15, Z16, K1, Z17
|
||||
VPSRAQ $13, Z12, K1, Z13
|
||||
VPSRAD X14, Z15, K1, Z16
|
||||
// Lane shuffles in and out.
|
||||
VPSHLDW $2, Z1, Z2, K1, Z3
|
||||
VPSHLDQ $4, Z3, Z4, K1, Z5
|
||||
VPSHRDW $6, Z5, Z6, K1, Z7
|
||||
VPSHRDQ $8, Z7, Z8, K1, Z9
|
||||
VPSHUFBITQMB Z9, Z10, K3
|
||||
VPTESTMB Z11, Z12, K4
|
||||
VPTESTNMQ Z13, Z14, K5
|
||||
MOVQ SI, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func avx512float(x float64) float64
|
||||
TEXT ·avx512float(SB), NOSPLIT, $0-16
|
||||
// Square roots, compares and the EXP2/RCP28 helpers.
|
||||
MOVQ x+0(FP), AX
|
||||
VSQRTPD Z1, K1, Z2
|
||||
VSQRTPS Z3, K1, Z4
|
||||
VSQRTSD X1, X2, K1, X3
|
||||
VSQRTSS X3, X4, X5
|
||||
VCOMISD X5, X6
|
||||
VUCOMISS X7, X8
|
||||
VEXP2PD Z5, K1, Z6
|
||||
VRCP28PD Z7, K1, Z8
|
||||
VRCP28SD X9, X8, K1, X10
|
||||
VRSQRT28PS Z11, K1, Z12
|
||||
VRSQRT28SS X11, X10, K1, X12
|
||||
VCVTSD2SS X1, X2, X3
|
||||
VCVTSS2SD X3, X2, K1, X4
|
||||
VFMADD132PD Z1, Z2, K1, Z3
|
||||
VFMADD231SD X1, X2, K1, X3
|
||||
VFMSUBADD213PS Z3, Z4, K1, Z5
|
||||
VFNMSUB231PD Z5, Z6, K1, Z7
|
||||
// Broadcasts and masked moves.
|
||||
VBROADCASTF32X2 X1, K1, Z2
|
||||
VBROADCASTI64X2 (SI), K1, Z3
|
||||
VMOVUPS Z1, K2, Z3
|
||||
VMOVSD X14, X5, K3, X22
|
||||
VMOVSS X18, X3, K2, X25
|
||||
VMOVHPS (SI), X18, X19
|
||||
VMOVHPS X20, 8(SI)
|
||||
VMOVLHPS X16, X5, X17
|
||||
VMOVNTDQ Z7, (SI)
|
||||
VMOVNTDQA 64(SI), Z8
|
||||
VMOVNTPD Z9, (SI)
|
||||
MOVQ SI, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func avx512mask(p *byte) uint64
|
||||
TEXT ·avx512mask(SB), NOSPLIT, $0-16
|
||||
MOVQ p+0(FP), SI
|
||||
// Omask broadcasts and the K register logic.
|
||||
VPBROADCASTMB2Q K1, Z2
|
||||
VPBROADCASTMW2D K3, Z4
|
||||
KUNPCKWD K6, K4, K1
|
||||
KADDB K2, K3, K5
|
||||
KORW K1, K2, K7
|
||||
// Gather and scatter prefetch hints.
|
||||
VGATHERPF0DPD K5, (SI)(Y29*8)
|
||||
VSCATTERPF1DPS K2, (SI)(Z28*4)
|
||||
// Masked gathers ride the EVEX spelling; the data length wins L'L.
|
||||
VGATHERDPD (SI)(X10*4), K7, Y22
|
||||
VPSCATTERDQ Y6, K2, (SI)(X4*1)
|
||||
// Lane extracts to general registers.
|
||||
VPEXTRB $3, X1, AX
|
||||
VPEXTRD $1, X2, DI
|
||||
VPINSRQ $1, SI, X3, X4
|
||||
VEXTRACTI32X4 $1, Z1, X5
|
||||
VINSERTI64X2 $1, X6, Z7, K2, Z8
|
||||
MOVQ SI, ret+8(FP)
|
||||
RET
|
||||
Reference in New Issue
Block a user