// The AVX-512 families behind the avx512enc gap: AES round ops, integer // VNNI and bit algorithms, word shifts and permutes with an immediate or a // register count, lane broadcasts and extracts, gather and scatter prefetch // hints, opmask broadcasts, the high/low half moves and the non-temporal // stores. Every result is folded back so no instruction is dead. #include "textflag.h" // func avx512int(p *byte, n int) uint64 TEXT ·avx512int(SB), NOSPLIT, $0-24 MOVQ p+0(FP), SI MOVQ n+16(FP), CX // AES rounds through the EVEX spellings, masks included. VAESENC Z20, Z21, Z22 VAESENCLAST Z23, Z24, Z25 VAESDEC (SI), Z26, Z27 VAESDECLAST Z28, Z29, Z30 // Integer VNNI and the bit algorithm group. VPDPBUSD Z1, Z2, K2, Z3 VPDPBUSDS Z4, Z5, K2, Z6 VPDPWSSD Z7, Z8, Z9 VPDPWSSDS Z10, Z11, K2, Z12 VPOPCNTW Z12, K3, Z13 VPOPCNTB Z14, Z15 VGF2P8MULB Z16, Z17, K4, Z18 VGF2P8AFFINEQB $7, Z18, Z19, K5, Z20 // Byte/word arithmetic with saturation and masks. VPADDSB Z1, Z2, K1, Z3 VPADDUSW Z3, Z4, K1, Z5 VPSUBSW Z5, Z6, K1, Z7 VPSUBUSB Z7, Z8, K1, Z9 VPSADBW Z9, Z10, Z11 VPMULHRSW Z11, Z12, Z13 VPMULHW Z13, Z14, Z15 VPUNPCKLBW Z15, Z16, K2, Z17 VPUNPCKHBW Z17, Z18, K2, Z19 VPUNPCKLWD Z19, Z20, K2, Z21 VPUNPCKHWD Z21, Z22, K2, Z23 VPCMPEQB Z23, Z24, K2, K3 VPCMPGTW Z25, Z26, K2, K3 VPCMPEQQ Z27, Z28, K2 VPMULTISHIFTQB Z29, Z30, K3, Z31 VDBPSADBW $3, Z1, Z2, K3, Z3 MOVQ CX, ret+16(FP) RET // func avx512perm(p *byte) uint64 TEXT ·avx512perm(SB), NOSPLIT, $0-16 MOVQ p+0(FP), SI // Permutations: immediate and register counts, ternary logic. VALIGNQ $3, Z1, Z2, K1, Z3 VPERMT2B Z3, Z4, K1, Z5 VPERMT2W Z5, Z6, K1, Z7 VPERMT2PS Z7, Z8, K1, Z9 VPERMI2W Z9, Z10, K1, Z11 VPERMI2PS Z11, Z12, K1, Z13 VPERMI2PD Z13, Z14, K1, Z15 VPERMB Z15, Z16, K1, Z17 VPERMW Z17, Z18, K1, Z19 VPERMPS Z19, Z20, Z21 VPERMD Z20, Z21, Z22 VPERMQ $1, Z1, K2, Z2 VPERMQ Z3, Z4, K2, Z5 VPERMPD $1, Z5, K2, Z6 VPERMPD Z7, Z8, K2, Z9 VPERMILPS $5, Z9, K2, Z10 VPERMILPS Z11, Z12, K2, Z13 VPERMILPD $1, Z13, K2, Z14 VPERMILPD Z15, Z16, K2, Z17 VPTERNLOGD $6, Z17, Z18, K2, Z19 VPTERNLOGQ $9, Z19, Z20, K2, Z21 // Lane shuffle and blend families. VSHUFPD $1, Z1, Z2, K1, Z3 VSHUFPS $2, Z4, Z5, K1, Z6 VBLENDMPD Z7, Z8, K1, Z9 VBLENDMPS Z9, Z10, K1, Z11 VPBLENDMB Z11, Z12, K1, Z13 VPBLENDMW Z13, Z14, K1, Z15 VPBLENDMD Z15, Z16, K1, Z17 VPBLENDMQ Z17, Z18, K1, Z19 // Conflicts and leading zero counts. VPCONFLICTD Z1, K1, Z2 VPCONFLICTQ Z3, K1, Z4 VPLZCNTD Z5, K1, Z6 VPLZCNTQ Z7, K1, Z8 // Compress and expand, byte and word widths. VPCOMPRESSB Z1, K1, (SI) VPCOMPRESSW Z2, K1, (SI) VPEXPANDB (SI), K1, Z3 VPEXPANDW (SI), K1, Z4 MOVQ SI, ret+8(FP) RET // func avx512shift(p *byte) uint64 TEXT ·avx512shift(SB), NOSPLIT, $0-16 MOVQ p+0(FP), SI // Variable shifts and shuffles with masks. VPSLLVW Z1, Z2, K1, Z3 VPSRLVW Z3, Z4, K1, Z5 VPSRAVW Z5, Z6, K1, Z7 VPSHLDVW Z7, Z8, K1, Z9 VPSHRDVW Z9, Z10, K1, Z11 VPSHLDVD Z11, Z12, K1, Z13 VPSHLDVQ Z13, Z14, K1, Z15 VPSHRDVD Z15, Z16, K1, Z17 VPSHRDVQ Z17, Z18, K1, Z19 // Immediate shifts, the word/byte-quad widths and masks. VPSLLW $3, Z1, K2, Z2 VPSRLW $5, Z3, K2, Z4 VPSRAW $7, Z5, K2, Z6 VPSLLDQ $9, Z7, Z8 VPSRLDQ $11, Z9, Z10 // Register-count shifts and their memory-count forms. VPSLLD X1, Z2, K1, Z3 VPSRLD 16(SI), Z4, K1, Z5 VPSLLQ X6, Z7, K1, Z8 VPSRLQ X9, Z10, K1, Z11 VPSLLW X12, Z13, K1, Z14 VPSRAW X15, Z16, K1, Z17 VPSRAQ $13, Z12, K1, Z13 VPSRAD X14, Z15, K1, Z16 // Lane shuffles in and out. VPSHLDW $2, Z1, Z2, K1, Z3 VPSHLDQ $4, Z3, Z4, K1, Z5 VPSHRDW $6, Z5, Z6, K1, Z7 VPSHRDQ $8, Z7, Z8, K1, Z9 VPSHUFBITQMB Z9, Z10, K3 VPTESTMB Z11, Z12, K4 VPTESTNMQ Z13, Z14, K5 MOVQ SI, ret+8(FP) RET // func avx512float(x float64) float64 TEXT ·avx512float(SB), NOSPLIT, $0-16 // Square roots, compares and the EXP2/RCP28 helpers. MOVQ x+0(FP), AX VSQRTPD Z1, K1, Z2 VSQRTPS Z3, K1, Z4 VSQRTSD X1, X2, K1, X3 VSQRTSS X3, X4, X5 VCOMISD X5, X6 VUCOMISS X7, X8 VEXP2PD Z5, K1, Z6 VRCP28PD Z7, K1, Z8 VRCP28SD X9, X8, K1, X10 VRSQRT28PS Z11, K1, Z12 VRSQRT28SS X11, X10, K1, X12 VCVTSD2SS X1, X2, X3 VCVTSS2SD X3, X2, K1, X4 VFMADD132PD Z1, Z2, K1, Z3 VFMADD231SD X1, X2, K1, X3 VFMSUBADD213PS Z3, Z4, K1, Z5 VFNMSUB231PD Z5, Z6, K1, Z7 // Broadcasts and masked moves. VBROADCASTF32X2 X1, K1, Z2 VBROADCASTI64X2 (SI), K1, Z3 VMOVUPS Z1, K2, Z3 VMOVSD X14, X5, K3, X22 VMOVSS X18, X3, K2, X25 VMOVHPS (SI), X18, X19 VMOVHPS X20, 8(SI) VMOVLHPS X16, X5, X17 VMOVNTDQ Z7, (SI) VMOVNTDQA 64(SI), Z8 VMOVNTPD Z9, (SI) MOVQ SI, ret+8(FP) RET // func avx512mask(p *byte) uint64 TEXT ·avx512mask(SB), NOSPLIT, $0-16 MOVQ p+0(FP), SI // Omask broadcasts and the K register logic. VPBROADCASTMB2Q K1, Z2 VPBROADCASTMW2D K3, Z4 KUNPCKWD K6, K4, K1 KADDB K2, K3, K5 KORW K1, K2, K7 // Gather and scatter prefetch hints. VGATHERPF0DPD K5, (SI)(Y29*8) VSCATTERPF1DPS K2, (SI)(Z28*4) // Masked gathers ride the EVEX spelling; the data length wins L'L. VGATHERDPD (SI)(X10*4), K7, Y22 VPSCATTERDQ Y6, K2, (SI)(X4*1) // Lane extracts to general registers. VPEXTRB $3, X1, AX VPEXTRD $1, X2, DI VPINSRQ $1, SI, X3, X4 VEXTRACTI32X4 $1, Z1, X5 VINSERTI64X2 $1, X6, Z7, K2, Z8 MOVQ SI, ret+8(FP) RET