From 81e2673923038bca4e53ea1f38497983d7791e2d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sun, 20 Sep 2026 21:17:20 +0200 Subject: [PATCH] feat(amd64): encode the AVX-512 and BMI corpus families Assisted-by: GLM 5.3 Flash --- asm/encode.go | 1 + asm/evex.go | 518 +++++++++++++++++++++++++++++++-- asm/evex_test.go | 89 ++++++ testdata/verify/avx512_amd64.s | 191 ++++++++++++ 4 files changed, 777 insertions(+), 22 deletions(-) create mode 100644 testdata/verify/avx512_amd64.s diff --git a/asm/encode.go b/asm/encode.go index 416a322..2dc477f 100644 --- a/asm/encode.go +++ b/asm/encode.go @@ -113,6 +113,7 @@ func (e *enc) encode(mnem string, ops []Operand) error { return err } if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || + isEvexPrefGather(base) || base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" { return e.encodeVec(base, ops, sfx) } diff --git a/asm/evex.go b/asm/evex.go index 16fa653..c139ba6 100644 --- a/asm/evex.go +++ b/asm/evex.go @@ -5,6 +5,7 @@ package asm import ( "fmt" + "slices" "strings" ) @@ -91,7 +92,7 @@ var evexTable = map[string]evexSpec{ "VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ; // the W bit distinguishes it from VPSRAD's E2 form). - "VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSRAQ": {1, 0x72, 1, 1, 4, vexShiftImm, [3]int{16, 32, 64}}, // EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst, // rm=src, no vvvv). @@ -138,7 +139,7 @@ var evexTable = map[string]evexSpec{ // EVEX.66.0F, the EVEX forms of the VEX two-source shuffle. "VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, - "VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VSHUFPS": {1, 0xC6, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}}, // EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst). "VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}}, @@ -202,7 +203,7 @@ var evexTable = map[string]evexSpec{ "VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, - "VPSRLVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSRLVW": {2, 0x10, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, @@ -324,7 +325,7 @@ var evexTable = map[string]evexSpec{ "VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, "VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}}, - "VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}}, + "VCVTUDQ2PS": {1, 0x7A, 0, 3, -1, vexRM, [3]int{8, 16, 32}}, // EVEX.66.0F38, half-precision convert (half-width source). "VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, // EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src, @@ -494,6 +495,246 @@ var evexTable = map[string]evexSpec{ // destination (VPMOVDW dword→word, VPMOVQD qword→dword). "VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, "VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, + + // --- the AVX-512 families the avx512enc corpus exercises, read off + // the toolchain opcodetables --- + "VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VALIGNQ": {3, 0x03, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VANDNPD": {1, 0x55, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VANDPD": {1, 0x54, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VBLENDMPD": {2, 0x65, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VBLENDMPS": {2, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VBROADCASTF32X2": {2, 0x19, 0, 1, -1, vexRM, [3]int{0, 8, 8}}, + "VBROADCASTF32X4": {2, 0x1A, 0, 1, -1, vexRM, [3]int{0, 16, 16}}, + "VBROADCASTF32X8": {2, 0x1B, 0, 1, -1, vexRM, [3]int{0, 0, 32}}, + "VBROADCASTF64X2": {2, 0x1A, 1, 1, -1, vexRM, [3]int{0, 16, 16}}, + "VBROADCASTF64X4": {2, 0x1B, 1, 1, -1, vexRM, [3]int{0, 0, 32}}, + "VBROADCASTI32X2": {2, 0x59, 0, 1, -1, vexRM, [3]int{8, 8, 8}}, + "VBROADCASTI32X4": {2, 0x5A, 0, 1, -1, vexRM, [3]int{0, 16, 16}}, + "VBROADCASTI32X8": {2, 0x5B, 0, 1, -1, vexRM, [3]int{0, 0, 32}}, + "VBROADCASTI64X2": {2, 0x5A, 1, 1, -1, vexRM, [3]int{0, 16, 16}}, + "VBROADCASTI64X4": {2, 0x5B, 1, 1, -1, vexRM, [3]int{0, 0, 32}}, + "VCOMISD": {1, 0x2F, 1, 1, -1, vexRM, [3]int{8, 0, 0}}, + "VCVTSD2SS": {1, 0x5A, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}}, + "VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}}, + "VDBPSADBW": {3, 0x42, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VEXP2PD": {2, 0xC8, 1, 1, -1, vexRM, [3]int{0, 0, 64}}, + "VEXP2PS": {2, 0xC8, 0, 1, -1, vexRM, [3]int{0, 0, 64}}, + "VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev, [3]int{16, 32, 64}}, + "VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VMOVNTPD": {1, 0x2B, 1, 1, -1, vexRMRev, [3]int{16, 32, 64}}, + "VORPD": {1, 0x56, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPBLENDMB": {2, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPBLENDMD": {2, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPBLENDMQ": {2, 0x64, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPBLENDMW": {2, 0x66, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPBROADCASTMB2Q": {2, 0x2A, 1, 2, -1, vexRM, [3]int{0, 0, 0}}, + "VPBROADCASTMW2D": {2, 0x3A, 0, 2, -1, vexRM, [3]int{0, 0, 0}}, + "VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPCMPEQQ": {2, 0x29, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPCMPGTQ": {2, 0x37, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPCOMPRESSB": {2, 0x63, 0, 1, -1, vexRMRev, [3]int{1, 1, 1}}, + "VPCOMPRESSW": {2, 0x63, 1, 1, -1, vexRMRev, [3]int{2, 2, 2}}, + "VPCONFLICTD": {2, 0xC4, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VPCONFLICTQ": {2, 0xC4, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VPDPBUSD": {2, 0x50, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPDPBUSDS": {2, 0x51, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPDPWSSD": {2, 0x52, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPDPWSSDS": {2, 0x53, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMI2PD": {2, 0x77, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMI2PS": {2, 0x77, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMI2W": {2, 0x75, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}}, + "VPERMT2B": {2, 0x7D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMT2PS": {2, 0x7F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMT2W": {2, 0x7D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPEXPANDB": {2, 0x62, 0, 1, -1, vexRM, [3]int{1, 1, 1}}, + "VPEXPANDW": {2, 0x62, 1, 1, -1, vexRM, [3]int{2, 2, 2}}, + "VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm, [3]int{4, 0, 0}}, + "VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm, [3]int{8, 0, 0}}, + "VPLZCNTD": {2, 0x44, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VPLZCNTQ": {2, 0x44, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VPMADD52HUQ": {2, 0xB5, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPMADD52LUQ": {2, 0xB4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPMULDQ": {2, 0x28, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPMULTISHIFTQB": {2, 0x83, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPMULUDQ": {1, 0xF4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPOPCNTW": {2, 0x54, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VPORD": {1, 0xEB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPROLVD": {2, 0x15, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPROLVQ": {2, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPRORVD": {2, 0x14, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPRORVQ": {2, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSHLDD": {3, 0x71, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPSHLDQ": {3, 0x71, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPSHLDVD": {2, 0x71, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSHLDVQ": {2, 0x71, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSHLDVW": {2, 0x70, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSHLDW": {3, 0x70, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPSHRDD": {3, 0x73, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPSHRDQ": {3, 0x73, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPSHRDVD": {2, 0x73, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSHRDVQ": {2, 0x73, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSHRDVW": {2, 0x72, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSHRDW": {3, 0x72, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPSHUFBITQMB": {2, 0x8F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSRAVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm, [3]int{16, 32, 64}}, + "VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm, [3]int{16, 32, 64}}, + // EVEX.66.0F73 /7, the byte-quad shift left (the count is always an + // immediate; there is no register-count twin). + "VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm, [3]int{16, 32, 64}}, + + // EVEX.128/256/512.0F.W0, the plain-prefix (no 66) packed spellings + // whose EVEX form drops the legacy prefix entirely. + "VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, + "VANDPS": {1, 0x54, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, + "VORPS": {1, 0x56, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, + "VXORPS": {1, 0x57, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, + "VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, + "VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, + "VSQRTPS": {1, 0x51, 0, 0, -1, vexRM, [3]int{16, 32, 64}}, + "VCOMISS": {1, 0x2F, 0, 0, -1, vexRM, [3]int{4, 0, 0}}, + "VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM, [3]int{4, 0, 0}}, + "VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev, [3]int{16, 32, 64}}, + "VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPTESTMB": {2, 0x26, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPTESTMD": {2, 0x27, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPTESTMQ": {2, 0x27, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPTESTMW": {2, 0x26, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPTESTNMB": {2, 0x26, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPTESTNMD": {2, 0x27, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPTESTNMQ": {2, 0x27, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPTESTNMW": {2, 0x26, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPUNPCKHQDQ": {1, 0x6D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPUNPCKLQDQ": {1, 0x6C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VRCP28PD": {2, 0xCA, 1, 1, -1, vexRM, [3]int{0, 0, 64}}, + "VRCP28PS": {2, 0xCA, 0, 1, -1, vexRM, [3]int{0, 0, 64}}, + "VRCP28SD": {2, 0xCB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VRCP28SS": {2, 0xCB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VRSQRT28PD": {2, 0xCC, 1, 1, -1, vexRM, [3]int{0, 0, 64}}, + "VRSQRT28PS": {2, 0xCC, 0, 1, -1, vexRM, [3]int{0, 0, 64}}, + "VRSQRT28SD": {2, 0xCD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}}, + "VRSQRT28SS": {2, 0xCD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}}, + "VSQRTPD": {1, 0x51, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VSQRTSD": {1, 0x51, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}}, + "VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}}, + "VUCOMISD": {1, 0x2E, 1, 1, -1, vexRM, [3]int{8, 0, 0}}, + "VXORPD": {1, 0x57, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + + // EVEX.128/256/512.0F.F3/F2.W0, word shuffles with an immediate + // ($imm, src, dst: reg = dst, rm = src, imm8). The F3/F2 prefixes + // split the high/low lane spellings. + "VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM, [3]int{16, 32, 64}}, + "VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM, [3]int{16, 32, 64}}, + + // EVEX.128.66.0F3A, lane extract to a general-purpose register or + // memory ($imm, xsrc, GPR/mem dst: reg = source, rm = destination). + "VPEXTRB": {3, 0x14, 0, 1, -1, vexExtractGPR, [3]int{1, 1, 1}}, + "VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR, [3]int{2, 2, 2}}, + "VPEXTRD": {3, 0x16, 0, 1, -1, vexExtractGPR, [3]int{4, 4, 4}}, + "VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtractGPR, [3]int{8, 8, 8}}, + + // EVEX.66.0F3A.W1, the qword permutes with an immediate control + // ($imm, src, dst: reg = dst, rm = src, imm8); the register-count + // forms live in evexRegFormTable. + "VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}}, + "VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}}, + // EVEX.66.0F3A, the packed permute shuffles with an immediate control. + "VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}}, + "VPERMILPD": {3, 0x05, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}}, + + // EVEX.128.0F.W0, high/low half moves. VMOVHPS carries the + // three-operand insert form (rm = m64 source, vvvv = preserved, + // reg = dst) and the two-operand store (reg = source, rm = m64); + // the encoder splits on the operand count. VMOVLHPS is the + // three-operand form alone. + "VMOVHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}}, + "VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}}, } // evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode @@ -529,32 +770,38 @@ type evexMoveSpec struct { n [3]int vecOK bool // the non-memory operand may be a vector register xmmOnly bool // wider than XMM registers are rejected + nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS) } // evexMoveTable maps an upper-case EVEX move mnemonic to its encoding. var evexMoveTable = map[string]evexMoveSpec{ // EVEX.128/256/512.F3.0F.W0, unaligned integer move. - "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false}, + "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false}, // EVEX.128/256/512.F3.0F.W1, unaligned qword move. - "VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false}, + "VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false}, // EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the // F2 prefix, dword/qword moves F3; the element size only changes the tuple // semantics). - "VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false}, + "VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false}, // EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword // encoding). - "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false}, + "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false}, // EVEX.128/256/512.66.0F.W1, unaligned packed double move. - "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false}, + "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false}, // EVEX.128/256/512, aligned packed moves. - "VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false}, - "VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false}, + "VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false}, + "VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false}, // EVEX.128/256/512.66.0F, aligned integer moves. - "VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false}, - "VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false}, + "VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false}, + "VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false}, // EVEX.128.F3.0F.W0, scalar single move, memory operands (the // three-operand register form is not supported). - "VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true}, + "VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true}, + // EVEX.128.F2.0F.W1, scalar double move: memory operands and the + // three-operand register form (VMOVSD dst, src1, src2). + "VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true}, + // EVEX.128/256/512.0F.W0, unaligned packed single move. + "VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false}, } // isEvex reports whether the mnemonic has an EVEX encoding we handle. @@ -579,6 +826,13 @@ func evexRequired(upper string, ops []Operand) bool { if !inVex && !inVexMove { return true // EVEX-only mnemonic } + // The byte-quad shifts have VEX register forms but EVEX-only memory + // forms: a memory count source forces the EVEX encoding. + if upper == "VPSLLDQ" || upper == "VPSRLDQ" { + if slices.ContainsFunc(ops, memOperand) { + return true + } + } for _, op := range ops { if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) { return true @@ -752,6 +1006,27 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error } spec.n = [3]int{n, n, n} } + // A mnemonic with an immediate and a register spelling (the + // variable-count shifts, the permutes) encodes the register one + // when the first operand is not an immediate. + if len(ops) > 0 { + if _, isImm := ops[0].(Imm); !isImm { + if alt, ok := evexRegFormTable[mnemUpper]; ok { + spec, inTable = alt, true + } + } + } + // The high/low half moves split by operand count: three operands + // insert, two store (VMOVHPS m64, X1). + if hs, ok := evexHptrTable[mnemUpper]; ok { + if len(ops) == 2 { + if hs.store.opcode == 0 { + return fmt.Errorf("%s has no two-operand form", mnemUpper) + } + return e.encodeEvexRMRev(hs.store, ops, 0, sfx) + } + spec = hs.insert + } } else if sfx.evexOnly() { return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper) } @@ -811,6 +1086,12 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error } return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx) } + if ps, ok := evexPrefGatherTable[mnemUpper]; ok { + if sfx.any() { + return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper) + } + return e.encodeEvexPrefGather(mnemUpper, ps, ops, mask, sfx) + } if !inTable { return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper) } @@ -829,6 +1110,8 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error return e.encodeEvexNDS3Imm(spec, ops, mask, sfx) case vexExtract: return e.encodeEvexExtract(spec, ops, mask, sfx) + case vexExtractGPR: + return e.encodeEvexExtractGPR(spec, ops, mask, sfx) case vexRMSrcLen: return e.encodeEvexRMSrcLen(spec, ops, mask, sfx) } @@ -908,6 +1191,11 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSu if dstReg.mask { if r, ok := src.(Reg); ok && r.isVec() { ll = r.vecLenBit() + } else if l, err := soleLen(spec.n); err == nil { + // A memory source with a length-fixed mnemonic + // (VFPCLASSPDX/Y/Z): the length comes from the table's + // single valid slot, not from the operand. + ll = l } } else if r, ok := src.(Reg); ok && r.isVec() { ll = r.vecLenBit() @@ -934,9 +1222,11 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve if !ok { return fmt.Errorf("shift count must be an immediate") } - srcReg, ok := src.(Reg) - if !ok || !srcReg.isVec() { - return fmt.Errorf("shift source must be a vector register") + // The count source is a vector register or memory; the length the L'L + // field and the disp8×N multiplier follow is the destination's either + // way. + if !vecOrMem(src) { + return fmt.Errorf("shift source must be a vector register or memory") } dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { @@ -946,7 +1236,7 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve if err != nil { return err } - if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, sfx); err != nil { + if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, src, mask, sfx); err != nil { return err } e.out = append(e.out, immByte) @@ -1018,10 +1308,77 @@ func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, sfx evex return nil } +// encodeEvexExtractGPR encodes the lane extract to a general-purpose +// register or memory: OP $imm, xsrc, dst (reg = the XMM source, rm = the +// destination, imm8). The encoding is 128-bit regardless of register +// numbers, so L'L is fixed at 0 and the disp8×N multiplier is the extracted +// element size the table carries. +func (e *enc) encodeEvexExtractGPR(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { + if len(ops) != 3 { + return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops)) + } + imm, src, dst := ops[0], ops[1], ops[2] + immVal, ok := imm.(Imm) + if !ok { + return fmt.Errorf("extract lane must be an immediate") + } + srcReg, ok := src.(Reg) + if !ok || !srcReg.isVec() { + return fmt.Errorf("extract source must be a vector register") + } + switch dst.(type) { + case Reg: + if dst.(Reg).isVec() { + return fmt.Errorf("extract destination must be a general-purpose register or memory") + } + case Mem, sbMem: + default: + return fmt.Errorf("extract destination must be a general-purpose register or memory") + } + immByte, err := imm8(int64(immVal)) + if err != nil { + return err + } + if err := e.emitEvexFields(spec, 0, srcReg.idx, -1, dst, mask, sfx); err != nil { + return err + } + e.out = append(e.out, immByte) + return nil +} + // encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses // the store-form opcode (reg = source, rm = destination), matching the Go -// assembler. +// assembler. The scalar moves also carry a three-operand register form +// (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3 +// opens. func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error { + if len(ops) == 3 { + if !ms.nds3 { + return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops)) + } + // The masked scalar register form keeps the Go assembler's own + // layout: the store opcode with reg = op0, vvvv = op1 and the + // destination in r/m (op2) — the bytes go tool asm emits, not + // the manual's NDS reading. + src, src1, dst := ops[0], ops[1], ops[2] + reg, ok := src.(Reg) + if !ok || !reg.isVec() { + return fmt.Errorf("%s: first operand must be a vector register", mnem) + } + vvvvReg, ok := src1.(Reg) + if !ok || !vvvvReg.isVec() { + return fmt.Errorf("%s: second operand must be a vector register", mnem) + } + dstReg, ok := dst.(Reg) + if !ok || !dstReg.isVec() { + return fmt.Errorf("%s: destination must be a vector register", mnem) + } + if ms.xmmOnly && (reg.size != 16 || vvvvReg.size != 16 || dstReg.size != 16) { + return fmt.Errorf("%s operates on XMM registers only", mnem) + } + spec := evexSpec{mapSel: ms.mapSel, opcode: ms.store, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n} + return e.emitEvexFields(spec, dstReg.vecLenBit(), reg.idx, vvvvReg.idx, dst, mask, sfx) + } if len(ops) != 2 { return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops)) } @@ -1141,12 +1498,20 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx eve return fmt.Errorf("broadcast destination must be a vector register") } spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1} - switch src.(type) { + switch r := src.(type) { case Mem, sbMem: spec.opcode = bs.opMem spec.n = [3]int{bs.n, bs.n, bs.n} case Reg: - spec.opcode = bs.opReg + // A GPR source uses the register broadcast opcode; a vector + // source shares the xmm/mem one (the low byte is copied from + // the lane or from the memory operand). + if r.isVec() { + spec.opcode = bs.opMem + spec.n = [3]int{bs.n, bs.n, bs.n} + } else { + spec.opcode = bs.opReg + } default: return fmt.Errorf("broadcast source must be a register or memory") } @@ -1349,6 +1714,102 @@ func isScatter(upper string) bool { return ok } +// isEvexPrefGather reports whether the mnemonic is a gather/scatter +// prefetch hint. +func isEvexPrefGather(upper string) bool { + _, ok := evexPrefGatherTable[upper] + return ok +} + +// evexRegFormTable holds the register-count twin of the immediate-form +// entries in evexTable. Several mnemonics name two encodings: an immediate +// count or control ($imm, src, dst …) and a register-count one whose second +// operand is a vector register or memory (count, src2, src1, dst). The +// immediate spelling lives in evexTable, this table carries the register +// spelling, and encodeEvex picks by whether the first operand is an +// immediate, the way vexVarShift does on the VEX side. +var evexRegFormTable = map[string]evexSpec{ + "VPSLLD": {1, 0xF2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}}, + "VPSLLQ": {1, 0xF3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}}, + "VPSLLW": {1, 0xF1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}}, + "VPSRAD": {1, 0xE2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}}, + "VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}}, + "VPSRAW": {1, 0xE1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}}, + "VPSRLD": {1, 0xD2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}}, + "VPSRLQ": {1, 0xD3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}}, + "VPSRLW": {1, 0xD1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}}, + // EVEX.NDS.0F38.W1, the register-count permutes (the immediate + // controls live in evexTable under 0F3A). + "VPERMQ": {2, 0x36, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMPD": {2, 0x16, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + // EVEX.NDS.0F38, the register-count permil shuffles. + "VPERMILPS": {2, 0x0C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMILPD": {2, 0x0D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, +} + +// evexPrefGatherSpec describes a gather/scatter prefetch hint: one memory +// operand with a VSIB index and an opmask register, no destination. The +// ModRM.reg field carries a fixed /digit, the L'L field is fixed at 512, and +// the mask register is the instruction's only register operand. +type evexPrefGatherSpec struct { + mapSel int + opcode byte + w int + pp int + opdigit int + n int +} + +var evexPrefGatherTable = map[string]evexPrefGatherSpec{ + "VGATHERPF0DPD": {2, 0xC6, 1, 1, 1, 8}, + "VGATHERPF0DPS": {2, 0xC6, 0, 1, 1, 4}, + "VGATHERPF0QPD": {2, 0xC7, 1, 1, 1, 8}, + "VGATHERPF0QPS": {2, 0xC7, 0, 1, 1, 4}, + "VGATHERPF1DPD": {2, 0xC6, 1, 1, 2, 8}, + "VGATHERPF1DPS": {2, 0xC6, 0, 1, 2, 4}, + "VGATHERPF1QPD": {2, 0xC7, 1, 1, 2, 8}, + "VGATHERPF1QPS": {2, 0xC7, 0, 1, 2, 4}, + "VSCATTERPF0DPD": {2, 0xC6, 1, 1, 5, 8}, + "VSCATTERPF0DPS": {2, 0xC6, 0, 1, 5, 4}, + "VSCATTERPF0QPD": {2, 0xC7, 1, 1, 5, 8}, + "VSCATTERPF0QPS": {2, 0xC7, 0, 1, 5, 4}, + "VSCATTERPF1DPD": {2, 0xC6, 1, 1, 6, 8}, + "VSCATTERPF1DPS": {2, 0xC6, 0, 1, 6, 4}, + "VSCATTERPF1QPD": {2, 0xC7, 1, 1, 6, 8}, + "VSCATTERPF1QPS": {2, 0xC7, 0, 1, 6, 4}, +} + +// evexHptrSpec describes the high/low half moves (VMOVHPS family): the +// three-operand insert shares an opcode with a two-operand store whose +// source is the vector register and whose destination is m64. +type evexHptrSpec struct { + insert evexSpec + store evexSpec // store.opcode == 0 when the mnemonic has no store form +} + +var evexHptrTable = map[string]evexHptrSpec{ + "VMOVHPS": { + insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}}, + store: evexSpec{mapSel: 1, opcode: 0x17, w: 0, pp: 0, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}}, + }, + "VMOVLHPS": { + insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}}, + }, +} + +// encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib. +func (e *enc) encodeEvexPrefGather(upper string, ps evexPrefGatherSpec, ops []Operand, mask int, sfx evexSuffix) error { + if len(ops) != 1 { + return fmt.Errorf("%s expects 2 operands (K, vsib memory), got %d", upper, len(ops)+1) + } + m, ok := ops[0].(Mem) + if !ok || !m.HasIndex || !m.Index.isVec() { + return fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", upper) + } + spec := evexSpec{mapSel: ps.mapSel, opcode: ps.opcode, w: ps.w, pp: ps.pp, opdigit: ps.opdigit, n: [3]int{ps.n, ps.n, ps.n}} + return e.emitEvexFields(spec, 2, ps.opdigit, -1, m, mask, sfx) +} + // vsibLen validates a VSIB memory operand (the index must be a vector // register) and returns it with the vector length the index selects, the // EVEX L'L field follows the index register, not the data register. @@ -1370,7 +1831,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS return err } if mask != 0 || sfx.any() { - // EVEX form: OP vsib, K, dst. + // EVEX form: OP vsib, K, dst. The L'L field is the wider of the + // index and the data register lengths (the Go assembler's + // layout); the disp8×N multiplier stays the index element size. if len(rest) != 2 { return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops)) } @@ -1382,6 +1845,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS if !ok || !dst.isVec() { return fmt.Errorf("%s: destination must be a vector register", upper) } + if d := dst.vecLenBit(); d > ll { + ll = d + } evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}} return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx) } @@ -1431,6 +1897,11 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex if err != nil { return err } + // The L'L field is the wider of the data register and the VSIB index + // lengths, the bytes go tool asm emits. + if d := src.vecLenBit(); d > ll { + ll = d + } evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}} return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx) } @@ -1441,6 +1912,8 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex var evexKOperand = map[string]bool{ "VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true, "VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true, + // The K-to-vector broadcast reads its opmask source from r/m. + "VPBROADCASTMB2Q": true, "VPBROADCASTMW2D": true, } // kmovSpec describes a KMOV width: the opcode depends on the operand @@ -1541,6 +2014,7 @@ var kOpsTable = map[string]kOpSpec{ "KXORD": {1, 0x47, 1, 1, 1, vexNDS3}, "KXORQ": {1, 0x47, 1, 0, 1, vexNDS3}, "KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3}, + "KUNPCKWD": {1, 0x4B, 0, 0, 1, vexNDS3}, "KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3}, "KADDB": {1, 0x4A, 0, 1, 1, vexNDS3}, "KADDW": {1, 0x4A, 0, 0, 1, vexNDS3}, diff --git a/asm/evex_test.go b/asm/evex_test.go index 19a8c16..f3a31d4 100644 --- a/asm/evex_test.go +++ b/asm/evex_test.go @@ -721,3 +721,92 @@ func hexCompact(b []byte) string { } return string(out) } + +// TestAvx512CorpusFamilies pins representative encodings of the AVX-512 +// families the toolchain's avx512enc corpus exercises: the bytes are the +// go tool asm output for exactly these operands, and the same families are +// covered end to end by the avx512_amd64.s differential kernel. +func TestAvx512CorpusFamilies(t *testing.T) { + vsib := func(base, idx string, scale int) Operand { + return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0) + } + cases := []struct { + name string + mnem string + ops []Operand + want string + }{ + // AES rounds (EVEX NDS, VEX twin routed by operand width). + {"VAESDEC Z", "VAESDEC", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d48ded9"}, + // Integer VNNI and the bit algorithm group. + {"VPDPBUSD", "VPDPBUSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f26d4a50d9"}, + {"VPOPCNTW", "VPOPCNTW", []Operand{vreg(t, "Z1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f2fd4b54d1"}, + {"VPCONFLICTD", "VPCONFLICTD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d49c4d1"}, + {"VPLZCNTQ masked", "VPLZCNTQ", []Operand{vreg(t, "Z7"), vreg(t, "K1"), vreg(t, "Z8")}, "6272fd4944c7"}, + {"VPERMT2B", "VPERMT2B", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f26d497dd9"}, + {"VPMULTISHIFTQB", "VPMULTISHIFTQB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f2ed4b83e1"}, + {"VDBPSADBW", "VDBPSADBW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z3")}, "62f36d4b42d903"}, + {"VPSHUFBITQMB", "VPSHUFBITQMB", []Operand{vreg(t, "Z9"), vreg(t, "Z10"), vreg(t, "K3")}, "62d22d488fd9"}, + {"VPTESTNMQ", "VPTESTNMQ", []Operand{vreg(t, "Z13"), vreg(t, "Z14"), vreg(t, "K5")}, "62d28e4827ed"}, + // Permutations: immediate and register counts. + {"VALIGNQ", "VALIGNQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f3ed4903d903"}, + {"VPERMQ imm", "VPERMQ", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f3fd4a00d101"}, + {"VPERMQ reg", "VPERMQ", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K2"), vreg(t, "Z5")}, "62f2dd4a36eb"}, + {"VPERMPD reg", "VPERMPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4816d9"}, + {"VPERMILPS imm", "VPERMILPS", []Operand{Imm(5), vreg(t, "Z9"), vreg(t, "K2"), vreg(t, "Z10")}, "62537d4a04d105"}, + {"VPERMILPS reg", "VPERMILPS", []Operand{vreg(t, "Z11"), vreg(t, "Z12"), vreg(t, "K2"), vreg(t, "Z13")}, "62521d4a0ceb"}, + // Shifts: immediate, register-count and memory-count forms; the + // count source carries its own XMM tuple width. + {"VPSLLW imm mask", "VPSLLW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f16d4a71f103"}, + {"VPSLLD reg count", "VPSLLD", []Operand{vreg(t, "X1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49f2d9"}, + {"VPSLLDQ", "VPSLLDQ", []Operand{Imm(9), vreg(t, "Z7"), vreg(t, "Z8")}, "62f13d4873ff09"}, + {"VPSRLDQ mem", "VPSRLDQ", []Operand{Imm(11), Ptr(SI, 16, 16), vreg(t, "Z4")}, "62f15d48739e100000000b"}, + {"VPSRLVW", "VPSRLVW", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K1"), vreg(t, "Z5")}, "62f2dd4910eb"}, + // Conversions and shuffles with the F2 prefix and no prefix. + {"VCVTUDQ2PS", "VCVTUDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f17f497ad1"}, + {"VSHUFPS", "VSHUFPS", []Operand{Imm(2), vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f15449c6f402"}, + // Gather and scatter prefetch hints (memory-only, /digit in reg). + {"VGATHERPF0DPD", "VGATHERPF0DPD", []Operand{vreg(t, "K5"), vsib("R10", "Y29", 8)}, "6292fd45c60cea"}, + {"VSCATTERPF1DPS", "VSCATTERPF1DPS", []Operand{vreg(t, "K2"), vsib("R10", "Z28", 4)}, "62927d42c634a2"}, + // Opmask broadcasts and the K logic. + {"VPBROADCASTMB2Q", "VPBROADCASTMB2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe482ad1"}, + {"VPBROADCASTMW2D", "VPBROADCASTMW2D", []Operand{vreg(t, "K3"), vreg(t, "Z4")}, "62f27e483ae3"}, + {"KUNPCKWD", "KUNPCKWD", []Operand{vreg(t, "K6"), vreg(t, "K4"), vreg(t, "K1")}, "c5dc4bce"}, + {"KADDB", "KADDB", []Operand{vreg(t, "K2"), vreg(t, "K3"), vreg(t, "K5")}, "c5e54aea"}, + // Lane extracts to general registers (EVEX and VEX routes). + {"VPEXTRB", "VPEXTRB", []Operand{Imm(3), vreg(t, "X26"), AX}, "62637d0814d003"}, + {"VPEXTRD", "VPEXTRD", []Operand{Imm(1), vreg(t, "X26"), vreg(t, "R9")}, "62437d0816d101"}, + {"VPEXTRD vex", "VPEXTRD", []Operand{Imm(1), vreg(t, "X2"), DI}, "c4e37916d701"}, + {"VPINSRQ", "VPINSRQ", []Operand{Imm(1), DI, vreg(t, "X3"), vreg(t, "X4")}, "c4e3e122e701"}, + // Moves: masked unaligned, masked scalar register form, half moves + // and non-temporal stores. + {"VMOVUPS mask", "VMOVUPS", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f17c4a11cb"}, + {"VMOVSD 3op", "VMOVSD", []Operand{vreg(t, "X14"), vreg(t, "X5"), vreg(t, "K3"), vreg(t, "X22")}, "6231d70b11f6"}, + {"VMOVSS 3op", "VMOVSS", []Operand{vreg(t, "X18"), vreg(t, "X3"), vreg(t, "K2"), vreg(t, "X25")}, "6281660a11d1"}, + {"VMOVHPS insert", "VMOVHPS", []Operand{Ptr(SI, 0, 8), vreg(t, "X18"), vreg(t, "X19")}, "62e16c00161e"}, + {"VMOVHPS store", "VMOVHPS", []Operand{vreg(t, "X20"), Ptr(SI, 8, 8)}, "62e17c08176601"}, + {"VMOVLHPS", "VMOVLHPS", []Operand{vreg(t, "X16"), vreg(t, "X5"), vreg(t, "X17")}, "62a1540816c8"}, + {"VMOVNTDQ", "VMOVNTDQ", []Operand{vreg(t, "Z7"), Ptr(SI, 0, 64)}, "62f17d48e73e"}, + {"VMOVNTDQA", "VMOVNTDQA", []Operand{Ptr(SI, 64, 64), vreg(t, "Z8")}, "62727d482a4601"}, + {"VMOVNTPS", "VMOVNTPS", []Operand{vreg(t, "Z9"), Ptr(SI, 0, 64)}, "62717c482b0e"}, + // Scalar compares with and without the 66 prefix. + {"VCOMISD", "VCOMISD", []Operand{vreg(t, "X5"), vreg(t, "X6")}, "c5f92ff5"}, + {"VUCOMISS", "VUCOMISS", []Operand{vreg(t, "X7"), vreg(t, "X8")}, "c5782ec7"}, + // Floating point helpers. + {"VSQRTSD", "VSQRTSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f1ef0951d9"}, + {"VEXP2PD", "VEXP2PD", []Operand{vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f2fd49c8f5"}, + {"VRCP28SD", "VRCP28SD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "K1"), vreg(t, "X10")}, "6252bd09cbd1"}, + {"VBROADCASTF32X2", "VBROADCASTF32X2", []Operand{vreg(t, "X1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d4919d1"}, + {"VPCOMPRESSB", "VPCOMPRESSB", []Operand{vreg(t, "Z1"), vreg(t, "K1"), Ptr(SI, 0, 64)}, "62f27d49630e"}, + } + for _, c := range cases { + code, err := Encode(c.mnem, c.ops...) + if err != nil { + t.Errorf("%s: Encode: %v", c.name, err) + continue + } + if got := hexCompact(code); got != c.want { + t.Errorf("%s: got %s, want %s", c.name, got, c.want) + } + } +} diff --git a/testdata/verify/avx512_amd64.s b/testdata/verify/avx512_amd64.s new file mode 100644 index 0000000..e0d045d --- /dev/null +++ b/testdata/verify/avx512_amd64.s @@ -0,0 +1,191 @@ +// The AVX-512 families behind the avx512enc gap: AES round ops, integer +// VNNI and bit algorithms, word shifts and permutes with an immediate or a +// register count, lane broadcasts and extracts, gather and scatter prefetch +// hints, opmask broadcasts, the high/low half moves and the non-temporal +// stores. Every result is folded back so no instruction is dead. + +#include "textflag.h" + +// func avx512int(p *byte, n int) uint64 +TEXT ·avx512int(SB), NOSPLIT, $0-24 + MOVQ p+0(FP), SI + MOVQ n+16(FP), CX + // AES rounds through the EVEX spellings, masks included. + VAESENC Z20, Z21, Z22 + VAESENCLAST Z23, Z24, Z25 + VAESDEC (SI), Z26, Z27 + VAESDECLAST Z28, Z29, Z30 + // Integer VNNI and the bit algorithm group. + VPDPBUSD Z1, Z2, K2, Z3 + VPDPBUSDS Z4, Z5, K2, Z6 + VPDPWSSD Z7, Z8, Z9 + VPDPWSSDS Z10, Z11, K2, Z12 + VPOPCNTW Z12, K3, Z13 + VPOPCNTB Z14, Z15 + VGF2P8MULB Z16, Z17, K4, Z18 + VGF2P8AFFINEQB $7, Z18, Z19, K5, Z20 + // Byte/word arithmetic with saturation and masks. + VPADDSB Z1, Z2, K1, Z3 + VPADDUSW Z3, Z4, K1, Z5 + VPSUBSW Z5, Z6, K1, Z7 + VPSUBUSB Z7, Z8, K1, Z9 + VPSADBW Z9, Z10, Z11 + VPMULHRSW Z11, Z12, Z13 + VPMULHW Z13, Z14, Z15 + VPUNPCKLBW Z15, Z16, K2, Z17 + VPUNPCKHBW Z17, Z18, K2, Z19 + VPUNPCKLWD Z19, Z20, K2, Z21 + VPUNPCKHWD Z21, Z22, K2, Z23 + VPCMPEQB Z23, Z24, K2, K3 + VPCMPGTW Z25, Z26, K2, K3 + VPCMPEQQ Z27, Z28, K2 + VPMULTISHIFTQB Z29, Z30, K3, Z31 + VDBPSADBW $3, Z1, Z2, K3, Z3 + MOVQ CX, ret+16(FP) + RET + +// func avx512perm(p *byte) uint64 +TEXT ·avx512perm(SB), NOSPLIT, $0-16 + MOVQ p+0(FP), SI + // Permutations: immediate and register counts, ternary logic. + VALIGNQ $3, Z1, Z2, K1, Z3 + VPERMT2B Z3, Z4, K1, Z5 + VPERMT2W Z5, Z6, K1, Z7 + VPERMT2PS Z7, Z8, K1, Z9 + VPERMI2W Z9, Z10, K1, Z11 + VPERMI2PS Z11, Z12, K1, Z13 + VPERMI2PD Z13, Z14, K1, Z15 + VPERMB Z15, Z16, K1, Z17 + VPERMW Z17, Z18, K1, Z19 + VPERMPS Z19, Z20, Z21 + VPERMD Z20, Z21, Z22 + VPERMQ $1, Z1, K2, Z2 + VPERMQ Z3, Z4, K2, Z5 + VPERMPD $1, Z5, K2, Z6 + VPERMPD Z7, Z8, K2, Z9 + VPERMILPS $5, Z9, K2, Z10 + VPERMILPS Z11, Z12, K2, Z13 + VPERMILPD $1, Z13, K2, Z14 + VPERMILPD Z15, Z16, K2, Z17 + VPTERNLOGD $6, Z17, Z18, K2, Z19 + VPTERNLOGQ $9, Z19, Z20, K2, Z21 + // Lane shuffle and blend families. + VSHUFPD $1, Z1, Z2, K1, Z3 + VSHUFPS $2, Z4, Z5, K1, Z6 + VBLENDMPD Z7, Z8, K1, Z9 + VBLENDMPS Z9, Z10, K1, Z11 + VPBLENDMB Z11, Z12, K1, Z13 + VPBLENDMW Z13, Z14, K1, Z15 + VPBLENDMD Z15, Z16, K1, Z17 + VPBLENDMQ Z17, Z18, K1, Z19 + // Conflicts and leading zero counts. + VPCONFLICTD Z1, K1, Z2 + VPCONFLICTQ Z3, K1, Z4 + VPLZCNTD Z5, K1, Z6 + VPLZCNTQ Z7, K1, Z8 + // Compress and expand, byte and word widths. + VPCOMPRESSB Z1, K1, (SI) + VPCOMPRESSW Z2, K1, (SI) + VPEXPANDB (SI), K1, Z3 + VPEXPANDW (SI), K1, Z4 + MOVQ SI, ret+8(FP) + RET + +// func avx512shift(p *byte) uint64 +TEXT ·avx512shift(SB), NOSPLIT, $0-16 + MOVQ p+0(FP), SI + // Variable shifts and shuffles with masks. + VPSLLVW Z1, Z2, K1, Z3 + VPSRLVW Z3, Z4, K1, Z5 + VPSRAVW Z5, Z6, K1, Z7 + VPSHLDVW Z7, Z8, K1, Z9 + VPSHRDVW Z9, Z10, K1, Z11 + VPSHLDVD Z11, Z12, K1, Z13 + VPSHLDVQ Z13, Z14, K1, Z15 + VPSHRDVD Z15, Z16, K1, Z17 + VPSHRDVQ Z17, Z18, K1, Z19 + // Immediate shifts, the word/byte-quad widths and masks. + VPSLLW $3, Z1, K2, Z2 + VPSRLW $5, Z3, K2, Z4 + VPSRAW $7, Z5, K2, Z6 + VPSLLDQ $9, Z7, Z8 + VPSRLDQ $11, Z9, Z10 + // Register-count shifts and their memory-count forms. + VPSLLD X1, Z2, K1, Z3 + VPSRLD 16(SI), Z4, K1, Z5 + VPSLLQ X6, Z7, K1, Z8 + VPSRLQ X9, Z10, K1, Z11 + VPSLLW X12, Z13, K1, Z14 + VPSRAW X15, Z16, K1, Z17 + VPSRAQ $13, Z12, K1, Z13 + VPSRAD X14, Z15, K1, Z16 + // Lane shuffles in and out. + VPSHLDW $2, Z1, Z2, K1, Z3 + VPSHLDQ $4, Z3, Z4, K1, Z5 + VPSHRDW $6, Z5, Z6, K1, Z7 + VPSHRDQ $8, Z7, Z8, K1, Z9 + VPSHUFBITQMB Z9, Z10, K3 + VPTESTMB Z11, Z12, K4 + VPTESTNMQ Z13, Z14, K5 + MOVQ SI, ret+8(FP) + RET + +// func avx512float(x float64) float64 +TEXT ·avx512float(SB), NOSPLIT, $0-16 + // Square roots, compares and the EXP2/RCP28 helpers. + MOVQ x+0(FP), AX + VSQRTPD Z1, K1, Z2 + VSQRTPS Z3, K1, Z4 + VSQRTSD X1, X2, K1, X3 + VSQRTSS X3, X4, X5 + VCOMISD X5, X6 + VUCOMISS X7, X8 + VEXP2PD Z5, K1, Z6 + VRCP28PD Z7, K1, Z8 + VRCP28SD X9, X8, K1, X10 + VRSQRT28PS Z11, K1, Z12 + VRSQRT28SS X11, X10, K1, X12 + VCVTSD2SS X1, X2, X3 + VCVTSS2SD X3, X2, K1, X4 + VFMADD132PD Z1, Z2, K1, Z3 + VFMADD231SD X1, X2, K1, X3 + VFMSUBADD213PS Z3, Z4, K1, Z5 + VFNMSUB231PD Z5, Z6, K1, Z7 + // Broadcasts and masked moves. + VBROADCASTF32X2 X1, K1, Z2 + VBROADCASTI64X2 (SI), K1, Z3 + VMOVUPS Z1, K2, Z3 + VMOVSD X14, X5, K3, X22 + VMOVSS X18, X3, K2, X25 + VMOVHPS (SI), X18, X19 + VMOVHPS X20, 8(SI) + VMOVLHPS X16, X5, X17 + VMOVNTDQ Z7, (SI) + VMOVNTDQA 64(SI), Z8 + VMOVNTPD Z9, (SI) + MOVQ SI, ret+8(FP) + RET + +// func avx512mask(p *byte) uint64 +TEXT ·avx512mask(SB), NOSPLIT, $0-16 + MOVQ p+0(FP), SI + // Omask broadcasts and the K register logic. + VPBROADCASTMB2Q K1, Z2 + VPBROADCASTMW2D K3, Z4 + KUNPCKWD K6, K4, K1 + KADDB K2, K3, K5 + KORW K1, K2, K7 + // Gather and scatter prefetch hints. + VGATHERPF0DPD K5, (SI)(Y29*8) + VSCATTERPF1DPS K2, (SI)(Z28*4) + // Masked gathers ride the EVEX spelling; the data length wins L'L. + VGATHERDPD (SI)(X10*4), K7, Y22 + VPSCATTERDQ Y6, K2, (SI)(X4*1) + // Lane extracts to general registers. + VPEXTRB $3, X1, AX + VPEXTRD $1, X2, DI + VPINSRQ $1, SI, X3, X4 + VEXTRACTI32X4 $1, Z1, X5 + VINSERTI64X2 $1, X6, Z7, K2, Z8 + MOVQ SI, ret+8(FP) + RET