feat(amd64): encode the AVX-512 and BMI corpus families

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 21:17:20 +02:00
parent 75e9fd771b
commit 81e2673923
4 changed files with 777 additions and 22 deletions
+1
View File
@@ -113,6 +113,7 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return err
}
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
isEvexPrefGather(base) ||
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
return e.encodeVec(base, ops, sfx)
}
+496 -22
View File
@@ -5,6 +5,7 @@ package asm
import (
"fmt"
"slices"
"strings"
)
@@ -91,7 +92,7 @@ var evexTable = map[string]evexSpec{
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
// the W bit distinguishes it from VPSRAD's E2 form).
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRAQ": {1, 0x72, 1, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
// rm=src, no vvvv).
@@ -138,7 +139,7 @@ var evexTable = map[string]evexSpec{
// EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFPS": {1, 0xC6, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
@@ -202,7 +203,7 @@ var evexTable = map[string]evexSpec{
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLVW": {2, 0x10, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -324,7 +325,7 @@ var evexTable = map[string]evexSpec{
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PS": {1, 0x7A, 0, 3, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F38, half-precision convert (half-width source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
@@ -494,6 +495,246 @@ var evexTable = map[string]evexSpec{
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
// --- the AVX-512 families the avx512enc corpus exercises, read off
// the toolchain opcodetables ---
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VALIGNQ": {3, 0x03, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VANDNPD": {1, 0x55, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VANDPD": {1, 0x54, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VBLENDMPD": {2, 0x65, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VBLENDMPS": {2, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VBROADCASTF32X2": {2, 0x19, 0, 1, -1, vexRM, [3]int{0, 8, 8}},
"VBROADCASTF32X4": {2, 0x1A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTF32X8": {2, 0x1B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
"VBROADCASTF64X2": {2, 0x1A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTF64X4": {2, 0x1B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
"VBROADCASTI32X2": {2, 0x59, 0, 1, -1, vexRM, [3]int{8, 8, 8}},
"VBROADCASTI32X4": {2, 0x5A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTI32X8": {2, 0x5B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
"VBROADCASTI64X2": {2, 0x5A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTI64X4": {2, 0x5B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
"VCOMISD": {1, 0x2F, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
"VCVTSD2SS": {1, 0x5A, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
"VDBPSADBW": {3, 0x42, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VEXP2PD": {2, 0xC8, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
"VEXP2PS": {2, 0xC8, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev, [3]int{16, 32, 64}},
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VMOVNTPD": {1, 0x2B, 1, 1, -1, vexRMRev, [3]int{16, 32, 64}},
"VORPD": {1, 0x56, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMB": {2, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMD": {2, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMQ": {2, 0x64, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMW": {2, 0x66, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBROADCASTMB2Q": {2, 0x2A, 1, 2, -1, vexRM, [3]int{0, 0, 0}},
"VPBROADCASTMW2D": {2, 0x3A, 0, 2, -1, vexRM, [3]int{0, 0, 0}},
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPEQQ": {2, 0x29, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTQ": {2, 0x37, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCOMPRESSB": {2, 0x63, 0, 1, -1, vexRMRev, [3]int{1, 1, 1}},
"VPCOMPRESSW": {2, 0x63, 1, 1, -1, vexRMRev, [3]int{2, 2, 2}},
"VPCONFLICTD": {2, 0xC4, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPCONFLICTQ": {2, 0xC4, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPDPBUSD": {2, 0x50, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPDPBUSDS": {2, 0x51, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPDPWSSD": {2, 0x52, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPDPWSSDS": {2, 0x53, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2PD": {2, 0x77, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2PS": {2, 0x77, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2W": {2, 0x75, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
"VPERMT2B": {2, 0x7D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2PS": {2, 0x7F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2W": {2, 0x7D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPEXPANDB": {2, 0x62, 0, 1, -1, vexRM, [3]int{1, 1, 1}},
"VPEXPANDW": {2, 0x62, 1, 1, -1, vexRM, [3]int{2, 2, 2}},
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm, [3]int{4, 0, 0}},
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm, [3]int{8, 0, 0}},
"VPLZCNTD": {2, 0x44, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPLZCNTQ": {2, 0x44, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPMADD52HUQ": {2, 0xB5, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMADD52LUQ": {2, 0xB4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULDQ": {2, 0x28, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULTISHIFTQB": {2, 0x83, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULUDQ": {1, 0xF4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPOPCNTW": {2, 0x54, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPORD": {1, 0xEB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPROLVD": {2, 0x15, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPROLVQ": {2, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPRORVD": {2, 0x14, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPRORVQ": {2, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDD": {3, 0x71, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHLDQ": {3, 0x71, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHLDVD": {2, 0x71, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDVQ": {2, 0x71, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDVW": {2, 0x70, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDW": {3, 0x70, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHRDD": {3, 0x73, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHRDQ": {3, 0x73, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHRDVD": {2, 0x73, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHRDVQ": {2, 0x73, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHRDVW": {2, 0x72, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHRDW": {3, 0x72, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHUFBITQMB": {2, 0x8F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRAVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.66.0F73 /7, the byte-quad shift left (the count is always an
// immediate; there is no register-count twin).
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0, the plain-prefix (no 66) packed spellings
// whose EVEX form drops the legacy prefix entirely.
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
"VCOMISS": {1, 0x2F, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev, [3]int{16, 32, 64}},
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMB": {2, 0x26, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMD": {2, 0x27, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMQ": {2, 0x27, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMW": {2, 0x26, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMB": {2, 0x26, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMD": {2, 0x27, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMQ": {2, 0x27, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMW": {2, 0x26, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHQDQ": {1, 0x6D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLQDQ": {1, 0x6C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VRCP28PD": {2, 0xCA, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRCP28PS": {2, 0xCA, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRCP28SD": {2, 0xCB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VRCP28SS": {2, 0xCB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VRSQRT28PD": {2, 0xCC, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRSQRT28PS": {2, 0xCC, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRSQRT28SD": {2, 0xCD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VRSQRT28SS": {2, 0xCD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VSQRTPD": {1, 0x51, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VSQRTSD": {1, 0x51, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
"VUCOMISD": {1, 0x2E, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
"VXORPD": {1, 0x57, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.F3/F2.W0, word shuffles with an immediate
// ($imm, src, dst: reg = dst, rm = src, imm8). The F3/F2 prefixes
// split the high/low lane spellings.
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM, [3]int{16, 32, 64}},
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.128.66.0F3A, lane extract to a general-purpose register or
// memory ($imm, xsrc, GPR/mem dst: reg = source, rm = destination).
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtractGPR, [3]int{1, 1, 1}},
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR, [3]int{2, 2, 2}},
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtractGPR, [3]int{4, 4, 4}},
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtractGPR, [3]int{8, 8, 8}},
// EVEX.66.0F3A.W1, the qword permutes with an immediate control
// ($imm, src, dst: reg = dst, rm = src, imm8); the register-count
// forms live in evexRegFormTable.
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.66.0F3A, the packed permute shuffles with an immediate control.
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPERMILPD": {3, 0x05, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.128.0F.W0, high/low half moves. VMOVHPS carries the
// three-operand insert form (rm = m64 source, vvvv = preserved,
// reg = dst) and the two-operand store (reg = source, rm = m64);
// the encoder splits on the operand count. VMOVLHPS is the
// three-operand form alone.
"VMOVHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
"VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
}
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
@@ -529,32 +770,38 @@ type evexMoveSpec struct {
n [3]int
vecOK bool // the non-memory operand may be a vector register
xmmOnly bool // wider than XMM registers are rejected
nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS)
}
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false},
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false},
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true},
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true},
// EVEX.128.F2.0F.W1, scalar double move: memory operands and the
// three-operand register form (VMOVSD dst, src1, src2).
"VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true},
// EVEX.128/256/512.0F.W0, unaligned packed single move.
"VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false},
}
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
@@ -579,6 +826,13 @@ func evexRequired(upper string, ops []Operand) bool {
if !inVex && !inVexMove {
return true // EVEX-only mnemonic
}
// The byte-quad shifts have VEX register forms but EVEX-only memory
// forms: a memory count source forces the EVEX encoding.
if upper == "VPSLLDQ" || upper == "VPSRLDQ" {
if slices.ContainsFunc(ops, memOperand) {
return true
}
}
for _, op := range ops {
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
return true
@@ -752,6 +1006,27 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
}
spec.n = [3]int{n, n, n}
}
// A mnemonic with an immediate and a register spelling (the
// variable-count shifts, the permutes) encodes the register one
// when the first operand is not an immediate.
if len(ops) > 0 {
if _, isImm := ops[0].(Imm); !isImm {
if alt, ok := evexRegFormTable[mnemUpper]; ok {
spec, inTable = alt, true
}
}
}
// The high/low half moves split by operand count: three operands
// insert, two store (VMOVHPS m64, X1).
if hs, ok := evexHptrTable[mnemUpper]; ok {
if len(ops) == 2 {
if hs.store.opcode == 0 {
return fmt.Errorf("%s has no two-operand form", mnemUpper)
}
return e.encodeEvexRMRev(hs.store, ops, 0, sfx)
}
spec = hs.insert
}
} else if sfx.evexOnly() {
return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper)
}
@@ -811,6 +1086,12 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
}
return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx)
}
if ps, ok := evexPrefGatherTable[mnemUpper]; ok {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
}
return e.encodeEvexPrefGather(mnemUpper, ps, ops, mask, sfx)
}
if !inTable {
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
}
@@ -829,6 +1110,8 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
return e.encodeEvexNDS3Imm(spec, ops, mask, sfx)
case vexExtract:
return e.encodeEvexExtract(spec, ops, mask, sfx)
case vexExtractGPR:
return e.encodeEvexExtractGPR(spec, ops, mask, sfx)
case vexRMSrcLen:
return e.encodeEvexRMSrcLen(spec, ops, mask, sfx)
}
@@ -908,6 +1191,11 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSu
if dstReg.mask {
if r, ok := src.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
} else if l, err := soleLen(spec.n); err == nil {
// A memory source with a length-fixed mnemonic
// (VFPCLASSPDX/Y/Z): the length comes from the table's
// single valid slot, not from the operand.
ll = l
}
} else if r, ok := src.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
@@ -934,9 +1222,11 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve
if !ok {
return fmt.Errorf("shift count must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("shift source must be a vector register")
// The count source is a vector register or memory; the length the L'L
// field and the disp8×N multiplier follow is the destination's either
// way.
if !vecOrMem(src) {
return fmt.Errorf("shift source must be a vector register or memory")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
@@ -946,7 +1236,7 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve
if err != nil {
return err
}
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, sfx); err != nil {
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, src, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
@@ -1018,10 +1308,77 @@ func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, sfx evex
return nil
}
// encodeEvexExtractGPR encodes the lane extract to a general-purpose
// register or memory: OP $imm, xsrc, dst (reg = the XMM source, rm = the
// destination, imm8). The encoding is 128-bit regardless of register
// numbers, so L'L is fixed at 0 and the disp8×N multiplier is the extracted
// element size the table carries.
func (e *enc) encodeEvexExtractGPR(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("extract source must be a vector register")
}
switch dst.(type) {
case Reg:
if dst.(Reg).isVec() {
return fmt.Errorf("extract destination must be a general-purpose register or memory")
}
case Mem, sbMem:
default:
return fmt.Errorf("extract destination must be a general-purpose register or memory")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, 0, srcReg.idx, -1, dst, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
// the store-form opcode (reg = source, rm = destination), matching the Go
// assembler.
// assembler. The scalar moves also carry a three-operand register form
// (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3
// opens.
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) == 3 {
if !ms.nds3 {
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
}
// The masked scalar register form keeps the Go assembler's own
// layout: the store opcode with reg = op0, vvvv = op1 and the
// destination in r/m (op2) — the bytes go tool asm emits, not
// the manual's NDS reading.
src, src1, dst := ops[0], ops[1], ops[2]
reg, ok := src.(Reg)
if !ok || !reg.isVec() {
return fmt.Errorf("%s: first operand must be a vector register", mnem)
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("%s: second operand must be a vector register", mnem)
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
if ms.xmmOnly && (reg.size != 16 || vvvvReg.size != 16 || dstReg.size != 16) {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
spec := evexSpec{mapSel: ms.mapSel, opcode: ms.store, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
return e.emitEvexFields(spec, dstReg.vecLenBit(), reg.idx, vvvvReg.idx, dst, mask, sfx)
}
if len(ops) != 2 {
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
}
@@ -1141,12 +1498,20 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx eve
return fmt.Errorf("broadcast destination must be a vector register")
}
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
switch src.(type) {
switch r := src.(type) {
case Mem, sbMem:
spec.opcode = bs.opMem
spec.n = [3]int{bs.n, bs.n, bs.n}
case Reg:
spec.opcode = bs.opReg
// A GPR source uses the register broadcast opcode; a vector
// source shares the xmm/mem one (the low byte is copied from
// the lane or from the memory operand).
if r.isVec() {
spec.opcode = bs.opMem
spec.n = [3]int{bs.n, bs.n, bs.n}
} else {
spec.opcode = bs.opReg
}
default:
return fmt.Errorf("broadcast source must be a register or memory")
}
@@ -1349,6 +1714,102 @@ func isScatter(upper string) bool {
return ok
}
// isEvexPrefGather reports whether the mnemonic is a gather/scatter
// prefetch hint.
func isEvexPrefGather(upper string) bool {
_, ok := evexPrefGatherTable[upper]
return ok
}
// evexRegFormTable holds the register-count twin of the immediate-form
// entries in evexTable. Several mnemonics name two encodings: an immediate
// count or control ($imm, src, dst …) and a register-count one whose second
// operand is a vector register or memory (count, src2, src1, dst). The
// immediate spelling lives in evexTable, this table carries the register
// spelling, and encodeEvex picks by whether the first operand is an
// immediate, the way vexVarShift does on the VEX side.
var evexRegFormTable = map[string]evexSpec{
"VPSLLD": {1, 0xF2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSLLQ": {1, 0xF3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSLLW": {1, 0xF1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRAD": {1, 0xE2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRAW": {1, 0xE1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRLD": {1, 0xD2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRLQ": {1, 0xD3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRLW": {1, 0xD1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
// EVEX.NDS.0F38.W1, the register-count permutes (the immediate
// controls live in evexTable under 0F3A).
"VPERMQ": {2, 0x36, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMPD": {2, 0x16, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.NDS.0F38, the register-count permil shuffles.
"VPERMILPS": {2, 0x0C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMILPD": {2, 0x0D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
}
// evexPrefGatherSpec describes a gather/scatter prefetch hint: one memory
// operand with a VSIB index and an opmask register, no destination. The
// ModRM.reg field carries a fixed /digit, the L'L field is fixed at 512, and
// the mask register is the instruction's only register operand.
type evexPrefGatherSpec struct {
mapSel int
opcode byte
w int
pp int
opdigit int
n int
}
var evexPrefGatherTable = map[string]evexPrefGatherSpec{
"VGATHERPF0DPD": {2, 0xC6, 1, 1, 1, 8},
"VGATHERPF0DPS": {2, 0xC6, 0, 1, 1, 4},
"VGATHERPF0QPD": {2, 0xC7, 1, 1, 1, 8},
"VGATHERPF0QPS": {2, 0xC7, 0, 1, 1, 4},
"VGATHERPF1DPD": {2, 0xC6, 1, 1, 2, 8},
"VGATHERPF1DPS": {2, 0xC6, 0, 1, 2, 4},
"VGATHERPF1QPD": {2, 0xC7, 1, 1, 2, 8},
"VGATHERPF1QPS": {2, 0xC7, 0, 1, 2, 4},
"VSCATTERPF0DPD": {2, 0xC6, 1, 1, 5, 8},
"VSCATTERPF0DPS": {2, 0xC6, 0, 1, 5, 4},
"VSCATTERPF0QPD": {2, 0xC7, 1, 1, 5, 8},
"VSCATTERPF0QPS": {2, 0xC7, 0, 1, 5, 4},
"VSCATTERPF1DPD": {2, 0xC6, 1, 1, 6, 8},
"VSCATTERPF1DPS": {2, 0xC6, 0, 1, 6, 4},
"VSCATTERPF1QPD": {2, 0xC7, 1, 1, 6, 8},
"VSCATTERPF1QPS": {2, 0xC7, 0, 1, 6, 4},
}
// evexHptrSpec describes the high/low half moves (VMOVHPS family): the
// three-operand insert shares an opcode with a two-operand store whose
// source is the vector register and whose destination is m64.
type evexHptrSpec struct {
insert evexSpec
store evexSpec // store.opcode == 0 when the mnemonic has no store form
}
var evexHptrTable = map[string]evexHptrSpec{
"VMOVHPS": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x17, w: 0, pp: 0, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
"VMOVLHPS": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
},
}
// encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib.
func (e *enc) encodeEvexPrefGather(upper string, ps evexPrefGatherSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 2 operands (K, vsib memory), got %d", upper, len(ops)+1)
}
m, ok := ops[0].(Mem)
if !ok || !m.HasIndex || !m.Index.isVec() {
return fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", upper)
}
spec := evexSpec{mapSel: ps.mapSel, opcode: ps.opcode, w: ps.w, pp: ps.pp, opdigit: ps.opdigit, n: [3]int{ps.n, ps.n, ps.n}}
return e.emitEvexFields(spec, 2, ps.opdigit, -1, m, mask, sfx)
}
// vsibLen validates a VSIB memory operand (the index must be a vector
// register) and returns it with the vector length the index selects, the
// EVEX L'L field follows the index register, not the data register.
@@ -1370,7 +1831,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
return err
}
if mask != 0 || sfx.any() {
// EVEX form: OP vsib, K, dst.
// EVEX form: OP vsib, K, dst. The L'L field is the wider of the
// index and the data register lengths (the Go assembler's
// layout); the disp8×N multiplier stays the index element size.
if len(rest) != 2 {
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
}
@@ -1382,6 +1845,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", upper)
}
if d := dst.vecLenBit(); d > ll {
ll = d
}
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
}
@@ -1431,6 +1897,11 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
if err != nil {
return err
}
// The L'L field is the wider of the data register and the VSIB index
// lengths, the bytes go tool asm emits.
if d := src.vecLenBit(); d > ll {
ll = d
}
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
}
@@ -1441,6 +1912,8 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
var evexKOperand = map[string]bool{
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
// The K-to-vector broadcast reads its opmask source from r/m.
"VPBROADCASTMB2Q": true, "VPBROADCASTMW2D": true,
}
// kmovSpec describes a KMOV width: the opcode depends on the operand
@@ -1541,6 +2014,7 @@ var kOpsTable = map[string]kOpSpec{
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
"KUNPCKWD": {1, 0x4B, 0, 0, 1, vexNDS3},
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
+89
View File
@@ -721,3 +721,92 @@ func hexCompact(b []byte) string {
}
return string(out)
}
// TestAvx512CorpusFamilies pins representative encodings of the AVX-512
// families the toolchain's avx512enc corpus exercises: the bytes are the
// go tool asm output for exactly these operands, and the same families are
// covered end to end by the avx512_amd64.s differential kernel.
func TestAvx512CorpusFamilies(t *testing.T) {
vsib := func(base, idx string, scale int) Operand {
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
}
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// AES rounds (EVEX NDS, VEX twin routed by operand width).
{"VAESDEC Z", "VAESDEC", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d48ded9"},
// Integer VNNI and the bit algorithm group.
{"VPDPBUSD", "VPDPBUSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f26d4a50d9"},
{"VPOPCNTW", "VPOPCNTW", []Operand{vreg(t, "Z1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f2fd4b54d1"},
{"VPCONFLICTD", "VPCONFLICTD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d49c4d1"},
{"VPLZCNTQ masked", "VPLZCNTQ", []Operand{vreg(t, "Z7"), vreg(t, "K1"), vreg(t, "Z8")}, "6272fd4944c7"},
{"VPERMT2B", "VPERMT2B", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f26d497dd9"},
{"VPMULTISHIFTQB", "VPMULTISHIFTQB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f2ed4b83e1"},
{"VDBPSADBW", "VDBPSADBW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z3")}, "62f36d4b42d903"},
{"VPSHUFBITQMB", "VPSHUFBITQMB", []Operand{vreg(t, "Z9"), vreg(t, "Z10"), vreg(t, "K3")}, "62d22d488fd9"},
{"VPTESTNMQ", "VPTESTNMQ", []Operand{vreg(t, "Z13"), vreg(t, "Z14"), vreg(t, "K5")}, "62d28e4827ed"},
// Permutations: immediate and register counts.
{"VALIGNQ", "VALIGNQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f3ed4903d903"},
{"VPERMQ imm", "VPERMQ", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f3fd4a00d101"},
{"VPERMQ reg", "VPERMQ", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K2"), vreg(t, "Z5")}, "62f2dd4a36eb"},
{"VPERMPD reg", "VPERMPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4816d9"},
{"VPERMILPS imm", "VPERMILPS", []Operand{Imm(5), vreg(t, "Z9"), vreg(t, "K2"), vreg(t, "Z10")}, "62537d4a04d105"},
{"VPERMILPS reg", "VPERMILPS", []Operand{vreg(t, "Z11"), vreg(t, "Z12"), vreg(t, "K2"), vreg(t, "Z13")}, "62521d4a0ceb"},
// Shifts: immediate, register-count and memory-count forms; the
// count source carries its own XMM tuple width.
{"VPSLLW imm mask", "VPSLLW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f16d4a71f103"},
{"VPSLLD reg count", "VPSLLD", []Operand{vreg(t, "X1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49f2d9"},
{"VPSLLDQ", "VPSLLDQ", []Operand{Imm(9), vreg(t, "Z7"), vreg(t, "Z8")}, "62f13d4873ff09"},
{"VPSRLDQ mem", "VPSRLDQ", []Operand{Imm(11), Ptr(SI, 16, 16), vreg(t, "Z4")}, "62f15d48739e100000000b"},
{"VPSRLVW", "VPSRLVW", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K1"), vreg(t, "Z5")}, "62f2dd4910eb"},
// Conversions and shuffles with the F2 prefix and no prefix.
{"VCVTUDQ2PS", "VCVTUDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f17f497ad1"},
{"VSHUFPS", "VSHUFPS", []Operand{Imm(2), vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f15449c6f402"},
// Gather and scatter prefetch hints (memory-only, /digit in reg).
{"VGATHERPF0DPD", "VGATHERPF0DPD", []Operand{vreg(t, "K5"), vsib("R10", "Y29", 8)}, "6292fd45c60cea"},
{"VSCATTERPF1DPS", "VSCATTERPF1DPS", []Operand{vreg(t, "K2"), vsib("R10", "Z28", 4)}, "62927d42c634a2"},
// Opmask broadcasts and the K logic.
{"VPBROADCASTMB2Q", "VPBROADCASTMB2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe482ad1"},
{"VPBROADCASTMW2D", "VPBROADCASTMW2D", []Operand{vreg(t, "K3"), vreg(t, "Z4")}, "62f27e483ae3"},
{"KUNPCKWD", "KUNPCKWD", []Operand{vreg(t, "K6"), vreg(t, "K4"), vreg(t, "K1")}, "c5dc4bce"},
{"KADDB", "KADDB", []Operand{vreg(t, "K2"), vreg(t, "K3"), vreg(t, "K5")}, "c5e54aea"},
// Lane extracts to general registers (EVEX and VEX routes).
{"VPEXTRB", "VPEXTRB", []Operand{Imm(3), vreg(t, "X26"), AX}, "62637d0814d003"},
{"VPEXTRD", "VPEXTRD", []Operand{Imm(1), vreg(t, "X26"), vreg(t, "R9")}, "62437d0816d101"},
{"VPEXTRD vex", "VPEXTRD", []Operand{Imm(1), vreg(t, "X2"), DI}, "c4e37916d701"},
{"VPINSRQ", "VPINSRQ", []Operand{Imm(1), DI, vreg(t, "X3"), vreg(t, "X4")}, "c4e3e122e701"},
// Moves: masked unaligned, masked scalar register form, half moves
// and non-temporal stores.
{"VMOVUPS mask", "VMOVUPS", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f17c4a11cb"},
{"VMOVSD 3op", "VMOVSD", []Operand{vreg(t, "X14"), vreg(t, "X5"), vreg(t, "K3"), vreg(t, "X22")}, "6231d70b11f6"},
{"VMOVSS 3op", "VMOVSS", []Operand{vreg(t, "X18"), vreg(t, "X3"), vreg(t, "K2"), vreg(t, "X25")}, "6281660a11d1"},
{"VMOVHPS insert", "VMOVHPS", []Operand{Ptr(SI, 0, 8), vreg(t, "X18"), vreg(t, "X19")}, "62e16c00161e"},
{"VMOVHPS store", "VMOVHPS", []Operand{vreg(t, "X20"), Ptr(SI, 8, 8)}, "62e17c08176601"},
{"VMOVLHPS", "VMOVLHPS", []Operand{vreg(t, "X16"), vreg(t, "X5"), vreg(t, "X17")}, "62a1540816c8"},
{"VMOVNTDQ", "VMOVNTDQ", []Operand{vreg(t, "Z7"), Ptr(SI, 0, 64)}, "62f17d48e73e"},
{"VMOVNTDQA", "VMOVNTDQA", []Operand{Ptr(SI, 64, 64), vreg(t, "Z8")}, "62727d482a4601"},
{"VMOVNTPS", "VMOVNTPS", []Operand{vreg(t, "Z9"), Ptr(SI, 0, 64)}, "62717c482b0e"},
// Scalar compares with and without the 66 prefix.
{"VCOMISD", "VCOMISD", []Operand{vreg(t, "X5"), vreg(t, "X6")}, "c5f92ff5"},
{"VUCOMISS", "VUCOMISS", []Operand{vreg(t, "X7"), vreg(t, "X8")}, "c5782ec7"},
// Floating point helpers.
{"VSQRTSD", "VSQRTSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f1ef0951d9"},
{"VEXP2PD", "VEXP2PD", []Operand{vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f2fd49c8f5"},
{"VRCP28SD", "VRCP28SD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "K1"), vreg(t, "X10")}, "6252bd09cbd1"},
{"VBROADCASTF32X2", "VBROADCASTF32X2", []Operand{vreg(t, "X1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d4919d1"},
{"VPCOMPRESSB", "VPCOMPRESSB", []Operand{vreg(t, "Z1"), vreg(t, "K1"), Ptr(SI, 0, 64)}, "62f27d49630e"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
}
+191
View File
@@ -0,0 +1,191 @@
// The AVX-512 families behind the avx512enc gap: AES round ops, integer
// VNNI and bit algorithms, word shifts and permutes with an immediate or a
// register count, lane broadcasts and extracts, gather and scatter prefetch
// hints, opmask broadcasts, the high/low half moves and the non-temporal
// stores. Every result is folded back so no instruction is dead.
#include "textflag.h"
// func avx512int(p *byte, n int) uint64
TEXT ·avx512int(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
MOVQ n+16(FP), CX
// AES rounds through the EVEX spellings, masks included.
VAESENC Z20, Z21, Z22
VAESENCLAST Z23, Z24, Z25
VAESDEC (SI), Z26, Z27
VAESDECLAST Z28, Z29, Z30
// Integer VNNI and the bit algorithm group.
VPDPBUSD Z1, Z2, K2, Z3
VPDPBUSDS Z4, Z5, K2, Z6
VPDPWSSD Z7, Z8, Z9
VPDPWSSDS Z10, Z11, K2, Z12
VPOPCNTW Z12, K3, Z13
VPOPCNTB Z14, Z15
VGF2P8MULB Z16, Z17, K4, Z18
VGF2P8AFFINEQB $7, Z18, Z19, K5, Z20
// Byte/word arithmetic with saturation and masks.
VPADDSB Z1, Z2, K1, Z3
VPADDUSW Z3, Z4, K1, Z5
VPSUBSW Z5, Z6, K1, Z7
VPSUBUSB Z7, Z8, K1, Z9
VPSADBW Z9, Z10, Z11
VPMULHRSW Z11, Z12, Z13
VPMULHW Z13, Z14, Z15
VPUNPCKLBW Z15, Z16, K2, Z17
VPUNPCKHBW Z17, Z18, K2, Z19
VPUNPCKLWD Z19, Z20, K2, Z21
VPUNPCKHWD Z21, Z22, K2, Z23
VPCMPEQB Z23, Z24, K2, K3
VPCMPGTW Z25, Z26, K2, K3
VPCMPEQQ Z27, Z28, K2
VPMULTISHIFTQB Z29, Z30, K3, Z31
VDBPSADBW $3, Z1, Z2, K3, Z3
MOVQ CX, ret+16(FP)
RET
// func avx512perm(p *byte) uint64
TEXT ·avx512perm(SB), NOSPLIT, $0-16
MOVQ p+0(FP), SI
// Permutations: immediate and register counts, ternary logic.
VALIGNQ $3, Z1, Z2, K1, Z3
VPERMT2B Z3, Z4, K1, Z5
VPERMT2W Z5, Z6, K1, Z7
VPERMT2PS Z7, Z8, K1, Z9
VPERMI2W Z9, Z10, K1, Z11
VPERMI2PS Z11, Z12, K1, Z13
VPERMI2PD Z13, Z14, K1, Z15
VPERMB Z15, Z16, K1, Z17
VPERMW Z17, Z18, K1, Z19
VPERMPS Z19, Z20, Z21
VPERMD Z20, Z21, Z22
VPERMQ $1, Z1, K2, Z2
VPERMQ Z3, Z4, K2, Z5
VPERMPD $1, Z5, K2, Z6
VPERMPD Z7, Z8, K2, Z9
VPERMILPS $5, Z9, K2, Z10
VPERMILPS Z11, Z12, K2, Z13
VPERMILPD $1, Z13, K2, Z14
VPERMILPD Z15, Z16, K2, Z17
VPTERNLOGD $6, Z17, Z18, K2, Z19
VPTERNLOGQ $9, Z19, Z20, K2, Z21
// Lane shuffle and blend families.
VSHUFPD $1, Z1, Z2, K1, Z3
VSHUFPS $2, Z4, Z5, K1, Z6
VBLENDMPD Z7, Z8, K1, Z9
VBLENDMPS Z9, Z10, K1, Z11
VPBLENDMB Z11, Z12, K1, Z13
VPBLENDMW Z13, Z14, K1, Z15
VPBLENDMD Z15, Z16, K1, Z17
VPBLENDMQ Z17, Z18, K1, Z19
// Conflicts and leading zero counts.
VPCONFLICTD Z1, K1, Z2
VPCONFLICTQ Z3, K1, Z4
VPLZCNTD Z5, K1, Z6
VPLZCNTQ Z7, K1, Z8
// Compress and expand, byte and word widths.
VPCOMPRESSB Z1, K1, (SI)
VPCOMPRESSW Z2, K1, (SI)
VPEXPANDB (SI), K1, Z3
VPEXPANDW (SI), K1, Z4
MOVQ SI, ret+8(FP)
RET
// func avx512shift(p *byte) uint64
TEXT ·avx512shift(SB), NOSPLIT, $0-16
MOVQ p+0(FP), SI
// Variable shifts and shuffles with masks.
VPSLLVW Z1, Z2, K1, Z3
VPSRLVW Z3, Z4, K1, Z5
VPSRAVW Z5, Z6, K1, Z7
VPSHLDVW Z7, Z8, K1, Z9
VPSHRDVW Z9, Z10, K1, Z11
VPSHLDVD Z11, Z12, K1, Z13
VPSHLDVQ Z13, Z14, K1, Z15
VPSHRDVD Z15, Z16, K1, Z17
VPSHRDVQ Z17, Z18, K1, Z19
// Immediate shifts, the word/byte-quad widths and masks.
VPSLLW $3, Z1, K2, Z2
VPSRLW $5, Z3, K2, Z4
VPSRAW $7, Z5, K2, Z6
VPSLLDQ $9, Z7, Z8
VPSRLDQ $11, Z9, Z10
// Register-count shifts and their memory-count forms.
VPSLLD X1, Z2, K1, Z3
VPSRLD 16(SI), Z4, K1, Z5
VPSLLQ X6, Z7, K1, Z8
VPSRLQ X9, Z10, K1, Z11
VPSLLW X12, Z13, K1, Z14
VPSRAW X15, Z16, K1, Z17
VPSRAQ $13, Z12, K1, Z13
VPSRAD X14, Z15, K1, Z16
// Lane shuffles in and out.
VPSHLDW $2, Z1, Z2, K1, Z3
VPSHLDQ $4, Z3, Z4, K1, Z5
VPSHRDW $6, Z5, Z6, K1, Z7
VPSHRDQ $8, Z7, Z8, K1, Z9
VPSHUFBITQMB Z9, Z10, K3
VPTESTMB Z11, Z12, K4
VPTESTNMQ Z13, Z14, K5
MOVQ SI, ret+8(FP)
RET
// func avx512float(x float64) float64
TEXT ·avx512float(SB), NOSPLIT, $0-16
// Square roots, compares and the EXP2/RCP28 helpers.
MOVQ x+0(FP), AX
VSQRTPD Z1, K1, Z2
VSQRTPS Z3, K1, Z4
VSQRTSD X1, X2, K1, X3
VSQRTSS X3, X4, X5
VCOMISD X5, X6
VUCOMISS X7, X8
VEXP2PD Z5, K1, Z6
VRCP28PD Z7, K1, Z8
VRCP28SD X9, X8, K1, X10
VRSQRT28PS Z11, K1, Z12
VRSQRT28SS X11, X10, K1, X12
VCVTSD2SS X1, X2, X3
VCVTSS2SD X3, X2, K1, X4
VFMADD132PD Z1, Z2, K1, Z3
VFMADD231SD X1, X2, K1, X3
VFMSUBADD213PS Z3, Z4, K1, Z5
VFNMSUB231PD Z5, Z6, K1, Z7
// Broadcasts and masked moves.
VBROADCASTF32X2 X1, K1, Z2
VBROADCASTI64X2 (SI), K1, Z3
VMOVUPS Z1, K2, Z3
VMOVSD X14, X5, K3, X22
VMOVSS X18, X3, K2, X25
VMOVHPS (SI), X18, X19
VMOVHPS X20, 8(SI)
VMOVLHPS X16, X5, X17
VMOVNTDQ Z7, (SI)
VMOVNTDQA 64(SI), Z8
VMOVNTPD Z9, (SI)
MOVQ SI, ret+8(FP)
RET
// func avx512mask(p *byte) uint64
TEXT ·avx512mask(SB), NOSPLIT, $0-16
MOVQ p+0(FP), SI
// Omask broadcasts and the K register logic.
VPBROADCASTMB2Q K1, Z2
VPBROADCASTMW2D K3, Z4
KUNPCKWD K6, K4, K1
KADDB K2, K3, K5
KORW K1, K2, K7
// Gather and scatter prefetch hints.
VGATHERPF0DPD K5, (SI)(Y29*8)
VSCATTERPF1DPS K2, (SI)(Z28*4)
// Masked gathers ride the EVEX spelling; the data length wins L'L.
VGATHERDPD (SI)(X10*4), K7, Y22
VPSCATTERDQ Y6, K2, (SI)(X4*1)
// Lane extracts to general registers.
VPEXTRB $3, X1, AX
VPEXTRD $1, X2, DI
VPINSRQ $1, SI, X3, X4
VEXTRACTI32X4 $1, Z1, X5
VINSERTI64X2 $1, X6, Z7, K2, Z8
MOVQ SI, ret+8(FP)
RET