feat(amd64): encode the AVX-512 and BMI corpus families
Assisted-by: GLM 5.3 Flash
This commit is contained in:
@@ -113,6 +113,7 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
||||||
|
isEvexPrefGather(base) ||
|
||||||
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
|
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
|
||||||
return e.encodeVec(base, ops, sfx)
|
return e.encodeVec(base, ops, sfx)
|
||||||
}
|
}
|
||||||
|
|||||||
+496
-22
@@ -5,6 +5,7 @@ package asm
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"slices"
|
||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -91,7 +92,7 @@ var evexTable = map[string]evexSpec{
|
|||||||
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
|
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
// EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
|
// EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
|
||||||
// the W bit distinguishes it from VPSRAD's E2 form).
|
// the W bit distinguishes it from VPSRAD's E2 form).
|
||||||
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPSRAQ": {1, 0x72, 1, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
// EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
|
// EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
|
||||||
// rm=src, no vvvv).
|
// rm=src, no vvvv).
|
||||||
@@ -138,7 +139,7 @@ var evexTable = map[string]evexSpec{
|
|||||||
|
|
||||||
// EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
|
// EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
|
||||||
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
"VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
"VSHUFPS": {1, 0xC6, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
// EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
|
// EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
|
||||||
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
|
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
|
||||||
@@ -202,7 +203,7 @@ var evexTable = map[string]evexSpec{
|
|||||||
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPSRLVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPSRLVW": {2, 0x10, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
@@ -324,7 +325,7 @@ var evexTable = map[string]evexSpec{
|
|||||||
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
|
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
|
"VCVTUDQ2PS": {1, 0x7A, 0, 3, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
// EVEX.66.0F38, half-precision convert (half-width source).
|
// EVEX.66.0F38, half-precision convert (half-width source).
|
||||||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
// EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
|
// EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
|
||||||
@@ -494,6 +495,246 @@ var evexTable = map[string]evexSpec{
|
|||||||
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
|
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
|
||||||
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
|
||||||
|
// --- the AVX-512 families the avx512enc corpus exercises, read off
|
||||||
|
// the toolchain opcodetables ---
|
||||||
|
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VALIGNQ": {3, 0x03, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VANDNPD": {1, 0x55, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VANDPD": {1, 0x54, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VBLENDMPD": {2, 0x65, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VBLENDMPS": {2, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VBROADCASTF32X2": {2, 0x19, 0, 1, -1, vexRM, [3]int{0, 8, 8}},
|
||||||
|
"VBROADCASTF32X4": {2, 0x1A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
|
||||||
|
"VBROADCASTF32X8": {2, 0x1B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
|
||||||
|
"VBROADCASTF64X2": {2, 0x1A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
|
||||||
|
"VBROADCASTF64X4": {2, 0x1B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
|
||||||
|
"VBROADCASTI32X2": {2, 0x59, 0, 1, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VBROADCASTI32X4": {2, 0x5A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
|
||||||
|
"VBROADCASTI32X8": {2, 0x5B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
|
||||||
|
"VBROADCASTI64X2": {2, 0x5A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
|
||||||
|
"VBROADCASTI64X4": {2, 0x5B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
|
||||||
|
"VCOMISD": {1, 0x2F, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
|
||||||
|
"VCVTSD2SS": {1, 0x5A, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VDBPSADBW": {3, 0x42, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VEXP2PD": {2, 0xC8, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||||
|
"VEXP2PS": {2, 0xC8, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||||
|
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev, [3]int{16, 32, 64}},
|
||||||
|
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VMOVNTPD": {1, 0x2B, 1, 1, -1, vexRMRev, [3]int{16, 32, 64}},
|
||||||
|
"VORPD": {1, 0x56, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPBLENDMB": {2, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPBLENDMD": {2, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPBLENDMQ": {2, 0x64, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPBLENDMW": {2, 0x66, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPBROADCASTMB2Q": {2, 0x2A, 1, 2, -1, vexRM, [3]int{0, 0, 0}},
|
||||||
|
"VPBROADCASTMW2D": {2, 0x3A, 0, 2, -1, vexRM, [3]int{0, 0, 0}},
|
||||||
|
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPEQQ": {2, 0x29, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPGTQ": {2, 0x37, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCOMPRESSB": {2, 0x63, 0, 1, -1, vexRMRev, [3]int{1, 1, 1}},
|
||||||
|
"VPCOMPRESSW": {2, 0x63, 1, 1, -1, vexRMRev, [3]int{2, 2, 2}},
|
||||||
|
"VPCONFLICTD": {2, 0xC4, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPCONFLICTQ": {2, 0xC4, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPDPBUSD": {2, 0x50, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPDPBUSDS": {2, 0x51, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPDPWSSD": {2, 0x52, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPDPWSSDS": {2, 0x53, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMI2PD": {2, 0x77, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMI2PS": {2, 0x77, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMI2W": {2, 0x75, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
|
||||||
|
"VPERMT2B": {2, 0x7D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMT2PS": {2, 0x7F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMT2W": {2, 0x7D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPEXPANDB": {2, 0x62, 0, 1, -1, vexRM, [3]int{1, 1, 1}},
|
||||||
|
"VPEXPANDW": {2, 0x62, 1, 1, -1, vexRM, [3]int{2, 2, 2}},
|
||||||
|
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm, [3]int{4, 0, 0}},
|
||||||
|
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm, [3]int{8, 0, 0}},
|
||||||
|
"VPLZCNTD": {2, 0x44, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPLZCNTQ": {2, 0x44, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMADD52HUQ": {2, 0xB5, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPMADD52LUQ": {2, 0xB4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPMULDQ": {2, 0x28, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPMULTISHIFTQB": {2, 0x83, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPMULUDQ": {1, 0xF4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPOPCNTW": {2, 0x54, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPORD": {1, 0xEB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPROLVD": {2, 0x15, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPROLVQ": {2, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPRORVD": {2, 0x14, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPRORVQ": {2, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHLDD": {3, 0x71, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPSHLDQ": {3, 0x71, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPSHLDVD": {2, 0x71, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHLDVQ": {2, 0x71, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHLDVW": {2, 0x70, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHLDW": {3, 0x70, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPSHRDD": {3, 0x73, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPSHRDQ": {3, 0x73, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPSHRDVD": {2, 0x73, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHRDVQ": {2, 0x73, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHRDVW": {2, 0x72, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHRDW": {3, 0x72, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPSHUFBITQMB": {2, 0x8F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSRAVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
|
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F73 /7, the byte-quad shift left (the count is always an
|
||||||
|
// immediate; there is no register-count twin).
|
||||||
|
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
|
// EVEX.128/256/512.0F.W0, the plain-prefix (no 66) packed spellings
|
||||||
|
// whose EVEX form drops the legacy prefix entirely.
|
||||||
|
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCOMISS": {1, 0x2F, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
|
||||||
|
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
|
||||||
|
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev, [3]int{16, 32, 64}},
|
||||||
|
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTMB": {2, 0x26, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTMD": {2, 0x27, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTMQ": {2, 0x27, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTMW": {2, 0x26, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTNMB": {2, 0x26, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTNMD": {2, 0x27, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTNMQ": {2, 0x27, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTNMW": {2, 0x26, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPUNPCKHQDQ": {1, 0x6D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPUNPCKLQDQ": {1, 0x6C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VRCP28PD": {2, 0xCA, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||||
|
"VRCP28PS": {2, 0xCA, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||||
|
"VRCP28SD": {2, 0xCB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VRCP28SS": {2, 0xCB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VRSQRT28PD": {2, 0xCC, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||||
|
"VRSQRT28PS": {2, 0xCC, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||||
|
"VRSQRT28SD": {2, 0xCD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VRSQRT28SS": {2, 0xCD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VSQRTPD": {1, 0x51, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VSQRTSD": {1, 0x51, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VUCOMISD": {1, 0x2E, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
|
||||||
|
"VXORPD": {1, 0x57, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
|
// EVEX.128/256/512.0F.F3/F2.W0, word shuffles with an immediate
|
||||||
|
// ($imm, src, dst: reg = dst, rm = src, imm8). The F3/F2 prefixes
|
||||||
|
// split the high/low lane spellings.
|
||||||
|
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
|
// EVEX.128.66.0F3A, lane extract to a general-purpose register or
|
||||||
|
// memory ($imm, xsrc, GPR/mem dst: reg = source, rm = destination).
|
||||||
|
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtractGPR, [3]int{1, 1, 1}},
|
||||||
|
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR, [3]int{2, 2, 2}},
|
||||||
|
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtractGPR, [3]int{4, 4, 4}},
|
||||||
|
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtractGPR, [3]int{8, 8, 8}},
|
||||||
|
|
||||||
|
// EVEX.66.0F3A.W1, the qword permutes with an immediate control
|
||||||
|
// ($imm, src, dst: reg = dst, rm = src, imm8); the register-count
|
||||||
|
// forms live in evexRegFormTable.
|
||||||
|
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F3A, the packed permute shuffles with an immediate control.
|
||||||
|
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VPERMILPD": {3, 0x05, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
|
// EVEX.128.0F.W0, high/low half moves. VMOVHPS carries the
|
||||||
|
// three-operand insert form (rm = m64 source, vvvv = preserved,
|
||||||
|
// reg = dst) and the two-operand store (reg = source, rm = m64);
|
||||||
|
// the encoder splits on the operand count. VMOVLHPS is the
|
||||||
|
// three-operand form alone.
|
||||||
|
"VMOVHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
}
|
}
|
||||||
|
|
||||||
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
|
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
|
||||||
@@ -529,32 +770,38 @@ type evexMoveSpec struct {
|
|||||||
n [3]int
|
n [3]int
|
||||||
vecOK bool // the non-memory operand may be a vector register
|
vecOK bool // the non-memory operand may be a vector register
|
||||||
xmmOnly bool // wider than XMM registers are rejected
|
xmmOnly bool // wider than XMM registers are rejected
|
||||||
|
nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS)
|
||||||
}
|
}
|
||||||
|
|
||||||
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
|
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
|
||||||
var evexMoveTable = map[string]evexMoveSpec{
|
var evexMoveTable = map[string]evexMoveSpec{
|
||||||
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
|
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
|
||||||
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
|
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
|
||||||
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
|
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
|
||||||
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
|
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
|
||||||
// semantics).
|
// semantics).
|
||||||
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
|
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
|
||||||
// encoding).
|
// encoding).
|
||||||
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
|
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
|
||||||
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false},
|
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128/256/512, aligned packed moves.
|
// EVEX.128/256/512, aligned packed moves.
|
||||||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false},
|
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false},
|
||||||
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false},
|
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128/256/512.66.0F, aligned integer moves.
|
// EVEX.128/256/512.66.0F, aligned integer moves.
|
||||||
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
|
||||||
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
|
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
|
||||||
// three-operand register form is not supported).
|
// three-operand register form is not supported).
|
||||||
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true},
|
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true},
|
||||||
|
// EVEX.128.F2.0F.W1, scalar double move: memory operands and the
|
||||||
|
// three-operand register form (VMOVSD dst, src1, src2).
|
||||||
|
"VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true},
|
||||||
|
// EVEX.128/256/512.0F.W0, unaligned packed single move.
|
||||||
|
"VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false},
|
||||||
}
|
}
|
||||||
|
|
||||||
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
||||||
@@ -579,6 +826,13 @@ func evexRequired(upper string, ops []Operand) bool {
|
|||||||
if !inVex && !inVexMove {
|
if !inVex && !inVexMove {
|
||||||
return true // EVEX-only mnemonic
|
return true // EVEX-only mnemonic
|
||||||
}
|
}
|
||||||
|
// The byte-quad shifts have VEX register forms but EVEX-only memory
|
||||||
|
// forms: a memory count source forces the EVEX encoding.
|
||||||
|
if upper == "VPSLLDQ" || upper == "VPSRLDQ" {
|
||||||
|
if slices.ContainsFunc(ops, memOperand) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
for _, op := range ops {
|
for _, op := range ops {
|
||||||
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
|
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
|
||||||
return true
|
return true
|
||||||
@@ -752,6 +1006,27 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
|||||||
}
|
}
|
||||||
spec.n = [3]int{n, n, n}
|
spec.n = [3]int{n, n, n}
|
||||||
}
|
}
|
||||||
|
// A mnemonic with an immediate and a register spelling (the
|
||||||
|
// variable-count shifts, the permutes) encodes the register one
|
||||||
|
// when the first operand is not an immediate.
|
||||||
|
if len(ops) > 0 {
|
||||||
|
if _, isImm := ops[0].(Imm); !isImm {
|
||||||
|
if alt, ok := evexRegFormTable[mnemUpper]; ok {
|
||||||
|
spec, inTable = alt, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The high/low half moves split by operand count: three operands
|
||||||
|
// insert, two store (VMOVHPS m64, X1).
|
||||||
|
if hs, ok := evexHptrTable[mnemUpper]; ok {
|
||||||
|
if len(ops) == 2 {
|
||||||
|
if hs.store.opcode == 0 {
|
||||||
|
return fmt.Errorf("%s has no two-operand form", mnemUpper)
|
||||||
|
}
|
||||||
|
return e.encodeEvexRMRev(hs.store, ops, 0, sfx)
|
||||||
|
}
|
||||||
|
spec = hs.insert
|
||||||
|
}
|
||||||
} else if sfx.evexOnly() {
|
} else if sfx.evexOnly() {
|
||||||
return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper)
|
return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper)
|
||||||
}
|
}
|
||||||
@@ -811,6 +1086,12 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
|||||||
}
|
}
|
||||||
return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx)
|
return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx)
|
||||||
}
|
}
|
||||||
|
if ps, ok := evexPrefGatherTable[mnemUpper]; ok {
|
||||||
|
if sfx.any() {
|
||||||
|
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
|
||||||
|
}
|
||||||
|
return e.encodeEvexPrefGather(mnemUpper, ps, ops, mask, sfx)
|
||||||
|
}
|
||||||
if !inTable {
|
if !inTable {
|
||||||
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
|
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
|
||||||
}
|
}
|
||||||
@@ -829,6 +1110,8 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
|||||||
return e.encodeEvexNDS3Imm(spec, ops, mask, sfx)
|
return e.encodeEvexNDS3Imm(spec, ops, mask, sfx)
|
||||||
case vexExtract:
|
case vexExtract:
|
||||||
return e.encodeEvexExtract(spec, ops, mask, sfx)
|
return e.encodeEvexExtract(spec, ops, mask, sfx)
|
||||||
|
case vexExtractGPR:
|
||||||
|
return e.encodeEvexExtractGPR(spec, ops, mask, sfx)
|
||||||
case vexRMSrcLen:
|
case vexRMSrcLen:
|
||||||
return e.encodeEvexRMSrcLen(spec, ops, mask, sfx)
|
return e.encodeEvexRMSrcLen(spec, ops, mask, sfx)
|
||||||
}
|
}
|
||||||
@@ -908,6 +1191,11 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSu
|
|||||||
if dstReg.mask {
|
if dstReg.mask {
|
||||||
if r, ok := src.(Reg); ok && r.isVec() {
|
if r, ok := src.(Reg); ok && r.isVec() {
|
||||||
ll = r.vecLenBit()
|
ll = r.vecLenBit()
|
||||||
|
} else if l, err := soleLen(spec.n); err == nil {
|
||||||
|
// A memory source with a length-fixed mnemonic
|
||||||
|
// (VFPCLASSPDX/Y/Z): the length comes from the table's
|
||||||
|
// single valid slot, not from the operand.
|
||||||
|
ll = l
|
||||||
}
|
}
|
||||||
} else if r, ok := src.(Reg); ok && r.isVec() {
|
} else if r, ok := src.(Reg); ok && r.isVec() {
|
||||||
ll = r.vecLenBit()
|
ll = r.vecLenBit()
|
||||||
@@ -934,9 +1222,11 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve
|
|||||||
if !ok {
|
if !ok {
|
||||||
return fmt.Errorf("shift count must be an immediate")
|
return fmt.Errorf("shift count must be an immediate")
|
||||||
}
|
}
|
||||||
srcReg, ok := src.(Reg)
|
// The count source is a vector register or memory; the length the L'L
|
||||||
if !ok || !srcReg.isVec() {
|
// field and the disp8×N multiplier follow is the destination's either
|
||||||
return fmt.Errorf("shift source must be a vector register")
|
// way.
|
||||||
|
if !vecOrMem(src) {
|
||||||
|
return fmt.Errorf("shift source must be a vector register or memory")
|
||||||
}
|
}
|
||||||
dstReg, ok := dst.(Reg)
|
dstReg, ok := dst.(Reg)
|
||||||
if !ok || !dstReg.isVec() {
|
if !ok || !dstReg.isVec() {
|
||||||
@@ -946,7 +1236,7 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, sfx); err != nil {
|
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, src, mask, sfx); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
e.out = append(e.out, immByte)
|
e.out = append(e.out, immByte)
|
||||||
@@ -1018,10 +1308,77 @@ func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, sfx evex
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// encodeEvexExtractGPR encodes the lane extract to a general-purpose
|
||||||
|
// register or memory: OP $imm, xsrc, dst (reg = the XMM source, rm = the
|
||||||
|
// destination, imm8). The encoding is 128-bit regardless of register
|
||||||
|
// numbers, so L'L is fixed at 0 and the disp8×N multiplier is the extracted
|
||||||
|
// element size the table carries.
|
||||||
|
func (e *enc) encodeEvexExtractGPR(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, src, dst := ops[0], ops[1], ops[2]
|
||||||
|
immVal, ok := imm.(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("extract lane must be an immediate")
|
||||||
|
}
|
||||||
|
srcReg, ok := src.(Reg)
|
||||||
|
if !ok || !srcReg.isVec() {
|
||||||
|
return fmt.Errorf("extract source must be a vector register")
|
||||||
|
}
|
||||||
|
switch dst.(type) {
|
||||||
|
case Reg:
|
||||||
|
if dst.(Reg).isVec() {
|
||||||
|
return fmt.Errorf("extract destination must be a general-purpose register or memory")
|
||||||
|
}
|
||||||
|
case Mem, sbMem:
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("extract destination must be a general-purpose register or memory")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(immVal))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if err := e.emitEvexFields(spec, 0, srcReg.idx, -1, dst, mask, sfx); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
e.out = append(e.out, immByte)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
|
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
|
||||||
// the store-form opcode (reg = source, rm = destination), matching the Go
|
// the store-form opcode (reg = source, rm = destination), matching the Go
|
||||||
// assembler.
|
// assembler. The scalar moves also carry a three-operand register form
|
||||||
|
// (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3
|
||||||
|
// opens.
|
||||||
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||||
|
if len(ops) == 3 {
|
||||||
|
if !ms.nds3 {
|
||||||
|
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
// The masked scalar register form keeps the Go assembler's own
|
||||||
|
// layout: the store opcode with reg = op0, vvvv = op1 and the
|
||||||
|
// destination in r/m (op2) — the bytes go tool asm emits, not
|
||||||
|
// the manual's NDS reading.
|
||||||
|
src, src1, dst := ops[0], ops[1], ops[2]
|
||||||
|
reg, ok := src.(Reg)
|
||||||
|
if !ok || !reg.isVec() {
|
||||||
|
return fmt.Errorf("%s: first operand must be a vector register", mnem)
|
||||||
|
}
|
||||||
|
vvvvReg, ok := src1.(Reg)
|
||||||
|
if !ok || !vvvvReg.isVec() {
|
||||||
|
return fmt.Errorf("%s: second operand must be a vector register", mnem)
|
||||||
|
}
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||||
|
}
|
||||||
|
if ms.xmmOnly && (reg.size != 16 || vvvvReg.size != 16 || dstReg.size != 16) {
|
||||||
|
return fmt.Errorf("%s operates on XMM registers only", mnem)
|
||||||
|
}
|
||||||
|
spec := evexSpec{mapSel: ms.mapSel, opcode: ms.store, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
|
||||||
|
return e.emitEvexFields(spec, dstReg.vecLenBit(), reg.idx, vvvvReg.idx, dst, mask, sfx)
|
||||||
|
}
|
||||||
if len(ops) != 2 {
|
if len(ops) != 2 {
|
||||||
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
|
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
|
||||||
}
|
}
|
||||||
@@ -1141,12 +1498,20 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx eve
|
|||||||
return fmt.Errorf("broadcast destination must be a vector register")
|
return fmt.Errorf("broadcast destination must be a vector register")
|
||||||
}
|
}
|
||||||
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
|
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
|
||||||
switch src.(type) {
|
switch r := src.(type) {
|
||||||
case Mem, sbMem:
|
case Mem, sbMem:
|
||||||
spec.opcode = bs.opMem
|
spec.opcode = bs.opMem
|
||||||
spec.n = [3]int{bs.n, bs.n, bs.n}
|
spec.n = [3]int{bs.n, bs.n, bs.n}
|
||||||
case Reg:
|
case Reg:
|
||||||
spec.opcode = bs.opReg
|
// A GPR source uses the register broadcast opcode; a vector
|
||||||
|
// source shares the xmm/mem one (the low byte is copied from
|
||||||
|
// the lane or from the memory operand).
|
||||||
|
if r.isVec() {
|
||||||
|
spec.opcode = bs.opMem
|
||||||
|
spec.n = [3]int{bs.n, bs.n, bs.n}
|
||||||
|
} else {
|
||||||
|
spec.opcode = bs.opReg
|
||||||
|
}
|
||||||
default:
|
default:
|
||||||
return fmt.Errorf("broadcast source must be a register or memory")
|
return fmt.Errorf("broadcast source must be a register or memory")
|
||||||
}
|
}
|
||||||
@@ -1349,6 +1714,102 @@ func isScatter(upper string) bool {
|
|||||||
return ok
|
return ok
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// isEvexPrefGather reports whether the mnemonic is a gather/scatter
|
||||||
|
// prefetch hint.
|
||||||
|
func isEvexPrefGather(upper string) bool {
|
||||||
|
_, ok := evexPrefGatherTable[upper]
|
||||||
|
return ok
|
||||||
|
}
|
||||||
|
|
||||||
|
// evexRegFormTable holds the register-count twin of the immediate-form
|
||||||
|
// entries in evexTable. Several mnemonics name two encodings: an immediate
|
||||||
|
// count or control ($imm, src, dst …) and a register-count one whose second
|
||||||
|
// operand is a vector register or memory (count, src2, src1, dst). The
|
||||||
|
// immediate spelling lives in evexTable, this table carries the register
|
||||||
|
// spelling, and encodeEvex picks by whether the first operand is an
|
||||||
|
// immediate, the way vexVarShift does on the VEX side.
|
||||||
|
var evexRegFormTable = map[string]evexSpec{
|
||||||
|
"VPSLLD": {1, 0xF2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSLLQ": {1, 0xF3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSLLW": {1, 0xF1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSRAD": {1, 0xE2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSRAW": {1, 0xE1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSRLD": {1, 0xD2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSRLQ": {1, 0xD3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSRLW": {1, 0xD1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
// EVEX.NDS.0F38.W1, the register-count permutes (the immediate
|
||||||
|
// controls live in evexTable under 0F3A).
|
||||||
|
"VPERMQ": {2, 0x36, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMPD": {2, 0x16, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.NDS.0F38, the register-count permil shuffles.
|
||||||
|
"VPERMILPS": {2, 0x0C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMILPD": {2, 0x0D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
}
|
||||||
|
|
||||||
|
// evexPrefGatherSpec describes a gather/scatter prefetch hint: one memory
|
||||||
|
// operand with a VSIB index and an opmask register, no destination. The
|
||||||
|
// ModRM.reg field carries a fixed /digit, the L'L field is fixed at 512, and
|
||||||
|
// the mask register is the instruction's only register operand.
|
||||||
|
type evexPrefGatherSpec struct {
|
||||||
|
mapSel int
|
||||||
|
opcode byte
|
||||||
|
w int
|
||||||
|
pp int
|
||||||
|
opdigit int
|
||||||
|
n int
|
||||||
|
}
|
||||||
|
|
||||||
|
var evexPrefGatherTable = map[string]evexPrefGatherSpec{
|
||||||
|
"VGATHERPF0DPD": {2, 0xC6, 1, 1, 1, 8},
|
||||||
|
"VGATHERPF0DPS": {2, 0xC6, 0, 1, 1, 4},
|
||||||
|
"VGATHERPF0QPD": {2, 0xC7, 1, 1, 1, 8},
|
||||||
|
"VGATHERPF0QPS": {2, 0xC7, 0, 1, 1, 4},
|
||||||
|
"VGATHERPF1DPD": {2, 0xC6, 1, 1, 2, 8},
|
||||||
|
"VGATHERPF1DPS": {2, 0xC6, 0, 1, 2, 4},
|
||||||
|
"VGATHERPF1QPD": {2, 0xC7, 1, 1, 2, 8},
|
||||||
|
"VGATHERPF1QPS": {2, 0xC7, 0, 1, 2, 4},
|
||||||
|
"VSCATTERPF0DPD": {2, 0xC6, 1, 1, 5, 8},
|
||||||
|
"VSCATTERPF0DPS": {2, 0xC6, 0, 1, 5, 4},
|
||||||
|
"VSCATTERPF0QPD": {2, 0xC7, 1, 1, 5, 8},
|
||||||
|
"VSCATTERPF0QPS": {2, 0xC7, 0, 1, 5, 4},
|
||||||
|
"VSCATTERPF1DPD": {2, 0xC6, 1, 1, 6, 8},
|
||||||
|
"VSCATTERPF1DPS": {2, 0xC6, 0, 1, 6, 4},
|
||||||
|
"VSCATTERPF1QPD": {2, 0xC7, 1, 1, 6, 8},
|
||||||
|
"VSCATTERPF1QPS": {2, 0xC7, 0, 1, 6, 4},
|
||||||
|
}
|
||||||
|
|
||||||
|
// evexHptrSpec describes the high/low half moves (VMOVHPS family): the
|
||||||
|
// three-operand insert shares an opcode with a two-operand store whose
|
||||||
|
// source is the vector register and whose destination is m64.
|
||||||
|
type evexHptrSpec struct {
|
||||||
|
insert evexSpec
|
||||||
|
store evexSpec // store.opcode == 0 when the mnemonic has no store form
|
||||||
|
}
|
||||||
|
|
||||||
|
var evexHptrTable = map[string]evexHptrSpec{
|
||||||
|
"VMOVHPS": {
|
||||||
|
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
|
||||||
|
store: evexSpec{mapSel: 1, opcode: 0x17, w: 0, pp: 0, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
|
||||||
|
},
|
||||||
|
"VMOVLHPS": {
|
||||||
|
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib.
|
||||||
|
func (e *enc) encodeEvexPrefGather(upper string, ps evexPrefGatherSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands (K, vsib memory), got %d", upper, len(ops)+1)
|
||||||
|
}
|
||||||
|
m, ok := ops[0].(Mem)
|
||||||
|
if !ok || !m.HasIndex || !m.Index.isVec() {
|
||||||
|
return fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", upper)
|
||||||
|
}
|
||||||
|
spec := evexSpec{mapSel: ps.mapSel, opcode: ps.opcode, w: ps.w, pp: ps.pp, opdigit: ps.opdigit, n: [3]int{ps.n, ps.n, ps.n}}
|
||||||
|
return e.emitEvexFields(spec, 2, ps.opdigit, -1, m, mask, sfx)
|
||||||
|
}
|
||||||
|
|
||||||
// vsibLen validates a VSIB memory operand (the index must be a vector
|
// vsibLen validates a VSIB memory operand (the index must be a vector
|
||||||
// register) and returns it with the vector length the index selects, the
|
// register) and returns it with the vector length the index selects, the
|
||||||
// EVEX L'L field follows the index register, not the data register.
|
// EVEX L'L field follows the index register, not the data register.
|
||||||
@@ -1370,7 +1831,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
|
|||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if mask != 0 || sfx.any() {
|
if mask != 0 || sfx.any() {
|
||||||
// EVEX form: OP vsib, K, dst.
|
// EVEX form: OP vsib, K, dst. The L'L field is the wider of the
|
||||||
|
// index and the data register lengths (the Go assembler's
|
||||||
|
// layout); the disp8×N multiplier stays the index element size.
|
||||||
if len(rest) != 2 {
|
if len(rest) != 2 {
|
||||||
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
|
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
|
||||||
}
|
}
|
||||||
@@ -1382,6 +1845,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
|
|||||||
if !ok || !dst.isVec() {
|
if !ok || !dst.isVec() {
|
||||||
return fmt.Errorf("%s: destination must be a vector register", upper)
|
return fmt.Errorf("%s: destination must be a vector register", upper)
|
||||||
}
|
}
|
||||||
|
if d := dst.vecLenBit(); d > ll {
|
||||||
|
ll = d
|
||||||
|
}
|
||||||
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
|
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
|
||||||
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
|
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
|
||||||
}
|
}
|
||||||
@@ -1431,6 +1897,11 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
// The L'L field is the wider of the data register and the VSIB index
|
||||||
|
// lengths, the bytes go tool asm emits.
|
||||||
|
if d := src.vecLenBit(); d > ll {
|
||||||
|
ll = d
|
||||||
|
}
|
||||||
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
|
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
|
||||||
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
|
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
|
||||||
}
|
}
|
||||||
@@ -1441,6 +1912,8 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
|
|||||||
var evexKOperand = map[string]bool{
|
var evexKOperand = map[string]bool{
|
||||||
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
|
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
|
||||||
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
|
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
|
||||||
|
// The K-to-vector broadcast reads its opmask source from r/m.
|
||||||
|
"VPBROADCASTMB2Q": true, "VPBROADCASTMW2D": true,
|
||||||
}
|
}
|
||||||
|
|
||||||
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
||||||
@@ -1541,6 +2014,7 @@ var kOpsTable = map[string]kOpSpec{
|
|||||||
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
|
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
|
||||||
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
|
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
|
||||||
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
|
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
|
||||||
|
"KUNPCKWD": {1, 0x4B, 0, 0, 1, vexNDS3},
|
||||||
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
|
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
|
||||||
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
|
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
|
||||||
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
|
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
|
||||||
|
|||||||
@@ -721,3 +721,92 @@ func hexCompact(b []byte) string {
|
|||||||
}
|
}
|
||||||
return string(out)
|
return string(out)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAvx512CorpusFamilies pins representative encodings of the AVX-512
|
||||||
|
// families the toolchain's avx512enc corpus exercises: the bytes are the
|
||||||
|
// go tool asm output for exactly these operands, and the same families are
|
||||||
|
// covered end to end by the avx512_amd64.s differential kernel.
|
||||||
|
func TestAvx512CorpusFamilies(t *testing.T) {
|
||||||
|
vsib := func(base, idx string, scale int) Operand {
|
||||||
|
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
|
||||||
|
}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
// AES rounds (EVEX NDS, VEX twin routed by operand width).
|
||||||
|
{"VAESDEC Z", "VAESDEC", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d48ded9"},
|
||||||
|
// Integer VNNI and the bit algorithm group.
|
||||||
|
{"VPDPBUSD", "VPDPBUSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f26d4a50d9"},
|
||||||
|
{"VPOPCNTW", "VPOPCNTW", []Operand{vreg(t, "Z1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f2fd4b54d1"},
|
||||||
|
{"VPCONFLICTD", "VPCONFLICTD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d49c4d1"},
|
||||||
|
{"VPLZCNTQ masked", "VPLZCNTQ", []Operand{vreg(t, "Z7"), vreg(t, "K1"), vreg(t, "Z8")}, "6272fd4944c7"},
|
||||||
|
{"VPERMT2B", "VPERMT2B", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f26d497dd9"},
|
||||||
|
{"VPMULTISHIFTQB", "VPMULTISHIFTQB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f2ed4b83e1"},
|
||||||
|
{"VDBPSADBW", "VDBPSADBW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z3")}, "62f36d4b42d903"},
|
||||||
|
{"VPSHUFBITQMB", "VPSHUFBITQMB", []Operand{vreg(t, "Z9"), vreg(t, "Z10"), vreg(t, "K3")}, "62d22d488fd9"},
|
||||||
|
{"VPTESTNMQ", "VPTESTNMQ", []Operand{vreg(t, "Z13"), vreg(t, "Z14"), vreg(t, "K5")}, "62d28e4827ed"},
|
||||||
|
// Permutations: immediate and register counts.
|
||||||
|
{"VALIGNQ", "VALIGNQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f3ed4903d903"},
|
||||||
|
{"VPERMQ imm", "VPERMQ", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f3fd4a00d101"},
|
||||||
|
{"VPERMQ reg", "VPERMQ", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K2"), vreg(t, "Z5")}, "62f2dd4a36eb"},
|
||||||
|
{"VPERMPD reg", "VPERMPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4816d9"},
|
||||||
|
{"VPERMILPS imm", "VPERMILPS", []Operand{Imm(5), vreg(t, "Z9"), vreg(t, "K2"), vreg(t, "Z10")}, "62537d4a04d105"},
|
||||||
|
{"VPERMILPS reg", "VPERMILPS", []Operand{vreg(t, "Z11"), vreg(t, "Z12"), vreg(t, "K2"), vreg(t, "Z13")}, "62521d4a0ceb"},
|
||||||
|
// Shifts: immediate, register-count and memory-count forms; the
|
||||||
|
// count source carries its own XMM tuple width.
|
||||||
|
{"VPSLLW imm mask", "VPSLLW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f16d4a71f103"},
|
||||||
|
{"VPSLLD reg count", "VPSLLD", []Operand{vreg(t, "X1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49f2d9"},
|
||||||
|
{"VPSLLDQ", "VPSLLDQ", []Operand{Imm(9), vreg(t, "Z7"), vreg(t, "Z8")}, "62f13d4873ff09"},
|
||||||
|
{"VPSRLDQ mem", "VPSRLDQ", []Operand{Imm(11), Ptr(SI, 16, 16), vreg(t, "Z4")}, "62f15d48739e100000000b"},
|
||||||
|
{"VPSRLVW", "VPSRLVW", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K1"), vreg(t, "Z5")}, "62f2dd4910eb"},
|
||||||
|
// Conversions and shuffles with the F2 prefix and no prefix.
|
||||||
|
{"VCVTUDQ2PS", "VCVTUDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f17f497ad1"},
|
||||||
|
{"VSHUFPS", "VSHUFPS", []Operand{Imm(2), vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f15449c6f402"},
|
||||||
|
// Gather and scatter prefetch hints (memory-only, /digit in reg).
|
||||||
|
{"VGATHERPF0DPD", "VGATHERPF0DPD", []Operand{vreg(t, "K5"), vsib("R10", "Y29", 8)}, "6292fd45c60cea"},
|
||||||
|
{"VSCATTERPF1DPS", "VSCATTERPF1DPS", []Operand{vreg(t, "K2"), vsib("R10", "Z28", 4)}, "62927d42c634a2"},
|
||||||
|
// Opmask broadcasts and the K logic.
|
||||||
|
{"VPBROADCASTMB2Q", "VPBROADCASTMB2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe482ad1"},
|
||||||
|
{"VPBROADCASTMW2D", "VPBROADCASTMW2D", []Operand{vreg(t, "K3"), vreg(t, "Z4")}, "62f27e483ae3"},
|
||||||
|
{"KUNPCKWD", "KUNPCKWD", []Operand{vreg(t, "K6"), vreg(t, "K4"), vreg(t, "K1")}, "c5dc4bce"},
|
||||||
|
{"KADDB", "KADDB", []Operand{vreg(t, "K2"), vreg(t, "K3"), vreg(t, "K5")}, "c5e54aea"},
|
||||||
|
// Lane extracts to general registers (EVEX and VEX routes).
|
||||||
|
{"VPEXTRB", "VPEXTRB", []Operand{Imm(3), vreg(t, "X26"), AX}, "62637d0814d003"},
|
||||||
|
{"VPEXTRD", "VPEXTRD", []Operand{Imm(1), vreg(t, "X26"), vreg(t, "R9")}, "62437d0816d101"},
|
||||||
|
{"VPEXTRD vex", "VPEXTRD", []Operand{Imm(1), vreg(t, "X2"), DI}, "c4e37916d701"},
|
||||||
|
{"VPINSRQ", "VPINSRQ", []Operand{Imm(1), DI, vreg(t, "X3"), vreg(t, "X4")}, "c4e3e122e701"},
|
||||||
|
// Moves: masked unaligned, masked scalar register form, half moves
|
||||||
|
// and non-temporal stores.
|
||||||
|
{"VMOVUPS mask", "VMOVUPS", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f17c4a11cb"},
|
||||||
|
{"VMOVSD 3op", "VMOVSD", []Operand{vreg(t, "X14"), vreg(t, "X5"), vreg(t, "K3"), vreg(t, "X22")}, "6231d70b11f6"},
|
||||||
|
{"VMOVSS 3op", "VMOVSS", []Operand{vreg(t, "X18"), vreg(t, "X3"), vreg(t, "K2"), vreg(t, "X25")}, "6281660a11d1"},
|
||||||
|
{"VMOVHPS insert", "VMOVHPS", []Operand{Ptr(SI, 0, 8), vreg(t, "X18"), vreg(t, "X19")}, "62e16c00161e"},
|
||||||
|
{"VMOVHPS store", "VMOVHPS", []Operand{vreg(t, "X20"), Ptr(SI, 8, 8)}, "62e17c08176601"},
|
||||||
|
{"VMOVLHPS", "VMOVLHPS", []Operand{vreg(t, "X16"), vreg(t, "X5"), vreg(t, "X17")}, "62a1540816c8"},
|
||||||
|
{"VMOVNTDQ", "VMOVNTDQ", []Operand{vreg(t, "Z7"), Ptr(SI, 0, 64)}, "62f17d48e73e"},
|
||||||
|
{"VMOVNTDQA", "VMOVNTDQA", []Operand{Ptr(SI, 64, 64), vreg(t, "Z8")}, "62727d482a4601"},
|
||||||
|
{"VMOVNTPS", "VMOVNTPS", []Operand{vreg(t, "Z9"), Ptr(SI, 0, 64)}, "62717c482b0e"},
|
||||||
|
// Scalar compares with and without the 66 prefix.
|
||||||
|
{"VCOMISD", "VCOMISD", []Operand{vreg(t, "X5"), vreg(t, "X6")}, "c5f92ff5"},
|
||||||
|
{"VUCOMISS", "VUCOMISS", []Operand{vreg(t, "X7"), vreg(t, "X8")}, "c5782ec7"},
|
||||||
|
// Floating point helpers.
|
||||||
|
{"VSQRTSD", "VSQRTSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f1ef0951d9"},
|
||||||
|
{"VEXP2PD", "VEXP2PD", []Operand{vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f2fd49c8f5"},
|
||||||
|
{"VRCP28SD", "VRCP28SD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "K1"), vreg(t, "X10")}, "6252bd09cbd1"},
|
||||||
|
{"VBROADCASTF32X2", "VBROADCASTF32X2", []Operand{vreg(t, "X1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d4919d1"},
|
||||||
|
{"VPCOMPRESSB", "VPCOMPRESSB", []Operand{vreg(t, "Z1"), vreg(t, "K1"), Ptr(SI, 0, 64)}, "62f27d49630e"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
Vendored
+191
@@ -0,0 +1,191 @@
|
|||||||
|
// The AVX-512 families behind the avx512enc gap: AES round ops, integer
|
||||||
|
// VNNI and bit algorithms, word shifts and permutes with an immediate or a
|
||||||
|
// register count, lane broadcasts and extracts, gather and scatter prefetch
|
||||||
|
// hints, opmask broadcasts, the high/low half moves and the non-temporal
|
||||||
|
// stores. Every result is folded back so no instruction is dead.
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func avx512int(p *byte, n int) uint64
|
||||||
|
TEXT ·avx512int(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ p+0(FP), SI
|
||||||
|
MOVQ n+16(FP), CX
|
||||||
|
// AES rounds through the EVEX spellings, masks included.
|
||||||
|
VAESENC Z20, Z21, Z22
|
||||||
|
VAESENCLAST Z23, Z24, Z25
|
||||||
|
VAESDEC (SI), Z26, Z27
|
||||||
|
VAESDECLAST Z28, Z29, Z30
|
||||||
|
// Integer VNNI and the bit algorithm group.
|
||||||
|
VPDPBUSD Z1, Z2, K2, Z3
|
||||||
|
VPDPBUSDS Z4, Z5, K2, Z6
|
||||||
|
VPDPWSSD Z7, Z8, Z9
|
||||||
|
VPDPWSSDS Z10, Z11, K2, Z12
|
||||||
|
VPOPCNTW Z12, K3, Z13
|
||||||
|
VPOPCNTB Z14, Z15
|
||||||
|
VGF2P8MULB Z16, Z17, K4, Z18
|
||||||
|
VGF2P8AFFINEQB $7, Z18, Z19, K5, Z20
|
||||||
|
// Byte/word arithmetic with saturation and masks.
|
||||||
|
VPADDSB Z1, Z2, K1, Z3
|
||||||
|
VPADDUSW Z3, Z4, K1, Z5
|
||||||
|
VPSUBSW Z5, Z6, K1, Z7
|
||||||
|
VPSUBUSB Z7, Z8, K1, Z9
|
||||||
|
VPSADBW Z9, Z10, Z11
|
||||||
|
VPMULHRSW Z11, Z12, Z13
|
||||||
|
VPMULHW Z13, Z14, Z15
|
||||||
|
VPUNPCKLBW Z15, Z16, K2, Z17
|
||||||
|
VPUNPCKHBW Z17, Z18, K2, Z19
|
||||||
|
VPUNPCKLWD Z19, Z20, K2, Z21
|
||||||
|
VPUNPCKHWD Z21, Z22, K2, Z23
|
||||||
|
VPCMPEQB Z23, Z24, K2, K3
|
||||||
|
VPCMPGTW Z25, Z26, K2, K3
|
||||||
|
VPCMPEQQ Z27, Z28, K2
|
||||||
|
VPMULTISHIFTQB Z29, Z30, K3, Z31
|
||||||
|
VDBPSADBW $3, Z1, Z2, K3, Z3
|
||||||
|
MOVQ CX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func avx512perm(p *byte) uint64
|
||||||
|
TEXT ·avx512perm(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ p+0(FP), SI
|
||||||
|
// Permutations: immediate and register counts, ternary logic.
|
||||||
|
VALIGNQ $3, Z1, Z2, K1, Z3
|
||||||
|
VPERMT2B Z3, Z4, K1, Z5
|
||||||
|
VPERMT2W Z5, Z6, K1, Z7
|
||||||
|
VPERMT2PS Z7, Z8, K1, Z9
|
||||||
|
VPERMI2W Z9, Z10, K1, Z11
|
||||||
|
VPERMI2PS Z11, Z12, K1, Z13
|
||||||
|
VPERMI2PD Z13, Z14, K1, Z15
|
||||||
|
VPERMB Z15, Z16, K1, Z17
|
||||||
|
VPERMW Z17, Z18, K1, Z19
|
||||||
|
VPERMPS Z19, Z20, Z21
|
||||||
|
VPERMD Z20, Z21, Z22
|
||||||
|
VPERMQ $1, Z1, K2, Z2
|
||||||
|
VPERMQ Z3, Z4, K2, Z5
|
||||||
|
VPERMPD $1, Z5, K2, Z6
|
||||||
|
VPERMPD Z7, Z8, K2, Z9
|
||||||
|
VPERMILPS $5, Z9, K2, Z10
|
||||||
|
VPERMILPS Z11, Z12, K2, Z13
|
||||||
|
VPERMILPD $1, Z13, K2, Z14
|
||||||
|
VPERMILPD Z15, Z16, K2, Z17
|
||||||
|
VPTERNLOGD $6, Z17, Z18, K2, Z19
|
||||||
|
VPTERNLOGQ $9, Z19, Z20, K2, Z21
|
||||||
|
// Lane shuffle and blend families.
|
||||||
|
VSHUFPD $1, Z1, Z2, K1, Z3
|
||||||
|
VSHUFPS $2, Z4, Z5, K1, Z6
|
||||||
|
VBLENDMPD Z7, Z8, K1, Z9
|
||||||
|
VBLENDMPS Z9, Z10, K1, Z11
|
||||||
|
VPBLENDMB Z11, Z12, K1, Z13
|
||||||
|
VPBLENDMW Z13, Z14, K1, Z15
|
||||||
|
VPBLENDMD Z15, Z16, K1, Z17
|
||||||
|
VPBLENDMQ Z17, Z18, K1, Z19
|
||||||
|
// Conflicts and leading zero counts.
|
||||||
|
VPCONFLICTD Z1, K1, Z2
|
||||||
|
VPCONFLICTQ Z3, K1, Z4
|
||||||
|
VPLZCNTD Z5, K1, Z6
|
||||||
|
VPLZCNTQ Z7, K1, Z8
|
||||||
|
// Compress and expand, byte and word widths.
|
||||||
|
VPCOMPRESSB Z1, K1, (SI)
|
||||||
|
VPCOMPRESSW Z2, K1, (SI)
|
||||||
|
VPEXPANDB (SI), K1, Z3
|
||||||
|
VPEXPANDW (SI), K1, Z4
|
||||||
|
MOVQ SI, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func avx512shift(p *byte) uint64
|
||||||
|
TEXT ·avx512shift(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ p+0(FP), SI
|
||||||
|
// Variable shifts and shuffles with masks.
|
||||||
|
VPSLLVW Z1, Z2, K1, Z3
|
||||||
|
VPSRLVW Z3, Z4, K1, Z5
|
||||||
|
VPSRAVW Z5, Z6, K1, Z7
|
||||||
|
VPSHLDVW Z7, Z8, K1, Z9
|
||||||
|
VPSHRDVW Z9, Z10, K1, Z11
|
||||||
|
VPSHLDVD Z11, Z12, K1, Z13
|
||||||
|
VPSHLDVQ Z13, Z14, K1, Z15
|
||||||
|
VPSHRDVD Z15, Z16, K1, Z17
|
||||||
|
VPSHRDVQ Z17, Z18, K1, Z19
|
||||||
|
// Immediate shifts, the word/byte-quad widths and masks.
|
||||||
|
VPSLLW $3, Z1, K2, Z2
|
||||||
|
VPSRLW $5, Z3, K2, Z4
|
||||||
|
VPSRAW $7, Z5, K2, Z6
|
||||||
|
VPSLLDQ $9, Z7, Z8
|
||||||
|
VPSRLDQ $11, Z9, Z10
|
||||||
|
// Register-count shifts and their memory-count forms.
|
||||||
|
VPSLLD X1, Z2, K1, Z3
|
||||||
|
VPSRLD 16(SI), Z4, K1, Z5
|
||||||
|
VPSLLQ X6, Z7, K1, Z8
|
||||||
|
VPSRLQ X9, Z10, K1, Z11
|
||||||
|
VPSLLW X12, Z13, K1, Z14
|
||||||
|
VPSRAW X15, Z16, K1, Z17
|
||||||
|
VPSRAQ $13, Z12, K1, Z13
|
||||||
|
VPSRAD X14, Z15, K1, Z16
|
||||||
|
// Lane shuffles in and out.
|
||||||
|
VPSHLDW $2, Z1, Z2, K1, Z3
|
||||||
|
VPSHLDQ $4, Z3, Z4, K1, Z5
|
||||||
|
VPSHRDW $6, Z5, Z6, K1, Z7
|
||||||
|
VPSHRDQ $8, Z7, Z8, K1, Z9
|
||||||
|
VPSHUFBITQMB Z9, Z10, K3
|
||||||
|
VPTESTMB Z11, Z12, K4
|
||||||
|
VPTESTNMQ Z13, Z14, K5
|
||||||
|
MOVQ SI, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func avx512float(x float64) float64
|
||||||
|
TEXT ·avx512float(SB), NOSPLIT, $0-16
|
||||||
|
// Square roots, compares and the EXP2/RCP28 helpers.
|
||||||
|
MOVQ x+0(FP), AX
|
||||||
|
VSQRTPD Z1, K1, Z2
|
||||||
|
VSQRTPS Z3, K1, Z4
|
||||||
|
VSQRTSD X1, X2, K1, X3
|
||||||
|
VSQRTSS X3, X4, X5
|
||||||
|
VCOMISD X5, X6
|
||||||
|
VUCOMISS X7, X8
|
||||||
|
VEXP2PD Z5, K1, Z6
|
||||||
|
VRCP28PD Z7, K1, Z8
|
||||||
|
VRCP28SD X9, X8, K1, X10
|
||||||
|
VRSQRT28PS Z11, K1, Z12
|
||||||
|
VRSQRT28SS X11, X10, K1, X12
|
||||||
|
VCVTSD2SS X1, X2, X3
|
||||||
|
VCVTSS2SD X3, X2, K1, X4
|
||||||
|
VFMADD132PD Z1, Z2, K1, Z3
|
||||||
|
VFMADD231SD X1, X2, K1, X3
|
||||||
|
VFMSUBADD213PS Z3, Z4, K1, Z5
|
||||||
|
VFNMSUB231PD Z5, Z6, K1, Z7
|
||||||
|
// Broadcasts and masked moves.
|
||||||
|
VBROADCASTF32X2 X1, K1, Z2
|
||||||
|
VBROADCASTI64X2 (SI), K1, Z3
|
||||||
|
VMOVUPS Z1, K2, Z3
|
||||||
|
VMOVSD X14, X5, K3, X22
|
||||||
|
VMOVSS X18, X3, K2, X25
|
||||||
|
VMOVHPS (SI), X18, X19
|
||||||
|
VMOVHPS X20, 8(SI)
|
||||||
|
VMOVLHPS X16, X5, X17
|
||||||
|
VMOVNTDQ Z7, (SI)
|
||||||
|
VMOVNTDQA 64(SI), Z8
|
||||||
|
VMOVNTPD Z9, (SI)
|
||||||
|
MOVQ SI, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func avx512mask(p *byte) uint64
|
||||||
|
TEXT ·avx512mask(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ p+0(FP), SI
|
||||||
|
// Omask broadcasts and the K register logic.
|
||||||
|
VPBROADCASTMB2Q K1, Z2
|
||||||
|
VPBROADCASTMW2D K3, Z4
|
||||||
|
KUNPCKWD K6, K4, K1
|
||||||
|
KADDB K2, K3, K5
|
||||||
|
KORW K1, K2, K7
|
||||||
|
// Gather and scatter prefetch hints.
|
||||||
|
VGATHERPF0DPD K5, (SI)(Y29*8)
|
||||||
|
VSCATTERPF1DPS K2, (SI)(Z28*4)
|
||||||
|
// Masked gathers ride the EVEX spelling; the data length wins L'L.
|
||||||
|
VGATHERDPD (SI)(X10*4), K7, Y22
|
||||||
|
VPSCATTERDQ Y6, K2, (SI)(X4*1)
|
||||||
|
// Lane extracts to general registers.
|
||||||
|
VPEXTRB $3, X1, AX
|
||||||
|
VPEXTRD $1, X2, DI
|
||||||
|
VPINSRQ $1, SI, X3, X4
|
||||||
|
VEXTRACTI32X4 $1, Z1, X5
|
||||||
|
VINSERTI64X2 $1, X6, Z7, K2, Z8
|
||||||
|
MOVQ SI, ret+8(FP)
|
||||||
|
RET
|
||||||
Reference in New Issue
Block a user