Files
gasm-sdk/asm/evex.go
T
2026-09-21 02:02:19 +02:00

2223 lines
94 KiB
Go
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"fmt"
"slices"
"strings"
)
// This file implements EVEX (AVX-512) instruction encoding: the four-byte
// EVEX prefix with 5-bit vector register fields (Z0-Z31, X/Y 16-31), the
// compressed disp8×N displacement, and the operand shapes the go-flac
// AVX-512 kernels use plus the common floating-point and conversion set.
// Masking follows the Go assembler's spelling: an explicit K1-K7 operand
// anywhere among the operands (merging) plus a ".Z" mnemonic suffix for
// zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are
// supported too.
// evexSpec describes one EVEX instruction's encoding parameters. The form
// field reuses the vexForm shapes, which carry over unchanged.
type evexSpec struct {
mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A
opcode byte
w int
pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2
opdigit int // ModRM.reg /digit, or -1 when reg is a register
form vexForm // vexNDS3, vexRM, vexShiftImm, vexNDS3Imm, vexExtract
n [3]int // disp8×N multiplier per vector length (128/256/512)
}
// evexTable maps an upper-case mnemonic to its EVEX encoding. Mnemonics
// that also have a VEX form (VPADDD, VMOVUPD, …) are dispatched here only
// when an operand demands EVEX (a ZMM or K register); EVEX-only mnemonics
// (VPXORD, VALIGND, …) always encode through this table. The N multipliers
// are taken from the Go assembler's opcode tables, which are authoritative
// for byte-for-byte agreement.
var evexTable = map[string]evexSpec{
// EVEX.128/256/512.66.0F, integer arithmetic / logic, NDS form.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBQ": {1, 0xFB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPXORD": {1, 0xEF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPXORQ": {1, 0xEF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1, packed double arithmetic.
"VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0, packed single arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1, packed double unpack.
"VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128.F2.0F.W1, scalar double arithmetic (the packed opcodes with
// an F2 pp; the EVEX forms exist for masked and zeroing use). The
// memory operand is a single double, so disp8×N = 8.
"VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VSUBSD": {1, 0x5C, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMULSD": {1, 0x59, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VDIVSD": {1, 0x5E, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128.F3.0F.W0, scalar single arithmetic (disp8×N = 4).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.512.66.0F3A, align (NDS + imm8).
"VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F, immediate shift (VPSRAD /4).
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
// the W bit distinguishes it from VPSRAD's E2 form).
"VPSRAQ": {1, 0x72, 1, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
// rm=src, no vvvv).
"VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1, duplicate the low double (reg=dst,
// rm=src, no vvvv): a 128-bit destination reads a single double from
// memory (disp8×8), the wider ones read the full operand.
"VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}},
// EVEX.128/256/512.0F.W0, signed dword to packed single (reg=dst,
// rm=src, no vvvv, no mandatory prefix, as in the VEX form).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0, packed single to packed double: the
// destination is twice the source width and sets the length; disp8×N
// follows the narrow memory source. No F3 prefix: the Go assembler
// emits this instruction with pp = 00 (Intel's maps would call that
// undefined) and gasm reproduces the Go assembler's bytes, its machine
// code is the oracle, not the manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.128/256/512.F3.0F.W0, signed dword to packed double (the EVEX
// form of the VEX instruction; the destination sets the length, disp8×N
// follows the narrow memory source).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
// EVEX packed double → dword conversions: the source is the wide
// operand and the mnemonic fixes the length, the bare names are
// 512-bit only (ZMM source, XMM destination), the X/Y spellings are
// EVEX-128/256. Exactly one slot of n is valid; it names the vector
// length (and the disp8×N multiplier) a register or memory source
// encodes.
"VCVTPD2DQ": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTTPD2DQ": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2DQX": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTPD2DQY": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
"VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.66.0F3A, ternary logic and lane shuffles (NDS + imm8).
"VPTERNLOGD": {3, 0x25, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPTERNLOGQ": {3, 0x25, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFI32X4": {3, 0x43, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFI64X2": {3, 0x43, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFF32X4": {3, 0x23, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFF64X2": {3, 0x23, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFPS": {1, 0xC6, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTF32X8": {3, 0x1A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
"VINSERTF64X2": {3, 0x18, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTF64X4": {3, 0x1A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
"VINSERTI32X4": {3, 0x38, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTI32X8": {3, 0x3A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
"VINSERTI64X2": {3, 0x38, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTI64X4": {3, 0x3A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
// EVEX.66.0F3A, lane extract (reg=source, rm=XMM/YMM destination,
// imm8).
"VEXTRACTF32X4": {3, 0x19, 0, 1, -1, vexExtract, [3]int{0, 16, 16}},
"VEXTRACTF32X8": {3, 0x1B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTF64X2": {3, 0x19, 1, 1, -1, vexExtract, [3]int{0, 16, 16}},
"VEXTRACTI32X4": {3, 0x39, 0, 1, -1, vexExtract, [3]int{0, 16, 16}},
"VEXTRACTI32X8": {3, 0x3B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTI64X2": {3, 0x39, 1, 1, -1, vexExtract, [3]int{0, 16, 16}},
// EVEX.66.0F, compare with an opmask destination ($imm, src2, src1,
// kdst): NDS3Imm with the K register in the reg field.
"VCMPPD": {1, 0xC2, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VCMPPS": {1, 0xC2, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}},
// EVEX.66.0F3A, integer compares with an opmask destination, the same
// NDS3Imm-with-k-reg shape as the floating-point compares; W selects the
// operand width (byte/word vs dword/qword), the opcode the signedness.
// The memory form takes a full vector, so disp8×N is 16/32/64.
"VPCMPB": {3, 0x3F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUB": {3, 0x3E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPW": {3, 0x3F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUW": {3, 0x3E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPD": {3, 0x1F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUD": {3, 0x1E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPQ": {3, 0x1F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F38, permutes (NDS form).
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2B": {2, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F38, population count (reg=dst, rm=src; W selects byte/word
// against dword/qword).
"VPOPCNTB": {2, 0x54, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPOPCNTD": {2, 0x55, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPOPCNTQ": {2, 0x55, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F.W1, the qword spelling of the packed OR (VPORQ has no VEX
// form in the Go assembler: it always encodes through EVEX).
"VPORQ": {1, 0xEB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F, the wider integer set (NDS form).
"VPMADDWD": {1, 0xF5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLVW": {2, 0x10, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKUSDW": {2, 0x2B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F38, absolute values and replicating moves (reg=dst,
// rm=src).
"VPABSB": {2, 0x1C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSW": {2, 0x1D, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSD": {2, 0x1E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSQ": {2, 0x1F, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.F3.0F, replicate even/odd singles.
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38, sign/zero-extending moves; the memory source is the
// narrow half (here byte to word).
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F, packed single conversions (reg=dst, rm=src).
"VCVTPS2DQ": {1, 0x5B, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTTPS2DQ": {1, 0x5B, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38, broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; disp8×N is the element size).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VBROADCASTSD": {2, 0x19, 1, 1, -1, vexRM, [3]int{0, 8, 8}},
// EVEX.66.0F38, expand loads (rm → vector register destination).
"VEXPANDPD": {2, 0x88, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
"VEXPANDPS": {2, 0x88, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VPEXPANDD": {2, 0x89, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VPEXPANDQ": {2, 0x89, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
// EVEX.66.0F38, compress stores (vector register source → rm), and the
// remaining narrowing stores.
"VCOMPRESSPD": {2, 0x8A, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}},
"VCOMPRESSPS": {2, 0x8A, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}},
"VPCOMPRESSD": {2, 0x8B, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}},
"VPCOMPRESSQ": {2, 0x8B, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}},
"VPMOVWB": {2, 0x30, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQB": {2, 0x32, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
// EVEX.66.0F, rotates (immediate form: /0 right, /1 left).
"VPRORD": {1, 0x72, 0, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
"VPRORQ": {1, 0x72, 1, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
"VPROLD": {1, 0x72, 0, 1, 1, vexShiftImm, [3]int{16, 32, 64}},
"VPROLQ": {1, 0x72, 1, 1, 1, vexShiftImm, [3]int{16, 32, 64}},
// EVEX word shifts.
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX W1 qword shifts.
"VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
"VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.66.0F38, floating-point helpers, packed (reg=dst, rm=src).
"VRCP14PD": {2, 0x4C, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRCP14PS": {2, 0x4C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRSQRT14PD": {2, 0x4E, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRSQRT14PS": {2, 0x4E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VGETEXPPD": {2, 0x42, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VGETEXPPS": {2, 0x42, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38, floating-point helpers, scalar (NDS form: src2 is
// rm, src1 is vvvv, the XMM destination is reg). Like the scalar 0F3A
// forms, these take the 66 prefix; W selects double/single.
"VRCP14SD": {2, 0x4D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VRCP14SS": {2, 0x4D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
"VRSQRT14SD": {2, 0x4F, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VRSQRT14SS": {2, 0x4F, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
"VGETEXPSD": {2, 0x43, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VGETEXPSS": {2, 0x43, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.66.0F38, scale by a power of two (NDS form).
"VSCALEFPD": {2, 0x2C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSCALEFPS": {2, 0x2C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSCALEFSD": {2, 0x2D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VSCALEFSS": {2, 0x2D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.66.0F3A, packed round/getmant/reduce ($imm, src, dst: reg=dst,
// rm=src, imm8).
"VRNDSCALEPD": {3, 0x09, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VRNDSCALEPS": {3, 0x08, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VGETMANTPD": {3, 0x26, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VGETMANTPS": {3, 0x26, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VREDUCEPD": {3, 0x56, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VREDUCEPS": {3, 0x56, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.66.0F3A, scalar round/getmant/reduce and fixup/range (NDS +
// imm8: $imm, src2, src1, dst). The scalar 0F3A forms all take the 66
// prefix; W selects double/single.
"VRNDSCALESD": {3, 0x0B, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VRNDSCALESS": {3, 0x0A, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
"VGETMANTSD": {3, 0x27, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VGETMANTSS": {3, 0x27, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
"VREDUCESD": {3, 0x57, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VREDUCESS": {3, 0x57, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
"VFIXUPIMMPD": {3, 0x54, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VFIXUPIMMPS": {3, 0x54, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VFIXUPIMMSD": {3, 0x55, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VFIXUPIMMSS": {3, 0x55, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
"VRANGEPD": {3, 0x50, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VRANGEPS": {3, 0x50, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VRANGESD": {3, 0x51, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VRANGESS": {3, 0x51, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
// EVEX.66.0F3A, floating-point class test ($imm, src, kdst): the
// reg field carries the opmask destination. The packed forms carry an
// explicit length in the mnemonic (X/Y/Z).
"VFPCLASSPDX": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{16, 0, 0}},
"VFPCLASSPDY": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 32, 0}},
"VFPCLASSPDZ": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 0, 64}},
"VFPCLASSPSX": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{16, 0, 0}},
"VFPCLASSPSY": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 32, 0}},
"VFPCLASSPSZ": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 0, 64}},
"VFPCLASSSD": {3, 0x67, 1, 1, -1, vexImmRM, [3]int{8, 0, 0}},
"VFPCLASSSS": {3, 0x67, 0, 1, -1, vexImmRM, [3]int{4, 0, 0}},
// EVEX, the remaining conversions. VCVTQQ2PS narrows (the 512-bit
// source sets the length); the rest follow the destination.
"VCVTQQ2PS": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2QQ": {1, 0x7B, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PS": {1, 0x7A, 0, 3, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F38, half-precision convert (half-width source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
// rm=dst, imm8, the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
// EVEX, unsigned and truncating conversions. The PD sources are the
// wide operand (the bare names are 512-bit only, the X/Y spellings fix
// the length); the PS/UQQ destinations are wide and follow the
// destination.
"VCVTPD2PS": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2PSX": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTPD2PSY": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
"VCVTPD2UDQ": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2UDQX": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTPD2UDQY": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
"VCVTTPD2UDQ": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTTPD2UDQX": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTTPD2UDQY": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
"VCVTTPD2UQQ": {1, 0x78, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTPS2UDQ": {1, 0x79, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
"VCVTTPS2UDQ": {1, 0x78, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
"VCVTPS2UQQ": {1, 0x79, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTTPS2UQQ": {1, 0x78, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTTPD2QQ": {1, 0x7A, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTTPS2QQ": {1, 0x7A, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUQQ2PD": {1, 0x7A, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
"VCVTUQQ2PS": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTUQQ2PSX": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTUQQ2PSY": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
"VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.66.0F38, the remaining sign/zero-extending moves (narrow
// source; disp8×N follows its size).
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.F3.0F38, the remaining narrowing stores (vector source in reg,
// narrow destination in r/m): signed, unsigned and the D/Q truncations.
"VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
"VPMOVSDW": {2, 0x23, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVSQW": {2, 0x24, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVSQD": {2, 0x25, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVSWB": {2, 0x20, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVUSWB": {2, 0x10, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVUSDB": {2, 0x11, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVUSQB": {2, 0x12, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
"VPMOVUSDW": {2, 0x13, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVUSQW": {2, 0x14, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVUSQD": {2, 0x15, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
// EVEX.F3.0F38, mask/vector conversions: M2* moves an opmask register
// into a vector (rm = K source, reg = vector destination), *2M does the
// reverse (reg = K destination, rm = vector source, the length follows
// the vector).
"VPMOVM2B": {2, 0x28, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVM2W": {2, 0x28, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVM2D": {2, 0x38, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVM2Q": {2, 0x38, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVB2M": {2, 0x29, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVW2M": {2, 0x29, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX, scalar conversions between vector and general-purpose
// registers. Vector to GPR (two operands: vec/mem source, GPR
// destination, vvvv unused): the signed and truncated pair, and the
// unsigned forms (EVEX only).
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTSD2USIL": {1, 0x79, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTSD2USIQ": {1, 0x79, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTSS2USIL": {1, 0x79, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTSS2USIQ": {1, 0x79, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTTSD2USIL": {1, 0x78, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTTSD2USIQ": {1, 0x78, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTTSS2USIL": {1, 0x78, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTTSS2USIQ": {1, 0x78, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
// vector source in vvvv, vector destination in reg).
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
"VCVTUSI2SDL": {1, 0x7B, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
"VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128/256/512.66.0F38.W0, sign-extend dwords to qwords; the memory
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
// the xmm/ymm/zmm destination lengths).
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.512.66.0F3A.W1, lane extract (reg=ZMM source, rm=YMM/memory
// destination, imm8).
"VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
// EVEX.66.0F38, more integer NDS forms (W distinguishes D/Q).
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
// EVEX.128/256/512, the wider integer set (AVX-512 F/BW): byte/word
// arithmetic, the bitwise ops with D/Q suffixes, min/max, averages and
// variable shifts. All NDS form; W distinguishes element size.
"VPADDB": {1, 0xFC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDW": {1, 0xFD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBB": {1, 0xF8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBW": {1, 0xF9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULLW": {1, 0xD5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPAVGB": {1, 0xE0, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPAVGW": {1, 0xE3, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINUB": {1, 0xDA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXUB": {1, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINSW": {1, 0xEA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXSW": {1, 0xEE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPANDD": {1, 0xDB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPANDQ": {1, 0xDB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPANDND": {1, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPANDNQ": {1, 0xDF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINSB": {2, 0x38, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXSB": {2, 0x3C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINSQ": {2, 0x39, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXSQ": {2, 0x3D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINUW": {2, 0x3A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXUW": {2, 0x3E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINSD": {2, 0x39, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXSD": {2, 0x3D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINUD": {2, 0x3B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXUD": {2, 0x3F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINUQ": {2, 0x3B, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXUQ": {2, 0x3F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSLLVD": {2, 0x47, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSLLVQ": {2, 0x47, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLVD": {2, 0x45, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLVQ": {2, 0x45, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRAVD": {2, 0x46, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRAVQ": {2, 0x46, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX forms of instructions that also exist in VEX (selected when a ZMM
// or K register, or indices 16-31, demand EVEX).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F, immediate shift (VPSLLD /6).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.F3.0F38.W0, narrowing stores: reg = wide source, rm = narrow
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
// --- the AVX-512 families the avx512enc corpus exercises, read off
// the toolchain opcodetables ---
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VALIGNQ": {3, 0x03, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VANDNPD": {1, 0x55, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VANDPD": {1, 0x54, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VBLENDMPD": {2, 0x65, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VBLENDMPS": {2, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VBROADCASTF32X2": {2, 0x19, 0, 1, -1, vexRM, [3]int{0, 8, 8}},
"VBROADCASTF32X4": {2, 0x1A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTF32X8": {2, 0x1B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
"VBROADCASTF64X2": {2, 0x1A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTF64X4": {2, 0x1B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
"VBROADCASTI32X2": {2, 0x59, 0, 1, -1, vexRM, [3]int{8, 8, 8}},
"VBROADCASTI32X4": {2, 0x5A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTI32X8": {2, 0x5B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
"VBROADCASTI64X2": {2, 0x5A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTI64X4": {2, 0x5B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
"VCOMISD": {1, 0x2F, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
"VCVTSD2SS": {1, 0x5A, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
"VDBPSADBW": {3, 0x42, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VEXP2PD": {2, 0xC8, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
"VEXP2PS": {2, 0xC8, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev, [3]int{16, 32, 64}},
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VMOVNTPD": {1, 0x2B, 1, 1, -1, vexRMRev, [3]int{16, 32, 64}},
"VORPD": {1, 0x56, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMB": {2, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMD": {2, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMQ": {2, 0x64, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMW": {2, 0x66, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBROADCASTMB2Q": {2, 0x2A, 1, 2, -1, vexRM, [3]int{0, 0, 0}},
"VPBROADCASTMW2D": {2, 0x3A, 0, 2, -1, vexRM, [3]int{0, 0, 0}},
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPEQQ": {2, 0x29, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTQ": {2, 0x37, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCOMPRESSB": {2, 0x63, 0, 1, -1, vexRMRev, [3]int{1, 1, 1}},
"VPCOMPRESSW": {2, 0x63, 1, 1, -1, vexRMRev, [3]int{2, 2, 2}},
"VPCONFLICTD": {2, 0xC4, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPCONFLICTQ": {2, 0xC4, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPDPBUSD": {2, 0x50, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPDPBUSDS": {2, 0x51, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPDPWSSD": {2, 0x52, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPDPWSSDS": {2, 0x53, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2PD": {2, 0x77, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2PS": {2, 0x77, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2W": {2, 0x75, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
"VPERMT2B": {2, 0x7D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2PS": {2, 0x7F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2W": {2, 0x7D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPEXPANDB": {2, 0x62, 0, 1, -1, vexRM, [3]int{1, 1, 1}},
"VPEXPANDW": {2, 0x62, 1, 1, -1, vexRM, [3]int{2, 2, 2}},
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm, [3]int{4, 0, 0}},
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm, [3]int{8, 0, 0}},
"VPLZCNTD": {2, 0x44, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPLZCNTQ": {2, 0x44, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPMADD52HUQ": {2, 0xB5, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMADD52LUQ": {2, 0xB4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULDQ": {2, 0x28, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULTISHIFTQB": {2, 0x83, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULUDQ": {1, 0xF4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPOPCNTW": {2, 0x54, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPORD": {1, 0xEB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPROLVD": {2, 0x15, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPROLVQ": {2, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPRORVD": {2, 0x14, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPRORVQ": {2, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDD": {3, 0x71, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHLDQ": {3, 0x71, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHLDVD": {2, 0x71, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDVQ": {2, 0x71, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDVW": {2, 0x70, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDW": {3, 0x70, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHRDD": {3, 0x73, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHRDQ": {3, 0x73, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHRDVD": {2, 0x73, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHRDVQ": {2, 0x73, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHRDVW": {2, 0x72, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHRDW": {3, 0x72, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHUFBITQMB": {2, 0x8F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRAVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.66.0F73 /7, the byte-quad shift left (the count is always an
// immediate; there is no register-count twin).
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0, the plain-prefix (no 66) packed spellings
// whose EVEX form drops the legacy prefix entirely.
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
"VCOMISS": {1, 0x2F, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev, [3]int{16, 32, 64}},
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMB": {2, 0x26, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMD": {2, 0x27, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMQ": {2, 0x27, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMW": {2, 0x26, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMB": {2, 0x26, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMD": {2, 0x27, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMQ": {2, 0x27, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMW": {2, 0x26, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHQDQ": {1, 0x6D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLQDQ": {1, 0x6C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VRCP28PD": {2, 0xCA, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRCP28PS": {2, 0xCA, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRCP28SD": {2, 0xCB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VRCP28SS": {2, 0xCB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VRSQRT28PD": {2, 0xCC, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRSQRT28PS": {2, 0xCC, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRSQRT28SD": {2, 0xCD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VRSQRT28SS": {2, 0xCD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VSQRTPD": {1, 0x51, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VSQRTSD": {1, 0x51, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
"VUCOMISD": {1, 0x2E, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
"VXORPD": {1, 0x57, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.F3/F2.W0, word shuffles with an immediate
// ($imm, src, dst: reg = dst, rm = src, imm8). The F3/F2 prefixes
// split the high/low lane spellings.
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM, [3]int{16, 32, 64}},
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.128.66.0F3A, lane extract to a general-purpose register or
// memory ($imm, xsrc, GPR/mem dst: reg = source, rm = destination).
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtractGPR, [3]int{1, 1, 1}},
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR, [3]int{2, 2, 2}},
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtractGPR, [3]int{4, 4, 4}},
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtractGPR, [3]int{8, 8, 8}},
// EVEX.66.0F3A.W1, the qword permutes with an immediate control
// ($imm, src, dst: reg = dst, rm = src, imm8); the register-count
// forms live in evexRegFormTable.
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.66.0F3A, the packed permute shuffles with an immediate control.
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPERMILPD": {3, 0x05, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.128.0F.W0, high/low half moves. VMOVHPS carries the
// three-operand insert form (rm = m64 source, vvvv = preserved,
// reg = dst) and the two-operand store (reg = source, rm = m64);
// the encoder splits on the operand count. VMOVLHPS is the
// three-operand form alone.
"VMOVHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
"VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
}
// evexQuad describes one quad-register instruction: the opcode under
// EVEX.0F38.W0 with the F2 mandatory prefix, and the width of the vector
// registers the bracketed list and the destination take (512-bit ZMM for
// the packed forms, 128-bit XMM for the scalar ones).
type evexQuad struct {
opcode byte
width int // register width in bytes: 64 (ZMM) or 16 (XMM)
}
// evexQuadTable maps the quad-register instructions (the 4FMAPS and 4VNNIW
// families) to their encoding. The operand shape is fixed: a single memory
// source in r/m, the bracketed register list whose LOW register travels the
// inverted 5-bit V'VVVV field, an optional opmask in aaa and the vector
// destination in reg. The vector length follows the destination (512-bit
// for the ZMM list forms, 128-bit for the scalar ones) while the disp8×N
// multiplier stays 16 for every member, the toolchain's own tuple choice.
var evexQuadTable = map[string]evexQuad{
"V4FMADDPS": {0x9A, 64},
"V4FMADDSS": {0x9B, 16},
"V4FNMADDPS": {0xAA, 64},
"V4FNMADDSS": {0xAB, 16},
"VP4DPWSSD": {0x52, 64},
"VP4DPWSSDS": {0x53, 64},
}
// isEvexQuad reports whether the mnemonic is a quad-register instruction.
func isEvexQuad(upper string) bool {
_, ok := evexQuadTable[upper]
return ok
}
// encodeEvexQuad encodes the quad-register form: OP mem, [Zn-Zn+3], (K), dst.
// The register list is the VVVV-side source: its low register fills the
// inverted V'VVVV bits, which is why an indexed memory source above Z15 (no
// spare EVEX.X bit once V' is taken) is refused. Masking rides the standard
// aaa field, zeroing keeps the usual requires-a-mask rule, and no other
// suffix applies.
func (e *enc) encodeEvexQuad(mnem string, q evexQuad, ops []Operand, sfx evexSuffix) error {
if len(ops) != 3 && len(ops) != 4 {
return fmt.Errorf("%s expects 3 or 4 operands (mem, [Zn-Zn+3], (K), dst), got %d", mnem, len(ops))
}
mem, lst := ops[0], ops[1]
dst := ops[len(ops)-1]
mask := 0
if len(ops) == 4 {
k, ok := ops[2].(Reg)
if !ok || !k.mask {
return fmt.Errorf("%s: third operand must be an opmask register", mnem)
}
if k.idx == 0 {
return fmt.Errorf("k0 is not a usable mask register")
}
mask = k.idx
}
list, ok := lst.(RegList)
if !ok {
return fmt.Errorf("%s: second operand must be a four-register list", mnem)
}
if list.Lo.size != q.width {
return fmt.Errorf("%s: the register list must hold %d-bit vector registers", mnem, q.width*8)
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
if dstReg.size != q.width {
return fmt.Errorf("%s: the destination must be a %d-bit vector register", mnem, q.width*8)
}
if !memOperand(mem) {
return fmt.Errorf("%s: the source must be a memory operand", mnem)
}
// The list owns V'VVVV; a scaled index in the EVEX-only half would fold
// its fifth bit into the same field the list's low register occupies.
if m, ok := mem.(Mem); ok && m.HasIndex && m.Index.idx >= 16 {
return fmt.Errorf("%s: an index register above Z15 has no EVEX bit free", mnem)
}
if sfx.zeroing && mask == 0 {
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnem)
}
spec := evexSpec{mapSel: 2, opcode: q.opcode, w: 0, pp: 3, opdigit: -1, n: [3]int{16, 16, 16}}
// The vector length follows the destination (512-bit for the ZMM forms,
// 128-bit for the scalar ones), exactly as the oracle encodes it.
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, list.Lo.idx, mem, mask, sfx)
}
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
// depends on the source kind, a GPR source uses opReg, a memory source uses
// opMem with a disp8×N of n.
type evexBcastSpec struct {
mapSel int
opReg byte
opMem byte
w int
n int
}
var evexBcastTable = map[string]evexBcastSpec{
// EVEX.128/256/512.66.0F38, broadcast a dword/qword to all lanes.
"VPBROADCASTD": {2, 0x7C, 0x58, 0, 4},
"VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8},
// EVEX.128/256/512.66.0F38, broadcast a byte/word (GPR or memory
// source) to all lanes.
"VPBROADCASTB": {2, 0x7A, 0x78, 0, 1},
"VPBROADCASTW": {2, 0x7B, 0x79, 0, 2},
}
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
// move table). vecOK and xmmOnly mirror the VEX twin's operand rules: a
// scalar move (vecOK false, xmmOnly true) takes XMM↔memory operands only.
type evexMoveSpec struct {
mapSel int
pp int
load byte // r/m → vector
store byte // vector → r/m
w int
n [3]int
vecOK bool // the non-memory operand may be a vector register
xmmOnly bool // wider than XMM registers are rejected
nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS)
}
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true},
// EVEX.128.F2.0F.W1, scalar double move: memory operands and the
// three-operand register form (VMOVSD dst, src1, src2).
"VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true},
// EVEX.128/256/512.0F.W0, unaligned packed single move.
"VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false},
}
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
func isEvex(mnemUpper string) bool {
if _, ok := evexTable[mnemUpper]; ok {
return true
}
if _, ok := evexBcastTable[mnemUpper]; ok {
return true
}
if _, ok := evexMoveTable[mnemUpper]; ok {
return true
}
return isEvexQuad(mnemUpper)
}
// evexRequired reports whether the operands force the EVEX encoding of a
// mnemonic that also has a VEX form: ZMM and K registers do, and so do
// register indices 16-31, which only EVEX can represent (X16-Y31 exist
// solely under AVX-512).
func evexRequired(upper string, ops []Operand) bool {
_, inVex := vexTable[upper]
_, inVexMove := vexMoveTable[upper]
if !inVex && !inVexMove {
return true // EVEX-only mnemonic
}
// The byte-quad shifts have VEX register forms but EVEX-only memory
// forms: a memory count source forces the EVEX encoding.
if upper == "VPSLLDQ" || upper == "VPSRLDQ" {
if slices.ContainsFunc(ops, memOperand) {
return true
}
}
for _, op := range ops {
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
return true
}
}
return false
}
// evexSuffix carries the EVEX mnemonic suffixes the Go assembler accepts:
// zeroing (.Z), a rounding mode (.RN_SAE, .RD_SAE, .RU_SAE, .RZ_SAE),
// suppress-all-exceptions (.SAE) and memory broadcast (.BCST). Masking is
// not a suffix, Go writes it as an explicit K operand.
type evexSuffix struct {
zeroing bool
sae bool
bcst bool
rounding int // -1 = none; otherwise the EVEX rc value (0 RN, 1 RD, 2 RU, 3 RZ)
}
// any reports whether any suffix is present.
func (s evexSuffix) any() bool {
return s.zeroing || s.sae || s.bcst || s.rounding >= 0
}
// evexOnly reports whether the suffix forces the EVEX encoding (everything
// but plain zeroing, which the dispatch checks separately).
func (s evexSuffix) evexOnly() bool {
return s.sae || s.bcst || s.rounding >= 0
}
// parseEvexSuffix splits the EVEX suffix chain off the mnemonic
// ("VADDPD.RN_SAE.Z" → base "VADDPD", rounding RN, zeroing), validating the
// combinations the Go assembler allows: .Z last, no duplicates, no
// broadcast together with rounding/SAE.
func parseEvexSuffix(mnem string) (string, evexSuffix, error) {
sfx := evexSuffix{rounding: -1}
before, after, ok := strings.Cut(mnem, ".")
if !ok {
return mnem, sfx, nil
}
base := before
parts := strings.Split(after, ".")
seen := map[string]bool{}
for j, p := range parts {
if seen[p] {
return "", sfx, fmt.Errorf("duplicate EVEX suffix %q", p)
}
seen[p] = true
switch p {
case "Z":
if j != len(parts)-1 {
return "", sfx, fmt.Errorf("the .Z suffix must come last in %q", after)
}
sfx.zeroing = true
case "SAE":
sfx.sae = true
case "BCST":
sfx.bcst = true
case "RN_SAE":
sfx.rounding = 0
case "RD_SAE":
sfx.rounding = 1
case "RU_SAE":
sfx.rounding = 2
case "RZ_SAE":
sfx.rounding = 3
default:
return "", sfx, fmt.Errorf("unsupported EVEX suffix %q", p)
}
}
if sfx.bcst && (sfx.sae || sfx.rounding >= 0) {
return "", sfx, fmt.Errorf("cannot combine .BCST with rounding or SAE in %q", after)
}
return base, sfx, nil
}
// evexRound lists the instructions that accept a rounding mode or .SAE.
var evexRound = map[string]bool{
"VADDPD": true, "VSUBPD": true, "VMULPD": true, "VDIVPD": true,
"VMINPD": true, "VMAXPD": true,
"VADDPS": true, "VSUBPS": true, "VMULPS": true, "VDIVPS": true,
"VMINPS": true, "VMAXPS": true,
"VADDSD": true, "VSUBSD": true, "VMULSD": true, "VDIVSD": true,
"VMINSD": true, "VMAXSD": true,
"VADDSS": true, "VSUBSS": true, "VMULSS": true, "VDIVSS": true,
"VMINSS": true, "VMAXSS": true,
"VSCALEFPD": true, "VSCALEFPS": true, "VSCALEFSD": true, "VSCALEFSS": true,
"VGETEXPPD": true, "VGETEXPPS": true, "VGETEXPSD": true, "VGETEXPSS": true,
"VCVTDQ2PS": true, "VCVTPS2QQ": true, "VCVTQQ2PS": true, "VCVTPD2UQQ": true,
"VCVTPD2PS": true, "VCVTPD2UDQ": true, "VCVTTPD2UDQ": true, "VCVTTPD2UQQ": true,
"VCVTPS2UDQ": true, "VCVTTPS2UDQ": true, "VCVTPS2UQQ": true, "VCVTTPS2UQQ": true,
"VCVTTPD2QQ": true, "VCVTTPS2QQ": true, "VCVTUQQ2PD": true, "VCVTUQQ2PS": true,
"VCVTSD2SI": true, "VCVTSD2SIQ": true, "VCVTSS2SI": true, "VCVTSS2SIQ": true,
"VCVTSD2USIL": true, "VCVTSD2USIQ": true, "VCVTSS2USIL": true, "VCVTSS2USIQ": true,
"VCVTTSD2SI": true, "VCVTTSD2SIQ": true, "VCVTTSS2SI": true, "VCVTTSS2SIQ": true,
"VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true,
"VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true,
"VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true,
}
// evexBcstN maps an instruction accepting .BCST to the broadcast element
// size, the disp8×N multiplier for its memory operand.
var evexBcstN = map[string]int{
"VADDPD": 8, "VSUBPD": 8, "VMULPD": 8, "VDIVPD": 8,
"VMINPD": 8, "VMAXPD": 8,
"VADDPS": 4, "VSUBPS": 4, "VMULPS": 4, "VDIVPS": 4,
"VMINPS": 4, "VMAXPS": 4,
"VRCP14PD": 8, "VRCP14PS": 4, "VRSQRT14PD": 8, "VRSQRT14PS": 4,
"VGETEXPPD": 8, "VGETEXPPS": 4,
"VSCALEFPD": 8, "VSCALEFPS": 4,
"VRNDSCALEPD": 8, "VRNDSCALEPS": 4,
"VGETMANTPD": 8, "VGETMANTPS": 4,
"VREDUCEPD": 8, "VREDUCEPS": 4,
"VFIXUPIMMPD": 8, "VFIXUPIMMPS": 4,
"VRANGEPD": 8, "VRANGEPS": 4,
"VCVTDQ2PS": 4, "VCVTPS2QQ": 4, "VCVTQQ2PS": 8,
"VCVTUDQ2PD": 4, "VCVTUDQ2PS": 4,
"VCVTPD2PS": 8, "VCVTPD2UDQ": 8, "VCVTTPD2UDQ": 8, "VCVTTPD2UQQ": 8,
"VCVTPS2UDQ": 4, "VCVTTPS2UDQ": 4, "VCVTPS2UQQ": 4, "VCVTTPS2UQQ": 4,
"VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8,
}
// splitMask extracts an explicit mask register (K1-K7) from the operand list,
// returning the remaining operands and the mask index. K0 is not a usable
// mask (aaa = 0 means "no mask"), matching the assembler.
func splitMask(ops []Operand) ([]Operand, int, error) {
var rest []Operand
mask := 0
for _, op := range ops {
if r, ok := op.(Reg); ok && r.mask {
if mask != 0 {
return nil, 0, fmt.Errorf("at most one mask register operand")
}
if r.idx == 0 {
return nil, 0, fmt.Errorf("k0 is not a usable mask register")
}
mask = r.idx
continue
}
rest = append(rest, op)
}
return rest, mask, nil
}
// encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The
// mask, when present, is an explicit K1-K7 operand anywhere among the
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
// broadcast.
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
// The mask/vector conversions take the K register as a genuine operand
// (source or destination), not as a mask, and accept no suffixes.
if evexKOperand[mnemUpper] {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
}
spec, ok := evexTable[mnemUpper]
if !ok {
return fmt.Errorf("unsupported instruction %q", mnemUpper)
}
return e.encodeEvexRM(spec, ops, 0, sfx)
}
spec, inTable := evexTable[mnemUpper]
if q, ok := evexQuadTable[mnemUpper]; ok {
// The quad-register family carries no rounding, SAE or broadcast;
// only masking and zeroing apply.
if sfx.sae || sfx.bcst || sfx.rounding >= 0 {
return fmt.Errorf("%s takes no rounding/SAE/broadcast suffix", mnemUpper)
}
return e.encodeEvexQuad(mnemUpper, q, ops, sfx)
}
if inTable {
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper)
}
if sfx.bcst {
n, ok := evexBcstN[mnemUpper]
if !ok {
return fmt.Errorf("%s: broadcast is not supported for this instruction", mnemUpper)
}
spec.n = [3]int{n, n, n}
}
// A mnemonic with an immediate and a register spelling (the
// variable-count shifts, the permutes) encodes the register one
// when the first operand is not an immediate.
if len(ops) > 0 {
if _, isImm := ops[0].(Imm); !isImm {
if alt, ok := evexRegFormTable[mnemUpper]; ok {
spec, inTable = alt, true
}
}
}
// The high/low half moves split by operand count: three operands
// insert, two store (VMOVHPS m64, X1).
if hs, ok := evexHptrTable[mnemUpper]; ok {
if len(ops) == 2 {
if hs.store.opcode == 0 {
return fmt.Errorf("%s has no two-operand form", mnemUpper)
}
return e.encodeEvexRMRev(hs.store, ops, 0, sfx)
}
spec = hs.insert
}
} else if sfx.evexOnly() {
return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper)
}
// Mask-destination comparisons (VPCMPEQD, VCMPPD $imm, …): the last
// operand is the destination K register, and any mask sits among the
// preceding operands.
kdst := func(encode func(evexSpec, []Operand, int, evexSuffix) error) error {
dst, ok := ops[len(ops)-1].(Reg)
if !ok || !dst.mask {
return nil // not a K-destination form; fall through
}
rest, mask, err := splitMask(ops[:len(ops)-1])
if err != nil {
return err
}
if sfx.zeroing && mask == 0 {
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper)
}
return encode(spec, append(rest, dst), mask, sfx)
}
if inTable && len(ops) > 0 {
switch spec.form {
case vexNDS3:
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
return kdst(e.encodeEvexNDS3)
}
case vexNDS3Imm:
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
return kdst(e.encodeEvexNDS3Imm)
}
case vexImmRM:
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
return kdst(e.encodeEvexImmRM)
}
}
}
rest, mask, err := splitMask(ops)
if err != nil {
return err
}
if sfx.zeroing && mask == 0 {
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper)
}
ops = rest
if bs, ok := evexBcastTable[mnemUpper]; ok {
if sfx.evexOnly() {
return fmt.Errorf("%s: broadcast instructions take no rounding/SAE/broadcast suffix", mnemUpper)
}
return e.encodeEvexBcast(bs, ops, mask, sfx)
}
if ms, ok := evexMoveTable[mnemUpper]; ok {
if sfx.evexOnly() {
return fmt.Errorf("%s: moves take no rounding/SAE/broadcast suffix", mnemUpper)
}
return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx)
}
if ps, ok := evexPrefGatherTable[mnemUpper]; ok {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
}
return e.encodeEvexPrefGather(mnemUpper, ps, ops, mask, sfx)
}
if !inTable {
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
}
switch spec.form {
case vexNDS3:
return e.encodeEvexNDS3(spec, ops, mask, sfx)
case vexRM:
return e.encodeEvexRM(spec, ops, mask, sfx)
case vexRMRev:
return e.encodeEvexRMRev(spec, ops, mask, sfx)
case vexImmRM:
return e.encodeEvexImmRM(spec, ops, mask, sfx)
case vexShiftImm:
return e.encodeEvexShiftImm(spec, ops, mask, sfx)
case vexNDS3Imm:
return e.encodeEvexNDS3Imm(spec, ops, mask, sfx)
case vexExtract:
return e.encodeEvexExtract(spec, ops, mask, sfx)
case vexExtractGPR:
return e.encodeEvexExtractGPR(spec, ops, mask, sfx)
case vexRMSrcLen:
return e.encodeEvexRMSrcLen(spec, ops, mask, sfx)
}
return fmt.Errorf("unhandled EVEX form for %s", mnemUpper)
}
// encodeEvexNDS3 encodes the three-operand NDS form: OP src2, src1, dst. The
// destination may be an opmask register (VPCMPEQD), in which case the vector
// length comes from the sources.
func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 3 {
return fmt.Errorf("EVEX NDS instruction expects 3 operands, got %d", len(ops))
}
src2, src1, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || (!dstReg.isVec() && !dstReg.mask) {
return fmt.Errorf("EVEX destination must be a vector or mask register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("EVEX vvvv operand must be a vector register")
}
ll := dstReg.vecLenBit()
if dstReg.mask {
ll = vvvvReg.vecLenBit()
if r, ok := src2.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
}
}
return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, sfx)
}
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
// no vvvv), e.g. VCVTQQ2PD. The destination may be an opmask register (the
// *2M mask conversions) or a general-purpose register (the scalar
// vector-to-GPR conversions); in both cases the vector length comes from
// the source.
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 2 {
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok {
return fmt.Errorf("EVEX destination must be a register")
}
ll := dstReg.vecLenBit()
if !dstReg.isVec() {
// Mask or GPR destination: the length follows the vector source
// (128 for a memory source).
ll = 0
if r, ok := src.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
}
}
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
}
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
// (reg = dst, rm = src, imm8), e.g. VPSHUFD. The destination may be an
// opmask register (VFPCLASS*), in which case the vector length comes from
// the source.
func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 3 {
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shuffle control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || (!dstReg.isVec() && !dstReg.mask) {
return fmt.Errorf("shuffle destination must be a vector or mask register")
}
ll := dstReg.vecLenBit()
if dstReg.mask {
if r, ok := src.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
} else if l, err := soleLen(spec.n); err == nil {
// A memory source with a length-fixed mnemonic
// (VFPCLASSPDX/Y/Z): the length comes from the table's
// single valid slot, not from the operand.
ll = l
}
} else if r, ok := src.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexShiftImm encodes an immediate shift: OP $imm, src, dst
// (ModRM.reg = /digit, vvvv = dst, rm = src, imm8), e.g. VPSRAD $31, Z3, Z5.
func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 3 {
return fmt.Errorf("EVEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shift count must be an immediate")
}
// The count source is a vector register or memory; the length the L'L
// field and the disp8×N multiplier follow is the destination's either
// way.
if !vecOrMem(src) {
return fmt.Errorf("shift source must be a vector register or memory")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("shift destination must be a vector register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, src, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexNDS3Imm encodes OP $imm, src2, src1, dst (reg=dst, vvvv=src1,
// rm=src2, imm8), e.g. VALIGND. The destination may be an opmask register
// (VCMPPD and friends), in which case the vector length comes from the
// sources.
func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 4 {
return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops))
}
imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shuffle control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || (!dstReg.isVec() && !dstReg.mask) {
return fmt.Errorf("destination must be a vector or mask register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("second source must be a vector register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
ll := dstReg.vecLenBit()
if dstReg.mask {
ll = vvvvReg.vecLenBit()
if r, ok := src2.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
}
}
if err := e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexExtract encodes OP $imm, zsrc, ydst (reg=ZMM source, rm=YMM/memory
// destination, imm8), e.g. VEXTRACTI64X4.
func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, zsrc, ydst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("extract source must be a vector register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexExtractGPR encodes the lane extract to a general-purpose
// register or memory: OP $imm, xsrc, dst (reg = the XMM source, rm = the
// destination, imm8). The encoding is 128-bit regardless of register
// numbers, so L'L is fixed at 0 and the disp8×N multiplier is the extracted
// element size the table carries.
func (e *enc) encodeEvexExtractGPR(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("extract source must be a vector register")
}
switch dst.(type) {
case Reg:
if dst.(Reg).isVec() {
return fmt.Errorf("extract destination must be a general-purpose register or memory")
}
case Mem, sbMem:
default:
return fmt.Errorf("extract destination must be a general-purpose register or memory")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, 0, srcReg.idx, -1, dst, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
// the store-form opcode (reg = source, rm = destination), matching the Go
// assembler. The scalar moves also carry a three-operand register form
// (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3
// opens.
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) == 3 {
if !ms.nds3 {
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
}
// The masked scalar register form keeps the Go assembler's own
// layout: the store opcode with reg = op0, vvvv = op1 and the
// destination in r/m (op2) — the bytes go tool asm emits, not
// the manual's NDS reading.
src, src1, dst := ops[0], ops[1], ops[2]
reg, ok := src.(Reg)
if !ok || !reg.isVec() {
return fmt.Errorf("%s: first operand must be a vector register", mnem)
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("%s: second operand must be a vector register", mnem)
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
if ms.xmmOnly && (reg.size != 16 || vvvvReg.size != 16 || dstReg.size != 16) {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
spec := evexSpec{mapSel: ms.mapSel, opcode: ms.store, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
return e.emitEvexFields(spec, dstReg.vecLenBit(), reg.idx, vvvvReg.idx, dst, mask, sfx)
}
if len(ops) != 2 {
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcIsVec := vecReg(src)
dstReg, dstIsVec := vecReg(dst)
op := ms.store
var reg Reg
var rm Operand
switch {
case srcIsVec && dstIsVec:
// A store-form reg-reg move, the layout the Go assembler uses; a
// scalar move has no two-register form at all (the register form
// takes three operands), matching the VEX twin's vecOK rule.
if !ms.vecOK {
return fmt.Errorf("%s does not take two vector registers", mnem)
}
reg, rm = srcReg, dst
case srcIsVec:
if !memOperand(dst) {
return fmt.Errorf("%s: invalid destination operand", mnem)
}
reg, rm = srcReg, dst
case dstIsVec:
if !memOperand(src) {
return fmt.Errorf("%s: invalid source operand", mnem)
}
op = ms.load
reg, rm = dstReg, src
default:
return fmt.Errorf("%s needs a vector register operand", mnem)
}
// The scalar move is 128-bit only, so the register the length follows
// must be an XMM (the VEX twin's xmmOnly rule; EVEX also reaches ZMM,
// hence the inequality rather than a YMM test).
if ms.xmmOnly && reg.size != 16 {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx)
}
// encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the length fixed by the mnemonic, the
// single valid slot of spec.n names the vector length (and the disp8×N
// multiplier) a register or memory source encodes.
func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 2 {
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("EVEX destination must be a vector register")
}
ll, err := soleLen(spec.n)
if err != nil {
return err
}
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
}
// soleLen returns the vector-length index of the single valid slot of n
// the length a length-fixed mnemonic (the EVEX conversion spellings) encodes
// regardless of its operands.
func soleLen(n [3]int) (int, error) {
ll := -1
for i, v := range n {
if v == 0 {
continue
}
if ll >= 0 {
return 0, fmt.Errorf("ambiguous vector-length table %v", n)
}
ll = i
}
if ll < 0 {
return 0, fmt.Errorf("empty vector-length table")
}
return ll, nil
}
// memOperand reports whether op is a memory reference (including a
// static-symbol reference).
func memOperand(op Operand) bool {
switch op.(type) {
case Mem, sbMem:
return true
}
return false
}
// encodeEvexRMRev encodes the narrowing-store form: OP src, dst with the wide
// source in the reg field and the narrow destination in r/m (VPMOVDW/QD).
func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 2 {
return fmt.Errorf("EVEX store instruction expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("EVEX source must be a vector register")
}
return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, sfx)
}
// encodeEvexBcast encodes VPBROADCASTD/Q: OP src, dst with the GPR or memory
// source broadcast to every lane of the vector destination.
func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 2 {
return fmt.Errorf("broadcast expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("broadcast destination must be a vector register")
}
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
switch r := src.(type) {
case Mem, sbMem:
spec.opcode = bs.opMem
spec.n = [3]int{bs.n, bs.n, bs.n}
case Reg:
// A GPR source uses the register broadcast opcode; a vector
// source shares the xmm/mem one (the low byte is copied from
// the lane or from the memory operand).
if r.isVec() {
spec.opcode = bs.opMem
spec.n = [3]int{bs.n, bs.n, bs.n}
} else {
spec.opcode = bs.opReg
}
default:
return fmt.Errorf("broadcast source must be a register or memory")
}
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, sfx)
}
// emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement
// (disp8×N compressed) for the given precomputed fields. regIdx is the
// unextended reg-field register index, or a /digit (0-7); vvvvIdx is the
// vvvv register index, or -1 when unused. mask (K1-K7, 0 = unmasked) and
// zeroing fill the aaa and z bits of the P2 byte.
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, sfx evexSuffix) error {
if ll > 2 {
return fmt.Errorf("invalid vector length")
}
// reg-field extension bits (R̄, R'̄), inverted.
rBar, rPrimeBar := 1, 1
if regIdx&8 != 0 {
rBar = 0
}
if regIdx&16 != 0 {
rPrimeBar = 0
}
// vvvv (inverted) and its extension bit V'̄.
vBar, vPrimeBar := 15, 1
if vvvvIdx >= 0 {
vBar = 15 - (vvvvIdx & 15)
if vvvvIdx&16 != 0 {
vPrimeBar = 0
}
}
var modrm, sib int
var disp []byte
xBar, bBar := 1, 1
var sb *sbRef
switch r := rm.(type) {
case Reg:
// ModRM.mod = 11: rm[3] extends via B̄, and rm[4] via X̄ (the EVEX
// register-register quirk).
modrm = 0xC0 | (regIdx&7)<<3 | (r.idx & 7)
sib = -1
if r.idx&8 != 0 {
bBar = 0
}
if r.idx&16 != 0 {
xBar = 0
}
case Mem:
var err error
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
if err != nil {
return err
}
// An indexed memory operand carries index[4] in V'̄ (Go folds it
// together with vvvv[4] into the same bit).
if r.HasIndex && r.Index.idx&16 != 0 {
vPrimeBar = 0
}
case sbMem:
// RIP-relative static-symbol reference; disp32 patched at link time
// (no disp8 scaling for RIP-relative addressing).
modrm = (regIdx&7)<<3 | 0x05
sib = -1
disp = le32(0)
sb = &sbRef{name: r.name, addend: r.addend}
default:
return fmt.Errorf("invalid EVEX r/m operand")
}
z := 0
if sfx.zeroing {
z = 1
}
// The b bit and the L'L field carry the rounding/SAE/broadcast mode:
// a rounding mode replaces L'L with the rc value, plain SAE and
// broadcast keep the vector length.
b := 0
switch {
case sfx.rounding >= 0:
b, ll = 1, sfx.rounding
case sfx.sae || sfx.bcst:
b = 1
}
p0 := byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4 | spec.mapSel)
p1 := byte(spec.w<<7 | vBar<<3 | 1<<2 | spec.pp)
p2 := byte(z<<7 | ll<<5 | b<<4 | vPrimeBar<<3 | mask) // z, L'L/rc, b, V', aaa
e.out = append(e.out, 0x62, p0, p1, p2, spec.opcode, byte(modrm))
if sib >= 0 {
e.out = append(e.out, byte(sib))
}
if sb != nil {
e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend})
}
e.out = append(e.out, disp...)
return nil
}
// memComponentsEvex computes the ModR/M byte (with the given reg field), the
// SIB byte (-1 if none), the displacement bytes and the (inverted sense)
// index/base extension bits for an EVEX memory operand. The displacement is
// compressed to disp8×N when it is a multiple of n and the quotient fits a
// signed byte; otherwise a full disp32 is used.
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
sib = -1
xBar, bBar = 1, 1 // inverted bits: 1 = no extension
// The disp32 fallback bounds the displacement by int32, and the
// compressed disp8 form reaches at most ±127×64, well inside it.
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
}
if !m.HasBase && !m.HasIndex {
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
}
needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4)
var mod int
switch {
case !m.HasBase:
mod = 0
disp = le32(m.Disp)
case m.Base.idx&7 == 5 && m.Disp == 0:
mod = 1
disp = []byte{0}
case m.Disp == 0:
mod = 0
case n > 0 && m.Disp%int64(n) == 0 && m.Disp/int64(n) >= -128 && m.Disp/int64(n) <= 127:
mod = 1
disp = []byte{byte(int8(m.Disp / int64(n)))}
default:
mod = 2
disp = le32(m.Disp)
}
if needSIB {
idxField := 4 // 100 = no index
if m.HasIndex {
idxField = m.Index.idx & 7
if m.Index.idx&8 != 0 {
xBar = 0
}
}
baseField := 5 // 101 = no base (with mod=00 → disp32)
if m.HasBase {
baseField = m.Base.idx & 7
if m.Base.idx&8 != 0 {
bBar = 0
}
}
return mod<<6 | regField<<3 | 0x04, scaleBits(m.Scale)<<6 | idxField<<3 | baseField, disp, xBar, bBar, nil
}
if m.Base.idx&8 != 0 {
bBar = 0
}
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil
}
// gatherSpec describes a gather/scatter family member: all live in
// 66.0F38; the opcode and W select the index and data element widths, and n
// is the data element size (the EVEX disp8×N multiplier).
type gatherSpec struct {
opcode byte
w int
n int
}
var gatherTable = map[string]gatherSpec{
"VGATHERDPS": {0x92, 0, 4},
"VGATHERDPD": {0x92, 1, 8},
"VGATHERQPS": {0x93, 0, 4},
"VGATHERQPD": {0x93, 1, 8},
"VPGATHERDD": {0x90, 0, 4},
"VPGATHERDQ": {0x90, 1, 8},
"VPGATHERQD": {0x91, 0, 4},
"VPGATHERQQ": {0x91, 1, 8},
}
var scatterTable = map[string]gatherSpec{
"VSCATTERDPS": {0xA2, 0, 4},
"VSCATTERDPD": {0xA2, 1, 8},
"VSCATTERQPS": {0xA3, 0, 4},
"VSCATTERQPD": {0xA3, 1, 8},
"VPSCATTERDD": {0xA0, 0, 4},
"VPSCATTERDQ": {0xA0, 1, 8},
"VPSCATTERQD": {0xA1, 0, 4},
"VPSCATTERQQ": {0xA1, 1, 8},
}
// isGather reports whether the mnemonic is a gather instruction.
func isGather(upper string) bool {
_, ok := gatherTable[upper]
return ok
}
// isScatter reports whether the mnemonic is a scatter instruction.
func isScatter(upper string) bool {
_, ok := scatterTable[upper]
return ok
}
// isEvexPrefGather reports whether the mnemonic is a gather/scatter
// prefetch hint.
func isEvexPrefGather(upper string) bool {
_, ok := evexPrefGatherTable[upper]
return ok
}
// evexRegFormTable holds the register-count twin of the immediate-form
// entries in evexTable. Several mnemonics name two encodings: an immediate
// count or control ($imm, src, dst …) and a register-count one whose second
// operand is a vector register or memory (count, src2, src1, dst). The
// immediate spelling lives in evexTable, this table carries the register
// spelling, and encodeEvex picks by whether the first operand is an
// immediate, the way vexVarShift does on the VEX side.
var evexRegFormTable = map[string]evexSpec{
"VPSLLD": {1, 0xF2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSLLQ": {1, 0xF3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSLLW": {1, 0xF1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRAD": {1, 0xE2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRAW": {1, 0xE1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRLD": {1, 0xD2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRLQ": {1, 0xD3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRLW": {1, 0xD1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
// EVEX.NDS.0F38.W1, the register-count permutes (the immediate
// controls live in evexTable under 0F3A).
"VPERMQ": {2, 0x36, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMPD": {2, 0x16, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.NDS.0F38, the register-count permil shuffles.
"VPERMILPS": {2, 0x0C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMILPD": {2, 0x0D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
}
// evexPrefGatherSpec describes a gather/scatter prefetch hint: one memory
// operand with a VSIB index and an opmask register, no destination. The
// ModRM.reg field carries a fixed /digit, the L'L field is fixed at 512, and
// the mask register is the instruction's only register operand.
type evexPrefGatherSpec struct {
mapSel int
opcode byte
w int
pp int
opdigit int
n int
}
var evexPrefGatherTable = map[string]evexPrefGatherSpec{
"VGATHERPF0DPD": {2, 0xC6, 1, 1, 1, 8},
"VGATHERPF0DPS": {2, 0xC6, 0, 1, 1, 4},
"VGATHERPF0QPD": {2, 0xC7, 1, 1, 1, 8},
"VGATHERPF0QPS": {2, 0xC7, 0, 1, 1, 4},
"VGATHERPF1DPD": {2, 0xC6, 1, 1, 2, 8},
"VGATHERPF1DPS": {2, 0xC6, 0, 1, 2, 4},
"VGATHERPF1QPD": {2, 0xC7, 1, 1, 2, 8},
"VGATHERPF1QPS": {2, 0xC7, 0, 1, 2, 4},
"VSCATTERPF0DPD": {2, 0xC6, 1, 1, 5, 8},
"VSCATTERPF0DPS": {2, 0xC6, 0, 1, 5, 4},
"VSCATTERPF0QPD": {2, 0xC7, 1, 1, 5, 8},
"VSCATTERPF0QPS": {2, 0xC7, 0, 1, 5, 4},
"VSCATTERPF1DPD": {2, 0xC6, 1, 1, 6, 8},
"VSCATTERPF1DPS": {2, 0xC6, 0, 1, 6, 4},
"VSCATTERPF1QPD": {2, 0xC7, 1, 1, 6, 8},
"VSCATTERPF1QPS": {2, 0xC7, 0, 1, 6, 4},
}
// evexHptrSpec describes the high/low half moves (VMOVHPS family): the
// three-operand insert shares an opcode with a two-operand store whose
// source is the vector register and whose destination is m64.
type evexHptrSpec struct {
insert evexSpec
store evexSpec // store.opcode == 0 when the mnemonic has no store form
}
var evexHptrTable = map[string]evexHptrSpec{
"VMOVHPS": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x17, w: 0, pp: 0, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
"VMOVLHPS": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
},
}
// encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib.
func (e *enc) encodeEvexPrefGather(upper string, ps evexPrefGatherSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 2 operands (K, vsib memory), got %d", upper, len(ops)+1)
}
m, ok := ops[0].(Mem)
if !ok || !m.HasIndex || !m.Index.isVec() {
return fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", upper)
}
spec := evexSpec{mapSel: ps.mapSel, opcode: ps.opcode, w: ps.w, pp: ps.pp, opdigit: ps.opdigit, n: [3]int{ps.n, ps.n, ps.n}}
return e.emitEvexFields(spec, 2, ps.opdigit, -1, m, mask, sfx)
}
// vsibLen validates a VSIB memory operand (the index must be a vector
// register) and returns it with the vector length the index selects, the
// EVEX L'L field follows the index register, not the data register.
func vsibLen(op Operand, what string) (Mem, int, error) {
m, ok := op.(Mem)
if !ok || !m.HasIndex || !m.Index.isVec() {
return Mem{}, 0, fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", what)
}
return m, m.Index.vecLenBit(), nil
}
// encodeGather encodes a gather. The VEX spelling carries the mask in a
// vector register (OP mask, vsib, dst: vvvv = mask, rm = vsib, reg = dst,
// L follows the data register); the EVEX spelling carries it in aaa (OP
// vsib, K, dst: rm = vsib, reg = dst, L follows the VSIB index).
func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexSuffix) error {
rest, mask, err := splitMask(ops)
if err != nil {
return err
}
if mask != 0 || sfx.any() {
// EVEX form: OP vsib, K, dst. The L'L field is the wider of the
// index and the data register lengths (the Go assembler's
// layout); the disp8×N multiplier stays the index element size.
if len(rest) != 2 {
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
}
vsib, ll, err := vsibLen(rest[0], upper)
if err != nil {
return err
}
dst, ok := rest[1].(Reg)
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", upper)
}
if d := dst.vecLenBit(); d > ll {
ll = d
}
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
}
// VEX form: OP mask, vsib, dst.
if len(rest) != 3 {
return fmt.Errorf("%s expects 3 operands (mask, vsib, dst), got %d", upper, len(rest))
}
maskReg, ok := rest[0].(Reg)
if !ok || !maskReg.isVec() {
return fmt.Errorf("%s: mask must be a vector register", upper)
}
vsib, _, err := vsibLen(rest[1], upper)
if err != nil {
return err
}
dst, ok := rest[2].(Reg)
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", upper)
}
spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1}
rBit := 0
if dst.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib)
}
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src,
// rm = the VSIB memory operand, the K mask in aaa and L following the VSIB
// index.
func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evexSuffix) error {
rest, mask, err := splitMask(ops)
if err != nil {
return err
}
if mask == 0 {
return fmt.Errorf("%s requires a K mask register", upper)
}
if len(rest) != 2 {
return fmt.Errorf("%s expects 3 operands (src, K, vsib), got %d", upper, len(ops))
}
src, ok := rest[0].(Reg)
if !ok || !src.isVec() {
return fmt.Errorf("%s: source must be a vector register", upper)
}
vsib, ll, err := vsibLen(rest[1], upper)
if err != nil {
return err
}
// The L'L field is the wider of the data register and the VSIB index
// lengths, the bytes go tool asm emits.
if d := src.vecLenBit(); d > ll {
ll = d
}
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
}
// evexKOperand lists the instructions whose K register is a genuine operand
// (the source or destination of a mask/vector conversion) rather than a
// mask modifier, the M2 and 2M conversions. They take no masking.
var evexKOperand = map[string]bool{
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
// The K-to-vector broadcast reads its opmask source from r/m.
"VPBROADCASTMB2Q": true, "VPBROADCASTMW2D": true,
}
// kmovSpec describes a KMOV width: the opcode depends on the operand
// direction, kk (k → k), kmem (k → mem), gprk (GPR/mem → k) and kgpr
// (k → GPR). Each direction group carries its own mandatory prefix and W:
// the k-destination/source forms share one pair, the GPR forms another.
type kmovSpec struct {
kk, kmem, gprk, kgpr byte
kPP, kW int // prefix and VEX.W for the k forms
gprPP, gprW int // prefix and VEX.W for the GPR forms
}
var kmovTable = map[string]kmovSpec{
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0, 0, 0},
"KMOVB": {0x90, 0x91, 0x92, 0x93, 1, 0, 1, 0},
"KMOVD": {0x90, 0x91, 0x92, 0x93, 1, 1, 3, 0},
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 0, 1, 3, 1},
}
// encodeKmov encodes a KMOV width, selecting the opcode by direction.
func (e *enc) encodeKmov(upper string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
ks := kmovTable[upper]
src, dst := ops[0], ops[1]
srcReg, srcIsReg := src.(Reg)
dstReg, dstIsReg := dst.(Reg)
srcK := srcIsReg && srcReg.mask
dstK := dstIsReg && dstReg.mask
switch {
case srcK && dstK:
// k ← k: reg = dst, rm = src.
spec := vexSpec{mapSel: 1, opcode: ks.kk, w: ks.kW, pp: ks.kPP, opdigit: -1}
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
case srcK && dstIsReg:
// GPR ← k: reg = dst, rm = src.
spec := vexSpec{mapSel: 1, opcode: ks.kgpr, w: ks.gprW, pp: ks.gprPP, opdigit: -1}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15, src)
case srcK:
if _, ok := dst.(Mem); !ok {
return fmt.Errorf("%s: invalid destination operand", upper)
}
// mem ← k: reg = src, rm = dst.
spec := vexSpec{mapSel: 1, opcode: ks.kmem, w: ks.kW, pp: ks.kPP, opdigit: -1}
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
case dstK:
// k ← GPR: reg = dst, rm = src. A memory source shares the k ← k
// opcode and prefix group (the ykmovb layout the Go assembler uses).
opcode, w, pp := ks.gprk, ks.gprW, ks.gprPP
if memOperand(src) {
opcode, w, pp = ks.kk, ks.kW, ks.kPP
}
spec := vexSpec{mapSel: 1, opcode: opcode, w: w, pp: pp, opdigit: -1}
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
}
return fmt.Errorf("%s requires a K register operand", upper)
}
// kOpSpec describes the VEX encoding of an opmask-register instruction: the
// L bit and the W/pp pair select the operand width, and the form the
// operand layout.
type kOpSpec struct {
mapSel int
opcode byte
w int
pp int
ll int
form vexForm
}
var kOpsTable = map[string]kOpSpec{
// k ← k OP k: reg = dst, vvvv = src1, rm = src2 (three opmask
// registers). Byte/word widths share W0 and differ by the 66 prefix;
// dword/qword share the W bit selection the Go assembler emits.
"KANDB": {1, 0x41, 0, 1, 1, vexNDS3},
"KANDW": {1, 0x41, 0, 0, 1, vexNDS3},
"KANDD": {1, 0x41, 1, 1, 1, vexNDS3},
"KANDQ": {1, 0x41, 1, 0, 1, vexNDS3},
"KANDNB": {1, 0x42, 0, 1, 1, vexNDS3},
"KANDNW": {1, 0x42, 0, 0, 1, vexNDS3},
"KANDND": {1, 0x42, 1, 1, 1, vexNDS3},
"KANDNQ": {1, 0x42, 1, 0, 1, vexNDS3},
"KORB": {1, 0x45, 0, 1, 1, vexNDS3},
"KORW": {1, 0x45, 0, 0, 1, vexNDS3},
"KORD": {1, 0x45, 1, 1, 1, vexNDS3},
"KORQ": {1, 0x45, 1, 0, 1, vexNDS3},
"KXNORB": {1, 0x46, 0, 1, 1, vexNDS3},
"KXNORW": {1, 0x46, 0, 0, 1, vexNDS3},
"KXNORD": {1, 0x46, 1, 1, 1, vexNDS3},
"KXNORQ": {1, 0x46, 1, 0, 1, vexNDS3},
"KXORB": {1, 0x47, 0, 1, 1, vexNDS3},
"KXORW": {1, 0x47, 0, 0, 1, vexNDS3},
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
"KUNPCKWD": {1, 0x4B, 0, 0, 1, vexNDS3},
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
"KADDD": {1, 0x4A, 1, 1, 1, vexNDS3},
"KADDQ": {1, 0x4A, 1, 0, 1, vexNDS3},
// k ← OP k (KNOT), k ← k AND~ k (KTEST-style RM) and flags ← k OP k
// (KORTEST): reg = dst, rm = src.
"KNOTB": {1, 0x44, 0, 1, 0, vexRM},
"KNOTW": {1, 0x44, 0, 0, 0, vexRM},
"KNOTD": {1, 0x44, 1, 1, 0, vexRM},
"KNOTQ": {1, 0x44, 1, 0, 0, vexRM},
"KORTESTB": {1, 0x98, 0, 1, 0, vexRM},
"KORTESTW": {1, 0x98, 0, 0, 0, vexRM},
"KORTESTD": {1, 0x98, 1, 1, 0, vexRM},
"KORTESTQ": {1, 0x98, 1, 0, 0, vexRM},
"KTESTB": {1, 0x99, 0, 1, 0, vexRM},
"KTESTW": {1, 0x99, 0, 0, 0, vexRM},
"KTESTD": {1, 0x99, 1, 1, 0, vexRM},
"KTESTQ": {1, 0x99, 1, 0, 0, vexRM},
// OP $imm, src, dst: reg = dst, rm = src, imm8. The opcodes split by
// direction (0x32/0x33 left, 0x30/0x31 right) and within each by
// element half (0x32 byte/word, 0x33 dword/qword); W picks byte/dword
// (W0) against word/qword (W1).
"KSHIFTLB": {3, 0x32, 0, 1, 0, vexImmRM},
"KSHIFTLW": {3, 0x32, 1, 1, 0, vexImmRM},
"KSHIFTLD": {3, 0x33, 0, 1, 0, vexImmRM},
"KSHIFTLQ": {3, 0x33, 1, 1, 0, vexImmRM},
"KSHIFTRB": {3, 0x30, 0, 1, 0, vexImmRM},
"KSHIFTRW": {3, 0x30, 1, 1, 0, vexImmRM},
"KSHIFTRD": {3, 0x31, 0, 1, 0, vexImmRM},
"KSHIFTRQ": {3, 0x31, 1, 1, 0, vexImmRM},
}
// isKOp reports whether the mnemonic is an opmask-register instruction.
func isKOp(upper string) bool {
_, ok := kOpsTable[upper]
return ok
}
// encodeKOp encodes an opmask-register instruction; every operand is a K
// register and the vector length is fixed by the instruction.
func (e *enc) encodeKOp(upper string, ops []Operand) error {
ks := kOpsTable[upper]
spec := vexSpec{mapSel: ks.mapSel, opcode: ks.opcode, w: ks.w, pp: ks.pp, opdigit: -1}
kreg := func(op Operand, what string) (Reg, error) {
r, ok := op.(Reg)
if !ok || !r.mask {
return Reg{}, fmt.Errorf("%s: %s must be an opmask register", upper, what)
}
return r, nil
}
switch ks.form {
case vexNDS3:
if len(ops) != 3 {
return fmt.Errorf("%s expects 3 operands, got %d", upper, len(ops))
}
src2, err := kreg(ops[0], "first source")
if err != nil {
return err
}
src1, err := kreg(ops[1], "second source")
if err != nil {
return err
}
dst, err := kreg(ops[2], "destination")
if err != nil {
return err
}
return e.emitVexFields(spec, ks.ll, dst.idx, 0, 15-src1.idx, src2)
case vexRM:
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
src, err := kreg(ops[0], "source")
if err != nil {
return err
}
dst, err := kreg(ops[1], "destination")
if err != nil {
return err
}
return e.emitVexFields(spec, ks.ll, dst.idx, 0, 15, src)
case vexImmRM:
if len(ops) != 3 {
return fmt.Errorf("%s expects 3 operands ($imm, src, dst), got %d", upper, len(ops))
}
immVal, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("%s: shift count must be an immediate", upper)
}
src, err := kreg(ops[1], "source")
if err != nil {
return err
}
dst, err := kreg(ops[2], "destination")
if err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitVexFields(spec, ks.ll, dst.idx, 0, 15, src); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
return fmt.Errorf("unhandled opmask form for %s", upper)
}