Files
gasm-sdk/asm/evex.go
T
petrbalvin 19a26e049b
Test / vet (push) Successful in 47s
Test / test (push) Successful in 2m35s
Test / build (push) Successful in 41s
feat(asm): add vpcmp compare, full opmask set, legacy sse integers and bswap
Assisted-by: GLM 5.3
2026-08-27 22:41:03 +02:00

1620 lines
64 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"fmt"
"strings"
)
// This file implements EVEX (AVX-512) instruction encoding: the four-byte
// EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the
// compressed disp8×N displacement, and the operand shapes the go-flac
// AVX-512 kernels use plus the common floating-point and conversion set.
// Masking follows the Go assembler's spelling: an explicit K1–K7 operand
// anywhere among the operands (merging) plus a ".Z" mnemonic suffix for
// zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are
// supported too.
// evexSpec describes one EVEX instruction's encoding parameters. The form
// field reuses the vexForm shapes, which carry over unchanged.
type evexSpec struct {
mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A
opcode byte
w int
pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2
opdigit int // ModRM.reg /digit, or -1 when reg is a register
form vexForm // vexNDS3, vexRM, vexShiftImm, vexNDS3Imm, vexExtract
n [3]int // disp8×N multiplier per vector length (128/256/512)
}
// evexTable maps an upper-case mnemonic to its EVEX encoding. Mnemonics
// that also have a VEX form (VPADDD, VMOVUPD, …) are dispatched here only
// when an operand demands EVEX (a ZMM or K register); EVEX-only mnemonics
// (VPXORD, VALIGND, …) always encode through this table. The N multipliers
// are taken from the Go assembler's opcode tables, which are authoritative
// for byte-for-byte agreement.
var evexTable = map[string]evexSpec{
// EVEX.128/256/512.66.0F — integer arithmetic / logic, NDS form.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBQ": {1, 0xFB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPXORD": {1, 0xEF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPXORQ": {1, 0xEF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — packed double arithmetic.
"VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0 — packed single arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — packed double unpack.
"VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128.F2.0F.W1 — scalar double arithmetic (the packed opcodes with
// an F2 pp; the EVEX forms exist for masked and zeroing use). The
// memory operand is a single double, so disp8×N = 8.
"VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VSUBSD": {1, 0x5C, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMULSD": {1, 0x59, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VDIVSD": {1, 0x5E, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128.F3.0F.W0 — scalar single arithmetic (disp8×N = 4).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.512.66.0F3A — align (NDS + imm8).
"VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F — immediate shift (VPSRAD /4).
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — variable shift with an XMM count (VPSRAQ;
// the W bit distinguishes it from VPSRAD's E2 form).
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst,
// rm=src, no vvvv).
"VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1 — duplicate the low double (reg=dst,
// rm=src, no vvvv): a 128-bit destination reads a single double from
// memory (disp8×8), the wider ones read the full operand.
"VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}},
// EVEX.128/256/512.0F.W0 — signed dword to packed single (reg=dst,
// rm=src, no vvvv, no mandatory prefix — as in the VEX form).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0 — packed single to packed double: the
// destination is twice the source width and sets the length; disp8×N
// follows the narrow memory source. No F3 prefix: the Go assembler
// emits this instruction with pp = 00 (Intel's maps would call that
// undefined) and gasm reproduces the Go assembler's bytes — its machine
// code is the oracle, not the manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.128/256/512.F3.0F.W0 — signed dword to packed double (the EVEX
// form of the VEX instruction; the destination sets the length, disp8×N
// follows the narrow memory source).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
// EVEX packed double → dword conversions: the source is the wide
// operand and the mnemonic fixes the length — the bare names are
// 512-bit only (ZMM source, XMM destination), the X/Y spellings are
// EVEX-128/256. Exactly one slot of n is valid; it names the vector
// length (and the disp8×N multiplier) a register or memory source
// encodes.
"VCVTPD2DQ": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTTPD2DQ": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2DQX": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTPD2DQY": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
"VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.66.0F3A — ternary logic and lane shuffles (NDS + imm8).
"VPTERNLOGD": {3, 0x25, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPTERNLOGQ": {3, 0x25, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFI32X4": {3, 0x43, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFI64X2": {3, 0x43, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFF32X4": {3, 0x23, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFF64X2": {3, 0x23, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F — the EVEX forms of the VEX two-source shuffle.
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F3A — lane insert ($imm, xsrc, zsrc1, zdst).
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTF32X8": {3, 0x1A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
"VINSERTF64X2": {3, 0x18, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTF64X4": {3, 0x1A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
"VINSERTI32X4": {3, 0x38, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTI32X8": {3, 0x3A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
"VINSERTI64X2": {3, 0x38, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTI64X4": {3, 0x3A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
// EVEX.66.0F3A — lane extract (reg=source, rm=XMM/YMM destination,
// imm8).
"VEXTRACTF32X4": {3, 0x19, 0, 1, -1, vexExtract, [3]int{0, 16, 16}},
"VEXTRACTF32X8": {3, 0x1B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTF64X2": {3, 0x19, 1, 1, -1, vexExtract, [3]int{0, 16, 16}},
"VEXTRACTI32X4": {3, 0x39, 0, 1, -1, vexExtract, [3]int{0, 16, 16}},
"VEXTRACTI32X8": {3, 0x3B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTI64X2": {3, 0x39, 1, 1, -1, vexExtract, [3]int{0, 16, 16}},
// EVEX.66.0F — compare with an opmask destination ($imm, src2, src1,
// kdst): NDS3Imm with the K register in the reg field.
"VCMPPD": {1, 0xC2, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VCMPPS": {1, 0xC2, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}},
// EVEX.66.0F3A — integer compares with an opmask destination, the same
// NDS3Imm-with-k-reg shape as the floating-point compares; W selects the
// operand width (byte/word vs dword/qword), the opcode the signedness.
// The memory form takes a full vector, so disp8×N is 16/32/64.
"VPCMPB": {3, 0x3F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUB": {3, 0x3E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPW": {3, 0x3F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUW": {3, 0x3E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPD": {3, 0x1F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUD": {3, 0x1E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPQ": {3, 0x1F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F38 — permutes (NDS form).
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F — the wider integer set (NDS form).
"VPMADDWD": {1, 0xF5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKUSDW": {2, 0x2B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F38 — absolute values and replicating moves (reg=dst,
// rm=src).
"VPABSB": {2, 0x1C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSW": {2, 0x1D, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSD": {2, 0x1E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSQ": {2, 0x1F, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.F3.0F — replicate even/odd singles.
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — sign/zero-extending moves; the memory source is the
// narrow half (here byte to word).
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F — packed single conversions (reg=dst, rm=src).
"VCVTPS2DQ": {1, 0x5B, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTTPS2DQ": {1, 0x5B, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; disp8×N is the element size).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VBROADCASTSD": {2, 0x19, 1, 1, -1, vexRM, [3]int{0, 8, 8}},
// EVEX.66.0F38 — expand loads (rm → vector register destination).
"VEXPANDPD": {2, 0x88, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
"VEXPANDPS": {2, 0x88, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VPEXPANDD": {2, 0x89, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VPEXPANDQ": {2, 0x89, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
// EVEX.66.0F38 — compress stores (vector register source → rm), and the
// remaining narrowing stores.
"VCOMPRESSPD": {2, 0x8A, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}},
"VCOMPRESSPS": {2, 0x8A, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}},
"VPCOMPRESSD": {2, 0x8B, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}},
"VPCOMPRESSQ": {2, 0x8B, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}},
"VPMOVWB": {2, 0x30, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQB": {2, 0x32, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
// EVEX.66.0F — rotates (immediate form: /0 right, /1 left).
"VPRORD": {1, 0x72, 0, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
"VPRORQ": {1, 0x72, 1, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
"VPROLD": {1, 0x72, 0, 1, 1, vexShiftImm, [3]int{16, 32, 64}},
"VPROLQ": {1, 0x72, 1, 1, 1, vexShiftImm, [3]int{16, 32, 64}},
// EVEX word shifts.
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX W1 qword shifts.
"VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
"VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.66.0F38 — floating-point helpers, packed (reg=dst, rm=src).
"VRCP14PD": {2, 0x4C, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRCP14PS": {2, 0x4C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRSQRT14PD": {2, 0x4E, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRSQRT14PS": {2, 0x4E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VGETEXPPD": {2, 0x42, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VGETEXPPS": {2, 0x42, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — floating-point helpers, scalar (NDS form: src2 is
// rm, src1 is vvvv, the XMM destination is reg). Like the scalar 0F3A
// forms, these take the 66 prefix; W selects double/single.
"VRCP14SD": {2, 0x4D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VRCP14SS": {2, 0x4D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
"VRSQRT14SD": {2, 0x4F, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VRSQRT14SS": {2, 0x4F, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
"VGETEXPSD": {2, 0x43, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VGETEXPSS": {2, 0x43, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.66.0F38 — scale by a power of two (NDS form).
"VSCALEFPD": {2, 0x2C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSCALEFPS": {2, 0x2C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSCALEFSD": {2, 0x2D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VSCALEFSS": {2, 0x2D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.66.0F3A — packed round/getmant/reduce ($imm, src, dst: reg=dst,
// rm=src, imm8).
"VRNDSCALEPD": {3, 0x09, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VRNDSCALEPS": {3, 0x08, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VGETMANTPD": {3, 0x26, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VGETMANTPS": {3, 0x26, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VREDUCEPD": {3, 0x56, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VREDUCEPS": {3, 0x56, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.66.0F3A — scalar round/getmant/reduce and fixup/range (NDS +
// imm8: $imm, src2, src1, dst). The scalar 0F3A forms all take the 66
// prefix; W selects double/single.
"VRNDSCALESD": {3, 0x0B, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VRNDSCALESS": {3, 0x0A, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
"VGETMANTSD": {3, 0x27, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VGETMANTSS": {3, 0x27, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
"VREDUCESD": {3, 0x57, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VREDUCESS": {3, 0x57, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
"VFIXUPIMMPD": {3, 0x54, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VFIXUPIMMPS": {3, 0x54, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VFIXUPIMMSD": {3, 0x55, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VFIXUPIMMSS": {3, 0x55, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
"VRANGEPD": {3, 0x50, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VRANGEPS": {3, 0x50, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VRANGESD": {3, 0x51, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VRANGESS": {3, 0x51, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
// EVEX.66.0F3A — floating-point class test ($imm, src, kdst): the
// reg field carries the opmask destination. The packed forms carry an
// explicit length in the mnemonic (X/Y/Z).
"VFPCLASSPDX": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{16, 0, 0}},
"VFPCLASSPDY": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 32, 0}},
"VFPCLASSPDZ": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 0, 64}},
"VFPCLASSPSX": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{16, 0, 0}},
"VFPCLASSPSY": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 32, 0}},
"VFPCLASSPSZ": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 0, 64}},
"VFPCLASSSD": {3, 0x67, 1, 1, -1, vexImmRM, [3]int{8, 0, 0}},
"VFPCLASSSS": {3, 0x67, 0, 1, -1, vexImmRM, [3]int{4, 0, 0}},
// EVEX — the remaining conversions. VCVTQQ2PS narrows (the 512-bit
// source sets the length); the rest follow the destination.
"VCVTQQ2PS": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2QQ": {1, 0x7B, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F38 — half-precision convert (half-width source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src,
// rm=dst, imm8 — the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
// EVEX — unsigned and truncating conversions. The PD sources are the
// wide operand (the bare names are 512-bit only, the X/Y spellings fix
// the length); the PS/UQQ destinations are wide and follow the
// destination.
"VCVTPD2PS": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2PSX": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTPD2PSY": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
"VCVTPD2UDQ": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2UDQX": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTPD2UDQY": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
"VCVTTPD2UDQ": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTTPD2UDQX": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTTPD2UDQY": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
"VCVTTPD2UQQ": {1, 0x78, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTPS2UDQ": {1, 0x79, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
"VCVTTPS2UDQ": {1, 0x78, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
"VCVTPS2UQQ": {1, 0x79, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTTPS2UQQ": {1, 0x78, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTTPD2QQ": {1, 0x7A, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTTPS2QQ": {1, 0x7A, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUQQ2PD": {1, 0x7A, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
"VCVTUQQ2PS": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTUQQ2PSX": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTUQQ2PSY": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
"VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.66.0F38 — the remaining sign/zero-extending moves (narrow
// source; disp8×N follows its size).
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.F3.0F38 — the remaining narrowing stores (vector source in reg,
// narrow destination in r/m): signed, unsigned and the D/Q truncations.
"VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
"VPMOVSDW": {2, 0x23, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVSQW": {2, 0x24, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVSQD": {2, 0x25, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVSWB": {2, 0x20, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVUSWB": {2, 0x10, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVUSDB": {2, 0x11, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVUSQB": {2, 0x12, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
"VPMOVUSDW": {2, 0x13, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVUSQW": {2, 0x14, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVUSQD": {2, 0x15, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
// EVEX.F3.0F38 — mask/vector conversions: M2* moves an opmask register
// into a vector (rm = K source, reg = vector destination), *2M does the
// reverse (reg = K destination, rm = vector source, the length follows
// the vector).
"VPMOVM2B": {2, 0x28, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVM2W": {2, 0x28, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVM2D": {2, 0x38, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVM2Q": {2, 0x38, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVB2M": {2, 0x29, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVW2M": {2, 0x29, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX — scalar conversions between vector and general-purpose
// registers. Vector to GPR (two operands: vec/mem source, GPR
// destination, vvvv unused): the signed and truncated pair, and the
// unsigned forms (EVEX only).
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTSD2USIL": {1, 0x79, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTSD2USIQ": {1, 0x79, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTSS2USIL": {1, 0x79, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTSS2USIQ": {1, 0x79, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTTSD2USIL": {1, 0x78, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTTSD2USIQ": {1, 0x78, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
"VCVTTSS2USIL": {1, 0x78, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
"VCVTTSS2USIQ": {1, 0x78, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
// vector source in vvvv, vector destination in reg).
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
"VCVTUSI2SDL": {1, 0x7B, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
"VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
// the xmm/ymm/zmm destination lengths).
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.512.66.0F3A.W1 — lane extract (reg=ZMM source, rm=YMM/memory
// destination, imm8).
"VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
// EVEX.66.0F38 — more integer NDS forms (W distinguishes D/Q).
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
// EVEX.128/256/512 — the wider integer set (AVX-512 F/BW): byte/word
// arithmetic, the bitwise ops with D/Q suffixes, min/max, averages and
// variable shifts. All NDS form; W distinguishes element size.
"VPADDB": {1, 0xFC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDW": {1, 0xFD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBB": {1, 0xF8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBW": {1, 0xF9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULLW": {1, 0xD5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPAVGB": {1, 0xE0, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPAVGW": {1, 0xE3, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINUB": {1, 0xDA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXUB": {1, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINSW": {1, 0xEA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXSW": {1, 0xEE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPANDD": {1, 0xDB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPANDQ": {1, 0xDB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPANDND": {1, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPANDNQ": {1, 0xDF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINSB": {2, 0x38, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXSB": {2, 0x3C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINSQ": {2, 0x39, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXSQ": {2, 0x3D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINUW": {2, 0x3A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXUW": {2, 0x3E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINSD": {2, 0x39, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXSD": {2, 0x3D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINUD": {2, 0x3B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXUD": {2, 0x3F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINUQ": {2, 0x3B, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXUQ": {2, 0x3F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSLLVD": {2, 0x47, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSLLVQ": {2, 0x47, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLVD": {2, 0x45, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLVQ": {2, 0x45, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRAVD": {2, 0x46, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRAVQ": {2, 0x46, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX forms of instructions that also exist in VEX (selected when a ZMM
// or K register, or indices 16–31, demand EVEX).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F — immediate shift (VPSLLD /6).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.F3.0F38.W0 — narrowing stores: reg = wide source, rm = narrow
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
}
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
// depends on the source kind — a GPR source uses opReg, a memory source uses
// opMem with a disp8×N of n.
type evexBcastSpec struct {
mapSel int
opReg byte
opMem byte
w int
n int
}
var evexBcastTable = map[string]evexBcastSpec{
// EVEX.128/256/512.66.0F38 — broadcast a dword/qword to all lanes.
"VPBROADCASTD": {2, 0x7C, 0x58, 0, 4},
"VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8},
// EVEX.128/256/512.66.0F38 — broadcast a byte/word (GPR or memory
// source) to all lanes.
"VPBROADCASTB": {2, 0x7A, 0x78, 0, 1},
"VPBROADCASTW": {2, 0x7B, 0x79, 0, 2},
}
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
// move table).
type evexMoveSpec struct {
mapSel int
pp int
load byte // r/m → vector
store byte // vector → r/m
w int
n [3]int
}
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0 — unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1 — unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W0 — unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1 — unaligned word move (shares the qword
// encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512 — aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F — aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128.F3.0F.W0 — scalar single move, memory operands (the
// three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}},
}
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
func isEvex(mnemUpper string) bool {
if _, ok := evexTable[mnemUpper]; ok {
return true
}
if _, ok := evexBcastTable[mnemUpper]; ok {
return true
}
_, ok := evexMoveTable[mnemUpper]
return ok
}
// evexRequired reports whether the operands force the EVEX encoding of a
// mnemonic that also has a VEX form: ZMM and K registers do, and so do
// register indices 16–31, which only EVEX can represent (X16–Y31 exist
// solely under AVX-512).
func evexRequired(upper string, ops []Operand) bool {
_, inVex := vexTable[upper]
_, inVexMove := vexMoveTable[upper]
if !inVex && !inVexMove {
return true // EVEX-only mnemonic
}
for _, op := range ops {
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
return true
}
}
return false
}
// evexSuffix carries the EVEX mnemonic suffixes the Go assembler accepts:
// zeroing (.Z), a rounding mode (.RN_SAE, .RD_SAE, .RU_SAE, .RZ_SAE),
// suppress-all-exceptions (.SAE) and memory broadcast (.BCST). Masking is
// not a suffix — Go writes it as an explicit K operand.
type evexSuffix struct {
zeroing bool
sae bool
bcst bool
rounding int // -1 = none; otherwise the EVEX rc value (0 RN, 1 RD, 2 RU, 3 RZ)
}
// any reports whether any suffix is present.
func (s evexSuffix) any() bool {
return s.zeroing || s.sae || s.bcst || s.rounding >= 0
}
// evexOnly reports whether the suffix forces the EVEX encoding (everything
// but plain zeroing, which the dispatch checks separately).
func (s evexSuffix) evexOnly() bool {
return s.sae || s.bcst || s.rounding >= 0
}
// parseEvexSuffix splits the EVEX suffix chain off the mnemonic
// ("VADDPD.RN_SAE.Z" → base "VADDPD", rounding RN, zeroing), validating the
// combinations the Go assembler allows: .Z last, no duplicates, no
// broadcast together with rounding/SAE.
func parseEvexSuffix(mnem string) (string, evexSuffix, error) {
sfx := evexSuffix{rounding: -1}
i := strings.IndexByte(mnem, '.')
if i < 0 {
return mnem, sfx, nil
}
base := mnem[:i]
parts := strings.Split(mnem[i+1:], ".")
seen := map[string]bool{}
for j, p := range parts {
if seen[p] {
return "", sfx, fmt.Errorf("duplicate EVEX suffix %q", p)
}
seen[p] = true
switch p {
case "Z":
if j != len(parts)-1 {
return "", sfx, fmt.Errorf("the .Z suffix must come last in %q", mnem[i+1:])
}
sfx.zeroing = true
case "SAE":
sfx.sae = true
case "BCST":
sfx.bcst = true
case "RN_SAE":
sfx.rounding = 0
case "RD_SAE":
sfx.rounding = 1
case "RU_SAE":
sfx.rounding = 2
case "RZ_SAE":
sfx.rounding = 3
default:
return "", sfx, fmt.Errorf("unsupported EVEX suffix %q", p)
}
}
if sfx.bcst && (sfx.sae || sfx.rounding >= 0) {
return "", sfx, fmt.Errorf("cannot combine .BCST with rounding or SAE in %q", mnem[i+1:])
}
return base, sfx, nil
}
// evexRound lists the instructions that accept a rounding mode or .SAE.
var evexRound = map[string]bool{
"VADDPD": true, "VSUBPD": true, "VMULPD": true, "VDIVPD": true,
"VMINPD": true, "VMAXPD": true,
"VADDPS": true, "VSUBPS": true, "VMULPS": true, "VDIVPS": true,
"VMINPS": true, "VMAXPS": true,
"VADDSD": true, "VSUBSD": true, "VMULSD": true, "VDIVSD": true,
"VMINSD": true, "VMAXSD": true,
"VADDSS": true, "VSUBSS": true, "VMULSS": true, "VDIVSS": true,
"VMINSS": true, "VMAXSS": true,
"VSCALEFPD": true, "VSCALEFPS": true, "VSCALEFSD": true, "VSCALEFSS": true,
"VGETEXPPD": true, "VGETEXPPS": true, "VGETEXPSD": true, "VGETEXPSS": true,
"VCVTDQ2PS": true, "VCVTPS2QQ": true, "VCVTQQ2PS": true, "VCVTPD2UQQ": true,
"VCVTPD2PS": true, "VCVTPD2UDQ": true, "VCVTTPD2UDQ": true, "VCVTTPD2UQQ": true,
"VCVTPS2UDQ": true, "VCVTTPS2UDQ": true, "VCVTPS2UQQ": true, "VCVTTPS2UQQ": true,
"VCVTTPD2QQ": true, "VCVTTPS2QQ": true, "VCVTUQQ2PD": true, "VCVTUQQ2PS": true,
"VCVTSD2SI": true, "VCVTSD2SIQ": true, "VCVTSS2SI": true, "VCVTSS2SIQ": true,
"VCVTSD2USIL": true, "VCVTSD2USIQ": true, "VCVTSS2USIL": true, "VCVTSS2USIQ": true,
"VCVTTSD2SI": true, "VCVTTSD2SIQ": true, "VCVTTSS2SI": true, "VCVTTSS2SIQ": true,
"VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true,
"VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true,
"VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true,
}
// evexBcstN maps an instruction accepting .BCST to the broadcast element
// size — the disp8×N multiplier for its memory operand.
var evexBcstN = map[string]int{
"VADDPD": 8, "VSUBPD": 8, "VMULPD": 8, "VDIVPD": 8,
"VMINPD": 8, "VMAXPD": 8,
"VADDPS": 4, "VSUBPS": 4, "VMULPS": 4, "VDIVPS": 4,
"VMINPS": 4, "VMAXPS": 4,
"VRCP14PD": 8, "VRCP14PS": 4, "VRSQRT14PD": 8, "VRSQRT14PS": 4,
"VGETEXPPD": 8, "VGETEXPPS": 4,
"VSCALEFPD": 8, "VSCALEFPS": 4,
"VRNDSCALEPD": 8, "VRNDSCALEPS": 4,
"VGETMANTPD": 8, "VGETMANTPS": 4,
"VREDUCEPD": 8, "VREDUCEPS": 4,
"VFIXUPIMMPD": 8, "VFIXUPIMMPS": 4,
"VRANGEPD": 8, "VRANGEPS": 4,
"VCVTDQ2PS": 4, "VCVTPS2QQ": 4, "VCVTQQ2PS": 8,
"VCVTUDQ2PD": 4, "VCVTUDQ2PS": 4,
"VCVTPD2PS": 8, "VCVTPD2UDQ": 8, "VCVTTPD2UDQ": 8, "VCVTTPD2UQQ": 8,
"VCVTPS2UDQ": 4, "VCVTTPS2UDQ": 4, "VCVTPS2UQQ": 4, "VCVTTPS2UQQ": 4,
"VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8,
}
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
// returning the remaining operands and the mask index. K0 is not a usable
// mask (aaa = 0 means "no mask"), matching the assembler.
func splitMask(ops []Operand) ([]Operand, int, error) {
var rest []Operand
mask := 0
for _, op := range ops {
if r, ok := op.(Reg); ok && r.mask {
if mask != 0 {
return nil, 0, fmt.Errorf("at most one mask register operand")
}
if r.idx == 0 {
return nil, 0, fmt.Errorf("K0 is not a usable mask register")
}
mask = r.idx
continue
}
rest = append(rest, op)
}
return rest, mask, nil
}
// encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The
// mask, when present, is an explicit K1–K7 operand anywhere among the
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
// broadcast.
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
// The mask/vector conversions take the K register as a genuine operand
// (source or destination), not as a mask, and accept no suffixes.
if evexKOperand[mnemUpper] {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
}
spec, ok := evexTable[mnemUpper]
if !ok {
return fmt.Errorf("unsupported instruction %q", mnemUpper)
}
return e.encodeEvexRM(spec, ops, 0, sfx)
}
spec, inTable := evexTable[mnemUpper]
if inTable {
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper)
}
if sfx.bcst {
n, ok := evexBcstN[mnemUpper]
if !ok {
return fmt.Errorf("%s: broadcast is not supported for this instruction", mnemUpper)
}
spec.n = [3]int{n, n, n}
}
} else if sfx.evexOnly() {
return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper)
}
// Mask-destination comparisons (VPCMPEQD, VCMPPD $imm, …): the last
// operand is the destination K register, and any mask sits among the
// preceding operands.
kdst := func(encode func(evexSpec, []Operand, int, evexSuffix) error) error {
dst, ok := ops[len(ops)-1].(Reg)
if !ok || !dst.mask {
return nil // not a K-destination form; fall through
}
rest, mask, err := splitMask(ops[:len(ops)-1])
if err != nil {
return err
}
if sfx.zeroing && mask == 0 {
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper)
}
return encode(spec, append(rest, dst), mask, sfx)
}
if inTable && len(ops) > 0 {
switch spec.form {
case vexNDS3:
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
return kdst(e.encodeEvexNDS3)
}
case vexNDS3Imm:
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
return kdst(e.encodeEvexNDS3Imm)
}
case vexImmRM:
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
return kdst(e.encodeEvexImmRM)
}
}
}
rest, mask, err := splitMask(ops)
if err != nil {
return err
}
if sfx.zeroing && mask == 0 {
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper)
}
ops = rest
if bs, ok := evexBcastTable[mnemUpper]; ok {
if sfx.evexOnly() {
return fmt.Errorf("%s: broadcast instructions take no rounding/SAE/broadcast suffix", mnemUpper)
}
return e.encodeEvexBcast(bs, ops, mask, sfx)
}
if ms, ok := evexMoveTable[mnemUpper]; ok {
if sfx.evexOnly() {
return fmt.Errorf("%s: moves take no rounding/SAE/broadcast suffix", mnemUpper)
}
return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx)
}
if !inTable {
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
}
switch spec.form {
case vexNDS3:
return e.encodeEvexNDS3(spec, ops, mask, sfx)
case vexRM:
return e.encodeEvexRM(spec, ops, mask, sfx)
case vexRMRev:
return e.encodeEvexRMRev(spec, ops, mask, sfx)
case vexImmRM:
return e.encodeEvexImmRM(spec, ops, mask, sfx)
case vexShiftImm:
return e.encodeEvexShiftImm(spec, ops, mask, sfx)
case vexNDS3Imm:
return e.encodeEvexNDS3Imm(spec, ops, mask, sfx)
case vexExtract:
return e.encodeEvexExtract(spec, ops, mask, sfx)
case vexRMSrcLen:
return e.encodeEvexRMSrcLen(spec, ops, mask, sfx)
}
return fmt.Errorf("unhandled EVEX form for %s", mnemUpper)
}
// encodeEvexNDS3 encodes the three-operand NDS form: OP src2, src1, dst. The
// destination may be an opmask register (VPCMPEQD), in which case the vector
// length comes from the sources.
func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 3 {
return fmt.Errorf("EVEX NDS instruction expects 3 operands, got %d", len(ops))
}
src2, src1, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || (!dstReg.isVec() && !dstReg.mask) {
return fmt.Errorf("EVEX destination must be a vector or mask register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("EVEX vvvv operand must be a vector register")
}
ll := dstReg.vecLenBit()
if dstReg.mask {
ll = vvvvReg.vecLenBit()
if r, ok := src2.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
}
}
return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, sfx)
}
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
// no vvvv), e.g. VCVTQQ2PD. The destination may be an opmask register (the
// *2M mask conversions) or a general-purpose register (the scalar
// vector-to-GPR conversions); in both cases the vector length comes from
// the source.
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 2 {
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok {
return fmt.Errorf("EVEX destination must be a register")
}
ll := dstReg.vecLenBit()
if !dstReg.isVec() {
// Mask or GPR destination: the length follows the vector source
// (128 for a memory source).
ll = 0
if r, ok := src.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
}
}
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
}
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
// (reg = dst, rm = src, imm8), e.g. VPSHUFD. The destination may be an
// opmask register (VFPCLASS*), in which case the vector length comes from
// the source.
func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 3 {
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shuffle control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || (!dstReg.isVec() && !dstReg.mask) {
return fmt.Errorf("shuffle destination must be a vector or mask register")
}
ll := dstReg.vecLenBit()
if dstReg.mask {
if r, ok := src.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
}
} else if r, ok := src.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexShiftImm encodes an immediate shift: OP $imm, src, dst
// (ModRM.reg = /digit, vvvv = dst, rm = src, imm8), e.g. VPSRAD $31, Z3, Z5.
func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 3 {
return fmt.Errorf("EVEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shift count must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("shift source must be a vector register")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("shift destination must be a vector register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexNDS3Imm encodes OP $imm, src2, src1, dst (reg=dst, vvvv=src1,
// rm=src2, imm8), e.g. VALIGND. The destination may be an opmask register
// (VCMPPD and friends), in which case the vector length comes from the
// sources.
func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 4 {
return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops))
}
imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shuffle control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || (!dstReg.isVec() && !dstReg.mask) {
return fmt.Errorf("destination must be a vector or mask register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("second source must be a vector register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
ll := dstReg.vecLenBit()
if dstReg.mask {
ll = vvvvReg.vecLenBit()
if r, ok := src2.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
}
}
if err := e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexExtract encodes OP $imm, zsrc, ydst (reg=ZMM source, rm=YMM/memory
// destination, imm8), e.g. VEXTRACTI64X4.
func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, zsrc, ydst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("extract source must be a vector register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
// the store-form opcode (reg = source, rm = destination), matching the Go
// assembler.
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 2 {
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcIsVec := vecReg(src)
dstReg, dstIsVec := vecReg(dst)
op := ms.store
var reg Reg
var rm Operand
switch {
case srcIsVec && dstIsVec:
reg, rm = srcReg, dst
case srcIsVec:
if !memOperand(dst) {
return fmt.Errorf("%s: invalid destination operand", mnem)
}
reg, rm = srcReg, dst
case dstIsVec:
if !memOperand(src) {
return fmt.Errorf("%s: invalid source operand", mnem)
}
op = ms.load
reg, rm = dstReg, src
default:
return fmt.Errorf("%s needs a vector register operand", mnem)
}
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx)
}
// encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the length fixed by the mnemonic — the
// single valid slot of spec.n names the vector length (and the disp8×N
// multiplier) a register or memory source encodes.
func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 2 {
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("EVEX destination must be a vector register")
}
ll, err := soleLen(spec.n)
if err != nil {
return err
}
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
}
// soleLen returns the vector-length index of the single valid slot of n —
// the length a length-fixed mnemonic (the EVEX conversion spellings) encodes
// regardless of its operands.
func soleLen(n [3]int) (int, error) {
ll := -1
for i, v := range n {
if v == 0 {
continue
}
if ll >= 0 {
return 0, fmt.Errorf("ambiguous vector-length table %v", n)
}
ll = i
}
if ll < 0 {
return 0, fmt.Errorf("empty vector-length table")
}
return ll, nil
}
// memOperand reports whether op is a memory reference (including a
// static-symbol reference).
func memOperand(op Operand) bool {
switch op.(type) {
case Mem, sbMem:
return true
}
return false
}
// encodeEvexRMRev encodes the narrowing-store form: OP src, dst with the wide
// source in the reg field and the narrow destination in r/m (VPMOVDW/QD).
func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 2 {
return fmt.Errorf("EVEX store instruction expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("EVEX source must be a vector register")
}
return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, sfx)
}
// encodeEvexBcast encodes VPBROADCASTD/Q: OP src, dst with the GPR or memory
// source broadcast to every lane of the vector destination.
func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 2 {
return fmt.Errorf("broadcast expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("broadcast destination must be a vector register")
}
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
switch src.(type) {
case Mem, sbMem:
spec.opcode = bs.opMem
spec.n = [3]int{bs.n, bs.n, bs.n}
case Reg:
spec.opcode = bs.opReg
default:
return fmt.Errorf("broadcast source must be a register or memory")
}
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, sfx)
}
// emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement
// (disp8×N compressed) for the given precomputed fields. regIdx is the
// unextended reg-field register index, or a /digit (0–7); vvvvIdx is the
// vvvv register index, or -1 when unused. mask (K1–K7, 0 = unmasked) and
// zeroing fill the aaa and z bits of the P2 byte.
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, sfx evexSuffix) error {
if ll > 2 {
return fmt.Errorf("invalid vector length")
}
// reg-field extension bits (R̄, R'̄), inverted.
rBar, rPrimeBar := 1, 1
if regIdx&8 != 0 {
rBar = 0
}
if regIdx&16 != 0 {
rPrimeBar = 0
}
// vvvv (inverted) and its extension bit V'̄.
vBar, vPrimeBar := 15, 1
if vvvvIdx >= 0 {
vBar = 15 - (vvvvIdx & 15)
if vvvvIdx&16 != 0 {
vPrimeBar = 0
}
}
var modrm, sib int
var disp []byte
xBar, bBar := 1, 1
var sb *sbRef
switch r := rm.(type) {
case Reg:
// ModRM.mod = 11: rm[3] extends via B̄, and rm[4] via X̄ (the EVEX
// register-register quirk).
modrm = 0xC0 | (regIdx&7)<<3 | (r.idx & 7)
sib = -1
if r.idx&8 != 0 {
bBar = 0
}
if r.idx&16 != 0 {
xBar = 0
}
if r.idx&16 != 0 {
xBar = 0
}
case Mem:
var err error
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
if err != nil {
return err
}
// An indexed memory operand carries index[4] in V'̄ (Go folds it
// together with vvvv[4] into the same bit).
if r.HasIndex && r.Index.idx&16 != 0 {
vPrimeBar = 0
}
case sbMem:
// RIP-relative static-symbol reference; disp32 patched at link time
// (no disp8 scaling for RIP-relative addressing).
modrm = (regIdx&7)<<3 | 0x05
sib = -1
disp = le32(0)
sb = &sbRef{name: r.name, addend: r.addend}
default:
return fmt.Errorf("invalid EVEX r/m operand")
}
z := 0
if sfx.zeroing {
z = 1
}
// The b bit and the L'L field carry the rounding/SAE/broadcast mode:
// a rounding mode replaces L'L with the rc value, plain SAE and
// broadcast keep the vector length.
b, ll := 0, ll
switch {
case sfx.rounding >= 0:
b, ll = 1, sfx.rounding
case sfx.sae || sfx.bcst:
b = 1
}
p0 := byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4 | spec.mapSel)
p1 := byte(spec.w<<7 | vBar<<3 | 1<<2 | spec.pp)
p2 := byte(z<<7 | ll<<5 | b<<4 | vPrimeBar<<3 | mask) // z, L'L/rc, b, V', aaa
e.out = append(e.out, 0x62, p0, p1, p2, spec.opcode, byte(modrm))
if sib >= 0 {
e.out = append(e.out, byte(sib))
}
if sb != nil {
e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend})
}
e.out = append(e.out, disp...)
return nil
}
// memComponentsEvex computes the ModR/M byte (with the given reg field), the
// SIB byte (-1 if none), the displacement bytes and the (inverted sense)
// index/base extension bits for an EVEX memory operand. The displacement is
// compressed to disp8×N when it is a multiple of n and the quotient fits a
// signed byte; otherwise a full disp32 is used.
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
sib = -1
xBar, bBar = 1, 1 // inverted bits: 1 = no extension
if !m.HasBase && !m.HasIndex {
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
}
needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4)
var mod int
switch {
case !m.HasBase:
mod = 0
disp = le32(m.Disp)
case m.Base.idx&7 == 5 && m.Disp == 0:
mod = 1
disp = []byte{0}
case m.Disp == 0:
mod = 0
case n > 0 && m.Disp%int64(n) == 0 && m.Disp/int64(n) >= -128 && m.Disp/int64(n) <= 127:
mod = 1
disp = []byte{byte(int8(m.Disp / int64(n)))}
default:
mod = 2
disp = le32(m.Disp)
}
if needSIB {
idxField := 4 // 100 = no index
if m.HasIndex {
idxField = m.Index.idx & 7
if m.Index.idx&8 != 0 {
xBar = 0
}
}
baseField := 5 // 101 = no base (with mod=00 → disp32)
if m.HasBase {
baseField = m.Base.idx & 7
if m.Base.idx&8 != 0 {
bBar = 0
}
}
return mod<<6 | regField<<3 | 0x04, scaleBits(m.Scale)<<6 | idxField<<3 | baseField, disp, xBar, bBar, nil
}
if m.Base.idx&8 != 0 {
bBar = 0
}
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil
}
// gatherSpec describes a gather/scatter family member: all live in
// 66.0F38; the opcode and W select the index and data element widths, and n
// is the data element size (the EVEX disp8×N multiplier).
type gatherSpec struct {
opcode byte
w int
n int
}
var gatherTable = map[string]gatherSpec{
"VGATHERDPS": {0x92, 0, 4},
"VGATHERDPD": {0x92, 1, 8},
"VGATHERQPS": {0x93, 0, 4},
"VGATHERQPD": {0x93, 1, 8},
"VPGATHERDD": {0x90, 0, 4},
"VPGATHERDQ": {0x90, 1, 8},
"VPGATHERQD": {0x91, 0, 4},
"VPGATHERQQ": {0x91, 1, 8},
}
var scatterTable = map[string]gatherSpec{
"VSCATTERDPS": {0xA2, 0, 4},
"VSCATTERDPD": {0xA2, 1, 8},
"VSCATTERQPS": {0xA3, 0, 4},
"VSCATTERQPD": {0xA3, 1, 8},
"VPSCATTERDD": {0xA0, 0, 4},
"VPSCATTERDQ": {0xA0, 1, 8},
"VPSCATTERQD": {0xA1, 0, 4},
"VPSCATTERQQ": {0xA1, 1, 8},
}
// isGather reports whether the mnemonic is a gather instruction.
func isGather(upper string) bool {
_, ok := gatherTable[upper]
return ok
}
// isScatter reports whether the mnemonic is a scatter instruction.
func isScatter(upper string) bool {
_, ok := scatterTable[upper]
return ok
}
// vsibLen validates a VSIB memory operand (the index must be a vector
// register) and returns it with the vector length the index selects — the
// EVEX L'L field follows the index register, not the data register.
func vsibLen(op Operand, what string) (Mem, int, error) {
m, ok := op.(Mem)
if !ok || !m.HasIndex || !m.Index.isVec() {
return Mem{}, 0, fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", what)
}
return m, m.Index.vecLenBit(), nil
}
// encodeGather encodes a gather. The VEX spelling carries the mask in a
// vector register (OP mask, vsib, dst: vvvv = mask, rm = vsib, reg = dst,
// L follows the data register); the EVEX spelling carries it in aaa (OP
// vsib, K, dst: rm = vsib, reg = dst, L follows the VSIB index).
func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexSuffix) error {
rest, mask, err := splitMask(ops)
if err != nil {
return err
}
if mask != 0 || sfx.any() {
// EVEX form: OP vsib, K, dst.
if len(rest) != 2 {
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
}
vsib, ll, err := vsibLen(rest[0], upper)
if err != nil {
return err
}
dst, ok := rest[1].(Reg)
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", upper)
}
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
}
// VEX form: OP mask, vsib, dst.
if len(rest) != 3 {
return fmt.Errorf("%s expects 3 operands (mask, vsib, dst), got %d", upper, len(rest))
}
maskReg, ok := rest[0].(Reg)
if !ok || !maskReg.isVec() {
return fmt.Errorf("%s: mask must be a vector register", upper)
}
vsib, _, err := vsibLen(rest[1], upper)
if err != nil {
return err
}
dst, ok := rest[2].(Reg)
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", upper)
}
spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1}
rBit := 0
if dst.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib)
}
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib — reg = src,
// rm = the VSIB memory operand, the K mask in aaa and L following the VSIB
// index.
func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evexSuffix) error {
rest, mask, err := splitMask(ops)
if err != nil {
return err
}
if mask == 0 {
return fmt.Errorf("%s requires a K mask register", upper)
}
if len(rest) != 2 {
return fmt.Errorf("%s expects 3 operands (src, K, vsib), got %d", upper, len(ops))
}
src, ok := rest[0].(Reg)
if !ok || !src.isVec() {
return fmt.Errorf("%s: source must be a vector register", upper)
}
vsib, ll, err := vsibLen(rest[1], upper)
if err != nil {
return err
}
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
}
// evexKOperand lists the instructions whose K register is a genuine operand
// (the source or destination of a mask/vector conversion) rather than a
// mask modifier — the M2 and 2M conversions. They take no masking.
var evexKOperand = map[string]bool{
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
}
// kmovSpec describes a KMOV width: the opcode depends on the operand
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory
// prefix and W for the wider widths.
type kmovSpec struct {
kk, kmem, gprk, kgpr byte
gprPP int
w int
}
var kmovTable = map[string]kmovSpec{
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0},
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 3, 1},
}
// encodeKmov encodes a KMOV width, selecting the opcode by direction.
func (e *enc) encodeKmov(upper string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
ks := kmovTable[upper]
src, dst := ops[0], ops[1]
srcReg, srcIsReg := src.(Reg)
dstReg, dstIsReg := dst.(Reg)
srcK := srcIsReg && srcReg.mask
dstK := dstIsReg && dstReg.mask
spec := vexSpec{mapSel: 1, w: ks.w, pp: 0, opdigit: -1}
switch {
case srcK && dstK:
spec.opcode = ks.kk // k ← k: reg = dst, rm = src
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
case srcK && dstIsReg:
spec.opcode = ks.kgpr // GPR ← k: reg = dst, rm = src
spec.pp = ks.gprPP
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15, src)
case srcK:
if _, ok := dst.(Mem); !ok {
return fmt.Errorf("%s: invalid destination operand", upper)
}
spec.opcode = ks.kmem // mem ← k: reg = src, rm = dst
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
case dstK:
spec.opcode = ks.gprk // k ← GPR/mem: reg = dst, rm = src
spec.pp = ks.gprPP
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
}
return fmt.Errorf("%s requires a K register operand", upper)
}
// kOpSpec describes the VEX encoding of an opmask-register instruction: the
// L bit and the W/pp pair select the operand width, and the form the
// operand layout.
type kOpSpec struct {
mapSel int
opcode byte
w int
pp int
ll int
form vexForm
}
var kOpsTable = map[string]kOpSpec{
// k ← k OP k: reg = dst, vvvv = src1, rm = src2 (three opmask
// registers). Byte/word widths share W0 and differ by the 66 prefix;
// dword/qword share the W bit selection the Go assembler emits.
"KANDB": {1, 0x41, 0, 1, 1, vexNDS3},
"KANDW": {1, 0x41, 0, 0, 1, vexNDS3},
"KANDD": {1, 0x41, 1, 1, 1, vexNDS3},
"KANDQ": {1, 0x41, 1, 0, 1, vexNDS3},
"KANDNB": {1, 0x42, 0, 1, 1, vexNDS3},
"KANDNW": {1, 0x42, 0, 0, 1, vexNDS3},
"KANDND": {1, 0x42, 1, 1, 1, vexNDS3},
"KANDNQ": {1, 0x42, 1, 0, 1, vexNDS3},
"KORB": {1, 0x45, 0, 1, 1, vexNDS3},
"KORW": {1, 0x45, 0, 0, 1, vexNDS3},
"KORD": {1, 0x45, 1, 1, 1, vexNDS3},
"KORQ": {1, 0x45, 1, 0, 1, vexNDS3},
"KXNORB": {1, 0x46, 0, 1, 1, vexNDS3},
"KXNORW": {1, 0x46, 0, 0, 1, vexNDS3},
"KXNORD": {1, 0x46, 1, 1, 1, vexNDS3},
"KXNORQ": {1, 0x46, 1, 0, 1, vexNDS3},
"KXORB": {1, 0x47, 0, 1, 1, vexNDS3},
"KXORW": {1, 0x47, 0, 0, 1, vexNDS3},
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
"KADDD": {1, 0x4A, 1, 1, 1, vexNDS3},
"KADDQ": {1, 0x4A, 1, 0, 1, vexNDS3},
// k ← OP k (KNOT), k ← k AND~ k (KTEST-style RM) and flags ← k OP k
// (KORTEST): reg = dst, rm = src.
"KNOTB": {1, 0x44, 0, 1, 0, vexRM},
"KNOTW": {1, 0x44, 0, 0, 0, vexRM},
"KNOTD": {1, 0x44, 1, 1, 0, vexRM},
"KNOTQ": {1, 0x44, 1, 0, 0, vexRM},
"KORTESTB": {1, 0x98, 0, 1, 0, vexRM},
"KORTESTW": {1, 0x98, 0, 0, 0, vexRM},
"KORTESTD": {1, 0x98, 1, 1, 0, vexRM},
"KORTESTQ": {1, 0x98, 1, 0, 0, vexRM},
"KTESTB": {1, 0x99, 0, 1, 0, vexRM},
"KTESTW": {1, 0x99, 0, 0, 0, vexRM},
"KTESTD": {1, 0x99, 1, 1, 0, vexRM},
"KTESTQ": {1, 0x99, 1, 0, 0, vexRM},
// OP $imm, src, dst: reg = dst, rm = src, imm8. The opcodes split by
// direction (0x32/0x33 left, 0x30/0x31 right) and within each by
// element half (0x32 byte/word, 0x33 dword/qword); W picks byte/dword
// (W0) against word/qword (W1).
"KSHIFTLB": {3, 0x32, 0, 1, 0, vexImmRM},
"KSHIFTLW": {3, 0x32, 1, 1, 0, vexImmRM},
"KSHIFTLD": {3, 0x33, 0, 1, 0, vexImmRM},
"KSHIFTLQ": {3, 0x33, 1, 1, 0, vexImmRM},
"KSHIFTRB": {3, 0x30, 0, 1, 0, vexImmRM},
"KSHIFTRW": {3, 0x30, 1, 1, 0, vexImmRM},
"KSHIFTRD": {3, 0x31, 0, 1, 0, vexImmRM},
"KSHIFTRQ": {3, 0x31, 1, 1, 0, vexImmRM},
}
// isKOp reports whether the mnemonic is an opmask-register instruction.
func isKOp(upper string) bool {
_, ok := kOpsTable[upper]
return ok
}
// encodeKOp encodes an opmask-register instruction; every operand is a K
// register and the vector length is fixed by the instruction.
func (e *enc) encodeKOp(upper string, ops []Operand) error {
ks := kOpsTable[upper]
spec := vexSpec{mapSel: ks.mapSel, opcode: ks.opcode, w: ks.w, pp: ks.pp, opdigit: -1}
kreg := func(op Operand, what string) (Reg, error) {
r, ok := op.(Reg)
if !ok || !r.mask {
return Reg{}, fmt.Errorf("%s: %s must be an opmask register", upper, what)
}
return r, nil
}
switch ks.form {
case vexNDS3:
if len(ops) != 3 {
return fmt.Errorf("%s expects 3 operands, got %d", upper, len(ops))
}
src2, err := kreg(ops[0], "first source")
if err != nil {
return err
}
src1, err := kreg(ops[1], "second source")
if err != nil {
return err
}
dst, err := kreg(ops[2], "destination")
if err != nil {
return err
}
return e.emitVexFields(spec, ks.ll, dst.idx, 0, 15-src1.idx, src2)
case vexRM:
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
src, err := kreg(ops[0], "source")
if err != nil {
return err
}
dst, err := kreg(ops[1], "destination")
if err != nil {
return err
}
return e.emitVexFields(spec, ks.ll, dst.idx, 0, 15, src)
case vexImmRM:
if len(ops) != 3 {
return fmt.Errorf("%s expects 3 operands ($imm, src, dst), got %d", upper, len(ops))
}
immVal, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("%s: shift count must be an immediate", upper)
}
src, err := kreg(ops[1], "source")
if err != nil {
return err
}
dst, err := kreg(ops[2], "destination")
if err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitVexFields(spec, ks.ll, dst.idx, 0, 15, src); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
return fmt.Errorf("unhandled opmask form for %s", upper)
}