1620 lines
63 KiB
Go
1620 lines
63 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||
// SPDX-License-Identifier: BSD-3-Clause
|
||
|
||
package asm
|
||
|
||
import (
|
||
"fmt"
|
||
"strings"
|
||
)
|
||
|
||
// This file implements EVEX (AVX-512) instruction encoding: the four-byte
|
||
// EVEX prefix with 5-bit vector register fields (Z0-Z31, X/Y 16-31), the
|
||
// compressed disp8×N displacement, and the operand shapes the go-flac
|
||
// AVX-512 kernels use plus the common floating-point and conversion set.
|
||
// Masking follows the Go assembler's spelling: an explicit K1-K7 operand
|
||
// anywhere among the operands (merging) plus a ".Z" mnemonic suffix for
|
||
// zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are
|
||
// supported too.
|
||
|
||
// evexSpec describes one EVEX instruction's encoding parameters. The form
|
||
// field reuses the vexForm shapes, which carry over unchanged.
|
||
type evexSpec struct {
|
||
mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A
|
||
opcode byte
|
||
w int
|
||
pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2
|
||
opdigit int // ModRM.reg /digit, or -1 when reg is a register
|
||
form vexForm // vexNDS3, vexRM, vexShiftImm, vexNDS3Imm, vexExtract
|
||
n [3]int // disp8×N multiplier per vector length (128/256/512)
|
||
}
|
||
|
||
// evexTable maps an upper-case mnemonic to its EVEX encoding. Mnemonics
|
||
// that also have a VEX form (VPADDD, VMOVUPD, …) are dispatched here only
|
||
// when an operand demands EVEX (a ZMM or K register); EVEX-only mnemonics
|
||
// (VPXORD, VALIGND, …) always encode through this table. The N multipliers
|
||
// are taken from the Go assembler's opcode tables, which are authoritative
|
||
// for byte-for-byte agreement.
|
||
var evexTable = map[string]evexSpec{
|
||
// EVEX.128/256/512.66.0F, integer arithmetic / logic, NDS form.
|
||
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSUBQ": {1, 0xFB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPXORD": {1, 0xEF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPXORQ": {1, 0xEF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.128/256/512.66.0F.W1, packed double arithmetic.
|
||
"VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.0F.W0, packed single arithmetic.
|
||
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.66.0F.W1, packed double unpack.
|
||
"VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.128.F2.0F.W1, scalar double arithmetic (the packed opcodes with
|
||
// an F2 pp; the EVEX forms exist for masked and zeroing use). The
|
||
// memory operand is a single double, so disp8×N = 8.
|
||
"VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VSUBSD": {1, 0x5C, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VMULSD": {1, 0x59, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VDIVSD": {1, 0x5E, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
|
||
// EVEX.128.F3.0F.W0, scalar single arithmetic (disp8×N = 4).
|
||
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
|
||
// EVEX.512.66.0F3A, align (NDS + imm8).
|
||
"VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.128/256/512.66.0F, immediate shift (VPSRAD /4).
|
||
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
|
||
// the W bit distinguishes it from VPSRAD's E2 form).
|
||
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
|
||
// rm=src, no vvvv).
|
||
"VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.F2.0F.W1, duplicate the low double (reg=dst,
|
||
// rm=src, no vvvv): a 128-bit destination reads a single double from
|
||
// memory (disp8×8), the wider ones read the full operand.
|
||
"VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}},
|
||
// EVEX.128/256/512.0F.W0, signed dword to packed single (reg=dst,
|
||
// rm=src, no vvvv, no mandatory prefix, as in the VEX form).
|
||
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.0F.W0, packed single to packed double: the
|
||
// destination is twice the source width and sets the length; disp8×N
|
||
// follows the narrow memory source. No F3 prefix: the Go assembler
|
||
// emits this instruction with pp = 00 (Intel's maps would call that
|
||
// undefined) and gasm reproduces the Go assembler's bytes, its machine
|
||
// code is the oracle, not the manual.
|
||
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
|
||
// EVEX.128/256/512.F3.0F.W0, signed dword to packed double (the EVEX
|
||
// form of the VEX instruction; the destination sets the length, disp8×N
|
||
// follows the narrow memory source).
|
||
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
|
||
// EVEX packed double → dword conversions: the source is the wide
|
||
// operand and the mnemonic fixes the length, the bare names are
|
||
// 512-bit only (ZMM source, XMM destination), the X/Y spellings are
|
||
// EVEX-128/256. Exactly one slot of n is valid; it names the vector
|
||
// length (and the disp8×N multiplier) a register or memory source
|
||
// encodes.
|
||
"VCVTPD2DQ": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||
"VCVTTPD2DQ": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||
"VCVTPD2DQX": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||
"VCVTPD2DQY": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||
"VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||
"VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||
|
||
// EVEX.66.0F3A, ternary logic and lane shuffles (NDS + imm8).
|
||
"VPTERNLOGD": {3, 0x25, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VPTERNLOGQ": {3, 0x25, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VSHUFI32X4": {3, 0x43, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VSHUFI64X2": {3, 0x43, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VSHUFF32X4": {3, 0x23, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VSHUFF64X2": {3, 0x23, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
|
||
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
|
||
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
|
||
"VINSERTF32X8": {3, 0x1A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
|
||
"VINSERTF64X2": {3, 0x18, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
|
||
"VINSERTF64X4": {3, 0x1A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
|
||
"VINSERTI32X4": {3, 0x38, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
|
||
"VINSERTI32X8": {3, 0x3A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
|
||
"VINSERTI64X2": {3, 0x38, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
|
||
"VINSERTI64X4": {3, 0x3A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
|
||
|
||
// EVEX.66.0F3A, lane extract (reg=source, rm=XMM/YMM destination,
|
||
// imm8).
|
||
"VEXTRACTF32X4": {3, 0x19, 0, 1, -1, vexExtract, [3]int{0, 16, 16}},
|
||
"VEXTRACTF32X8": {3, 0x1B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
|
||
"VEXTRACTF64X2": {3, 0x19, 1, 1, -1, vexExtract, [3]int{0, 16, 16}},
|
||
"VEXTRACTI32X4": {3, 0x39, 0, 1, -1, vexExtract, [3]int{0, 16, 16}},
|
||
"VEXTRACTI32X8": {3, 0x3B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
|
||
"VEXTRACTI64X2": {3, 0x39, 1, 1, -1, vexExtract, [3]int{0, 16, 16}},
|
||
|
||
// EVEX.66.0F, compare with an opmask destination ($imm, src2, src1,
|
||
// kdst): NDS3Imm with the K register in the reg field.
|
||
"VCMPPD": {1, 0xC2, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VCMPPS": {1, 0xC2, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||
"VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||
|
||
// EVEX.66.0F3A, integer compares with an opmask destination, the same
|
||
// NDS3Imm-with-k-reg shape as the floating-point compares; W selects the
|
||
// operand width (byte/word vs dword/qword), the opcode the signedness.
|
||
// The memory form takes a full vector, so disp8×N is 16/32/64.
|
||
"VPCMPB": {3, 0x3F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VPCMPUB": {3, 0x3E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VPCMPW": {3, 0x3F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VPCMPUW": {3, 0x3E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VPCMPD": {3, 0x1F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VPCMPUD": {3, 0x1E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VPCMPQ": {3, 0x1F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.66.0F38, permutes (NDS form).
|
||
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.66.0F, the wider integer set (NDS form).
|
||
"VPMADDWD": {1, 0xF5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSRLVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPACKUSDW": {2, 0x2B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.66.0F38, absolute values and replicating moves (reg=dst,
|
||
// rm=src).
|
||
"VPABSB": {2, 0x1C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VPABSW": {2, 0x1D, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VPABSD": {2, 0x1E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VPABSQ": {2, 0x1F, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
// EVEX.F3.0F, replicate even/odd singles.
|
||
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
// EVEX.66.0F38, sign/zero-extending moves; the memory source is the
|
||
// narrow half (here byte to word).
|
||
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||
// EVEX.66.0F, packed single conversions (reg=dst, rm=src).
|
||
"VCVTPS2DQ": {1, 0x5B, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VCVTTPS2DQ": {1, 0x5B, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
// EVEX.66.0F38, broadcast a single/double to all lanes (reg=dst,
|
||
// rm=scalar memory; disp8×N is the element size).
|
||
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
|
||
"VBROADCASTSD": {2, 0x19, 1, 1, -1, vexRM, [3]int{0, 8, 8}},
|
||
|
||
// EVEX.66.0F38, expand loads (rm → vector register destination).
|
||
"VEXPANDPD": {2, 0x88, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
|
||
"VEXPANDPS": {2, 0x88, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
|
||
"VPEXPANDD": {2, 0x89, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
|
||
"VPEXPANDQ": {2, 0x89, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
|
||
|
||
// EVEX.66.0F38, compress stores (vector register source → rm), and the
|
||
// remaining narrowing stores.
|
||
"VCOMPRESSPD": {2, 0x8A, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}},
|
||
"VCOMPRESSPS": {2, 0x8A, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}},
|
||
"VPCOMPRESSD": {2, 0x8B, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}},
|
||
"VPCOMPRESSQ": {2, 0x8B, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}},
|
||
"VPMOVWB": {2, 0x30, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||
"VPMOVQB": {2, 0x32, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
|
||
|
||
// EVEX.66.0F, rotates (immediate form: /0 right, /1 left).
|
||
"VPRORD": {1, 0x72, 0, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
|
||
"VPRORQ": {1, 0x72, 1, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
|
||
"VPROLD": {1, 0x72, 0, 1, 1, vexShiftImm, [3]int{16, 32, 64}},
|
||
"VPROLQ": {1, 0x72, 1, 1, 1, vexShiftImm, [3]int{16, 32, 64}},
|
||
// EVEX word shifts.
|
||
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
|
||
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
|
||
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
|
||
// EVEX W1 qword shifts.
|
||
"VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
|
||
"VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.66.0F38, floating-point helpers, packed (reg=dst, rm=src).
|
||
"VRCP14PD": {2, 0x4C, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VRCP14PS": {2, 0x4C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VRSQRT14PD": {2, 0x4E, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VRSQRT14PS": {2, 0x4E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VGETEXPPD": {2, 0x42, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VGETEXPPS": {2, 0x42, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
// EVEX.66.0F38, floating-point helpers, scalar (NDS form: src2 is
|
||
// rm, src1 is vvvv, the XMM destination is reg). Like the scalar 0F3A
|
||
// forms, these take the 66 prefix; W selects double/single.
|
||
"VRCP14SD": {2, 0x4D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VRCP14SS": {2, 0x4D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VRSQRT14SD": {2, 0x4F, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VRSQRT14SS": {2, 0x4F, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VGETEXPSD": {2, 0x43, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VGETEXPSS": {2, 0x43, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
// EVEX.66.0F38, scale by a power of two (NDS form).
|
||
"VSCALEFPD": {2, 0x2C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VSCALEFPS": {2, 0x2C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VSCALEFSD": {2, 0x2D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VSCALEFSS": {2, 0x2D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
|
||
// EVEX.66.0F3A, packed round/getmant/reduce ($imm, src, dst: reg=dst,
|
||
// rm=src, imm8).
|
||
"VRNDSCALEPD": {3, 0x09, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||
"VRNDSCALEPS": {3, 0x08, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||
"VGETMANTPD": {3, 0x26, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||
"VGETMANTPS": {3, 0x26, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||
"VREDUCEPD": {3, 0x56, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||
"VREDUCEPS": {3, 0x56, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||
// EVEX.66.0F3A, scalar round/getmant/reduce and fixup/range (NDS +
|
||
// imm8: $imm, src2, src1, dst). The scalar 0F3A forms all take the 66
|
||
// prefix; W selects double/single.
|
||
"VRNDSCALESD": {3, 0x0B, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||
"VRNDSCALESS": {3, 0x0A, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||
"VGETMANTSD": {3, 0x27, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||
"VGETMANTSS": {3, 0x27, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||
"VREDUCESD": {3, 0x57, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||
"VREDUCESS": {3, 0x57, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||
"VFIXUPIMMPD": {3, 0x54, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VFIXUPIMMPS": {3, 0x54, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VFIXUPIMMSD": {3, 0x55, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||
"VFIXUPIMMSS": {3, 0x55, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||
"VRANGEPD": {3, 0x50, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VRANGEPS": {3, 0x50, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
"VRANGESD": {3, 0x51, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||
"VRANGESS": {3, 0x51, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||
|
||
// EVEX.66.0F3A, floating-point class test ($imm, src, kdst): the
|
||
// reg field carries the opmask destination. The packed forms carry an
|
||
// explicit length in the mnemonic (X/Y/Z).
|
||
"VFPCLASSPDX": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{16, 0, 0}},
|
||
"VFPCLASSPDY": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 32, 0}},
|
||
"VFPCLASSPDZ": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 0, 64}},
|
||
"VFPCLASSPSX": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{16, 0, 0}},
|
||
"VFPCLASSPSY": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 32, 0}},
|
||
"VFPCLASSPSZ": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 0, 64}},
|
||
"VFPCLASSSD": {3, 0x67, 1, 1, -1, vexImmRM, [3]int{8, 0, 0}},
|
||
"VFPCLASSSS": {3, 0x67, 0, 1, -1, vexImmRM, [3]int{4, 0, 0}},
|
||
|
||
// EVEX, the remaining conversions. VCVTQQ2PS narrows (the 512-bit
|
||
// source sets the length); the rest follow the destination.
|
||
"VCVTQQ2PS": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||
"VCVTPD2QQ": {1, 0x7B, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
|
||
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
|
||
// EVEX.66.0F38, half-precision convert (half-width source).
|
||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||
// EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
|
||
// rm=dst, imm8, the extract layout).
|
||
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
|
||
|
||
// EVEX, unsigned and truncating conversions. The PD sources are the
|
||
// wide operand (the bare names are 512-bit only, the X/Y spellings fix
|
||
// the length); the PS/UQQ destinations are wide and follow the
|
||
// destination.
|
||
"VCVTPD2PS": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||
"VCVTPD2PSX": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||
"VCVTPD2PSY": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||
"VCVTPD2UDQ": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||
"VCVTPD2UDQX": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||
"VCVTPD2UDQY": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||
"VCVTTPD2UDQ": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||
"VCVTTPD2UDQX": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||
"VCVTTPD2UDQY": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||
"VCVTTPD2UQQ": {1, 0x78, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VCVTPS2UDQ": {1, 0x79, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VCVTTPS2UDQ": {1, 0x78, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VCVTPS2UQQ": {1, 0x79, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||
"VCVTTPS2UQQ": {1, 0x78, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||
"VCVTTPD2QQ": {1, 0x7A, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VCVTTPS2QQ": {1, 0x7A, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||
"VCVTUQQ2PD": {1, 0x7A, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VCVTUQQ2PS": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||
"VCVTUQQ2PSX": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||
"VCVTUQQ2PSY": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||
"VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||
"VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||
|
||
// EVEX.66.0F38, the remaining sign/zero-extending moves (narrow
|
||
// source; disp8×N follows its size).
|
||
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
|
||
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
|
||
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||
|
||
// EVEX.F3.0F38, the remaining narrowing stores (vector source in reg,
|
||
// narrow destination in r/m): signed, unsigned and the D/Q truncations.
|
||
"VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||
"VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
|
||
"VPMOVSDW": {2, 0x23, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||
"VPMOVSQW": {2, 0x24, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||
"VPMOVSQD": {2, 0x25, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||
"VPMOVSWB": {2, 0x20, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||
"VPMOVUSWB": {2, 0x10, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||
"VPMOVUSDB": {2, 0x11, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||
"VPMOVUSQB": {2, 0x12, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
|
||
"VPMOVUSDW": {2, 0x13, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||
"VPMOVUSQW": {2, 0x14, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||
"VPMOVUSQD": {2, 0x15, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||
"VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||
"VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||
|
||
// EVEX.F3.0F38, mask/vector conversions: M2* moves an opmask register
|
||
// into a vector (rm = K source, reg = vector destination), *2M does the
|
||
// reverse (reg = K destination, rm = vector source, the length follows
|
||
// the vector).
|
||
"VPMOVM2B": {2, 0x28, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VPMOVM2W": {2, 0x28, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VPMOVM2D": {2, 0x38, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VPMOVM2Q": {2, 0x38, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VPMOVB2M": {2, 0x29, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VPMOVW2M": {2, 0x29, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
"VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
|
||
// EVEX, scalar conversions between vector and general-purpose
|
||
// registers. Vector to GPR (two operands: vec/mem source, GPR
|
||
// destination, vvvv unused): the signed and truncated pair, and the
|
||
// unsigned forms (EVEX only).
|
||
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||
"VCVTSD2USIL": {1, 0x79, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||
"VCVTSD2USIQ": {1, 0x79, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||
"VCVTSS2USIL": {1, 0x79, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||
"VCVTSS2USIQ": {1, 0x79, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||
"VCVTTSD2USIL": {1, 0x78, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||
"VCVTTSD2USIQ": {1, 0x78, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||
"VCVTTSS2USIL": {1, 0x78, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||
"VCVTTSS2USIQ": {1, 0x78, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
||
// vector source in vvvv, vector destination in reg).
|
||
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VCVTUSI2SDL": {1, 0x7B, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
// EVEX.128/256/512.66.0F38.W0, sign-extend dwords to qwords; the memory
|
||
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
||
// the xmm/ymm/zmm destination lengths).
|
||
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||
|
||
// EVEX.512.66.0F3A.W1, lane extract (reg=ZMM source, rm=YMM/memory
|
||
// destination, imm8).
|
||
"VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
|
||
"VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
|
||
|
||
// EVEX.66.0F38, more integer NDS forms (W distinguishes D/Q).
|
||
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
|
||
|
||
// EVEX.128/256/512, the wider integer set (AVX-512 F/BW): byte/word
|
||
// arithmetic, the bitwise ops with D/Q suffixes, min/max, averages and
|
||
// variable shifts. All NDS form; W distinguishes element size.
|
||
"VPADDB": {1, 0xFC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPADDW": {1, 0xFD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSUBB": {1, 0xF8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSUBW": {1, 0xF9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMULLW": {1, 0xD5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPAVGB": {1, 0xE0, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPAVGW": {1, 0xE3, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINUB": {1, 0xDA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXUB": {1, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINSW": {1, 0xEA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXSW": {1, 0xEE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPANDD": {1, 0xDB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPANDQ": {1, 0xDB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPANDND": {1, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPANDNQ": {1, 0xDF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINSB": {2, 0x38, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXSB": {2, 0x3C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINSQ": {2, 0x39, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXSQ": {2, 0x3D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINUW": {2, 0x3A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXUW": {2, 0x3E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINSD": {2, 0x39, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXSD": {2, 0x3D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINUD": {2, 0x3B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXUD": {2, 0x3F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINUQ": {2, 0x3B, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXUQ": {2, 0x3F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSLLVD": {2, 0x47, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSLLVQ": {2, 0x47, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSRLVD": {2, 0x45, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSRLVQ": {2, 0x45, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSRAVD": {2, 0x46, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSRAVQ": {2, 0x46, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
|
||
// EVEX forms of instructions that also exist in VEX (selected when a ZMM
|
||
// or K register, or indices 16-31, demand EVEX).
|
||
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.66.0F, immediate shift (VPSLLD /6).
|
||
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.F3.0F38.W0, narrowing stores: reg = wide source, rm = narrow
|
||
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
|
||
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||
}
|
||
|
||
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
|
||
// depends on the source kind, a GPR source uses opReg, a memory source uses
|
||
// opMem with a disp8×N of n.
|
||
type evexBcastSpec struct {
|
||
mapSel int
|
||
opReg byte
|
||
opMem byte
|
||
w int
|
||
n int
|
||
}
|
||
|
||
var evexBcastTable = map[string]evexBcastSpec{
|
||
// EVEX.128/256/512.66.0F38, broadcast a dword/qword to all lanes.
|
||
"VPBROADCASTD": {2, 0x7C, 0x58, 0, 4},
|
||
"VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8},
|
||
// EVEX.128/256/512.66.0F38, broadcast a byte/word (GPR or memory
|
||
// source) to all lanes.
|
||
"VPBROADCASTB": {2, 0x7A, 0x78, 0, 1},
|
||
"VPBROADCASTW": {2, 0x7B, 0x79, 0, 2},
|
||
}
|
||
|
||
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
|
||
// move table).
|
||
type evexMoveSpec struct {
|
||
mapSel int
|
||
pp int
|
||
load byte // r/m → vector
|
||
store byte // vector → r/m
|
||
w int
|
||
n [3]int
|
||
}
|
||
|
||
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
|
||
var evexMoveTable = map[string]evexMoveSpec{
|
||
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
|
||
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
|
||
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
|
||
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
|
||
// semantics).
|
||
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
|
||
// encoding).
|
||
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
|
||
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512, aligned packed moves.
|
||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}},
|
||
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.66.0F, aligned integer moves.
|
||
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
||
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
||
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
|
||
// three-operand register form is not supported).
|
||
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}},
|
||
}
|
||
|
||
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
||
func isEvex(mnemUpper string) bool {
|
||
if _, ok := evexTable[mnemUpper]; ok {
|
||
return true
|
||
}
|
||
if _, ok := evexBcastTable[mnemUpper]; ok {
|
||
return true
|
||
}
|
||
_, ok := evexMoveTable[mnemUpper]
|
||
return ok
|
||
}
|
||
|
||
// evexRequired reports whether the operands force the EVEX encoding of a
|
||
// mnemonic that also has a VEX form: ZMM and K registers do, and so do
|
||
// register indices 16-31, which only EVEX can represent (X16-Y31 exist
|
||
// solely under AVX-512).
|
||
func evexRequired(upper string, ops []Operand) bool {
|
||
_, inVex := vexTable[upper]
|
||
_, inVexMove := vexMoveTable[upper]
|
||
if !inVex && !inVexMove {
|
||
return true // EVEX-only mnemonic
|
||
}
|
||
for _, op := range ops {
|
||
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
|
||
return true
|
||
}
|
||
}
|
||
return false
|
||
}
|
||
|
||
// evexSuffix carries the EVEX mnemonic suffixes the Go assembler accepts:
|
||
// zeroing (.Z), a rounding mode (.RN_SAE, .RD_SAE, .RU_SAE, .RZ_SAE),
|
||
// suppress-all-exceptions (.SAE) and memory broadcast (.BCST). Masking is
|
||
// not a suffix, Go writes it as an explicit K operand.
|
||
type evexSuffix struct {
|
||
zeroing bool
|
||
sae bool
|
||
bcst bool
|
||
rounding int // -1 = none; otherwise the EVEX rc value (0 RN, 1 RD, 2 RU, 3 RZ)
|
||
}
|
||
|
||
// any reports whether any suffix is present.
|
||
func (s evexSuffix) any() bool {
|
||
return s.zeroing || s.sae || s.bcst || s.rounding >= 0
|
||
}
|
||
|
||
// evexOnly reports whether the suffix forces the EVEX encoding (everything
|
||
// but plain zeroing, which the dispatch checks separately).
|
||
func (s evexSuffix) evexOnly() bool {
|
||
return s.sae || s.bcst || s.rounding >= 0
|
||
}
|
||
|
||
// parseEvexSuffix splits the EVEX suffix chain off the mnemonic
|
||
// ("VADDPD.RN_SAE.Z" → base "VADDPD", rounding RN, zeroing), validating the
|
||
// combinations the Go assembler allows: .Z last, no duplicates, no
|
||
// broadcast together with rounding/SAE.
|
||
func parseEvexSuffix(mnem string) (string, evexSuffix, error) {
|
||
sfx := evexSuffix{rounding: -1}
|
||
before, after, ok := strings.Cut(mnem, ".")
|
||
if !ok {
|
||
return mnem, sfx, nil
|
||
}
|
||
base := before
|
||
parts := strings.Split(after, ".")
|
||
seen := map[string]bool{}
|
||
for j, p := range parts {
|
||
if seen[p] {
|
||
return "", sfx, fmt.Errorf("duplicate EVEX suffix %q", p)
|
||
}
|
||
seen[p] = true
|
||
switch p {
|
||
case "Z":
|
||
if j != len(parts)-1 {
|
||
return "", sfx, fmt.Errorf("the .Z suffix must come last in %q", after)
|
||
}
|
||
sfx.zeroing = true
|
||
case "SAE":
|
||
sfx.sae = true
|
||
case "BCST":
|
||
sfx.bcst = true
|
||
case "RN_SAE":
|
||
sfx.rounding = 0
|
||
case "RD_SAE":
|
||
sfx.rounding = 1
|
||
case "RU_SAE":
|
||
sfx.rounding = 2
|
||
case "RZ_SAE":
|
||
sfx.rounding = 3
|
||
default:
|
||
return "", sfx, fmt.Errorf("unsupported EVEX suffix %q", p)
|
||
}
|
||
}
|
||
if sfx.bcst && (sfx.sae || sfx.rounding >= 0) {
|
||
return "", sfx, fmt.Errorf("cannot combine .BCST with rounding or SAE in %q", after)
|
||
}
|
||
return base, sfx, nil
|
||
}
|
||
|
||
// evexRound lists the instructions that accept a rounding mode or .SAE.
|
||
var evexRound = map[string]bool{
|
||
"VADDPD": true, "VSUBPD": true, "VMULPD": true, "VDIVPD": true,
|
||
"VMINPD": true, "VMAXPD": true,
|
||
"VADDPS": true, "VSUBPS": true, "VMULPS": true, "VDIVPS": true,
|
||
"VMINPS": true, "VMAXPS": true,
|
||
"VADDSD": true, "VSUBSD": true, "VMULSD": true, "VDIVSD": true,
|
||
"VMINSD": true, "VMAXSD": true,
|
||
"VADDSS": true, "VSUBSS": true, "VMULSS": true, "VDIVSS": true,
|
||
"VMINSS": true, "VMAXSS": true,
|
||
"VSCALEFPD": true, "VSCALEFPS": true, "VSCALEFSD": true, "VSCALEFSS": true,
|
||
"VGETEXPPD": true, "VGETEXPPS": true, "VGETEXPSD": true, "VGETEXPSS": true,
|
||
"VCVTDQ2PS": true, "VCVTPS2QQ": true, "VCVTQQ2PS": true, "VCVTPD2UQQ": true,
|
||
"VCVTPD2PS": true, "VCVTPD2UDQ": true, "VCVTTPD2UDQ": true, "VCVTTPD2UQQ": true,
|
||
"VCVTPS2UDQ": true, "VCVTTPS2UDQ": true, "VCVTPS2UQQ": true, "VCVTTPS2UQQ": true,
|
||
"VCVTTPD2QQ": true, "VCVTTPS2QQ": true, "VCVTUQQ2PD": true, "VCVTUQQ2PS": true,
|
||
"VCVTSD2SI": true, "VCVTSD2SIQ": true, "VCVTSS2SI": true, "VCVTSS2SIQ": true,
|
||
"VCVTSD2USIL": true, "VCVTSD2USIQ": true, "VCVTSS2USIL": true, "VCVTSS2USIQ": true,
|
||
"VCVTTSD2SI": true, "VCVTTSD2SIQ": true, "VCVTTSS2SI": true, "VCVTTSS2SIQ": true,
|
||
"VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true,
|
||
"VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true,
|
||
"VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true,
|
||
}
|
||
|
||
// evexBcstN maps an instruction accepting .BCST to the broadcast element
|
||
// size, the disp8×N multiplier for its memory operand.
|
||
var evexBcstN = map[string]int{
|
||
"VADDPD": 8, "VSUBPD": 8, "VMULPD": 8, "VDIVPD": 8,
|
||
"VMINPD": 8, "VMAXPD": 8,
|
||
"VADDPS": 4, "VSUBPS": 4, "VMULPS": 4, "VDIVPS": 4,
|
||
"VMINPS": 4, "VMAXPS": 4,
|
||
"VRCP14PD": 8, "VRCP14PS": 4, "VRSQRT14PD": 8, "VRSQRT14PS": 4,
|
||
"VGETEXPPD": 8, "VGETEXPPS": 4,
|
||
"VSCALEFPD": 8, "VSCALEFPS": 4,
|
||
"VRNDSCALEPD": 8, "VRNDSCALEPS": 4,
|
||
"VGETMANTPD": 8, "VGETMANTPS": 4,
|
||
"VREDUCEPD": 8, "VREDUCEPS": 4,
|
||
"VFIXUPIMMPD": 8, "VFIXUPIMMPS": 4,
|
||
"VRANGEPD": 8, "VRANGEPS": 4,
|
||
"VCVTDQ2PS": 4, "VCVTPS2QQ": 4, "VCVTQQ2PS": 8,
|
||
"VCVTUDQ2PD": 4, "VCVTUDQ2PS": 4,
|
||
"VCVTPD2PS": 8, "VCVTPD2UDQ": 8, "VCVTTPD2UDQ": 8, "VCVTTPD2UQQ": 8,
|
||
"VCVTPS2UDQ": 4, "VCVTTPS2UDQ": 4, "VCVTPS2UQQ": 4, "VCVTTPS2UQQ": 4,
|
||
"VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8,
|
||
}
|
||
|
||
// splitMask extracts an explicit mask register (K1-K7) from the operand list,
|
||
// returning the remaining operands and the mask index. K0 is not a usable
|
||
// mask (aaa = 0 means "no mask"), matching the assembler.
|
||
func splitMask(ops []Operand) ([]Operand, int, error) {
|
||
var rest []Operand
|
||
mask := 0
|
||
for _, op := range ops {
|
||
if r, ok := op.(Reg); ok && r.mask {
|
||
if mask != 0 {
|
||
return nil, 0, fmt.Errorf("at most one mask register operand")
|
||
}
|
||
if r.idx == 0 {
|
||
return nil, 0, fmt.Errorf("k0 is not a usable mask register")
|
||
}
|
||
mask = r.idx
|
||
continue
|
||
}
|
||
rest = append(rest, op)
|
||
}
|
||
return rest, mask, nil
|
||
}
|
||
|
||
// encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The
|
||
// mask, when present, is an explicit K1-K7 operand anywhere among the
|
||
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
|
||
// broadcast.
|
||
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
|
||
// The mask/vector conversions take the K register as a genuine operand
|
||
// (source or destination), not as a mask, and accept no suffixes.
|
||
if evexKOperand[mnemUpper] {
|
||
if sfx.any() {
|
||
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
|
||
}
|
||
spec, ok := evexTable[mnemUpper]
|
||
if !ok {
|
||
return fmt.Errorf("unsupported instruction %q", mnemUpper)
|
||
}
|
||
return e.encodeEvexRM(spec, ops, 0, sfx)
|
||
}
|
||
spec, inTable := evexTable[mnemUpper]
|
||
if inTable {
|
||
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
||
return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper)
|
||
}
|
||
if sfx.bcst {
|
||
n, ok := evexBcstN[mnemUpper]
|
||
if !ok {
|
||
return fmt.Errorf("%s: broadcast is not supported for this instruction", mnemUpper)
|
||
}
|
||
spec.n = [3]int{n, n, n}
|
||
}
|
||
} else if sfx.evexOnly() {
|
||
return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper)
|
||
}
|
||
|
||
// Mask-destination comparisons (VPCMPEQD, VCMPPD $imm, …): the last
|
||
// operand is the destination K register, and any mask sits among the
|
||
// preceding operands.
|
||
kdst := func(encode func(evexSpec, []Operand, int, evexSuffix) error) error {
|
||
dst, ok := ops[len(ops)-1].(Reg)
|
||
if !ok || !dst.mask {
|
||
return nil // not a K-destination form; fall through
|
||
}
|
||
rest, mask, err := splitMask(ops[:len(ops)-1])
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if sfx.zeroing && mask == 0 {
|
||
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper)
|
||
}
|
||
return encode(spec, append(rest, dst), mask, sfx)
|
||
}
|
||
if inTable && len(ops) > 0 {
|
||
switch spec.form {
|
||
case vexNDS3:
|
||
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
||
return kdst(e.encodeEvexNDS3)
|
||
}
|
||
case vexNDS3Imm:
|
||
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
||
return kdst(e.encodeEvexNDS3Imm)
|
||
}
|
||
case vexImmRM:
|
||
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
||
return kdst(e.encodeEvexImmRM)
|
||
}
|
||
}
|
||
}
|
||
|
||
rest, mask, err := splitMask(ops)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if sfx.zeroing && mask == 0 {
|
||
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper)
|
||
}
|
||
ops = rest
|
||
|
||
if bs, ok := evexBcastTable[mnemUpper]; ok {
|
||
if sfx.evexOnly() {
|
||
return fmt.Errorf("%s: broadcast instructions take no rounding/SAE/broadcast suffix", mnemUpper)
|
||
}
|
||
return e.encodeEvexBcast(bs, ops, mask, sfx)
|
||
}
|
||
if ms, ok := evexMoveTable[mnemUpper]; ok {
|
||
if sfx.evexOnly() {
|
||
return fmt.Errorf("%s: moves take no rounding/SAE/broadcast suffix", mnemUpper)
|
||
}
|
||
return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx)
|
||
}
|
||
if !inTable {
|
||
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
|
||
}
|
||
switch spec.form {
|
||
case vexNDS3:
|
||
return e.encodeEvexNDS3(spec, ops, mask, sfx)
|
||
case vexRM:
|
||
return e.encodeEvexRM(spec, ops, mask, sfx)
|
||
case vexRMRev:
|
||
return e.encodeEvexRMRev(spec, ops, mask, sfx)
|
||
case vexImmRM:
|
||
return e.encodeEvexImmRM(spec, ops, mask, sfx)
|
||
case vexShiftImm:
|
||
return e.encodeEvexShiftImm(spec, ops, mask, sfx)
|
||
case vexNDS3Imm:
|
||
return e.encodeEvexNDS3Imm(spec, ops, mask, sfx)
|
||
case vexExtract:
|
||
return e.encodeEvexExtract(spec, ops, mask, sfx)
|
||
case vexRMSrcLen:
|
||
return e.encodeEvexRMSrcLen(spec, ops, mask, sfx)
|
||
}
|
||
return fmt.Errorf("unhandled EVEX form for %s", mnemUpper)
|
||
}
|
||
|
||
// encodeEvexNDS3 encodes the three-operand NDS form: OP src2, src1, dst. The
|
||
// destination may be an opmask register (VPCMPEQD), in which case the vector
|
||
// length comes from the sources.
|
||
func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("EVEX NDS instruction expects 3 operands, got %d", len(ops))
|
||
}
|
||
src2, src1, dst := ops[0], ops[1], ops[2]
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || (!dstReg.isVec() && !dstReg.mask) {
|
||
return fmt.Errorf("EVEX destination must be a vector or mask register")
|
||
}
|
||
vvvvReg, ok := src1.(Reg)
|
||
if !ok || !vvvvReg.isVec() {
|
||
return fmt.Errorf("EVEX vvvv operand must be a vector register")
|
||
}
|
||
ll := dstReg.vecLenBit()
|
||
if dstReg.mask {
|
||
ll = vvvvReg.vecLenBit()
|
||
if r, ok := src2.(Reg); ok && r.isVec() {
|
||
ll = r.vecLenBit()
|
||
}
|
||
}
|
||
return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, sfx)
|
||
}
|
||
|
||
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
|
||
// no vvvv), e.g. VCVTQQ2PD. The destination may be an opmask register (the
|
||
// *2M mask conversions) or a general-purpose register (the scalar
|
||
// vector-to-GPR conversions); in both cases the vector length comes from
|
||
// the source.
|
||
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok {
|
||
return fmt.Errorf("EVEX destination must be a register")
|
||
}
|
||
ll := dstReg.vecLenBit()
|
||
if !dstReg.isVec() {
|
||
// Mask or GPR destination: the length follows the vector source
|
||
// (128 for a memory source).
|
||
ll = 0
|
||
if r, ok := src.(Reg); ok && r.isVec() {
|
||
ll = r.vecLenBit()
|
||
}
|
||
}
|
||
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
|
||
}
|
||
|
||
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
|
||
// (reg = dst, rm = src, imm8), e.g. VPSHUFD. The destination may be an
|
||
// opmask register (VFPCLASS*), in which case the vector length comes from
|
||
// the source.
|
||
func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||
}
|
||
imm, src, dst := ops[0], ops[1], ops[2]
|
||
immVal, ok := imm.(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("shuffle control must be an immediate")
|
||
}
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || (!dstReg.isVec() && !dstReg.mask) {
|
||
return fmt.Errorf("shuffle destination must be a vector or mask register")
|
||
}
|
||
ll := dstReg.vecLenBit()
|
||
if dstReg.mask {
|
||
if r, ok := src.(Reg); ok && r.isVec() {
|
||
ll = r.vecLenBit()
|
||
}
|
||
} else if r, ok := src.(Reg); ok && r.isVec() {
|
||
ll = r.vecLenBit()
|
||
}
|
||
immByte, err := imm8(int64(immVal))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if err := e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx); err != nil {
|
||
return err
|
||
}
|
||
e.out = append(e.out, immByte)
|
||
return nil
|
||
}
|
||
|
||
// encodeEvexShiftImm encodes an immediate shift: OP $imm, src, dst
|
||
// (ModRM.reg = /digit, vvvv = dst, rm = src, imm8), e.g. VPSRAD $31, Z3, Z5.
|
||
func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("EVEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||
}
|
||
imm, src, dst := ops[0], ops[1], ops[2]
|
||
immVal, ok := imm.(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("shift count must be an immediate")
|
||
}
|
||
srcReg, ok := src.(Reg)
|
||
if !ok || !srcReg.isVec() {
|
||
return fmt.Errorf("shift source must be a vector register")
|
||
}
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || !dstReg.isVec() {
|
||
return fmt.Errorf("shift destination must be a vector register")
|
||
}
|
||
immByte, err := imm8(int64(immVal))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, sfx); err != nil {
|
||
return err
|
||
}
|
||
e.out = append(e.out, immByte)
|
||
return nil
|
||
}
|
||
|
||
// encodeEvexNDS3Imm encodes OP $imm, src2, src1, dst (reg=dst, vvvv=src1,
|
||
// rm=src2, imm8), e.g. VALIGND. The destination may be an opmask register
|
||
// (VCMPPD and friends), in which case the vector length comes from the
|
||
// sources.
|
||
func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||
if len(ops) != 4 {
|
||
return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops))
|
||
}
|
||
imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
|
||
immVal, ok := imm.(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("shuffle control must be an immediate")
|
||
}
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || (!dstReg.isVec() && !dstReg.mask) {
|
||
return fmt.Errorf("destination must be a vector or mask register")
|
||
}
|
||
vvvvReg, ok := src1.(Reg)
|
||
if !ok || !vvvvReg.isVec() {
|
||
return fmt.Errorf("second source must be a vector register")
|
||
}
|
||
immByte, err := imm8(int64(immVal))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
ll := dstReg.vecLenBit()
|
||
if dstReg.mask {
|
||
ll = vvvvReg.vecLenBit()
|
||
if r, ok := src2.(Reg); ok && r.isVec() {
|
||
ll = r.vecLenBit()
|
||
}
|
||
}
|
||
if err := e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, sfx); err != nil {
|
||
return err
|
||
}
|
||
e.out = append(e.out, immByte)
|
||
return nil
|
||
}
|
||
|
||
// encodeEvexExtract encodes OP $imm, zsrc, ydst (reg=ZMM source, rm=YMM/memory
|
||
// destination, imm8), e.g. VEXTRACTI64X4.
|
||
func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("extract expects 3 operands ($imm, zsrc, ydst), got %d", len(ops))
|
||
}
|
||
imm, src, dst := ops[0], ops[1], ops[2]
|
||
immVal, ok := imm.(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("extract lane must be an immediate")
|
||
}
|
||
srcReg, ok := src.(Reg)
|
||
if !ok || !srcReg.isVec() {
|
||
return fmt.Errorf("extract source must be a vector register")
|
||
}
|
||
immByte, err := imm8(int64(immVal))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, sfx); err != nil {
|
||
return err
|
||
}
|
||
e.out = append(e.out, immByte)
|
||
return nil
|
||
}
|
||
|
||
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
|
||
// the store-form opcode (reg = source, rm = destination), matching the Go
|
||
// assembler.
|
||
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
srcReg, srcIsVec := vecReg(src)
|
||
dstReg, dstIsVec := vecReg(dst)
|
||
|
||
op := ms.store
|
||
var reg Reg
|
||
var rm Operand
|
||
switch {
|
||
case srcIsVec && dstIsVec:
|
||
reg, rm = srcReg, dst
|
||
case srcIsVec:
|
||
if !memOperand(dst) {
|
||
return fmt.Errorf("%s: invalid destination operand", mnem)
|
||
}
|
||
reg, rm = srcReg, dst
|
||
case dstIsVec:
|
||
if !memOperand(src) {
|
||
return fmt.Errorf("%s: invalid source operand", mnem)
|
||
}
|
||
op = ms.load
|
||
reg, rm = dstReg, src
|
||
default:
|
||
return fmt.Errorf("%s needs a vector register operand", mnem)
|
||
}
|
||
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
|
||
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx)
|
||
}
|
||
|
||
// encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
|
||
// the destination always XMM and the length fixed by the mnemonic, the
|
||
// single valid slot of spec.n names the vector length (and the disp8×N
|
||
// multiplier) a register or memory source encodes.
|
||
func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || !dstReg.isVec() {
|
||
return fmt.Errorf("EVEX destination must be a vector register")
|
||
}
|
||
ll, err := soleLen(spec.n)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
|
||
}
|
||
|
||
// soleLen returns the vector-length index of the single valid slot of n
|
||
// the length a length-fixed mnemonic (the EVEX conversion spellings) encodes
|
||
// regardless of its operands.
|
||
func soleLen(n [3]int) (int, error) {
|
||
ll := -1
|
||
for i, v := range n {
|
||
if v == 0 {
|
||
continue
|
||
}
|
||
if ll >= 0 {
|
||
return 0, fmt.Errorf("ambiguous vector-length table %v", n)
|
||
}
|
||
ll = i
|
||
}
|
||
if ll < 0 {
|
||
return 0, fmt.Errorf("empty vector-length table")
|
||
}
|
||
return ll, nil
|
||
}
|
||
|
||
// memOperand reports whether op is a memory reference (including a
|
||
// static-symbol reference).
|
||
func memOperand(op Operand) bool {
|
||
switch op.(type) {
|
||
case Mem, sbMem:
|
||
return true
|
||
}
|
||
return false
|
||
}
|
||
|
||
// encodeEvexRMRev encodes the narrowing-store form: OP src, dst with the wide
|
||
// source in the reg field and the narrow destination in r/m (VPMOVDW/QD).
|
||
func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("EVEX store instruction expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
srcReg, ok := src.(Reg)
|
||
if !ok || !srcReg.isVec() {
|
||
return fmt.Errorf("EVEX source must be a vector register")
|
||
}
|
||
return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, sfx)
|
||
}
|
||
|
||
// encodeEvexBcast encodes VPBROADCASTD/Q: OP src, dst with the GPR or memory
|
||
// source broadcast to every lane of the vector destination.
|
||
func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("broadcast expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || !dstReg.isVec() {
|
||
return fmt.Errorf("broadcast destination must be a vector register")
|
||
}
|
||
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
|
||
switch src.(type) {
|
||
case Mem, sbMem:
|
||
spec.opcode = bs.opMem
|
||
spec.n = [3]int{bs.n, bs.n, bs.n}
|
||
case Reg:
|
||
spec.opcode = bs.opReg
|
||
default:
|
||
return fmt.Errorf("broadcast source must be a register or memory")
|
||
}
|
||
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, sfx)
|
||
}
|
||
|
||
// emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement
|
||
// (disp8×N compressed) for the given precomputed fields. regIdx is the
|
||
// unextended reg-field register index, or a /digit (0-7); vvvvIdx is the
|
||
// vvvv register index, or -1 when unused. mask (K1-K7, 0 = unmasked) and
|
||
// zeroing fill the aaa and z bits of the P2 byte.
|
||
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, sfx evexSuffix) error {
|
||
if ll > 2 {
|
||
return fmt.Errorf("invalid vector length")
|
||
}
|
||
// reg-field extension bits (R̄, R'̄), inverted.
|
||
rBar, rPrimeBar := 1, 1
|
||
if regIdx&8 != 0 {
|
||
rBar = 0
|
||
}
|
||
if regIdx&16 != 0 {
|
||
rPrimeBar = 0
|
||
}
|
||
// vvvv (inverted) and its extension bit V'̄.
|
||
vBar, vPrimeBar := 15, 1
|
||
if vvvvIdx >= 0 {
|
||
vBar = 15 - (vvvvIdx & 15)
|
||
if vvvvIdx&16 != 0 {
|
||
vPrimeBar = 0
|
||
}
|
||
}
|
||
|
||
var modrm, sib int
|
||
var disp []byte
|
||
xBar, bBar := 1, 1
|
||
var sb *sbRef
|
||
switch r := rm.(type) {
|
||
case Reg:
|
||
// ModRM.mod = 11: rm[3] extends via B̄, and rm[4] via X̄ (the EVEX
|
||
// register-register quirk).
|
||
modrm = 0xC0 | (regIdx&7)<<3 | (r.idx & 7)
|
||
sib = -1
|
||
if r.idx&8 != 0 {
|
||
bBar = 0
|
||
}
|
||
if r.idx&16 != 0 {
|
||
xBar = 0
|
||
}
|
||
if r.idx&16 != 0 {
|
||
xBar = 0
|
||
}
|
||
case Mem:
|
||
var err error
|
||
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
|
||
if err != nil {
|
||
return err
|
||
}
|
||
// An indexed memory operand carries index[4] in V'̄ (Go folds it
|
||
// together with vvvv[4] into the same bit).
|
||
if r.HasIndex && r.Index.idx&16 != 0 {
|
||
vPrimeBar = 0
|
||
}
|
||
case sbMem:
|
||
// RIP-relative static-symbol reference; disp32 patched at link time
|
||
// (no disp8 scaling for RIP-relative addressing).
|
||
modrm = (regIdx&7)<<3 | 0x05
|
||
sib = -1
|
||
disp = le32(0)
|
||
sb = &sbRef{name: r.name, addend: r.addend}
|
||
default:
|
||
return fmt.Errorf("invalid EVEX r/m operand")
|
||
}
|
||
|
||
z := 0
|
||
if sfx.zeroing {
|
||
z = 1
|
||
}
|
||
// The b bit and the L'L field carry the rounding/SAE/broadcast mode:
|
||
// a rounding mode replaces L'L with the rc value, plain SAE and
|
||
// broadcast keep the vector length.
|
||
b := 0
|
||
switch {
|
||
case sfx.rounding >= 0:
|
||
b, ll = 1, sfx.rounding
|
||
case sfx.sae || sfx.bcst:
|
||
b = 1
|
||
}
|
||
p0 := byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4 | spec.mapSel)
|
||
p1 := byte(spec.w<<7 | vBar<<3 | 1<<2 | spec.pp)
|
||
p2 := byte(z<<7 | ll<<5 | b<<4 | vPrimeBar<<3 | mask) // z, L'L/rc, b, V', aaa
|
||
e.out = append(e.out, 0x62, p0, p1, p2, spec.opcode, byte(modrm))
|
||
if sib >= 0 {
|
||
e.out = append(e.out, byte(sib))
|
||
}
|
||
if sb != nil {
|
||
e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend})
|
||
}
|
||
e.out = append(e.out, disp...)
|
||
return nil
|
||
}
|
||
|
||
// memComponentsEvex computes the ModR/M byte (with the given reg field), the
|
||
// SIB byte (-1 if none), the displacement bytes and the (inverted sense)
|
||
// index/base extension bits for an EVEX memory operand. The displacement is
|
||
// compressed to disp8×N when it is a multiple of n and the quotient fits a
|
||
// signed byte; otherwise a full disp32 is used.
|
||
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
|
||
sib = -1
|
||
xBar, bBar = 1, 1 // inverted bits: 1 = no extension
|
||
if !m.HasBase && !m.HasIndex {
|
||
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
|
||
}
|
||
|
||
needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4)
|
||
|
||
var mod int
|
||
switch {
|
||
case !m.HasBase:
|
||
mod = 0
|
||
disp = le32(m.Disp)
|
||
case m.Base.idx&7 == 5 && m.Disp == 0:
|
||
mod = 1
|
||
disp = []byte{0}
|
||
case m.Disp == 0:
|
||
mod = 0
|
||
case n > 0 && m.Disp%int64(n) == 0 && m.Disp/int64(n) >= -128 && m.Disp/int64(n) <= 127:
|
||
mod = 1
|
||
disp = []byte{byte(int8(m.Disp / int64(n)))}
|
||
default:
|
||
mod = 2
|
||
disp = le32(m.Disp)
|
||
}
|
||
|
||
if needSIB {
|
||
idxField := 4 // 100 = no index
|
||
if m.HasIndex {
|
||
idxField = m.Index.idx & 7
|
||
if m.Index.idx&8 != 0 {
|
||
xBar = 0
|
||
}
|
||
}
|
||
baseField := 5 // 101 = no base (with mod=00 → disp32)
|
||
if m.HasBase {
|
||
baseField = m.Base.idx & 7
|
||
if m.Base.idx&8 != 0 {
|
||
bBar = 0
|
||
}
|
||
}
|
||
return mod<<6 | regField<<3 | 0x04, scaleBits(m.Scale)<<6 | idxField<<3 | baseField, disp, xBar, bBar, nil
|
||
}
|
||
|
||
if m.Base.idx&8 != 0 {
|
||
bBar = 0
|
||
}
|
||
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil
|
||
}
|
||
|
||
// gatherSpec describes a gather/scatter family member: all live in
|
||
// 66.0F38; the opcode and W select the index and data element widths, and n
|
||
// is the data element size (the EVEX disp8×N multiplier).
|
||
type gatherSpec struct {
|
||
opcode byte
|
||
w int
|
||
n int
|
||
}
|
||
|
||
var gatherTable = map[string]gatherSpec{
|
||
"VGATHERDPS": {0x92, 0, 4},
|
||
"VGATHERDPD": {0x92, 1, 8},
|
||
"VGATHERQPS": {0x93, 0, 4},
|
||
"VGATHERQPD": {0x93, 1, 8},
|
||
"VPGATHERDD": {0x90, 0, 4},
|
||
"VPGATHERDQ": {0x90, 1, 8},
|
||
"VPGATHERQD": {0x91, 0, 4},
|
||
"VPGATHERQQ": {0x91, 1, 8},
|
||
}
|
||
|
||
var scatterTable = map[string]gatherSpec{
|
||
"VSCATTERDPS": {0xA2, 0, 4},
|
||
"VSCATTERDPD": {0xA2, 1, 8},
|
||
"VSCATTERQPS": {0xA3, 0, 4},
|
||
"VSCATTERQPD": {0xA3, 1, 8},
|
||
"VPSCATTERDD": {0xA0, 0, 4},
|
||
"VPSCATTERDQ": {0xA0, 1, 8},
|
||
"VPSCATTERQD": {0xA1, 0, 4},
|
||
"VPSCATTERQQ": {0xA1, 1, 8},
|
||
}
|
||
|
||
// isGather reports whether the mnemonic is a gather instruction.
|
||
func isGather(upper string) bool {
|
||
_, ok := gatherTable[upper]
|
||
return ok
|
||
}
|
||
|
||
// isScatter reports whether the mnemonic is a scatter instruction.
|
||
func isScatter(upper string) bool {
|
||
_, ok := scatterTable[upper]
|
||
return ok
|
||
}
|
||
|
||
// vsibLen validates a VSIB memory operand (the index must be a vector
|
||
// register) and returns it with the vector length the index selects, the
|
||
// EVEX L'L field follows the index register, not the data register.
|
||
func vsibLen(op Operand, what string) (Mem, int, error) {
|
||
m, ok := op.(Mem)
|
||
if !ok || !m.HasIndex || !m.Index.isVec() {
|
||
return Mem{}, 0, fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", what)
|
||
}
|
||
return m, m.Index.vecLenBit(), nil
|
||
}
|
||
|
||
// encodeGather encodes a gather. The VEX spelling carries the mask in a
|
||
// vector register (OP mask, vsib, dst: vvvv = mask, rm = vsib, reg = dst,
|
||
// L follows the data register); the EVEX spelling carries it in aaa (OP
|
||
// vsib, K, dst: rm = vsib, reg = dst, L follows the VSIB index).
|
||
func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexSuffix) error {
|
||
rest, mask, err := splitMask(ops)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if mask != 0 || sfx.any() {
|
||
// EVEX form: OP vsib, K, dst.
|
||
if len(rest) != 2 {
|
||
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
|
||
}
|
||
vsib, ll, err := vsibLen(rest[0], upper)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
dst, ok := rest[1].(Reg)
|
||
if !ok || !dst.isVec() {
|
||
return fmt.Errorf("%s: destination must be a vector register", upper)
|
||
}
|
||
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
|
||
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
|
||
}
|
||
// VEX form: OP mask, vsib, dst.
|
||
if len(rest) != 3 {
|
||
return fmt.Errorf("%s expects 3 operands (mask, vsib, dst), got %d", upper, len(rest))
|
||
}
|
||
maskReg, ok := rest[0].(Reg)
|
||
if !ok || !maskReg.isVec() {
|
||
return fmt.Errorf("%s: mask must be a vector register", upper)
|
||
}
|
||
vsib, _, err := vsibLen(rest[1], upper)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
dst, ok := rest[2].(Reg)
|
||
if !ok || !dst.isVec() {
|
||
return fmt.Errorf("%s: destination must be a vector register", upper)
|
||
}
|
||
spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1}
|
||
rBit := 0
|
||
if dst.idx >= 8 {
|
||
rBit = 1
|
||
}
|
||
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib)
|
||
}
|
||
|
||
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src,
|
||
// rm = the VSIB memory operand, the K mask in aaa and L following the VSIB
|
||
// index.
|
||
func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evexSuffix) error {
|
||
rest, mask, err := splitMask(ops)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if mask == 0 {
|
||
return fmt.Errorf("%s requires a K mask register", upper)
|
||
}
|
||
if len(rest) != 2 {
|
||
return fmt.Errorf("%s expects 3 operands (src, K, vsib), got %d", upper, len(ops))
|
||
}
|
||
src, ok := rest[0].(Reg)
|
||
if !ok || !src.isVec() {
|
||
return fmt.Errorf("%s: source must be a vector register", upper)
|
||
}
|
||
vsib, ll, err := vsibLen(rest[1], upper)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
|
||
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
|
||
}
|
||
|
||
// evexKOperand lists the instructions whose K register is a genuine operand
|
||
// (the source or destination of a mask/vector conversion) rather than a
|
||
// mask modifier, the M2 and 2M conversions. They take no masking.
|
||
var evexKOperand = map[string]bool{
|
||
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
|
||
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
|
||
}
|
||
|
||
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
||
// direction, kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
||
// gprk (GPR/mem → K), kgpr (K → GPR), and the GPR forms carry a mandatory
|
||
// prefix and W for the wider widths.
|
||
type kmovSpec struct {
|
||
kk, kmem, gprk, kgpr byte
|
||
gprPP int
|
||
w int
|
||
}
|
||
|
||
var kmovTable = map[string]kmovSpec{
|
||
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0},
|
||
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 3, 1},
|
||
}
|
||
|
||
// encodeKmov encodes a KMOV width, selecting the opcode by direction.
|
||
func (e *enc) encodeKmov(upper string, ops []Operand) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||
}
|
||
ks := kmovTable[upper]
|
||
src, dst := ops[0], ops[1]
|
||
srcReg, srcIsReg := src.(Reg)
|
||
dstReg, dstIsReg := dst.(Reg)
|
||
srcK := srcIsReg && srcReg.mask
|
||
dstK := dstIsReg && dstReg.mask
|
||
spec := vexSpec{mapSel: 1, w: ks.w, pp: 0, opdigit: -1}
|
||
switch {
|
||
case srcK && dstK:
|
||
spec.opcode = ks.kk // k ← k: reg = dst, rm = src
|
||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
||
case srcK && dstIsReg:
|
||
spec.opcode = ks.kgpr // GPR ← k: reg = dst, rm = src
|
||
spec.pp = ks.gprPP
|
||
rBit := 0
|
||
if dstReg.idx >= 8 {
|
||
rBit = 1
|
||
}
|
||
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15, src)
|
||
case srcK:
|
||
if _, ok := dst.(Mem); !ok {
|
||
return fmt.Errorf("%s: invalid destination operand", upper)
|
||
}
|
||
spec.opcode = ks.kmem // mem ← k: reg = src, rm = dst
|
||
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
|
||
case dstK:
|
||
spec.opcode = ks.gprk // k ← GPR/mem: reg = dst, rm = src
|
||
spec.pp = ks.gprPP
|
||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
||
}
|
||
return fmt.Errorf("%s requires a K register operand", upper)
|
||
}
|
||
|
||
// kOpSpec describes the VEX encoding of an opmask-register instruction: the
|
||
// L bit and the W/pp pair select the operand width, and the form the
|
||
// operand layout.
|
||
type kOpSpec struct {
|
||
mapSel int
|
||
opcode byte
|
||
w int
|
||
pp int
|
||
ll int
|
||
form vexForm
|
||
}
|
||
|
||
var kOpsTable = map[string]kOpSpec{
|
||
// k ← k OP k: reg = dst, vvvv = src1, rm = src2 (three opmask
|
||
// registers). Byte/word widths share W0 and differ by the 66 prefix;
|
||
// dword/qword share the W bit selection the Go assembler emits.
|
||
"KANDB": {1, 0x41, 0, 1, 1, vexNDS3},
|
||
"KANDW": {1, 0x41, 0, 0, 1, vexNDS3},
|
||
"KANDD": {1, 0x41, 1, 1, 1, vexNDS3},
|
||
"KANDQ": {1, 0x41, 1, 0, 1, vexNDS3},
|
||
"KANDNB": {1, 0x42, 0, 1, 1, vexNDS3},
|
||
"KANDNW": {1, 0x42, 0, 0, 1, vexNDS3},
|
||
"KANDND": {1, 0x42, 1, 1, 1, vexNDS3},
|
||
"KANDNQ": {1, 0x42, 1, 0, 1, vexNDS3},
|
||
"KORB": {1, 0x45, 0, 1, 1, vexNDS3},
|
||
"KORW": {1, 0x45, 0, 0, 1, vexNDS3},
|
||
"KORD": {1, 0x45, 1, 1, 1, vexNDS3},
|
||
"KORQ": {1, 0x45, 1, 0, 1, vexNDS3},
|
||
"KXNORB": {1, 0x46, 0, 1, 1, vexNDS3},
|
||
"KXNORW": {1, 0x46, 0, 0, 1, vexNDS3},
|
||
"KXNORD": {1, 0x46, 1, 1, 1, vexNDS3},
|
||
"KXNORQ": {1, 0x46, 1, 0, 1, vexNDS3},
|
||
"KXORB": {1, 0x47, 0, 1, 1, vexNDS3},
|
||
"KXORW": {1, 0x47, 0, 0, 1, vexNDS3},
|
||
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
|
||
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
|
||
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
|
||
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
|
||
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
|
||
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
|
||
"KADDD": {1, 0x4A, 1, 1, 1, vexNDS3},
|
||
"KADDQ": {1, 0x4A, 1, 0, 1, vexNDS3},
|
||
// k ← OP k (KNOT), k ← k AND~ k (KTEST-style RM) and flags ← k OP k
|
||
// (KORTEST): reg = dst, rm = src.
|
||
"KNOTB": {1, 0x44, 0, 1, 0, vexRM},
|
||
"KNOTW": {1, 0x44, 0, 0, 0, vexRM},
|
||
"KNOTD": {1, 0x44, 1, 1, 0, vexRM},
|
||
"KNOTQ": {1, 0x44, 1, 0, 0, vexRM},
|
||
"KORTESTB": {1, 0x98, 0, 1, 0, vexRM},
|
||
"KORTESTW": {1, 0x98, 0, 0, 0, vexRM},
|
||
"KORTESTD": {1, 0x98, 1, 1, 0, vexRM},
|
||
"KORTESTQ": {1, 0x98, 1, 0, 0, vexRM},
|
||
"KTESTB": {1, 0x99, 0, 1, 0, vexRM},
|
||
"KTESTW": {1, 0x99, 0, 0, 0, vexRM},
|
||
"KTESTD": {1, 0x99, 1, 1, 0, vexRM},
|
||
"KTESTQ": {1, 0x99, 1, 0, 0, vexRM},
|
||
// OP $imm, src, dst: reg = dst, rm = src, imm8. The opcodes split by
|
||
// direction (0x32/0x33 left, 0x30/0x31 right) and within each by
|
||
// element half (0x32 byte/word, 0x33 dword/qword); W picks byte/dword
|
||
// (W0) against word/qword (W1).
|
||
"KSHIFTLB": {3, 0x32, 0, 1, 0, vexImmRM},
|
||
"KSHIFTLW": {3, 0x32, 1, 1, 0, vexImmRM},
|
||
"KSHIFTLD": {3, 0x33, 0, 1, 0, vexImmRM},
|
||
"KSHIFTLQ": {3, 0x33, 1, 1, 0, vexImmRM},
|
||
"KSHIFTRB": {3, 0x30, 0, 1, 0, vexImmRM},
|
||
"KSHIFTRW": {3, 0x30, 1, 1, 0, vexImmRM},
|
||
"KSHIFTRD": {3, 0x31, 0, 1, 0, vexImmRM},
|
||
"KSHIFTRQ": {3, 0x31, 1, 1, 0, vexImmRM},
|
||
}
|
||
|
||
// isKOp reports whether the mnemonic is an opmask-register instruction.
|
||
func isKOp(upper string) bool {
|
||
_, ok := kOpsTable[upper]
|
||
return ok
|
||
}
|
||
|
||
// encodeKOp encodes an opmask-register instruction; every operand is a K
|
||
// register and the vector length is fixed by the instruction.
|
||
func (e *enc) encodeKOp(upper string, ops []Operand) error {
|
||
ks := kOpsTable[upper]
|
||
spec := vexSpec{mapSel: ks.mapSel, opcode: ks.opcode, w: ks.w, pp: ks.pp, opdigit: -1}
|
||
kreg := func(op Operand, what string) (Reg, error) {
|
||
r, ok := op.(Reg)
|
||
if !ok || !r.mask {
|
||
return Reg{}, fmt.Errorf("%s: %s must be an opmask register", upper, what)
|
||
}
|
||
return r, nil
|
||
}
|
||
switch ks.form {
|
||
case vexNDS3:
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("%s expects 3 operands, got %d", upper, len(ops))
|
||
}
|
||
src2, err := kreg(ops[0], "first source")
|
||
if err != nil {
|
||
return err
|
||
}
|
||
src1, err := kreg(ops[1], "second source")
|
||
if err != nil {
|
||
return err
|
||
}
|
||
dst, err := kreg(ops[2], "destination")
|
||
if err != nil {
|
||
return err
|
||
}
|
||
return e.emitVexFields(spec, ks.ll, dst.idx, 0, 15-src1.idx, src2)
|
||
case vexRM:
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||
}
|
||
src, err := kreg(ops[0], "source")
|
||
if err != nil {
|
||
return err
|
||
}
|
||
dst, err := kreg(ops[1], "destination")
|
||
if err != nil {
|
||
return err
|
||
}
|
||
return e.emitVexFields(spec, ks.ll, dst.idx, 0, 15, src)
|
||
case vexImmRM:
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("%s expects 3 operands ($imm, src, dst), got %d", upper, len(ops))
|
||
}
|
||
immVal, ok := ops[0].(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("%s: shift count must be an immediate", upper)
|
||
}
|
||
src, err := kreg(ops[1], "source")
|
||
if err != nil {
|
||
return err
|
||
}
|
||
dst, err := kreg(ops[2], "destination")
|
||
if err != nil {
|
||
return err
|
||
}
|
||
immByte, err := imm8(int64(immVal))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if err := e.emitVexFields(spec, ks.ll, dst.idx, 0, 15, src); err != nil {
|
||
return err
|
||
}
|
||
e.out = append(e.out, immByte)
|
||
return nil
|
||
}
|
||
return fmt.Errorf("unhandled opmask form for %s", upper)
|
||
}
|