813 lines
29 KiB
Go
813 lines
29 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||
// SPDX-License-Identifier: BSD-3-Clause
|
||
|
||
package asm
|
||
|
||
import (
|
||
"fmt"
|
||
"strings"
|
||
)
|
||
|
||
// This file implements EVEX (AVX-512) instruction encoding: the four-byte
|
||
// EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the
|
||
// compressed disp8×N displacement, and the operand shapes the go-flac
|
||
// AVX-512 kernels use plus the common floating-point and conversion set.
|
||
// Masking follows the Go assembler's spelling: an explicit K1–K7 operand
|
||
// anywhere among the operands (merging) plus a ".Z" mnemonic suffix for
|
||
// zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are
|
||
// supported too.
|
||
|
||
// evexSpec describes one EVEX instruction's encoding parameters. The form
|
||
// field reuses the vexForm shapes, which carry over unchanged.
|
||
type evexSpec struct {
|
||
mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A
|
||
opcode byte
|
||
w int
|
||
pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2
|
||
opdigit int // ModRM.reg /digit, or -1 when reg is a register
|
||
form vexForm // vexNDS3, vexRM, vexShiftImm, vexNDS3Imm, vexExtract
|
||
n [3]int // disp8×N multiplier per vector length (128/256/512)
|
||
}
|
||
|
||
// evexTable maps an upper-case mnemonic to its EVEX encoding. Mnemonics
|
||
// that also have a VEX form (VPADDD, VMOVUPD, …) are dispatched here only
|
||
// when an operand demands EVEX (a ZMM or K register); EVEX-only mnemonics
|
||
// (VPXORD, VALIGND, …) always encode through this table. The N multipliers
|
||
// are taken from the Go assembler's opcode tables, which are authoritative
|
||
// for byte-for-byte agreement.
|
||
var evexTable = map[string]evexSpec{
|
||
// EVEX.128/256/512.66.0F — integer arithmetic / logic, NDS form.
|
||
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSUBQ": {1, 0xFB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPXORD": {1, 0xEF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPXORQ": {1, 0xEF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.128/256/512.66.0F.W1 — packed double arithmetic.
|
||
"VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.66.0F.W1 — packed double unpack.
|
||
"VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.128.F2.0F.W1 — scalar double arithmetic (the packed opcodes with
|
||
// an F2 pp; the EVEX forms exist for masked and zeroing use). The
|
||
// memory operand is a single double, so disp8×N = 8.
|
||
"VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VSUBSD": {1, 0x5C, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VMULSD": {1, 0x59, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VDIVSD": {1, 0x5E, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
"VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||
|
||
// EVEX.128.F3.0F.W0 — scalar single arithmetic (disp8×N = 4).
|
||
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||
|
||
// EVEX.512.66.0F3A — align (NDS + imm8).
|
||
"VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.128/256/512.66.0F — immediate shift (VPSRAD /4).
|
||
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.66.0F.W1 — variable shift with an XMM count (VPSRAQ;
|
||
// the W bit distinguishes it from VPSRAD's E2 form).
|
||
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst,
|
||
// rm=src, no vvvv).
|
||
"VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.F2.0F.W1 — duplicate the low double (reg=dst,
|
||
// rm=src, no vvvv): a 128-bit destination reads a single double from
|
||
// memory (disp8×8), the wider ones read the full operand.
|
||
"VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}},
|
||
// EVEX.128/256/512.0F.W0 — signed dword to packed single (reg=dst,
|
||
// rm=src, no vvvv, no mandatory prefix — as in the VEX form).
|
||
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.0F.W0 — packed single to packed double: the
|
||
// destination is twice the source width and sets the length; disp8×N
|
||
// follows the narrow memory source. No F3 prefix: the Go assembler
|
||
// emits this instruction with pp = 00 (Intel's maps would call that
|
||
// undefined) and gasm reproduces the Go assembler's bytes — its machine
|
||
// code is the oracle, not the manual.
|
||
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
|
||
// EVEX.128/256/512.F3.0F.W0 — signed dword to packed double (the EVEX
|
||
// form of the VEX instruction; the destination sets the length, disp8×N
|
||
// follows the narrow memory source).
|
||
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
|
||
// EVEX packed double → dword conversions: the source is the wide
|
||
// operand and the mnemonic fixes the length — the bare names are
|
||
// 512-bit only (ZMM source, XMM destination), the X/Y spellings are
|
||
// EVEX-128/256. Exactly one slot of n is valid; it names the vector
|
||
// length (and the disp8×N multiplier) a register or memory source
|
||
// encodes.
|
||
"VCVTPD2DQ": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||
"VCVTTPD2DQ": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||
"VCVTPD2DQX": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||
"VCVTPD2DQY": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||
"VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||
"VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
||
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
||
// the xmm/ymm/zmm destination lengths).
|
||
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||
|
||
// EVEX.512.66.0F3A.W1 — lane extract (reg=ZMM source, rm=YMM/memory
|
||
// destination, imm8).
|
||
"VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
|
||
"VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
|
||
|
||
// EVEX.66.0F38 — more integer NDS forms (W distinguishes D/Q).
|
||
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
|
||
|
||
// EVEX.128/256/512 — the wider integer set (AVX-512 F/BW): byte/word
|
||
// arithmetic, the bitwise ops with D/Q suffixes, min/max, averages and
|
||
// variable shifts. All NDS form; W distinguishes element size.
|
||
"VPADDB": {1, 0xFC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPADDW": {1, 0xFD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSUBB": {1, 0xF8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSUBW": {1, 0xF9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMULLW": {1, 0xD5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPAVGB": {1, 0xE0, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPAVGW": {1, 0xE3, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINUB": {1, 0xDA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXUB": {1, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINSW": {1, 0xEA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXSW": {1, 0xEE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPANDD": {1, 0xDB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPANDQ": {1, 0xDB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPANDND": {1, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPANDNQ": {1, 0xDF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINSB": {2, 0x38, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXSB": {2, 0x3C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINSQ": {2, 0x39, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXSQ": {2, 0x3D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINUW": {2, 0x3A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXUW": {2, 0x3E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINSD": {2, 0x39, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXSD": {2, 0x3D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINUD": {2, 0x3B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXUD": {2, 0x3F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMINUQ": {2, 0x3B, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPMAXUQ": {2, 0x3F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSLLVD": {2, 0x47, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSLLVQ": {2, 0x47, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSRLVD": {2, 0x45, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSRLVQ": {2, 0x45, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSRAVD": {2, 0x46, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
"VPSRAVQ": {2, 0x46, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
|
||
// EVEX forms of instructions that also exist in VEX (selected when a ZMM
|
||
// or K register, or indices 16–31, demand EVEX).
|
||
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.66.0F — immediate shift (VPSLLD /6).
|
||
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
|
||
|
||
// EVEX.F3.0F38.W0 — narrowing stores: reg = wide source, rm = narrow
|
||
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
|
||
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||
}
|
||
|
||
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
|
||
// depends on the source kind — a GPR source uses opReg, a memory source uses
|
||
// opMem with a disp8×N of n.
|
||
type evexBcastSpec struct {
|
||
mapSel int
|
||
opReg byte
|
||
opMem byte
|
||
w int
|
||
n int
|
||
}
|
||
|
||
var evexBcastTable = map[string]evexBcastSpec{
|
||
// EVEX.128/256/512.66.0F38 — broadcast a dword/qword to all lanes.
|
||
"VPBROADCASTD": {2, 0x7C, 0x58, 0, 4},
|
||
"VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8},
|
||
}
|
||
|
||
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
|
||
// move table).
|
||
type evexMoveSpec struct {
|
||
mapSel int
|
||
pp int
|
||
load byte // r/m → vector
|
||
store byte // vector → r/m
|
||
w int
|
||
n [3]int
|
||
}
|
||
|
||
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
|
||
var evexMoveTable = map[string]evexMoveSpec{
|
||
// EVEX.128/256/512.F3.0F.W0 — unaligned integer move.
|
||
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.F3.0F.W1 — unaligned qword move.
|
||
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.F2.0F.W0 — unaligned byte move (byte/word moves use the
|
||
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
|
||
// semantics).
|
||
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.F2.0F.W1 — unaligned word move (shares the qword
|
||
// encoding).
|
||
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
||
// EVEX.128/256/512.66.0F.W1 — unaligned packed double move.
|
||
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
|
||
}
|
||
|
||
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
||
func isEvex(mnemUpper string) bool {
|
||
if _, ok := evexTable[mnemUpper]; ok {
|
||
return true
|
||
}
|
||
if _, ok := evexBcastTable[mnemUpper]; ok {
|
||
return true
|
||
}
|
||
_, ok := evexMoveTable[mnemUpper]
|
||
return ok
|
||
}
|
||
|
||
// evexRequired reports whether the operands force the EVEX encoding of a
|
||
// mnemonic that also has a VEX form: ZMM and K registers do, and so do
|
||
// register indices 16–31, which only EVEX can represent (X16–Y31 exist
|
||
// solely under AVX-512).
|
||
func evexRequired(upper string, ops []Operand) bool {
|
||
_, inVex := vexTable[upper]
|
||
_, inVexMove := vexMoveTable[upper]
|
||
if !inVex && !inVexMove {
|
||
return true // EVEX-only mnemonic
|
||
}
|
||
for _, op := range ops {
|
||
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
|
||
return true
|
||
}
|
||
}
|
||
return false
|
||
}
|
||
|
||
// stripEvexSuffix splits a ".Z" zeroing suffix off the mnemonic. It is the
|
||
// only EVEX suffix supported; Go writes masking as an explicit K operand, not
|
||
// a suffix.
|
||
func stripEvexSuffix(mnem string) (base string, zeroing bool, err error) {
|
||
i := strings.LastIndexByte(mnem, '.')
|
||
if i < 0 {
|
||
return mnem, false, nil
|
||
}
|
||
if mnem[i+1:] == "Z" {
|
||
return mnem[:i], true, nil
|
||
}
|
||
return "", false, fmt.Errorf("unsupported EVEX suffix %q", mnem[i+1:])
|
||
}
|
||
|
||
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
|
||
// returning the remaining operands and the mask index. K0 is not a usable
|
||
// mask (aaa = 0 means "no mask"), matching the assembler.
|
||
func splitMask(ops []Operand) ([]Operand, int, error) {
|
||
var rest []Operand
|
||
mask := 0
|
||
for _, op := range ops {
|
||
if r, ok := op.(Reg); ok && r.mask {
|
||
if mask != 0 {
|
||
return nil, 0, fmt.Errorf("at most one mask register operand")
|
||
}
|
||
if r.idx == 0 {
|
||
return nil, 0, fmt.Errorf("K0 is not a usable mask register")
|
||
}
|
||
mask = r.idx
|
||
continue
|
||
}
|
||
rest = append(rest, op)
|
||
}
|
||
return rest, mask, nil
|
||
}
|
||
|
||
// encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The
|
||
// mask, when present, is an explicit K1–K7 operand anywhere among the
|
||
// operands; zeroing comes from the .Z mnemonic suffix and requires a mask.
|
||
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, zeroing bool) error {
|
||
// Mask-destination comparisons (VPCMPEQD …, K1): the last operand is the
|
||
// destination K register, and any mask sits among the preceding operands.
|
||
if spec, ok := evexTable[mnemUpper]; ok && spec.form == vexNDS3 && len(ops) > 0 {
|
||
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
||
rest, mask, err := splitMask(ops[:len(ops)-1])
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if zeroing && mask == 0 {
|
||
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper)
|
||
}
|
||
return e.encodeEvexNDS3(spec, append(rest, dst), mask, zeroing)
|
||
}
|
||
}
|
||
|
||
rest, mask, err := splitMask(ops)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if zeroing && mask == 0 {
|
||
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper)
|
||
}
|
||
ops = rest
|
||
|
||
if bs, ok := evexBcastTable[mnemUpper]; ok {
|
||
return e.encodeEvexBcast(bs, ops, mask, zeroing)
|
||
}
|
||
if ms, ok := evexMoveTable[mnemUpper]; ok {
|
||
return e.encodeEvexMove(mnemUpper, ms, ops, mask, zeroing)
|
||
}
|
||
spec, ok := evexTable[mnemUpper]
|
||
if !ok {
|
||
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
|
||
}
|
||
switch spec.form {
|
||
case vexNDS3:
|
||
return e.encodeEvexNDS3(spec, ops, mask, zeroing)
|
||
case vexRM:
|
||
return e.encodeEvexRM(spec, ops, mask, zeroing)
|
||
case vexRMRev:
|
||
return e.encodeEvexRMRev(spec, ops, mask, zeroing)
|
||
case vexImmRM:
|
||
return e.encodeEvexImmRM(spec, ops, mask, zeroing)
|
||
case vexShiftImm:
|
||
return e.encodeEvexShiftImm(spec, ops, mask, zeroing)
|
||
case vexNDS3Imm:
|
||
return e.encodeEvexNDS3Imm(spec, ops, mask, zeroing)
|
||
case vexExtract:
|
||
return e.encodeEvexExtract(spec, ops, mask, zeroing)
|
||
case vexRMSrcLen:
|
||
return e.encodeEvexRMSrcLen(spec, ops, mask, zeroing)
|
||
}
|
||
return fmt.Errorf("unhandled EVEX form for %s", mnemUpper)
|
||
}
|
||
|
||
// encodeEvexNDS3 encodes the three-operand NDS form: OP src2, src1, dst. The
|
||
// destination may be an opmask register (VPCMPEQD), in which case the vector
|
||
// length comes from the sources.
|
||
func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("EVEX NDS instruction expects 3 operands, got %d", len(ops))
|
||
}
|
||
src2, src1, dst := ops[0], ops[1], ops[2]
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || (!dstReg.isVec() && !dstReg.mask) {
|
||
return fmt.Errorf("EVEX destination must be a vector or mask register")
|
||
}
|
||
vvvvReg, ok := src1.(Reg)
|
||
if !ok || !vvvvReg.isVec() {
|
||
return fmt.Errorf("EVEX vvvv operand must be a vector register")
|
||
}
|
||
ll := dstReg.vecLenBit()
|
||
if dstReg.mask {
|
||
ll = vvvvReg.vecLenBit()
|
||
if r, ok := src2.(Reg); ok && r.isVec() {
|
||
ll = r.vecLenBit()
|
||
}
|
||
}
|
||
return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, zeroing)
|
||
}
|
||
|
||
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
|
||
// no vvvv), e.g. VCVTQQ2PD.
|
||
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || !dstReg.isVec() {
|
||
return fmt.Errorf("EVEX destination must be a vector register")
|
||
}
|
||
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, zeroing)
|
||
}
|
||
|
||
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
|
||
// (reg = dst, rm = src, imm8), e.g. VPSHUFD.
|
||
func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||
}
|
||
imm, src, dst := ops[0], ops[1], ops[2]
|
||
immVal, ok := imm.(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("shuffle control must be an immediate")
|
||
}
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || !dstReg.isVec() {
|
||
return fmt.Errorf("shuffle destination must be a vector register")
|
||
}
|
||
ll := dstReg.vecLenBit()
|
||
if r, ok := src.(Reg); ok && r.isVec() {
|
||
ll = r.vecLenBit()
|
||
}
|
||
immByte, err := imm8(int64(immVal))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if err := e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, zeroing); err != nil {
|
||
return err
|
||
}
|
||
e.out = append(e.out, immByte)
|
||
return nil
|
||
}
|
||
|
||
// encodeEvexShiftImm encodes an immediate shift: OP $imm, src, dst
|
||
// (ModRM.reg = /digit, vvvv = dst, rm = src, imm8), e.g. VPSRAD $31, Z3, Z5.
|
||
func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("EVEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||
}
|
||
imm, src, dst := ops[0], ops[1], ops[2]
|
||
immVal, ok := imm.(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("shift count must be an immediate")
|
||
}
|
||
srcReg, ok := src.(Reg)
|
||
if !ok || !srcReg.isVec() {
|
||
return fmt.Errorf("shift source must be a vector register")
|
||
}
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || !dstReg.isVec() {
|
||
return fmt.Errorf("shift destination must be a vector register")
|
||
}
|
||
immByte, err := imm8(int64(immVal))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, zeroing); err != nil {
|
||
return err
|
||
}
|
||
e.out = append(e.out, immByte)
|
||
return nil
|
||
}
|
||
|
||
// encodeEvexNDS3Imm encodes OP $imm, src2, src1, dst (reg=dst, vvvv=src1,
|
||
// rm=src2, imm8), e.g. VALIGND.
|
||
func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||
if len(ops) != 4 {
|
||
return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops))
|
||
}
|
||
imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
|
||
immVal, ok := imm.(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("shuffle control must be an immediate")
|
||
}
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || !dstReg.isVec() {
|
||
return fmt.Errorf("destination must be a vector register")
|
||
}
|
||
vvvvReg, ok := src1.(Reg)
|
||
if !ok || !vvvvReg.isVec() {
|
||
return fmt.Errorf("second source must be a vector register")
|
||
}
|
||
immByte, err := imm8(int64(immVal))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, vvvvReg.idx, src2, mask, zeroing); err != nil {
|
||
return err
|
||
}
|
||
e.out = append(e.out, immByte)
|
||
return nil
|
||
}
|
||
|
||
// encodeEvexExtract encodes OP $imm, zsrc, ydst (reg=ZMM source, rm=YMM/memory
|
||
// destination, imm8), e.g. VEXTRACTI64X4.
|
||
func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("extract expects 3 operands ($imm, zsrc, ydst), got %d", len(ops))
|
||
}
|
||
imm, src, dst := ops[0], ops[1], ops[2]
|
||
immVal, ok := imm.(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("extract lane must be an immediate")
|
||
}
|
||
srcReg, ok := src.(Reg)
|
||
if !ok || !srcReg.isVec() {
|
||
return fmt.Errorf("extract source must be a vector register")
|
||
}
|
||
immByte, err := imm8(int64(immVal))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, zeroing); err != nil {
|
||
return err
|
||
}
|
||
e.out = append(e.out, immByte)
|
||
return nil
|
||
}
|
||
|
||
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
|
||
// the store-form opcode (reg = source, rm = destination), matching the Go
|
||
// assembler.
|
||
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, zeroing bool) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
srcReg, srcIsVec := vecReg(src)
|
||
dstReg, dstIsVec := vecReg(dst)
|
||
|
||
op := ms.store
|
||
var reg Reg
|
||
var rm Operand
|
||
switch {
|
||
case srcIsVec && dstIsVec:
|
||
reg, rm = srcReg, dst
|
||
case srcIsVec:
|
||
if !memOperand(dst) {
|
||
return fmt.Errorf("%s: invalid destination operand", mnem)
|
||
}
|
||
reg, rm = srcReg, dst
|
||
case dstIsVec:
|
||
if !memOperand(src) {
|
||
return fmt.Errorf("%s: invalid source operand", mnem)
|
||
}
|
||
op = ms.load
|
||
reg, rm = dstReg, src
|
||
default:
|
||
return fmt.Errorf("%s needs a vector register operand", mnem)
|
||
}
|
||
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
|
||
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, zeroing)
|
||
}
|
||
|
||
// encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
|
||
// the destination always XMM and the length fixed by the mnemonic — the
|
||
// single valid slot of spec.n names the vector length (and the disp8×N
|
||
// multiplier) a register or memory source encodes.
|
||
func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || !dstReg.isVec() {
|
||
return fmt.Errorf("EVEX destination must be a vector register")
|
||
}
|
||
ll, err := soleLen(spec.n)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, zeroing)
|
||
}
|
||
|
||
// soleLen returns the vector-length index of the single valid slot of n —
|
||
// the length a length-fixed mnemonic (the EVEX conversion spellings) encodes
|
||
// regardless of its operands.
|
||
func soleLen(n [3]int) (int, error) {
|
||
ll := -1
|
||
for i, v := range n {
|
||
if v == 0 {
|
||
continue
|
||
}
|
||
if ll >= 0 {
|
||
return 0, fmt.Errorf("ambiguous vector-length table %v", n)
|
||
}
|
||
ll = i
|
||
}
|
||
if ll < 0 {
|
||
return 0, fmt.Errorf("empty vector-length table")
|
||
}
|
||
return ll, nil
|
||
}
|
||
|
||
// memOperand reports whether op is a memory reference (including a
|
||
// static-symbol reference).
|
||
func memOperand(op Operand) bool {
|
||
switch op.(type) {
|
||
case Mem, sbMem:
|
||
return true
|
||
}
|
||
return false
|
||
}
|
||
|
||
// encodeEvexRMRev encodes the narrowing-store form: OP src, dst with the wide
|
||
// source in the reg field and the narrow destination in r/m (VPMOVDW/QD).
|
||
func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("EVEX store instruction expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
srcReg, ok := src.(Reg)
|
||
if !ok || !srcReg.isVec() {
|
||
return fmt.Errorf("EVEX source must be a vector register")
|
||
}
|
||
return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, zeroing)
|
||
}
|
||
|
||
// encodeEvexBcast encodes VPBROADCASTD/Q: OP src, dst with the GPR or memory
|
||
// source broadcast to every lane of the vector destination.
|
||
func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, zeroing bool) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("broadcast expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || !dstReg.isVec() {
|
||
return fmt.Errorf("broadcast destination must be a vector register")
|
||
}
|
||
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
|
||
switch src.(type) {
|
||
case Mem, sbMem:
|
||
spec.opcode = bs.opMem
|
||
spec.n = [3]int{bs.n, bs.n, bs.n}
|
||
case Reg:
|
||
spec.opcode = bs.opReg
|
||
default:
|
||
return fmt.Errorf("broadcast source must be a register or memory")
|
||
}
|
||
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, zeroing)
|
||
}
|
||
|
||
// emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement
|
||
// (disp8×N compressed) for the given precomputed fields. regIdx is the
|
||
// unextended reg-field register index, or a /digit (0–7); vvvvIdx is the
|
||
// vvvv register index, or -1 when unused. mask (K1–K7, 0 = unmasked) and
|
||
// zeroing fill the aaa and z bits of the P2 byte.
|
||
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, zeroing bool) error {
|
||
if ll > 2 {
|
||
return fmt.Errorf("invalid vector length")
|
||
}
|
||
// reg-field extension bits (R̄, R'̄), inverted.
|
||
rBar, rPrimeBar := 1, 1
|
||
if regIdx&8 != 0 {
|
||
rBar = 0
|
||
}
|
||
if regIdx&16 != 0 {
|
||
rPrimeBar = 0
|
||
}
|
||
// vvvv (inverted) and its extension bit V'̄.
|
||
vBar, vPrimeBar := 15, 1
|
||
if vvvvIdx >= 0 {
|
||
vBar = 15 - (vvvvIdx & 15)
|
||
if vvvvIdx&16 != 0 {
|
||
vPrimeBar = 0
|
||
}
|
||
}
|
||
|
||
var modrm, sib int
|
||
var disp []byte
|
||
xBar, bBar := 1, 1
|
||
var sb *sbRef
|
||
switch r := rm.(type) {
|
||
case Reg:
|
||
// ModRM.mod = 11: rm[3] extends via B̄, and rm[4] via X̄ (the EVEX
|
||
// register-register quirk).
|
||
modrm = 0xC0 | (regIdx&7)<<3 | (r.idx & 7)
|
||
sib = -1
|
||
if r.idx&8 != 0 {
|
||
bBar = 0
|
||
}
|
||
if r.idx&16 != 0 {
|
||
xBar = 0
|
||
}
|
||
if r.idx&16 != 0 {
|
||
xBar = 0
|
||
}
|
||
case Mem:
|
||
var err error
|
||
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
|
||
if err != nil {
|
||
return err
|
||
}
|
||
// An indexed memory operand carries index[4] in V'̄ (Go folds it
|
||
// together with vvvv[4] into the same bit).
|
||
if r.HasIndex && r.Index.idx&16 != 0 {
|
||
vPrimeBar = 0
|
||
}
|
||
case sbMem:
|
||
// RIP-relative static-symbol reference; disp32 patched at link time
|
||
// (no disp8 scaling for RIP-relative addressing).
|
||
modrm = (regIdx&7)<<3 | 0x05
|
||
sib = -1
|
||
disp = le32(0)
|
||
sb = &sbRef{name: r.name, addend: r.addend}
|
||
default:
|
||
return fmt.Errorf("invalid EVEX r/m operand")
|
||
}
|
||
|
||
z := 0
|
||
if zeroing {
|
||
z = 1
|
||
}
|
||
p0 := byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4 | spec.mapSel)
|
||
p1 := byte(spec.w<<7 | vBar<<3 | 1<<2 | spec.pp)
|
||
p2 := byte(z<<7 | ll<<5 | vPrimeBar<<3 | mask) // z, L'L, b=0, V', aaa
|
||
e.out = append(e.out, 0x62, p0, p1, p2, spec.opcode, byte(modrm))
|
||
if sib >= 0 {
|
||
e.out = append(e.out, byte(sib))
|
||
}
|
||
if sb != nil {
|
||
e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend})
|
||
}
|
||
e.out = append(e.out, disp...)
|
||
return nil
|
||
}
|
||
|
||
// memComponentsEvex computes the ModR/M byte (with the given reg field), the
|
||
// SIB byte (-1 if none), the displacement bytes and the (inverted sense)
|
||
// index/base extension bits for an EVEX memory operand. The displacement is
|
||
// compressed to disp8×N when it is a multiple of n and the quotient fits a
|
||
// signed byte; otherwise a full disp32 is used.
|
||
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
|
||
sib = -1
|
||
xBar, bBar = 1, 1 // inverted bits: 1 = no extension
|
||
if !m.HasBase && !m.HasIndex {
|
||
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
|
||
}
|
||
|
||
needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4)
|
||
|
||
var mod int
|
||
switch {
|
||
case !m.HasBase:
|
||
mod = 0
|
||
disp = le32(m.Disp)
|
||
case m.Base.idx&7 == 5 && m.Disp == 0:
|
||
mod = 1
|
||
disp = []byte{0}
|
||
case m.Disp == 0:
|
||
mod = 0
|
||
case n > 0 && m.Disp%int64(n) == 0 && m.Disp/int64(n) >= -128 && m.Disp/int64(n) <= 127:
|
||
mod = 1
|
||
disp = []byte{byte(int8(m.Disp / int64(n)))}
|
||
default:
|
||
mod = 2
|
||
disp = le32(m.Disp)
|
||
}
|
||
|
||
if needSIB {
|
||
idxField := 4 // 100 = no index
|
||
if m.HasIndex {
|
||
idxField = m.Index.idx & 7
|
||
if m.Index.idx&8 != 0 {
|
||
xBar = 0
|
||
}
|
||
}
|
||
baseField := 5 // 101 = no base (with mod=00 → disp32)
|
||
if m.HasBase {
|
||
baseField = m.Base.idx & 7
|
||
if m.Base.idx&8 != 0 {
|
||
bBar = 0
|
||
}
|
||
}
|
||
return mod<<6 | regField<<3 | 0x04, scaleBits(m.Scale)<<6 | idxField<<3 | baseField, disp, xBar, bBar, nil
|
||
}
|
||
|
||
if m.Base.idx&8 != 0 {
|
||
bBar = 0
|
||
}
|
||
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil
|
||
}
|
||
|
||
// encodeKmovw encodes KMOVW, whose opcode depends on the operand direction:
|
||
// 90 (k/mem → K), 91 (K → mem), 92 (GPR → K), 93 (K → GPR); k → k uses 90.
|
||
func (e *enc) encodeKmovw(ops []Operand) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("KMOVW expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
srcReg, srcIsReg := src.(Reg)
|
||
dstReg, dstIsReg := dst.(Reg)
|
||
srcK := srcIsReg && srcReg.mask
|
||
dstK := dstIsReg && dstReg.mask
|
||
spec := vexSpec{mapSel: 1, w: 0, pp: 0, opdigit: -1}
|
||
switch {
|
||
case srcK && dstK:
|
||
spec.opcode = 0x90 // k ← k: reg = dst, rm = src
|
||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
||
case srcK && dstIsReg:
|
||
spec.opcode = 0x93 // GPR ← k: reg = dst, rm = src
|
||
rBit := 0
|
||
if dstReg.idx >= 8 {
|
||
rBit = 1
|
||
}
|
||
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15, src)
|
||
case srcK:
|
||
if _, ok := dst.(Mem); !ok {
|
||
return fmt.Errorf("KMOVW: invalid destination operand")
|
||
}
|
||
spec.opcode = 0x91 // mem ← k: reg = src, rm = dst
|
||
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
|
||
case dstK:
|
||
spec.opcode = 0x92 // k ← GPR/mem: reg = dst, rm = src
|
||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
||
}
|
||
return fmt.Errorf("KMOVW requires a K register operand")
|
||
}
|