style: replace em and en dashes across sources

This commit is contained in:
2026-09-14 18:22:18 +02:00
parent 2db563be07
commit 1691c81095
27 changed files with 229 additions and 229 deletions
+1 -1
View File
@@ -29,7 +29,7 @@ func arm64Registers() []Register {
regs = append(regs, Register{Name: name, Class: class, Desc: desc}) regs = append(regs, Register{Name: name, Class: class, Desc: desc})
} }
// General-purpose integer registers R0–R30. // General-purpose integer registers R0-R30.
for i := 0; i <= 30; i++ { for i := 0; i <= 30; i++ {
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register") add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
} }
+3 -3
View File
@@ -9,7 +9,7 @@ package asm
// an opcode constant, and the format selects the bit layout. The opcode // an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own arm64 // constants and formats are transcribed from the Go toolchain's own arm64
// backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm` // backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm`
// exactly — the ground-truth oracle for the verify suite. // exactly, the ground-truth oracle for the verify suite.
// //
// All AArch64 instructions are 32 bits, little-endian. The formats used here // All AArch64 instructions are 32 bits, little-endian. The formats used here
// (per the ARM Architecture Reference Manual): // (per the ARM Architecture Reference Manual):
@@ -28,7 +28,7 @@ package asm
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd // ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
// arm64RegNum returns the 5-bit register number for an AArch64 register name: // arm64RegNum returns the 5-bit register number for an AArch64 register name:
// R0–R30 (integer), F0–F31 (floating point), and the ABI aliases the // R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
// runtime's assembly uses. Returns -1 for an unrecognised name. // runtime's assembly uses. Returns -1 for an unrecognised name.
func arm64RegNum(name string) int { func arm64RegNum(name string) int {
switch name { switch name {
@@ -99,7 +99,7 @@ func arm64RegNum(name string) int {
case "SP": case "SP":
return 31 // SP and ZR share encoding 31; context determines meaning return 31 // SP and ZR share encoding 31; context determines meaning
} }
// F0–F31. // F0-F31.
if len(name) >= 1 && name[0] == 'F' { if len(name) >= 1 && name[0] == 'F' {
n := 0 n := 0
for i := 1; i < len(name); i++ { for i := 1; i < len(name); i++ {
+3 -3
View File
@@ -12,7 +12,7 @@ import (
// Image: a .text section holding the function bodies, a .data section // Image: a .text section holding the function bodies, a .data section
// holding the GLOBL initialisers, a symbol table with one symbol per TEXT // holding the GLOBL initialisers, a symbol table with one symbol per TEXT
// and GLOBL (file-local <> symbols are STB_LOCAL, the rest STB_GLOBAL), and // and GLOBL (file-local <> symbols are STB_LOCAL, the rest STB_GLOBAL), and
// a .rela.text relocation table — one R_X86_64_PC32 entry per static-symbol // a .rela.text relocation table, one R_X86_64_PC32 entry per static-symbol
// reference, internal references resolving against the local data symbols // reference, internal references resolving against the local data symbols
// and external ones against undefined globals. The output links with the // and external ones against undefined globals. The output links with the
// system toolchain (cc/ld) the way a hand-assembled .o would. // system toolchain (cc/ld) the way a hand-assembled .o would.
@@ -71,7 +71,7 @@ func (img *Image) ELFObject() ([]byte, error) {
// Build the symbol table: the null entry and the two section symbols // Build the symbol table: the null entry and the two section symbols
// come first, then the local symbols (static TEXT and GLOBL), then the // come first, then the local symbols (static TEXT and GLOBL), then the
// globals (exported TEXT and GLOBL, and the undefined externals) — ELF // globals (exported TEXT and GLOBL, and the undefined externals), ELF
// requires every local to precede every global, and sh_info records the // requires every local to precede every global, and sh_info records the
// boundary. symIdx maps a symbol name to its index for the relocations. // boundary. symIdx maps a symbol name to its index for the relocations.
var locals, globals []elfSym var locals, globals []elfSym
@@ -218,7 +218,7 @@ func (img *Image) ELFObject() ([]byte, error) {
shstrOff := len(out) shstrOff := len(out)
out = append(out, stSections.bytes()...) out = append(out, stSections.bytes()...)
// DWARF debug sections (no relocations — the linker resolves DWARF fixups). // DWARF debug sections (no relocations, the linker resolves DWARF fixups).
dwAlign := func(n int) { dwAlign := func(n int) {
for len(out)%n != 0 { for len(out)%n != 0 {
out = append(out, 0) out = append(out, 0)
+1 -1
View File
@@ -282,7 +282,7 @@ func dwarfBuildFrameSection(img *Image) []byte {
// Patch CIE length. // Patch CIE length.
le.PutUint32(b[cieStart:], uint32(len(b)-cieStart-4)) le.PutUint32(b[cieStart:], uint32(len(b)-cieStart-4))
// FDEs (Frame Description Entries) — one per function. // FDEs (Frame Description Entries), one per function.
for _, fn := range img.Funcs { for _, fn := range img.Funcs {
fdeStart := len(b) fdeStart := len(b)
b = append(b, 0, 0, 0, 0) // length (placeholder) b = append(b, 0, 0, 0, 0) // length (placeholder)
+1 -1
View File
@@ -284,7 +284,7 @@ func setRM(i *instr, reg Reg, rm Operand, opSize int) error {
} }
// setRMDigit fills in the ModR/M for an instruction whose reg field is an // setRMDigit fills in the ModR/M for an instruction whose reg field is an
// opcode /digit extension (0–7), which carries none of the register REX rules. // opcode /digit extension (0-7), which carries none of the register REX rules.
func setRMDigit(i *instr, digit int, rm Operand, opSize int) error { func setRMDigit(i *instr, digit int, rm Operand, opSize int) error {
return setRMReg(i, digit, false, false, rm, opSize) return setRMReg(i, digit, false, false, rm, opSize)
} }
+82 -82
View File
@@ -9,10 +9,10 @@ import (
) )
// This file implements EVEX (AVX-512) instruction encoding: the four-byte // This file implements EVEX (AVX-512) instruction encoding: the four-byte
// EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the // EVEX prefix with 5-bit vector register fields (Z0-Z31, X/Y 16-31), the
// compressed disp8×N displacement, and the operand shapes the go-flac // compressed disp8×N displacement, and the operand shapes the go-flac
// AVX-512 kernels use plus the common floating-point and conversion set. // AVX-512 kernels use plus the common floating-point and conversion set.
// Masking follows the Go assembler's spelling: an explicit K1–K7 operand // Masking follows the Go assembler's spelling: an explicit K1-K7 operand
// anywhere among the operands (merging) plus a ".Z" mnemonic suffix for // anywhere among the operands (merging) plus a ".Z" mnemonic suffix for
// zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are // zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are
// supported too. // supported too.
@@ -36,7 +36,7 @@ type evexSpec struct {
// are taken from the Go assembler's opcode tables, which are authoritative // are taken from the Go assembler's opcode tables, which are authoritative
// for byte-for-byte agreement. // for byte-for-byte agreement.
var evexTable = map[string]evexSpec{ var evexTable = map[string]evexSpec{
// EVEX.128/256/512.66.0F — integer arithmetic / logic, NDS form. // EVEX.128/256/512.66.0F, integer arithmetic / logic, NDS form.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -48,25 +48,25 @@ var evexTable = map[string]evexSpec{
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — packed double arithmetic. // EVEX.128/256/512.66.0F.W1, packed double arithmetic.
"VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0 — packed single arithmetic. // EVEX.128/256/512.0F.W0, packed single arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, "VADDPS": {1, 0x58, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, "VMULPS": {1, 0x59, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, "VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, "VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, "VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, "VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — packed double unpack. // EVEX.128/256/512.66.0F.W1, packed double unpack.
"VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128.F2.0F.W1 — scalar double arithmetic (the packed opcodes with // EVEX.128.F2.0F.W1, scalar double arithmetic (the packed opcodes with
// an F2 pp; the EVEX forms exist for masked and zeroing use). The // an F2 pp; the EVEX forms exist for masked and zeroing use). The
// memory operand is a single double, so disp8×N = 8. // memory operand is a single double, so disp8×N = 8.
"VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, "VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
@@ -76,7 +76,7 @@ var evexTable = map[string]evexSpec{
"VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, "VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, "VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128.F3.0F.W0 — scalar single arithmetic (disp8×N = 4). // EVEX.128.F3.0F.W0, scalar single arithmetic (disp8×N = 4).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
@@ -84,38 +84,38 @@ var evexTable = map[string]evexSpec{
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.512.66.0F3A — align (NDS + imm8). // EVEX.512.66.0F3A, align (NDS + imm8).
"VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F — immediate shift (VPSRAD /4). // EVEX.128/256/512.66.0F, immediate shift (VPSRAD /4).
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}}, "VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — variable shift with an XMM count (VPSRAQ; // EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
// the W bit distinguishes it from VPSRAD's E2 form). // the W bit distinguishes it from VPSRAD's E2 form).
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst, // EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
// rm=src, no vvvv). // rm=src, no vvvv).
"VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, "VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1 — duplicate the low double (reg=dst, // EVEX.128/256/512.F2.0F.W1, duplicate the low double (reg=dst,
// rm=src, no vvvv): a 128-bit destination reads a single double from // rm=src, no vvvv): a 128-bit destination reads a single double from
// memory (disp8×8), the wider ones read the full operand. // memory (disp8×8), the wider ones read the full operand.
"VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}}, "VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}},
// EVEX.128/256/512.0F.W0 — signed dword to packed single (reg=dst, // EVEX.128/256/512.0F.W0, signed dword to packed single (reg=dst,
// rm=src, no vvvv, no mandatory prefix — as in the VEX form). // rm=src, no vvvv, no mandatory prefix, as in the VEX form).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}}, "VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0 — packed single to packed double: the // EVEX.128/256/512.0F.W0, packed single to packed double: the
// destination is twice the source width and sets the length; disp8×N // destination is twice the source width and sets the length; disp8×N
// follows the narrow memory source. No F3 prefix: the Go assembler // follows the narrow memory source. No F3 prefix: the Go assembler
// emits this instruction with pp = 00 (Intel's maps would call that // emits this instruction with pp = 00 (Intel's maps would call that
// undefined) and gasm reproduces the Go assembler's bytes — its machine // undefined) and gasm reproduces the Go assembler's bytes, its machine
// code is the oracle, not the manual. // code is the oracle, not the manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}}, "VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.128/256/512.F3.0F.W0 — signed dword to packed double (the EVEX // EVEX.128/256/512.F3.0F.W0, signed dword to packed double (the EVEX
// form of the VEX instruction; the destination sets the length, disp8×N // form of the VEX instruction; the destination sets the length, disp8×N
// follows the narrow memory source). // follows the narrow memory source).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}}, "VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
// EVEX packed double → dword conversions: the source is the wide // EVEX packed double → dword conversions: the source is the wide
// operand and the mnemonic fixes the length — the bare names are // operand and the mnemonic fixes the length, the bare names are
// 512-bit only (ZMM source, XMM destination), the X/Y spellings are // 512-bit only (ZMM source, XMM destination), the X/Y spellings are
// EVEX-128/256. Exactly one slot of n is valid; it names the vector // EVEX-128/256. Exactly one slot of n is valid; it names the vector
// length (and the disp8×N multiplier) a register or memory source // length (and the disp8×N multiplier) a register or memory source
@@ -127,7 +127,7 @@ var evexTable = map[string]evexSpec{
"VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}}, "VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}}, "VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.66.0F3A — ternary logic and lane shuffles (NDS + imm8). // EVEX.66.0F3A, ternary logic and lane shuffles (NDS + imm8).
"VPTERNLOGD": {3, 0x25, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VPTERNLOGD": {3, 0x25, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPTERNLOGQ": {3, 0x25, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VPTERNLOGQ": {3, 0x25, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFI32X4": {3, 0x43, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VSHUFI32X4": {3, 0x43, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
@@ -136,11 +136,11 @@ var evexTable = map[string]evexSpec{
"VSHUFF64X2": {3, 0x23, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VSHUFF64X2": {3, 0x23, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F — the EVEX forms of the VEX two-source shuffle. // EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F3A — lane insert ($imm, xsrc, zsrc1, zdst). // EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}}, "VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTF32X8": {3, 0x1A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}}, "VINSERTF32X8": {3, 0x1A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
"VINSERTF64X2": {3, 0x18, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}}, "VINSERTF64X2": {3, 0x18, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
@@ -150,7 +150,7 @@ var evexTable = map[string]evexSpec{
"VINSERTI64X2": {3, 0x38, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}}, "VINSERTI64X2": {3, 0x38, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTI64X4": {3, 0x3A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}}, "VINSERTI64X4": {3, 0x3A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
// EVEX.66.0F3A — lane extract (reg=source, rm=XMM/YMM destination, // EVEX.66.0F3A, lane extract (reg=source, rm=XMM/YMM destination,
// imm8). // imm8).
"VEXTRACTF32X4": {3, 0x19, 0, 1, -1, vexExtract, [3]int{0, 16, 16}}, "VEXTRACTF32X4": {3, 0x19, 0, 1, -1, vexExtract, [3]int{0, 16, 16}},
"VEXTRACTF32X8": {3, 0x1B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}}, "VEXTRACTF32X8": {3, 0x1B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
@@ -159,14 +159,14 @@ var evexTable = map[string]evexSpec{
"VEXTRACTI32X8": {3, 0x3B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}}, "VEXTRACTI32X8": {3, 0x3B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTI64X2": {3, 0x39, 1, 1, -1, vexExtract, [3]int{0, 16, 16}}, "VEXTRACTI64X2": {3, 0x39, 1, 1, -1, vexExtract, [3]int{0, 16, 16}},
// EVEX.66.0F — compare with an opmask destination ($imm, src2, src1, // EVEX.66.0F, compare with an opmask destination ($imm, src2, src1,
// kdst): NDS3Imm with the K register in the reg field. // kdst): NDS3Imm with the K register in the reg field.
"VCMPPD": {1, 0xC2, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VCMPPD": {1, 0xC2, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VCMPPS": {1, 0xC2, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VCMPPS": {1, 0xC2, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}}, "VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}}, "VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}},
// EVEX.66.0F3A — integer compares with an opmask destination, the same // EVEX.66.0F3A, integer compares with an opmask destination, the same
// NDS3Imm-with-k-reg shape as the floating-point compares; W selects the // NDS3Imm-with-k-reg shape as the floating-point compares; W selects the
// operand width (byte/word vs dword/qword), the opcode the signedness. // operand width (byte/word vs dword/qword), the opcode the signedness.
// The memory form takes a full vector, so disp8×N is 16/32/64. // The memory form takes a full vector, so disp8×N is 16/32/64.
@@ -179,7 +179,7 @@ var evexTable = map[string]evexSpec{
"VPCMPQ": {3, 0x1F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VPCMPQ": {3, 0x1F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F38 — permutes (NDS form). // EVEX.66.0F38, permutes (NDS form).
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -188,7 +188,7 @@ var evexTable = map[string]evexSpec{
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F — the wider integer set (NDS form). // EVEX.66.0F, the wider integer set (NDS form).
"VPMADDWD": {1, 0xF5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMADDWD": {1, 0xF5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -199,34 +199,34 @@ var evexTable = map[string]evexSpec{
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKUSDW": {2, 0x2B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPACKUSDW": {2, 0x2B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F38 — absolute values and replicating moves (reg=dst, // EVEX.66.0F38, absolute values and replicating moves (reg=dst,
// rm=src). // rm=src).
"VPABSB": {2, 0x1C, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VPABSB": {2, 0x1C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSW": {2, 0x1D, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VPABSW": {2, 0x1D, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSD": {2, 0x1E, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VPABSD": {2, 0x1E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSQ": {2, 0x1F, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, "VPABSQ": {2, 0x1F, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.F3.0F — replicate even/odd singles. // EVEX.F3.0F, replicate even/odd singles.
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, "VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, "VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — sign/zero-extending moves; the memory source is the // EVEX.66.0F38, sign/zero-extending moves; the memory source is the
// narrow half (here byte to word). // narrow half (here byte to word).
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F — packed single conversions (reg=dst, rm=src). // EVEX.66.0F, packed single conversions (reg=dst, rm=src).
"VCVTPS2DQ": {1, 0x5B, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VCVTPS2DQ": {1, 0x5B, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTTPS2DQ": {1, 0x5B, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, "VCVTTPS2DQ": {1, 0x5B, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — broadcast a single/double to all lanes (reg=dst, // EVEX.66.0F38, broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; disp8×N is the element size). // rm=scalar memory; disp8×N is the element size).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM, [3]int{4, 4, 4}}, "VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VBROADCASTSD": {2, 0x19, 1, 1, -1, vexRM, [3]int{0, 8, 8}}, "VBROADCASTSD": {2, 0x19, 1, 1, -1, vexRM, [3]int{0, 8, 8}},
// EVEX.66.0F38 — expand loads (rm → vector register destination). // EVEX.66.0F38, expand loads (rm → vector register destination).
"VEXPANDPD": {2, 0x88, 1, 1, -1, vexRM, [3]int{8, 8, 8}}, "VEXPANDPD": {2, 0x88, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
"VEXPANDPS": {2, 0x88, 0, 1, -1, vexRM, [3]int{4, 4, 4}}, "VEXPANDPS": {2, 0x88, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VPEXPANDD": {2, 0x89, 0, 1, -1, vexRM, [3]int{4, 4, 4}}, "VPEXPANDD": {2, 0x89, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VPEXPANDQ": {2, 0x89, 1, 1, -1, vexRM, [3]int{8, 8, 8}}, "VPEXPANDQ": {2, 0x89, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
// EVEX.66.0F38 — compress stores (vector register source → rm), and the // EVEX.66.0F38, compress stores (vector register source → rm), and the
// remaining narrowing stores. // remaining narrowing stores.
"VCOMPRESSPD": {2, 0x8A, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}}, "VCOMPRESSPD": {2, 0x8A, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}},
"VCOMPRESSPS": {2, 0x8A, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}}, "VCOMPRESSPS": {2, 0x8A, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}},
@@ -235,7 +235,7 @@ var evexTable = map[string]evexSpec{
"VPMOVWB": {2, 0x30, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, "VPMOVWB": {2, 0x30, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQB": {2, 0x32, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}}, "VPMOVQB": {2, 0x32, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
// EVEX.66.0F — rotates (immediate form: /0 right, /1 left). // EVEX.66.0F, rotates (immediate form: /0 right, /1 left).
"VPRORD": {1, 0x72, 0, 1, 0, vexShiftImm, [3]int{16, 32, 64}}, "VPRORD": {1, 0x72, 0, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
"VPRORQ": {1, 0x72, 1, 1, 0, vexShiftImm, [3]int{16, 32, 64}}, "VPRORQ": {1, 0x72, 1, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
"VPROLD": {1, 0x72, 0, 1, 1, vexShiftImm, [3]int{16, 32, 64}}, "VPROLD": {1, 0x72, 0, 1, 1, vexShiftImm, [3]int{16, 32, 64}},
@@ -248,14 +248,14 @@ var evexTable = map[string]evexSpec{
"VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}}, "VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
"VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}}, "VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.66.0F38 — floating-point helpers, packed (reg=dst, rm=src). // EVEX.66.0F38, floating-point helpers, packed (reg=dst, rm=src).
"VRCP14PD": {2, 0x4C, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, "VRCP14PD": {2, 0x4C, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRCP14PS": {2, 0x4C, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VRCP14PS": {2, 0x4C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRSQRT14PD": {2, 0x4E, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, "VRSQRT14PD": {2, 0x4E, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRSQRT14PS": {2, 0x4E, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VRSQRT14PS": {2, 0x4E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VGETEXPPD": {2, 0x42, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, "VGETEXPPD": {2, 0x42, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VGETEXPPS": {2, 0x42, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VGETEXPPS": {2, 0x42, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — floating-point helpers, scalar (NDS form: src2 is // EVEX.66.0F38, floating-point helpers, scalar (NDS form: src2 is
// rm, src1 is vvvv, the XMM destination is reg). Like the scalar 0F3A // rm, src1 is vvvv, the XMM destination is reg). Like the scalar 0F3A
// forms, these take the 66 prefix; W selects double/single. // forms, these take the 66 prefix; W selects double/single.
"VRCP14SD": {2, 0x4D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}}, "VRCP14SD": {2, 0x4D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
@@ -264,13 +264,13 @@ var evexTable = map[string]evexSpec{
"VRSQRT14SS": {2, 0x4F, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}}, "VRSQRT14SS": {2, 0x4F, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
"VGETEXPSD": {2, 0x43, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}}, "VGETEXPSD": {2, 0x43, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VGETEXPSS": {2, 0x43, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}}, "VGETEXPSS": {2, 0x43, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.66.0F38 — scale by a power of two (NDS form). // EVEX.66.0F38, scale by a power of two (NDS form).
"VSCALEFPD": {2, 0x2C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VSCALEFPD": {2, 0x2C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSCALEFPS": {2, 0x2C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VSCALEFPS": {2, 0x2C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSCALEFSD": {2, 0x2D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}}, "VSCALEFSD": {2, 0x2D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VSCALEFSS": {2, 0x2D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}}, "VSCALEFSS": {2, 0x2D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.66.0F3A — packed round/getmant/reduce ($imm, src, dst: reg=dst, // EVEX.66.0F3A, packed round/getmant/reduce ($imm, src, dst: reg=dst,
// rm=src, imm8). // rm=src, imm8).
"VRNDSCALEPD": {3, 0x09, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}}, "VRNDSCALEPD": {3, 0x09, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VRNDSCALEPS": {3, 0x08, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}}, "VRNDSCALEPS": {3, 0x08, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
@@ -278,7 +278,7 @@ var evexTable = map[string]evexSpec{
"VGETMANTPS": {3, 0x26, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}}, "VGETMANTPS": {3, 0x26, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VREDUCEPD": {3, 0x56, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}}, "VREDUCEPD": {3, 0x56, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VREDUCEPS": {3, 0x56, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}}, "VREDUCEPS": {3, 0x56, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.66.0F3A — scalar round/getmant/reduce and fixup/range (NDS + // EVEX.66.0F3A, scalar round/getmant/reduce and fixup/range (NDS +
// imm8: $imm, src2, src1, dst). The scalar 0F3A forms all take the 66 // imm8: $imm, src2, src1, dst). The scalar 0F3A forms all take the 66
// prefix; W selects double/single. // prefix; W selects double/single.
"VRNDSCALESD": {3, 0x0B, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}}, "VRNDSCALESD": {3, 0x0B, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
@@ -296,7 +296,7 @@ var evexTable = map[string]evexSpec{
"VRANGESD": {3, 0x51, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}}, "VRANGESD": {3, 0x51, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VRANGESS": {3, 0x51, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}}, "VRANGESS": {3, 0x51, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
// EVEX.66.0F3A — floating-point class test ($imm, src, kdst): the // EVEX.66.0F3A, floating-point class test ($imm, src, kdst): the
// reg field carries the opmask destination. The packed forms carry an // reg field carries the opmask destination. The packed forms carry an
// explicit length in the mnemonic (X/Y/Z). // explicit length in the mnemonic (X/Y/Z).
"VFPCLASSPDX": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{16, 0, 0}}, "VFPCLASSPDX": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{16, 0, 0}},
@@ -308,7 +308,7 @@ var evexTable = map[string]evexSpec{
"VFPCLASSSD": {3, 0x67, 1, 1, -1, vexImmRM, [3]int{8, 0, 0}}, "VFPCLASSSD": {3, 0x67, 1, 1, -1, vexImmRM, [3]int{8, 0, 0}},
"VFPCLASSSS": {3, 0x67, 0, 1, -1, vexImmRM, [3]int{4, 0, 0}}, "VFPCLASSSS": {3, 0x67, 0, 1, -1, vexImmRM, [3]int{4, 0, 0}},
// EVEX — the remaining conversions. VCVTQQ2PS narrows (the 512-bit // EVEX, the remaining conversions. VCVTQQ2PS narrows (the 512-bit
// source sets the length); the rest follow the destination. // source sets the length); the rest follow the destination.
"VCVTQQ2PS": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}}, "VCVTQQ2PS": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2QQ": {1, 0x7B, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, "VCVTPD2QQ": {1, 0x7B, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
@@ -316,13 +316,13 @@ var evexTable = map[string]evexSpec{
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}}, "VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}}, "VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F38 — half-precision convert (half-width source). // EVEX.66.0F38, half-precision convert (half-width source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src, // EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
// rm=dst, imm8 — the extract layout). // rm=dst, imm8, the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}}, "VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
// EVEX — unsigned and truncating conversions. The PD sources are the // EVEX, unsigned and truncating conversions. The PD sources are the
// wide operand (the bare names are 512-bit only, the X/Y spellings fix // wide operand (the bare names are 512-bit only, the X/Y spellings fix
// the length); the PS/UQQ destinations are wide and follow the // the length); the PS/UQQ destinations are wide and follow the
// destination. // destination.
@@ -349,7 +349,7 @@ var evexTable = map[string]evexSpec{
"VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}}, "VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}}, "VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.66.0F38 — the remaining sign/zero-extending moves (narrow // EVEX.66.0F38, the remaining sign/zero-extending moves (narrow
// source; disp8×N follows its size). // source; disp8×N follows its size).
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}}, "VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}}, "VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
@@ -361,7 +361,7 @@ var evexTable = map[string]evexSpec{
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}}, "VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.F3.0F38 — the remaining narrowing stores (vector source in reg, // EVEX.F3.0F38, the remaining narrowing stores (vector source in reg,
// narrow destination in r/m): signed, unsigned and the D/Q truncations. // narrow destination in r/m): signed, unsigned and the D/Q truncations.
"VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}}, "VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}}, "VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
@@ -378,7 +378,7 @@ var evexTable = map[string]evexSpec{
"VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}}, "VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}}, "VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
// EVEX.F3.0F38 — mask/vector conversions: M2* moves an opmask register // EVEX.F3.0F38, mask/vector conversions: M2* moves an opmask register
// into a vector (rm = K source, reg = vector destination), *2M does the // into a vector (rm = K source, reg = vector destination), *2M does the
// reverse (reg = K destination, rm = vector source, the length follows // reverse (reg = K destination, rm = vector source, the length follows
// the vector). // the vector).
@@ -391,7 +391,7 @@ var evexTable = map[string]evexSpec{
"VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, "VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, "VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX — scalar conversions between vector and general-purpose // EVEX, scalar conversions between vector and general-purpose
// registers. Vector to GPR (two operands: vec/mem source, GPR // registers. Vector to GPR (two operands: vec/mem source, GPR
// destination, vvvv unused): the signed and truncated pair, and the // destination, vvvv unused): the signed and truncated pair, and the
// unsigned forms (EVEX only). // unsigned forms (EVEX only).
@@ -421,22 +421,22 @@ var evexTable = map[string]evexSpec{
"VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, "VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}}, "VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory // EVEX.128/256/512.66.0F38.W0, sign-extend dwords to qwords; the memory
// operand is the narrow source, so disp8×N follows its size (8/16/32 for // operand is the narrow source, so disp8×N follows its size (8/16/32 for
// the xmm/ymm/zmm destination lengths). // the xmm/ymm/zmm destination lengths).
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.512.66.0F3A.W1 — lane extract (reg=ZMM source, rm=YMM/memory // EVEX.512.66.0F3A.W1, lane extract (reg=ZMM source, rm=YMM/memory
// destination, imm8). // destination, imm8).
"VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}}, "VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}}, "VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
// EVEX.66.0F38 — more integer NDS forms (W distinguishes D/Q). // EVEX.66.0F38, more integer NDS forms (W distinguishes D/Q).
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}}, "VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
// EVEX.128/256/512 — the wider integer set (AVX-512 F/BW): byte/word // EVEX.128/256/512, the wider integer set (AVX-512 F/BW): byte/word
// arithmetic, the bitwise ops with D/Q suffixes, min/max, averages and // arithmetic, the bitwise ops with D/Q suffixes, min/max, averages and
// variable shifts. All NDS form; W distinguishes element size. // variable shifts. All NDS form; W distinguishes element size.
"VPADDB": {1, 0xFC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPADDB": {1, 0xFC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -474,21 +474,21 @@ var evexTable = map[string]evexSpec{
"VPSRAVQ": {2, 0x46, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSRAVQ": {2, 0x46, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX forms of instructions that also exist in VEX (selected when a ZMM // EVEX forms of instructions that also exist in VEX (selected when a ZMM
// or K register, or indices 16–31, demand EVEX). // or K register, or indices 16-31, demand EVEX).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}}, "VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F — immediate shift (VPSLLD /6). // EVEX.66.0F, immediate shift (VPSLLD /6).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}}, "VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.F3.0F38.W0 — narrowing stores: reg = wide source, rm = narrow // EVEX.F3.0F38.W0, narrowing stores: reg = wide source, rm = narrow
// destination (VPMOVDW dword→word, VPMOVQD qword→dword). // destination (VPMOVDW dword→word, VPMOVQD qword→dword).
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, "VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, "VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
} }
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode // evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
// depends on the source kind — a GPR source uses opReg, a memory source uses // depends on the source kind, a GPR source uses opReg, a memory source uses
// opMem with a disp8×N of n. // opMem with a disp8×N of n.
type evexBcastSpec struct { type evexBcastSpec struct {
mapSel int mapSel int
@@ -499,10 +499,10 @@ type evexBcastSpec struct {
} }
var evexBcastTable = map[string]evexBcastSpec{ var evexBcastTable = map[string]evexBcastSpec{
// EVEX.128/256/512.66.0F38 — broadcast a dword/qword to all lanes. // EVEX.128/256/512.66.0F38, broadcast a dword/qword to all lanes.
"VPBROADCASTD": {2, 0x7C, 0x58, 0, 4}, "VPBROADCASTD": {2, 0x7C, 0x58, 0, 4},
"VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8}, "VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8},
// EVEX.128/256/512.66.0F38 — broadcast a byte/word (GPR or memory // EVEX.128/256/512.66.0F38, broadcast a byte/word (GPR or memory
// source) to all lanes. // source) to all lanes.
"VPBROADCASTB": {2, 0x7A, 0x78, 0, 1}, "VPBROADCASTB": {2, 0x7A, 0x78, 0, 1},
"VPBROADCASTW": {2, 0x7B, 0x79, 0, 2}, "VPBROADCASTW": {2, 0x7B, 0x79, 0, 2},
@@ -521,26 +521,26 @@ type evexMoveSpec struct {
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding. // evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{ var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0 — unaligned integer move. // EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1 — unaligned qword move. // EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, "VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W0 — unaligned byte move (byte/word moves use the // EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple // F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics). // semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, "VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1 — unaligned word move (shares the qword // EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding). // encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — unaligned packed double move. // EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}}, "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512 — aligned packed moves. // EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}}, "VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}}, "VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F — aligned integer moves. // EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, "VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, "VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128.F3.0F.W0 — scalar single move, memory operands (the // EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported). // three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}}, "VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}},
} }
@@ -559,7 +559,7 @@ func isEvex(mnemUpper string) bool {
// evexRequired reports whether the operands force the EVEX encoding of a // evexRequired reports whether the operands force the EVEX encoding of a
// mnemonic that also has a VEX form: ZMM and K registers do, and so do // mnemonic that also has a VEX form: ZMM and K registers do, and so do
// register indices 16–31, which only EVEX can represent (X16–Y31 exist // register indices 16-31, which only EVEX can represent (X16-Y31 exist
// solely under AVX-512). // solely under AVX-512).
func evexRequired(upper string, ops []Operand) bool { func evexRequired(upper string, ops []Operand) bool {
_, inVex := vexTable[upper] _, inVex := vexTable[upper]
@@ -578,7 +578,7 @@ func evexRequired(upper string, ops []Operand) bool {
// evexSuffix carries the EVEX mnemonic suffixes the Go assembler accepts: // evexSuffix carries the EVEX mnemonic suffixes the Go assembler accepts:
// zeroing (.Z), a rounding mode (.RN_SAE, .RD_SAE, .RU_SAE, .RZ_SAE), // zeroing (.Z), a rounding mode (.RN_SAE, .RD_SAE, .RU_SAE, .RZ_SAE),
// suppress-all-exceptions (.SAE) and memory broadcast (.BCST). Masking is // suppress-all-exceptions (.SAE) and memory broadcast (.BCST). Masking is
// not a suffix — Go writes it as an explicit K operand. // not a suffix, Go writes it as an explicit K operand.
type evexSuffix struct { type evexSuffix struct {
zeroing bool zeroing bool
sae bool sae bool
@@ -668,7 +668,7 @@ var evexRound = map[string]bool{
} }
// evexBcstN maps an instruction accepting .BCST to the broadcast element // evexBcstN maps an instruction accepting .BCST to the broadcast element
// size — the disp8×N multiplier for its memory operand. // size, the disp8×N multiplier for its memory operand.
var evexBcstN = map[string]int{ var evexBcstN = map[string]int{
"VADDPD": 8, "VSUBPD": 8, "VMULPD": 8, "VDIVPD": 8, "VADDPD": 8, "VSUBPD": 8, "VMULPD": 8, "VDIVPD": 8,
"VMINPD": 8, "VMAXPD": 8, "VMINPD": 8, "VMAXPD": 8,
@@ -689,7 +689,7 @@ var evexBcstN = map[string]int{
"VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8, "VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8,
} }
// splitMask extracts an explicit mask register (K1–K7) from the operand list, // splitMask extracts an explicit mask register (K1-K7) from the operand list,
// returning the remaining operands and the mask index. K0 is not a usable // returning the remaining operands and the mask index. K0 is not a usable
// mask (aaa = 0 means "no mask"), matching the assembler. // mask (aaa = 0 means "no mask"), matching the assembler.
func splitMask(ops []Operand) ([]Operand, int, error) { func splitMask(ops []Operand) ([]Operand, int, error) {
@@ -712,7 +712,7 @@ func splitMask(ops []Operand) ([]Operand, int, error) {
} }
// encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The // encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The
// mask, when present, is an explicit K1–K7 operand anywhere among the // mask, when present, is an explicit K1-K7 operand anywhere among the
// operands; the mnemonic suffix carries zeroing, rounding/SAE and // operands; the mnemonic suffix carries zeroing, rounding/SAE and
// broadcast. // broadcast.
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error { func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
@@ -1042,7 +1042,7 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
} }
// encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with // encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the length fixed by the mnemonic — the // the destination always XMM and the length fixed by the mnemonic, the
// single valid slot of spec.n names the vector length (and the disp8×N // single valid slot of spec.n names the vector length (and the disp8×N
// multiplier) a register or memory source encodes. // multiplier) a register or memory source encodes.
func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
@@ -1061,7 +1061,7 @@ func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx eve
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx) return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
} }
// soleLen returns the vector-length index of the single valid slot of n — // soleLen returns the vector-length index of the single valid slot of n
// the length a length-fixed mnemonic (the EVEX conversion spellings) encodes // the length a length-fixed mnemonic (the EVEX conversion spellings) encodes
// regardless of its operands. // regardless of its operands.
func soleLen(n [3]int) (int, error) { func soleLen(n [3]int) (int, error) {
@@ -1131,8 +1131,8 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx eve
// emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement // emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement
// (disp8×N compressed) for the given precomputed fields. regIdx is the // (disp8×N compressed) for the given precomputed fields. regIdx is the
// unextended reg-field register index, or a /digit (0–7); vvvvIdx is the // unextended reg-field register index, or a /digit (0-7); vvvvIdx is the
// vvvv register index, or -1 when unused. mask (K1–K7, 0 = unmasked) and // vvvv register index, or -1 when unused. mask (K1-K7, 0 = unmasked) and
// zeroing fill the aaa and z bits of the P2 byte. // zeroing fill the aaa and z bits of the P2 byte.
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, sfx evexSuffix) error { func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, sfx evexSuffix) error {
if ll > 2 { if ll > 2 {
@@ -1324,7 +1324,7 @@ func isScatter(upper string) bool {
} }
// vsibLen validates a VSIB memory operand (the index must be a vector // vsibLen validates a VSIB memory operand (the index must be a vector
// register) and returns it with the vector length the index selects — the // register) and returns it with the vector length the index selects, the
// EVEX L'L field follows the index register, not the data register. // EVEX L'L field follows the index register, not the data register.
func vsibLen(op Operand, what string) (Mem, int, error) { func vsibLen(op Operand, what string) (Mem, int, error) {
m, ok := op.(Mem) m, ok := op.(Mem)
@@ -1383,7 +1383,7 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib) return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib)
} }
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib — reg = src, // encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src,
// rm = the VSIB memory operand, the K mask in aaa and L following the VSIB // rm = the VSIB memory operand, the K mask in aaa and L following the VSIB
// index. // index.
func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evexSuffix) error { func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evexSuffix) error {
@@ -1411,15 +1411,15 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
// evexKOperand lists the instructions whose K register is a genuine operand // evexKOperand lists the instructions whose K register is a genuine operand
// (the source or destination of a mask/vector conversion) rather than a // (the source or destination of a mask/vector conversion) rather than a
// mask modifier — the M2 and 2M conversions. They take no masking. // mask modifier, the M2 and 2M conversions. They take no masking.
var evexKOperand = map[string]bool{ var evexKOperand = map[string]bool{
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true, "VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true, "VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
} }
// kmovSpec describes a KMOV width: the opcode depends on the operand // kmovSpec describes a KMOV width: the opcode depends on the operand
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem), // direction, kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory // gprk (GPR/mem → K), kgpr (K → GPR), and the GPR forms carry a mandatory
// prefix and W for the wider widths. // prefix and W for the wider widths.
type kmovSpec struct { type kmovSpec struct {
kk, kmem, gprk, kgpr byte kk, kmem, gprk, kgpr byte
+4 -4
View File
@@ -14,8 +14,8 @@ import (
"sync" "sync"
) )
// This file emits GOOBJ — the Go toolchain's object format, which cmd/link // This file emits GOOBJ, the Go toolchain's object format, which cmd/link
// consumes directly — so gasm-assembled functions drop into a go build // consumes directly, so gasm-assembled functions drop into a go build
// without the Go assembler. The layout follows cmd/internal/goobj: a // without the Go assembler. The layout follows cmd/internal/goobj: a
// toolchain preamble ("go object ...\n!\n"), the go120ld header with its // toolchain preamble ("go object ...\n!\n"), the go120ld header with its
// block offsets, a string table, symbol definitions, the relocation / // block offsets, a string table, symbol definitions, the relocation /
@@ -178,7 +178,7 @@ type dwarfRelocSet struct {
// does with its -p flag). srcPath names the source file recorded in the // does with its -p flag). srcPath names the source file recorded in the
// object's file table and line tables. The toolchain's object preamble is // object's file table and line tables. The toolchain's object preamble is
// captured from the installed go tool asm, so the output links with the // captured from the installed go tool asm, so the output links with the
// toolchain it was produced on — exactly like a real assembly object. // toolchain it was produced on, exactly like a real assembly object.
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) { func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreamble() pre, err := toolchainObjectPreamble()
if err != nil { if err != nil {
@@ -198,7 +198,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)") return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
} }
// The non-package definitions first — the DWARF symbols reference the // The non-package definitions first, the DWARF symbols reference the
// functions by these indices: per function the four pc-value tables // functions by these indices: per function the four pc-value tables
// and the function itself, as cmd/asm lays them out. // and the function itself, as cmd/asm lays them out.
type npSym struct { type npSym struct {
+1 -1
View File
@@ -84,7 +84,7 @@ func sortedPkgRefs(refs map[string][]string) []pkgRef {
for pkg, syms := range refs { for pkg, syms := range refs {
pkgs = append(pkgs, pkgRef{pkg, syms}) pkgs = append(pkgs, pkgRef{pkg, syms})
} }
// Simple insertion sort — the list is tiny (usually 1–3 packages). // Simple insertion sort, the list is tiny (usually 1-3 packages).
for i := 1; i < len(pkgs); i++ { for i := 1; i < len(pkgs); i++ {
for j := i; j > 0 && pkgs[j-1].path > pkgs[j].path; j-- { for j := i; j > 0 && pkgs[j-1].path > pkgs[j].path; j-- {
pkgs[j-1], pkgs[j] = pkgs[j], pkgs[j-1] pkgs[j-1], pkgs[j] = pkgs[j], pkgs[j-1]
+2 -2
View File
@@ -13,9 +13,9 @@ import (
) )
// GOObjectLOONG64 emits a GOOBJ object file for LoongArch. The layout is // GOObjectLOONG64 emits a GOOBJ object file for LoongArch. The layout is
// the shared one in goobj.go — the toolchain preamble, the go120ld header // the shared one in goobj.go, the toolchain preamble, the go120ld header
// with its block offsets, the string table, the symbol definitions and the // with its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the loong64 preamble, the MinLC of 4 // reloc/aux/data index arrays, with the loong64 preamble, the MinLC of 4
// for the pc-value deltas, and R_LOONG64_ADDR_HI/LO relocation types for // for the pc-value deltas, and R_LOONG64_ADDR_HI/LO relocation types for
// the pcalau12i+addi.d address pairs. // the pcalau12i+addi.d address pairs.
func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) { func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
+2 -2
View File
@@ -13,9 +13,9 @@ import (
) )
// GOObjectRISCV emits a GOOBJ object file for RISC-V. The layout is the // GOObjectRISCV emits a GOOBJ object file for RISC-V. The layout is the
// shared one in goobj.go — the toolchain preamble, the go120ld header with // shared one in goobj.go, the toolchain preamble, the go120ld header with
// its block offsets, the string table, the symbol definitions and the // its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the RISC-V preamble, the MinLC of 2 for // reloc/aux/data index arrays, with the RISC-V preamble, the MinLC of 2 for
// the pc-value deltas, and the single R_RISCV_PCREL_ITYPE/STYPE relocation // the pc-value deltas, and the single R_RISCV_PCREL_ITYPE/STYPE relocation
// per AUIPC pair, matching `go tool asm`'s model (each pair is one 8-byte // per AUIPC pair, matching `go tool asm`'s model (each pair is one 8-byte
// relocation, not the ELF HI20/LO12 pair). // relocation, not the ELF HI20/LO12 pair).
+14 -14
View File
@@ -21,7 +21,7 @@ var aluOp = map[string]struct {
} }
// unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use // unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use
// the 0xFE/0xFF group (the short 0x40–0x4F forms are REX prefixes in 64-bit // the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes in 64-bit
// mode); NEG/NOT use the 0xF6/0xF7 group. // mode); NEG/NOT use the 0xF6/0xF7 group.
var unaryOp = map[string]struct { var unaryOp = map[string]struct {
digit int digit int
@@ -33,7 +33,7 @@ var unaryOp = map[string]struct {
"NEG": {3, 0xF7}, "NEG": {3, 0xF7},
} }
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0–0xD3 group. // shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0-0xD3 group.
var shiftOp = map[string]int{ var shiftOp = map[string]int{
"SHL": 4, "SHL": 4,
"SHR": 5, "SHR": 5,
@@ -50,7 +50,7 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
// Integer scalar XMM moves: MOVQ with an XMM operand is the SSE2 // Integer scalar XMM moves: MOVQ with an XMM operand is the SSE2
// packed-quadword move, NOT a GPR move: mem→xmm encodes as F3 0F 7E // packed-quadword move, NOT a GPR move: mem→xmm encodes as F3 0F 7E
// (reg = dst, no REX.W — the Go assembler's form), xmm→mem as // (reg = dst, no REX.W, the Go assembler's form), xmm→mem as
// 66 0F D6 (rm = xmm). Register forms against a GPR use the MOVD // 66 0F D6 (rm = xmm). Register forms against a GPR use the MOVD
// opcodes with REX.W instead: 66 REX.W 0F 6E (gpr→xmm) and // opcodes with REX.W instead: 66 REX.W 0F 6E (gpr→xmm) and
// 66 REX.W 0F 7E (xmm→gpr); the memory opcodes with a register r/m // 66 REX.W 0F 7E (xmm→gpr); the memory opcodes with a register r/m
@@ -103,7 +103,7 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
switch src := src.(type) { switch src := src.(type) {
case Reg: case Reg:
if dstIsReg { if dstIsReg {
// MOV r/m, r: 0x88/0x89, reg=src, rm=dst — the form the Go // MOV r/m, r: 0x88/0x89, reg=src, rm=dst, the form the Go
// assembler emits for register-to-register moves. // assembler emits for register-to-register moves.
i := newInstr(size, []byte{movRM(size)}) i := newInstr(size, []byte{movRM(size)})
if err := setRM(i, src, dst, size); err != nil { if err := setRM(i, src, dst, size); err != nil {
@@ -147,7 +147,7 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
// a signed int32, choosing per sign: // a signed int32, choosing per sign:
// v >= 0: B8+rd imm32 without REX.W (zero-extended by the // v >= 0: B8+rd imm32 without REX.W (zero-extended by the
// hardware, REX.B still emitted for R8-R15); // hardware, REX.B still emitted for R8-R15);
// v < 0: REX.W C7 /0 imm32 (sign-extended — the plain B8+rd // v < 0: REX.W C7 /0 imm32 (sign-extended, the plain B8+rd
// form would zero-extend and corrupt the value). // form would zero-extend and corrupt the value).
// Out-of-range immediates keep the B8+rd imm64 form. // Out-of-range immediates keep the B8+rd imm64 form.
if size == 8 && v >= 0 && v <= (1<<31)-1 { if size == 8 && v >= 0 && v <= (1<<31)-1 {
@@ -226,8 +226,8 @@ func (e *enc) encodeALU(op struct {
return e.encodeALUImm(op.digit, dst, int64(imm), size) return e.encodeALUImm(op.digit, dst, int64(imm), size)
} }
// CMP accepts the immediate in the second position too — CMPL CX, $31 is // CMP accepts the immediate in the second position too, CMPL CX, $31 is
// the form the Go assembler itself accepts — and encodes it identically // the form the Go assembler itself accepts, and encodes it identically
// (CMP r/m, imm sets the flags as first − second). No other ALU op takes // (CMP r/m, imm sets the flags as first − second). No other ALU op takes
// an immediate destination. // an immediate destination.
if imm, ok := dst.(Imm); ok { if imm, ok := dst.(Imm); ok {
@@ -313,7 +313,7 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
i.imm = []byte{byte(int8(imm))} i.imm = []byte{byte(int8(imm))}
return e.emit(i) return e.emit(i)
} }
// 0x81 /digit, imm16/imm32 — or the Go assembler's accumulator short // 0x81 /digit, imm16/imm32, or the Go assembler's accumulator short
// form (opcode+5, no ModR/M) when the destination is AX/AL, which it // form (opcode+5, no ModR/M) when the destination is AX/AL, which it
// prefers over the generic form exactly here. // prefers over the generic form exactly here.
if r, ok := dst.(Reg); ok && r.idx == 0 { if r, ok := dst.(Reg); ok && r.idx == 0 {
@@ -339,7 +339,7 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
} }
src, dst := ops[0], ops[1] src, dst := ops[0], ops[1]
if imm, ok := src.(Imm); ok { if imm, ok := src.(Imm); ok {
// TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0 — but the Go assembler // TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0, but the Go assembler
// always uses the accumulator forms (A8/A9, no ModR/M) when the // always uses the accumulator forms (A8/A9, no ModR/M) when the
// register operand is AL/AX, whatever the immediate's width. // register operand is AL/AX, whatever the immediate's width.
if r, ok := dst.(Reg); ok && r.idx == 0 { if r, ok := dst.(Reg); ok && r.idx == 0 {
@@ -678,7 +678,7 @@ func (e *enc) encodeCmov(upper string, ops []Operand) error {
} }
// encodeSet encodes a conditional byte set: SET + condition (SETNE, SETEQ, …), // encodeSet encodes a conditional byte set: SET + condition (SETNE, SETEQ, …),
// always a byte write — 0F 90+cc /0 into a register or memory operand. // always a byte write, 0F 90+cc /0 into a register or memory operand.
func (e *enc) encodeSet(upper string, ops []Operand) error { func (e *enc) encodeSet(upper string, ops []Operand) error {
if len(ops) != 1 { if len(ops) != 1 {
return fmt.Errorf("SETcc expects 1 operand, got %d", len(ops)) return fmt.Errorf("SETcc expects 1 operand, got %d", len(ops))
@@ -711,8 +711,8 @@ var countOp = map[string]struct {
"POPCNT": {0xB8, 0xF3}, "POPCNT": {0xB8, 0xF3},
} }
// encodeCount encodes the bit-scan and bit-count family — BSF (0F BC), // encodeCount encodes the bit-scan and bit-count family, BSF (0F BC),
// BSR (0F BD), TZCNT (F3 0F BC), LZCNT (F3 0F BD) and POPCNT (F3 0F B8) — // BSR (0F BD), TZCNT (F3 0F BC), LZCNT (F3 0F BD) and POPCNT (F3 0F B8)
// with reg = dst and rm = src. The size suffix selects the operand width // with reg = dst and rm = src. The size suffix selects the operand width
// (BSFQ, TZCNTL, …). Note BSF/BSR leave the destination undefined when the // (BSFQ, TZCNTL, …). Note BSF/BSR leave the destination undefined when the
// source is zero (unlike their F3-prefixed counterparts); callers must // source is zero (unlike their F3-prefixed counterparts); callers must
@@ -801,8 +801,8 @@ type sseMove struct {
} }
var sseMoveTable = map[string]sseMove{ var sseMoveTable = map[string]sseMove{
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU — unaligned octa "MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA — aligned octa "MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single "MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single "MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double "MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
+14 -14
View File
@@ -9,7 +9,7 @@ package asm
// an opcode constant, and the format selects the bit layout. The opcode // an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own loong64 // constants and formats are transcribed from the Go toolchain's own loong64
// backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm` // backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm`
// exactly — the ground-truth oracle for the verify suite. // exactly, the ground-truth oracle for the verify suite.
// //
// All LoongArch instructions are 32 bits, little-endian. The formats used // All LoongArch instructions are 32 bits, little-endian. The formats used
// here (per the LoongArch Volume I specification): // here (per the LoongArch Volume I specification):
@@ -33,8 +33,8 @@ package asm
import "maps" import "maps"
// loong64RegNum returns the 5-bit register number for a LoongArch register // loong64RegNum returns the 5-bit register number for a LoongArch register
// name: R0–R31 (integer), F0–F31 (floating point), FCC0–FCC7 (condition // name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
// flags), FCSR0–FCSR31 (control/status) and the ABI aliases the runtime's // flags), FCSR0-FCSR31 (control/status) and the ABI aliases the runtime's
// assembly uses. Returns -1 for an unrecognised name. // assembly uses. Returns -1 for an unrecognised name.
func loong64RegNum(name string) int { func loong64RegNum(name string) int {
switch name { switch name {
@@ -103,7 +103,7 @@ func loong64RegNum(name string) int {
case "R31", "S8": case "R31", "S8":
return 31 return 31
} }
// F0–F31, FCC0–FCC7, FCSR0–FCSR31. // F0-F31, FCC0-FCC7, FCSR0-FCSR31.
if len(name) >= 4 && name[:4] == "FCSR" { if len(name) >= 4 && name[:4] == "FCSR" {
return loong64RegSpecial(name[4:], 31) return loong64RegSpecial(name[4:], 31)
} }
@@ -199,7 +199,7 @@ func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
} }
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd. // l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
// The msb/lsb fields are 6 bits wide (0–63) and are validated by the caller. // The msb/lsb fields are 6 bits wide (0-63) and are validated by the caller.
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 { func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f) return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
} }
@@ -280,7 +280,7 @@ var l64DualTable = map[string]l64DualEnc{}
var l64InstrTable = map[string]l64Enc{} var l64InstrTable = map[string]l64Enc{}
func init() { func init() {
// 3R — integer. // 3R, integer.
rrr := map[string]uint32{ rrr := map[string]uint32{
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15, "ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15, "SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
@@ -300,7 +300,7 @@ func init() {
"CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15, "CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15,
"CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15, "CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15,
} }
// 3R — floating point. // 3R, floating point.
rrr["MULF"] = 0x209 << 15 rrr["MULF"] = 0x209 << 15
rrr["MULD"] = 0x20a << 15 rrr["MULD"] = 0x20a << 15
rrr["DIVF"] = 0x20d << 15 rrr["DIVF"] = 0x20d << 15
@@ -390,12 +390,12 @@ func init() {
"ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true}, "ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true},
}) })
// 2RI12 — pure immediate arithmetic (LU52ID has no register form). // 2RI12, pure immediate arithmetic (LU52ID has no register form).
l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22} l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22}
// ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16. // ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16.
l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26} l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26}
// 2RI14 — LL/SC are aliased by the Go assembler to the pointer loads and // 2RI14, LL/SC are aliased by the Go assembler to the pointer loads and
// stores (ldptr/stptr), with the offset scaled by 4. // stores (ldptr/stptr), with the offset scaled by 4.
l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w
l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d
@@ -414,7 +414,7 @@ func init() {
// LUI is the Plan 9 spelling of lu12i.w. // LUI is the Plan 9 spelling of lu12i.w.
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25} l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
// 4R — fused multiply-add. // 4R, fused multiply-add.
rrrr := map[string]uint32{ rrrr := map[string]uint32{
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20, "FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20, "FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
@@ -425,7 +425,7 @@ func init() {
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op} l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
} }
// IRIR — bit-field insert/extract. // IRIR, bit-field insert/extract.
irir := map[string]uint32{ irir := map[string]uint32{
"BSTRINSW": 0x3<<21 | 0x0<<15, "BSTRINSW": 0x3<<21 | 0x0<<15,
"BSTRINSV": 0x2 << 22, "BSTRINSV": 0x2 << 22,
@@ -436,7 +436,7 @@ func init() {
l64InstrTable[m] = l64Enc{format: l64Firir, op: op} l64InstrTable[m] = l64Enc{format: l64Firir, op: op}
} }
// 3RI2 — ALSL. // 3RI2, ALSL.
irrr := map[string]uint32{ irrr := map[string]uint32{
"ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17, "ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17,
} }
@@ -452,7 +452,7 @@ func init() {
// PRELD. // PRELD.
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22} l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
// Atomics — 3R with the AM field order (rk=value, rj=address, rd=result). // Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
am := map[string]uint32{ am := map[string]uint32{
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15, "AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15, "AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
@@ -477,7 +477,7 @@ func init() {
} }
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the // l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
// register move between the integer and floating-point register banks — the // register move between the integer and floating-point register banks, the
// MOVW/MOVV specials the Go assembler accepts. // MOVW/MOVV specials the Go assembler accepts.
var l64FpMovTable = map[string]uint32{ var l64FpMovTable = map[string]uint32{
"MOVV.R.F": 0x452a << 10, // movgr2fr.d "MOVV.R.F": 0x452a << 10, // movgr2fr.d
+3 -3
View File
@@ -98,13 +98,13 @@ func loong64Return(fi loong64FrameInfo) []byte {
var ws []uint32 var ws []uint32
if fi.autosize != 0 { if fi.autosize != 0 {
if !fi.leaf { if !fi.leaf {
// MOVV 0(R3), R1 — restore the link register. // MOVV 0(R3), R1, restore the link register.
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1)) ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1))
} }
// ADDV $autosize, R3 — close the frame. // ADDV $autosize, R3, close the frame.
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3)) ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
} }
// jirl r0, r1, 0 — return. // jirl r0, r1, 0, return.
ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0)) ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0))
return l64WordsLE(ws...) return l64WordsLE(ws...)
} }
+9 -9
View File
@@ -12,33 +12,33 @@ import "maps"
import "strings" import "strings"
// Reg is an x86-64 register. In Plan 9 assembly the classic names (AX, BX, …) // Reg is an x86-64 register. In Plan 9 assembly the classic names (AX, BX, …)
// are size-agnostic — the instruction suffix (MOVQ vs MOVL) fixes the width — // are size-agnostic, the instruction suffix (MOVQ vs MOVL) fixes the width
// so the encoder keys off the register's index and lets the mnemonic supply the // so the encoder keys off the register's index and lets the mnemonic supply the
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which // size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4–7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share // occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one. The mask flag marks the AVX-512 opmask // those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0–K7. // registers K0-K7.
type Reg struct { type Reg struct {
idx int idx int
size int // informational width implied by the name; the mnemonic decides size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH high bool // AH/CH/DH/BH
mask bool // K0–K7 opmask register mask bool // K0-K7 opmask register
} }
// Index returns the register number (0–15 for GPRs, 0–31 for vectors). // Index returns the register number (0-15 for GPRs, 0-31 for vectors).
func (r Reg) Index() int { return r.idx } func (r Reg) Index() int { return r.idx }
// Size returns the width in bytes implied by the register's name. // Size returns the width in bytes implied by the register's name.
func (r Reg) Size() int { return r.size } func (r Reg) Size() int { return r.size }
// IsMask reports whether r is an AVX-512 opmask register (K0–K7). // IsMask reports whether r is an AVX-512 opmask register (K0-K7).
func (r Reg) IsMask() bool { return r.mask } func (r Reg) IsMask() bool { return r.mask }
func (r Reg) isOperand() {} func (r Reg) isOperand() {}
// needsREX reports whether this register forces a REX prefix at the given // needsREX reports whether this register forces a REX prefix at the given
// operand size: the extended registers R8–R15 always do, and at byte size the // operand size: the extended registers R8-R15 always do, and at byte size the
// low registers SPL/BPL/SIL/DIL (indices 4–7, not high) do as well. // low registers SPL/BPL/SIL/DIL (indices 4-7, not high) do as well.
func (r Reg) needsREX(opSize int) bool { func (r Reg) needsREX(opSize int) bool {
if r.idx >= 8 { if r.idx >= 8 {
return true return true
@@ -133,7 +133,7 @@ func buildRegByName() map[string]Reg {
} }
// Vector: X0..X31 (128-bit, size 16), Y0..Y31 (256-bit, size 32), // Vector: X0..X31 (128-bit, size 16), Y0..Y31 (256-bit, size 32),
// Z0..Z31 (512-bit, size 64). Indices 16–31 are only encodable in EVEX // Z0..Z31 (512-bit, size 64). Indices 16-31 are only encodable in EVEX
// (AVX-512) instructions; the encoder validates that through its tables. // (AVX-512) instructions; the encoder validates that through its tables.
for i := 0; i <= 31; i++ { for i := 0; i <= 31; i++ {
m["X"+itoa(i)] = Reg{idx: i, size: 16} m["X"+itoa(i)] = Reg{idx: i, size: 16}
+10 -10
View File
@@ -192,7 +192,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
if relocs != nil { if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelRISCVJal, Addend: op.Addr.Sym.Offset}) *relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelRISCVJal, Addend: op.Addr.Sym.Offset})
} }
word = riscvJType(1, 0) // JAL X1, 0 — the linker fills the offset word = riscvJType(1, 0) // JAL X1, 0, the linker fills the offset
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
case "JMP": case "JMP":
// JMP = JAL X0, target. The Go assembler never compresses this to // JMP = JAL X0, target. The Go assembler never compresses this to
@@ -395,7 +395,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
} }
word = riscvSType(enc, rs1, rs2, imm) word = riscvSType(enc, rs1, rs2, imm)
// LR (load-reserved): INSTR (addr), dst — 2 operands. // LR (load-reserved): INSTR (addr), dst, 2 operands.
case len(ops) == 2 && isLRInstr(mnem): case len(ops) == 2 && isLRInstr(mnem):
rs1, _ := memFromOperandWithFrame(ops[0], fi) rs1, _ := memFromOperandWithFrame(ops[0], fi)
rd := regFromOperand(ops[1]) rd := regFromOperand(ops[1])
@@ -404,7 +404,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
} }
word = riscvAMOType(enc, rd, rs1, 0) // rs2=0 for LR word = riscvAMOType(enc, rd, rs1, 0) // rs2=0 for LR
// SC (store-conditional): INSTR src, (addr), dst — 3 operands. // SC (store-conditional): INSTR src, (addr), dst, 3 operands.
case len(ops) == 3 && isSCInstr(mnem): case len(ops) == 3 && isSCInstr(mnem):
rs2 := regFromOperand(ops[0]) rs2 := regFromOperand(ops[0])
rs1, _ := memFromOperandWithFrame(ops[1], fi) rs1, _ := memFromOperandWithFrame(ops[1], fi)
@@ -443,7 +443,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
} }
return encodeRISCVItypeImmediate(mnem, enc, rd, rd, imm) return encodeRISCVItypeImmediate(mnem, enc, rd, rd, imm)
// Loads: rd, offset(rs1) — Plan 9 order is LD src, dst. // Loads: rd, offset(rs1), Plan 9 order is LD src, dst.
case len(ops) == 2 && isLoadInstr(mnem): case len(ops) == 2 && isLoadInstr(mnem):
rd := regFromOperand(ops[1]) // destination (last operand) rd := regFromOperand(ops[1]) // destination (last operand)
rs1, imm := memFromOperandWithFrame(ops[0], fi) // memory source (first operand) rs1, imm := memFromOperandWithFrame(ops[0], fi) // memory source (first operand)
@@ -538,7 +538,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
// Immediate → register. // Immediate → register.
if isImmOperand(src) { if isImmOperand(src) {
// MOV $sym(SB), rd — load address of a static symbol or external. // MOV $sym(SB), rd, load address of a static symbol or external.
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
rd := regFromOperand(dst) rd := regFromOperand(dst)
if rd < 0 { if rd < 0 {
@@ -546,7 +546,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
} }
return encodeRISCVSBAddr(src.Imm.Sym, rd, relocs), nil return encodeRISCVSBAddr(src.Imm.Sym, rd, relocs), nil
} }
// MOV $sym(FP/SP), rd — not supported: immediate symbol references // MOV $sym(FP/SP), rd, not supported: immediate symbol references
// other than SB cannot be encoded as a simple immediate. // other than SB cannot be encoded as a simple immediate.
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo != "" { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo != "" {
return nil, fmt.Errorf("MOV $%s(%s): unsupported immediate symbol reference (only SB is supported)", src.Imm.Sym.Name, src.Imm.Sym.Pseudo) return nil, fmt.Errorf("MOV $%s(%s): unsupported immediate symbol reference (only SB is supported)", src.Imm.Sym.Name, src.Imm.Sym.Pseudo)
@@ -562,7 +562,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
// Memory → register (load). // Memory → register (load).
if isMemOperand(src) && !isMemOperand(dst) { if isMemOperand(src) && !isMemOperand(dst) {
rd := regFromOperand(dst) rd := regFromOperand(dst)
// MOV sym(SB), rd — load from static data. // MOV sym(SB), rd, load from static data.
if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" { if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" {
if rd < 0 { if rd < 0 {
return nil, fmt.Errorf("MOV sym(SB): invalid destination register") return nil, fmt.Errorf("MOV sym(SB): invalid destination register")
@@ -580,7 +580,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
// Register → memory (store). // Register → memory (store).
if !isMemOperand(src) && isMemOperand(dst) { if !isMemOperand(src) && isMemOperand(dst) {
rs2 := regFromOperand(src) rs2 := regFromOperand(src)
// MOV rd, sym(SB) — store to static data. // MOV rd, sym(SB), store to static data.
if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" { if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" {
if rs2 < 0 { if rs2 < 0 {
return nil, fmt.Errorf("MOV rd, sym(SB): invalid source register") return nil, fmt.Errorf("MOV rd, sym(SB): invalid source register")
@@ -1006,7 +1006,7 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
} }
case "ADDW", "SUBW": case "ADDW", "SUBW":
// C.ADDW (0x27,1) / C.SUBW (0x27,0) — CA-type, prime regs. // C.ADDW (0x27,1) / C.SUBW (0x27,0), CA-type, prime regs.
if len(ops) == 3 { if len(ops) == 3 {
funct2 := uint32(0x0) funct2 := uint32(0x0)
if mnem == "ADDW" { if mnem == "ADDW" {
@@ -1314,7 +1314,7 @@ func suggestLabel(target string, offsets map[string]int) string {
} }
// Only suggest if the distance is small enough. // Only suggest if the distance is small enough.
if bestDist <= 3 && bestDist < len(target)/2+1 { if bestDist <= 3 && bestDist < len(target)/2+1 {
return fmt.Sprintf(" — did you mean %q?", best) return fmt.Sprintf("; did you mean %q?", best)
} }
return "" return ""
} }
+14 -14
View File
@@ -154,7 +154,7 @@ type riscvEnc struct {
// riscvInstrTable maps RISC-V mnemonics to their encoding. // riscvInstrTable maps RISC-V mnemonics to their encoding.
var riscvInstrTable = map[string]riscvEnc{ var riscvInstrTable = map[string]riscvEnc{
// RV64I — R-type arithmetic/logic. // RV64I, R-type arithmetic/logic.
"ADD": {0x33, 0x0, 0x00}, "ADD": {0x33, 0x0, 0x00},
"SUB": {0x33, 0x0, 0x20}, "SUB": {0x33, 0x0, 0x20},
"SLL": {0x33, 0x1, 0x00}, "SLL": {0x33, 0x1, 0x00},
@@ -165,20 +165,20 @@ var riscvInstrTable = map[string]riscvEnc{
"SRA": {0x33, 0x5, 0x20}, "SRA": {0x33, 0x5, 0x20},
"OR": {0x33, 0x6, 0x00}, "OR": {0x33, 0x6, 0x00},
"AND": {0x33, 0x7, 0x00}, "AND": {0x33, 0x7, 0x00},
// RV64I — 32-bit variants (W suffix). // RV64I, 32-bit variants (W suffix).
"ADDW": {0x3B, 0x0, 0x00}, "ADDW": {0x3B, 0x0, 0x00},
"SUBW": {0x3B, 0x0, 0x20}, "SUBW": {0x3B, 0x0, 0x20},
"SLLW": {0x3B, 0x1, 0x00}, "SLLW": {0x3B, 0x1, 0x00},
"SRLW": {0x3B, 0x5, 0x00}, "SRLW": {0x3B, 0x5, 0x00},
"SRAW": {0x3B, 0x5, 0x20}, "SRAW": {0x3B, 0x5, 0x20},
// RV64I — I-type shift-immediate (shamt in rs2 field). // RV64I, I-type shift-immediate (shamt in rs2 field).
"SLLI": {0x13, 0x1, 0x00}, "SLLI": {0x13, 0x1, 0x00},
"SRLI": {0x13, 0x5, 0x00}, "SRLI": {0x13, 0x5, 0x00},
"SRAI": {0x13, 0x5, 0x20}, "SRAI": {0x13, 0x5, 0x20},
"SLLIW": {0x1B, 0x1, 0x00}, "SLLIW": {0x1B, 0x1, 0x00},
"SRLIW": {0x1B, 0x5, 0x00}, "SRLIW": {0x1B, 0x5, 0x00},
"SRAIW": {0x1B, 0x5, 0x20}, "SRAIW": {0x1B, 0x5, 0x20},
// RV64M — multiply/divide. // RV64M, multiply/divide.
"MUL": {0x33, 0x0, 0x01}, "MUL": {0x33, 0x0, 0x01},
"MULH": {0x33, 0x1, 0x01}, "MULH": {0x33, 0x1, 0x01},
"MULHSU": {0x33, 0x2, 0x01}, "MULHSU": {0x33, 0x2, 0x01},
@@ -187,13 +187,13 @@ var riscvInstrTable = map[string]riscvEnc{
"DIVU": {0x33, 0x5, 0x01}, "DIVU": {0x33, 0x5, 0x01},
"REM": {0x33, 0x6, 0x01}, "REM": {0x33, 0x6, 0x01},
"REMU": {0x33, 0x7, 0x01}, "REMU": {0x33, 0x7, 0x01},
// RV64M — 32-bit variants. // RV64M, 32-bit variants.
"MULW": {0x3B, 0x0, 0x01}, "MULW": {0x3B, 0x0, 0x01},
"DIVW": {0x3B, 0x4, 0x01}, "DIVW": {0x3B, 0x4, 0x01},
"DIVUW": {0x3B, 0x5, 0x01}, "DIVUW": {0x3B, 0x5, 0x01},
"REMW": {0x3B, 0x6, 0x01}, "REMW": {0x3B, 0x6, 0x01},
"REMUW": {0x3B, 0x7, 0x01}, "REMUW": {0x3B, 0x7, 0x01},
// RV64I — I-type arithmetic. // RV64I, I-type arithmetic.
"ADDI": {0x13, 0x0, 0x00}, "ADDI": {0x13, 0x0, 0x00},
"ADDIW": {0x1B, 0x0, 0x00}, "ADDIW": {0x1B, 0x0, 0x00},
"SLTI": {0x13, 0x2, 0x00}, "SLTI": {0x13, 0x2, 0x00},
@@ -228,10 +228,10 @@ var riscvInstrTable = map[string]riscvEnc{
"ECALL": {0x73, 0x0, 0x00}, "ECALL": {0x73, 0x0, 0x00},
"EBREAK": {0x73, 0x0, 0x00}, "EBREAK": {0x73, 0x0, 0x00},
"FENCE": {0x0F, 0x0, 0x00}, "FENCE": {0x0F, 0x0, 0x00},
// JALR — indirect jump/call (I-type). // JALR, indirect jump/call (I-type).
"JALR": {0x67, 0x0, 0x00}, "JALR": {0x67, 0x0, 0x00},
// RV64A — atomics (AMO opcode 0x2F). // RV64A, atomics (AMO opcode 0x2F).
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27]. // funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2}, "AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2}, "AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
@@ -252,7 +252,7 @@ var riscvInstrTable = map[string]riscvEnc{
"AMOMINUW": {0x2F, 0x2, 0x18 << 2}, "AMOMINUW": {0x2F, 0x2, 0x18 << 2},
"AMOMINUD": {0x2F, 0x3, 0x18 << 2}, "AMOMINUD": {0x2F, 0x3, 0x18 << 2},
// RV64F/D — floating-point arithmetic. // RV64F/D, floating-point arithmetic.
"FADDS": {0x53, 0x0, 0x00}, "FADDS": {0x53, 0x0, 0x00},
"FSUBS": {0x53, 0x0, 0x04}, "FSUBS": {0x53, 0x0, 0x04},
"FMULS": {0x53, 0x0, 0x08}, "FMULS": {0x53, 0x0, 0x08},
@@ -274,13 +274,13 @@ var riscvInstrTable = map[string]riscvEnc{
"FMIND": {0x53, 0x0, 0x15}, "FMIND": {0x53, 0x0, 0x15},
"FMAXD": {0x53, 0x1, 0x15}, "FMAXD": {0x53, 0x1, 0x15},
// RV64A — load-reserved / store-conditional (funct5 0x02 / 0x03). // RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
"LRW": {0x2F, 0x2, 0x02 << 2}, "LRW": {0x2F, 0x2, 0x02 << 2},
"LRD": {0x2F, 0x3, 0x02 << 2}, "LRD": {0x2F, 0x3, 0x02 << 2},
"SCW": {0x2F, 0x2, 0x03 << 2}, "SCW": {0x2F, 0x2, 0x03 << 2},
"SCD": {0x2F, 0x3, 0x03 << 2}, "SCD": {0x2F, 0x3, 0x03 << 2},
// FP compare — result in integer register (funct7 0x50/0x51). // FP compare, result in integer register (funct7 0x50/0x51).
"FEQS": {0x53, 0x2, 0x50}, "FEQS": {0x53, 0x2, 0x50},
"FLTS": {0x53, 0x1, 0x50}, "FLTS": {0x53, 0x1, 0x50},
"FLES": {0x53, 0x0, 0x50}, "FLES": {0x53, 0x0, 0x50},
@@ -442,10 +442,10 @@ func riscvJType(rd int, offset int32) uint32 {
// ---- RVC (compressed) encoding helpers ---- // ---- RVC (compressed) encoding helpers ----
// isRVCIntReg reports whether a register number can be encoded in the 3-bit // isRVCIntReg reports whether a register number can be encoded in the 3-bit
// prime register field used by compressed instructions (x8–x15). // prime register field used by compressed instructions (x8-x15).
func isRVCIntReg(r int) bool { return r >= 8 && r <= 15 } func isRVCIntReg(r int) bool { return r >= 8 && r <= 15 }
// rvcReg3 returns the 3-bit encoding for registers x8–x15 (0–7). // rvcReg3 returns the 3-bit encoding for registers x8-x15 (0-7).
func rvcReg3(r int) uint32 { return uint32(r - 8) } func rvcReg3(r int) uint32 { return uint32(r - 8) }
// rvcCR encodes a CR-type (register) compressed instruction. // rvcCR encodes a CR-type (register) compressed instruction.
@@ -455,7 +455,7 @@ func rvcCR(funct4, rd, rs2 uint32) uint16 {
} }
// rvcCI encodes a CI-type (immediate) compressed instruction. // rvcCI encodes a CI-type (immediate) compressed instruction.
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW — linear 6-bit immediate. // Used for C.ADDI, C.LI, C.LUI, C.ADDIW, linear 6-bit immediate.
func rvcCI(funct3, rd uint32, imm uint32) uint16 { func rvcCI(funct3, rd uint32, imm uint32) uint16 {
return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x1) return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x1)
} }
+8 -8
View File
@@ -58,12 +58,12 @@ func riscvIsLeaf(t *ast.Text) bool {
case "CALL": case "CALL":
return false return false
case "JAL": case "JAL":
// JAL rd, target — a call only when rd is the link register. // JAL rd, target, a call only when rd is the link register.
if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 { if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 {
return false return false
} }
case "JALR": case "JALR":
// JALR rs1, rd — a call when rd is X1; JALR offset(rs1) always // JALR rs1, rd, a call when rd is X1; JALR offset(rs1) always
// links to X1. // links to X1.
if len(in.Operands) == 1 { if len(in.Operands) == 1 {
return false return false
@@ -84,12 +84,12 @@ func riscvPrologue(fi riscvFrameInfo) []byte {
return nil return nil
} }
var out []byte var out []byte
// MOV LR, -autosize(SP) — SD X1, -autosize(X2). The negative offset is // MOV LR, -autosize(SP), SD X1, -autosize(X2). The negative offset is
// not compressible to C.SDSP (unsigned), so it stays 4 bytes. // not compressible to C.SDSP (unsigned), so it stays 4 bytes.
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...) out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
// ADDI $-autosize, SP, SP — open the frame (C.ADDI when it fits). // ADDI $-autosize, SP, SP, open the frame (C.ADDI when it fits).
out = append(out, riscvSPAdjust(int32(-fi.autosize))...) out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
// MOV LR, 0(SP) — SD X1, 0(X2) → C.SDSP X1, 0. // MOV LR, 0(SP), SD X1, 0(X2) → C.SDSP X1, 0.
c := rvcSSP(0x7, 1, 0) c := rvcSSP(0x7, 1, 0)
out = append(out, byte(c), byte(c>>8)) out = append(out, byte(c), byte(c>>8))
return out return out
@@ -101,10 +101,10 @@ func riscvPrologue(fi riscvFrameInfo) []byte {
func riscvReturn(fi riscvFrameInfo) []byte { func riscvReturn(fi riscvFrameInfo) []byte {
var out []byte var out []byte
if fi.autosize != 0 { if fi.autosize != 0 {
// MOV 0(SP), LR — LD X1, 0(X2) → C.LDSP X1, 0. // MOV 0(SP), LR, LD X1, 0(X2) → C.LDSP X1, 0.
c := rvcLSP(0x3, 1, 0) c := rvcLSP(0x3, 1, 0)
out = append(out, byte(c), byte(c>>8)) out = append(out, byte(c), byte(c>>8))
// ADDI $autosize, SP, SP — close the frame (C.ADDI when it fits). // ADDI $autosize, SP, SP, close the frame (C.ADDI when it fits).
out = append(out, riscvSPAdjust(int32(fi.autosize))...) out = append(out, riscvSPAdjust(int32(fi.autosize))...)
} }
// JALR X0, 0(X1). // JALR X0, 0(X1).
@@ -143,7 +143,7 @@ func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
} }
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to // riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final JALR — the point where SP is restored. // (but not including) the final JALR, the point where SP is restored.
func riscvReturnEpilogueLen(fi riscvFrameInfo) int { func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
if fi.autosize == 0 { if fi.autosize == 0 {
return 0 return 0
+43 -43
View File
@@ -36,11 +36,11 @@ const (
vexNDS3Imm vexNDS3Imm
// vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg = // vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg =
// ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM // ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM
// source lives in the reg field, the destination in r/m — the PEXTR-style // source lives in the reg field, the destination in r/m, the PEXTR-style
// layout. VEXTRACTI128 and VEXTRACTF128 use this shape. // layout. VEXTRACTI128 and VEXTRACTF128 use this shape.
vexExtract vexExtract
// vexRMRev is the reversed two-operand form `OP src, dst` with the source // vexRMRev is the reversed two-operand form `OP src, dst` with the source
// in ModRM.reg and the destination in r/m — the layout of the EVEX // in ModRM.reg and the destination in r/m, the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD). // narrowing stores (VPMOVDW, VPMOVQD).
vexRMRev vexRMRev
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose // vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
@@ -68,7 +68,7 @@ type vexSpec struct {
// incrementally; every entry is covered by a byte-for-byte ground-truth test // incrementally; every entry is covered by a byte-for-byte ground-truth test
// against the Go assembler. // against the Go assembler.
var vexTable = map[string]vexSpec{ var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG — integer arithmetic / logic / compare. // VEX.128/256.66.0F.WIG, integer arithmetic / logic / compare.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3}, "VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3},
"VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3}, "VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3}, "VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3},
@@ -82,7 +82,7 @@ var vexTable = map[string]vexSpec{
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3}, "VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3},
"VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3}, "VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3},
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3}, "VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3},
// VEX.256.66.0F38.W0 — dword permute (three-operand NDS form). // VEX.256.66.0F38.W0, dword permute (three-operand NDS form).
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3}, "VPERMD": {2, 0x36, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG. // VEX.128/256.66.0F38.WIG.
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3}, "VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3},
@@ -90,14 +90,14 @@ var vexTable = map[string]vexSpec{
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3}, "VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3},
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3}, "VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic. // VEX.128/256.66.0F.WIG, packed double-precision arithmetic / logic.
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3}, "VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3}, "VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3}, "VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3}, "VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3}, "VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3}, "VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG — packed single-precision arithmetic. // VEX.128/256.0F.WIG, packed single-precision arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3}, "VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3}, "VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3}, "VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
@@ -107,7 +107,7 @@ var vexTable = map[string]vexSpec{
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3}, "VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3}, "VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3}, "VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
// VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed // VEX.128.F2.0F.WIG, scalar double-precision arithmetic (the packed
// opcodes with an F2 pp). // opcodes with an F2 pp).
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3}, "VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3}, "VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
@@ -115,7 +115,7 @@ var vexTable = map[string]vexSpec{
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3}, "VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3}, "VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3}, "VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
// VEX.128.F3.0F.WIG — scalar single-precision arithmetic (the packed // VEX.128.F3.0F.WIG, scalar single-precision arithmetic (the packed
// opcodes with an F3 pp). // opcodes with an F3 pp).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3}, "VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3}, "VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
@@ -123,10 +123,10 @@ var vexTable = map[string]vexSpec{
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3}, "VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3}, "VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3}, "VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
// VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form). // VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3}, "VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG — sign/zero extend and broadcast (reg=dst, rm=src, // VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
// no vvvv). // no vvvv).
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM}, "VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM}, "VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
@@ -143,70 +143,70 @@ var vexTable = map[string]vexSpec{
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM}, "VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
"VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM}, "VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM},
"VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM}, "VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM},
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion // VEX.128/256.F3.0F.WIG, signed dword to packed double conversion
// (reg=dst, rm=src, no vvvv; the length follows the destination). // (reg=dst, rm=src, no vvvv; the length follows the destination).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM}, "VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
// VEX.128/256.0F.WIG — signed dword to packed single conversion // VEX.128/256.0F.WIG, signed dword to packed single conversion
// (reg=dst, rm=src, no vvvv, no mandatory prefix). // (reg=dst, rm=src, no vvvv, no mandatory prefix).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM}, "VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
// VEX.128/256.0F.WIG — packed single to packed double conversion // VEX.128/256.0F.WIG, packed single to packed double conversion
// (reg=dst, rm=src; the destination is the wide operand and sets the // (reg=dst, rm=src; the destination is the wide operand and sets the
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but // length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
// the Go assembler emits the instruction with pp = 00, and gasm follows // the Go assembler emits the instruction with pp = 00, and gasm follows
// the Go assembler's bytes — its machine code is the oracle, not the // the Go assembler's bytes, its machine code is the oracle, not the
// manual. // manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM}, "VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
// VEX.128.F2.0F.WIG — duplicate the low double of each 128-bit lane // VEX.128.F2.0F.WIG, duplicate the low double of each 128-bit lane
// (reg=dst, rm=src, no vvvv; the length follows the destination). // (reg=dst, rm=src, no vvvv; the length follows the destination).
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM}, "VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
// VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src). // VEX.128/256.66.0F.WIG, move mask to a GPR (reg=gpr dst, rm=vec src).
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM}, "VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD) "VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
// VEX.128/256.66.0F.WIG — immediate shifts (opdigit selects the shift). // VEX.128/256.66.0F.WIG, immediate shifts (opdigit selects the shift).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm}, "VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm},
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm}, "VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm},
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm}, "VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm},
"VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm}, "VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm},
"VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm}, "VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm},
// VEX.128/256.66.0F.WIG — immediate shuffle (reg=dst, rm=src, imm8). // VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM}, "VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
// VEX.256.66.0F3A.W1 — qword permute (reg=dst, rm=src, imm8). // VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8).
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM}, "VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
// VEX.128/256.66.0F.WIG — two-source shuffle (reg=dst, vvvv=src1, rm=src2, // VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// imm8). // imm8).
"VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm}, "VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm},
// VEX.256.66.0F3A.W0 — permute / insert (same shape; VINSERTI128's rm is // VEX.256.66.0F3A.W0, permute / insert (same shape; VINSERTI128's rm is
// the XMM or memory source). // the XMM or memory source).
"VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm}, "VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm},
"VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm}, "VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm},
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8). // VEX.256.66.0F3A.W0, lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract}, "VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract}, "VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
// VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst: // VEX.128/256.66.0F3A.W0, half-precision convert back ($imm, src, dst:
// reg=src, rm=XMM/memory dst, imm8 — the extract layout). // reg=src, rm=XMM/memory dst, imm8, the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract}, "VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
// VEX.128.0F.W0 — no operands. // VEX.128.0F.W0, no operands.
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero}, "VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
// VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src). // VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM}, "KTESTW": {1, 0x99, 0, 0, -1, vexRM},
// VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst, // VEX.66.0F38.W0, broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; SD is 256-bit only). // rm=scalar memory; SD is 256-bit only).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM}, "VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM}, "VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
// VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width // VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
// source). // source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM}, "VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src). // VEX.F3.0F.WIG, replicate even/odd singles (reg=dst, rm=src).
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM}, "VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM}, "VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y // VEX.66.0F.WIG, packed double to packed single conversion, the X/Y
// spellings: the destination is always XMM and the spelling fixes the // spellings: the destination is always XMM and the spelling fixes the
// source length (X = 128, Y = 256). // source length (X = 128, Y = 256).
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen}, "VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
@@ -230,14 +230,14 @@ var vexTable = map[string]vexSpec{
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3}, "VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3}, "VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift). // VEX.128/256.66.0F.WIG, word shifts (opdigit selects the shift).
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm}, "VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm}, "VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm}, "VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
// VEX.F2.0F — packed double to packed dword conversions, truncating and // VEX.F2.0F, packed double to packed dword conversions, truncating and
// non-truncating. The destination is always XMM; the X/Y spellings fix // non-truncating. The destination is always XMM; the X/Y spellings fix
// the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen. // the source length (XMM/YMM), and VEX.L follows it, see vexSrcLen.
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen}, "VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen}, "VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen}, "VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
@@ -257,7 +257,7 @@ var vexSrcLen = map[string]int{
"VCVTPD2PSY": 1, "VCVTPD2PSY": 1,
} }
// vexVarShift maps the shift mnemonics to their variable-count opcode — the // vexVarShift maps the shift mnemonics to their variable-count opcode, the
// form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8), // form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8),
// an ordinary NDS encoding rather than the /digit immediate form above. // an ordinary NDS encoding rather than the /digit immediate form above.
var vexVarShift = map[string]byte{ var vexVarShift = map[string]byte{
@@ -288,20 +288,20 @@ type vexMoveSpec struct {
// vexMoveTable maps an upper-case move mnemonic to its encoding. // vexMoveTable maps an upper-case move mnemonic to its encoding.
var vexMoveTable = map[string]vexMoveSpec{ var vexMoveTable = map[string]vexMoveSpec{
// VEX.128/256.F3.0F.WIG — unaligned integer move. // VEX.128/256.F3.0F.WIG, unaligned integer move.
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false}, "VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG — unaligned packed double move. // VEX.128/256.66.0F.WIG, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false}, "VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
// VEX.128.66.0F.W0 — 32-bit GPR/memory ↔ XMM. // VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
"VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true}, "VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true},
// VMOVQ — 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm). // VMOVQ, 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true}, "VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true},
// VEX.128.F2.0F.WIG — scalar double move, memory operands only (the // VEX.128.F2.0F.WIG, scalar double move, memory operands only (the
// register form takes three operands and is not supported yet). // register form takes three operands and is not supported yet).
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true}, "VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128.F3.0F.WIG — scalar single move, memory operands only. // VEX.128.F3.0F.WIG, scalar single move, memory operands only.
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true}, "VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128/256 — aligned packed moves. // VEX.128/256, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false}, "VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false}, "VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
} }
@@ -317,7 +317,7 @@ func isVex(mnemUpper string) bool {
// encodeVex encodes a VEX instruction with operands in Plan 9 order. // encodeVex encodes a VEX instruction with operands in Plan 9 order.
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error { func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
// Vector register indices 16–31 exist only in EVEX encodings; fail // Vector register indices 16-31 exist only in EVEX encodings; fail
// loudly rather than silently truncating the index. // loudly rather than silently truncating the index.
for _, op := range ops { for _, op := range ops {
if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 { if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 {
@@ -420,7 +420,7 @@ func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
} }
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with // encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the VEX.L bit following the source — fixed // the destination always XMM and the VEX.L bit following the source, fixed
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when // by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
// the source is memory. // the source is memory.
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error { func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
+1 -1
View File
@@ -17,7 +17,7 @@ type File struct {
Orphans []Stmt // labels/instructions seen before any TEXT directive Orphans []Stmt // labels/instructions seen before any TEXT directive
// Macros holds the names introduced by #define directives in this file. // Macros holds the names introduced by #define directives in this file.
// The linter uses it to avoid flagging macro invocations as unknown // The linter uses it to avoid flagging macro invocations as unknown
// instructions (macro expansion itself is out of scope — see the docs). // instructions (macro expansion itself is out of scope, see the docs).
Macros map[string]bool Macros map[string]bool
} }
+4 -4
View File
@@ -1,7 +1,7 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause // SPDX-License-Identifier: BSD-3-Clause
// Package format implements a canonical formatter for GAsm source — the // Package format implements a canonical formatter for GAsm source, the
// equivalent of gofmt for Plan 9 assembly. It works on the token stream // equivalent of gofmt for Plan 9 assembly. It works on the token stream
// rather than the AST so that every line (including comments and blanks) is // rather than the AST so that every line (including comments and blanks) is
// preserved; it only normalises indentation, operand spacing and per-function // preserved; it only normalises indentation, operand spacing and per-function
@@ -92,7 +92,7 @@ func Source(src string) string {
case kInstr: case kInstr:
out = renderInstr(line, maxWidth[inf.funcID]) out = renderInstr(line, maxWidth[inf.funcID])
// A RET ends the body for indentation purposes: comments that // A RET ends the body for indentation purposes: comments that
// follow it — typically the next function's doc comment — belong // follow it, typically the next function's doc comment, belong
// at column 0, not inside the finished function. // at column 0, not inside the finished function.
if strings.EqualFold(line[0].Text, "RET") { if strings.EqualFold(line[0].Text, "RET") {
inBody = false inBody = false
@@ -120,8 +120,8 @@ type outLine struct {
} }
// normalizeSpacing enforces the canonical blank-line layout: runs of blank // normalizeSpacing enforces the canonical blank-line layout: runs of blank
// lines collapse to one, and a new block — a label, or a TEXT or GLOBL // lines collapse to one, and a new block, a label, or a TEXT or GLOBL
// directive — is preceded by exactly one blank line. Comments immediately // directive, is preceded by exactly one blank line. Comments immediately
// above a block belong to it, so the blank line is inserted before them. No // above a block belong to it, so the blank line is inserted before them. No
// blank line is forced at the top of the file, right after a TEXT (the // blank line is forced at the top of the file, right after a TEXT (the
// function's first label), or between stacked labels that share an address. // function's first label), or between stacked labels that share an address.
+1 -1
View File
@@ -38,7 +38,7 @@ func enterJITChecked(fn uintptr, stack uintptr)
// leaveJITCheckedRaw is the raw return trampoline for ABI checks. Its // leaveJITCheckedRaw is the raw return trampoline for ABI checks. Its
// address is obtained from the GLOBL in abi_amd64.s (leaveCheckedPtr), // address is obtained from the GLOBL in abi_amd64.s (leaveCheckedPtr),
// which points to the .abi0 code — NOT the ABIInternal wrapper that this // which points to the .abi0 code, NOT the ABIInternal wrapper that this
// declaration would generate. The declaration exists solely to satisfy // declaration would generate. The declaration exists solely to satisfy
// go vet's "missing Go declaration" check. // go vet's "missing Go declaration" check.
// //
+1 -1
View File
@@ -22,7 +22,7 @@ func enterJITChecked(fn uintptr, stack uintptr)
// leaveJITCheckedRaw is the raw return trampoline for ABI checks. Its // leaveJITCheckedRaw is the raw return trampoline for ABI checks. Its
// address is obtained from the GLOBL in abi_arm64.s (leaveCheckedPtr), // address is obtained from the GLOBL in abi_arm64.s (leaveCheckedPtr),
// which points to the .abi0 code — NOT the ABIInternal wrapper that this // which points to the .abi0 code, NOT the ABIInternal wrapper that this
// declaration would generate. The declaration exists solely to satisfy // declaration would generate. The declaration exists solely to satisfy
// go vet's "missing Go declaration" check. // go vet's "missing Go declaration" check.
// //
+1 -1
View File
@@ -22,7 +22,7 @@ func enterJITChecked(fn uintptr, stack uintptr)
// leaveJITCheckedRaw is the raw return trampoline for ABI checks. Its // leaveJITCheckedRaw is the raw return trampoline for ABI checks. Its
// address is obtained from the GLOBL in abi_loong64.s (leaveCheckedPtr), // address is obtained from the GLOBL in abi_loong64.s (leaveCheckedPtr),
// which points to the .abi0 code — NOT the ABIInternal wrapper that this // which points to the .abi0 code, NOT the ABIInternal wrapper that this
// declaration would generate. The declaration exists solely to satisfy // declaration would generate. The declaration exists solely to satisfy
// go vet's "missing Go declaration" check. // go vet's "missing Go declaration" check.
// //
+1 -1
View File
@@ -22,7 +22,7 @@ func enterJITChecked(fn uintptr, stack uintptr)
// leaveJITCheckedRaw is the raw return trampoline for ABI checks. Its // leaveJITCheckedRaw is the raw return trampoline for ABI checks. Its
// address is obtained from the GLOBL in abi_riscv64.s (leaveCheckedPtr), // address is obtained from the GLOBL in abi_riscv64.s (leaveCheckedPtr),
// which points to the .abi0 code — NOT the ABIInternal wrapper that this // which points to the .abi0 code, NOT the ABIInternal wrapper that this
// declaration would generate. The declaration exists solely to satisfy // declaration would generate. The declaration exists solely to satisfy
// go vet's "missing Go declaration" check. // go vet's "missing Go declaration" check.
// //
+1 -1
View File
@@ -57,7 +57,7 @@ const stackPad = 64
// (the ABI0 convention shares the argument area for inputs and outputs). // (the ABI0 convention shares the argument area for inputs and outputs).
// //
// The function must be NOSPLIT (no stack growth) and must not reference // The function must be NOSPLIT (no stack growth) and must not reference
// external symbols — the image is self-contained. // external symbols, the image is self-contained.
func Call(fnAddr uintptr, args []byte) ([]byte, error) { func Call(fnAddr uintptr, args []byte) ([]byte, error) {
// Prepare the stack: [padding][leaveJIT addr][args...] // Prepare the stack: [padding][leaveJIT addr][args...]
stackSize := stackPad + 8 + len(args) + 64 // padding + ret + args + safety stackSize := stackPad + 8 + len(args) + 64 // padding + ret + args + safety
+2 -2
View File
@@ -33,7 +33,7 @@ func (r FuzzResult) String() string {
if r.OK() { if r.OK() {
return fmt.Sprintf("%s: %d/%d iterations match", r.Func, r.Matches, r.Iterations) return fmt.Sprintf("%s: %d/%d iterations match", r.Func, r.Matches, r.Iterations)
} }
s := fmt.Sprintf("%s: %d/%d match, %d MISMATCH — %s", s := fmt.Sprintf("%s: %d/%d match, %d MISMATCH: %s",
r.Func, r.Matches, r.Iterations, r.Mismatches, r.FirstFail) r.Func, r.Matches, r.Iterations, r.Mismatches, r.FirstFail)
if len(r.CrashInput) > 0 { if len(r.CrashInput) > 0 {
s += fmt.Sprintf("\n input: %x", r.CrashInput) s += fmt.Sprintf("\n input: %x", r.CrashInput)
@@ -295,7 +295,7 @@ func genDualArgs(rng *rand.Rand, sig funcSig, argSize int) (gasmArgs, goArgs []b
bufs = append(bufs, buf1, buf2) bufs = append(bufs, buf1, buf2)
putPtr(gasmArgs, off, unsafe.Pointer(&buf1[0])) putPtr(gasmArgs, off, unsafe.Pointer(&buf1[0]))
putPtr(goArgs, off, unsafe.Pointer(&buf2[0])) putPtr(goArgs, off, unsafe.Pointer(&buf2[0]))
// len and cap both equal declaredLen — the buffer is guaranteed // len and cap both equal declaredLen, the buffer is guaranteed
// to hold at least declaredLen elements plus safety margin. // to hold at least declaredLen elements plus safety margin.
putU64(gasmArgs, off+8, uint64(declaredLen)) putU64(gasmArgs, off+8, uint64(declaredLen))
putU64(gasmArgs, off+16, uint64(declaredLen)) putU64(gasmArgs, off+16, uint64(declaredLen))
+2 -2
View File
@@ -19,7 +19,7 @@ func TestFuzzResultString(t *testing.T) {
t.Run("mismatch", func(t *testing.T) { t.Run("mismatch", func(t *testing.T) {
r := FuzzResult{Func: "mul", Iterations: 100, Matches: 95, Mismatches: 5, FirstFail: "iter 23"} r := FuzzResult{Func: "mul", Iterations: 100, Matches: 95, Mismatches: 5, FirstFail: "iter 23"}
s := r.String() s := r.String()
if s != "mul: 95/100 match, 5 MISMATCH — iter 23" { if s != "mul: 95/100 match, 5 MISMATCH: iter 23" {
t.Errorf("String() = %q", s) t.Errorf("String() = %q", s)
} }
}) })
@@ -27,7 +27,7 @@ func TestFuzzResultString(t *testing.T) {
t.Run("crash", func(t *testing.T) { t.Run("crash", func(t *testing.T) {
r := FuzzResult{Func: "dec", Iterations: 100, Matches: 99, Mismatches: 1, FirstFail: "SIGSEGV", CrashInput: []byte{0x01, 0x02}} r := FuzzResult{Func: "dec", Iterations: 100, Matches: 99, Mismatches: 1, FirstFail: "SIGSEGV", CrashInput: []byte{0x01, 0x02}}
s := r.String() s := r.String()
if s != "dec: 99/100 match, 1 MISMATCH — SIGSEGV\n input: 0102" { if s != "dec: 99/100 match, 1 MISMATCH: SIGSEGV\n input: 0102" {
t.Errorf("String() = %q", s) t.Errorf("String() = %q", s)
} }
}) })