Compare commits

..
16 Commits
Author SHA1 Message Date
petrbalvin 51a2854d7f feat(verify): add analyzeO2/Res and decodeMono24 differential tests
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin c234c3dd5b feat(verify): add analyzeO1Range and fastStereoSums differential tests
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin 9262990ce5 feat(verify): extend differential tests to go-flac and AVX-512, add --abi/--profile CLI flags
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin d9f6167a4d feat(verify): add basic-block enumeration and path-diversity profiling
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin f5088c52fc feat(verify): add runtime ABI checks with sentinel registers and red-zone canary
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin f52e23f1bc feat(verify): add differential fuzz testing against a portable Go reference
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin c9775c2b95 feat(verify): add the JIT execution substrate and gasm verify subcommand
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin 5af12e15ac feat(asm): add the GPR-interchanging conversions, completing the amd64 EVEX set
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin db8e3fc160 docs: record the deferred GOOBJ external-symbols decision 2026-08-02 23:11:30 +02:00
petrbalvin 1312122a99 feat(asm): complete the EVEX conversions, narrowing and mask-vector moves
Assisted-by: Qwen 3.8 Max Preview
2026-07-20 16:05:08 +02:00
petrbalvin 11f962fbcc feat(asm): add the EVEX FP helper tail and gather/scatter with VSIB
Assisted-by: Qwen 3.8 Max Preview
2026-07-19 15:58:48 +02:00
petrbalvin ee68859beb feat(asm): add the wider EVEX set and the rounding, SAE and broadcast suffixes
Assisted-by: Qwen 3.8 Max Preview
2026-07-18 15:47:59 +02:00
petrbalvin 0920edb092 feat(asm): emit GOOBJ objects that link directly with the Go toolchain
Assisted-by: Qwen 3.8 Max Preview
2026-07-17 18:57:04 +02:00
petrbalvin 900c9772b1 feat(asm): emit linkable ELF and Mach-O objects with external symbols
Assisted-by: Qwen 3.8 Max Preview
2026-07-16 20:52:20 +02:00
petrbalvin b914c0e390 feat(asm): add the EVEX floating-point and conversion set
Assisted-by: Qwen 3.8 Max Preview
2026-07-15 17:13:28 +02:00
petrbalvin 0f3146ff2c feat(asm): add EVEX masking, zeroing and the AVX-512 F/BW integer set
Assisted-by: Qwen 3.8 Max Preview
2026-07-14 21:03:26 +02:00
39 changed files with 6629 additions and 207 deletions
+10
View File
@@ -156,6 +156,16 @@ func (t *Table) Lookup(mnemonic string) (Instr, bool) {
}
}
}
// amd64 EVEX instructions take a .Z zeroing suffix (masking is written as
// an explicit K operand rather than a suffix); strip it so the base
// instruction is still recognised.
if t.Arch == AMD64 {
if base, ok := strings.CutSuffix(key, ".Z"); ok {
if in, found := t.instrs[base]; found {
return in, true
}
}
}
return Instr{}, false
}
+43 -10
View File
@@ -23,16 +23,19 @@ import (
// operands require relocations and are not yet supported; the SIMD (VEX/AVX2)
// integer and shuffle/extract/permute/move set is in.
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
code, _, labels, err := assemble(t, nil)
code, _, labels, _, err := assemble(t, nil)
return code, labels, err
}
// linkInfo carries file-level symbol context into a single-function assembly:
// the set of static symbols a GLOBL in the same file defines. A nil link
// rejects SB operands outright (single-function assembly cannot resolve
// them).
// them). When allowExternal is set, a reference to a symbol no GLOBL in the
// file defines is recorded as an external relocation instead of failing —
// the object-file emitters resolve it at link time.
type linkInfo struct {
symbols map[string]bool
symbols map[string]bool
allowExternal bool
}
// sbPatch is a function-relative static-symbol relocation: the disp32 field
@@ -45,9 +48,19 @@ type sbPatch struct {
addend int64
}
// spadjStep is one stack-adjustment boundary within a function: Value is the
// SP delta from the entry state (just below the return address) in effect
// from PC (function-relative) until the next step. The steps feed the
// pcsp table of the object-file emitters.
type spadjStep struct {
pc int
value int
}
// assemble encodes a TEXT body, returning the machine code, the static-symbol
// patch sites (for the file-level layout to resolve) and the label table.
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, error) {
// patch sites (for the file-level layout to resolve), the label table and the
// stack-adjustment boundaries.
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, error) {
fi := computeFrame(t)
chain := jumpChain(t)
resolve := func(name string) string {
@@ -71,7 +84,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, e
case *ast.Instr:
sz, err := instrSize(s, fi, long[i], link)
if err != nil {
return nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
return nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
}
sizes[i] = sz
pcs[i] = pos
@@ -111,24 +124,42 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, e
// Pass 2: emit.
out := append([]byte(nil), fi.prologue...)
var patches []sbPatch
var steps []spadjStep
if fi.useFP {
// PUSHQ BP saves the return-address-relative base (+8); the MOVQ
// changes nothing; SUBQ $size, SP completes the frame.
steps = append(steps,
spadjStep{1, 8},
spadjStep{len(fi.prologue), 8 + fi.size},
)
}
pos := len(fi.prologue)
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
if !ok {
continue
}
if strings.ToUpper(s.Mnemonic.Text) == "RET" && fi.useFP {
// The RET's epilogue prefix unwinds: ADDQ $size, SP restores
// the saved-BP-only stack, POPQ BP the entry state.
epi := len(fi.epilogue)
steps = append(steps,
spadjStep{pos + epi - 1, 8},
spadjStep{pos + epi, 0},
)
}
code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
if err != nil {
return nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
return nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
}
if len(code) != sizes[i] {
return nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
return nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
}
patches = append(patches, ps...)
out = append(out, code...)
pos += len(code)
}
return out, patches, offsets, nil
return out, patches, offsets, steps, nil
}
// jumpChain precomputes jump-to-jump folding: a label whose first instruction
@@ -425,7 +456,9 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op
if a.Sym.Static {
return nil, fmt.Errorf("undefined symbol %q", a.Sym.Name)
}
return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name)
if !link.allowExternal {
return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name)
}
}
return sbMem{size: size, name: a.Sym.Name, addend: a.Sym.Offset}, nil
}
+301
View File
@@ -0,0 +1,301 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// This file emits ELF64 relocatable objects (ET_REL) from an assembled
// Image: a .text section holding the function bodies, a .data section
// holding the GLOBL initialisers, a symbol table with one symbol per TEXT
// and GLOBL (file-local <> symbols are STB_LOCAL, the rest STB_GLOBAL), and
// a .rela.text relocation table — one R_X86_64_PC32 entry per static-symbol
// reference, internal references resolving against the local data symbols
// and external ones against undefined globals. The output links with the
// system toolchain (cc/ld) the way a hand-assembled .o would.
// ELF constants (ELF64, little-endian, System V).
const (
elfClass64 = 2
elfDataLSB = 1
elfVersion = 1
etREL = 1 // relocatable object
emX8664 = 62
shtNull = 0
shtProgbits = 1
shtSymtab = 2
shtStrtab = 3
shtRela = 4
shfWrite = 1
shfAlloc = 2
shfExecInstr = 4
stbLocal = 0
stbGlobal = 1
sttNotype = 0
sttObject = 1
sttFunc = 2
sttSection = 3
stInfoShift = 4
shnUndef = 0
rX8664PC32 = 2
)
// elfSym is one symbol-table entry in construction.
type elfSym struct {
name string
info byte
shndx uint16
value uint64
size uint64
}
// ELFObject returns the image as an ELF64 relocatable object file, ready for
// the system linker. Symbol names are the TEXT and GLOBL identifiers as
// written (the middle dot stripped); a package prefix, when present, is
// joined with a dot. Every static-symbol reference becomes an
// R_X86_64_PC32 relocation, so the code is position-independent and links
// at any address.
func (img *Image) ELFObject() ([]byte, error) {
le := binary.LittleEndian
// Section indices: 0 NULL, 1 .text, 2 .data; the tables follow.
const (
secText = 1
secData = 2
)
// Build the symbol table: the null entry and the two section symbols
// come first, then the local symbols (static TEXT and GLOBL), then the
// globals (exported TEXT and GLOBL, and the undefined externals) — ELF
// requires every local to precede every global, and sh_info records the
// boundary. symIdx maps a symbol name to its index for the relocations.
var locals, globals []elfSym
for _, fn := range img.Funcs {
s := elfSym{
name: objectName(fn.Pkg, fn.Name),
info: sttFunc,
shndx: secText,
value: uint64(fn.Offset),
size: uint64(fn.Size),
}
if fn.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := elfSym{
name: objectName(d.Pkg, d.Name),
info: sttObject,
shndx: secData,
value: uint64(d.Offset),
size: uint64(d.Size),
}
if d.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, name := range img.Externals {
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
}
syms := []elfSym{
{}, // the mandatory null entry
{name: ".text", info: sttSection, shndx: secText},
{name: ".data", info: sttSection, shndx: secData},
}
syms = append(syms, locals...)
shInfo := len(syms) // first global symbol
syms = append(syms, globals...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Build the relocations.
type elfRela struct {
off uint64
sym int
addend int64
}
var relas []elfRela
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
sym: idx,
// R_X86_64_PC32 computes S + A − P with P the patch site; the
// assembler measures the symbol from the instruction end,
// After − Off bytes past the field, so the addend carries
// that distance with a negative sign.
addend: r.Addend - int64(r.After-r.Off),
})
}
}
// Serialise the string tables.
stNames := newElfStrtab()
for _, s := range syms {
stNames.add(s.name)
}
stSections := newElfStrtab()
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
// Section presence: .rela.text only when there are relocations.
hasRela := len(relas) > 0
nSections := 6 // NULL, .text, .data, .symtab, .strtab, .shstrtab
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Lay the file out: header, section data, section headers.
var out []byte
out = append(out, make([]byte, 64)...) // ELF header, filled last
align := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
align(16)
textOff := len(out)
out = append(out, img.Code...)
align(16)
dataOff := len(out)
out = append(out, img.Data...)
align(8)
symtabOff := len(out)
for _, s := range syms {
var b [24]byte
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
b[4] = s.info
b[5] = 0 // st_other
le.PutUint16(b[6:], s.shndx)
le.PutUint64(b[8:], s.value)
le.PutUint64(b[16:], s.size)
out = append(out, b[:]...)
}
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
if hasRela {
align(8)
relaOff = len(out)
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|rX8664PC32)
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
align(8)
shoff := len(out)
// Section headers.
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
var b [64]byte
le.PutUint32(b[0:], uint32(stSections.at(name)))
le.PutUint32(b[4:], uint32(typ))
le.PutUint64(b[8:], flags)
le.PutUint64(b[16:], 0) // sh_addr
le.PutUint64(b[24:], uint64(off))
le.PutUint64(b[32:], uint64(size))
le.PutUint32(b[40:], uint32(link))
le.PutUint32(b[44:], uint32(info))
le.PutUint64(b[48:], alignV)
le.PutUint64(b[56:], entsize)
out = append(out, b[:]...)
}
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// The ELF header.
hdr := out[:64]
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
le.PutUint16(hdr[16:], etREL)
le.PutUint16(hdr[18:], emX8664)
le.PutUint32(hdr[20:], elfVersion)
le.PutUint64(hdr[24:], 0) // e_entry
le.PutUint64(hdr[32:], 0) // e_phoff
le.PutUint64(hdr[40:], uint64(shoff)) // e_shoff
le.PutUint32(hdr[48:], 0) // e_flags
le.PutUint16(hdr[52:], 64) // e_ehsize
le.PutUint16(hdr[54:], 0) // e_phentsize
le.PutUint16(hdr[56:], 0) // e_phnum
le.PutUint16(hdr[58:], 64) // e_shentsize
le.PutUint16(hdr[60:], uint16(nSections))
le.PutUint16(hdr[62:], uint16(secShstr))
return out, nil
}
// objectName renders a symbol's object-file name: the identifier as written,
// with an explicit package prefix joined by a dot.
func objectName(pkg, name string) string {
if pkg == "" {
return name
}
return pkg + "." + name
}
// elfStrtab is an ELF string table under construction.
type elfStrtab struct {
buf []byte
off map[string]int
}
func newElfStrtab() *elfStrtab {
return &elfStrtab{buf: []byte{0}, off: map[string]int{"": 0}}
}
func (s *elfStrtab) add(name string) {
if _, ok := s.off[name]; ok {
return
}
s.off[name] = len(s.buf)
s.buf = append(s.buf, name...)
s.buf = append(s.buf, 0)
}
func (s *elfStrtab) at(name string) int { return s.off[name] }
func (s *elfStrtab) bytes() []byte { return s.buf }
+310
View File
@@ -0,0 +1,310 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"os"
"os/exec"
"path/filepath"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// The object-file tests share one source: two exported functions, one
// file-local constant reached through a relocation, and one external symbol
// the linker must resolve. The functions take their arguments in the System
// V registers (not the Go stack ABI) so a C driver can call them directly.
const elfTestSrc = `
#include "textflag.h"
TEXT ·addq(SB), NOSPLIT, $0
LEAQ (DI)(SI*1), AX
RET
TEXT ·getanswer(SB), NOSPLIT, $0
MOVQ answer<>(SB), AX
RET
TEXT ·useextern(SB), NOSPLIT, $0
MOVQ extvar(SB), AX
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`
func elfTestImage(t *testing.T) *Image {
t.Helper()
f, errs := parser.Parse("t_amd64.s", elfTestSrc)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
return img
}
// TestAssembleFileExternals checks that a reference to a symbol no GLOBL
// defines is recorded as an external relocation instead of failing — the
// raw image leaves the displacement zero, the object emitters carry it.
func TestAssembleFileExternals(t *testing.T) {
img := elfTestImage(t)
if len(img.Externals) != 1 || img.Externals[0] != "extvar" {
t.Fatalf("Externals = %v, want [extvar]", img.Externals)
}
var ext, local int
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.External {
ext++
if r.Name != "extvar" {
t.Errorf("external reloc names %q, want extvar", r.Name)
}
} else {
local++
if r.Name != "answer" {
t.Errorf("local reloc names %q, want answer", r.Name)
}
}
}
}
if ext != 1 || local != 1 {
t.Errorf("relocs = %d external, %d local; want 1 and 1", ext, local)
}
}
// TestELFObject checks the structure of the emitted ELF64 relocatable
// object: sections, the symbol table (bindings, types, values, sizes) and
// the .rela.text relocations, parsed back with debug/elf.
func TestELFObject(t *testing.T) {
img := elfTestImage(t)
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
f, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer f.Close()
if f.Type != elf.ET_REL || f.Machine != elf.EM_X86_64 {
t.Errorf("type/machine = %v/%v, want ET_REL/EM_X86_64", f.Type, f.Machine)
}
text := f.Section(".text")
data := f.Section(".data")
if text == nil || data == nil {
t.Fatal("missing .text or .data section")
}
if text.Flags&elf.SHF_EXECINSTR == 0 || text.Flags&elf.SHF_ALLOC == 0 {
t.Errorf(".text flags = %v", text.Flags)
}
if data.Flags&elf.SHF_WRITE == 0 {
t.Errorf(".data flags = %v", data.Flags)
}
textData, err := text.Data()
if err != nil {
t.Fatal(err)
}
if !bytes.Equal(textData, img.Code) {
t.Errorf(".text contents differ from the image code")
}
syms, err := f.Symbols()
if err != nil {
t.Fatalf("symbols: %v", err)
}
byName := map[string]elf.Symbol{}
for _, s := range syms {
byName[s.Name] = s
}
wantSym := func(name string, bind elf.SymBind, typ elf.SymType, section elf.SectionIndex, size uint64) {
t.Helper()
s, ok := byName[name]
if !ok {
t.Errorf("symbol %q not found", name)
return
}
if elf.ST_BIND(s.Info) != bind || elf.ST_TYPE(s.Info) != typ {
t.Errorf("%s: bind/type = %v/%v, want %v/%v", name, elf.ST_BIND(s.Info), elf.ST_TYPE(s.Info), bind, typ)
}
if s.Section != section {
t.Errorf("%s: section = %v, want %v", name, s.Section, section)
}
if s.Size != size {
t.Errorf("%s: size = %d, want %d", name, s.Size, size)
}
}
// The emitted layout is fixed: 0 NULL, 1 .text, 2 .data.
if f.Sections[1].Name != ".text" || f.Sections[2].Name != ".data" {
t.Fatalf("section layout = %s, %s; want .text, .data", f.Sections[1].Name, f.Sections[2].Name)
}
textIdx := elf.SectionIndex(1)
dataIdx := elf.SectionIndex(2)
wantSym("addq", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 5)
wantSym("getanswer", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 8)
wantSym("useextern", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 8)
wantSym("answer", elf.STB_LOCAL, elf.STT_OBJECT, dataIdx, 8)
wantSym("extvar", elf.STB_GLOBAL, elf.STT_NOTYPE, elf.SHN_UNDEF, 0)
// Relocations: one for the file-local constant (resolving against the
// local data symbol) and one for the external (against the undefined
// global), both R_X86_64_PC32 with the −4 addend the PC-relative form
// needs. debug/elf does not surface rela entries, so read the section
// directly.
relaSec := f.Section(".rela.text")
if relaSec == nil {
t.Fatal("missing .rela.text")
}
raw, err := relaSec.Data()
if err != nil {
t.Fatal(err)
}
if len(raw)%24 != 0 || len(raw)/24 != 2 {
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
}
// Symbol names straight from the raw tables: r_info carries an index
// into .symtab including the null entry, which debug/elf's Symbols()
// slice may not mirror.
symtabRaw, err := f.Section(".symtab").Data()
if err != nil {
t.Fatal(err)
}
strtabRaw, err := f.Section(".strtab").Data()
if err != nil {
t.Fatal(err)
}
symName := func(idx int) string {
stName := binary.LittleEndian.Uint32(symtabRaw[idx*24:])
end := bytes.IndexByte(strtabRaw[stName:], 0)
return string(strtabRaw[stName : int(stName)+end])
}
for i := 0; i < 2; i++ {
e := raw[i*24 : (i+1)*24]
off := binary.LittleEndian.Uint64(e[0:])
info := binary.LittleEndian.Uint64(e[8:])
addend := int64(binary.LittleEndian.Uint64(e[16:]))
typ := info & 0xffffffff
sym := int(info >> 32)
if typ != uint64(elf.R_X86_64_PC32) {
t.Errorf("reloc %d: type %d, want R_X86_64_PC32", i, typ)
}
if addend != -4 {
t.Errorf("reloc %d: addend %d, want -4", i, addend)
}
if name := symName(sym); name != "answer" && name != "extvar" {
t.Errorf("reloc %d: symbol %q, want answer or extvar", i, name)
}
// The relocation offset lands on the disp32 field: the four bytes
// before a RET-terminated eight-byte MOVQ.
if off+4 > uint64(len(textData)) {
t.Errorf("reloc %d: offset %d outside .text", i, off)
}
}
}
// TestELFObjectNoRelocations checks a file with no static-symbol references
// emits a valid object without a .rela.text section.
func TestELFObjectNoRelocations(t *testing.T) {
f, errs := parser.Parse("n_amd64.s", `
#include "textflag.h"
TEXT ·nop(SB), NOSPLIT, $0
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Section(".rela.text") != nil {
t.Error("unexpected .rela.text section")
}
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
found := false
for _, s := range syms {
if s.Name == "nop" && elf.ST_TYPE(s.Info) == elf.STT_FUNC {
found = true
}
}
if !found {
t.Error("function symbol nop not found")
}
}
// TestELFLinkAndRun is the end-to-end check: assemble the test functions,
// link the emitted object with a C driver that defines the external symbol,
// and run the result. Skipped when no C compiler is available.
func TestELFLinkAndRun(t *testing.T) {
cc, err := exec.LookPath("cc")
if err != nil {
t.Skip("no C compiler available")
}
dir := t.TempDir()
img := elfTestImage(t)
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
objPath := filepath.Join(dir, "t.o")
if err := os.WriteFile(objPath, obj, 0o644); err != nil {
t.Fatal(err)
}
const driver = `
#include <stdio.h>
long addq(long a, long b);
long getanswer(void);
long useextern(void);
long extvar = 7;
int main(void) {
printf("%ld %ld %ld\n", addq(41, 1), getanswer(), useextern());
return 0;
}
`
driverPath := filepath.Join(dir, "driver.c")
if err := os.WriteFile(driverPath, []byte(driver), 0o644); err != nil {
t.Fatal(err)
}
// -no-pie: the encoder emits R_X86_64_PC32 for external references,
// which a position-independent executable would reject (it wants
// PLT32/GOT relocations, a future increment).
appPath := filepath.Join(dir, "app")
out, err := exec.Command(cc, "-no-pie", "-o", appPath, driverPath, objPath).CombinedOutput()
if err != nil {
t.Fatalf("link failed: %v\n%s", err, out)
}
run, err := exec.Command(appPath).CombinedOutput()
if err != nil {
t.Fatalf("run failed: %v\n%s", err, run)
}
if got := string(run); got != "42 42 7\n" {
t.Errorf("output %q, want \"42 42 7\\n\"", got)
}
}
+34 -8
View File
@@ -51,9 +51,17 @@ func (e *enc) encode(mnem string, ops []Operand) error {
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
// before splitSize.
if isVex(upper) || isEvex(upper) || upper == "KMOVW" {
return e.encodeVec(upper, ops)
// before splitSize. EVEX suffixes (.Z, .SAE, rounding, .BCST) split
// off the mnemonic too.
base, sfx, err := parseEvexSuffix(upper)
if err != nil {
return err
}
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
return e.encodeVec(base, ops, sfx)
}
if sfx.any() {
return fmt.Errorf("%s: the suffix requires an EVEX instruction", mnem)
}
// CMOVcc and SETcc carry the condition in the mnemonic (CMOVLGT, SETNE).
@@ -121,14 +129,32 @@ func splitSize(upper string) (base string, size int) {
// its own direction-dependent opcodes; KTESTW is always VEX; everything else
// takes EVEX when an operand demands it (a ZMM or K register, or an
// EVEX-only mnemonic) and VEX otherwise.
func (e *enc) encodeVec(upper string, ops []Operand) error {
if upper == "KMOVW" {
return e.encodeKmovw(ops)
func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
if gs, ok := gatherTable[upper]; ok {
return e.encodeGather(upper, gs, ops, sfx)
}
if upper == "KTESTW" || !evexRequired(upper, ops) {
if ss, ok := scatterTable[upper]; ok {
return e.encodeScatter(upper, ss, ops, sfx)
}
if upper == "KMOVW" || upper == "KMOVQ" {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", upper)
}
return e.encodeKmov(upper, ops)
}
if isKOp(upper) {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", upper)
}
return e.encodeKOp(upper, ops)
}
if upper == "KTESTW" || (!evexRequired(upper, ops) && !sfx.evexOnly()) {
if sfx.any() {
return fmt.Errorf("%s: the .Z suffix requires an EVEX instruction", upper)
}
return e.encodeVex(upper, ops)
}
return e.encodeEvex(upper, ops)
return e.encodeEvex(upper, ops, sfx)
}
// --- instruction components -------------------------------------------------
+1066 -54
View File
File diff suppressed because it is too large Load Diff
+516 -1
View File
@@ -65,6 +65,23 @@ func TestEvexGroundTruth(t *testing.T) {
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
// The wider AVX-512 F/BW integer set.
{"VPADDB Z1,Z2,Z3", "VPADDB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48fcd9"},
{"VPSUBW Z1,Z2,Z3", "VPSUBW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48f9d9"},
{"VPANDQ Z1,Z2,Z3", "VPANDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed48dbd9"},
{"VPANDND Z1,Z2,Z3", "VPANDND", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48dfd9"},
{"VPMULLW Z1,Z2,Z3", "VPMULLW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48d5d9"},
{"VPMINUB Z1,Z2,Z3", "VPMINUB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48dad9"},
{"VPMAXUQ Z1,Z2,Z3", "VPMAXUQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed483fd9"},
{"VPAVGW Z1,Z2,Z3", "VPAVGW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48e3d9"},
{"VPSLLVQ Z3,Z1,Z2", "VPSLLVQ", []Operand{vreg(t, "Z3"), vreg(t, "Z1"), vreg(t, "Z2")}, "62f2f54847d3"},
{"VPSRAVQ Z3,Z1,Z2", "VPSRAVQ", []Operand{vreg(t, "Z3"), vreg(t, "Z1"), vreg(t, "Z2")}, "62f2f54846d3"},
{"VPSHUFD $0x1B,Z1,Z2", "VPSHUFD", []Operand{Imm(0x1B), vreg(t, "Z1"), vreg(t, "Z2")}, "62f17d4870d11b"},
{"VPSHUFB Z1,Z2,Z3", "VPSHUFB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4800d9"},
{"VMOVDQU8 Z1,Z2", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17f487fca"},
{"VMOVDQU16 Z1,Z2", "VMOVDQU16", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff487fca"},
// Indices 16–31: rm[4] rides in X̄ for register operands.
{"VPSHUFD $1,X16,X17", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}, "62a17d0870c801"},
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
// Conversions and narrowing stores (reg = wide source).
@@ -84,6 +101,32 @@ func TestEvexGroundTruth(t *testing.T) {
{"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"},
// Register indices 16–31 exist only in EVEX encodings.
{"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"},
// Packed double arithmetic / unpack (EVEX forms carry W=1).
{"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"},
{"VDIVPD Z4,Z5,Z6", "VDIVPD", []Operand{vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "Z6")}, "62f1d5485ef4"},
{"VMINPD Z7,Z8,Z9", "VMINPD", []Operand{vreg(t, "Z7"), vreg(t, "Z8"), vreg(t, "Z9")}, "6271bd485dcf"},
{"VMAXPD Z10,Z11,Z12", "VMAXPD", []Operand{vreg(t, "Z10"), vreg(t, "Z11"), vreg(t, "Z12")}, "6251a5485fe2"},
{"VUNPCKLPD Z1,Z2,Z3", "VUNPCKLPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4814d9"},
{"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"},
{"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"},
{"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"},
// VMOVDDUP — duplicate the low double; disp8×N = 64 at 512 bits, and
// X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄).
{"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"},
{"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"},
{"VMOVDDUP X16,X17", "VMOVDDUP", []Operand{vreg(t, "X16"), vreg(t, "X17")}, "62a1ff0812c8"},
// Conversions: DQ→PS, PS→PD (pp = 00, the Go assembler's choice),
// DQ→PD (the destination sets the length).
{"VCVTDQ2PS Z1,Z2", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"},
{"VCVTPS2PD Y1,Z2", "VCVTPS2PD", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17c485ad1"},
{"VCVTPS2PD 32(AX),Z2", "VCVTPS2PD", []Operand{Ptr(AX, 32, 32), vreg(t, "Z2")}, "62f17c485a5001"},
{"VCVTDQ2PD Y1,Z2", "VCVTDQ2PD", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17e48e6d1"},
// PD→DQ conversions: the source is the wide operand and fixes the
// length (ZMM source → L'L = 10 even with an XMM destination; a
// memory source takes the length the mnemonic's spelling implies).
{"VCVTPD2DQ Z1,Y2", "VCVTPD2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff48e6d1"},
{"VCVTPD2DQ 64(AX),Y2", "VCVTPD2DQ", []Operand{Ptr(AX, 64, 64), vreg(t, "Y2")}, "62f1ff48e65001"},
{"VCVTTPD2DQ Z3,Y4", "VCVTTPD2DQ", []Operand{vreg(t, "Z3"), vreg(t, "Y4")}, "62f1fd48e6e3"},
}
for _, c := range cases {
want := strings.ReplaceAll(c.want, " ", "")
@@ -110,6 +153,478 @@ func TestEvexGroundTruth(t *testing.T) {
}
}
// TestEvexMasking checks the AVX-512 mask operand (K1–K7, placed freely among
// the operands) and the .Z zeroing suffix, byte for byte against the Go
// assembler.
func TestEvexMasking(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// Masked arithmetic: K anywhere among the operands; .Z sets the z bit.
{"VPADDD.Z merging+zeroing", "VPADDD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f16dcafed9"},
{"VPADDD merging", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49fed9"},
{"VADDPD.Z", "VADDPD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1edca58d9"},
{"VPMINSD.Z", "VPMINSD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K5"), vreg(t, "Z3")}, "62f26dcd39d9"},
{"VPMINSQ.Z", "VPMINSQ.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K5"), vreg(t, "Z3")}, "62f2edcd39d9"},
// Masked immediate shift (K before the destination).
{"VPSRAD.Z", "VPSRAD.Z", []Operand{Imm(1), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f165c972e201"},
{"VPSLLD merge", "VPSLLD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1654a72f104"},
// Masked align.
{"VALIGND", "VALIGND", []Operand{Imm(12), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f36d4b03e10c"},
// Masked conversion and extract.
{"VCVTQQ2PD.Z", "VCVTQQ2PD.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1fecae6d9"},
{"VEXTRACTI64X4", "VEXTRACTI64X4", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f3fd4a3bcb01"},
// Masked moves: K sits between the register and memory operands.
{"VMOVDQU8 store", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "K3"), Ptr(SI, 0, 64)}, "62f17f4b7f0e"},
{"VMOVDQU32 load", "VMOVDQU32", []Operand{Ptr(SI, 0, 64), vreg(t, "K4"), vreg(t, "Z1")}, "62f17e4c6f0e"},
{"VMOVDQU32 store", "VMOVDQU32", []Operand{vreg(t, "Z1"), vreg(t, "K4"), Ptr(DI, 0, 64)}, "62f17e4c7f0f"},
// Masked comparison with a K destination: dst K1, mask K2.
{"VPCMPEQD k-dst+mask", "VPCMPEQD", []Operand{vreg(t, "Z0"), vreg(t, "Z3"), vreg(t, "K2"), vreg(t, "K1")}, "62f1654a76c8"},
// Masked floating point: packed double, the scalar SD/SS forms (which
// exist under EVEX only for masked and zeroing use) and conversions.
{"VSUBPD.Z", "VSUBPD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f1edcb5ce1"},
{"VADDSD merge", "VADDSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K3"), vreg(t, "X4")}, "62f1ef0b58e1"},
{"VSUBSD.Z", "VSUBSD.Z", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K5"), vreg(t, "X3")}, "62f1ef8d5cd9"},
{"VADDSS merge", "VADDSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f16e0958d9"},
{"VCVTPD2DQ merge", "VCVTPD2DQ", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f1ff4ae6d9"},
{"VCVTTPD2DQ.Z", "VCVTTPD2DQ.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f1fdcae6d9"},
{"VCVTDQ2PS.Z", "VCVTDQ2PS.Z", []Operand{vreg(t, "Z1"), vreg(t, "K4"), vreg(t, "Z2")}, "62f17ccc5bd1"},
{"VCVTDQ2PD merge", "VCVTDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "K2"), vreg(t, "X3")}, "62f17e0ae6d9"},
{"VCVTDQ2PD.Z", "VCVTDQ2PD.Z", []Operand{vreg(t, "Y1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f17ecae6d1"},
{"VCVTPS2PD.Z", "VCVTPS2PD.Z", []Operand{vreg(t, "Y1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f17ccb5ad1"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
want := c.mnem
if i := len(want) - 2; i > 0 && want[i:] == ".Z" {
want = want[:i]
}
if inst.Op.String() != want {
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
}
}
// Error cases.
bad := []struct {
name string
mnem string
ops []Operand
}{
{"zeroing without mask", "VPADDD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"K0 mask", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K0"), vreg(t, "Z3")}},
{"two masks", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "K2"), vreg(t, "Z3")}},
{".Z on VEX-only", "VPSHUFD.Z", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}},
{"broadcast unsupported", "VPXORD.BCST", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"rounding unsupported", "VPXORD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"bcst with rounding", "VADDPD.BCST.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"Z not last", "VADDPD.Z.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"duplicate suffix", "VADDPD.Z.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"KMOVW.Z", "KMOVW.Z", []Operand{vreg(t, "K1"), vreg(t, "K2")}},
}
for _, c := range bad {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set — ternary
// logic, lane shuffles/inserts/extracts, compares with a K destination,
// permutes, the wider integer families, expand/compress, broadcasts,
// rotates and word shifts, the opmask instructions, the EVEX suffixes
// (rounding/SAE/broadcast) and the aligned/scalar moves — byte for byte
// against the Go assembler.
func TestEvexExtendedGroundTruth(t *testing.T) {
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// Ternary logic and lane shuffles (NDS + imm8).
{"VPTERNLOGD", "VPTERNLOGD", []Operand{Imm(0xE8), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4825d9e8"},
{"VPTERNLOGQ", "VPTERNLOGQ", []Operand{Imm(0x96), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4825d996"},
{"VSHUFI32X4", "VSHUFI32X4", []Operand{Imm(0x4E), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "62f36d2843d94e"},
{"VSHUFF64X2", "VSHUFF64X2", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4823d901"},
{"VPALIGNR", "VPALIGNR", []Operand{Imm(7), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d480fd907"},
// Permutes.
{"VPERMB", "VPERMB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d488dd9"},
{"VPERMW", "VPERMW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed488dd9"},
{"VPERMI2D", "VPERMI2D", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4876d9"},
{"VPERMT2PD", "VPERMT2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed487fd9"},
// Compare with a K destination (and an immediate predicate).
{"VCMPPD", "VCMPPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3")}, "62f1ed48c2d904"},
{"VCMPPS", "VCMPPS", []Operand{Imm(0), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "K4")}, "62f16c28c2e100"},
{"VCMPSD", "VCMPSD", []Operand{Imm(17), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K5")}, "62f1ef08c2e911"},
// Rounding / SAE / broadcast suffixes.
{"VADDPD.RN_SAE", "VADDPD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed1858d9"},
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
// Packed single arithmetic (same opcodes, no mandatory prefix) —
// ZMM, YMM and XMM widths, rounding and broadcast.
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
{"VMAXPS", "VMAXPS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e85fd9"},
{"VDIVPS.RD_SAE", "VDIVPS.RD_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c385ed9"},
{"VADDPS.BCST", "VADDPS.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f174585810"},
// Compress / expand.
{"VCOMPRESSPD", "VCOMPRESSPD", []Operand{vreg(t, "Z1"), mem64(DI)}, "62f2fd488a0f"},
{"VEXPANDPS", "VEXPANDPS", []Operand{mem64(SI), vreg(t, "Y2")}, "62f27d288816"},
{"VPCOMPRESSD.Z", "VPCOMPRESSD.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), mem64(DI)}, "62f27dca8b0f"},
// Broadcasts.
{"VPBROADCASTB gpr", "VPBROADCASTB", []Operand{BX, vreg(t, "Z1")}, "62f27d487acb"},
{"VPBROADCASTW mem", "VPBROADCASTW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d487910"},
{"VBROADCASTSS", "VBROADCASTSS", []Operand{mem64(AX), vreg(t, "Y3")}, "c4e27d1818"},
{"VBROADCASTSD", "VBROADCASTSD", []Operand{mem64(AX), vreg(t, "Z4")}, "62f2fd481920"},
// Wider integer families.
{"VPMADDWD", "VPMADDWD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48f5d9"},
{"VPMADDUBSW", "VPMADDUBSW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4804d9"},
{"VPMULHUW", "VPMULHUW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48e4d9"},
{"VPSLLVW", "VPSLLVW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4812d9"},
{"VPACKSSWB", "VPACKSSWB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d4863d9"},
{"VPACKUSDW", "VPACKUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482bd9"},
// Absolute values and replicating moves.
{"VPABSD", "VPABSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d481ed1"},
{"VPABSQ mem", "VPABSQ", []Operand{mem64(AX), vreg(t, "Z2")}, "62f2fd481f10"},
{"VMOVSLDUP", "VMOVSLDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa12d1"},
{"VMOVSHDUP", "VMOVSHDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e4816d1"},
// Rotates and word/qword shifts.
{"VPROLD", "VPROLD", []Operand{Imm(5), vreg(t, "Z1"), vreg(t, "Z2")}, "62f16d4872c905"},
{"VPRORQ", "VPRORQ", []Operand{Imm(63), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4872c13f"},
{"VPSLLW", "VPSLLW", []Operand{Imm(9), vreg(t, "X1"), vreg(t, "X2")}, "c5e971f109"},
{"VPSRLQ", "VPSRLQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4873d103"},
// Opmask instructions (VEX-encoded, the width in the L/W/pp bits).
{"KANDW", "KANDW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec41d9"},
{"KORD", "KORD", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d545f4"},
{"KXNORQ", "KXNORQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec46d9"},
{"KNOTB", "KNOTB", []Operand{vreg(t, "K4"), vreg(t, "K5")}, "c5f944ec"},
{"KUNPCKBW", "KUNPCKBW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ed4bd9"},
{"KSHIFTLW", "KSHIFTLW", []Operand{Imm(2), vreg(t, "K1"), vreg(t, "K2")}, "c4e3f932d102"},
{"KADDQ", "KADDQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec4ad9"},
{"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"},
{"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"},
{"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"},
// Lane extract / insert.
{"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"},
{"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"},
{"VINSERTF32X8", "VINSERTF32X8", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d481ad901"},
{"VINSERTI64X4", "VINSERTI64X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed483ad901"},
// Aligned moves and the scalar single move.
{"VMOVAPS", "VMOVAPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4829ca"},
{"VMOVDQA64 mem", "VMOVDQA64", []Operand{mem64(AX), vreg(t, "Z2")}, "62f1fd486f10"},
{"VMOVSS mem", "VMOVSS", []Operand{mem64(AX), vreg(t, "X2")}, "c5fa1010"},
// Conversions and extending/narrowing moves.
{"VCVTPS2DQ", "VCVTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17d485bd1"},
{"VCVTTPS2DQ", "VCVTTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e485bd1"},
{"VPMOVZXBW", "VPMOVZXBW", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d30d1"},
{"VPMOVSXBW mem", "VPMOVSXBW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d482010"},
{"VPMOVWB", "VPMOVWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4830ca"},
{"VPMOVQB", "VPMOVQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4832ca"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
want := c.mnem
if i := strings.IndexByte(want, '.'); i > 0 {
want = want[:i]
}
if inst.Op.String() != want {
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
}
}
}
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef,
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions —
// plus gather/scatter with VSIB addressing, byte for byte against the Go
// assembler.
func TestEvexHelperGroundTruth(t *testing.T) {
vsib := func(base, idx string, scale int) Operand {
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
}
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// Reciprocals and rsqrt (packed RM, scalar NDS).
{"VRCP14PD", "VRCP14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484cd1"},
{"VRCP14PS", "VRCP14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484cd1"},
{"VRCP14SD", "VRCP14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084dd9"},
{"VRCP14SS", "VRCP14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084dd9"},
{"VRSQRT14PD", "VRSQRT14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484ed1"},
{"VRSQRT14PS", "VRSQRT14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484ed1"},
{"VRSQRT14SD", "VRSQRT14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084fd9"},
{"VRSQRT14SS", "VRSQRT14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084fd9"},
// Getexp (packed RM, scalar NDS).
{"VGETEXPPD", "VGETEXPPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd4842d1"},
{"VGETEXPPS", "VGETEXPPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d4842d1"},
{"VGETEXPSD", "VGETEXPSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed0843d9"},
{"VGETEXPSS", "VGETEXPSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d0843d9"},
// Scalef (NDS).
{"VSCALEFPD", "VSCALEFPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed482cd9"},
{"VSCALEFPS", "VSCALEFPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482cd9"},
{"VSCALEFSD", "VSCALEFSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed082dd9"},
{"VSCALEFSS", "VSCALEFSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d082dd9"},
// Rndscale / getmant / reduce (packed $imm,src,dst; scalar NDS+imm).
{"VRNDSCALEPD", "VRNDSCALEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4809d104"},
{"VRNDSCALEPS", "VRNDSCALEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4808d104"},
{"VRNDSCALESD", "VRNDSCALESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed080bd904"},
{"VRNDSCALESS", "VRNDSCALESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d080ad904"},
{"VGETMANTPD", "VGETMANTPD", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4826d103"},
{"VGETMANTPS", "VGETMANTPS", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4826d103"},
{"VGETMANTSD", "VGETMANTSD", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0827d903"},
{"VGETMANTSS", "VGETMANTSS", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0827d903"},
{"VREDUCEPD", "VREDUCEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4856d104"},
{"VREDUCEPS", "VREDUCEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4856d104"},
{"VREDUCESD", "VREDUCESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0857d904"},
{"VREDUCESS", "VREDUCESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0857d904"},
// Fixupimm / range (NDS + imm8).
{"VFIXUPIMMPD", "VFIXUPIMMPD", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4854d902"},
{"VFIXUPIMMPS", "VFIXUPIMMPS", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4854d902"},
{"VFIXUPIMMSD", "VFIXUPIMMSD", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0855d902"},
{"VFIXUPIMMSS", "VFIXUPIMMSS", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0855d902"},
{"VRANGEPD", "VRANGEPD", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4850d901"},
{"VRANGEPS", "VRANGEPS", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4850d901"},
{"VRANGESD", "VRANGESD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0851d901"},
{"VRANGESS", "VRANGESS", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0851d901"},
// FP class test ($imm, src, kdst; packed forms carry the length in
// the X/Y/Z mnemonic suffix the decoder drops).
{"VFPCLASSPDZ", "VFPCLASSPDZ", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "K2")}, "62f3fd4866d104"},
{"VFPCLASSPSY", "VFPCLASSPSY", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "K2")}, "62f37d2866d104"},
{"VFPCLASSSD", "VFPCLASSSD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f3fd0867d104"},
{"VFPCLASSSS", "VFPCLASSSS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f37d0867d104"},
// Gather: VEX spelling (mask register, VSIB, destination) and EVEX
// spelling (VSIB, K mask, destination; L'L follows the VSIB index).
{"VGATHERDPS vex", "VGATHERDPS", []Operand{vreg(t, "X2"), vsib("SI", "X1", 4), vreg(t, "X3")}, "c4e269921c8e"},
{"VPGATHERDD vex", "VPGATHERDD", []Operand{vreg(t, "Y2"), vsib("SI", "Y1", 4), vreg(t, "Y3")}, "c4e26d901c8e"},
{"VGATHERDPS evex", "VGATHERDPS", []Operand{vsib("SI", "X1", 4), vreg(t, "K2"), vreg(t, "X3")}, "62f27d0a921c8e"},
{"VPGATHERQD evex", "VPGATHERQD", []Operand{vsib("SI", "Z1", 8), vreg(t, "K2"), vreg(t, "Y3")}, "62f27d4a911cce"},
// Scatter (EVEX only: source, K mask, VSIB).
{"VSCATTERDPS", "VSCATTERDPS", []Operand{vreg(t, "X3"), vreg(t, "K1"), vsib("SI", "X1", 4)}, "62f27d09a21c8e"},
{"VSCATTERQPD", "VSCATTERQPD", []Operand{vreg(t, "Z3"), vreg(t, "K1"), vsib("SI", "Z1", 8)}, "62f2fd49a31cce"},
// The remaining conversions.
{"VCVTDQ2PS", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"},
{"VCVTQQ2PS", "VCVTQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc485bd1"},
{"VCVTPD2QQ", "VCVTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487bd1"},
{"VCVTPS2QQ", "VCVTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487bd1"},
{"VCVTUDQ2PD", "VCVTUDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "62f17e287ad1"},
{"VCVTPH2PS", "VCVTPH2PS", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f27d4813d1"},
{"VCVTPS2PH", "VCVTPS2PH", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "X2")}, "c4e37d1dca04"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
want := c.mnem
got := inst.Op.String()
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
t.Errorf("%s: decoded as %s", c.name, got)
}
}
}
// TestEvexGprGroundTruth covers the scalar conversions between vector and
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
// for byte against the Go assembler, including memory sources and extended
// GPRs.
func TestEvexGprGroundTruth(t *testing.T) {
mem := func(b Reg) Operand { return Ptr(b, 0, 8) }
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"VCVTSD2SI", "VCVTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2dc1"},
{"VCVTSD2SIQ", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2dc1"},
{"VCVTSS2SI", "VCVTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2dc1"},
{"VCVTSS2SIQ", "VCVTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2dc1"},
{"VCVTTSD2SI", "VCVTTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2cc1"},
{"VCVTTSD2SIQ", "VCVTTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2cc1"},
{"VCVTTSS2SI", "VCVTTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2cc1"},
{"VCVTTSS2SIQ", "VCVTTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2cc1"},
{"VCVTSD2USIL", "VCVTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0879c1"},
{"VCVTSD2USIQ", "VCVTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0879c1"},
{"VCVTSS2USIL", "VCVTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0879c1"},
{"VCVTSS2USIQ", "VCVTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0879c1"},
{"VCVTTSD2USIL", "VCVTTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0878c1"},
{"VCVTTSD2USIQ", "VCVTTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0878c1"},
{"VCVTTSS2USIL", "VCVTTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0878c1"},
{"VCVTTSS2USIQ", "VCVTTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0878c1"},
{"VCVTSI2SDL", "VCVTSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f32ad0"},
{"VCVTSI2SDQ", "VCVTSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32ad0"},
{"VCVTSI2SSL", "VCVTSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f22ad0"},
{"VCVTSI2SSQ", "VCVTSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f22ad0"},
{"VCVTUSI2SDL", "VCVTUSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f177087bd0"},
{"VCVTUSI2SDQ", "VCVTUSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f7087bd0"},
{"VCVTUSI2SSL", "VCVTUSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f176087bd0"},
{"VCVTUSI2SSQ", "VCVTUSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f6087bd0"},
{"VCVTSD2SI mem", "VCVTSD2SI", []Operand{mem(AX), BX}, "c5fb2d18"},
{"VCVTSI2SDQ mem", "VCVTSI2SDQ", []Operand{mem(BX), vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32a13"},
{"VCVTSD2SIQ hi gpr", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), vreg(t, "R9")}, "c461fb2dc9"},
{"VCVTSI2SDQ hi gpr", "VCVTSI2SDQ", []Operand{vreg(t, "R10"), vreg(t, "X1"), vreg(t, "X2")}, "c4c1f32ad2"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
// The decoder does not distinguish the Plan 9 SIQ spelling (the
// 64-bit GPR destination) from the base name; the W bit carries it.
want := c.mnem
got := inst.Op.String()
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
t.Errorf("%s: decoded as %s", c.name, got)
}
}
}
// TestEvexConversionGroundTruth covers the unsigned and truncating VCVT*
// conversions, the remaining sign/zero-extending moves, the signed/unsigned
// narrowing stores and the mask/vector conversions, byte for byte against
// the Go assembler.
func TestEvexConversionGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// Unsigned and truncating conversions.
{"VCVTPD2PS", "VCVTPD2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fd485ad1"},
{"VCVTPD2PSX", "VCVTPD2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f95ad1"},
{"VCVTPD2PSY", "VCVTPD2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5fd5ad1"},
{"VCVTPD2UDQ", "VCVTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4879d1"},
{"VCVTPD2UDQX", "VCVTPD2UDQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc0879d1"},
{"VCVTTPD2UDQ", "VCVTTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4878d1"},
{"VCVTTPD2UDQY", "VCVTTPD2UDQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc2878d1"},
{"VCVTTPD2UQQ", "VCVTTPD2UQQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd4878d1"},
{"VCVTPS2UDQ", "VCVTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4879d1"},
{"VCVTTPS2UDQ", "VCVTTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4878d1"},
{"VCVTPS2UQQ", "VCVTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4879d1"},
{"VCVTTPS2UQQ", "VCVTTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4878d1"},
{"VCVTTPD2QQ", "VCVTTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487ad1"},
{"VCVTTPS2QQ", "VCVTTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487ad1"},
{"VCVTUQQ2PD", "VCVTUQQ2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487ad1"},
{"VCVTUQQ2PS", "VCVTUQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff487ad1"},
{"VCVTUQQ2PSX", "VCVTUQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1ff087ad1"},
{"VCVTQQ2PSX", "VCVTQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc085bd1"},
{"VCVTQQ2PSY", "VCVTQQ2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc285bd1"},
// The remaining sign/zero-extending moves.
{"VPMOVSXBD", "VPMOVSXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d21d1"},
{"VPMOVSXBQ evex", "VPMOVSXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4822d1"},
{"VPMOVSXWQ", "VPMOVSXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d24d1"},
{"VPMOVSXWD", "VPMOVSXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d23d1"},
{"VPMOVZXBD", "VPMOVZXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d31d1"},
{"VPMOVZXBQ evex", "VPMOVZXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4832d1"},
{"VPMOVZXWD", "VPMOVZXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d33d1"},
{"VPMOVZXWQ", "VPMOVZXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d34d1"},
// Signed narrowing stores.
{"VPMOVSDB", "VPMOVSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4821ca"},
{"VPMOVSDW", "VPMOVSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4823ca"},
{"VPMOVSQB", "VPMOVSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4822ca"},
{"VPMOVSQD", "VPMOVSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4825ca"},
{"VPMOVSQW", "VPMOVSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4824ca"},
{"VPMOVSWB", "VPMOVSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4820ca"},
// Unsigned narrowing stores.
{"VPMOVUSDB", "VPMOVUSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4811ca"},
{"VPMOVUSDW", "VPMOVUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4813ca"},
{"VPMOVUSQB", "VPMOVUSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4812ca"},
{"VPMOVUSQD", "VPMOVUSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4815ca"},
{"VPMOVUSQW", "VPMOVUSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4814ca"},
{"VPMOVUSWB", "VPMOVUSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4810ca"},
{"VPMOVDB", "VPMOVDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4831ca"},
{"VPMOVQW", "VPMOVQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4834ca"},
// Mask/vector conversions (the K register is an operand, not a
// mask).
{"VPMOVM2B", "VPMOVM2B", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0828d1"},
{"VPMOVM2W", "VPMOVM2W", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f2fe0828d1"},
{"VPMOVM2D", "VPMOVM2D", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0838d1"},
{"VPMOVM2Q", "VPMOVM2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe4838d1"},
{"VPMOVB2M", "VPMOVB2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f27e0829d1"},
{"VPMOVW2M", "VPMOVW2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f2fe0829d1"},
{"VPMOVD2M", "VPMOVD2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f27e4839d1"},
{"VPMOVQ2M", "VPMOVQ2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f2fe4839d1"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
want := c.mnem
got := inst.Op.String()
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
t.Errorf("%s: decoded as %s", c.name, got)
}
}
}
// TestEvexErrors checks the EVEX-specific error paths.
func TestEvexErrors(t *testing.T) {
cases := []struct {
@@ -125,7 +640,7 @@ func TestEvexErrors(t *testing.T) {
{"VPMOVDW src", "VPMOVDW", []Operand{AX, vreg(t, "Y0")}},
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
// VEX-only mnemonics reject registers only EVEX can encode.
{"VPSHUFD X16", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}},
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
+453
View File
@@ -0,0 +1,453 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"fmt"
"os"
"os/exec"
"path/filepath"
"sync"
)
// This file emits GOOBJ — the Go toolchain's object format, which cmd/link
// consumes directly — so gasm-assembled functions drop into a go build
// without the Go assembler. The layout follows cmd/internal/goobj: a
// toolchain preamble ("go object ...\n!\n"), the go120ld header with its
// block offsets, a string table, symbol definitions, the relocation /
// aux / data index arrays, and the three blocks themselves.
//
// The object carries what the linker requires of an assembly object: the
// functions (non-package symbols, as cmd/asm emits them), the GLOBL data,
// one FuncInfo per function, and the pc-value tables (pcsp, pcfile,
// pcline, pcinline). DWARF and the implicit funcdata symbols are omitted;
// the linker fills their defaults.
// GOOBJ block indices (cmd/internal/goobj).
const (
blkAutolib = iota
blkPkgIdx
blkFile
blkSymdef
blkHashed64def
blkHasheddef
blkNonpkgdef
blkNonpkgref
blkRefFlags
blkHash64
blkHash
blkRelocIdx
blkAuxIdx
blkDataIdx
blkReloc
blkAux
blkData
blkRefName
blkEnd
)
// Symbol kinds used by assembly objects (cmd/internal/objabi).
const (
kindSTEXT = 1
kindSRODATA = 3
kindSDATA = 7
)
// Symbol flags (cmd/internal/goobj).
const (
symFlagDupok = 0x01
symFlagNoSplit = 0x10
symFlag2Link = 0x10 // asm objects flag every named symbol as linkname
symABIStatic = 0xffff
)
// Aux entry types (cmd/internal/goobj).
const (
auxFuncInfo = 1
auxPcsp = 7
auxPcfile = 8
auxPcline = 9
auxPcinline = 10
)
// FuncInfo flags (internal/abi).
const (
funcFlagSPWrite = 2
funcFlagAsm = 4
)
// Relocation types (cmd/internal/objabi).
const relocPCRel = 14
// Special package indices for symbol references.
const (
pkgIdxNone = 0x7fffffff
pkgIdxSelf = 0x7ffffffb
)
const goobjMagic = "\x00go120ld"
// goSym is one symbol definition under construction.
type goSym struct {
name string
abi uint16
typ uint8
flag uint8
flag2 uint8
size uint32
align uint32
}
func (s goSym) append(b []byte, strOff map[string]uint32) []byte {
b = binary.LittleEndian.AppendUint32(b, uint32(len(s.name)))
b = binary.LittleEndian.AppendUint32(b, strOff[s.name])
b = binary.LittleEndian.AppendUint16(b, s.abi)
b = append(b, s.typ, s.flag, s.flag2)
b = binary.LittleEndian.AppendUint32(b, s.size)
return binary.LittleEndian.AppendUint32(b, s.align)
}
// GOObject returns the image as a GOOBJ object file for the given package
// path (the linker qualifies the exported symbols with it, the way cmd/asm
// does with its -p flag). srcPath names the source file recorded in the
// object's file table and line tables. The toolchain's object preamble is
// captured from the installed go tool asm, so the output links with the
// toolchain it was produced on — exactly like a real assembly object.
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
if pkgPath == "" {
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
}
pre, err := toolchainObjectPreamble()
if err != nil {
return nil, err
}
// The symbol tables. Package definitions: the GLOBL symbols, then one
// anonymous FuncInfo symbol per function. Non-package definitions: the
// pc-value tables and the functions themselves, as cmd/asm lays them
// out. defIdx maps a GLOBL's bare name to its definition index for the
// relocations; fnNpIdx maps a function to its non-package index.
var defs []goSym
var defData [][]byte
defIdx := map[string]int{}
for _, d := range img.DataSyms {
name := d.Name
if !d.Static {
name = pkgPath + "." + name
}
typ := uint8(kindSDATA)
if d.Rodata {
typ = kindSRODATA
}
flag := uint8(0)
if d.Dupok {
flag = symFlagDupok
}
abi := uint16(0)
if d.Static {
abi = symABIStatic
}
defIdx[d.Name] = len(defs)
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)})
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
}
fnFiIdx := make([]int, len(img.Funcs))
for i := range img.Funcs {
data := marshalFuncInfo(img.Funcs[i])
fnFiIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))})
defData = append(defData, data)
}
type npSym struct {
sym goSym
data []byte
}
var nps []npSym
type pcRefs struct{ sp, file, line, inl int }
pcIdx := make([]pcRefs, len(img.Funcs))
fnNpIdx := make([]int, len(img.Funcs))
for i, fn := range img.Funcs {
tables := []struct {
data []byte
dst *int
}{
{pcspTable(fn), &pcIdx[i].sp},
{pcValueFlat(0, fn.Size), &pcIdx[i].file},
{pcValueFlat(int32(fn.Line), fn.Size), &pcIdx[i].line},
{pcValueFlat(-1, fn.Size), &pcIdx[i].inl},
}
for _, t := range tables {
*t.dst = len(nps)
nps = append(nps, npSym{
sym: goSym{typ: kindSRODATA, size: uint32(len(t.data)), align: 1},
data: t.data,
})
}
name := fn.Name
abi := uint16(0)
if fn.Static {
abi = symABIStatic
} else {
name = pkgPath + "." + name
}
flag := uint8(0)
if fn.NoSplit {
flag |= symFlagNoSplit
}
fnNpIdx[i] = len(nps)
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
// The linker writes the resolved displacement into the field;
// leave it zero, as cmd/asm's object does.
if r.Off >= 0 && r.Off+4 <= len(code) {
code[r.Off], code[r.Off+1], code[r.Off+2], code[r.Off+3] = 0, 0, 0, 0
}
}
nps = append(nps, npSym{
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
data: code,
})
}
// Relocations, per defined symbol in definition order (package defs,
// then non-package defs). Only file-local GLOBL references resolve;
// external symbols need the import machinery of a later increment.
nsyms := len(defs) + len(nps)
symRelocs := make([][]byte, nsyms) // flat 23-byte records
for i, fn := range img.Funcs {
si := len(defs) + fnNpIdx[i]
for _, r := range fn.Relocs {
if r.External {
return nil, fmt.Errorf("GOOBJ emission: external symbol %q is not supported yet", r.Name)
}
di, ok := defIdx[r.Name]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
}
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = 4 // field width
binary.LittleEndian.PutUint16(rec[5:], relocPCRel)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
symRelocs[si] = append(symRelocs[si], rec[:]...)
}
}
// Aux entries per function: FuncInfo, then the four pc tables.
// References into the non-package table use pkgIdxNone.
symAux := make([][]byte, nsyms)
for i := range img.Funcs {
si := len(defs) + fnNpIdx[i]
aux := func(typ uint8, pkg, idx uint32) {
var rec [9]byte
rec[0] = typ
binary.LittleEndian.PutUint32(rec[1:], pkg)
binary.LittleEndian.PutUint32(rec[5:], idx)
symAux[si] = append(symAux[si], rec[:]...)
}
aux(auxFuncInfo, pkgIdxSelf, uint32(fnFiIdx[i]))
aux(auxPcsp, pkgIdxNone, uint32(len(defs)+pcIdx[i].sp))
aux(auxPcfile, pkgIdxNone, uint32(len(defs)+pcIdx[i].file))
aux(auxPcline, pkgIdxNone, uint32(len(defs)+pcIdx[i].line))
aux(auxPcinline, pkgIdxNone, uint32(len(defs)+pcIdx[i].inl))
}
// The string table. Absolute offsets: it starts right after the
// 96-byte header (magic, fingerprint, flags, the 19 block offsets).
const headerSize = 8 + 8 + 4 + 4*(blkEnd+1)
strTab := []byte{}
strOff := map[string]uint32{}
addStr := func(s string) {
if _, ok := strOff[s]; ok {
return
}
strOff[s] = uint32(headerSize + len(strTab))
strTab = append(strTab, s...)
}
addStr("")
addStr(srcPath)
for _, s := range defs {
addStr(s.name)
}
for _, s := range nps {
addStr(s.sym.name)
}
stringRef := func(b []byte, s string) []byte {
b = binary.LittleEndian.AppendUint32(b, uint32(len(s)))
return binary.LittleEndian.AppendUint32(b, strOff[s])
}
// Serialise the block bodies.
var symdefBlk, npdefBlk []byte
for _, s := range defs {
symdefBlk = s.append(symdefBlk, strOff)
}
for _, s := range nps {
npdefBlk = s.sym.append(npdefBlk, strOff)
}
pkgIdxBlk := stringRef(nil, "") // index 0: the dummy invalid package
fileBlk := stringRef(nil, srcPath)
var relocBlk, auxBlk, dataBlk []byte
relocIdxBlk := make([]byte, 0, 4*(nsyms+1))
auxIdxBlk := make([]byte, 0, 4*(nsyms+1))
dataIdxBlk := make([]byte, 0, 4*(nsyms+1))
var nr, na, nd uint32
for si := 0; si < nsyms; si++ {
relocIdxBlk = binary.LittleEndian.AppendUint32(relocIdxBlk, nr)
auxIdxBlk = binary.LittleEndian.AppendUint32(auxIdxBlk, na)
dataIdxBlk = binary.LittleEndian.AppendUint32(dataIdxBlk, nd)
relocBlk = append(relocBlk, symRelocs[si]...)
auxBlk = append(auxBlk, symAux[si]...)
var d []byte
if si < len(defData) {
d = defData[si]
} else {
d = nps[si-len(defData)].data
}
dataBlk = append(dataBlk, d...)
nr += uint32(len(symRelocs[si])) / 23
na += uint32(len(symAux[si])) / 9
nd += uint32(len(d))
}
relocIdxBlk = binary.LittleEndian.AppendUint32(relocIdxBlk, nr)
auxIdxBlk = binary.LittleEndian.AppendUint32(auxIdxBlk, na)
dataIdxBlk = binary.LittleEndian.AppendUint32(dataIdxBlk, nd)
blocks := [blkEnd][]byte{
blkPkgIdx: pkgIdxBlk,
blkFile: fileBlk,
blkSymdef: symdefBlk,
blkNonpkgdef: npdefBlk,
blkRelocIdx: relocIdxBlk,
blkAuxIdx: auxIdxBlk,
blkDataIdx: dataIdxBlk,
blkReloc: relocBlk,
blkAux: auxBlk,
blkData: dataBlk,
}
// Assemble the payload: header (offsets filled once known), string
// table, blocks in order.
payload := make([]byte, headerSize)
copy(payload, goobjMagic)
// The fingerprint stays zero, as cmd/asm leaves it.
binary.LittleEndian.PutUint32(payload[16:], 4) // ObjFlagFromAssembly
off := uint32(headerSize + len(strTab))
for i := 0; i < blkEnd; i++ {
binary.LittleEndian.PutUint32(payload[20+4*i:], off)
off += uint32(len(blocks[i]))
}
binary.LittleEndian.PutUint32(payload[20+4*blkEnd:], off)
payload = append(payload, strTab...)
for _, blk := range blocks {
payload = append(payload, blk...)
}
out := make([]byte, 0, len(pre)+len(payload))
out = append(out, pre...)
return append(out, payload...), nil
}
// marshalFuncInfo serialises a function's goobj.FuncInfo: sizes, flags,
// start line, the one-element file table and an empty inline tree.
func marshalFuncInfo(fn FuncLayout) []byte {
flag := uint8(funcFlagAsm)
if fn.SPWrite {
flag |= funcFlagSPWrite
}
b := make([]byte, 0, 28)
b = binary.LittleEndian.AppendUint32(b, uint32(fn.Args))
b = binary.LittleEndian.AppendUint32(b, uint32(fn.Frame))
b = append(b, 0, flag, 0, 0) // FuncID normal, flags, padding
b = binary.LittleEndian.AppendUint32(b, uint32(int32(fn.Line)))
b = binary.LittleEndian.AppendUint32(b, 1) // one file
b = binary.LittleEndian.AppendUint32(b, 0) // file index 0
b = binary.LittleEndian.AppendUint32(b, 0) // no inline tree
return b
}
// pcValueFlat encodes a pc-value table holding v over the whole function.
func pcValueFlat(v int32, size int) []byte {
// The table is delta-encoded from an implicit value of -1: a varint
// value delta, an unsigned pc delta to the end, and a zero terminator.
out := binary.AppendVarint(nil, int64(v)+1)
out = binary.AppendUvarint(out, uint64(size))
return append(out, 0)
}
// pcspTable encodes the stack-adjustment table: the SP delta in effect at
// every pc, from the function's prologue and epilogue boundaries.
func pcspTable(fn FuncLayout) []byte {
if len(fn.Spadj) == 0 {
return pcValueFlat(0, fn.Size)
}
pts := make([]SpadjStep, 0, len(fn.Spadj)+1)
pts = append(pts, SpadjStep{PC: 0, Value: 0})
pts = append(pts, fn.Spadj...)
out := binary.AppendVarint(nil, int64(pts[0].Value)+1)
cur, old := pts[0].PC, pts[0].Value
for _, p := range pts[1:] {
out = binary.AppendUvarint(out, uint64(p.PC-cur))
out = binary.AppendVarint(out, int64(p.Value-old))
cur, old = p.PC, p.Value
}
out = binary.AppendUvarint(out, uint64(fn.Size-cur))
return append(out, 0)
}
// toolchainObjectPreamble returns the "go object ...\n!\n" header the
// installed go tool asm writes, captured by assembling a one-instruction
// probe. The linker compares this string verbatim against its own, so it
// must come from the toolchain itself, not be reconstructed.
var (
preambleOnce sync.Once
preamble []byte
preambleErr error
)
func toolchainObjectPreamble() ([]byte, error) {
preambleOnce.Do(func() {
goBin, err := exec.LookPath("go")
if err != nil {
preambleErr = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
return
}
dir, err := os.MkdirTemp("", "gasm-preamble")
if err != nil {
preambleErr = err
return
}
defer os.RemoveAll(dir)
src := filepath.Join(dir, "probe_amd64.s")
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
preambleErr = err
return
}
obj := filepath.Join(dir, "probe.o")
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
cmd.Env = append(os.Environ(), "GOARCH=amd64")
if out, err := cmd.CombinedOutput(); err != nil {
preambleErr = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
return
}
data, err := os.ReadFile(obj)
if err != nil {
preambleErr = err
return
}
i := bytes.Index(data, []byte("\n!\n"))
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
preambleErr = fmt.Errorf("unrecognised assembler object layout")
return
}
preamble = data[:i+3]
})
return preamble, preambleErr
}
+477
View File
@@ -0,0 +1,477 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// goobjView is a minimal parsed view of a GOOBJ payload, enough to check
// the emitter's output block by block.
type goobjView struct {
t *testing.T
b []byte
offs [blkEnd + 1]uint32
strOff uint32
}
func openGoobj(t *testing.T, data []byte) *goobjView {
t.Helper()
i := bytes.Index(data, []byte(goobjMagic))
if i < 0 {
t.Fatal("no GOOBJ magic in output")
}
v := &goobjView{t: t, b: data[i:], strOff: uint32(i + 96)}
for j := 0; j <= blkEnd; j++ {
v.offs[j] = binary.LittleEndian.Uint32(v.b[20+4*j:])
}
return v
}
func (v *goobjView) blk(i int) []byte { return v.b[v.offs[i]:v.offs[i+1]] }
func (v *goobjView) str(off, ln uint32) string {
return string(v.b[off : off+ln])
}
type goobjSymView struct {
name string
abi uint16
typ uint8
flag uint8
flag2 uint8
size uint32
align uint32
}
func (v *goobjView) syms(i int) []goobjSymView {
var out []goobjSymView
for x := v.blk(i); len(x) >= 21; x = x[21:] {
le := binary.LittleEndian
out = append(out, goobjSymView{
name: v.str(le.Uint32(x[4:]), le.Uint32(x[0:])),
abi: le.Uint16(x[8:]),
typ: x[10],
flag: x[11],
flag2: x[12],
size: le.Uint32(x[13:]),
align: le.Uint32(x[17:]),
})
}
return out
}
// TestGOObjectStructure checks the emitted object's blocks against the
// ground truth captured from go tool asm: the symbol tables, the FuncInfo
// contents, the pc-value tables, the relocation and the aux wiring.
func TestGOObjectStructure(t *testing.T) {
f, errs := parser.Parse("t_amd64.s", `
#include "textflag.h"
TEXT ·addq(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), CX
ADDQ CX, AX
MOVQ AX, ret+16(FP)
RET
TEXT ·loadmask(SB), NOSPLIT, $0-8
VMOVDQU mask<>(SB), X0
VPMOVMSKB X0, AX
MOVQ AX, ret+0(FP)
RET
GLOBL mask<>(SB), RODATA, $16
DATA mask<>+0(SB)/8, $0x0807060504030201
DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
obj, err := img.GOObject("testpkg", "t_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
v := openGoobj(t, obj)
if flags := binary.LittleEndian.Uint32(v.b[16:]); flags != 4 {
t.Errorf("flags = %#x, want ObjFlagFromAssembly (4)", flags)
}
// Package defs: the static GLOBL, then one anonymous FuncInfo per
// function.
defs := v.syms(blkSymdef)
if len(defs) != 3 {
t.Fatalf("symdefs = %d, want 3", len(defs))
}
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
t.Errorf("mask symbol = %+v", defs[0])
}
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
t.Errorf("funcinfo symbol = %+v", defs[1])
}
// Non-package defs: four pc tables and the function, per function.
nps := v.syms(blkNonpkgdef)
if len(nps) != 10 {
t.Fatalf("nonpkgdefs = %d, want 10", len(nps))
}
fn := nps[4]
if fn.name != "testpkg.addq" || fn.typ != kindSTEXT || fn.flag != symFlagNoSplit || fn.size != 19 {
t.Errorf("addq symbol = %+v", fn)
}
for i, s := range []int{0, 1, 2, 3, 5, 6, 7, 8} {
if nps[s].typ != kindSRODATA || nps[s].align != 1 || nps[s].name != "" {
t.Errorf("pc table %d = %+v", i, nps[s])
}
}
// FuncInfo: args 24, FuncFlag Asm, one file, no inline tree.
le := binary.LittleEndian
data := v.blk(blkData)
fi := data[16:44]
if le.Uint32(fi[0:]) != 24 || le.Uint32(fi[4:]) != 0 || fi[8] != 0 || fi[9] != funcFlagAsm ||
le.Uint32(fi[16:]) != 1 || le.Uint32(fi[20:]) != 0 || le.Uint32(fi[24:]) != 0 {
t.Errorf("funcinfo bytes %x", fi)
}
// pcsp: a flat zero over the whole function (zero-frame NOSPLIT).
if got := data[72:75]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
t.Errorf("pcsp = %x, want 021300", got)
}
// pcinline: a flat -1.
if got := data[81:84]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) {
t.Errorf("pcinline = %x, want 001300", got)
}
// The one relocation: R_PCREL, four bytes wide, against the GLOBL,
// with the field in the function code left zero. The loadmask code's
// offset comes from the data index (symbol 3 defs + 9 non-package).
relocs := v.blk(blkReloc)
if len(relocs) != 23 {
t.Fatalf("relocs = %d bytes, want one 23-byte entry", len(relocs))
}
off := int32(le.Uint32(relocs[0:]))
if off != 4 || relocs[4] != 4 || le.Uint16(relocs[5:]) != relocPCRel ||
le.Uint64(relocs[7:]) != 0 || le.Uint32(relocs[15:]) != pkgIdxSelf || le.Uint32(relocs[19:]) != 0 {
t.Errorf("reloc = %x", relocs)
}
didx := v.blk(blkDataIdx)
lm := le.Uint32(didx[4*(3+9):])
code := data[lm : lm+18]
if !bytes.Equal(code[4:8], []byte{0, 0, 0, 0}) {
t.Errorf("relocated field = %x, want zeroed", code[4:8])
}
// Aux wiring: FuncInfo (package symbol), then the four pc tables
// (non-package symbols).
auxs := v.blk(blkAux)
if len(auxs) != 2*5*9 {
t.Fatalf("aux = %d bytes, want 10 entries", len(auxs))
}
wantAux := []struct {
typ uint8
pkg uint32
idx uint32
}{
{auxFuncInfo, pkgIdxSelf, 1},
{auxPcsp, pkgIdxNone, uint32(len(defs) + 0)},
{auxPcfile, pkgIdxNone, uint32(len(defs) + 1)},
{auxPcline, pkgIdxNone, uint32(len(defs) + 2)},
{auxPcinline, pkgIdxNone, uint32(len(defs) + 3)},
{auxFuncInfo, pkgIdxSelf, 2},
{auxPcsp, pkgIdxNone, uint32(len(defs) + 5)},
{auxPcfile, pkgIdxNone, uint32(len(defs) + 6)},
{auxPcline, pkgIdxNone, uint32(len(defs) + 7)},
{auxPcinline, pkgIdxNone, uint32(len(defs) + 8)},
}
for i, w := range wantAux {
e := auxs[i*9:]
if e[0] != w.typ || le.Uint32(e[1:]) != w.pkg || le.Uint32(e[5:]) != w.idx {
t.Errorf("aux[%d] = {%d,%d,%d}, want {%d,%d,%d}", i, e[0], le.Uint32(e[1:]), le.Uint32(e[5:]), w.typ, w.pkg, w.idx)
}
}
}
// decodePCValues decodes a pc-value table into (pc, value) steps. The
// table ends with a final unsigned pc delta covering the rest of the
// function, followed by a zero byte that carries no value delta.
func decodePCValues(b []byte) (pcs, vals []int64) {
val, n := binary.Varint(b)
b = b[n:]
val-- // the first delta is against the implicit -1
var pc int64
pcs = append(pcs, pc)
vals = append(vals, val)
for {
pcd, n := binary.Uvarint(b)
b = b[n:]
if pcd == 0 { // zero pc delta terminates the table
break
}
pc += int64(pcd)
if len(b) == 1 && b[0] == 0 { // final coverage, no value change
break
}
vd, n := binary.Varint(b)
b = b[n:]
val += vd
pcs = append(pcs, pc)
vals = append(vals, val)
}
return pcs, vals
}
// TestGOObjectPcspFrame checks the pcsp table of a frame-pointer function:
// the prologue raises the stack delta to 8+frame, the RET's epilogue
// restores it to zero.
func TestGOObjectPcspFrame(t *testing.T) {
f, errs := parser.Parse("frame_amd64.s", `
#include "textflag.h"
TEXT ·framed(SB), NOSPLIT, $8-0
MOVQ BP, AX
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
fn := img.Funcs[0]
pcs, vals := decodePCValues(pcspTable(fn))
// Prologue: PUSHQ BP (1 byte, +8), MOVQ SP, BP (3 bytes, no change),
// SUBQ $8, SP (4 bytes, +16 in total); the RET's epilogue unwinds
// ADDQ $8, SP (+8) then POPQ BP (0).
wantPCs := []int64{0, 1, 8}
wantVals := []int64{0, 8, 16}
if len(pcs) < len(wantPCs) {
t.Fatalf("pcsp pcs = %v vals = %v", pcs, vals)
}
for i := range wantPCs {
if pcs[i] != wantPCs[i] || vals[i] != wantVals[i] {
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d) — all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
}
}
// The last two steps unwind the epilogue to zero.
n := len(pcs)
if vals[n-1] != 0 || vals[n-2] != 8 {
t.Errorf("epilogue steps = %v %v, want …8, 0", pcs, vals)
}
// The table covers the whole function.
if last := pcs[n-1]; last >= int64(fn.Size) {
t.Errorf("last pc %d beyond function size %d", last, fn.Size)
}
}
// TestGOObjectExternalRejected checks that a reference to a symbol no GLOBL
// defines is reported: GOOBJ emission resolves only file-local symbols so
// far.
func TestGOObjectExternalRejected(t *testing.T) {
f, errs := parser.Parse("ext_amd64.s", `
#include "textflag.h"
TEXT ·useext(SB), NOSPLIT, $0-8
MOVQ elsewhere(SB), AX
MOVQ AX, ret+0(FP)
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
if _, err := img.GOObject("p", "ext_amd64.s"); err == nil || !strings.Contains(err.Error(), "external") {
t.Errorf("error = %v, want an external-symbol error", err)
}
}
// TestGOObjectLinkAndRun is the end-to-end check: assemble the test
// functions to a GOOBJ, swap it into a go build in place of the toolchain's
// assembly object, link, and run — the output must match the baseline
// binary the Go assembler produced. Skipped when no Go toolchain is
// available.
func TestGOObjectLinkAndRun(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
const asmSrc = `
#include "textflag.h"
TEXT ·addq(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), CX
ADDQ CX, AX
MOVQ AX, ret+16(FP)
RET
TEXT ·loadmask(SB), NOSPLIT, $0-8
VMOVDQU mask<>(SB), X0
VPMOVMSKB X0, AX
MOVQ AX, ret+0(FP)
RET
GLOBL mask<>(SB), RODATA, $16
DATA mask<>+0(SB)/8, $0x0807060504030201
DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
`
const mainSrc = `package main
func addq(a, b int64) int64
func loadmask() int64
func main() {
println(addq(41, 1))
println(loadmask())
}
`
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module goobjtest\n\ngo 1.26\n"), 0o644); err != nil {
t.Fatal(err)
}
// Baseline build with the toolchain's assembler; keep the work
// directory and the commands the build used.
cmd := exec.Command(goBin, "build", "-x", "-work", "-o", "app", ".")
cmd.Dir = dir
buildLog, err := cmd.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work string
var asmObj, pkgArch, linkLine string
for _, line := range strings.Split(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "-o ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || pkgArch == "" || linkLine == "" {
t.Fatalf("could not locate the build steps:\n%s", buildLog)
}
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
// The baseline's answer.
baseOut, err := exec.Command(filepath.Join(dir, "app")).CombinedOutput()
if err != nil {
t.Fatalf("run baseline: %v\n%s", err, baseOut)
}
// Assemble the same source with gasm and swap the object in.
pf, perrs := parser.Parse(filepath.Join(dir, "main_amd64.s"), asmSrc)
if len(perrs) > 0 {
t.Fatalf("parse: %v", perrs)
}
img, err := AssembleFile(pf)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
obj, err := img.GOObject("main", filepath.Join(dir, "main_amd64.s"))
if err != nil {
t.Fatalf("GOObject: %v", err)
}
if err := os.WriteFile(asmObj, obj, 0o644); err != nil {
t.Fatal(err)
}
// Rebuild the package archive with our object in place of the
// toolchain's (go tool pack has no replace-in-place that dedupes, so
// extract, substitute and repack).
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract.Dir = membersDir
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for _, m := range strings.Fields(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, filepath.Join(membersDir, m))
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
// Link with our archive. The link line carries a GOROOT assignment
// and $WORK placeholders; run it through the shell with the
// GOEXPERIMENT the toolchain expects (the linker compares the object
// header against its own, experiments included).
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
link := exec.Command("sh", "-c", linkLine)
link.Dir = dir
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)))
if out, err := link.CombinedOutput(); err != nil {
t.Fatalf("link with gasm object: %v\n%s", err, out)
}
got, err := exec.Command(filepath.Join(dir, "app2")).CombinedOutput()
if err != nil {
t.Fatalf("run gasm-linked binary: %v\n%s", err, got)
}
if !bytes.Equal(got, baseOut) {
t.Errorf("gasm-linked output %q, want baseline %q", got, baseOut)
}
}
// fieldAfter returns the whitespace-delimited field following the first
// occurrence of flag in line.
func fieldAfter(line, flag string) string {
fields := strings.Fields(line)
for i, f := range fields {
if f == flag && i+1 < len(fields) {
return fields[i+1]
}
}
return ""
}
+179 -43
View File
@@ -5,27 +5,73 @@ package asm
import (
"fmt"
"sort"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// Image is an assembled file: the function bodies laid out in source order,
// followed by the file's static data section (GLOBL/DATA). Static-symbol
// references are encoded RIP-relative and resolved within the image, so the
// bytes are self-consistent and executable at any base address.
// followed by the file's static data section (GLOBL/DATA). References to
// file-local static symbols are encoded RIP-relative and resolved within the
// image, so the raw bytes are self-consistent and executable at any base
// address; references to external symbols are recorded as relocations
// (Funcs[i].Relocs, Externals) and left unresolved — the object-file
// emitters turn them into linker relocations.
type Image struct {
Code []byte // concatenated function bodies
Data []byte // static data section
Funcs []FuncLayout // function positions, in source order
Symbols map[string]int // static symbol → byte offset within the image
Code []byte // concatenated function bodies
Data []byte // static data section
Funcs []FuncLayout // function positions, in source order
Symbols map[string]int // static symbol → byte offset within the image
DataSyms []DataSymbol // GLOBL symbols, in layout order
Externals []string // referenced but undefined symbols, sorted
}
// FuncLayout describes one assembled function within an Image.
type FuncLayout struct {
Name string
Pkg string // explicit package prefix ("" = the current package)
Static bool // the <> marker: file-local, not exported
Offset int // start offset within the image (== offset within Code)
Size int
Args int // declared argument/result area (the TEXT size suffix)
Frame int // local frame size (the TEXT $framesize)
NoSplit bool // the NOSPLIT flag
SPWrite bool // the SPWRITE flag: writes an arbitrary value to SP
Line int // source line of the TEXT directive
Labels map[string]int // local labels, function-relative
Relocs []Reloc // static-symbol references, in emission order
Spadj []SpadjStep // stack-adjustment boundaries, ascending by PC
}
// SpadjStep is one stack-adjustment boundary: Value is the SP delta from the
// entry state in effect from PC (function-relative) until the next step.
type SpadjStep struct {
PC int
Value int
}
// Reloc is one static-symbol reference within a function body: the disp32
// field at Off (function-relative) must reach the symbol plus Addend,
// measured from After, the address just past the instruction. An External
// relocation names a symbol no GLOBL in the file defines; the object-file
// emitters carry it into the output's relocation table.
type Reloc struct {
Off int
After int
Name string
Addend int64
External bool
}
// DataSymbol describes one GLOBL symbol laid out in the data section.
type DataSymbol struct {
Name string
Offset int // start offset within the image (== offset within Code)
Pkg string // explicit package prefix ("" = the current package)
Offset int // byte offset within Data
Size int
Labels map[string]int // local labels, function-relative
Static bool // the <> marker: file-local, not exported
Rodata bool // the RODATA flag: read-only data
Dupok bool // the DUPOK flag: duplicate-OK
}
// Bytes returns the whole image: code, then data.
@@ -37,19 +83,21 @@ func (img *Image) Bytes() []byte {
// AssembleFile assembles every TEXT function of a parsed file and lays out
// its static symbols (GLOBL/DATA) in a data section behind the code. Each
// static-symbol reference becomes a RIP-relative load whose displacement is
// resolved against that layout. External (non-file-local) symbol references
// are rejected: they need object-file emission.
// reference to a file-local static symbol becomes a RIP-relative load whose
// displacement is resolved against that layout; a reference to a symbol no
// GLOBL defines is recorded as an external relocation (Externals) with its
// displacement left zero — the object-file emitters resolve it at link
// time, while the raw image (Bytes) cannot represent it.
func AssembleFile(f *ast.File) (*Image, error) {
syms, order, err := collectData(f)
dataSyms, err := collectData(f)
if err != nil {
return nil, err
}
known := make(map[string]bool, len(syms))
for name := range syms {
known[name] = true
known := make(map[string]bool, len(dataSyms))
for _, d := range dataSyms {
known[d.name] = true
}
link := &linkInfo{symbols: known}
link := &linkInfo{symbols: known, allowExternal: true}
img := &Image{Symbols: map[string]int{}}
type asmFunc struct {
@@ -62,50 +110,100 @@ func AssembleFile(f *ast.File) (*Image, error) {
if !ok {
continue
}
code, patches, labels, err := assemble(t, link)
code, patches, labels, steps, err := assemble(t, link)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
img.Funcs = append(img.Funcs, FuncLayout{
fl := FuncLayout{
Name: t.Name.Name,
Pkg: t.Name.Pkg,
Static: t.Name.Static,
Offset: len(img.Code),
Size: len(code),
Frame: frameSize(t),
Args: argsSize(t),
Line: t.Pos().Line,
Labels: labels,
})
}
for _, f := range t.Flags {
switch f {
case "NOSPLIT":
fl.NoSplit = true
case "SPWRITE":
fl.SPWrite = true
}
}
for _, s := range steps {
fl.Spadj = append(fl.Spadj, SpadjStep{PC: s.pc, Value: s.value})
}
img.Funcs = append(img.Funcs, fl)
img.Code = append(img.Code, code...)
funcs = append(funcs, asmFunc{name: t.Name.Name, patches: patches})
}
// Lay out the data section behind the code, each symbol 16-aligned.
dataStart := len(img.Code)
for _, name := range order {
for _, d := range dataSyms {
if pos := dataStart + len(img.Data); pos != align16(pos) {
img.Data = append(img.Data, make([]byte, align16(pos)-pos)...)
}
img.Symbols[name] = dataStart + len(img.Data)
img.Data = append(img.Data, syms[name]...)
img.Symbols[d.name] = dataStart + len(img.Data)
img.DataSyms = append(img.DataSyms, DataSymbol{
Name: d.name,
Pkg: d.pkg,
Offset: len(img.Data),
Size: len(d.buf),
Static: d.static,
Rodata: d.rodata,
Dupok: d.dupok,
})
img.Data = append(img.Data, d.buf...)
}
// Resolve the RIP-relative displacements now that every address is known.
// Resolve the RIP-relative displacements of file-local references now
// that every address is known, and record every reference (resolved or
// external) for the object-file emitters.
externals := map[string]bool{}
for i, fn := range funcs {
base := img.Funcs[i].Offset
code := img.Code[base : base+img.Funcs[i].Size]
for _, p := range fn.patches {
rel := int64(img.Symbols[p.name]) + p.addend - int64(base+p.after)
if rel < -1<<31 || rel >= 1<<31 {
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend}
if imgOff, ok := img.Symbols[p.name]; ok {
rel := int64(imgOff) + p.addend - int64(base+p.after)
if rel < -1<<31 || rel >= 1<<31 {
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
}
copy(code[p.off:p.off+4], le32(rel))
} else {
reloc.External = true
externals[p.name] = true
}
copy(code[p.off:p.off+4], le32(rel))
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
}
}
for name := range externals {
img.Externals = append(img.Externals, name)
}
sort.Strings(img.Externals)
return img, nil
}
// dataSym is one GLOBL symbol and its DATA initialiser.
type dataSym struct {
name string
pkg string
buf []byte
static bool
rodata bool
dupok bool
}
// collectData gathers the file's static symbols (GLOBL) and their initial
// contents (DATA) into byte buffers, in declaration order.
func collectData(f *ast.File) (map[string][]byte, []string, error) {
syms := map[string][]byte{}
var order []string
func collectData(f *ast.File) ([]dataSym, error) {
index := map[string]int{}
var syms []dataSym
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Globl:
@@ -113,50 +211,88 @@ func collectData(f *ast.File) (map[string][]byte, []string, error) {
continue
}
name := dd.Name.Name
if _, dup := syms[name]; dup {
return nil, nil, fmt.Errorf("duplicate GLOBL %q", name)
if _, dup := index[name]; dup {
return nil, fmt.Errorf("duplicate GLOBL %q", name)
}
size := 0
if dd.Size != nil && dd.Size.Imm.HasVal {
size = int(dd.Size.Imm.Val)
}
syms[name] = make([]byte, size)
order = append(order, name)
index[name] = len(syms)
ds := dataSym{
name: name,
pkg: dd.Name.Pkg,
buf: make([]byte, size),
static: dd.Name.Static,
}
for _, f := range dd.Flags {
switch f {
case "RODATA":
ds.rodata = true
case "DUPOK":
ds.dupok = true
case "1":
ds.dupok = true
case "8":
ds.rodata = true
case "9":
ds.dupok = true
ds.rodata = true
}
}
syms = append(syms, ds)
case *ast.Data:
if dd.Name == nil || dd.Name.Pseudo != "SB" {
continue
}
buf, ok := syms[dd.Name.Name]
i, ok := index[dd.Name.Name]
if !ok {
return nil, nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
}
if dd.Value == nil || !dd.Value.Imm.HasVal {
return nil, nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name)
return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name)
}
w := dd.Width
switch w {
case 1, 2, 4, 8:
default:
return nil, nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
}
off := dd.Name.Offset
buf := syms[i].buf
if off < 0 || off+int64(w) > int64(len(buf)) {
return nil, nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
}
v := dd.Value.Imm.Val
if dd.Value.Imm.Neg {
v = -v
}
for i := 0; i < w; i++ {
buf[off+int64(i)] = byte(v >> (8 * i))
for j := 0; j < w; j++ {
buf[off+int64(j)] = byte(v >> (8 * j))
}
}
}
return syms, order, nil
return syms, nil
}
// align16 rounds n up to the next multiple of 16.
func align16(n int) int {
return (n + 15) &^ 15
}
// frameSize returns the local frame size declared on the TEXT directive.
func frameSize(t *ast.Text) int {
if t.Frame != nil && t.Frame.Imm.HasVal {
return int(t.Frame.Imm.Val)
}
return 0
}
// argsSize returns the argument/result area declared on the TEXT directive.
func argsSize(t *ast.Text) int {
if t.Args != nil && t.Args.Imm.HasVal {
return int(t.Args.Imm.Val)
}
return 0
}
+258
View File
@@ -0,0 +1,258 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// This file emits Mach-O x86-64 objects (MH_OBJECT) from an assembled
// Image, in the shape the Darwin assembler produces: one unnamed segment
// carrying a __TEXT,__text and a __DATA,__data section laid out back to
// back at addresses zero and len(code), a symbol table (locals first, then
// exported definitions, then undefined externals) and one relocation entry
// per static-symbol reference, of type X86_64_RELOC_SIGNED.
//
// The image's own address space carries straight over — the data section
// starts immediately after the code, and the layout padding already lives
// inside Image.Data — so every symbol keeps its image address as its
// n_value, and a local (non-external) relocation leaves the displacement
// the assembler resolved in place: the linker only adjusts it by the
// section's final movement.
// Mach-O constants.
const (
machoMagic64 = 0xfeedfacf
machoCPUamd64 = 0x01000007 // CPU_TYPE_X86_64
machoCPUSubAll = 3 // CPU_SUBTYPE_X86_64_ALL
machoObj = 1 // MH_OBJECT
machoSegment64 = 0x19 // LC_SEGMENT_64
machoSymtab = 0x2 // LC_SYMTAB
machoSectTextFlags = 0x80000400 // S_ATTR_PURE_INSTRUCTIONS | S_ATTR_SOME_INSTRUCTIONS
nUndf = 0x00 // undefined symbol
nSect = 0x0e // defined in section number n_sect
nExt = 0x01 // external (exported or undefined-global) bit
x8664RelocSigned = 1
)
// MachOObject returns the image as a Mach-O x86-64 relocatable object
// (MH_OBJECT), the shape the Darwin toolchain links. Symbol names follow
// the same rules as the ELF output. Every static-symbol reference becomes
// an X86_64_RELOC_SIGNED relocation: external references against their
// undefined symbol, file-local ones against the __DATA section with the
// resolved displacement carried in the instruction bytes.
func (img *Image) MachOObject() ([]byte, error) {
le := binary.LittleEndian
// Section ordinals (1-based, as Mach-O numbers them).
const (
sectText = 1
sectData = 2
)
// Object address space: code at 0, data immediately after (the layout
// padding is already part of img.Data, so image addresses are object
// addresses).
textAddr := uint64(0)
dataAddr := uint64(len(img.Code))
vmsize := dataAddr + uint64(len(img.Data))
// The code, with external displacements primed to addend − 4: the
// linker adds the symbol's address to the field as it stands. Local
// displacements stay as the assembler resolved them.
code := append([]byte(nil), img.Code...)
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.External {
// Prime the field to the addend measured from the patch
// site: the assembler records it from the instruction end,
// After − Off bytes past the field.
copy(code[fn.Offset+r.Off:], le32(r.Addend-int64(r.After-r.Off)))
}
}
}
// Symbols: locals first, then exported definitions, then undefined
// externals — the order the classic link editor expects.
type machoSym struct {
name string
typ byte
sect byte
value uint64
}
var locals, globals, undefs []machoSym
for _, fn := range img.Funcs {
s := machoSym{name: objectName(fn.Pkg, fn.Name), typ: nSect, sect: sectText, value: textAddr + uint64(fn.Offset)}
if fn.Static {
locals = append(locals, s)
} else {
s.typ |= nExt
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := machoSym{name: objectName(d.Pkg, d.Name), typ: nSect, sect: sectData, value: dataAddr + uint64(d.Offset)}
if d.Static {
locals = append(locals, s)
} else {
s.typ |= nExt
globals = append(globals, s)
}
}
for _, name := range img.Externals {
undefs = append(undefs, machoSym{name: name, typ: nUndf | nExt})
}
syms := append(append(locals, globals...), undefs...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Relocations, attached to the __text section.
type machoReloc struct {
addr uint32
symnum uint32
extern bool
}
var relocs []machoReloc
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
rel := machoReloc{addr: uint32(fn.Offset + r.Off)}
if r.External {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
rel.symnum = uint32(idx)
rel.extern = true
} else {
// Section-relative: r_symbolnum carries the section number
// and the resolved displacement stays in the bytes.
rel.symnum = sectData
}
relocs = append(relocs, rel)
}
}
// The string table opens with the conventional " \0".
strtab := []byte{' ', 0}
strOff := map[string]int{}
for _, s := range syms {
if _, ok := strOff[s.name]; ok {
continue
}
strOff[s.name] = len(strtab)
strtab = append(strtab, s.name...)
strtab = append(strtab, 0)
}
// File layout: header, the two load commands, section data (code,
// data), the relocation table, the symbol table, the string table.
const (
hdrSize = 32
segCmdSize = 72 + 2*80 // segment command with two sections
symCmdSize = 24
)
sizeofcmds := segCmdSize + symCmdSize
dataOff := hdrSize + sizeofcmds
reloff := dataOff + len(code) + len(img.Data)
symoff := reloff + 8*len(relocs)
stroff := symoff + 16*len(syms)
out := make([]byte, stroff+len(strtab))
// mach_header_64.
le.PutUint32(out[0:], machoMagic64)
le.PutUint32(out[4:], machoCPUamd64)
le.PutUint32(out[8:], machoCPUSubAll)
le.PutUint32(out[12:], machoObj)
le.PutUint32(out[16:], 2) // ncmds
le.PutUint32(out[20:], uint32(sizeofcmds))
le.PutUint32(out[24:], 0) // flags
le.PutUint32(out[28:], 0) // reserved
// LC_SEGMENT_64 with the two sections.
p := hdrSize
le.PutUint32(out[p:], machoSegment64)
le.PutUint32(out[p+4:], segCmdSize)
// segname: the empty string, zero-padded to 16 bytes.
le.PutUint64(out[p+8:], 0)
le.PutUint64(out[p+16:], 0)
le.PutUint64(out[p+24:], 0) // vmaddr
le.PutUint64(out[p+32:], vmsize)
le.PutUint64(out[p+40:], uint64(dataOff))
le.PutUint64(out[p+48:], vmsize)
le.PutUint32(out[p+56:], 7) // maxprot rwx
le.PutUint32(out[p+60:], 7) // initprot rwx
le.PutUint32(out[p+64:], 2) // nsects
le.PutUint32(out[p+68:], 0) // flags
// __TEXT,__text
s := p + 72
copy(out[s:], "__text")
copy(out[s+16:], "__TEXT")
le.PutUint64(out[s+32:], textAddr)
le.PutUint64(out[s+40:], uint64(len(code)))
le.PutUint32(out[s+48:], uint32(dataOff))
le.PutUint32(out[s+52:], 4) // align 2^4
le.PutUint32(out[s+56:], uint32(reloff))
le.PutUint32(out[s+60:], uint32(len(relocs)))
le.PutUint32(out[s+64:], machoSectTextFlags)
// __DATA,__data
s += 80
copy(out[s:], "__data")
copy(out[s+16:], "__DATA")
le.PutUint64(out[s+32:], dataAddr)
le.PutUint64(out[s+40:], uint64(len(img.Data)))
le.PutUint32(out[s+48:], uint32(dataOff+len(code)))
le.PutUint32(out[s+52:], 4) // align 2^4
// LC_SYMTAB.
p = hdrSize + segCmdSize
le.PutUint32(out[p:], machoSymtab)
le.PutUint32(out[p+4:], symCmdSize)
le.PutUint32(out[p+8:], uint32(symoff))
le.PutUint32(out[p+12:], uint32(len(syms)))
le.PutUint32(out[p+16:], uint32(stroff))
le.PutUint32(out[p+20:], uint32(len(strtab)))
// Section data.
copy(out[dataOff:], code)
copy(out[dataOff+len(code):], img.Data)
// Relocation entries.
for i, r := range relocs {
e := out[reloff+i*8:]
le.PutUint32(e[0:], r.addr)
bits := r.symnum & 0x00ffffff
bits |= 1 << 24 // r_pcrel
bits |= 2 << 25 // r_length = 4 bytes
if r.extern {
bits |= 1 << 27 // r_extern
}
bits |= x8664RelocSigned << 28
le.PutUint32(e[4:], bits)
}
// nlist_64 entries.
for i, s := range syms {
e := out[symoff+i*16:]
le.PutUint32(e[0:], uint32(strOff[s.name]))
e[4] = s.typ
e[5] = s.sect
le.PutUint16(e[6:], 0) // n_desc
le.PutUint64(e[8:], s.value)
}
// String table.
copy(out[stroff:], strtab)
return out, nil
}
+127
View File
@@ -0,0 +1,127 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/macho"
"encoding/binary"
"testing"
)
// TestMachOObject checks the structure of the emitted MH_OBJECT: the two
// sections and their addresses, the symbol table (types, sections, values)
// and the __text relocation entries, parsed back with debug/macho. No
// Darwin toolchain is available on the test hosts, so the check is
// structural — the ELF output carries the end-to-end link-and-run proof of
// the shared symbol and relocation model.
func TestMachOObject(t *testing.T) {
img := elfTestImage(t)
obj, err := img.MachOObject()
if err != nil {
t.Fatalf("MachOObject: %v", err)
}
f, err := macho.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer f.Close()
if f.Type != macho.TypeObj {
t.Errorf("file type = %v, want MH_OBJECT", f.Type)
}
if f.Cpu != macho.CpuAmd64 {
t.Errorf("cpu = %v, want CpuAmd64", f.Cpu)
}
text := f.Section("__text")
data := f.Section("__data")
if text == nil || data == nil {
t.Fatal("missing __text or __data section")
}
if text.Addr != 0 || text.Size != uint64(len(img.Code)) {
t.Errorf("__text addr/size = %#x/%d, want 0/%d", text.Addr, text.Size, len(img.Code))
}
if data.Addr != uint64(len(img.Code)) {
t.Errorf("__data addr = %#x, want %#x", data.Addr, len(img.Code))
}
// Symbol table: locals, exported definitions, undefined externals.
syms := f.Symtab.Syms
byName := map[string]macho.Symbol{}
for _, s := range syms {
byName[s.Name] = s
}
wantSym := func(name string, typ, sect uint8, value uint64) {
t.Helper()
s, ok := byName[name]
if !ok {
t.Errorf("symbol %q not found", name)
return
}
if s.Type != typ || s.Sect != sect || s.Value != value {
t.Errorf("%s: type/sect/value = %#x/%d/%#x, want %#x/%d/%#x",
name, s.Type, s.Sect, s.Value, typ, sect, value)
}
}
const (
defined = nSect | nExt
local = nSect
undefined = nUndf | nExt
)
wantSym("addq", defined, 1, 0)
wantSym("getanswer", defined, 1, 5)
wantSym("useextern", defined, 1, 13)
answer := byName["answer"]
if answer.Type != local || answer.Sect != 2 {
t.Errorf("answer: type/sect = %#x/%d, want %#x/2", answer.Type, answer.Sect, local)
}
wantSym("extvar", undefined, 0, 0)
// Relocations: both X86_64_RELOC_SIGNED, PC-relative, 4 bytes wide.
// The local one carries its section number in Value, the external one
// its symbol number.
if len(text.Relocs) != 2 {
t.Fatalf("__text relocs = %d, want 2", len(text.Relocs))
}
var sawLocal, sawExternal bool
for _, r := range text.Relocs {
if !r.Pcrel || r.Len != 2 || r.Type != x8664RelocSigned {
t.Errorf("reloc at %#x: pcrel/len/type = %v/%d/%d", r.Addr, r.Pcrel, r.Len, r.Type)
}
switch {
case r.Extern:
if name := syms[r.Value].Name; name != "extvar" {
t.Errorf("external reloc at %#x names %q, want extvar", r.Addr, name)
}
sawExternal = true
default:
if r.Value != 2 { // __data, the second section
t.Errorf("local reloc at %#x: section %d, want 2 (__data)", r.Addr, r.Value)
}
sawLocal = true
}
}
if !sawLocal || !sawExternal {
t.Errorf("relocs seen: local=%v external=%v, want both", sawLocal, sawExternal)
}
// The __text bytes are the image code, with the external displacement
// primed to addend − 4 and the local one left resolved.
textData, err := text.Data()
if err != nil {
t.Fatal(err)
}
want := append([]byte(nil), img.Code...)
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.Name == "extvar" {
binary.LittleEndian.PutUint32(want[fn.Offset+r.Off:], 0xfffffffc) // −4
}
}
}
if !bytes.Equal(textData, want) {
t.Errorf("__text bytes %x, want %x", textData, want)
}
}
+150 -2
View File
@@ -43,6 +43,13 @@ const (
// in ModRM.reg and the destination in r/m — the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD).
vexRMRev
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
// vector length follows the source: the packed-double → dword
// conversions (VCVTPD2DQ/VCVTTPD2DQ and their X/Y spellings) narrow into
// an XMM destination, so the L bit rides with the wider source. The
// mnemonic's spelling fixes the length (X = 128, Y = 256), which also
// covers a memory source. ModRM.reg = dst, ModRM.rm = src, no vvvv.
vexRMSrcLen
// vexZero is the no-operand form (VZEROUPPER).
vexZero
)
@@ -84,14 +91,38 @@ var vexTable = map[string]vexSpec{
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic.
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG — packed single-precision arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3},
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3},
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3},
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
// VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed
// opcodes with an F2 pp).
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
"VMULSD": {1, 0x59, 0, 3, -1, vexNDS3},
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
// VEX.128.F3.0F.WIG — scalar single-precision arithmetic (the packed
// opcodes with an F3 pp).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3},
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
// VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form).
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
@@ -99,12 +130,33 @@ var vexTable = map[string]vexSpec{
// no vvvv).
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM},
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM},
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM},
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM},
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM},
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM},
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
// VEX.128/256.0F.WIG — signed dword to packed single conversion
// (reg=dst, rm=src, no vvvv, no mandatory prefix).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
// VEX.128/256.0F.WIG — packed single to packed double conversion
// (reg=dst, rm=src; the destination is the wide operand and sets the
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
// the Go assembler emits the instruction with pp = 00, and gasm follows
// the Go assembler's bytes — its machine code is the oracle, not the
// manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
// VEX.128.F2.0F.WIG — duplicate the low double of each 128-bit lane
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
// VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src).
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
@@ -132,12 +184,75 @@ var vexTable = map[string]vexSpec{
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
// VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst:
// reg=src, rm=XMM/memory dst, imm8 — the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
// VEX.128.0F.W0 — no operands.
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
// VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
// VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; SD is 256-bit only).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
// VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width
// source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y
// spellings: the destination is always XMM and the spelling fixes the
// source length (X = 128, Y = 256).
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
// VEX scalar conversions between vector and general-purpose registers.
// Vector to GPR (two operands: vec/mem source, GPR destination, vvvv
// unused; the length follows the source).
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM},
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM},
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM},
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM},
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM},
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM},
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM},
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM},
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
// vector source in vvvv, vector destination in reg).
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3},
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3},
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
// VEX.F2.0F — packed double to packed dword conversions, truncating and
// non-truncating. The destination is always XMM; the X/Y spellings fix
// the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen.
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
}
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
// the packed-double → dword conversions) to its fixed vector length:
// X = 128 (L = 0), Y = 256 (L = 1). The spelling fixes the length even for
// a memory source, matching the Go assembler's ytab.
var vexSrcLen = map[string]int{
"VCVTPD2DQX": 0,
"VCVTPD2DQY": 1,
"VCVTTPD2DQX": 0,
"VCVTTPD2DQY": 1,
"VCVTPD2PSX": 0,
"VCVTPD2PSY": 1,
}
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
@@ -182,6 +297,11 @@ var vexMoveTable = map[string]vexMoveSpec{
// VEX.128.F2.0F.WIG — scalar double move, memory operands only (the
// register form takes three operands and is not supported yet).
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128.F3.0F.WIG — scalar single move, memory operands only.
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128/256 — aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
}
// isVex reports whether the mnemonic is a VEX-encoded instruction we handle.
@@ -230,6 +350,8 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3Imm(spec, ops)
case vexExtract:
return e.encodeVexExtract(spec, ops)
case vexRMSrcLen:
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero:
return e.encodeVexZero(mnemUpper, spec, ops)
}
@@ -295,6 +417,32 @@ func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
return e.emitVexFields(spec, l, regField, rBit, 15, src)
}
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the VEX.L bit following the source — fixed
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
// the source is memory.
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("VEX destination must be a vector register")
}
ll, ok := vexSrcLen[mnem]
if !ok {
return fmt.Errorf("no fixed vector length for %s", mnem)
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
// An unused vvvv field must be stored as all ones (v̄vvv = 1111).
return e.emitVexFields(spec, ll, regField, rBit, 15, src)
}
// encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst.
// The destination is carried in VEX.vvvv, the source in ModRM.rm, and the
// shift kind in the ModRM.reg /digit.
+113 -67
View File
@@ -39,11 +39,14 @@ func TestVexNDS3(t *testing.T) {
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(% x): %v", mnem, code, err)
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
continue
}
if inst.Op.String() != mnem {
t.Errorf("%s: decoded as %s (% x)", mnem, inst.Op.String(), code)
// The decoder folds the Plan 9 L/Q GPR-width spellings (VCVTSI2SDL/
// SDQ, SSL/SSQ) onto the base name; the W bit carries the width.
got := inst.Op.String()
if got != mnem && !(len(mnem) > len(got) && mnem[:len(got)] == got) {
t.Errorf("%s: decoded as %s (% x)", mnem, got, code)
}
}
}
@@ -156,82 +159,121 @@ func TestVexShiftImm(t *testing.T) {
// as well as every new operand form.
func TestVexGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
name string
mnem string
ops []Operand
want string
wantOp string // decoded mnemonic, when it differs from mnem (the X/Y spellings)
}{
// Three-operand NDS form.
{"VPADDQ Y8,Y9,Y8", "VPADDQ", []Operand{vreg(t, "Y8"), vreg(t, "Y9"), vreg(t, "Y8")}, "c44135d4c0"},
{"VPADDQ X9,X8,X8", "VPADDQ", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c44139d4c1"},
{"VPXOR X7,X7,X7", "VPXOR", []Operand{vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7")}, "c5c1efff"},
{"VPSHUFB Y1,Y2,Y3", "VPSHUFB", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d00d9"},
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9"},
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec"},
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9"},
{"VPADDQ Y8,Y9,Y8", "VPADDQ", []Operand{vreg(t, "Y8"), vreg(t, "Y9"), vreg(t, "Y8")}, "c44135d4c0", ""},
{"VPADDQ X9,X8,X8", "VPADDQ", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c44139d4c1", ""},
{"VPXOR X7,X7,X7", "VPXOR", []Operand{vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7")}, "c5c1efff", ""},
{"VPSHUFB Y1,Y2,Y3", "VPSHUFB", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d00d9", ""},
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""},
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""},
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""},
// Floating point (packed and scalar) and FMA — same NDS form, the pp
// bits and map select the operation.
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1"},
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9"},
{"VMULPD Y12,Y12,Y12", "VMULPD", []Operand{vreg(t, "Y12"), vreg(t, "Y12"), vreg(t, "Y12")}, "c4411d59e4"},
{"VXORPD Y8,Y8,Y8", "VXORPD", []Operand{vreg(t, "Y8"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d57c0"},
{"VUNPCKHPD X8,X8,X9", "VUNPCKHPD", []Operand{vreg(t, "X8"), vreg(t, "X8"), vreg(t, "X9")}, "c4413915c8"},
{"VADDSD X9,X8,X8", "VADDSD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c4413b58c1"},
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8"},
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6"},
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807"},
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""},
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""},
{"VMULPD Y12,Y12,Y12", "VMULPD", []Operand{vreg(t, "Y12"), vreg(t, "Y12"), vreg(t, "Y12")}, "c4411d59e4", ""},
{"VXORPD Y8,Y8,Y8", "VXORPD", []Operand{vreg(t, "Y8"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d57c0", ""},
{"VUNPCKHPD X8,X8,X9", "VUNPCKHPD", []Operand{vreg(t, "X8"), vreg(t, "X8"), vreg(t, "X9")}, "c4413915c8", ""},
{"VADDSD X9,X8,X8", "VADDSD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c4413b58c1", ""},
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
// Two-operand reg/rm form (v̄vvv must be 1111).
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0"},
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306"},
{"VPBROADCASTD X0,Y15", "VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, "c4627d58f8"},
{"VCVTDQ2PD X12,Y12", "VCVTDQ2PD", []Operand{vreg(t, "X12"), vreg(t, "Y12")}, "c4417ee6e4"},
{"VCVTDQ2PD (SI),Y4", "VCVTDQ2PD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y4")}, "c5fee626"},
{"VPMOVMSKB X11,AX", "VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, "c4c179d7c3"},
{"VMOVMSKPS Y7,AX", "VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, "c5fc50c7"},
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
{"VPBROADCASTD X0,Y15", "VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, "c4627d58f8", ""},
{"VCVTDQ2PD X12,Y12", "VCVTDQ2PD", []Operand{vreg(t, "X12"), vreg(t, "Y12")}, "c4417ee6e4", ""},
{"VCVTDQ2PD (SI),Y4", "VCVTDQ2PD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y4")}, "c5fee626", ""},
{"VPMOVMSKB X11,AX", "VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, "c4c179d7c3", ""},
{"VMOVMSKPS Y7,AX", "VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, "c5fc50c7", ""},
// Immediate shifts.
{"VPSLLD $1,Y3,Y4", "VPSLLD", []Operand{Imm(1), vreg(t, "Y3"), vreg(t, "Y4")}, "c5dd72f301"},
{"VPSRLQ $2,Y5,Y6", "VPSRLQ", []Operand{Imm(2), vreg(t, "Y5"), vreg(t, "Y6")}, "c5cd73d502"},
{"VPSLLD $1,Y3,Y4", "VPSLLD", []Operand{Imm(1), vreg(t, "Y3"), vreg(t, "Y4")}, "c5dd72f301", ""},
{"VPSRLQ $2,Y5,Y6", "VPSRLQ", []Operand{Imm(2), vreg(t, "Y5"), vreg(t, "Y6")}, "c5cd73d502", ""},
// Variable-count shifts: the count lives in an XMM register or memory
// and the instruction takes the NDS form.
{"VPSRLQ X0,Y8,Y8", "VPSRLQ", []Operand{vreg(t, "X0"), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd3c0"},
{"VPSRLQ (AX),Y8,Y8", "VPSRLQ", []Operand{Ptr(AX, 0, 16), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd300"},
{"VPSLLD X0,Y1,Y2", "VPSLLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f2d0"},
{"VPSRLD X0,Y1,Y2", "VPSRLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5d2d0"},
{"VPSRAD X0,Y1,Y2", "VPSRAD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5e2d0"},
{"VPSLLQ X0,Y1,Y2", "VPSLLQ", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f3d0"},
{"VPSRLQ X0,Y8,Y8", "VPSRLQ", []Operand{vreg(t, "X0"), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd3c0", ""},
{"VPSRLQ (AX),Y8,Y8", "VPSRLQ", []Operand{Ptr(AX, 0, 16), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd300", ""},
{"VPSLLD X0,Y1,Y2", "VPSLLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f2d0", ""},
{"VPSRLD X0,Y1,Y2", "VPSRLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5d2d0", ""},
{"VPSRAD X0,Y1,Y2", "VPSRAD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5e2d0", ""},
{"VPSLLQ X0,Y1,Y2", "VPSLLQ", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f3d0", ""},
// Immediate shuffle (reg=dst, rm=src, imm8).
{"VPSHUFD $0xEE,X8,X9", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")}, "c4417970c8ee"},
{"VPSHUFD $0xEE,Y1,Y2", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "Y1"), vreg(t, "Y2")}, "c5fd70d1ee"},
{"VPERMQ $0x1B,Y1,Y2", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e3fd00d11b"},
{"VPERMQ $0x1B,Y11,Y12", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y11"), vreg(t, "Y12")}, "c443fd00e31b"},
{"VPSHUFD $0xEE,X8,X9", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")}, "c4417970c8ee", ""},
{"VPSHUFD $0xEE,Y1,Y2", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "Y1"), vreg(t, "Y2")}, "c5fd70d1ee", ""},
{"VPERMQ $0x1B,Y1,Y2", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e3fd00d11b", ""},
{"VPERMQ $0x1B,Y11,Y12", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y11"), vreg(t, "Y12")}, "c443fd00e31b", ""},
// Three-operand + immediate (reg=dst, vvvv=src1, rm=src2, imm8).
{"VSHUFPD $1,X1,X2,X3", "VSHUFPD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e9c6d901"},
{"VSHUFPD $1,Y1,Y2,Y3", "VSHUFPD", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5edc6d901"},
{"VPERM2I128 $0x31,Y1,Y2,Y3", "VPERM2I128", []Operand{Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e36d46d931"},
{"VINSERTI128 $1,X5,Y1,Y2", "VINSERTI128", []Operand{Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37538d501"},
{"VSHUFPD $1,X1,X2,X3", "VSHUFPD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e9c6d901", ""},
{"VSHUFPD $1,Y1,Y2,Y3", "VSHUFPD", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5edc6d901", ""},
{"VPERM2I128 $0x31,Y1,Y2,Y3", "VPERM2I128", []Operand{Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e36d46d931", ""},
{"VINSERTI128 $1,X5,Y1,Y2", "VINSERTI128", []Operand{Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37538d501", ""},
// Lane extract (reg=YMM source, rm=XMM/memory destination, imm8).
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101"},
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701"},
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101"},
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""},
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""},
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""},
// Moves — each direction picks its own opcode and VEX.W.
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e"},
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f"},
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca"},
{"VMOVUPD (DI),Y14", "VMOVUPD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y14")}, "c57d1037"},
{"VMOVUPD Y14,(DI)", "VMOVUPD", []Operand{vreg(t, "Y14"), Ptr(DI, 0, 32)}, "c57d1137"},
{"VMOVUPD X1,X2", "VMOVUPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f911ca"},
{"VMOVQ X8,AX", "VMOVQ", []Operand{vreg(t, "X8"), AX}, "c461f97ec0"},
{"VMOVQ AX,X9", "VMOVQ", []Operand{AX, vreg(t, "X9")}, "c461f96ec8"},
{"VMOVQ X8,(DI)", "VMOVQ", []Operand{vreg(t, "X8"), Ptr(DI, 0, 8)}, "c461f97e07"},
{"VMOVQ (SI),X9", "VMOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X9")}, "c461f96e0e"},
{"VMOVQ X8,X2", "VMOVQ", []Operand{vreg(t, "X8"), vreg(t, "X2")}, "c579d6c2"},
{"VMOVQ X2,X8", "VMOVQ", []Operand{vreg(t, "X2"), vreg(t, "X8")}, "c4c179d6d0"},
{"VMOVD X0,(SI)", "VMOVD", []Operand{vreg(t, "X0"), Ptr(SI, 0, 4)}, "c5f97e06"},
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0"},
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006"},
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106"},
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""},
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""},
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""},
{"VMOVUPD (DI),Y14", "VMOVUPD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y14")}, "c57d1037", ""},
{"VMOVUPD Y14,(DI)", "VMOVUPD", []Operand{vreg(t, "Y14"), Ptr(DI, 0, 32)}, "c57d1137", ""},
{"VMOVUPD X1,X2", "VMOVUPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f911ca", ""},
{"VMOVQ X8,AX", "VMOVQ", []Operand{vreg(t, "X8"), AX}, "c461f97ec0", ""},
{"VMOVQ AX,X9", "VMOVQ", []Operand{AX, vreg(t, "X9")}, "c461f96ec8", ""},
{"VMOVQ X8,(DI)", "VMOVQ", []Operand{vreg(t, "X8"), Ptr(DI, 0, 8)}, "c461f97e07", ""},
{"VMOVQ (SI),X9", "VMOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X9")}, "c461f96e0e", ""},
{"VMOVQ X8,X2", "VMOVQ", []Operand{vreg(t, "X8"), vreg(t, "X2")}, "c579d6c2", ""},
{"VMOVQ X2,X8", "VMOVQ", []Operand{vreg(t, "X2"), vreg(t, "X8")}, "c4c179d6d0", ""},
{"VMOVD X0,(SI)", "VMOVD", []Operand{vreg(t, "X0"), Ptr(SI, 0, 4)}, "c5f97e06", ""},
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""},
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""},
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""},
// Packed double arithmetic and unpack — the NDS form, the opcode
// selects the operation.
{"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""},
{"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""},
{"VMINPD Y1,Y2,Y3", "VMINPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5dd9", ""},
{"VMAXPD X4,X5,X6", "VMAXPD", []Operand{vreg(t, "X4"), vreg(t, "X5"), vreg(t, "X6")}, "c5d15ff4", ""},
{"VUNPCKLPD X1,X2,X3", "VUNPCKLPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e914d9", ""},
{"VUNPCKLPD Y1,Y2,Y3", "VUNPCKLPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed14d9", ""},
{"VSUBPD (AX),X1,X2", "VSUBPD", []Operand{Ptr(AX, 0, 16), vreg(t, "X1"), vreg(t, "X2")}, "c5f15c10", ""},
// Scalar double and single arithmetic (F2 / F3 pp, 128-bit only).
{"VSUBSD X1,X2,X3", "VSUBSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5eb5cd9", ""},
{"VDIVSD X7,X1,X2", "VDIVSD", []Operand{vreg(t, "X7"), vreg(t, "X1"), vreg(t, "X2")}, "c5f35ed7", ""},
{"VMINSD X1,X2,X3", "VMINSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5eb5dd9", ""},
{"VMAXSD X3,X4,X5", "VMAXSD", []Operand{vreg(t, "X3"), vreg(t, "X4"), vreg(t, "X5")}, "c5db5feb", ""},
{"VADDSS X1,X2,X3", "VADDSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea58d9", ""},
{"VSUBSS X1,X2,X3", "VSUBSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5cd9", ""},
{"VMULSS X9,X10,X11", "VMULSS", []Operand{vreg(t, "X9"), vreg(t, "X10"), vreg(t, "X11")}, "c4412a59d9", ""},
{"VDIVSS X1,X2,X3", "VDIVSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5ed9", ""},
{"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""},
{"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""},
{"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""},
// VMOVDDUP — duplicate the low double (reg=dst, rm=src, F2 pp).
{"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""},
{"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""},
{"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""},
// Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3
// prefix — see the table comment), DQ→PD.
{"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""},
{"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""},
{"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""},
{"VCVTPS2PD X1,Y2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c5fc5ad1", ""},
// PD→DQ conversions: the X/Y spellings fix the source length and the
// destination is always XMM; the decoder reports the base mnemonic.
{"VCVTPD2DQX X1,X2", "VCVTPD2DQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fbe6d1", "VCVTPD2DQ"},
{"VCVTPD2DQY Y1,X2", "VCVTPD2DQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5ffe6d1", "VCVTPD2DQ"},
{"VCVTTPD2DQX X3,X4", "VCVTTPD2DQX", []Operand{vreg(t, "X3"), vreg(t, "X4")}, "c5f9e6e3", "VCVTTPD2DQ"},
{"VCVTTPD2DQY Y5,X6", "VCVTTPD2DQY", []Operand{vreg(t, "Y5"), vreg(t, "X6")}, "c5fde6f5", "VCVTTPD2DQ"},
{"VCVTPD2DQY (AX),X1", "VCVTPD2DQY", []Operand{Ptr(AX, 0, 32), vreg(t, "X1")}, "c5ffe608", "VCVTPD2DQ"},
// No-operand.
{"VZEROUPPER", "VZEROUPPER", nil, "c5f877"},
{"VZEROUPPER", "VZEROUPPER", nil, "c5f877", ""},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
@@ -251,7 +293,11 @@ func TestVexGroundTruth(t *testing.T) {
if inst.Len != len(code) {
t.Errorf("%s: Decode consumed %d of %d bytes", c.name, inst.Len, len(code))
}
if inst.Op.String() != c.mnem {
wantOp := c.wantOp
if wantOp == "" {
wantOp = c.mnem
}
if inst.Op.String() != wantOp {
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
}
}
+135 -9
View File
@@ -24,11 +24,12 @@ import (
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
"sourcedock.dev/petrbalvin/gasm-devkit/lsp"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
)
// version is the release version, stamped at build time via
// -ldflags "-X main.version=…" (defaulting to the current release).
var version = "0.8.0"
var version = "0.23.0"
func main() {
if len(os.Args) < 2 {
@@ -46,6 +47,8 @@ func main() {
os.Exit(cmdLint(os.Args[2:]))
case "asm":
os.Exit(cmdAsm(os.Args[2:]))
case "verify":
os.Exit(cmdVerify(os.Args[2:]))
case "lsp":
os.Exit(cmdLSP(os.Args[2:]))
case "version", "--version", "-V":
@@ -80,6 +83,7 @@ Commands:
fmt canonicalise formatting (gofmt for assembly)
lint run static checks
asm assemble .s files to machine code (amd64)
verify JIT-assemble and run dynamic checks (amd64)
lsp run the language server over stdio
version print the version (same as --version)
@@ -93,6 +97,8 @@ Examples:
gasm fmt reformat every .s below the current directory
gasm lint go-flac/*.s run static checks over the kernels
gasm asm -o k.bin kern_amd64.s
gasm asm --format elf -o k.o kern_amd64.s
gasm asm --format goobj -p pkg/path -o k.o kern_amd64.s
`, version)
}
@@ -339,17 +345,26 @@ hover, document symbols, diagnostics and semantic-token highlighting.
}
func cmdAsm(args []string) int {
fs := newCommand("asm", "gasm asm [-o out.bin] <file>", `
fs := newCommand("asm", "gasm asm [--format raw|elf|macho|goobj] [-p pkg] [-o out] <file>", `
Assemble FILE (amd64) without the Go toolchain: every TEXT function is
encoded to machine code — scalar, VEX/AVX2 and EVEX/AVX-512 instructions,
FP/SP frame mapping, local labels and file-local static symbols (GLOBL/DATA)
resolved RIP-relative — and printed as a hex dump. With -o the concatenated
image (functions followed by the data section) is written to a file instead.
resolved RIP-relative — and printed as a hex dump.
With -o the output is written to a file instead. The --format flag selects
what is written: raw (the default) concatenates the functions and the data
section into one self-consistent image; elf and macho emit a relocatable
object (.text/.data sections, a symbol table and one PC32 relocation per
static-symbol reference) that links with the system toolchain; goobj emits
the Go toolchain's own object format, which cmd/link consumes directly (it
requires -p, the package path, and the installed Go toolchain).
`)
out := fs.String("o", "", "write the concatenated machine code to this file")
out := fs.String("o", "", "write the output to this file")
format := fs.String("format", "raw", "output format: raw (concatenated image), elf, macho or goobj (Go object)")
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm asm [-o out.bin] <file>")
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|macho|goobj] [-p pkg] [-o out] <file>")
return 2
}
path := fs.Arg(0)
@@ -420,12 +435,123 @@ image (functions followed by the data section) is written to a file instead.
}
}
if *out != "" {
all := img.Bytes()
if err := os.WriteFile(*out, all, 0o644); err != nil {
var obj []byte
var err error
var kind string
switch *format {
case "raw":
if len(img.Externals) > 0 {
fmt.Fprintf(os.Stderr, "gasm asm: external symbol %q needs an object file (use --format elf or --format macho)\n", img.Externals[0])
return 1
}
obj, kind = img.Bytes(), "raw image"
case "elf":
obj, err = img.ELFObject()
kind = "ELF object"
case "macho":
obj, err = img.MachOObject()
kind = "Mach-O object"
case "goobj":
obj, err = img.GOObject(*pkg, path)
kind = "Go object"
default:
fmt.Fprintf(os.Stderr, "gasm asm: unknown format %q (want raw, elf, macho or goobj)\n", *format)
return 2
}
if err != nil {
fmt.Fprintln(os.Stderr, "gasm asm:", err)
return 1
}
fmt.Printf("wrote %d bytes to %s\n", len(all), *out)
if err := os.WriteFile(*out, obj, 0o644); err != nil {
fmt.Fprintln(os.Stderr, "gasm asm:", err)
return 1
}
fmt.Printf("wrote %d bytes to %s (%s)\n", len(obj), *out, kind)
}
return 0
}
func cmdVerify(args []string) int {
fs := newCommand("verify", "gasm verify [-smoke] [-abi] [-profile] <file.s>", `
Assemble FILE (amd64), map it into executable memory and report the available
functions. This confirms the assembled image is self-consistent (no
unresolved external symbols) and executable — the prerequisite for dynamic
testing.
With -smoke, each NOSPLIT function is called with a zeroed argument block to
confirm the JIT trampoline works end-to-end. This is safe only for functions
that tolerate nil pointers and zero lengths in their arguments.
With -abi, each function is called with sentinel values in the callee-saved
registers (BP, R14) and a red-zone canary below SP; violations are reported.
With -profile, the static basic-block structure is listed for each function.
`)
smoke := fs.Bool("smoke", false, "call each NOSPLIT function with zeroed args")
abi := fs.Bool("abi", false, "run ABI-checking calls (sentinel registers + red zone)")
profile := fs.Bool("profile", false, "list basic-block structure per function")
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm verify [-smoke] [-abi] [-profile] <file.s>")
return 2
}
path := fs.Arg(0)
if arch.FromFilename(path) != arch.AMD64 {
fmt.Fprintln(os.Stderr, "gasm verify: only amd64 is supported")
return 1
}
k, err := verify.Load(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
defer k.Close()
names := k.FuncNames()
fmt.Printf("%s: %d functions JIT-loaded\n", path, len(names))
rc := 0
for _, name := range names {
fl, _ := k.Func(name)
flags := ""
if fl.NoSplit {
flags = " NOSPLIT"
}
fmt.Printf(" %s: %d bytes, args=%d, frame=%d%s\n", name, fl.Size, fl.Args, fl.Frame, flags)
if *profile {
blocks, err := k.Blocks(name)
if err != nil {
fmt.Printf(" profile: %v\n", err)
} else {
fmt.Printf(" blocks: %d\n", len(blocks))
}
}
if *smoke && fl.NoSplit {
args := make([]byte, fl.Args)
_, err := k.CallFunc(name, args)
if err != nil {
fmt.Printf(" smoke: FAIL — %v\n", err)
rc = 1
} else {
fmt.Printf(" smoke: OK\n")
}
}
if *abi && fl.NoSplit {
args := make([]byte, fl.Args)
_, report, err := k.CallFuncChecked(name, args)
if err != nil {
fmt.Printf(" abi: FAIL — %v\n", err)
rc = 1
} else if !report.OK() {
fmt.Printf(" abi: %s\n", report)
rc = 1
} else {
fmt.Printf(" abi: clean\n")
}
}
}
return rc
}
+84 -11
View File
@@ -142,7 +142,7 @@ Two deeper analyses sit on top of the AST:
has no System V style callee-saved registers (amd64 `BX`, `R12`–`R15` and
the like are caller-saved or permanent scratch, and hand-written kernels may
clobber them freely). The audited set is the frame pointer and the
goroutine pointer per architecture (amd64 `BP`/`R14`, arm64 `R18`/`R28`/
the frame pointer, the goroutine pointer per architecture (amd64 `BP`/`R14`, arm64 `R18`/`R28`/
`R29`, riscv64 `X27`, loong64 `R22`); the goroutine pointer is reported only
when the function can reach the runtime — it is not `NOSPLIT` or makes a
call — since the ABI0 transition machinery restores it on those paths, and
@@ -213,15 +213,46 @@ three-operand-plus-immediate form (`VSHUFPD`,
`VPERM2I128`, `VINSERTI128`), the lane-extract form (`VEXTRACTI128`,
`VEXTRACTF128`, where the YMM source occupies the reg field and the XMM or
memory destination r/m), the direction-sensitive moves (`VMOVDQU`, `VMOVUPD`,
`VMOVD`, `VMOVQ`, `VMOVSD`), the floating-point and FMA arithmetic (`VADDPD`,
`VMULPD`, `VXORPD`, `VUNPCKHPD`, the scalar `VADDSD`/`VMULSD`, `VCVTDQ2PD`,
`VFMADD231PD`) and the no-operand `VZEROUPPER` — together with `VPERMD` and
`VMOVD`, `VMOVQ`, `VMOVSD`), the floating-point and FMA arithmetic — the
packed double operations (`VADDPD`/`VSUBPD`/`VMULPD`/`VDIVPD`/`VMINPD`/
`VMAXPD`), the unpacks (`VUNPCKHPD`/`VUNPCKLPD`), the scalar SD and SS
operations, `VMOVDDUP`, `VXORPD`, the width-changing conversions
(`VCVTDQ2PS`, `VCVTPS2PD`, `VCVTDQ2PD`, and the `VCVTPD2DQX`/`Y` and
`VCVTTPD2DQX`/`Y` spellings, whose length follows the wider source) and
`VFMADD231PD` — and the no-operand `VZEROUPPER`, together with `VPERMD` and
the scalar families (`CMOVcc`, `SETcc`, `LZCNT`/`TZCNT`, the extending moves,
`CVTSx2SD`, `IMUL3`) and the EVEX (AVX-512) prefix — the four-byte prefix with
5-bit register fields (Z0–Z31, X/Y 16–31), opmask registers as operands and
mask destinations, and the compressed disp8×N displacement, whose multiplier
follows the memory operand's size — covering every instruction the go-flac
AVX2 and AVX-512 kernels use. Every encoding is validated two ways: by
5-bit register fields (Z0–Z31, X/Y 16–31, with the mod=11 quirk that carries
rm[4] in X̄), opmask registers (K0–K7 as operands, mask destinations and
explicit merging/zeroing masks — written the way Go writes them, as a K
operand among the operands plus a `.Z` mnemonic suffix), and the compressed
disp8×N displacement, whose multiplier follows the memory operand's size —
covering every instruction the go-flac and go-lz4 AVX2/AVX-512 kernels use,
plus the common AVX-512 F/BW integer set, the floating-point and conversion
set (the packed double and single arithmetic, the scalar SD/SS forms —
whose EVEX encodings serve masked and zeroing use — `VMOVDDUP`, the
replicating moves, and the width-changing conversions, including the
`VCVTPD2DQ`/`VCVTTPD2DQ` family whose length follows the wider source
operand), and the wider AVX-512 set: ternary logic, lane shuffles, inserts
and extracts, compares with an opmask destination, the permutes, the
expand/compress family, the broadcasts, the opmask-register instructions
(KAND/KOR/KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST and KMOVQ), the aligned
moves and the remaining extending/narrowing moves, the floating-point
helper and conversion tail (VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*,
VSCALEF*, VRNDSCALE*, VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with an
opmask destination, and the VCVT* conversions — signed, unsigned and
truncating, including the length-suffixed X/Y spellings and the
mask/vector conversions VPMOVM2*/VPMOV*2M, and the scalar conversions
between vector and general-purpose registers (VCVT{,T}S{D,S}2SI{,Q} and
the unsigned forms, VCVTSI2*/VCVTUSI2*), and gather/scatter with VSIB addressing — both the
VEX spelling with a vector mask register and the EVEX spelling with an
explicit K mask, where the EVEX length follows the VSIB index register,
not the data register. The EVEX mnemonic
suffixes — rounding modes (.RN_SAE/.RD_SAE/.RU_SAE/.RZ_SAE),
suppress-all-exceptions (.SAE) and memory broadcast (.BCST) — set the EVEX
b bit and the L'L rounding-control field (broadcast keeps the vector length
and scales disp8 by the element size), and combine with the .Z zeroing
suffix. Every encoding is validated two ways: by
round-trip decoding through `golang.org/x/arch`, and byte-for-byte against
the machine code the real Go assembler emits — a comparison that holds for
whole functions: all 27 functions of both kernels assemble to exactly the Go
@@ -232,9 +263,51 @@ File-level assembly (`AssembleFile`) goes beyond single functions: it
materialises the file's static symbols (`GLOBL`/`DATA`) in a data section
behind the code and resolves references to them (`mask<>(SB)`) to
RIP-relative loads whose displacements point inside the resulting image, so
the bytes are self-consistent at any base address. External (non-file-local)
symbols are rejected: they need object-file emission, which — together with
EVEX masking/zeroing and the other architectures — is the rest of Phase 2.
the bytes are self-consistent at any base address. References to symbols no
`GLOBL` defines are kept as relocations on the function layout, and the
object-file emitters turn the whole image into a linkable object: the ELF
and Mach-O writers (`gasm asm --format elf|macho`) lay the code and data out
as `.text`/`.data` (or `__text`/`__data`) sections, export a symbol per
`TEXT` and `GLOBL` (the `<>` ones local, the rest global) and emit one
PC-relative relocation per static-symbol reference — undefined external
symbols included, so the output links with the system toolchain. The GOOBJ
emitter (`gasm asm --format goobj`) writes the format the Go linker consumes
directly: the functions as non-package symbols (the way `cmd/asm` records
assembly symbols), the `GLOBL` data, one `FuncInfo` per function and the
pc-value tables — `pcsp` built from the prologue and epilogue stack
boundaries, plus flat `pcfile`, `pcline` and `pcinline` tables — so a
gasm-assembled object drops into a `go build` in place of the toolchain's.
The object preamble (the version-and-experiment header the linker compares
verbatim) is captured from the installed `go tool asm`, so the output is
always consistent with the toolchain that links it. External cross-package
references and the implicit funcdata/DWARF symbols remain future work (the
linker fills the latter's defaults); the rest of Phase 2 is those, the
remaining EVEX forms and the other architectures.
### `verify`
The dynamic-analysis substrate (Phase 3). It JIT-loads assembled images into
executable memory and invokes them directly, enabling differential testing,
runtime ABI checks and coverage profiling.
The execution model is pure Go (stdlib only). `Map` copies machine code into
an anonymous `syscall.Mmap` mapping and enforces W^X (write the bytes, then
`mprotect` to read-execute). `Call` prepares a stack whose first word is the
address of an assembly trampoline (`leaveJIT`), lays the ABI0 argument
block after it, switches to that stack via `enterJIT` (which saves the Go
stack pointer in a package global and jumps to the target), and recovers
control when the function RETs into `leaveJIT` (which restores the Go stack
and returns). A 64-byte pad below the return address accommodates the
ABIInternal wrapper that the Go runtime interposes on assembly functions.
`Load` / `LoadSource` / `LoadAST` parse, assemble and map a `.s` file in one
step, returning a `Kernel` whose `CallFunc` method marshals the argument block
by name. The image must be self-contained (no external relocations); the
assembler’s `Image.Bytes()` provides the code-and-data concatenation.
The `gasm verify` CLI subcommand exposes this: it loads a file, reports the
available functions and (with `-smoke`) calls each NOSPLIT function with zeroed
arguments to confirm the trampoline round-trips.
## Extension points
+56
View File
@@ -0,0 +1,56 @@
# Deferred decisions
Design decisions deliberately postponed, with enough context to pick them up
again without re-deriving the analysis. Each entry records what is deferred,
why, the options on the table, and the trigger that should reopen it.
---
## GOOBJ external (cross-package) symbol references
**Status:** deferred (v0.15.0, 2026-08-02). The GOOBJ emitter resolves only
symbols defined in the file being assembled; a reference to any other symbol
is rejected.
**Why it is deferred.** GOOBJ symbol references are *positional*: a
reference is a `{PkgIdx, SymIdx}` pair, where `SymIdx` is the index of the
symbol in the *referenced package's* symbol-definition table. That ordering
is not derivable from the reference site — it lives in the referenced
package's gc export data (the iexport binary format, which evolves with the
toolchain). `cmd/asm` reads it with `cmd/internal` readers gasm cannot
import, so emitting external references means either parsing export data
ourselves or taking a dependency that does.
**What works today.** Single-package objects: every symbol the file defines
(as `TEXT` or `GLOBL`, static or exported) and every reference to them.
This covers the production use case — the go-flac / go-lz4 kernels carry no
`FUNCDATA`/`PCDATA`, hence no references into `runtime`, and the Go side
references the assembly symbols, never the reverse. Such a package builds
with its assembly object replaced by a gasm-emitted one.
**The options, when we return.**
1. **`golang.org/x/tools/go/gcexportdata` as a production dependency.**
The straightforward path: read each imported package's export file
(paths from `-importcfg` or `go list -export`), assign symbol indices in
its symbol order, write `PkgIndex`/`Autolib` entries (fingerprints from
the export files' build IDs) and positional references. Robust across
toolchain versions — `x/tools` tracks the format. **Cost:** the first
production dependency beyond the standard library, an explicit deviation
from the "production code depends only on the standard library"
principle in the README. Requires the user's explicit agreement.
2. **A minimal iexport parser of our own.** Preserves self-containment.
Substantial effort and inherently fragile: the format is an internal
contract that changes with Go releases, so the parser needs a
version-gated fallback and regression tests against several toolchains.
3. **Shell out to the toolchain for symbol metadata.** Consistent with the
existing GOOBJ preamble probe (which already runs `go tool asm`), but no
toolchain command exposes a package's symbols *in definition-index
order* — `go tool nm` sorts differently — so this does not solve the
core problem on its own; it would only feed option 1 or 2.
**Trigger to reopen.** An assembly file that needs a cross-package
reference — in practice `FUNCDATA $…, runtime·…(SB)` (stack maps / GC
metadata written in assembly), or any kernel that calls into another
package directly. Until then, option 3's limitation is moot and the
single-package emitter suffices.
+1 -1
View File
@@ -3,7 +3,7 @@
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
version := "0.8.0"
version := "0.23.0"
default:
@just --list
+23 -1
View File
@@ -242,7 +242,7 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
}
}
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) {
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) && !maskedEvex(mnem, st.Operands) {
if in, ok := tab.Lookup(mnem); ok && in.MinOps >= 0 {
n := len(st.Operands)
if n < in.MinOps || n > in.MaxOps {
@@ -438,6 +438,28 @@ func isMacroInvocation(mnem string, macros map[string]bool) bool {
return strings.Contains(mnem, "_") || macros[mnem]
}
// maskedEvex reports whether the instruction is a masked EVEX form: the
// mnemonic carries a .Z suffix, or the operand list contains an opmask
// register (K1–K7). Either way the operand count differs from the unmasked
// form, so count checks are skipped.
func maskedEvex(mnem string, ops []*ast.Operand) bool {
if strings.Contains(mnem, ".") {
return true
}
for _, op := range ops {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Base == "" &&
op.Addr.Index == "" && op.Addr.Sym.Pseudo == "" && isMaskReg(op.Addr.Sym.Name) {
return true
}
}
return false
}
// isMaskReg reports whether name is an opmask register K0–K7.
func isMaskReg(name string) bool {
return len(name) == 2 && name[0] == 'K' && name[1] >= '0' && name[1] <= '7'
}
// isConditionalDirective reports whether a preprocessor directive (the text
// after '#') is a conditional-compilation directive whose branches the parser
// cannot resolve.
+20
View File
@@ -187,6 +187,26 @@ done:
}
}
// TestEvexMaskingRecognised checks that masked EVEX forms — the .Z suffix and
// an explicit K operand — are recognised and exempt from operand-count
// checks.
func TestEvexMaskingRecognised(t *testing.T) {
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
VPADDD.Z Z1, Z2, K2, Z3
VPMINSD Z1, Z2, K5, Z3
VMOVDQU8 Z1, K3, (SI)
RET
`)
if codes(diags)[CodeUnknownInstr] != 0 {
t.Fatalf("masked EVEX must be recognised: %+v", diags)
}
if codes(diags)[CodeOperandCount] != 0 {
t.Fatalf("masked operand counts must not be flagged: %+v", diags)
}
}
func TestArm64AddressingSuffix(t *testing.T) {
// .W (pre-index) and .P (post-index) suffixes must resolve to the base
// instruction.
+5
View File
@@ -375,6 +375,11 @@ func parseImmediate(g []token.Token) ast.Immediate {
if v, ok := tryInt(text); ok {
imm.Val = v
imm.HasVal = true
} else if u, err := strconv.ParseUint(text, 0, 64); err == nil && !imm.Neg {
// Unsigned 64-bit literals (DATA mask<>+8(SB)/8, $0x8000…)
// overflow int64; keep the bit pattern.
imm.Val = int64(u)
imm.HasVal = true
} else {
imm.Float = text
}
+28
View File
@@ -0,0 +1,28 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// func cleanAdd(a, b int64) int64
// A well-behaved function that preserves all callee-saved registers.
TEXT ·cleanAdd(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
ADDQ b+8(FP), AX
MOVQ AX, ret+16(FP)
RET
// func dirtyBP(a int64) int64
// Deliberately clobbers BP (an ABI violation for a NOSPLIT frame=0 function).
TEXT ·dirtyBP(SB), NOSPLIT, $0-16
MOVQ $0x1234, BP
MOVQ a+0(FP), AX
MOVQ AX, ret+8(FP)
RET
// func dirtyR14(a int64) int64
// Deliberately clobbers R14 (the goroutine pointer — a serious ABI violation).
TEXT ·dirtyR14(SB), NOSPLIT, $0-16
MOVQ $0x5678, R14
MOVQ a+0(FP), AX
MOVQ AX, ret+8(FP)
RET
+67
View File
@@ -0,0 +1,67 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// func add(a, b int64) int64
TEXT ·add(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
ADDQ b+8(FP), AX
MOVQ AX, ret+16(FP)
RET
// func sum(data []int64) int64
// Sums all elements of the slice.
TEXT ·sum(SB), NOSPLIT, $0-32
MOVQ data_base+0(FP), SI
MOVQ data_len+8(FP), CX
XORQ AX, AX
TESTQ CX, CX
JZ sum_done
sum_loop:
ADDQ (SI), AX
ADDQ $8, SI
DECQ CX
JNZ sum_loop
sum_done:
MOVQ AX, ret+24(FP)
RET
// func wideCopy(dst, src []byte)
// Non-overlapping copy of min(len(dst), len(src)) bytes using 32-byte moves.
TEXT ·wideCopy(SB), NOSPLIT, $0-48
MOVQ dst_base+0(FP), DI
MOVQ dst_len+8(FP), BX
MOVQ src_base+24(FP), SI
MOVQ src_len+32(FP), R8
CMPQ BX, R8
JLE wc_have_n
MOVQ R8, BX
wc_have_n:
CMPQ BX, $32
JB wc_small
VMOVDQU (SI), Y0
VMOVDQU Y0, (DI)
VMOVDQU -32(SI)(BX*1), Y0
VMOVDQU Y0, -32(DI)(BX*1)
VZEROUPPER
RET
wc_small:
TESTQ BX, BX
JZ wc_done
wc_byte:
MOVB (SI), R8B
MOVB R8B, (DI)
INCQ SI
INCQ DI
DECQ BX
JNZ wc_byte
wc_done:
RET
+130
View File
@@ -0,0 +1,130 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build amd64
package verify
import (
"encoding/binary"
"fmt"
"syscall"
"unsafe"
)
// abiResult records register-clobber violations detected by the ABI-checking
// trampoline. Bit 0: BP clobbered. Bit 1: R14 clobbered.
var abiResult uint64
// savedBP holds the caller's frame pointer across the ABI-checked JIT call.
// Referenced by enterJITChecked to satisfy go vet's save-before-clobber rule.
var savedBP uintptr
// leaveCheckedPtr is initialised by the linker from the GLOBL/DATA in
// abi_amd64.s: it holds the raw address of leaveJITCheckedRaw (which has
// no ABIInternal wrapper, so the JIT function RETs directly into it).
var leaveCheckedPtr uintptr
// enterJITChecked sets sentinels in BP and R14, switches to the prepared
// stack and jumps to fn.
//
//go:nosplit
func enterJITChecked(fn uintptr, stack uintptr)
// leaveJITCheckedRaw is the raw return trampoline for ABI checks. Its
// address is obtained from the GLOBL in abi_amd64.s (leaveCheckedPtr),
// which points to the .abi0 code — NOT the ABIInternal wrapper that this
// declaration would generate. The declaration exists solely to satisfy
// go vet's "missing Go declaration" check.
//
//go:nosplit
func leaveJITCheckedRaw()
// ABIReport describes the result of an ABI-checking call.
type ABIReport struct {
BPClobbered bool // BP was modified by the function
R14Clobbered bool // R14 (goroutine pointer) was modified
RedZoneHit bool // the 128-byte red zone below SP was written
}
// OK returns true when no violations were detected.
func (r ABIReport) OK() bool {
return !r.BPClobbered && !r.R14Clobbered && !r.RedZoneHit
}
// String returns a human-readable summary.
func (r ABIReport) String() string {
if r.OK() {
return "ABI clean"
}
s := "ABI violation:"
if r.BPClobbered {
s += " BP clobbered"
}
if r.R14Clobbered {
s += " R14 clobbered"
}
if r.RedZoneHit {
s += " red-zone written"
}
return s
}
// redZoneSize is the System V AMD64 red zone: 128 bytes below SP that a
// leaf function may use without adjusting SP. Go does not use the red zone,
// so any write there is a bug.
const redZoneSize = 128
// redZoneFill is the byte pattern used to detect red-zone writes.
const redZoneFill = 0xA5
// CallChecked invokes the function with ABI sentinels and a red-zone
// canary, returning both the argument block (with results) and an ABIReport.
func CallChecked(fnAddr uintptr, args []byte) ([]byte, ABIReport, error) {
report := ABIReport{}
// Reset the global result.
abiResult = 0
// Prepare the stack: [red-zone canary][padding][leaveJITCheckedRaw][args...]
// The red zone sits below the initial SP, so the function would have to
// write below SP to corrupt it.
totalSize := redZoneSize + stackPad + 8 + len(args) + 64
stackMem, err := syscall.Mmap(-1, 0, totalSize,
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, report, fmt.Errorf("verify: stack mmap: %w", err)
}
defer syscall.Munmap(stackMem)
// Fill the red zone with the canary pattern.
for i := 0; i < redZoneSize; i++ {
stackMem[i] = redZoneFill
}
// Return address and args after the red zone and padding.
retOff := redZoneSize + stackPad
binary.LittleEndian.PutUint64(stackMem[retOff:retOff+8], uint64(leaveCheckedPtr))
copy(stackMem[retOff+8:], args)
stackBase := uintptr(unsafe.Pointer(&stackMem[retOff]))
enterJITChecked(fnAddr, stackBase)
// Read the register-clobber result.
res := abiResult
report.BPClobbered = res&1 != 0
report.R14Clobbered = res&2 != 0
// Check the red zone.
for i := 0; i < redZoneSize; i++ {
if stackMem[i] != redZoneFill {
report.RedZoneHit = true
break
}
}
// Copy out the argument area.
out := make([]byte, len(args))
copy(out, stackMem[retOff+8:retOff+8+len(args)])
return out, report, nil
}
+61
View File
@@ -0,0 +1,61 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// ABI-checking trampoline. Sets sentinel values in the callee-saved
// registers (BP, R14) before entering the JIT function and checks whether
// they survived on return.
//
// The return trampoline (leaveJITCheckedRaw) is a raw TEXT symbol with no
// Go function declaration, so the toolchain does NOT interpose an
// ABIInternal wrapper — the JIT function RETs directly into the check code,
// which sees the registers exactly as the function left them.
//
// Go ABI0 on amd64 guarantees:
// - BP is callee-saved (NOSPLIT frame=0 functions must not touch it).
// - R14 holds the goroutine pointer and must survive across any call.
// Sentinel values chosen to be unlikely in normal execution.
#define SENTINEL_BP 0xDEADBEEFCAFEF00D
#define SENTINEL_R14 0x0BADF00DDEADBEEF
// GLOBL holding the raw address of the leave trampoline, read by Go.
GLOBL ·leaveCheckedPtr(SB), NOPTR, $8
DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB)
// func enterJITChecked(fn uintptr, stack uintptr)
// Sets sentinels in BP and R14, switches to the prepared stack and jumps
// to fn. The prepared stack's return address must be leaveJITCheckedRaw
// (read from leaveCheckedPtr).
TEXT ·enterJITChecked(SB), NOSPLIT, $0-16
MOVQ fn+0(FP), AX // target (before SP switch)
MOVQ SP, ·savedSP(SB) // preserve Go stack
MOVQ BP, ·savedBP(SB) // preserve frame pointer (vet requires save before clobber)
MOVQ $SENTINEL_BP, BP // sentinel in BP
MOVQ $SENTINEL_R14, R14 // sentinel in R14
MOVQ stack+8(FP), SP // switch to prepared stack
JMP AX
// leaveJITCheckedRaw is the raw return trampoline. It has NO Go function
// declaration, so no ABIInternal wrapper is generated — the JIT function's
// RET lands here directly, seeing BP and R14 exactly as the function left
// them. It checks the sentinels, records violations in abiResult, then
// restores the Go stack and returns.
TEXT ·leaveJITCheckedRaw(SB), NOSPLIT, $0-0
// Check BP against the sentinel.
MOVQ $SENTINEL_BP, CX
CMPQ BP, CX
JEQ bp_ok
ORQ $1, ·abiResult(SB)
bp_ok:
// Check R14 against the sentinel.
MOVQ $SENTINEL_R14, CX
CMPQ R14, CX
JEQ r14_ok
ORQ $2, ·abiResult(SB)
r14_ok:
MOVQ ·savedSP(SB), SP
RET
+26
View File
@@ -0,0 +1,26 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build !amd64
package verify
import "fmt"
// ABIReport describes the result of an ABI-checking call.
type ABIReport struct {
BPClobbered bool
R14Clobbered bool
RedZoneHit bool
}
// OK returns true when no violations were detected.
func (r ABIReport) OK() bool { return false }
// String returns a human-readable summary.
func (r ABIReport) String() string { return "verify: ABI checks require amd64" }
// CallChecked is unavailable on non-amd64 architectures.
func CallChecked(fnAddr uintptr, args []byte) ([]byte, ABIReport, error) {
return nil, ABIReport{}, fmt.Errorf("verify: ABI checks require amd64")
}
+145
View File
@@ -0,0 +1,145 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"testing"
"unsafe"
)
func loadABIKernel(t *testing.T) *Kernel {
t.Helper()
k, err := Load("../testdata/verify/abi_amd64.s")
if err != nil {
t.Fatalf("Load: %v", err)
}
t.Cleanup(k.Close)
return k
}
func TestABIClean(t *testing.T) {
k := loadABIKernel(t)
args := make([]byte, 24)
PutUint64(args, 0, 3)
PutUint64(args, 8, 4)
out, report, err := k.CallFuncChecked("cleanAdd", args)
if err != nil {
t.Fatalf("CallFuncChecked: %v", err)
}
if got := int64(GetUint64(out, 16)); got != 7 {
t.Errorf("cleanAdd(3, 4) = %d, want 7", got)
}
if !report.OK() {
t.Errorf("cleanAdd: %s", report)
}
}
func TestABIBPClobbered(t *testing.T) {
k := loadABIKernel(t)
args := make([]byte, 16)
PutUint64(args, 0, 42)
out, report, err := k.CallFuncChecked("dirtyBP", args)
if err != nil {
t.Fatalf("CallFuncChecked: %v", err)
}
if got := int64(GetUint64(out, 8)); got != 42 {
t.Errorf("dirtyBP(42) = %d, want 42", got)
}
if !report.BPClobbered {
t.Error("dirtyBP: expected BP clobbered, but report says clean")
}
if report.R14Clobbered {
t.Error("dirtyBP: R14 should not be clobbered")
}
}
func TestABIR14Clobbered(t *testing.T) {
k := loadABIKernel(t)
args := make([]byte, 16)
PutUint64(args, 0, 99)
out, report, err := k.CallFuncChecked("dirtyR14", args)
if err != nil {
t.Fatalf("CallFuncChecked: %v", err)
}
if got := int64(GetUint64(out, 8)); got != 99 {
t.Errorf("dirtyR14(99) = %d, want 99", got)
}
if !report.R14Clobbered {
t.Error("dirtyR14: expected R14 clobbered, but report says clean")
}
if report.BPClobbered {
t.Error("dirtyR14: BP should not be clobbered")
}
}
// TestABILZ4Kernels verifies that the production go-lz4 kernels are ABI-clean:
// they preserve BP and R14 and do not write into the red zone.
func TestABILZ4Kernels(t *testing.T) {
k := loadLZ4Kernel(t)
// wideCopyAVX2 with a real copy.
src := make([]byte, 128)
for i := range src {
src[i] = byte(i)
}
dst := make([]byte, 128)
args := make([]byte, 48)
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
PutUint64(args, 8, 128)
PutUint64(args, 16, 128)
PutPtr(args, 24, unsafe.Pointer(&src[0]))
PutUint64(args, 32, 128)
PutUint64(args, 40, 128)
_, report, err := k.CallFuncChecked("wideCopyAVX2", args)
if err != nil {
t.Fatalf("CallFuncChecked(wideCopyAVX2): %v", err)
}
if !report.OK() {
t.Errorf("wideCopyAVX2: %s", report)
}
// decodeBlockAVX2 with a simple block.
decSrc := []byte{0x50, 'H', 'e', 'l', 'l', 'o'}
decDst := make([]byte, 64)
decArgs := make([]byte, 64)
PutPtr(decArgs, 0, unsafe.Pointer(&decSrc[0]))
PutUint64(decArgs, 8, uint64(len(decSrc)))
PutUint64(decArgs, 16, uint64(cap(decSrc)))
PutPtr(decArgs, 24, unsafe.Pointer(&decDst[0]))
PutUint64(decArgs, 32, uint64(len(decDst)))
PutUint64(decArgs, 40, uint64(cap(decDst)))
_, report, err = k.CallFuncChecked("decodeBlockAVX2", decArgs)
if err != nil {
t.Fatalf("CallFuncChecked(decodeBlockAVX2): %v", err)
}
if !report.OK() {
t.Errorf("decodeBlockAVX2: %s", report)
}
}
func TestCallFuncCheckedErrors(t *testing.T) {
k := loadABIKernel(t)
// Nonexistent function.
_, _, err := k.CallFuncChecked("nope", make([]byte, 8))
if err == nil {
t.Fatal("expected error for nonexistent function")
}
// Arg block too small.
_, _, err = k.CallFuncChecked("cleanAdd", make([]byte, 8))
if err == nil {
t.Fatal("expected error for too-small arg block")
}
}
+144
View File
@@ -0,0 +1,144 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"math/rand"
"os"
"testing"
"unsafe"
)
const lz4AVX512Path = "../../go-libraries/go-lz4/avx512_amd64.s"
func loadLZ4AVX512Kernel(t *testing.T) *Kernel {
t.Helper()
if _, err := os.Stat(lz4AVX512Path); err != nil {
t.Skipf("sibling kernel not available: %v", err)
}
k, err := Load(lz4AVX512Path)
if err != nil {
t.Fatalf("Load(%s): %v", lz4AVX512Path, err)
}
t.Cleanup(k.Close)
return k
}
func TestAVX512DecodeKnownAnswers(t *testing.T) {
k := loadLZ4AVX512Kernel(t)
tests := []struct {
name string
src []byte
wantN int
wantCode int
}{
{"literals_only", []byte{0x50, 'H', 'e', 'l', 'l', 'o'}, 5, 0},
{"literals_and_match", []byte{0x54, 'A', 'A', 'A', 'A', 'A', 0x05, 0x00, 0x30, 'B', 'B', 'B'}, 16, 0},
{"overlapping", []byte{0x14, 'X', 0x01, 0x00, 0x10, 'Y'}, 10, 0},
{"malformed", []byte{0x50, 'H', 'e'}, 0, 1},
{"zero_offset", []byte{0x14, 'X', 0x00, 0x00}, 0, 2},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
dst := make([]byte, 64)
args := make([]byte, 64)
PutPtr(args, 0, unsafe.Pointer(&tt.src[0]))
PutUint64(args, 8, uint64(len(tt.src)))
PutUint64(args, 16, uint64(cap(tt.src)))
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
PutUint64(args, 32, uint64(len(dst)))
PutUint64(args, 40, uint64(cap(dst)))
out, err := k.CallFunc("decodeBlockAVX512", args)
if err != nil {
t.Fatalf("CallFunc: %v", err)
}
n := int(GetUint64(out, 48))
code := int(GetUint64(out, 56))
if n != tt.wantN || code != tt.wantCode {
t.Errorf("got (n=%d, code=%d), want (n=%d, code=%d)", n, code, tt.wantN, tt.wantCode)
}
})
}
}
func TestAVX512DifferentialFuzz(t *testing.T) {
k := loadLZ4AVX512Kernel(t)
rng := rand.New(rand.NewSource(77))
for i := 0; i < 3000; i++ {
wantSize := 1 + rng.Intn(4096)
src := genLZ4Block(rng, wantSize)
dstSize := wantSize + 64
goDst := make([]byte, dstSize)
goN, goCode := decodeBlockGo(src, goDst)
jitDst := make([]byte, dstSize)
args := make([]byte, 64)
if len(src) > 0 {
PutPtr(args, 0, unsafe.Pointer(&src[0]))
}
PutUint64(args, 8, uint64(len(src)))
PutUint64(args, 16, uint64(cap(src)))
if dstSize > 0 {
PutPtr(args, 24, unsafe.Pointer(&jitDst[0]))
}
PutUint64(args, 32, uint64(dstSize))
PutUint64(args, 40, uint64(cap(jitDst)))
out, err := k.CallFunc("decodeBlockAVX512", args)
if err != nil {
t.Fatalf("iter %d: %v", i, err)
}
jitN := int(GetUint64(out, 48))
jitCode := int(GetUint64(out, 56))
if jitCode != goCode {
t.Fatalf("iter %d: code mismatch: JIT=%d Go=%d", i, jitCode, goCode)
}
if jitCode != 0 {
continue
}
if jitN != goN {
t.Fatalf("iter %d: n mismatch: JIT=%d Go=%d", i, jitN, goN)
}
if !bytes.Equal(jitDst[:jitN], goDst[:goN]) {
t.Fatalf("iter %d: output mismatch (n=%d)", i, jitN)
}
}
}
func TestAVX512WideCopy(t *testing.T) {
k := loadLZ4AVX512Kernel(t)
sizes := []int{0, 1, 31, 32, 63, 64, 65, 127, 128, 256, 1024}
for _, n := range sizes {
src := make([]byte, n)
for i := range src {
src[i] = byte(i*11 + 3)
}
dst := make([]byte, n)
args := make([]byte, 48)
if n > 0 {
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
PutPtr(args, 24, unsafe.Pointer(&src[0]))
}
PutUint64(args, 8, uint64(n))
PutUint64(args, 16, uint64(n))
PutUint64(args, 32, uint64(n))
PutUint64(args, 40, uint64(n))
_, err := k.CallFunc("wideCopyAVX512", args)
if err != nil {
t.Fatalf("wideCopyAVX512(n=%d): %v", n, err)
}
if !bytes.Equal(dst, src) {
t.Errorf("wideCopyAVX512(n=%d): mismatch", n)
}
}
}
+77
View File
@@ -0,0 +1,77 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build amd64
package verify
import (
"encoding/binary"
"fmt"
"reflect"
"syscall"
"unsafe"
)
// savedSP holds the Go stack pointer while a JIT call is in flight.
// Referenced by the assembly trampoline (trampoline_amd64.s).
var savedSP uintptr
// enterJIT switches to the prepared stack and jumps to fn.
// It does not return normally; the JIT function's RET transfers control
// to leaveJIT, which restores the Go stack.
//
//go:nosplit
func enterJIT(fn uintptr, stack uintptr)
// leaveJIT restores the Go stack after a JIT function returns.
// Its address is placed as the return address on the prepared stack.
//
//go:nosplit
func leaveJIT()
// leaveJITAddr is the machine address of leaveJIT, resolved once at init.
var leaveJITAddr uintptr
func init() {
leaveJITAddr = reflect.ValueOf(leaveJIT).Pointer()
}
// stackPad is padding below the return address on the prepared stack.
// The ABIInternal wrapper that leaveJIT's address resolves to executes
// PUSHQ BP and CALL before reaching the raw assembly, writing up to 16
// bytes below the return-address slot. 64 bytes of headroom is ample.
const stackPad = 64
// Call invokes the assembled function at fnAddr with the given ABI0 argument
// block (the raw bytes that would appear at FP+0). It returns the argument
// block after the call, which contains any results the function wrote back
// (the ABI0 convention shares the argument area for inputs and outputs).
//
// The function must be NOSPLIT (no stack growth) and must not reference
// external symbols — the image is self-contained.
func Call(fnAddr uintptr, args []byte) ([]byte, error) {
// Prepare the stack: [padding][leaveJIT addr][args...]
stackSize := stackPad + 8 + len(args) + 64 // padding + ret + args + safety
stackMem, err := syscall.Mmap(-1, 0, stackSize,
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, fmt.Errorf("verify: stack mmap: %w", err)
}
defer syscall.Munmap(stackMem)
// The return address sits after the padding; the function's SP will
// point here, leaving stackPad bytes below for the wrapper's pushes.
retOff := stackPad
binary.LittleEndian.PutUint64(stackMem[retOff:retOff+8], uint64(leaveJITAddr))
// The ABI0 argument area follows the return address.
copy(stackMem[retOff+8:], args)
stackBase := uintptr(unsafe.Pointer(&stackMem[retOff]))
enterJIT(fnAddr, stackBase)
// Copy out the (possibly modified) argument area.
out := make([]byte, len(args))
copy(out, stackMem[retOff+8:retOff+8+len(args)])
return out, nil
}
+13
View File
@@ -0,0 +1,13 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build !amd64
package verify
import "fmt"
// Call is unavailable on non-amd64 architectures.
func Call(fnAddr uintptr, args []byte) ([]byte, error) {
return nil, fmt.Errorf("verify: JIT execution requires amd64")
}
+103
View File
@@ -0,0 +1,103 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"fmt"
"sort"
)
// Block describes one basic block within a function: a maximal sequence of
// instructions with a single entry point (a label or the function start) and
// a single exit (a jump, conditional jump or RET).
type Block struct {
Offset int // byte offset within the function
Label string // label name ("" for the entry block)
}
// Blocks identifies the basic blocks of a function from its local labels.
// Each label is a potential jump target and therefore a block boundary; the
// function entry (offset 0) is always a block. The blocks are returned in
// ascending offset order.
func (k *Kernel) Blocks(name string) ([]Block, error) {
idx, ok := k.funcs[name]
if !ok {
return nil, fmt.Errorf("verify: function %q not found", name)
}
fl := k.img.Funcs[idx]
blocks := []Block{{Offset: 0, Label: "(entry)"}}
// Build a reverse map: offset → label name.
offToLabel := make(map[int]string, len(fl.Labels))
for label, off := range fl.Labels {
if off > 0 && off < fl.Size {
offToLabel[off] = label
}
}
// Collect and sort offsets.
offsets := make([]int, 0, len(offToLabel))
for off := range offToLabel {
offsets = append(offsets, off)
}
sort.Ints(offsets)
for _, off := range offsets {
blocks = append(blocks, Block{Offset: off, Label: offToLabel[off]})
}
return blocks, nil
}
// BlockCount returns the number of identified basic blocks for the function.
func (k *Kernel) BlockCount(name string) (int, error) {
blocks, err := k.Blocks(name)
if err != nil {
return 0, err
}
return len(blocks), nil
}
// PathFingerprint is the observable output of one function execution: the
// values written back into the result slots of the argument block. Two
// executions that produce the same fingerprint took observationally
// equivalent paths (though they may differ internally).
type PathFingerprint struct {
Results []uint64 // the result words from the arg block
}
// ProfilePaths runs the function with each of the given argument blocks and
// collects the distinct output fingerprints. This measures path diversity:
// how many observationally different execution paths the input corpus
// exercises. Combined with Blocks (the static block count), it gives a
// lower bound on code coverage.
func (k *Kernel) ProfilePaths(name string, argSets [][]byte, resultOffsets []int) ([]PathFingerprint, error) {
idx, ok := k.funcs[name]
if !ok {
return nil, fmt.Errorf("verify: function %q not found", name)
}
fl := k.img.Funcs[idx]
seen := map[string]bool{}
var paths []PathFingerprint
for _, args := range argSets {
if len(args) < fl.Args {
return nil, fmt.Errorf("verify: %s: arg block too small", name)
}
out, err := k.CallFunc(name, args)
if err != nil {
return nil, err
}
fp := PathFingerprint{}
key := ""
for _, off := range resultOffsets {
v := GetUint64(out, off)
fp.Results = append(fp.Results, v)
key += fmt.Sprintf("%016x", v)
}
if !seen[key] {
seen[key] = true
paths = append(paths, fp)
}
}
return paths, nil
}
+85
View File
@@ -0,0 +1,85 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"testing"
"unsafe"
)
func TestBlocks(t *testing.T) {
k := loadBasic(t)
// The "sum" function has labels: sum_done, sum_loop.
blocks, err := k.Blocks("sum")
if err != nil {
t.Fatalf("Blocks(sum): %v", err)
}
if len(blocks) < 3 {
t.Errorf("sum: expected at least 3 blocks (entry + 2 labels), got %d", len(blocks))
}
if blocks[0].Offset != 0 {
t.Errorf("first block offset = %d, want 0", blocks[0].Offset)
}
t.Logf("sum blocks: %v", blocks)
}
func TestBlockCount(t *testing.T) {
k := loadLZ4Kernel(t)
n, err := k.BlockCount("decodeBlockAVX2")
if err != nil {
t.Fatalf("BlockCount: %v", err)
}
// The decoder has many labels (dec_loop, dec_malformed, etc.).
if n < 10 {
t.Errorf("decodeBlockAVX2: expected at least 10 blocks, got %d", n)
}
t.Logf("decodeBlockAVX2: %d basic blocks", n)
}
func TestProfilePaths(t *testing.T) {
k := loadLZ4Kernel(t)
// Build a corpus of varied LZ4 blocks.
var argSets [][]byte
blocks := []struct {
src []byte
dstSize int
}{
{[]byte{0x00}, 16}, // empty
{[]byte{0x50, 'H', 'e', 'l', 'l', 'o'}, 16}, // literals only
{[]byte{0x54, 'A', 'A', 'A', 'A', 'A', 5, 0, 0x30, 'B', 'B', 'B'}, 32}, // match
{[]byte{0x14, 'X', 1, 0, 0x10, 'Y'}, 16}, // overlapping
{[]byte{0x50, 'H'}, 16}, // malformed
{[]byte{0x14, 'X', 0, 0}, 16}, // zero offset
}
for _, b := range blocks {
args := make([]byte, 64)
if len(b.src) > 0 {
PutPtr(args, 0, unsafe.Pointer(&b.src[0]))
}
PutUint64(args, 8, uint64(len(b.src)))
PutUint64(args, 16, uint64(cap(b.src)))
dst := make([]byte, b.dstSize)
if len(dst) > 0 {
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
}
PutUint64(args, 32, uint64(len(dst)))
PutUint64(args, 40, uint64(cap(dst)))
argSets = append(argSets, args)
}
// Result offsets: n+48 and code+56.
paths, err := k.ProfilePaths("decodeBlockAVX2", argSets, []int{48, 56})
if err != nil {
t.Fatalf("ProfilePaths: %v", err)
}
// We expect at least 3 distinct paths: success (various n), malformed, zero offset.
if len(paths) < 3 {
t.Errorf("expected at least 3 distinct paths, got %d", len(paths))
}
t.Logf("decodeBlockAVX2: %d distinct output paths from %d inputs", len(paths), len(argSets))
}
+295
View File
@@ -0,0 +1,295 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"math/rand"
"testing"
"unsafe"
)
// decodeBlockGo is a minimal portable LZ4 block decoder used as the
// differential-testing oracle. It mirrors the contract of
// go-lz4's decodeBlockGo: (bytesWritten, code) where code is
// 0 = ok, 1 = malformed, 2 = zero offset.
func decodeBlockGo(src, dst []byte) (int, int) {
if len(src) == 0 {
return 0, 1
}
si, di := 0, 0
for {
if si >= len(src) {
return 0, 1 // truncated: no token
}
token := int(src[si])
si++
// Literals.
lLen := token >> 4
if lLen == 15 {
for {
if si >= len(src) {
return 0, 1
}
b := int(src[si])
si++
lLen += b
if b != 255 {
break
}
}
}
if si+lLen > len(src) {
return 0, 1 // truncated literals
}
if di+lLen > len(dst) {
return 0, 1 // destination overflow
}
copy(dst[di:di+lLen], src[si:si+lLen])
di += lLen
si += lLen
// End of block.
if si >= len(src) {
return di, 0
}
// Match offset.
if si+2 > len(src) {
return 0, 1
}
offset := int(src[si]) | int(src[si+1])<<8
si += 2
if offset == 0 {
return 0, 2
}
// Match length.
mLen := token & 15
if mLen == 15 {
for {
if si >= len(src) {
return 0, 1
}
b := int(src[si])
si++
mLen += b
if b != 255 {
break
}
}
}
mLen += 4
// Copy match (overlapping-safe).
if di-offset < 0 {
return 0, 1 // offset reaches before dst start
}
if di+mLen > len(dst) {
return 0, 1 // destination overflow
}
for i := 0; i < mLen; i++ {
dst[di+i] = dst[di-offset+i]
}
di += mLen
}
}
// genLZ4Block generates a random valid LZ4 block that decompresses into
// approximately wantSize bytes. The block is always well-formed (ends with
// a literals-only sequence).
func genLZ4Block(rng *rand.Rand, wantSize int) []byte {
var block []byte
produced := 0
for produced < wantSize {
remaining := wantSize - produced
// Decide: emit a literals+match sequence or the final literals.
if remaining <= 8 || rng.Intn(4) == 0 {
// Final literals-only sequence.
lLen := remaining
if lLen > 60 {
lLen = 1 + rng.Intn(60)
}
block = appendToken(block, lLen, 0)
for i := 0; i < lLen; i++ {
block = append(block, byte(rng.Intn(256)))
}
produced += lLen
break
}
// Literals + match.
lLen := rng.Intn(min(16, remaining))
if produced+lLen == 0 {
lLen = 1 // must have at least 1 literal before the first match
}
mLenRaw := rng.Intn(12) // match length = mLenRaw + 4
mLen := mLenRaw + 4
if produced+mLen > remaining {
mLen = remaining - produced
if mLen < 4 {
// Not enough room for a match; emit final literals.
lLen = remaining
block = appendToken(block, lLen, 0)
for i := 0; i < lLen; i++ {
block = append(block, byte(rng.Intn(256)))
}
break
}
mLenRaw = mLen - 4
}
block = appendToken(block, lLen, mLenRaw)
for i := 0; i < lLen; i++ {
block = append(block, byte(rng.Intn(256)))
}
produced += lLen
// Offset: must be <= produced (can't reference before start).
maxOff := produced
if maxOff > 65535 {
maxOff = 65535
}
offset := 1 + rng.Intn(maxOff)
block = append(block, byte(offset), byte(offset>>8))
produced += mLen
}
return block
}
// appendToken appends a token (and extension bytes if needed) for the given
// literal and match lengths.
func appendToken(block []byte, lLen, mLenRaw int) []byte {
lit4 := lLen
if lit4 > 15 {
lit4 = 15
}
ml4 := mLenRaw
if ml4 > 15 {
ml4 = 15
}
block = append(block, byte(lit4<<4|ml4))
// Literal extension bytes.
rem := lLen - 15
for rem >= 255 {
block = append(block, 255)
rem -= 255
}
if lLen >= 15 {
block = append(block, byte(rem))
}
// Match extension bytes.
rem = mLenRaw - 15
for rem >= 255 {
block = append(block, 255)
rem -= 255
}
if mLenRaw >= 15 {
block = append(block, byte(rem))
}
return block
}
func min(a, b int) int {
if a < b {
return a
}
return b
}
// TestDifferentialLZ4Fuzz drives the JIT-assembled decodeBlockAVX2 with
// random valid LZ4 blocks and compares the output bit-for-bit against the
// portable Go reference.
func TestDifferentialLZ4Fuzz(t *testing.T) {
k := loadLZ4Kernel(t)
const iterations = 5000
rng := rand.New(rand.NewSource(42))
for i := 0; i < iterations; i++ {
wantSize := 1 + rng.Intn(4096)
src := genLZ4Block(rng, wantSize)
dstSize := wantSize + 64 // generous destination
// Go reference.
goDst := make([]byte, dstSize)
goN, goCode := decodeBlockGo(src, goDst)
// JIT kernel.
jitDst := make([]byte, dstSize)
jitN, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
if jitCode != goCode {
t.Fatalf("iter %d: code mismatch: JIT=%d, Go=%d (src len=%d)",
i, jitCode, goCode, len(src))
}
if jitCode != 0 {
continue // both agree it's malformed/zero-offset
}
if jitN != goN {
t.Fatalf("iter %d: n mismatch: JIT=%d, Go=%d (src len=%d)",
i, jitN, goN, len(src))
}
if !bytes.Equal(jitDst[:jitN], goDst[:goN]) {
t.Fatalf("iter %d: output mismatch (n=%d, src len=%d)", i, jitN, len(src))
}
}
}
// TestDifferentialLZ4Hostile drives the kernel with random garbage to check
// that error codes agree with the Go reference (no crashes, same classification).
func TestDifferentialLZ4Hostile(t *testing.T) {
k := loadLZ4Kernel(t)
const iterations = 2000
rng := rand.New(rand.NewSource(99))
for i := 0; i < iterations; i++ {
srcLen := rng.Intn(128)
src := make([]byte, srcLen)
rng.Read(src)
dstSize := rng.Intn(512)
dst := make([]byte, dstSize)
// Go reference.
goDst := make([]byte, dstSize)
copy(goDst, dst)
_, goCode := decodeBlockGo(src, goDst)
// JIT kernel.
jitDst := make([]byte, dstSize)
copy(jitDst, dst)
_, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
if jitCode != goCode {
t.Fatalf("iter %d: hostile code mismatch: JIT=%d, Go=%d (srcLen=%d, dstSize=%d)",
i, jitCode, goCode, srcLen, dstSize)
}
}
}
// callDecodeBlockAVX2Raw is like callDecodeBlockAVX2 but accepts explicit
// dst size (for hostile tests where dst may be smaller than the output).
func callDecodeBlockAVX2Raw(t *testing.T, k *Kernel, src, dst []byte) (int, int) {
t.Helper()
args := make([]byte, 64)
if len(src) > 0 {
PutPtr(args, 0, unsafe.Pointer(&src[0]))
}
PutUint64(args, 8, uint64(len(src)))
PutUint64(args, 16, uint64(cap(src)))
if len(dst) > 0 {
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
}
PutUint64(args, 32, uint64(len(dst)))
PutUint64(args, 40, uint64(cap(dst)))
out, err := k.CallFunc("decodeBlockAVX2", args)
if err != nil {
t.Fatalf("CallFunc(decodeBlockAVX2): %v", err)
}
return int(GetUint64(out, 48)), int(GetUint64(out, 56))
}
+479
View File
@@ -0,0 +1,479 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"math/rand"
"os"
"testing"
"unsafe"
)
const flacKernelPath = "../../go-libraries/go-flac/avx2_amd64.s"
func loadFLACKernel(t *testing.T) *Kernel {
t.Helper()
if _, err := os.Stat(flacKernelPath); err != nil {
t.Skipf("sibling kernel not available: %v", err)
}
k, err := Load(flacKernelPath)
if err != nil {
t.Fatalf("Load(%s): %v", flacKernelPath, err)
}
t.Cleanup(k.Close)
return k
}
// --- Portable Go references (from go-flac/simd.go) ---
func decodeMono16Go(src []byte, dst []int32) {
for i := 0; i < len(dst); i++ {
dst[i] = int32(int16(uint16(src[2*i]) | uint16(src[2*i+1])<<8))
}
}
func pack16Go(dst []byte, src []int32) {
for i, v := range src {
dst[2*i] = byte(v)
dst[2*i+1] = byte(v >> 8)
}
}
func decorrelateLeftSideGo(left, right, out []int32) {
for i := range left {
l := left[i]
out[2*i] = l
out[2*i+1] = l - right[i]
}
}
func decorrelateSideRightGo(left, right, out []int32) {
for i := range left {
side := left[i]
rch := right[i]
out[2*i] = rch + side
out[2*i+1] = rch
}
}
func decorrelateMidSideGo(left, right, out []int32) {
for i := range left {
mid := left[i]
side := right[i]
mid2 := mid<<1 | (side & 1)
out[2*i] = (mid2 + side) >> 1
out[2*i+1] = (mid2 - side) >> 1
}
}
func decorrelateInterleaveGo(left, right, out []int32) {
for i := range left {
out[2*i] = left[i]
out[2*i+1] = right[i]
}
}
func analyzeO1RangeGo(swin []int32, dstP []uint32, hist *[32]uint16) (partSum uint64, overflow bool) {
swin = swin[:len(dstP)+1]
for j := 0; j+1 < len(swin); j++ {
r := swin[j+1] - swin[j]
if r == -2147483648 { // math.MinInt32
overflow = true
}
f := uint32(r<<1) ^ uint32(r>>31)
dstP[j] = f
partSum += uint64(f)
bl := 0
for v := f; v > 0; v >>= 1 {
bl++
}
if bl > 31 {
bl = 31
}
hist[bl]++
}
return
}
func analyzeO2RangeGo(swin []int32, dstP []uint32, hist *[32]uint16) (partSum uint64, overflow bool) {
swin = swin[:len(dstP)+2]
for j := 0; j+2 < len(swin); j++ {
r := swin[j+2] - 2*swin[j+1] + swin[j]
if r == -2147483648 {
overflow = true
}
f := uint32(r<<1) ^ uint32(r>>31)
dstP[j] = f
partSum += uint64(f)
bl := 0
for v := f; v > 0; v >>= 1 {
bl++
}
if bl > 31 {
bl = 31
}
hist[bl]++
}
return
}
func analyzeResRangeGo(swin []int32, dstP []uint32, hist *[32]uint16) (partSum uint64, overflow bool) {
for j := 0; j < len(swin); j++ {
r := swin[j]
if r == -2147483648 {
overflow = true
}
f := uint32(r<<1) ^ uint32(r>>31)
dstP[j] = f
partSum += uint64(f)
bl := 0
for v := f; v > 0; v >>= 1 {
bl++
}
if bl > 31 {
bl = 31
}
hist[bl]++
}
return
}
func decodeMono24Go(src []byte, dst []int32) {
for i := 0; i < len(dst); i++ {
off := 3 * i
u := uint32(src[off]) | uint32(src[off+1])<<8 | uint32(src[off+2])<<16
dst[i] = int32(u<<8) >> 8
}
}
// --- Differential tests ---
func TestFLACDecodeMono16(t *testing.T) {
k := loadFLACKernel(t)
rng := rand.New(rand.NewSource(7))
for iter := 0; iter < 500; iter++ {
n := rng.Intn(256)
src := make([]byte, 2*n)
rng.Read(src)
goDst := make([]int32, n)
decodeMono16Go(src, goDst)
jitDst := make([]int32, n)
args := make([]byte, 48)
if len(src) > 0 {
PutPtr(args, 0, unsafe.Pointer(&src[0]))
}
PutUint64(args, 8, uint64(len(src)))
PutUint64(args, 16, uint64(cap(src)))
if n > 0 {
PutPtr(args, 24, unsafe.Pointer(&jitDst[0]))
}
PutUint64(args, 32, uint64(n))
PutUint64(args, 40, uint64(cap(jitDst)))
_, err := k.CallFunc("decodeMono16AVX2", args)
if err != nil {
t.Fatalf("iter %d: %v", iter, err)
}
for i := range goDst {
if jitDst[i] != goDst[i] {
t.Fatalf("iter %d: mismatch at [%d]: JIT=%d Go=%d", iter, i, jitDst[i], goDst[i])
}
}
}
}
func TestFLACPack16(t *testing.T) {
k := loadFLACKernel(t)
rng := rand.New(rand.NewSource(13))
for iter := 0; iter < 500; iter++ {
n := rng.Intn(256)
src := make([]int32, n)
for i := range src {
src[i] = int32(rng.Intn(65536) - 32768)
}
goDst := make([]byte, 2*n)
pack16Go(goDst, src)
jitDst := make([]byte, 2*n)
args := make([]byte, 48)
if len(jitDst) > 0 {
PutPtr(args, 0, unsafe.Pointer(&jitDst[0]))
}
PutUint64(args, 8, uint64(len(jitDst)))
PutUint64(args, 16, uint64(cap(jitDst)))
if n > 0 {
PutPtr(args, 24, unsafe.Pointer(&src[0]))
}
PutUint64(args, 32, uint64(n))
PutUint64(args, 40, uint64(cap(src)))
_, err := k.CallFunc("pack16AVX2", args)
if err != nil {
t.Fatalf("iter %d: %v", iter, err)
}
if !bytes.Equal(jitDst, goDst) {
t.Fatalf("iter %d: output mismatch (n=%d)", iter, n)
}
}
}
func TestFLACDecorrelate(t *testing.T) {
k := loadFLACKernel(t)
rng := rand.New(rand.NewSource(21))
kernels := []struct {
name string
ref func(left, right, out []int32)
}{
{"decorrelateLeftSideAVX2", decorrelateLeftSideGo},
{"decorrelateSideRightAVX2", decorrelateSideRightGo},
{"decorrelateMidSideAVX2", decorrelateMidSideGo},
{"decorrelateInterleaveAVX2", decorrelateInterleaveGo},
}
for _, kk := range kernels {
t.Run(kk.name, func(t *testing.T) {
for iter := 0; iter < 200; iter++ {
n := rng.Intn(128)
left := make([]int32, n)
right := make([]int32, n)
for i := range left {
left[i] = int32(rng.Intn(1<<24) - 1<<23)
right[i] = int32(rng.Intn(1<<24) - 1<<23)
}
goOut := make([]int32, 2*n)
kk.ref(left, right, goOut)
jitOut := make([]int32, 2*n)
args := make([]byte, 72)
if n > 0 {
PutPtr(args, 0, unsafe.Pointer(&left[0]))
PutPtr(args, 24, unsafe.Pointer(&right[0]))
PutPtr(args, 48, unsafe.Pointer(&jitOut[0]))
}
PutUint64(args, 8, uint64(n))
PutUint64(args, 16, uint64(cap(left)))
PutUint64(args, 32, uint64(n))
PutUint64(args, 40, uint64(cap(right)))
PutUint64(args, 56, uint64(2*n))
PutUint64(args, 64, uint64(cap(jitOut)))
_, err := k.CallFunc(kk.name, args)
if err != nil {
t.Fatalf("iter %d: %v", iter, err)
}
for i := range goOut {
if jitOut[i] != goOut[i] {
t.Fatalf("iter %d: mismatch at [%d]: JIT=%d Go=%d", iter, i, jitOut[i], goOut[i])
}
}
}
})
}
}
func TestFLACAnalyzeO1Range(t *testing.T) {
k := loadFLACKernel(t)
rng := rand.New(rand.NewSource(33))
for iter := 0; iter < 300; iter++ {
n := 1 + rng.Intn(128) // partition size
swin := make([]int32, n+1)
for i := range swin {
swin[i] = int32(rng.Intn(1<<20) - 1<<19)
}
goDstP := make([]uint32, n)
var goHist [32]uint16
goSum, goOvf := analyzeO1RangeGo(swin, goDstP, &goHist)
jitDstP := make([]uint32, n)
var jitHist [32]uint16
args := make([]byte, 72) // 65 rounded up
PutPtr(args, 0, unsafe.Pointer(&swin[0]))
PutUint64(args, 8, uint64(len(swin)))
PutUint64(args, 16, uint64(cap(swin)))
PutPtr(args, 24, unsafe.Pointer(&jitDstP[0]))
PutUint64(args, 32, uint64(n))
PutUint64(args, 40, uint64(cap(jitDstP)))
PutPtr(args, 48, unsafe.Pointer(&jitHist[0]))
out, err := k.CallFunc("analyzeO1RangeAVX2", args)
if err != nil {
t.Fatalf("iter %d: %v", iter, err)
}
jitSum := GetUint64(out, 56)
jitOvf := out[64] != 0
if jitSum != goSum {
t.Fatalf("iter %d: partSum mismatch: JIT=%d Go=%d", iter, jitSum, goSum)
}
if jitOvf != goOvf {
t.Fatalf("iter %d: overflow mismatch: JIT=%v Go=%v", iter, jitOvf, goOvf)
}
for i := range goDstP {
if jitDstP[i] != goDstP[i] {
t.Fatalf("iter %d: dstP[%d] mismatch: JIT=%d Go=%d", iter, i, jitDstP[i], goDstP[i])
}
}
if jitHist != goHist {
t.Fatalf("iter %d: hist mismatch: JIT=%v Go=%v", iter, jitHist, goHist)
}
}
}
func TestFLACFastStereoSums(t *testing.T) {
k := loadFLACKernel(t)
rng := rand.New(rand.NewSource(44))
for iter := 0; iter < 300; iter++ {
n := 1 + rng.Intn(256)
left := make([]int32, n)
right := make([]int32, n)
for i := range left {
left[i] = int32(rng.Intn(1<<24) - 1<<23)
right[i] = int32(rng.Intn(1<<24) - 1<<23)
}
// Go reference: compute the four sums.
var goSums [4]uint64
for i := 0; i < n; i++ {
l := left[i]
r := right[i]
side := l - r
mid := (l + r) >> 1
goSums[0] += foldAbs(l) + foldAbs(r)
goSums[1] += foldAbs(l) + foldAbs(side)
goSums[2] += foldAbs(side) + foldAbs(r)
goSums[3] += foldAbs(mid) + foldAbs(side)
}
var jitSums [4]uint64
args := make([]byte, 56)
PutPtr(args, 0, unsafe.Pointer(&left[0]))
PutUint64(args, 8, uint64(n))
PutUint64(args, 16, uint64(cap(left)))
PutPtr(args, 24, unsafe.Pointer(&right[0]))
PutUint64(args, 32, uint64(n))
PutUint64(args, 40, uint64(cap(right)))
PutPtr(args, 48, unsafe.Pointer(&jitSums[0]))
_, err := k.CallFunc("fastStereoSumsAVX2", args)
if err != nil {
t.Fatalf("iter %d: %v", iter, err)
}
if jitSums != goSums {
t.Fatalf("iter %d: sums mismatch:\n JIT=%v\n Go =%v", iter, jitSums, goSums)
}
}
}
func foldAbs(v int32) uint64 {
return uint64(uint32(v<<1) ^ uint32(v>>31))
}
// runAnalyzeTest is the shared harness for the analyzeO*Range family.
func runAnalyzeTest(t *testing.T, k *Kernel, name string, order int, ref func([]int32, []uint32, *[32]uint16) (uint64, bool)) {
t.Helper()
rng := rand.New(rand.NewSource(int64(order)*100 + 7))
for iter := 0; iter < 200; iter++ {
n := 1 + rng.Intn(128)
swin := make([]int32, n+order)
for i := range swin {
swin[i] = int32(rng.Intn(1<<20) - 1<<19)
}
goDstP := make([]uint32, n)
var goHist [32]uint16
goSum, goOvf := ref(swin, goDstP, &goHist)
jitDstP := make([]uint32, n)
var jitHist [32]uint16
args := make([]byte, 72)
PutPtr(args, 0, unsafe.Pointer(&swin[0]))
PutUint64(args, 8, uint64(len(swin)))
PutUint64(args, 16, uint64(cap(swin)))
PutPtr(args, 24, unsafe.Pointer(&jitDstP[0]))
PutUint64(args, 32, uint64(n))
PutUint64(args, 40, uint64(cap(jitDstP)))
PutPtr(args, 48, unsafe.Pointer(&jitHist[0]))
out, err := k.CallFunc(name, args)
if err != nil {
t.Fatalf("iter %d: %v", iter, err)
}
jitSum := GetUint64(out, 56)
jitOvf := out[64] != 0
if jitSum != goSum {
t.Fatalf("iter %d: partSum: JIT=%d Go=%d", iter, jitSum, goSum)
}
if jitOvf != goOvf {
t.Fatalf("iter %d: overflow: JIT=%v Go=%v", iter, jitOvf, goOvf)
}
for i := range goDstP {
if jitDstP[i] != goDstP[i] {
t.Fatalf("iter %d: dstP[%d]: JIT=%d Go=%d", iter, i, jitDstP[i], goDstP[i])
}
}
if jitHist != goHist {
t.Fatalf("iter %d: hist mismatch", iter)
}
}
}
func TestFLACAnalyzeO2Range(t *testing.T) {
k := loadFLACKernel(t)
runAnalyzeTest(t, k, "analyzeO2RangeAVX2", 2, analyzeO2RangeGo)
}
func TestFLACAnalyzeResRange(t *testing.T) {
k := loadFLACKernel(t)
// analyzeResRange has order 0: swin IS the residual (no prediction).
runAnalyzeTest(t, k, "analyzeResRangeAVX2", 0, analyzeResRangeGo)
}
func TestFLACDecodeMono24(t *testing.T) {
k := loadFLACKernel(t)
rng := rand.New(rand.NewSource(55))
for iter := 0; iter < 500; iter++ {
n := rng.Intn(256)
src := make([]byte, 3*n)
rng.Read(src)
goDst := make([]int32, n)
decodeMono24Go(src, goDst)
jitDst := make([]int32, n)
args := make([]byte, 48)
if len(src) > 0 {
PutPtr(args, 0, unsafe.Pointer(&src[0]))
}
PutUint64(args, 8, uint64(len(src)))
PutUint64(args, 16, uint64(cap(src)))
if n > 0 {
PutPtr(args, 24, unsafe.Pointer(&jitDst[0]))
}
PutUint64(args, 32, uint64(n))
PutUint64(args, 40, uint64(cap(jitDst)))
_, err := k.CallFunc("decodeMono24AVX2", args)
if err != nil {
t.Fatalf("iter %d: %v", iter, err)
}
for i := range goDst {
if jitDst[i] != goDst[i] {
t.Fatalf("iter %d: dst[%d]: JIT=%d Go=%d", iter, i, jitDst[i], goDst[i])
}
}
}
}
+88
View File
@@ -0,0 +1,88 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package verify provides the dynamic-analysis substrate for gasm: it
// JIT-assembles Plan 9 amd64 kernels into executable memory and calls them
// directly, enabling differential testing against portable Go references,
// runtime ABI checks and basic-block coverage profiling.
//
// The execution model is pure Go (stdlib only): machine code is mapped with
// syscall.Mmap and invoked through an assembly trampoline that switches to a
// prepared ABI0 stack. No cgo, no external toolchain.
package verify
import (
"encoding/binary"
"fmt"
"syscall"
"unsafe"
)
// Executable maps a copy of code into a read-execute memory region suitable
// for direct invocation. The mapping is anonymous and private; the original
// slice is not retained. Call Unmap to release the region.
type Executable struct {
addr uintptr // base address of the mapping
size int
mem []byte // the mmap'd slice (for Unmap)
}
// Map copies code into a freshly allocated RX region and returns it.
// The mapping is PROT_READ|PROT_EXEC; writes are not permitted after the
// copy, matching W^X policy.
func Map(code []byte) (*Executable, error) {
size := len(code)
if size == 0 {
return nil, fmt.Errorf("verify: cannot map zero-length code")
}
// Round up to the page size.
const pageSize = 4096
mapSize := (size + pageSize - 1) &^ (pageSize - 1)
mem, err := syscall.Mmap(-1, 0, mapSize,
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, fmt.Errorf("verify: mmap: %w", err)
}
copy(mem, code)
// Remove write permission (W^X).
if err := syscall.Mprotect(mem, syscall.PROT_READ|syscall.PROT_EXEC); err != nil {
syscall.Munmap(mem)
return nil, fmt.Errorf("verify: mprotect: %w", err)
}
return &Executable{
addr: uintptr(unsafe.Pointer(&mem[0])),
size: size,
mem: mem,
}, nil
}
// Unmap releases the executable region.
func (e *Executable) Unmap() {
if e.mem != nil {
syscall.Munmap(e.mem)
e.mem = nil
}
}
// FuncAddr returns the absolute address of a function at the given offset
// within the mapped image.
func (e *Executable) FuncAddr(offset int) uintptr {
return e.addr + uintptr(offset)
}
// PutUint64 writes v into buf at byte offset off (little-endian).
func PutUint64(buf []byte, off int, v uint64) {
binary.LittleEndian.PutUint64(buf[off:off+8], v)
}
// GetUint64 reads a little-endian uint64 from buf at byte offset off.
func GetUint64(buf []byte, off int) uint64 {
return binary.LittleEndian.Uint64(buf[off : off+8])
}
// PutPtr writes a pointer value into buf at byte offset off.
func PutPtr(buf []byte, off int, p unsafe.Pointer) {
binary.LittleEndian.PutUint64(buf[off:off+8], uint64(uintptr(p)))
}
+215
View File
@@ -0,0 +1,215 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"testing"
"unsafe"
)
func loadBasic(t *testing.T) *Kernel {
t.Helper()
k, err := Load("../testdata/verify/basic_amd64.s")
if err != nil {
t.Fatalf("Load: %v", err)
}
t.Cleanup(k.Close)
return k
}
func TestJITAdd(t *testing.T) {
k := loadBasic(t)
tests := []struct {
a, b, want int64
}{
{0, 0, 0},
{1, 2, 3},
{-1, 1, 0},
{1 << 62, 1 << 62, -9223372036854775808}, // overflow wraps (MinInt64)
{-100, -200, -300},
}
for _, tt := range tests {
args := make([]byte, 24)
PutUint64(args, 0, uint64(tt.a))
PutUint64(args, 8, uint64(tt.b))
out, err := k.CallFunc("add", args)
if err != nil {
t.Fatalf("CallFunc(add, %d, %d): %v", tt.a, tt.b, err)
}
got := int64(GetUint64(out, 16))
if got != tt.want {
t.Errorf("add(%d, %d) = %d, want %d", tt.a, tt.b, got, tt.want)
}
}
}
func TestJITSum(t *testing.T) {
k := loadBasic(t)
tests := []struct {
data []int64
want int64
}{
{nil, 0},
{[]int64{1}, 1},
{[]int64{1, 2, 3, 4, 5}, 15},
{[]int64{-10, 20, -30, 40}, 20},
}
for _, tt := range tests {
args := make([]byte, 32)
if len(tt.data) > 0 {
PutPtr(args, 0, unsafe.Pointer(&tt.data[0]))
}
PutUint64(args, 8, uint64(len(tt.data)))
PutUint64(args, 16, uint64(cap(tt.data)))
out, err := k.CallFunc("sum", args)
if err != nil {
t.Fatalf("CallFunc(sum, %v): %v", tt.data, err)
}
got := int64(GetUint64(out, 24))
if got != tt.want {
t.Errorf("sum(%v) = %d, want %d", tt.data, got, tt.want)
}
}
}
func TestJITWideCopy(t *testing.T) {
k := loadBasic(t)
tests := []struct {
name string
n int
}{
{"empty", 0},
{"tiny", 7},
{"exact32", 32},
{"overlap_range", 48},
{"exact64", 64},
{"unaligned", 45},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
src := make([]byte, tt.n)
for i := range src {
src[i] = byte(i * 7)
}
dst := make([]byte, tt.n)
args := make([]byte, 48)
if tt.n > 0 {
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
PutPtr(args, 24, unsafe.Pointer(&src[0]))
}
PutUint64(args, 8, uint64(tt.n)) // dst_len
PutUint64(args, 16, uint64(tt.n)) // dst_cap
PutUint64(args, 32, uint64(tt.n)) // src_len
PutUint64(args, 40, uint64(tt.n)) // src_cap
_, err := k.CallFunc("wideCopy", args)
if err != nil {
t.Fatalf("CallFunc(wideCopy): %v", err)
}
if !bytes.Equal(dst, src) {
t.Errorf("wideCopy: dst ≠ src\n got %x\n want %x", dst, src)
}
})
}
}
func TestKernelFuncNames(t *testing.T) {
k := loadBasic(t)
names := k.FuncNames()
want := []string{"add", "sum", "wideCopy"}
if len(names) != len(want) {
t.Fatalf("FuncNames() = %v, want %v", names, want)
}
for i, n := range names {
if n != want[i] {
t.Errorf("FuncNames()[%d] = %q, want %q", i, n, want[i])
}
}
}
func TestKernelFuncNotFound(t *testing.T) {
k := loadBasic(t)
_, err := k.CallFunc("nonexistent", make([]byte, 8))
if err == nil {
t.Fatal("expected error for nonexistent function")
}
}
func TestKernelArgTooSmall(t *testing.T) {
k := loadBasic(t)
_, err := k.CallFunc("add", make([]byte, 8)) // needs 24
if err == nil {
t.Fatal("expected error for too-small arg block")
}
}
func TestMapZeroLength(t *testing.T) {
_, err := Map(nil)
if err == nil {
t.Fatal("expected error for zero-length code")
}
}
func TestLoadSourceError(t *testing.T) {
_, err := LoadSource("bad.s", "TEXT ·f(SB), NOSPLIT\n\tBADINSTRUCTION\n")
// The parser may or may not error on unknown instructions (it's
// error-tolerant), but the assembler will reject it.
if err == nil {
t.Log("LoadSource succeeded unexpectedly (parser is error-tolerant)")
}
}
func TestLoadSourceParseError(t *testing.T) {
// A completely invalid file that the parser rejects.
_, err := LoadSource("empty.s", "")
if err != nil {
t.Logf("expected: %v", err)
}
}
func TestFuncLookup(t *testing.T) {
k := loadBasic(t)
fl, err := k.Func("add")
if err != nil {
t.Fatalf("Func(add): %v", err)
}
if fl.Name != "add" {
t.Errorf("Func(add).Name = %q, want %q", fl.Name, "add")
}
if fl.Args != 24 {
t.Errorf("Func(add).Args = %d, want 24", fl.Args)
}
_, err = k.Func("nonexistent")
if err == nil {
t.Fatal("expected error for nonexistent function")
}
}
func TestABIReportString(t *testing.T) {
r := ABIReport{}
if r.String() != "ABI clean" {
t.Errorf("clean report = %q", r.String())
}
r.BPClobbered = true
if r.OK() {
t.Error("expected not OK with BP clobbered")
}
s := r.String()
if s == "ABI clean" {
t.Error("expected violation string, got clean")
}
r.R14Clobbered = true
r.RedZoneHit = true
s = r.String()
if s == "ABI clean" {
t.Error("expected violation string for all flags")
}
}
+160
View File
@@ -0,0 +1,160 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"os"
"testing"
"unsafe"
)
// lz4KernelPath is the sibling repository's AVX2 kernel, used for
// integration testing. The test is skipped when the file is absent
// (e.g. in CI without the sibling checkout).
const lz4KernelPath = "../../go-libraries/go-lz4/avx2_amd64.s"
func loadLZ4Kernel(t *testing.T) *Kernel {
t.Helper()
if _, err := os.Stat(lz4KernelPath); err != nil {
t.Skipf("sibling kernel not available: %v", err)
}
k, err := Load(lz4KernelPath)
if err != nil {
t.Fatalf("Load(%s): %v", lz4KernelPath, err)
}
t.Cleanup(k.Close)
return k
}
// callDecodeBlockAVX2 invokes the JIT-assembled decodeBlockAVX2 with the
// given src and dst buffers, returning (n, code).
func callDecodeBlockAVX2(t *testing.T, k *Kernel, src, dst []byte) (int, int) {
t.Helper()
args := make([]byte, 64)
if len(src) > 0 {
PutPtr(args, 0, unsafe.Pointer(&src[0]))
}
PutUint64(args, 8, uint64(len(src)))
PutUint64(args, 16, uint64(cap(src)))
if len(dst) > 0 {
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
}
PutUint64(args, 32, uint64(len(dst)))
PutUint64(args, 40, uint64(cap(dst)))
out, err := k.CallFunc("decodeBlockAVX2", args)
if err != nil {
t.Fatalf("CallFunc(decodeBlockAVX2): %v", err)
}
return int(GetUint64(out, 48)), int(GetUint64(out, 56))
}
func TestLZ4DecodeKnownAnswers(t *testing.T) {
k := loadLZ4Kernel(t)
tests := []struct {
name string
src []byte
dstSize int
wantDst []byte
wantN int
wantCode int
}{
{
name: "literals_only",
src: []byte{0x50, 'H', 'e', 'l', 'l', 'o'},
dstSize: 16,
wantDst: []byte("Hello"),
wantN: 5,
wantCode: 0,
},
{
name: "literals_and_match",
src: []byte{0x54, 'A', 'A', 'A', 'A', 'A', 0x05, 0x00, 0x30, 'B', 'B', 'B'},
dstSize: 32,
wantDst: []byte("AAAAAAAAAAAAABBB"),
wantN: 16,
wantCode: 0,
},
{
name: "overlapping_match",
// 1 literal 'X', then match offset=1 length=4+4=8 → "XXXXXXXXX",
// then final 1 literal 'Y'.
src: []byte{0x14, 'X', 0x01, 0x00, 0x10, 'Y'},
dstSize: 16,
wantDst: []byte("XXXXXXXXXY"),
wantN: 10,
wantCode: 0,
},
{
name: "malformed_truncated",
src: []byte{0x50, 'H', 'e'}, // claims 5 literals, has 2
dstSize: 16,
wantN: 0,
wantCode: 1,
},
{
name: "zero_offset",
src: []byte{0x14, 'X', 0x00, 0x00},
dstSize: 16,
wantN: 0,
wantCode: 2,
},
{
name: "empty_token",
src: []byte{0x00}, // 0 literals, end of block
dstSize: 16,
wantDst: nil,
wantN: 0,
wantCode: 0,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
dst := make([]byte, tt.dstSize)
n, code := callDecodeBlockAVX2(t, k, tt.src, dst)
if n != tt.wantN || code != tt.wantCode {
t.Fatalf("decodeBlockAVX2: got (n=%d, code=%d), want (n=%d, code=%d)",
n, code, tt.wantN, tt.wantCode)
}
if tt.wantCode == 0 && tt.wantDst != nil {
if !bytes.Equal(dst[:n], tt.wantDst) {
t.Errorf("output mismatch:\n got %q\n want %q", dst[:n], tt.wantDst)
}
}
})
}
}
func TestLZ4WideCopyAVX2(t *testing.T) {
k := loadLZ4Kernel(t)
sizes := []int{0, 1, 15, 16, 31, 32, 33, 63, 64, 100, 256, 1024}
for _, n := range sizes {
src := make([]byte, n)
for i := range src {
src[i] = byte(i*13 + 7)
}
dst := make([]byte, n)
args := make([]byte, 48)
if n > 0 {
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
PutPtr(args, 24, unsafe.Pointer(&src[0]))
}
PutUint64(args, 8, uint64(n))
PutUint64(args, 16, uint64(n))
PutUint64(args, 32, uint64(n))
PutUint64(args, 40, uint64(n))
_, err := k.CallFunc("wideCopyAVX2", args)
if err != nil {
t.Fatalf("wideCopyAVX2(n=%d): %v", n, err)
}
if !bytes.Equal(dst, src) {
t.Errorf("wideCopyAVX2(n=%d): output mismatch", n)
}
}
}
+30
View File
@@ -0,0 +1,30 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// ABI0 JIT trampoline. enterJIT switches from the Go stack to a prepared
// stack and jumps to the assembled function; when the function RETs, control
// lands in leaveJIT, which restores the Go stack and returns to the Go caller.
//
// The prepared stack must begin with the address of leaveJIT (the return
// address the JIT function will pop), followed by the function's ABI0
// argument area.
//
// Single-threaded: savedSP is a package global, so only one JIT call may be
// in flight at a time. gasm verify runs sequentially.
// func enterJIT(fn uintptr, stack uintptr)
// Switches to the prepared stack and jumps to fn. Does not return normally;
// the JIT function's RET transfers control to leaveJIT.
TEXT ·enterJIT(SB), NOSPLIT, $0-16
MOVQ fn+0(FP), AX // target function address (before SP switch)
MOVQ SP, ·savedSP(SB) // preserve the Go stack pointer
MOVQ stack+8(FP), SP // switch to the prepared stack
JMP AX
// func leaveJIT()
// Restores the Go stack pointer and returns to enterJIT's caller.
TEXT ·leaveJIT(SB), NOSPLIT, $0-0
MOVQ ·savedSP(SB), SP
RET
+122
View File
@@ -0,0 +1,122 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"fmt"
"os"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// Kernel is a JIT-loaded assembly image ready for direct invocation.
// It wraps an executable memory mapping and the function layout metadata
// needed to marshal ABI0 calls.
type Kernel struct {
exec *Executable
img *asm.Image
funcs map[string]int // function name → index into img.Funcs
}
// Load parses, assembles and maps a .s file into executable memory.
// The returned Kernel is ready for Call. The caller must call Close to
// release the mapping.
func Load(path string) (*Kernel, error) {
src, err := os.ReadFile(path)
if err != nil {
return nil, fmt.Errorf("verify: %w", err)
}
return LoadSource(path, string(src))
}
// LoadSource parses, assembles and maps assembly source into executable memory.
func LoadSource(filename, src string) (*Kernel, error) {
file, errs := parser.Parse(filename, src)
if len(errs) > 0 {
return nil, fmt.Errorf("verify: parse %s: %v", filename, errs[0])
}
return LoadAST(file)
}
// LoadAST assembles a parsed AST file and maps the result into executable
// memory.
func LoadAST(file *ast.File) (*Kernel, error) {
img, err := asm.AssembleFile(file)
if err != nil {
return nil, fmt.Errorf("verify: assemble: %w", err)
}
if len(img.Externals) > 0 {
return nil, fmt.Errorf("verify: unresolved external symbols: %v", img.Externals)
}
code := img.Bytes()
exec, err := Map(code)
if err != nil {
return nil, err
}
funcs := make(map[string]int, len(img.Funcs))
for i, f := range img.Funcs {
funcs[f.Name] = i
}
return &Kernel{exec: exec, img: img, funcs: funcs}, nil
}
// Func returns the layout metadata for the named function.
func (k *Kernel) Func(name string) (asm.FuncLayout, error) {
idx, ok := k.funcs[name]
if !ok {
return asm.FuncLayout{}, fmt.Errorf("verify: function %q not found", name)
}
return k.img.Funcs[idx], nil
}
// FuncNames returns the names of all functions in the kernel, in source order.
func (k *Kernel) FuncNames() []string {
names := make([]string, len(k.img.Funcs))
for i, f := range k.img.Funcs {
names[i] = f.Name
}
return names
}
// CallFunc invokes the named function with the given ABI0 argument block.
// The arg block is the raw bytes of the function's argument/result area
// (as declared by the TEXT $frame-args suffix). Returns the arg block
// after the call (with any results written back by the function).
func (k *Kernel) CallFunc(name string, args []byte) ([]byte, error) {
idx, ok := k.funcs[name]
if !ok {
return nil, fmt.Errorf("verify: function %q not found", name)
}
fl := k.img.Funcs[idx]
if len(args) < fl.Args {
return nil, fmt.Errorf("verify: %s: arg block too small: got %d, need %d", name, len(args), fl.Args)
}
fnAddr := k.exec.FuncAddr(fl.Offset)
return Call(fnAddr, args)
}
// CallFuncChecked invokes the named function with ABI sentinels and a
// red-zone canary, returning the argument block and an ABIReport that
// records any callee-saved register or red-zone violations.
func (k *Kernel) CallFuncChecked(name string, args []byte) ([]byte, ABIReport, error) {
idx, ok := k.funcs[name]
if !ok {
return nil, ABIReport{}, fmt.Errorf("verify: function %q not found", name)
}
fl := k.img.Funcs[idx]
if len(args) < fl.Args {
return nil, ABIReport{}, fmt.Errorf("verify: %s: arg block too small: got %d, need %d", name, len(args), fl.Args)
}
fnAddr := k.exec.FuncAddr(fl.Offset)
return CallChecked(fnAddr, args)
}
// Close releases the executable mapping.
func (k *Kernel) Close() {
if k.exec != nil {
k.exec.Unmap()
}
}