Compare commits

...
11 Commits
Author SHA1 Message Date
petrbalvin ee68859beb feat(asm): add the wider EVEX set and the rounding, SAE and broadcast suffixes
Assisted-by: Qwen 3.8 Max Preview
2026-07-18 15:47:59 +02:00
petrbalvin 0920edb092 feat(asm): emit GOOBJ objects that link directly with the Go toolchain
Assisted-by: Qwen 3.8 Max Preview
2026-07-17 18:57:04 +02:00
petrbalvin 900c9772b1 feat(asm): emit linkable ELF and Mach-O objects with external symbols
Assisted-by: Qwen 3.8 Max Preview
2026-07-16 20:52:20 +02:00
petrbalvin b914c0e390 feat(asm): add the EVEX floating-point and conversion set
Assisted-by: Qwen 3.8 Max Preview
2026-07-15 17:13:28 +02:00
petrbalvin 0f3146ff2c feat(asm): add EVEX masking, zeroing and the AVX-512 F/BW integer set
Assisted-by: Qwen 3.8 Max Preview
2026-07-14 21:03:26 +02:00
petrbalvin 9370f9c3ee feat(cli): standard --help and --version with per-command usage
Assisted-by: Qwen 3.8 Max Preview
2026-07-13 19:50:38 +02:00
petrbalvin e98680597d feat(fmt): go-fmt-style recursive formatting and canonical blank-line layout
Assisted-by: Qwen 3.8 Max Preview
2026-07-12 21:24:41 +02:00
petrbalvin 1a01870695 fix(lint): calibrate register-clobber to the Go ABI and add legacy SSE moves
Assisted-by: Qwen 3.8 Max Preview
2026-07-11 17:36:52 +02:00
petrbalvin 458cfb626e feat(asm): add EVEX/AVX-512 encoding and assemble the AVX-512 kernel byte-identically
Assisted-by: Qwen 3.8 Max Preview
2026-07-10 13:20:49 +02:00
petrbalvin 56ecc39539 feat(asm): assemble static symbols and the whole go-flac AVX2 kernel byte-identically
Assisted-by: Qwen 3.8 Max Preview
2026-07-09 15:56:03 +02:00
petrbalvin a82f575aee feat(asm): byte-identical go-flac AVX2 assembly with scalar families and jump relaxation
Assisted-by: Qwen 3.8 Max Preview
2026-07-08 12:51:35 +02:00
32 changed files with 6200 additions and 367 deletions
+10
View File
@@ -156,6 +156,16 @@ func (t *Table) Lookup(mnemonic string) (Instr, bool) {
}
}
}
// amd64 EVEX instructions take a .Z zeroing suffix (masking is written as
// an explicit K operand rather than a suffix); strip it so the base
// instruction is still recognised.
if t.Arch == AMD64 {
if base, ok := strings.CutSuffix(key, ".Z"); ok {
if in, found := t.instrs[base]; found {
return in, true
}
}
}
return Instr{}, false
}
+250 -49
View File
@@ -13,55 +13,205 @@ import (
// Assemble encodes the body of a TEXT function into x86-64 machine code,
// resolving local labels to relative jump offsets and translating the FP/SP
// pseudo-registers onto the hardware stack pointer (matching the Go
// assembler's default frame-pointer behaviour). Jumps always use the 32-bit
// relative form so instruction sizes are fixed and offsets resolve in a single
// layout pass.
// assembler's default frame-pointer behaviour). Jumps start in the short
// (rel8) form and expand to rel32 when the settled displacement does not fit;
// sizes only grow, so the layout reaches a fixed point in a few passes. CALL
// has no short form and is always rel32.
//
// Supported operands: registers, memory (real base register), immediates,
// FP/SP frame-relative operands, and local-label jumps. SB (global symbol)
// operands require relocations and are not yet supported; the SIMD (VEX/AVX2)
// integer and shuffle/extract/permute/move set is in.
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
fi := computeFrame(t)
code, _, labels, _, err := assemble(t, nil)
return code, labels, err
}
// Pass 1: lay out instructions (including prologue/epilogue) to fix label
// offsets.
offsets := map[string]int{}
// linkInfo carries file-level symbol context into a single-function assembly:
// the set of static symbols a GLOBL in the same file defines. A nil link
// rejects SB operands outright (single-function assembly cannot resolve
// them). When allowExternal is set, a reference to a symbol no GLOBL in the
// file defines is recorded as an external relocation instead of failing —
// the object-file emitters resolve it at link time.
type linkInfo struct {
symbols map[string]bool
allowExternal bool
}
// sbPatch is a function-relative static-symbol relocation: the disp32 field
// at off must become the symbol's address minus after, where after is the
// function-relative address just past the instruction.
type sbPatch struct {
off int
after int
name string
addend int64
}
// spadjStep is one stack-adjustment boundary within a function: Value is the
// SP delta from the entry state (just below the return address) in effect
// from PC (function-relative) until the next step. The steps feed the
// pcsp table of the object-file emitters.
type spadjStep struct {
pc int
value int
}
// assemble encodes a TEXT body, returning the machine code, the static-symbol
// patch sites (for the file-level layout to resolve), the label table and the
// stack-adjustment boundaries.
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, error) {
fi := computeFrame(t)
chain := jumpChain(t)
resolve := func(name string) string {
if r, ok := chain[name]; ok {
return r
}
return name
}
// Layout: iterate jump sizes to a fixed point.
long := make([]bool, len(t.Body))
sizes := make([]int, len(t.Body))
pos := len(fi.prologue)
for i, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
sz, err := instrSize(s, fi)
if err != nil {
return nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
offsets := map[string]int{}
pcs := make([]int, len(t.Body))
for {
pos := len(fi.prologue)
for i, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
sz, err := instrSize(s, fi, long[i], link)
if err != nil {
return nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
}
sizes[i] = sz
pcs[i] = pos
pos += sz
}
sizes[i] = sz
pos += sz
}
// Expand any short jump whose displacement no longer fits rel8.
changed := false
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
if !ok {
continue
}
mnem := strings.ToUpper(s.Mnemonic.Text)
if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] {
continue
}
name, ok := labelName(s.Operands[0])
if !ok {
continue // reported during emission
}
target, ok := offsets[resolve(name)]
if !ok {
continue // reported during emission
}
rel := int64(target - (pcs[i] + jumpSize(mnem, false)))
if !fits8(rel) {
long[i] = true
changed = true
}
}
if !changed {
break
}
}
// Pass 2: emit.
out := append([]byte(nil), fi.prologue...)
pos = len(fi.prologue)
var patches []sbPatch
var steps []spadjStep
if fi.useFP {
// PUSHQ BP saves the return-address-relative base (+8); the MOVQ
// changes nothing; SUBQ $size, SP completes the frame.
steps = append(steps,
spadjStep{1, 8},
spadjStep{len(fi.prologue), 8 + fi.size},
)
}
pos := len(fi.prologue)
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
if !ok {
continue
}
code, err := encodeInstr(s, pos, offsets, fi)
if strings.ToUpper(s.Mnemonic.Text) == "RET" && fi.useFP {
// The RET's epilogue prefix unwinds: ADDQ $size, SP restores
// the saved-BP-only stack, POPQ BP the entry state.
epi := len(fi.epilogue)
steps = append(steps,
spadjStep{pos + epi - 1, 8},
spadjStep{pos + epi, 0},
)
}
code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
if err != nil {
return nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
return nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
}
if len(code) != sizes[i] {
return nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
return nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
}
patches = append(patches, ps...)
out = append(out, code...)
pos += len(code)
}
return out, offsets, nil
return out, patches, offsets, steps, nil
}
// jumpChain precomputes jump-to-jump folding: a label whose first instruction
// is an unconditional local jump redirects its own jumpers to the ultimate
// target. The Go toolchain chases exactly these chains (the linker's xfol
// pass) before it encodes branches, so matching its bytes requires the same
// redirection.
func jumpChain(t *ast.Text) map[string]string {
// label → the target of its leading unconditional local JMP, if any.
leadsTo := map[string]string{}
for i, stmt := range t.Body {
l, ok := stmt.(*ast.Label)
if !ok {
continue
}
// Stacked labels share an address: skip to the first instruction.
j := i + 1
for j < len(t.Body) {
if _, isLabel := t.Body[j].(*ast.Label); !isLabel {
break
}
j++
}
if j >= len(t.Body) {
continue
}
in, ok := t.Body[j].(*ast.Instr)
if !ok || strings.ToUpper(in.Mnemonic.Text) != "JMP" || len(in.Operands) != 1 {
continue
}
if name, ok := labelName(in.Operands[0]); ok {
leadsTo[l.Name.Text] = name
}
}
// Chase each chain to its end, guarding against cycles.
chain := map[string]string{}
for name := range leadsTo {
visited := map[string]bool{name: true}
cur := name
for {
next, ok := leadsTo[cur]
if !ok || visited[next] {
break
}
visited[next] = true
cur = next
}
if cur != name {
chain[name] = cur
}
}
return chain
}
// frameInfo carries the frame layout derived from the TEXT directive.
@@ -119,15 +269,15 @@ func addSP(size int) []byte { // ADDQ $size, SP
return append([]byte{0x48, 0x81, 0xC4}, le32(int64(size))...)
}
// instrSize returns the encoded length of an instruction (pass 1). encodeInstr
// already includes the epilogue for a RET in a frame-pointer function; jumps use
// a fixed rel32 size (no epilogue).
func instrSize(s *ast.Instr, fi frameInfo) (int, error) {
// instrSize returns the encoded length of an instruction (layout pass).
// encodeInstr already includes the epilogue for a RET in a frame-pointer
// function; jumps use their short or long form (never an epilogue).
func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, error) {
mnem := strings.ToUpper(s.Mnemonic.Text)
if isJumpMnemonic(mnem) {
return jumpSize(mnem), nil
return jumpSize(mnem, long), nil
}
code, err := encodeInstr(s, 0, nil, fi)
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
if err != nil {
return 0, err
}
@@ -142,18 +292,27 @@ func isJumpMnemonic(mnem string) bool {
return ok
}
// jumpSize returns the fixed length of a rel32 jump instruction.
func jumpSize(mnem string) int {
if mnem == "JMP" || mnem == "CALL" {
// jumpSize returns the length of a jump instruction in the requested form:
// short (rel8) where available, otherwise the rel32 form. CALL is always
// rel32.
func jumpSize(mnem string, long bool) int {
if mnem == "CALL" {
return 5 // opcode + rel32
}
if !long {
return 2 // opcode + rel8
}
if mnem == "JMP" {
return 5 // E9 + rel32
}
return 6 // 0x0F 0x8x + rel32
}
// encodeInstr encodes one instruction, resolving jump targets against offsets
// (relative to pc, the instruction's own offset). A RET in a frame-pointer
// function is prefixed with the epilogue.
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo) ([]byte, error) {
// function is prefixed with the epilogue. resolve, when non-nil, redirects a
// jump label through the jump-to-jump chain before the offset lookup.
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, error) {
mnem := strings.ToUpper(s.Mnemonic.Text)
var prefix []byte
@@ -162,37 +321,53 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo) ([]
}
var code []byte
var ps []sbPatch
var err error
if isJumpMnemonic(mnem) {
code, err = encodeJump(s, mnem, pc+len(prefix), offsets)
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
} else {
code, err = encodeNormal(s, fi)
code, ps, err = encodeNormal(s, fi, link)
}
if err != nil {
return nil, err
return nil, nil, err
}
return append(prefix, code...), nil
// Anchor the patch fields at function-relative positions: off indexes the
// disp32 field, after is the address just past the instruction.
body := pc + len(prefix)
for i := range ps {
ps[i].off += body
ps[i].after = body + len(code)
}
return append(prefix, code...), ps, nil
}
func encodeNormal(s *ast.Instr, fi frameInfo) ([]byte, error) {
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) {
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
if size == 0 {
size = 8
}
ops := make([]Operand, len(s.Operands))
for i, op := range s.Operands {
o, err := operandFromAST(op, size, fi)
o, err := operandFromAST(op, size, fi, link)
if err != nil {
return nil, err
return nil, nil, err
}
ops[i] = o
}
return Encode(s.Mnemonic.Text, ops...)
e := &enc{}
if err := e.encode(s.Mnemonic.Text, ops); err != nil {
return nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
}
return e.out, ps, nil
}
// encodeJump encodes a JMP/CALL/Jcc with a rel32 offset resolved from the
// target label.
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int) ([]byte, error) {
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
// target label, in the short (rel8) or long (rel32) form.
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long bool, resolve func(string) string) ([]byte, error) {
if len(s.Operands) != 1 {
return nil, fmt.Errorf("jump expects 1 operand, got %d", len(s.Operands))
}
@@ -200,12 +375,25 @@ func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int) ([]by
if !ok {
return nil, fmt.Errorf("jump target must be a local label")
}
if resolve != nil && mnem != "CALL" {
name = resolve(name)
}
target, ok := offsets[name]
if !ok {
return nil, fmt.Errorf("undefined label %q", name)
}
rel := int64(target - (pc + jumpSize(mnem)))
rel := int64(target - (pc + jumpSize(mnem, long)))
if !long {
if !fits8(rel) {
return nil, fmt.Errorf("jump to %q does not fit the short form", name)
}
if mnem == "JMP" {
return []byte{0xEB, byte(int8(rel))}, nil
}
cc, _ := condCode(mnem)
return []byte{0x70 + byte(cc), byte(int8(rel))}, nil
}
switch mnem {
case "JMP":
return append([]byte{0xE9}, le32(rel)...), nil
@@ -231,7 +419,7 @@ var spReg = Reg{idx: 4, size: 8}
// operandFromAST converts a parsed operand into an encoder Operand, applying
// the frame translation to FP/SP pseudo-register operands.
func operandFromAST(op *ast.Operand, size int, fi frameInfo) (Operand, error) {
func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) {
switch op.Kind {
case ast.OpImmediate:
if op.Imm.HasVal {
@@ -257,9 +445,22 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo) (Operand, error) {
off := fi.spAdjust + a.Sym.Offset
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
}
// SB (global symbol) needs a relocation — not yet supported.
// SB (global symbol): a symbol defined in the same file (GLOBL) is
// encoded RIP-relative and resolved by the file-level layout;
// anything not defined here needs object-file emission.
if a.Sym != nil && a.Sym.Pseudo == "SB" {
return nil, fmt.Errorf("SB (global symbol) operands need relocation support (pending)")
if link == nil || link.symbols == nil {
return nil, fmt.Errorf("symbol %q needs file-level assembly (AssembleFile)", a.Sym.Name)
}
if !link.symbols[a.Sym.Name] {
if a.Sym.Static {
return nil, fmt.Errorf("undefined symbol %q", a.Sym.Name)
}
if !link.allowExternal {
return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name)
}
}
return sbMem{size: size, name: a.Sym.Name, addend: a.Sym.Offset}, nil
}
// Memory with a real base register: (base), off(base), (base)(index*scale).
+73
View File
@@ -244,3 +244,76 @@ TEXT ·hsum(SB), NOSPLIT, $0
t.Errorf("VEX kernel mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
// TestAssembleShortJumps checks that a tight loop settles on the short (rel8)
// jump forms, byte for byte with the Go assembler.
func TestAssembleShortJumps(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·loop(SB), NOSPLIT, $0
XORQ AX, AX
l1:
ADDQ $1, AX
CMPQ AX, $10
JLT l1
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
// From the Go-assembled function:
// XORQ AX, AX 4831c0
// ADDQ $1, AX 4883c001
// CMPQ AX, $10 4883f80a
// JLT l1 7cf6 (short, rel8)
// RET c3
want := []byte{
0x48, 0x31, 0xc0,
0x48, 0x83, 0xc0, 0x01,
0x48, 0x83, 0xf8, 0x0a,
0x7c, 0xf6,
0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("short-jump mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
// TestAssembleJumpFolding checks jump-to-jump folding: a conditional jump to a
// label that only holds an unconditional jump is redirected to the ultimate
// target, exactly as the Go toolchain does before it encodes branches.
func TestAssembleJumpFolding(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·fold(SB), NOSPLIT, $0
XORQ AX, AX
JGE done
INCQ AX
done:
JMP end
end:
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
// From the Go-assembled function: the JGE skips past the done: trampoline
// straight to end:
// XORQ AX, AX 4831c0
// JGE end 7d05 (folded past done)
// INCQ AX 48ffc0
// JMP end eb00
// RET c3
want := []byte{
0x48, 0x31, 0xc0,
0x7d, 0x05,
0x48, 0xff, 0xc0,
0xeb, 0x00,
0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("jump-folding mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
+301
View File
@@ -0,0 +1,301 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// This file emits ELF64 relocatable objects (ET_REL) from an assembled
// Image: a .text section holding the function bodies, a .data section
// holding the GLOBL initialisers, a symbol table with one symbol per TEXT
// and GLOBL (file-local <> symbols are STB_LOCAL, the rest STB_GLOBAL), and
// a .rela.text relocation table — one R_X86_64_PC32 entry per static-symbol
// reference, internal references resolving against the local data symbols
// and external ones against undefined globals. The output links with the
// system toolchain (cc/ld) the way a hand-assembled .o would.
// ELF constants (ELF64, little-endian, System V).
const (
elfClass64 = 2
elfDataLSB = 1
elfVersion = 1
etREL = 1 // relocatable object
emX8664 = 62
shtNull = 0
shtProgbits = 1
shtSymtab = 2
shtStrtab = 3
shtRela = 4
shfWrite = 1
shfAlloc = 2
shfExecInstr = 4
stbLocal = 0
stbGlobal = 1
sttNotype = 0
sttObject = 1
sttFunc = 2
sttSection = 3
stInfoShift = 4
shnUndef = 0
rX8664PC32 = 2
)
// elfSym is one symbol-table entry in construction.
type elfSym struct {
name string
info byte
shndx uint16
value uint64
size uint64
}
// ELFObject returns the image as an ELF64 relocatable object file, ready for
// the system linker. Symbol names are the TEXT and GLOBL identifiers as
// written (the middle dot stripped); a package prefix, when present, is
// joined with a dot. Every static-symbol reference becomes an
// R_X86_64_PC32 relocation, so the code is position-independent and links
// at any address.
func (img *Image) ELFObject() ([]byte, error) {
le := binary.LittleEndian
// Section indices: 0 NULL, 1 .text, 2 .data; the tables follow.
const (
secText = 1
secData = 2
)
// Build the symbol table: the null entry and the two section symbols
// come first, then the local symbols (static TEXT and GLOBL), then the
// globals (exported TEXT and GLOBL, and the undefined externals) — ELF
// requires every local to precede every global, and sh_info records the
// boundary. symIdx maps a symbol name to its index for the relocations.
var locals, globals []elfSym
for _, fn := range img.Funcs {
s := elfSym{
name: objectName(fn.Pkg, fn.Name),
info: sttFunc,
shndx: secText,
value: uint64(fn.Offset),
size: uint64(fn.Size),
}
if fn.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := elfSym{
name: objectName(d.Pkg, d.Name),
info: sttObject,
shndx: secData,
value: uint64(d.Offset),
size: uint64(d.Size),
}
if d.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, name := range img.Externals {
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
}
syms := []elfSym{
{}, // the mandatory null entry
{name: ".text", info: sttSection, shndx: secText},
{name: ".data", info: sttSection, shndx: secData},
}
syms = append(syms, locals...)
shInfo := len(syms) // first global symbol
syms = append(syms, globals...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Build the relocations.
type elfRela struct {
off uint64
sym int
addend int64
}
var relas []elfRela
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
sym: idx,
// R_X86_64_PC32 computes S + A − P with P the patch site; the
// assembler measures the symbol from the instruction end,
// After − Off bytes past the field, so the addend carries
// that distance with a negative sign.
addend: r.Addend - int64(r.After-r.Off),
})
}
}
// Serialise the string tables.
stNames := newElfStrtab()
for _, s := range syms {
stNames.add(s.name)
}
stSections := newElfStrtab()
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
// Section presence: .rela.text only when there are relocations.
hasRela := len(relas) > 0
nSections := 6 // NULL, .text, .data, .symtab, .strtab, .shstrtab
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Lay the file out: header, section data, section headers.
var out []byte
out = append(out, make([]byte, 64)...) // ELF header, filled last
align := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
align(16)
textOff := len(out)
out = append(out, img.Code...)
align(16)
dataOff := len(out)
out = append(out, img.Data...)
align(8)
symtabOff := len(out)
for _, s := range syms {
var b [24]byte
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
b[4] = s.info
b[5] = 0 // st_other
le.PutUint16(b[6:], s.shndx)
le.PutUint64(b[8:], s.value)
le.PutUint64(b[16:], s.size)
out = append(out, b[:]...)
}
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
if hasRela {
align(8)
relaOff = len(out)
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|rX8664PC32)
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
align(8)
shoff := len(out)
// Section headers.
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
var b [64]byte
le.PutUint32(b[0:], uint32(stSections.at(name)))
le.PutUint32(b[4:], uint32(typ))
le.PutUint64(b[8:], flags)
le.PutUint64(b[16:], 0) // sh_addr
le.PutUint64(b[24:], uint64(off))
le.PutUint64(b[32:], uint64(size))
le.PutUint32(b[40:], uint32(link))
le.PutUint32(b[44:], uint32(info))
le.PutUint64(b[48:], alignV)
le.PutUint64(b[56:], entsize)
out = append(out, b[:]...)
}
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// The ELF header.
hdr := out[:64]
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
le.PutUint16(hdr[16:], etREL)
le.PutUint16(hdr[18:], emX8664)
le.PutUint32(hdr[20:], elfVersion)
le.PutUint64(hdr[24:], 0) // e_entry
le.PutUint64(hdr[32:], 0) // e_phoff
le.PutUint64(hdr[40:], uint64(shoff)) // e_shoff
le.PutUint32(hdr[48:], 0) // e_flags
le.PutUint16(hdr[52:], 64) // e_ehsize
le.PutUint16(hdr[54:], 0) // e_phentsize
le.PutUint16(hdr[56:], 0) // e_phnum
le.PutUint16(hdr[58:], 64) // e_shentsize
le.PutUint16(hdr[60:], uint16(nSections))
le.PutUint16(hdr[62:], uint16(secShstr))
return out, nil
}
// objectName renders a symbol's object-file name: the identifier as written,
// with an explicit package prefix joined by a dot.
func objectName(pkg, name string) string {
if pkg == "" {
return name
}
return pkg + "." + name
}
// elfStrtab is an ELF string table under construction.
type elfStrtab struct {
buf []byte
off map[string]int
}
func newElfStrtab() *elfStrtab {
return &elfStrtab{buf: []byte{0}, off: map[string]int{"": 0}}
}
func (s *elfStrtab) add(name string) {
if _, ok := s.off[name]; ok {
return
}
s.off[name] = len(s.buf)
s.buf = append(s.buf, name...)
s.buf = append(s.buf, 0)
}
func (s *elfStrtab) at(name string) int { return s.off[name] }
func (s *elfStrtab) bytes() []byte { return s.buf }
+310
View File
@@ -0,0 +1,310 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"os"
"os/exec"
"path/filepath"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// The object-file tests share one source: two exported functions, one
// file-local constant reached through a relocation, and one external symbol
// the linker must resolve. The functions take their arguments in the System
// V registers (not the Go stack ABI) so a C driver can call them directly.
const elfTestSrc = `
#include "textflag.h"
TEXT ·addq(SB), NOSPLIT, $0
LEAQ (DI)(SI*1), AX
RET
TEXT ·getanswer(SB), NOSPLIT, $0
MOVQ answer<>(SB), AX
RET
TEXT ·useextern(SB), NOSPLIT, $0
MOVQ extvar(SB), AX
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`
func elfTestImage(t *testing.T) *Image {
t.Helper()
f, errs := parser.Parse("t_amd64.s", elfTestSrc)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
return img
}
// TestAssembleFileExternals checks that a reference to a symbol no GLOBL
// defines is recorded as an external relocation instead of failing — the
// raw image leaves the displacement zero, the object emitters carry it.
func TestAssembleFileExternals(t *testing.T) {
img := elfTestImage(t)
if len(img.Externals) != 1 || img.Externals[0] != "extvar" {
t.Fatalf("Externals = %v, want [extvar]", img.Externals)
}
var ext, local int
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.External {
ext++
if r.Name != "extvar" {
t.Errorf("external reloc names %q, want extvar", r.Name)
}
} else {
local++
if r.Name != "answer" {
t.Errorf("local reloc names %q, want answer", r.Name)
}
}
}
}
if ext != 1 || local != 1 {
t.Errorf("relocs = %d external, %d local; want 1 and 1", ext, local)
}
}
// TestELFObject checks the structure of the emitted ELF64 relocatable
// object: sections, the symbol table (bindings, types, values, sizes) and
// the .rela.text relocations, parsed back with debug/elf.
func TestELFObject(t *testing.T) {
img := elfTestImage(t)
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
f, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer f.Close()
if f.Type != elf.ET_REL || f.Machine != elf.EM_X86_64 {
t.Errorf("type/machine = %v/%v, want ET_REL/EM_X86_64", f.Type, f.Machine)
}
text := f.Section(".text")
data := f.Section(".data")
if text == nil || data == nil {
t.Fatal("missing .text or .data section")
}
if text.Flags&elf.SHF_EXECINSTR == 0 || text.Flags&elf.SHF_ALLOC == 0 {
t.Errorf(".text flags = %v", text.Flags)
}
if data.Flags&elf.SHF_WRITE == 0 {
t.Errorf(".data flags = %v", data.Flags)
}
textData, err := text.Data()
if err != nil {
t.Fatal(err)
}
if !bytes.Equal(textData, img.Code) {
t.Errorf(".text contents differ from the image code")
}
syms, err := f.Symbols()
if err != nil {
t.Fatalf("symbols: %v", err)
}
byName := map[string]elf.Symbol{}
for _, s := range syms {
byName[s.Name] = s
}
wantSym := func(name string, bind elf.SymBind, typ elf.SymType, section elf.SectionIndex, size uint64) {
t.Helper()
s, ok := byName[name]
if !ok {
t.Errorf("symbol %q not found", name)
return
}
if elf.ST_BIND(s.Info) != bind || elf.ST_TYPE(s.Info) != typ {
t.Errorf("%s: bind/type = %v/%v, want %v/%v", name, elf.ST_BIND(s.Info), elf.ST_TYPE(s.Info), bind, typ)
}
if s.Section != section {
t.Errorf("%s: section = %v, want %v", name, s.Section, section)
}
if s.Size != size {
t.Errorf("%s: size = %d, want %d", name, s.Size, size)
}
}
// The emitted layout is fixed: 0 NULL, 1 .text, 2 .data.
if f.Sections[1].Name != ".text" || f.Sections[2].Name != ".data" {
t.Fatalf("section layout = %s, %s; want .text, .data", f.Sections[1].Name, f.Sections[2].Name)
}
textIdx := elf.SectionIndex(1)
dataIdx := elf.SectionIndex(2)
wantSym("addq", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 5)
wantSym("getanswer", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 8)
wantSym("useextern", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 8)
wantSym("answer", elf.STB_LOCAL, elf.STT_OBJECT, dataIdx, 8)
wantSym("extvar", elf.STB_GLOBAL, elf.STT_NOTYPE, elf.SHN_UNDEF, 0)
// Relocations: one for the file-local constant (resolving against the
// local data symbol) and one for the external (against the undefined
// global), both R_X86_64_PC32 with the −4 addend the PC-relative form
// needs. debug/elf does not surface rela entries, so read the section
// directly.
relaSec := f.Section(".rela.text")
if relaSec == nil {
t.Fatal("missing .rela.text")
}
raw, err := relaSec.Data()
if err != nil {
t.Fatal(err)
}
if len(raw)%24 != 0 || len(raw)/24 != 2 {
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
}
// Symbol names straight from the raw tables: r_info carries an index
// into .symtab including the null entry, which debug/elf's Symbols()
// slice may not mirror.
symtabRaw, err := f.Section(".symtab").Data()
if err != nil {
t.Fatal(err)
}
strtabRaw, err := f.Section(".strtab").Data()
if err != nil {
t.Fatal(err)
}
symName := func(idx int) string {
stName := binary.LittleEndian.Uint32(symtabRaw[idx*24:])
end := bytes.IndexByte(strtabRaw[stName:], 0)
return string(strtabRaw[stName : int(stName)+end])
}
for i := 0; i < 2; i++ {
e := raw[i*24 : (i+1)*24]
off := binary.LittleEndian.Uint64(e[0:])
info := binary.LittleEndian.Uint64(e[8:])
addend := int64(binary.LittleEndian.Uint64(e[16:]))
typ := info & 0xffffffff
sym := int(info >> 32)
if typ != uint64(elf.R_X86_64_PC32) {
t.Errorf("reloc %d: type %d, want R_X86_64_PC32", i, typ)
}
if addend != -4 {
t.Errorf("reloc %d: addend %d, want -4", i, addend)
}
if name := symName(sym); name != "answer" && name != "extvar" {
t.Errorf("reloc %d: symbol %q, want answer or extvar", i, name)
}
// The relocation offset lands on the disp32 field: the four bytes
// before a RET-terminated eight-byte MOVQ.
if off+4 > uint64(len(textData)) {
t.Errorf("reloc %d: offset %d outside .text", i, off)
}
}
}
// TestELFObjectNoRelocations checks a file with no static-symbol references
// emits a valid object without a .rela.text section.
func TestELFObjectNoRelocations(t *testing.T) {
f, errs := parser.Parse("n_amd64.s", `
#include "textflag.h"
TEXT ·nop(SB), NOSPLIT, $0
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Section(".rela.text") != nil {
t.Error("unexpected .rela.text section")
}
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
found := false
for _, s := range syms {
if s.Name == "nop" && elf.ST_TYPE(s.Info) == elf.STT_FUNC {
found = true
}
}
if !found {
t.Error("function symbol nop not found")
}
}
// TestELFLinkAndRun is the end-to-end check: assemble the test functions,
// link the emitted object with a C driver that defines the external symbol,
// and run the result. Skipped when no C compiler is available.
func TestELFLinkAndRun(t *testing.T) {
cc, err := exec.LookPath("cc")
if err != nil {
t.Skip("no C compiler available")
}
dir := t.TempDir()
img := elfTestImage(t)
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
objPath := filepath.Join(dir, "t.o")
if err := os.WriteFile(objPath, obj, 0o644); err != nil {
t.Fatal(err)
}
const driver = `
#include <stdio.h>
long addq(long a, long b);
long getanswer(void);
long useextern(void);
long extvar = 7;
int main(void) {
printf("%ld %ld %ld\n", addq(41, 1), getanswer(), useextern());
return 0;
}
`
driverPath := filepath.Join(dir, "driver.c")
if err := os.WriteFile(driverPath, []byte(driver), 0o644); err != nil {
t.Fatal(err)
}
// -no-pie: the encoder emits R_X86_64_PC32 for external references,
// which a position-independent executable would reject (it wants
// PLT32/GOT relocations, a future increment).
appPath := filepath.Join(dir, "app")
out, err := exec.Command(cc, "-no-pie", "-o", appPath, driverPath, objPath).CombinedOutput()
if err != nil {
t.Fatalf("link failed: %v\n%s", err, out)
}
run, err := exec.Command(appPath).CombinedOutput()
if err != nil {
t.Fatalf("run failed: %v\n%s", err, run)
}
if got := string(run); got != "42 42 7\n" {
t.Errorf("output %q, want \"42 42 7\\n\"", got)
}
}
+88 -6
View File
@@ -19,7 +19,16 @@ func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
}
type enc struct {
out []byte
out []byte
patches []encPatch // disp32 fields awaiting static-symbol resolution
}
// encPatch marks a 4-byte displacement field in enc.out that must receive the
// RIP-relative offset of a static symbol once the file layout is settled.
type encPatch struct {
off int
name string
addend int64
}
func (e *enc) encode(mnem string, ops []Operand) error {
@@ -40,10 +49,27 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeJcc(cc, ops)
}
// VEX (AVX/AVX2) instructions: the trailing B/W/L/Q/D is part of the
// mnemonic, not a size suffix, so dispatch before splitSize.
if isVex(upper) {
return e.encodeVex(upper, ops)
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
// before splitSize. EVEX suffixes (.Z, .SAE, rounding, .BCST) split
// off the mnemonic too.
base, sfx, err := parseEvexSuffix(upper)
if err != nil {
return err
}
if isVex(base) || isEvex(base) || isKOp(base) || base == "KMOVW" || base == "KMOVQ" {
return e.encodeVec(base, ops, sfx)
}
if sfx.any() {
return fmt.Errorf("%s: the suffix requires an EVEX instruction", mnem)
}
// CMOVcc and SETcc carry the condition in the mnemonic (CMOVLGT, SETNE).
if strings.HasPrefix(upper, "CMOV") {
return e.encodeCmov(upper, ops)
}
if strings.HasPrefix(upper, "SET") {
return e.encodeSet(upper, ops)
}
base, size := splitSize(upper)
@@ -63,12 +89,20 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeUnary(unaryOp[base], ops, size)
case "SHL", "SHR", "SAR":
return e.encodeShift(shiftOp[base], ops, size)
case "IMUL":
case "IMUL", "IMUL3":
return e.encodeImul(ops, size)
case "PUSH":
return e.encodePushPop(ops, true)
case "POP":
return e.encodePushPop(ops, false)
case "LZCNT", "TZCNT":
return e.encodeCount(base, ops, size)
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX":
return e.encodeMovExtend(base, ops)
case "CVTSL2SD", "CVTSQ2SD":
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
return e.encodeSSEMove(sseMoveTable[base], ops)
}
return fmt.Errorf("unsupported instruction %q", mnem)
}
@@ -91,6 +125,32 @@ func splitSize(upper string) (base string, size int) {
return upper, 0
}
// encodeVec dispatches a VEX/EVEX mnemonic to the right encoding: KMOVW has
// its own direction-dependent opcodes; KTESTW is always VEX; everything else
// takes EVEX when an operand demands it (a ZMM or K register, or an
// EVEX-only mnemonic) and VEX otherwise.
func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
if upper == "KMOVW" || upper == "KMOVQ" {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", upper)
}
return e.encodeKmov(upper, ops)
}
if isKOp(upper) {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", upper)
}
return e.encodeKOp(upper, ops)
}
if upper == "KTESTW" || (!evexRequired(upper, ops) && !sfx.evexOnly()) {
if sfx.any() {
return fmt.Errorf("%s: the .Z suffix requires an EVEX instruction", upper)
}
return e.encodeVex(upper, ops)
}
return e.encodeEvex(upper, ops, sfx)
}
// --- instruction components -------------------------------------------------
type instr struct {
@@ -100,17 +160,29 @@ type instr struct {
rexX bool
rexB bool
rexForced bool // REX needed even with all bits zero (8-bit low registers)
prefix byte // legacy 0xF2/0xF3 prefix (0 = none); emitted after 0x66
opcode []byte
modrm int // -1 if absent
sib int // -1 if absent
disp []byte
imm []byte
sb *sbRef // static-symbol displacement in disp, awaiting resolution
}
// sbRef records that an instruction's displacement refers to a static symbol
// rather than holding a literal value.
type sbRef struct {
name string
addend int64
}
func (e *enc) emit(i *instr) error {
if i.opSize16 {
e.out = append(e.out, 0x66)
}
if i.prefix != 0 {
e.out = append(e.out, i.prefix)
}
rex := byte(0)
if i.rexW {
rex |= 0x08
@@ -134,6 +206,9 @@ func (e *enc) emit(i *instr) error {
if i.sib >= 0 {
e.out = append(e.out, byte(i.sib))
}
if i.sb != nil {
e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend})
}
e.out = append(e.out, i.disp...)
e.out = append(e.out, i.imm...)
return nil
@@ -181,6 +256,13 @@ func setRMReg(i *instr, regField int, rexR, regForced bool, rm Operand, opSize i
return nil
case Mem:
return setMem(i, regField, r)
case sbMem:
// RIP-relative reference; the displacement is patched once the static
// symbol's address is known.
i.modrm = regField<<3 | 0x05 // mod=00, rm=101 → (RIP)+disp32
i.disp = le32(0)
i.sb = &sbRef{name: r.name, addend: r.addend}
return nil
default:
return fmt.Errorf("invalid r/m operand %T", rm)
}
+161 -1
View File
@@ -4,6 +4,7 @@
package asm
import (
"strings"
"testing"
"golang.org/x/arch/x86/x86asm"
@@ -70,7 +71,7 @@ func TestALU(t *testing.T) {
checkSyntax(t, "and rbx, 0x7", "ANDQ", Imm(7), BX)
checkSyntax(t, "or rcx, rbx", "ORQ", BX, CX)
checkSyntax(t, "xor rax, rax", "XORQ", AX, AX)
checkSyntax(t, "cmp r10, rsi", "CMPQ", SI, Reg{idx: 10, size: 8})
checkSyntax(t, "cmp rsi, r10", "CMPQ", SI, Reg{idx: 10, size: 8})
checkSyntax(t, "add rbx, qword ptr [rax]", "ADDQ", Ptr(AX, 0, 8), BX)
checkSyntax(t, "add qword ptr [rax], rbx", "ADDQ", BX, Ptr(AX, 0, 8))
checkSyntax(t, "cmp rbx, -0x20", "CMPQ", Imm(-32), BX)
@@ -126,6 +127,52 @@ func TestControl(t *testing.T) {
checkOp(t, x86asm.JBE, "JLS", Imm(0))
}
// TestSSEMoveGroundTruth checks the legacy (non-VEX) SSE moves byte for byte
// against the Go assembler. wantOp is the decoder's name, which differs from
// the Plan 9 spelling for the octa moves (MOVOU = MOVDQU, MOVO = MOVDQA).
func TestSSEMoveGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
wantOp string
}{
{"MOVOU (SI),X1", "MOVOU", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "f30f6f0e", "MOVDQU"},
{"MOVOU X3,(DI)", "MOVOU", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "f30f7f1f", "MOVDQU"},
{"MOVOU X1,X2", "MOVOU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f30f6fd1", "MOVDQU"},
{"MOVOU (SI)(BX*4),X9", "MOVOU", []Operand{Idx(SI, BX, 4, 0, 16), vreg(t, "X9")}, "f3440f6f0c9e", "MOVDQU"},
{"MOVO (SI),X1", "MOVO", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "660f6f0e", "MOVDQA"},
{"MOVO X3,(DI)", "MOVO", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "660f7f1f", "MOVDQA"},
{"MOVUPS (SI),X1", "MOVUPS", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "0f100e", "MOVUPS"},
{"MOVAPS X3,(DI)", "MOVAPS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "0f291f", "MOVAPS"},
{"MOVUPD (SI),X1", "MOVUPD", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "660f100e", "MOVUPD"},
{"MOVAPD X3,(DI)", "MOVAPD", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "660f291f", "MOVAPD"},
{"MOVSD (SI),X1", "MOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X1")}, "f20f100e", "MOVSD_XMM"},
{"MOVSD X1,X2", "MOVSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f20f10d1", "MOVSD_XMM"},
{"MOVSS X3,(DI)", "MOVSS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 4)}, "f30f111f", "MOVSS"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
if inst.Op.String() != c.wantOp {
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
}
}
}
// TestGoFlacScalarTail encodes the scalar tail of an analyze kernel to confirm
// the encoder handles a realistic instruction sequence.
func TestGoFlacScalarTail(t *testing.T) {
@@ -134,3 +181,116 @@ func TestGoFlacScalarTail(t *testing.T) {
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
checkSyntax(t, "and r10, -0x8", "ANDQ", Imm(-8), Reg{idx: 10, size: 8})
}
// TestScalarGroundTruth checks the scalar instruction families the go-flac
// kernels use beyond the basic set, byte for byte against the Go assembler's
// machine code. wantOp is the x86 decoder's name, which differs from the
// Plan 9 spelling for some of these (CMOVLGT → CMOVG, MOVBLZX → MOVZX, …).
func TestScalarGroundTruth(t *testing.T) {
r8 := Reg{idx: 8, size: 8}
r9 := Reg{idx: 9, size: 8}
r9w := Reg{idx: 9, size: 2}
r8w := Reg{idx: 8, size: 2}
r13 := Reg{idx: 13, size: 8}
cases := []struct {
name string
mnem string
ops []Operand
want string
wantOp string
}{
{"LZCNTL AX,CX", "LZCNTL", []Operand{AX, CX}, "f30fbdc8", "LZCNT"},
{"LZCNTQ R8,R9", "LZCNTQ", []Operand{r8, r9}, "f34d0fbdc8", "LZCNT"},
{"LZCNTW AX,CX", "LZCNTW", []Operand{AX, CX}, "66f30fbdc8", "LZCNT"},
{"TZCNTL AX,CX", "TZCNTL", []Operand{AX, CX}, "f30fbcc8", "TZCNT"},
{"CMOVLGT CX,AX", "CMOVLGT", []Operand{CX, AX}, "0f4fc1", "CMOVG"},
{"CMOVLEQ CX,AX", "CMOVLEQ", []Operand{CX, AX}, "0f44c1", "CMOVE"},
{"CMOVQGT R9,R8", "CMOVQGT", []Operand{r9, r8}, "4d0f4fc1", "CMOVG"},
{"CMOVWLS R9W,R8W", "CMOVWLS", []Operand{r9w, r8w}, "66450f46c1", "CMOVBE"},
{"SETNE AL", "SETNE", []Operand{AL}, "0f95c0", "SETNE"},
{"SETNE (AX)", "SETNE", []Operand{Ptr(AX, 0, 1)}, "0f9500", "SETNE"},
{"MOVBLZX AL,CX", "MOVBLZX", []Operand{AL, CX}, "0fb6c8", "MOVZX"},
{"MOVBLZX (SI),CX", "MOVBLZX", []Operand{Ptr(SI, 0, 1), CX}, "0fb60e", "MOVZX"},
{"MOVWLSX (SI)(AX*1),CX", "MOVWLSX", []Operand{Idx(SI, AX, 1, 0, 2), CX}, "0fbf0c06", "MOVSX"},
{"MOVLQSX CX,R8", "MOVLQSX", []Operand{CX, r8}, "4c63c1", "MOVSXD"},
{"MOVBQZX AL,R8", "MOVBQZX", []Operand{AL, r8}, "4c0fb6c0", "MOVZX"},
{"MOVWLZX AX,CX", "MOVWLZX", []Operand{AX, CX}, "0fb7c8", "MOVZX"},
{"MOVWQZX AX,R8", "MOVWQZX", []Operand{AX, r8}, "4c0fb7c0", "MOVZX"},
{"CVTSL2SD R8,X13", "CVTSL2SD", []Operand{r8, vreg(t, "X13")}, "f2450f2ae8", "CVTSI2SD"},
{"CVTSL2SD AX,X0", "CVTSL2SD", []Operand{AX, vreg(t, "X0")}, "f20f2ac0", "CVTSI2SD"},
{"CVTSQ2SD R8,X13", "CVTSQ2SD", []Operand{r8, vreg(t, "X13")}, "f24d0f2ae8", "CVTSI2SD"},
{"INCW (R13)(AX*2)", "INCW", []Operand{Idx(r13, AX, 2, 0, 2)}, "6641ff444500", "INC"},
// The traditional three-operand IMUL spelling.
{"IMUL3L $31,CX,DX", "IMUL3L", []Operand{Imm(31), CX, DX}, "6bd11f", "IMUL"},
{"IMUL3L $256,CX,DX", "IMUL3L", []Operand{Imm(256), CX, DX}, "69d100010000", "IMUL"},
{"IMUL3Q $7,R9,R8", "IMUL3Q", []Operand{Imm(7), r9, r8}, "4d6bc107", "IMUL"},
{"IMUL3W $5,CX,DX", "IMUL3W", []Operand{Imm(5), CX, DX}, "666bd105", "IMUL"},
// Negative displacement with base + index (regression: the parser
// used to drop the whole address).
{"LEAQ -4(DX)(R9*4),R9", "LEAQ", []Operand{Idx(DX, r9, 4, -4, 8), r9}, "4e8d4c8afc", "LEA"},
{"LEAQ 16(SI)(BX*4),R10", "LEAQ", []Operand{Idx(SI, BX, 4, 16, 8), Reg{idx: 10, size: 8}}, "4c8d549e10", "LEA"},
// Register-to-register MOV uses the r/m←r opcode (reg = source), the
// Go assembler's choice.
{"MOVQ BX,R10", "MOVQ", []Operand{BX, Reg{idx: 10, size: 8}}, "4989da", "MOV"},
{"MOVQ AX,BX", "MOVQ", []Operand{AX, BX}, "4889c3", "MOV"},
{"MOVL AX,BX", "MOVL", []Operand{AX, BX}, "89c3", "MOV"},
{"MOVB AL,BL", "MOVB", []Operand{AL, BL}, "88c3", "MOV"},
{"MOVW AX,BX", "MOVW", []Operand{AX, BX}, "6689c3", "MOV"},
{"MOVQ R12,R13", "MOVQ", []Operand{Reg{idx: 12, size: 8}, Reg{idx: 13, size: 8}}, "4d89e5", "MOV"},
// CMP must record first − second: with a register second operand the
// first goes in r/m, with a memory second operand the first goes in reg.
{"CMPQ SI,R10", "CMPQ", []Operand{SI, Reg{idx: 10, size: 8}}, "4c39d6", "CMP"},
{"CMPQ SI,(AX)", "CMPQ", []Operand{SI, Ptr(AX, 0, 8)}, "483b30", "CMP"},
{"CMPQ (AX),SI", "CMPQ", []Operand{Ptr(AX, 0, 8), SI}, "483930", "CMP"},
{"CMPL CX,(AX)", "CMPL", []Operand{CX, Ptr(AX, 0, 4)}, "3b08", "CMP"},
{"CMPB AL,(BX)", "CMPB", []Operand{AL, Ptr(BX, 0, 1)}, "3a03", "CMP"},
{"CMPW AX,BX", "CMPW", []Operand{AX, BX}, "6639d8", "CMP"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := strings.ReplaceAll(hexBytes(code), " ", ""); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
continue
}
if inst.Op.String() != c.wantOp {
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
}
}
}
// TestScalarErrors checks that malformed conditional / extend / convert
// instructions are rejected.
func TestScalarErrors(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
}{
{"CMOV arity", "CMOVLGT", []Operand{AX}},
{"CMOV bare", "CMOV", []Operand{AX, BX}},
{"CMOV bad size", "CMOVBGT", []Operand{AX, BX}},
{"CMOV bad condition", "CMOVLXX", []Operand{AX, BX}},
{"CMOV mem dst", "CMOVLGT", []Operand{AX, Ptr(BX, 0, 4)}},
{"SET arity", "SETNE", []Operand{AL, BL}},
{"SET bad condition", "SETXX", []Operand{AL}},
{"SET bare", "SET", []Operand{AL}},
{"LZCNT arity", "LZCNTL", []Operand{AX}},
{"LZCNT mem dst", "LZCNTL", []Operand{AX, Ptr(BX, 0, 4)}},
{"MOVBLZX mem dst", "MOVBLZX", []Operand{AL, Ptr(BX, 0, 4)}},
{"CVTSL2SD gpr dst", "CVTSL2SD", []Operand{AX, BX}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
+1200
View File
File diff suppressed because it is too large Load Diff
+479
View File
@@ -0,0 +1,479 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"os"
"strings"
"testing"
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestEvexGroundTruth checks the EVEX (AVX-512) encodings byte for byte
// against machine code extracted from the Go toolchain's assembly of the
// same instructions, covering every operand shape the go-flac AVX-512
// kernels use: NDS arithmetic, immediate and variable shifts, shuffles with
// an immediate, lane extracts, narrowing stores, broadcasts from a GPR or
// memory, mask destinations, mask moves, disp8×N compression and the 5-bit
// register fields (X/Y 16–31, Z 0–31).
func TestEvexGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// NDS integer arithmetic / logic.
{"VPXORD Z12,Z12,Z12", "VPXORD", []Operand{vreg(t, "Z12"), vreg(t, "Z12"), vreg(t, "Z12")}, "62511d48efe4"},
{"VPXORQ Z8,Z9,Z10", "VPXORQ", []Operand{vreg(t, "Z8"), vreg(t, "Z9"), vreg(t, "Z10")}, "6251b548efd0"},
{"VPADDD Z1,Z0,Z0", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z0"), vreg(t, "Z0")}, "62f17d48fec1"},
{"VPSUBQ Z8,Z11,Z11", "VPSUBQ", []Operand{vreg(t, "Z8"), vreg(t, "Z11"), vreg(t, "Z11")}, "6251a548fbd8"},
{"VPUNPCKLDQ Z5,Z3,Z6", "VPUNPCKLDQ", []Operand{vreg(t, "Z5"), vreg(t, "Z3"), vreg(t, "Z6")}, "62f1654862f5"},
{"VPUNPCKHDQ Z5,Z3,Z7", "VPUNPCKHDQ", []Operand{vreg(t, "Z5"), vreg(t, "Z3"), vreg(t, "Z7")}, "62f165486afd"},
{"VPMULLQ Z9,Z10,Z10", "VPMULLQ", []Operand{vreg(t, "Z9"), vreg(t, "Z10"), vreg(t, "Z10")}, "6252ad4840d1"},
{"VPMULLD Z13,Z11,Z2", "VPMULLD", []Operand{vreg(t, "Z13"), vreg(t, "Z11"), vreg(t, "Z2")}, "62d2254840d5"},
{"VPERMD Z0,Z15,Z8", "VPERMD", []Operand{vreg(t, "Z0"), vreg(t, "Z15"), vreg(t, "Z8")}, "6272054836c0"},
// Packed-double arithmetic (EVEX forms carry W=1).
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
// Align (NDS + imm8).
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
// Shifts: immediate (/digit) and variable (XMM count).
{"VPSRAD $31,Z3,Z5", "VPSRAD", []Operand{Imm(31), vreg(t, "Z3"), vreg(t, "Z5")}, "62f1554872e31f"},
{"VPSLLD $1,Z3,Z4", "VPSLLD", []Operand{Imm(1), vreg(t, "Z3"), vreg(t, "Z4")}, "62f15d4872f301"},
{"VPSRAQ X31,Z8,Z8", "VPSRAQ", []Operand{vreg(t, "X31"), vreg(t, "Z8"), vreg(t, "Z8")}, "6211bd48e2c7"},
// Mask destinations (the K register occupies the reg field).
{"VPCMPEQD Z0,Z3,K1", "VPCMPEQD", []Operand{vreg(t, "Z0"), vreg(t, "Z3"), vreg(t, "K1")}, "62f1654876c8"},
{"VPCMPEQD Y30,Y11,K1", "VPCMPEQD", []Operand{vreg(t, "Y30"), vreg(t, "Y11"), vreg(t, "K1")}, "6291252876ce"},
// Mask moves and test (VEX-encoded).
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
// Moves, incl. disp8×N (64 for a 512-bit operand).
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
{"VMOVDQU32 16(SI)(R15*4),Z4", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 16, 64), vreg(t, "Z4")}, "62b17e486fa4be10000000"},
{"VMOVDQU32 Z0,4(SI)(AX*1)", "VMOVDQU32", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f17e487f840604000000"},
{"VMOVDQU32 Z3,(DI)(R15*4)", "VMOVDQU32", []Operand{vreg(t, "Z3"), Idx(DI, vreg(t, "R15"), 4, 0, 64)}, "62b17e487f1cbf"},
// VMOVDQU64 — the W1 qword variant.
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
// The wider AVX-512 F/BW integer set.
{"VPADDB Z1,Z2,Z3", "VPADDB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48fcd9"},
{"VPSUBW Z1,Z2,Z3", "VPSUBW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48f9d9"},
{"VPANDQ Z1,Z2,Z3", "VPANDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed48dbd9"},
{"VPANDND Z1,Z2,Z3", "VPANDND", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48dfd9"},
{"VPMULLW Z1,Z2,Z3", "VPMULLW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48d5d9"},
{"VPMINUB Z1,Z2,Z3", "VPMINUB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48dad9"},
{"VPMAXUQ Z1,Z2,Z3", "VPMAXUQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed483fd9"},
{"VPAVGW Z1,Z2,Z3", "VPAVGW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48e3d9"},
{"VPSLLVQ Z3,Z1,Z2", "VPSLLVQ", []Operand{vreg(t, "Z3"), vreg(t, "Z1"), vreg(t, "Z2")}, "62f2f54847d3"},
{"VPSRAVQ Z3,Z1,Z2", "VPSRAVQ", []Operand{vreg(t, "Z3"), vreg(t, "Z1"), vreg(t, "Z2")}, "62f2f54846d3"},
{"VPSHUFD $0x1B,Z1,Z2", "VPSHUFD", []Operand{Imm(0x1B), vreg(t, "Z1"), vreg(t, "Z2")}, "62f17d4870d11b"},
{"VPSHUFB Z1,Z2,Z3", "VPSHUFB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4800d9"},
{"VMOVDQU8 Z1,Z2", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17f487fca"},
{"VMOVDQU16 Z1,Z2", "VMOVDQU16", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff487fca"},
// Indices 16–31: rm[4] rides in X̄ for register operands.
{"VPSHUFD $1,X16,X17", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}, "62a17d0870c801"},
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
// Conversions and narrowing stores (reg = wide source).
{"VCVTQQ2PD Z12,Z12", "VCVTQQ2PD", []Operand{vreg(t, "Z12"), vreg(t, "Z12")}, "6251fe48e6e4"},
{"VCVTQQ2PD X13,X13", "VCVTQQ2PD", []Operand{vreg(t, "X13"), vreg(t, "X13")}, "6251fe08e6ed"},
{"VPMOVSXDQ 32(SI),Z12", "VPMOVSXDQ", []Operand{Ptr(SI, 32, 32), vreg(t, "Z12")}, "62727d48256601"},
{"VPMOVDW Z0,Y0", "VPMOVDW", []Operand{vreg(t, "Z0"), vreg(t, "Y0")}, "62f27e4833c0"},
{"VPMOVQD Z11,Y11", "VPMOVQD", []Operand{vreg(t, "Z11"), vreg(t, "Y11")}, "62527e4835db"},
// Lane extracts.
{"VEXTRACTI64X4 $1,Z8,Y9", "VEXTRACTI64X4", []Operand{Imm(1), vreg(t, "Z8"), vreg(t, "Y9")}, "6253fd483bc101"},
{"VEXTRACTF64X4 $1,Z10,Y11", "VEXTRACTF64X4", []Operand{Imm(1), vreg(t, "Z10"), vreg(t, "Y11")}, "6253fd481bd301"},
// Broadcasts: GPR source (0x7C) vs memory source (0x58/0x59, disp8×4/8).
{"VPBROADCASTD AX,Z15", "VPBROADCASTD", []Operand{AX, vreg(t, "Z15")}, "62727d487cf8"},
{"VPBROADCASTD (SI),Z8", "VPBROADCASTD", []Operand{Ptr(SI, 0, 4), vreg(t, "Z8")}, "62727d485806"},
{"VPBROADCASTD 4(SI),Z10", "VPBROADCASTD", []Operand{Ptr(SI, 4, 4), vreg(t, "Z10")}, "62727d48585601"},
{"VPBROADCASTQ R8,X31", "VPBROADCASTQ", []Operand{vreg(t, "R8"), vreg(t, "X31")}, "6242fd087cf8"},
{"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"},
// Register indices 16–31 exist only in EVEX encodings.
{"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"},
// Packed double arithmetic / unpack (EVEX forms carry W=1).
{"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"},
{"VDIVPD Z4,Z5,Z6", "VDIVPD", []Operand{vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "Z6")}, "62f1d5485ef4"},
{"VMINPD Z7,Z8,Z9", "VMINPD", []Operand{vreg(t, "Z7"), vreg(t, "Z8"), vreg(t, "Z9")}, "6271bd485dcf"},
{"VMAXPD Z10,Z11,Z12", "VMAXPD", []Operand{vreg(t, "Z10"), vreg(t, "Z11"), vreg(t, "Z12")}, "6251a5485fe2"},
{"VUNPCKLPD Z1,Z2,Z3", "VUNPCKLPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4814d9"},
{"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"},
{"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"},
{"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"},
// VMOVDDUP — duplicate the low double; disp8×N = 64 at 512 bits, and
// X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄).
{"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"},
{"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"},
{"VMOVDDUP X16,X17", "VMOVDDUP", []Operand{vreg(t, "X16"), vreg(t, "X17")}, "62a1ff0812c8"},
// Conversions: DQ→PS, PS→PD (pp = 00, the Go assembler's choice),
// DQ→PD (the destination sets the length).
{"VCVTDQ2PS Z1,Z2", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"},
{"VCVTPS2PD Y1,Z2", "VCVTPS2PD", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17c485ad1"},
{"VCVTPS2PD 32(AX),Z2", "VCVTPS2PD", []Operand{Ptr(AX, 32, 32), vreg(t, "Z2")}, "62f17c485a5001"},
{"VCVTDQ2PD Y1,Z2", "VCVTDQ2PD", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17e48e6d1"},
// PD→DQ conversions: the source is the wide operand and fixes the
// length (ZMM source → L'L = 10 even with an XMM destination; a
// memory source takes the length the mnemonic's spelling implies).
{"VCVTPD2DQ Z1,Y2", "VCVTPD2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff48e6d1"},
{"VCVTPD2DQ 64(AX),Y2", "VCVTPD2DQ", []Operand{Ptr(AX, 64, 64), vreg(t, "Y2")}, "62f1ff48e65001"},
{"VCVTTPD2DQ Z3,Y4", "VCVTTPD2DQ", []Operand{vreg(t, "Z3"), vreg(t, "Y4")}, "62f1fd48e6e3"},
}
for _, c := range cases {
want := strings.ReplaceAll(c.want, " ", "")
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != want {
t.Errorf("%s: bytes %s, want %s", c.name, got, want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
if inst.Len != len(code) {
t.Errorf("%s: Decode consumed %d of %d bytes", c.name, inst.Len, len(code))
}
if inst.Op.String() != c.mnem {
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
}
}
}
// TestEvexMasking checks the AVX-512 mask operand (K1–K7, placed freely among
// the operands) and the .Z zeroing suffix, byte for byte against the Go
// assembler.
func TestEvexMasking(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// Masked arithmetic: K anywhere among the operands; .Z sets the z bit.
{"VPADDD.Z merging+zeroing", "VPADDD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f16dcafed9"},
{"VPADDD merging", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49fed9"},
{"VADDPD.Z", "VADDPD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1edca58d9"},
{"VPMINSD.Z", "VPMINSD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K5"), vreg(t, "Z3")}, "62f26dcd39d9"},
{"VPMINSQ.Z", "VPMINSQ.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K5"), vreg(t, "Z3")}, "62f2edcd39d9"},
// Masked immediate shift (K before the destination).
{"VPSRAD.Z", "VPSRAD.Z", []Operand{Imm(1), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f165c972e201"},
{"VPSLLD merge", "VPSLLD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1654a72f104"},
// Masked align.
{"VALIGND", "VALIGND", []Operand{Imm(12), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f36d4b03e10c"},
// Masked conversion and extract.
{"VCVTQQ2PD.Z", "VCVTQQ2PD.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1fecae6d9"},
{"VEXTRACTI64X4", "VEXTRACTI64X4", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f3fd4a3bcb01"},
// Masked moves: K sits between the register and memory operands.
{"VMOVDQU8 store", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "K3"), Ptr(SI, 0, 64)}, "62f17f4b7f0e"},
{"VMOVDQU32 load", "VMOVDQU32", []Operand{Ptr(SI, 0, 64), vreg(t, "K4"), vreg(t, "Z1")}, "62f17e4c6f0e"},
{"VMOVDQU32 store", "VMOVDQU32", []Operand{vreg(t, "Z1"), vreg(t, "K4"), Ptr(DI, 0, 64)}, "62f17e4c7f0f"},
// Masked comparison with a K destination: dst K1, mask K2.
{"VPCMPEQD k-dst+mask", "VPCMPEQD", []Operand{vreg(t, "Z0"), vreg(t, "Z3"), vreg(t, "K2"), vreg(t, "K1")}, "62f1654a76c8"},
// Masked floating point: packed double, the scalar SD/SS forms (which
// exist under EVEX only for masked and zeroing use) and conversions.
{"VSUBPD.Z", "VSUBPD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f1edcb5ce1"},
{"VADDSD merge", "VADDSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K3"), vreg(t, "X4")}, "62f1ef0b58e1"},
{"VSUBSD.Z", "VSUBSD.Z", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K5"), vreg(t, "X3")}, "62f1ef8d5cd9"},
{"VADDSS merge", "VADDSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f16e0958d9"},
{"VCVTPD2DQ merge", "VCVTPD2DQ", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f1ff4ae6d9"},
{"VCVTTPD2DQ.Z", "VCVTTPD2DQ.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f1fdcae6d9"},
{"VCVTDQ2PS.Z", "VCVTDQ2PS.Z", []Operand{vreg(t, "Z1"), vreg(t, "K4"), vreg(t, "Z2")}, "62f17ccc5bd1"},
{"VCVTDQ2PD merge", "VCVTDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "K2"), vreg(t, "X3")}, "62f17e0ae6d9"},
{"VCVTDQ2PD.Z", "VCVTDQ2PD.Z", []Operand{vreg(t, "Y1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f17ecae6d1"},
{"VCVTPS2PD.Z", "VCVTPS2PD.Z", []Operand{vreg(t, "Y1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f17ccb5ad1"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
want := c.mnem
if i := len(want) - 2; i > 0 && want[i:] == ".Z" {
want = want[:i]
}
if inst.Op.String() != want {
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
}
}
// Error cases.
bad := []struct {
name string
mnem string
ops []Operand
}{
{"zeroing without mask", "VPADDD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"K0 mask", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K0"), vreg(t, "Z3")}},
{"two masks", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "K2"), vreg(t, "Z3")}},
{".Z on VEX-only", "VPSHUFD.Z", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}},
{"broadcast unsupported", "VPXORD.BCST", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"rounding unsupported", "VPXORD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"bcst with rounding", "VADDPD.BCST.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"Z not last", "VADDPD.Z.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"duplicate suffix", "VADDPD.Z.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"KMOVW.Z", "KMOVW.Z", []Operand{vreg(t, "K1"), vreg(t, "K2")}},
}
for _, c := range bad {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set — ternary
// logic, lane shuffles/inserts/extracts, compares with a K destination,
// permutes, the wider integer families, expand/compress, broadcasts,
// rotates and word shifts, the opmask instructions, the EVEX suffixes
// (rounding/SAE/broadcast) and the aligned/scalar moves — byte for byte
// against the Go assembler.
func TestEvexExtendedGroundTruth(t *testing.T) {
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// Ternary logic and lane shuffles (NDS + imm8).
{"VPTERNLOGD", "VPTERNLOGD", []Operand{Imm(0xE8), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4825d9e8"},
{"VPTERNLOGQ", "VPTERNLOGQ", []Operand{Imm(0x96), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4825d996"},
{"VSHUFI32X4", "VSHUFI32X4", []Operand{Imm(0x4E), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "62f36d2843d94e"},
{"VSHUFF64X2", "VSHUFF64X2", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4823d901"},
{"VPALIGNR", "VPALIGNR", []Operand{Imm(7), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d480fd907"},
// Permutes.
{"VPERMB", "VPERMB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d488dd9"},
{"VPERMW", "VPERMW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed488dd9"},
{"VPERMI2D", "VPERMI2D", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4876d9"},
{"VPERMT2PD", "VPERMT2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed487fd9"},
// Compare with a K destination (and an immediate predicate).
{"VCMPPD", "VCMPPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3")}, "62f1ed48c2d904"},
{"VCMPPS", "VCMPPS", []Operand{Imm(0), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "K4")}, "62f16c28c2e100"},
{"VCMPSD", "VCMPSD", []Operand{Imm(17), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K5")}, "62f1ef08c2e911"},
// Rounding / SAE / broadcast suffixes.
{"VADDPD.RN_SAE", "VADDPD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed1858d9"},
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
// Packed single arithmetic (same opcodes, no mandatory prefix) —
// ZMM, YMM and XMM widths, rounding and broadcast.
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
{"VMAXPS", "VMAXPS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e85fd9"},
{"VDIVPS.RD_SAE", "VDIVPS.RD_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c385ed9"},
{"VADDPS.BCST", "VADDPS.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f174585810"},
// Compress / expand.
{"VCOMPRESSPD", "VCOMPRESSPD", []Operand{vreg(t, "Z1"), mem64(DI)}, "62f2fd488a0f"},
{"VEXPANDPS", "VEXPANDPS", []Operand{mem64(SI), vreg(t, "Y2")}, "62f27d288816"},
{"VPCOMPRESSD.Z", "VPCOMPRESSD.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), mem64(DI)}, "62f27dca8b0f"},
// Broadcasts.
{"VPBROADCASTB gpr", "VPBROADCASTB", []Operand{BX, vreg(t, "Z1")}, "62f27d487acb"},
{"VPBROADCASTW mem", "VPBROADCASTW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d487910"},
{"VBROADCASTSS", "VBROADCASTSS", []Operand{mem64(AX), vreg(t, "Y3")}, "c4e27d1818"},
{"VBROADCASTSD", "VBROADCASTSD", []Operand{mem64(AX), vreg(t, "Z4")}, "62f2fd481920"},
// Wider integer families.
{"VPMADDWD", "VPMADDWD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48f5d9"},
{"VPMADDUBSW", "VPMADDUBSW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4804d9"},
{"VPMULHUW", "VPMULHUW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48e4d9"},
{"VPSLLVW", "VPSLLVW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4812d9"},
{"VPACKSSWB", "VPACKSSWB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d4863d9"},
{"VPACKUSDW", "VPACKUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482bd9"},
// Absolute values and replicating moves.
{"VPABSD", "VPABSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d481ed1"},
{"VPABSQ mem", "VPABSQ", []Operand{mem64(AX), vreg(t, "Z2")}, "62f2fd481f10"},
{"VMOVSLDUP", "VMOVSLDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa12d1"},
{"VMOVSHDUP", "VMOVSHDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e4816d1"},
// Rotates and word/qword shifts.
{"VPROLD", "VPROLD", []Operand{Imm(5), vreg(t, "Z1"), vreg(t, "Z2")}, "62f16d4872c905"},
{"VPRORQ", "VPRORQ", []Operand{Imm(63), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4872c13f"},
{"VPSLLW", "VPSLLW", []Operand{Imm(9), vreg(t, "X1"), vreg(t, "X2")}, "c5e971f109"},
{"VPSRLQ", "VPSRLQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4873d103"},
// Opmask instructions (VEX-encoded, the width in the L/W/pp bits).
{"KANDW", "KANDW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec41d9"},
{"KORD", "KORD", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d545f4"},
{"KXNORQ", "KXNORQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec46d9"},
{"KNOTB", "KNOTB", []Operand{vreg(t, "K4"), vreg(t, "K5")}, "c5f944ec"},
{"KUNPCKBW", "KUNPCKBW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ed4bd9"},
{"KSHIFTLW", "KSHIFTLW", []Operand{Imm(2), vreg(t, "K1"), vreg(t, "K2")}, "c4e3f932d102"},
{"KADDQ", "KADDQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec4ad9"},
{"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"},
{"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"},
{"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"},
// Lane extract / insert.
{"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"},
{"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"},
{"VINSERTF32X8", "VINSERTF32X8", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d481ad901"},
{"VINSERTI64X4", "VINSERTI64X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed483ad901"},
// Aligned moves and the scalar single move.
{"VMOVAPS", "VMOVAPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4829ca"},
{"VMOVDQA64 mem", "VMOVDQA64", []Operand{mem64(AX), vreg(t, "Z2")}, "62f1fd486f10"},
{"VMOVSS mem", "VMOVSS", []Operand{mem64(AX), vreg(t, "X2")}, "c5fa1010"},
// Conversions and extending/narrowing moves.
{"VCVTPS2DQ", "VCVTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17d485bd1"},
{"VCVTTPS2DQ", "VCVTTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e485bd1"},
{"VPMOVZXBW", "VPMOVZXBW", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d30d1"},
{"VPMOVSXBW mem", "VPMOVSXBW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d482010"},
{"VPMOVWB", "VPMOVWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4830ca"},
{"VPMOVQB", "VPMOVQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4832ca"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
want := c.mnem
if i := strings.IndexByte(want, '.'); i > 0 {
want = want[:i]
}
if inst.Op.String() != want {
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
}
}
}
// TestEvexErrors checks the EVEX-specific error paths.
func TestEvexErrors(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
}{
{"NDS arity", "VPXORD", []Operand{vreg(t, "Z0"), vreg(t, "Z1")}},
{"KMOVW arity", "KMOVW", []Operand{vreg(t, "K1")}},
{"KMOVW no K", "KMOVW", []Operand{AX, CX}},
{"VMOVUPD Z gpr", "VMOVUPD", []Operand{AX, vreg(t, "Z1")}},
{"broadcast src", "VPBROADCASTD", []Operand{Imm(1), vreg(t, "Z1")}},
{"VPMOVDW src", "VPMOVDW", []Operand{AX, vreg(t, "Y0")}},
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
// VEX-only mnemonics reject registers only EVEX can encode.
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
// kernel — all functions plus the file-global idx16 constant — and checks
// that the static-symbol load resolves to the right bytes in the image.
// Skipped when the sibling repository is not checked out.
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx512_amd64.s"
if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit")
}
src, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
if len(img.Funcs) != 10 {
t.Errorf("functions = %d, want 10", len(img.Funcs))
}
// idx16 as the DATA directives define it: dwords 1..16.
idx := make([]byte, 0, 64)
for i := 1; i <= 16; i++ {
idx = append(idx, byte(i), 0, 0, 0)
}
image := img.Bytes()
base := img.Symbols["idx16"]
if base == 0 {
t.Fatal("idx16 not laid out")
}
if got := image[base : base+64]; hexCompact(got) != hexCompact(idx) {
t.Errorf("idx16 contents %x, want %x", got, idx)
}
// The VMOVDQU32 idx16(SB), Z13 load (62 71 7e 48 6f 2d + rel32) must
// resolve to idx16 within the image.
loads := 0
for _, fn := range img.Funcs {
code := img.Code[fn.Offset : fn.Offset+fn.Size]
pat := []byte{0x62, 0x71, 0x7e, 0x48, 0x6f, 0x2d}
for pos := 0; ; {
i := indexOf(code[pos:], pat)
if i < 0 {
break
}
i += pos
rel := int32(uint32(code[i+6]) | uint32(code[i+7])<<8 | uint32(code[i+8])<<16 | uint32(code[i+9])<<24)
target := fn.Offset + i + 10 + int(rel)
if target != base {
t.Errorf("%s: idx16 load at +%d targets 0x%x, want 0x%x", fn.Name, i, target, base)
}
loads++
pos = i + 10
}
}
if loads != 1 {
t.Errorf("idx16 loads found = %d, want 1", loads)
}
}
// hexCompact renders bytes as a lowercase hex string without separators.
func hexCompact(b []byte) string {
const hexdig = "0123456789abcdef"
out := make([]byte, len(b)*2)
for i, c := range b {
out[i*2] = hexdig[c>>4]
out[i*2+1] = hexdig[c&0xf]
}
return string(out)
}
// indexOf returns the index of the first occurrence of pat in b, or -1.
func indexOf(b, pat []byte) int {
for i := 0; i+len(pat) <= len(b); i++ {
j := 0
for j < len(pat) && b[i+j] == pat[j] {
j++
}
if j == len(pat) {
return i
}
}
return -1
}
+453
View File
@@ -0,0 +1,453 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"fmt"
"os"
"os/exec"
"path/filepath"
"sync"
)
// This file emits GOOBJ — the Go toolchain's object format, which cmd/link
// consumes directly — so gasm-assembled functions drop into a go build
// without the Go assembler. The layout follows cmd/internal/goobj: a
// toolchain preamble ("go object ...\n!\n"), the go120ld header with its
// block offsets, a string table, symbol definitions, the relocation /
// aux / data index arrays, and the three blocks themselves.
//
// The object carries what the linker requires of an assembly object: the
// functions (non-package symbols, as cmd/asm emits them), the GLOBL data,
// one FuncInfo per function, and the pc-value tables (pcsp, pcfile,
// pcline, pcinline). DWARF and the implicit funcdata symbols are omitted;
// the linker fills their defaults.
// GOOBJ block indices (cmd/internal/goobj).
const (
blkAutolib = iota
blkPkgIdx
blkFile
blkSymdef
blkHashed64def
blkHasheddef
blkNonpkgdef
blkNonpkgref
blkRefFlags
blkHash64
blkHash
blkRelocIdx
blkAuxIdx
blkDataIdx
blkReloc
blkAux
blkData
blkRefName
blkEnd
)
// Symbol kinds used by assembly objects (cmd/internal/objabi).
const (
kindSTEXT = 1
kindSRODATA = 3
kindSDATA = 7
)
// Symbol flags (cmd/internal/goobj).
const (
symFlagDupok = 0x01
symFlagNoSplit = 0x10
symFlag2Link = 0x10 // asm objects flag every named symbol as linkname
symABIStatic = 0xffff
)
// Aux entry types (cmd/internal/goobj).
const (
auxFuncInfo = 1
auxPcsp = 7
auxPcfile = 8
auxPcline = 9
auxPcinline = 10
)
// FuncInfo flags (internal/abi).
const (
funcFlagSPWrite = 2
funcFlagAsm = 4
)
// Relocation types (cmd/internal/objabi).
const relocPCRel = 14
// Special package indices for symbol references.
const (
pkgIdxNone = 0x7fffffff
pkgIdxSelf = 0x7ffffffb
)
const goobjMagic = "\x00go120ld"
// goSym is one symbol definition under construction.
type goSym struct {
name string
abi uint16
typ uint8
flag uint8
flag2 uint8
size uint32
align uint32
}
func (s goSym) append(b []byte, strOff map[string]uint32) []byte {
b = binary.LittleEndian.AppendUint32(b, uint32(len(s.name)))
b = binary.LittleEndian.AppendUint32(b, strOff[s.name])
b = binary.LittleEndian.AppendUint16(b, s.abi)
b = append(b, s.typ, s.flag, s.flag2)
b = binary.LittleEndian.AppendUint32(b, s.size)
return binary.LittleEndian.AppendUint32(b, s.align)
}
// GOObject returns the image as a GOOBJ object file for the given package
// path (the linker qualifies the exported symbols with it, the way cmd/asm
// does with its -p flag). srcPath names the source file recorded in the
// object's file table and line tables. The toolchain's object preamble is
// captured from the installed go tool asm, so the output links with the
// toolchain it was produced on — exactly like a real assembly object.
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
if pkgPath == "" {
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
}
pre, err := toolchainObjectPreamble()
if err != nil {
return nil, err
}
// The symbol tables. Package definitions: the GLOBL symbols, then one
// anonymous FuncInfo symbol per function. Non-package definitions: the
// pc-value tables and the functions themselves, as cmd/asm lays them
// out. defIdx maps a GLOBL's bare name to its definition index for the
// relocations; fnNpIdx maps a function to its non-package index.
var defs []goSym
var defData [][]byte
defIdx := map[string]int{}
for _, d := range img.DataSyms {
name := d.Name
if !d.Static {
name = pkgPath + "." + name
}
typ := uint8(kindSDATA)
if d.Rodata {
typ = kindSRODATA
}
flag := uint8(0)
if d.Dupok {
flag = symFlagDupok
}
abi := uint16(0)
if d.Static {
abi = symABIStatic
}
defIdx[d.Name] = len(defs)
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)})
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
}
fnFiIdx := make([]int, len(img.Funcs))
for i := range img.Funcs {
data := marshalFuncInfo(img.Funcs[i])
fnFiIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))})
defData = append(defData, data)
}
type npSym struct {
sym goSym
data []byte
}
var nps []npSym
type pcRefs struct{ sp, file, line, inl int }
pcIdx := make([]pcRefs, len(img.Funcs))
fnNpIdx := make([]int, len(img.Funcs))
for i, fn := range img.Funcs {
tables := []struct {
data []byte
dst *int
}{
{pcspTable(fn), &pcIdx[i].sp},
{pcValueFlat(0, fn.Size), &pcIdx[i].file},
{pcValueFlat(int32(fn.Line), fn.Size), &pcIdx[i].line},
{pcValueFlat(-1, fn.Size), &pcIdx[i].inl},
}
for _, t := range tables {
*t.dst = len(nps)
nps = append(nps, npSym{
sym: goSym{typ: kindSRODATA, size: uint32(len(t.data)), align: 1},
data: t.data,
})
}
name := fn.Name
abi := uint16(0)
if fn.Static {
abi = symABIStatic
} else {
name = pkgPath + "." + name
}
flag := uint8(0)
if fn.NoSplit {
flag |= symFlagNoSplit
}
fnNpIdx[i] = len(nps)
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
// The linker writes the resolved displacement into the field;
// leave it zero, as cmd/asm's object does.
if r.Off >= 0 && r.Off+4 <= len(code) {
code[r.Off], code[r.Off+1], code[r.Off+2], code[r.Off+3] = 0, 0, 0, 0
}
}
nps = append(nps, npSym{
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
data: code,
})
}
// Relocations, per defined symbol in definition order (package defs,
// then non-package defs). Only file-local GLOBL references resolve;
// external symbols need the import machinery of a later increment.
nsyms := len(defs) + len(nps)
symRelocs := make([][]byte, nsyms) // flat 23-byte records
for i, fn := range img.Funcs {
si := len(defs) + fnNpIdx[i]
for _, r := range fn.Relocs {
if r.External {
return nil, fmt.Errorf("GOOBJ emission: external symbol %q is not supported yet", r.Name)
}
di, ok := defIdx[r.Name]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
}
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = 4 // field width
binary.LittleEndian.PutUint16(rec[5:], relocPCRel)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
symRelocs[si] = append(symRelocs[si], rec[:]...)
}
}
// Aux entries per function: FuncInfo, then the four pc tables.
// References into the non-package table use pkgIdxNone.
symAux := make([][]byte, nsyms)
for i := range img.Funcs {
si := len(defs) + fnNpIdx[i]
aux := func(typ uint8, pkg, idx uint32) {
var rec [9]byte
rec[0] = typ
binary.LittleEndian.PutUint32(rec[1:], pkg)
binary.LittleEndian.PutUint32(rec[5:], idx)
symAux[si] = append(symAux[si], rec[:]...)
}
aux(auxFuncInfo, pkgIdxSelf, uint32(fnFiIdx[i]))
aux(auxPcsp, pkgIdxNone, uint32(len(defs)+pcIdx[i].sp))
aux(auxPcfile, pkgIdxNone, uint32(len(defs)+pcIdx[i].file))
aux(auxPcline, pkgIdxNone, uint32(len(defs)+pcIdx[i].line))
aux(auxPcinline, pkgIdxNone, uint32(len(defs)+pcIdx[i].inl))
}
// The string table. Absolute offsets: it starts right after the
// 96-byte header (magic, fingerprint, flags, the 19 block offsets).
const headerSize = 8 + 8 + 4 + 4*(blkEnd+1)
strTab := []byte{}
strOff := map[string]uint32{}
addStr := func(s string) {
if _, ok := strOff[s]; ok {
return
}
strOff[s] = uint32(headerSize + len(strTab))
strTab = append(strTab, s...)
}
addStr("")
addStr(srcPath)
for _, s := range defs {
addStr(s.name)
}
for _, s := range nps {
addStr(s.sym.name)
}
stringRef := func(b []byte, s string) []byte {
b = binary.LittleEndian.AppendUint32(b, uint32(len(s)))
return binary.LittleEndian.AppendUint32(b, strOff[s])
}
// Serialise the block bodies.
var symdefBlk, npdefBlk []byte
for _, s := range defs {
symdefBlk = s.append(symdefBlk, strOff)
}
for _, s := range nps {
npdefBlk = s.sym.append(npdefBlk, strOff)
}
pkgIdxBlk := stringRef(nil, "") // index 0: the dummy invalid package
fileBlk := stringRef(nil, srcPath)
var relocBlk, auxBlk, dataBlk []byte
relocIdxBlk := make([]byte, 0, 4*(nsyms+1))
auxIdxBlk := make([]byte, 0, 4*(nsyms+1))
dataIdxBlk := make([]byte, 0, 4*(nsyms+1))
var nr, na, nd uint32
for si := 0; si < nsyms; si++ {
relocIdxBlk = binary.LittleEndian.AppendUint32(relocIdxBlk, nr)
auxIdxBlk = binary.LittleEndian.AppendUint32(auxIdxBlk, na)
dataIdxBlk = binary.LittleEndian.AppendUint32(dataIdxBlk, nd)
relocBlk = append(relocBlk, symRelocs[si]...)
auxBlk = append(auxBlk, symAux[si]...)
var d []byte
if si < len(defData) {
d = defData[si]
} else {
d = nps[si-len(defData)].data
}
dataBlk = append(dataBlk, d...)
nr += uint32(len(symRelocs[si])) / 23
na += uint32(len(symAux[si])) / 9
nd += uint32(len(d))
}
relocIdxBlk = binary.LittleEndian.AppendUint32(relocIdxBlk, nr)
auxIdxBlk = binary.LittleEndian.AppendUint32(auxIdxBlk, na)
dataIdxBlk = binary.LittleEndian.AppendUint32(dataIdxBlk, nd)
blocks := [blkEnd][]byte{
blkPkgIdx: pkgIdxBlk,
blkFile: fileBlk,
blkSymdef: symdefBlk,
blkNonpkgdef: npdefBlk,
blkRelocIdx: relocIdxBlk,
blkAuxIdx: auxIdxBlk,
blkDataIdx: dataIdxBlk,
blkReloc: relocBlk,
blkAux: auxBlk,
blkData: dataBlk,
}
// Assemble the payload: header (offsets filled once known), string
// table, blocks in order.
payload := make([]byte, headerSize)
copy(payload, goobjMagic)
// The fingerprint stays zero, as cmd/asm leaves it.
binary.LittleEndian.PutUint32(payload[16:], 4) // ObjFlagFromAssembly
off := uint32(headerSize + len(strTab))
for i := 0; i < blkEnd; i++ {
binary.LittleEndian.PutUint32(payload[20+4*i:], off)
off += uint32(len(blocks[i]))
}
binary.LittleEndian.PutUint32(payload[20+4*blkEnd:], off)
payload = append(payload, strTab...)
for _, blk := range blocks {
payload = append(payload, blk...)
}
out := make([]byte, 0, len(pre)+len(payload))
out = append(out, pre...)
return append(out, payload...), nil
}
// marshalFuncInfo serialises a function's goobj.FuncInfo: sizes, flags,
// start line, the one-element file table and an empty inline tree.
func marshalFuncInfo(fn FuncLayout) []byte {
flag := uint8(funcFlagAsm)
if fn.SPWrite {
flag |= funcFlagSPWrite
}
b := make([]byte, 0, 28)
b = binary.LittleEndian.AppendUint32(b, uint32(fn.Args))
b = binary.LittleEndian.AppendUint32(b, uint32(fn.Frame))
b = append(b, 0, flag, 0, 0) // FuncID normal, flags, padding
b = binary.LittleEndian.AppendUint32(b, uint32(int32(fn.Line)))
b = binary.LittleEndian.AppendUint32(b, 1) // one file
b = binary.LittleEndian.AppendUint32(b, 0) // file index 0
b = binary.LittleEndian.AppendUint32(b, 0) // no inline tree
return b
}
// pcValueFlat encodes a pc-value table holding v over the whole function.
func pcValueFlat(v int32, size int) []byte {
// The table is delta-encoded from an implicit value of -1: a varint
// value delta, an unsigned pc delta to the end, and a zero terminator.
out := binary.AppendVarint(nil, int64(v)+1)
out = binary.AppendUvarint(out, uint64(size))
return append(out, 0)
}
// pcspTable encodes the stack-adjustment table: the SP delta in effect at
// every pc, from the function's prologue and epilogue boundaries.
func pcspTable(fn FuncLayout) []byte {
if len(fn.Spadj) == 0 {
return pcValueFlat(0, fn.Size)
}
pts := make([]SpadjStep, 0, len(fn.Spadj)+1)
pts = append(pts, SpadjStep{PC: 0, Value: 0})
pts = append(pts, fn.Spadj...)
out := binary.AppendVarint(nil, int64(pts[0].Value)+1)
cur, old := pts[0].PC, pts[0].Value
for _, p := range pts[1:] {
out = binary.AppendUvarint(out, uint64(p.PC-cur))
out = binary.AppendVarint(out, int64(p.Value-old))
cur, old = p.PC, p.Value
}
out = binary.AppendUvarint(out, uint64(fn.Size-cur))
return append(out, 0)
}
// toolchainObjectPreamble returns the "go object ...\n!\n" header the
// installed go tool asm writes, captured by assembling a one-instruction
// probe. The linker compares this string verbatim against its own, so it
// must come from the toolchain itself, not be reconstructed.
var (
preambleOnce sync.Once
preamble []byte
preambleErr error
)
func toolchainObjectPreamble() ([]byte, error) {
preambleOnce.Do(func() {
goBin, err := exec.LookPath("go")
if err != nil {
preambleErr = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
return
}
dir, err := os.MkdirTemp("", "gasm-preamble")
if err != nil {
preambleErr = err
return
}
defer os.RemoveAll(dir)
src := filepath.Join(dir, "probe_amd64.s")
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
preambleErr = err
return
}
obj := filepath.Join(dir, "probe.o")
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
cmd.Env = append(os.Environ(), "GOARCH=amd64")
if out, err := cmd.CombinedOutput(); err != nil {
preambleErr = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
return
}
data, err := os.ReadFile(obj)
if err != nil {
preambleErr = err
return
}
i := bytes.Index(data, []byte("\n!\n"))
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
preambleErr = fmt.Errorf("unrecognised assembler object layout")
return
}
preamble = data[:i+3]
})
return preamble, preambleErr
}
+477
View File
@@ -0,0 +1,477 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// goobjView is a minimal parsed view of a GOOBJ payload, enough to check
// the emitter's output block by block.
type goobjView struct {
t *testing.T
b []byte
offs [blkEnd + 1]uint32
strOff uint32
}
func openGoobj(t *testing.T, data []byte) *goobjView {
t.Helper()
i := bytes.Index(data, []byte(goobjMagic))
if i < 0 {
t.Fatal("no GOOBJ magic in output")
}
v := &goobjView{t: t, b: data[i:], strOff: uint32(i + 96)}
for j := 0; j <= blkEnd; j++ {
v.offs[j] = binary.LittleEndian.Uint32(v.b[20+4*j:])
}
return v
}
func (v *goobjView) blk(i int) []byte { return v.b[v.offs[i]:v.offs[i+1]] }
func (v *goobjView) str(off, ln uint32) string {
return string(v.b[off : off+ln])
}
type goobjSymView struct {
name string
abi uint16
typ uint8
flag uint8
flag2 uint8
size uint32
align uint32
}
func (v *goobjView) syms(i int) []goobjSymView {
var out []goobjSymView
for x := v.blk(i); len(x) >= 21; x = x[21:] {
le := binary.LittleEndian
out = append(out, goobjSymView{
name: v.str(le.Uint32(x[4:]), le.Uint32(x[0:])),
abi: le.Uint16(x[8:]),
typ: x[10],
flag: x[11],
flag2: x[12],
size: le.Uint32(x[13:]),
align: le.Uint32(x[17:]),
})
}
return out
}
// TestGOObjectStructure checks the emitted object's blocks against the
// ground truth captured from go tool asm: the symbol tables, the FuncInfo
// contents, the pc-value tables, the relocation and the aux wiring.
func TestGOObjectStructure(t *testing.T) {
f, errs := parser.Parse("t_amd64.s", `
#include "textflag.h"
TEXT ·addq(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), CX
ADDQ CX, AX
MOVQ AX, ret+16(FP)
RET
TEXT ·loadmask(SB), NOSPLIT, $0-8
VMOVDQU mask<>(SB), X0
VPMOVMSKB X0, AX
MOVQ AX, ret+0(FP)
RET
GLOBL mask<>(SB), RODATA, $16
DATA mask<>+0(SB)/8, $0x0807060504030201
DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
obj, err := img.GOObject("testpkg", "t_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
v := openGoobj(t, obj)
if flags := binary.LittleEndian.Uint32(v.b[16:]); flags != 4 {
t.Errorf("flags = %#x, want ObjFlagFromAssembly (4)", flags)
}
// Package defs: the static GLOBL, then one anonymous FuncInfo per
// function.
defs := v.syms(blkSymdef)
if len(defs) != 3 {
t.Fatalf("symdefs = %d, want 3", len(defs))
}
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
t.Errorf("mask symbol = %+v", defs[0])
}
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
t.Errorf("funcinfo symbol = %+v", defs[1])
}
// Non-package defs: four pc tables and the function, per function.
nps := v.syms(blkNonpkgdef)
if len(nps) != 10 {
t.Fatalf("nonpkgdefs = %d, want 10", len(nps))
}
fn := nps[4]
if fn.name != "testpkg.addq" || fn.typ != kindSTEXT || fn.flag != symFlagNoSplit || fn.size != 19 {
t.Errorf("addq symbol = %+v", fn)
}
for i, s := range []int{0, 1, 2, 3, 5, 6, 7, 8} {
if nps[s].typ != kindSRODATA || nps[s].align != 1 || nps[s].name != "" {
t.Errorf("pc table %d = %+v", i, nps[s])
}
}
// FuncInfo: args 24, FuncFlag Asm, one file, no inline tree.
le := binary.LittleEndian
data := v.blk(blkData)
fi := data[16:44]
if le.Uint32(fi[0:]) != 24 || le.Uint32(fi[4:]) != 0 || fi[8] != 0 || fi[9] != funcFlagAsm ||
le.Uint32(fi[16:]) != 1 || le.Uint32(fi[20:]) != 0 || le.Uint32(fi[24:]) != 0 {
t.Errorf("funcinfo bytes %x", fi)
}
// pcsp: a flat zero over the whole function (zero-frame NOSPLIT).
if got := data[72:75]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
t.Errorf("pcsp = %x, want 021300", got)
}
// pcinline: a flat -1.
if got := data[81:84]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) {
t.Errorf("pcinline = %x, want 001300", got)
}
// The one relocation: R_PCREL, four bytes wide, against the GLOBL,
// with the field in the function code left zero. The loadmask code's
// offset comes from the data index (symbol 3 defs + 9 non-package).
relocs := v.blk(blkReloc)
if len(relocs) != 23 {
t.Fatalf("relocs = %d bytes, want one 23-byte entry", len(relocs))
}
off := int32(le.Uint32(relocs[0:]))
if off != 4 || relocs[4] != 4 || le.Uint16(relocs[5:]) != relocPCRel ||
le.Uint64(relocs[7:]) != 0 || le.Uint32(relocs[15:]) != pkgIdxSelf || le.Uint32(relocs[19:]) != 0 {
t.Errorf("reloc = %x", relocs)
}
didx := v.blk(blkDataIdx)
lm := le.Uint32(didx[4*(3+9):])
code := data[lm : lm+18]
if !bytes.Equal(code[4:8], []byte{0, 0, 0, 0}) {
t.Errorf("relocated field = %x, want zeroed", code[4:8])
}
// Aux wiring: FuncInfo (package symbol), then the four pc tables
// (non-package symbols).
auxs := v.blk(blkAux)
if len(auxs) != 2*5*9 {
t.Fatalf("aux = %d bytes, want 10 entries", len(auxs))
}
wantAux := []struct {
typ uint8
pkg uint32
idx uint32
}{
{auxFuncInfo, pkgIdxSelf, 1},
{auxPcsp, pkgIdxNone, uint32(len(defs) + 0)},
{auxPcfile, pkgIdxNone, uint32(len(defs) + 1)},
{auxPcline, pkgIdxNone, uint32(len(defs) + 2)},
{auxPcinline, pkgIdxNone, uint32(len(defs) + 3)},
{auxFuncInfo, pkgIdxSelf, 2},
{auxPcsp, pkgIdxNone, uint32(len(defs) + 5)},
{auxPcfile, pkgIdxNone, uint32(len(defs) + 6)},
{auxPcline, pkgIdxNone, uint32(len(defs) + 7)},
{auxPcinline, pkgIdxNone, uint32(len(defs) + 8)},
}
for i, w := range wantAux {
e := auxs[i*9:]
if e[0] != w.typ || le.Uint32(e[1:]) != w.pkg || le.Uint32(e[5:]) != w.idx {
t.Errorf("aux[%d] = {%d,%d,%d}, want {%d,%d,%d}", i, e[0], le.Uint32(e[1:]), le.Uint32(e[5:]), w.typ, w.pkg, w.idx)
}
}
}
// decodePCValues decodes a pc-value table into (pc, value) steps. The
// table ends with a final unsigned pc delta covering the rest of the
// function, followed by a zero byte that carries no value delta.
func decodePCValues(b []byte) (pcs, vals []int64) {
val, n := binary.Varint(b)
b = b[n:]
val-- // the first delta is against the implicit -1
var pc int64
pcs = append(pcs, pc)
vals = append(vals, val)
for {
pcd, n := binary.Uvarint(b)
b = b[n:]
if pcd == 0 { // zero pc delta terminates the table
break
}
pc += int64(pcd)
if len(b) == 1 && b[0] == 0 { // final coverage, no value change
break
}
vd, n := binary.Varint(b)
b = b[n:]
val += vd
pcs = append(pcs, pc)
vals = append(vals, val)
}
return pcs, vals
}
// TestGOObjectPcspFrame checks the pcsp table of a frame-pointer function:
// the prologue raises the stack delta to 8+frame, the RET's epilogue
// restores it to zero.
func TestGOObjectPcspFrame(t *testing.T) {
f, errs := parser.Parse("frame_amd64.s", `
#include "textflag.h"
TEXT ·framed(SB), NOSPLIT, $8-0
MOVQ BP, AX
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
fn := img.Funcs[0]
pcs, vals := decodePCValues(pcspTable(fn))
// Prologue: PUSHQ BP (1 byte, +8), MOVQ SP, BP (3 bytes, no change),
// SUBQ $8, SP (4 bytes, +16 in total); the RET's epilogue unwinds
// ADDQ $8, SP (+8) then POPQ BP (0).
wantPCs := []int64{0, 1, 8}
wantVals := []int64{0, 8, 16}
if len(pcs) < len(wantPCs) {
t.Fatalf("pcsp pcs = %v vals = %v", pcs, vals)
}
for i := range wantPCs {
if pcs[i] != wantPCs[i] || vals[i] != wantVals[i] {
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d) — all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
}
}
// The last two steps unwind the epilogue to zero.
n := len(pcs)
if vals[n-1] != 0 || vals[n-2] != 8 {
t.Errorf("epilogue steps = %v %v, want …8, 0", pcs, vals)
}
// The table covers the whole function.
if last := pcs[n-1]; last >= int64(fn.Size) {
t.Errorf("last pc %d beyond function size %d", last, fn.Size)
}
}
// TestGOObjectExternalRejected checks that a reference to a symbol no GLOBL
// defines is reported: GOOBJ emission resolves only file-local symbols so
// far.
func TestGOObjectExternalRejected(t *testing.T) {
f, errs := parser.Parse("ext_amd64.s", `
#include "textflag.h"
TEXT ·useext(SB), NOSPLIT, $0-8
MOVQ elsewhere(SB), AX
MOVQ AX, ret+0(FP)
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
if _, err := img.GOObject("p", "ext_amd64.s"); err == nil || !strings.Contains(err.Error(), "external") {
t.Errorf("error = %v, want an external-symbol error", err)
}
}
// TestGOObjectLinkAndRun is the end-to-end check: assemble the test
// functions to a GOOBJ, swap it into a go build in place of the toolchain's
// assembly object, link, and run — the output must match the baseline
// binary the Go assembler produced. Skipped when no Go toolchain is
// available.
func TestGOObjectLinkAndRun(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
const asmSrc = `
#include "textflag.h"
TEXT ·addq(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), CX
ADDQ CX, AX
MOVQ AX, ret+16(FP)
RET
TEXT ·loadmask(SB), NOSPLIT, $0-8
VMOVDQU mask<>(SB), X0
VPMOVMSKB X0, AX
MOVQ AX, ret+0(FP)
RET
GLOBL mask<>(SB), RODATA, $16
DATA mask<>+0(SB)/8, $0x0807060504030201
DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
`
const mainSrc = `package main
func addq(a, b int64) int64
func loadmask() int64
func main() {
println(addq(41, 1))
println(loadmask())
}
`
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module goobjtest\n\ngo 1.26\n"), 0o644); err != nil {
t.Fatal(err)
}
// Baseline build with the toolchain's assembler; keep the work
// directory and the commands the build used.
cmd := exec.Command(goBin, "build", "-x", "-work", "-o", "app", ".")
cmd.Dir = dir
buildLog, err := cmd.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work string
var asmObj, pkgArch, linkLine string
for _, line := range strings.Split(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "-o ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || pkgArch == "" || linkLine == "" {
t.Fatalf("could not locate the build steps:\n%s", buildLog)
}
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
// The baseline's answer.
baseOut, err := exec.Command(filepath.Join(dir, "app")).CombinedOutput()
if err != nil {
t.Fatalf("run baseline: %v\n%s", err, baseOut)
}
// Assemble the same source with gasm and swap the object in.
pf, perrs := parser.Parse(filepath.Join(dir, "main_amd64.s"), asmSrc)
if len(perrs) > 0 {
t.Fatalf("parse: %v", perrs)
}
img, err := AssembleFile(pf)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
obj, err := img.GOObject("main", filepath.Join(dir, "main_amd64.s"))
if err != nil {
t.Fatalf("GOObject: %v", err)
}
if err := os.WriteFile(asmObj, obj, 0o644); err != nil {
t.Fatal(err)
}
// Rebuild the package archive with our object in place of the
// toolchain's (go tool pack has no replace-in-place that dedupes, so
// extract, substitute and repack).
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract.Dir = membersDir
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for _, m := range strings.Fields(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, filepath.Join(membersDir, m))
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
// Link with our archive. The link line carries a GOROOT assignment
// and $WORK placeholders; run it through the shell with the
// GOEXPERIMENT the toolchain expects (the linker compares the object
// header against its own, experiments included).
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
link := exec.Command("sh", "-c", linkLine)
link.Dir = dir
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)))
if out, err := link.CombinedOutput(); err != nil {
t.Fatalf("link with gasm object: %v\n%s", err, out)
}
got, err := exec.Command(filepath.Join(dir, "app2")).CombinedOutput()
if err != nil {
t.Fatalf("run gasm-linked binary: %v\n%s", err, got)
}
if !bytes.Equal(got, baseOut) {
t.Errorf("gasm-linked output %q, want baseline %q", got, baseOut)
}
}
// fieldAfter returns the whitespace-delimited field following the first
// occurrence of flag in line.
func fieldAfter(line, flag string) string {
fields := strings.Fields(line)
for i, f := range fields {
if f == flag && i+1 < len(fields) {
return fields[i+1]
}
}
return ""
}
+256 -6
View File
@@ -52,9 +52,10 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
switch src := src.(type) {
case Reg:
if dstIsReg {
// MOV r, r/m: 0x8A/0x8B, reg=dst, rm=src.
i := newInstr(size, []byte{movRR(size)})
if err := setRM(i, dstReg, src, size); err != nil {
// MOV r/m, r: 0x88/0x89, reg=src, rm=dst — the form the Go
// assembler emits for register-to-register moves.
i := newInstr(size, []byte{movRM(size)})
if err := setRM(i, src, dst, size); err != nil {
return err
}
return e.emit(i)
@@ -77,6 +78,17 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
}
return e.emit(i)
case sbMem:
if !dstIsReg {
return fmt.Errorf("MOV: two memory operands")
}
// MOV r, r/m: reg=dst, rm=src(static symbol).
i := newInstr(size, []byte{movRR(size)})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
case Imm:
if dstIsReg {
// MOV r, imm: 0xB0+reg (8-bit) / 0xB8+reg (16/32/64, imm64 for Q).
@@ -147,9 +159,37 @@ func (e *enc) encodeALU(op struct {
return e.encodeALUImm(op.digit, src, int64(imm), size)
}
// CMP records first − second without writing anywhere, so the first
// operand must land as the minuend; every other ALU op writes its second
// operand and follows the forms below.
cmp := op.rr == 0x39
dstReg, dstIsReg := dst.(Reg)
srcReg, srcIsReg := src.(Reg)
switch {
case cmp && dstIsReg:
// CMP x, reg: OP r/m, r (0x38/0x39) with rm = first operand, reg =
// second, matching the Go assembler.
opc := op.rr
if size == 1 {
opc = op.rr - 1
}
i := newInstr(size, []byte{opc})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
case cmp && srcIsReg:
// CMP reg, mem: OP r, r/m (0x3A/0x3B) with reg = first operand, rm =
// second.
opc := op.rr + 2
if size == 1 {
opc = op.rr + 1
}
i := newInstr(size, []byte{opc})
if err := setRM(i, srcReg, dst, size); err != nil {
return err
}
return e.emit(i)
case srcIsReg:
// OP r/m, r: reg=src, rm=dst (dst is a register or memory). This is the
// form the Go assembler prefers when the source is a register.
@@ -251,12 +291,13 @@ func (e *enc) encodeLea(ops []Operand, size int) error {
if !ok {
return fmt.Errorf("LEA: destination must be a register")
}
mem, ok := src.(Mem)
if !ok {
switch src.(type) {
case Mem, sbMem:
default:
return fmt.Errorf("LEA: source must be a memory operand")
}
i := newInstr(size, []byte{0x8D})
if err := setRM(i, dstReg, mem, size); err != nil {
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
@@ -497,3 +538,212 @@ func immediate(v int64, size int, full64 bool) []byte {
return le32(v) // sign-extended imm32
}
}
// --- CMOVcc / SETcc ---------------------------------------------------------
// encodeCmov encodes a conditional move: CMOV + size (W/L/Q) + condition
// (CMOVLGT, CMOVQEQ, …). The condition reads exactly like the Jcc spellings;
// the instruction is 0F 40+cc with reg = dst, rm = src.
func (e *enc) encodeCmov(upper string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("CMOVcc expects 2 operands, got %d", len(ops))
}
rest := upper[len("CMOV"):]
if len(rest) < 2 {
return fmt.Errorf("unsupported instruction %q", upper)
}
var size int
switch rest[0] {
case 'W':
size = 2
case 'L':
size = 4
case 'Q':
size = 8
default:
return fmt.Errorf("unsupported instruction %q", upper)
}
cc, ok := jccMap[rest[1:]]
if !ok {
return fmt.Errorf("unsupported instruction %q", upper)
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok {
return fmt.Errorf("CMOVcc destination must be a register")
}
i := newInstr(size, []byte{0x0F, byte(0x40 + cc)})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
}
// encodeSet encodes a conditional byte set: SET + condition (SETNE, SETEQ, …),
// always a byte write — 0F 90+cc /0 into a register or memory operand.
func (e *enc) encodeSet(upper string, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("SETcc expects 1 operand, got %d", len(ops))
}
cond := upper[len("SET"):]
cc, ok := jccMap[cond]
if !ok || cond == "" {
return fmt.Errorf("unsupported instruction %q", upper)
}
i := &instr{opcode: []byte{0x0F, byte(0x90 + cc)}, modrm: -1, sib: -1}
if err := setRMDigit(i, 0, ops[0], 1); err != nil {
return err
}
return e.emit(i)
}
// --- LZCNT / TZCNT ----------------------------------------------------------
// encodeCount encodes LZCNT/TZCNT (leading / trailing zero count): F3 0F BD
// or F3 0F BC, with reg = dst and rm = src. The size suffix selects the
// operand width (LZCNTW/LZCNTL/LZCNTQ).
func (e *enc) encodeCount(base string, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
}
op := byte(0xBD)
if base == "TZCNT" {
op = 0xBC
}
dstReg, ok := ops[1].(Reg)
if !ok {
return fmt.Errorf("%s destination must be a register", base)
}
i := newInstr(size, []byte{0x0F, op})
i.prefix = 0xF3
if err := setRM(i, dstReg, ops[0], size); err != nil {
return err
}
return e.emit(i)
}
// --- mixed-width sign/zero-extending moves -----------------------------------
// movExtendOp maps Go's mixed-width move names to their opcode and destination
// width. The source is narrower than the destination, so the plain size-suffix
// convention does not apply to these names.
var movExtendOp = map[string]struct {
op []byte
dst64 bool
}{
"MOVBLZX": {[]byte{0x0F, 0xB6}, false}, // byte → long, zero-extend
"MOVBQZX": {[]byte{0x0F, 0xB6}, true}, // byte → quad, zero-extend
"MOVWLZX": {[]byte{0x0F, 0xB7}, false}, // word → long, zero-extend
"MOVWQZX": {[]byte{0x0F, 0xB7}, true}, // word → quad, zero-extend
"MOVWLSX": {[]byte{0x0F, 0xBF}, false}, // word → long, sign-extend
"MOVLQSX": {[]byte{0x63}, true}, // long → quad, sign-extend (MOVSXD)
}
// encodeMovExtend encodes a mixed-width extending move: reg = dst (the wider
// operand), rm = src.
func (e *enc) encodeMovExtend(base string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
}
spec := movExtendOp[base]
dstReg, ok := ops[1].(Reg)
if !ok {
return fmt.Errorf("%s destination must be a register", base)
}
size := 4
if spec.dst64 {
size = 8
}
i := newInstr(size, spec.op)
if err := setRM(i, dstReg, ops[0], size); err != nil {
return err
}
return e.emit(i)
}
// --- legacy SSE moves --------------------------------------------------------
// sseMove describes a legacy (non-VEX) SSE move: a mandatory prefix plus a
// load opcode (reg = destination, rm = source) and a store opcode (the
// reverse). The Plan 9 names MOVOU/MOVO are the integer unaligned/aligned
// octa moves (MOVDQU/MOVDQA), not the packed-single ones.
type sseMove struct {
prefix byte // 0, 0x66, 0xF2 or 0xF3
load byte
store byte
}
var sseMoveTable = map[string]sseMove{
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU — unaligned octa
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA — aligned octa
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
"MOVAPD": {0x66, 0x28, 0x29}, // aligned packed double
"MOVSD": {0xF2, 0x10, 0x11}, // scalar double
"MOVSS": {0xF3, 0x10, 0x11}, // scalar single
}
// encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the
// load form (reg = destination), matching the Go assembler.
func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("SSE move expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcVec := vecReg(src)
dstReg, dstVec := vecReg(dst)
op := m.store
var reg Reg
var rm Operand
switch {
case srcVec && dstVec:
op = m.load
reg, rm = dstReg, src
case srcVec:
if _, ok := dst.(Mem); !ok {
return fmt.Errorf("SSE move: invalid destination operand")
}
reg, rm = srcReg, dst
case dstVec:
if _, ok := src.(Mem); !ok {
return fmt.Errorf("SSE move: invalid source operand")
}
op = m.load
reg, rm = dstReg, src
default:
return fmt.Errorf("SSE move needs a vector register operand")
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
if err := setRM(i, reg, rm, 8); err != nil {
return err
}
return e.emit(i)
}
// --- CVTSL2SD / CVTSQ2SD -----------------------------------------------------
// encodeCvtsi2sd encodes a signed integer to scalar double conversion
// (CVTSL2SD from a 32-bit, CVTSQ2SD from a 64-bit source): F2 0F 2A with
// reg = XMM dst, rm = GPR/memory src. The Go assembler emits the legacy SSE
// encoding here, not the VEX form, so we match it byte for byte.
func (e *enc) encodeCvtsi2sd(quad bool, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("CVTSx2SD expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("CVTSx2SD destination must be a vector register")
}
size := 4
if quad {
size = 8
}
i := newInstr(size, []byte{0x0F, 0x2A})
i.prefix = 0xF2
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
}
+298
View File
@@ -0,0 +1,298 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"fmt"
"sort"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// Image is an assembled file: the function bodies laid out in source order,
// followed by the file's static data section (GLOBL/DATA). References to
// file-local static symbols are encoded RIP-relative and resolved within the
// image, so the raw bytes are self-consistent and executable at any base
// address; references to external symbols are recorded as relocations
// (Funcs[i].Relocs, Externals) and left unresolved — the object-file
// emitters turn them into linker relocations.
type Image struct {
Code []byte // concatenated function bodies
Data []byte // static data section
Funcs []FuncLayout // function positions, in source order
Symbols map[string]int // static symbol → byte offset within the image
DataSyms []DataSymbol // GLOBL symbols, in layout order
Externals []string // referenced but undefined symbols, sorted
}
// FuncLayout describes one assembled function within an Image.
type FuncLayout struct {
Name string
Pkg string // explicit package prefix ("" = the current package)
Static bool // the <> marker: file-local, not exported
Offset int // start offset within the image (== offset within Code)
Size int
Args int // declared argument/result area (the TEXT size suffix)
Frame int // local frame size (the TEXT $framesize)
NoSplit bool // the NOSPLIT flag
SPWrite bool // the SPWRITE flag: writes an arbitrary value to SP
Line int // source line of the TEXT directive
Labels map[string]int // local labels, function-relative
Relocs []Reloc // static-symbol references, in emission order
Spadj []SpadjStep // stack-adjustment boundaries, ascending by PC
}
// SpadjStep is one stack-adjustment boundary: Value is the SP delta from the
// entry state in effect from PC (function-relative) until the next step.
type SpadjStep struct {
PC int
Value int
}
// Reloc is one static-symbol reference within a function body: the disp32
// field at Off (function-relative) must reach the symbol plus Addend,
// measured from After, the address just past the instruction. An External
// relocation names a symbol no GLOBL in the file defines; the object-file
// emitters carry it into the output's relocation table.
type Reloc struct {
Off int
After int
Name string
Addend int64
External bool
}
// DataSymbol describes one GLOBL symbol laid out in the data section.
type DataSymbol struct {
Name string
Pkg string // explicit package prefix ("" = the current package)
Offset int // byte offset within Data
Size int
Static bool // the <> marker: file-local, not exported
Rodata bool // the RODATA flag: read-only data
Dupok bool // the DUPOK flag: duplicate-OK
}
// Bytes returns the whole image: code, then data.
func (img *Image) Bytes() []byte {
out := make([]byte, 0, len(img.Code)+len(img.Data))
out = append(out, img.Code...)
return append(out, img.Data...)
}
// AssembleFile assembles every TEXT function of a parsed file and lays out
// its static symbols (GLOBL/DATA) in a data section behind the code. Each
// reference to a file-local static symbol becomes a RIP-relative load whose
// displacement is resolved against that layout; a reference to a symbol no
// GLOBL defines is recorded as an external relocation (Externals) with its
// displacement left zero — the object-file emitters resolve it at link
// time, while the raw image (Bytes) cannot represent it.
func AssembleFile(f *ast.File) (*Image, error) {
dataSyms, err := collectData(f)
if err != nil {
return nil, err
}
known := make(map[string]bool, len(dataSyms))
for _, d := range dataSyms {
known[d.name] = true
}
link := &linkInfo{symbols: known, allowExternal: true}
img := &Image{Symbols: map[string]int{}}
type asmFunc struct {
name string
patches []sbPatch
}
var funcs []asmFunc
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
continue
}
code, patches, labels, steps, err := assemble(t, link)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
fl := FuncLayout{
Name: t.Name.Name,
Pkg: t.Name.Pkg,
Static: t.Name.Static,
Offset: len(img.Code),
Size: len(code),
Frame: frameSize(t),
Args: argsSize(t),
Line: t.Pos().Line,
Labels: labels,
}
for _, f := range t.Flags {
switch f {
case "NOSPLIT":
fl.NoSplit = true
case "SPWRITE":
fl.SPWrite = true
}
}
for _, s := range steps {
fl.Spadj = append(fl.Spadj, SpadjStep{PC: s.pc, Value: s.value})
}
img.Funcs = append(img.Funcs, fl)
img.Code = append(img.Code, code...)
funcs = append(funcs, asmFunc{name: t.Name.Name, patches: patches})
}
// Lay out the data section behind the code, each symbol 16-aligned.
dataStart := len(img.Code)
for _, d := range dataSyms {
if pos := dataStart + len(img.Data); pos != align16(pos) {
img.Data = append(img.Data, make([]byte, align16(pos)-pos)...)
}
img.Symbols[d.name] = dataStart + len(img.Data)
img.DataSyms = append(img.DataSyms, DataSymbol{
Name: d.name,
Pkg: d.pkg,
Offset: len(img.Data),
Size: len(d.buf),
Static: d.static,
Rodata: d.rodata,
Dupok: d.dupok,
})
img.Data = append(img.Data, d.buf...)
}
// Resolve the RIP-relative displacements of file-local references now
// that every address is known, and record every reference (resolved or
// external) for the object-file emitters.
externals := map[string]bool{}
for i, fn := range funcs {
base := img.Funcs[i].Offset
code := img.Code[base : base+img.Funcs[i].Size]
for _, p := range fn.patches {
reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend}
if imgOff, ok := img.Symbols[p.name]; ok {
rel := int64(imgOff) + p.addend - int64(base+p.after)
if rel < -1<<31 || rel >= 1<<31 {
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
}
copy(code[p.off:p.off+4], le32(rel))
} else {
reloc.External = true
externals[p.name] = true
}
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
}
}
for name := range externals {
img.Externals = append(img.Externals, name)
}
sort.Strings(img.Externals)
return img, nil
}
// dataSym is one GLOBL symbol and its DATA initialiser.
type dataSym struct {
name string
pkg string
buf []byte
static bool
rodata bool
dupok bool
}
// collectData gathers the file's static symbols (GLOBL) and their initial
// contents (DATA) into byte buffers, in declaration order.
func collectData(f *ast.File) ([]dataSym, error) {
index := map[string]int{}
var syms []dataSym
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Globl:
if dd.Name == nil || dd.Name.Pseudo != "SB" {
continue
}
name := dd.Name.Name
if _, dup := index[name]; dup {
return nil, fmt.Errorf("duplicate GLOBL %q", name)
}
size := 0
if dd.Size != nil && dd.Size.Imm.HasVal {
size = int(dd.Size.Imm.Val)
}
index[name] = len(syms)
ds := dataSym{
name: name,
pkg: dd.Name.Pkg,
buf: make([]byte, size),
static: dd.Name.Static,
}
for _, f := range dd.Flags {
switch f {
case "RODATA":
ds.rodata = true
case "DUPOK":
ds.dupok = true
case "1":
ds.dupok = true
case "8":
ds.rodata = true
case "9":
ds.dupok = true
ds.rodata = true
}
}
syms = append(syms, ds)
case *ast.Data:
if dd.Name == nil || dd.Name.Pseudo != "SB" {
continue
}
i, ok := index[dd.Name.Name]
if !ok {
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
}
if dd.Value == nil || !dd.Value.Imm.HasVal {
return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name)
}
w := dd.Width
switch w {
case 1, 2, 4, 8:
default:
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
}
off := dd.Name.Offset
buf := syms[i].buf
if off < 0 || off+int64(w) > int64(len(buf)) {
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
}
v := dd.Value.Imm.Val
if dd.Value.Imm.Neg {
v = -v
}
for j := 0; j < w; j++ {
buf[off+int64(j)] = byte(v >> (8 * j))
}
}
}
return syms, nil
}
// align16 rounds n up to the next multiple of 16.
func align16(n int) int {
return (n + 15) &^ 15
}
// frameSize returns the local frame size declared on the TEXT directive.
func frameSize(t *ast.Text) int {
if t.Frame != nil && t.Frame.Imm.HasVal {
return int(t.Frame.Imm.Val)
}
return 0
}
// argsSize returns the argument/result area declared on the TEXT directive.
func argsSize(t *ast.Text) int {
if t.Args != nil && t.Args.Imm.HasVal {
return int(t.Args.Imm.Val)
}
return 0
}
+196
View File
@@ -0,0 +1,196 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"os"
"strings"
"testing"
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestAssembleFileStaticData checks the whole-image layout — code, padding
// and the data section — and that the RIP-relative displacements of static
// symbol loads resolve to the right bytes.
func TestAssembleFileStaticData(t *testing.T) {
f, errs := parser.Parse("d_amd64.s", `
#include "textflag.h"
TEXT ·load(SB), NOSPLIT, $0
VMOVDQU mask<>(SB), X15
MOVL small<>(SB), AX
RET
GLOBL mask<>(SB), RODATA, $16
DATA mask<>+0(SB)/4, $0x80020100
DATA mask<>+4(SB)/4, $0x80050403
DATA mask<>+8(SB)/4, $0x80080706
DATA mask<>+12(SB)/4, $0x800B0A09
GLOBL small<>(SB), RODATA, $4
DATA small<>+0(SB)/4, $0x1234
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
// Code (15 bytes) + 1 pad byte to align the data section to 16:
// VMOVDQU mask<>(SB), X15 c5 7a 6f 3d 08 00 00 00 (disp = 16 − 8)
// MOVL small<>(SB), AX 8b 05 12 00 00 00 (disp = 32 − 14)
// RET c3
// Data: pad, mask (16 bytes), small (4 bytes).
want := "c57a6f3d080000008b0512000000c300" +
"000102800304058006070880090a0b80" +
"34120000"
if got := strings.ReplaceAll(hexBytes(img.Bytes()), " ", ""); got != want {
t.Errorf("image bytes:\n got %s\n want %s", got, want)
}
if img.Symbols["mask"] != 16 || img.Symbols["small"] != 32 {
t.Errorf("symbol offsets = %v, want mask=16 small=32", img.Symbols)
}
if len(img.Funcs) != 1 || img.Funcs[0].Name != "load" || img.Funcs[0].Size != 15 {
t.Errorf("funcs = %+v", img.Funcs)
}
}
// TestAssembleFileErrors checks the static-symbol error paths.
func TestAssembleFileErrors(t *testing.T) {
cases := []struct {
name string
src string
want string // substring of the error
}{
{
"undefined symbol",
`
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
VMOVDQU nope<>(SB), X0
RET
`,
"undefined symbol",
},
{
"DATA without GLOBL",
`
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
RET
DATA orphan<>+0(SB)/4, $1
`,
"no matching GLOBL",
},
{
"DATA exceeds size",
`
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
RET
GLOBL tiny<>(SB), RODATA, $4
DATA tiny<>+0(SB)/8, $1
`,
"exceeds GLOBL size",
},
{
"DATA bad width",
`
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
RET
GLOBL odd<>(SB), RODATA, $4
DATA odd<>+0(SB)/3, $1
`,
"invalid width",
},
}
for _, c := range cases {
f, errs := parser.Parse("e_amd64.s", c.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", c.name, errs)
}
if _, err := AssembleFile(f); err == nil || !strings.Contains(err.Error(), c.want) {
t.Errorf("%s: error %v, want substring %q", c.name, err, c.want)
}
}
// A static-symbol operand is unresolvable in single-function assembly.
fn := firstText(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
MOVQ x<>(SB), AX
RET
GLOBL x<>(SB), RODATA, $8
DATA x<>+0(SB)/4, $1
`)
if _, _, err := Assemble(fn); err == nil || !strings.Contains(err.Error(), "file-level assembly") {
t.Errorf("single-function SB: error %v, want a file-level-assembly error", err)
}
}
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel —
// all functions plus the file-local mask24 constant — and checks that every
// static-symbol load resolves to the right bytes in the image. Skipped when
// the sibling repository is not checked out.
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx2_amd64.s"
if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit")
}
src, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
if len(img.Funcs) != 17 {
t.Errorf("functions = %d, want 17", len(img.Funcs))
}
// mask24 as the DATA directives define it.
mask := []byte{
0x00, 0x01, 0x02, 0x80, 0x03, 0x04, 0x05, 0x80,
0x06, 0x07, 0x08, 0x80, 0x09, 0x0a, 0x0b, 0x80,
}
image := img.Bytes()
if got := image[img.Symbols["mask24"] : img.Symbols["mask24"]+16]; !bytes.Equal(got, mask) {
t.Errorf("mask24 contents %x, want %x", got, mask)
}
// Every VMOVDQU mask24<>(SB), X15 (c5 7a 6f 3d + rel32, i.e. a VMOVDQU
// with a RIP-relative r/m) must land on the mask bytes within the image.
loads := 0
for _, fn := range img.Funcs {
code := img.Code[fn.Offset : fn.Offset+fn.Size]
for pc := 0; pc < len(code); {
inst, err := x86asm.Decode(code[pc:], 64)
if err != nil {
t.Fatalf("%s: decode at +%d: %v", fn.Name, pc, err)
}
// mod=00, rm=101 → RIP-relative.
if inst.Op == x86asm.VMOVDQU && inst.Len == 8 && code[pc+3]&0xC7 == 0x05 {
rel := int32(uint32(code[pc+4]) | uint32(code[pc+5])<<8 | uint32(code[pc+6])<<16 | uint32(code[pc+7])<<24)
target := fn.Offset + pc + 8 + int(rel)
if !bytes.Equal(image[target:target+16], mask) {
t.Errorf("%s: mask load at +%d lands on %x, want %x", fn.Name, pc, image[target:target+16], mask)
}
loads++
}
pc += inst.Len
}
}
if loads != 2 {
t.Errorf("mask loads found = %d, want 2", loads)
}
}
+258
View File
@@ -0,0 +1,258 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// This file emits Mach-O x86-64 objects (MH_OBJECT) from an assembled
// Image, in the shape the Darwin assembler produces: one unnamed segment
// carrying a __TEXT,__text and a __DATA,__data section laid out back to
// back at addresses zero and len(code), a symbol table (locals first, then
// exported definitions, then undefined externals) and one relocation entry
// per static-symbol reference, of type X86_64_RELOC_SIGNED.
//
// The image's own address space carries straight over — the data section
// starts immediately after the code, and the layout padding already lives
// inside Image.Data — so every symbol keeps its image address as its
// n_value, and a local (non-external) relocation leaves the displacement
// the assembler resolved in place: the linker only adjusts it by the
// section's final movement.
// Mach-O constants.
const (
machoMagic64 = 0xfeedfacf
machoCPUamd64 = 0x01000007 // CPU_TYPE_X86_64
machoCPUSubAll = 3 // CPU_SUBTYPE_X86_64_ALL
machoObj = 1 // MH_OBJECT
machoSegment64 = 0x19 // LC_SEGMENT_64
machoSymtab = 0x2 // LC_SYMTAB
machoSectTextFlags = 0x80000400 // S_ATTR_PURE_INSTRUCTIONS | S_ATTR_SOME_INSTRUCTIONS
nUndf = 0x00 // undefined symbol
nSect = 0x0e // defined in section number n_sect
nExt = 0x01 // external (exported or undefined-global) bit
x8664RelocSigned = 1
)
// MachOObject returns the image as a Mach-O x86-64 relocatable object
// (MH_OBJECT), the shape the Darwin toolchain links. Symbol names follow
// the same rules as the ELF output. Every static-symbol reference becomes
// an X86_64_RELOC_SIGNED relocation: external references against their
// undefined symbol, file-local ones against the __DATA section with the
// resolved displacement carried in the instruction bytes.
func (img *Image) MachOObject() ([]byte, error) {
le := binary.LittleEndian
// Section ordinals (1-based, as Mach-O numbers them).
const (
sectText = 1
sectData = 2
)
// Object address space: code at 0, data immediately after (the layout
// padding is already part of img.Data, so image addresses are object
// addresses).
textAddr := uint64(0)
dataAddr := uint64(len(img.Code))
vmsize := dataAddr + uint64(len(img.Data))
// The code, with external displacements primed to addend − 4: the
// linker adds the symbol's address to the field as it stands. Local
// displacements stay as the assembler resolved them.
code := append([]byte(nil), img.Code...)
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.External {
// Prime the field to the addend measured from the patch
// site: the assembler records it from the instruction end,
// After − Off bytes past the field.
copy(code[fn.Offset+r.Off:], le32(r.Addend-int64(r.After-r.Off)))
}
}
}
// Symbols: locals first, then exported definitions, then undefined
// externals — the order the classic link editor expects.
type machoSym struct {
name string
typ byte
sect byte
value uint64
}
var locals, globals, undefs []machoSym
for _, fn := range img.Funcs {
s := machoSym{name: objectName(fn.Pkg, fn.Name), typ: nSect, sect: sectText, value: textAddr + uint64(fn.Offset)}
if fn.Static {
locals = append(locals, s)
} else {
s.typ |= nExt
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := machoSym{name: objectName(d.Pkg, d.Name), typ: nSect, sect: sectData, value: dataAddr + uint64(d.Offset)}
if d.Static {
locals = append(locals, s)
} else {
s.typ |= nExt
globals = append(globals, s)
}
}
for _, name := range img.Externals {
undefs = append(undefs, machoSym{name: name, typ: nUndf | nExt})
}
syms := append(append(locals, globals...), undefs...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Relocations, attached to the __text section.
type machoReloc struct {
addr uint32
symnum uint32
extern bool
}
var relocs []machoReloc
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
rel := machoReloc{addr: uint32(fn.Offset + r.Off)}
if r.External {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
rel.symnum = uint32(idx)
rel.extern = true
} else {
// Section-relative: r_symbolnum carries the section number
// and the resolved displacement stays in the bytes.
rel.symnum = sectData
}
relocs = append(relocs, rel)
}
}
// The string table opens with the conventional " \0".
strtab := []byte{' ', 0}
strOff := map[string]int{}
for _, s := range syms {
if _, ok := strOff[s.name]; ok {
continue
}
strOff[s.name] = len(strtab)
strtab = append(strtab, s.name...)
strtab = append(strtab, 0)
}
// File layout: header, the two load commands, section data (code,
// data), the relocation table, the symbol table, the string table.
const (
hdrSize = 32
segCmdSize = 72 + 2*80 // segment command with two sections
symCmdSize = 24
)
sizeofcmds := segCmdSize + symCmdSize
dataOff := hdrSize + sizeofcmds
reloff := dataOff + len(code) + len(img.Data)
symoff := reloff + 8*len(relocs)
stroff := symoff + 16*len(syms)
out := make([]byte, stroff+len(strtab))
// mach_header_64.
le.PutUint32(out[0:], machoMagic64)
le.PutUint32(out[4:], machoCPUamd64)
le.PutUint32(out[8:], machoCPUSubAll)
le.PutUint32(out[12:], machoObj)
le.PutUint32(out[16:], 2) // ncmds
le.PutUint32(out[20:], uint32(sizeofcmds))
le.PutUint32(out[24:], 0) // flags
le.PutUint32(out[28:], 0) // reserved
// LC_SEGMENT_64 with the two sections.
p := hdrSize
le.PutUint32(out[p:], machoSegment64)
le.PutUint32(out[p+4:], segCmdSize)
// segname: the empty string, zero-padded to 16 bytes.
le.PutUint64(out[p+8:], 0)
le.PutUint64(out[p+16:], 0)
le.PutUint64(out[p+24:], 0) // vmaddr
le.PutUint64(out[p+32:], vmsize)
le.PutUint64(out[p+40:], uint64(dataOff))
le.PutUint64(out[p+48:], vmsize)
le.PutUint32(out[p+56:], 7) // maxprot rwx
le.PutUint32(out[p+60:], 7) // initprot rwx
le.PutUint32(out[p+64:], 2) // nsects
le.PutUint32(out[p+68:], 0) // flags
// __TEXT,__text
s := p + 72
copy(out[s:], "__text")
copy(out[s+16:], "__TEXT")
le.PutUint64(out[s+32:], textAddr)
le.PutUint64(out[s+40:], uint64(len(code)))
le.PutUint32(out[s+48:], uint32(dataOff))
le.PutUint32(out[s+52:], 4) // align 2^4
le.PutUint32(out[s+56:], uint32(reloff))
le.PutUint32(out[s+60:], uint32(len(relocs)))
le.PutUint32(out[s+64:], machoSectTextFlags)
// __DATA,__data
s += 80
copy(out[s:], "__data")
copy(out[s+16:], "__DATA")
le.PutUint64(out[s+32:], dataAddr)
le.PutUint64(out[s+40:], uint64(len(img.Data)))
le.PutUint32(out[s+48:], uint32(dataOff+len(code)))
le.PutUint32(out[s+52:], 4) // align 2^4
// LC_SYMTAB.
p = hdrSize + segCmdSize
le.PutUint32(out[p:], machoSymtab)
le.PutUint32(out[p+4:], symCmdSize)
le.PutUint32(out[p+8:], uint32(symoff))
le.PutUint32(out[p+12:], uint32(len(syms)))
le.PutUint32(out[p+16:], uint32(stroff))
le.PutUint32(out[p+20:], uint32(len(strtab)))
// Section data.
copy(out[dataOff:], code)
copy(out[dataOff+len(code):], img.Data)
// Relocation entries.
for i, r := range relocs {
e := out[reloff+i*8:]
le.PutUint32(e[0:], r.addr)
bits := r.symnum & 0x00ffffff
bits |= 1 << 24 // r_pcrel
bits |= 2 << 25 // r_length = 4 bytes
if r.extern {
bits |= 1 << 27 // r_extern
}
bits |= x8664RelocSigned << 28
le.PutUint32(e[4:], bits)
}
// nlist_64 entries.
for i, s := range syms {
e := out[symoff+i*16:]
le.PutUint32(e[0:], uint32(strOff[s.name]))
e[4] = s.typ
e[5] = s.sect
le.PutUint16(e[6:], 0) // n_desc
le.PutUint64(e[8:], s.value)
}
// String table.
copy(out[stroff:], strtab)
return out, nil
}
+127
View File
@@ -0,0 +1,127 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/macho"
"encoding/binary"
"testing"
)
// TestMachOObject checks the structure of the emitted MH_OBJECT: the two
// sections and their addresses, the symbol table (types, sections, values)
// and the __text relocation entries, parsed back with debug/macho. No
// Darwin toolchain is available on the test hosts, so the check is
// structural — the ELF output carries the end-to-end link-and-run proof of
// the shared symbol and relocation model.
func TestMachOObject(t *testing.T) {
img := elfTestImage(t)
obj, err := img.MachOObject()
if err != nil {
t.Fatalf("MachOObject: %v", err)
}
f, err := macho.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer f.Close()
if f.Type != macho.TypeObj {
t.Errorf("file type = %v, want MH_OBJECT", f.Type)
}
if f.Cpu != macho.CpuAmd64 {
t.Errorf("cpu = %v, want CpuAmd64", f.Cpu)
}
text := f.Section("__text")
data := f.Section("__data")
if text == nil || data == nil {
t.Fatal("missing __text or __data section")
}
if text.Addr != 0 || text.Size != uint64(len(img.Code)) {
t.Errorf("__text addr/size = %#x/%d, want 0/%d", text.Addr, text.Size, len(img.Code))
}
if data.Addr != uint64(len(img.Code)) {
t.Errorf("__data addr = %#x, want %#x", data.Addr, len(img.Code))
}
// Symbol table: locals, exported definitions, undefined externals.
syms := f.Symtab.Syms
byName := map[string]macho.Symbol{}
for _, s := range syms {
byName[s.Name] = s
}
wantSym := func(name string, typ, sect uint8, value uint64) {
t.Helper()
s, ok := byName[name]
if !ok {
t.Errorf("symbol %q not found", name)
return
}
if s.Type != typ || s.Sect != sect || s.Value != value {
t.Errorf("%s: type/sect/value = %#x/%d/%#x, want %#x/%d/%#x",
name, s.Type, s.Sect, s.Value, typ, sect, value)
}
}
const (
defined = nSect | nExt
local = nSect
undefined = nUndf | nExt
)
wantSym("addq", defined, 1, 0)
wantSym("getanswer", defined, 1, 5)
wantSym("useextern", defined, 1, 13)
answer := byName["answer"]
if answer.Type != local || answer.Sect != 2 {
t.Errorf("answer: type/sect = %#x/%d, want %#x/2", answer.Type, answer.Sect, local)
}
wantSym("extvar", undefined, 0, 0)
// Relocations: both X86_64_RELOC_SIGNED, PC-relative, 4 bytes wide.
// The local one carries its section number in Value, the external one
// its symbol number.
if len(text.Relocs) != 2 {
t.Fatalf("__text relocs = %d, want 2", len(text.Relocs))
}
var sawLocal, sawExternal bool
for _, r := range text.Relocs {
if !r.Pcrel || r.Len != 2 || r.Type != x8664RelocSigned {
t.Errorf("reloc at %#x: pcrel/len/type = %v/%d/%d", r.Addr, r.Pcrel, r.Len, r.Type)
}
switch {
case r.Extern:
if name := syms[r.Value].Name; name != "extvar" {
t.Errorf("external reloc at %#x names %q, want extvar", r.Addr, name)
}
sawExternal = true
default:
if r.Value != 2 { // __data, the second section
t.Errorf("local reloc at %#x: section %d, want 2 (__data)", r.Addr, r.Value)
}
sawLocal = true
}
}
if !sawLocal || !sawExternal {
t.Errorf("relocs seen: local=%v external=%v, want both", sawLocal, sawExternal)
}
// The __text bytes are the image code, with the external displacement
// primed to addend − 4 and the local one left resolved.
textData, err := text.Data()
if err != nil {
t.Fatal(err)
}
want := append([]byte(nil), img.Code...)
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.Name == "extvar" {
binary.LittleEndian.PutUint32(want[fn.Offset+r.Off:], 0xfffffffc) // −4
}
}
}
if !bytes.Equal(textData, want) {
t.Errorf("__text bytes %x, want %x", textData, want)
}
}
+12
View File
@@ -41,3 +41,15 @@ func Idx(base, index Reg, scale int, disp int64, size int) Mem {
func Rip(disp int64, size int) Mem {
return Mem{Disp: disp, Size: size}
}
// sbMem is a memory operand that references a static (SB) symbol. It encodes
// as a RIP-relative reference with a placeholder displacement; the encoder
// records a patch site so the file-level layout can fill in the true rel32
// once the symbol's address is known.
type sbMem struct {
size int
name string // static symbol name (the GLOBL identifier)
addend int64 // byte offset within the symbol
}
func (sbMem) isOperand() {}
+69 -55
View File
@@ -14,19 +14,24 @@ import "strings"
// so the encoder keys off the register's index and lets the mnemonic supply the
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4–7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one.
// those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0–K7.
type Reg struct {
idx int
size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH
mask bool // K0–K7 opmask register
}
// Index returns the register number (0–15).
// Index returns the register number (0–15 for GPRs, 0–31 for vectors).
func (r Reg) Index() int { return r.idx }
// Size returns the width in bytes implied by the register's name.
func (r Reg) Size() int { return r.size }
// IsMask reports whether r is an AVX-512 opmask register (K0–K7).
func (r Reg) IsMask() bool { return r.mask }
func (r Reg) isOperand() {}
// needsREX reports whether this register forces a REX prefix at the given
@@ -41,45 +46,45 @@ func (r Reg) needsREX(opSize int) bool {
// Register constants (the size is the width the name implies).
var (
AL = Reg{0, 1, false}
CL = Reg{1, 1, false}
DL = Reg{2, 1, false}
BL = Reg{3, 1, false}
AH = Reg{4, 1, true}
CH = Reg{5, 1, true}
DH = Reg{6, 1, true}
BH = Reg{7, 1, true}
SPL = Reg{4, 1, false}
BPL = Reg{5, 1, false}
SIL = Reg{6, 1, false}
DIL = Reg{7, 1, false}
AL = Reg{idx: 0, size: 1}
CL = Reg{idx: 1, size: 1}
DL = Reg{idx: 2, size: 1}
BL = Reg{idx: 3, size: 1}
AH = Reg{idx: 4, size: 1, high: true}
CH = Reg{idx: 5, size: 1, high: true}
DH = Reg{idx: 6, size: 1, high: true}
BH = Reg{idx: 7, size: 1, high: true}
SPL = Reg{idx: 4, size: 1}
BPL = Reg{idx: 5, size: 1}
SIL = Reg{idx: 6, size: 1}
DIL = Reg{idx: 7, size: 1}
AX = Reg{0, 2, false}
CX = Reg{1, 2, false}
DX = Reg{2, 2, false}
BX = Reg{3, 2, false}
SP = Reg{4, 2, false}
BP = Reg{5, 2, false}
SI = Reg{6, 2, false}
DI = Reg{7, 2, false}
AX = Reg{idx: 0, size: 2}
CX = Reg{idx: 1, size: 2}
DX = Reg{idx: 2, size: 2}
BX = Reg{idx: 3, size: 2}
SP = Reg{idx: 4, size: 2}
BP = Reg{idx: 5, size: 2}
SI = Reg{idx: 6, size: 2}
DI = Reg{idx: 7, size: 2}
EAX = Reg{0, 4, false}
ECX = Reg{1, 4, false}
EDX = Reg{2, 4, false}
EBX = Reg{3, 4, false}
ESP = Reg{4, 4, false}
EBP = Reg{5, 4, false}
ESI = Reg{6, 4, false}
EDI = Reg{7, 4, false}
EAX = Reg{idx: 0, size: 4}
ECX = Reg{idx: 1, size: 4}
EDX = Reg{idx: 2, size: 4}
EBX = Reg{idx: 3, size: 4}
ESP = Reg{idx: 4, size: 4}
EBP = Reg{idx: 5, size: 4}
ESI = Reg{idx: 6, size: 4}
EDI = Reg{idx: 7, size: 4}
RAX = Reg{0, 8, false}
RCX = Reg{1, 8, false}
RDX = Reg{2, 8, false}
RBX = Reg{3, 8, false}
RSP = Reg{4, 8, false}
RBP = Reg{5, 8, false}
RSI = Reg{6, 8, false}
RDI = Reg{7, 8, false}
RAX = Reg{idx: 0, size: 8}
RCX = Reg{idx: 1, size: 8}
RDX = Reg{idx: 2, size: 8}
RBX = Reg{idx: 3, size: 8}
RSP = Reg{idx: 4, size: 8}
RBP = Reg{idx: 5, size: 8}
RSI = Reg{idx: 6, size: 8}
RDI = Reg{idx: 7, size: 8}
)
// regByName maps an assembly register name (case-insensitive) to a Reg.
@@ -91,28 +96,28 @@ func buildRegByName() map[string]Reg {
// 64-bit: RAX..RDI, R8..R15.
r64 := []string{"RAX", "RCX", "RDX", "RBX", "RSP", "RBP", "RSI", "RDI"}
for i, n := range r64 {
m[n] = Reg{i, 8, false}
m[n] = Reg{idx: i, size: 8}
}
for i := 8; i <= 15; i++ {
m["R"+itoa(i)] = Reg{i, 8, false}
m["R"+itoa(i)] = Reg{idx: i, size: 8}
}
// 32-bit: EAX..EDI, R8D..R15D.
e32 := []string{"EAX", "ECX", "EDX", "EBX", "ESP", "EBP", "ESI", "EDI"}
for i, n := range e32 {
m[n] = Reg{i, 4, false}
m[n] = Reg{idx: i, size: 4}
}
for i := 8; i <= 15; i++ {
m["R"+itoa(i)+"D"] = Reg{i, 4, false}
m["R"+itoa(i)+"D"] = Reg{idx: i, size: 4}
}
// 16-bit: AX..DI, R8W..R15W.
w16 := []string{"AX", "CX", "DX", "BX", "SP", "BP", "SI", "DI"}
for i, n := range w16 {
m[n] = Reg{i, 2, false}
m[n] = Reg{idx: i, size: 2}
}
for i := 8; i <= 15; i++ {
m["R"+itoa(i)+"W"] = Reg{i, 2, false}
m["R"+itoa(i)+"W"] = Reg{idx: i, size: 2}
}
// 8-bit: AL..BH, SPL..DIL, R8B..R15B.
@@ -124,25 +129,34 @@ func buildRegByName() map[string]Reg {
m[n] = r
}
for i := 8; i <= 15; i++ {
m["R"+itoa(i)+"B"] = Reg{i, 1, false}
m["R"+itoa(i)+"B"] = Reg{idx: i, size: 1}
}
// Vector: X0..X15 (128-bit, encoded size 16), Y0..Y15 (256-bit, size 32).
// Z (512-bit) and K (mask) registers arrive with EVEX/AVX-512 support.
for i := 0; i <= 15; i++ {
m["X"+itoa(i)] = Reg{i, 16, false}
m["Y"+itoa(i)] = Reg{i, 32, false}
// Vector: X0..X31 (128-bit, size 16), Y0..Y31 (256-bit, size 32),
// Z0..Z31 (512-bit, size 64). Indices 16–31 are only encodable in EVEX
// (AVX-512) instructions; the encoder validates that through its tables.
for i := 0; i <= 31; i++ {
m["X"+itoa(i)] = Reg{idx: i, size: 16}
m["Y"+itoa(i)] = Reg{idx: i, size: 32}
m["Z"+itoa(i)] = Reg{idx: i, size: 64}
}
// Opmask: K0..K7.
for i := 0; i <= 7; i++ {
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
}
return m
}
// isVec reports whether r is an XMM/YMM vector register.
func (r Reg) isVec() bool { return r.size == 16 || r.size == 32 }
// isVec reports whether r is an XMM/YMM/ZMM vector register.
func (r Reg) isVec() bool { return r.size == 16 || r.size == 32 || r.size == 64 }
// vecLenBit returns the VEX.L bit for a vector register (X=0/128-bit,
// Y=1/256-bit).
// vecLenBit returns the vector-length field for a vector register:
// 0 (128-bit, VEX.L / EVEX.L'L=00), 1 (256-bit) or 2 (512-bit, EVEX only).
func (r Reg) vecLenBit() int {
if r.size == 32 {
switch r.size {
case 64:
return 2
case 32:
return 1
}
return 0
+173 -3
View File
@@ -39,6 +39,17 @@ const (
// source lives in the reg field, the destination in r/m — the PEXTR-style
// layout. VEXTRACTI128 and VEXTRACTF128 use this shape.
vexExtract
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
// in ModRM.reg and the destination in r/m — the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD).
vexRMRev
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
// vector length follows the source: the packed-double → dword
// conversions (VCVTPD2DQ/VCVTTPD2DQ and their X/Y spellings) narrow into
// an XMM destination, so the L bit rides with the wider source. The
// mnemonic's spelling fixes the length (X = 128, Y = 256), which also
// covers a memory source. ModRM.reg = dst, ModRM.rm = src, no vvvv.
vexRMSrcLen
// vexZero is the no-operand form (VZEROUPPER).
vexZero
)
@@ -80,14 +91,38 @@ var vexTable = map[string]vexSpec{
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic.
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG — packed single-precision arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3},
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3},
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3},
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
// VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed
// opcodes with an F2 pp).
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
"VMULSD": {1, 0x59, 0, 3, -1, vexNDS3},
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
// VEX.128.F3.0F.WIG — scalar single-precision arithmetic (the packed
// opcodes with an F3 pp).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3},
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
// VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form).
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
@@ -95,12 +130,27 @@ var vexTable = map[string]vexSpec{
// no vvvv).
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM},
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
// VEX.128/256.0F.WIG — signed dword to packed single conversion
// (reg=dst, rm=src, no vvvv, no mandatory prefix).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
// VEX.128/256.0F.WIG — packed single to packed double conversion
// (reg=dst, rm=src; the destination is the wide operand and sets the
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
// the Go assembler emits the instruction with pp = 00, and gasm follows
// the Go assembler's bytes — its machine code is the oracle, not the
// manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
// VEX.128.F2.0F.WIG — duplicate the low double of each 128-bit lane
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
// VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src).
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
@@ -131,6 +181,52 @@ var vexTable = map[string]vexSpec{
// VEX.128.0F.W0 — no operands.
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
// VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
// VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; SD is 256-bit only).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
// VEX.F2.0F — packed double to packed dword conversions, truncating and
// non-truncating. The destination is always XMM; the X/Y spellings fix
// the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen.
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
}
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
// the packed-double → dword conversions) to its fixed vector length:
// X = 128 (L = 0), Y = 256 (L = 1). The spelling fixes the length even for
// a memory source, matching the Go assembler's ytab.
var vexSrcLen = map[string]int{
"VCVTPD2DQX": 0,
"VCVTPD2DQY": 1,
"VCVTTPD2DQX": 0,
"VCVTTPD2DQY": 1,
}
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
// form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8),
// an ordinary NDS encoding rather than the /digit immediate form above.
var vexVarShift = map[string]byte{
"VPSLLD": 0xF2,
"VPSLLQ": 0xF3,
"VPSRAD": 0xE2,
"VPSRLD": 0xD2,
"VPSRLQ": 0xD3,
}
// vexMoveSpec describes a VEX move, which takes different opcodes (and
@@ -164,6 +260,11 @@ var vexMoveTable = map[string]vexMoveSpec{
// VEX.128.F2.0F.WIG — scalar double move, memory operands only (the
// register form takes three operands and is not supported yet).
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128.F3.0F.WIG — scalar single move, memory operands only.
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128/256 — aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
}
// isVex reports whether the mnemonic is a VEX-encoded instruction we handle.
@@ -177,9 +278,27 @@ func isVex(mnemUpper string) bool {
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
// Vector register indices 16–31 exist only in EVEX encodings; fail
// loudly rather than silently truncating the index.
for _, op := range ops {
if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 {
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
}
}
if ms, ok := vexMoveTable[mnemUpper]; ok {
return e.encodeVexMove(mnemUpper, ms, ops)
}
// The shifts come in two shapes under one mnemonic: an immediate count
// ($imm, src, dst) and a variable count in an XMM register or memory
// (count, src, dst), the latter an ordinary NDS form.
if op, ok := vexVarShift[mnemUpper]; ok && len(ops) == 3 {
if _, isImm := ops[0].(Imm); !isImm {
if !vecOrMem(ops[0]) {
return fmt.Errorf("%s: shift count must be an immediate, a vector register or memory", mnemUpper)
}
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
}
}
spec := vexTable[mnemUpper]
switch spec.form {
case vexNDS3:
@@ -194,6 +313,8 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3Imm(spec, ops)
case vexExtract:
return e.encodeVexExtract(spec, ops)
case vexRMSrcLen:
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero:
return e.encodeVexZero(mnemUpper, spec, ops)
}
@@ -259,6 +380,32 @@ func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
return e.emitVexFields(spec, l, regField, rBit, 15, src)
}
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the VEX.L bit following the source — fixed
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
// the source is memory.
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("VEX destination must be a vector register")
}
ll, ok := vexSrcLen[mnem]
if !ok {
return fmt.Errorf("no fixed vector length for %s", mnem)
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
// An unused vvvv field must be stored as all ones (v̄vvv = 1111).
return e.emitVexFields(spec, ll, regField, rBit, 15, src)
}
// encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst.
// The destination is carried in VEX.vvvv, the source in ModRM.rm, and the
// shift kind in the ModRM.reg /digit.
@@ -483,11 +630,21 @@ func vecReg(op Operand) (Reg, bool) {
return r, ok && r.isVec()
}
// vecOrMem reports whether op is a vector register or a memory reference.
func vecOrMem(op Operand) bool {
switch op.(type) {
case Mem, sbMem:
return true
}
r, ok := op.(Reg)
return ok && r.isVec()
}
// validMoveOther reports whether the non-vector operand of a move is
// acceptable: memory always is, a GPR only for VMOVD/VMOVQ.
func validMoveOther(ms vexMoveSpec, op Operand) bool {
switch o := op.(type) {
case Mem:
case Mem, sbMem:
return true
case Reg:
return ms.gprOK && !o.isVec()
@@ -499,9 +656,13 @@ func validMoveOther(ms vexMoveSpec, op Operand) bool {
// the given precomputed fields. It is shared by every register/rm VEX form;
// immediate bytes are appended by the caller.
func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Operand) error {
if l > 1 {
return fmt.Errorf("ZMM operand requires an EVEX instruction")
}
var modrm, sib int
var disp []byte
var xBit, bBit int
var sb *sbRef
switch r := rm.(type) {
case Reg:
modrm = 0xC0 | regField<<3 | (r.idx & 7)
@@ -515,6 +676,12 @@ func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Ope
if err != nil {
return err
}
case sbMem:
// RIP-relative static-symbol reference; disp32 patched at link time.
modrm = regField<<3 | 0x05
sib = -1
disp = le32(0)
sb = &sbRef{name: r.name, addend: r.addend}
default:
return fmt.Errorf("invalid VEX r/m operand")
}
@@ -530,6 +697,9 @@ func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Ope
if sib >= 0 {
e.out = append(e.out, byte(sib))
}
if sb != nil {
e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend})
}
e.out = append(e.out, disp...)
return nil
}
+109 -58
View File
@@ -156,74 +156,121 @@ func TestVexShiftImm(t *testing.T) {
// as well as every new operand form.
func TestVexGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
name string
mnem string
ops []Operand
want string
wantOp string // decoded mnemonic, when it differs from mnem (the X/Y spellings)
}{
// Three-operand NDS form.
{"VPADDQ Y8,Y9,Y8", "VPADDQ", []Operand{vreg(t, "Y8"), vreg(t, "Y9"), vreg(t, "Y8")}, "c44135d4c0"},
{"VPADDQ X9,X8,X8", "VPADDQ", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c44139d4c1"},
{"VPXOR X7,X7,X7", "VPXOR", []Operand{vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7")}, "c5c1efff"},
{"VPSHUFB Y1,Y2,Y3", "VPSHUFB", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d00d9"},
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9"},
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec"},
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9"},
{"VPADDQ Y8,Y9,Y8", "VPADDQ", []Operand{vreg(t, "Y8"), vreg(t, "Y9"), vreg(t, "Y8")}, "c44135d4c0", ""},
{"VPADDQ X9,X8,X8", "VPADDQ", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c44139d4c1", ""},
{"VPXOR X7,X7,X7", "VPXOR", []Operand{vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7")}, "c5c1efff", ""},
{"VPSHUFB Y1,Y2,Y3", "VPSHUFB", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d00d9", ""},
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""},
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""},
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""},
// Floating point (packed and scalar) and FMA — same NDS form, the pp
// bits and map select the operation.
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1"},
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9"},
{"VMULPD Y12,Y12,Y12", "VMULPD", []Operand{vreg(t, "Y12"), vreg(t, "Y12"), vreg(t, "Y12")}, "c4411d59e4"},
{"VXORPD Y8,Y8,Y8", "VXORPD", []Operand{vreg(t, "Y8"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d57c0"},
{"VUNPCKHPD X8,X8,X9", "VUNPCKHPD", []Operand{vreg(t, "X8"), vreg(t, "X8"), vreg(t, "X9")}, "c4413915c8"},
{"VADDSD X9,X8,X8", "VADDSD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c4413b58c1"},
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8"},
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6"},
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807"},
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""},
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""},
{"VMULPD Y12,Y12,Y12", "VMULPD", []Operand{vreg(t, "Y12"), vreg(t, "Y12"), vreg(t, "Y12")}, "c4411d59e4", ""},
{"VXORPD Y8,Y8,Y8", "VXORPD", []Operand{vreg(t, "Y8"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d57c0", ""},
{"VUNPCKHPD X8,X8,X9", "VUNPCKHPD", []Operand{vreg(t, "X8"), vreg(t, "X8"), vreg(t, "X9")}, "c4413915c8", ""},
{"VADDSD X9,X8,X8", "VADDSD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c4413b58c1", ""},
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
// Two-operand reg/rm form (v̄vvv must be 1111).
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0"},
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306"},
{"VPBROADCASTD X0,Y15", "VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, "c4627d58f8"},
{"VCVTDQ2PD X12,Y12", "VCVTDQ2PD", []Operand{vreg(t, "X12"), vreg(t, "Y12")}, "c4417ee6e4"},
{"VCVTDQ2PD (SI),Y4", "VCVTDQ2PD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y4")}, "c5fee626"},
{"VPMOVMSKB X11,AX", "VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, "c4c179d7c3"},
{"VMOVMSKPS Y7,AX", "VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, "c5fc50c7"},
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
{"VPBROADCASTD X0,Y15", "VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, "c4627d58f8", ""},
{"VCVTDQ2PD X12,Y12", "VCVTDQ2PD", []Operand{vreg(t, "X12"), vreg(t, "Y12")}, "c4417ee6e4", ""},
{"VCVTDQ2PD (SI),Y4", "VCVTDQ2PD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y4")}, "c5fee626", ""},
{"VPMOVMSKB X11,AX", "VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, "c4c179d7c3", ""},
{"VMOVMSKPS Y7,AX", "VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, "c5fc50c7", ""},
// Immediate shifts.
{"VPSLLD $1,Y3,Y4", "VPSLLD", []Operand{Imm(1), vreg(t, "Y3"), vreg(t, "Y4")}, "c5dd72f301"},
{"VPSRLQ $2,Y5,Y6", "VPSRLQ", []Operand{Imm(2), vreg(t, "Y5"), vreg(t, "Y6")}, "c5cd73d502"},
{"VPSLLD $1,Y3,Y4", "VPSLLD", []Operand{Imm(1), vreg(t, "Y3"), vreg(t, "Y4")}, "c5dd72f301", ""},
{"VPSRLQ $2,Y5,Y6", "VPSRLQ", []Operand{Imm(2), vreg(t, "Y5"), vreg(t, "Y6")}, "c5cd73d502", ""},
// Variable-count shifts: the count lives in an XMM register or memory
// and the instruction takes the NDS form.
{"VPSRLQ X0,Y8,Y8", "VPSRLQ", []Operand{vreg(t, "X0"), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd3c0", ""},
{"VPSRLQ (AX),Y8,Y8", "VPSRLQ", []Operand{Ptr(AX, 0, 16), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd300", ""},
{"VPSLLD X0,Y1,Y2", "VPSLLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f2d0", ""},
{"VPSRLD X0,Y1,Y2", "VPSRLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5d2d0", ""},
{"VPSRAD X0,Y1,Y2", "VPSRAD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5e2d0", ""},
{"VPSLLQ X0,Y1,Y2", "VPSLLQ", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f3d0", ""},
// Immediate shuffle (reg=dst, rm=src, imm8).
{"VPSHUFD $0xEE,X8,X9", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")}, "c4417970c8ee"},
{"VPSHUFD $0xEE,Y1,Y2", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "Y1"), vreg(t, "Y2")}, "c5fd70d1ee"},
{"VPERMQ $0x1B,Y1,Y2", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e3fd00d11b"},
{"VPERMQ $0x1B,Y11,Y12", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y11"), vreg(t, "Y12")}, "c443fd00e31b"},
{"VPSHUFD $0xEE,X8,X9", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")}, "c4417970c8ee", ""},
{"VPSHUFD $0xEE,Y1,Y2", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "Y1"), vreg(t, "Y2")}, "c5fd70d1ee", ""},
{"VPERMQ $0x1B,Y1,Y2", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e3fd00d11b", ""},
{"VPERMQ $0x1B,Y11,Y12", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y11"), vreg(t, "Y12")}, "c443fd00e31b", ""},
// Three-operand + immediate (reg=dst, vvvv=src1, rm=src2, imm8).
{"VSHUFPD $1,X1,X2,X3", "VSHUFPD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e9c6d901"},
{"VSHUFPD $1,Y1,Y2,Y3", "VSHUFPD", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5edc6d901"},
{"VPERM2I128 $0x31,Y1,Y2,Y3", "VPERM2I128", []Operand{Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e36d46d931"},
{"VINSERTI128 $1,X5,Y1,Y2", "VINSERTI128", []Operand{Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37538d501"},
{"VSHUFPD $1,X1,X2,X3", "VSHUFPD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e9c6d901", ""},
{"VSHUFPD $1,Y1,Y2,Y3", "VSHUFPD", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5edc6d901", ""},
{"VPERM2I128 $0x31,Y1,Y2,Y3", "VPERM2I128", []Operand{Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e36d46d931", ""},
{"VINSERTI128 $1,X5,Y1,Y2", "VINSERTI128", []Operand{Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37538d501", ""},
// Lane extract (reg=YMM source, rm=XMM/memory destination, imm8).
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101"},
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701"},
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101"},
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""},
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""},
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""},
// Moves — each direction picks its own opcode and VEX.W.
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e"},
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f"},
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca"},
{"VMOVUPD (DI),Y14", "VMOVUPD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y14")}, "c57d1037"},
{"VMOVUPD Y14,(DI)", "VMOVUPD", []Operand{vreg(t, "Y14"), Ptr(DI, 0, 32)}, "c57d1137"},
{"VMOVUPD X1,X2", "VMOVUPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f911ca"},
{"VMOVQ X8,AX", "VMOVQ", []Operand{vreg(t, "X8"), AX}, "c461f97ec0"},
{"VMOVQ AX,X9", "VMOVQ", []Operand{AX, vreg(t, "X9")}, "c461f96ec8"},
{"VMOVQ X8,(DI)", "VMOVQ", []Operand{vreg(t, "X8"), Ptr(DI, 0, 8)}, "c461f97e07"},
{"VMOVQ (SI),X9", "VMOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X9")}, "c461f96e0e"},
{"VMOVQ X8,X2", "VMOVQ", []Operand{vreg(t, "X8"), vreg(t, "X2")}, "c579d6c2"},
{"VMOVQ X2,X8", "VMOVQ", []Operand{vreg(t, "X2"), vreg(t, "X8")}, "c4c179d6d0"},
{"VMOVD X0,(SI)", "VMOVD", []Operand{vreg(t, "X0"), Ptr(SI, 0, 4)}, "c5f97e06"},
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0"},
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006"},
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106"},
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""},
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""},
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""},
{"VMOVUPD (DI),Y14", "VMOVUPD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y14")}, "c57d1037", ""},
{"VMOVUPD Y14,(DI)", "VMOVUPD", []Operand{vreg(t, "Y14"), Ptr(DI, 0, 32)}, "c57d1137", ""},
{"VMOVUPD X1,X2", "VMOVUPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f911ca", ""},
{"VMOVQ X8,AX", "VMOVQ", []Operand{vreg(t, "X8"), AX}, "c461f97ec0", ""},
{"VMOVQ AX,X9", "VMOVQ", []Operand{AX, vreg(t, "X9")}, "c461f96ec8", ""},
{"VMOVQ X8,(DI)", "VMOVQ", []Operand{vreg(t, "X8"), Ptr(DI, 0, 8)}, "c461f97e07", ""},
{"VMOVQ (SI),X9", "VMOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X9")}, "c461f96e0e", ""},
{"VMOVQ X8,X2", "VMOVQ", []Operand{vreg(t, "X8"), vreg(t, "X2")}, "c579d6c2", ""},
{"VMOVQ X2,X8", "VMOVQ", []Operand{vreg(t, "X2"), vreg(t, "X8")}, "c4c179d6d0", ""},
{"VMOVD X0,(SI)", "VMOVD", []Operand{vreg(t, "X0"), Ptr(SI, 0, 4)}, "c5f97e06", ""},
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""},
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""},
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""},
// Packed double arithmetic and unpack — the NDS form, the opcode
// selects the operation.
{"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""},
{"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""},
{"VMINPD Y1,Y2,Y3", "VMINPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5dd9", ""},
{"VMAXPD X4,X5,X6", "VMAXPD", []Operand{vreg(t, "X4"), vreg(t, "X5"), vreg(t, "X6")}, "c5d15ff4", ""},
{"VUNPCKLPD X1,X2,X3", "VUNPCKLPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e914d9", ""},
{"VUNPCKLPD Y1,Y2,Y3", "VUNPCKLPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed14d9", ""},
{"VSUBPD (AX),X1,X2", "VSUBPD", []Operand{Ptr(AX, 0, 16), vreg(t, "X1"), vreg(t, "X2")}, "c5f15c10", ""},
// Scalar double and single arithmetic (F2 / F3 pp, 128-bit only).
{"VSUBSD X1,X2,X3", "VSUBSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5eb5cd9", ""},
{"VDIVSD X7,X1,X2", "VDIVSD", []Operand{vreg(t, "X7"), vreg(t, "X1"), vreg(t, "X2")}, "c5f35ed7", ""},
{"VMINSD X1,X2,X3", "VMINSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5eb5dd9", ""},
{"VMAXSD X3,X4,X5", "VMAXSD", []Operand{vreg(t, "X3"), vreg(t, "X4"), vreg(t, "X5")}, "c5db5feb", ""},
{"VADDSS X1,X2,X3", "VADDSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea58d9", ""},
{"VSUBSS X1,X2,X3", "VSUBSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5cd9", ""},
{"VMULSS X9,X10,X11", "VMULSS", []Operand{vreg(t, "X9"), vreg(t, "X10"), vreg(t, "X11")}, "c4412a59d9", ""},
{"VDIVSS X1,X2,X3", "VDIVSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5ed9", ""},
{"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""},
{"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""},
{"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""},
// VMOVDDUP — duplicate the low double (reg=dst, rm=src, F2 pp).
{"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""},
{"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""},
{"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""},
// Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3
// prefix — see the table comment), DQ→PD.
{"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""},
{"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""},
{"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""},
{"VCVTPS2PD X1,Y2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c5fc5ad1", ""},
// PD→DQ conversions: the X/Y spellings fix the source length and the
// destination is always XMM; the decoder reports the base mnemonic.
{"VCVTPD2DQX X1,X2", "VCVTPD2DQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fbe6d1", "VCVTPD2DQ"},
{"VCVTPD2DQY Y1,X2", "VCVTPD2DQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5ffe6d1", "VCVTPD2DQ"},
{"VCVTTPD2DQX X3,X4", "VCVTTPD2DQX", []Operand{vreg(t, "X3"), vreg(t, "X4")}, "c5f9e6e3", "VCVTTPD2DQ"},
{"VCVTTPD2DQY Y5,X6", "VCVTTPD2DQY", []Operand{vreg(t, "Y5"), vreg(t, "X6")}, "c5fde6f5", "VCVTTPD2DQ"},
{"VCVTPD2DQY (AX),X1", "VCVTPD2DQY", []Operand{Ptr(AX, 0, 32), vreg(t, "X1")}, "c5ffe608", "VCVTPD2DQ"},
// No-operand.
{"VZEROUPPER", "VZEROUPPER", nil, "c5f877"},
{"VZEROUPPER", "VZEROUPPER", nil, "c5f877", ""},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
@@ -243,7 +290,11 @@ func TestVexGroundTruth(t *testing.T) {
if inst.Len != len(code) {
t.Errorf("%s: Decode consumed %d of %d bytes", c.name, inst.Len, len(code))
}
if inst.Op.String() != c.mnem {
wantOp := c.wantOp
if wantOp == "" {
wantOp = c.mnem
}
if inst.Op.String() != wantOp {
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
}
}
+234 -46
View File
@@ -11,7 +11,9 @@ import (
"flag"
"fmt"
"io"
"io/fs"
"os"
"path/filepath"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
@@ -26,7 +28,7 @@ import (
// version is the release version, stamped at build time via
// -ldflags "-X main.version=…" (defaulting to the current release).
var version = "0.2.0"
var version = "0.13.0"
func main() {
if len(os.Args) < 2 {
@@ -47,30 +49,73 @@ func main() {
case "lsp":
os.Exit(cmdLSP(os.Args[2:]))
case "version", "--version", "-V":
fmt.Printf("gasm %s\n", version)
case "help", "-h", "--help":
os.Exit(cmdVersion())
case "help", "--help", "-h":
usage(os.Stdout)
default:
fmt.Fprintf(os.Stderr, "gasm: unknown command %q\n\n", os.Args[1])
usage(os.Stderr)
fmt.Fprintf(os.Stderr, "gasm: unknown command %q — run \"gasm --help\" for usage\n", os.Args[1])
os.Exit(2)
}
}
// cmdVersion prints the release version.
func cmdVersion() int {
fmt.Printf("gasm %s\n", version)
return 0
}
func usage(w io.Writer) {
fmt.Fprintf(w, `gasm %s — developer tooling for Go's Plan 9 assembler
fmt.Fprintf(w, `gasm %s — developer tooling for Go's Plan 9 assembler (GAsm)
gasm bundles a lexer, parser, formatter, linter, standalone assembler and
language server for Plan 9 assembly into one self-contained binary.
Usage:
gasm tokens <file> print the lexical token stream
gasm parse <file> parse and report syntax errors
gasm fmt [-w] <file...> canonicalise formatting (-w writes in place)
gasm lint <file...> run static checks
gasm asm [-o out.bin] <file> assemble to machine code (amd64, Phase 2)
gasm lsp run the language server over stdio
gasm version print the version
gasm <command> [arguments]
gasm [flags]
Commands:
tokens print the lexical token stream
parse parse and report syntax errors
fmt canonicalise formatting (gofmt for assembly)
lint run static checks
asm assemble .s files to machine code (amd64)
lsp run the language server over stdio
version print the version (same as --version)
Flags:
-h, --help show this help
-V, --version print the version
Run "gasm <command> -h" for a command's usage and flags.
Examples:
gasm fmt reformat every .s below the current directory
gasm lint go-flac/*.s run static checks over the kernels
gasm asm -o k.bin kern_amd64.s
gasm asm --format elf -o k.o kern_amd64.s
gasm asm --format goobj -p pkg/path -o k.o kern_amd64.s
`, version)
}
// newCommand returns the FlagSet of a subcommand whose -h/--help prints a
// proper usage block: the one-line usage, the long description and the flag
// defaults. The flag package routes -h/--help to fs.Usage and exits 0.
func newCommand(name, usageLine, long string) *flag.FlagSet {
fs := flag.NewFlagSet(name, flag.ExitOnError)
fs.Usage = func() {
w := fs.Output()
fmt.Fprintf(w, "Usage: %s\n\n%s\n", usageLine, strings.TrimSpace(long))
hasFlags := false
fs.VisitAll(func(*flag.Flag) { hasFlags = true })
if hasFlags {
fmt.Fprintln(w, "\nFlags:")
fs.PrintDefaults()
}
}
return fs
}
// readSource returns the contents of path, or stdin when path is "-".
func readSource(path string) (string, error) {
if path == "-" {
@@ -82,7 +127,10 @@ func readSource(path string) (string, error) {
}
func cmdTokens(args []string) int {
fs := flag.NewFlagSet("tokens", flag.ExitOnError)
fs := newCommand("tokens", "gasm tokens <file>", `
Print the lexical token stream of FILE: position, token kind and text, one
token per line. FILE may be "-" to read standard input.
`)
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm tokens <file>")
@@ -100,7 +148,11 @@ func cmdTokens(args []string) int {
}
func cmdParse(args []string) int {
fs := flag.NewFlagSet("parse", flag.ExitOnError)
fs := newCommand("parse", "gasm parse <file>", `
Parse FILE and report syntax errors on stderr. On success, print how many
declarations and TEXT functions the file contains. FILE may be "-" to read
standard input.
`)
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm parse <file>")
@@ -130,15 +182,49 @@ func cmdParse(args []string) int {
}
func cmdFmt(args []string) int {
fs := flag.NewFlagSet("fmt", flag.ExitOnError)
fs := newCommand("fmt", "gasm fmt [-w] [path...]", `
Canonicalise the formatting of Plan 9 assembly sources: indentation, operand
spacing, per-function mnemonic alignment and blank-line layout (exactly one
blank line before each label, TEXT and GLOBL block). Formatting is
idempotent and preserves every line, comments included.
With no paths — or a directory path — every .s file below it is reformatted
in place and the changed files are listed, the way go fmt does; "." and "_"
directories are skipped. Explicit file paths print to stdout unless -w is
given.
`)
write := fs.Bool("w", false, "write result to the source file")
fs.Parse(args)
if fs.NArg() == 0 {
fmt.Fprintln(os.Stderr, "usage: gasm fmt [-w] <file...>")
return 2
// Like go fmt: with no arguments, or with a directory argument, every .s
// file below the directory is formatted in place and the names of the
// changed files are listed; explicit file arguments keep the -w / stdout
// behaviour.
paths := fs.Args()
dirMode := len(paths) == 0
if dirMode {
paths = []string{"."}
}
var files []string
for _, p := range paths {
info, err := os.Stat(p)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
return 1
}
if info.IsDir() {
dirMode = true
found, err := asmFiles(p)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
return 1
}
files = append(files, found...)
continue
}
files = append(files, p)
}
rc := 0
for _, path := range fs.Args() {
for _, path := range files {
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
@@ -146,11 +232,15 @@ func cmdFmt(args []string) int {
continue
}
out := format.Source(path, src)
if *write {
if dirMode || *write {
if out != src {
if err := os.WriteFile(path, []byte(out), 0o644); err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
rc = 1
continue
}
if dirMode {
fmt.Println(path)
}
}
continue
@@ -160,8 +250,40 @@ func cmdFmt(args []string) int {
return rc
}
// asmFiles collects the .s files below dir, skipping directories whose name
// starts with "." or "_" — as the go tooling does, which keeps .git and
// scratch or reference trees (e.g. _refs) untouched.
func asmFiles(dir string) ([]string, error) {
var out []string
err := filepath.WalkDir(dir, func(path string, d fs.DirEntry, err error) error {
if err != nil {
return err
}
if d.IsDir() {
if path != dir && (strings.HasPrefix(d.Name(), ".") || strings.HasPrefix(d.Name(), "_")) {
return filepath.SkipDir
}
return nil
}
if strings.HasSuffix(d.Name(), ".s") {
out = append(out, path)
}
return nil
})
return out, err
}
func cmdLint(args []string) int {
fs := flag.NewFlagSet("lint", flag.ExitOnError)
fs := newCommand("lint", "gasm lint <file...>", `
Run the static checks over the given files and print diagnostics as
"file:line:col: severity: message [code]". The exit status is non-zero when
an error-severity diagnostic is found; warnings (e.g. the register-clobber
audit) do not affect it.
Rules include unknown-instruction, operand-count, undefined-label,
duplicate-label, missing-ret, missing-textflag-include, abi-argsize,
unreachable-code, register-clobber and funcdata-pcdata.
`)
disable := fs.String("disable", "", "comma-separated rule codes to disable")
fs.Parse(args)
if fs.NArg() == 0 {
@@ -202,7 +324,13 @@ func cmdLint(args []string) int {
}
func cmdLSP(args []string) int {
fs := flag.NewFlagSet("lsp", flag.ExitOnError)
fs := newCommand("lsp", "gasm lsp", `
Run the language server over standard input/output: JSON-RPC 2.0 with
Content-Length framing. Point an LSP-capable editor at the binary and
associate it with .s files; the target architecture is inferred from the file
suffix (_amd64.s, _arm64.s, _riscv64.s, _loong64.s). Provides completion,
hover, document symbols, diagnostics and semantic-token highlighting.
`)
fs.Parse(args)
srv := lsp.New(os.Stdin, os.Stdout)
if err := srv.Run(); err != nil {
@@ -213,11 +341,26 @@ func cmdLSP(args []string) int {
}
func cmdAsm(args []string) int {
fs := flag.NewFlagSet("asm", flag.ExitOnError)
out := fs.String("o", "", "write the concatenated machine code to this file")
fs := newCommand("asm", "gasm asm [--format raw|elf|macho|goobj] [-p pkg] [-o out] <file>", `
Assemble FILE (amd64) without the Go toolchain: every TEXT function is
encoded to machine code — scalar, VEX/AVX2 and EVEX/AVX-512 instructions,
FP/SP frame mapping, local labels and file-local static symbols (GLOBL/DATA)
resolved RIP-relative — and printed as a hex dump.
With -o the output is written to a file instead. The --format flag selects
what is written: raw (the default) concatenates the functions and the data
section into one self-consistent image; elf and macho emit a relocatable
object (.text/.data sections, a symbol table and one PC32 relocation per
static-symbol reference) that links with the system toolchain; goobj emits
the Go toolchain's own object format, which cmd/link consumes directly (it
requires -p, the package path, and the installed Go toolchain).
`)
out := fs.String("o", "", "write the output to this file")
format := fs.String("format", "raw", "output format: raw (concatenated image), elf, macho or goobj (Go object)")
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm asm [-o out.bin] <file>")
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|macho|goobj] [-p pkg] [-o out] <file>")
return 2
}
path := fs.Arg(0)
@@ -238,20 +381,18 @@ func cmdAsm(args []string) int {
return 1
}
var all []byte
functions := 0
for _, d := range f.Decls {
txt, ok := d.(*ast.Text)
if !ok {
continue
}
code, _, err := asm.Assemble(txt)
if err != nil {
fmt.Fprintf(os.Stderr, "%s: %s: %v\n", path, txt.Name.Name, err)
return 1
}
functions++
fmt.Printf("%s: %d bytes\n", txt.Name.Name, len(code))
img, err := asm.AssembleFile(f)
if err != nil {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
return 1
}
if len(img.Funcs) == 0 {
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions found")
return 1
}
for _, fn := range img.Funcs {
code := img.Code[fn.Offset : fn.Offset+fn.Size]
fmt.Printf("%s: %d bytes\n", fn.Name, fn.Size)
for i := 0; i < len(code); i += 16 {
end := i + 16
if end > len(code) {
@@ -263,18 +404,65 @@ func cmdAsm(args []string) int {
}
fmt.Println()
}
all = append(all, code...)
}
if functions == 0 {
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions found")
return 1
if len(img.Data) > 0 {
fmt.Printf("data: %d bytes at 0x%x\n", len(img.Data), len(img.Code))
for _, d := range f.Decls {
g, ok := d.(*ast.Globl)
if !ok || g.Name == nil || g.Name.Pseudo != "SB" {
continue
}
size := 0
if g.Size != nil && g.Size.Imm.HasVal {
size = int(g.Size.Imm.Val)
}
fmt.Printf(" %s: %d bytes at 0x%x\n", g.Name.Name, size, img.Symbols[g.Name.Name])
}
for i := 0; i < len(img.Data); i += 16 {
end := i + 16
if end > len(img.Data) {
end = len(img.Data)
}
fmt.Printf(" %04x:", len(img.Code)+i)
for _, b := range img.Data[i:end] {
fmt.Printf(" %02x", b)
}
fmt.Println()
}
}
if *out != "" {
if err := os.WriteFile(*out, all, 0o644); err != nil {
var obj []byte
var err error
var kind string
switch *format {
case "raw":
if len(img.Externals) > 0 {
fmt.Fprintf(os.Stderr, "gasm asm: external symbol %q needs an object file (use --format elf or --format macho)\n", img.Externals[0])
return 1
}
obj, kind = img.Bytes(), "raw image"
case "elf":
obj, err = img.ELFObject()
kind = "ELF object"
case "macho":
obj, err = img.MachOObject()
kind = "Mach-O object"
case "goobj":
obj, err = img.GOObject(*pkg, path)
kind = "Go object"
default:
fmt.Fprintf(os.Stderr, "gasm asm: unknown format %q (want raw, elf, macho or goobj)\n", *format)
return 2
}
if err != nil {
fmt.Fprintln(os.Stderr, "gasm asm:", err)
return 1
}
fmt.Printf("wrote %d bytes to %s\n", len(all), *out)
if err := os.WriteFile(*out, obj, 0o644); err != nil {
fmt.Fprintln(os.Stderr, "gasm asm:", err)
return 1
}
fmt.Printf("wrote %d bytes to %s (%s)\n", len(obj), *out, kind)
}
return 0
}
+70 -5
View File
@@ -52,6 +52,54 @@ func capture(fn func() int) (stdout, stderr string, code int) {
return string(ob), string(eb), code
}
// TestCmdFmtRecursive checks the go-fmt-style directory mode: with no
// arguments every .s file below the working directory is formatted in place
// ("." and "_" directories skipped), changed files are listed, and a second
// run is a no-op.
func TestCmdFmtRecursive(t *testing.T) {
tmp := t.TempDir()
t.Chdir(tmp)
unformatted := []byte("TEXT ·f(SB),NOSPLIT,$0\nRET\n")
write := func(path string) {
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(path, unformatted, 0o644); err != nil {
t.Fatal(err)
}
}
write("a_amd64.s")
write(filepath.Join("sub", "b_amd64.s"))
write(filepath.Join("_refs", "c_amd64.s"))
write(filepath.Join(".git", "d_amd64.s"))
out, errOut, code := capture(func() int { return cmdFmt(nil) })
if code != 0 {
t.Fatalf("code = %d (%s)", code, errOut)
}
if out != "a_amd64.s\n"+filepath.Join("sub", "b_amd64.s")+"\n" {
t.Errorf("listed files unexpected:\n%s", out)
}
for _, p := range []string{"a_amd64.s", filepath.Join("sub", "b_amd64.s")} {
b, _ := os.ReadFile(p)
if !strings.Contains(string(b), "\tRET") {
t.Errorf("%s not formatted in place:\n%s", p, b)
}
}
for _, p := range []string{filepath.Join("_refs", "c_amd64.s"), filepath.Join(".git", "d_amd64.s")} {
b, _ := os.ReadFile(p)
if string(b) != string(unformatted) {
t.Errorf("%s must not be touched:\n%s", p, b)
}
}
// Second pass: everything is canonical, nothing is listed.
out, _, code = capture(func() int { return cmdFmt(nil) })
if code != 0 || out != "" {
t.Errorf("second pass: code=%d out=%q, want a no-op", code, out)
}
}
func TestCmdTokens(t *testing.T) {
path := writeTemp(t, "f_amd64.s", clean)
out, _, code := capture(func() int { return cmdTokens([]string{path}) })
@@ -152,15 +200,32 @@ func TestCmdFmtWrite(t *testing.T) {
func TestUsage(t *testing.T) {
var b bytes.Buffer
usage(&b)
if !strings.Contains(b.String(), "gasm") {
t.Errorf("usage text unexpected:\n%s", b.String())
out := b.String()
for _, want := range []string{
"gasm", "Commands:", "Flags:", "--help", "--version",
"tokens", "parse", "fmt", "lint", "asm", "lsp", "version",
} {
if !strings.Contains(out, want) {
t.Errorf("usage text missing %q:\n%s", want, out)
}
}
}
func TestCmdVersion(t *testing.T) {
out, _, code := capture(func() int { return cmdVersion() })
if code != 0 {
t.Fatalf("code = %d", code)
}
if !strings.Contains(out, version) {
t.Errorf("version output %q does not mention %q", out, version)
}
}
func TestCmdArgErrors(t *testing.T) {
// Missing file arguments produce a usage error (code 2).
if _, _, code := capture(func() int { return cmdFmt(nil) }); code != 2 {
t.Errorf("cmdFmt() code = %d, want 2", code)
// A missing path is an error (code 1); cmdFmt with no arguments is the
// recursive mode now, covered by TestCmdFmtRecursive.
if _, _, code := capture(func() int { return cmdFmt([]string{"no/such/path"}) }); code != 1 {
t.Errorf("cmdFmt(missing path) code = %d, want 1", code)
}
if _, _, code := capture(func() int { return cmdLint(nil) }); code != 2 {
t.Errorf("cmdLint() code = %d, want 2", code)
+93 -29
View File
@@ -136,13 +136,18 @@ Two deeper analyses sit on top of the AST:
control-flow graph (basic blocks split at labels and after branches, with
fall-through and jump-target edges), computes a conservative per-instruction
register def/use, and runs the standard backward liveness iteration to a fixed
point. On top of that it flags a **callee-saved register that is written but
never saved and restored** — the per-architecture callee-saved set is amd64
`BX/BP/R12–R15`, arm64 `R19–R30`, riscv64 `X1/X8/X9/X18–X27`, loong64
`R1/R22–R31`. This is an *audit*: the runtime's own assembly clobbers these
registers freely (it controls both sides of the call), so the rule is
advisory there, but in hand-written kernels called from ordinary Go code a
clobber is a genuine ABI violation. It runs only on macro-free files, where
point. On top of that it flags writes to the registers the **Go ABI** fixes
across calls that are never saved and restored — calibrated from
`cmd/compile/abi-internal.md`, *not* the platform ABI: Go's stack-based ABI0
has no System V style callee-saved registers (amd64 `BX`, `R12`–`R15` and
the like are caller-saved or permanent scratch, and hand-written kernels may
clobber them freely). The audited set is the frame pointer and the
the frame pointer, the goroutine pointer per architecture (amd64 `BP`/`R14`, arm64 `R18`/`R28`/
`R29`, riscv64 `X27`, loong64 `R22`); the goroutine pointer is reported only
when the function can reach the runtime — it is not `NOSPLIT` or makes a
call — since the ABI0 transition machinery restores it on those paths, and
NOSPLIT call-free leaves may use it (the runtime's own assembly does). It
runs only on macro-free files, where
no opaque macro can perform the save/restore.
- **`funcdata-pcdata`.** `FUNCDATA $idx, sym(SB)` and `PCDATA $idx, $val` are
checked for well-formed operands (arity, immediate index and value, symbol
@@ -152,9 +157,15 @@ Two deeper analyses sit on top of the AST:
### `format`
The formatter works on the **token stream, not the AST**, so it preserves
every line — comments and blanks included. It only normalises indentation,
operand spacing and per-function mnemonic alignment. It is idempotent and its
output always round-trips through the parser.
every line — comments and blanks included. It normalises indentation, operand
spacing, per-function mnemonic alignment and blank-line layout: a new block
(a label, `TEXT` or `GLOBL`) is preceded by exactly one blank line (comments
leading a block stay with it), runs of blanks collapse to one, and a `RET`
terminates the body so the next function's doc comment stays at column 0. It
is idempotent and its output always round-trips through the parser. With a
directory argument — or none — it reformats every `.s` file below it in
place and lists the files changed, the way `go fmt` does (`.` and `_`
directories are skipped).
### `lsp`
@@ -182,33 +193,86 @@ Every encoding is validated by decoding it again with `golang.org/x/arch` — th
one module dependency, used in tests only and never linked into the binary.
On top of the encoder, `Assemble` walks a parsed `TEXT` body, converts each
operand to an encoder operand, and lays the instructions out in two passes so
local labels resolve to fixed rel32 jump offsets. The `FP`/`SP` pseudo-
operand to an encoder operand, and lays the instructions out so local labels
resolve to relative jump offsets: jumps start in the short (rel8) form and
expand to rel32 when the settled displacement does not fit, iterating to a
fixed point, and jump-to-jump chains are folded (a conditional jump to a label
whose only instruction is an unconditional jump is redirected to the ultimate
target) exactly as the Go toolchain's linker does before it encodes branches.
The `FP`/`SP` pseudo-
registers are translated onto the hardware stack pointer — `x+N(FP)` becomes
`(N+8)(SP)` for a zero-frame function and `(N+frame+16)(SP)` once a frame
pointer is set up, with the matching Go prologue/epilogue generated — so the
output is byte-identical to the Go assembler for these cases. SIMD is handled
by a VEX (AVX/AVX2) encoder — the two- and three-byte VEX prefixes with XMM/YMM
registers — across seven operand forms: the three-operand NDS form, the
two-operand reg/rm form, the immediate-shift form, the immediate shuffle form
(`VPSHUFD`, `VPERMQ`), the three-operand-plus-immediate form (`VSHUFPD`,
registers — across eight operand forms: the three-operand NDS form, the
two-operand reg/rm form, the immediate-shift form (plus the variable-count
shifts, which share the NDS shape with the count in an XMM register or
memory), the immediate shuffle form (`VPSHUFD`, `VPERMQ`), the
three-operand-plus-immediate form (`VSHUFPD`,
`VPERM2I128`, `VINSERTI128`), the lane-extract form (`VEXTRACTI128`,
lane-extract form (`VEXTRACTI128`,
`VEXTRACTF128`, where the YMM source occupies the reg field and the XMM or
memory destination r/m), the direction-sensitive moves (`VMOVDQU`, `VMOVUPD`,
`VMOVD`, `VMOVQ`, `VMOVSD`), the floating-point and FMA arithmetic (`VADDPD`,
`VMULPD`, `VXORPD`, `VUNPCKHPD`, the scalar `VADDSD`/`VMULSD`, `VCVTDQ2PD`,
`VFMADD231PD`) and the no-operand `VZEROUPPER` — together with `VPERMD`,
covering every integer, shuffle and FP instruction the go-flac AVX2 kernels
use. Every encoding is validated two ways: by round-trip decoding
through `golang.org/x/arch`, and byte-for-byte against the machine code the
real Go assembler emits (which also locks the v̄vvv = 1111 rule for unused
vvvv fields — a value the hardware rejects with #UD and the decoder silently
ignores). This increment covers register / memory / immediate / FP-frame
operands, local-label jumps and these VEX SIMD forms; EVEX / AVX-512, `SB`
(global symbol) operands (relocations), a handful of scalar gaps the kernels
hit (`CMOVcc`, `SETcc`, `LZCNT`, `MOVSX`/`MOVZX`) and object-file emission
are the rest of Phase 2.
`VMOVD`, `VMOVQ`, `VMOVSD`), the floating-point and FMA arithmetic — the
packed double operations (`VADDPD`/`VSUBPD`/`VMULPD`/`VDIVPD`/`VMINPD`/
`VMAXPD`), the unpacks (`VUNPCKHPD`/`VUNPCKLPD`), the scalar SD and SS
operations, `VMOVDDUP`, `VXORPD`, the width-changing conversions
(`VCVTDQ2PS`, `VCVTPS2PD`, `VCVTDQ2PD`, and the `VCVTPD2DQX`/`Y` and
`VCVTTPD2DQX`/`Y` spellings, whose length follows the wider source) and
`VFMADD231PD` — and the no-operand `VZEROUPPER`, together with `VPERMD` and
the scalar families (`CMOVcc`, `SETcc`, `LZCNT`/`TZCNT`, the extending moves,
`CVTSx2SD`, `IMUL3`) and the EVEX (AVX-512) prefix — the four-byte prefix with
5-bit register fields (Z0–Z31, X/Y 16–31, with the mod=11 quirk that carries
rm[4] in X̄), opmask registers (K0–K7 as operands, mask destinations and
explicit merging/zeroing masks — written the way Go writes them, as a K
operand among the operands plus a `.Z` mnemonic suffix), and the compressed
disp8×N displacement, whose multiplier follows the memory operand's size —
covering every instruction the go-flac and go-lz4 AVX2/AVX-512 kernels use,
plus the common AVX-512 F/BW integer set, the floating-point and conversion
set (the packed double and single arithmetic, the scalar SD/SS forms —
whose EVEX encodings serve masked and zeroing use — `VMOVDDUP`, the
replicating moves, and the width-changing conversions, including the
`VCVTPD2DQ`/`VCVTTPD2DQ` family whose length follows the wider source
operand), and the wider AVX-512 set: ternary logic, lane shuffles, inserts
and extracts, compares with an opmask destination, the permutes, the
expand/compress family, the broadcasts, the opmask-register instructions
(KAND/KOR/KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST and KMOVQ), the aligned
moves and the remaining extending/narrowing moves. The EVEX mnemonic
suffixes — rounding modes (.RN_SAE/.RD_SAE/.RU_SAE/.RZ_SAE),
suppress-all-exceptions (.SAE) and memory broadcast (.BCST) — set the EVEX
b bit and the L'L rounding-control field (broadcast keeps the vector length
and scales disp8 by the element size), and combine with the .Z zeroing
suffix. Every encoding is validated two ways: by
round-trip decoding through `golang.org/x/arch`, and byte-for-byte against
the machine code the real Go assembler emits — a comparison that holds for
whole functions: all 27 functions of both kernels assemble to exactly the Go
toolchain's bytes, the lone exception being the displacements of the
static-constant loads, which the Go linker fills at link time.
File-level assembly (`AssembleFile`) goes beyond single functions: it
materialises the file's static symbols (`GLOBL`/`DATA`) in a data section
behind the code and resolves references to them (`mask<>(SB)`) to
RIP-relative loads whose displacements point inside the resulting image, so
the bytes are self-consistent at any base address. References to symbols no
`GLOBL` defines are kept as relocations on the function layout, and the
object-file emitters turn the whole image into a linkable object: the ELF
and Mach-O writers (`gasm asm --format elf|macho`) lay the code and data out
as `.text`/`.data` (or `__text`/`__data`) sections, export a symbol per
`TEXT` and `GLOBL` (the `<>` ones local, the rest global) and emit one
PC-relative relocation per static-symbol reference — undefined external
symbols included, so the output links with the system toolchain. The GOOBJ
emitter (`gasm asm --format goobj`) writes the format the Go linker consumes
directly: the functions as non-package symbols (the way `cmd/asm` records
assembly symbols), the `GLOBL` data, one `FuncInfo` per function and the
pc-value tables — `pcsp` built from the prologue and epilogue stack
boundaries, plus flat `pcfile`, `pcline` and `pcinline` tables — so a
gasm-assembled object drops into a `go build` in place of the toolchain's.
The object preamble (the version-and-experiment header the linker compares
verbatim) is captured from the installed `go tool asm`, so the output is
always consistent with the toolchain that links it. External cross-package
references and the implicit funcdata/DWARF symbols remain future work (the
linker fills the latter's defaults); the rest of Phase 2 is those, the
remaining EVEX forms and the other architectures.
## Extension points
+88 -13
View File
@@ -27,14 +27,6 @@ func Source(path, src string) string {
mnemLen int
funcID int
}
const (
kBlank = iota
kComment
kPreproc
kDirective
kLabel
kInstr
)
infos := make([]info, len(lines))
funcID := -1
@@ -70,8 +62,8 @@ func Source(path, src string) string {
infos[i] = inf
}
// Second pass: render.
var b strings.Builder
// Second pass: render each line.
outs := make([]outLine, 0, len(lines))
inBody := false
for i, line := range lines {
inf := infos[i]
@@ -99,11 +91,94 @@ func Source(path, src string) string {
}
case kInstr:
out = renderInstr(line, maxWidth[inf.funcID])
// A RET ends the body for indentation purposes: comments that
// follow it — typically the next function's doc comment — belong
// at column 0, not inside the finished function.
if strings.EqualFold(line[0].Text, "RET") {
inBody = false
}
}
b.WriteString(strings.TrimRight(out, " \t"))
b.WriteByte('\n')
outs = append(outs, outLine{kind: inf.kind, text: strings.TrimRight(out, " \t")})
}
return b.String()
return normalizeSpacing(outs)
}
// Line classification, shared by the formatting passes.
const (
kBlank = iota
kComment
kPreproc
kDirective
kLabel
kInstr
)
// outLine is one rendered line together with its classification.
type outLine struct {
kind int
text string
}
// normalizeSpacing enforces the canonical blank-line layout: runs of blank
// lines collapse to one, and a new block — a label, or a TEXT or GLOBL
// directive — is preceded by exactly one blank line. Comments immediately
// above a block belong to it, so the blank line is inserted before them. No
// blank line is forced at the top of the file, right after a TEXT (the
// function's first label), or between stacked labels that share an address.
func normalizeSpacing(outs []outLine) string {
blockStart := func(ol outLine) bool {
switch ol.kind {
case kLabel:
return true
case kDirective:
// TEXT and GLOBL open a block; DATA continues a GLOBL block.
return strings.HasPrefix(ol.text, "TEXT") || strings.HasPrefix(ol.text, "GLOBL")
}
return false
}
insert := make([]bool, len(outs))
for i, ol := range outs {
if !blockStart(ol) {
continue
}
j := i
for j > 0 && outs[j-1].kind == kComment {
j--
}
if j == 0 {
continue // top of file
}
switch prev := outs[j-1]; {
case prev.kind == kBlank, prev.kind == kLabel:
continue // already separated, or stacked labels
case prev.kind == kDirective && strings.HasPrefix(prev.text, "TEXT"):
continue // the function's first label
}
insert[j] = true
}
var b strings.Builder
prevBlank := true // also suppresses leading blanks
for i, ol := range outs {
if insert[i] && !prevBlank {
b.WriteByte('\n')
}
if ol.kind == kBlank {
if !prevBlank {
b.WriteByte('\n')
}
prevBlank = true
continue
}
b.WriteString(ol.text)
b.WriteByte('\n')
prevBlank = false
}
out := strings.TrimRight(b.String(), "\n")
if out == "" {
return ""
}
return out + "\n"
}
// renderInstr renders an instruction line: a tab, the mnemonic padded to the
+96
View File
@@ -39,6 +39,102 @@ func TestGolden(t *testing.T) {
}
}
// TestDocCommentIndent checks that a doc comment preceding a TEXT directive
// sits at column 0 even when another function (ending in RET) precedes it —
// the RET must terminate the previous body for indentation purposes.
func TestDocCommentIndent(t *testing.T) {
in := "#include \"textflag.h\"\n" +
"\n" +
"// func first()\n" +
"TEXT ·first(SB), NOSPLIT, $0\n" +
"XORQ AX, AX\n" +
"RET\n" +
"\n" +
"// func second()\n" +
"TEXT ·second(SB), NOSPLIT, $0\n" +
"RET\n"
want := "#include \"textflag.h\"\n" +
"\n" +
"// func first()\n" +
"TEXT ·first(SB), NOSPLIT, $0\n" +
"\tXORQ AX, AX\n" +
"\tRET\n" +
"\n" +
"// func second()\n" +
"TEXT ·second(SB), NOSPLIT, $0\n" +
"\tRET\n"
got := Source("d_amd64.s", in)
if got != want {
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
}
// Body comments stay indented.
body := "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n// inside the body\nXORQ AX, AX\nRET\n"
gotBody := Source("b_amd64.s", body)
if !strings.Contains(gotBody, "\t// inside the body\n") {
t.Fatalf("body comment must stay indented:\n%q", gotBody)
}
}
// TestBlankLines checks the blank-line canonicalisation: exactly one blank
// line before a new block (a label, or TEXT/GLOBL), runs of blanks collapsed
// to one, and no blank forced after TEXT, between stacked labels, or at the
// top of the file. Leading comments belong to the block they precede.
func TestBlankLines(t *testing.T) {
in := "#include \"textflag.h\"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"first:\n" + // first label: no blank after TEXT
"XORQ AX, AX\n" +
"JMP next\n" + // unlabeled glue: fmt inserts a blank before next:
"next:\n" +
"stacked:\n" + // stacked labels share an address: no blank between
"INCQ AX\n" +
"\n" +
"\n" + // two blanks collapse to one
"// separated block\n" + // comment belongs to the label below
"later:\n" +
"RET\n" +
"// func g()\n" + // doc comment: blank goes before it
"TEXT ·g(SB), NOSPLIT, $0\n" +
"RET\n" +
"GLOBL ·mask(SB), RODATA, $8\n" + // blank before GLOBL…
"DATA ·mask+0(SB)/4, $1\n" + // …but not before DATA
"\n" +
"\n" +
"\n" // trailing blanks dropped
want := "#include \"textflag.h\"\n" +
"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"first:\n" +
"\tXORQ AX, AX\n" +
"\tJMP next\n" +
"\n" +
"next:\n" +
"stacked:\n" +
"\tINCQ AX\n" +
"\n" +
"\t// separated block\n" + // body comment before a label stays indented
"later:\n" +
"\tRET\n" +
"\n" +
"// func g()\n" +
"TEXT ·g(SB), NOSPLIT, $0\n" +
"\tRET\n" +
"\n" +
"GLOBL ·mask(SB), RODATA, $8\n" +
"DATA ·mask+0(SB)/4, $1\n"
got := Source("b_amd64.s", in)
if got != want {
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
}
if again := Source("b_amd64.s", got); again != got {
t.Fatalf("not idempotent:\n%q", again)
}
}
func TestOperandSpacing(t *testing.T) {
cases := map[string]string{
"4(SI)": "4(SI)",
+1 -1
View File
@@ -3,7 +3,7 @@
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
version := "0.2.0"
version := "0.13.0"
default:
@just --list
+61 -7
View File
@@ -242,7 +242,7 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
}
}
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) {
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) && !maskedEvex(mnem, st.Operands) {
if in, ok := tab.Lookup(mnem); ok && in.MinOps >= 0 {
n := len(st.Operands)
if n < in.MinOps || n > in.MaxOps {
@@ -319,18 +319,27 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
}
}
// Register liveness: a callee-saved register that is written but never
// saved and restored is clobbered across the call. The check runs over the
// control-flow graph and is skipped for macro-using files, where an opaque
// macro may perform the save/restore.
// Register liveness: a register the Go ABI fixes across calls that is
// written but never saved and restored is clobbered. The check runs over
// the control-flow graph and is skipped for macro-using files, where an
// opaque macro may perform the save/restore.
if doLabelChecks && archKnown && !cfg.Disable[CodeRegisterClobber] {
live := analyzeLiveness(t, cfg.Arch)
if clobbered := clobberedCalleeSaved(live, cfg.Arch); len(clobbered) > 0 {
always, rt := clobberedGoFixed(live, cfg.Arch, reachesRuntime(t))
if len(always) > 0 {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeRegisterClobber,
Message: fmt.Sprintf("callee-saved register(s) %s written but never saved/restored", strings.Join(clobbered, ", ")),
Message: fmt.Sprintf("register(s) %s written but never saved/restored: fixed by the Go ABI (frame/goroutine pointer)", strings.Join(always, ", ")),
})
}
if len(rt) > 0 {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeRegisterClobber,
Message: fmt.Sprintf("goroutine-pointer register(s) %s written but never saved/restored in a function that can reach the Go runtime", strings.Join(rt, ", ")),
})
}
}
@@ -341,6 +350,29 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
return out
}
// reachesRuntime reports whether a function can reach the Go runtime: it is
// not NOSPLIT (so the stack-split and traceback machinery runs) or it makes a
// CALL. Goroutine-pointer registers must survive such functions; a NOSPLIT
// leaf may clobber them, since the ABI0 transition restores them (the
// runtime's own assembly relies on this, e.g. R14 on amd64).
func reachesRuntime(t *ast.Text) bool {
nosplit := false
for _, f := range t.Flags {
if strings.EqualFold(f, "NOSPLIT") {
nosplit = true
}
}
for _, s := range t.Body {
if in, ok := s.(*ast.Instr); ok {
switch strings.ToUpper(in.Mnemonic.Text) {
case "CALL", "BL", "JAL": // amd64, arm64/loong64, riscv64 calls
return true
}
}
}
return !nosplit
}
// usesFPArgs reports whether a function references its arguments through the FP
// pseudo-register — i.e. it uses the stack-based ABI0 layout, where the
// declared argument size must match the signature.
@@ -406,6 +438,28 @@ func isMacroInvocation(mnem string, macros map[string]bool) bool {
return strings.Contains(mnem, "_") || macros[mnem]
}
// maskedEvex reports whether the instruction is a masked EVEX form: the
// mnemonic carries a .Z suffix, or the operand list contains an opmask
// register (K1–K7). Either way the operand count differs from the unmasked
// form, so count checks are skipped.
func maskedEvex(mnem string, ops []*ast.Operand) bool {
if strings.Contains(mnem, ".") {
return true
}
for _, op := range ops {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Base == "" &&
op.Addr.Index == "" && op.Addr.Sym.Pseudo == "" && isMaskReg(op.Addr.Sym.Name) {
return true
}
}
return false
}
// isMaskReg reports whether name is an opmask register K0–K7.
func isMaskReg(name string) bool {
return len(name) == 2 && name[0] == 'K' && name[1] >= '0' && name[1] <= '7'
}
// isConditionalDirective reports whether a preprocessor directive (the text
// after '#') is a conditional-compilation directive whose branches the parser
// cannot resolve.
+25 -5
View File
@@ -48,11 +48,11 @@ func TestFixtureIsClean(t *testing.T) {
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
// The fixture mirrors the go-flac kernels, which use callee-saved registers
// (BX, R13) without saving them; the register-clobber audit flags that by
// design. This test targets the other rules, so the audit is disabled here
// (it is covered by TestRegisterClobber).
diags := File(f, Config{Arch: arch.AMD64, Disable: map[string]bool{CodeRegisterClobber: true}})
// The fixture mirrors the go-flac kernels, which write the Go ABI0
// scratch registers (BX, R13) without saving them — legal under Go's
// stack-based ABI, so the register-clobber audit stays silent and the
// fixture must lint entirely clean.
diags := File(f, Config{Arch: arch.AMD64})
if len(diags) != 0 {
t.Fatalf("expected no diagnostics on the fixture, got %+v", diags)
}
@@ -187,6 +187,26 @@ done:
}
}
// TestEvexMaskingRecognised checks that masked EVEX forms — the .Z suffix and
// an explicit K operand — are recognised and exempt from operand-count
// checks.
func TestEvexMaskingRecognised(t *testing.T) {
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
VPADDD.Z Z1, Z2, K2, Z3
VPMINSD Z1, Z2, K5, Z3
VMOVDQU8 Z1, K3, (SI)
RET
`)
if codes(diags)[CodeUnknownInstr] != 0 {
t.Fatalf("masked EVEX must be recognised: %+v", diags)
}
if codes(diags)[CodeOperandCount] != 0 {
t.Fatalf("masked operand counts must not be flagged: %+v", diags)
}
}
func TestArm64AddressingSuffix(t *testing.T) {
// .W (pre-index) and .P (post-index) suffixes must resolve to the base
// instruction.
+54 -53
View File
@@ -4,7 +4,6 @@
package lint
import (
"fmt"
"sort"
"strings"
@@ -248,7 +247,7 @@ func instrEffect(in *ast.Instr, a arch.Arch) regEffect {
}
compare := isCompare(mnem)
dstIdx := dstIndex(in, a)
dstIdx := dstIndex(in)
for i, op := range in.Operands {
r := gprName(op, a)
@@ -281,13 +280,11 @@ func instrEffect(in *ast.Instr, a arch.Arch) regEffect {
return eff
}
// dstIndex returns the operand index of the destination register: last for the
// Plan 9 (amd64) spelling, first for arm64/riscv64/loong64.
func dstIndex(in *ast.Instr, a arch.Arch) int {
if a == arch.AMD64 {
return len(in.Operands) - 1
}
return 0
// dstIndex returns the operand index of the destination register: in Plan 9
// notation the destination is the last operand on every architecture Go
// supports (amd64, arm64, riscv64 and loong64 alike).
func dstIndex(in *ast.Instr) int {
return len(in.Operands) - 1
}
// isCompare reports whether the mnemonic only reads its operands (setting flags).
@@ -372,41 +369,37 @@ func sameSet(a, b map[string]bool) bool {
return true
}
// calleeSavedGPRs returns the general-purpose registers an assembly function
// must preserve for its caller, using the register names the assembler accepts
// for each architecture.
func calleeSavedGPRs(a arch.Arch) map[string]bool {
// goFixedGPRs returns the general-purpose registers the Go ABI designates as
// fixed across calls — the ones hand-written assembly must not permanently
// clobber. This follows cmd/compile/abi-internal.md, not the platform ABI:
// Go's stack-based ABI0 (which hand-written assembly uses) has no System V
// style callee-saved registers, so clobbering the argument and scratch
// registers (amd64 BX, R12, R13, R15, …) is legal.
//
// Two groups are returned. always holds registers whose loss is never safe.
// runtime holds registers that survive an ABI0 leaf only because the
// transition machinery restores them (on amd64 the g pointer is reloaded
// from TLS): clobbering them is safe exactly in NOSPLIT functions that make
// no calls, which is how the runtime's own assembly uses them.
func goFixedGPRs(a arch.Arch) (always, runtime map[string]bool) {
switch a {
case arch.AMD64:
return gprSet("BX", "BP", "R12", "R13", "R14", "R15")
// BP maintains the frame chain; R14 holds the current goroutine.
// R15 is scratch except in dynamically linked binaries, so it is not
// flagged.
return gprSet("BP"), gprSet("R14")
case arch.ARM64:
names := []string{"R29", "R30"} // FP, LR
for i := 19; i <= 28; i++ {
names = append(names, fmt.Sprintf("R%d", i))
}
return gprSet(names...)
// R18 is reserved for the OS on some platforms, R28 holds the current
// goroutine, R29 is the frame pointer.
return gprSet("R18", "R28", "R29"), nil
case arch.RISCV:
// RA (X1) and the S registers (X8, X9, X18–X27) are callee-saved.
names := []string{"X1", "RA", "X8", "X9", "S0", "S1", "FP"}
for i := 18; i <= 27; i++ {
names = append(names, fmt.Sprintf("X%d", i))
}
for i := 2; i <= 11; i++ {
names = append(names, fmt.Sprintf("S%d", i))
}
return gprSet(names...)
// X27 holds the current goroutine.
return gprSet("X27"), nil
case arch.LOONG64:
// RA (R1), FP (R22) and S0–S8 (R23–R31) are callee-saved.
names := []string{"R1", "RA", "R22", "FP"}
for i := 23; i <= 31; i++ {
names = append(names, fmt.Sprintf("R%d", i))
}
for i := 0; i <= 8; i++ {
names = append(names, fmt.Sprintf("S%d", i))
}
return gprSet(names...)
// R22 holds the current goroutine.
return gprSet("R22"), nil
}
return nil
return nil, nil
}
func gprSet(names ...string) map[string]bool {
@@ -417,15 +410,16 @@ func gprSet(names ...string) map[string]bool {
return m
}
// clobberedCalleeSaved returns the callee-saved registers a function writes
// without also saving and restoring them — i.e. registers whose caller-owned
// value is lost across the call. It walks the blocks of the liveness analysis
// (so the control-flow graph is what supplies the instruction set) and
// aggregates each instruction's register effects.
func clobberedCalleeSaved(l *liveness, a arch.Arch) []string {
callee := calleeSavedGPRs(a)
if len(callee) == 0 {
return nil
// clobberedGoFixed returns the Go-ABI-fixed registers a function writes
// without also saving and restoring them. The first result lists registers
// whose loss is never safe; the second lists the goroutine-pointer class,
// whose loss is reported only when reachesRuntime is true (a non-NOSPLIT
// function, or one that makes calls — the ABI0 transition machinery restores
// the g pointer only on such paths).
func clobberedGoFixed(l *liveness, a arch.Arch, reachesRuntime bool) (always, runtime []string) {
alwaysSet, runtimeSet := goFixedGPRs(a)
if len(alwaysSet) == 0 && len(runtimeSet) == 0 {
return nil, nil
}
def := map[string]bool{}
saved := map[string]bool{}
@@ -444,12 +438,19 @@ func clobberedCalleeSaved(l *liveness, a arch.Arch) []string {
}
}
}
var out []string
for r := range callee {
if def[r] && !(saved[r] && restored[r]) {
out = append(out, r)
clobbered := func(set map[string]bool) []string {
var out []string
for r := range set {
if def[r] && !(saved[r] && restored[r]) {
out = append(out, r)
}
}
sort.Strings(out)
return out
}
sort.Strings(out)
return out
always = clobbered(alwaysSet)
if reachesRuntime {
runtime = clobbered(runtimeSet)
}
return always, runtime
}
+112 -16
View File
@@ -5,36 +5,132 @@ package lint
import "testing"
// TestRegisterClobber detects writes to callee-saved registers that are not
// saved and restored.
// TestRegisterClobber checks the register-clobber audit is calibrated to the
// Go ABI (cmd/compile/abi-internal.md), not the platform ABI: Go's
// stack-based ABI0 — which hand-written assembly uses — has no System V
// style callee-saved registers, so argument and scratch registers may be
// clobbered freely. Only the registers the ABI fixes across calls (the
// frame pointer, the goroutine pointer, OS-reserved registers) are audited.
func TestRegisterClobber(t *testing.T) {
// BX (callee-saved on amd64) is written but never saved → clobbered.
clob := lintSrc(t, "#include \"textflag.h\"\n"+
// amd64: BX, R12, R13 and R15 are argument/permanent-scratch registers in
// Go ABI0 — writing them unsaved is legal (a System V calibration would
// report all of these).
scratch := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVQ CX, BX\n"+
"\tXORL R12, R12\n"+
"\tXORL R13, R13\n"+
"\tXORL R15, R15\n"+
"\tRET\n")
if codes(clob)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved callee-saved write should be flagged: %+v", clob)
if codes(scratch)[CodeRegisterClobber] != 0 {
t.Fatalf("Go ABI0 scratch registers must not be flagged: %+v", scratch)
}
// Saved and restored → preserved.
// amd64: R14 (the goroutine pointer) in a NOSPLIT function without calls
// is the runtime's own pattern — the ABI0 transition restores it — so it
// is not flagged.
leaf := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tXORL R14, R14\n"+
"\tRET\n")
if codes(leaf)[CodeRegisterClobber] != 0 {
t.Fatalf("R14 in a NOSPLIT leaf must not be flagged: %+v", leaf)
}
// amd64: R14 in a function that makes a call is a genuine hazard.
withCall := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tXORL R14, R14\n"+
"\tCALL ·g(SB)\n"+
"\tRET\n")
if codes(withCall)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved R14 with a call should be flagged: %+v", withCall)
}
// amd64: R14 in a non-NOSPLIT function is a hazard regardless of calls.
split := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), $0\n"+
"\tMOVQ CX, R14\n"+
"\tRET\n")
if codes(split)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved R14 in a non-NOSPLIT function should be flagged: %+v", split)
}
// amd64: R14 saved and restored around the call is preserved.
saved := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $8\n"+
"\tPUSHQ BX\n"+
"\tMOVQ CX, BX\n"+
"\tPOPQ BX\n"+
"\tPUSHQ R14\n"+
"\tXORL R14, R14\n"+
"\tCALL ·g(SB)\n"+
"\tPOPQ R14\n"+
"\tRET\n")
if codes(saved)[CodeRegisterClobber] != 0 {
t.Fatalf("saved/restored register must not be flagged: %+v", saved)
t.Fatalf("saved/restored R14 must not be flagged: %+v", saved)
}
// A caller-saved register (CX) is fine to write.
caller := lintSrc(t, "#include \"textflag.h\"\n"+
// amd64: BP maintains the frame chain and is always audited.
bp := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVQ $1, CX\n"+
"\tMOVQ CX, BP\n"+
"\tRET\n")
if codes(caller)[CodeRegisterClobber] != 0 {
t.Fatalf("caller-saved register must not be flagged: %+v", caller)
if codes(bp)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved BP write should be flagged: %+v", bp)
}
// arm64: R20 is scratch; R28 (goroutine pointer) and R18 (OS-reserved)
// are fixed by the Go ABI.
armScratch := lintSrcArch(t, "t_arm64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVD R0, R20\n"+
"\tRET\n")
if codes(armScratch)[CodeRegisterClobber] != 0 {
t.Fatalf("arm64 scratch register must not be flagged: %+v", armScratch)
}
armG := lintSrcArch(t, "t_arm64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVD R0, R28\n"+
"\tRET\n")
if codes(armG)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved arm64 R28 write should be flagged: %+v", armG)
}
armReserved := lintSrcArch(t, "t_arm64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVD R0, R18\n"+
"\tRET\n")
if codes(armReserved)[CodeRegisterClobber] != 1 {
t.Fatalf("arm64 R18 write should be flagged: %+v", armReserved)
}
// riscv64: X27 holds the goroutine; X5–X7 are scratch.
riscScratch := lintSrcArch(t, "t_riscv64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOV X5, X6\n"+
"\tRET\n")
if codes(riscScratch)[CodeRegisterClobber] != 0 {
t.Fatalf("riscv64 scratch register must not be flagged: %+v", riscScratch)
}
riscG := lintSrcArch(t, "t_riscv64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOV X5, X27\n"+
"\tRET\n")
if codes(riscG)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved riscv64 X27 write should be flagged: %+v", riscG)
}
// loong64: R22 holds the goroutine; R5–R19 are argument/scratch.
loongScratch := lintSrcArch(t, "t_loong64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVV R5, R6\n"+
"\tRET\n")
if codes(loongScratch)[CodeRegisterClobber] != 0 {
t.Fatalf("loong64 scratch register must not be flagged: %+v", loongScratch)
}
loongG := lintSrcArch(t, "t_loong64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVV R5, R22\n"+
"\tRET\n")
if codes(loongG)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved loong64 R22 write should be flagged: %+v", loongG)
}
}
+26 -14
View File
@@ -375,6 +375,11 @@ func parseImmediate(g []token.Token) ast.Immediate {
if v, ok := tryInt(text); ok {
imm.Val = v
imm.HasVal = true
} else if u, err := strconv.ParseUint(text, 0, 64); err == nil && !imm.Neg {
// Unsigned 64-bit literals (DATA mask<>+8(SB)/8, $0x8000…)
// overflow int64; keep the bit pattern.
imm.Val = int64(u)
imm.HasVal = true
} else {
imm.Float = text
}
@@ -400,22 +405,29 @@ func parseAddress(g []token.Token) ast.Address {
}
i := 0
// Optional leading displacement before a '(' base group.
if isSignedNumber(g, i) && i+1 < len(g) && g[i+1].Kind == token.LParen {
neg := false
if g[i].Kind == token.Minus {
neg = true
i++
} else if g[i].Kind == token.Plus {
i++
// Optional leading displacement before a '(' base group. A sign pushes
// the parenthesis one token further out: -4(DX) has it at i+2.
if isSignedNumber(g, i) {
paren := i + 1
if g[i].Kind == token.Minus || g[i].Kind == token.Plus {
paren = i + 2
}
if i < len(g) && g[i].Kind == token.Number {
addr.Offset = parseInt(g[i].Text)
addr.HasOff = true
if neg {
addr.Offset = -addr.Offset
if paren < len(g) && g[paren].Kind == token.LParen {
neg := false
if g[i].Kind == token.Minus {
neg = true
i++
} else if g[i].Kind == token.Plus {
i++
}
if i < len(g) && g[i].Kind == token.Number {
addr.Offset = parseInt(g[i].Text)
addr.HasOff = true
if neg {
addr.Offset = -addr.Offset
}
i++
}
i++
}
}
// First parenthesised group: the base register.
+40
View File
@@ -34,6 +34,46 @@ func texts(f *ast.File) []*ast.Text {
return out
}
// TestNegativeDisplacement is a regression test for a leading negative
// displacement with a base and index: the sign pushed the parenthesis one
// token further out than the lookahead expected, and the whole address used
// to parse empty.
func TestNegativeDisplacement(t *testing.T) {
f, errs := Parse("neg_amd64.s", `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
LEAQ -4(DX)(R9*4), R9
MOVQ +8(AX), BX
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
fn := texts(f)[0]
var leaq, movq *ast.Instr
for _, s := range fn.Body {
if in, ok := s.(*ast.Instr); ok {
switch in.Mnemonic.Text {
case "LEAQ":
leaq = in
case "MOVQ":
movq = in
}
}
}
if leaq == nil || movq == nil {
t.Fatalf("instructions not parsed: leaq=%v movq=%v", leaq, movq)
}
a := leaq.Operands[0].Addr
if a.Base != "DX" || a.Index != "R9" || a.Scale != 4 || a.Offset != -4 || !a.HasOff {
t.Errorf("LEAQ addr = %+v, want -4(DX)(R9*4)", a)
}
b := movq.Operands[0].Addr
if b.Base != "AX" || b.Offset != 8 || !b.HasOff {
t.Errorf("MOVQ addr = %+v, want +8(AX)", b)
}
}
func TestParseSample(t *testing.T) {
f := mustParse(t, "../testdata/sample_amd64.s")