Compare commits

..
11 Commits
Author SHA1 Message Date
petrbalvin cf6bc6987e fix(ci): pass the upload file to curl, not its interpolation
Test / test (push) Successful in 2m26s
Release / gates (push) Successful in 2m25s
Release / build (amd64, linux) (push) Successful in 1m15s
Release / build (arm64, linux) (push) Successful in 1m16s
Release / build (loong64, linux) (push) Successful in 1m16s
Release / build (riscv64, linux) (push) Successful in 1m16s
Release / release (push) Successful in 34s
2026-09-22 01:31:49 +02:00
petrbalvin ff7b1452b1 docs: name 0.35.0 as the supported release
Test / test (push) Successful in 2m33s
Release / gates (push) Successful in 2m29s
Release / build (amd64, linux) (push) Successful in 1m18s
Release / build (arm64, linux) (push) Successful in 1m20s
Release / build (loong64, linux) (push) Successful in 1m17s
Release / build (riscv64, linux) (push) Successful in 1m26s
Release / release (push) Failing after 35s
2026-09-22 00:52:56 +02:00
petrbalvin 517c1cea25 chore: prepare release v0.35.0
Test / test (push) Successful in 2m33s
Release / gates (push) Failing after 46s
Release / build (amd64, linux) (push) Skipped
Release / build (arm64, linux) (push) Skipped
Release / build (loong64, linux) (push) Skipped
Release / build (riscv64, linux) (push) Skipped
Release / release (push) Skipped
2026-09-22 00:44:10 +02:00
petrbalvin a3e3010e0f fix(cmd): resolve the runtime header test GOROOT from the go command
Test / test (push) Successful in 2m39s
2026-09-21 22:46:07 +02:00
petrbalvin 057c4eb545 docs: complete the release delta in the changelog and readme 2026-09-21 22:45:56 +02:00
petrbalvin f720381d43 feat(asm): the segment-absolute and crash-store forms GOROOT writes
Test / test (push) Failing after 2m28s
Assisted-by: GLM 5.3 Flash
2026-09-21 22:19:53 +02:00
petrbalvin 2c9042d62c feat(asm): PCALIGN alignment on amd64
Assisted-by: GLM 5.3 Flash
2026-09-21 22:00:30 +02:00
petrbalvin 82ef289d3a feat(asm): the immediate multiply and arm64 indirect branches GOROOT writes
Assisted-by: GLM 5.3 Flash
2026-09-21 21:50:11 +02:00
petrbalvin 7246b0e002 feat(asm): the TLS access pair in the toolchain's one-instruction form
Assisted-by: GLM 5.3 Flash
2026-09-21 21:35:15 +02:00
petrbalvin 8cfd40aac8 feat(asm): the operand forms and defines GOROOT writes
Assisted-by: GLM 5.3 Flash
2026-09-21 21:17:34 +02:00
petrbalvin 5382c9a8e4 feat(audit): list every corpus failure per architecture 2026-09-21 21:17:34 +02:00
22 changed files with 847 additions and 123 deletions
+4 -1
View File
@@ -342,7 +342,10 @@ jobs:
my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}}, my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}},
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}}, q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
q{-H}, q{Content-Type: application/octet-stream}, q{-H}, q{Content-Type: application/octet-stream},
q{-X}, q{POST}, q{--data-binary}, qq{@$path}, # The @ must not sit inside a qq{} string: there it starts an
# array interpolation and the upload body collapses to empty,
# which Gitea stores as a 201-created zero-byte attachment.
q{-X}, q{POST}, q{--data-binary}, q{@} . $path,
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name}); qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name});
open(my $curl, q{-|}, @cmd) or die qq{curl: $!}; open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
my $code = <$curl>; my $code = <$curl>;
+96 -40
View File
@@ -9,31 +9,31 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added ### Added
- **Per-architecture reference pages.** [docs/asm/](docs/asm/README.md) -
gains AMD64, ARM64, RISCV64 and LOONG64: the register files and the
roles the ABI fixes, addressing, operand order with every special form, ## [0.35.0] - 2026-09-22
constants and materialisation, alignment, fences and the relocations
each target emits. An instruction inventory appendix per architecture ### Added
is generated from the toolchain's own tables by `just gen`, and the
regenerated tables recognise 147 more mnemonics than the previous - **The go_asm.h generator.** `gasm asm` generates the package's go_asm.h
release carried (arm64 107, riscv64 31, loong64 9). itself when an assembly file includes it: the Go files beside the source
- **The Plan 9 assembly language reference.** [docs/asm/](docs/asm/README.md) are type-checked for the target architecture and the constants and field
opens the complete language reference with its common core: the lexicon, offsets become assembler defines, so package-context files assemble with
statement structure and constant expressions, the operand grammar with no compiler and no `go build` in the loop. `-GOOS` selects the
the pseudo-registers and symbol naming, the directives and the function type-checking GOOS for GOOS-specific files, and the corpus audit derives
flag vocabulary, preprocessing with `#define` and `#include`, and the the GOOS from the file name.
Go-embedded layer (ABI0, prototypes, `go_asm.h`, `funcdata.h` and the - **ELF data relocations on arm64, riscv64 and loong64.** `gasm asm
runtime contract). Every claim is verified against `go tool asm` of --format elf` emits `.rela.data` for symbol-valued DATA initialisers on
Go 1.27.1 and gasm's differential tests; the per-architecture pages and every architecture (amd64 carried them already), so standalone ELF
generated instruction appendices follow. objects link on all four targets.
- **GOOBJ format specification.** [docs/GOOBJ.md](docs/GOOBJ.md) - **Corpus failure listing.** `gasm audit-instructions --corpus --list`
documents the Go object file format in full: both containers, the 96 prints every failing file with its failure reason, per architecture,
byte header and all 19 blocks, every structure with its byte instead of one representative file per reason.
offsets, symbol kinds and flag bits, all 106 relocation types with - **DATA with symbol values and relaxed symbol spellings.** DATA
the weak variants, aux symbols, the FuncInfo payload, the pc-value initialisers accept `$symbol(SB)` values, laid down as an absolute
table encoding, the content hashes and the builtin table, all relocation at the data field (GOOBJ on all four architectures and ELF
verified byte for byte against objects produced by Go 1.27.1's own on all four as of this release), and U+2215 is accepted inside symbol
tools. package paths.
- **Macro expansion and include splicing.** `gasm asm`, `gasm diff` and - **Macro expansion and include splicing.** `gasm asm`, `gasm diff` and
`gasm audit-instructions` now preprocess assembly the way the `gasm audit-instructions` now preprocess assembly the way the
toolchain does: object and parameterised `#define` macros expand at toolchain does: object and parameterised `#define` macros expand at
@@ -74,21 +74,77 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
ranges, index-only VSIB memory operands and bare trailing immediates; ranges, index-only VSIB memory operands and bare trailing immediates;
macro substitution reaches parameters used with element suffixes macro substitution reaches parameters used with element suffixes
(`A.S4`), and `;` separates statements in plain files. (`A.S4`), and `;` separates statements in plain files.
- **`gasm asm -GOOS`.** The go_asm.h generator type-checks per target - **Per-architecture reference pages.** [docs/asm/](docs/asm/README.md)
GOOS, so the darwin-only and windows-only runtime files assemble with gains AMD64, ARM64, RISCV64 and LOONG64: the register files and the
their own defines; the corpus audit derives the GOOS from the file roles the ABI fixes, addressing, operand order with every special form,
name. DATA initialisers accept `$symbol(SB)` values (an absolute constants and materialisation, alignment, fences and the relocations
relocation at the data field, GOOBJ on all four architectures and ELF each target emits. An instruction inventory appendix per architecture
on amd64, arm64, riscv64 and loong64), and U+2215 is accepted inside is generated from the toolchain's own tables by `just gen`, and the
symbol package paths. regenerated tables recognise 147 more mnemonics than the previous
- **The corpus audit measures honestly.** Files named for Go ports gasm release carried (arm64 107, riscv64 31, loong64 9).
does not target (arm, 386, s390x, ...) are no longer attempted for the - **The Plan 9 assembly language reference.** [docs/asm/](docs/asm/README.md)
four supported architectures (no supported build compiles them), and opens the complete language reference with its common core: the lexicon,
the headline rate is reported over attemptable files: 136 of 433 on statement structure and constant expressions, the operand grammar with
the full corpus (31.4 %), 135 of 383 on real code (35.2 %), from the the pseudo-registers and symbol naming, the directives and the function
127 that the previous release measured. The probe battery that flag vocabulary, preprocessing with `#define` and `#include`, and the
decides encodability gained the operand shapes the new families use. Go-embedded layer (ABI0, prototypes, `go_asm.h`, `funcdata.h` and the
- runtime contract). Every claim is verified against `go tool asm` of
Go 1.27.1 and gasm's differential tests; the per-architecture pages and
generated instruction appendices follow.
- **GOOBJ format specification.** [docs/GOOBJ.md](docs/GOOBJ.md)
documents the Go object file format in full: both containers, the 96
byte header and all 19 blocks, every structure with its byte
offsets, symbol kinds and flag bits, all 106 relocation types with
the weak variants, aux symbols, the FuncInfo payload, the pc-value
table encoding, the content hashes and the builtin table, all
verified byte for byte against objects produced by Go 1.27.1's own
tools.
### Changed
- **The corpus audit measures like a build.** Files named for a Go port
gasm does not target (arm, 386, s390x, ...) are never attempted, because
no supported build compiles them; the GOOS comes from the file name; and
each target's go_asm.h is generated on the fly. The headline is reported
over attemptable files: 291 of 353 on the full corpus (82.4 %) assemble
for every target architecture and 295 of 303 on real code (97.4 %),
against 108 of 627 over all files (17.2 %) that the previous release
measured.
### Fixed
- **The operand forms GOROOT writes.** Numeric PC-relative jumps
(`JEQ 2(PC)`, the park loop `JMP 0(PC)`) resolve with the toolchain's
own instruction counting and fold jump-to-jump chains exactly as its
branch optimiser does; symbol immediates (`MOVQ $sym(SB), AX`)
assemble to the toolchain's RIP-relative LEA with an R_PCREL
relocation; negated constant expressions in operands (`ADJSP
$-(REGS - 8)`, the shape the cgo ABI macros write) fold; the immediate
multiply (`IMULQ $1000000000, AX`) encodes with the toolchain's
0x69/0x6B selection; the TLS access pair assembles as the toolchain's
one-instruction form (the bare `MOVQ TLS, r` load nops out and
`off(r)(TLS*1)` folds to the segment-prefixed absolute whose disp32
carries the R_TLSLE relocation, per-GOOS); arm64 accepts the
bare-register indirect branch (`BL R9` beside `BL (R9)`, both BLR) and
the zero-immediate store (`MOVD $0, mem` through the zero register,
rejecting non-zero immediates as the toolchain does); `PCALIGN` now
aligns on amd64, padding with the toolchain's greedy
single-instruction NOPs; the segment-absolute forms (`MOVQ 0x30(GS),
AX` and the store direction) and the absolute crash-store
(`MOVL $0xf1, 0xf1`) encode; and `gasm asm` predefines the
`GOARCH_<arch>` and `GOOS_<goos>` macros the go command passes to
`go tool asm`, so GOROOT headers' `#ifdef GOARCH_amd64` platform
blocks (`go_tls.h`'s `get_tls` and friends) select as intended. The
GOROOT corpus measure moves to 291 of 353 files assembling for every
target architecture (82.4 %), 97.4 % of the real-code corpus, from
70.8 % and 82.2 %.
- **Tool corrections across the pipeline.** The formatter keeps square
brackets in SIMD operands, statement separators and canonical macro
bodies; the linter drops false positives on shift counts, SETcc
spellings and ABIInternal references; the lexer treats a trailing
carriage return as a line end so comment text stays idempotent; and
arm64 rejects bare BTI with a diagnostic while accepting the full
family.
## [0.34.0] - 2026-09-20 ## [0.34.0] - 2026-09-20
+9 -6
View File
@@ -81,7 +81,10 @@ to give that syntax the tooling it deserves.
GOOBJ format, which needs the installed toolchain and which `go build` GOOBJ format, which needs the installed toolchain and which `go build`
consumes in place of the toolchain's output. Framed functions get the consumes in place of the toolchain's output. Framed functions get the
stack-split guard and the morestack block, byte-identical to the stack-split guard and the morestack block, byte-identical to the
toolchain's, so split functions link too. toolchain's, so split functions link too. The assembler preprocesses
like the toolchain (`#define`, `#include` with `-I`, `#ifdef`), generates
`go_asm.h` from the package's Go files, and carries `PCALIGN`, the
`LOCK`/`REP` prefixes and the literal-data pseudo-ops.
- **Disassembler.** `gasm dis` lists a `.s` file's functions at their real - **Disassembler.** `gasm dis` lists a `.s` file's functions at their real
offsets after assembling, or disassembles raw bytes from a file or stdin. offsets after assembling, or disassembles raw bytes from a file or stdin.
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into - **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
@@ -110,9 +113,9 @@ Four architectures, the four that matter in practice:
| Architecture | GOARCH | File suffix | Instructions recognised | | Architecture | GOARCH | File suffix | Instructions recognised |
|--------------|-------------|--------------|---------------------------------------------| |--------------|-------------|--------------|---------------------------------------------|
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases | | AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
| ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes | | ARM64 | `arm64` | `_arm64.s` | 645 + common opcodes |
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes | | RISC-V | `riscv64` | `_riscv64.s` | 992 + common opcodes |
| LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes | | LoongArch | `loong64` | `_loong64.s` | 808 + common opcodes |
"Common opcodes" are the instructions shared by every architecture (`RET`, "Common opcodes" are the instructions shared by every architecture (`RET`,
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally `JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
@@ -124,8 +127,8 @@ can emit today is narrower, and a recognised but unencodable instruction is
reported as an explicit error, never as a wrong byte. reported as an explicit error, never as a wrong byte.
The same measurement runs over GOROOT's whole assembly corpus: The same measurement runs over GOROOT's whole assembly corpus:
`gasm audit-instructions --corpus` reports 136 of 433 attemptable files `gasm audit-instructions --corpus` reports 291 of 353 attemptable files
(31.4 %) assembling for every target architecture today (files named for (82.4 %) assembling for every target architecture today (files named for
other Go ports are counted but never attempted), with the top failure other Go ports are counted but never attempted), with the top failure
reasons per architecture; the number moves with every release. reasons per architecture; the number moves with every release.
+1 -1
View File
@@ -7,7 +7,7 @@ releases do not receive them.
| Version | Supported | | Version | Supported |
|---|---| |---|---|
| 0.34.0 | yes | | 0.35.0 | yes |
| older releases | no | | older releases | no |
## Reporting a vulnerability ## Reporting a vulnerability
+23
View File
@@ -637,6 +637,20 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
return a64wordLE(a64UncondBranch(opc, uint32(rn), 0)), nil return a64wordLE(a64UncondBranch(opc, uint32(rn), 0)), nil
} }
// The bare spelling BL R9 is the same indirect branch: the parser reads
// a bare identifier as a symbol, and one named for a register is an
// indirect branch through it, which the toolchain accepts alongside the
// parenthesised form (BL (R3) and BL R3 both encode BLR R3).
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" && op.Addr.Index == "" {
if rn := arm64RegNum(op.Addr.Sym.Name); rn >= 0 {
opc := uint32(0) // BR
if link {
opc = 1 // BLR
}
return a64wordLE(a64UncondBranch(opc, uint32(rn), 0)), nil
}
}
// Symbol reference: BL sym(SB), or B sym(SB) for a tail call, against a // Symbol reference: BL sym(SB), or B sym(SB) for a tail call, against a
// relocation (R_CALLARM64 either way). // relocation (R_CALLARM64 either way).
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" { if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
@@ -1454,6 +1468,15 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo,
} }
return encodeARM64SBAddr(src.Imm.Sym, rd, relocs), nil return encodeARM64SBAddr(src.Imm.Sym, rd, relocs), nil
} }
// Immediate → memory: only storing zero is encodable (the ZR
// register); the toolchain rejects any other immediate-to-memory
// combination ("illegal combination").
if isMemOperand(dst) {
if arm64Imm64(src) != 0 {
return nil, fmt.Errorf("%s: illegal combination: an immediate store must be zero", mnem)
}
return encodeARM64MemOp(mnem, dst, 31, false, fi, "")
}
rd := arm64RegNum(operandRegName(dst)) rd := arm64RegNum(operandRegName(dst))
if rd < 0 { if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem) return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
+343 -10
View File
@@ -38,10 +38,25 @@ func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
// rejects SB operands outright (single-function assembly cannot resolve // rejects SB operands outright (single-function assembly cannot resolve
// them). When allowExternal is set, a reference to a symbol no GLOBL in the // them). When allowExternal is set, a reference to a symbol no GLOBL in the
// file defines is recorded as an external relocation instead of failing // file defines is recorded as an external relocation instead of failing
// the object-file emitters resolve it at link time. // the object-file emitters resolve it at link time. goos selects the TLS
// access form: the empty default behaves as linux.
type linkInfo struct { type linkInfo struct {
symbols map[string]bool symbols map[string]bool
allowExternal bool allowExternal bool
goos string
}
// tlsOneInsn reports the one-instruction TLS form, obj6.go's
// CanUse1InsnTLS for the GOOS gasm supports: the bare TLS load nops out and
// the (TLS*1) index folds to a segment-absolute access. Windows and plan9
// keep the two-instruction form; shared linux does too, which gasm's raw
// path does not model and therefore does not select.
func (l *linkInfo) tlsOneInsn() bool {
switch l.goos {
case "", "linux", "freebsd":
return true
}
return false
} }
// sbPatch is a function-relative static-symbol relocation: the disp32 field // sbPatch is a function-relative static-symbol relocation: the disp32 field
@@ -86,6 +101,10 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
// outgrows the short form. // outgrows the short form.
long := make([]bool, len(t.Body)) long := make([]bool, len(t.Body))
sizes := make([]int, len(t.Body)) sizes := make([]int, len(t.Body))
numTargets := make([]int, len(t.Body))
for i := range numTargets {
numTargets[i] = -1
}
offsets := map[string]int{} offsets := map[string]int{}
pcs := make([]int, len(t.Body)) pcs := make([]int, len(t.Body))
var guardJBlong, guardJBElong, moreJMPlong bool var guardJBlong, guardJBElong, moreJMPlong bool
@@ -94,23 +113,106 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
for { for {
guard := fi.guardLen(guardJBlong, guardJBElong) guard := fi.guardLen(guardJBlong, guardJBElong)
pos := guard + len(fi.prologue) pos := guard + len(fi.prologue)
for i := range numTargets {
numTargets[i] = -1
}
idxAtPc := map[int]int{}
for i, stmt := range t.Body { for i, stmt := range t.Body {
switch s := stmt.(type) { switch s := stmt.(type) {
case *ast.Label: case *ast.Label:
offsets[s.Name.Text] = pos offsets[s.Name.Text] = pos
case *ast.Instr: case *ast.Instr:
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
// The alignment pseudo-statement: its size is the
// padding to the next boundary at this very position,
// filled with NOPs at emission.
pad, err := pcAlignPad(pcAlignValue(s), pos)
if err != nil {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("PCALIGN: %w", err)
}
sizes[i] = pad
pcs[i] = pos
pos += pad
continue
}
sz, err := instrSize(s, fi, long[i], link) sz, err := instrSize(s, fi, long[i], link)
if err != nil { if err != nil {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err) return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
} }
sizes[i] = sz sizes[i] = sz
pcs[i] = pos pcs[i] = pos
idxAtPc[pos] = i
pos += sz pos += sz
} }
} }
bodyLen := pos - (guard + len(fi.prologue)) bodyLen := pos - (guard + len(fi.prologue))
// Expand any short jump whose displacement no longer fits rel8. // Expand any short jump whose displacement no longer fits rel8.
changed := false changed := false
// Numeric ±N(PC) jumps resolve against this iteration's layout; the
// emission pass reads the same table after the loop converges. A
// target that is itself an unconditional local JMP is chased to the
// ultimate target: the toolchain's brloop pass collapses branch-to-
// branch chains before it encodes, so matching its bytes requires
// the same redirection.
for i := range numTargets {
numTargets[i] = -1
}
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
if !ok {
continue
}
if len(s.Operands) == 1 {
if n, isNum := pcJumpOffset(s.Operands[0]); isNum {
if target, okT := pcJumpTarget(t, i, n, pcs); okT {
numTargets[i] = target
}
}
}
}
for i := range numTargets {
if numTargets[i] < 0 {
continue
}
tgt := numTargets[i]
for hop := 0; hop < len(t.Body); hop++ {
idx, ok := idxAtPc[tgt]
if !ok {
break
}
in, ok := t.Body[idx].(*ast.Instr)
if !ok || strings.ToUpper(in.Mnemonic.Text) != "JMP" || len(in.Operands) != 1 {
break
}
if name, isLabel := labelName(in.Operands[0]); isLabel {
tgt = offsets[resolve(name)]
continue
}
if n, isNum := pcJumpOffset(in.Operands[0]); isNum {
next, okT := pcJumpTarget(t, idx, n, pcs)
if !okT {
break
}
tgt = next
continue
}
break // JMP through a register or memory: the chain ends
}
numTargets[i] = tgt
}
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
if !ok {
continue
}
if numTargets[i] >= 0 && !long[i] {
rel := int64(numTargets[i] - (pcs[i] + jumpSize(strings.ToUpper(s.Mnemonic.Text), false)))
if !fits8(rel) {
long[i] = true
changed = true
}
}
}
for i, stmt := range t.Body { for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr) s, ok := stmt.(*ast.Instr)
if !ok { if !ok {
@@ -232,7 +334,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
spadjStep{pos + epi, 0}, spadjStep{pos + epi, 0},
) )
} }
code, ps, pool, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link) code, ps, pool, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link, numTargets[i])
if err != nil { if err != nil {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err) return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
} }
@@ -428,6 +530,83 @@ func computeFrame(t *ast.Text) frameInfo {
return fi return fi
} }
// pcJumpOffset recognises the numeric relative jump operand ±N(PC) and
// returns N: the toolchain counts instructions, not bytes, so +2(PC) targets
// the second instruction boundary after the branch.
func pcJumpOffset(op *ast.Operand) (int, bool) {
if op.Kind != ast.OpAddr || op.Addr.Base != "PC" {
return 0, false
}
return int(op.Addr.Offset), true
}
// pcJumpTarget resolves a numeric jump at statement index j: N counts the
// instruction statements after the jump itself (N = 0 is the jump's own
// address, the classic park loop), and the target is the start of the Nth
// one. It reports false when the count runs past the end of the function.
func pcJumpTarget(t *ast.Text, j, n int, pcs []int) (int, bool) {
if n == 0 {
return pcs[j], true
}
seen := 0
for k := j + 1; k < len(t.Body); k++ {
if _, ok := t.Body[k].(*ast.Instr); !ok {
continue
}
seen++
if seen == n {
return pcs[k], true
}
}
return 0, false
}
// x86 NOP encodings, single-instruction no-ops of lengths 1 to 9 (the
// toolchain's asm6.go nop table); longer padding repeats the largest that
// fits, greedy from the end.
var x86Nops = [][]byte{
{0x90},
{0x66, 0x90},
{0x0F, 0x1F, 0x00},
{0x0F, 0x1F, 0x40, 0x00},
{0x0F, 0x1F, 0x44, 0x00, 0x00},
{0x66, 0x0F, 0x1F, 0x44, 0x00, 0x00},
{0x0F, 0x1F, 0x80, 0x00, 0x00, 0x00, 0x00},
{0x0F, 0x1F, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00},
{0x66, 0x0F, 0x1F, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00},
}
// fillNOPs fills p with the greedy largest single-instruction NOPs, exactly
// the toolchain's fillnop.
func fillNOPs(p []byte) {
for len(p) > 0 {
m := min(len(p), len(x86Nops))
copy(p[:m], x86Nops[m-1])
p = p[m:]
}
}
// pcAlignPad computes the padding PCALIGN $align inserts at pos: the
// alignment must be a power of two in [8, 2048] and the padding runs to the
// next boundary (zero when the position is already aligned).
func pcAlignPad(align, pos int) (int, error) {
if align <= 0 || align&(align-1) != 0 || align < 8 || align > 2048 {
return 0, fmt.Errorf("alignment value of an instruction must be a power of two and in the range [8, 2048], got %d", align)
}
if lob := pos & (align - 1); lob != 0 {
return align - lob, nil
}
return 0, nil
}
// pcAlignValue reads a PCALIGN statement's alignment operand.
func pcAlignValue(s *ast.Instr) int {
if len(s.Operands) == 1 && s.Operands[0].Kind == ast.OpImmediate && s.Operands[0].Imm.HasVal {
return int(s.Operands[0].Imm.Val)
}
return 0 // rejected by pcAlignPad's range check
}
// hasCall reports whether the function body contains a CALL instruction. // hasCall reports whether the function body contains a CALL instruction.
func hasCall(t *ast.Text) bool { func hasCall(t *ast.Text) bool {
for _, stmt := range t.Body { for _, stmt := range t.Body {
@@ -602,7 +781,7 @@ func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, erro
} }
return jumpSize(mnem, long), nil return jumpSize(mnem, long), nil
} }
code, _, _, err := encodeInstr(s, 0, nil, fi, false, nil, link) code, _, _, err := encodeInstr(s, 0, nil, fi, false, nil, link, -1)
if err != nil { if err != nil {
return 0, err return 0, err
} }
@@ -637,9 +816,21 @@ func jumpSize(mnem string, long bool) int {
// (relative to pc, the instruction's own offset). A RET in a frame-pointer // (relative to pc, the instruction's own offset). A RET in a frame-pointer
// function is prefixed with the epilogue. resolve, when non-nil, redirects a // function is prefixed with the epilogue. resolve, when non-nil, redirects a
// jump label through the jump-to-jump chain before the offset lookup. // jump label through the jump-to-jump chain before the offset lookup.
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) { func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo, numTarget int) ([]byte, []sbPatch, []floatPoolEntry, error) {
mnem := strings.ToUpper(s.Mnemonic.Text) mnem := strings.ToUpper(s.Mnemonic.Text)
if mnem == "PCALIGN" {
// The layout pass already accounted the padding; emit the same
// amount of NOP bytes for the statement's own position.
pad, err := pcAlignPad(pcAlignValue(s), pc)
if err != nil {
return nil, nil, nil, err
}
out := make([]byte, pad)
fillNOPs(out)
return out, nil, nil, nil
}
var prefix []byte var prefix []byte
if mnem == "RET" && fi.useFP { if mnem == "RET" && fi.useFP {
prefix = fi.epilogue prefix = fi.epilogue
@@ -677,7 +868,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
} }
return append(prefix, code...), nil, nil, nil return append(prefix, code...), nil, nil, nil
} }
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve) code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve, numTarget)
} else { } else {
code, ps, pool, err = encodeNormal(s, fi, link) code, ps, pool, err = encodeNormal(s, fi, link)
} }
@@ -703,6 +894,41 @@ func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch
} }
return code, nil, nil, nil return code, nil, nil, nil
} }
// MOVQ $sym±off(SB), r64: the toolchain assembles a symbol immediate as
// LEAQ disp32(RIP), r64 with an R_PCREL relocation at the disp32 field,
// never as a 64-bit absolute immediate (verified against go tool asm).
// MOVD is the MOVQ alias; the narrower widths reject the form outright.
if (mnemUpper == "MOVQ" || mnemUpper == "MOVD") && len(s.Operands) == 2 &&
s.Operands[0].Kind == ast.OpImmediate && s.Operands[0].Imm.Sym != nil &&
s.Operands[0].Imm.Sym.Pseudo == "SB" {
mem := &ast.Operand{Kind: ast.OpAddr, Addr: ast.Address{Sym: s.Operands[0].Imm.Sym}}
src, err := operandFromAST(mnemUpper, mem, 8, fi, link)
if err != nil {
return nil, nil, nil, err
}
dst, err := operandFromAST(mnemUpper, s.Operands[1], 8, fi, link)
if err != nil {
return nil, nil, nil, err
}
e := &enc{}
if err := e.encodeLea([]Operand{src, dst}, 8); err != nil {
return nil, nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
}
return e.out, ps, nil, nil
}
// MOVQ/MOVL TLS, r: the bare TLS load. The toolchain's progedit nops
// it out on the one-instruction TLS systems (linux and freebsd, not
// shared) and encodes the segment-prefixed load elsewhere; get_tls(r),
// the macro GOROOT's go_tls.h defines, expands to exactly this
// statement, and the toolchain's pairing pass removes it whenever the
// following instruction's (TLS*1) index folds.
if (mnemUpper == "MOVQ" || mnemUpper == "MOVL") && len(s.Operands) == 2 && isBareTLS(s.Operands[0]) {
return encodeTLSBaseLoad(s, fi, link)
}
_, size := splitSize(mnemUpper) _, size := splitSize(mnemUpper)
if size == 0 { if size == 0 {
size = 8 size = 8
@@ -722,10 +948,67 @@ func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch
ps := make([]sbPatch, len(e.patches)) ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches { for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend} ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
if p.tls {
ps[i].kind = RelTLSLE
}
} }
return e.out, ps, e.floatPoolList(), nil return e.out, ps, e.floatPoolList(), nil
} }
// isBareTLS reports whether the operand is the bare TLS pseudo-register
// load source, the expansion of go_tls.h's get_tls(r) macro.
func isBareTLS(op *ast.Operand) bool {
return op.Kind == ast.OpAddr && op.Addr.Sym != nil &&
op.Addr.Sym.Pseudo == "" && op.Addr.Sym.Name == "TLS" &&
op.Addr.Base == "" && op.Addr.Index == ""
}
// encodeTLSBaseLoad assembles MOVQ/MOVL TLS, r. On the one-instruction TLS
// systems (linux and freebsd outside -shared, obj6.go's CanUse1InsnTLS) the
// statement nops out: the following (TLS*1) access folds to a direct
// segment-absolute load. The two-instruction systems keep the segment load,
// nine bytes with the R_TLSLE patch site at the disp32.
func encodeTLSBaseLoad(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
if size == 0 {
size = 8
}
dst, err := operandFromAST("MOVQ", s.Operands[1], 8, fi, link)
if err != nil {
return nil, nil, nil, err
}
reg, ok := dst.(Reg)
if !ok || reg.isVec() {
return nil, nil, nil, fmt.Errorf("TLS: destination must be a general register")
}
if link == nil || link.tlsOneInsn() {
return nil, nil, nil, nil // noped out
}
seg := byte(0x64) // FS
if link.goos == "windows" {
seg = 0x65 // GS
}
e := &enc{}
i := &instr{
prefix: seg,
rexW: size == 8,
rexR: reg.idx >= 8,
opcode: []byte{0x8B},
modrm: 0x04 | (reg.idx&7)<<3,
sib: 0x25,
disp: le32(0),
tls: true,
}
if err := e.emit(i); err != nil {
return nil, nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend, kind: RelTLSLE}
}
return e.out, ps, nil, nil
}
// encodeBookkeeping accepts-and-ignores FUNCDATA and PCDATA at the statement // encodeBookkeeping accepts-and-ignores FUNCDATA and PCDATA at the statement
// level, before operand conversion: the toolchain's shapes are FUNCDATA // level, before operand conversion: the toolchain's shapes are FUNCDATA
// $n, sym(SB) and PCDATA $n, $m, and neither contributes a byte to the // $n, sym(SB) and PCDATA $n, $m, and neither contributes a byte to the
@@ -753,22 +1036,30 @@ func encodeBookkeeping(upper string, s *ast.Instr) ([]byte, error) {
} }
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the // encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
// target label, in the short (rel8) or long (rel32) form. // target label or from a numeric ±N(PC) instruction count, in the short
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long bool, resolve func(string) string) ([]byte, error) { // (rel8) or long (rel32) form. numTarget is the resolved byte offset of a
// numeric operand, negative when the operand is not one.
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long bool, resolve func(string) string, numTarget int) ([]byte, error) {
if len(s.Operands) != 1 { if len(s.Operands) != 1 {
return nil, fmt.Errorf("jump expects 1 operand, got %d", len(s.Operands)) return nil, fmt.Errorf("jump expects 1 operand, got %d", len(s.Operands))
} }
name, ok := labelName(s.Operands[0]) name, isLabel := labelName(s.Operands[0])
if !ok { if !isLabel && numTarget < 0 {
return nil, fmt.Errorf("jump target must be a local label") return nil, fmt.Errorf("jump target must be a local label")
} }
var target int
if isLabel {
if resolve != nil && mnem != "CALL" { if resolve != nil && mnem != "CALL" {
name = resolve(name) name = resolve(name)
} }
target, ok := offsets[name] t, ok := offsets[name]
if !ok { if !ok {
return nil, fmt.Errorf("undefined label %q", name) return nil, fmt.Errorf("undefined label %q", name)
} }
target = t
} else {
target = numTarget
}
rel := int64(target - (pc + jumpSize(mnem, long))) rel := int64(target - (pc + jumpSize(mnem, long)))
if !long { if !long {
@@ -841,6 +1132,11 @@ func indirectJumpTarget(s *ast.Instr) bool {
return false return false
} }
a := s.Operands[0].Addr a := s.Operands[0].Addr
// ±N(PC) is the numeric relative form, the PC counts instructions from
// the branch: relative, not indirect.
if a.Base == "PC" || a.Index == "PC" {
return false
}
if a.Base != "" || a.Index != "" { if a.Base != "" || a.Index != "" {
return true return true
} }
@@ -958,12 +1254,44 @@ func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, l
// Memory with a real base register: (base), off(base), (base)(index*scale). // Memory with a real base register: (base), off(base), (base)(index*scale).
if a.Base != "" { if a.Base != "" {
// Segment-absolute: 0x30(GS) and 0x28(FS), the windows TLS
// spellings. The segment override prefixes a disp32 absolute
// reference with no relocation.
if a.Base == "GS" || a.Base == "FS" {
seg := byte(0x64)
if a.Base == "GS" {
seg = 0x65
}
return SegAbs{Disp: a.Offset, Size: size, Seg: seg}, nil
}
base, ok := ParseReg(a.Base) base, ok := ParseReg(a.Base)
if !ok { if !ok {
return nil, fmt.Errorf("unknown base register %q", a.Base) return nil, fmt.Errorf("unknown base register %q", a.Base)
} }
m := Mem{Base: base, Disp: a.Offset, HasBase: true, Size: size} m := Mem{Base: base, Disp: a.Offset, HasBase: true, Size: size}
if a.Index != "" { if a.Index != "" {
if a.Index == "TLS" {
// off(base)(TLS*1): the thread-local annotation. The
// one-instruction TLS form folds it to off(TLS), the
// segment-prefixed absolute whose disp32 carries an
// R_TLS_LE patch site; the base register disappears
// from the encoding, exactly as the toolchain's
// progedit rewrites the address.
seg := byte(0x64) // FS on linux, freebsd, plan9
if link != nil && link.goos == "windows" {
seg = 0x65 // GS
}
return TLSMem{Disp: a.Offset, Size: size, Seg: seg}, nil
}
if a.Index == "GS" || a.Index == "FS" {
// 0(CX)(GS): the segment annotation rides the base
// access as the override prefix.
m.Seg = 0x64
if a.Index == "GS" {
m.Seg = 0x65
}
return m, nil
}
idx, ok := ParseReg(a.Index) idx, ok := ParseReg(a.Index)
if !ok { if !ok {
return nil, fmt.Errorf("unknown index register %q", a.Index) return nil, fmt.Errorf("unknown index register %q", a.Index)
@@ -984,6 +1312,11 @@ func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, l
} }
return Mem{Index: idx, Scale: a.Scale, Disp: a.Offset, HasIndex: true, Size: size}, nil return Mem{Index: idx, Scale: a.Scale, Disp: a.Offset, HasIndex: true, Size: size}, nil
} }
// A bare displacement with no base: the absolute address form,
// MOVL $0xf1, 0xf1. No segment and no relocation.
if a.Sym == nil && a.Base == "" && a.Index == "" && a.HasOff {
return SegAbs{Disp: a.Offset, Size: size}, nil
}
// Bare register. // Bare register.
if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" { if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" {
if r, ok := ParseReg(a.Sym.Name); ok { if r, ok := ParseReg(a.Sym.Name); ok {
+33
View File
@@ -64,6 +64,7 @@ type encPatch struct {
off int off int
name string name string
addend int64 addend int64
tls bool // a TLS slot offset: the patch is R_TLSLE with no symbol
} }
func (e *enc) encode(mnem string, ops []Operand) error { func (e *enc) encode(mnem string, ops []Operand) error {
@@ -578,6 +579,7 @@ type instr struct {
disp []byte disp []byte
imm []byte imm []byte
sb *sbRef // static-symbol displacement in disp, awaiting resolution sb *sbRef // static-symbol displacement in disp, awaiting resolution
tls bool // the displacement is a TLS slot offset, patched R_TLSLE
} }
// sbRef records that an instruction's displacement refers to a static symbol // sbRef records that an instruction's displacement refers to a static symbol
@@ -620,6 +622,9 @@ func (e *enc) emit(i *instr) error {
if i.sb != nil { if i.sb != nil {
e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend}) e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend})
} }
if i.tls {
e.patches = append(e.patches, encPatch{off: len(e.out), tls: true})
}
e.out = append(e.out, i.disp...) e.out = append(e.out, i.disp...)
e.out = append(e.out, i.imm...) e.out = append(e.out, i.imm...)
return nil return nil
@@ -674,12 +679,30 @@ func setRMReg(i *instr, regField int, rexR, regForced bool, rm Operand, opSize i
i.disp = le32(0) i.disp = le32(0)
i.sb = &sbRef{name: r.name, addend: r.addend} i.sb = &sbRef{name: r.name, addend: r.addend}
return nil return nil
case TLSMem:
// off(TLS): the segment-prefixed absolute access, mod=00 with the
// SIB escape's disp32 absolute form. The displacement is the TLS
// slot offset, patched by the linker's TLS relocation.
i.prefix = r.Seg
i.modrm = 0x04 | regField<<3
i.sib = 0x25
i.disp = le32(r.Disp)
i.tls = true
return nil
case SegAbs:
// 0x30(GS): the segment override with the SIB escape's disp32
// absolute form, no relocation.
setSegAbs(i, regField, r)
return nil
default: default:
return fmt.Errorf("invalid r/m operand %T", rm) return fmt.Errorf("invalid r/m operand %T", rm)
} }
} }
func setMem(i *instr, regField int, m Mem) error { func setMem(i *instr, regField int, m Mem) error {
if m.Seg != 0 {
i.prefix = m.Seg
}
modrm, sib, disp, xBit, bBit, err := memComponents(regField, m) modrm, sib, disp, xBit, bBit, err := memComponents(regField, m)
if err != nil { if err != nil {
return err return err
@@ -692,6 +715,16 @@ func setMem(i *instr, regField int, m Mem) error {
return nil return nil
} }
// setSegAbs assembles a segment-absolute operand, 0x30(GS): the segment
// override with the mod=00 SIB escape's disp32 absolute form and no
// relocation.
func setSegAbs(i *instr, regField int, m SegAbs) {
i.prefix = m.Seg
i.modrm = 0x04 | regField<<3
i.sib = 0x25
i.disp = le32(m.Disp)
}
// memComponents computes the ModR/M byte (with the given reg field), the SIB // memComponents computes the ModR/M byte (with the given reg field), the SIB
// byte (-1 if none), the displacement bytes, and the high index/base bits, for // byte (-1 if none), the displacement bytes, and the high index/base bits, for
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths. // a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
+4 -4
View File
@@ -307,12 +307,12 @@ func TestStackGuardBytesLOONG64(t *testing.T) {
func TestStackGuardGOObjInternalCall(t *testing.T) { func TestStackGuardGOObjInternalCall(t *testing.T) {
for _, tt := range []struct { for _, tt := range []struct {
src string src string
assemble func(*ast.File) (*Image, error) assemble func(*ast.File, ...AssembleOption) (*Image, error)
}{ }{
{"g_amd64.s", AssembleFile}, {"g_amd64.s", AssembleFile},
{"g_arm64.s", AssembleFileARM64}, {"g_arm64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileARM64(f) }},
{"g_riscv64.s", AssembleFileRISCV}, {"g_riscv64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileRISCV(f) }},
{"g_loong64.s", AssembleFileLOONG64}, {"g_loong64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileLOONG64(f) }},
} { } {
f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n") f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 { if len(errs) > 0 {
+60 -7
View File
@@ -192,6 +192,30 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
} }
return e.emit(i) return e.emit(i)
case TLSMem:
if !dstIsReg {
return fmt.Errorf("MOV: two memory operands")
}
// MOV r, off(TLS): the segment-prefixed absolute load, reg=dst,
// rm=src(tlsMem) through the SIB escape; the disp32 is the TLS slot
// offset with its R_TLSLE patch site.
i := newInstr(size, []byte{movRR(size)})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
case SegAbs:
if !dstIsReg {
return fmt.Errorf("MOV: two memory operands")
}
// MOV r, 0x30(GS): the segment-absolute load.
i := newInstr(size, []byte{movRR(size)})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
case Imm: case Imm:
if dstIsReg { if dstIsReg {
v := int64(src) v := int64(src)
@@ -232,11 +256,24 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
i.imm = imm i.imm = imm
return e.emit(i) return e.emit(i)
} }
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0. // MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0. An immediate in the
// destination slot is the absolute-address crash-store spelling,
// MOVL $0xf1, 0xf1: the parser reads the trailing bare constant
// as an immediate, and the store's disp32 carries the address.
op := byte(0xC7) op := byte(0xC7)
if size == 1 { if size == 1 {
op = 0xC6 op = 0xC6
} }
if d, ok := dst.(Imm); ok {
i := newInstr(size, []byte{op})
setSegAbs(i, 0, SegAbs{Disp: int64(d)})
immBytes, err := immediate(int64(src), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
i := newInstr(size, []byte{op}) i := newInstr(size, []byte{op})
if err := setRMDigit(i, 0, dst, size); err != nil { if err := setRMDigit(i, 0, dst, size); err != nil {
return err return err
@@ -627,7 +664,16 @@ func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error {
func (e *enc) encodeImul(ops []Operand, size int) error { func (e *enc) encodeImul(ops []Operand, size int) error {
switch len(ops) { switch len(ops) {
case 2: case 2:
// IMUL r, r/m: 0x0F 0xAF. // Two shapes. The leading-immediate spelling IMUL $imm, r multiplies
// r in place (dst = rm = r): the shape GOROOT's clock code writes.
// Otherwise IMUL r, r/m: 0x0F 0xAF.
if imm, ok := ops[0].(Imm); ok {
dstReg, isReg := ops[1].(Reg)
if !isReg {
return fmt.Errorf("IMUL: destination must be a register")
}
return e.encodeImulImm(imm, dstReg, dstReg, size)
}
dstReg, ok := ops[1].(Reg) dstReg, ok := ops[1].(Reg)
if !ok { if !ok {
return fmt.Errorf("IMUL: destination must be a register") return fmt.Errorf("IMUL: destination must be a register")
@@ -647,17 +693,26 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
if !ok { if !ok {
return fmt.Errorf("IMUL: immediate operand expected first") return fmt.Errorf("IMUL: immediate operand expected first")
} }
// Plan 9 order: IMUL $imm, src, dst. // Plan 9 order: IMUL $imm, src, dst; the source stays a general
// r/m operand (setRM takes registers and memory alike).
return e.encodeImulImm(imm, ops[1], dstReg, size)
}
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
}
// encodeImulImm emits the immediate multiply: 0x6B with a sign-extended imm8
// when the value fits, 0x69 with a 32-bit immediate otherwise.
func (e *enc) encodeImulImm(imm Imm, rm Operand, dst Reg, size int) error {
if fits8(int64(imm)) { if fits8(int64(imm)) {
i := newInstr(size, []byte{0x6B}) i := newInstr(size, []byte{0x6B})
if err := setRM(i, dstReg, ops[1], size); err != nil { if err := setRM(i, dst, rm, size); err != nil {
return err return err
} }
i.imm = []byte{byte(int8(imm))} i.imm = []byte{byte(int8(imm))}
return e.emit(i) return e.emit(i)
} }
i := newInstr(size, []byte{0x69}) i := newInstr(size, []byte{0x69})
if err := setRM(i, dstReg, ops[1], size); err != nil { if err := setRM(i, dst, rm, size); err != nil {
return err return err
} }
immBytes, err := immediate(int64(imm), size, false) immBytes, err := immediate(int64(imm), size, false)
@@ -667,8 +722,6 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
i.imm = immBytes i.imm = immBytes
return e.emit(i) return e.emit(i)
} }
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
}
// --- PUSH / POP ------------------------------------------------------------- // --- PUSH / POP -------------------------------------------------------------
+1
View File
@@ -130,6 +130,7 @@ func TestDifferentialKernels(t *testing.T) {
{filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false}, {filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false}, {filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false}, {filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "forms_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true}, {filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true},
{filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true}, {filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true},
} { } {
+23 -1
View File
@@ -150,6 +150,18 @@ func (img *Image) Bytes() []byte {
return append(out, img.Data...) return append(out, img.Data...)
} }
// AssembleOption adjusts the file-level assembly context.
type AssembleOption func(*linkInfo)
// WithGOOS selects the target operating system for the forms that depend on
// it, the TLS access shape above all: linux and freebsd take the
// one-instruction form, windows and plan9 keep the two-instruction load.
func WithGOOS(goos string) AssembleOption {
return func(l *linkInfo) {
l.goos = goos
}
}
// AssembleFile assembles every TEXT function of a parsed file and lays out // AssembleFile assembles every TEXT function of a parsed file and lays out
// its static symbols (GLOBL/DATA) in a data section behind the code. Each // its static symbols (GLOBL/DATA) in a data section behind the code. Each
// reference to a file-local static symbol becomes a RIP-relative load whose // reference to a file-local static symbol becomes a RIP-relative load whose
@@ -157,7 +169,7 @@ func (img *Image) Bytes() []byte {
// GLOBL defines is recorded as an external relocation (Externals) with its // GLOBL defines is recorded as an external relocation (Externals) with its
// displacement left zero, the object-file emitters resolve it at link // displacement left zero, the object-file emitters resolve it at link
// time, while the raw image (Bytes) cannot represent it. // time, while the raw image (Bytes) cannot represent it.
func AssembleFile(f *ast.File) (*Image, error) { func AssembleFile(f *ast.File, opts ...AssembleOption) (*Image, error) {
dataSyms, err := collectData(f) dataSyms, err := collectData(f)
if err != nil { if err != nil {
return nil, err return nil, err
@@ -166,7 +178,17 @@ func AssembleFile(f *ast.File) (*Image, error) {
for _, d := range dataSyms { for _, d := range dataSyms {
known[d.name] = true known[d.name] = true
} }
// TEXT symbols are file-level definitions too: a symbol immediate
// ($fn(SB)) may name one, exactly as a data reference names a GLOBL.
for _, d := range f.Decls {
if t, ok := d.(*ast.Text); ok {
known[t.Name.Name] = true
}
}
link := &linkInfo{symbols: known, allowExternal: true} link := &linkInfo{symbols: known, allowExternal: true}
for _, o := range opts {
o(link)
}
poolSeen := map[string]bool{} poolSeen := map[string]bool{}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path} img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
+24
View File
@@ -36,6 +36,29 @@ type FloatImm struct {
func (FloatImm) isOperand() {} func (FloatImm) isOperand() {}
// TLSMem is a thread-local access, the source form off(base)(TLS*1) with the
// base dropped: the toolchain's one-instruction TLS rewrite assembles it as
// the segment-prefixed absolute whose disp32 carries an R_TLS_LE patch site
// (the linker fills the TLS slot offset).
type TLSMem struct {
Disp int64
Size int
Seg byte // the segment override: FS (0x64) or GS (0x65) on windows
}
func (TLSMem) isOperand() {}
// SegAbs is a segment-absolute access, 0x30(GS): the segment override
// prefixes a disp32 absolute reference with no relocation. The base
// register spellings GS and FS produce it.
type SegAbs struct {
Disp int64
Size int
Seg byte // 0x64 FS, 0x65 GS
}
func (SegAbs) isOperand() {}
// Mem is a memory operand of the form disp(base)(index*scale). // Mem is a memory operand of the form disp(base)(index*scale).
type Mem struct { type Mem struct {
Base Reg Base Reg
@@ -45,6 +68,7 @@ type Mem struct {
Size int // operand width in bytes Size int // operand width in bytes
HasBase bool HasBase bool
HasIndex bool HasIndex bool
Seg byte // segment override prefix (0x64 FS, 0x65 GS); 0 = none
} }
func (Mem) isOperand() {} func (Mem) isOperand() {}
+11 -3
View File
@@ -5,6 +5,7 @@ package main
import ( import (
"os" "os"
"os/exec"
"path/filepath" "path/filepath"
"strings" "strings"
"testing" "testing"
@@ -393,13 +394,20 @@ func TestRunCorpusAuditGOOS(t *testing.T) {
} }
// TestGenerateGoAsmHeaderRuntime pins the generator against the real thing: // TestGenerateGoAsmHeaderRuntime pins the generator against the real thing:
// the runtime package, whose header the toolchain's own -asmhdr output was // the runtime package of the ambient toolchain, whose header the toolchain's
// sampled from. Skipped in short mode: it type-checks the whole package. // own -asmhdr output was sampled from. Skipped in short mode: it type-checks
// the whole package. The GOROOT comes from the go command itself, so the
// test follows whatever toolchain the host provides.
func TestGenerateGoAsmHeaderRuntime(t *testing.T) { func TestGenerateGoAsmHeaderRuntime(t *testing.T) {
if testing.Short() { if testing.Short() {
t.Skip("type-checks the whole runtime package") t.Skip("type-checks the whole runtime package")
} }
dir, err := generateGoAsmHeader("/usr/local/go/src/runtime", "", "amd64", t.TempDir()) out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil {
t.Skipf("no Go toolchain: %v", err)
}
runtimeDir := filepath.Join(strings.TrimSpace(string(out)), "src", "runtime")
dir, err := generateGoAsmHeader(runtimeDir, "", "amd64", t.TempDir())
if err != nil { if err != nil {
t.Fatalf("generateGoAsmHeader(runtime): %v", err) t.Fatalf("generateGoAsmHeader(runtime): %v", err)
} }
+43 -15
View File
@@ -38,7 +38,7 @@ import (
// construction and are excluded from the diff; the other architectures list // construction and are excluded from the diff; the other architectures list
// their conditional branches outright. // their conditional branches outright.
func cmdAuditInstructions(args []string) error { func cmdAuditInstructions(args []string) error {
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [-I dir] [amd64|arm64|riscv64|loong64]", ` fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [--list] [-I dir] [amd64|arm64|riscv64|loong64]", `
Compare the gasm encoder for the given architecture (default amd64) against Compare the gasm encoder for the given architecture (default amd64) against
go tool asm and print the diff: superset encodings (gasm-only, shippable via go tool asm and print the diff: superset encodings (gasm-only, shippable via
gasm asm --format goobj) and known-but-unencodable names (the backlog). The gasm asm --format goobj) and known-but-unencodable names (the backlog). The
@@ -55,16 +55,19 @@ toolchain probing. A file whose name carries a recognisable _arch suffix is
attempted for that architecture; a file without one is attempted for all attempted for that architecture; a file without one is attempted for all
four, exactly as a GOARCH build would compile it. The report gives the four, exactly as a GOARCH build would compile it. The report gives the
per-architecture pass rates and the most common failure reasons, which drive per-architecture pass rates and the most common failure reasons, which drive
the encodability backlog by frequency rather than by table order. the encodability backlog by frequency rather than by table order. With
-list the report also prints every failing file with its reason, per
architecture.
`) `)
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons") corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
list := fs.Bool("list", false, "with --corpus, list every failing file with its reason, per architecture")
var dirs includeDirs var dirs includeDirs
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)") fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
if err := fs.Parse(args); err != nil { if err := fs.Parse(args); err != nil {
return err return err
} }
if *corpus { if *corpus {
return cmdAuditCorpus(fs.Args(), dirs) return cmdAuditCorpus(fs.Args(), dirs, *list)
} }
archName := "amd64" archName := "amd64"
switch n := len(fs.Args()); { switch n := len(fs.Args()); {
@@ -403,20 +406,30 @@ type corpusTally struct {
assembled int assembled int
reasons map[string]int // failure reason → count reasons map[string]int // failure reason → count
example map[string]string // failure reason → one representative file example map[string]string // failure reason → one representative file
fails []corpusFailure // every failure, in file order, for --list
} }
func (t *corpusTally) fail(path, reason string) { // corpusFailure is one failed attempt, recorded for the --list report.
type corpusFailure struct {
path string
reason string
detail string
}
func (t *corpusTally) fail(path string, err error) {
reason := corpusReason(err)
t.reasons[reason]++ t.reasons[reason]++
if t.example[reason] == "" { if t.example[reason] == "" {
t.example[reason] = path t.example[reason] = path
} }
t.fails = append(t.fails, corpusFailure{path: path, reason: reason, detail: firstLine(err.Error())})
} }
// cmdAuditCorpus implements audit-instructions --corpus. The include // cmdAuditCorpus implements audit-instructions --corpus. The include
// directories carry #include resolution over a corpus whose files refer to // directories carry #include resolution over a corpus whose files refer to
// headers such as GOROOT/pkg/include, the same -I a toolchain comparison // headers such as GOROOT/pkg/include, the same -I a toolchain comparison
// needs. // needs.
func cmdAuditCorpus(args []string, dirs includeDirs) error { func cmdAuditCorpus(args []string, dirs includeDirs, list bool) error {
if len(args) > 1 { if len(args) > 1 {
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")} return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
} }
@@ -454,7 +467,7 @@ func cmdAuditCorpus(args []string, dirs includeDirs) error {
if err != nil { if err != nil {
return err return err
} }
printCorpusStats(stats) printCorpusStats(stats, list)
return nil return nil
} }
@@ -645,21 +658,22 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
hdrDir, err := hdr.dirFor(pkgDir, goos, goarchName(tg.a)) hdrDir, err := hdr.dirFor(pkgDir, goos, goarchName(tg.a))
if err != nil { if err != nil {
ok = false ok = false
t.fail(path, corpusReason(err)) t.fail(path, err)
continue continue
} }
f, errs := parser.ParseWithOptions(path, src, parser.Options{ f, errs := parser.ParseWithOptions(path, src, parser.Options{
Expand: true, Expand: true,
IncludeDirs: append(slices.Clone(dirs), hdrDir), IncludeDirs: append(slices.Clone(dirs), hdrDir),
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
}) })
if len(errs) > 0 { if len(errs) > 0 {
ok = false ok = false
t.fail(path, corpusReason(errs[0])) t.fail(path, errs[0])
continue continue
} }
if _, err := assembleFile(tg.a, f); err != nil { if _, err := assembleFile(tg.a, f, goos); err != nil {
ok = false ok = false
t.fail(path, corpusReason(err)) t.fail(path, err)
continue continue
} }
t.assembled++ t.assembled++
@@ -670,21 +684,28 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
continue continue
} }
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
ok := true ok := true
for _, i := range wanted { for _, i := range wanted {
tg, t := targets[i], tallies[i] tg, t := targets[i], tallies[i]
t.attempted++ t.attempted++
// The parse carries the target's platform predefines, so it
// cannot be shared across targets the way a header-free file's
// could: a #ifdef GOARCH_arm block must be live on arm64 and
// dead everywhere else.
f, errs := parser.ParseWithOptions(path, src, parser.Options{
Expand: true,
IncludeDirs: dirs,
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
})
var err error var err error
if len(errs) > 0 { if len(errs) > 0 {
err = errs[0] // a parse failure is a failure for every target err = errs[0] // a parse failure is a failure for every target
} else { } else {
_, err = assembleFile(tg.a, f) _, err = assembleFile(tg.a, f, goos)
} }
if err != nil { if err != nil {
ok = false ok = false
t.fail(path, corpusReason(err)) t.fail(path, err)
continue continue
} }
t.assembled++ t.assembled++
@@ -706,7 +727,7 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
} }
// printCorpusStats renders the corpus audit report. // printCorpusStats renders the corpus audit report.
func printCorpusStats(s *corpusStats) { func printCorpusStats(s *corpusStats, list bool) {
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d named for other Go ports, never attempted)\n", s.root, s.files, s.generic, s.otherPort) fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d named for other Go ports, never attempted)\n", s.root, s.files, s.generic, s.otherPort)
// The rate is over the files a supported build would attempt: the // The rate is over the files a supported build would attempt: the
// other ports' files sit in the count for completeness but can never // other ports' files sit in the count for completeness but can never
@@ -721,6 +742,13 @@ func printCorpusStats(s *corpusStats) {
fmt.Printf(" %4d %s\n", t.reasons[r], r) fmt.Printf(" %4d %s\n", t.reasons[r], r)
fmt.Printf(" e.g. %s\n", t.example[r]) fmt.Printf(" e.g. %s\n", t.example[r])
} }
if !list {
continue
}
for _, f := range t.fails {
fmt.Printf(" FAIL %s\n", f.path)
fmt.Printf(" %s: %s\n", f.reason, f.detail)
}
} }
} }
+1 -1
View File
@@ -84,7 +84,7 @@ func disSource(path string, target arch.Arch) int {
if len(errs) > 0 { if len(errs) > 0 {
return 1 return 1
} }
img, err := assembleFile(target, f) img, err := assembleFile(target, f, "")
if err != nil { if err != nil {
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err) fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
return 1 return 1
+34 -11
View File
@@ -574,7 +574,7 @@ naming the package.
defer cleanup() defer cleanup()
dirs = append(dirs, hdrDir) dirs = append(dirs, hdrDir)
} }
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs}) f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs, Predefines: platformPredefinesFor(string(targetArch), goos)})
for _, e := range errs { for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e) fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
} }
@@ -582,7 +582,7 @@ naming the package.
return 1 return 1
} }
img, err := assembleFile(targetArch, f) img, err := assembleFile(targetArch, f, goos)
if err != nil { if err != nil {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err) fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
return 1 return 1
@@ -789,11 +789,34 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
return 1 return 1
} }
// platformPredefines mirrors the go command's assembler invocation, which
// defines GOOS_<goos> and GOARCH_<arch> as -D macros: GOROOT headers
// (go_tls.h, asm_riscv64.h) select their platform blocks with #ifdef on
// exactly those names, so an assembler without them cannot see the platform
// definitions at all.
func platformPredefines(goarch, goos string) map[string]string {
return map[string]string{
"GOARCH_" + goarch: "1",
"GOOS_" + goos: "1",
}
}
// platformPredefinesFor resolves the ambient GOOS the way a build would: a
// file whose name carries one (sys_darwin_arm64.s) is compiled for that GOOS
// and nothing else.
func platformPredefinesFor(goarch string, fileGoos string) map[string]string {
goos := fileGoos
if goos == "" {
goos = runtime.GOOS
}
return platformPredefines(goarch, goos)
}
// assembleFile assembles a parsed file for the given architecture and returns the image. // assembleFile assembles a parsed file for the given architecture and returns the image.
func assembleFile(targetArch arch.Arch, f *ast.File) (*asm.Image, error) { func assembleFile(targetArch arch.Arch, f *ast.File, goos string) (*asm.Image, error) {
switch targetArch { switch targetArch {
case arch.AMD64: case arch.AMD64:
return asm.AssembleFile(f) return asm.AssembleFile(f, asm.WithGOOS(goos))
case arch.RISCV: case arch.RISCV:
return asm.AssembleFileRISCV(f) return asm.AssembleFileRISCV(f)
case arch.ARM64: case arch.ARM64:
@@ -813,18 +836,18 @@ func assemblePath(path string, forced arch.Arch, dirs includeDirs) (*asm.Image,
if err != nil { if err != nil {
return nil, err return nil, err
} }
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs}) target := forced
if target == arch.Unknown {
target = arch.FromFilename(path)
}
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs, Predefines: platformPredefinesFor(string(target), "")})
for _, e := range errs { for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e) fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
} }
if len(errs) > 0 { if len(errs) > 0 {
return nil, fmt.Errorf("parse errors") return nil, fmt.Errorf("parse errors")
} }
target := forced return assembleFile(target, f, "")
if target == arch.Unknown {
target = arch.FromFilename(path)
}
return assembleFile(target, f)
} }
// printByteDiff shows the first few byte differences between two code blocks. // printByteDiff shows the first few byte differences between two code blocks.
@@ -920,7 +943,7 @@ func cmdVerifyNonJIT(path string, targetArch arch.Arch, groundTruth, profile boo
if len(errs) > 0 { if len(errs) > 0 {
return 1 return 1
} }
img, err := assembleFile(targetArch, f) img, err := assembleFile(targetArch, f, "")
if err != nil { if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err) fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1 return 1
+14
View File
@@ -119,6 +119,20 @@ identifier is a register or a label is an *architecture* question, so it is
left to `arch` and resolved in the lint/lsp layers. This keeps the parser left to `arch` and resolved in the lint/lsp layers. This keeps the parser
arch-agnostic and its output deterministic. arch-agnostic and its output deterministic.
### Optional preprocessing
With `Options{Expand: true}` the parser runs a pre-parse pass
(`preproc.go`) that splices `#include` files (the source directory, then the
`-I` directories), expands object and parameterised `#define` macros,
applies `#undef` and the `#ifdef`/`#ifndef`/`#else`/`#endif` family, and
folds constant expressions left in operands. The go command's platform
macros (`GOARCH_<arch>`, `GOOS_<goos>`) arrive through `Options.Predefines`.
The assembly path (`asm`, `diff`, `audit`) expands; `lint`, `fmt` and the
language server read the raw file. The command layer adds the go_asm.h
generator (`asmhdr.go`): a file that includes go_asm.h gets the package's
defines type-checked out of its Go files for the target architecture and
GOOS, with no compiler in the loop.
### `arch` ### `arch`
Register files are generated programmatically (the regular `R8`-`R15`, Register files are generated programmatically (the regular `R8`-`R15`,
+4 -2
View File
@@ -365,7 +365,7 @@ add: 16 bytes, args=24, frame=0 NOSPLIT
## audit-instructions ## audit-instructions
```text ```text
Usage: gasm audit-instructions [--corpus [dir]] [-I dir] [amd64|arm64|riscv64|loong64] Usage: gasm audit-instructions [--corpus [dir]] [--list] [-I dir] [amd64|arm64|riscv64|loong64]
``` ```
Compare the gasm encoder for the given architecture (default amd64) against the Compare the gasm encoder for the given architecture (default amd64) against the
@@ -403,7 +403,9 @@ architecture; a file without one is attempted for all four, exactly as a
The report gives the headline number (files The report gives the headline number (files
that assemble for every target architecture), the per-architecture pass rates that assemble for every target architecture), the per-architecture pass rates
and the most common failure reasons with one representative file each, which and the most common failure reasons with one representative file each, which
drive the encodability backlog by frequency rather than by table order. A run drive the encodability backlog by frequency rather than by table order. With
`--list` the report additionally prints every failing file with its failure
reason, per architecture. A run
over GOROOT takes under a second. over GOROOT takes under a second.
```sh ```sh
+8 -2
View File
@@ -1,8 +1,8 @@
.TH GASM-AUDIT-INSTRUCTIONS 1 "2026-09-19" "gasm" "User Commands" .TH GASM-AUDIT-INSTRUCTIONS 1 "2026-09-21" "gasm" "User Commands"
.SH NAME .SH NAME
gasm-audit-instructions \- diff the encoder against the Go toolchain, or measure a corpus gasm-audit-instructions \- diff the encoder against the Go toolchain, or measure a corpus
.SH SYNOPSIS .SH SYNOPSIS
.B gasm audit\-instructions [\-\-corpus [\fIdir\fR]] [\-I dir] [amd64|arm64|riscv64|loong64] .B gasm audit\-instructions [\-\-corpus [\fIdir\fR]] [\-\-list] [\-I dir] [amd64|arm64|riscv64|loong64]
.SH DESCRIPTION .SH DESCRIPTION
Compare the gasm encoder for the given architecture (default amd64) Compare the gasm encoder for the given architecture (default amd64)
against against
@@ -39,6 +39,12 @@ second.
Assemble a corpus of .s files and report pass rates and failure Assemble a corpus of .s files and report pass rates and failure
reasons. reasons.
.TP .TP
.B \-\-list
With
.BR \-\-corpus ,
print every failing file with its failure reason, per architecture,
instead of one representative file per reason.
.TP
.B \-I \fIdir\fR .B \-I \fIdir\fR
Directory to search for #include files; may be repeated, searched in Directory to search for #include files; may be repeated, searched in
order after the source directory. A corpus run whose files include order after the source directory. A corpus run whose files include
+31
View File
@@ -489,6 +489,17 @@ func parseImmediate(g []token.Token) ast.Immediate {
} else if g[i].Kind == token.Plus { } else if g[i].Kind == token.Plus {
i++ i++
} }
// A constant expression after the sign: $-(R - 8), $+(32-shift). The
// toolchain folds the negated value in place (the cgo ABI macros write
// ADJSP $-(REGS_HOST_TO_ABI0_STACK - 8)), so the sign applies to the
// folded value exactly as it does to a bare literal.
if i < len(g) && (g[i].Kind == token.LParen || g[i].Kind == token.Tilde) {
if v, rest, ok := foldExpr(g[i:]); ok && len(rest) == 0 {
imm.Val = v
imm.HasVal = true
return imm
}
}
if i < len(g) && g[i].Kind == token.Number { if i < len(g) && g[i].Kind == token.Number {
text := g[i].Text text := g[i].Text
if v, ok := tryInt(text); ok { if v, ok := tryInt(text); ok {
@@ -673,6 +684,26 @@ func parseAddress(g []token.Token) ast.Address {
if i > 0 && i < len(g) { if i > 0 && i < len(g) {
addr.Shift = joinRaw(g[i:]) addr.Shift = joinRaw(g[i:])
} }
// A lone (possibly signed) number is an absolute address: MOVL $0xf1,
// 0xf1 stores through the bare displacement with no base at all. In
// operand position a number without $ is an address, never a value.
if addr.Sym == nil && addr.Base == "" && addr.Index == "" && !addr.HasOff {
neg := false
j := 0
if j < len(g) && (g[j].Kind == token.Minus || g[j].Kind == token.Plus) {
neg = g[j].Kind == token.Minus
j++
}
if j == len(g)-1 && g[j].Kind == token.Number {
v := parseInt(g[j].Text)
if neg {
v = -v
}
addr.Offset = v
addr.HasOff = true
return addr
}
}
return addr return addr
} }
+9
View File
@@ -34,6 +34,12 @@ type Options struct {
// Expand enables macro expansion, include splicing and the // Expand enables macro expansion, include splicing and the
// statement-separator reading of ';' that the expanded bodies rely on. // statement-separator reading of ';' that the expanded bodies rely on.
Expand bool Expand bool
// Predefines names the macros defined before the file is read. The
// go command drives go tool asm with -D GOOS_<goos> -D GOARCH_<arch>,
// and GOROOT's own headers (go_tls.h, asm_riscv64.h) select their
// platform blocks with #ifdef on exactly those names, so an assembler
// without them cannot see the platform definitions at all.
Predefines map[string]string
} }
// ParseWithOptions parses src like Parse, optionally preprocessing it first. // ParseWithOptions parses src like Parse, optionally preprocessing it first.
@@ -44,6 +50,9 @@ func ParseWithOptions(path, src string, opts Options) (*ast.File, []error) {
var errs []error var errs []error
if opts.Expand { if opts.Expand {
pp := &preproc{opts: opts, macros: map[string]*macroDef{}} pp := &preproc{opts: opts, macros: map[string]*macroDef{}}
for name, value := range opts.Predefines {
pp.macros[name] = &macroDef{name: name, body: lexer.Tokenize(value)}
}
lines = pp.fileLines(path, tokens, token.Position{}) lines = pp.fileLines(path, tokens, token.Position{})
errs = pp.errs errs = pp.errs
} else { } else {
+52
View File
@@ -0,0 +1,52 @@
// Kernel: the operand forms the GOROOT campaign surfaced — numeric
// PC-relative jumps, symbol-immediate materialisation (the toolchain rewrites
// MOVQ $sym(SB) into a RIP-relative LEA) and the negated constant-expression
// ADJSP the cgo ABI macros write. Bytes are pinned against go tool asm by
// TestDifferentialKernels.
#include "textflag.h"
DATA sd<>(SB)/4, $7
GLOBL sd<>(SB), RODATA, $4
// func Jumps(flag int64) int64
TEXT ·Jumps(SB), NOSPLIT, $0-16
MOVQ flag+0(FP), AX
TESTQ AX, AX
JEQ 2(PC)
MOVQ $1, AX
JMP 3(PC)
MOVQ $2, AX
MOVQ AX, ret+0(FP)
RET
// func SymImm() int64
TEXT ·SymImm(SB), NOSPLIT, $0-16
MOVQ $sd<>(SB), AX
MOVQ $·SymImm(SB), CX
MOVQ AX, ret+0(FP)
RET
// func Frame()
TEXT ·Frame(SB), NOSPLIT, $0
PUSHFQ
CLD
ADJSP $(64 - 8)
ADJSP $-(64 - 8)
POPFQ
RET
// func Tls() int64
TEXT ·Tls(SB), NOSPLIT, $0-8
MOVQ TLS, BX
MOVQ 0(BX)(TLS*1), AX
MOVQ AX, ret+0(FP)
RET
// func Aligned() int64
TEXT ·Aligned(SB), NOSPLIT, $0-8
MOVQ $1, AX
PCALIGN $16
MOVQ $2, AX
PCALIGN $32
MOVQ AX, ret+0(FP)
RET