Compare commits

...
26 Commits
Author SHA1 Message Date
petrbalvin 8054fff9ac chore: prepare release v0.31.1
Test / vet (push) Successful in 47s
Release / build (amd64, linux) (push) Successful in 55s
Release / build (arm64, linux) (push) Successful in 51s
Release / build (loong64, linux) (push) Successful in 46s
Release / build (riscv64, linux) (push) Successful in 42s
Test / test (push) Successful in 2m29s
Release / release (push) Successful in 19s
Test / build (push) Successful in 42s
Assisted-by: MiMo V2.5 Pro
2026-08-20 22:35:11 +02:00
petrbalvin 6a79c35bf7 fix(version): bump version to 0.31.0 in justfile and main.go
Assisted-by: MiMo V2.5 Pro
2026-08-20 22:35:11 +02:00
petrbalvin 6f4f2096e9 chore: prepare release v0.31.0
Release / build (amd64, linux) (push) Successful in 42s
Release / build (arm64, linux) (push) Successful in 40s
Release / build (loong64, linux) (push) Successful in 42s
Release / build (riscv64, linux) (push) Successful in 45s
Release / release (push) Successful in 18s
2026-08-20 16:24:33 +02:00
petrbalvin 56630f8624 chore(toolchain): upgrade to Go 1.27
Test / vet (push) Successful in 1m5s
Test / test (push) Successful in 2m33s
Test / build (push) Successful in 40s
2026-08-20 16:03:26 +02:00
petrbalvin 459f4a2b6e fix(test): add arm64 encoding tests for Go 1.26 coverage compatibility
Assisted-by: MiMo V2.5 Pro
2026-08-20 15:44:23 +02:00
petrbalvin 5d66343488 fix(goobj): make R_DWTXTADDR_U4 relocation type Go-version-aware
The relocation type number shifted between Go 1.26 (103) and Go 1.27
(106)
because new LoongArch relocations were inserted. Detect the Go version
at
runtime and use the correct value.
2026-08-20 15:35:39 +02:00
petrbalvin 48334c4d5a docs: remove completed roadmap phases, fix licence description 2026-08-20 15:01:57 +02:00
petrbalvin 7629963cab chore: fix project conventions — .gitignore, CHANGELOG categories, docs naming
Assisted-by: MiMo V2.5 Pro
2026-08-20 14:47:39 +02:00
petrbalvin 97951cbeb6 feat(asm): extend arm64 encoder with atomics, bitfield, SIMD and more test kernels
Assisted-by: MiMo V2.5 Pro
2026-08-20 14:31:15 +02:00
petrbalvin 6e73f59e78 feat(asm): extend arm64 encoder with FP, conditional select, CRC32 and tests
Assisted-by: MiMo V2.5 Pro
2026-08-20 14:07:12 +02:00
petrbalvin 4221ec5741 feat(asm): add AArch64 arm64 encoder with ground-truth verification
Assisted-by: MiMo V2.5 Pro
2026-08-20 13:33:39 +02:00
petrbalvin 01dcc3b86e revert(toolchain): restore go1.26 in CI
Test / vet (push) Successful in 43s
Test / test (push) Failing after 1m55s
Test / build (push) Skipped
Release / build (amd64, linux) (push) Successful in 40s
Release / build (arm64, linux) (push) Successful in 37s
Release / build (loong64, linux) (push) Successful in 38s
Release / build (riscv64, linux) (push) Successful in 58s
Release / release (push) Successful in 22s
2026-08-13 18:34:20 +02:00
petrbalvin b08885bd31 chore(release): prepare v0.30.0
Test / vet (push) Failing after 17s
Test / test (push) Skipped
Test / build (push) Skipped
Assisted-by: DeepSeek V4 Pro
2026-08-13 18:27:12 +02:00
petrbalvin c05c53452f fix(asm): encode RISC-V CALL sym(SB) as JAL
Test / vet (push) Successful in 47s
Test / test (push) Failing after 2m9s
Test / build (push) Skipped
Assisted-by: DeepSeek V4 Pro
2026-08-13 18:12:22 +02:00
petrbalvin 681a449c01 fix(asm): match RISC-V branch and jump encodings 2026-08-13 17:57:10 +02:00
petrbalvin 31a2cee382 fix(asm): materialise RISC-V MOV immediates
Test / vet (push) Successful in 44s
Test / test (push) Failing after 1m56s
Test / build (push) Skipped
2026-08-13 17:41:16 +02:00
petrbalvin 3bc7c18bc3 fix(asm): materialise large RISC-V immediates
Assisted-by: DeepSeek V4 Pro
2026-08-13 16:04:08 +02:00
petrbalvin f0512a4e1c fix(asm): complete RISC-V compressed loads/stores and word arithmetic
Assisted-by: DeepSeek V4 Pro
2026-08-13 15:42:38 +02:00
petrbalvin 373c09f725 fix(asm): correct RISC-V operand order and complete RVC compression
Assisted-by: DeepSeek V4 Pro
2026-08-13 15:13:31 +02:00
petrbalvin b78b6c5004 fix(asm): correct RISC-V frame layout and RVC encodings
Assisted-by: GLM 5.2
2026-08-13 14:41:57 +02:00
petrbalvin 9b5c878f9e feat(asm): emit RISC-V GOOBJ with the shared emitter
Assisted-by: DeepSeek V4 Pro
2026-08-13 12:07:29 +02:00
petrbalvin 31ee8e7941 feat(asm): add LoongArch encoder with ELF and GOOBJ emission
Assisted-by: DeepSeek V4 Pro
2026-08-13 11:24:44 +02:00
petrbalvin 2d1176e045 feat(asm): resolve external GOOBJ symbols from archive data
Test / vet (push) Successful in 1m2s
Test / test (push) Failing after 2m5s
Test / build (push) Skipped
2026-08-08 16:18:56 +02:00
petrbalvin 2a27a3a52b docs: add BSD-3-Clause headers to generated files and update CI docs 2026-08-07 22:43:40 +02:00
petrbalvin cde7d0f96a docs: document watchpoint slot tracking and update debugger commands 2026-08-07 22:27:49 +02:00
petrbalvin d7ee1b78d4 feat: drop Mach-O and macOS support, Linux-only 2026-08-07 22:20:26 +02:00
79 changed files with 11169 additions and 1788 deletions
+1 -1
View File
@@ -25,7 +25,7 @@ jobs:
- uses: actions/setup-go@v6
with:
go-version: "1.26"
go-version: "1.27"
- name: Download dependencies
run: go mod download
+3 -3
View File
@@ -15,7 +15,7 @@ jobs:
- uses: actions/setup-go@v6
with:
go-version: "1.26"
go-version: "1.27"
- name: Download dependencies
run: go mod download
@@ -41,7 +41,7 @@ jobs:
- uses: actions/setup-go@v6
with:
go-version: "1.26"
go-version: "1.27"
- name: Download dependencies
run: go mod download
@@ -84,7 +84,7 @@ jobs:
- uses: actions/setup-go@v6
with:
go-version: "1.26"
go-version: "1.27"
- name: Download dependencies
run: go mod download
+3 -4
View File
@@ -7,9 +7,8 @@
coverage.out
*.test
# Editor detritus
*.swp
.DS_Store
# Scratch / temporary work
_scratch/
# ZCode workspace
.zcode
+1 -1
View File
@@ -55,7 +55,7 @@ No body, no footers, no trailing period on the subject.
## Code Style
Language: Go 1.26 (`toolchain go1.26.5`).
Language: Go 1.27 (`toolchain go1.27.0`).
### Formatter
+149 -55
View File
@@ -9,6 +9,151 @@ and this project adheres to [Conventional Commits](https://www.conventionalcommi
Unreleased changes on the `development` branch.
### Added
-
## [0.31.1] — 2026-08-20
### Fixed
- **Version stamp.** The v0.31.0 release binary reported itself as `0.30.0`
because the version variables in `justfile` and `cmd/gasm/main.go` were not
bumped during the release commit.
## [0.31.0] — 2026-08-20
The arm64 encoder (Phase 5 — complete) ships with ELF64 and GOOBJ emission,
verified byte-for-byte against `GOARCH=arm64 go tool asm` and linked into a
real `go build`. The encoder covers the full integer instruction set, FP
arithmetic, conditional select, CRC32, and the MOV pseudo-instruction with
bitmask immediate encoding. The project now requires Go 1.27.
### Added
- **arm64 encoder (Phase 5 — complete).** `gasm asm` can now assemble `_arm64.s`
files: the AArch64 integer instruction set with the MOV pseudo-instruction and
its immediate-constant expansions (MOVZ/MOVN/MOVK for wide immediates, ORR with
logical bitmask encoding for values like `$1`), data-processing (shifted
register and immediate forms), load/store (scaled unsigned and unscaled9-bit
immediate), conditional and unconditional branches, FP/SP frame mapping,
SB/global symbol references (ADRP+ADD pairs with `R_ADDRARM64` relocations),
jump chain folding, and ELF64 emission (`gasm asm --format elf`). Ground-truth
verification against `GOARCH=arm64 go tool asm` matches byte-for-byte. Phase 5
(the other architectures — RISC-V, LoongArch, arm64) is now complete.
### Changed
- **Go 1.27 required.** The project now requires Go 1.27 (`toolchain go1.27.0`).
The `R_DWTXTADDR_U4` relocation type is detected at runtime for backward
compatibility.
## [0.30.0] — 2026-08-13
The LoongArch encoder (Phase 5) ships with ELF64 and GOOBJ emission, verified
byte-for-byte against `GOARCH=loong64 go tool asm` and linked into a real
`go build`; the shared GOOBJ emitter now writes the per-function DWARF symbols
the linker's DWARF pass reads. The RISC-V encoder reaches byte-for-byte parity
with `go tool asm`: the frame model, operand ordering, RVC compression,
large-immediate and `MOV $imm` materialisation, branch/jump encodings, and
`CALL sym(SB)` (now a `JAL` with an `R_RISCV_JAL` relocation). The debugger
tracks four hardware watchpoint slots, and the toolkit is Linux-only.
### Added
- **LoongArch encoder (Phase 5).** `gasm asm` can now assemble `_loong64.s`
files: the full LoongArch64 instruction set with the dual-form arithmetic
mnemonics, the 16/21-bit branch families, the MOV pseudo-instruction and
its immediate-constant expansions, FP/SP frame mapping, SB/global symbol
references (pcalau12i pairs) and ELF64 emission
(`gasm asm --format elf`). Ground-truth verification against
`GOARCH=loong64 go tool asm` matches byte-for-byte; GOOBJ emission
(`gasm asm --format goobj`) is proven end-to-end by linking the object
into a cross-compiled `go build`.
- **GOOBJ DWARF symbols.** The GOOBJ emitters now write the per-function
DWARF symbols the linker requires (the subprogram DIE and the `.debug_line`
program, byte-identical to `cmd/asm`'s), and the pc-value table deltas are
in the architecture's MinLC units as the runtime expects — the amd64 link
test now genuinely substitutes the gasm object, and the amd64/loong64
end-to-end GOOBJ link tests pass.
- **RISC-V GOOBJ emission via the shared emitter.** RISC-V GOOBJ output is
now written by the same shared emitter as amd64 and LoongArch, modelling
each AUIPC + second-instruction pair as a single R_RISCV_PCREL_ITYPE/STYPE
relocation (the layout `cmd/asm` writes, not the ELF HI20/LO12 pair), so the
object links into a cross-compiled `go build` for `GOARCH=riscv64`. An
end-to-end link test substitutes the gasm object and reads the symbol back
with `go tool nm`; the rewrite also corrects the relocation `after` field.
### Fixed
- **RISC-V frame model and RVC encodings.** The riscv64 frame layout now
matches `go tool asm`: the prologue/epilogue save and restore the link
register (LR) instead of S0, with the correct autosize (locals + 8) and the
RVC-compressed prologue/epilogue instructions; `RET` emits the uncompressed
`JALR X0, 0(X1)` the toolchain writes; the `C.ADDI`/`C.LI`/`C.LUI`/`C.ADDIW`
opcode bit and the `C.ADD` CR-type encoding are fixed; and the `LR`/`TMP`
register aliases now resolve to X1 and X31. The pcsp/pcfile/pcline tables
are populated from the recorded stack-adjustment and source-line data, and
a byte-exact ground-truth test compares framed and leaf functions against
`GOARCH=riscv64 go tool asm`.
- **RISC-V operand ordering and RVC compression.** R-type instructions now
take `rs2, rs1, rd` and I-type arithmetic instructions take `imm12, rs1,
rd`, matching the Go assembler's documented operand order (previously both
were reversed, so non-commutative R-type instructions such as `SUB` encoded
the wrong operation). The two-operand ternary forms (`ADD rs2, rd`,
`ADDI $imm, rd`, `SLLI $shamt, rd`) are now accepted. RVC compression is
completed for `C.ADDI16SP`, `C.SLLI`, `C.SRLI`, `C.SRAI`, `C.ANDI`,
`C.NOP`, `C.EBREAK`, `C.MV` (from `ADDI`/`ADD`) and the commutative
`AND`/`OR`/`XOR` forms; the byte-exact ground-truth test now covers these.
- **RISC-V compressed loads/stores and word arithmetic.** RVC compression
now also covers the register-relative `C.LW`/`C.SW`/`C.LD`/`C.SD`/
`C.FLD`/`C.FSD` forms (in addition to the stack-relative `C.LWSP`/`C.SWSP`/
`C.LDSP`/`C.SDSP`), plus `C.ADDI4SPN`, `C.ADDW` and `C.SUBW`. The
byte-exact ground-truth test exercises these against `GOARCH=riscv64
go tool asm`.
- **RISC-V large-immediate materialisation.** `ADDI`/`ANDI`/`ORI`/`XORI`
with a 32-bit immediate that does not fit 12 bits now expand exactly as
`cmd/asm`: two `ADDI`s for the small `ADDI` split range, and
`LUI`+`ADDIW`+`<op>` otherwise, with the `LUI` and `ADDIW` compressed to
`C.LUI`/`C.ADDIW` when their immediate fits six signed bits. The
byte-exact ground-truth test covers positive, negative, and out-of-range
immediates against `GOARCH=riscv64 go tool asm`.
- **RISC-V `MOV $imm, rd` materialisation.** The immediate-loading
pseudo-instruction now uses the toolchain's `Split32BitImmediate` split
(previously it rounded the upper 20 bits, producing wrong results for
negative and bit-11-set immediates) and compresses the emitted
`ADDI`/`LUI`/`ADDIW` to `C.LI`/`C.LUI`/`C.ADDIW` when their immediate
fits six signed bits. A byte-exact ground-truth test covers zero, small,
negative, and 32-bit immediates against `GOARCH=riscv64 go tool asm`.
- **RISC-V branch/jump compression.** `JMP`/`JAL` were being compressed to
`C.J` and `BEQ`/`BNE` (with `X0`) to `C.BEQZ`/`C.BNEZ`, but `go tool asm`
never emits these compressed forms. They now emit the 32-bit `JAL` and
branch encodings the toolchain writes; the dead `C.J`/`C.BEQZ`/`C.BNEZ`
encoders were removed, and the `C.LUI` direct-instruction compression now
uses the correct six-bit signed range. A byte-exact ground-truth test
covers the branch family and jumps against `GOARCH=riscv64 go tool asm`.
- **RISC-V `CALL sym(SB)`.** The call pseudo-instruction now emits the
toolchain's `JAL X1, sym(SB)` with a single `R_RISCV_JAL` relocation
(previously it emitted an `AUIPC`+`JALR` pair against a local branch
label, a form `go tool asm` rejects). The GOOBJ and ELF emitters now map
that relocation (Go objabi 59 / ELF `R_RISCV_JAL` 17, a 4-byte field), and
relocation offsets are recorded relative to the function start (including
the prologue). A byte-exact ground-truth test covers a call against
`GOARCH=riscv64 go tool asm`.
- **Debugger watchpoint slots.** `gasm debug`'s `watch` command always used
hardware watchpoint slot 0, so a second `watch` call silently overwrote
the first. Watchpoint slots are now tracked in the `Session` (DR0–DR3);
`watch` picks the first free slot and reports an error if all four are in
use, and `unwatch <slot>` clears one (no argument clears all).
### Changed
- **Linux only.** The toolkit, its CI and the released binaries are now
Linux-only; cross-compiled to linux/{amd64,arm64,riscv64,loong64}.
- **Phase 4 closed.** README's "Remaining" list for the debugger is gone;
disassembly at PC, memory-write, watchpoints, and source-line mapping are
all shipped.
## [0.29.0] — 2026-08-07
RISC-V GOOBJ emission, YMM vector register display, named buffer allocation
@@ -60,20 +205,13 @@ new CLI commands. A signature-parser fix corrects grouped Go parameters.
prevent GC from collecting heap objects whose addresses were passed to JIT
code via `unsafe.Pointer`; all verify tests pass 100/100 under `-race`.
### Cleaned up
### Changed
- **Removed external kernel test dependencies** — the verify test suite no
longer references production kernels from the separate go-libraries project.
The remaining test suite uses only `testdata/verify/*.s` kernels, which are
part of this repository. Coverage is identical locally and in CI (80.3 %).
### Verified
- `gasm diff` detects byte-level differences; `--map` pairs differently-named
functions for comparison.
- `gasm verify --call` invokes functions with user-supplied buffers; the arg
block is printed before and after the call, showing return values.
- LSP go-to-definition resolves labels across functions and files.
## [0.28.0] — 2026-08-03
@@ -105,10 +243,6 @@ ground-truth verification against `GOARCH=riscv64 go tool asm`.
- RVC: C.LDSP/C.SDSP/FLDSP/FSDSP immediate encoding now matches Go toolchain
(bit-interleaved format).
### Verified
- 118 RISC-V tests, asm coverage 83.3%.
- Ground-truth: C.LDSP, C.SDSP, C.FLDSP, C.FSDSP byte-exact vs Go toolchain.
## [0.27.0] — 2026-08-01
@@ -141,7 +275,7 @@ area bit-for-bit.
analyze, autocorr) pass; partial functions (decoders that fault on malformed
input) should use `--ground-truth` instead.
### Known limitation
### Fixed
`--fuzz` crashes the process for partial functions (e.g. LZ4 decoders) whose
over-copy paths read past the buffer on random garbage input. Subprocess
@@ -204,11 +338,6 @@ The remaining go-flac encoder kernels join the differential suite.
frames: the four zigzag-fold entropy sums compared against the scalar
loop).
### Verified
- `gasm fmt` doc-comment indentation confirmed correct: comments before
every TEXT are at column 0 (the RET-detection logic handles multi-exit
functions).
## [0.21.0] — 2026-07-26
@@ -246,7 +375,7 @@ test corpus exercises.
argument blocks and collect distinct output fingerprints (the result
words); reports path diversity as a lower bound on code coverage.
### Note
### Changed
INT3-based per-block hit counting was prototyped but deferred: Go's runtime
signal management (sigaltstack, handler re-installation) makes raw
@@ -311,11 +440,6 @@ toolchain.
known-answer LZ4 blocks decode bit-for-bit, wide copies of 0–1024 bytes
match, malformed input returns the correct error codes.
### Verified
- `just test` (race, 84.6 % total coverage, verify 82.2 %).
- `gasm verify` on both go-lz4 kernels: all functions JIT-load and
smoke-test clean.
## [0.16.0] — 2026-07-21
@@ -441,13 +565,6 @@ Go assembler.
(`DATA mask<>+8(SB)/8, $0x800f…`) parse as unsigned and keep their bit
pattern, instead of being rejected as non-integer.
### Verified
- End-to-end: a gasm-emitted GOOBJ swapped into a `go build` in place of
the toolchain's assembly object links and runs with output identical to
the baseline binary (stack-argument calls and a `GLOBL` relocation
resolved by the Go linker). All 17 go-flac AVX2 kernel functions emit as
a GOOBJ that `go tool nm` reads back with every symbol intact.
## [0.11.0] — 2026-07-16
@@ -506,7 +623,7 @@ verified byte for byte against the Go assembler.
arithmetic, the unpacks, VMOVDDUP and the conversions all accept the
explicit K1–K7 operand and the `.Z` suffix the way Go writes them.
### Documented
### Changed
- VCVTPS2PD follows the Go assembler's encoding, which omits the F3
mandatory prefix (VEX.pp / EVEX.pp = 00) that Intel's maps prescribe; the
@@ -514,15 +631,6 @@ verified byte for byte against the Go assembler.
gasm reproduces it exactly (and round-trips through the x86 decoder, which
shares the convention).
### Verified
- 58 new ground-truth cases — every instruction extracted from the Go
toolchain's own assembly (go build + an executable-segment dump), checked
byte for byte and round-tripped through the decoder, covering disp8×N for
the scalar (×8/×4), duplication (×8/×32/×64) and conversion (×8/×16/×32)
memory operands, the 5-bit register fields and the masked/zeroing P2
byte. All four go-flac/go-lz4 kernels still assemble byte-identically
and lint clean.
## [0.9.0] — 2026-07-14
@@ -658,13 +766,6 @@ the Go toolchain, completing the production-kernel coverage.
- `asm`: the VEX encoder now rejects vector register indices 16–31 instead of
encoding a truncated (wrong) register.
### Verified
- All 10 functions of the go-flac `avx512_amd64.s` kernel assemble
byte-identically to the Go toolchain's machine code (the disp32 of the one
`VMOVDQU32 idx16(SB), Z13` load is linker-filled in Go and resolved within
gasm's own image — checked to reach the right constant bytes). The AVX2
kernel's 17 functions remain byte-identical.
## [0.4.0] — 2026-07-09
@@ -685,13 +786,6 @@ machine code byte for byte.
- `gasm asm` prints the data section and symbol map alongside the functions
and writes the whole image (code + data) with `-o`.
### Verified
- All 17 functions of the go-flac `avx2_amd64.s` kernel assemble
byte-identically to the Go toolchain's machine code; the only differing
bytes are the displacements of the two `VMOVDQU mask24<>(SB), X15` loads,
which the Go linker fills at link time and gasm resolves within its own
image (checked to reach the right constant bytes).
## [0.3.0] — 2026-07-08
+12 -5
View File
@@ -2,9 +2,9 @@
## Prerequisites
- Go 1.26 or later (`toolchain go1.26.5`)
- Go 1.27 or later (`toolchain go1.27.0`)
- `just` command runner
- A Linux, FreeBSD, or macOS host on amd64 or arm64
- A Linux host on amd64, arm64, riscv64 or loong64
## Development Setup
@@ -68,8 +68,15 @@ See [AGENTS.md](AGENTS.md) for the full style guide. Key points:
## CI
There is no CI pipeline in this repository. The Definition of Done
(`just build` + `just test` + `just fmt`) is enforced locally.
CI runs on every push to `development` and on pull requests:
- **Test** (`test.yml`) — `gofmt` check, `go vet`, `go test -race` and the
80 % coverage gate.
- **Release** (`release.yml`) — cross-compiles release binaries for
linux/{amd64,arm64,riscv64,loong64} on version tags and publishes them.
The Definition of Done (`just build` + `just test` + `just fmt`) must still
pass locally before pushing.
## AI-Assisted Contributions
@@ -85,7 +92,7 @@ Attribute agent authorship in issues and pull requests on one trailing
line:
```
_Assisted-by: DeepSeek V4 Pro_
_Assisted-by: Qwen 3.8 Max_
```
## Questions
+21 -239
View File
@@ -2,8 +2,6 @@
Developer tooling for **GAsm** — Go's built-in Plan 9 assembler.
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
Go ships an assembler but no tooling for it. There is no syntax highlighting,
no autocomplete, no linter, no static analyser, no formatter, no standalone
assembler and no debugger for `.s` files. Developers write assembly blind,
@@ -18,23 +16,13 @@ gasm parse parse and report syntax errors
gasm fmt canonicalise formatting (gofmt for assembly)
gasm lint static checks
gasm lsp language server (completion, hover, symbols, diagnostics, highlighting)
gasm asm standalone assembler (Phase 2)
gasm verify dynamic analysis & verification (Phase 3)
gasm debug source-level debugger (Phase 4)
gasm asm standalone assembler
gasm verify dynamic analysis & verification
gasm debug source-level debugger
gasm diff compare machine code of two .s files
gasm profile show basic-block structure of functions
```
> **Status: Phase 4 — done, Phase 5 underway.** Phase 1 (the language
> foundation, linter, formatter and language server) shipped in v0.1.0;
> Phase 2 (the standalone assembler — the full amd64 instruction set plus
> ELF, Mach-O and GOOBJ object emission) in v0.12.0; Phase 3 (dynamic
> analysis — JIT execution, differential testing, ABI checks and coverage
> profiling) in v0.25.0; Phase 4 (interactive debugger — ptrace-based,
> breakpoints, watchpoints, stepping, vector register display, named buffer
> allocation) in v0.27.0; RISC-V encoder (RV64IMAFDC + RVC, ELF emission,
> ground-truth, GOOBJ) in v0.28.0–v0.29.0. See [Roadmap](#roadmap).
## Architecture support
gasm-devkit targets every architecture Go's assembler speaks. The instruction
@@ -56,216 +44,13 @@ carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
command — `just gen` — and requires only a Go installation; the committed
output has no runtime dependency on the toolchain.
The target *architectures* above are what the toolkit analyses. The toolkit
itself is portable Go and builds on Linux, FreeBSD and macOS, on amd64 and
arm64 hosts.
## Supported Platforms
## Roadmap
The toolkit runs on Linux. All four Linux architectures are supported as
hosts — amd64, arm64, riscv64 and loong64 — and the release matrix
cross-compiles the same four targets.
The work is delivered in four phases. Each phase is completed and hardened
before the next begins. The ordering follows a dependency chain: understand
the code statically (Phase 1), make it runnable (Phase 2), then run it and
observe or control it (Phases 3–4).
### Phase 1 — language foundation, editor tooling and static analysis · *done*
Everything needed to read, understand, check, format and highlight GAsm —
without executing it.
| Capability | Status |
|------------|--------|
| Lexer — permissive, position-aware scanner for all four architectures | done |
| Parser — line-oriented, error-tolerant, full AST with source positions | done |
| Instruction + register tables for amd64, arm64, riscv64, loong64 (generated, complete) | done |
| Linter — `unknown-instruction`, `operand-count`, `undefined-label`, `duplicate-label`, `missing-ret`, `missing-textflag-include`, `abi-argsize`, `unreachable-code`, `register-clobber`, `funcdata-pcdata` | done |
| Formatter — idempotent, comment-preserving, per-function alignment; a `RET` terminates the body for indentation, so the next function's doc comment stays at column 0; exactly one blank line before every block (label, `TEXT`, `GLOBL`) and runs of blanks collapsed; directory / no-argument mode reformats every `.s` in place, `go fmt`-style | done |
| Language server — completion, hover, document symbols, diagnostics, semantic-token highlighting | done |
| CLI — `gasm tokens / parse / fmt / lint / lsp` | done |
| Real-world validation against production AVX2 / AVX-512 kernels | done |
| Lint hardening — zero false positives across the Go runtime corpus (90 files, all four architectures): macro-invocation handling, branch aliases (`B`/`BL`/`JAL`), addressing suffixes (`.P`/`.W`), terminal `UNDEF` | done |
| Static analysis — `abi-argsize` (argument/result area computed from the `// func` signature under Go's ABI0 layout and checked against the TEXT declaration) and `unreachable-code` (dead code after `RET`, suppressed where reachability is undecidable: PC-relative jumps, register-indirect branches, `#ifdef`) | done |
| Static analysis — register liveness (CFG construction + per-instruction def/use + iterative backward dataflow) driving `register-clobber`, calibrated to the **Go ABI** (not System V): flags writes to the registers Go fixes across calls — the frame pointer and the goroutine pointer (`R14` on amd64, `R28`/`R29` on arm64, `X27` on riscv64, `R22` on loong64, plus the OS-reserved `R18` on arm64) — that are never saved/restored; the goroutine pointer is reported only when the function can reach the runtime (not `NOSPLIT`, or makes calls), matching how the runtime's own assembly uses it. `funcdata-pcdata` structural validation of `FUNCDATA`/`PCDATA` operands and indices | done |
> **Limitation — macros.** gasm-devkit reads `.s` source as written; it does
> **not** run the C preprocessor, so `#define` macros are not expanded. Files
> that use macros (the runtime's `asm_*.s`, `race_*.s`, `sys_*.s`, …) parse
> cleanly, and macro *invocations* are recognised and never flagged, but the
> `undefined-label` and `missing-ret` heuristics are suppressed in macro-using
> files because labels a macro defines are invisible without expansion. Full
> macro expansion is future work (it pairs naturally with the Phase 2
> assembler). Hand-written, macro-free kernels — such as everything in
> `go-libraries` — are analysed in full.
### Phase 2 — standalone assembler · *done*
Assembly without the Go toolchain in the loop.
- **`gasm asm`:** a standalone assembler that turns a `.s` file into machine
code directly — pure Go, no `go build`, no external toolchain. Useful for
fast iteration, for environments without a full Go installation, and as the
execution substrate that Phases 3 and 4 build on.
Done so far:
- An amd64 (x86-64) **instruction encoder** — REX/ModR-M/SIB/displacement/
immediate machinery and the scalar instruction set (MOV, the ALU group, TEST,
LEA, INC/DEC/NEG/NOT, shifts, IMUL and IMUL3, PUSH/POP, JMP/CALL/Jcc,
CMOVcc, SETcc, LZCNT/TZCNT, the sign/zero-extending moves — MOVBLZX and
friends, MOVLQSX — and CVTSL2SD/CVTSQ2SD), validated by round-tripping
every encoding through `golang.org/x/arch`'s decoder and byte-for-byte
against the Go assembler.
- An **assembler** that drives the parser's AST into the encoder with local-
label resolution — jumps start in the short (rel8) form and expand to rel32
when the displacement does not fit, and jump-to-jump chains are folded the
way the Go toolchain folds them — so `gasm asm <file>` emits machine code
for each `TEXT` function.
- **File-level assembly with static data** — `GLOBL`/`DATA` symbols are laid
out in a data section behind the code and references to them (`mask<>(SB)`)
are encoded RIP-relative with the displacement resolved within the image,
so the output is self-consistent and position-independent. References to
symbols no `GLOBL` in the file defines are recorded as relocations and
carried into the object-file output.
- **GOOBJ emission** — `gasm asm --format goobj -p <pkgpath>` writes the Go
toolchain's own object format (the one `cmd/link` consumes directly), so
gasm-assembled kernels drop into a `go build` without the Go assembler:
the functions as non-package symbols, `GLOBL` data, one `FuncInfo` per
function and the pc-value tables (`pcsp` with the real prologue/epilogue
stack deltas, `pcfile`, `pcline`, `pcinline`). Verified end-to-end by
swapping a gasm-emitted object into a `go build` in place of the
toolchain's, linking and running — bit-identical behaviour.
- **Object-file emission** — `gasm asm --format elf` / `--format macho`
writes a relocatable object (a `.text` and a `.data` section, a symbol
table — file-local `<>` symbols local, the rest global — and one
`R_X86_64_PC32` / `X86_64_RELOC_SIGNED` relocation per static-symbol
reference) that links with the system toolchain: external references
resolve against undefined symbols, file-local ones against the data
section. Verified end-to-end by linking a gasm-emitted object with a C
driver and running it.
- **`FP`/`SP` frame mapping** — the pseudo-registers are translated onto the
hardware stack pointer (`x+N(FP)` → `(N+8)(SP)` for a zero frame, `(N+frame+
16)(SP)` with a frame pointer; locals via `x-N(SP)`), and the Go-style
prologue/epilogue is generated for functions with a frame. The output is
**byte-identical to the Go assembler** for these cases (verified against
`go tool objdump`).
- **SIMD (VEX / AVX2)** — the VEX prefix machinery (2-byte C5 and 3-byte C4)
with XMM/YMM vector registers, validated by round-trip decoding **and**
byte-for-byte against the Go assembler's machine code, across eight operand
forms: the three-operand NDS form (VPADDD/Q, VPSUBD/Q, VPXOR, VPOR, VPAND/N,
VPCMPEQD, VPCMPGTQ, VPUNPCK*, VPMULLD, VPMULDQ, VPSHUFB, VPACKSSDW,
VPERMD), the two-operand reg/rm form (VPMOVSXWD/DQ, VPMOVZXDQ,
VPBROADCASTD/Q, VPMOVMSKB, VMOVMSKPS, VCVTDQ2PD), the immediate-shift and
variable-count shifts (VPSLLD/Q, VPSRAD, VPSRLD/Q with an immediate or an
XMM/memory count), the immediate shuffle (VPSHUFD, VPERMQ), the
three-operand-plus-immediate form (VSHUFPD, VPERM2I128, VINSERTI128), the
lane extract (VEXTRACTI128, VEXTRACTF128), the direction-sensitive moves
(VMOVDQU, VMOVUPD, VMOVD, VMOVQ, VMOVSD), the no-operand VZEROUPPER, and
the floating-point set: the packed double arithmetic
(VADDPD/VSUBPD/VMULPD/VDIVPD/VMINPD/VMAXPD), the unpacks
(VUNPCKHPD/VUNPCKLPD), the scalar SD and SS operations, VMOVDDUP, the
width-changing conversions (VCVTDQ2PS, VCVTPS2PD, VCVTDQ2PD and the
VCVTPD2DQX/Y / VCVTTPD2DQX/Y spellings, whose VEX.L follows the wider
source) and VFMADD231PD.
- **SIMD (EVEX / AVX-512)** — the four-byte EVEX prefix with the 5-bit
register fields (Z0–Z31, X/Y 16–31), opmask registers (K0–K7 as operands
and mask destinations, KMOVW, KTESTW) and the compressed disp8×N
displacement, covering every AVX-512 instruction the go-flac kernels use:
VPXORD/Q, VPADDD, VPSUBD/Q, VPUNPCK*DQ, VPMULLD/Q, VPERMD, VPSLLD/VPSRAD/
VPSRAQ, VALIGND, VPCMPEQD (with a K destination), VMOVDQU32, VMOVUPD,
VCVTQQ2PD, VPMOVSXDQ, the narrowing stores VPMOVDW/VPMOVQD, the lane
extracts VEXTRACTI64X4/VEXTRACTF64X4, VFMADD231PD, VADDPD, VMULPD,
VMOVDQU64 and the broadcasts VPBROADCASTD/Q from a GPR or memory, plus the
wider AVX-512 F/BW integer set (VPADDB/W, VPSUBB/W, VPANDD/Q/ND/NQ, VPMULLW,
VPMIN*/VPMAX* for B/W/D/Q elements, signed and unsigned, VPAVGB/W, the variable
shifts VPSLLV*/VPSRLV*/VPSRAV*, VMOVDQU8/16), the common floating-point
and conversion set (the packed double and single arithmetic
VADD/VSUB/VMUL/VDIV/VMIN/VMAX PD and PS, the scalar SD/SS operations —
whose EVEX forms exist for masked and zeroing use — the VUNPCK{L,H}PD
unpacks, VMOVDDUP, VMOVSLDUP/VMOVSHDUP and the VCVT* conversions), and
the wider AVX-512 set: ternary logic (VPTERNLOGD/Q), lane shuffles,
inserts and extracts (VSHUF{F,I}{32,64}X{2,4}, the VINSERT*/VEXTRACT*
{F,I}{32,64}X{2,4,8} family, VPALIGNR), compares with an opmask
destination (VCMPPD/PS/SD/SS), the permutes (VPERMB/W, VPERMI2/T2
D/Q/PD), the wider integer families (VPMADDWD/UBSW, VPMULHUW, VPACK*,
VPABS*, the VPROL*/VPROR* rotates and the word shifts), expand/compress
(VEXPAND*/VCOMPRESS*, VPEXPAND*/VPCOMPRESS*), the broadcasts
(VPBROADCASTB/W, VBROADCASTSS/SD), the opmask instructions (KAND/KOR/
KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST, KMOVQ), the aligned moves
(VMOVAPS/APD, VMOVDQA32/64, VMOVSS) and the remaining extending and
narrowing moves, the floating-point helper and conversion tail
(VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*, VSCALEF*, VRNDSCALE*,
VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with a K destination, and the
VCVT* conversions VCVTQQ2PS, VCVTPD2QQ/UQQ, VCVTPS2QQ, VCVTUDQ2PD/PS,
VCVTPH2PS, VCVTPS2PH), and gather/scatter with VSIB addressing
(VGATHER*/VPGATHER* in both the VEX mask-register spelling and the EVEX
K-mask spelling — where the L'L field follows the VSIB index — plus
VSCATTER*/VPSCATTER*). The EVEX mnemonic suffixes the Go assembler
accepts are honoured: rounding modes (.RN_SAE, .RD_SAE, .RU_SAE,
.RZ_SAE), suppress-all-exceptions (.SAE) and memory broadcast (.BCST,
with the element-sized disp8×N), each combinable with the .Z zeroing
suffix. Masking is supported the way
Go writes it — an explicit K1–K7 operand placed among the operands, and a
`.Z` mnemonic suffix for zeroing.
- **Legacy SSE moves** — `MOVOU`/`MOVO` (the Plan 9 names for MOVDQU/MOVDQA),
`MOVUPS`/`MOVAPS`/`MOVUPD`/`MOVAPD` and the scalar `MOVSD`/`MOVSS`.
- **Both go-flac kernels — all 17 AVX2 and all 10 AVX-512 functions —
assemble byte-identically to the Go toolchain's machine code**; the only
differing bytes are the displacements of the static-constant loads, which
the Go linker fills at link time and gasm resolves within its own image
(verified to reach the right constant bytes).
Remaining for Phase 2:
- External (cross-package) symbol references in the GOOBJ output —
**deferred** with a recorded decision and three options; see
[`docs/DEFERRED.md`](docs/DEFERRED.md). Single-package objects (no
cross-package references) work today, which covers the production
kernels. With that item deferred, the amd64 instruction set — scalar,
VEX/AVX2 and the full EVEX/AVX-512 set including GPR-interchanging
conversions — is complete, and RISC-V encoding (RV64IMAFDC + RVC)
including ELF and GOOBJ emission is complete.
### Phase 3 — dynamic analysis · *done*
Run the code and check what static analysis cannot. The oracle is the
portable Go implementation every kernel is derived from.
- **`gasm verify`:**
- **JIT execution substrate** — *done.* Assemble the kernel, map it into
executable memory (`syscall.Mmap`, W^X) and call it through an ABI0
trampoline; pure Go, no cgo, no external toolchain.
- **Differential testing** — *done.* The JIT-assembled kernel is fuzzed
against a portable Go reference, comparing the result bit-for-bit;
the automated form of the project's bit-identical contract.
- **Runtime ABI checks** — *done.* The ABI-checking trampoline sets
sentinels in BP and R14, verifies they survive the call, and fills a
128-byte red-zone canary below SP.
- **Coverage / basic-block profiling** — *done.* Static block enumeration
from the assembler's label map plus multi-input path-diversity
measurement: how many observationally distinct execution paths a
test corpus exercises.
### Phase 4 — debugger · *done*
- **`gasm debug`:** single-step a GAsm function, inspect registers (including
YMM vector registers), set breakpoints on labels, allocate and fill named
buffers, and hex-dump memory — the interactive counterpart to Phase 3's
execution substrate.
- **MVP** — *done.* ptrace-based debuggee subprocess (PTRACE_TRACEME +
LockOSThread), entry breakpoint (auto-run to function start),
single-step, register inspection (GPR + YMM/XMM via PTRACE_GETFPREGS),
label resolution, breakpoint management via `/proc/pid/mem`, named
buffer allocation with pattern filling (`--buf`), and an interactive REPL.
- **Remaining:** disassembly at PC (x86asm decode), memory-write support,
watchpoints, source-line mapping, and multi-platform support
(FreeBSD/macOS ptrace variants).
### Phase 5 — the other architectures · *in progress*
- **RISC-V encoding — done.** RV64IMAFDC instruction set, RVC compression,
MOV pseudo-instruction, SB/global symbols (AUIPC pairs), ELF64 and GOOBJ
emission, and ground-truth verification against `go tool asm`.
- **Remaining:** arm64 and loong64 encoding, plus the same encode-and-verify
treatment for each (instruction tables already generated from the toolchain).
**FreeBSD support is planned for a future release.**
## Principles
@@ -277,8 +62,9 @@ portable Go implementation every kernel is derived from.
dependency, `golang.org/x/arch`, is used **only in tests** to validate the
instruction encoder by round-trip decoding — it is never linked into the
`gasm` binary.
- **Portable.** Builds and runs on Linux, FreeBSD and macOS; amd64 and arm64
hosts. Latest stable Go only.
- **Linux-only.** Runs natively on amd64, arm64, riscv64 and loong64 Linux
hosts; the release matrix cross-compiles the same four targets. Latest
stable Go only.
- **No vendor lock-in.** The integration surface is the Language Server
Protocol and a command-line interface — both open standards. No cloud
service, no proprietary API, no dependence on any one editor's internals.
@@ -297,17 +83,16 @@ portable Go implementation every kernel is derived from.
| `arch` | amd64, arm64, riscv64 and loong64 register files and instruction tables. |
| `lint` | Conservative static checks. |
| `format` | A canonical formatter — `gofmt` for assembly. |
| `asm` | The standalone assembler: amd64 and RISC-V encoders, linker, object-file emitters (ELF, Mach-O, GOOBJ). |
| `verify` | JIT execution substrate for dynamic analysis, combined ABI+fuzz differential testing (Phase 3). |
| `debug` | Interactive ptrace debugger with GPR/YMM register display and named buffer allocation (Phase 4). |
| `asm` | The standalone assembler: amd64, RISC-V and LoongArch encoders, linker, object-file emitters (ELF, GOOBJ). |
| `verify` | JIT execution substrate for dynamic analysis, combined ABI+fuzz differential testing. |
| `debug` | Interactive ptrace debugger with GPR/YMM register display and named buffer allocation. |
| `lsp` | Language Server Protocol server. |
| `cmd/gasm` | The `gasm` binary tying it all together. |
| `_gen` | The generator that rebuilds the instruction tables from the Go toolchain. |
See [`docs/ARCHITECTURE.md`](docs/ARCHITECTURE.md) for the design rationale and
data flow, [`docs/ZED.md`](docs/ZED.md) for the editor-integration story, and
[`docs/DEFERRED.md`](docs/DEFERRED.md) for design decisions deliberately
postponed (with the analysis needed to pick them up again).
data flow, and [`docs/DECISIONS.md`](docs/DECISIONS.md) for design decisions
deliberately postponed (with the analysis needed to pick them up again).
## Quick start
@@ -341,8 +126,8 @@ gasm profile k.s # show basic-block structure
```
See [CONTRIBUTING.md](CONTRIBUTING.md) for the full development workflow,
[docs/cli.md](docs/cli.md) for the command reference, and
[docs/development.md](docs/development.md) for setup and recipes.
[docs/CLI.md](docs/CLI.md) for the command reference, and
[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for setup and recipes.
## Editor integration
@@ -353,10 +138,7 @@ binary and associate it with `.s` files. Syntax highlighting is delivered as
infers the target architecture from the file-name suffix
(`_amd64.s` / `_arm64.s` / `_riscv64.s` / `_loong64.s`).
Zed users should read [`docs/ZED.md`](docs/ZED.md): Zed's native highlighting
engine (Tree-sitter, C/WASM) cannot be fed from pure Go, so the pure-Go path
into Zed is the language server and its semantic tokens.
## License
## Licence
BSD-3-Clause — the same licence as Go itself. See [`LICENSE`](LICENSE).
BSD-3-Clause — see [LICENSE](LICENSE).
Copyright © 2026 [Petr Balvín](https://petrbalvin.org)
+8 -2
View File
@@ -86,7 +86,10 @@ func filterCommon(names []string) []string {
func writeCommon(names []string) error {
var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n\n")
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
b.WriteString("// SPDX-License-Identifier: BSD-3-Clause\n\n")
b.WriteString("package arch\n\n")
b.WriteString("// commonGeneratedInstrs is the set of opcodes shared by every architecture\n")
b.WriteString("// (RET, JMP, NOP, CALL, TEXT, FUNCDATA, PCDATA, …).\n")
@@ -154,7 +157,10 @@ func stringLit(elt ast.Expr) string {
func writeGen(arch, sub string, names []string) error {
var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n\n")
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
b.WriteString("// SPDX-License-Identifier: BSD-3-Clause\n\n")
b.WriteString("package arch\n\n")
b.WriteString("// " + arch + "GeneratedInstrs is the complete set of " + arch +
" mnemonics accepted by\n// Go's Plan 9 assembler.\n")
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/util.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
+181
View File
@@ -0,0 +1,181 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGOObjectAARCH64Structure checks the basic structure of the emitted
// AArch64 GOOBJ: the preamble, the magic, the block offsets and the
// non-package symbol definitions.
func TestGOObjectAARCH64Structure(t *testing.T) {
f, errs := parser.Parse("k_arm64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.GOObjectAARCH64("testpkg", "k_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
// Check preamble.
idx := strings.Index(string(obj), "\n!\n")
if idx < 0 {
t.Fatal("missing preamble separator")
}
preamble := string(obj[:idx])
if !strings.HasPrefix(preamble, "go object") {
t.Errorf("preamble = %q, want 'go object ...'", preamble)
}
// Check GOOBJ magic.
magicIdx := idx + 3
if magicIdx+8 > len(obj) || string(obj[magicIdx:magicIdx+8]) != "\x00go120ld" {
t.Error("missing GOOBJ magic")
}
// The object should contain the function's code.
if len(img.Code) == 0 {
t.Error("no code generated")
}
}
// TestGOObjectAARCH64Link does an end-to-end link test: it cross-compiles a
// Go program for arm64, substitutes the gasm-produced object into the package
// archive, re-links with cmd/link, and verifies the symbol appears in the
// resulting binary. The binary is not executed (no arm64 host or qemu).
// Skipped when no Go toolchain is available.
func TestGOObjectAARCH64Link(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
asmSrc := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
`
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
mainSrc := `package main
func add(a, b int64) int64
func main() {
if add(20, 22) != 42 {
panic("bad add")
}
}
`
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module a64link\n\ngo 1.21\n"), 0o644); err != nil {
t.Fatal(err)
}
// Capture the cross build (GOARCH=arm64): the package archive and the
// link line.
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
build.Env = append(os.Environ(), "GOARCH=arm64")
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var pkgArch, work, linkLine, asmObj string
for _, line := range strings.Split(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
}
defer os.RemoveAll(work)
// Expand $WORK in the object path.
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
// Read the toolchain-produced object and assemble the same source with gasm.
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse("main_arm64.s", string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
gasmObj, err := img.GOObjectAARCH64("a64link", "main_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
// Replace the toolchain-produced object with gasm's.
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
// Re-link.
if linkLine == "" {
t.Skip("could not find link command in build log")
}
// Expand $WORK in the link command.
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
// Verify the binary exists and contains the symbol.
binPath := filepath.Join(dir, "prog")
if _, err := os.Stat(binPath); err != nil {
t.Fatalf("binary not found: %v", err)
}
binData, err := os.ReadFile(binPath)
if err != nil {
t.Fatalf("read binary: %v", err)
}
if !strings.Contains(string(binData), "add") && !strings.Contains(string(binData), "a64link") {
t.Error("binary does not contain expected symbol")
}
}
File diff suppressed because it is too large Load Diff
+764
View File
@@ -0,0 +1,764 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// arm64 (AArch64) instruction encoding.
//
// The encoder is data-driven: each mnemonic maps to an instruction format and
// an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own arm64
// backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm`
// exactly — the ground-truth oracle for the verify suite.
//
// All AArch64 instructions are 32 bits, little-endian. The formats used here
// (per the ARM Architecture Reference Manual):
//
// DP-shifted-reg sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
// DP-immediate sf<<31 | op<<30 | S<<29 | 0x11<<24 | imm12<<10 | Rn<<5 | Rd
// Logical-imm sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd
// Move-wide sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd
// Load/store size<<30 | 0x7<<27 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt
// LDST-unscaled size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<5 | Rt (actually imm9<<12 | Rn<<5 | Rt)
// LDST-pair opc<<30 | 0x5<<27 | V<<26 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt
// Branch-imm 0<<31 | 0x5<<26 | imm26 (B)
// Branch-imm 1<<31 | 0x5<<26 | imm26 (BL)
// Branch-cond 0x2A<<25 | imm19<<5 | cond (B.cond)
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
// R0–R30 (integer), F0–F31 (floating point), and the ABI aliases the
// runtime's assembly uses. Returns -1 for an unrecognised name.
func arm64RegNum(name string) int {
switch name {
case "R0":
return 0
case "R1":
return 1
case "R2":
return 2
case "R3":
return 3
case "R4":
return 4
case "R5":
return 5
case "R6":
return 6
case "R7":
return 7
case "R8":
return 8
case "R9":
return 9
case "R10":
return 10
case "R11":
return 11
case "R12":
return 12
case "R13":
return 13
case "R14":
return 14
case "R15":
return 15
case "R16":
return 16
case "R17":
return 17
case "R18":
return 18
case "R19":
return 19
case "R20":
return 20
case "R21":
return 21
case "R22":
return 22
case "R23":
return 23
case "R24":
return 24
case "R25":
return 25
case "R26", "REGCTXT", "CTXT":
return 26
case "R27", "REGTMP", "TMP":
return 27
case "R28", "REGG", "g":
return 28
case "R29", "FP":
return 29
case "R30", "LR", "LINK":
return 30
case "R31", "ZR":
return 31
case "SP":
return 31 // SP and ZR share encoding 31; context determines meaning
}
// F0–F31.
if len(name) >= 1 && name[0] == 'F' {
n := 0
for i := 1; i < len(name); i++ {
if name[i] < '0' || name[i] > '9' {
return -1
}
n = n*10 + int(name[i]-'0')
}
if n <= 31 {
return n
}
}
return -1
}
// arm64IsSP reports whether a register operand is the stack pointer (R31/SP),
// which uses a different encoding path for some instructions.
func arm64IsSP(name string) bool {
return name == "SP"
}
// ---- format helpers ----
// a64wordLE encodes a uint32 as 4 little-endian bytes.
func a64wordLE(w uint32) []byte {
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
}
// a64WordsLE concatenates one or more instruction words as little-endian bytes.
func a64WordsLE(ws ...uint32) []byte {
var out []byte
for _, w := range ws {
out = append(out, a64wordLE(w)...)
}
return out
}
// ---- data-processing (shifted register) ----
// a64DPSR encodes a data-processing (shifted register) instruction:
// sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd.
func a64DPSR(sf, op, S, shift, rm, imm6, rn, rd uint32) uint32 {
return sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | rm<<16 | imm6<<10 | rn<<5 | rd
}
// ---- data-processing (immediate) ----
// a64AddSub encodes an ADD/SUB (immediate) instruction:
// sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | Rn<<5 | Rd.
func a64AddSub(sf, op, S, sh, imm12, rn, rd uint32) uint32 {
return sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | rn<<5 | rd
}
// ---- logical (immediate) ----
// a64LogicalImm encodes a logical (immediate) instruction:
// sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd.
func a64LogicalImm(sf, opc, N, immr, imms, rn, rd uint32) uint32 {
return sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | rn<<5 | rd
}
// ---- move wide ----
// a64MoveWide encodes a MOVZ/MOVK/MOVN instruction:
// sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd.
func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 {
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
}
// ---- load/store (unsigned immediate, scaled) ----
// a64LSU encodes a load/store register (unsigned immediate, scaled):
// size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt.
// (0x39<<24 encodes bits 29:24 = 111001, the scaled unsigned offset form.)
func a64LSU(size, V, opc, imm12, rn, rt uint32) uint32 {
return size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | rn<<5 | rt
}
// ---- load/store (unscaled immediate) ----
// a64LSUnscaled encodes a load/store register (unscaled immediate, 9-bit signed):
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<12 | Rn<<5 | Rt.
// Note: the 0<<24 distinguishes unscaled from the pre/post-index forms.
func a64LSUnscaled(size, V, opc int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | uint32(opc)<<22 |
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// ---- load/store pair ----
// a64LSP encodes a load/store pair instruction (signed offset):
// opc<<30 | 0x5<<27 | V<<26 | 2<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
// opc: 0=32-bit, 1=reserved, 2=64-bit. V: 0=integer, 1=FP/SIMD.
// L: 0=store, 1=load. imm7 is the signed scaled offset (÷8 for 64-bit pairs).
func a64LSP(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
return opc<<30 | 5<<27 | V<<26 | 2<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
}
// ---- load/store pair (pre-index) ----
// a64LSPPre encodes a load/store pair (pre-index):
// opc<<30 | 0x5<<27 | V<<26 | 0b11<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
func a64LSPPre(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
return opc<<30 | 5<<27 | V<<26 | 3<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
}
// ---- load/store pair (post-index) ----
// a64LSPPost encodes a load/store pair (post-index):
// opc<<30 | 0x5<<27 | V<<26 | 0b01<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
func a64LSPPost(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
return opc<<30 | 5<<27 | V<<26 | 1<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
}
// ---- pre-index load/store ----
// a64LSPreIndex encodes a load/store register (pre-index):
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func a64LSPreIndex(size, V, opc uint32, imm9 int32, rn, rt uint32) uint32 {
return size<<30 | 7<<27 | V<<26 | opc<<22 | 3<<10 | (uint32(imm9)&0x1FF)<<12 | rn<<5 | rt
}
// ---- post-index load/store ----
// a64LSPostIndex encodes a load/store register (post-index):
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func a64LSPostIndex(size, V, opc uint32, imm9 int32, rn, rt uint32) uint32 {
return size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<10 | (uint32(imm9)&0x1FF)<<12 | rn<<5 | rt
}
// ---- branches ----
// a64Branch encodes an unconditional branch (B/BL):
// op<<31 | 0x5<<26 | imm26.
func a64Branch(op uint32, imm26 int32) uint32 {
return op<<31 | 5<<26 | (uint32(imm26) & 0x03FFFFFF)
}
// a64BranchCond encodes a conditional branch (B.cond):
// 0x2A<<25 | imm19<<5 | cond.
func a64BranchCond(imm19 int32, cond uint32) uint32 {
return 0x2A<<25 | (uint32(imm19)&0x7FFFF)<<5 | cond&0xF
}
// a64UncondBranch encodes an unconditional branch register (BR/BLR/RET):
// 0x6B<<25 | opc<<21 | 0x1F<<16 | Rn<<5 | Rd.
// opc: 0=BR, 1=BLR, 2=RET. For RET, Rn defaults to LR(30).
func a64UncondBranch(opc, rn, rd uint32) uint32 {
return 0x6B<<25 | opc<<21 | 0x1F<<16 | rn<<5 | rd
}
// ---- ADR/ADRP ----
// a64ADR encodes an ADR instruction (p=0) or ADRP instruction (p=1):
// p<<31 | immlo<<29 | 0x10<<24 | immhi<<5 | Rd.
func a64ADR(p uint32, immhi int32, immlo uint32, rd uint32) uint32 {
return p<<31 | immlo<<29 | 0x10<<24 | (uint32(immhi)&0x7FFFF)<<5 | rd
}
// ---- EXTR ----
// a64EXTR encodes an EXTR instruction:
// sf<<31 | 0<<29 | 0x27<<23 | N<<22 | 0<<21 | Rm<<16 | imms<<10 | Rn<<5 | Rd.
func a64EXTR(sf, N, rm, imms, rn, rd uint32) uint32 {
return sf<<31 | 0x27<<23 | N<<22 | rm<<16 | imms<<10 | rn<<5 | rd
}
// ---- system ----
// a64NOP encodes a NOP: 0xd503201f.
const a64NOP uint32 = 0xd503201f
// a64BRK encodes a BRK instruction: 0xd4200000 | imm16<<5.
func a64BRK(imm16 uint32) uint32 {
return 0xd4200000 | imm16<<5
}
// ---- condition codes ----
const (
a64CondEQ = 0x0
a64CondNE = 0x1
a64CondCS = 0x2
a64CondHS = 0x2
a64CondCC = 0x3
a64CondLO = 0x3
a64CondMI = 0x4
a64CondPL = 0x5
a64CondVS = 0x6
a64CondVC = 0x7
a64CondHI = 0x8
a64CondLS = 0x9
a64CondGE = 0xa
a64CondLT = 0xb
a64CondGT = 0xc
a64CondLE = 0xd
a64CondAL = 0xe
a64CondNV = 0xf
)
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
var arm64CondMap = map[string]uint32{
"EQ": a64CondEQ,
"NE": a64CondNE,
"CS": a64CondCS,
"HS": a64CondHS,
"CC": a64CondCC,
"LO": a64CondLO,
"MI": a64CondMI,
"PL": a64CondPL,
"VS": a64CondVS,
"VC": a64CondVC,
"HI": a64CondHI,
"LS": a64CondLS,
"GE": a64CondGE,
"LT": a64CondLT,
"GT": a64CondGT,
"LE": a64CondLE,
}
// ---- instruction format tags ----
type a64Format uint8
const (
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
a64FDPIR // data-processing (immediate): ADD/SUB $imm
a64FLogImm // logical (immediate): AND/ORR/EOR $imm
a64FMovWide // move wide: MOVZ, MOVN, MOVK
a64FLSU // load/store (unsigned immediate, scaled)
a64FLSUnscaled // load/store (unscaled immediate)
a64FLSPair // load/store pair
a64FBranch // unconditional branch (B/BL)
a64FBranchCond // conditional branch (B.cond)
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
a64FADR // ADR/ADRP
a64FEXTR // EXTR
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
a64FSystem // system: NOP, BRK, etc.
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
a64FFMovGR // FMOV between GP and FP registers
a64FCRC32 // CRC32
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
)
// a64Enc is one instruction's encoding: its bit layout (format) and the
// opcode constant, positioned at its exact bit range.
type a64Enc struct {
format a64Format
op uint32 // the pre-positioned opcode bits
size int // 4 for most, 8 for DP-imm with shift, etc.
}
// a64InstrTable maps AArch64 mnemonics (as the Go assembler spells them) to
// their encoding. The base integer, memory, floating-point and SIMD
// instruction sets are covered.
var a64InstrTable = map[string]a64Enc{}
func init() {
// ---- data-processing (shifted register) ----
// Format: sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
dpsr := map[string]uint32{
// Add/Sub
"ADD": 1<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=1, op=0, S=0 (64-bit default)
"ADDW": 0<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=0
"ADDS": 1<<31 | 0<<30 | 1<<29 | 0x0b<<24,
"ADDSW": 0<<31 | 0<<30 | 1<<29 | 0x0b<<24,
"SUB": 1<<31 | 1<<30 | 0<<29 | 0x0b<<24,
"SUBW": 0<<31 | 1<<30 | 0<<29 | 0x0b<<24,
"SUBS": 1<<31 | 1<<30 | 1<<29 | 0x0b<<24,
"SUBSW": 0<<31 | 1<<30 | 1<<29 | 0x0b<<24,
// Logical (shifted register)
"AND": 1<<31 | 0<<29 | 0x0a<<24,
"ANDW": 0<<31 | 0<<29 | 0x0a<<24,
"BIC": 1<<31 | 0<<29 | 0x0a<<24 | 1<<21,
"BICW": 0<<31 | 0<<29 | 0x0a<<24 | 1<<21,
"ORR": 1<<31 | 1<<29 | 0x0a<<24,
"ORRW": 0<<31 | 1<<29 | 0x0a<<24,
"ORN": 1<<31 | 1<<29 | 0x0a<<24 | 1<<21,
"ORNW": 0<<31 | 1<<29 | 0x0a<<24 | 1<<21,
"EOR": 1<<31 | 2<<29 | 0x0a<<24,
"EORW": 0<<31 | 2<<29 | 0x0a<<24,
"EON": 1<<31 | 2<<29 | 0x0a<<24 | 1<<21,
"EONW": 0<<31 | 2<<29 | 0x0a<<24 | 1<<21,
"ANDS": 1<<31 | 3<<29 | 0x0a<<24,
"ANDSW": 0<<31 | 3<<29 | 0x0a<<24,
"BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21,
"BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21,
// Shift
"LSL": 1<<31 | 0<<29 | 0x0a<<24, // alias of UBFM
"LSLW": 0<<31 | 0<<29 | 0x0a<<24,
"LSR": 1<<31 | 0<<29 | 0x0a<<24,
"LSRW": 0<<31 | 0<<29 | 0x0a<<24,
"ASR": 1<<31 | 0<<29 | 0x0a<<24,
"ASRW": 0<<31 | 0<<29 | 0x0a<<24,
"ROR": 1<<31 | 0<<29 | 0x0a<<24,
"RORW": 0<<31 | 0<<29 | 0x0a<<24,
// Multiply
"MADD": 1<<31 | 0<<29 | 0x1b<<24 | 0<<21,
"MADDW": 0<<31 | 0<<29 | 0x1b<<24 | 0<<21,
"MSUB": 1<<31 | 0<<29 | 0x1b<<24 | 1<<21,
"MSUBW": 0<<31 | 0<<29 | 0x1b<<24 | 1<<21,
// Divide
"SDIV": 1<<31 | 0<<29 | 0x0d<<24,
"SDIVW": 0<<31 | 0<<29 | 0x0d<<24,
"UDIV": 1<<31 | 0<<29 | 0x0d<<24 | 1<<10,
"UDIVW": 0<<31 | 0<<29 | 0x0d<<24 | 1<<10,
// CRC
"CRC32B": 0<<31 | 0<<29 | 0x1b<<24 | 4<<10,
"CRC32H": 0<<31 | 0<<29 | 0x1b<<24 | 5<<10,
"CRC32W": 0<<31 | 0<<29 | 0x1b<<24 | 6<<10,
"CRC32X": 1<<31 | 0<<29 | 0x1b<<24 | 7<<10,
// Conditional select
"CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10,
"CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10,
"CSINC": 1<<31 | 0<<29 | 0x1d<<24 | 1<<10,
"CSINCW": 0<<31 | 0<<29 | 0x1d<<24 | 1<<10,
"CSINV": 1<<31 | 0<<29 | 0x1d<<24 | 2<<10,
"CSINVW": 0<<31 | 0<<29 | 0x1d<<24 | 2<<10,
"CSNEG": 1<<31 | 0<<29 | 0x1d<<24 | 3<<10,
"CSNEGW": 0<<31 | 0<<29 | 0x1d<<24 | 3<<10,
}
for m, op := range dpsr {
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
}
// Aliases that map to the same encoding as their target.
a64InstrTable["CMP"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
a64InstrTable["CMPW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBSW"]}
a64InstrTable["CMN"] = a64Enc{format: a64FDPSR, op: dpsr["ADDS"]}
a64InstrTable["CMNW"] = a64Enc{format: a64FDPSR, op: dpsr["ADDSW"]}
a64InstrTable["TST"] = a64Enc{format: a64FDPSR, op: dpsr["ANDS"]}
a64InstrTable["TSTW"] = a64Enc{format: a64FDPSR, op: dpsr["ANDSW"]}
a64InstrTable["NEG"] = a64Enc{format: a64FDPSR, op: dpsr["SUB"]}
a64InstrTable["NEGW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBW"]}
a64InstrTable["NEGS"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
a64InstrTable["MVN"] = a64Enc{format: a64FDPSR, op: dpsr["ORN"]}
a64InstrTable["MVNW"] = a64Enc{format: a64FDPSR, op: dpsr["ORNW"]}
a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]}
a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]}
// ---- data-processing (immediate) ----
// ADD/SUB $imm, Rn, Rd
a64InstrTable["ADDImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 0<<29 | 0x11<<24}
a64InstrTable["ADDWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 0<<30 | 0<<29 | 0x11<<24}
a64InstrTable["SUBImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 0<<29 | 0x11<<24}
a64InstrTable["SUBWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 1<<30 | 0<<29 | 0x11<<24}
a64InstrTable["ADDSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 1<<29 | 0x11<<24}
a64InstrTable["SUBSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 1<<29 | 0x11<<24}
// ---- move wide ----
// MOVZ/MOVN/MOVK
a64InstrTable["MOVZ"] = a64Enc{format: a64FMovWide, op: 1<<31 | 2<<29 | 0x25<<23}
a64InstrTable["MOVZW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 2<<29 | 0x25<<23}
a64InstrTable["MOVN"] = a64Enc{format: a64FMovWide, op: 1<<31 | 0<<29 | 0x25<<23}
a64InstrTable["MOVNW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 0<<29 | 0x25<<23}
a64InstrTable["MOVK"] = a64Enc{format: a64FMovWide, op: 1<<31 | 3<<29 | 0x25<<23}
a64InstrTable["MOVKW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 3<<29 | 0x25<<23}
// ---- ADR/ADRP ----
a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0}
a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1}
// ---- load/store (unsigned immediate) ----
a64InstrTable["MOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<22} // LDR 64-bit
a64InstrTable["MOVWU"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<22} // LDR 32-bit unsigned
a64InstrTable["MOVHU"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 1<<22} // LDRH unsigned
a64InstrTable["MOVBU"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 1<<22} // LDRB unsigned
a64InstrTable["MOVW"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 2<<22} // LDRSW (signed 32→64)
a64InstrTable["MOVH"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 2<<22} // LDRSH (signed half)
a64InstrTable["MOVB"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 2<<22} // LDRSB (signed byte)
a64InstrTable["FMOVS"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 32-bit FP
a64InstrTable["FMOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 64-bit FP
// Store opcodes (load ^ (1<<22)):
// STR 64-bit: size=3, V=0, opc=00 → 3<<30 | 7<<27 | 0<<22
// STR 32-bit: size=2, V=0, opc=00 → 2<<30 | 7<<27 | 0<<22
// STRH: size=1, V=0, opc=00 → 1<<30 | 7<<27 | 0<<22
// STRB: size=0, V=0, opc=00 → 0<<30 | 7<<27 | 0<<22
// ---- branches ----
a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26}
a64InstrTable["BL"] = a64Enc{format: a64FBranch, op: 1<<31 | 5<<26}
// Conditional branches.
condBranches := map[string]uint32{
"BEQ": 0x0, "BNE": 0x1, "BCS": 0x2, "BHS": 0x2,
"BCC": 0x3, "BLO": 0x3, "BMI": 0x4, "BPL": 0x5,
"BVS": 0x6, "BVC": 0x7, "BHI": 0x8, "BLS": 0x9,
"BGE": 0xa, "BLT": 0xb, "BGT": 0xc, "BLE": 0xd,
}
for name, cond := range condBranches {
a64InstrTable[name] = a64Enc{format: a64FBranchCond, op: 0x2A<<25 | cond}
}
// Unconditional branch register (BR/BLR/RET).
a64InstrTable["BR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 0<<21}
a64InstrTable["BLR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 1<<21}
a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21}
// ---- system ----
a64InstrTable["NOP"] = a64Enc{format: a64FSystem, op: a64NOP}
a64InstrTable["NOOP"] = a64Enc{format: a64FSystem, op: a64NOP}
a64InstrTable["BRK"] = a64Enc{format: a64FSystem, op: 0xd4200000}
a64InstrTable["UNDEF"] = a64Enc{format: a64FSystem, op: a64BRK(0)}
// ---- EXTR ----
a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22}
a64InstrTable["EXTRW"] = a64Enc{format: a64FEXTR, op: 0<<31 | 0x27<<23 | 0<<22}
// ---- bitfield ----
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
a64InstrTable["UBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
a64InstrTable["BFI"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
// ---- FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL ----
fp3 := map[string]uint32{
"FADDS": 0x1e202800, "FADDD": 0x1e602800,
"FSUBS": 0x1e203800, "FSUBD": 0x1e603800,
"FMULS": 0x1e200800, "FMULD": 0x1e600800,
"FDIVS": 0x1e201800, "FDIVD": 0x1e601800,
"FMAXS": 0x1e204800, "FMAXD": 0x1e604800,
"FMINS": 0x1e205800, "FMIND": 0x1e605800,
"FMAXNMS": 0x1e206800, "FMAXNMD": 0x1e606800,
"FMINNMS": 0x1e207800, "FMINNMD": 0x1e607800,
"FNMULS": 0x1e208800, "FNMULD": 0x1e608800,
}
for m, op := range fp3 {
a64InstrTable[m] = a64Enc{format: a64FFP3, op: op}
}
// ---- FP unary (Rn, Rd): FMOV reg-reg, FABS, FNEG, FSQRT, FCVT, FRINT* ----
fp1 := map[string]uint32{
"FMOVS": 0x1e204000, "FMOVD": 0x1e604000,
"FABSS": 0x1e20c000, "FABSD": 0x1e60c000,
"FNEGS": 0x1e214000, "FNEGD": 0x1e614000,
"FSQRTS": 0x1e21c000, "FSQRTD": 0x1e61c000,
"FCVTSD": 0x1e22c000, "FCVTDS": 0x1e624000,
"FRINTNS": 0x1e244000, "FRINTND": 0x1e644000,
"FRINTPS": 0x1e24c000, "FRINTPD": 0x1e64c000,
"FRINTMS": 0x1e254000, "FRINTMD": 0x1e654000,
"FRINTZS": 0x1e25c000, "FRINTZD": 0x1e65c000,
"FRINTAS": 0x1e264000, "FRINTAD": 0x1e664000,
"FRINTXS": 0x1e274000, "FRINTXD": 0x1e674000,
"FRINTIS": 0x1e27c000, "FRINTID": 0x1e67c000,
}
for m, op := range fp1 {
a64InstrTable[m] = a64Enc{format: a64FFPUnary, op: op}
}
// ---- FP 4-operand FMA (Ra, Rm, Rn, Rd) ----
fp4 := map[string]uint32{
"FMADDS": 0x1f000000, "FMADDD": 0x1f400000,
"FMSUBS": 0x1f008000, "FMSUBD": 0x1f408000,
"FNMADDS": 0x1f200000, "FNMADDD": 0x1f600000,
"FNMSUBS": 0x1f208000, "FNMSUBD": 0x1f608000,
}
for m, op := range fp4 {
a64InstrTable[m] = a64Enc{format: a64FFP4, op: op}
}
// ---- FP compare (Rm, Rn or #0, Rn) ----
fpcmp := map[string]uint32{
"FCMPS": 0x1e202000, "FCMPD": 0x1e602000,
"FCMPES": 0x1e202010, "FCMPED": 0x1e602010,
}
for m, op := range fpcmp {
a64InstrTable[m] = a64Enc{format: a64FFPCmp, op: op}
}
// ---- FP conditional compare (Rm, Rn, #nzcv, cond) ----
fpccmp := map[string]uint32{
"FCCMPS": 0x1e200400, "FCCMPD": 0x1e600400,
"FCCMPES": 0x1e200410, "FCCMPED": 0x1e600410,
}
for m, op := range fpccmp {
a64InstrTable[m] = a64Enc{format: a64FFPCCmp, op: op}
}
// ---- FP conditional select (Rm, Rn, Rd, cond) ----
a64InstrTable["FCSELS"] = a64Enc{format: a64FFPSel, op: 0x1e200c00}
a64InstrTable["FCSELD"] = a64Enc{format: a64FFPSel, op: 0x1e600c00}
// ---- FP ↔ integer conversion ----
fpcvt := map[string]uint32{
"FCVTZSD": 0x9e780000, "FCVTZSDW": 0x1e780000,
"FCVTZSS": 0x9e380000, "FCVTZSSW": 0x1e380000,
"FCVTZUD": 0x9e790000, "FCVTZUDW": 0x1e790000,
"FCVTZUS": 0x9e390000, "FCVTZUSW": 0x1e390000,
"SCVTFD": 0x9e620000, "SCVTFS": 0x9e220000,
"SCVTFWD": 0x1e620000, "SCVTFWS": 0x1e220000,
"UCVTFD": 0x9e630000, "UCVTFS": 0x9e230000,
"UCVTFWD": 0x1e630000, "UCVTFWS": 0x1e230000,
}
for m, op := range fpcvt {
a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op}
}
// ---- FMOV between GP and FP registers ----
a64InstrTable["FMOVGR"] = a64Enc{format: a64FFMovGR, op: 0x1e260000} // placeholder, actual encoding depends on direction
// ---- conditional select: CSEL, CSINC, CSINV, CSNEG ----
csel := map[string]uint32{
"CSEL": 0x9a800000, "CSELW": 0x1a800000,
"CSINC": 0x9a800400, "CSINCW": 0x1a800400,
"CSINV": 0xda800000, "CSINVW": 0x5a800000,
"CSNEG": 0xda800400, "CSNEGW": 0x5a800400,
}
for m, op := range csel {
a64InstrTable[m] = a64Enc{format: a64FCSEL, op: op}
}
// Aliases
a64InstrTable["CSET"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
a64InstrTable["CSETW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
a64InstrTable["CSETM"] = a64Enc{format: a64FCSEL, op: 0xda800000}
a64InstrTable["CSETMW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
a64InstrTable["CINC"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
a64InstrTable["CINCW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
a64InstrTable["CINV"] = a64Enc{format: a64FCSEL, op: 0xda800000}
a64InstrTable["CINVW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
a64InstrTable["CNEG"] = a64Enc{format: a64FCSEL, op: 0xda800400}
a64InstrTable["CNEGW"] = a64Enc{format: a64FCSEL, op: 0x5a800400}
// ---- CRC32 ----
crc32 := map[string]uint32{
"CRC32B": 0x1ac04000, "CRC32H": 0x1ac04400,
"CRC32W": 0x1ac04800, "CRC32X": 0x9ac04c00,
"CRC32CB": 0x1ac05000, "CRC32CH": 0x1ac05400,
"CRC32CW": 0x1ac05800, "CRC32CX": 0x9ac05c00,
}
for m, op := range crc32 {
a64InstrTable[m] = a64Enc{format: a64FCRC32, op: op}
}
// ---- exclusive load/store ----
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
a64InstrTable["LDXRW"] = a64Enc{format: a64FExcl, op: 0x885f7c00}
a64InstrTable["LDAXR"] = a64Enc{format: a64FExcl, op: 0xc85ffc00}
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
a64InstrTable["STXRW"] = a64Enc{format: a64FExcl, op: 0x88007c00}
a64InstrTable["STLXR"] = a64Enc{format: a64FExcl, op: 0xc800fc00}
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
// ---- LSE atomics ----
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["LDADDW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["LDADDB"] = a64Enc{format: a64FLSE, op: 0<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["LDADDH"] = a64Enc{format: a64FLSE, op: 1<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["CASD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x45<<21 | 0x1f<<10}
a64InstrTable["CASW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x45<<21 | 0x1f<<10}
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
// ---- SIMD basics ----
a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400}
a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400}
a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00}
}
// ---- load/store helper tables ----
// a64LSType describes the load/store parameters for a MOV width mnemonic.
type a64LSType struct {
size int // 0=byte, 1=half, 2=word, 3=dword
V int // 0=integer, 1=FP
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
}
// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters.
// For loads, opc selects signed vs unsigned; for stores, we flip the opc.
var a64LoadTable = map[string]a64LSType{
"MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
}
// a64StoreOpc returns the store opc for a given load type.
// For integer: store opc = 00 (the load opc bits cleared).
// For FP: store opc = 00 (same pattern).
func a64StoreOpc(t a64LSType) int {
if t.V == 1 {
return 0 // FP store
}
return 0 // integer store
}
// a64MovRegTable maps register-to-register MOV mnemonic expansions.
// The Go toolchain encodes MOV Rn, Rd as ORR Rn, ZR, Rd.
var a64MovRegTable = map[string]uint32{
"MOVD": 1<<31 | 1<<29 | 0x0a<<24, // ORR 64-bit
"MOVW": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVB": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit (byte move)
"MOVBU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVH": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVHU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVWU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
}
// arm64RegClass discriminates integer (R), floating-point (F) registers for
// the MOV pseudo-instruction.
type arm64RegClass int
const (
arm64ClsNone arm64RegClass = iota
arm64ClsGR
arm64ClsFP
)
// arm64RegClassOf reports the register class of a register operand name.
func arm64RegClassOf(name string) arm64RegClass {
switch {
case name == "":
return arm64ClsNone
case len(name) >= 1 && name[0] == 'F':
return arm64ClsFP
default:
return arm64ClsGR
}
}
// arm64Movcon returns the shift (in units of 16 bits) at which a non-zero
// 16-bit chunk of v sits, or -1 if v cannot be represented as a single
// MOVZ/MOVN immediate. This is the Go toolchain's movcon function.
func arm64Movcon(v int64) int {
for s := 0; s < 64; s += 16 {
if (uint64(v) &^ (uint64(0xFFFF) << uint(s))) == 0 {
return s
}
}
return -1
}
+574
View File
@@ -0,0 +1,574 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
func TestArm64LDRSTREncoding(t *testing.T) {
tests := []struct {
name string
got uint32
want uint32
}{
{"LDR X4, [SP, #56]", a64LSU(3, 0, 1, 7, 31, 4), 0xf9401fe4},
{"STR X4, [SP, #64]", a64LSU(3, 0, 0, 8, 31, 4), 0xf90023e4},
{"STR X5, [SP, #32]", a64LSU(3, 0, 0, 4, 31, 5), 0xf90013e5},
{"LDR X6, [SP, #32]", a64LSU(3, 0, 1, 4, 31, 6), 0xf94013e6},
}
for _, tt := range tests {
if tt.got != tt.want {
t.Errorf("%s: got %08x, want %08x", tt.name, tt.got, tt.want)
}
}
}
func TestArm64PrologueEncoding(t *testing.T) {
fi := arm64FrameInfo{autosize: 48, frame: 32, leaf: false}
pro := arm64Prologue(fi)
if len(pro) != 12 {
t.Fatalf("prologue length: got %d, want 12", len(pro))
}
expected := []uint32{0xf81d0ffe, 0xf81f83fd, 0xd10023fd}
for i, w := range leWords(pro) {
if w != expected[i] {
t.Errorf("prologue word %d: got %08x, want %08x", i, w, expected[i])
}
}
}
func TestArm64EpilogueSmallEncoding(t *testing.T) {
fi := arm64FrameInfo{autosize: 48, frame: 32, leaf: false}
ret := arm64Return(fi)
if len(ret) != 12 {
t.Fatalf("epilogue length: got %d, want 12", len(ret))
}
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #48; RET
expected := []uint32{0xf85f83fd, 0xf84307fe, 0xd65f03c0}
for i, w := range leWords(ret) {
if w != expected[i] {
t.Errorf("epilogue word %d: got %08x, want %08x", i, w, expected[i])
}
}
}
func TestArm64LargeFrameEncoding(t *testing.T) {
fi := arm64FrameInfo{autosize: 272, frame: 256, leaf: false}
pro := arm64Prologue(fi)
if len(pro) != 16 {
t.Fatalf("prologue length: got %d, want 16", len(pro))
}
expected := []uint32{0xd10443f4, 0xa93ffa9d, 0x9100029f, 0xd10023fd}
for i, w := range leWords(pro) {
if w != expected[i] {
t.Errorf("prologue word %d: got %08x, want %08x", i, w, expected[i])
}
}
epi := arm64Return(fi)
if len(epi) != 12 {
t.Fatalf("epilogue length: got %d, want 12", len(epi))
}
eexpected := []uint32{0xa97ffbfd, 0x910443ff, 0xd65f03c0}
for i, w := range leWords(epi) {
if w != eexpected[i] {
t.Errorf("epilogue word %d: got %08x, want %08x", i, w, eexpected[i])
}
}
}
func TestArm64NoFrame(t *testing.T) {
fi := arm64FrameInfo{autosize: 0, frame: 0, leaf: true}
pro := arm64Prologue(fi)
if len(pro) != 0 {
t.Errorf("no-frame prologue: got %d bytes, want 0", len(pro))
}
ret := arm64Return(fi)
if len(ret) != 4 {
t.Fatalf("no-frame return: got %d bytes, want 4", len(ret))
}
if leWord(ret) != 0xd65f03c0 {
t.Errorf("no-frame RET: got %08x, want d65f03c0", leWord(ret))
}
}
func TestArm64RegNum(t *testing.T) {
tests := []struct {
name string
want int
}{
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
{"F0", 0}, {"F4", 4}, {"F31", 31},
{"INVALID", -1}, {"X0", -1}, {"", -1},
}
for _, tt := range tests {
got := arm64RegNum(tt.name)
if got != tt.want {
t.Errorf("arm64RegNum(%q) = %d, want %d", tt.name, got, tt.want)
}
}
}
func TestArm64ComputeFrame(t *testing.T) {
src := "TEXT ·f(SB), NOSPLIT, $32-0\n\tADD\tR4, R5\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
fi := arm64ComputeFrame(f.Decls[0].(*ast.Text))
if fi.frame != 32 {
t.Errorf("frame: got %d, want 32", fi.frame)
}
if fi.autosize != 48 { // 32+8=40, aligned to48
t.Errorf("autosize: got %d, want 48", fi.autosize)
}
// ADD + RET with no CALL/BL → leaf
if !fi.leaf {
t.Error("expected leaf")
}
}
func TestArm64IsLeaf(t *testing.T) {
src := "TEXT ·f(SB), NOSPLIT, $0-0\n\tADD\tR4, R5\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if !arm64IsLeaf(f.Decls[0].(*ast.Text)) {
t.Error("expected leaf")
}
src2 := "TEXT ·f(SB), NOSPLIT, $0-0\n\tBL\tother(SB)\n\tRET\n"
f2, errs := parser.Parse("test_arm64.s", src2)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if arm64IsLeaf(f2.Decls[0].(*ast.Text)) {
t.Error("expected non-leaf")
}
}
func TestArm64Bitmask(t *testing.T) {
tests := []struct {
v uint64
sf int
N, immr, imms uint32
ok bool
}{
{1, 1, 1, 0, 0, true}, // single bit at pos 0
{2, 1, 1, 63, 0, true}, // single bit at pos 1 (immr = esize-1)
{0, 1, 0, 0, 0, false}, // zero is not a bitmask
{0xFFFFFFFFFFFFFFFF, 1, 0, 0, 0, false}, // all ones is not a bitmask
{0x5555555555555555, 1, 0, 0, 0x3E, true}, // alternating bits (esize=2, ones=1)
{0xFFFFFFFF00000000, 1, 1, 32, 31, true}, // upper 32 bits set (esize=64, ones=32)
}
for _, tt := range tests {
N, immr, imms, ok := arm64Bitmask(tt.v, tt.sf)
if ok != tt.ok {
t.Errorf("arm64Bitmask(%#x, %d): ok=%v, want %v", tt.v, tt.sf, ok, tt.ok)
continue
}
if ok && (N != tt.N || immr != tt.immr || imms != tt.imms) {
t.Errorf("arm64Bitmask(%#x, %d): N=%d immr=%d imms=%d, want N=%d immr=%d imms=%d",
tt.v, tt.sf, N, immr, imms, tt.N, tt.immr, tt.imms)
}
}
}
func TestArm64AssembleFile(t *testing.T) {
src := `#include "textflag.h"
TEXT ·simple(SB), NOSPLIT, $0-0
MOV R4, R5
ADD R4, R5, R6
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if len(img.Funcs) != 1 {
t.Fatalf("got %d funcs, want 1", len(img.Funcs))
}
fn := img.Funcs[0]
if fn.Name != "simple" {
t.Errorf("func name: got %q, want %q", fn.Name, "simple")
}
//3 instructions ×4 bytes =12
if fn.Size != 12 {
t.Errorf("func size: got %d, want 12", fn.Size)
}
}
func TestArm64AssembleFileWithFrame(t *testing.T) {
src := `#include "textflag.h"
TEXT ·framed(SB), NOSPLIT, $16-8
MOVD arg+0(FP), R4
ADD $1, R4, R4
MOVD R4, ret+0(FP)
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if len(img.Funcs) != 1 {
t.Fatalf("got %d funcs, want 1", len(img.Funcs))
}
fn := img.Funcs[0]
if fn.Frame != 16 {
t.Errorf("frame: got %d, want 16", fn.Frame)
}
// Prologue (3×4=12) + body (3×4=12) + RET epilogue (3×4=12) = 36
if fn.Size != 36 {
t.Errorf("func size: got %d, want 36", fn.Size)
}
}
func TestArm64AssembleFileWithBranches(t *testing.T) {
src := `#include "textflag.h"
TEXT ·branch(SB), NOSPLIT, $0-0
BEQ done
BNE skip
skip:
ADD R4, R5
done:
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
fn := img.Funcs[0]
if fn.Size != 16 {
t.Errorf("func size: got %d, want 16", fn.Size)
}
}
func TestArm64AssembleFileWithJumpChain(t *testing.T) {
src := `#include "textflag.h"
TEXT ·chain(SB), NOSPLIT, $0-0
BNE skip
ADD R4, R5
RET
skip:
B target
target:
ADD R6, R7
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
// BNE should be redirected past skip→target to target directly.
if img.Funcs[0].Size != 24 {
t.Errorf("func size: got %d, want 24", img.Funcs[0].Size)
}
}
func TestArm64AssembleErrors(t *testing.T) {
tests := []struct {
name string
src string
}{
{"unsupported", "TEXT ·f(SB), NOSPLIT, $0-0\n\tINVALID\tR4, R5\n\tRET\n"},
{"undefined label", "TEXT ·f(SB), NOSPLIT, $0-0\n\tB\tnosuch\n\tRET\n"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
f, errs := parser.Parse("test_arm64.s", tt.src)
if len(errs) > 0 {
return // parse error, that's fine
}
_, err := AssembleFileARM64(f)
if err == nil {
t.Error("expected error, got nil")
}
})
}
}
func TestArm64Movcon(t *testing.T) {
tests := []struct {
v int64
want int
}{
{0, 0}, // 0 fits at shift 0
{1, 0}, // single bit at shift 0
{0x10000, 16}, // single bit at shift 16
{0x100000000, 32}, // single bit at shift 32
{0xFF, 0}, // 0xFF fits at shift 0
{0x12345, -1}, // multiple chunks, not movcon
}
for _, tt := range tests {
got := arm64Movcon(tt.v)
if got != tt.want {
t.Errorf("arm64Movcon(%#x) = %d, want %d", tt.v, got, tt.want)
}
}
}
func TestArm64RegClassOf(t *testing.T) {
if arm64RegClassOf("R4") != arm64ClsGR {
t.Error("R4 should be GR")
}
if arm64RegClassOf("F4") != arm64ClsFP {
t.Error("F4 should be FP")
}
if arm64RegClassOf("") != arm64ClsNone {
t.Error("empty should be None")
}
}
func TestArm64ResolvePseudo(t *testing.T) {
fi := arm64FrameInfo{autosize: 48, frame: 32}
// FP: offset = sym.Offset + autosize +8
base, off := arm64ResolvePseudo(&ast.Symbol{Pseudo: "FP", Offset: 0}, fi)
if base != 31 || off != 56 {
t.Errorf("FP: base=%d off=%d, want 31, 56", base, off)
}
// SP: offset = sym.Offset + frame +8
base, off = arm64ResolvePseudo(&ast.Symbol{Pseudo: "SP", Offset: -8}, fi)
if base != 31 || off != 32 {
t.Errorf("SP: base=%d off=%d, want 31, 32", base, off)
}
// SB: unresolved
base, _ = arm64ResolvePseudo(&ast.Symbol{Pseudo: "SB"}, fi)
if base != -1 {
t.Errorf("SB: base=%d, want -1", base)
}
}
// TestArm64FPSel tests FP conditional select encoding.
func TestArm64FPSel(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
FCSELD GE, F10, F11, F12
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
// FCSELD should be 4 bytes + RET 4 bytes = 8
if img.Funcs[0].Size != 8 {
t.Errorf("size: got %d, want 8", img.Funcs[0].Size)
}
}
// TestArm64FPCvt tests FP conversion encoding.
func TestArm64FPCvt(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
FCVTZSD F4, R0
SCVTFD R4, F8
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 12 {
t.Errorf("size: got %d, want 12", img.Funcs[0].Size)
}
}
// TestArm64CSEL tests conditional select encoding.
func TestArm64CSEL(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
CSEL EQ, R0, R1, R2
CSET NE, R3
CINC GE, R4, R5
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 16 {
t.Errorf("size: got %d, want 16", img.Funcs[0].Size)
}
}
// TestArm64CRC32 tests CRC32 encoding.
func TestArm64CRC32(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
CRC32B R0, R2
CRC32W R6, R8
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 12 {
t.Errorf("size: got %d, want 12", img.Funcs[0].Size)
}
}
// TestArm64Bitfield tests bitfield/shift encoding.
func TestArm64Bitfield(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
ASR $4, R0, R1
LSL $12, R4, R5
EXTR $8, R0, R1, R2
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 16 {
t.Errorf("size: got %d, want 16", img.Funcs[0].Size)
}
}
// TestArm64SIMD tests SIMD encoding (via the instruction table).
func TestArm64SIMD(t *testing.T) {
// Verify SIMD instructions are in the table.
for _, mnem := range []string{"VADD", "VSUB", "VMUL"} {
if _, ok := a64InstrTable[mnem]; !ok {
t.Errorf("%s not in instruction table", mnem)
}
}
}
// TestArm64LoadImm64 tests 64-bit immediate loading.
func TestArm64LoadImm64(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
MOVD $0x123456789ABCDEF0, R0
MOVD $0, R1
MOVD $1, R2
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
// $0x123456789ABCDEF0 needs 4 MOVZ/MOVK instructions (16 bytes)
// $0 is 1 instruction (4 bytes)
// $1 is 1 bitmask instruction (4 bytes)
// RET is 1 instruction (4 bytes)
if img.Funcs[0].Size != 28 {
t.Errorf("size: got %d, want 28", img.Funcs[0].Size)
}
}
// TestArm64BranchCond tests conditional branch encoding.
func TestArm64BranchCond(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
BEQ done
BNE done
BGE done
BLT done
ADD R4, R5
done:
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
// 4 branches + 1 ADD + 1 RET = 24 bytes
if img.Funcs[0].Size != 24 {
t.Errorf("size: got %d, want 24", img.Funcs[0].Size)
}
}
// TestArm64Errors tests error paths.
func TestArm64Errors(t *testing.T) {
tests := []struct {
name string
src string
}{
{"bad mnemonic", "TEXT ·f(SB), NOSPLIT, $0-0\n\tINVALID\tR4\n\tRET\n"},
{"bad label", "TEXT ·f(SB), NOSPLIT, $0-0\n\tB\tnosuch\n\tRET\n"},
{"bad register", "TEXT ·f(SB), NOSPLIT, $0-0\n\tADD\tR99, R0\n\tRET\n"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
f, errs := parser.Parse("test_arm64.s", tt.src)
if len(errs) > 0 {
return
}
_, err := AssembleFileARM64(f)
if err == nil {
t.Error("expected error, got nil")
}
})
}
}
// leWord reads a little-endian uint32 from b.
func leWord(b []byte) uint32 {
return uint32(b[0]) | uint32(b[1])<<8 | uint32(b[2])<<16 | uint32(b[3])<<24
}
// leWords reads all little-endian uint32s from b.
func leWords(b []byte) []uint32 {
n := len(b) / 4
w := make([]uint32, n)
for i := range w {
w[i] = leWord(b[i*4:])
}
return w
}
+237
View File
@@ -0,0 +1,237 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// arm64 frame mapping, matching the Go toolchain's arm64 backend.
//
// Go's arm64 functions use R29 as the frame pointer (FP) and R30 as the link
// register (LR). R31 is the stack pointer (SP). FP and SP in the source
// are synthetic pseudo-registers resolved against the hardware SP and the
// frame size.
//
// The autosize is the real stack adjustment: the declared local frame plus
// 8 bytes for the saved link register, rounded up to a 16-byte multiple.
// The toolchain adds an "extrasize" to align: if autosize%16 == 8, add 8;
// if autosize%16 == 0, add 16.
//
// Prologue (autosize > 0, small frame ≤ 0xf0):
//
// MOVD.W LR, -autosize(SP) // pre-index: SP -= autosize, store LR at SP
// MOVD FP, -8(SP) // store FP at SP-8
// SUB $8, SP, FP // FP = SP - 8
//
// Prologue (autosize > 0, large frame > 0xf0):
//
// SUB $autosize, SP, R20 // R20 = SP - autosize
// STP (FP, LR), -8(R20) // store FP,LR at R20-8
// MOVD R20, SP // SP = R20
// SUB $8, SP, FP // FP = SP - 8
//
// Epilogue (non-leaf, small frame):
//
// ADD $autosize-8, SP, FP // restore FP
// ADD $autosize, SP, SP // deallocate frame
// MOVD -8(SP), FP // (actually the reverse of prologue)
// Actually:
// MOVD -8(SP), FP // load FP from SP-8
// MOVD.P autosize(SP), LR // post-index: load LR, SP += autosize
//
// Epilogue (non-leaf, large frame):
// ADD $autosize-8, SP, FP
// ADD $autosize, SP, SP
// Actually:
// LDP -8(SP), (FP, LR) // load FP,LR
// ADD $autosize, SP, SP // deallocate frame
//
// Epilogue (leaf with frame):
// ADD $autosize-8, SP, FP
// ADD $autosize, SP, SP
//
// RET always emits as BR LR (0xd65f03c0).
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
type arm64FrameInfo struct {
autosize int // the real SP adjustment (locals + saved LR + alignment)
frame int // the declared $framesize
args int // the declared -argsize
noSplit bool // the NOSPLIT flag
leaf bool // no call instructions in the body
}
// arm64ComputeFrame derives the frame layout for a TEXT function.
func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
fi := arm64FrameInfo{
frame: frameSize(t),
args: argsSize(t),
}
for _, f := range t.Flags {
if f == "NOSPLIT" {
fi.noSplit = true
}
}
fi.leaf = arm64IsLeaf(t)
if fi.frame != 0 || !fi.leaf {
fi.autosize = fi.frame + 8 // space for the saved LR
if fi.autosize%16 != 0 {
// The toolchain aligns to 16: if autosize%16 == 8, add 8;
// otherwise add whatever is needed.
fi.autosize += 16 - (fi.autosize % 16)
}
}
return fi
}
// arm64IsLeaf reports whether a function contains no call instructions
// (BL/CALL), matching the toolchain's LEAF mark.
func arm64IsLeaf(t *ast.Text) bool {
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "BL", "CALL":
return false
}
}
return true
}
// arm64Prologue returns the prologue bytes for an arm64 function.
func arm64Prologue(fi arm64FrameInfo) []byte {
if fi.autosize == 0 {
return nil
}
if fi.autosize <= 0xf0 {
// Small frame: MOVD.W LR, -autosize(SP); MOVD FP, -8(SP); SUB $8, SP, FP
return a64WordsLE(
arm64PreStoreImm(3, 0, int32(-fi.autosize), 31, 30), // STR.W LR, -autosize(SP) (pre-index store)
arm64UnscaledStore(3, 0, -8, 31, 29), // STUR FP, [SP, #-8]
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
)
}
// Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP
return a64WordsLE(
a64AddSub(1, 1, 0, 0, uint32(fi.autosize), 31, 20), // SUB $autosize, SP, R20
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
)
}
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
// deallocate the frame when present) followed by RET (BR LR).
func arm64Return(fi arm64FrameInfo) []byte {
var ws []uint32
if fi.autosize != 0 {
if fi.leaf {
// Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP
ws = append(ws,
a64AddSub(1, 0, 0, 0, uint32(fi.autosize-8), 31, 29), // ADD $autosize-8, SP, FP
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
)
} else if fi.autosize <= 0xf0 {
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize
ws = append(ws,
arm64UnscaledLoad(3, 0, -8, 31, 29), // LDR FP, [SP, #-8]
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
)
} else {
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
ws = append(ws,
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
)
}
}
// RET: BR LR (0xd65f03c0)
ws = append(ws, a64UncondBranch(2, 30, 0)) // opc=2(RET), Rn=LR(30), Rd=0
return a64WordsLE(ws...)
}
// arm64PrologueSpadjPC returns the function-relative byte offset where the
// prologue has finished decrementing SP (the delta becomes autosize).
func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
if fi.autosize == 0 {
return 0
}
if fi.autosize <= 0xf0 {
return 4 // MOVD.W instruction decrements SP
}
return 8 // SUB + STP + MOVD (3 instructions, SP updated at the MOVD)
}
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final RET instruction.
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
if fi.autosize == 0 {
return 0
}
if fi.leaf {
return 8 // ADD + ADD
}
if fi.autosize <= 0xf0 {
return 8 // LDR + LDR.P
}
return 8 // LDP + ADD
}
// arm64ResolvePseudo translates a pseudo-register memory reference into a
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
// x+N(SP) → (N + frame + 8)(SP). Returns base = -1 for an unresolvable
// reference (SB: static data, handled by the relocation path).
//
// The Go toolchain resolves all pseudo-register references against the
// hardware stack pointer (R31/SP): FP references add autosize+8 (the
// distance from SP after the prologue to the caller's argument area),
// SP references add frame+8 (the distance to the local area).
func arm64ResolvePseudo(sym *ast.Symbol, fi arm64FrameInfo) (base int, off int32) {
if sym == nil {
return -1, 0
}
switch sym.Pseudo {
case "FP":
return 31, int32(sym.Offset) + int32(fi.autosize) + 8
case "SP":
return 31, int32(sym.Offset) + int32(fi.frame) + 8
case "SB":
return -1, int32(sym.Offset)
}
return -1, 0
}
// arm64PreStoreImm encodes a pre-index store (STR with writeback):
// size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func arm64PreStoreImm(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 |
3<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// arm64UnscaledStore encodes an unscaled store (STUR):
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt.
func arm64UnscaledStore(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 |
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// arm64UnscaledLoad encodes an unscaled load (LDUR):
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt.
func arm64UnscaledLoad(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// arm64PostLoad encodes a post-index load (LDR with post-increment):
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
+224
View File
@@ -0,0 +1,224 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// AArch64 ELF64 relocatable object emission.
const (
emAARCH64 = 183 // EM_AARCH64
// AArch64 relocation types (the ELF psABI).
rArm64PrelPgHi21 = 275 // R_AARCH64_ADR_PREL_PG_HI21 (ADRP page)
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD/STR/LDR page offset)
)
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
// optional .rela.text.
func (img *Image) ELFAARCH64Object() ([]byte, error) {
le := binary.LittleEndian
const (
secText = 1
secData = 2
)
// Build symbol table.
var locals, globals []elfSym
for _, fn := range img.Funcs {
s := elfSym{
name: objectName(fn.Pkg, fn.Name),
info: sttFunc,
shndx: secText,
value: uint64(fn.Offset),
size: uint64(fn.Size),
}
if fn.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := elfSym{
name: objectName(d.Pkg, d.Name),
info: sttObject,
shndx: secData,
value: uint64(d.Offset),
size: uint64(d.Size),
}
if d.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, name := range img.Externals {
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
}
syms := []elfSym{
{},
{name: ".text", info: sttSection, shndx: secText},
{name: ".data", info: sttSection, shndx: secData},
}
syms = append(syms, locals...)
shInfo := len(syms)
syms = append(syms, globals...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Build relocations. Each SB reference is an ADRP pair:
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21
// ADD/LDR/STR → R_AARCH64_ADD_ABS_LO12_NC
type elfRela struct {
off uint64
typ uint32
sym int
addend int64
}
var relas []elfRela
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
typ := uint32(rArm64PrelPgHi21)
if r.Kind == RelArm64Addr && r.Off%4 == 4 {
// The second instruction in an ADRP pair uses ADD_ABS_LO12_NC.
typ = rArm64AddAbsLo12NC
}
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
typ: typ,
sym: idx,
addend: r.Addend - int64(r.After-r.Off),
})
}
}
// String tables.
stNames := newElfStrtab()
for _, s := range syms {
stNames.add(s.name)
}
stSections := newElfStrtab()
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout.
var out []byte
out = append(out, make([]byte, 64)...)
align := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
align(16)
textOff := len(out)
out = append(out, img.Code...)
align(16)
dataOff := len(out)
out = append(out, img.Data...)
align(8)
symtabOff := len(out)
for _, s := range syms {
var b [24]byte
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
b[4] = s.info
b[5] = 0
le.PutUint16(b[6:], s.shndx)
le.PutUint64(b[8:], s.value)
le.PutUint64(b[16:], s.size)
out = append(out, b[:]...)
}
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
if hasRela {
align(8)
relaOff = len(out)
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
align(8)
shoff := len(out)
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
var b [64]byte
le.PutUint32(b[0:], uint32(stSections.at(name)))
le.PutUint32(b[4:], uint32(typ))
le.PutUint64(b[8:], flags)
le.PutUint64(b[16:], 0)
le.PutUint64(b[24:], uint64(off))
le.PutUint64(b[32:], uint64(size))
le.PutUint32(b[40:], uint32(link))
le.PutUint32(b[44:], uint32(info))
le.PutUint64(b[48:], alignV)
le.PutUint64(b[56:], entsize)
out = append(out, b[:]...)
}
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// ELF header.
hdr := out[:64]
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
le.PutUint16(hdr[16:], etREL)
le.PutUint16(hdr[18:], emAARCH64)
le.PutUint32(hdr[20:], elfVersion)
le.PutUint64(hdr[24:], 0)
le.PutUint64(hdr[32:], 0)
le.PutUint64(hdr[40:], uint64(shoff))
le.PutUint32(hdr[48:], 0)
le.PutUint16(hdr[52:], 64)
le.PutUint16(hdr[54:], 0)
le.PutUint16(hdr[56:], 0)
le.PutUint16(hdr[58:], 64)
le.PutUint16(hdr[60:], uint16(nSections))
le.PutUint16(hdr[62:], uint16(secShstr))
return out, nil
}
+142
View File
@@ -0,0 +1,142 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
// relocatable object: sections, the symbol table (bindings, types, values,
// sizes) and the .rela.text relocation pair for the static-symbol load,
// parsed back with debug/elf.
func TestELFAARCH64Object(t *testing.T) {
f, errs := parser.Parse("k_arm64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
TEXT ·getanswer(SB), NOSPLIT, $0-8
MOVD answer<>(SB), R4
MOVD R4, ret+0(FP)
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.ELFAARCH64Object()
if err != nil {
t.Fatalf("ELFAARCH64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_AARCH64 {
t.Errorf("type/machine = %v/%v, want ET_REL/EM_AARCH64", ef.Type, ef.Machine)
}
text := ef.Section(".text")
data := ef.Section(".data")
if text == nil || data == nil {
t.Fatal("missing .text or .data section")
}
if text.Size == 0 {
t.Error(".text section is empty")
}
syms, err := ef.Symbols()
if err != nil {
t.Fatalf("symbols: %v", err)
}
foundAdd, foundGetanswer, foundAnswer := false, false, false
for _, s := range syms {
switch s.Name {
case "add":
foundAdd = true
if elf.SymType(s.Info&0xf) != elf.STT_FUNC || elf.SymBind(s.Info>>4) != elf.STB_GLOBAL {
t.Errorf("add: info=0x%02x, want STT_FUNC|STB_GLOBAL", s.Info)
}
case "getanswer":
foundGetanswer = true
if elf.SymType(s.Info&0xf) != elf.STT_FUNC || elf.SymBind(s.Info>>4) != elf.STB_GLOBAL {
t.Errorf("getanswer: info=0x%02x, want STT_FUNC|STB_GLOBAL", s.Info)
}
case "answer":
foundAnswer = true
if elf.SymType(s.Info&0xf) != elf.STT_OBJECT || elf.SymBind(s.Info>>4) != elf.STB_LOCAL {
t.Errorf("answer: info=0x%02x, want STT_OBJECT|STB_LOCAL", s.Info)
}
}
}
if !foundAdd {
t.Error("symbol 'add' not found")
}
if !foundGetanswer {
t.Error("symbol 'getanswer' not found")
}
if !foundAnswer {
t.Error("symbol 'answer' not found")
}
// Check that .rela.text exists (getanswer has SB reference).
relaText := ef.Section(".rela.text")
if relaText == nil {
t.Error("missing .rela.text section")
}
}
// TestELFAARCH64ObjectNoRelocations checks the ELF output when there are no
// static-symbol references (no .rela.text section).
func TestELFAARCH64ObjectNoRelocations(t *testing.T) {
f, errs := parser.Parse("k_arm64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.ELFAARCH64Object()
if err != nil {
t.Fatalf("ELFAARCH64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Section(".rela.text") != nil {
t.Error("unexpected .rela.text section when there are no relocations")
}
}
+223
View File
@@ -0,0 +1,223 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// LoongArch ELF64 relocatable object emission.
const (
emLOONGARCH = 258 // EM_LOONGARCH
// LoongArch relocation types (the ELF psABI).
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
)
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
// optional .rela.text.
func (img *Image) ELFLOONG64Object() ([]byte, error) {
le := binary.LittleEndian
const (
secText = 1
secData = 2
)
// Build symbol table.
var locals, globals []elfSym
for _, fn := range img.Funcs {
s := elfSym{
name: objectName(fn.Pkg, fn.Name),
info: sttFunc,
shndx: secText,
value: uint64(fn.Offset),
size: uint64(fn.Size),
}
if fn.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := elfSym{
name: objectName(d.Pkg, d.Name),
info: sttObject,
shndx: secData,
value: uint64(d.Offset),
size: uint64(d.Size),
}
if d.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, name := range img.Externals {
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
}
syms := []elfSym{
{},
{name: ".text", info: sttSection, shndx: secText},
{name: ".data", info: sttSection, shndx: secData},
}
syms = append(syms, locals...)
shInfo := len(syms)
syms = append(syms, globals...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Build relocations. Each SB reference is a pcalau12i pair:
// pcalau12i rd, 0 → R_LARCH_PCALA_HI20
// addi.d/ld/st → R_LARCH_PCALA_LO12
type elfRela struct {
off uint64
typ uint32
sym int
addend int64
}
var relas []elfRela
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
typ := uint32(rLarchPCALAHI20)
if r.Kind == RelLoong64AddrLo {
typ = rLarchPCALALO12
}
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
typ: typ,
sym: idx,
addend: r.Addend - int64(r.After-r.Off),
})
}
}
// String tables.
stNames := newElfStrtab()
for _, s := range syms {
stNames.add(s.name)
}
stSections := newElfStrtab()
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout.
var out []byte
out = append(out, make([]byte, 64)...)
align := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
align(16)
textOff := len(out)
out = append(out, img.Code...)
align(16)
dataOff := len(out)
out = append(out, img.Data...)
align(8)
symtabOff := len(out)
for _, s := range syms {
var b [24]byte
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
b[4] = s.info
b[5] = 0
le.PutUint16(b[6:], s.shndx)
le.PutUint64(b[8:], s.value)
le.PutUint64(b[16:], s.size)
out = append(out, b[:]...)
}
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
if hasRela {
align(8)
relaOff = len(out)
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
align(8)
shoff := len(out)
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
var b [64]byte
le.PutUint32(b[0:], uint32(stSections.at(name)))
le.PutUint32(b[4:], uint32(typ))
le.PutUint64(b[8:], flags)
le.PutUint64(b[16:], 0)
le.PutUint64(b[24:], uint64(off))
le.PutUint64(b[32:], uint64(size))
le.PutUint32(b[40:], uint32(link))
le.PutUint32(b[44:], uint32(info))
le.PutUint64(b[48:], alignV)
le.PutUint64(b[56:], entsize)
out = append(out, b[:]...)
}
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// ELF header.
hdr := out[:64]
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
le.PutUint16(hdr[16:], etREL)
le.PutUint16(hdr[18:], emLOONGARCH)
le.PutUint32(hdr[20:], elfVersion)
le.PutUint64(hdr[24:], 0)
le.PutUint64(hdr[32:], 0)
le.PutUint64(hdr[40:], uint64(shoff))
le.PutUint32(hdr[48:], 0)
le.PutUint16(hdr[52:], 64)
le.PutUint16(hdr[54:], 0)
le.PutUint16(hdr[56:], 0)
le.PutUint16(hdr[58:], 64)
le.PutUint16(hdr[60:], uint16(nSections))
le.PutUint16(hdr[62:], uint16(secShstr))
return out, nil
}
+200
View File
@@ -0,0 +1,200 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
// relocatable object: sections, the symbol table (bindings, types, values,
// sizes) and the .rela.text relocation pair for the static-symbol load,
// parsed back with debug/elf.
func TestELFLOONG64Object(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
TEXT ·getanswer(SB), NOSPLIT, $0-8
MOVV answer<>(SB), R4
MOVV R4, ret+0(FP)
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.ELFLOONG64Object()
if err != nil {
t.Fatalf("ELFLOONG64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_LOONGARCH {
t.Errorf("type/machine = %v/%v, want ET_REL/EM_LOONGARCH", ef.Type, ef.Machine)
}
text := ef.Section(".text")
data := ef.Section(".data")
if text == nil || data == nil {
t.Fatal("missing .text or .data section")
}
if text.Flags&elf.SHF_EXECINSTR == 0 || text.Flags&elf.SHF_ALLOC == 0 {
t.Errorf(".text flags = %v", text.Flags)
}
if data.Flags&elf.SHF_WRITE == 0 {
t.Errorf(".data flags = %v", data.Flags)
}
textData, err := text.Data()
if err != nil {
t.Fatal(err)
}
if !bytes.Equal(textData, img.Code) {
t.Errorf(".text contents differ from the image code")
}
dataData, err := data.Data()
if err != nil {
t.Fatal(err)
}
syms, err := ef.Symbols()
if err != nil {
t.Fatalf("symbols: %v", err)
}
byName := map[string]elf.Symbol{}
for _, s := range syms {
byName[s.Name] = s
}
wantSym := func(name string, bind elf.SymBind, typ elf.SymType, section elf.SectionIndex, size uint64) {
t.Helper()
s, ok := byName[name]
if !ok {
t.Errorf("symbol %q not found", name)
return
}
if elf.ST_BIND(s.Info) != bind || elf.ST_TYPE(s.Info) != typ {
t.Errorf("%s: bind/type = %v/%v, want %v/%v", name, elf.ST_BIND(s.Info), elf.ST_TYPE(s.Info), bind, typ)
}
if s.Section != section {
t.Errorf("%s: section = %v, want %v", name, s.Section, section)
}
if s.Size != size {
t.Errorf("%s: size = %d, want %d", name, s.Size, size)
}
}
if ef.Sections[1].Name != ".text" || ef.Sections[2].Name != ".data" {
t.Fatalf("section layout = %s, %s; want .text, .data", ef.Sections[1].Name, ef.Sections[2].Name)
}
textIdx := elf.SectionIndex(1)
dataIdx := elf.SectionIndex(2)
wantSym("add", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 20)
wantSym("getanswer", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 16)
wantSym("answer", elf.STB_LOCAL, elf.STT_OBJECT, dataIdx, 8)
// The data section carries 16-byte alignment padding; the answer
// symbol sits at its padded offset.
ans := byName["answer"]
if ans.Value+8 > uint64(len(dataData)) {
t.Fatalf("answer value %d outside .data (%d bytes)", ans.Value, len(dataData))
}
if got := dataData[ans.Value : ans.Value+8]; !bytes.Equal(got, []byte{42, 0, 0, 0, 0, 0, 0, 0}) {
t.Errorf("answer data = % x, want $42", got)
}
// Relocations: the static-symbol load is a pcalau12i+ld.d pair, so one
// R_LARCH_PCALA_HI20 and one R_LARCH_PCALA_LO12, both against the local
// data symbol. debug/elf does not surface rela entries, so read the
// section directly.
relaSec := ef.Section(".rela.text")
if relaSec == nil {
t.Fatal("missing .rela.text")
}
raw, err := relaSec.Data()
if err != nil {
t.Fatal(err)
}
if len(raw)%24 != 0 || len(raw)/24 != 2 {
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
}
le := binary.LittleEndian
for i := 0; i < 2; i++ {
e := raw[i*24 : (i+1)*24]
off := le.Uint64(e[0:])
info := le.Uint64(e[8:])
typ := info & 0xffffffff
sym := int(info >> 32)
if i == 0 && (typ != uint64(elf.R_LARCH_PCALA_HI20) || off != 20) {
t.Errorf("reloc %d: type %d off %d, want R_LARCH_PCALA_HI20 at 20", i, typ, off)
}
if i == 1 && (typ != uint64(elf.R_LARCH_PCALA_LO12) || off != 24) {
t.Errorf("reloc %d: type %d off %d, want R_LARCH_PCALA_LO12 at 24", i, typ, off)
}
if sym != 3 { // NULL, .text, .data, then the first local: answer
t.Errorf("reloc %d: symbol index %d, want 3 (answer)", i, sym)
}
}
}
// TestELFLOONG64ObjectNoRelocations checks a file with no static-symbol
// references emits a valid object without a .rela.text section.
func TestELFLOONG64ObjectNoRelocations(t *testing.T) {
f, errs := parser.Parse("n_loong64.s", `
#include "textflag.h"
TEXT ·nop(SB), NOSPLIT, $0
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.ELFLOONG64Object()
if err != nil {
t.Fatalf("ELFLOONG64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Section(".rela.text") != nil {
t.Error("unexpected .rela.text section")
}
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
found := false
for _, s := range syms {
if s.Name == "nop" && elf.ST_TYPE(s.Info) == elf.STT_FUNC {
found = true
}
}
if !found {
t.Error("function symbol nop not found")
}
}
+22 -18
View File
@@ -15,6 +15,7 @@ const (
// RISC-V relocation types.
rRISCV32 = 1
rRISCVJAL = 17 // R_RISCV_JAL
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
@@ -79,11 +80,12 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
symIdx[s.name] = i
}
// Build relocations. Each SB reference produces a pair:
// AUIPC rd, 0 → R_RISCV_PCREL_HI20
// ADDI/LD/SD → R_RISCV_PCREL_LO12_I or _S
// For now we record them as individual entries; at link time
// the linker must pair HI20 with its matching LO12.
// Build relocations. Each SB reference is an AUIPC + second-instruction
// pair carrying a single relocation kind; the ELF writer expands it into
// the R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I/S pair the psABI expects.
// The HI20 carries the symbol addend; the LO12 addend is zero, matching
// cmd/link's own ELF conversion (the LO12 resolves against the HI20's
// AUIPC location).
type elfRela struct {
off uint64
typ uint32
@@ -97,22 +99,24 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
// Determine relocation type from the relocation kind.
typ := uint32(rRISCVPCRELHI20) // default: AUIPC
switch r.Kind {
case RelPCRelLO12:
typ = rRISCVPCRELLO12I
case RelPCRelLO12S:
typ = rRISCVPCRELLO12S
case RelRISCVPCRELIType:
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: idx, addend: 0},
)
case RelRISCVPCRELSType:
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: idx, addend: 0},
)
case RelRISCVJal:
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVJAL, sym: idx, addend: r.Addend})
case RelPCRelAbs:
typ = rRISCV32
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCV32, sym: idx, addend: r.Addend})
default:
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
}
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
typ: typ,
sym: idx,
addend: r.Addend - int64(r.After-r.Off),
})
}
}
+253 -86
View File
@@ -10,6 +10,7 @@ import (
"os"
"os/exec"
"path/filepath"
"strings"
"sync"
)
@@ -22,9 +23,15 @@ import (
//
// The object carries what the linker requires of an assembly object: the
// functions (non-package symbols, as cmd/asm emits them), the GLOBL data,
// one FuncInfo per function, and the pc-value tables (pcsp, pcfile,
// pcline, pcinline). DWARF and the implicit funcdata symbols are omitted;
// the linker fills their defaults.
// one FuncInfo per function, the per-function DWARF symbols (the
// .debug_line program and the subprogram DIE, which the linker's DWARF
// pass reads verbatim), and the pc-value tables (pcsp, pcfile, pcline,
// pcinline). The implicit funcdata symbols are omitted; the linker fills
// their defaults.
//
// emitGOObject is architecture-agnostic; the per-architecture GOObject*
// methods supply the toolchain preamble, the MinLC (pc-value delta unit)
// and the relocation-type mapping for code relocations.
// GOOBJ block indices (cmd/internal/goobj).
const (
@@ -51,9 +58,11 @@ const (
// Symbol kinds used by assembly objects (cmd/internal/objabi).
const (
kindSTEXT = 1
kindSRODATA = 3
kindSDATA = 7
kindSTEXT = 1
kindSRODATA = 3
kindSDATA = 7
kindSDWARFFCN = 14
kindSDWARFLINES = 20
)
// Symbol flags (cmd/internal/goobj).
@@ -66,11 +75,13 @@ const (
// Aux entry types (cmd/internal/goobj).
const (
auxFuncInfo = 1
auxPcsp = 7
auxPcfile = 8
auxPcline = 9
auxPcinline = 10
auxFuncInfo = 1
auxDwarfInfo = 3
auxDwarfLines = 6
auxPcsp = 7
auxPcfile = 8
auxPcline = 9
auxPcinline = 10
)
// FuncInfo flags (internal/abi).
@@ -80,7 +91,49 @@ const (
)
// Relocation types (cmd/internal/objabi).
const relocPCRel = 14
// R_PCREL and R_ADDR are stable across Go versions.
const (
relocPCRel = 14 // R_PCREL
relocAddr = 1 // R_ADDR
)
// relocDWTXTADDRU4 returns the R_DWTXTADDR_U4 relocation type for the
// installed Go toolchain. The value shifted between Go 1.26 (103) and
// Go 1.27 (106) because new LoongArch relocations were inserted before it.
func relocDWTXTADDRU4() uint16 {
if isGo127OrLater() {
return 106
}
return 103
}
var (
goVersionOnce sync.Once
goVersionGT26 bool
)
// isGo127OrLater reports whether the installed Go toolchain is 1.27 or later.
func isGo127OrLater() bool {
goVersionOnce.Do(func() {
goBin, err := exec.LookPath("go")
if err != nil {
return
}
out, err := exec.Command(goBin, "version").Output()
if err != nil {
return
}
// "go version go1.27rc1 linux/amd64"
s := string(out)
for _, prefix := range []string{"go version go1.27", "go version go1.28", "go version go1.29", "go version go2."} {
if strings.Contains(s, prefix) {
goVersionGT26 = true
return
}
}
})
return goVersionGT26
}
// Special package indices for symbol references.
const (
@@ -110,6 +163,13 @@ func (s goSym) append(b []byte, strOff map[string]uint32) []byte {
return binary.LittleEndian.AppendUint32(b, s.align)
}
// dwarfRelocSet attaches emitter-generated relocations (the DWARF
// lines/info symbols' address references) to a definition index.
type dwarfRelocSet struct {
si int
relocs []goobjReloc
}
// GOObject returns the image as a GOOBJ object file for the given package
// path (the linker qualifies the exported symbols with it, the way cmd/asm
// does with its -p flag). srcPath names the source file recorded in the
@@ -117,19 +177,90 @@ func (s goSym) append(b []byte, strOff map[string]uint32) []byte {
// captured from the installed go tool asm, so the output links with the
// toolchain it was produced on — exactly like a real assembly object.
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
if pkgPath == "" {
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
}
pre, err := toolchainObjectPreamble()
if err != nil {
return nil, err
}
// amd64: MinLC 1, R_PCREL for the code relocations.
return img.emitGOObject(pkgPath, srcPath, pre, 1, func(Reloc) (uint16, uint8) { return relocPCRel, 4 })
}
// The symbol tables. Package definitions: the GLOBL symbols, then one
// anonymous FuncInfo symbol per function. Non-package definitions: the
// pc-value tables and the functions themselves, as cmd/asm lays them
// out. defIdx maps a GLOBL's bare name to its definition index for the
// relocations; fnNpIdx maps a function to its non-package index.
// emitGOObject assembles the GOOBJ payload for any architecture. pre is
// the toolchain's object preamble; minLC is the architecture's minimum
// instruction length, the unit of the pc-value table deltas; relocField
// maps a code relocation to its objabi relocation type and the width of
// the instruction field the linker writes.
func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, relocField func(Reloc) (uint16, uint8)) ([]byte, error) {
if pkgPath == "" {
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
}
// The non-package definitions first — the DWARF symbols reference the
// functions by these indices: per function the four pc-value tables
// and the function itself, as cmd/asm lays them out.
type npSym struct {
sym goSym
data []byte
}
var nps []npSym
type pcRefs struct{ sp, file, line, inl int }
pcIdx := make([]pcRefs, len(img.Funcs))
fnNpIdx := make([]int, len(img.Funcs))
for i, fn := range img.Funcs {
tables := []struct {
data []byte
dst *int
}{
{pcspTable(fn, minLC), &pcIdx[i].sp},
{pcValueFlat(0, fn.Size, minLC), &pcIdx[i].file},
{pcValueFlat(int32(fn.Line), fn.Size, minLC), &pcIdx[i].line},
{pcValueFlat(-1, fn.Size, minLC), &pcIdx[i].inl},
}
for _, t := range tables {
*t.dst = len(nps)
nps = append(nps, npSym{
sym: goSym{typ: kindSRODATA, size: uint32(len(t.data)), align: 1},
data: t.data,
})
}
name := fn.Name
abi := uint16(0)
if fn.Static {
abi = symABIStatic
} else {
name = pkgPath + "." + name
}
flag := uint8(0)
if fn.NoSplit {
flag |= symFlagNoSplit
}
fnNpIdx[i] = len(nps)
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
// Only the amd64 encoder resolves file-local static symbols
// into a disp32 field at assemble time; GOOBJ must leave that
// field zero for the linker to fill. The RISC-V and LoongArch
// encoders emit zero immediates with a relocation instead, and
// their relocations cover whole AUIPC/pcalau12i pairs, so
// zeroing r.Off would erase the opcode/register bits the linker
// preserves when it patches only the immediate.
if r.Kind != RelPCRel32 {
continue
}
if r.Off >= 0 && r.Off+4 <= len(code) {
code[r.Off], code[r.Off+1], code[r.Off+2], code[r.Off+3] = 0, 0, 0, 0
}
}
nps = append(nps, npSym{
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
data: code,
})
}
// The package definitions: the GLOBL symbols, then, per function, the
// FuncInfo and the two DWARF symbols (the .debug_line program and the
// subprogram DIE). defIdx maps a GLOBL's bare name to its definition
// index for the code relocations.
var defs []goSym
var defData [][]byte
defIdx := map[string]int{}
@@ -155,74 +286,82 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
}
fnFiIdx := make([]int, len(img.Funcs))
for i := range img.Funcs {
data := marshalFuncInfo(img.Funcs[i])
fnLinesIdx := make([]int, len(img.Funcs))
fnDIEIdx := make([]int, len(img.Funcs))
var dwarfRelocs []dwarfRelocSet
for i, fn := range img.Funcs {
data := marshalFuncInfo(fn)
fnFiIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))})
defData = append(defData, data)
}
type npSym struct {
sym goSym
data []byte
}
var nps []npSym
type pcRefs struct{ sp, file, line, inl int }
pcIdx := make([]pcRefs, len(img.Funcs))
fnNpIdx := make([]int, len(img.Funcs))
for i, fn := range img.Funcs {
tables := []struct {
data []byte
dst *int
}{
{pcspTable(fn), &pcIdx[i].sp},
{pcValueFlat(0, fn.Size), &pcIdx[i].file},
{pcValueFlat(int32(fn.Line), fn.Size), &pcIdx[i].line},
{pcValueFlat(-1, fn.Size), &pcIdx[i].inl},
}
for _, t := range tables {
*t.dst = len(nps)
nps = append(nps, npSym{
sym: goSym{typ: kindSRODATA, size: uint32(len(t.data)), align: 1},
data: t.data,
})
}
name := fn.Name
abi := uint16(0)
if fn.Static {
abi = symABIStatic
} else {
if !fn.Static {
name = pkgPath + "." + name
}
flag := uint8(0)
if fn.NoSplit {
flag |= symFlagNoSplit
// The DWARF symbols: the .debug_line state-machine program and the
// subprogram DIE, both referencing the function by its non-package
// index (package definitions, like cmd/asm's).
lines, lrel := goobjDwarfLines(fn, fnNpIdx[i])
fnLinesIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDWARFLINES, size: uint32(len(lines))})
defData = append(defData, lines)
die, drel := goobjDwarfInfo(fn, name, fnNpIdx[i])
fnDIEIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDWARFFCN, size: uint32(len(die))})
defData = append(defData, die)
dwarfRelocs = append(dwarfRelocs,
dwarfRelocSet{si: fnLinesIdx[i], relocs: lrel},
dwarfRelocSet{si: fnDIEIdx[i], relocs: drel},
)
}
// Resolve external symbol references (cross-package). Build the
// package index table and determine each external symbol's SymIdx
// by reading the target package's export data.
var extPkgTable []string
var extPkgIdx map[string]int
var extSymIdx map[string]int
if len(img.Externals) > 0 {
var err error
extPkgTable, extPkgIdx, extSymIdx, err = resolveExternalSymbols(img.Externals)
if err != nil {
return nil, fmt.Errorf("GOOBJ emission: resolving external symbols: %w", err)
}
fnNpIdx[i] = len(nps)
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
// The linker writes the resolved displacement into the field;
// leave it zero, as cmd/asm's object does.
if r.Off >= 0 && r.Off+4 <= len(code) {
code[r.Off], code[r.Off+1], code[r.Off+2], code[r.Off+3] = 0, 0, 0, 0
}
}
nps = append(nps, npSym{
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
data: code,
})
}
// Relocations, per defined symbol in definition order (package defs,
// then non-package defs). Only file-local GLOBL references resolve;
// external symbols need the import machinery of a later increment.
// then non-package defs).
nsyms := len(defs) + len(nps)
symRelocs := make([][]byte, nsyms) // flat 23-byte records
for i, fn := range img.Funcs {
si := len(defs) + fnNpIdx[i]
for _, r := range fn.Relocs {
typ, size := relocField(r)
if r.External {
return nil, fmt.Errorf("GOOBJ emission: external symbol %q is not supported yet", r.Name)
// Split package-qualified name: "runtime·morestack" → runtime, morestack.
pkg, name := splitQualified(r.Name)
if pkg == "" {
return nil, fmt.Errorf("GOOBJ emission: external symbol %q has no package prefix", r.Name)
}
pIdx, ok := extPkgIdx[pkg]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: package %q not resolved", pkg)
}
sIdx, ok := extSymIdx[pkg+"·"+name]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: symbol %s·%s not resolved", pkg, name)
}
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size // field width
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], uint32(pIdx))
binary.LittleEndian.PutUint32(rec[19:], uint32(sIdx))
symRelocs[si] = append(symRelocs[si], rec[:]...)
continue
}
di, ok := defIdx[r.Name]
if !ok {
@@ -230,17 +369,30 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
}
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = 4 // field width
binary.LittleEndian.PutUint16(rec[5:], relocPCRel)
rec[4] = size // field width
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
symRelocs[si] = append(symRelocs[si], rec[:]...)
}
}
// The DWARF symbols' own relocations (the function address references).
for _, ds := range dwarfRelocs {
for _, r := range ds.relocs {
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(r.off))
rec[4] = r.siz
binary.LittleEndian.PutUint16(rec[5:], r.typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.add))
binary.LittleEndian.PutUint32(rec[15:], r.pkg)
binary.LittleEndian.PutUint32(rec[19:], r.sym)
symRelocs[ds.si] = append(symRelocs[ds.si], rec[:]...)
}
}
// Aux entries per function: FuncInfo, then the four pc tables.
// References into the non-package table use pkgIdxNone.
// Aux entries per function: FuncInfo, the DWARF symbols, then the four
// pc tables. References into the non-package table use pkgIdxNone.
symAux := make([][]byte, nsyms)
for i := range img.Funcs {
si := len(defs) + fnNpIdx[i]
@@ -252,10 +404,14 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
symAux[si] = append(symAux[si], rec[:]...)
}
aux(auxFuncInfo, pkgIdxSelf, uint32(fnFiIdx[i]))
aux(auxPcsp, pkgIdxNone, uint32(len(defs)+pcIdx[i].sp))
aux(auxPcfile, pkgIdxNone, uint32(len(defs)+pcIdx[i].file))
aux(auxPcline, pkgIdxNone, uint32(len(defs)+pcIdx[i].line))
aux(auxPcinline, pkgIdxNone, uint32(len(defs)+pcIdx[i].inl))
aux(auxDwarfInfo, pkgIdxSelf, uint32(fnDIEIdx[i]))
aux(auxDwarfLines, pkgIdxSelf, uint32(fnLinesIdx[i]))
// The pc-table references are 0-based within the non-package
// definitions; the loader adds the package-definition count itself.
aux(auxPcsp, pkgIdxNone, uint32(pcIdx[i].sp))
aux(auxPcfile, pkgIdxNone, uint32(pcIdx[i].file))
aux(auxPcline, pkgIdxNone, uint32(pcIdx[i].line))
aux(auxPcinline, pkgIdxNone, uint32(pcIdx[i].inl))
}
// The string table. Absolute offsets: it starts right after the
@@ -291,7 +447,16 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
for _, s := range nps {
npdefBlk = s.sym.append(npdefBlk, strOff)
}
pkgIdxBlk := stringRef(nil, "") // index 0: the dummy invalid package
// Package index table: index 0 is the dummy invalid package.
// External packages follow, in pkgIdx order.
for _, pkg := range extPkgTable {
addStr(pkg)
}
pkgIdxBlk := stringRef(nil, "") // index 0: dummy
for _, pkg := range extPkgTable {
pkgIdxBlk = stringRef(pkgIdxBlk, pkg)
}
fileBlk := stringRef(nil, srcPath)
var relocBlk, auxBlk, dataBlk []byte
@@ -374,19 +539,21 @@ func marshalFuncInfo(fn FuncLayout) []byte {
}
// pcValueFlat encodes a pc-value table holding v over the whole function.
func pcValueFlat(v int32, size int) []byte {
// The pc deltas are in MinLC units (the runtime scales them by the
// architecture's minimum instruction length).
func pcValueFlat(v int32, size, minLC int) []byte {
// The table is delta-encoded from an implicit value of -1: a varint
// value delta, an unsigned pc delta to the end, and a zero terminator.
out := binary.AppendVarint(nil, int64(v)+1)
out = binary.AppendUvarint(out, uint64(size))
out = binary.AppendUvarint(out, uint64(size/minLC))
return append(out, 0)
}
// pcspTable encodes the stack-adjustment table: the SP delta in effect at
// every pc, from the function's prologue and epilogue boundaries.
func pcspTable(fn FuncLayout) []byte {
func pcspTable(fn FuncLayout, minLC int) []byte {
if len(fn.Spadj) == 0 {
return pcValueFlat(0, fn.Size)
return pcValueFlat(0, fn.Size, minLC)
}
pts := make([]SpadjStep, 0, len(fn.Spadj)+1)
pts = append(pts, SpadjStep{PC: 0, Value: 0})
@@ -394,11 +561,11 @@ func pcspTable(fn FuncLayout) []byte {
out := binary.AppendVarint(nil, int64(pts[0].Value)+1)
cur, old := pts[0].PC, pts[0].Value
for _, p := range pts[1:] {
out = binary.AppendUvarint(out, uint64(p.PC-cur))
out = binary.AppendUvarint(out, uint64((p.PC-cur)/minLC))
out = binary.AppendVarint(out, int64(p.Value-old))
cur, old = p.PC, p.Value
}
out = binary.AppendUvarint(out, uint64(fn.Size-cur))
out = binary.AppendUvarint(out, uint64((fn.Size-cur)/minLC))
return append(out, 0)
}
+188
View File
@@ -0,0 +1,188 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
)
// This file generates the per-function DWARF symbols the linker's DWARF
// pass requires of an assembly object, byte-identical to what cmd/asm
// emits: the .debug_line state-machine program (SDWARFLINES) and the
// subprogram DIE (SDWARFFCN). The linker copies the DIE and line-program
// bytes verbatim into .debug_info and .debug_line, fixing up their
// relocations, so the formats here must match cmd/internal/dwarf's
// DW_ABRV_FUNCTION and generateDebugLinesSymbol exactly.
//
// DWARF5 is assumed throughout (the toolchain's default on Linux and the
// other non-Darwin targets gasm supports).
// Line-program parameters (cmd/internal/obj/dwarf.go).
const (
dwLineBase = -4
dwLineRange = 10
dwOpcodeBase = 11
dwPCRange = (255 - dwOpcodeBase) / dwLineRange
)
// goobjReloc is one relocation attached to an emitter-generated symbol
// (the DWARF lines/info symbols), in goobj's on-disk encoding fields.
type goobjReloc struct {
off int32
siz uint8
typ uint16
add int64
pkg uint32
sym uint32
}
// goobjDwarfLines builds the function's .debug_line state-machine program:
// an LNE_set_address extended opcode establishing the function's start
// address (carrying the R_ADDR relocation), one row per source line
// change across the function's instructions, an advance to the end of the
// function and an end-of-sequence opcode. The linker appends these bytes
// after the unit's line header, so they must start with the address and
// leave the state machine terminated.
func goobjDwarfLines(fn FuncLayout, fnNpIdx int) ([]byte, []goobjReloc) {
// Rows: the prologue, if any, then the body instructions (fn.Lines
// covers the body only). The first body offset > 0 means a prologue
// precedes it; the toolchain reports the prologue on the TEXT line.
pts := make([]LineEntry, 0, len(fn.Lines)+1)
if len(fn.Lines) == 0 || fn.Lines[0].Offset > 0 {
pts = append(pts, LineEntry{Offset: 0, Line: fn.Line})
}
pts = append(pts, fn.Lines...)
out := []byte{0, 9, 2, 0, 0, 0, 0, 0, 0, 0, 0} // LNE_set_address, address zeroed
relocs := []goobjReloc{{
off: 3, siz: 8, typ: relocAddr,
pkg: pkgIdxNone, sym: uint32(fnNpIdx),
}}
// The state machine starts at line 1, pc 0 (function-relative); the
// implicit initial pc is the function entry, so the first pc delta is
// against 0.
line := int64(1)
pc := uint64(0)
for _, p := range pts {
if p.Line == 0 || uint64(p.Offset) < pc {
continue
}
// Rows mark source-line changes only; the pc delta is measured from
// the previous row, not the previous instruction.
if int64(p.Line) == line {
continue
}
deltaPC := uint64(p.Offset) - pc
deltaLC := int64(p.Line) - line
out = dwPutPCLCDelta(out, deltaPC, deltaLC)
line, pc = int64(p.Line), uint64(p.Offset)
}
// Cover the rest of the function and close the sequence.
if end := uint64(fn.Size) - pc; end > 0 {
out = append(out, 2) // DW_LNS_advance_pc
out = binary.AppendUvarint(out, end)
}
out = append(out, 0, 1, 1) // LNE_end_sequence
return out, relocs
}
// dwPutPCLCDelta encodes one (pcDelta, lineDelta) step as the shortest
// special opcode plus any standard-opcode remainder, exactly like
// cmd/internal/obj's putpclcdelta.
func dwPutPCLCDelta(b []byte, deltaPC uint64, deltaLC int64) []byte {
opcode := dwSelectOpcode(deltaPC, deltaLC)
deltaPC -= uint64((opcode - dwOpcodeBase) / dwLineRange)
deltaLC -= (opcode-dwOpcodeBase)%dwLineRange + dwLineBase
// The remainder: standard opcodes first, then the special opcode
// (which emits the row).
if deltaPC != 0 {
switch {
case deltaPC <= uint64(dwPCRange):
opcode -= dwLineRange * int64(uint64(dwPCRange)-deltaPC)
b = append(b, 8) // DW_LNS_const_add_pc
case (1<<14) <= deltaPC && deltaPC < (1<<16):
b = append(b, 9) // DW_LNS_fixed_advance_pc
b = binary.LittleEndian.AppendUint16(b, uint16(deltaPC))
default:
b = append(b, 2) // DW_LNS_advance_pc
b = binary.AppendUvarint(b, deltaPC)
}
}
if deltaLC != 0 {
b = append(b, 3) // DW_LNS_advance_line
b = binary.AppendVarint(b, deltaLC)
}
return append(b, byte(opcode))
}
// dwSelectOpcode picks the special opcode for (deltaPC, deltaLC) per
// cmd/internal/obj's putpclcdelta selection logic.
func dwSelectOpcode(deltaPC uint64, deltaLC int64) int64 {
switch {
case deltaLC < dwLineBase:
if deltaPC >= uint64(dwPCRange) {
return dwOpcodeBase + dwLineRange*dwPCRange
}
return dwOpcodeBase + dwLineRange*int64(deltaPC)
case deltaLC < dwLineBase+dwLineRange:
if deltaPC >= uint64(dwPCRange) {
op := int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*dwPCRange
if op > 255 {
op -= dwLineRange
}
return op
}
return int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*int64(deltaPC)
default:
if deltaPC <= uint64(dwPCRange) {
op := int64(dwOpcodeBase) + (dwLineRange - 1) + dwLineRange*int64(deltaPC)
if op > 255 {
op = 255
}
return op
}
switch deltaPC - uint64(dwPCRange) {
case uint64(dwPCRange), (1 << 7) - 1, (1 << 16) - 1, (1 << 21) - 1,
(1 << 28) - 1, (1 << 35) - 1, (1 << 42) - 1, (1 << 49) - 1,
(1 << 56) - 1, (1 << 63) - 1:
return 255
default:
// 250: the toolchain's "249" comment is stale.
return dwOpcodeBase + dwLineRange*dwPCRange - 1
}
}
}
// goobjDwarfInfo builds the function's DWARF5 subprogram DIE (abbrev
// DW_ABRV_FUNCTION): name, low_pc as a .debug_addr index (the
// R_DWTXTADDR_U4 relocation), high_pc as the size, the call-frame-CFA
// frame base, the decl file/line and the external flag. name is the
// symbol's object name (package-qualified unless static).
func goobjDwarfInfo(fn FuncLayout, name string, fnNpIdx int) ([]byte, []goobjReloc) {
out := []byte{3} // DW_ABRV_FUNCTION
out = append(out, name...)
out = append(out, 0)
addrx := len(out)
out = append(out, 0, 0, 0, 0) // DW_AT_low_pc: addrx slot, zeroed
out = binary.AppendUvarint(out, uint64(fn.Size))
out = append(out, 1, 0x9c) // DW_AT_frame_base: block1, DW_OP_call_frame_cfa
out = binary.LittleEndian.AppendUint32(out, 1)
out = binary.AppendUvarint(out, uint64(fn.Line))
if fn.Static {
out = append(out, 0)
} else {
out = append(out, 1) // DW_AT_external
}
out = append(out, 0) // end of children
relocs := []goobjReloc{{
off: int32(addrx), siz: 4, typ: relocDWTXTADDRU4(),
pkg: pkgIdxNone, sym: uint32(fnNpIdx),
}}
return out, relocs
}
+214
View File
@@ -0,0 +1,214 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"testing"
)
// TestDWSelectOpcode checks the special-opcode selection against
// hand-computed values for the boundary cases: line deltas below, inside
// and above the line range, and pc deltas at and beyond PC_RANGE (24).
func TestDWSelectOpcode(t *testing.T) {
cases := []struct {
deltaPC uint64
deltaLC int64
want int64
}{
{0, 2, 17}, // the common single-instruction step
{0, -4, 11}, // deltaLC == LINE_BASE
{0, -5, 11}, // deltaLC below LINE_BASE: opcode adds nothing
{0, 6, 20}, // deltaLC == LINE_BASE+LINE_RANGE, remainder via advance_line
{4, 1, 56}, // the 4-byte loong64 instruction step
{23, 1, 246}, // deltaPC == PC_RANGE-1
{24, 1, 246}, // deltaPC == PC_RANGE: wraps past 255
{25, 1, 246}, // deltaPC past PC_RANGE (the const_add_pc remainder adjusts it later)
{100, 1, 246},
{151, 10, 255}, // deltaPC-PC_RANGE == (1<<7)-1, large line delta
{100, 10, 250}, // deltaPC-PC_RANGE not on a switch boundary
{23, 10, 250}, // large line delta inside PC_RANGE
}
for _, c := range cases {
if got := dwSelectOpcode(c.deltaPC, c.deltaLC); got != c.want {
t.Errorf("dwSelectOpcode(%d, %d) = %d, want %d", c.deltaPC, c.deltaLC, got, c.want)
}
}
}
// decodeDWLineProgram decodes a .debug_line state-machine program (as
// emitted by goobjDwarfLines) into (pc, line) rows.
func decodeDWLineProgram(t *testing.T, b []byte) (pcs []uint64, lines []int64) {
t.Helper()
pc, line := uint64(0), int64(1)
emit := func() {
if len(pcs) == 0 || pcs[len(pcs)-1] != pc || lines[len(lines)-1] != line {
pcs = append(pcs, pc)
lines = append(lines, line)
}
}
advancePC := func(delta uint64) { pc += delta }
advanceLine := func(delta int64) { line += delta }
for i := 0; i < len(b); {
op := b[i]
i++
switch {
case op == 0: // extended opcode
ln, n := binary.Uvarint(b[i:])
i += n
sub := b[i]
i++
_ = ln
switch sub {
case 2: // DW_LNE_set_address: 8-byte address
pc = binary.LittleEndian.Uint64(b[i:])
i += 8
case 1: // DW_LNE_end_sequence
// terminates the sequence; no new row
}
case op == 2: // DW_LNS_advance_pc
v, n := binary.Uvarint(b[i:])
i += n
advancePC(v)
case op == 3: // DW_LNS_advance_line
v, n := binary.Varint(b[i:])
i += n
advanceLine(v)
case op == 8: // DW_LNS_const_add_pc
advancePC(uint64(dwPCRange))
case op == 9: // DW_LNS_fixed_advance_pc
advancePC(uint64(binary.LittleEndian.Uint16(b[i:])))
i += 2
case op >= dwOpcodeBase: // special opcode
advancePC(uint64((int64(op) - dwOpcodeBase) / dwLineRange))
advanceLine((int64(op)-dwOpcodeBase)%dwLineRange + dwLineBase)
emit()
}
}
return pcs, lines
}
// TestGoobjDwarfLinesRows checks the emitted line program's rows for
// synthetic functions: a zero-frame function with one instruction per
// line, a framed function (the prologue row is prepended on the TEXT
// line), instructions sharing a line, and a function with a large pc gap
// (the const_add_pc remainder path).
func TestGoobjDwarfLinesRows(t *testing.T) {
cases := []struct {
name string
fn FuncLayout
want [][2]int64 // (pc, line)
}{
{
"one instruction per line",
FuncLayout{Size: 20, Line: 2, Lines: []LineEntry{
{0, 3}, {4, 4}, {8, 5}, {12, 6}, {16, 7},
}},
[][2]int64{{0, 3}, {4, 4}, {8, 5}, {12, 6}, {16, 7}},
},
{
"framed: prologue row on the TEXT line",
FuncLayout{Size: 24, Line: 2, Lines: []LineEntry{
{12, 3}, {16, 4},
}},
[][2]int64{{0, 2}, {12, 3}, {16, 4}},
},
{
"instructions sharing a line fold into one row",
FuncLayout{Size: 16, Line: 2, Lines: []LineEntry{
{0, 3}, {4, 3}, {8, 4}, {12, 4},
}},
[][2]int64{{0, 3}, {8, 4}},
},
{
"large gap crosses PC_RANGE",
FuncLayout{Size: 60, Line: 2, Lines: []LineEntry{
{0, 3}, {40, 4},
}},
[][2]int64{{0, 3}, {40, 4}},
},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
prog, relocs := goobjDwarfLines(c.fn, 0)
if len(relocs) != 1 || relocs[0].off != 3 || relocs[0].siz != 8 || relocs[0].typ != relocAddr || relocs[0].sym != 0 {
t.Fatalf("relocs = %+v", relocs)
}
pcs, lines := decodeDWLineProgram(t, prog)
if len(pcs) != len(c.want) {
t.Fatalf("rows = %d (%v / %v), want %d", len(pcs), pcs, lines, len(c.want))
}
for i, w := range c.want {
if pcs[i] != uint64(w[0]) || lines[i] != w[1] {
t.Errorf("row %d = (%d, %d), want (%d, %d)", i, pcs[i], lines[i], w[0], w[1])
}
}
})
}
}
// TestGoobjDwarfInfo checks the subprogram DIE for an exported and a
// static function: the abbrev, name, high_pc, frame base, decl file/line,
// the external flag and the addrx relocation position.
func TestGoobjDwarfInfo(t *testing.T) {
fn := FuncLayout{Size: 20, Line: 2}
die, relocs := goobjDwarfInfo(fn, "pkg.f", 3)
want := []byte{
0x03,
'p', 'k', 'g', '.', 'f', 0,
0, 0, 0, 0, // addrx slot at offset 7
0x14, // high_pc: 20
0x01, 0x9c, // frame_base
0x01, 0, 0, 0, // decl_file 1
0x02, // decl_line 2
0x01, // external
0x00, // end of children
}
if !bytes.Equal(die, want) {
t.Errorf("DIE = %x, want %x", die, want)
}
if len(relocs) != 1 || relocs[0].off != 7 || relocs[0].siz != 4 || relocs[0].typ != relocDWTXTADDRU4() || relocs[0].sym != 3 {
t.Errorf("relocs = %+v", relocs)
}
// A static function carries no external flag and no package prefix.
fn.Static = true
die, _ = goobjDwarfInfo(fn, "f", 1)
if die[len(die)-2] != 0 {
t.Errorf("static external flag = %d, want 0", die[len(die)-2])
}
}
// TestDwPutPCLCDeltaRemainders checks the standard-opcode remainders:
// const_add_pc and fixed_advance_pc after a special opcode.
func TestDwPutPCLCDeltaRemainders(t *testing.T) {
// deltaPC 25 past PC_RANGE: opcode 26 covers (1, 1), const_add_pc
// covers the remaining 23 pc and 0 line.
got := dwPutPCLCDelta(nil, 25, 1)
if !bytes.Equal(got, []byte{8, 26}) {
t.Errorf("25/1 = %x, want [8 1a]", got)
}
// deltaPC 20000: opcode 246 covers 23, fixed_advance_pc covers the
// remaining 19977.
got = dwPutPCLCDelta(nil, 20000, 1)
if got[0] != 9 || binary.LittleEndian.Uint16(got[1:]) != 19977 || got[3] != 246 {
t.Errorf("20000/1 = %x, want fixed_advance_pc 19977 then 246", got)
}
// Line remainder: deltaLC 10 leaves 5 past the opcode's reach, encoded
// as advance_line 5 (zigzag 0x0a) before opcode 250.
got = dwPutPCLCDelta(nil, 23, 10)
if !bytes.Equal(got, []byte{3, 0x0a, 250}) {
t.Errorf("23/10 = %x, want [03 0a fa]", got)
}
// Negative line remainder: deltaLC -5 leaves advance_line -1 (zigzag
// 0x01) after opcode 11.
got = dwPutPCLCDelta(nil, 0, -5)
if !bytes.Equal(got, []byte{3, 1, 11}) {
t.Errorf("0/-5 = %x, want [03 01 0b]", got)
}
}
+348
View File
@@ -0,0 +1,348 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"fmt"
"os"
"os/exec"
"strings"
)
// readGOOBJSymbols reads the GOOBJ symbol definitions from a compiled Go
// package's export file. The file is an ar archive containing a __.PKGDEF
// member whose payload is the "go object ...\n!\n" preamble followed by the
// GOOBJ data. The function returns the symbol names in definition order
// (the order they appear in blkSymdef), which matches the SymIdx the linker
// expects for cross-package references.
func readGOOBJSymbols(exportPath string) ([]string, error) {
data, err := os.ReadFile(exportPath)
if err != nil {
return nil, err
}
goobj, err := extractGOOBJ(data)
if err != nil {
return nil, fmt.Errorf("%s: %w", exportPath, err)
}
return goobj.symbols(), nil
}
// exportPath returns the export file path for a given import path by running
// "go list -export". The result is cached so repeated calls for the same
// package are fast.
func exportPath(importPath string) (string, error) {
cmd := exec.Command("go", "list", "-json", "-export", importPath)
out, err := cmd.Output()
if err != nil {
return "", fmt.Errorf("go list %s: %w", importPath, err)
}
// Quick JSON extraction: find "Export": "…"
const key = `"Export": "`
i := bytes.Index(out, []byte(key))
if i < 0 {
return "", fmt.Errorf("go list %s: no Export field", importPath)
}
start := i + len(key)
end := bytes.IndexByte(out[start:], '"')
if end < 0 {
return "", fmt.Errorf("go list %s: malformed Export field", importPath)
}
return string(out[start : start+end]), nil
}
// resolveExternalGOOBJ resolves a set of external symbol references into
// (package index, symbol index) pairs suitable for GOOBJ emission.
//
// refs maps package import paths to the symbol names referenced from that
// package. The returned pkgIdx maps each import path to its position in
// the blkPkgIdx table (0-based), and symIdx gives each symbol's index within
// its package.
func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symIdx map[string]int, err error) {
pkgIdx = make(map[string]int, len(refs))
symIdx = make(map[string]int)
// Assign package indices in sorted order for determinism.
packages := sortedPkgRefs(refs)
for i, pkg := range packages {
pkgIdx[pkg.path] = i
exp, err := exportPath(pkg.path)
if err != nil {
return nil, nil, err
}
data, err := os.ReadFile(exp)
if err != nil {
return nil, nil, err
}
gobj, err := extractGOOBJ(data)
if err != nil {
return nil, nil, fmt.Errorf("%s: %w", pkg.path, err)
}
for _, name := range pkg.syms {
idx := gobj.findSymbol(pkg.path, name)
if idx < 0 {
return nil, nil, fmt.Errorf("symbol %s·%s not found in export data of %s", pkg.path, name, pkg.path)
}
symIdx[pkg.path+"·"+name] = idx
}
}
return pkgIdx, symIdx, nil
}
type pkgRef struct {
path string
syms []string
}
func sortedPkgRefs(refs map[string][]string) []pkgRef {
var pkgs []pkgRef
for pkg, syms := range refs {
pkgs = append(pkgs, pkgRef{pkg, syms})
}
// Simple insertion sort — the list is tiny (usually 1–3 packages).
for i := 1; i < len(pkgs); i++ {
for j := i; j > 0 && pkgs[j-1].path > pkgs[j].path; j-- {
pkgs[j-1], pkgs[j] = pkgs[j], pkgs[j-1]
}
}
return pkgs
}
// extractGOOBJ finds the GOOBJ data in an ar archive and returns a parsed
// goobjFile. The archive member _go_.o contains the "go object …\n!\n"
// preamble followed by the GOOBJ payload; __.PKGDEF is the compiler export
// data (type information) and is not the GOOBJ object.
func extractGOOBJ(data []byte) (*goobjFile, error) {
if len(data) < 8 || string(data[:8]) != "!<arch>\n" {
return nil, fmt.Errorf("not an ar archive")
}
pos := 8
for pos+60 <= len(data) {
hdr := data[pos : pos+60]
pos += 60
// Parse ar header fields.
name := strings.TrimRight(string(hdr[:16]), " /")
size := parseArDecimal(hdr[48:58])
if size < 0 {
return nil, fmt.Errorf("invalid ar header: bad size")
}
if pos+size > len(data) {
return nil, fmt.Errorf("ar entry %q extends past end of file", name)
}
body := data[pos : pos+size]
pos += size
// ar pads to even bytes.
if pos%2 != 0 {
pos++
}
if name == "_go_.o" {
return parseGOOBJ(body)
}
}
return nil, fmt.Errorf("archive contains no _go_.o member")
}
// parseArDecimal parses a decimal number from a space-padded field.
func parseArDecimal(b []byte) int {
v := 0
for _, c := range b {
if c == ' ' {
continue
}
if c < '0' || c > '9' {
return -1
}
v = v*10 + int(c-'0')
}
return v
}
// goobjFile is a parsed GOOBJ file: the string table and the symbol-definition
// block.
type goobjFile struct {
strTab []byte // string table, at headerSize + n
symdef []byte // blkSymdef raw block
npdef []byte // blkNonpkgdef raw block
}
// symbols returns all symbol names in definition order by scanning the
// symdef and nonpkgdef blocks and resolving each name through the string
// table. Package definitions (blkSymdef) use fully-qualified names like
// "runtime.morestack"; non-package definitions (blkNonpkgdef) use bare
// names like "morestack". This combined list matches the index the
// linker expects for cross-package references.
func (f *goobjFile) symbols() []string {
return append(f.defNames(), f.npdefNames()...)
}
// findSymbol returns the index of a symbol within the combined symbol list,
// or -1 if not found. It first tries the fully-qualified name (pkg.name),
// then the bare name.
func (f *goobjFile) findSymbol(pkg, name string) int {
qualified := pkg + "." + name
syms := f.symbols()
for i, s := range syms {
if s == qualified {
return i
}
}
// Try bare name (for non-package definitions).
for i, s := range syms {
if s == name {
return i
}
}
return -1
}
// defNames returns names from blkSymdef only.
func (f *goobjFile) defNames() []string {
return f.readSymNames(f.symdef)
}
// npdefNames returns names from blkNonpkgdef.
func (f *goobjFile) npdefNames() []string {
return f.readSymNames(f.npdef)
}
// readSymNames reads symbol names from a symdef/nonpkgdef block. Each record
// is 21 bytes: nameLen (u32), nameOff (u32), abi (u16), typ, flag, flag2,
// size (u32), align (u32). nameOff is an absolute offset into the string
// table.
func (f *goobjFile) readSymNames(block []byte) []string {
const recSize = 21
if len(block) < recSize {
return nil
}
n := len(block) / recSize
names := make([]string, 0, n)
for i := 0; i < n; i++ {
rec := block[i*recSize : (i+1)*recSize]
nameLen := binary.LittleEndian.Uint32(rec[0:4])
nameOff := binary.LittleEndian.Uint32(rec[4:8])
// nameOff is an absolute offset into the GOOBJ payload. The string
// table we have starts at goobjHeaderSize, so we subtract that.
if nameOff < goobjHeaderSize {
continue
}
relOff := nameOff - goobjHeaderSize
if relOff >= uint32(len(f.strTab)) || relOff+nameLen > uint32(len(f.strTab)) {
continue
}
names = append(names, string(f.strTab[relOff:relOff+nameLen]))
}
return names
}
const goobjHeaderSize = 8 + 8 + 4 + 4*(blkEnd+1) // magic + fingerprint + flags + 19 block offsets
// parseGOOBJ parses a raw GOOBJ payload (the data after the "\n!\n" preamble).
func parseGOOBJ(data []byte) (*goobjFile, error) {
// Find the "\n!\n" separator.
sep := []byte("\n!\n")
i := bytes.Index(data, sep)
if i < 0 {
// Maybe the data has no preamble (e.g. a raw .o file).
i = -3 // treat as if preamble starts before the data
}
payload := data[i+len(sep):]
if len(payload) < goobjHeaderSize {
return nil, fmt.Errorf("GOOBJ payload too short (%d bytes)", len(payload))
}
if string(payload[:8]) != goobjMagic {
return nil, fmt.Errorf("bad GOOBJ magic: %q", payload[:8])
}
// Read block offsets. The header layout is:
// [0:8] magic
// [8:16] fingerprint
// [16:20] flags
// [20:96] 19 × uint32 offsets
var offs [blkEnd + 1]uint32
for i := 0; i <= blkEnd; i++ {
offs[i] = binary.LittleEndian.Uint32(payload[20+4*i:])
}
// The string table lives at headerSize.
strTabStart := uint32(goobjHeaderSize)
f := &goobjFile{
strTab: payload[strTabStart:offs[0]],
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
}
return f, nil
}
// blockSlice extracts a block from the payload using its offset pair.
func blockSlice(payload []byte, offs [blkEnd + 1]uint32, start, end int) []byte {
if start < 0 || end > blkEnd || offs[end] < offs[start] {
return nil
}
beg := offs[start]
fin := offs[end]
if int(fin) > len(payload) || int(beg) > int(fin) {
return nil
}
return payload[beg:fin]
}
// resolveExternalSymbols is the high-level entry point for GOOBJ emission.
// Given a list of external symbol names (e.g. ["runtime·morestack",
// "runtime·g0"]), it returns the package-index table entries and a map from
// full symbol name to GOOBJ {pkgIdx, symIdx}.
//
// The package table entries should be written into blkPkgIdx, and the
// returned indices should replace pkgIdxSelf / placeholder values in the
// relocation records.
func resolveExternalSymbols(externals []string) (pkgTable []string, pkgIdxMap map[string]int, symIdxMap map[string]int, err error) {
// Group references by package.
refs := make(map[string]map[string]bool)
for _, full := range externals {
pkg, name := splitQualified(full)
if refs[pkg] == nil {
refs[pkg] = make(map[string]bool)
}
refs[pkg][name] = true
}
// Convert maps to slices.
r := make(map[string][]string, len(refs))
for pkg, names := range refs {
for name := range names {
r[pkg] = append(r[pkg], name)
}
}
pkgIdx1, symIdx1, err := resolveExternalGOOBJ(r)
if err != nil {
return nil, nil, nil, err
}
// Build the package table in pkgIdx order.
pkgTable = make([]string, len(pkgIdx1))
for pkg, idx := range pkgIdx1 {
pkgTable[idx] = pkg
}
return pkgTable, pkgIdx1, symIdx1, nil
}
// splitQualified splits a qualified Go symbol name (pkgpath·name) into its
// package path and local name. The separator is the middle dot (U+00B7).
// If no separator is found, the symbol is assumed to be in the current
// package (empty pkg).
func splitQualified(full string) (pkg, name string) {
if idx := strings.IndexByte(full, '\u00b7'); idx >= 0 {
return full[:idx], full[idx+len("\u00b7"):]
}
if idx := strings.IndexByte(full, '.'); idx >= 0 {
return full[:idx], full[idx+1:]
}
return "", full
}
+66
View File
@@ -0,0 +1,66 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"os"
"os/exec"
"testing"
)
// TestReadRuntimeSymbols verifies the GOOBJ reader can extract and find
// symbols from the runtime package's compiled archive.
func TestReadRuntimeSymbols(t *testing.T) {
exp, err := exportPath("runtime")
if err != nil {
t.Skipf("cannot find runtime export: %v (need Go toolchain)", err)
}
data, err := os.ReadFile(exp)
if err != nil {
t.Skipf("cannot read runtime export: %v", err)
}
gobj, err := extractGOOBJ(data)
if err != nil {
t.Fatalf("extractGOOBJ: %v", err)
}
t.Logf("runtime: %d symbols", len(gobj.symbols()))
// Verify we can find well-known runtime symbols.
for _, tc := range []struct{ pkg, name string }{
{"runtime", "g0"},
{"runtime", "morestack"},
{"runtime", "newstack"},
} {
idx := gobj.findSymbol(tc.pkg, tc.name)
if idx < 0 {
t.Errorf("findSymbol(%q, %q) = -1", tc.pkg, tc.name)
} else {
t.Logf("findSymbol(%q, %q) = %d", tc.pkg, tc.name, idx)
}
}
}
// TestResolveExternalSymbols verifies end-to-end resolution of external
// symbol references.
func TestResolveExternalSymbols(t *testing.T) {
if _, err := exec.LookPath("go"); err != nil {
t.Skip("go toolchain not available")
}
refs := map[string][]string{
"runtime": {"g0"},
}
pkgIdx, symIdx, err := resolveExternalGOOBJ(refs)
if err != nil {
t.Fatalf("resolveExternalGOOBJ: %v", err)
}
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 0 {
t.Errorf("pkgIdx = %v, want runtime→0", pkgIdx)
}
if _, ok := symIdx["runtime·g0"]; !ok {
t.Errorf("symIdx missing runtime·g0, got %v", symIdx)
}
t.Logf("runtime·g0 → SymIdx=%d", symIdx["runtime·g0"])
}
+75 -39
View File
@@ -111,17 +111,26 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
t.Errorf("flags = %#x, want ObjFlagFromAssembly (4)", flags)
}
// Package defs: the static GLOBL, then one anonymous FuncInfo per
// function.
// Package defs: the static GLOBL, then per function the FuncInfo and the
// two DWARF symbols (debug_line program, subprogram DIE).
defs := v.syms(blkSymdef)
if len(defs) != 3 {
t.Fatalf("symdefs = %d, want 3", len(defs))
if len(defs) != 7 {
t.Fatalf("symdefs = %d, want 7", len(defs))
}
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
t.Errorf("mask symbol = %+v", defs[0])
}
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
t.Errorf("funcinfo symbol = %+v", defs[1])
t.Errorf("addq funcinfo symbol = %+v", defs[1])
}
if defs[2].name != "" || defs[2].typ != kindSDWARFLINES || defs[2].size == 0 {
t.Errorf("addq lines symbol = %+v", defs[2])
}
if defs[3].name != "" || defs[3].typ != kindSDWARFFCN || defs[3].size == 0 {
t.Errorf("addq DIE symbol = %+v", defs[3])
}
if defs[4].name != "" || defs[4].typ != kindSDATA || defs[4].size != 28 {
t.Errorf("loadmask funcinfo symbol = %+v", defs[4])
}
// Non-package defs: four pc tables and the function, per function.
@@ -142,45 +151,62 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
// FuncInfo: args 24, FuncFlag Asm, one file, no inline tree.
le := binary.LittleEndian
data := v.blk(blkData)
didx := v.blk(blkDataIdx)
fi := data[16:44]
if le.Uint32(fi[0:]) != 24 || le.Uint32(fi[4:]) != 0 || fi[8] != 0 || fi[9] != funcFlagAsm ||
le.Uint32(fi[16:]) != 1 || le.Uint32(fi[20:]) != 0 || le.Uint32(fi[24:]) != 0 {
t.Errorf("funcinfo bytes %x", fi)
}
// pcsp: a flat zero over the whole function (zero-frame NOSPLIT).
if got := data[72:75]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
// The pc-value tables of addq (non-package indices 0–3, so global
// indices 7–10): pcsp a flat zero over the whole function, pcinline a
// flat -1, both with the pc delta in MinLC (1) units.
pcsp := data[le.Uint32(didx[4*7:]):]
if got := pcsp[:3]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
t.Errorf("pcsp = %x, want 021300", got)
}
// pcinline: a flat -1.
if got := data[81:84]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) {
pcinl := data[le.Uint32(didx[4*10:]):]
if got := pcinl[:3]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) {
t.Errorf("pcinline = %x, want 001300", got)
}
// The one relocation: R_PCREL, four bytes wide, against the GLOBL,
// with the field in the function code left zero. The loadmask code's
// offset comes from the data index (symbol 3 defs + 9 non-package).
// Relocations: the four DWARF address references (two per function, in
// definition order), then the loadmask code's R_PCREL against the
// GLOBL, with the field in the function code left zero. The loadmask
// code's offset comes from the data index (7 defs + 9 non-package).
relocs := v.blk(blkReloc)
if len(relocs) != 23 {
t.Fatalf("relocs = %d bytes, want one 23-byte entry", len(relocs))
if len(relocs) != 5*23 {
t.Fatalf("relocs = %d bytes, want 5 entries", len(relocs))
}
off := int32(le.Uint32(relocs[0:]))
if off != 4 || relocs[4] != 4 || le.Uint16(relocs[5:]) != relocPCRel ||
le.Uint64(relocs[7:]) != 0 || le.Uint32(relocs[15:]) != pkgIdxSelf || le.Uint32(relocs[19:]) != 0 {
t.Errorf("reloc = %x", relocs)
// addq's DWARF references (defs 2 and 3) against the function, which
// is non-package index 4.
lr := relocs[:23]
if int32(le.Uint32(lr[0:])) != 3 || lr[4] != 8 || le.Uint16(lr[5:]) != relocAddr ||
le.Uint32(lr[15:]) != pkgIdxNone || le.Uint32(lr[19:]) != 4 {
t.Errorf("addq lines reloc = %x", lr)
}
didx := v.blk(blkDataIdx)
lm := le.Uint32(didx[4*(3+9):])
dr := relocs[23:46]
if dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4() ||
le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 {
t.Errorf("addq DIE reloc = %x", dr)
}
cr := relocs[4*23:]
off := int32(le.Uint32(cr[0:]))
if off != 4 || cr[4] != 4 || le.Uint16(cr[5:]) != relocPCRel ||
le.Uint64(cr[7:]) != 0 || le.Uint32(cr[15:]) != pkgIdxSelf || le.Uint32(cr[19:]) != 0 {
t.Errorf("loadmask reloc = %x", cr)
}
lm := le.Uint32(didx[4*16:])
code := data[lm : lm+18]
if !bytes.Equal(code[4:8], []byte{0, 0, 0, 0}) {
t.Errorf("relocated field = %x, want zeroed", code[4:8])
}
// Aux wiring: FuncInfo (package symbol), then the four pc tables
// (non-package symbols).
// Aux wiring: FuncInfo, the two DWARF symbols (package symbols), then
// the four pc tables (non-package symbols).
auxs := v.blk(blkAux)
if len(auxs) != 2*5*9 {
t.Fatalf("aux = %d bytes, want 10 entries", len(auxs))
if len(auxs) != 2*7*9 {
t.Fatalf("aux = %d bytes, want 14 entries", len(auxs))
}
wantAux := []struct {
typ uint8
@@ -188,15 +214,19 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
idx uint32
}{
{auxFuncInfo, pkgIdxSelf, 1},
{auxPcsp, pkgIdxNone, uint32(len(defs) + 0)},
{auxPcfile, pkgIdxNone, uint32(len(defs) + 1)},
{auxPcline, pkgIdxNone, uint32(len(defs) + 2)},
{auxPcinline, pkgIdxNone, uint32(len(defs) + 3)},
{auxFuncInfo, pkgIdxSelf, 2},
{auxPcsp, pkgIdxNone, uint32(len(defs) + 5)},
{auxPcfile, pkgIdxNone, uint32(len(defs) + 6)},
{auxPcline, pkgIdxNone, uint32(len(defs) + 7)},
{auxPcinline, pkgIdxNone, uint32(len(defs) + 8)},
{auxDwarfInfo, pkgIdxSelf, 3},
{auxDwarfLines, pkgIdxSelf, 2},
{auxPcsp, pkgIdxNone, 0},
{auxPcfile, pkgIdxNone, 1},
{auxPcline, pkgIdxNone, 2},
{auxPcinline, pkgIdxNone, 3},
{auxFuncInfo, pkgIdxSelf, 4},
{auxDwarfInfo, pkgIdxSelf, 6},
{auxDwarfLines, pkgIdxSelf, 5},
{auxPcsp, pkgIdxNone, 5},
{auxPcfile, pkgIdxNone, 6},
{auxPcline, pkgIdxNone, 7},
{auxPcinline, pkgIdxNone, 8},
}
for i, w := range wantAux {
e := auxs[i*9:]
@@ -253,7 +283,7 @@ TEXT ·framed(SB), NOSPLIT, $8-0
t.Fatalf("AssembleFile: %v", err)
}
fn := img.Funcs[0]
pcs, vals := decodePCValues(pcspTable(fn))
pcs, vals := decodePCValues(pcspTable(fn, 1))
// Prologue: PUSHQ BP (1 byte, +8), MOVQ SP, BP (3 bytes, no change),
// SUBQ $8, SP (4 bytes, +16 in total); the RET's epilogue unwinds
// ADDQ $8, SP (+8) then POPQ BP (0).
@@ -349,7 +379,7 @@ func main() {
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module goobjtest\n\ngo 1.26\n"), 0o644); err != nil {
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module goobjtest\n\ngo 1.27\n"), 0o644); err != nil {
t.Fatal(err)
}
@@ -401,13 +431,12 @@ func main() {
if err != nil {
t.Fatalf("GOObject: %v", err)
}
if err := os.WriteFile(asmObj, obj, 0o644); err != nil {
t.Fatal(err)
}
// Rebuild the package archive with our object in place of the
// toolchain's (go tool pack has no replace-in-place that dedupes, so
// extract, substitute and repack).
// extract, substitute and repack). The archive member holding the
// assembler's output is named after the asm object file, e.g.
// main_amd64.o.
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
@@ -417,6 +446,13 @@ func main() {
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
member := filepath.Join(membersDir, filepath.Base(asmObj))
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, obj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listOut, err := listCmd.CombinedOutput()
if err != nil {
+84
View File
@@ -0,0 +1,84 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"fmt"
"os"
"os/exec"
"path/filepath"
"sync"
)
// GOObjectAARCH64 emits a GOOBJ object file for AArch64. The layout is
// the shared one in goobj.go — the toolchain preamble, the go120ld header
// with its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the arm64 preamble, the MinLC of 4
// for the pc-value deltas, and R_ADDRARM64 relocation types for the
// ADRP+ADD/LDR/STR address pairs.
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreambleAARCH64()
if err != nil {
return nil, err
}
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
return relocArm64Addr, 4
})
}
// arm64 relocation types (cmd/internal/objabi). R_ADDRARM64 resolves an
// ADRP+ADD/LDR/STR pair to a symbol's address.
const (
relocArm64Addr = 9 // R_ADDRARM64
)
// toolchainObjectPreambleAARCH64 returns the "go object ...\n!\n" header
// the installed go tool asm writes for arm64, captured by assembling a
// one-instruction probe.
var (
preambleAARCH64Once sync.Once
preambleAARCH64 []byte
preambleAARCH64Err error
)
func toolchainObjectPreambleAARCH64() ([]byte, error) {
preambleAARCH64Once.Do(func() {
goBin, err := exec.LookPath("go")
if err != nil {
preambleAARCH64Err = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
return
}
dir, err := os.MkdirTemp("", "gasm-preamble-arm64")
if err != nil {
preambleAARCH64Err = err
return
}
defer os.RemoveAll(dir)
src := filepath.Join(dir, "probe_arm64.s")
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
preambleAARCH64Err = err
return
}
obj := filepath.Join(dir, "probe.o")
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
cmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := cmd.CombinedOutput(); err != nil {
preambleAARCH64Err = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
return
}
data, err := os.ReadFile(obj)
if err != nil {
preambleAARCH64Err = err
return
}
i := bytes.Index(data, []byte("\n!\n"))
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
preambleAARCH64Err = fmt.Errorf("unrecognised assembler object layout")
return
}
preambleAARCH64 = data[:i+3]
})
return preambleAARCH64, preambleAARCH64Err
}
+91
View File
@@ -0,0 +1,91 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"fmt"
"os"
"os/exec"
"path/filepath"
"sync"
)
// GOObjectLOONG64 emits a GOOBJ object file for LoongArch. The layout is
// the shared one in goobj.go — the toolchain preamble, the go120ld header
// with its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the loong64 preamble, the MinLC of 4
// for the pc-value deltas, and R_LOONG64_ADDR_HI/LO relocation types for
// the pcalau12i+addi.d address pairs.
func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreambleLOONG64()
if err != nil {
return nil, err
}
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
// A pcalau12i+addi.d pair: the high part carries
// R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO.
if r.Kind == RelLoong64AddrLo {
return relocLoong64AddrLo, 4
}
return relocLoong64AddrHi, 4
})
}
// Loong64 relocation types (cmd/internal/objabi). R_LOONG64_ADDR_HI
// resolves the high 20 bits of a PC-relative address into pcalau12i;
// R_LOONG64_ADDR_LO the low 12 bits into addi.d/ld/st.
const (
relocLoong64AddrHi = 77 // R_LOONG64_ADDR_HI
relocLoong64AddrLo = 78 // R_LOONG64_ADDR_LO
)
// toolchainObjectPreambleLOONG64 returns the "go object ...\n!\n" header
// the installed go tool asm writes for loong64, captured by assembling a
// one-instruction probe (see toolchainObjectPreamble).
var (
preambleLOONG64Once sync.Once
preambleLOONG64 []byte
preambleLOONG64Err error
)
func toolchainObjectPreambleLOONG64() ([]byte, error) {
preambleLOONG64Once.Do(func() {
goBin, err := exec.LookPath("go")
if err != nil {
preambleLOONG64Err = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
return
}
dir, err := os.MkdirTemp("", "gasm-preamble-loong64")
if err != nil {
preambleLOONG64Err = err
return
}
defer os.RemoveAll(dir)
src := filepath.Join(dir, "probe_loong64.s")
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
preambleLOONG64Err = err
return
}
obj := filepath.Join(dir, "probe.o")
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
cmd.Env = append(os.Environ(), "GOARCH=loong64")
if out, err := cmd.CombinedOutput(); err != nil {
preambleLOONG64Err = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
return
}
data, err := os.ReadFile(obj)
if err != nil {
preambleLOONG64Err = err
return
}
i := bytes.Index(data, []byte("\n!\n"))
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
preambleLOONG64Err = fmt.Errorf("unrecognised assembler object layout")
return
}
preambleLOONG64 = data[:i+3]
})
return preambleLOONG64, preambleLOONG64Err
}
+25 -282
View File
@@ -5,7 +5,6 @@ package asm
import (
"bytes"
"encoding/binary"
"fmt"
"os"
"os/exec"
@@ -13,298 +12,42 @@ import (
"sync"
)
// GOObjectRISCV emits a GOOBJ object file for RISC-V.
// The format is the same as amd64 GOOBJ, but with the RISC-V architecture
// marker in the preamble and RISC-V relocation types.
// GOObjectRISCV emits a GOOBJ object file for RISC-V. The layout is the
// shared one in goobj.go — the toolchain preamble, the go120ld header with
// its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the RISC-V preamble, the MinLC of 2 for
// the pc-value deltas, and the single R_RISCV_PCREL_ITYPE/STYPE relocation
// per AUIPC pair, matching `go tool asm`'s model (each pair is one 8-byte
// relocation, not the ELF HI20/LO12 pair).
func (img *Image) GOObjectRISCV(pkgPath, srcPath string) ([]byte, error) {
if pkgPath == "" {
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
}
pre, err := toolchainObjectPreambleRISCV()
if err != nil {
return nil, err
}
// The symbol tables. Package definitions: the GLOBL symbols, then one
// anonymous FuncInfo symbol per function. Non-package definitions: the
// pc-value tables and the functions themselves, as cmd/asm lays them
// out. defIdx maps a GLOBL's bare name to its definition index for the
// relocations; fnNpIdx maps a function to its non-package index.
var defs []goSym
var defData [][]byte
defIdx := map[string]int{}
for _, d := range img.DataSyms {
name := d.Name
if !d.Static {
name = pkgPath + "." + name
return img.emitGOObject(pkgPath, srcPath, pre, 2, func(r Reloc) (uint16, uint8) {
switch r.Kind {
case RelRISCVPCRELSType:
return relocRISCVPcrelStype, 8
case RelRISCVJal:
return relocRISCVJal, 4
default:
return relocRISCVPcrelItype, 8
}
typ := uint8(kindSDATA)
if d.Rodata {
typ = kindSRODATA
}
flag := uint8(0)
if d.Dupok {
flag = symFlagDupok
}
abi := uint16(0)
if d.Static {
abi = symABIStatic
}
defIdx[d.Name] = len(defs)
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)})
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
}
fnFiIdx := make([]int, len(img.Funcs))
for i := range img.Funcs {
data := marshalFuncInfo(img.Funcs[i])
fnFiIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))})
defData = append(defData, data)
}
type npSym struct {
sym goSym
data []byte
}
var nps []npSym
type pcRefs struct{ sp, file, line, inl int }
pcIdx := make([]pcRefs, len(img.Funcs))
fnNpIdx := make([]int, len(img.Funcs))
for i, fn := range img.Funcs {
tables := []struct {
data []byte
dst *int
}{
{pcspTable(fn), &pcIdx[i].sp},
{pcValueFlat(0, fn.Size), &pcIdx[i].file},
{pcValueFlat(int32(fn.Line), fn.Size), &pcIdx[i].line},
{pcValueFlat(-1, fn.Size), &pcIdx[i].inl},
}
for _, t := range tables {
*t.dst = len(nps)
nps = append(nps, npSym{
sym: goSym{typ: kindSRODATA, size: uint32(len(t.data)), align: 1},
data: t.data,
})
}
name := fn.Name
abi := uint16(0)
if fn.Static {
abi = symABIStatic
} else {
name = pkgPath + "." + name
}
flag := uint8(0)
if fn.NoSplit {
flag |= symFlagNoSplit
}
fnNpIdx[i] = len(nps)
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
// The linker writes the resolved displacement into the field;
// leave it zero, as cmd/asm's object does.
if r.Off >= 0 && r.Off+4 <= len(code) {
code[r.Off], code[r.Off+1], code[r.Off+2], code[r.Off+3] = 0, 0, 0, 0
}
}
nps = append(nps, npSym{
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
data: code,
})
}
// Relocations, per defined symbol in definition order (package defs,
// then non-package defs). Only file-local GLOBL references resolve;
// external symbols need the import machinery of a later increment.
nsyms := len(defs) + len(nps)
symRelocs := make([][]byte, nsyms) // flat 23-byte records
for i, fn := range img.Funcs {
si := len(defs) + fnNpIdx[i]
for _, r := range fn.Relocs {
if r.External {
return nil, fmt.Errorf("GOOBJ emission: external symbol %q is not supported yet", r.Name)
}
di, ok := defIdx[r.Name]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
}
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = 4 // field width
binary.LittleEndian.PutUint16(rec[5:], relocRISCVPcrelHi20)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
symRelocs[si] = append(symRelocs[si], rec[:]...)
}
}
// Aux entries per function: FuncInfo, then the four pc tables.
// References into the non-package table use pkgIdxNone.
symAux := make([][]byte, nsyms)
for i := range img.Funcs {
si := len(defs) + fnNpIdx[i]
aux := func(typ uint8, pkg, idx uint32) {
var rec [9]byte
rec[0] = typ
binary.LittleEndian.PutUint32(rec[1:], pkg)
binary.LittleEndian.PutUint32(rec[5:], idx)
symAux[si] = append(symAux[si], rec[:]...)
}
aux(auxFuncInfo, pkgIdxSelf, uint32(fnFiIdx[i]))
aux(auxPcsp, pkgIdxNone, uint32(len(defs)+pcIdx[i].sp))
aux(auxPcfile, pkgIdxNone, uint32(len(defs)+pcIdx[i].file))
aux(auxPcline, pkgIdxNone, uint32(len(defs)+pcIdx[i].line))
aux(auxPcinline, pkgIdxNone, uint32(len(defs)+pcIdx[i].inl))
}
// --- Serialise ---
// String table: all symbol names, NUL-terminated.
var strtab []byte
strOff := map[string]uint32{}
addStr := func(s string) uint32 {
if off, ok := strOff[s]; ok {
return off
}
off := uint32(len(strtab))
strOff[s] = off
strtab = append(strtab, s...)
strtab = append(strtab, 0)
return off
}
for _, s := range defs {
addStr(s.name)
}
for _, s := range nps {
addStr(s.sym.name)
}
// Symbol definition records (21 bytes each).
var symdef, nonpkgdef []byte
for _, s := range defs {
symdef = s.append(symdef, strOff)
}
for _, s := range nps {
nonpkgdef = s.sym.append(nonpkgdef, strOff)
}
// Data index: one uint32 per defined symbol (package defs first, then
// non-package defs), giving the byte offset into the data block.
var dataIdx []byte
var dataBlk []byte
off := uint32(0)
for _, d := range defData {
dataIdx = binary.LittleEndian.AppendUint32(dataIdx, off)
dataBlk = append(dataBlk, d...)
off += uint32(len(d))
}
for _, s := range nps {
dataIdx = binary.LittleEndian.AppendUint32(dataIdx, off)
dataBlk = append(dataBlk, s.data...)
off += uint32(len(s.data))
}
dataIdx = binary.LittleEndian.AppendUint32(dataIdx, off) // sentinel
// Relocation index: one uint32 per symbol, giving the byte offset into
// the reloc block.
var relocIdx []byte
roff := uint32(0)
for i := 0; i < nsyms; i++ {
relocIdx = binary.LittleEndian.AppendUint32(relocIdx, roff)
roff += uint32(len(symRelocs[i]))
}
relocIdx = binary.LittleEndian.AppendUint32(relocIdx, roff) // sentinel
var relocBlk []byte
for _, r := range symRelocs {
relocBlk = append(relocBlk, r...)
}
// Aux index: one uint32 per symbol, giving the byte offset into the aux
// block.
var auxIdx []byte
aoff := uint32(0)
for i := 0; i < nsyms; i++ {
auxIdx = binary.LittleEndian.AppendUint32(auxIdx, aoff)
aoff += uint32(len(symAux[i]))
}
auxIdx = binary.LittleEndian.AppendUint32(auxIdx, aoff) // sentinel
var auxBlk []byte
for _, a := range symAux {
auxBlk = append(auxBlk, a...)
}
// File table: one entry, the source file.
var fileBlk []byte
fileOff := addStr(srcPath)
fileBlk = binary.LittleEndian.AppendUint32(fileBlk, uint32(len(srcPath)))
fileBlk = binary.LittleEndian.AppendUint32(fileBlk, fileOff)
// Assemble the object.
var out bytes.Buffer
out.Write(pre)
out.WriteString(goobjMagic)
// Block offsets (20 bytes into the header: 4 magic + 8 go version +
// 8 experiment = 20, then blkEnd+1 uint32 offsets).
// We'll fill these in after we know the sizes.
hdrStart := out.Len()
out.Write(make([]byte, 4*(blkEnd+1)))
writeBlock := func(data []byte) {
out.Write(data)
}
// Blocks in order: autolib, pkgidx, file, symdef, hashed64def, hasheddef,
// nonpkgdef, nonpkgref, refflags, hash64, hash, relocidx, auxidx, dataidx,
// reloc, aux, data, refname.
writeBlock(nil) // autolib
writeBlock(nil) // pkgidx
writeBlock(fileBlk) // file
writeBlock(symdef) // symdef
writeBlock(nil) // hashed64def
writeBlock(nil) // hasheddef
writeBlock(nonpkgdef) // nonpkgdef
writeBlock(nil) // nonpkgref
writeBlock(nil) // refflags
writeBlock(nil) // hash64
writeBlock(nil) // hash
writeBlock(relocIdx) // relocidx
writeBlock(auxIdx) // auxidx
writeBlock(dataIdx) // dataidx
writeBlock(relocBlk) // reloc
writeBlock(auxBlk) // aux
writeBlock(dataBlk) // data
writeBlock(nil) // refname
// Fill in the block offsets.
le := binary.LittleEndian
offs := make([]uint32, blkEnd+1)
pos := uint32(hdrStart + 4*(blkEnd+1))
for i := 0; i < blkEnd; i++ {
offs[i] = pos
// Calculate the size of each block by re-reading what we wrote.
// This is a simplification; a real implementation would track sizes.
}
offs[blkEnd] = uint32(out.Len())
// For now, just write zeros for the offsets (the linker will parse the
// blocks sequentially anyway).
for i := 0; i <= blkEnd; i++ {
le.PutUint32(out.Bytes()[hdrStart+4*i:], offs[i])
}
return out.Bytes(), nil
})
}
// RISC-V relocation types (cmd/internal/objabi).
// RISC-V relocation types (cmd/internal/objabi). The Go linker applies
// R_RISCV_PCREL_ITYPE/STYPE to an AUIPC + I/S-type instruction pair as a
// single 8-byte field; R_RISCV_JAL covers a single 4-byte J-type instruction.
const (
relocRISCVPcrelHi20 = 23
relocRISCVPcrelLo12I = 24
relocRISCVPcrelLo12S = 25
relocRISCVJal = 59 // R_RISCV_JAL
relocRISCVPcrelItype = 62 // R_RISCV_PCREL_ITYPE
relocRISCVPcrelStype = 63 // R_RISCV_PCREL_STYPE
)
// toolchainObjectPreambleRISCV returns the RISC-V object preamble.
// toolchainObjectPreambleRISCV returns the "go object ...\n!\n" header
// the installed go tool asm writes for riscv64, captured by assembling a
// one-instruction probe (see toolchainObjectPreamble).
var (
preambleRISCVOnce sync.Once
preambleRISCV []byte
+326
View File
@@ -0,0 +1,326 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
// the symbol tables, the function code bytes and the relocation wiring.
func TestGOObjectLOONG64Structure(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
GLOBL ·table<>(SB), RODATA, $8
DATA ·table<>+0(SB)/8, $0x1122334455667788
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.GOObjectLOONG64("testpkg", "k_loong64.s")
if err != nil {
t.Fatalf("GOObjectLOONG64: %v", err)
}
v := openGoobj(t, obj)
// Package defs: the static GLOBL, then the FuncInfo and the two DWARF
// symbols (debug_line program, subprogram DIE).
defs := v.syms(blkSymdef)
if len(defs) != 4 {
t.Fatalf("symdefs = %d, want 4", len(defs))
}
if defs[0].name != "table" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 8 {
t.Errorf("table symbol = %+v", defs[0])
}
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
t.Errorf("funcinfo symbol = %+v", defs[1])
}
if defs[2].name != "" || defs[2].typ != kindSDWARFLINES || defs[2].size == 0 {
t.Errorf("lines symbol = %+v", defs[2])
}
if defs[3].name != "" || defs[3].typ != kindSDWARFFCN || defs[3].size == 0 {
t.Errorf("DIE symbol = %+v", defs[3])
}
// Non-package defs: four pc tables and the function.
nps := v.syms(blkNonpkgdef)
if len(nps) != 5 {
t.Fatalf("nonpkgdefs = %d, want 5", len(nps))
}
fn := nps[4]
if fn.name != "testpkg.add" || fn.typ != kindSTEXT || fn.flag != symFlagNoSplit || fn.size != 20 {
t.Errorf("add symbol = %+v", fn)
}
// The function code: 20 bytes, the ground-truth encoding. It sits
// after the GLOBL, FuncInfo, two DWARF symbols and four pc tables.
dataIdx := v.blk(blkDataIdx)
dataBlk := v.blk(blkData)
le := binary.LittleEndian
dOff := le.Uint32(dataIdx[8*4:])
code := dataBlk[dOff : dOff+20]
want := []byte{
0x64, 0x20, 0xc0, 0x28, // ld.d r4, 8(r3)
0x65, 0x40, 0xc0, 0x28, // ld.d r5, 16(r3)
0x84, 0x94, 0x10, 0x00, // add.d r4, r4, r5
0x64, 0x60, 0xc0, 0x29, // st.d r4, 24(r3)
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
}
for i := range want {
if code[i] != want[i] {
t.Fatalf("code byte %d = %02x, want %02x", i, code[i], want[i])
}
}
// The debug_line program: LNE_set_address (the R_ADDR relocation
// carries the function address), then one row per line change — the
// TEXT is on line 4 (a leading blank line precedes the include), the
// instructions on lines 5–9 — an advance to the 20-byte end and an
// end-of-sequence.
linesOff := le.Uint32(dataIdx[4*2:])
lines := dataBlk[linesOff : linesOff+21]
wantLines := []byte{
0x00, 0x09, 0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, // LNE_set_address
0x13, // pc 0, line 5
0x38, // pc 4, line 6
0x38, // pc 8, line 7
0x38, // pc 12, line 8
0x38, // pc 16, line 9
0x02, 0x04, // advance_pc to 20
0x00, 0x01, 0x01, // end_sequence
}
for i := range wantLines {
if lines[i] != wantLines[i] {
t.Fatalf("lines byte %d = %02x, want %02x", i, lines[i], wantLines[i])
}
}
// The subprogram DIE: abbrev 3 (FUNCTION), the qualified name, the
// addrx low_pc slot (R_DWTXTADDR_U4), the size as high_pc, the
// call-frame-CFA frame base, decl file/line and the external flag.
dieOff := le.Uint32(dataIdx[4*3:])
die := dataBlk[dieOff : dieOff+27]
wantDie := []byte{
0x03,
't', 'e', 's', 't', 'p', 'k', 'g', '.', 'a', 'd', 'd', 0,
0x00, 0x00, 0x00, 0x00, // low_pc: addrx slot
0x14, // high_pc: 20
0x01, 0x9c, // frame_base: DW_OP_call_frame_cfa
0x01, 0x00, 0x00, 0x00, // decl_file: 1
0x04, // decl_line: 4
0x01, // external
0x00, // end of children
}
for i := range wantDie {
if die[i] != wantDie[i] {
t.Fatalf("DIE byte %d = %02x, want %02x", i, die[i], wantDie[i])
}
}
// The DWARF symbols carry the function-address references: R_ADDR for
// the line program's set_address, R_DWTXTADDR_U4 for the DIE's addrx
// slot, both against the function's non-package index. The reloc
// index counts relocations, not bytes.
relocIdx := v.blk(blkRelocIdx)
relocs := v.blk(blkReloc)
if le.Uint32(relocIdx[4*2:]) != 0 || le.Uint32(relocIdx[4*3:]) != 1 || le.Uint32(relocIdx[4*4:]) != 2 {
t.Fatalf("dwarf reloc index ranges: %d %d %d", le.Uint32(relocIdx[4*2:]), le.Uint32(relocIdx[4*3:]), le.Uint32(relocIdx[4*4:]))
}
lr := relocs[:23]
if int32(le.Uint32(lr[0:])) != 3 || lr[4] != 8 || le.Uint16(lr[5:]) != relocAddr ||
le.Uint32(lr[15:]) != pkgIdxNone || le.Uint32(lr[19:]) != 4 {
t.Errorf("lines reloc = %x", lr)
}
dr := relocs[23:46]
if int32(le.Uint32(dr[0:])) != 13 || dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4() ||
le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 {
t.Errorf("die reloc = %x", dr)
}
// The pc-value deltas are in MinLC (4) units: the flat pcsp covers
// the whole 20-byte function with a delta of 5.
pcspOff := le.Uint32(dataIdx[4*4:])
if got := dataBlk[pcspOff : pcspOff+3]; !bytes.Equal(got, []byte{0x02, 0x05, 0x00}) {
t.Errorf("pcsp = %x, want 020500", got)
}
}
// TestGOObjectLOONG64Link cross-compiles a Go program with the gasm-produced
// object substituted into the package archive, proving cmd/link accepts the
// emitted GOOBJ. The binary is not executed (no LoongArch host or qemu).
// Skipped when no Go toolchain is available.
func TestGOObjectLOONG64Link(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
asmSrc := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
`
if err := os.WriteFile(filepath.Join(dir, "main_loong64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
mainSrc := `package main
func add(a, b int64) int64
func main() {
if add(20, 22) != 42 {
panic("bad add")
}
}
`
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module l64link\n\ngo 1.21\n"), 0o644); err != nil {
t.Fatal(err)
}
// Capture the cross build (GOARCH=loong64): the package archive and the
// link line.
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
build.Env = append(os.Environ(), "GOARCH=loong64")
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var pkgArch, work, linkLine, asmObj string
for _, line := range strings.Split(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_loong64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if pkgArch == "" || linkLine == "" || asmObj == "" {
t.Skip("could not locate the archive, asm output or link line in the build log")
}
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
// The archive member holding the assembler's output is named after the
// asm object file (main_loong64.o), as cmd/go packs it with `pack r`.
asmMember := filepath.Base(strings.ReplaceAll(asmObj, "$WORK", work))
// Assemble the same source with gasm and swap the object in.
pf, perrs := parser.Parse(filepath.Join(dir, "main_loong64.s"), asmSrc)
if len(perrs) > 0 {
t.Fatalf("parse: %v", perrs)
}
pimg, err := AssembleFileLOONG64(pf)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := pimg.GOObjectLOONG64("main", filepath.Join(dir, "main_loong64.s"))
if err != nil {
t.Fatalf("GOObjectLOONG64: %v", err)
}
// Extract the archive, substitute the object member, repack.
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
extract.Dir = membersDir
extract.Env = append(os.Environ(), "GOARCH=loong64")
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
// Substitute the gasm object for the assembler's archive member (pack
// extracts members read-only).
member := filepath.Join(membersDir, asmMember)
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, obj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listCmd.Env = append(os.Environ(), "GOARCH=loong64")
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for _, m := range strings.Fields(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, filepath.Join(membersDir, m))
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
pack.Env = append(os.Environ(), "GOARCH=loong64")
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
// Re-link with our archive in place of the toolchain's. The link line
// carries a GOROOT assignment and $WORK placeholders; run it through the
// shell with the GOEXPERIMENT and GOARCH the toolchain expects (the
// linker compares the object header against its own, experiments
// included).
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
link := exec.Command("sh", "-c", linkLine)
link.Dir = dir
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)), "GOARCH=loong64")
if out, err := link.CombinedOutput(); err != nil {
t.Fatalf("link with gasm object: %v\n%s", err, out)
}
// The binary is not executed: there is no LoongArch host or qemu here.
// The link itself and the symbol table prove cmd/link accepted the gasm
// object and laid out the function.
nm := exec.Command(goBin, "tool", "nm", filepath.Join(dir, "app2"))
nm.Env = append(os.Environ(), "GOARCH=loong64")
nmOut, err := nm.CombinedOutput()
if err != nil {
t.Fatalf("nm gasm-linked binary: %v\n%s", err, nmOut)
}
if !strings.Contains(string(nmOut), "main.add") {
t.Errorf("main.add not found in linked binary:\n%s", nmOut)
}
}
+83 -7
View File
@@ -89,11 +89,14 @@ func (fl *FuncLayout) LineAt(offset int) int {
type RelocKind int
const (
RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64)
RelPCRelHI20 // R_RISCV_PCREL_HI20 (AUIPC)
RelPCRelLO12 // R_RISCV_PCREL_LO12_I (ADDI, LD)
RelPCRelLO12S // R_RISCV_PCREL_LO12_S (SD)
RelPCRelAbs // 32-bit absolute (R_RISCV_32)
RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64)
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
RelRISCVJal // R_RISCV_JAL (J-type call)
RelPCRelAbs // 32-bit absolute (R_RISCV_32)
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
RelArm64Addr // R_ADDRARM64 (ADRP + ADD/LDR/STR pair)
)
type Reloc struct {
@@ -248,7 +251,7 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
if !ok {
continue
}
code, labels, relocs, err := assembleRISCV(t)
code, labels, relocs, lines, spadj, err := assembleRISCV(t)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
@@ -262,6 +265,8 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
Args: argsSize(t),
Line: t.Pos().Line,
Labels: labels,
Lines: lines,
Spadj: spadj,
Relocs: relocs,
}
for _, f := range t.Flags {
@@ -289,7 +294,78 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
img.DataSyms = append(img.DataSyms, DataSymbol{
Name: d.name,
Pkg: d.pkg,
Offset: pos,
Offset: len(img.Data) - len(d.buf), // relative to the data section
Size: d.size,
Static: d.static,
Rodata: d.rodata,
Dupok: d.dupok,
})
}
return img, nil
}
// AssembleFileLOONG64 assembles every TEXT function of a parsed loong64 file
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
// code. SB references in the code are encoded as pcalau12i pairs with zero
// immediates; the object-file emitters record R_LOONG64_ADDR_HI/LO
// relocations for the linker.
func AssembleFileLOONG64(f *ast.File) (*Image, error) {
dataSyms, err := collectData(f)
if err != nil {
return nil, err
}
img := &Image{Symbols: map[string]int{}}
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
continue
}
code, labels, relocs, lines, spadj, err := assembleLOONG64(t)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
fl := FuncLayout{
Name: t.Name.Name,
Pkg: t.Name.Pkg,
Static: t.Name.Static,
Offset: len(img.Code),
Size: len(code),
Frame: frameSize(t),
Args: argsSize(t),
Line: t.Pos().Line,
Labels: labels,
Lines: lines,
Spadj: spadj,
Relocs: relocs,
}
for _, f := range t.Flags {
switch f {
case "NOSPLIT":
fl.NoSplit = true
case "SPWRITE":
fl.SPWrite = true
}
}
img.Funcs = append(img.Funcs, fl)
img.Code = append(img.Code, code...)
}
// Lay out the data section behind the code, 16-aligned.
dataStart := len(img.Code)
for _, d := range dataSyms {
pos := dataStart + len(img.Data)
for pos%16 != 0 {
img.Data = append(img.Data, 0)
pos++
}
img.Symbols[d.name] = pos
img.Data = append(img.Data, d.buf...)
img.DataSyms = append(img.DataSyms, DataSymbol{
Name: d.name,
Pkg: d.pkg,
Offset: len(img.Data) - len(d.buf), // relative to the data section
Size: d.size,
Static: d.static,
Rodata: d.rodata,
File diff suppressed because it is too large Load Diff
+595
View File
@@ -0,0 +1,595 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// loong64 (LoongArch) instruction encoding.
//
// The encoder is data-driven: each mnemonic maps to an instruction format and
// an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own loong64
// backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm`
// exactly — the ground-truth oracle for the verify suite.
//
// All LoongArch instructions are 32 bits, little-endian. The formats used
// here (per the LoongArch Volume I specification):
//
// 3R opcode[31:15] | rk[4:0] | rj[4:0] | rd[4:0]
// 2R opcode[31:15] | rj[4:0] | rd[4:0]
// 2RI12 opcode[31:22] | si12[11:0] | rj[4:0] | rd[4:0]
// 2RI14 opcode[31:18] | si14[13:0] | rj[4:0] | rd[4:0]
// 2RI16 opcode[31:22] | si16[15:0] | rj[4:0] | rd[4:0]
// 2RI20 opcode[31:25] | si20[19:0] | rd[4:0]
// 1RI21 opcode[31:26] | si21[20:0] | rj[4:0] (BEQZ/BNEZ, B*Z, BC*Z)
// B/BL opcode[31:26] | offs[25:0]
// 4R opcode[31:20] | r1[4:0] | r2[4:0] | r3[4:0] | r4[4:0]
// IRIR opcode[31:22] | msb[4:0] | rj[4:0] | lsb[4:0] | rd[4:0]
// 3RI2 opcode[31:17] | sa2[1:0] | rk[4:0] | rj[4:0] | rd[4:0]
//
// The opcode constants are pre-positioned (they include the zero bit ranges
// of the immediate and register fields), mirroring the toolchain's OP_*
// helpers, so each l64* function only ORs its fields in.
// loong64RegNum returns the 5-bit register number for a LoongArch register
// name: R0–R31 (integer), F0–F31 (floating point), FCC0–FCC7 (condition
// flags), FCSR0–FCSR31 (control/status) and the ABI aliases the runtime's
// assembly uses. Returns -1 for an unrecognised name.
func loong64RegNum(name string) int {
switch name {
case "R0", "ZERO":
return 0
case "R1", "RA", "LINK":
return 1
case "R2", "TP":
return 2
case "R3", "SP":
return 3
case "R4", "A0":
return 4
case "R5", "A1":
return 5
case "R6", "A2":
return 6
case "R7", "A3":
return 7
case "R8", "A4":
return 8
case "R9", "A5":
return 9
case "R10", "A6":
return 10
case "R11", "A7":
return 11
case "R12", "T0":
return 12
case "R13", "T1":
return 13
case "R14", "T2":
return 14
case "R15", "T3":
return 15
case "R16", "T4":
return 16
case "R17", "T5":
return 17
case "R18", "T6":
return 18
case "R19", "T7":
return 19
case "R20", "T8":
return 20
case "R21":
return 21
case "R22", "G", "g", "FP":
return 22
case "R23", "S0":
return 23
case "R24", "S1":
return 24
case "R25", "S2":
return 25
case "R26", "S3":
return 26
case "R27", "S4":
return 27
case "R28", "S5":
return 28
case "R29", "S6", "CTXT":
return 29
case "R30", "S7", "TMP":
return 30
case "R31", "S8":
return 31
}
// F0–F31, FCC0–FCC7, FCSR0–FCSR31.
if len(name) >= 4 && name[:4] == "FCSR" {
return loong64RegSpecial(name[4:], "FCSR", 31)
}
if len(name) >= 3 && name[:3] == "FCC" {
return loong64RegSpecial(name[3:], "FCC", 7)
}
if len(name) < 2 {
return -1
}
prefix, digits := name[:1], name[1:]
if digits[0] < '0' || digits[0] > '9' {
return -1
}
n := 0
for i := 0; i < len(digits); i++ {
if digits[i] < '0' || digits[i] > '9' {
return -1
}
n = n*10 + int(digits[i]-'0')
}
if prefix == "F" && n <= 31 {
return n
}
return -1
}
// loong64RegSpecial parses a numbered FCC/FCSR register.
func loong64RegSpecial(digits, prefix string, max int) int {
if digits == "" {
return -1
}
n := 0
for i := 0; i < len(digits); i++ {
if digits[i] < '0' || digits[i] > '9' {
return -1
}
n = n*10 + int(digits[i]-'0')
}
if n <= max {
return n
}
return -1
}
// ---- format helpers ----
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
func l64rrr(op uint32, rk, rj, rd int) uint32 {
return op | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64rr encodes a 2R instruction: op | rj<<5 | rd.
func l64rr(op uint32, rj, rd int) uint32 {
return op | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64irr encodes a 2RI12 instruction: op | si12<<10 | rj<<5 | rd.
func l64irr(op uint32, imm, rj, rd int) uint32 {
return op | (uint32(imm)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64irr14 encodes a 2RI14 instruction: op | si14<<10 | rj<<5 | rd.
func l64irr14(op uint32, imm, rj, rd int) uint32 {
return op | (uint32(imm)&0x3FFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64irr16 encodes a 2RI16 instruction: op | si16<<10 | rj<<5 | rd.
func l64irr16(op uint32, imm, rj, rd int) uint32 {
return op | (uint32(imm)&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64ir encodes a 2RI20 instruction: op | si20<<5 | rd.
func l64ir(op uint32, imm, rd int) uint32 {
return op | (uint32(imm)&0xFFFFF)<<5 | uint32(rd&0x1f)
}
// l64bbl encodes a B/BL instruction: op | offs[25:0], where offs is the
// 4-byte-aligned word distance (the toolchain stores the shifted value).
func l64bbl(op uint32, offs int) uint32 {
return op | (uint32(offs)&0xFFFF)<<10 | (uint32(offs)>>16)&0x3FF
}
// l64ir21 encodes a 1RI21 branch (BEQZ/BNEZ, BLTZ/BGEZ/BLEZ/BGTZ, BFPT/BFPF):
// op | si21[15:0]<<10 | rj<<5 | si21[20:16].
func l64ir21(op uint32, offs, rj int) uint32 {
v := uint32(offs)
return op | (v&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | (v>>16)&0x1F
}
// l64rrrr encodes a 4R instruction: op | r1<<15 | r2<<10 | r3<<5 | r4.
func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
return op | uint32(r1&0x1f)<<15 | uint32(r2&0x1f)<<10 | uint32(r3&0x1f)<<5 | uint32(r4&0x1f)
}
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
// The msb/lsb fields are 6 bits wide (0–63) and are validated by the caller.
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
}
// l64irrr encodes a 3RI2 instruction (ALSL): op | sa<<15 | rk<<10 | rj<<5 | rd.
func l64irrr(op uint32, sa, rk, rj, rd int) uint32 {
return op | uint32(sa&0x3)<<15 | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64i15 encodes a no-operand system instruction with a 15-bit code field
// (SYSCALL, BREAK, DBAR): op | code[14:0].
func l64i15(op uint32, code int) uint32 {
return op | uint32(code)&0x7FFF
}
// l64irr5i encodes PRELD: op | offs<<10 | rj<<5 | hint.
func l64irr5i(op uint32, offs, rj, hint int) uint32 {
return op | (uint32(offs)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(hint&0x1f)
}
// l64wordLE encodes a uint32 as 4 little-endian bytes.
func l64wordLE(w uint32) []byte {
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
}
// l64WordsLE concatenates one or more instruction words as little-endian bytes.
func l64WordsLE(ws ...uint32) []byte {
var out []byte
for _, w := range ws {
out = append(out, l64wordLE(w)...)
}
return out
}
// ---- instruction formats ----
type l64Format uint8
const (
l64Frrr l64Format = iota // 3R (integer and FP arithmetic)
l64Frr // 2R
l64Firr // 2RI12 (arithmetic with 12-bit immediate)
l64Firr14 // 2RI14 (ldptr/stptr)
l64Firr16 // 2RI16 (addu16i.d)
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub)
l64Firir // bstrins/bstrpick
l64Firrr // alsl
l64Fi15 // syscall/break/dbar
l64Fam // atomic (3R with the AM field order)
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
l64Fshift // 2RI12 with a 5/6-bit shift immediate
l64Fpreld // preld (2RI12 + 5-bit hint)
)
// l64Enc is one instruction's encoding: its bit layout (format) and the
// opcode constant, positioned at its exact bit range.
type l64Enc struct {
format l64Format
op uint32
}
// l64DualEnc holds both forms of a dual-form mnemonic: the 3R register form
// and the 2RI12 immediate form (which is a shift for the shift mnemonics).
type l64DualEnc struct {
rrr uint32 // 3R register form
imm uint32 // 2RI12 immediate form
shift bool // the immediate form is a 5/6-bit shift amount
}
// l64DualTable maps the dual-form arithmetic/logic mnemonics to both
// encodings; the assembler picks by operand kind.
var l64DualTable = map[string]l64DualEnc{}
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered
// yet; the base integer, memory and floating-point ISA is complete.
var l64InstrTable = map[string]l64Enc{}
func init() {
// 3R — integer.
rrr := map[string]uint32{
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
"SGT": 0x24 << 15, "SGTU": 0x25 << 15,
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15,
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
"ORN": 0x2c << 15, "ANDN": 0x2d << 15,
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
"SLLV": 0x31 << 15, "SRLV": 0x32 << 15, "SRAV": 0x33 << 15,
"ROTR": 0x36 << 15, "ROTRV": 0x37 << 15,
"MUL": 0x38 << 15, "MULW": 0x38 << 15, "MULH": 0x39 << 15, "MULHU": 0x3a << 15,
"MULV": 0x3b << 15, "MULVU": 0x3b << 15, "MULHV": 0x3c << 15, "MULHVU": 0x3d << 15,
"MULWVW": 0x3e << 15, "MULWVWU": 0x3f << 15,
"DIV": 0x40 << 15, "DIVW": 0x40 << 15, "REM": 0x41 << 15, "REMW": 0x41 << 15,
"DIVU": 0x42 << 15, "DIVWU": 0x42 << 15, "REMU": 0x43 << 15, "REMWU": 0x43 << 15,
"DIVV": 0x44 << 15, "REMV": 0x45 << 15, "DIVVU": 0x46 << 15, "REMVU": 0x47 << 15,
"CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15,
"CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15,
}
// 3R — floating point.
rrr["MULF"] = 0x209 << 15
rrr["MULD"] = 0x20a << 15
rrr["DIVF"] = 0x20d << 15
rrr["DIVD"] = 0x20e << 15
rrr["SUBF"] = 0x205 << 15
rrr["SUBD"] = 0x206 << 15
rrr["ADDF"] = 0x201 << 15
rrr["ADDD"] = 0x202 << 15
rrr["CMPEQF"] = 0x0c1<<20 | 0x4<<15
rrr["CMPEQD"] = 0x0c2<<20 | 0x4<<15
rrr["CMPGED"] = 0x0c2<<20 | 0x7<<15
rrr["CMPGEF"] = 0x0c1<<20 | 0x7<<15
rrr["CMPGTD"] = 0x0c2<<20 | 0x3<<15
rrr["CMPGTF"] = 0x0c1<<20 | 0x3<<15
rrr["FMINF"] = 0x215 << 15
rrr["FMIND"] = 0x216 << 15
rrr["FMAXF"] = 0x211 << 15
rrr["FMAXD"] = 0x212 << 15
rrr["FMAXAF"] = 0x219 << 15
rrr["FMAXAD"] = 0x21a << 15
rrr["FMINAF"] = 0x21d << 15
rrr["FMINAD"] = 0x21e << 15
rrr["FSCALEBF"] = 0x221 << 15
rrr["FSCALEBD"] = 0x222 << 15
rrr["FCOPYSGF"] = 0x225 << 15
rrr["FCOPYSGD"] = 0x226 << 15
for m, op := range rrr {
l64InstrTable[m] = l64Enc{format: l64Frrr, op: op}
}
// 2R.
rr := map[string]uint32{
"CLOW": 0x4 << 10, "CLZW": 0x5 << 10, "CTOW": 0x6 << 10, "CTZW": 0x7 << 10,
"CLOV": 0x8 << 10, "CLZV": 0x9 << 10, "CTOV": 0xa << 10, "CTZV": 0xb << 10,
"REVB2H": 0xc << 10, "REVB4H": 0xd << 10, "REVB2W": 0xe << 10, "REVBV": 0xf << 10,
"REVH2W": 0x10 << 10, "REVHV": 0x11 << 10,
"BITREV4B": 0x12 << 10, "BITREV8B": 0x13 << 10, "BITREVW": 0x14 << 10, "BITREVV": 0x15 << 10,
"EXTWH": 0x16 << 10, "EXTWB": 0x17 << 10, "CPUCFG": 0x1b << 10,
"TRUNCFV": 0x46a9 << 10, "TRUNCDV": 0x46aa << 10, "TRUNCFW": 0x46a1 << 10, "TRUNCDW": 0x46a2 << 10,
"MOVWF": 0x4744 << 10, "MOVVF": 0x4746 << 10, "MOVWD": 0x4748 << 10, "MOVVD": 0x474a << 10,
"MOVFW": 0x46c1 << 10, "MOVDW": 0x46c2 << 10, "MOVFV": 0x46c9 << 10, "MOVDV": 0x46ca << 10,
"FRINTF": 0x4791 << 10, "FRINTD": 0x4792 << 10,
"MOVDF": 0x4646 << 10, "MOVFD": 0x4649 << 10,
"ABSF": 0x4501 << 10, "ABSD": 0x4502 << 10,
"MOVF": 0x4525 << 10, "MOVD": 0x4526 << 10,
"NEGF": 0x4505 << 10, "NEGD": 0x4506 << 10,
"SQRTF": 0x4511 << 10, "SQRTD": 0x4512 << 10,
"FLOGBF": 0x4509 << 10, "FLOGBD": 0x450a << 10,
"FCLASSF": 0x450d << 10, "FCLASSD": 0x450e << 10,
"FTINTRMWF": 0x4681 << 10, "FTINTRMWD": 0x4682 << 10,
"FTINTRMVF": 0x4689 << 10, "FTINTRMVD": 0x468a << 10,
"FTINTRPWF": 0x4691 << 10, "FTINTRPWD": 0x4692 << 10,
"FTINTRPVF": 0x4699 << 10, "FTINTRPVD": 0x469a << 10,
"FTINTRZWF": 0x46a1 << 10, "FTINTRZWD": 0x46a2 << 10,
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
}
for m, op := range rr {
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
}
// RDTIME is a 2R instruction with rd and rj in swapped positions.
l64InstrTable["RDTIMELW"] = l64Enc{format: l64Frdtime, op: 0x18 << 10}
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
// selected by the operand kind; the shift mnemonics pair the 3R form
// with a 5/6-bit shift immediate.
for m, e := range map[string]l64DualEnc{
"ADD": {rrr: 0x20 << 15, imm: 0x00a << 22},
"ADDW": {rrr: 0x20 << 15, imm: 0x00a << 22},
"ADDV": {rrr: 0x21 << 15, imm: 0x00b << 22},
"ADDVU": {rrr: 0x21 << 15, imm: 0x00b << 22},
"AND": {rrr: 0x29 << 15, imm: 0x00d << 22},
"OR": {rrr: 0x2a << 15, imm: 0x00e << 22},
"XOR": {rrr: 0x2b << 15, imm: 0x00f << 22},
"SGT": {rrr: 0x24 << 15, imm: 0x008 << 22},
"SGTU": {rrr: 0x25 << 15, imm: 0x009 << 22},
"SLL": {rrr: 0x2e << 15, imm: 0x00081 << 15, shift: true},
"SRL": {rrr: 0x2f << 15, imm: 0x00089 << 15, shift: true},
"SRA": {rrr: 0x30 << 15, imm: 0x00091 << 15, shift: true},
"ROTR": {rrr: 0x36 << 15, imm: 0x00099 << 15, shift: true},
"SLLV": {rrr: 0x31 << 15, imm: 0x0041 << 16, shift: true},
"SRLV": {rrr: 0x32 << 15, imm: 0x0045 << 16, shift: true},
"SRAV": {rrr: 0x33 << 15, imm: 0x0049 << 16, shift: true},
"ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true},
} {
l64DualTable[m] = e
}
// 2RI12 — pure immediate arithmetic (LU52ID has no register form).
l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22}
// ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16.
l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26}
// 2RI14 — LL/SC are aliased by the Go assembler to the pointer loads and
// stores (ldptr/stptr), with the offset scaled by 4.
l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w
l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d
l64InstrTable["SC"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w
l64InstrTable["SCW"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w
l64InstrTable["SCV"] = l64Enc{format: l64Firr14, op: 0x23 << 24} // sc.d
l64InstrTable["LL"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w)
l64InstrTable["LLW"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w)
l64InstrTable["LLV"] = l64Enc{format: l64Firr14, op: 0x22 << 24} // ldptr.d (ll.d)
// 2RI20.
l64InstrTable["LU12IW"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
l64InstrTable["LU32ID"] = l64Enc{format: l64Fir20, op: 0x0b << 25}
l64InstrTable["PCALAU12I"] = l64Enc{format: l64Fir20, op: 0x0d << 25}
l64InstrTable["PCADDU12I"] = l64Enc{format: l64Fir20, op: 0x0e << 25}
// LUI is the Plan 9 spelling of lu12i.w.
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
// 4R — fused multiply-add.
rrrr := map[string]uint32{
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
}
for m, op := range rrrr {
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
}
// IRIR — bit-field insert/extract.
irir := map[string]uint32{
"BSTRINSW": 0x3<<21 | 0x0<<15,
"BSTRINSV": 0x2 << 22,
"BSTRPICKW": 0x3<<21 | 0x1<<15,
"BSTRPICKV": 0x3 << 22,
}
for m, op := range irir {
l64InstrTable[m] = l64Enc{format: l64Firir, op: op}
}
// 3RI2 — ALSL.
irrr := map[string]uint32{
"ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17,
}
for m, op := range irrr {
l64InstrTable[m] = l64Enc{format: l64Firrr, op: op}
}
// 0-operand system instructions.
l64InstrTable["SYSCALL"] = l64Enc{format: l64Fi15, op: 0x56 << 15}
l64InstrTable["BREAK"] = l64Enc{format: l64Fi15, op: 0x54 << 15}
l64InstrTable["DBAR"] = l64Enc{format: l64Fi15, op: 0x70e4 << 15}
// PRELD.
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
// Atomics — 3R with the AM field order (rk=value, rj=address, rd=result).
am := map[string]uint32{
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
"AMCASB": 0x070B0 << 15, "AMCASH": 0x070B1 << 15,
"AMCASW": 0x070B2 << 15, "AMCASV": 0x070B3 << 15,
"AMADDW": 0x070C2 << 15, "AMADDV": 0x070C3 << 15,
"AMANDW": 0x070C4 << 15, "AMANDV": 0x070C5 << 15,
"AMORW": 0x070C6 << 15, "AMORV": 0x070C7 << 15,
"AMXORW": 0x070C8 << 15, "AMXORV": 0x070C9 << 15,
"AMMAXW": 0x070CA << 15, "AMMAXV": 0x070CB << 15,
"AMMINW": 0x070CC << 15, "AMMINV": 0x070CD << 15,
"AMMAXWU": 0x070CE << 15, "AMMAXVU": 0x070CF << 15,
"AMMINWU": 0x070D0 << 15, "AMMINVU": 0x070D1 << 15,
"AMSWAPDBB": 0x070BC << 15, "AMSWAPDBH": 0x070BD << 15,
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
}
for m, op := range am {
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
}
}
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
// register move between the integer and floating-point register banks — the
// MOVW/MOVV specials the Go assembler accepts.
var l64FpMovTable = map[string]uint32{
"MOVV.R.F": 0x452a << 10, // movgr2fr.d
"MOVV.R.FCC": 0x4536 << 10, // movgr2cf
"MOVV.R.FCSR": 0x4530 << 10, // movgr2fcsr
"MOVV.F.R": 0x452e << 10, // movfr2gr.d
"MOVV.F.FCC": 0x4534 << 10, // movfr2cf
"MOVV.FCC.R": 0x4537 << 10, // movcf2gr
"MOVV.FCC.F": 0x4535 << 10, // movcf2fr
"MOVV.FCSR.R": 0x4532 << 10, // movfcsr2gr
"MOVW.R.F": 0x4529 << 10, // movgr2fr.w
"MOVW.F.R": 0x452d << 10, // movfr2gr.s
}
// l64branchTable holds the 16-bit branch and jump encodings (2RI16).
var l64branchTable = map[string]uint32{
"BEQ": 0x16 << 26,
"BNE": 0x17 << 26,
"BLT": 0x18 << 26,
"BGE": 0x19 << 26,
"BLTU": 0x1a << 26,
"BGEU": 0x1b << 26,
"JIRL": 0x13 << 26,
}
// l64branch21Table holds the single-register branches with 21-bit offsets:
// the negative opcode constants the toolchain uses for the short forms.
var l64branch21Table = map[string]uint32{
"BEQZ": 0x10 << 26, // beq r0, rj → beqz
"BNEZ": 0x11 << 26, // bne r0, rj → bnez
"BLTZ": 0x18 << 26, // blt rj, r0 → bltz
"BGEZ": 0x19 << 26, // bge rj, r0 → bgez
"BGTZ": 0x18 << 26, // blt r0, rj → bgtz
"BLEZ": 0x19 << 26, // bge r0, rj → blez
"BFPT": 0x12<<26 | 0x1<<8,
"BFPF": 0x12<<26 | 0x0<<8,
}
// l64jumpTable maps the jump pseudo-instructions and their aliases to the
// B/BL opcode constants.
var l64jumpTable = map[string]uint32{
"JMP": 0x14 << 26, // b
"B": 0x14 << 26, // b
"JAL": 0x15 << 26, // bl
"CALL": 0x15 << 26, // bl
"BL": 0x15 << 26, // bl
}
// l64loadStoreTable maps the MOV width mnemonics to their load and store
// 2RI12 opcodes. The load opcode is the negated store opcode, exactly as
// the toolchain derives it.
var l64loadStoreTable = map[string]struct{ ld, st uint32 }{
"MOVB": {0x0a0 << 22, 0x0a4 << 22},
"MOVH": {0x0a1 << 22, 0x0a5 << 22},
"MOVW": {0x0a2 << 22, 0x0a6 << 22},
"MOVV": {0x0a3 << 22, 0x0a7 << 22},
"MOVBU": {0x0a8 << 22, 0x0a4 << 22},
"MOVHU": {0x0a9 << 22, 0x0a5 << 22},
"MOVWU": {0x0aa << 22, 0x0a6 << 22},
"MOVF": {0x0ac << 22, 0x0ad << 22},
"MOVD": {0x0ae << 22, 0x0af << 22},
}
// l64movRegTable maps a register-to-register MOV mnemonic to its expansion,
// matching the toolchain's case-1 encoding: MOVB → ext.w.b, MOVH → ext.w.h,
// MOVW → sll.w, MOVV → or, MOVBU → andi. MOVHU/MOVWU expand to bstrpick.d
// and are handled separately in the assembler.
type l64MovRegEnc struct {
rr bool // 2R format (ext.w.b/ext.w.h)
op uint32 // opcode constant (rr forms) or 3R/2RI12 opcode
imm int // 2RI12 immediate for MOVBU's andi
}
var l64movRegTable = map[string]l64MovRegEnc{
"MOVB": {true, 0x17 << 10, 0}, // ext.w.b rd, rj
"MOVH": {true, 0x16 << 10, 0}, // ext.w.h rd, rj
"MOVW": {false, 0x2e << 15, 0}, // sll.w rd, rj, r0
"MOVV": {false, 0x2a << 15, 0}, // or rd, rj, r0
"MOVBU": {false, 0x00d << 22, 0xff}, // andi rd, rj, $0xff
}
// l64movFpRegTable maps a floating-point register move mnemonic to its 2R
// opcode (fmov.s / fmov.d), used when both operands are F registers.
var l64movFpRegTable = map[string]uint32{
"MOVF": 0x4525 << 10,
"MOVD": 0x4526 << 10,
}
// l64RegClass discriminates integer (R), floating-point (F) and condition
// (FCC) registers for the MOV pseudo-instruction's register-move encoding.
type l64RegClass int
const (
l64ClsNone l64RegClass = iota
l64ClsGR
l64ClsFP
l64ClsFCC
l64ClsFCSR
)
// loong64RegClass reports the register class of a register operand name.
func loong64RegClass(name string) l64RegClass {
switch {
case name == "":
return l64ClsNone
case len(name) >= 3 && name[:3] == "FCC":
return l64ClsFCC
case len(name) >= 4 && name[:4] == "FCSR":
return l64ClsFCSR
case name[0] == 'F':
return l64ClsFP
default:
return l64ClsGR
}
}
+293
View File
@@ -0,0 +1,293 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// firstTextLOONG64 parses assembly source and returns the first TEXT body.
func firstTextLOONG64(t *testing.T, src string) *ast.Text {
t.Helper()
f, errs := parser.Parse("f_loong64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
for _, d := range f.Decls {
if fn, ok := d.(*ast.Text); ok {
return fn
}
}
t.Fatal("no TEXT found")
return nil
}
// assembleLOONG64Helper assembles one TEXT function and returns its bytes.
func assembleLOONG64Helper(t *testing.T, fn *ast.Text) []byte {
t.Helper()
code, _, _, _, _, err := assembleLOONG64(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
return code
}
// wantWords checks that code matches the expected little-endian words.
func wantWords(t *testing.T, code []byte, want ...uint32) {
t.Helper()
got := make([]uint32, 0, len(code)/4)
for i := 0; i+4 <= len(code); i += 4 {
got = append(got, binary.LittleEndian.Uint32(code[i:]))
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
func TestLOONG64_add(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
`)
code := assembleLOONG64Helper(t, fn)
// 5 instructions: two ld.d, add.d, st.d, jirl r0, r1, 0.
wantWords(t, code,
0x28C02064, // ld.d r4, 8(r3)
0x28C04065, // ld.d r5, 16(r3)
0x00109484, // add.d r4, r4, r5
0x29C06064, // st.d r4, 24(r3)
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_arithmetic(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·arith(SB), NOSPLIT, $0
ADDV R4, R5, R6
SUBV R7, R8, R9
MULV R10, R11, R12
DIVV R13, R14, R15
AND R16, R17, R18
OR R18, R19, R20
XOR R20, R21, R2
SLLV R2, R23, R24
SRLV R24, R25, R26
SRAV R26, R27, R28
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x001090A6, // add.d r6, r5, r4
0x00119D09, // sub.d r9, r8, r7
0x001DA96C, // mul.d r12, r11, r10
0x002235CF, // div.d r15, r14, r13
0x0014C232, // and r18, r17, r16
0x00154A74, // or r20, r19, r18
0x0015D2A2, // xor r2, r21, r20
0x00188AF8, // sll.d r24, r23, r2
0x0019633A, // srl.d r26, r25, r24
0x0019EB7C, // sra.d r28, r27, r26
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_immediates(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·imm(SB), NOSPLIT, $0
ADDV $42, R4, R5
ADDV $-8, R6
AND $0xff, R7, R8
OR $1, R9, R10
SGT $100, R13, R14
SLLV $4, R15, R16
MOVV $0x12345, R17
MOVV $0, R18
MOVW $0, R19
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x02C0A885, // addi.d r5, r4, 42
0x02FFE0C6, // addi.d r6, r6, -8
0x0343FCE8, // andi r8, r7, 0xff
0x0380052A, // ori r10, r9, 1
0x020191AE, // slti r14, r13, 100
0x004111F0, // slli.d r16, r15, 4
0x14000251, // lu12i.w r17, 0x12
0x038D1631, // ori r17, r17, 0x345
0x00150012, // or r18, r0, r0
0x00170013, // sll.w r19, r0, r0
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_loadStore(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·mem(SB), NOSPLIT, $0
MOVV (R4), R5
MOVV R5, (R6)
MOVW 8(R7), R8
MOVB R9, -4(R10)
MOVV (R11)(R12), R13
MOVV R14, (R15)(R16)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x28C00085, // ld.d r5, 0(r4)
0x29C000C5, // st.d r5, 0(r6)
0x288020E8, // ld.w r8, 8(r7)
0x293FF149, // st.b r9, -4(r10)
0x380C316D, // ldx.d r13, r11, r12
0x381C41EE, // stx.d r14, r15, r16
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_branches(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·br(SB), NOSPLIT, $0
BEQ R4, R5, done
BNE R6, R7, skip
BLT R8, R9, done
BGE R10, R11, done
BLTU R12, R13, done
BGEU R14, R15, done
skip:
JMP done
done:
RET
`)
code := assembleLOONG64Helper(t, fn)
// skip is at 0x18 (6 words), done at 0x1c.
wantWords(t, code,
0x58001C85, // beq r5, r4, +7
0x5C0018C7, // bne r7, r6, +6
0x60001509, // blt r9, r8, +5
0x6400114B, // bge r11, r10, +4
0x68000D8D, // bltu r13, r12, +3
0x6C0009CF, // bgeu r15, r14, +2
0x50000400, // b done (+1, chain-folded through skip)
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_frame(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $32-8
MOVV R4, R5
MOVV arg+0(FP), R6
MOVV R7, local-8(SP)
MOVV local-8(SP), R8
MOVV R9, ret+0(FP)
RET
`)
code := assembleLOONG64Helper(t, fn)
// autosize = align8(32+8) = 40; prologue stores LR at -40(SP),
// opens the frame, stores LR again at 0(SP). The function is a leaf
// (no calls), so the epilogue skips the LR restore. FP args are at
// autosize+8; SP locals at autosize+offset.
wantWords(t, code,
0x29FF6061, // st.d r1, -40(r3)
0x02FF6063, // addi.d r3, r3, -40
0x29C00061, // st.d r1, 0(r3)
0x00150085, // or r5, r4, r0
0x28C0C066, // ld.d r6, 48(r3) arg+0(FP) → 0+40+8
0x29C08067, // st.d r7, 32(r3) local-8(SP) → 40-8
0x28C08068, // ld.d r8, 32(r3)
0x29C0C069, // st.d r9, 48(r3) ret+0(FP) → 0+40+8
0x02C0A063, // addi.d r3, r3, 40
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_jumpChain(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·jc(SB), NOSPLIT, $0
JMP a
a:
JMP b
b:
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x50000800, // b +2 (a, chain-folded to b)
0x50000400, // b +1 (b)
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_dconClasses(t *testing.T) {
cases := []struct {
v int64
word int // expected word count
}{
{0x123456789, 3}, // lu12i.w + ori + lu32i.d
{-1, 2}, // addi.d + lu52i.d (the MOV path handles -1 earlier)
{0x1000000000000, 2}, // addi.w + lu32i.d
{0x123456789abcdef0, 4}, // full sequence
{0xFFFFFFFFF, 2}, // lu12i.w + ori
{0x1234567800000000, 3}, // addi.w + lu32i.d + lu52i.d
}
for _, c := range cases {
if n := len(l64DconMovWords(0, c.v)); n != c.word {
t.Errorf("0x%x: %d words, want %d", c.v, n, c.word)
}
}
}
func TestLOONG64_regNames(t *testing.T) {
cases := map[string]int{
"R0": 0, "R31": 31, "F0": 0, "F31": 31, "FCC0": 0, "FCC7": 7,
"FCSR0": 0, "FCSR3": 3, "ZERO": 0, "RA": 1, "SP": 3, "g": 22, "G": 22,
"R32": -1, "FCC8": -1, "X0": -1, "R": -1, "TMP": 30, "CTXT": 29,
}
for name, want := range cases {
if got := loong64RegNum(name); got != want {
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
}
}
}
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
// A spot-check that assembleLOONG64 emits the same bytes the Go
// toolchain does for a small kernel (the full comparison lives in
// verify's TestGroundTruthLOONG64).
src := `#include "textflag.h"
TEXT ·k(SB), NOSPLIT, $0-0
ADDV R4, R5, R6
MOVV $0x100000, R7
BEQ R6, R7, done
JMP done
done:
RET
`
fn := firstTextLOONG64(t, src)
code := assembleLOONG64Helper(t, fn)
want := []byte{
0xa6, 0x90, 0x10, 0x00, // add.d r6, r5, r4
0x07, 0x20, 0x00, 0x14, // lu12i.w r7, 0x100
0xc7, 0x08, 0x00, 0x58, // beq r7, r6, +2 (done)
0x00, 0x04, 0x00, 0x50, // b +1 (done)
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
+129
View File
@@ -0,0 +1,129 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// Loong64 frame mapping, matching the Go toolchain's loong64 backend.
//
// Go's loong64 functions have no frame pointer: FP and SP are synthetic
// registers resolved against the hardware stack pointer (R3) and the frame
// size. The return address lives in R1 (the link register).
//
// The autosize is the real stack adjustment: the declared local frame plus
// the 8 bytes for the saved link register, rounded up to a multiple of 8
// (the toolchain aligns frames with `if autosize&4 != 0 { autosize += 4 }`).
// A leaf function (no calls) with a zero frame gets no prologue at all.
//
// Prologue (autosize > 0), byte-identical to the toolchain:
//
// MOVV R1, -autosize(R3) // save LR below the new SP (traceback-safe)
// ADDV $-autosize, R3 // open the frame
// MOVV R1, 0(R3) // save LR again at SP (signal-safety)
//
// Epilogue: MOVV 0(R3), R1; ADDV $autosize, R3 (non-leaf only for the LR
// restore); the RET's jirl r0, r1, 0 follows.
// loong64FrameInfo holds the frame layout derived from a TEXT directive.
type loong64FrameInfo struct {
autosize int // the real SP adjustment (locals + saved LR, aligned)
frame int // the declared $framesize
args int // the declared -argsize
noSplit bool // the NOSPLIT flag
leaf bool // no call instructions in the body
}
// loong64ComputeFrame derives the frame layout for a TEXT function.
func loong64ComputeFrame(t *ast.Text) loong64FrameInfo {
fi := loong64FrameInfo{
frame: frameSize(t),
args: argsSize(t),
}
for _, f := range t.Flags {
if f == "NOSPLIT" {
fi.noSplit = true
}
}
fi.leaf = loong64IsLeaf(t)
if fi.frame != 0 {
fi.autosize = fi.frame + 8 // space for the saved LR
if fi.autosize&4 != 0 {
fi.autosize += 4
}
} else if !fi.leaf {
// A zero-frame non-leaf function still opens an 8-byte frame for LR.
fi.autosize = 8
}
return fi
}
// loong64IsLeaf reports whether a function contains no call instructions
// (JAL/BL/CALL), matching the toolchain's LEAF mark, which drives the frame
// and the epilogue shape.
func loong64IsLeaf(t *ast.Text) bool {
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "JAL", "CALL", "BL":
return false
}
}
return true
}
// loong64Prologue returns the prologue bytes for a loong64 function.
func loong64Prologue(fi loong64FrameInfo) []byte {
if fi.autosize == 0 {
return nil
}
addiD := l64DualTable["ADDV"].imm
return l64WordsLE(
l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, 3, 1), // MOVV R1, -autosize(R3)
l64irr(addiD, -fi.autosize, 3, 3), // ADDV $-autosize, R3
l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1), // MOVV R1, 0(R3)
)
}
// loong64Return returns the bytes for a RET: the epilogue (restore LR and
// deallocate the frame when present) followed by jirl r0, r1, 0.
func loong64Return(fi loong64FrameInfo) []byte {
var ws []uint32
if fi.autosize != 0 {
if !fi.leaf {
// MOVV 0(R3), R1 — restore the link register.
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1))
}
// ADDV $autosize, R3 — close the frame.
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
}
// jirl r0, r1, 0 — return.
ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0))
return l64WordsLE(ws...)
}
// loong64ResolvePseudo translates a pseudo-register memory reference into a
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
// x-N(SP) → (autosize - N)(SP). Returns base = -1 for an unresolvable
// reference (SB: static data, handled by the relocation path).
func loong64ResolvePseudo(sym *ast.Symbol, fi loong64FrameInfo) (base int, off int32) {
if sym == nil {
return -1, 0
}
switch sym.Pseudo {
case "FP":
return 3, int32(sym.Offset) + int32(fi.autosize) + 8
case "SP":
return 3, int32(fi.autosize) + int32(sym.Offset)
case "SB":
return -1, int32(sym.Offset)
}
return -1, 0
}
+367
View File
@@ -0,0 +1,367 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestLOONG64_sys exercises the no-operand system instructions and the
// bare-data pseudo-instructions. The words match `go tool asm`
// (GOARCH=loong64) for the same source.
func TestLOONG64_sys(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·sys(SB), NOSPLIT, $0
NOOP
UNDEF
WORD $0x12345678
SYSCALL $0x10
BREAK $0x20
DBAR $1
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x03400000, // andi r0, r0, 0 (NOOP)
0x002A0000, // break 0 (UNDEF)
0x12345678, // WORD
0x002B0010, // syscall 0x10
0x002A0020, // break 0x20
0x38720001, // dbar 1
0x4C000020, // jirl r0, r1, 0
)
}
// TestLOONG64_branches21 exercises the single-register branch forms: the
// 21-bit BEQZ/BNEZ/BLTZ/BGEZ and the rd-field BGTZ/BLEZ.
func TestLOONG64_branches21(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·b21(SB), NOSPLIT, $0
BEQZ R4, done
BNEZ R5, done
BLTZ R6, done
BGEZ R7, done
BGTZ R8, done
BLEZ R9, done
done:
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x40001880, // beqz r4, +6
0x440014A0, // bnez r5, +5
0x600010C0, // bltz r6, +4
0x64000CE0, // bgez r7, +3
0x60000808, // bgtz r8, +2 (register in the rd field)
0x64000409, // blez r9, +1
0x4C000020, // jirl r0, r1, 0
)
}
// TestLOONG64_fma exercises the four fused multiply-add forms (4 and 3
// operand spellings).
func TestLOONG64_fma(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·fma(SB), NOSPLIT, $0
FMADDD F0, F1, F2, F3
FMSUBD F4, F5, F6
FNMADDD F7, F8, F9, F10
FNMSUBD F11, F12, F13
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x08200443, // fmadd.d f3, f2, f1, f0
0x086214C6, // fmsub.d f6, f5, f5, f4
0x08A3A12A, // fnmadd.d f10, f9, f8, f7
0x08E5B1AD, // fnmsub.d f13, f12, f12, f11
0x4C000020,
)
}
// TestLOONG64_bitops exercises BSTRINS/BSTRPICK (the 6-bit msb/lsb fields)
// and ALSL (the sa−1 shift field).
func TestLOONG64_bitops(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·bits(SB), NOSPLIT, $0
BSTRINSW $3, R4, $0, R5
BSTRINSV $3, R4, $1, R6
BSTRPICKW $3, R4, $0, R5
BSTRPICKV $6, R7, $0, R8
ALSLW $1, R4, R5, R6
ALSLW $4, R7, R8, R9
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x00630085, // bstrins.w r5, r4, $3, $0
0x00830486, // bstrins.d r6, r4, $3, $1
0x00638085, // bstrpick.w r5, r4, $3, $0
0x00C600E8, // bstrpick.d r8, r7, $6, $0
0x00041486, // alsl.w r6, r5, r4, $1 (sa-1)
0x0005A0E9, // alsl.w r9, r8, r7, $4
0x4C000020,
)
}
// TestLOONG64_ptr exercises the 14-bit-offset memory forms (LL/SC/MOVWP/
// MOVVP with the offset scaled by 4) and PRELD.
func TestLOONG64_ptr(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·ptr(SB), NOSPLIT, $0
LLW 8(R14), R15
SCW R16, -4(R17)
MOVWP 16(R18), R19
MOVVP R20, 24(R21)
PRELD 32(R22), $0
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x200009CF, // ll.w r15, 8(r14)
0x21FFFE30, // sc.w r16, -4(r17)
0x24001253, // ldptr.w r19, 16(r18)
0x27001AB4, // stptr.d r20, 24(r21)
0x2AC082C0, // preld 32(r22), 0
0x4C000020,
)
}
// TestLOONG64_atomics exercises the AM* read-modify-write forms and
// RDTIME, plus the MOVV FP→GP move.
func TestLOONG64_atomics(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·atoms(SB), NOSPLIT, $0
AMADDW R4, (R5), R6
RDTIMED R7, R8
MOVV F1, R2
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x386110A6, // amadd.w r6, r5, r4
0x000068E8, // rdtime.d r8, r7
0x0114B822, // movfr2gr.d r2, f1
0x4C000020,
)
}
// TestLOONG64_lu52 exercises the LU52I.D immediate form (a gasm extension
// the toolchain reaches only through its MOVV expansion).
func TestLOONG64_lu52(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·lu52(SB), NOSPLIT, $0
LU52ID $0x345, R10
LU52ID $0x123, R11, R12
ADDV16 $0x10000, R13
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x030D154A, // lu52i.d r10, r10, 0x345
0x03048D6C, // lu52i.d r12, r11, 0x123
0x100005AD, // addu16i.d r13, r13, 0x10000>>16
0x4C000020,
)
}
// TestLOONG64_sbRefs checks the static-symbol reference forms through the
// full file assembly: each pcalau12i+addi.d/ld/st pair carries the
// R_LOONG64_ADDR_HI/LO relocation pair, and the immediate fields are left
// zero for the linker.
func TestLOONG64_sbRefs(t *testing.T) {
f, errs := parser.Parse("sb_loong64.s", `#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0
MOVV $·table(SB), R4
MOVV ·table+8(SB), R5
MOVV R6, ·table(SB)
RET
GLOBL ·table(SB), RODATA, $8
DATA ·table+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
fn := img.Funcs[0]
if fn.Size != 28 {
t.Fatalf("function size = %d, want 28", fn.Size)
}
var hi, lo int
// The three references: $·table (0), ·table+8 (8), ·table (0).
wantAdd := []int64{0, 0, 8, 8, 0, 0}
for i, r := range fn.Relocs {
wantKind := RelLoong64AddrHi
wantOff := (i / 2) * 8
if i%2 == 1 {
wantKind = RelLoong64AddrLo
wantOff += 4
}
if r.Kind != wantKind || r.Off != wantOff || r.Name != "table" || r.Addend != wantAdd[i] {
t.Errorf("reloc %d = {kind %v off %d name %q addend %d}", i, r.Kind, r.Off, r.Name, r.Addend)
}
if r.Kind == RelLoong64AddrHi {
hi++
} else {
lo++
}
}
if hi != 3 || lo != 3 {
t.Errorf("relocs = %d hi + %d lo, want 3 + 3", hi, lo)
}
// The image carries the zero-immediate pair encodings (the linker
// fills the immediate fields from the relocations).
code := img.Code[fn.Offset : fn.Offset+fn.Size]
wantWords(t, code,
0x1A000004, // pcalau12i r4, 0
0x02C00084, // addi.d r4, r4, 0
0x1A00001E, // pcalau12i r30, 0
0x28C003C5, // ld.d r5, 0(r30)
0x1A00001E, // pcalau12i r30, 0
0x29C003C6, // st.d r6, 0(r30)
0x4C000020, // jirl r0, r1, 0
)
}
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
// invalid register operands and operand-count mismatches.
func TestLOONG64_errors(t *testing.T) {
cases := []string{
`TEXT ·e(SB), NOSPLIT, $0
JMP nowhere
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
BEQZ X0, done
done:
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
ADDV R4
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
FMADDD F0, F1
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
AMADDW R4, R5
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
WORD
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
PRELD 32(R4)
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
ALSLW $5, R4, R5, R6
RET
`,
}
for i, src := range cases {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_pcsp checks the stack-adjustment table of a framed function:
// the prologue raises the SP delta by autosize (in effect from the third
// instruction) and the RET's epilogue restores it to zero, with the pc deltas
// in MinLC (4) units — byte-identical to `go tool asm`.
func TestLOONG64_pcsp(t *testing.T) {
cases := []struct {
name string
src string
want []byte
}{
{
"leaf",
`#include "textflag.h"
TEXT ·leaf(SB), NOSPLIT, $8-0
MOVV R4, R5
RET
`,
[]byte{0x02, 0x02, 0x20, 0x03, 0x1f, 0x01, 0x00},
},
{
"nonleaf",
`#include "textflag.h"
TEXT ·nonleaf(SB), NOSPLIT, $8-0
MOVV R4, R5
JAL (R12)
RET
`,
[]byte{0x02, 0x02, 0x20, 0x05, 0x1f, 0x01, 0x00},
},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
f, errs := parser.Parse("pcsp_loong64.s", c.src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
if got := pcspTable(img.Funcs[0], 4); !bytes.Equal(got, c.want) {
t.Errorf("pcsp = % x, want % x", got, c.want)
}
})
}
}
// TestLOONG64_sbRefsUndefined checks that a reference to a symbol no GLOBL
// defines assembles into a relocation and is rejected at object emission.
func TestLOONG64_sbRefsUndefined(t *testing.T) {
f, errs := parser.Parse("sb_loong64.s", `#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0
MOVV missing(SB), R4
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
if len(img.Funcs[0].Relocs) != 2 {
t.Fatalf("relocs = %d, want the HI/LO pair", len(img.Funcs[0].Relocs))
}
if _, err := img.GOObjectLOONG64("p", "sb_loong64.s"); err == nil {
t.Error("expected an unknown-symbol error at emission")
}
}
// TestLOONG64_movImmToFp checks the immediate-to-FP move forms.
func TestLOONG64_movImmToFp(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·fpmov(SB), NOSPLIT, $0
MOVV $0x1, F0
MOVW $0x2, F4
RET
`)
code := assembleLOONG64Helper(t, fn)
want := []byte{
0x00, 0x04, 0x80, 0x03, // ori f0, r0, 1
0x04, 0x08, 0x80, 0x03, // ori f4, r0, 2
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
-258
View File
@@ -1,258 +0,0 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// This file emits Mach-O x86-64 objects (MH_OBJECT) from an assembled
// Image, in the shape the Darwin assembler produces: one unnamed segment
// carrying a __TEXT,__text and a __DATA,__data section laid out back to
// back at addresses zero and len(code), a symbol table (locals first, then
// exported definitions, then undefined externals) and one relocation entry
// per static-symbol reference, of type X86_64_RELOC_SIGNED.
//
// The image's own address space carries straight over — the data section
// starts immediately after the code, and the layout padding already lives
// inside Image.Data — so every symbol keeps its image address as its
// n_value, and a local (non-external) relocation leaves the displacement
// the assembler resolved in place: the linker only adjusts it by the
// section's final movement.
// Mach-O constants.
const (
machoMagic64 = 0xfeedfacf
machoCPUamd64 = 0x01000007 // CPU_TYPE_X86_64
machoCPUSubAll = 3 // CPU_SUBTYPE_X86_64_ALL
machoObj = 1 // MH_OBJECT
machoSegment64 = 0x19 // LC_SEGMENT_64
machoSymtab = 0x2 // LC_SYMTAB
machoSectTextFlags = 0x80000400 // S_ATTR_PURE_INSTRUCTIONS | S_ATTR_SOME_INSTRUCTIONS
nUndf = 0x00 // undefined symbol
nSect = 0x0e // defined in section number n_sect
nExt = 0x01 // external (exported or undefined-global) bit
x8664RelocSigned = 1
)
// MachOObject returns the image as a Mach-O x86-64 relocatable object
// (MH_OBJECT), the shape the Darwin toolchain links. Symbol names follow
// the same rules as the ELF output. Every static-symbol reference becomes
// an X86_64_RELOC_SIGNED relocation: external references against their
// undefined symbol, file-local ones against the __DATA section with the
// resolved displacement carried in the instruction bytes.
func (img *Image) MachOObject() ([]byte, error) {
le := binary.LittleEndian
// Section ordinals (1-based, as Mach-O numbers them).
const (
sectText = 1
sectData = 2
)
// Object address space: code at 0, data immediately after (the layout
// padding is already part of img.Data, so image addresses are object
// addresses).
textAddr := uint64(0)
dataAddr := uint64(len(img.Code))
vmsize := dataAddr + uint64(len(img.Data))
// The code, with external displacements primed to addend − 4: the
// linker adds the symbol's address to the field as it stands. Local
// displacements stay as the assembler resolved them.
code := append([]byte(nil), img.Code...)
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.External {
// Prime the field to the addend measured from the patch
// site: the assembler records it from the instruction end,
// After − Off bytes past the field.
copy(code[fn.Offset+r.Off:], le32(r.Addend-int64(r.After-r.Off)))
}
}
}
// Symbols: locals first, then exported definitions, then undefined
// externals — the order the classic link editor expects.
type machoSym struct {
name string
typ byte
sect byte
value uint64
}
var locals, globals, undefs []machoSym
for _, fn := range img.Funcs {
s := machoSym{name: objectName(fn.Pkg, fn.Name), typ: nSect, sect: sectText, value: textAddr + uint64(fn.Offset)}
if fn.Static {
locals = append(locals, s)
} else {
s.typ |= nExt
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := machoSym{name: objectName(d.Pkg, d.Name), typ: nSect, sect: sectData, value: dataAddr + uint64(d.Offset)}
if d.Static {
locals = append(locals, s)
} else {
s.typ |= nExt
globals = append(globals, s)
}
}
for _, name := range img.Externals {
undefs = append(undefs, machoSym{name: name, typ: nUndf | nExt})
}
syms := append(append(locals, globals...), undefs...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Relocations, attached to the __text section.
type machoReloc struct {
addr uint32
symnum uint32
extern bool
}
var relocs []machoReloc
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
rel := machoReloc{addr: uint32(fn.Offset + r.Off)}
if r.External {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
rel.symnum = uint32(idx)
rel.extern = true
} else {
// Section-relative: r_symbolnum carries the section number
// and the resolved displacement stays in the bytes.
rel.symnum = sectData
}
relocs = append(relocs, rel)
}
}
// The string table opens with the conventional " \0".
strtab := []byte{' ', 0}
strOff := map[string]int{}
for _, s := range syms {
if _, ok := strOff[s.name]; ok {
continue
}
strOff[s.name] = len(strtab)
strtab = append(strtab, s.name...)
strtab = append(strtab, 0)
}
// File layout: header, the two load commands, section data (code,
// data), the relocation table, the symbol table, the string table.
const (
hdrSize = 32
segCmdSize = 72 + 2*80 // segment command with two sections
symCmdSize = 24
)
sizeofcmds := segCmdSize + symCmdSize
dataOff := hdrSize + sizeofcmds
reloff := dataOff + len(code) + len(img.Data)
symoff := reloff + 8*len(relocs)
stroff := symoff + 16*len(syms)
out := make([]byte, stroff+len(strtab))
// mach_header_64.
le.PutUint32(out[0:], machoMagic64)
le.PutUint32(out[4:], machoCPUamd64)
le.PutUint32(out[8:], machoCPUSubAll)
le.PutUint32(out[12:], machoObj)
le.PutUint32(out[16:], 2) // ncmds
le.PutUint32(out[20:], uint32(sizeofcmds))
le.PutUint32(out[24:], 0) // flags
le.PutUint32(out[28:], 0) // reserved
// LC_SEGMENT_64 with the two sections.
p := hdrSize
le.PutUint32(out[p:], machoSegment64)
le.PutUint32(out[p+4:], segCmdSize)
// segname: the empty string, zero-padded to 16 bytes.
le.PutUint64(out[p+8:], 0)
le.PutUint64(out[p+16:], 0)
le.PutUint64(out[p+24:], 0) // vmaddr
le.PutUint64(out[p+32:], vmsize)
le.PutUint64(out[p+40:], uint64(dataOff))
le.PutUint64(out[p+48:], vmsize)
le.PutUint32(out[p+56:], 7) // maxprot rwx
le.PutUint32(out[p+60:], 7) // initprot rwx
le.PutUint32(out[p+64:], 2) // nsects
le.PutUint32(out[p+68:], 0) // flags
// __TEXT,__text
s := p + 72
copy(out[s:], "__text")
copy(out[s+16:], "__TEXT")
le.PutUint64(out[s+32:], textAddr)
le.PutUint64(out[s+40:], uint64(len(code)))
le.PutUint32(out[s+48:], uint32(dataOff))
le.PutUint32(out[s+52:], 4) // align 2^4
le.PutUint32(out[s+56:], uint32(reloff))
le.PutUint32(out[s+60:], uint32(len(relocs)))
le.PutUint32(out[s+64:], machoSectTextFlags)
// __DATA,__data
s += 80
copy(out[s:], "__data")
copy(out[s+16:], "__DATA")
le.PutUint64(out[s+32:], dataAddr)
le.PutUint64(out[s+40:], uint64(len(img.Data)))
le.PutUint32(out[s+48:], uint32(dataOff+len(code)))
le.PutUint32(out[s+52:], 4) // align 2^4
// LC_SYMTAB.
p = hdrSize + segCmdSize
le.PutUint32(out[p:], machoSymtab)
le.PutUint32(out[p+4:], symCmdSize)
le.PutUint32(out[p+8:], uint32(symoff))
le.PutUint32(out[p+12:], uint32(len(syms)))
le.PutUint32(out[p+16:], uint32(stroff))
le.PutUint32(out[p+20:], uint32(len(strtab)))
// Section data.
copy(out[dataOff:], code)
copy(out[dataOff+len(code):], img.Data)
// Relocation entries.
for i, r := range relocs {
e := out[reloff+i*8:]
le.PutUint32(e[0:], r.addr)
bits := r.symnum & 0x00ffffff
bits |= 1 << 24 // r_pcrel
bits |= 2 << 25 // r_length = 4 bytes
if r.extern {
bits |= 1 << 27 // r_extern
}
bits |= x8664RelocSigned << 28
le.PutUint32(e[4:], bits)
}
// nlist_64 entries.
for i, s := range syms {
e := out[symoff+i*16:]
le.PutUint32(e[0:], uint32(strOff[s.name]))
e[4] = s.typ
e[5] = s.sect
le.PutUint16(e[6:], 0) // n_desc
le.PutUint64(e[8:], s.value)
}
// String table.
copy(out[stroff:], strtab)
return out, nil
}
-127
View File
@@ -1,127 +0,0 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/macho"
"encoding/binary"
"testing"
)
// TestMachOObject checks the structure of the emitted MH_OBJECT: the two
// sections and their addresses, the symbol table (types, sections, values)
// and the __text relocation entries, parsed back with debug/macho. No
// Darwin toolchain is available on the test hosts, so the check is
// structural — the ELF output carries the end-to-end link-and-run proof of
// the shared symbol and relocation model.
func TestMachOObject(t *testing.T) {
img := elfTestImage(t)
obj, err := img.MachOObject()
if err != nil {
t.Fatalf("MachOObject: %v", err)
}
f, err := macho.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer f.Close()
if f.Type != macho.TypeObj {
t.Errorf("file type = %v, want MH_OBJECT", f.Type)
}
if f.Cpu != macho.CpuAmd64 {
t.Errorf("cpu = %v, want CpuAmd64", f.Cpu)
}
text := f.Section("__text")
data := f.Section("__data")
if text == nil || data == nil {
t.Fatal("missing __text or __data section")
}
if text.Addr != 0 || text.Size != uint64(len(img.Code)) {
t.Errorf("__text addr/size = %#x/%d, want 0/%d", text.Addr, text.Size, len(img.Code))
}
if data.Addr != uint64(len(img.Code)) {
t.Errorf("__data addr = %#x, want %#x", data.Addr, len(img.Code))
}
// Symbol table: locals, exported definitions, undefined externals.
syms := f.Symtab.Syms
byName := map[string]macho.Symbol{}
for _, s := range syms {
byName[s.Name] = s
}
wantSym := func(name string, typ, sect uint8, value uint64) {
t.Helper()
s, ok := byName[name]
if !ok {
t.Errorf("symbol %q not found", name)
return
}
if s.Type != typ || s.Sect != sect || s.Value != value {
t.Errorf("%s: type/sect/value = %#x/%d/%#x, want %#x/%d/%#x",
name, s.Type, s.Sect, s.Value, typ, sect, value)
}
}
const (
defined = nSect | nExt
local = nSect
undefined = nUndf | nExt
)
wantSym("addq", defined, 1, 0)
wantSym("getanswer", defined, 1, 5)
wantSym("useextern", defined, 1, 13)
answer := byName["answer"]
if answer.Type != local || answer.Sect != 2 {
t.Errorf("answer: type/sect = %#x/%d, want %#x/2", answer.Type, answer.Sect, local)
}
wantSym("extvar", undefined, 0, 0)
// Relocations: both X86_64_RELOC_SIGNED, PC-relative, 4 bytes wide.
// The local one carries its section number in Value, the external one
// its symbol number.
if len(text.Relocs) != 2 {
t.Fatalf("__text relocs = %d, want 2", len(text.Relocs))
}
var sawLocal, sawExternal bool
for _, r := range text.Relocs {
if !r.Pcrel || r.Len != 2 || r.Type != x8664RelocSigned {
t.Errorf("reloc at %#x: pcrel/len/type = %v/%d/%d", r.Addr, r.Pcrel, r.Len, r.Type)
}
switch {
case r.Extern:
if name := syms[r.Value].Name; name != "extvar" {
t.Errorf("external reloc at %#x names %q, want extvar", r.Addr, name)
}
sawExternal = true
default:
if r.Value != 2 { // __data, the second section
t.Errorf("local reloc at %#x: section %d, want 2 (__data)", r.Addr, r.Value)
}
sawLocal = true
}
}
if !sawLocal || !sawExternal {
t.Errorf("relocs seen: local=%v external=%v, want both", sawLocal, sawExternal)
}
// The __text bytes are the image code, with the external displacement
// primed to addend − 4 and the local one left resolved.
textData, err := text.Data()
if err != nil {
t.Fatal(err)
}
want := append([]byte(nil), img.Code...)
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.Name == "extvar" {
binary.LittleEndian.PutUint32(want[fn.Offset+r.Off:], 0xfffffffc) // −4
}
}
}
if !bytes.Equal(textData, want) {
t.Errorf("__text bytes %x, want %x", textData, want)
}
}
+386 -149
View File
@@ -5,17 +5,25 @@ package asm
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// assembleRISCV assembles a RISC-V TEXT function body into machine code.
// It handles the full RV64IMAFDC instruction set including RVC compression.
func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, error) {
func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
fi := riscvComputeFrame(t)
prologue := riscvPrologue(fi)
var relocs []Reloc
var spadj []SpadjStep
// The prologue raises the SP delta by autosize; the boundary is reported
// at the pc just past its ADDI, exactly as the toolchain's pctospadj does.
if fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: riscvPrologueSpadjPC(fi), Value: fi.autosize})
}
// Pass 1: collect instructions and compute label offsets assuming 4 bytes
// per instruction (or 8 for MOV $large-imm). No encoding yet.
@@ -33,7 +41,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, error) {
offsets[s.Name.Text] = pos
case *ast.Instr:
recs = append(recs, instrRec{instr: s})
pos += riscvInstrSize(s)
pos += riscvInstrSize(s, fi)
}
}
@@ -42,7 +50,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, error) {
for i := range recs {
code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil) // no relocs in Pass 2
if err != nil {
return nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err)
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err)
}
recs[i].code = code
pc += len(code)
@@ -78,36 +86,52 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, error) {
out := append([]byte(nil), prologue...)
pc = len(prologue)
preCount := len(relocs)
var lines []LineEntry
for _, r := range recs {
lines = append(lines, LineEntry{Offset: pc, Line: r.instr.Pos().Line})
if r.compressed && !isBranchLike(r.instr.Mnemonic.Text) {
out = append(out, r.code...)
pc += len(r.code)
} else {
code, err := encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs)
if err != nil {
return nil, nil, nil, err
return nil, nil, nil, nil, nil, err
}
if c16, ok := tryCompressRVC(r.instr, fi); ok {
code = []byte{byte(c16), byte(c16 >> 8)}
}
// Make newly added relocation offsets absolute (subtract prologue to make
// them function-relative, then the caller adds fn.Offset).
// Make newly added relocation offsets function-relative. Each
// instruction records its reloc offset relative to its own start;
// the current pc is that instruction's offset from the function
// start (which includes the prologue). After is the address just
// past the relocated field, shifted by the same amount.
for j := preCount; j < len(relocs); j++ {
relocs[j].Off += pc - len(prologue)
relocs[j].Off += pc
relocs[j].After += pc
}
preCount = len(relocs)
// The RET's epilogue closes the frame: the SP delta returns to zero
// after its ADDI (restore LR + ADDI).
if strings.ToUpper(r.instr.Mnemonic.Text) == "RET" && fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: pc + riscvReturnEpilogueLen(fi), Value: 0})
}
out = append(out, code...)
pc += len(code)
}
}
return out, offsets, relocs, nil
return out, offsets, relocs, lines, spadj, nil
}
// riscvInstrSize returns the encoded size in bytes of a RISC-V instruction.
// Most instructions are 4 bytes; MOV with a large immediate is 8 (LUI+ADDIW).
func riscvInstrSize(instr *ast.Instr) int {
// Most instructions are 4 bytes; MOV with a large immediate and I-type
// arithmetic with a large immediate expand to several (possibly compressed)
// instructions.
func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
mnem := instr.Mnemonic.Text
ops := instr.Operands
if mnem == "RET" {
return len(riscvReturn(fi))
}
if mnem == "MOV" && len(ops) == 2 {
// MOV $sym(SB), rd → 8 bytes (AUIPC + ADDI).
if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil && ops[0].Imm.Sym.Pseudo == "SB" {
@@ -121,14 +145,15 @@ func riscvInstrSize(instr *ast.Instr) int {
if isMemOperand(ops[1]) && ops[1].Addr.Sym != nil && ops[1].Addr.Sym.Pseudo == "SB" {
return 8
}
// MOV $imm, rd → large immediate needs LUI+ADDIW.
if isImmOperand(ops[0]) {
imm := immFromOperand(ops[0])
if imm < -2048 || imm > 2047 {
return 8
}
// MOV $imm, rd → size depends on the immediate and RVC compression.
if isImmOperand(ops[0]) && ops[0].Imm.Sym == nil {
return riscvMovImmSize(regFromOperand(ops[1]), immFromOperand(ops[0]))
}
}
// I-type arithmetic with a large immediate expands to several instructions.
if (mnem == "ADDI" || mnem == "ANDI" || mnem == "ORI" || mnem == "XORI") && len(ops) >= 1 && isImmOperand(ops[0]) {
return riscvItypeImmediateSize(mnem, immFromOperand(ops[0]))
}
return 4
}
@@ -151,36 +176,27 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
// Handle pseudo-instructions and special cases first.
switch mnem {
case "RET":
// RET = JALR X0, 0(X1)
word = riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
// RET = epilogue (restore LR and close the frame when present) +
// uncompressed JALR X0, 0(X1) (the toolchain never compresses RET).
return riscvReturn(fi), nil
case "CALL":
// CALL target → AUIPC X1, %pcrel_hi + JALR X1, %pcrel_lo(X1).
// For now, emit AUIPC X1, 0 + JALR X1, 0(X1) with zero offsets.
// The relocation system will fill the actual offsets.
if len(ops) >= 1 {
target := labelFromOperand(ops[0])
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
offset := int32(targetOff - pc)
// AUIPC X1, upper 20 bits
hi := (offset + 0x800) >> 12
word1 := riscvUType(riscvEnc{0x17, 0x0, 0x00}, 1, hi<<12)
// JALR X1, lower 12 bits(X1)
lo := offset - (hi << 12)
word2 := riscvIType(riscvEnc{0x67, 0x0, 0x00}, 1, 1, lo)
var out []byte
out = append(out, byte(word1), byte(word1>>8), byte(word1>>16), byte(word1>>24))
out = append(out, byte(word2), byte(word2>>8), byte(word2>>16), byte(word2>>24))
return out, nil
// CALL sym(SB) → JAL X1, sym(SB) with a single R_RISCV_JAL
// relocation. The Go assembler rejects CALL to a local branch label.
if len(ops) != 1 {
return nil, fmt.Errorf("CALL expects 1 operand, got %d", len(ops))
}
// CALL with no target: encode as NOP (unsupported).
word = riscvIType(riscvEnc{0x13, 0x0, 0x00}, 0, 0, 0)
op := ops[0]
if op.Addr.Sym == nil || op.Addr.Sym.Pseudo != "SB" {
return nil, fmt.Errorf("CALL: local branch target is not supported (use CALL sym(SB))")
}
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelRISCVJal, Addend: op.Addr.Sym.Offset})
}
word = riscvJType(1, 0) // JAL X1, 0 — the linker fills the offset
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
case "JMP":
// JMP = JAL X0, target. Try C.J compression.
// JMP = JAL X0, target. The Go assembler never compresses this to
// C.J, so always emit the 32-bit JAL.
var target string
if len(ops) >= 1 {
target = labelFromOperand(ops[0])
@@ -190,11 +206,6 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
offset := int32(targetOff - pc)
// C.J: funct3=0x5, offset in ±2 KB, bit 0 must be 0.
if offset >= -2048 && offset <= 2046 && offset%2 == 0 {
c16 := rvcCJ(0x5, offset)
return []byte{byte(c16), byte(c16 >> 8)}, nil
}
word = riscvJType(0, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
case "JAL":
@@ -211,11 +222,6 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
offset := int32(targetOff - pc)
// JAL X0, target → C.J when offset fits.
if rd == 0 && offset >= -2048 && offset <= 2046 && offset%2 == 0 {
c16 := rvcCJ(0x5, offset)
return []byte{byte(c16), byte(c16 >> 8)}, nil
}
word = riscvJType(rd, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
@@ -304,26 +310,44 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
}
switch {
// R-type: Plan 9 order is INSTR src1, src2, dst (destination last).
// R-type: Go reverses the ISA order, writing rs2, rs1, rd (destination
// last); the two-operand form INSTR rs2, rd uses rd as rs1.
case len(ops) == 3 && isRTypeInstr(mnem):
rs1 := regFromOperand(ops[0]) // source 1 (first operand)
rs2 := regFromOperand(ops[1]) // source 2 (second operand)
rs2 := regFromOperand(ops[0]) // first operand = rs2
rs1 := regFromOperand(ops[1]) // second operand = rs1
rd := regFromOperand(ops[2]) // destination (last operand)
if rd < 0 || rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
word = riscvRType(enc, rd, rs1, rs2)
// I-type shift (SLLI, SRLI, SRAI): INSTR rs, $shamt, rd.
case len(ops) == 2 && isRTypeInstr(mnem):
rs2 := regFromOperand(ops[0]) // source (first operand)
rd := regFromOperand(ops[1]) // destination (second operand)
if rd < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
word = riscvRType(enc, rd, rd, rs2)
// I-type shift (SLLI, SRLI, SRAI): INSTR $shamt, rs1, rd; the two-operand
// form INSTR $shamt, rd uses rd as the source.
case len(ops) == 3 && isShiftImmInstr(mnem):
rs1 := regFromOperand(ops[0])
shamt := int(immFromOperand(ops[1]))
shamt := int(immFromOperand(ops[0]))
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
word = riscvRType(enc, rd, rs1, shamt)
case len(ops) == 2 && isShiftImmInstr(mnem):
shamt := int(immFromOperand(ops[0]))
rd := regFromOperand(ops[1])
if rd < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
word = riscvRType(enc, rd, rd, shamt)
// AMO atomics: Plan 9 order is INSTR src, (addr), dst.
case len(ops) == 3 && isAMOInstr(mnem):
rs2 := regFromOperand(ops[0]) // source value
@@ -334,10 +358,10 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
}
word = riscvAMOType(enc, rd, rs1, rs2)
// FP arithmetic: Plan 9 order is INSTR src1, src2, dst.
// FP arithmetic: Go reverses the ISA order, writing rs2, rs1, rd.
case len(ops) == 3 && isFPArithInstr(mnem):
rs1 := regFromOperand(ops[0])
rs2 := regFromOperand(ops[1])
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd < 0 || rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid FP register in %s", mnem)
@@ -390,25 +414,34 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
}
word = riscvAMOType(enc, rd, rs1, rs2)
// FP compare: INSTR src1, src2, dst(int) — result in integer register.
// FP compare: Go reverses the ISA order, writing rs2, rs1, rd.
case len(ops) == 3 && isFPCmpInstr(mnem):
rs1 := regFromOperand(ops[0])
rs2 := regFromOperand(ops[1])
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd < 0 || rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
word = riscvRType(enc, rd, rs1, rs2)
// I-type with immediate: Plan 9 order is INSTR src, imm, dst.
// I-type with immediate: Plan 9 order is INSTR $imm, rs1, rd; the
// two-operand form INSTR $imm, rd uses rd as the source.
case len(ops) == 3 && isITypeInstr(mnem):
rs1 := regFromOperand(ops[0]) // source register
imm := immFromOperand(ops[1]) // immediate
imm := immFromOperand(ops[0]) // immediate
rs1 := regFromOperand(ops[1]) // source register
rd := regFromOperand(ops[2]) // destination
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
word = riscvIType(enc, rd, rs1, imm)
return encodeRISCVItypeImmediate(mnem, enc, rd, rs1, imm)
case len(ops) == 2 && isITypeInstr(mnem):
imm := immFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rd < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
return encodeRISCVItypeImmediate(mnem, enc, rd, rd, imm)
// Loads: rd, offset(rs1) — Plan 9 order is LD src, dst.
case len(ops) == 2 && isLoadInstr(mnem):
@@ -442,18 +475,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return nil, fmt.Errorf("invalid register in %s", mnem)
}
// Try C.BEQZ / C.BNEZ compression.
if (mnem == "BEQ" || mnem == "BNE") && rs2 == 0 && isRVCIntReg(rs1) {
if cOff := offset; cOff >= -256 && cOff <= 254 && cOff%2 == 0 {
funct3 := uint32(0x6) // C.BEQZ
if mnem == "BNE" {
funct3 = 0x7 // C.BNEZ
}
c16 := rvcCB(funct3, rvcReg3(rs1), offset)
return []byte{byte(c16), byte(c16 >> 8)}, nil
}
}
// The Go assembler never compresses branches to C.BEQZ/C.BNEZ.
word = riscvBType(enc, rs1, rs2, offset)
// U-type: rd, imm.
@@ -585,62 +607,200 @@ func encodeRISCVMov(instr *ast.Instr, offsets map[string]int, fi riscvFrameInfo,
}
}
// encodeRISCVLoadImm encodes loading an immediate into a register.
// For 12-bit immediates: ADDI $imm, ZERO, rd.
// For larger: LUI $hi, rd + ADDIW $lo, rd, rd.
// encodeRISCVLoadImm encodes loading an immediate into a register (MOV $imm,
// rd), matching the toolchain's instructionsForMOVConst. For 12-bit
// immediates it emits ADDI $imm, ZERO, rd (compressed to C.LI when it fits
// six signed bits); for larger immediates it emits LUI + [ADDIW], with the LUI
// and ADDIW compressed to C.LUI / C.ADDIW when their immediate fits.
func encodeRISCVLoadImm(rd int, imm int32) []byte {
if imm >= -2048 && imm <= 2047 {
word := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, 0, imm)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
if rd != 0 && imm >= -32 && imm <= 31 {
return word16(rvcCI(0x2, uint32(rd), uint32(imm)&0x3F)) // C.LI
}
return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, 0, imm))
}
// LUI + ADDIW for larger constants.
low, high := splitRISCV32Imm(imm)
var out []byte
hi := int32((uint32(imm)+0x800)>>12) << 12 // LUI loads upper 20 bits
lo := imm - hi
wordLUI := riscvUType(riscvEnc{0x37, 0x0, 0x00}, rd, hi)
out = append(out, byte(wordLUI), byte(wordLUI>>8), byte(wordLUI>>16), byte(wordLUI>>24))
if lo != 0 {
wordADDIW := riscvIType(riscvEnc{0x1B, 0x0, 0x00}, rd, rd, lo)
out = append(out, byte(wordADDIW), byte(wordADDIW>>8), byte(wordADDIW>>16), byte(wordADDIW>>24))
if rd != 0 && rd != 2 && high >= -32 && high <= 31 {
out = append(out, word16(rvcCI(0x3, uint32(rd), uint32(high)&0x3F))...) // C.LUI
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, rd, high<<12))...)
}
if low != 0 {
if low >= -32 && low <= 31 {
out = append(out, word16(rvcCI(0x1, uint32(rd), uint32(low)&0x3F))...) // C.ADDIW
} else {
out = append(out, wordLE(riscvIType(riscvEnc{0x1B, 0x0, 0x00}, rd, rd, low))...)
}
}
return out
}
// riscvMovImmSize returns the encoded byte length of MOV $imm, rd, mirroring
// encodeRISCVLoadImm's expansion and compression.
func riscvMovImmSize(rd int, imm int32) int {
if imm >= -2048 && imm <= 2047 {
if rd != 0 && imm >= -32 && imm <= 31 {
return 2 // C.LI
}
return 4 // ADDI
}
low, high := splitRISCV32Imm(imm)
size := 0
if rd != 0 && rd != 2 && high >= -32 && high <= 31 {
size += 2 // C.LUI
} else {
size += 4 // LUI
}
if low != 0 {
if low >= -32 && low <= 31 {
size += 2 // C.ADDIW
} else {
size += 4 // ADDIW
}
}
return size
}
// splitRISCV32Imm splits a signed 32-bit immediate into a signed 12-bit low
// part and a signed 20-bit high part, mirroring cmd/internal/obj/riscv's
// Split32BitImmediate. The high part is returned unshifted; callers place it
// in the upper bits of LUI (or its compressed C.LUI form).
func splitRISCV32Imm(imm int32) (low, high int32) {
if imm >= -2048 && imm <= 2047 {
return imm, 0
}
h := int64(imm) >> 12
if imm&(1<<11) != 0 {
h++
}
low = int32((int64(imm) << 52) >> 52) // sign extend 12 bits
high = int32((h << 44) >> 44) // sign extend 20 bits
return low, high
}
// encodeRISCVItypeImmediate encodes an I-type arithmetic instruction, expanding
// large immediates for ADDI/ANDI/ORI/XORI into LUI+ADDIW+op (or two ADDIs for
// ADDI), matching the Go assembler.
func encodeRISCVItypeImmediate(mnem string, enc riscvEnc, rd, rs1 int, imm int32) ([]byte, error) {
if imm >= -2048 && imm <= 2047 {
return wordLE(riscvIType(enc, rd, rs1, imm)), nil
}
var opMn string
switch mnem {
case "ADDI":
opMn = "ADD"
case "ANDI":
opMn = "AND"
case "ORI":
opMn = "OR"
case "XORI":
opMn = "XOR"
default:
return nil, fmt.Errorf("%s: immediate %d does not fit 12 bits", mnem, imm)
}
// ADDI with a small-ish immediate splits into two ADDIs.
if mnem == "ADDI" && imm >= -4096 && imm < 4095 {
imm0 := imm / 2
imm1 := imm - imm0
var out []byte
out = append(out, wordLE(riscvIType(enc, rd, rs1, imm0))...)
out = append(out, wordLE(riscvIType(enc, rd, rd, imm1))...)
return out, nil
}
// LUI $high, TMP; [ADDIW $low, TMP, TMP]; op TMP, rs1, rd. The LUI and
// ADDIW compress to their RVC forms (C.LUI / C.ADDIW) when the immediate
// fits 6 signed bits, matching the toolchain's compress pass.
low, high := splitRISCV32Imm(imm)
tmp := 31 // X31 = T6 = TMP
var out []byte
if high != 0 && high >= -32 && high <= 31 {
out = append(out, word16(rvcCI(0x3, uint32(tmp), uint32(high)&0x3F))...)
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, tmp, high<<12))...)
}
if low != 0 {
if low >= -32 && low <= 31 {
out = append(out, word16(rvcCI(0x1, uint32(tmp), uint32(low)&0x3F))...)
} else {
out = append(out, wordLE(riscvIType(riscvEnc{0x1B, 0x0, 0x00}, tmp, tmp, low))...)
}
}
opEnc, ok := riscvInstrTable[opMn]
if !ok {
return nil, fmt.Errorf("%s: unsupported operation %q", mnem, opMn)
}
out = append(out, wordLE(riscvRType(opEnc, rd, rs1, tmp))...)
return out, nil
}
// riscvItypeImmediateSize returns the encoded byte length of an I-type
// immediate instruction, accounting for the large-immediate expansion.
func riscvItypeImmediateSize(mnem string, imm int32) int {
if imm >= -2048 && imm <= 2047 {
return 4
}
switch mnem {
case "ADDI", "ANDI", "ORI", "XORI":
default:
return 4
}
if mnem == "ADDI" && imm >= -4096 && imm < 4095 {
return 8
}
low, high := splitRISCV32Imm(imm)
size := 4 // the R-type op (TMP is X31, never compressed)
if high != 0 && high >= -32 && high <= 31 {
size += 2 // C.LUI
} else {
size += 4 // LUI
}
if low != 0 {
if low >= -32 && low <= 31 {
size += 2 // C.ADDIW
} else {
size += 4 // ADDIW
}
}
return size
}
// encodeRISCVSBAddr emits AUIPC + ADDI to load the address of a static
// symbol into rd. Records R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I relocs.
// symbol into rd, recording the single R_RISCV_PCREL_ITYPE relocation the Go
// toolchain uses for the pair (the object-file emitters expand or map it).
func encodeRISCVSBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
name := sym.Name
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: name, Kind: RelPCRelHI20})
*relocs = append(*relocs, Reloc{Off: 4, After: 4, Name: name, Kind: RelPCRelLO12})
*relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELIType, Addend: sym.Offset})
}
auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, rd, 0)
addi := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rd, 0)
return append(wordLE(auipc), wordLE(addi)...)
}
// encodeRISCVSBLoad emits AUIPC + LD to load from a static symbol into rd.
// Records R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I relocs.
// encodeRISCVSBLoad emits AUIPC + LD to load from a static symbol into rd,
// recording the single R_RISCV_PCREL_ITYPE relocation for the pair.
func encodeRISCVSBLoad(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
name := sym.Name
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: name, Kind: RelPCRelHI20})
*relocs = append(*relocs, Reloc{Off: 4, After: 4, Name: name, Kind: RelPCRelLO12})
*relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELIType, Addend: sym.Offset})
}
auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, rd, 0)
ld := riscvIType(riscvEnc{0x03, 0x3, 0x00}, rd, rd, 0)
return append(wordLE(auipc), wordLE(ld)...)
}
// encodeRISCVSBStore emits AUIPC + SD to store a register into a static symbol.
// Records R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_S relocs.
// encodeRISCVSBStore emits AUIPC + SD to store a register into a static symbol,
// recording the single R_RISCV_PCREL_STYPE relocation for the pair.
func encodeRISCVSBStore(sym *ast.Symbol, rs2 int, relocs *[]Reloc) []byte {
tmp := 31 // X31 = T6
name := sym.Name
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: name, Kind: RelPCRelHI20})
*relocs = append(*relocs, Reloc{Off: 4, After: 4, Name: name, Kind: RelPCRelLO12S})
*relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELSType, Addend: sym.Offset})
}
auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, tmp, 0)
sd := riscvSType(riscvEnc{0x23, 0x3, 0x00}, tmp, rs2, 0)
@@ -655,6 +815,11 @@ func wordLE(w uint32) []byte {
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
}
// word16 encodes a uint16 as 2 little-endian bytes.
func word16(w uint16) []byte {
return []byte{byte(w), byte(w >> 8)}
}
// encodeRISCVJALR encodes the JALR indirect jump/call instruction.
// Plan 9: JALR rs1, rd (2 regs) or JALR offset(rs1) (memory → rd=X1).
func encodeRISCVJALR(instr *ast.Instr, fi riscvFrameInfo) ([]byte, error) {
@@ -686,10 +851,6 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
ops := instr.Operands
switch mnem {
case "RET":
// RET = JALR X0, 0(X1) → C.JR RA (CR-type: funct4=0x8, rd=0, rs2=1)
return rvcCR(0x8, 0, 1), true
case "LD", "MOV":
// LD rd, offset(SP) → C.LDSP when rd≠0 and uimm[8:3] fits.
// MOV name+off(FP), rd → load, same compression.
@@ -708,6 +869,10 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcLSP(0x3, uint32(rd), uint32(imm)), true
}
// Register-relative C.LD: both in prime regs, 8-byte scaled offset.
if rs1 != -1 && rd != -1 && isRVCIntReg(rd) && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCL(0x3, rvcReg3(rd), rvcReg3(rs1), uint32(imm)), true
}
// MOV reg, mem → store, try C.SDSP.
if mnem == "MOV" && len(ops) == 2 && !isMemOperand(ops[0]) && isMemOperand(ops[1]) {
rs2, rs1, imm := extractSDParams(instr, fi)
@@ -722,16 +887,46 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcSSP(0x7, uint32(rs2), uint32(imm)), true
}
// Register-relative C.SD: base and source in prime regs.
if rs1 != -1 && rs2 != -1 && isRVCIntReg(rs1) && isRVCIntReg(rs2) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCS(0x7, rvcReg3(rs2), rvcReg3(rs1), uint32(imm)), true
}
case "LW":
rd, rs1, imm := extractLDParams(instr, fi)
if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 256 && imm%4 == 0 {
return rvcLSP(0x2, uint32(rd), uint32(imm)), true
}
if rs1 != -1 && rd != -1 && isRVCIntReg(rd) && isRVCIntReg(rs1) && imm >= 0 && imm < 128 && imm%4 == 0 {
return rvcCL(0x2, rvcReg3(rd), rvcReg3(rs1), uint32(imm)), true
}
case "SW":
rs2, rs1, imm := extractSDParams(instr, fi)
if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 256 && imm%4 == 0 {
return rvcSSP(0x6, uint32(rs2), uint32(imm)), true
}
if rs1 != -1 && rs2 != -1 && isRVCIntReg(rs1) && isRVCIntReg(rs2) && imm >= 0 && imm < 128 && imm%4 == 0 {
return rvcCS(0x6, rvcReg3(rs2), rvcReg3(rs1), uint32(imm)), true
}
case "ADDI":
rd, rs1, imm := extractITypeParams(instr, fi)
if rd == -1 || rs1 == -1 {
return 0, false
}
if rd == 2 && rs1 == 2 && imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
// C.ADDI16SP: ADDI to SP by a nonzero 16-byte multiple.
return rvcADDI16SP(2, imm), true
}
if rd == rs1 && rd != 0 && imm != 0 && imm >= -32 && imm <= 31 {
// C.ADDI: funct3=0x0, rs1/rd, nzimm[5:0]
return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true
}
if isRVCIntReg(rd) && rs1 == 2 && imm != 0 && imm >= 0 && imm < 1024 && imm%4 == 0 {
// C.ADDI4SPN: ADDI $imm, SP, rd for a prime rd.
return rvcCIW(0x0, rvcReg3(rd), uint32(imm)), true
}
if rs1 == 0 && rd != 0 && imm >= -32 && imm <= 31 {
// C.LI: funct3=0x2, rd, imm[5:0]
return rvcCI(0x2, uint32(rd), uint32(imm)&0x3F), true
@@ -740,43 +935,44 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
// C.MV: funct4=0x8, rd, rs1 (CR-type)
return rvcCR(0x8, uint32(rd), uint32(rs1)), true
}
case "JAL":
// JAL X0, target → C.J when offset fits in ±2KB.
if len(ops) >= 1 {
// For JAL with implicit rd=0 (JMP alias), check target.
// C.J: funct3=0x5
// Offset is computed at encode time — we can't check it here.
return 0, false
if rd == 0 && rs1 == 0 && imm == 0 {
// C.NOP
return 0x0001, true
}
case "JAL":
// JAL/JMP are never compressed to C.J by the Go assembler.
return 0, false
case "JMP":
// C.J — handled in encodeRISCVInstr with actual offset.
// JAL/JMP are never compressed to C.J by the Go assembler.
return 0, false
case "BEQ":
// C.BEQZ — handled in encodeRISCVInstr with actual offset.
// Branches are never compressed to C.BEQZ/C.BNEZ.
return 0, false
case "BNE":
// C.BNEZ — handled in encodeRISCVInstr with actual offset.
// Branches are never compressed to C.BEQZ/C.BNEZ.
return 0, false
case "ADD":
// ADD rd, rs2 → C.ADD when rd == rs1 and both in prime regs (rd ≠ 0).
// ADD is commutative: if rd == rs2, swap.
// ADD rs2, rs1, rd → C.ADD (CR-type, funct4=0x9) when rd == rs1; ADD
// is commutative, so if rd == rs2, swap. ADD rs2, X0, rd is C.MV.
if len(ops) == 3 {
rs1 := regFromOperand(ops[0])
rs2 := regFromOperand(ops[1])
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd != -1 && rs1 != -1 && rs2 != -1 && rd != 0 {
if rd == rs1 && isRVCIntReg(rd) && isRVCIntReg(rs2) && rs2 != 0 {
// C.ADD: funct6=0x27, funct2=0x0 (CA-type)
return rvcCA(0x27, 0x0, rvcReg3(rd), rvcReg3(rs2)), true
if rd == rs1 && rs2 != 0 {
return rvcCR(0x9, uint32(rd), uint32(rs2)), true
}
if rd == rs2 && isRVCIntReg(rd) && isRVCIntReg(rs1) && rs1 != 0 {
// Swap: C.ADD rd, rs1
return rvcCA(0x27, 0x0, rvcReg3(rd), rvcReg3(rs1)), true
if rd == rs2 && rs1 != 0 {
return rvcCR(0x9, uint32(rd), uint32(rs1)), true
}
if rs1 == 0 && rs2 != 0 {
// ADD rs2, X0, rd → C.MV rd, rs2.
return rvcCR(0x8, uint32(rd), uint32(rs2)), true
}
}
}
@@ -795,13 +991,38 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
case "AND":
funct2 = 0x3
}
rs1 := regFromOperand(ops[0])
rs2 := regFromOperand(ops[1])
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd != -1 && rs1 != -1 && rs2 != -1 && rd != 0 {
if rd == rs1 && isRVCIntReg(rd) && isRVCIntReg(rs2) && rs2 != 0 {
return rvcCA(0x23, funct2, rvcReg3(rd), rvcReg3(rs2)), true
}
// AND/OR/XOR are commutative; SUB is not.
if mnem != "SUB" && rd == rs2 && isRVCIntReg(rd) && isRVCIntReg(rs1) && rs1 != 0 {
return rvcCA(0x23, funct2, rvcReg3(rd), rvcReg3(rs1)), true
}
}
}
case "ADDW", "SUBW":
// C.ADDW (0x27,1) / C.SUBW (0x27,0) — CA-type, prime regs.
if len(ops) == 3 {
funct2 := uint32(0x0)
if mnem == "ADDW" {
funct2 = 0x1
}
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd != -1 && rs1 != -1 && rs2 != -1 && isRVCIntReg(rd) {
if rd == rs1 && isRVCIntReg(rs2) {
return rvcCA(0x27, funct2, rvcReg3(rd), rvcReg3(rs2)), true
}
// ADDW is commutative; SUBW is not.
if mnem == "ADDW" && isRVCIntReg(rs1) && rd == rs2 {
return rvcCA(0x27, funct2, rvcReg3(rd), rvcReg3(rs1)), true
}
}
}
@@ -811,6 +1032,10 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
if rs1 == 2 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcLSP(0x1, uint32(rd), uint32(imm)), true
}
// Register-relative C.FLD: rd in F8-F15, base in X8-X15.
if rs1 != -1 && rd != -1 && rd >= 8 && rd <= 15 && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCL(0x1, uint32(rd-8), rvcReg3(rs1), uint32(imm)), true
}
case "FSD":
// FSD rs2, imm(SP) → C.FSDSP (CSS-type, funct3=0x5).
@@ -818,13 +1043,18 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcSSP(0x5, uint32(rs2), uint32(imm)), true
}
// Register-relative C.FSD: source in F8-F15, base in X8-X15.
if rs1 != -1 && rs2 != -1 && rs2 >= 8 && rs2 <= 15 && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCS(0x5, uint32(rs2-8), rvcReg3(rs1), uint32(imm)), true
}
case "LUI":
// LUI rd, imm → C.LUI when rd≠0, rd≠SP, imm nonzero and fits in 6 bits.
// LUI rd, imm → C.LUI when rd≠0, rd≠SP, imm nonzero and fits in six
// signed bits (matching the toolchain's compress pass).
if len(ops) == 2 {
rd := regFromOperand(ops[0])
imm := immFromOperand(ops[1])
if rd != -1 && rd != 0 && rd != 2 && imm != 0 && imm >= 1 && imm <= 63 {
if rd != -1 && rd != 0 && rd != 2 && imm != 0 && imm >= -32 && imm <= 31 {
return rvcCI(0x3, uint32(rd), uint32(imm)&0x3F), true
}
}
@@ -836,33 +1066,32 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
}
case "SLLI", "SRLI", "SRAI":
// C.SLLI (funct3=0x0), C.SRLI (funct3=0x4, funct2=0), C.SRAI (funct3=0x4, funct2=1).
rd, rs1, imm := extractITypeParams(instr, fi)
if rd == rs1 && rd != 0 && imm != 0 && imm >= 1 && imm <= 63 {
if mnem == "SLLI" {
// C.SLLI: funct3=0, CI-type with shamt in bits [12|6:2].
// For simplicity, use the standard CI format — the shamt is in imm[5:0].
return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true
// C.SLLI: funct3=0, op=10 quadrant, shamt in bits [12|6:2].
return rvcSLLI(uint32(rd), uint32(imm)&0x3F), true
}
if isRVCIntReg(rd) {
funct2 := uint32(0x0)
if mnem == "SRAI" {
funct2 = 0x1
}
// CB-format shift: funct3=0x4, shamt in bits [12|6:2].
// Use simplified encoding for now.
_ = funct2
return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true
// C.SRLI/C.SRAI: CB-type, funct3=0x4.
return rvcCBShift(funct2, rvcReg3(rd), uint32(imm)&0x3F), true
}
}
case "ANDI":
rd, rs1, imm := extractITypeParams(instr, fi)
if isRVCIntReg(rd) && rd == rs1 && imm >= -32 && imm <= 31 {
// C.ANDI: funct3=0x4, funct2=0x2 (CB-type).
// Simplified encoding for now.
return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true
// C.ANDI: CB-type, funct3=0x4, funct2=0x2.
return rvcCBShift(0x2, rvcReg3(rd), uint32(imm)&0x3F), true
}
case "EBREAK":
// C.EBREAK: CR-type, funct4=0x9, rd=0, rs2=0.
return rvcCR(0x9, 0, 0), true
}
return 0, false
@@ -899,15 +1128,23 @@ func extractSDParams(instr *ast.Instr, fi riscvFrameInfo) (rs2, rs1 int, imm int
return
}
// extractITypeParams extracts rd, rs1, and immediate for an I-type instruction.
// extractITypeParams extracts rd, rs1, and immediate for an I-type
// instruction. The Plan 9 order is INSTR $imm, rs1, rd (3 operands) or
// INSTR $imm, rd (2 operands, rd is also the source).
func extractITypeParams(instr *ast.Instr, fi riscvFrameInfo) (rd, rs1 int, imm int32) {
ops := instr.Operands
if len(ops) != 3 {
switch len(ops) {
case 3:
imm = immFromOperand(ops[0])
rs1 = regFromOperand(ops[1])
rd = regFromOperand(ops[2])
case 2:
imm = immFromOperand(ops[0])
rd = regFromOperand(ops[1])
rs1 = rd
default:
return -1, -1, 0
}
rs1 = regFromOperand(ops[0])
imm = immFromOperand(ops[1])
rd = regFromOperand(ops[2])
return
}
+79 -52
View File
@@ -13,7 +13,7 @@ func riscvRegNum(name string) int {
// Numbered integer registers.
case "X0", "ZERO":
return 0
case "X1", "RA":
case "X1", "RA", "LR":
return 1
case "X2", "SP":
return 2
@@ -21,9 +21,9 @@ func riscvRegNum(name string) int {
return 3
case "X4", "TP":
return 4
case "X5", "T0", "LR":
case "X5", "T0":
return 5
case "X6", "T1", "TMP":
case "X6", "T1":
return 6
case "X7", "T2":
return 7
@@ -73,7 +73,7 @@ func riscvRegNum(name string) int {
return 29
case "X30", "T5":
return 30
case "X31", "T6":
case "X31", "T6", "TMP":
return 31
// Floating-point registers (F0-F31).
case "F0", "FT0":
@@ -457,63 +457,84 @@ func rvcCR(funct4, rd, rs2 uint32) uint16 {
// rvcCI encodes a CI-type (immediate) compressed instruction.
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW — linear 6-bit immediate.
func rvcCI(funct3, rd uint32, imm uint32) uint16 {
return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x2)
return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x1)
}
// rvcLSP encodes a CI-type stack-relative load: C.LDSP (funct3=3) or
// C.FLDSP (funct3=1). offset is the full byte offset; the immediate bits
// are interleaved per the RISC-V spec: [5:3|8:6].
func rvcLSP(funct3, rd uint32, offset uint32) uint16 {
// Bit interleave offset bits [5,4,3,8,7,6] → packed value.
// rvcSLLI encodes C.SLLI, which shares funct3=0 with C.ADDI but lives in the
// op=10 quadrant (unlike C.ADDI's op=01).
func rvcSLLI(rd, shamt uint32) uint16 {
return uint16(((shamt>>5)&1)<<12 | (rd << 7) | (shamt&0x1F)<<2 | 0x2)
}
// encodeRVCPattern extracts the bits listed in pattern (MSB first) from imm
// into a packed value, matching cmd/internal/obj/riscv's encodeBitPattern.
func encodeRVCPattern(imm uint32, pattern []int) uint32 {
packed := uint32(0)
for i, b := range []int{5, 4, 3, 8, 7, 6} {
for _, bit := range pattern {
packed = packed<<1 | (imm>>bit)&1
}
return packed
}
// rvcLSP encodes a stack-relative compressed load (op=10 quadrant): C.LWSP
// (funct3=2, 4-byte scale), C.LDSP (funct3=3) or C.FLDSP (funct3=1, 8-byte
// scale). offset is the full byte offset.
func rvcLSP(funct3, rd uint32, offset uint32) uint16 {
pattern := []int{5, 4, 3, 8, 7, 6}
if funct3 == 0x2 {
pattern = []int{5, 4, 3, 2, 7, 6}
}
packed := uint32(0)
for i, b := range pattern {
packed |= ((offset >> b) & 1) << (5 - i)
}
return uint16((funct3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x2)
}
// rvcSSP encodes a CSS-type stack-relative store: C.SDSP (funct3=7) or
// C.FSDSP (funct3=5). offset is the full byte offset; the immediate bits
// are interleaved per the RISC-V spec: [5:3|8:6].
// rvcSSP encodes a stack-relative compressed store (op=10 quadrant): C.SWSP
// (funct3=6, 4-byte scale), C.SDSP (funct3=7) or C.FSDSP (funct3=5, 8-byte
// scale). offset is the full byte offset.
func rvcSSP(funct3, rs2 uint32, offset uint32) uint16 {
// Bit interleave offset bits [5,4,3,8,7,6] → packed value.
pattern := []int{5, 4, 3, 8, 7, 6}
if funct3 == 0x6 {
pattern = []int{5, 4, 3, 2, 7, 6}
}
packed := uint32(0)
for i, b := range []int{5, 4, 3, 8, 7, 6} {
for i, b := range pattern {
packed |= ((offset >> b) & 1) << (5 - i)
}
return uint16((funct3 << 13) | (packed << 7) | (rs2 << 2) | 0x2)
}
// rvcCSS encodes a CSS-type (stack store) compressed instruction.
func rvcCSS(funct3, rs2 uint32, imm uint32) uint16 {
return uint16((funct3 << 13) | (imm << 7) | (rs2 << 2) | 0x2)
}
// rvcCL encodes a CL-type (load) compressed instruction.
// imm layout: [5:3] in bits [12:10], [2|6] in bits [6:5].
// rvcCL encodes a register-relative compressed load (op=00 quadrant): C.LW
// (funct3=2), C.LD (funct3=3) or C.FLD (funct3=1). imm is the full byte
// offset; the immediate bits are extracted per the RISC-V CL format.
func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 {
bits := uint16((funct3 << 13) | ((imm>>3)&0x7)<<10 | (rs1 << 7) | ((imm & 0x7) << 5) | (rd << 2) | 0x0)
return bits
pattern := []int{5, 4, 3, 7, 6}
if funct3 == 0x2 {
pattern = []int{5, 4, 3, 2, 6}
}
packed := encodeRVCPattern(imm, pattern)
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rd << 2))
}
// rvcCS encodes a CS-type (store) compressed instruction.
// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
// offset; the immediate bits are extracted per the RISC-V CS format.
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
return uint16((funct3 << 13) | ((imm>>3)&0x7)<<10 | (rs1 << 7) | ((imm & 0x7) << 5) | (rs2 << 2) | 0x0)
pattern := []int{5, 3, 7, 6}
if funct3 == 0x6 {
pattern = []int{5, 3, 2, 6}
}
packed := encodeRVCPattern(imm, pattern)
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
}
// rvcCJ encodes a CJ-type (jump) compressed instruction.
// offset is a 12-bit signed offset (bit 0 is always 0).
func rvcCJ(funct3 uint32, offset int32) uint16 {
uoff := uint32(offset) & 0xFFE
bits := ((uoff >> 11) & 1) << 10
bits |= ((uoff >> 4) & 1) << 9
bits |= ((uoff >> 9) & 0x3) << 7
bits |= ((uoff >> 10) & 1) << 6
bits |= ((uoff >> 6) & 1) << 5
bits |= ((uoff >> 7) & 1) << 4
bits |= ((uoff >> 1) & 0x7) << 1
bits |= ((uoff >> 5) & 1)
return uint16((funct3 << 13) | (bits << 2) | 0x1)
// rvcCIW encodes a CIW-type compressed immediate wide instruction: C.ADDI4SPN
// (funct3=0). imm is the raw byte offset.
func rvcCIW(funct3, rd uint32, imm uint32) uint16 {
packed := encodeRVCPattern(imm, []int{5, 4, 9, 8, 7, 6, 2, 3})
return uint16((funct3 << 13) | (packed << 5) | (rd << 2))
}
// rvcCA encodes a CA-type (arithmetic) compressed instruction.
@@ -522,16 +543,22 @@ func rvcCA(funct6, funct2, rd, rs2 uint32) uint16 {
return uint16((funct6 << 10) | (rd << 7) | (funct2 << 5) | (rs2 << 2) | 0x1)
}
// rvcCB encodes a CB-type (branch) compressed instruction.
// Format: funct3[15:13] | offset[8|4:3] | rs1'[9:7] | offset[7:6|2:1|5] | op=01.
// Bit pattern for offset: [8|4:3|7:6|2:1|5]
func rvcCB(funct3, rs1 uint32, offset int32) uint16 {
uoff := uint32(offset) & 0x1FE // bits [8:1]
offBits := uint32(0)
offBits |= ((uoff >> 8) & 1) << 10 // bit 10 = offset[8]
offBits |= ((uoff >> 3) & 0x3) << 8 // bits 9:8 = offset[4:3]
offBits |= ((uoff >> 6) & 0x3) << 6 // bits 7:6 = offset[7:6]
offBits |= ((uoff >> 1) & 0x3) << 3 // bits 4:3 = offset[2:1]
offBits |= ((uoff >> 5) & 1) << 2 // bit 2 = offset[5]
return uint16((funct3 << 13) | offBits | (rs1 << 7) | 0x1)
// rvcCBShift encodes a CB-type shift/immediate compressed instruction
// (C.SRLI, C.SRAI, C.ANDI). rd is the 3-bit prime-register index; imm is
// the 6-bit shamt/immediate; funct2 selects the operation (0=SRLI, 1=SRAI,
// 2=ANDI).
func rvcCBShift(funct2, rd, imm uint32) uint16 {
return uint16((0x4 << 13) | ((imm>>5)&1)<<12 | (funct2 << 10) | (rd << 7) | (imm&0x1F)<<2 | 0x1)
}
// rvcADDI16SP encodes C.ADDI16SP: ADDI rd, imm, rd for the stack pointer
// with a 10-bit signed, 16-byte-scaled immediate. imm is the raw byte
// offset; the immediate bits are extracted in the order [9|4|6|8:7|5].
func rvcADDI16SP(rd uint32, imm int32) uint16 {
u := uint32(imm)
packed := uint32(0)
for _, bit := range []uint{9, 4, 6, 8, 7, 5} {
packed = packed<<1 | (u>>bit)&1
}
return uint16((0x3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x1)
}
+171 -157
View File
@@ -4,6 +4,7 @@
package asm
import (
"bytes"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -29,7 +30,7 @@ func firstTextRISCV(t *testing.T, src string) *ast.Text {
// assembleRISCVHelper assembles one TEXT function and returns its code bytes.
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
t.Helper()
code, _, _, err := assembleRISCV(fn)
code, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
@@ -65,9 +66,9 @@ TEXT ·arith(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
// 5 R-type instructions + RET compressed = 5*4 + 2 = 22
if len(code) != 22 {
t.Errorf("expected 22 bytes, got %d", len(code))
// 5 R-type instructions + RET = 5*4 + 4 = 24
if len(code) != 24 {
t.Errorf("expected 24 bytes, got %d", len(code))
}
}
@@ -81,35 +82,35 @@ TEXT ·mem(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
// 4 loads/stores (4B each) + C.JR RET (2B) = 18
if len(code) != 18 {
t.Errorf("expected 18 bytes, got %d", len(code))
// Four register-relative loads/stores compress (2B each) + JALR (4B) = 12.
if len(code) != 12 {
t.Errorf("expected 12 bytes, got %d", len(code))
}
}
func TestRISCV_immediate(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·imm(SB), NOSPLIT, $0
ADDI X10, $42, X11
ANDI X11, $0xFF, X12
ORI X12, $1, X13
XORI X13, $0, X14
ADDI $42, X10, X11
ANDI $0xFF, X11, X12
ORI $1, X12, X13
XORI $0, X13, X14
RET
`)
code := assembleRISCVHelper(t, fn)
// 4 I-type + C.JR = 4*4 + 2 = 18
if len(code) != 18 {
t.Errorf("expected 18 bytes, got %d", len(code))
// 4 I-type + JALR = 4*4 + 4 = 20
if len(code) != 20 {
t.Errorf("expected 20 bytes, got %d", len(code))
}
}
func TestRISCV_branches(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·br(SB), NOSPLIT, $0
ADDI X10, $1, X10
ADDI $1, X10, X10
loop:
BEQ X10, X11, done
ADDI X10, $1, X10
ADDI $1, X10, X10
JMP loop
done:
RET
@@ -122,28 +123,28 @@ done:
}
func TestRISCV_MOV_imm_small(t *testing.T) {
// MOV $42, rd → ADDI (fits in 12 bits). Not RVC-compressed (treated as MOV, not ADDI).
// MOV $42, rd → ADDI (fits in 12 bits, but not C.LI's 6-bit immediate).
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·small(SB), NOSPLIT, $0
MOV $42, X10
RET
`)
code := assembleRISCVHelper(t, fn)
// ADDI (4B) + C.JR (2B) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes, got %d", len(code))
// ADDI (4B) + JALR (4B) = 8
if len(code) != 8 {
t.Errorf("expected 8 bytes, got %d", len(code))
}
}
func TestRISCV_MOV_imm_large(t *testing.T) {
// MOV $0x12345, rd → LUI + ADDIW (8 bytes total)
// MOV $0x12345, rd → C.LUI $18 (2B) + ADDIW $837 (4B).
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·large(SB), NOSPLIT, $0
MOV $0x12345, X10
RET
`)
code := assembleRISCVHelper(t, fn)
// LUI (4B) + ADDIW (4B) + C.JR (2B) = 10
// C.LUI (2B) + ADDIW (4B) + JALR (4B) = 10
if len(code) != 10 {
t.Errorf("expected 10 bytes, got %d", len(code))
}
@@ -157,9 +158,9 @@ TEXT ·reg(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
// C.MV (2B) + C.JR (2B) = 4
if len(code) != 4 {
t.Errorf("expected 4 bytes, got %d (% x)", len(code), code)
// C.MV (2B) + JALR (4B) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes, got %d (% x)", len(code), code)
}
}
@@ -172,9 +173,9 @@ TEXT ·frame(SB), NOSPLIT, $0-8
RET
`)
code := assembleRISCVHelper(t, fn)
// C.LDSP (2B) + C.SDSP (2B) + C.JR (2B) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes, got %d", len(code))
// C.LDSP (2B) + C.SDSP (2B) + JALR (4B) = 8
if len(code) != 8 {
t.Errorf("expected 8 bytes, got %d", len(code))
}
}
@@ -187,9 +188,9 @@ TEXT ·rvcstore(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
// C.LDSP (2B) + C.SDSP (2B) + C.JR (2B) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes, got %d (% x)", len(code), code)
// C.LDSP (2B) + C.SDSP (2B) + JALR (4B) = 8
if len(code) != 8 {
t.Errorf("expected 8 bytes, got %d (% x)", len(code), code)
}
}
@@ -202,9 +203,9 @@ TEXT ·amo(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
// 3 AMO instructions (4B each) + C.JR (2B) = 14
if len(code) != 14 {
t.Errorf("expected 14 bytes, got %d", len(code))
// 3 AMO instructions (4B each) + JALR (4B) = 16
if len(code) != 16 {
t.Errorf("expected 16 bytes, got %d", len(code))
}
}
@@ -219,9 +220,9 @@ TEXT ·fpadd(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
// 5 FP instructions (4B each) + C.JR (2B) = 22
if len(code) != 22 {
t.Errorf("expected 22 bytes, got %d (%d)", len(code), len(code))
// 5 FP instructions (4B each) + JALR (4B) = 24
if len(code) != 24 {
t.Errorf("expected 24 bytes, got %d (%d)", len(code), len(code))
}
}
@@ -234,9 +235,9 @@ TEXT ·csrtest(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
// 3 CSR instructions (4B each) + C.JR (2B) = 14
if len(code) != 14 {
t.Errorf("expected 14 bytes, got %d", len(code))
// 3 CSR instructions (4B each) + JALR (4B) = 16
if len(code) != 16 {
t.Errorf("expected 16 bytes, got %d", len(code))
}
}
@@ -250,9 +251,9 @@ TEXT ·fmatest(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
// 4 FMA instructions (4B each) + C.JR (2B) = 18
if len(code) != 18 {
t.Errorf("expected 18 bytes, got %d", len(code))
// 4 FMA instructions (4B each) + JALR (4B) = 20
if len(code) != 20 {
t.Errorf("expected 20 bytes, got %d", len(code))
}
}
@@ -266,9 +267,9 @@ TEXT ·cvt(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
// 4 conversion instructions (4B each) + C.JR (2B) = 18
if len(code) != 18 {
t.Errorf("expected 18 bytes, got %d", len(code))
// 4 conversion instructions (4B each) + JALR (4B) = 20
if len(code) != 20 {
t.Errorf("expected 20 bytes, got %d", len(code))
}
}
@@ -281,9 +282,9 @@ TEXT ·cmp(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
// 3 FP compare (4B each) + C.JR (2B) = 14
if len(code) != 14 {
t.Errorf("expected 14 bytes, got %d", len(code))
// 3 FP compare (4B each) + JALR (4B) = 16
if len(code) != 16 {
t.Errorf("expected 16 bytes, got %d", len(code))
}
}
@@ -291,9 +292,9 @@ func TestRISCV_forwardBranch(t *testing.T) {
// Forward label reference — must not fail.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·fwd(SB), NOSPLIT, $0
ADDI X10, $1, X10
ADDI $1, X10, X10
BEQ X10, X11, done
ADDI X10, $1, X10
ADDI $1, X10, X10
done:
RET
`)
@@ -308,13 +309,13 @@ func TestRISCV_RVC_ADDI(t *testing.T) {
// ADDI where rd=rs1 and small imm → C.ADDI
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·caddi(SB), NOSPLIT, $0
ADDI X10, $5, X10
ADDI $5, X10, X10
RET
`)
code := assembleRISCVHelper(t, fn)
// C.ADDI (2B) + C.JR (2B) = 4
if len(code) != 4 {
t.Errorf("expected 4 bytes, got %d", len(code))
// C.ADDI (2B) + JALR (4B) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes, got %d", len(code))
}
}
@@ -322,13 +323,13 @@ func TestRISCV_RVC_LI(t *testing.T) {
// ADDI X0, $imm, rd → C.LI
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cli(SB), NOSPLIT, $0
ADDI X0, $7, X10
ADDI $7, X0, X10
RET
`)
code := assembleRISCVHelper(t, fn)
// C.LI (2B) + C.JR (2B) = 4
if len(code) != 4 {
t.Errorf("expected 4 bytes, got %d", len(code))
// C.LI (2B) + JALR (4B) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes, got %d", len(code))
}
}
@@ -340,9 +341,9 @@ TEXT ·clui(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
// C.LUI (2B) + C.JR (2B) = 4
if len(code) != 4 {
t.Errorf("expected 4 bytes, got %d", len(code))
// C.LUI (2B) + JALR (4B) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes, got %d", len(code))
}
}
@@ -368,13 +369,13 @@ TEXT ·sub(SB), NOSPLIT, $0
if len(img.Funcs) != 2 {
t.Fatalf("expected 2 functions, got %d", len(img.Funcs))
}
// func add: C.LDSP(2) + C.JR(2) = 4
if img.Funcs[0].Size != 4 {
t.Errorf("add: expected 4 bytes, got %d", img.Funcs[0].Size)
// func add: C.LDSP(2) + JALR(4) = 6
if img.Funcs[0].Size != 6 {
t.Errorf("add: expected 6 bytes, got %d", img.Funcs[0].Size)
}
// func sub: SUB(4) + C.JR(2) = 6
if img.Funcs[1].Size != 6 {
t.Errorf("sub: expected 6 bytes, got %d", img.Funcs[1].Size)
// func sub: SUB(4) + JALR(4) = 8
if img.Funcs[1].Size != 8 {
t.Errorf("sub: expected 8 bytes, got %d", img.Funcs[1].Size)
}
}
@@ -384,34 +385,34 @@ func TestRISCV_encodings(t *testing.T) {
name, src string
wantBytes int
}{
{"ADD", "ADD X10, X11, X12\nRET\n", 6},
{"SUBW", "SUBW X10, X11, X12\nRET\n", 6},
{"MUL", "MUL X10, X11, X12\nRET\n", 6},
{"DIVW", "DIVW X10, X11, X12\nRET\n", 6},
{"REMUW", "REMUW X10, X11, X12\nRET\n", 6},
{"ADDIW", "ADDIW X10, $5, X11\nRET\n", 6},
{"SLLI", "SLLI X10, $3, X11\nRET\n", 6}, // ADDI+SLLI? No, SLLI uses I-type
{"SRLI", "SRLI X10, $2, X11\nRET\n", 6},
{"SRAI", "SRAI X10, $1, X11\nRET\n", 6},
{"LB", "LB (X10), X11\nRET\n", 6},
{"LBU", "LBU (X10), X11\nRET\n", 6},
{"LH", "LH (X10), X11\nRET\n", 6},
{"LHU", "LHU (X10), X11\nRET\n", 6},
{"LWU", "LWU (X10), X11\nRET\n", 6},
{"SB", "SB X10, (X11)\nRET\n", 6},
{"SH", "SH X10, (X11)\nRET\n", 6},
{"ADD", "ADD X10, X11, X12\nRET\n", 8},
{"SUBW", "SUBW X10, X11, X12\nRET\n", 8},
{"MUL", "MUL X10, X11, X12\nRET\n", 8},
{"DIVW", "DIVW X10, X11, X12\nRET\n", 8},
{"REMUW", "REMUW X10, X11, X12\nRET\n", 8},
{"ADDIW", "ADDIW $5, X10, X11\nRET\n", 8},
{"SLLI", "SLLI $3, X10, X11\nRET\n", 8}, // ADDI+SLLI? No, SLLI uses I-type
{"SRLI", "SRLI $2, X10, X11\nRET\n", 8},
{"SRAI", "SRAI $1, X10, X11\nRET\n", 8},
{"LB", "LB (X10), X11\nRET\n", 8},
{"LBU", "LBU (X10), X11\nRET\n", 8},
{"LH", "LH (X10), X11\nRET\n", 8},
{"LHU", "LHU (X10), X11\nRET\n", 8},
{"LWU", "LWU (X10), X11\nRET\n", 8},
{"SB", "SB X10, (X11)\nRET\n", 8},
{"SH", "SH X10, (X11)\nRET\n", 8},
{"SW", "SW X10, (X11)\nRET\n", 6},
{"LUI", "LUI X10, $0x12345\nRET\n", 6},
{"AUIPC", "AUIPC X10, $0\nRET\n", 6},
{"FLW", "FLW (X10), F10\nRET\n", 6},
{"FSW", "FSW F10, (X11)\nRET\n", 6},
{"FADDS", "FADDS F10, F11, F12\nRET\n", 6},
{"FMINS", "FMINS F10, F11, F12\nRET\n", 6},
{"FMAXD", "FMAXD F10, F11, F12\nRET\n", 6},
{"FCVTSD", "FCVTSD F10, F11\nRET\n", 6},
{"FCVTDS", "FCVTDS F10, F11\nRET\n", 6},
{"FMVXW", "FMVXW F10, X10\nRET\n", 6},
{"FMADD_S", "FMADDS F10, F11, F12, F13\nRET\n", 6},
{"LUI", "LUI X10, $0x12345\nRET\n", 8},
{"AUIPC", "AUIPC X10, $0\nRET\n", 8},
{"FLW", "FLW (X10), F10\nRET\n", 8},
{"FSW", "FSW F10, (X11)\nRET\n", 8},
{"FADDS", "FADDS F10, F11, F12\nRET\n", 8},
{"FMINS", "FMINS F10, F11, F12\nRET\n", 8},
{"FMAXD", "FMAXD F10, F11, F12\nRET\n", 8},
{"FCVTSD", "FCVTSD F10, F11\nRET\n", 8},
{"FCVTDS", "FCVTDS F10, F11\nRET\n", 8},
{"FMVXW", "FMVXW F10, X10\nRET\n", 8},
{"FMADD_S", "FMADDS F10, F11, F12, F13\nRET\n", 8},
}
for _, tt := range tests {
@@ -428,24 +429,24 @@ TEXT ·`+tt.name+`(SB), NOSPLIT, $0
}
func TestRISCV_RVC_branch(t *testing.T) {
// BEQ rs, X0, target → C.BEQZ when rs is in prime regs and offset fits.
// Branches are never RVC-compressed (no C.BEQZ/C.BNEZ), matching go tool asm.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cbeqz(SB), NOSPLIT, $0
ADDI X10, $1, X10
ADDI $1, X10, X10
BEQ X10, X0, done
ADDI X10, $1, X10
ADDI $1, X10, X10
done:
RET
`)
code := assembleRISCVHelper(t, fn)
// C.ADDI(2) + C.BEQZ(2) + C.ADDI(2) + C.JR(2) = 8 (all compress)
if len(code) != 8 {
t.Errorf("expected 8 bytes with C.BEQZ, got %d", len(code))
// C.ADDI(2) + BEQ(4) + C.ADDI(2) + JALR(4) = 12
if len(code) != 12 {
t.Errorf("expected 12 bytes with uncompressed BEQ, got %d", len(code))
}
}
func TestRISCV_RVC_CJ(t *testing.T) {
// JMP target → C.J when offset fits.
// JMP target → JAL X0 (never compressed to C.J), matching go tool asm.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cj(SB), NOSPLIT, $0
JMP done
@@ -453,9 +454,9 @@ func TestRISCV_RVC_CJ(t *testing.T) {
RET
`)
code := assembleRISCVHelper(t, fn)
// C.J(2) + C.JR(2) = 4
if len(code) != 4 {
t.Errorf("expected 4 bytes with C.J, got %d", len(code))
// JAL(4) + JALR(4) = 8
if len(code) != 8 {
t.Errorf("expected 8 bytes with uncompressed JMP, got %d", len(code))
}
}
@@ -467,9 +468,9 @@ func TestRISCV_RVC_CADD(t *testing.T) {
RET
`)
code := assembleRISCVHelper(t, fn)
// C.ADD(2) + C.JR(2) = 4
if len(code) != 4 {
t.Errorf("expected 4 bytes with C.ADD, got %d", len(code))
// C.ADD(2) + JALR(4) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.ADD, got %d", len(code))
}
}
@@ -481,35 +482,23 @@ func TestRISCV_RVC_CADD_commute(t *testing.T) {
RET
`)
code := assembleRISCVHelper(t, fn)
// C.ADD(2) + C.JR(2) = 4
if len(code) != 4 {
t.Errorf("expected 4 bytes with C.ADD (commuted), got %d", len(code))
// C.ADD(2) + JALR(4) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.ADD (commuted), got %d", len(code))
}
}
func TestRISCV_RVC_CSUB(t *testing.T) {
// SUB where rd==rs1 and both in prime regs → C.SUB.
// SUB rs2, rs1, rd → C.SUB when rd == rs1 and both in prime regs.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·csub(SB), NOSPLIT, $0
SUB X11, X10, X10
RET
`)
code := assembleRISCVHelper(t, fn)
// SUB X11,X10,X10 → rd=X10, rs1=X11 ≠ rd → no C.SUB.
// Plan9: INSTR src1, src2, dst. For C.SUB: rd must equal rs1.
// So: SUB X10, X11, X10 → rd=10, rs1=10, rs2=11 ✓
if len(code) == 4 {
return // compressed
}
// Try with correct operand order.
fn2 := firstTextRISCV(t, `#include "textflag.h"
TEXT ·csub2(SB), NOSPLIT, $0
SUB X10, X11, X10
RET
`)
code2 := assembleRISCVHelper(t, fn2)
if len(code2) != 4 {
t.Errorf("expected 4 bytes with C.SUB, got %d (% x)", len(code2), code2)
// SUB X11, X10, X10 → rs2=X11, rs1=X10, rd=X10; rd==rs1 → C.SUB (2B) + JALR (4B) = 6.
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.SUB, got %d (% x)", len(code), code)
}
}
@@ -520,8 +509,8 @@ TEXT ·cxor(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
if len(code) != 4 {
t.Errorf("expected 4 bytes with C.XOR, got %d", len(code))
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.XOR, got %d", len(code))
}
}
@@ -532,8 +521,8 @@ TEXT ·cor(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
if len(code) != 4 {
t.Errorf("expected 4 bytes with C.OR, got %d", len(code))
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.OR, got %d", len(code))
}
}
@@ -544,8 +533,8 @@ TEXT ·cand(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
if len(code) != 4 {
t.Errorf("expected 4 bytes with C.AND, got %d", len(code))
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.AND, got %d", len(code))
}
}
@@ -556,9 +545,9 @@ TEXT ·cfldsp(SB), NOSPLIT, $0-8
RET
`)
code := assembleRISCVHelper(t, fn)
// C.FLDSP(2) + C.JR(2) = 4
if len(code) != 4 {
t.Errorf("expected 4 bytes with C.FLDSP, got %d", len(code))
// C.FLDSP(2) + JALR(4) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.FLDSP, got %d", len(code))
}
}
@@ -569,9 +558,9 @@ TEXT ·cfsdsp(SB), NOSPLIT, $0-8
RET
`)
code := assembleRISCVHelper(t, fn)
// C.FSDSP(2) + C.JR(2) = 4
if len(code) != 4 {
t.Errorf("expected 4 bytes with C.FSDSP, got %d", len(code))
// C.FSDSP(2) + JALR(4) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.FSDSP, got %d", len(code))
}
}
@@ -592,9 +581,9 @@ DATA answer<>+0(SB)/8, $42
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
// AUIPC(4) + ADDI(4) + C.JR(2) = 10
if img.Funcs[0].Size != 10 {
t.Errorf("expected 10 bytes, got %d", img.Funcs[0].Size)
// AUIPC(4) + ADDI(4) + JALR(4) = 12
if img.Funcs[0].Size != 12 {
t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
}
}
@@ -614,9 +603,9 @@ GLOBL result<>(SB), NOPTR, $8
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
// AUIPC X31(4) + SD X10,0(X31)(4) + C.JR(2) = 10
if img.Funcs[0].Size != 10 {
t.Errorf("expected 10 bytes, got %d", img.Funcs[0].Size)
// AUIPC X31(4) + SD X10,0(X31)(4) + JALR(4) = 12
if img.Funcs[0].Size != 12 {
t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
}
}
@@ -696,9 +685,9 @@ DATA answer<>+0(SB)/8, $42
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
// AUIPC(4) + LD(4) + C.JR(2) = 10
if img.Funcs[0].Size != 10 {
t.Errorf("expected 10 bytes, got %d", img.Funcs[0].Size)
// AUIPC(4) + LD(4) + JALR(4) = 12
if img.Funcs[0].Size != 12 {
t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
}
}
@@ -712,7 +701,7 @@ TEXT ·sys(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
// 3 system instructions × 4 bytes + C.JR(2) = 14
// FENCE(4) + ECALL(4) + C.EBREAK(2) + JALR(4) = 14
if len(code) != 14 {
t.Errorf("expected 14 bytes, got %d (% x)", len(code), code)
}
@@ -725,25 +714,50 @@ TEXT ·badfp(SB), NOSPLIT, $0
MOV $arg(FP), X10
RET
`)
_, _, _, err := assembleRISCV(fn)
_, _, _, _, _, err := assembleRISCV(fn)
if err == nil {
t.Error("expected error for MOV $arg(FP), got nil")
}
}
func TestRISCV_CALL(t *testing.T) {
// CALL target → AUIPC + JALR (8 bytes).
// CALL sym(SB) → JAL X1, sym(SB) with a single R_RISCV_JAL relocation.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·calltest(SB), NOSPLIT, $0
CALL sub
done:
CALL ext(SB)
RET
`)
code, _, relocs, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
// prologue (8) + JAL (4) + epilogue+JALR (8) = 20
if len(code) != 20 {
t.Fatalf("expected 20 bytes with CALL sym(SB), got %d", len(code))
}
if len(relocs) != 1 {
t.Fatalf("relocs = %d, want 1", len(relocs))
}
r := relocs[0]
if r.Kind != RelRISCVJal || r.Name != "ext" || r.Off != 8 || r.After != 12 || r.Addend != 0 {
t.Errorf("reloc = {kind %v off %d after %d name %q addend %d}", r.Kind, r.Off, r.After, r.Name, r.Addend)
}
// The JAL instruction itself is JAL X1, 0 at function offset 8.
wantJAL := wordLE(riscvJType(1, 0))
if !bytes.Equal(code[8:12], wantJAL) {
t.Errorf("JAL = % x, want % x", code[8:12], wantJAL)
}
}
func TestRISCV_CALL_local_error(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·calllocal(SB), NOSPLIT, $0
CALL sub
sub:
RET
`)
code := assembleRISCVHelper(t, fn)
// CALL(8) + C.JR(2) + C.JR(2) = 12
if len(code) != 12 {
t.Errorf("expected 12 bytes with CALL, got %d", len(code))
_, _, _, _, _, err := assembleRISCV(fn)
if err == nil {
t.Error("expected error for CALL to local label, got nil")
}
}
+149 -84
View File
@@ -3,115 +3,180 @@
package asm
import "sourcedock.dev/petrbalvin/gasm-devkit/ast"
import (
"strings"
// RISC-V frame mapping: translates Go's FP/SP pseudo-register addressing
// into real RISC-V memory accesses.
//
// In Go's ABI0 (used by assembly functions), arguments are passed on the
// stack. At function entry the return address sits at SP, so the frame
// pointer FP == SP+8 and the first argument is at FP+0 == SP+8.
//
// On RISC-V the hardware registers are:
// SP = X2 (stack pointer)
// FP = S0 = X8 (frame pointer, by convention)
//
// For NOSPLIT $0 functions the prologue is omitted and arguments are read
// directly from SP+8+offset.
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// riscvFrameInfo holds the frame parameters computed from a TEXT directive.
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
//
// Go's riscv64 functions have no hardware frame pointer: FP and SP are
// synthetic registers resolved against the hardware stack pointer (X2) and
// the frame size. The return address lives in the link register (X1, RA/LR).
//
// The autosize is the real stack adjustment: the declared local frame plus
// the 8 bytes for the saved link register (the toolchain's FixedFrameSize).
// A leaf function with a zero frame gets no prologue at all.
//
// Prologue (autosize > 0), byte-identical to the toolchain:
//
// MOV LR, -autosize(SP) // save LR below the new SP (traceback-safe)
// ADDI $-autosize, SP, SP // open the frame
// MOV LR, 0(SP) // save LR again at SP (signal-safety)
//
// Epilogue (autosize > 0): MOV 0(SP), LR; ADDI $autosize, SP, SP; the RET's
// uncompressed JALR X0, 0(X1) follows. The toolchain restores LR on every
// frame, leaf or not.
// riscvFrameInfo holds the frame layout derived from a TEXT directive.
type riscvFrameInfo struct {
frameSize int // the $framesize from TEXT
argsSize int // the -argsize from TEXT
noSplit bool // the NOSPLIT flag
autosize int // the real SP adjustment (locals + saved LR)
}
// riscvComputeFrame extracts frame information from a TEXT directive.
// riscvComputeFrame derives the frame layout for a TEXT function.
func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
fi := riscvFrameInfo{}
fi.frameSize = frameSize(t)
fi.argsSize = argsSize(t)
for _, f := range t.Flags {
if f == "NOSPLIT" {
fi.noSplit = true
frame := frameSize(t)
if frame != 0 || !riscvIsLeaf(t) {
// FixedFrameSize = 8: space for the saved link register. A
// zero-frame non-leaf function still opens an 8-byte frame for LR.
return riscvFrameInfo{autosize: frame + 8}
}
return riscvFrameInfo{}
}
// riscvIsLeaf reports whether a function contains no call instructions.
// CALL always links; JAL/JALR link only when their destination register is
// the link register (X1), matching cmd/internal/obj/riscv's containsCall.
func riscvIsLeaf(t *ast.Text) bool {
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "CALL":
return false
case "JAL":
// JAL rd, target — a call only when rd is the link register.
if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 {
return false
}
case "JALR":
// JALR rs1, rd — a call when rd is X1; JALR offset(rs1) always
// links to X1.
if len(in.Operands) == 1 {
return false
}
if len(in.Operands) >= 2 && regFromOperand(in.Operands[1]) == 1 {
return false
}
}
}
return fi
return true
}
// riscvPrologue returns the prologue bytes for a RISC-V function.
// For NOSPLIT $0 functions there is no prologue. For functions with a
// frame, we emit: ADDI SP, SP, -framesize; SD S0, (framesize-8)(SP); ...
// riscvPrologue returns the prologue bytes for a RISC-V function, matching
// the toolchain's compression: the SP adjustment compresses to C.ADDI when
// the immediate fits, and the second LR save compresses to C.SDSP.
func riscvPrologue(fi riscvFrameInfo) []byte {
if fi.noSplit && fi.frameSize == 0 {
return nil // no prologue for NOSPLIT $0
}
var out []byte
if fi.frameSize > 0 {
// ADDI SP, SP, -framesize
out = append(out, riscvITypeLE(0x13, 0x0, 2, 2, int32(-fi.frameSize))...)
// Save the frame pointer (S0 = X8) at the top of the new frame.
// SD S0, (framesize-8)(SP)
out = append(out, riscvSTypeLE(0x23, 0x3, 2, 8, int32(fi.frameSize-8))...)
}
return out
}
// riscvEpilogue returns the epilogue bytes for a RISC-V function.
func riscvEpilogue(fi riscvFrameInfo) []byte {
if fi.noSplit && fi.frameSize == 0 {
if fi.autosize == 0 {
return nil
}
var out []byte
if fi.frameSize > 0 {
// Restore the frame pointer: LD S0, (framesize-8)(SP)
out = append(out, riscvITypeLE(0x03, 0x3, 8, 2, int32(fi.frameSize-8))...)
// ADDI SP, SP, framesize
out = append(out, riscvITypeLE(0x13, 0x0, 2, 2, int32(fi.frameSize))...)
}
// MOV LR, -autosize(SP) — SD X1, -autosize(X2). The negative offset is
// not compressible to C.SDSP (unsigned), so it stays 4 bytes.
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
// ADDI $-autosize, SP, SP — open the frame (C.ADDI when it fits).
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
// MOV LR, 0(SP) — SD X1, 0(X2) → C.SDSP X1, 0.
c := rvcSSP(0x7, 1, 0)
out = append(out, byte(c), byte(c>>8))
return out
}
// riscvReturn returns the bytes for a RET: the epilogue (restore LR and
// deallocate the frame when present) followed by the uncompressed JALR X0,
// 0(X1) the toolchain emits for RET (it never compresses RET to C.JR).
func riscvReturn(fi riscvFrameInfo) []byte {
var out []byte
if fi.autosize != 0 {
// MOV 0(SP), LR — LD X1, 0(X2) → C.LDSP X1, 0.
c := rvcLSP(0x3, 1, 0)
out = append(out, byte(c), byte(c>>8))
// ADDI $autosize, SP, SP — close the frame (C.ADDI when it fits).
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
}
// JALR X0, 0(X1).
return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...)
}
// riscvSPAdjust emits an ADDI rd, imm, rd for the stack pointer (rd = rs1 =
// X2), compressed to C.ADDI16SP when the immediate is a nonzero 16-byte
// multiple, else C.ADDI when it fits 6-bit signed.
func riscvSPAdjust(imm int32) []byte {
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
c := rvcADDI16SP(2, imm)
return []byte{byte(c), byte(c >> 8)}
}
if riscvFitsCAddi(imm) {
c := rvcCI(0x0, 2, uint32(imm)&0x3F)
return []byte{byte(c), byte(c >> 8)}
}
return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 2, 2, imm))
}
// riscvFitsCAddi reports whether imm compresses to C.ADDI (a nonzero 6-bit
// signed immediate).
func riscvFitsCAddi(imm int32) bool {
return imm != 0 && imm >= -32 && imm <= 31
}
// riscvPrologueSpadjPC returns the function-relative byte offset where the
// prologue has finished decrementing SP (the delta becomes autosize).
func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
if fi.autosize == 0 {
return 0
}
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 4 + riscvSPAdjustLen(int32(-fi.autosize))
}
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final JALR — the point where SP is restored.
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
if fi.autosize == 0 {
return 0
}
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 2 + riscvSPAdjustLen(int32(fi.autosize))
}
func riscvSPAdjustLen(imm int32) int {
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
return 2
}
if riscvFitsCAddi(imm) {
return 2
}
return 4
}
// riscvResolvePseudo translates a pseudo-register memory reference into a
// real base register and offset. It handles name+offset(FP) and
// name+offset(SP).
//
// Returns the base register number and the adjusted offset.
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
// x+N(SP) → (N + autosize)(SP). Returns base = -1 for an unresolvable
// reference (SB: static data, handled by the relocation path).
func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32) {
if sym == nil {
return -1, 0
}
offset := int32(sym.Offset)
switch sym.Pseudo {
case "FP":
// FP == SP+8 for NOSPLIT $0; arguments are at SP+8+offset.
if fi.noSplit && fi.frameSize == 0 {
return 2, 8 + offset // SP + 8 + argOffset
}
// With a frame, FP points to the saved frame; args are at FP+offset.
return 8, offset // S0 + argOffset
return 2, int32(sym.Offset) + int32(fi.autosize) + 8
case "SP":
// SP-relative; the offset is from the current SP.
return 2, offset
return 2, int32(fi.autosize) + int32(sym.Offset)
case "SB":
// Static data reference — needs a relocation (not yet supported).
return -1, offset
default:
return -1, offset
return -1, int32(sym.Offset)
}
}
// riscvITypeLE encodes an I-type instruction and returns little-endian bytes.
func riscvITypeLE(opcode, funct3 uint32, rd, rs1 int, imm int32) []byte {
word := (uint32(imm&0xFFF) << 20) | (uint32(rs1) << 15) |
(funct3 << 12) | (uint32(rd) << 7) | opcode
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
}
// riscvSTypeLE encodes an S-type instruction and returns little-endian bytes.
func riscvSTypeLE(opcode, funct3 uint32, rs1, rs2 int, imm int32) []byte {
immU := uint32(imm) & 0xFFF
word := ((immU >> 5) << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
(funct3 << 12) | ((immU & 0x1F) << 7) | opcode
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
return -1, 0
}
+81
View File
@@ -0,0 +1,81 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestRISCVFrameSpadjAndLines checks that a framed function records its
// stack-adjustment boundaries and source-line table, the inputs the GOOBJ
// emitter turns into the pcsp/pcfile/pcline tables.
func TestRISCVFrameSpadjAndLines(t *testing.T) {
f, errs := parser.Parse("frame_riscv64.s", `#include "textflag.h"
TEXT ·framed(SB), NOSPLIT, $16-16
MOV a+0(FP), X10
MOV b+8(FP), X11
ADD X11, X10, X10
MOV X10, ret+16(FP)
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
fn := img.Funcs[0]
if fn.Size != 24 {
t.Fatalf("size = %d, want 24", fn.Size)
}
// autosize = 16 + 8 = 24; the prologue boundary is just past its C.ADDI
// (SD 4 + C.ADDI 2 = 6), and the RET restores SP just past its C.ADDI
// (RET starts at 16; C.LDSP 2 + C.ADDI 2 = 20).
wantSpadj := []SpadjStep{{PC: 6, Value: 24}, {PC: 20, Value: 0}}
if len(fn.Spadj) != len(wantSpadj) {
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
}
for i := range wantSpadj {
if fn.Spadj[i] != wantSpadj[i] {
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
}
}
// One line entry per instruction, in emission order.
wantLines := []LineEntry{
{Offset: 8, Line: 4},
{Offset: 10, Line: 5},
{Offset: 12, Line: 6},
{Offset: 14, Line: 7},
{Offset: 16, Line: 8},
}
if len(fn.Lines) != len(wantLines) {
t.Fatalf("lines = %v, want %v", fn.Lines, wantLines)
}
for i := range wantLines {
if fn.Lines[i] != wantLines[i] {
t.Errorf("lines[%d] = %v, want %v", i, fn.Lines[i], wantLines[i])
}
}
}
// TestRISCVRegAliases checks the Go ABI register aliases that the toolchain
// defines: LR is the link register (X1) and TMP is the assembler scratch
// register (X31/T6).
func TestRISCVRegAliases(t *testing.T) {
for name, want := range map[string]int{
"X1": 1, "RA": 1, "LR": 1,
"X31": 31, "T6": 31, "TMP": 31,
"X2": 2, "SP": 2,
} {
if got := riscvRegNum(name); got != want {
t.Errorf("riscvRegNum(%q) = %d, want %d", name, got, want)
}
}
}
+346
View File
@@ -0,0 +1,346 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
// instruction carrying an R_RISCV_JAL relocation (4-byte field) in both the
// GOOBJ and ELF object emitters.
func TestGOObjectRISCVCallReloc(t *testing.T) {
f, errs := parser.Parse("k_riscv64.s", `
#include "textflag.h"
TEXT ·c(SB), NOSPLIT, $0-0
CALL callee<>(SB)
RET
GLOBL callee<>(SB), RODATA, $8
DATA callee<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
fn := img.Funcs[0]
if len(fn.Relocs) != 1 {
t.Fatalf("relocs = %d, want 1", len(fn.Relocs))
}
r := fn.Relocs[0]
if r.Kind != RelRISCVJal || r.Off != 8 || r.After != 12 || r.Name != "callee" || r.Addend != 0 || r.External {
t.Errorf("reloc = {kind %v off %d after %d name %q addend %d external %v}", r.Kind, r.Off, r.After, r.Name, r.Addend, r.External)
}
obj, err := img.GOObjectRISCV("testpkg", "k_riscv64.s")
if err != nil {
t.Fatalf("GOObjectRISCV: %v", err)
}
v := openGoobj(t, obj)
relocIdx := v.blk(blkRelocIdx)
relocs := v.blk(blkReloc)
// The function is the last non-package symbol: 4 package defs, then the
// 4 pc tables and the function.
first := int(binary.LittleEndian.Uint32(relocIdx[(4+4)*4:]))
if (first+1)*23 > len(relocs) {
t.Fatalf("reloc block too short: first=%d len=%d", first, len(relocs))
}
e := relocs[first*23:]
le := binary.LittleEndian
if int32(le.Uint32(e[0:])) != 8 || e[4] != 4 || le.Uint16(e[5:]) != relocRISCVJal || le.Uint32(e[15:]) != pkgIdxSelf || le.Uint32(e[19:]) != 0 {
t.Errorf("GOOBJ reloc = off %d size %d type %d pkg %d sym %d", int32(le.Uint32(e[0:])), e[4], le.Uint16(e[5:]), le.Uint32(e[15:]), le.Uint32(e[19:]))
}
// The ELF object must carry a single R_RISCV_JAL relocation in .rela.text.
elfObj, err := img.ELFRISCVObject()
if err != nil {
t.Fatalf("ELFRISCVObject: %v", err)
}
if !hasELFRISCVJAL(t, elfObj) {
t.Error("ELF object missing R_RISCV_JAL relocation")
}
}
func TestGOObjectRISCVStructure(t *testing.T) {
f, errs := parser.Parse("k_riscv64.s", `
#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0-0
MOV $answer<>(SB), X10
MOV answer<>(SB), X11
MOV X12, answer<>(SB)
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
fn := img.Funcs[0]
if fn.Size != 28 {
t.Fatalf("function size = %d, want 28", fn.Size)
}
if len(fn.Relocs) != 3 {
t.Fatalf("relocs = %d, want 3", len(fn.Relocs))
}
wantKind := []RelocKind{RelRISCVPCRELIType, RelRISCVPCRELIType, RelRISCVPCRELSType}
wantOff := []int{0, 8, 16}
for i, r := range fn.Relocs {
if r.Kind != wantKind[i] || r.Off != wantOff[i] || r.After != r.Off+8 || r.Name != "answer" || r.Addend != 0 {
t.Errorf("reloc %d = {kind %v off %d after %d name %q addend %d}", i, r.Kind, r.Off, r.After, r.Name, r.Addend)
}
}
obj, err := img.GOObjectRISCV("testpkg", "k_riscv64.s")
if err != nil {
t.Fatalf("GOObjectRISCV: %v", err)
}
v := openGoobj(t, obj)
// Package defs: the static GLOBL, the FuncInfo, then the two DWARF
// symbols.
defs := v.syms(blkSymdef)
if len(defs) != 4 {
t.Fatalf("symdefs = %d, want 4", len(defs))
}
if defs[0].name != "answer" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 8 {
t.Errorf("answer symbol = %+v", defs[0])
}
if defs[2].typ != kindSDWARFLINES || defs[3].typ != kindSDWARFFCN {
t.Errorf("dwarf symbols = %+v, %+v", defs[2], defs[3])
}
// The three code relocations, in definition order: ITYPE, ITYPE, STYPE,
// each 8 bytes wide against the GLOBL (package symbol 0).
relocIdx := v.blk(blkRelocIdx)
relocs := v.blk(blkReloc)
if len(relocs) != 5*23 {
t.Fatalf("relocs = %d bytes, want 5 entries", len(relocs))
}
// The function is the last non-package symbol; its relocs start after
// the DWARF symbols' (defs 2 and 3 each carry one).
le := binary.LittleEndian
first := int(le.Uint32(relocIdx[4*(4+4):]))
wantType := []uint16{relocRISCVPcrelItype, relocRISCVPcrelItype, relocRISCVPcrelStype}
wantOffAbs := []int{0, 8, 16}
for i := 0; i < 3; i++ {
e := relocs[(first+i)*23:]
if int32(le.Uint32(e[0:])) != int32(wantOffAbs[i]) || e[4] != 8 || le.Uint16(e[5:]) != wantType[i] ||
le.Uint32(e[15:]) != pkgIdxSelf || le.Uint32(e[19:]) != 0 {
t.Errorf("reloc %d = off %d size %d type %d pkg %d sym %d", i, int32(le.Uint32(e[0:])), e[4], le.Uint16(e[5:]), le.Uint32(e[15:]), le.Uint32(e[19:]))
}
}
// The function code: three AUIPC+second-instruction pairs with zero
// immediates, then the uncompressed JALR X0, 0(X1) the toolchain emits
// for RET.
code := img.Code[fn.Offset : fn.Offset+fn.Size]
want := append(wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 10, 0)), wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 10, 10, 0))...)
want = append(want, wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 11, 0))...)
want = append(want, wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 11, 11, 0))...)
want = append(want, wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 31, 0))...)
want = append(want, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 12, 0))...)
want = append(want, 0x67, 0x80, 0x00, 0x00) // JALR X0, 0(X1)
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
// The same bytes must survive into the object's data block intact: the
// linker patches only the immediate fields of the AUIPC pairs, so the
// opcode/register bits of every instruction must not be zeroed.
dataIdx := v.blk(blkDataIdx)
dataBlk := v.blk(blkData)
dOff := int(le.Uint32(dataIdx[8*4:])) // the function is the last symbol
emitted := dataBlk[dOff : dOff+fn.Size]
if !bytes.Equal(emitted, want) {
t.Errorf("emitted data = % x\nwant % x", emitted, want)
}
}
// TestGOObjectRISCVLink cross-compiles a Go program with the gasm-produced
// object substituted into the package archive, proving cmd/link accepts the
// emitted RISC-V GOOBJ. The binary is not executed (no riscv64 host or
// qemu). Skipped when no Go toolchain is available.
func TestGOObjectRISCVLink(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
asmSrc := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOV a+0(FP), X10
MOV b+8(FP), X11
ADD X11, X10, X10
MOV X10, ret+16(FP)
RET
`
if err := os.WriteFile(filepath.Join(dir, "main_riscv64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
mainSrc := `package main
func add(a, b int64) int64
func main() {
if add(20, 22) != 42 {
panic("bad add")
}
}
`
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module rvlink\n\ngo 1.21\n"), 0o644); err != nil {
t.Fatal(err)
}
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
build.Env = append(os.Environ(), "GOARCH=riscv64")
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var pkgArch, work, linkLine, asmObj string
for _, line := range strings.Split(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_riscv64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if pkgArch == "" || linkLine == "" || asmObj == "" {
t.Skip("could not locate the archive, asm output or link line in the build log")
}
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
asmMember := filepath.Base(strings.ReplaceAll(asmObj, "$WORK", work))
pf, perrs := parser.Parse(filepath.Join(dir, "main_riscv64.s"), asmSrc)
if len(perrs) > 0 {
t.Fatalf("parse: %v", perrs)
}
pimg, err := AssembleFileRISCV(pf)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
obj, err := pimg.GOObjectRISCV("main", filepath.Join(dir, "main_riscv64.s"))
if err != nil {
t.Fatalf("GOObjectRISCV: %v", err)
}
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
extract.Dir = membersDir
extract.Env = append(os.Environ(), "GOARCH=riscv64")
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
member := filepath.Join(membersDir, asmMember)
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, obj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listCmd.Env = append(os.Environ(), "GOARCH=riscv64")
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for _, m := range strings.Fields(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, filepath.Join(membersDir, m))
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
pack.Env = append(os.Environ(), "GOARCH=riscv64")
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
link := exec.Command("sh", "-c", linkLine)
link.Dir = dir
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)), "GOARCH=riscv64")
if out, err := link.CombinedOutput(); err != nil {
t.Fatalf("link with gasm object: %v\n%s", err, out)
}
nm := exec.Command(goBin, "tool", "nm", filepath.Join(dir, "app2"))
nm.Env = append(os.Environ(), "GOARCH=riscv64")
nmOut, err := nm.CombinedOutput()
if err != nil {
t.Fatalf("nm gasm-linked binary: %v\n%s", err, nmOut)
}
if !strings.Contains(string(nmOut), "main.add") {
t.Errorf("main.add not found in linked binary:\n%s", nmOut)
}
}
// hasELFRISCVJAL reports whether the ELF object carries an R_RISCV_JAL
// relocation in its .rela.text section.
func hasELFRISCVJAL(t *testing.T, data []byte) bool {
t.Helper()
f, err := elf.NewFile(bytes.NewReader(data))
if err != nil {
t.Fatalf("parse ELF: %v", err)
}
defer f.Close()
rela := f.Section(".rela.text")
if rela == nil {
return false
}
b, err := rela.Data()
if err != nil {
t.Fatalf(".rela.text data: %v", err)
}
const rRISCVJAL = 17
for i := 0; i+24 <= len(b); i += 24 {
info := binary.LittleEndian.Uint64(b[i+8:])
if uint32(info) == rRISCVJAL {
return true
}
}
return false
}
+244 -45
View File
@@ -34,7 +34,7 @@ import (
// version is the release version, stamped at build time via
// -ldflags "-X main.version=…" (defaulting to the current release).
var version = "0.29.0"
var version = "0.31.1"
func main() {
if len(os.Args) < 2 {
@@ -395,34 +395,30 @@ hover, document symbols, diagnostics and semantic-token highlighting.
}
func cmdAsm(args []string) int {
fs := newCommand("asm", "gasm asm [--format raw|elf|macho|goobj] [-p pkg] [-o out] <file>", `
Assemble FILE (amd64 or riscv64) without the Go toolchain: every TEXT function is
encoded to machine code — scalar, VEX/AVX2 and EVEX/AVX-512 instructions,
FP/SP frame mapping, local labels and file-local static symbols (GLOBL/DATA)
resolved RIP-relative — and printed as a hex dump.
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>", `
Assemble FILE without the Go toolchain: every TEXT function is encoded to
machine code and printed as a hex dump. Supported architectures: amd64
(including VEX/AVX2 and EVEX/AVX-512), riscv64 (RV64IMAFDC + RVC) and
loong64 (LoongArch base ISA); arm64 encoding is not yet implemented.
With -o the output is written to a file instead. The --format flag selects
what is written: raw (the default) concatenates the functions and the data
section into one self-consistent image; elf and macho emit a relocatable
object (.text/.data sections, a symbol table and one PC32 relocation per
section into one self-consistent image; elf emits a relocatable object
(.text/.data sections, a symbol table and one PC32 relocation per
static-symbol reference) that links with the system toolchain; goobj emits
the Go toolchain's own object format, which cmd/link consumes directly (it
requires -p, the package path, and the installed Go toolchain).
`)
out := fs.String("o", "", "write the output to this file")
format := fs.String("format", "raw", "output format: raw (concatenated image), elf, macho or goobj (Go object)")
format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|macho|goobj] [-p pkg] [-o out] <file>")
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>")
return 2
}
path := fs.Arg(0)
targetArch := arch.FromFilename(path)
if targetArch != arch.AMD64 && targetArch != arch.RISCV {
fmt.Fprintln(os.Stderr, "gasm asm: only amd64 and riscv64 are supported")
return 1
}
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
@@ -436,12 +432,7 @@ requires -p, the package path, and the installed Go toolchain).
return 1
}
var img *asm.Image
if targetArch == arch.RISCV {
img, err = asm.AssembleFileRISCV(f)
} else {
img, err = asm.AssembleFile(f)
}
img, err := assembleFile(path, targetArch, f)
if err != nil {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
return 1
@@ -497,29 +488,36 @@ requires -p, the package path, and the installed Go toolchain).
switch *format {
case "raw":
if len(img.Externals) > 0 {
fmt.Fprintf(os.Stderr, "gasm asm: external symbol %q needs an object file (use --format elf or --format macho)\n", img.Externals[0])
fmt.Fprintf(os.Stderr, "gasm asm: external symbol %q needs an object file (use --format elf)\n", img.Externals[0])
return 1
}
obj, kind = img.Bytes(), "raw image"
case "elf":
if targetArch == arch.RISCV {
switch targetArch {
case arch.RISCV:
obj, err = img.ELFRISCVObject()
} else {
case arch.LOONG64:
obj, err = img.ELFLOONG64Object()
case arch.ARM64:
obj, err = img.ELFAARCH64Object()
default:
obj, err = img.ELFObject()
}
kind = "ELF object"
case "macho":
obj, err = img.MachOObject()
kind = "Mach-O object"
case "goobj":
if targetArch == arch.RISCV {
switch targetArch {
case arch.RISCV:
obj, err = img.GOObjectRISCV(*pkg, path)
} else {
case arch.LOONG64:
obj, err = img.GOObjectLOONG64(*pkg, path)
case arch.ARM64:
obj, err = img.GOObjectAARCH64(*pkg, path)
default:
obj, err = img.GOObject(*pkg, path)
}
kind = "Go object"
default:
fmt.Fprintf(os.Stderr, "gasm asm: unknown format %q (want raw, elf, macho or goobj)\n", *format)
fmt.Fprintf(os.Stderr, "gasm asm: unknown format %q (want raw, elf or goobj)\n", *format)
return 2
}
if err != nil {
@@ -568,12 +566,12 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
}
// Assemble both files.
img1, err := assembleFile(path1)
img1, err := assemblePath(path1)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err)
return 1
}
img2, err := assembleFile(path2)
img2, err := assemblePath(path2)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err)
return 1
@@ -632,8 +630,24 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
return 1
}
// assembleFile assembles a file and returns the image.
func assembleFile(path string) (*asm.Image, error) {
// assembleFile assembles a parsed file for the given architecture and returns the image.
func assembleFile(path string, targetArch arch.Arch, f *ast.File) (*asm.Image, error) {
switch targetArch {
case arch.AMD64:
return asm.AssembleFile(f)
case arch.RISCV:
return asm.AssembleFileRISCV(f)
case arch.ARM64:
return asm.AssembleFileARM64(f)
case arch.LOONG64:
return asm.AssembleFileLOONG64(f)
default:
return nil, fmt.Errorf("unsupported architecture %q", targetArch)
}
}
// assemblePath reads, parses and assembles a file (used by cmdDiff).
func assemblePath(path string) (*asm.Image, error) {
src, err := readSource(path)
if err != nil {
return nil, err
@@ -645,11 +659,7 @@ func assembleFile(path string) (*asm.Image, error) {
if len(errs) > 0 {
return nil, fmt.Errorf("parse errors")
}
targetArch := arch.FromFilename(path)
if targetArch == arch.RISCV {
return asm.AssembleFileRISCV(f)
}
return asm.AssembleFile(f)
return assembleFile(path, arch.FromFilename(path), f)
}
// printByteDiff shows the first few byte differences between two code blocks.
@@ -822,6 +832,188 @@ func cmdVerifyRISCV(path string, groundTruth, profile bool) int {
return 0
}
// cmdVerifyLOONG64 verifies a loong64 source file against `go tool asm`
// (GOARCH=loong64) — the ground-truth oracle — since gasm cannot JIT-load
// LoongArch code on an amd64 host. Relocation sites are masked before the
// byte comparison, as the toolchain leaves them zero for the linker.
func cmdVerifyLOONG64(path string, groundTruth, profile bool) int {
src, err := readSource(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
f, errs := parser.Parse(path, src)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return 1
}
img, err := asm.AssembleFileLOONG64(f)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
if groundTruth {
gt, err := verify.GroundTruthLOONG64(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: ground truth: %v\n", err)
return 1
}
matched, total := 0, 0
for _, fn := range img.Funcs {
gasmCode := img.Code[fn.Offset : fn.Offset+fn.Size]
goCode, ok := gt[fn.Name]
if !ok {
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", fn.Name)
continue
}
total++
gasmCmp := make([]byte, len(gasmCode))
goCmp := make([]byte, len(goCode))
copy(gasmCmp, gasmCode)
copy(goCmp, goCode)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ {
gasmCmp[j] = 0
}
for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ {
goCmp[j] = 0
}
}
if bytes.Equal(gasmCmp, goCmp) {
matched++
if len(fn.Relocs) > 0 {
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
} else {
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
}
} else {
fmt.Printf(" %s: MISMATCH (%d vs %d bytes)\n", fn.Name, fn.Size, len(goCode))
for i := 0; i < len(gasmCode) || i < len(goCode); i += 16 {
var gb, gs string
for j := i; j < i+16 && j < len(gasmCode); j++ {
gb += fmt.Sprintf(" %02x", gasmCode[j])
}
for j := i; j < i+16 && j < len(goCode); j++ {
gs += fmt.Sprintf(" %02x", goCode[j])
}
fmt.Printf(" %04x: gasm:%s\n", i, gb)
fmt.Printf(" %04x: gt: %s\n", i, gs)
}
}
}
fmt.Printf("%s: %d/%d matched\n", path, matched, total)
if matched < total {
return 1
}
return 0
}
if profile {
for _, fn := range img.Funcs {
fmt.Printf("%s: %d bytes, labels: %v\n", fn.Name, fn.Size, fn.Labels)
}
return 0
}
fmt.Printf("%s: %d functions assembled\n", path, len(img.Funcs))
for _, fn := range img.Funcs {
fmt.Printf(" %s: %d bytes\n", fn.Name, fn.Size)
}
return 0
}
func cmdVerifyARM64(path string, groundTruth, profile bool) int {
src, err := readSource(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
f, errs := parser.Parse(path, src)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return 1
}
img, err := asm.AssembleFileARM64(f)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
if groundTruth {
gt, err := verify.GroundTruthARM64(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: ground truth: %v\n", err)
return 1
}
matched, total := 0, 0
for _, fn := range img.Funcs {
gasmCode := img.Code[fn.Offset : fn.Offset+fn.Size]
goCode, ok := gt[fn.Name]
if !ok {
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", fn.Name)
continue
}
total++
gasmCmp := make([]byte, len(gasmCode))
goCmp := make([]byte, len(goCode))
copy(gasmCmp, gasmCode)
copy(goCmp, goCode)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ {
gasmCmp[j] = 0
}
for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ {
goCmp[j] = 0
}
}
if bytes.Equal(gasmCmp, goCmp) {
matched++
if len(fn.Relocs) > 0 {
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
} else {
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
}
} else {
fmt.Printf(" %s: MISMATCH (%d vs %d bytes)\n", fn.Name, fn.Size, len(goCode))
for i := 0; i < len(gasmCode) || i < len(goCode); i += 16 {
var gb, gs string
for j := i; j < i+16 && j < len(gasmCode); j++ {
gb += fmt.Sprintf(" %02x", gasmCode[j])
}
for j := i; j < i+16 && j < len(goCode); j++ {
gs += fmt.Sprintf(" %02x", goCode[j])
}
fmt.Printf(" %04x: gasm:%s\n", i, gb)
fmt.Printf(" %04x: gt: %s\n", i, gs)
}
}
}
fmt.Printf("%s: %d/%d matched\n", path, matched, total)
if matched < total {
return 1
}
return 0
}
if profile {
for _, fn := range img.Funcs {
fmt.Printf("%s: %d bytes, labels: %v\n", fn.Name, fn.Size, fn.Labels)
}
return 0
}
fmt.Printf("%s: %d functions assembled\n", path, len(img.Funcs))
for _, fn := range img.Funcs {
fmt.Printf(" %s: %d bytes\n", fn.Name, fn.Size)
}
return 0
}
func cmdVerify(args []string) int {
fs := newCommand("verify", "gasm verify [-smoke] [-abi] [-fuzz] [-ground-truth] [-profile] [-call] <file.s>", `
Assemble FILE (amd64), map it into executable memory and report the available
@@ -867,14 +1059,21 @@ decoders) that crash on random input but should succeed on valid data.
}
path := fs.Arg(0)
targetArch := arch.FromFilename(path)
if targetArch != arch.AMD64 && targetArch != arch.RISCV {
fmt.Fprintln(os.Stderr, "gasm verify: only amd64 and riscv64 are supported")
return 1
}
// RISC-V: ground-truth only (no JIT on non-RISC-V hosts).
if targetArch == arch.RISCV {
switch targetArch {
case arch.AMD64:
// JIT-based verification below.
case arch.RISCV:
// RISC-V: ground-truth only (no JIT on non-RISC-V hosts).
return cmdVerifyRISCV(path, *groundTruth, *profile)
case arch.LOONG64:
// LoongArch: ground-truth only (no JIT on non-LoongArch hosts).
return cmdVerifyLOONG64(path, *groundTruth, *profile)
case arch.ARM64:
// AArch64: ground-truth only (no JIT on non-ARM64 hosts).
return cmdVerifyARM64(path, *groundTruth, *profile)
default:
fmt.Fprintln(os.Stderr, "gasm verify: only amd64, riscv64 and loong64 are supported")
return 1
}
k, err := verify.Load(path)
+50
View File
@@ -271,3 +271,53 @@ func TestBreakpointInfo(t *testing.T) {
t.Errorf("Info %q does not contain label", info)
}
}
func TestWatchpointSlotTracking(t *testing.T) {
s := &Session{}
// All four slots are free initially.
for i := 0; i < 4; i++ {
if s.IsWatchpointSlotUsed(i) {
t.Errorf("slot %d should be free initially", i)
}
}
if got := s.FindFreeWatchpointSlot(); got != 0 {
t.Errorf("FindFreeWatchpointSlot() = %d, want 0", got)
}
// Manually mark slots 0 and 2 as used (simulating successful SetWatchpoint).
s.wpSlots[0] = true
s.wpSlots[2] = true
if !s.IsWatchpointSlotUsed(0) {
t.Error("slot 0 should be in use")
}
if s.IsWatchpointSlotUsed(1) {
t.Error("slot 1 should be free")
}
if !s.IsWatchpointSlotUsed(2) {
t.Error("slot 2 should be in use")
}
if s.IsWatchpointSlotUsed(3) {
t.Error("slot 3 should be free")
}
if got := s.FindFreeWatchpointSlot(); got != 1 {
t.Errorf("FindFreeWatchpointSlot() = %d, want 1", got)
}
// Out-of-range slot queries return false.
if s.IsWatchpointSlotUsed(-1) {
t.Error("slot -1 should be reported as free (out of range)")
}
if s.IsWatchpointSlotUsed(4) {
t.Error("slot 4 should be reported as free (out of range)")
}
// Mark all slots used: FindFreeWatchpointSlot returns -1.
for i := 0; i < 4; i++ {
s.wpSlots[i] = true
}
if got := s.FindFreeWatchpointSlot(); got != -1 {
t.Errorf("FindFreeWatchpointSlot() with all slots used = %d, want -1", got)
}
}
+2 -1
View File
@@ -25,7 +25,8 @@ type Session struct {
cmd *exec.Cmd
stopped bool
exited bool
codeBase uint64 // base address of the JIT code in the debuggee
codeBase uint64 // base address of the JIT code in the debuggee
wpSlots [4]bool // watchpoint slot occupancy (DR0-DR3)
}
// Launch starts the debuggee subprocess (gasm debug --target ...) and
+25 -14
View File
@@ -384,8 +384,8 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Println(` break <label|addr> [if <reg> <op> <val>] set a breakpoint
delete <label|addr> remove a breakpoint
info break list all breakpoints
watch <addr> [r|w] set a hardware watchpoint (write by default)
unwatch clear all watchpoints
watch <addr> [r|w] [size] set a hardware watchpoint (write by default)
unwatch [<slot>] clear one or all watchpoints
step [n], s single-step n instructions
next, n step over CALL
continue, c run until breakpoint or exit
@@ -457,28 +457,39 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
if len(parts) > 3 {
size, _ = strconv.Atoi(parts[3])
}
// Find a free slot (0-3).
slot := -1
for i := 0; i < 4; i++ {
// Simple: use slot 0 for now.
slot = i
break
}
slot := s.FindFreeWatchpointSlot()
if slot < 0 {
fmt.Println("no free watchpoint slots")
fmt.Println("no free watchpoint slots (use 'unwatch <slot>' to clear one)")
continue
}
if err := s.SetWatchpoint(slot, addr, typ, size); err != nil {
fmt.Printf("watch: %v\n", err)
} else {
fmt.Printf("watchpoint %d set: %#x (%s, %d bytes)\n", slot, addr, parts[2], size)
typStr := "w"
if typ == WatchRead {
typStr = "r"
}
fmt.Printf("watchpoint %d set: %#x (%s, %d bytes)\n", slot, addr, typStr, size)
}
case "unwatch":
if err := s.ClearAllWatchpoints(); err != nil {
fmt.Printf("unwatch: %v\n", err)
if len(parts) >= 2 {
slot, err := strconv.Atoi(parts[1])
if err != nil || slot < 0 || slot > 3 {
fmt.Println("usage: unwatch [<slot>]")
continue
}
if err := s.ClearWatchpoint(slot); err != nil {
fmt.Printf("unwatch: %v\n", err)
} else {
fmt.Printf("watchpoint %d cleared\n", slot)
}
} else {
fmt.Println("all watchpoints cleared")
if err := s.ClearAllWatchpoints(); err != nil {
fmt.Printf("unwatch: %v\n", err)
} else {
fmt.Println("all watchpoints cleared")
}
}
default:
+36 -4
View File
@@ -25,12 +25,34 @@ const (
WatchRead WatchpointType = 3 // trigger on read or write
)
// FindFreeWatchpointSlot returns the index of the first free watchpoint slot
// (0-3), or -1 if all four hardware watchpoints are in use.
func (s *Session) FindFreeWatchpointSlot() int {
for i := 0; i < 4; i++ {
if !s.wpSlots[i] {
return i
}
}
return -1
}
// IsWatchpointSlotUsed reports whether slot (0-3) currently holds a watchpoint.
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot > 3 {
return false
}
return s.wpSlots[slot]
}
// SetWatchpoint installs a hardware watchpoint on the given address.
// slot is 0-3 (four hardware watchpoints available).
// slot is 0-3 (four hardware watchpoints available); the slot must be free.
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
if slot < 0 || slot > 3 {
return fmt.Errorf("debug: watchpoint slot must be 0-3")
}
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
// Determine the length encoding.
var lenBits uint64
@@ -82,6 +104,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
return fmt.Errorf("debug: set DR7: %w", err)
}
s.wpSlots[slot] = true
return nil
}
@@ -90,20 +113,29 @@ func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot > 3 {
return fmt.Errorf("debug: watchpoint slot must be 0-3")
}
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
// Read DR7, clear the enable bit for this slot.
dr7, err := ptracePeekUser(s.pid, 0x38)
if err != nil {
return err
}
dr7 &^= uint64(1) << (2 * slot) // disable
return ptracePokeUser(s.pid, 0x38, dr7)
if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
return err
}
s.wpSlots[slot] = false
return nil
}
// ClearAllWatchpoints removes all hardware watchpoints.
func (s *Session) ClearAllWatchpoints() error {
for slot := 0; slot < 4; slot++ {
if err := s.ClearWatchpoint(slot); err != nil {
return err
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
}
}
}
return nil
+40 -8
View File
@@ -201,6 +201,28 @@ R_RISCV_PCREL_HI20/LO12 relocations). The encoder compresses eligible
instructions to 16-bit RVC forms and is validated byte-for-byte against
`GOARCH=riscv64 go tool asm`.
A **LoongArch encoder** (Phase 5, LoongArch64) encodes the integer and
floating-point instruction sets with the dual-form arithmetic mnemonics (3R
vs 2RI12), the 16/21-bit branch families, the MOV pseudo-instruction and its
constant materialisation (the dcon classification driving lu12i.w/ori/lu32i.d/
lu52i.d expansions), the FP/SP frame mapping (autosize = align8(frame+8),
prologue storing the link register before and after the SP decrement) and
SB/global symbol references (pcalau12i pairs with R_LOONG64_ADDR_HI/LO
relocations). Like the RISC-V encoder it is validated byte-for-byte against
`GOARCH=loong64 go tool asm`, and its GOOBJ output is proven end-to-end by
substituting it into a cross-compiled `go build` and linking with `cmd/link`.
An **AArch64 encoder** (Phase 5, arm64) encodes the integer instruction set
with the data-processing (shifted register and immediate forms), load/store
(scaled unsigned immediate and unscaled9-bit immediate), conditional and
unconditional branches, the MOV pseudo-instruction and its constant
materialisation (MOVZ/MOVN/MOVK for wide immediates, ORR with logical bitmask
encoding for values like `$1`), the FP/SP frame mapping (autosize =
align16(frame+8), prologue using pre-index store for small frames and
STP+SUB for large frames) and SB/global symbol references (ADRP+ADD pairs with
R_ADDRARM64 relocations). Like the other encoders it is validated
byte-for-byte against `GOARCH=arm64 go tool asm`.
On top of the encoder, `Assemble` walks a parsed `TEXT` body, converts each
operand to an encoder operand, and lays the instructions out so local labels
resolve to relative jump offsets: jumps start in the short (rel8) form and
@@ -275,8 +297,8 @@ RIP-relative loads whose displacements point inside the resulting image, so
the bytes are self-consistent at any base address. References to symbols no
`GLOBL` defines are kept as relocations on the function layout, and the
object-file emitters turn the whole image into a linkable object: the ELF
and Mach-O writers (`gasm asm --format elf|macho`) lay the code and data out
as `.text`/`.data` (or `__text`/`__data`) sections, export a symbol per
writer (`gasm asm --format elf`) lays the code and data out as `.text`/`.data`
sections, exports a symbol per
`TEXT` and `GLOBL` (the `<>` ones local, the rest global) and emit one
PC-relative relocation per static-symbol reference — undefined external
symbols included, so the output links with the system toolchain. The GOOBJ
@@ -288,12 +310,22 @@ boundaries, plus flat `pcfile`, `pcline` and `pcinline` tables — so a
gasm-assembled object drops into a `go build` in place of the toolchain's.
The object preamble (the version-and-experiment header the linker compares
verbatim) is captured from the installed `go tool asm`, so the output is
always consistent with the toolchain that links it. RISC-V GOOBJ emission
uses the same format with the RISC-V architecture marker and RISC-V relocation
types. External cross-package
references and the implicit funcdata/DWARF symbols remain future work (the
linker fills the latter's defaults); the rest of Phase 2 is those, the
remaining EVEX forms and the other architectures.
always consistent with the toolchain that links it. RISC-V and LoongArch
GOOBJ emission share this emitter: the loong64 marker with
R_LOONG64_ADDR_HI/LO relocation types, and the riscv64 marker with a single
R_RISCV_PCREL_ITYPE/STYPE relocation per AUIPC pair (plus `R_RISCV_JAL` for
`CALL sym(SB)`) — the model `cmd/asm`
writes, not the ELF HI20/LO12 pair — and both link into a real `go build` for
their `GOARCH`. Per function, the emitter also writes the two DWARF
symbols the linker's DWARF pass reads verbatim — the subprogram DIE
(`SDWARFFCN`) and the `.debug_line` state-machine program (`SDWARFLINES`),
both built the way `cmd/asm` builds them (the DIE carries the
R_DWTXTADDR_U4 address reference; the line program one row per source-line
change, in the same special-opcode encoding) — and the pc-value deltas are
in the architecture's MinLC units, as the runtime's `pcvalue` expects.
External cross-package references remain future work (the amd64 and RISC-V
paths resolve them; LoongArch does not yet); the rest of Phase 2 is those
and the remaining EVEX forms.
### `verify`
+20 -7
View File
@@ -50,13 +50,13 @@ Rules: `unknown-instruction`, `operand-count`, `undefined-label`,
`abi-argsize`, `unreachable-code`, `register-clobber`,
`funcdata-pcdata`.
## `gasm asm [--format raw|elf|macho|goobj] [-p pkg] [-o out] <file>`
## `gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>`
Assemble FILE (amd64) to machine code.
| Flag | Description |
|------|-------------|
| `--format` | Output format: `raw` (default), `elf`, `macho`, `goobj` |
| `--format` | Output format: `raw` (default), `elf`, `goobj` |
| `-p` | Package path (required for `--format goobj`) |
| `-o` | Write output to file (default: hex dump to stdout) |
@@ -102,13 +102,26 @@ REPL commands:
| Command | Description |
|---------|-------------|
| `break <label\|addr>` | Set a breakpoint |
| `step [n]` | Single-step n instructions |
| `continue` | Run until next breakpoint or exit |
| `break <label\|addr> [if <reg> <op> <val>]` | Set a breakpoint, optionally conditional |
| `delete <label\|addr>` | Remove a breakpoint |
| `info break` | List all breakpoints |
| `step [n]`, `s` | Single-step n instructions |
| `next`, `n` | Step over CALL |
| `finish`, `fin` | Run until the function returns |
| `continue`, `c` | Run until breakpoint, watchpoint or exit |
| `disas [n]`, `u` | Disassemble n instructions at PC |
| `regs` | Print general-purpose + YMM/XMM vector registers |
| `where` | Show source line and nearest label at PC |
| `stack` | Show stack near RSP (return address + ABI0 args) |
| `bt`, `backtrace` | Backtrace (current frame + return address) |
| `x [addr] [len]` | Hex-dump memory |
| `labels` | List function labels and offsets |
| `quit` | Kill the debuggee and exit |
| `w <addr> <val...>` | Write bytes to memory |
| `set <reg> <value>` | Set a register |
| `watch <addr> [r\|w] [size]` | Set a hardware watchpoint (write by default) |
| `unwatch [<slot>]` | Clear one or all watchpoints |
| `labels`, `l` | List function labels and offsets |
| `help`, `h`, `?` | Show command help |
| `quit`, `q` | Kill the debuggee and exit |
## `gasm diff [--map old=new,...] <file1.s> <file2.s>`
+34
View File
@@ -0,0 +1,34 @@
# Deferred decisions
Design decisions deliberately postponed, with enough context to pick them up
again without re-deriving the analysis. Each entry records what is deferred,
why, the options on the table, and the trigger that should reopen it.
---
## GOOBJ external (cross-package) symbol references
**Status:** resolved (v0.29.0+, 2026-08-07).
**Approach taken.** Instead of parsing the compiler's iexport data (which
would have required either `golang.org/x/tools` or an in-house parser), the
resolver reads the **GOOBJ data directly** from the target package's `.a`
archive. The `.a` file contains a `_go_.o` member whose GOOBJ format is the
same one gasm writes — the parser reuses the same layout (`blkSymdef`,
`blkNonpkgdef`, the string table), so no new dependency was needed.
**How it works.**
1. `go list -json -export <pkg>` finds the target package's `.a` file.
2. `extractGOOBJ` reads the ar archive, finds the `_go_.o` member, skips
the `"go object …\n!\n"` preamble and parses the GOOBJ header.
3. `goobjFile.symbols()` walks `blkSymdef` and `blkNonpkgdef` in definition
order — the same order the linker uses — to build the symbol → index
mapping.
4. `resolveExternalSymbols` wires the resolved `{PkgIdx, SymIdx}` into the
GOOBJ emission.
The resolver is invoked automatically when `img.Externals` is non-empty; it
runs `go list` as a subprocess (consistent with `toolchainObjectPreamble`
which already calls `go tool asm`). All symbol data is cached per package
for the lifetime of the GOOBJ emission.
-56
View File
@@ -1,56 +0,0 @@
# Deferred decisions
Design decisions deliberately postponed, with enough context to pick them up
again without re-deriving the analysis. Each entry records what is deferred,
why, the options on the table, and the trigger that should reopen it.
---
## GOOBJ external (cross-package) symbol references
**Status:** deferred (v0.15.0, 2026-08-02). The GOOBJ emitter resolves only
symbols defined in the file being assembled; a reference to any other symbol
is rejected.
**Why it is deferred.** GOOBJ symbol references are *positional*: a
reference is a `{PkgIdx, SymIdx}` pair, where `SymIdx` is the index of the
symbol in the *referenced package's* symbol-definition table. That ordering
is not derivable from the reference site — it lives in the referenced
package's gc export data (the iexport binary format, which evolves with the
toolchain). `cmd/asm` reads it with `cmd/internal` readers gasm cannot
import, so emitting external references means either parsing export data
ourselves or taking a dependency that does.
**What works today.** Single-package objects: every symbol the file defines
(as `TEXT` or `GLOBL`, static or exported) and every reference to them.
This covers the production use case — the go-flac / go-lz4 kernels carry no
`FUNCDATA`/`PCDATA`, hence no references into `runtime`, and the Go side
references the assembly symbols, never the reverse. Such a package builds
with its assembly object replaced by a gasm-emitted one.
**The options, when we return.**
1. **`golang.org/x/tools/go/gcexportdata` as a production dependency.**
The straightforward path: read each imported package's export file
(paths from `-importcfg` or `go list -export`), assign symbol indices in
its symbol order, write `PkgIndex`/`Autolib` entries (fingerprints from
the export files' build IDs) and positional references. Robust across
toolchain versions — `x/tools` tracks the format. **Cost:** the first
production dependency beyond the standard library, an explicit deviation
from the "production code depends only on the standard library"
principle in the README. Requires the user's explicit agreement.
2. **A minimal iexport parser of our own.** Preserves self-containment.
Substantial effort and inherently fragile: the format is an internal
contract that changes with Go releases, so the parser needs a
version-gated fallback and regression tests against several toolchains.
3. **Shell out to the toolchain for symbol metadata.** Consistent with the
existing GOOBJ preamble probe (which already runs `go tool asm`), but no
toolchain command exposes a package's symbols *in definition-index
order* — `go tool nm` sorts differently — so this does not solve the
core problem on its own; it would only feed option 1 or 2.
**Trigger to reopen.** An assembly file that needs a cross-package
reference — in practice `FUNCDATA $…, runtime·…(SB)` (stack maps / GC
metadata written in assembly), or any kernel that calls into another
package directly. Until then, option 3's limitation is moot and the
single-package emitter suffices.
+1 -1
View File
@@ -4,7 +4,7 @@ Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrb
## Prerequisites
- **Go** 1.26+ with `toolchain go1.26.5`
- **Go** 1.27+ with `toolchain go1.27.0`
- **just** — the command runner; every task below is a just recipe
- No external dependencies beyond the Go toolchain
-79
View File
@@ -1,79 +0,0 @@
# Using gasm-devkit with Zed
This document is deliberately blunt, because the situation is a genuine
conflict between two of the project's own commitments, and papering over it
would be dishonest.
## The conflict
gasm-devkit is **pure Go, no C, no cgo, no JavaScript runtimes, no native
binaries, no vendor lock-in, no platform-specific IDE internals.**
Zed's extension model, as verified against Zed's own documentation, is:
- Extensions are written in **Rust** and compiled to **WebAssembly**
(`wasm32-wasip2`).
- Syntax highlighting is provided by **Tree-sitter** grammars, which are
**C** compiled to WebAssembly with the wasi-sdk, from a grammar written in a
**JavaScript** DSL.
- A *new* language cannot be registered through configuration alone. Defining
a language requires an extension, and every language extension must name a
Tree-sitter grammar. (Zed's `lsp` settings section configures
already-registered servers; it does not register an arbitrary external binary
for a brand-new language.)
There is therefore **no pure-Go path into Zed's extension host.** This is a
property of Zed, not of gasm-devkit: no language tooling author can feed Zed a
pure-Go highlighting grammar, because Zed's highlighting engine is Tree-sitter
and its plugin runtime is Rust/WASM.
## What gasm-devkit gives Zed regardless
The toolkit's integration surface is the **Language Server Protocol**, an open
standard. Through `gasm lsp` it provides, with zero editor-specific code:
- autocomplete (instructions, registers, pseudo-registers, labels),
- hover documentation,
- diagnostics (the linter, pushed as you type),
- document outline (functions and labels),
- **syntax highlighting, delivered as LSP semantic tokens.**
That last point matters: Zed can render highlighting entirely from LSP semantic
tokens (`"semantic_tokens": "full"` replaces Tree-sitter highlighting for a
language). So the highlighting *capability* exists in pure Go; what Zed needs
is merely to be told that `.s` files are a language served by `gasm lsp`.
## The honest options
1. **Use an editor that registers an external LSP by configuration.**
Neovim, Helix, VS Code and Sublime all let you associate `.s` with the
`gasm lsp` binary and use its semantic tokens — no Rust, no C, no lock-in.
This is the option that satisfies every stated constraint with no
exception.
2. **Treat a Zed adapter as one quarantined exception.** A minimal Zed
extension — a few lines of Rust that register the language and launch
`gasm lsp` — plus either a Tree-sitter grammar or `"full"` semantic tokens
for highlighting. Crucially, this adapter is the *editor's plugin format*;
it is sandboxed inside Zed and never linked into, compiled into, or shipped
with the Go toolkit. gasm-devkit itself stays pure Go. But producing it
uses the Rust/wasi-sdk/Tree-sitter toolchain, which the project constraints
forbid — so it must be a conscious, explicit decision, not a silent one.
The author's philosophy — digital sovereignty, no dependency on toolchains he
does not control — is the tie-breaker, and it is a value judgement rather than
a technical one. gasm-devkit is built so that **either** choice keeps the
toolkit itself clean: the pure-Go core and the LSP are the product; a Zed
adapter, if ever wanted, is a thin, separable leaf.
## Wiring the LSP (editor-agnostic)
Run the server and point an LSP client at it:
```sh
go run ./cmd/gasm lsp # or: go install ./cmd/gasm && gasm lsp
```
Associate the command with `*.s` (and `*_amd64.s` / `*_arm64.s`) in whichever
editor you use. The server infers the target architecture from the file-name
suffix and selects the amd64 or arm64 instruction tables accordingly.
+2 -2
View File
@@ -1,7 +1,7 @@
module sourcedock.dev/petrbalvin/gasm-devkit
go 1.26
go 1.27
toolchain go1.26.5
toolchain go1.27.0
require golang.org/x/arch v0.29.0
+1 -1
View File
@@ -3,7 +3,7 @@
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
version := "0.29.0"
version := "0.31.1"
default:
@just --list
+5 -1
View File
@@ -298,10 +298,14 @@ func parseSymbolPrefix(g []token.Token) (*ast.Symbol, int) {
sym.Static = true
i += 2
}
if i < len(g) && g[i].Kind == token.Plus {
if i < len(g) && (g[i].Kind == token.Plus || g[i].Kind == token.Minus) {
neg := g[i].Kind == token.Minus
i++
if i < len(g) && g[i].Kind == token.Number {
sym.Offset, sym.HasOff = parseInt(g[i].Text), true
if neg {
sym.Offset = -sym.Offset
}
i++
}
}
+57
View File
@@ -0,0 +1,57 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// add returns a + b.
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
// arith exercises the register-register integer set.
TEXT ·arith(SB), NOSPLIT, $0-0
ADD R4, R5, R6
SUB R7, R8, R9
AND R10, R11, R12
ORR R12, R13, R14
EOR R14, R15, R16
CMP R16, R17
ADD R4, R5
SUB R6, R7
RET
// branch exercises conditional and unconditional control flow.
TEXT ·branch(SB), NOSPLIT, $0-0
BEQ done
BNE skip
BGE done
BLT done
BGT done
BLE done
skip:
B loop
loop:
ADD R4, R5
RET
done:
RET
// mov exercises the MOV pseudo-instruction.
TEXT ·mov(SB), NOSPLIT, $0-16
MOVD $0, R4
MOVD $1, R5
MOVD $42, R6
MOVD a+0(FP), R7
MOVD R7, ret+0(FP)
MOVW $100, R8
RET
// frame exercises the prologue/epilogue of a function with a real frame.
TEXT ·frame(SB), NOSPLIT, $32-8
MOVD arg+0(FP), R4
ADD $1, R4, R4
MOVD R4, ret+0(FP)
RET
+54
View File
@@ -0,0 +1,54 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// add returns a + b.
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
// arith exercises the 3R integer and FP set.
TEXT ·arith(SB), NOSPLIT, $0-0
ADDV R4, R5, R6
SUBV R7, R8, R9
MULV R10, R11, R12
DIVV R13, R14, R15
AND R16, R17, R18
OR R18, R19, R20
XOR R20, R21, R2
SLLV R2, R23, R24
SRLV R24, R25, R26
SRAV R26, R27, R28
RET
// imm exercises the immediate forms.
TEXT ·imm(SB), NOSPLIT, $0-0
ADDV $42, R4, R5
ADDV $-8, R6
AND $0xff, R7, R8
OR $1, R9, R10
XOR $0, R11, R12
SGT $100, R13, R14
SLLV $4, R15, R16
MOVV $0x12345, R17
RET
// branch exercises conditional and unconditional control flow.
TEXT ·branch(SB), NOSPLIT, $0-0
BEQ R4, R5, done
BNE R6, R7, skip
BLT R8, R9, done
BGE R10, R11, done
BLTU R12, R13, done
BGEU R14, R15, done
skip:
JMP loop
loop:
JAL skip
RET
done:
RET
+18
View File
@@ -0,0 +1,18 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·framed(SB), NOSPLIT, $16-16
MOV a+0(FP), X10
MOV b+8(FP), X11
ADD X11, X10, X10
MOV X10, ret+16(FP)
RET
TEXT ·leaf(SB), NOSPLIT, $0-16
MOV a+0(FP), X10
MOV b+8(FP), X11
ADD X11, X10, X10
MOV X10, ret+16(FP)
RET
+27
View File
@@ -0,0 +1,27 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// branch exercises all conditional branch forms and jump chain folding.
TEXT ·branch(SB), NOSPLIT, $0-0
BEQ done
BNE skip
BGE done
BLT done
BGT done
BLE done
BCS done
BCC done
BMI done
BPL done
BVS done
BVC done
BHI done
BLS done
skip:
B loop
loop:
ADD R4, R5
done:
RET
+35
View File
@@ -0,0 +1,35 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·branches(SB), NOSPLIT, $0
ADDI $1, X10, X10
BEQ X10, X11, beq_done
ADDI $2, X10, X10
beq_done:
BNE X10, X11, bne_done
ADDI $3, X10, X10
bne_done:
BLT X10, X11, blt_done
ADDI $4, X10, X10
blt_done:
BGE X10, X11, bge_done
ADDI $5, X10, X10
bge_done:
BLTU X10, X11, bltu_done
ADDI $6, X10, X10
bltu_done:
BGEU X10, X11, bgeu_done
ADDI $7, X10, X10
bgeu_done:
RET
TEXT ·jumps(SB), NOSPLIT, $0
JMP done
ADDI $1, X10, X10
done:
JAL X11, skip
ADDI $2, X10, X10
skip:
RET
+10
View File
@@ -0,0 +1,10 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// caller exercises BL to an external symbol (produces a relocation).
TEXT ·caller(SB), NOSPLIT, $0-0
BL other(SB)
ADD R4, R5
RET
+8
View File
@@ -0,0 +1,8 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·call(SB), NOSPLIT, $0
CALL callee(SB)
RET
+119
View File
@@ -0,0 +1,119 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// fparith exercises the FP arithmetic set.
TEXT ·fparith(SB), NOSPLIT, $0-0
FADDD F0, F1, F2
FSUBD F3, F4, F5
FMULD F6, F7, F8
FDIVD F9, F10, F11
FADDS F12, F13, F14
FSUBS F15, F16, F17
FMULS F18, F19, F20
FDIVS F21, F22, F23
FSQRTD F24, F25
FSQRTS F26, F27
FNEGD F28, F29
FNEGS F30, F31
FABSD F0, F1
FABSS F2, F3
FNMULD F4, F5, F6
FNMULS F7, F8, F9
FMIND F10, F11, F12
FMAXD F13, F14, F15
FMINS F16, F17, F18
FMAXS F19, F20, F21
RET
// fpfma exercises fused multiply-add.
TEXT ·fpfma(SB), NOSPLIT, $0-0
FMADDD F0, F1, F2, F3
FMSUBD F4, F5, F6, F7
FNMADDD F8, F9, F10, F11
FNMSUBD F12, F13, F14, F15
FMADDS F16, F17, F18, F19
FMSUBS F20, F21, F22, F23
FNMADDS F24, F25, F26, F27
FNMSUBS F28, F29, F30, F0
RET
// fpconv exercises FP↔integer conversion and cross-precision.
// Syntax: FCVTZSD Fd, Rn (float→int: FP source first, int dest second)
// SCVTFD Rn, Fd (int→float: int source first, FP dest second)
TEXT ·fpconv(SB), NOSPLIT, $0-0
FCVTSD F0, F1
FCVTDS F2, F3
FCVTZSD F4, R0
FCVTZSS F5, R1
FCVTZUD F6, R2
FCVTZUS F7, R3
SCVTFD R4, F8
SCVTFS R5, F9
UCVTFD R6, F10
UCVTFS R7, F11
SCVTFWD R0, F12
SCVTFWS R1, F13
UCVTFWD R2, F14
UCVTFWS R3, F15
FMOVS F14, R20
FMOVS R21, F15
FMOVD F16, R22
FMOVD R23, F17
RET
// fpcmp exercises FP compare and conditional compare.
// FCCMP syntax: FCCMP cond, Fn, Fm, $nzcv
// FCSEL syntax: FCSEL cond, Fn, Fm, Fd
TEXT ·fpcmp(SB), NOSPLIT, $0-0
FCMPS F0, F1
FCMPD F2, F3
FCMPS $0.0, F4
FCMPD $0.0, F5
FCCMPS EQ, F6, F7, $0
FCCMPD NE, F8, F9, $0
FCSELS GE, F10, F11, F12
FCSELD LT, F13, F14, F15
RET
// frint exercises FP rounding.
TEXT ·frint(SB), NOSPLIT, $0-0
FRINTND F0, F1
FRINTNS F2, F3
FRINTPD F4, F5
FRINTPS F6, F7
FRINTMD F8, F9
FRINTMS F10, F11
FRINTZD F12, F13
FRINTZS F14, F15
FRINTAD F16, F17
FRINTAS F18, F19
FRINTXD F20, F21
FRINTXS F22, F23
FRINTID F24, F25
FRINTIS F26, F27
FMOVD F0, F1
FMOVS F2, F3
RET
// condsel exercises conditional select and CRC32.
TEXT ·condsel(SB), NOSPLIT, $0-0
CSEL EQ, R0, R1, R2
CSINC NE, R3, R4, R5
CSINV GE, R6, R7, R8
CSNEG LT, R9, R10, R11
CSET EQ, R12
CSETM NE, R13
CINC EQ, R14, R15
CINV NE, R16, R17
CNEG GE, R19, R20
CRC32B R0, R2
CRC32H R3, R5
CRC32W R6, R8
CRC32X R9, R11
CRC32CB R12, R14
CRC32CH R15, R0
CRC32CW R1, R3
CRC32CX R4, R6
RET
+143
View File
@@ -0,0 +1,143 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// fp exercises the floating-point set: 3R arithmetic, 2R unary, compares
// into FCC, fused multiply-add and the register moves.
TEXT ·fp(SB), NOSPLIT, $0-0
ADDD F4, F5, F6
SUBD F7, F8, F9
MULD F9, F10, F11
DIVD F11, F12, F13
MULF F13, F14, F15
ADDF F15, F16, F17
SQRTD F17, F18
SQRTF F18, F19
ABSD F19, F20
NEGD F20, F21
MOVD F21, F22
CMPEQD F22, F23, FCC0
CMPGTF F23, F24, FCC1
CMPGED F24, F25, FCC2
FMADDD F0, F1, F2, F3
FMSUBF F3, F4, F5, F6
FNMADDD F6, F7, F8, F9
FNMSUBF F9, F10, F11, F12
FMAXD F12, F13, F14
FMINF F14, F15, F16
FMAXAD F16, F17, F18
FMINAF F18, F19, F20
FSCALEBF F20, F21, F22
FCOPYSGD F22, F23, F24
MOVV F25, R25
MOVV R26, F27
MOVW R28, F29
MOVW F30, R31
RET
// mov forms: register moves, immediates (12/32/64-bit), memory with FP/SP
// pseudo-registers and the register-indexed forms.
TEXT ·mov(SB), NOSPLIT, $0-16
MOVV R4, R5
MOVW R6, R7
MOVB R8, R9
MOVBU R10, R11
MOVHU R12, R13
MOVWU R14, R15
MOVV $42, R16
MOVV $0x12345, R17
MOVV $0x100000, R18
MOVW $-100, R19
MOVV $0x123456789, R20
MOVV a+0(FP), R21
MOVV R23, b+8(FP)
MOVW c+16(FP), R24
MOVV (R24)(R25), R26
MOVV R27, (R28)(R29)
RET
// frame exercises the prologue/epilogue of a function with a real frame.
TEXT ·frame(SB), NOSPLIT, $32-8
MOVV R4, R5
MOVV arg+0(FP), R6
MOVV R7, local-8(SP)
MOVV local-8(SP), R8
MOVV R9, ret+0(FP)
RET
// branches21 exercises the single-register and zero-register branch forms
// with 21-bit offsets.
TEXT ·branches21(SB), NOSPLIT, $0-0
BEQ R0, R4, l1
BEQ R5, R0, l2
BNE R0, R6, l3
BNE R7, R0, l4
BLTZ R8, l5
BGEZ R9, l6
BLEZ R10, l7
BGTZ R11, l8
JMP l9
l1:
JMP l10
l2:
JMP l11
l3:
JMP l12
l4:
JMP l13
l5:
JMP l14
l6:
JMP l15
l7:
JMP l16
l8:
JMP l16
l9:
MOVV R1, R2
l10:
LL (R12), R13
LLV (R14), R15
SC R16, (R17)
SCV R18, (R19)
RDTIMED R20, R21
SYSCALL
DBAR
RET
l11:
JAL (R30)
RET
l12:
BSTRINSV $7, R4, $0, R5
BSTRPICKV $63, R6, $32, R7
ALSLV $2, R8, R9, R10
ADDV16 $65536, R11, R12
RET
l13:
MOVV $0xffffffffffffffff, R13
RET
l14:
CPUCFG R14, R14
RET
l15:
NOR R15, R16, R17
ORN R18, R19, R20
ANDN R21, R24, R25
RET
l16:
MOVB R26, (R27)
MOVB (R28), R29
RET
// sbdata loads and stores a static symbol with relocations (the relocation
// fields are masked before comparison).
GLOBL ·table(SB), RODATA, $16
DATA ·table+0(SB)/8, $0x1122334455667788
DATA ·table+8(SB)/8, $0x8877665544332211
TEXT ·sbdata(SB), NOSPLIT, $0-0
MOVV $·table(SB), R4
MOVV ·table(SB), R5
MOVV R6, ·table+8(SB)
RET
+12
View File
@@ -0,0 +1,12 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·largeimm(SB), NOSPLIT, $0
ADDI $2048, X5
ADDI $4095, X5, X6
ANDI $4095, X5, X6
ORI $-4096, X5, X6
XORI $0x12345, X5, X6
RET
+24
View File
@@ -0,0 +1,24 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·ldst(SB), NOSPLIT, $0
LD (X8), X9
SD X9, (X8)
LW (X8), X9
SW X9, (X8)
LD 8(X2), X10
SD X10, 16(X2)
LW 4(X2), X11
SW X11, 8(X2)
RET
TEXT ·addi4spn(SB), NOSPLIT, $0
ADDI $16, X2, X8
RET
TEXT ·wordarith(SB), NOSPLIT, $0
ADDW X9, X8, X8
SUBW X9, X8, X8
RET
+21
View File
@@ -0,0 +1,21 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// movimm exercises MOV with various immediate values.
TEXT ·movimm(SB), NOSPLIT, $0-0
MOVD $0, R0
MOVD $1, R1
MOVD $42, R2
MOVD $255, R3
MOVD $256, R4
MOVD $0xFFFF, R5
MOVD $0x12345678, R6
MOVD $0x123456789ABCDEF0, R7
MOVD $-1, R8
MOVD $-2, R9
MOVW $0, R10
MOVW $100, R11
MOVW $0x12345, R12
RET
+19
View File
@@ -0,0 +1,19 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·movimm(SB), NOSPLIT, $0
MOV $0, X10
MOV $5, X10
MOV $42, X10
MOV $-1, X10
MOV $-2048, X10
MOV $-2049, X10
MOV $2047, X10
MOV $2048, X10
MOV $4095, X10
MOV $-4096, X10
MOV $0x12345, X10
MOV $2147483647, X10
RET
+32
View File
@@ -0,0 +1,32 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·shifts(SB), NOSPLIT, $0
SLLI $3, X10, X10
SRLI $2, X10, X10
SRAI $1, X10, X10
RET
TEXT ·logic(SB), NOSPLIT, $0
AND X11, X10, X10
OR X11, X10, X10
XOR X11, X10, X10
ANDI $7, X10, X10
RET
TEXT ·mv(SB), NOSPLIT, $0
ADDI $0, X11, X10
RET
TEXT ·nop(SB), NOSPLIT, $0
ADDI $0, X0
RET
TEXT ·bigframe(SB), NOSPLIT, $24-0
RET
TEXT ·ebreak(SB), NOSPLIT, $0
EBREAK
RET
+87
View File
@@ -0,0 +1,87 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"os"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGroundTruthARM64 assembles the arm64 test kernels with gasm and
// compares them byte-for-byte against `go tool asm` (GOARCH=arm64). The
// relocation fields of static-symbol references are masked before the
// comparison, since the toolchain leaves them zero for the linker.
func TestGroundTruthARM64(t *testing.T) {
for _, path := range []string{
"../testdata/verify/basic_arm64.s",
"../testdata/verify/fp_arm64.s",
"../testdata/verify/movimm_arm64.s",
"../testdata/verify/branch_arm64.s",
"../testdata/verify/call_arm64.s",
} {
t.Run(path, func(t *testing.T) {
src, err := os.ReadFile(path)
if err != nil {
t.Fatalf("read: %v", err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := asm.AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
gt, err := GroundTruthARM64(path)
if err != nil {
t.Fatalf("GroundTruthARM64: %v", err)
}
matched := 0
for _, fn := range img.Funcs {
gasmCode := maskRelocs(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
goCode, ok := gt[fn.Name]
if !ok {
t.Errorf("%s: not in ground truth (%d functions)", fn.Name, len(gt))
continue
}
goCode = maskRelocs(goCode, fn.Relocs)
// The Go toolchain may add zero padding at the end of
// functions. Compare up to the shorter length, then
// verify any trailing bytes are zero.
cmpLen := len(gasmCode)
if len(goCode) < cmpLen {
cmpLen = len(goCode)
}
if !bytes.Equal(gasmCode[:cmpLen], goCode[:cmpLen]) {
t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\n%s", fn.Name, len(gasmCode), len(goCode), diffHex(gasmCode, goCode))
continue
}
// Check trailing padding is zero.
trailingOK := true
if len(goCode) > len(gasmCode) {
for _, b := range goCode[len(gasmCode):] {
if b != 0 {
trailingOK = false
break
}
}
}
if !trailingOK {
t.Errorf("%s: non-zero trailing bytes in go tool asm output", fn.Name)
continue
}
matched++
t.Logf("%s: MATCH (%d bytes, go=%d)", fn.Name, fn.Size, len(goCode))
}
if matched == 0 {
t.Fatal("no functions matched")
}
})
}
}
+14
View File
@@ -32,6 +32,18 @@ func GroundTruthRISCV(path string) (map[string][]byte, error) {
return groundTruthArch(path, "riscv64")
}
// GroundTruthLOONG64 assembles the given .s file with the Go toolchain in
// LoongArch cross-assembly mode (GOARCH=loong64).
func GroundTruthLOONG64(path string) (map[string][]byte, error) {
return groundTruthArch(path, "loong64")
}
// GroundTruthARM64 assembles the given .s file with the Go toolchain in
// AArch64 cross-assembly mode (GOARCH=arm64).
func GroundTruthARM64(path string) (map[string][]byte, error) {
return groundTruthArch(path, "arm64")
}
func groundTruthArch(path, goarch string) (map[string][]byte, error) {
goroot := runtime.GOROOT()
asmBin := filepath.Join(goroot, "pkg", "tool", runtime.GOOS+"_"+runtime.GOARCH, "asm")
@@ -51,6 +63,8 @@ func groundTruthArch(path, goarch string) (map[string][]byte, error) {
pkg := strings.TrimSuffix(base, ".s")
pkg = strings.TrimSuffix(pkg, "_amd64")
pkg = strings.TrimSuffix(pkg, "_riscv64")
pkg = strings.TrimSuffix(pkg, "_loong64")
pkg = strings.TrimSuffix(pkg, "_arm64")
cmd := exec.Command(asmBin, "-I", includeDir, "-p", pkg, "-o", objPath, path)
if goarch != "" {
+110
View File
@@ -0,0 +1,110 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"fmt"
"os"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGroundTruthLOONG64 assembles the loong64 test kernels with gasm and
// compares them byte-for-byte against `go tool asm` (GOARCH=loong64). The
// relocation fields of static-symbol references are masked before the
// comparison, since the toolchain leaves them zero for the linker.
func TestGroundTruthLOONG64(t *testing.T) {
for _, path := range []string{
"../testdata/verify/basic_loong64.s",
"../testdata/verify/fp_loong64.s",
} {
t.Run(path, func(t *testing.T) {
src, err := os.ReadFile(path)
if err != nil {
t.Fatalf("read: %v", err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := asm.AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
gt, err := GroundTruthLOONG64(path)
if err != nil {
t.Fatalf("GroundTruthLOONG64: %v", err)
}
matched := 0
for _, fn := range img.Funcs {
gasmCode := maskRelocs(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
goCode, ok := gt[fn.Name]
if !ok {
t.Errorf("%s: not in ground truth (%d functions)", fn.Name, len(gt))
continue
}
goCode = maskRelocs(goCode, fn.Relocs)
if !bytes.Equal(gasmCode, goCode) {
t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\n%s", fn.Name, len(gasmCode), len(goCode), diffHex(gasmCode, goCode))
continue
}
matched++
t.Logf("%s: MATCH (%d bytes)", fn.Name, fn.Size)
}
if matched == 0 {
t.Fatal("no functions matched")
}
})
}
}
// maskRelocs zeroes the 4-byte immediate fields of the relocation sites.
func maskRelocs(code []byte, relocs []asm.Reloc) []byte {
for _, r := range relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
return code
}
func diffHex(a, b []byte) string {
var out bytes.Buffer
n := len(a)
if len(b) > n {
n = len(b)
}
for i := 0; i < n; i += 4 {
ab, bb := "??", "??"
if i+4 <= len(a) {
ab = fmt.Sprintf("%02x%02x%02x%02x", a[i], a[i+1], a[i+2], a[i+3])
} else if i < len(a) {
var sb strings.Builder
for j := i; j < len(a); j++ {
fmt.Fprintf(&sb, "%02x", a[j])
}
ab = sb.String()
}
if i+4 <= len(b) {
bb = fmt.Sprintf("%02x%02x%02x%02x", b[i], b[i+1], b[i+2], b[i+3])
} else if i < len(b) {
var sb strings.Builder
for j := i; j < len(b); j++ {
fmt.Fprintf(&sb, "%02x", b[j])
}
bb = sb.String()
}
mark := " "
if ab != bb {
mark = "!"
}
fmt.Fprintf(&out, "%04x: %s %s %s\n", i, ab, bb, mark)
}
return out.String()
}
+72
View File
@@ -0,0 +1,72 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"os"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGroundTruthRISCV assembles the riscv64 test kernels with gasm and
// compares them byte-for-byte against `go tool asm` (GOARCH=riscv64). The
// relocation fields of static-symbol references are masked before the
// comparison, since the toolchain leaves them zero for the linker.
func TestGroundTruthRISCV(t *testing.T) {
for _, path := range []string{
"../testdata/verify/basic_riscv64.s",
"../testdata/verify/rvc_riscv64.s",
"../testdata/verify/loadstore_riscv64.s",
"../testdata/verify/largeimm_riscv64.s",
"../testdata/verify/movimm_riscv64.s",
"../testdata/verify/branch_riscv64.s",
"../testdata/verify/call_riscv64.s",
} {
t.Run(path, func(t *testing.T) {
testGroundTruthRISCVFile(t, path)
})
}
}
func testGroundTruthRISCVFile(t *testing.T, path string) {
src, err := os.ReadFile(path)
if err != nil {
t.Fatalf("read: %v", err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := asm.AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
gt, err := GroundTruthRISCV(path)
if err != nil {
t.Fatalf("GroundTruthRISCV: %v", err)
}
matched := 0
for _, fn := range img.Funcs {
gasmCode := maskRelocs(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
goCode, ok := gt[fn.Name]
if !ok {
t.Errorf("%s: not in ground truth (%d functions)", fn.Name, len(gt))
continue
}
goCode = maskRelocs(goCode, fn.Relocs)
if !bytes.Equal(gasmCode, goCode) {
t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\n%s", fn.Name, len(gasmCode), len(goCode), diffHex(gasmCode, goCode))
continue
}
matched++
t.Logf("%s: MATCH (%d bytes)", fn.Name, fn.Size)
}
if matched == 0 {
t.Fatal("no functions matched")
}
}