Compare commits

...
24 Commits
Author SHA1 Message Date
petrbalvin 6f4f2096e9 chore: prepare release v0.31.0
Release / build (amd64, linux) (push) Successful in 42s
Release / build (arm64, linux) (push) Successful in 40s
Release / build (loong64, linux) (push) Successful in 42s
Release / build (riscv64, linux) (push) Successful in 45s
Release / release (push) Successful in 18s
2026-08-20 16:24:33 +02:00
petrbalvin 56630f8624 chore(toolchain): upgrade to Go 1.27
Test / vet (push) Successful in 1m5s
Test / test (push) Successful in 2m33s
Test / build (push) Successful in 40s
2026-08-20 16:03:26 +02:00
petrbalvin 459f4a2b6e fix(test): add arm64 encoding tests for Go 1.26 coverage compatibility
Assisted-by: MiMo V2.5 Pro
2026-08-20 15:44:23 +02:00
petrbalvin 5d66343488 fix(goobj): make R_DWTXTADDR_U4 relocation type Go-version-aware
The relocation type number shifted between Go 1.26 (103) and Go 1.27
(106)
because new LoongArch relocations were inserted. Detect the Go version
at
runtime and use the correct value.
2026-08-20 15:35:39 +02:00
petrbalvin 48334c4d5a docs: remove completed roadmap phases, fix licence description 2026-08-20 15:01:57 +02:00
petrbalvin 7629963cab chore: fix project conventions — .gitignore, CHANGELOG categories, docs naming
Assisted-by: MiMo V2.5 Pro
2026-08-20 14:47:39 +02:00
petrbalvin 97951cbeb6 feat(asm): extend arm64 encoder with atomics, bitfield, SIMD and more test kernels
Assisted-by: MiMo V2.5 Pro
2026-08-20 14:31:15 +02:00
petrbalvin 6e73f59e78 feat(asm): extend arm64 encoder with FP, conditional select, CRC32 and tests
Assisted-by: MiMo V2.5 Pro
2026-08-20 14:07:12 +02:00
petrbalvin 4221ec5741 feat(asm): add AArch64 arm64 encoder with ground-truth verification
Assisted-by: MiMo V2.5 Pro
2026-08-20 13:33:39 +02:00
petrbalvin 01dcc3b86e revert(toolchain): restore go1.26 in CI
Test / vet (push) Successful in 43s
Test / test (push) Failing after 1m55s
Test / build (push) Skipped
Release / build (amd64, linux) (push) Successful in 40s
Release / build (arm64, linux) (push) Successful in 37s
Release / build (loong64, linux) (push) Successful in 38s
Release / build (riscv64, linux) (push) Successful in 58s
Release / release (push) Successful in 22s
2026-08-13 18:34:20 +02:00
petrbalvin b08885bd31 chore(release): prepare v0.30.0
Test / vet (push) Failing after 17s
Test / test (push) Skipped
Test / build (push) Skipped
Assisted-by: DeepSeek V4 Pro
2026-08-13 18:27:12 +02:00
petrbalvin c05c53452f fix(asm): encode RISC-V CALL sym(SB) as JAL
Test / vet (push) Successful in 47s
Test / test (push) Failing after 2m9s
Test / build (push) Skipped
Assisted-by: DeepSeek V4 Pro
2026-08-13 18:12:22 +02:00
petrbalvin 681a449c01 fix(asm): match RISC-V branch and jump encodings 2026-08-13 17:57:10 +02:00
petrbalvin 31a2cee382 fix(asm): materialise RISC-V MOV immediates
Test / vet (push) Successful in 44s
Test / test (push) Failing after 1m56s
Test / build (push) Skipped
2026-08-13 17:41:16 +02:00
petrbalvin 3bc7c18bc3 fix(asm): materialise large RISC-V immediates
Assisted-by: DeepSeek V4 Pro
2026-08-13 16:04:08 +02:00
petrbalvin f0512a4e1c fix(asm): complete RISC-V compressed loads/stores and word arithmetic
Assisted-by: DeepSeek V4 Pro
2026-08-13 15:42:38 +02:00
petrbalvin 373c09f725 fix(asm): correct RISC-V operand order and complete RVC compression
Assisted-by: DeepSeek V4 Pro
2026-08-13 15:13:31 +02:00
petrbalvin b78b6c5004 fix(asm): correct RISC-V frame layout and RVC encodings
Assisted-by: GLM 5.2
2026-08-13 14:41:57 +02:00
petrbalvin 9b5c878f9e feat(asm): emit RISC-V GOOBJ with the shared emitter
Assisted-by: DeepSeek V4 Pro
2026-08-13 12:07:29 +02:00
petrbalvin 31ee8e7941 feat(asm): add LoongArch encoder with ELF and GOOBJ emission
Assisted-by: DeepSeek V4 Pro
2026-08-13 11:24:44 +02:00
petrbalvin 2d1176e045 feat(asm): resolve external GOOBJ symbols from archive data
Test / vet (push) Successful in 1m2s
Test / test (push) Failing after 2m5s
Test / build (push) Skipped
2026-08-08 16:18:56 +02:00
petrbalvin 2a27a3a52b docs: add BSD-3-Clause headers to generated files and update CI docs 2026-08-07 22:43:40 +02:00
petrbalvin cde7d0f96a docs: document watchpoint slot tracking and update debugger commands 2026-08-07 22:27:49 +02:00
petrbalvin d7ee1b78d4 feat: drop Mach-O and macOS support, Linux-only 2026-08-07 22:20:26 +02:00
79 changed files with 11161 additions and 1788 deletions
+1 -1
View File
@@ -25,7 +25,7 @@ jobs:
- uses: actions/setup-go@v6 - uses: actions/setup-go@v6
with: with:
go-version: "1.26" go-version: "1.27"
- name: Download dependencies - name: Download dependencies
run: go mod download run: go mod download
+3 -3
View File
@@ -15,7 +15,7 @@ jobs:
- uses: actions/setup-go@v6 - uses: actions/setup-go@v6
with: with:
go-version: "1.26" go-version: "1.27"
- name: Download dependencies - name: Download dependencies
run: go mod download run: go mod download
@@ -41,7 +41,7 @@ jobs:
- uses: actions/setup-go@v6 - uses: actions/setup-go@v6
with: with:
go-version: "1.26" go-version: "1.27"
- name: Download dependencies - name: Download dependencies
run: go mod download run: go mod download
@@ -84,7 +84,7 @@ jobs:
- uses: actions/setup-go@v6 - uses: actions/setup-go@v6
with: with:
go-version: "1.26" go-version: "1.27"
- name: Download dependencies - name: Download dependencies
run: go mod download run: go mod download
+3 -4
View File
@@ -7,9 +7,8 @@
coverage.out coverage.out
*.test *.test
# Editor detritus
*.swp
.DS_Store
# Scratch / temporary work # Scratch / temporary work
_scratch/ _scratch/
# ZCode workspace
.zcode
+1 -1
View File
@@ -55,7 +55,7 @@ No body, no footers, no trailing period on the subject.
## Code Style ## Code Style
Language: Go 1.26 (`toolchain go1.26.5`). Language: Go 1.27 (`toolchain go1.27.0`).
### Formatter ### Formatter
+141 -55
View File
@@ -9,6 +9,143 @@ and this project adheres to [Conventional Commits](https://www.conventionalcommi
Unreleased changes on the `development` branch. Unreleased changes on the `development` branch.
### Added
-
## [0.31.0] — 2026-08-20
The arm64 encoder (Phase 5 — complete) ships with ELF64 and GOOBJ emission,
verified byte-for-byte against `GOARCH=arm64 go tool asm` and linked into a
real `go build`. The encoder covers the full integer instruction set, FP
arithmetic, conditional select, CRC32, and the MOV pseudo-instruction with
bitmask immediate encoding. The project now requires Go 1.27.
### Added
- **arm64 encoder (Phase 5 — complete).** `gasm asm` can now assemble `_arm64.s`
files: the AArch64 integer instruction set with the MOV pseudo-instruction and
its immediate-constant expansions (MOVZ/MOVN/MOVK for wide immediates, ORR with
logical bitmask encoding for values like `$1`), data-processing (shifted
register and immediate forms), load/store (scaled unsigned and unscaled9-bit
immediate), conditional and unconditional branches, FP/SP frame mapping,
SB/global symbol references (ADRP+ADD pairs with `R_ADDRARM64` relocations),
jump chain folding, and ELF64 emission (`gasm asm --format elf`). Ground-truth
verification against `GOARCH=arm64 go tool asm` matches byte-for-byte. Phase 5
(the other architectures — RISC-V, LoongArch, arm64) is now complete.
### Changed
- **Go 1.27 required.** The project now requires Go 1.27 (`toolchain go1.27.0`).
The `R_DWTXTADDR_U4` relocation type is detected at runtime for backward
compatibility.
## [0.30.0] — 2026-08-13
The LoongArch encoder (Phase 5) ships with ELF64 and GOOBJ emission, verified
byte-for-byte against `GOARCH=loong64 go tool asm` and linked into a real
`go build`; the shared GOOBJ emitter now writes the per-function DWARF symbols
the linker's DWARF pass reads. The RISC-V encoder reaches byte-for-byte parity
with `go tool asm`: the frame model, operand ordering, RVC compression,
large-immediate and `MOV $imm` materialisation, branch/jump encodings, and
`CALL sym(SB)` (now a `JAL` with an `R_RISCV_JAL` relocation). The debugger
tracks four hardware watchpoint slots, and the toolkit is Linux-only.
### Added
- **LoongArch encoder (Phase 5).** `gasm asm` can now assemble `_loong64.s`
files: the full LoongArch64 instruction set with the dual-form arithmetic
mnemonics, the 16/21-bit branch families, the MOV pseudo-instruction and
its immediate-constant expansions, FP/SP frame mapping, SB/global symbol
references (pcalau12i pairs) and ELF64 emission
(`gasm asm --format elf`). Ground-truth verification against
`GOARCH=loong64 go tool asm` matches byte-for-byte; GOOBJ emission
(`gasm asm --format goobj`) is proven end-to-end by linking the object
into a cross-compiled `go build`.
- **GOOBJ DWARF symbols.** The GOOBJ emitters now write the per-function
DWARF symbols the linker requires (the subprogram DIE and the `.debug_line`
program, byte-identical to `cmd/asm`'s), and the pc-value table deltas are
in the architecture's MinLC units as the runtime expects — the amd64 link
test now genuinely substitutes the gasm object, and the amd64/loong64
end-to-end GOOBJ link tests pass.
- **RISC-V GOOBJ emission via the shared emitter.** RISC-V GOOBJ output is
now written by the same shared emitter as amd64 and LoongArch, modelling
each AUIPC + second-instruction pair as a single R_RISCV_PCREL_ITYPE/STYPE
relocation (the layout `cmd/asm` writes, not the ELF HI20/LO12 pair), so the
object links into a cross-compiled `go build` for `GOARCH=riscv64`. An
end-to-end link test substitutes the gasm object and reads the symbol back
with `go tool nm`; the rewrite also corrects the relocation `after` field.
### Fixed
- **RISC-V frame model and RVC encodings.** The riscv64 frame layout now
matches `go tool asm`: the prologue/epilogue save and restore the link
register (LR) instead of S0, with the correct autosize (locals + 8) and the
RVC-compressed prologue/epilogue instructions; `RET` emits the uncompressed
`JALR X0, 0(X1)` the toolchain writes; the `C.ADDI`/`C.LI`/`C.LUI`/`C.ADDIW`
opcode bit and the `C.ADD` CR-type encoding are fixed; and the `LR`/`TMP`
register aliases now resolve to X1 and X31. The pcsp/pcfile/pcline tables
are populated from the recorded stack-adjustment and source-line data, and
a byte-exact ground-truth test compares framed and leaf functions against
`GOARCH=riscv64 go tool asm`.
- **RISC-V operand ordering and RVC compression.** R-type instructions now
take `rs2, rs1, rd` and I-type arithmetic instructions take `imm12, rs1,
rd`, matching the Go assembler's documented operand order (previously both
were reversed, so non-commutative R-type instructions such as `SUB` encoded
the wrong operation). The two-operand ternary forms (`ADD rs2, rd`,
`ADDI $imm, rd`, `SLLI $shamt, rd`) are now accepted. RVC compression is
completed for `C.ADDI16SP`, `C.SLLI`, `C.SRLI`, `C.SRAI`, `C.ANDI`,
`C.NOP`, `C.EBREAK`, `C.MV` (from `ADDI`/`ADD`) and the commutative
`AND`/`OR`/`XOR` forms; the byte-exact ground-truth test now covers these.
- **RISC-V compressed loads/stores and word arithmetic.** RVC compression
now also covers the register-relative `C.LW`/`C.SW`/`C.LD`/`C.SD`/
`C.FLD`/`C.FSD` forms (in addition to the stack-relative `C.LWSP`/`C.SWSP`/
`C.LDSP`/`C.SDSP`), plus `C.ADDI4SPN`, `C.ADDW` and `C.SUBW`. The
byte-exact ground-truth test exercises these against `GOARCH=riscv64
go tool asm`.
- **RISC-V large-immediate materialisation.** `ADDI`/`ANDI`/`ORI`/`XORI`
with a 32-bit immediate that does not fit 12 bits now expand exactly as
`cmd/asm`: two `ADDI`s for the small `ADDI` split range, and
`LUI`+`ADDIW`+`<op>` otherwise, with the `LUI` and `ADDIW` compressed to
`C.LUI`/`C.ADDIW` when their immediate fits six signed bits. The
byte-exact ground-truth test covers positive, negative, and out-of-range
immediates against `GOARCH=riscv64 go tool asm`.
- **RISC-V `MOV $imm, rd` materialisation.** The immediate-loading
pseudo-instruction now uses the toolchain's `Split32BitImmediate` split
(previously it rounded the upper 20 bits, producing wrong results for
negative and bit-11-set immediates) and compresses the emitted
`ADDI`/`LUI`/`ADDIW` to `C.LI`/`C.LUI`/`C.ADDIW` when their immediate
fits six signed bits. A byte-exact ground-truth test covers zero, small,
negative, and 32-bit immediates against `GOARCH=riscv64 go tool asm`.
- **RISC-V branch/jump compression.** `JMP`/`JAL` were being compressed to
`C.J` and `BEQ`/`BNE` (with `X0`) to `C.BEQZ`/`C.BNEZ`, but `go tool asm`
never emits these compressed forms. They now emit the 32-bit `JAL` and
branch encodings the toolchain writes; the dead `C.J`/`C.BEQZ`/`C.BNEZ`
encoders were removed, and the `C.LUI` direct-instruction compression now
uses the correct six-bit signed range. A byte-exact ground-truth test
covers the branch family and jumps against `GOARCH=riscv64 go tool asm`.
- **RISC-V `CALL sym(SB)`.** The call pseudo-instruction now emits the
toolchain's `JAL X1, sym(SB)` with a single `R_RISCV_JAL` relocation
(previously it emitted an `AUIPC`+`JALR` pair against a local branch
label, a form `go tool asm` rejects). The GOOBJ and ELF emitters now map
that relocation (Go objabi 59 / ELF `R_RISCV_JAL` 17, a 4-byte field), and
relocation offsets are recorded relative to the function start (including
the prologue). A byte-exact ground-truth test covers a call against
`GOARCH=riscv64 go tool asm`.
- **Debugger watchpoint slots.** `gasm debug`'s `watch` command always used
hardware watchpoint slot 0, so a second `watch` call silently overwrote
the first. Watchpoint slots are now tracked in the `Session` (DR0–DR3);
`watch` picks the first free slot and reports an error if all four are in
use, and `unwatch <slot>` clears one (no argument clears all).
### Changed
- **Linux only.** The toolkit, its CI and the released binaries are now
Linux-only; cross-compiled to linux/{amd64,arm64,riscv64,loong64}.
- **Phase 4 closed.** README's "Remaining" list for the debugger is gone;
disassembly at PC, memory-write, watchpoints, and source-line mapping are
all shipped.
## [0.29.0] — 2026-08-07 ## [0.29.0] — 2026-08-07
RISC-V GOOBJ emission, YMM vector register display, named buffer allocation RISC-V GOOBJ emission, YMM vector register display, named buffer allocation
@@ -60,20 +197,13 @@ new CLI commands. A signature-parser fix corrects grouped Go parameters.
prevent GC from collecting heap objects whose addresses were passed to JIT prevent GC from collecting heap objects whose addresses were passed to JIT
code via `unsafe.Pointer`; all verify tests pass 100/100 under `-race`. code via `unsafe.Pointer`; all verify tests pass 100/100 under `-race`.
### Cleaned up ### Changed
- **Removed external kernel test dependencies** — the verify test suite no - **Removed external kernel test dependencies** — the verify test suite no
longer references production kernels from the separate go-libraries project. longer references production kernels from the separate go-libraries project.
The remaining test suite uses only `testdata/verify/*.s` kernels, which are The remaining test suite uses only `testdata/verify/*.s` kernels, which are
part of this repository. Coverage is identical locally and in CI (80.3 %). part of this repository. Coverage is identical locally and in CI (80.3 %).
### Verified
- `gasm diff` detects byte-level differences; `--map` pairs differently-named
functions for comparison.
- `gasm verify --call` invokes functions with user-supplied buffers; the arg
block is printed before and after the call, showing return values.
- LSP go-to-definition resolves labels across functions and files.
## [0.28.0] — 2026-08-03 ## [0.28.0] — 2026-08-03
@@ -105,10 +235,6 @@ ground-truth verification against `GOARCH=riscv64 go tool asm`.
- RVC: C.LDSP/C.SDSP/FLDSP/FSDSP immediate encoding now matches Go toolchain - RVC: C.LDSP/C.SDSP/FLDSP/FSDSP immediate encoding now matches Go toolchain
(bit-interleaved format). (bit-interleaved format).
### Verified
- 118 RISC-V tests, asm coverage 83.3%.
- Ground-truth: C.LDSP, C.SDSP, C.FLDSP, C.FSDSP byte-exact vs Go toolchain.
## [0.27.0] — 2026-08-01 ## [0.27.0] — 2026-08-01
@@ -141,7 +267,7 @@ area bit-for-bit.
analyze, autocorr) pass; partial functions (decoders that fault on malformed analyze, autocorr) pass; partial functions (decoders that fault on malformed
input) should use `--ground-truth` instead. input) should use `--ground-truth` instead.
### Known limitation ### Fixed
`--fuzz` crashes the process for partial functions (e.g. LZ4 decoders) whose `--fuzz` crashes the process for partial functions (e.g. LZ4 decoders) whose
over-copy paths read past the buffer on random garbage input. Subprocess over-copy paths read past the buffer on random garbage input. Subprocess
@@ -204,11 +330,6 @@ The remaining go-flac encoder kernels join the differential suite.
frames: the four zigzag-fold entropy sums compared against the scalar frames: the four zigzag-fold entropy sums compared against the scalar
loop). loop).
### Verified
- `gasm fmt` doc-comment indentation confirmed correct: comments before
every TEXT are at column 0 (the RET-detection logic handles multi-exit
functions).
## [0.21.0] — 2026-07-26 ## [0.21.0] — 2026-07-26
@@ -246,7 +367,7 @@ test corpus exercises.
argument blocks and collect distinct output fingerprints (the result argument blocks and collect distinct output fingerprints (the result
words); reports path diversity as a lower bound on code coverage. words); reports path diversity as a lower bound on code coverage.
### Note ### Changed
INT3-based per-block hit counting was prototyped but deferred: Go's runtime INT3-based per-block hit counting was prototyped but deferred: Go's runtime
signal management (sigaltstack, handler re-installation) makes raw signal management (sigaltstack, handler re-installation) makes raw
@@ -311,11 +432,6 @@ toolchain.
known-answer LZ4 blocks decode bit-for-bit, wide copies of 0–1024 bytes known-answer LZ4 blocks decode bit-for-bit, wide copies of 0–1024 bytes
match, malformed input returns the correct error codes. match, malformed input returns the correct error codes.
### Verified
- `just test` (race, 84.6 % total coverage, verify 82.2 %).
- `gasm verify` on both go-lz4 kernels: all functions JIT-load and
smoke-test clean.
## [0.16.0] — 2026-07-21 ## [0.16.0] — 2026-07-21
@@ -441,13 +557,6 @@ Go assembler.
(`DATA mask<>+8(SB)/8, $0x800f…`) parse as unsigned and keep their bit (`DATA mask<>+8(SB)/8, $0x800f…`) parse as unsigned and keep their bit
pattern, instead of being rejected as non-integer. pattern, instead of being rejected as non-integer.
### Verified
- End-to-end: a gasm-emitted GOOBJ swapped into a `go build` in place of
the toolchain's assembly object links and runs with output identical to
the baseline binary (stack-argument calls and a `GLOBL` relocation
resolved by the Go linker). All 17 go-flac AVX2 kernel functions emit as
a GOOBJ that `go tool nm` reads back with every symbol intact.
## [0.11.0] — 2026-07-16 ## [0.11.0] — 2026-07-16
@@ -506,7 +615,7 @@ verified byte for byte against the Go assembler.
arithmetic, the unpacks, VMOVDDUP and the conversions all accept the arithmetic, the unpacks, VMOVDDUP and the conversions all accept the
explicit K1–K7 operand and the `.Z` suffix the way Go writes them. explicit K1–K7 operand and the `.Z` suffix the way Go writes them.
### Documented ### Changed
- VCVTPS2PD follows the Go assembler's encoding, which omits the F3 - VCVTPS2PD follows the Go assembler's encoding, which omits the F3
mandatory prefix (VEX.pp / EVEX.pp = 00) that Intel's maps prescribe; the mandatory prefix (VEX.pp / EVEX.pp = 00) that Intel's maps prescribe; the
@@ -514,15 +623,6 @@ verified byte for byte against the Go assembler.
gasm reproduces it exactly (and round-trips through the x86 decoder, which gasm reproduces it exactly (and round-trips through the x86 decoder, which
shares the convention). shares the convention).
### Verified
- 58 new ground-truth cases — every instruction extracted from the Go
toolchain's own assembly (go build + an executable-segment dump), checked
byte for byte and round-tripped through the decoder, covering disp8×N for
the scalar (×8/×4), duplication (×8/×32/×64) and conversion (×8/×16/×32)
memory operands, the 5-bit register fields and the masked/zeroing P2
byte. All four go-flac/go-lz4 kernels still assemble byte-identically
and lint clean.
## [0.9.0] — 2026-07-14 ## [0.9.0] — 2026-07-14
@@ -658,13 +758,6 @@ the Go toolchain, completing the production-kernel coverage.
- `asm`: the VEX encoder now rejects vector register indices 16–31 instead of - `asm`: the VEX encoder now rejects vector register indices 16–31 instead of
encoding a truncated (wrong) register. encoding a truncated (wrong) register.
### Verified
- All 10 functions of the go-flac `avx512_amd64.s` kernel assemble
byte-identically to the Go toolchain's machine code (the disp32 of the one
`VMOVDQU32 idx16(SB), Z13` load is linker-filled in Go and resolved within
gasm's own image — checked to reach the right constant bytes). The AVX2
kernel's 17 functions remain byte-identical.
## [0.4.0] — 2026-07-09 ## [0.4.0] — 2026-07-09
@@ -685,13 +778,6 @@ machine code byte for byte.
- `gasm asm` prints the data section and symbol map alongside the functions - `gasm asm` prints the data section and symbol map alongside the functions
and writes the whole image (code + data) with `-o`. and writes the whole image (code + data) with `-o`.
### Verified
- All 17 functions of the go-flac `avx2_amd64.s` kernel assemble
byte-identically to the Go toolchain's machine code; the only differing
bytes are the displacements of the two `VMOVDQU mask24<>(SB), X15` loads,
which the Go linker fills at link time and gasm resolves within its own
image (checked to reach the right constant bytes).
## [0.3.0] — 2026-07-08 ## [0.3.0] — 2026-07-08
+12 -5
View File
@@ -2,9 +2,9 @@
## Prerequisites ## Prerequisites
- Go 1.26 or later (`toolchain go1.26.5`) - Go 1.27 or later (`toolchain go1.27.0`)
- `just` command runner - `just` command runner
- A Linux, FreeBSD, or macOS host on amd64 or arm64 - A Linux host on amd64, arm64, riscv64 or loong64
## Development Setup ## Development Setup
@@ -68,8 +68,15 @@ See [AGENTS.md](AGENTS.md) for the full style guide. Key points:
## CI ## CI
There is no CI pipeline in this repository. The Definition of Done CI runs on every push to `development` and on pull requests:
(`just build` + `just test` + `just fmt`) is enforced locally.
- **Test** (`test.yml`) — `gofmt` check, `go vet`, `go test -race` and the
80 % coverage gate.
- **Release** (`release.yml`) — cross-compiles release binaries for
linux/{amd64,arm64,riscv64,loong64} on version tags and publishes them.
The Definition of Done (`just build` + `just test` + `just fmt`) must still
pass locally before pushing.
## AI-Assisted Contributions ## AI-Assisted Contributions
@@ -85,7 +92,7 @@ Attribute agent authorship in issues and pull requests on one trailing
line: line:
``` ```
_Assisted-by: DeepSeek V4 Pro_ _Assisted-by: Qwen 3.8 Max_
``` ```
## Questions ## Questions
+21 -239
View File
@@ -2,8 +2,6 @@
Developer tooling for **GAsm** — Go's built-in Plan 9 assembler. Developer tooling for **GAsm** — Go's built-in Plan 9 assembler.
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
Go ships an assembler but no tooling for it. There is no syntax highlighting, Go ships an assembler but no tooling for it. There is no syntax highlighting,
no autocomplete, no linter, no static analyser, no formatter, no standalone no autocomplete, no linter, no static analyser, no formatter, no standalone
assembler and no debugger for `.s` files. Developers write assembly blind, assembler and no debugger for `.s` files. Developers write assembly blind,
@@ -18,23 +16,13 @@ gasm parse parse and report syntax errors
gasm fmt canonicalise formatting (gofmt for assembly) gasm fmt canonicalise formatting (gofmt for assembly)
gasm lint static checks gasm lint static checks
gasm lsp language server (completion, hover, symbols, diagnostics, highlighting) gasm lsp language server (completion, hover, symbols, diagnostics, highlighting)
gasm asm standalone assembler (Phase 2) gasm asm standalone assembler
gasm verify dynamic analysis & verification (Phase 3) gasm verify dynamic analysis & verification
gasm debug source-level debugger (Phase 4) gasm debug source-level debugger
gasm diff compare machine code of two .s files gasm diff compare machine code of two .s files
gasm profile show basic-block structure of functions gasm profile show basic-block structure of functions
``` ```
> **Status: Phase 4 — done, Phase 5 underway.** Phase 1 (the language
> foundation, linter, formatter and language server) shipped in v0.1.0;
> Phase 2 (the standalone assembler — the full amd64 instruction set plus
> ELF, Mach-O and GOOBJ object emission) in v0.12.0; Phase 3 (dynamic
> analysis — JIT execution, differential testing, ABI checks and coverage
> profiling) in v0.25.0; Phase 4 (interactive debugger — ptrace-based,
> breakpoints, watchpoints, stepping, vector register display, named buffer
> allocation) in v0.27.0; RISC-V encoder (RV64IMAFDC + RVC, ELF emission,
> ground-truth, GOOBJ) in v0.28.0–v0.29.0. See [Roadmap](#roadmap).
## Architecture support ## Architecture support
gasm-devkit targets every architecture Go's assembler speaks. The instruction gasm-devkit targets every architecture Go's assembler speaks. The instruction
@@ -56,216 +44,13 @@ carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
command — `just gen` — and requires only a Go installation; the committed command — `just gen` — and requires only a Go installation; the committed
output has no runtime dependency on the toolchain. output has no runtime dependency on the toolchain.
The target *architectures* above are what the toolkit analyses. The toolkit ## Supported Platforms
itself is portable Go and builds on Linux, FreeBSD and macOS, on amd64 and
arm64 hosts.
## Roadmap The toolkit runs on Linux. All four Linux architectures are supported as
hosts — amd64, arm64, riscv64 and loong64 — and the release matrix
cross-compiles the same four targets.
The work is delivered in four phases. Each phase is completed and hardened **FreeBSD support is planned for a future release.**
before the next begins. The ordering follows a dependency chain: understand
the code statically (Phase 1), make it runnable (Phase 2), then run it and
observe or control it (Phases 3–4).
### Phase 1 — language foundation, editor tooling and static analysis · *done*
Everything needed to read, understand, check, format and highlight GAsm —
without executing it.
| Capability | Status |
|------------|--------|
| Lexer — permissive, position-aware scanner for all four architectures | done |
| Parser — line-oriented, error-tolerant, full AST with source positions | done |
| Instruction + register tables for amd64, arm64, riscv64, loong64 (generated, complete) | done |
| Linter — `unknown-instruction`, `operand-count`, `undefined-label`, `duplicate-label`, `missing-ret`, `missing-textflag-include`, `abi-argsize`, `unreachable-code`, `register-clobber`, `funcdata-pcdata` | done |
| Formatter — idempotent, comment-preserving, per-function alignment; a `RET` terminates the body for indentation, so the next function's doc comment stays at column 0; exactly one blank line before every block (label, `TEXT`, `GLOBL`) and runs of blanks collapsed; directory / no-argument mode reformats every `.s` in place, `go fmt`-style | done |
| Language server — completion, hover, document symbols, diagnostics, semantic-token highlighting | done |
| CLI — `gasm tokens / parse / fmt / lint / lsp` | done |
| Real-world validation against production AVX2 / AVX-512 kernels | done |
| Lint hardening — zero false positives across the Go runtime corpus (90 files, all four architectures): macro-invocation handling, branch aliases (`B`/`BL`/`JAL`), addressing suffixes (`.P`/`.W`), terminal `UNDEF` | done |
| Static analysis — `abi-argsize` (argument/result area computed from the `// func` signature under Go's ABI0 layout and checked against the TEXT declaration) and `unreachable-code` (dead code after `RET`, suppressed where reachability is undecidable: PC-relative jumps, register-indirect branches, `#ifdef`) | done |
| Static analysis — register liveness (CFG construction + per-instruction def/use + iterative backward dataflow) driving `register-clobber`, calibrated to the **Go ABI** (not System V): flags writes to the registers Go fixes across calls — the frame pointer and the goroutine pointer (`R14` on amd64, `R28`/`R29` on arm64, `X27` on riscv64, `R22` on loong64, plus the OS-reserved `R18` on arm64) — that are never saved/restored; the goroutine pointer is reported only when the function can reach the runtime (not `NOSPLIT`, or makes calls), matching how the runtime's own assembly uses it. `funcdata-pcdata` structural validation of `FUNCDATA`/`PCDATA` operands and indices | done |
> **Limitation — macros.** gasm-devkit reads `.s` source as written; it does
> **not** run the C preprocessor, so `#define` macros are not expanded. Files
> that use macros (the runtime's `asm_*.s`, `race_*.s`, `sys_*.s`, …) parse
> cleanly, and macro *invocations* are recognised and never flagged, but the
> `undefined-label` and `missing-ret` heuristics are suppressed in macro-using
> files because labels a macro defines are invisible without expansion. Full
> macro expansion is future work (it pairs naturally with the Phase 2
> assembler). Hand-written, macro-free kernels — such as everything in
> `go-libraries` — are analysed in full.
### Phase 2 — standalone assembler · *done*
Assembly without the Go toolchain in the loop.
- **`gasm asm`:** a standalone assembler that turns a `.s` file into machine
code directly — pure Go, no `go build`, no external toolchain. Useful for
fast iteration, for environments without a full Go installation, and as the
execution substrate that Phases 3 and 4 build on.
Done so far:
- An amd64 (x86-64) **instruction encoder** — REX/ModR-M/SIB/displacement/
immediate machinery and the scalar instruction set (MOV, the ALU group, TEST,
LEA, INC/DEC/NEG/NOT, shifts, IMUL and IMUL3, PUSH/POP, JMP/CALL/Jcc,
CMOVcc, SETcc, LZCNT/TZCNT, the sign/zero-extending moves — MOVBLZX and
friends, MOVLQSX — and CVTSL2SD/CVTSQ2SD), validated by round-tripping
every encoding through `golang.org/x/arch`'s decoder and byte-for-byte
against the Go assembler.
- An **assembler** that drives the parser's AST into the encoder with local-
label resolution — jumps start in the short (rel8) form and expand to rel32
when the displacement does not fit, and jump-to-jump chains are folded the
way the Go toolchain folds them — so `gasm asm <file>` emits machine code
for each `TEXT` function.
- **File-level assembly with static data** — `GLOBL`/`DATA` symbols are laid
out in a data section behind the code and references to them (`mask<>(SB)`)
are encoded RIP-relative with the displacement resolved within the image,
so the output is self-consistent and position-independent. References to
symbols no `GLOBL` in the file defines are recorded as relocations and
carried into the object-file output.
- **GOOBJ emission** — `gasm asm --format goobj -p <pkgpath>` writes the Go
toolchain's own object format (the one `cmd/link` consumes directly), so
gasm-assembled kernels drop into a `go build` without the Go assembler:
the functions as non-package symbols, `GLOBL` data, one `FuncInfo` per
function and the pc-value tables (`pcsp` with the real prologue/epilogue
stack deltas, `pcfile`, `pcline`, `pcinline`). Verified end-to-end by
swapping a gasm-emitted object into a `go build` in place of the
toolchain's, linking and running — bit-identical behaviour.
- **Object-file emission** — `gasm asm --format elf` / `--format macho`
writes a relocatable object (a `.text` and a `.data` section, a symbol
table — file-local `<>` symbols local, the rest global — and one
`R_X86_64_PC32` / `X86_64_RELOC_SIGNED` relocation per static-symbol
reference) that links with the system toolchain: external references
resolve against undefined symbols, file-local ones against the data
section. Verified end-to-end by linking a gasm-emitted object with a C
driver and running it.
- **`FP`/`SP` frame mapping** — the pseudo-registers are translated onto the
hardware stack pointer (`x+N(FP)` → `(N+8)(SP)` for a zero frame, `(N+frame+
16)(SP)` with a frame pointer; locals via `x-N(SP)`), and the Go-style
prologue/epilogue is generated for functions with a frame. The output is
**byte-identical to the Go assembler** for these cases (verified against
`go tool objdump`).
- **SIMD (VEX / AVX2)** — the VEX prefix machinery (2-byte C5 and 3-byte C4)
with XMM/YMM vector registers, validated by round-trip decoding **and**
byte-for-byte against the Go assembler's machine code, across eight operand
forms: the three-operand NDS form (VPADDD/Q, VPSUBD/Q, VPXOR, VPOR, VPAND/N,
VPCMPEQD, VPCMPGTQ, VPUNPCK*, VPMULLD, VPMULDQ, VPSHUFB, VPACKSSDW,
VPERMD), the two-operand reg/rm form (VPMOVSXWD/DQ, VPMOVZXDQ,
VPBROADCASTD/Q, VPMOVMSKB, VMOVMSKPS, VCVTDQ2PD), the immediate-shift and
variable-count shifts (VPSLLD/Q, VPSRAD, VPSRLD/Q with an immediate or an
XMM/memory count), the immediate shuffle (VPSHUFD, VPERMQ), the
three-operand-plus-immediate form (VSHUFPD, VPERM2I128, VINSERTI128), the
lane extract (VEXTRACTI128, VEXTRACTF128), the direction-sensitive moves
(VMOVDQU, VMOVUPD, VMOVD, VMOVQ, VMOVSD), the no-operand VZEROUPPER, and
the floating-point set: the packed double arithmetic
(VADDPD/VSUBPD/VMULPD/VDIVPD/VMINPD/VMAXPD), the unpacks
(VUNPCKHPD/VUNPCKLPD), the scalar SD and SS operations, VMOVDDUP, the
width-changing conversions (VCVTDQ2PS, VCVTPS2PD, VCVTDQ2PD and the
VCVTPD2DQX/Y / VCVTTPD2DQX/Y spellings, whose VEX.L follows the wider
source) and VFMADD231PD.
- **SIMD (EVEX / AVX-512)** — the four-byte EVEX prefix with the 5-bit
register fields (Z0–Z31, X/Y 16–31), opmask registers (K0–K7 as operands
and mask destinations, KMOVW, KTESTW) and the compressed disp8×N
displacement, covering every AVX-512 instruction the go-flac kernels use:
VPXORD/Q, VPADDD, VPSUBD/Q, VPUNPCK*DQ, VPMULLD/Q, VPERMD, VPSLLD/VPSRAD/
VPSRAQ, VALIGND, VPCMPEQD (with a K destination), VMOVDQU32, VMOVUPD,
VCVTQQ2PD, VPMOVSXDQ, the narrowing stores VPMOVDW/VPMOVQD, the lane
extracts VEXTRACTI64X4/VEXTRACTF64X4, VFMADD231PD, VADDPD, VMULPD,
VMOVDQU64 and the broadcasts VPBROADCASTD/Q from a GPR or memory, plus the
wider AVX-512 F/BW integer set (VPADDB/W, VPSUBB/W, VPANDD/Q/ND/NQ, VPMULLW,
VPMIN*/VPMAX* for B/W/D/Q elements, signed and unsigned, VPAVGB/W, the variable
shifts VPSLLV*/VPSRLV*/VPSRAV*, VMOVDQU8/16), the common floating-point
and conversion set (the packed double and single arithmetic
VADD/VSUB/VMUL/VDIV/VMIN/VMAX PD and PS, the scalar SD/SS operations —
whose EVEX forms exist for masked and zeroing use — the VUNPCK{L,H}PD
unpacks, VMOVDDUP, VMOVSLDUP/VMOVSHDUP and the VCVT* conversions), and
the wider AVX-512 set: ternary logic (VPTERNLOGD/Q), lane shuffles,
inserts and extracts (VSHUF{F,I}{32,64}X{2,4}, the VINSERT*/VEXTRACT*
{F,I}{32,64}X{2,4,8} family, VPALIGNR), compares with an opmask
destination (VCMPPD/PS/SD/SS), the permutes (VPERMB/W, VPERMI2/T2
D/Q/PD), the wider integer families (VPMADDWD/UBSW, VPMULHUW, VPACK*,
VPABS*, the VPROL*/VPROR* rotates and the word shifts), expand/compress
(VEXPAND*/VCOMPRESS*, VPEXPAND*/VPCOMPRESS*), the broadcasts
(VPBROADCASTB/W, VBROADCASTSS/SD), the opmask instructions (KAND/KOR/
KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST, KMOVQ), the aligned moves
(VMOVAPS/APD, VMOVDQA32/64, VMOVSS) and the remaining extending and
narrowing moves, the floating-point helper and conversion tail
(VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*, VSCALEF*, VRNDSCALE*,
VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with a K destination, and the
VCVT* conversions VCVTQQ2PS, VCVTPD2QQ/UQQ, VCVTPS2QQ, VCVTUDQ2PD/PS,
VCVTPH2PS, VCVTPS2PH), and gather/scatter with VSIB addressing
(VGATHER*/VPGATHER* in both the VEX mask-register spelling and the EVEX
K-mask spelling — where the L'L field follows the VSIB index — plus
VSCATTER*/VPSCATTER*). The EVEX mnemonic suffixes the Go assembler
accepts are honoured: rounding modes (.RN_SAE, .RD_SAE, .RU_SAE,
.RZ_SAE), suppress-all-exceptions (.SAE) and memory broadcast (.BCST,
with the element-sized disp8×N), each combinable with the .Z zeroing
suffix. Masking is supported the way
Go writes it — an explicit K1–K7 operand placed among the operands, and a
`.Z` mnemonic suffix for zeroing.
- **Legacy SSE moves** — `MOVOU`/`MOVO` (the Plan 9 names for MOVDQU/MOVDQA),
`MOVUPS`/`MOVAPS`/`MOVUPD`/`MOVAPD` and the scalar `MOVSD`/`MOVSS`.
- **Both go-flac kernels — all 17 AVX2 and all 10 AVX-512 functions —
assemble byte-identically to the Go toolchain's machine code**; the only
differing bytes are the displacements of the static-constant loads, which
the Go linker fills at link time and gasm resolves within its own image
(verified to reach the right constant bytes).
Remaining for Phase 2:
- External (cross-package) symbol references in the GOOBJ output —
**deferred** with a recorded decision and three options; see
[`docs/DEFERRED.md`](docs/DEFERRED.md). Single-package objects (no
cross-package references) work today, which covers the production
kernels. With that item deferred, the amd64 instruction set — scalar,
VEX/AVX2 and the full EVEX/AVX-512 set including GPR-interchanging
conversions — is complete, and RISC-V encoding (RV64IMAFDC + RVC)
including ELF and GOOBJ emission is complete.
### Phase 3 — dynamic analysis · *done*
Run the code and check what static analysis cannot. The oracle is the
portable Go implementation every kernel is derived from.
- **`gasm verify`:**
- **JIT execution substrate** — *done.* Assemble the kernel, map it into
executable memory (`syscall.Mmap`, W^X) and call it through an ABI0
trampoline; pure Go, no cgo, no external toolchain.
- **Differential testing** — *done.* The JIT-assembled kernel is fuzzed
against a portable Go reference, comparing the result bit-for-bit;
the automated form of the project's bit-identical contract.
- **Runtime ABI checks** — *done.* The ABI-checking trampoline sets
sentinels in BP and R14, verifies they survive the call, and fills a
128-byte red-zone canary below SP.
- **Coverage / basic-block profiling** — *done.* Static block enumeration
from the assembler's label map plus multi-input path-diversity
measurement: how many observationally distinct execution paths a
test corpus exercises.
### Phase 4 — debugger · *done*
- **`gasm debug`:** single-step a GAsm function, inspect registers (including
YMM vector registers), set breakpoints on labels, allocate and fill named
buffers, and hex-dump memory — the interactive counterpart to Phase 3's
execution substrate.
- **MVP** — *done.* ptrace-based debuggee subprocess (PTRACE_TRACEME +
LockOSThread), entry breakpoint (auto-run to function start),
single-step, register inspection (GPR + YMM/XMM via PTRACE_GETFPREGS),
label resolution, breakpoint management via `/proc/pid/mem`, named
buffer allocation with pattern filling (`--buf`), and an interactive REPL.
- **Remaining:** disassembly at PC (x86asm decode), memory-write support,
watchpoints, source-line mapping, and multi-platform support
(FreeBSD/macOS ptrace variants).
### Phase 5 — the other architectures · *in progress*
- **RISC-V encoding — done.** RV64IMAFDC instruction set, RVC compression,
MOV pseudo-instruction, SB/global symbols (AUIPC pairs), ELF64 and GOOBJ
emission, and ground-truth verification against `go tool asm`.
- **Remaining:** arm64 and loong64 encoding, plus the same encode-and-verify
treatment for each (instruction tables already generated from the toolchain).
## Principles ## Principles
@@ -277,8 +62,9 @@ portable Go implementation every kernel is derived from.
dependency, `golang.org/x/arch`, is used **only in tests** to validate the dependency, `golang.org/x/arch`, is used **only in tests** to validate the
instruction encoder by round-trip decoding — it is never linked into the instruction encoder by round-trip decoding — it is never linked into the
`gasm` binary. `gasm` binary.
- **Portable.** Builds and runs on Linux, FreeBSD and macOS; amd64 and arm64 - **Linux-only.** Runs natively on amd64, arm64, riscv64 and loong64 Linux
hosts. Latest stable Go only. hosts; the release matrix cross-compiles the same four targets. Latest
stable Go only.
- **No vendor lock-in.** The integration surface is the Language Server - **No vendor lock-in.** The integration surface is the Language Server
Protocol and a command-line interface — both open standards. No cloud Protocol and a command-line interface — both open standards. No cloud
service, no proprietary API, no dependence on any one editor's internals. service, no proprietary API, no dependence on any one editor's internals.
@@ -297,17 +83,16 @@ portable Go implementation every kernel is derived from.
| `arch` | amd64, arm64, riscv64 and loong64 register files and instruction tables. | | `arch` | amd64, arm64, riscv64 and loong64 register files and instruction tables. |
| `lint` | Conservative static checks. | | `lint` | Conservative static checks. |
| `format` | A canonical formatter — `gofmt` for assembly. | | `format` | A canonical formatter — `gofmt` for assembly. |
| `asm` | The standalone assembler: amd64 and RISC-V encoders, linker, object-file emitters (ELF, Mach-O, GOOBJ). | | `asm` | The standalone assembler: amd64, RISC-V and LoongArch encoders, linker, object-file emitters (ELF, GOOBJ). |
| `verify` | JIT execution substrate for dynamic analysis, combined ABI+fuzz differential testing (Phase 3). | | `verify` | JIT execution substrate for dynamic analysis, combined ABI+fuzz differential testing. |
| `debug` | Interactive ptrace debugger with GPR/YMM register display and named buffer allocation (Phase 4). | | `debug` | Interactive ptrace debugger with GPR/YMM register display and named buffer allocation. |
| `lsp` | Language Server Protocol server. | | `lsp` | Language Server Protocol server. |
| `cmd/gasm` | The `gasm` binary tying it all together. | | `cmd/gasm` | The `gasm` binary tying it all together. |
| `_gen` | The generator that rebuilds the instruction tables from the Go toolchain. | | `_gen` | The generator that rebuilds the instruction tables from the Go toolchain. |
See [`docs/ARCHITECTURE.md`](docs/ARCHITECTURE.md) for the design rationale and See [`docs/ARCHITECTURE.md`](docs/ARCHITECTURE.md) for the design rationale and
data flow, [`docs/ZED.md`](docs/ZED.md) for the editor-integration story, and data flow, and [`docs/DECISIONS.md`](docs/DECISIONS.md) for design decisions
[`docs/DEFERRED.md`](docs/DEFERRED.md) for design decisions deliberately deliberately postponed (with the analysis needed to pick them up again).
postponed (with the analysis needed to pick them up again).
## Quick start ## Quick start
@@ -341,8 +126,8 @@ gasm profile k.s # show basic-block structure
``` ```
See [CONTRIBUTING.md](CONTRIBUTING.md) for the full development workflow, See [CONTRIBUTING.md](CONTRIBUTING.md) for the full development workflow,
[docs/cli.md](docs/cli.md) for the command reference, and [docs/CLI.md](docs/CLI.md) for the command reference, and
[docs/development.md](docs/development.md) for setup and recipes. [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for setup and recipes.
## Editor integration ## Editor integration
@@ -353,10 +138,7 @@ binary and associate it with `.s` files. Syntax highlighting is delivered as
infers the target architecture from the file-name suffix infers the target architecture from the file-name suffix
(`_amd64.s` / `_arm64.s` / `_riscv64.s` / `_loong64.s`). (`_amd64.s` / `_arm64.s` / `_riscv64.s` / `_loong64.s`).
Zed users should read [`docs/ZED.md`](docs/ZED.md): Zed's native highlighting ## License
engine (Tree-sitter, C/WASM) cannot be fed from pure Go, so the pure-Go path
into Zed is the language server and its semantic tokens.
## Licence BSD-3-Clause — see [LICENSE](LICENSE).
Copyright © 2026 [Petr Balvín](https://petrbalvin.org)
BSD-3-Clause — the same licence as Go itself. See [`LICENSE`](LICENSE).
+8 -2
View File
@@ -86,7 +86,10 @@ func filterCommon(names []string) []string {
func writeCommon(names []string) error { func writeCommon(names []string) error {
var b strings.Builder var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n") b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n\n") b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
b.WriteString("// SPDX-License-Identifier: BSD-3-Clause\n\n")
b.WriteString("package arch\n\n") b.WriteString("package arch\n\n")
b.WriteString("// commonGeneratedInstrs is the set of opcodes shared by every architecture\n") b.WriteString("// commonGeneratedInstrs is the set of opcodes shared by every architecture\n")
b.WriteString("// (RET, JMP, NOP, CALL, TEXT, FUNCDATA, PCDATA, …).\n") b.WriteString("// (RET, JMP, NOP, CALL, TEXT, FUNCDATA, PCDATA, …).\n")
@@ -154,7 +157,10 @@ func stringLit(elt ast.Expr) string {
func writeGen(arch, sub string, names []string) error { func writeGen(arch, sub string, names []string) error {
var b strings.Builder var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n") b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n\n") b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
b.WriteString("// SPDX-License-Identifier: BSD-3-Clause\n\n")
b.WriteString("package arch\n\n") b.WriteString("package arch\n\n")
b.WriteString("// " + arch + "GeneratedInstrs is the complete set of " + arch + b.WriteString("// " + arch + "GeneratedInstrs is the complete set of " + arch +
" mnemonics accepted by\n// Go's Plan 9 assembler.\n") " mnemonics accepted by\n// Go's Plan 9 assembler.\n")
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT. // Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain. // Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch package arch
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT. // Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain. // Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch package arch
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT. // Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/util.go from the Go toolchain. // Source: cmd/internal/obj/util.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch package arch
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT. // Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain. // Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch package arch
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT. // Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain. // Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch package arch
+181
View File
@@ -0,0 +1,181 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGOObjectAARCH64Structure checks the basic structure of the emitted
// AArch64 GOOBJ: the preamble, the magic, the block offsets and the
// non-package symbol definitions.
func TestGOObjectAARCH64Structure(t *testing.T) {
f, errs := parser.Parse("k_arm64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.GOObjectAARCH64("testpkg", "k_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
// Check preamble.
idx := strings.Index(string(obj), "\n!\n")
if idx < 0 {
t.Fatal("missing preamble separator")
}
preamble := string(obj[:idx])
if !strings.HasPrefix(preamble, "go object") {
t.Errorf("preamble = %q, want 'go object ...'", preamble)
}
// Check GOOBJ magic.
magicIdx := idx + 3
if magicIdx+8 > len(obj) || string(obj[magicIdx:magicIdx+8]) != "\x00go120ld" {
t.Error("missing GOOBJ magic")
}
// The object should contain the function's code.
if len(img.Code) == 0 {
t.Error("no code generated")
}
}
// TestGOObjectAARCH64Link does an end-to-end link test: it cross-compiles a
// Go program for arm64, substitutes the gasm-produced object into the package
// archive, re-links with cmd/link, and verifies the symbol appears in the
// resulting binary. The binary is not executed (no arm64 host or qemu).
// Skipped when no Go toolchain is available.
func TestGOObjectAARCH64Link(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
asmSrc := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
`
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
mainSrc := `package main
func add(a, b int64) int64
func main() {
if add(20, 22) != 42 {
panic("bad add")
}
}
`
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module a64link\n\ngo 1.21\n"), 0o644); err != nil {
t.Fatal(err)
}
// Capture the cross build (GOARCH=arm64): the package archive and the
// link line.
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
build.Env = append(os.Environ(), "GOARCH=arm64")
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var pkgArch, work, linkLine, asmObj string
for _, line := range strings.Split(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
}
defer os.RemoveAll(work)
// Expand $WORK in the object path.
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
// Read the toolchain-produced object and assemble the same source with gasm.
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse("main_arm64.s", string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
gasmObj, err := img.GOObjectAARCH64("a64link", "main_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
// Replace the toolchain-produced object with gasm's.
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
// Re-link.
if linkLine == "" {
t.Skip("could not find link command in build log")
}
// Expand $WORK in the link command.
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
// Verify the binary exists and contains the symbol.
binPath := filepath.Join(dir, "prog")
if _, err := os.Stat(binPath); err != nil {
t.Fatalf("binary not found: %v", err)
}
binData, err := os.ReadFile(binPath)
if err != nil {
t.Fatalf("read binary: %v", err)
}
if !strings.Contains(string(binData), "add") && !strings.Contains(string(binData), "a64link") {
t.Error("binary does not contain expected symbol")
}
}
File diff suppressed because it is too large Load Diff
+764
View File
@@ -0,0 +1,764 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// arm64 (AArch64) instruction encoding.
//
// The encoder is data-driven: each mnemonic maps to an instruction format and
// an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own arm64
// backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm`
// exactly — the ground-truth oracle for the verify suite.
//
// All AArch64 instructions are 32 bits, little-endian. The formats used here
// (per the ARM Architecture Reference Manual):
//
// DP-shifted-reg sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
// DP-immediate sf<<31 | op<<30 | S<<29 | 0x11<<24 | imm12<<10 | Rn<<5 | Rd
// Logical-imm sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd
// Move-wide sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd
// Load/store size<<30 | 0x7<<27 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt
// LDST-unscaled size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<5 | Rt (actually imm9<<12 | Rn<<5 | Rt)
// LDST-pair opc<<30 | 0x5<<27 | V<<26 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt
// Branch-imm 0<<31 | 0x5<<26 | imm26 (B)
// Branch-imm 1<<31 | 0x5<<26 | imm26 (BL)
// Branch-cond 0x2A<<25 | imm19<<5 | cond (B.cond)
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
// R0–R30 (integer), F0–F31 (floating point), and the ABI aliases the
// runtime's assembly uses. Returns -1 for an unrecognised name.
func arm64RegNum(name string) int {
switch name {
case "R0":
return 0
case "R1":
return 1
case "R2":
return 2
case "R3":
return 3
case "R4":
return 4
case "R5":
return 5
case "R6":
return 6
case "R7":
return 7
case "R8":
return 8
case "R9":
return 9
case "R10":
return 10
case "R11":
return 11
case "R12":
return 12
case "R13":
return 13
case "R14":
return 14
case "R15":
return 15
case "R16":
return 16
case "R17":
return 17
case "R18":
return 18
case "R19":
return 19
case "R20":
return 20
case "R21":
return 21
case "R22":
return 22
case "R23":
return 23
case "R24":
return 24
case "R25":
return 25
case "R26", "REGCTXT", "CTXT":
return 26
case "R27", "REGTMP", "TMP":
return 27
case "R28", "REGG", "g":
return 28
case "R29", "FP":
return 29
case "R30", "LR", "LINK":
return 30
case "R31", "ZR":
return 31
case "SP":
return 31 // SP and ZR share encoding 31; context determines meaning
}
// F0–F31.
if len(name) >= 1 && name[0] == 'F' {
n := 0
for i := 1; i < len(name); i++ {
if name[i] < '0' || name[i] > '9' {
return -1
}
n = n*10 + int(name[i]-'0')
}
if n <= 31 {
return n
}
}
return -1
}
// arm64IsSP reports whether a register operand is the stack pointer (R31/SP),
// which uses a different encoding path for some instructions.
func arm64IsSP(name string) bool {
return name == "SP"
}
// ---- format helpers ----
// a64wordLE encodes a uint32 as 4 little-endian bytes.
func a64wordLE(w uint32) []byte {
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
}
// a64WordsLE concatenates one or more instruction words as little-endian bytes.
func a64WordsLE(ws ...uint32) []byte {
var out []byte
for _, w := range ws {
out = append(out, a64wordLE(w)...)
}
return out
}
// ---- data-processing (shifted register) ----
// a64DPSR encodes a data-processing (shifted register) instruction:
// sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd.
func a64DPSR(sf, op, S, shift, rm, imm6, rn, rd uint32) uint32 {
return sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | rm<<16 | imm6<<10 | rn<<5 | rd
}
// ---- data-processing (immediate) ----
// a64AddSub encodes an ADD/SUB (immediate) instruction:
// sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | Rn<<5 | Rd.
func a64AddSub(sf, op, S, sh, imm12, rn, rd uint32) uint32 {
return sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | rn<<5 | rd
}
// ---- logical (immediate) ----
// a64LogicalImm encodes a logical (immediate) instruction:
// sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd.
func a64LogicalImm(sf, opc, N, immr, imms, rn, rd uint32) uint32 {
return sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | rn<<5 | rd
}
// ---- move wide ----
// a64MoveWide encodes a MOVZ/MOVK/MOVN instruction:
// sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd.
func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 {
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
}
// ---- load/store (unsigned immediate, scaled) ----
// a64LSU encodes a load/store register (unsigned immediate, scaled):
// size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt.
// (0x39<<24 encodes bits 29:24 = 111001, the scaled unsigned offset form.)
func a64LSU(size, V, opc, imm12, rn, rt uint32) uint32 {
return size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | rn<<5 | rt
}
// ---- load/store (unscaled immediate) ----
// a64LSUnscaled encodes a load/store register (unscaled immediate, 9-bit signed):
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<12 | Rn<<5 | Rt.
// Note: the 0<<24 distinguishes unscaled from the pre/post-index forms.
func a64LSUnscaled(size, V, opc int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | uint32(opc)<<22 |
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// ---- load/store pair ----
// a64LSP encodes a load/store pair instruction (signed offset):
// opc<<30 | 0x5<<27 | V<<26 | 2<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
// opc: 0=32-bit, 1=reserved, 2=64-bit. V: 0=integer, 1=FP/SIMD.
// L: 0=store, 1=load. imm7 is the signed scaled offset (÷8 for 64-bit pairs).
func a64LSP(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
return opc<<30 | 5<<27 | V<<26 | 2<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
}
// ---- load/store pair (pre-index) ----
// a64LSPPre encodes a load/store pair (pre-index):
// opc<<30 | 0x5<<27 | V<<26 | 0b11<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
func a64LSPPre(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
return opc<<30 | 5<<27 | V<<26 | 3<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
}
// ---- load/store pair (post-index) ----
// a64LSPPost encodes a load/store pair (post-index):
// opc<<30 | 0x5<<27 | V<<26 | 0b01<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
func a64LSPPost(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
return opc<<30 | 5<<27 | V<<26 | 1<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
}
// ---- pre-index load/store ----
// a64LSPreIndex encodes a load/store register (pre-index):
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func a64LSPreIndex(size, V, opc uint32, imm9 int32, rn, rt uint32) uint32 {
return size<<30 | 7<<27 | V<<26 | opc<<22 | 3<<10 | (uint32(imm9)&0x1FF)<<12 | rn<<5 | rt
}
// ---- post-index load/store ----
// a64LSPostIndex encodes a load/store register (post-index):
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func a64LSPostIndex(size, V, opc uint32, imm9 int32, rn, rt uint32) uint32 {
return size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<10 | (uint32(imm9)&0x1FF)<<12 | rn<<5 | rt
}
// ---- branches ----
// a64Branch encodes an unconditional branch (B/BL):
// op<<31 | 0x5<<26 | imm26.
func a64Branch(op uint32, imm26 int32) uint32 {
return op<<31 | 5<<26 | (uint32(imm26) & 0x03FFFFFF)
}
// a64BranchCond encodes a conditional branch (B.cond):
// 0x2A<<25 | imm19<<5 | cond.
func a64BranchCond(imm19 int32, cond uint32) uint32 {
return 0x2A<<25 | (uint32(imm19)&0x7FFFF)<<5 | cond&0xF
}
// a64UncondBranch encodes an unconditional branch register (BR/BLR/RET):
// 0x6B<<25 | opc<<21 | 0x1F<<16 | Rn<<5 | Rd.
// opc: 0=BR, 1=BLR, 2=RET. For RET, Rn defaults to LR(30).
func a64UncondBranch(opc, rn, rd uint32) uint32 {
return 0x6B<<25 | opc<<21 | 0x1F<<16 | rn<<5 | rd
}
// ---- ADR/ADRP ----
// a64ADR encodes an ADR instruction (p=0) or ADRP instruction (p=1):
// p<<31 | immlo<<29 | 0x10<<24 | immhi<<5 | Rd.
func a64ADR(p uint32, immhi int32, immlo uint32, rd uint32) uint32 {
return p<<31 | immlo<<29 | 0x10<<24 | (uint32(immhi)&0x7FFFF)<<5 | rd
}
// ---- EXTR ----
// a64EXTR encodes an EXTR instruction:
// sf<<31 | 0<<29 | 0x27<<23 | N<<22 | 0<<21 | Rm<<16 | imms<<10 | Rn<<5 | Rd.
func a64EXTR(sf, N, rm, imms, rn, rd uint32) uint32 {
return sf<<31 | 0x27<<23 | N<<22 | rm<<16 | imms<<10 | rn<<5 | rd
}
// ---- system ----
// a64NOP encodes a NOP: 0xd503201f.
const a64NOP uint32 = 0xd503201f
// a64BRK encodes a BRK instruction: 0xd4200000 | imm16<<5.
func a64BRK(imm16 uint32) uint32 {
return 0xd4200000 | imm16<<5
}
// ---- condition codes ----
const (
a64CondEQ = 0x0
a64CondNE = 0x1
a64CondCS = 0x2
a64CondHS = 0x2
a64CondCC = 0x3
a64CondLO = 0x3
a64CondMI = 0x4
a64CondPL = 0x5
a64CondVS = 0x6
a64CondVC = 0x7
a64CondHI = 0x8
a64CondLS = 0x9
a64CondGE = 0xa
a64CondLT = 0xb
a64CondGT = 0xc
a64CondLE = 0xd
a64CondAL = 0xe
a64CondNV = 0xf
)
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
var arm64CondMap = map[string]uint32{
"EQ": a64CondEQ,
"NE": a64CondNE,
"CS": a64CondCS,
"HS": a64CondHS,
"CC": a64CondCC,
"LO": a64CondLO,
"MI": a64CondMI,
"PL": a64CondPL,
"VS": a64CondVS,
"VC": a64CondVC,
"HI": a64CondHI,
"LS": a64CondLS,
"GE": a64CondGE,
"LT": a64CondLT,
"GT": a64CondGT,
"LE": a64CondLE,
}
// ---- instruction format tags ----
type a64Format uint8
const (
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
a64FDPIR // data-processing (immediate): ADD/SUB $imm
a64FLogImm // logical (immediate): AND/ORR/EOR $imm
a64FMovWide // move wide: MOVZ, MOVN, MOVK
a64FLSU // load/store (unsigned immediate, scaled)
a64FLSUnscaled // load/store (unscaled immediate)
a64FLSPair // load/store pair
a64FBranch // unconditional branch (B/BL)
a64FBranchCond // conditional branch (B.cond)
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
a64FADR // ADR/ADRP
a64FEXTR // EXTR
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
a64FSystem // system: NOP, BRK, etc.
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
a64FFMovGR // FMOV between GP and FP registers
a64FCRC32 // CRC32
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
)
// a64Enc is one instruction's encoding: its bit layout (format) and the
// opcode constant, positioned at its exact bit range.
type a64Enc struct {
format a64Format
op uint32 // the pre-positioned opcode bits
size int // 4 for most, 8 for DP-imm with shift, etc.
}
// a64InstrTable maps AArch64 mnemonics (as the Go assembler spells them) to
// their encoding. The base integer, memory, floating-point and SIMD
// instruction sets are covered.
var a64InstrTable = map[string]a64Enc{}
func init() {
// ---- data-processing (shifted register) ----
// Format: sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
dpsr := map[string]uint32{
// Add/Sub
"ADD": 1<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=1, op=0, S=0 (64-bit default)
"ADDW": 0<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=0
"ADDS": 1<<31 | 0<<30 | 1<<29 | 0x0b<<24,
"ADDSW": 0<<31 | 0<<30 | 1<<29 | 0x0b<<24,
"SUB": 1<<31 | 1<<30 | 0<<29 | 0x0b<<24,
"SUBW": 0<<31 | 1<<30 | 0<<29 | 0x0b<<24,
"SUBS": 1<<31 | 1<<30 | 1<<29 | 0x0b<<24,
"SUBSW": 0<<31 | 1<<30 | 1<<29 | 0x0b<<24,
// Logical (shifted register)
"AND": 1<<31 | 0<<29 | 0x0a<<24,
"ANDW": 0<<31 | 0<<29 | 0x0a<<24,
"BIC": 1<<31 | 0<<29 | 0x0a<<24 | 1<<21,
"BICW": 0<<31 | 0<<29 | 0x0a<<24 | 1<<21,
"ORR": 1<<31 | 1<<29 | 0x0a<<24,
"ORRW": 0<<31 | 1<<29 | 0x0a<<24,
"ORN": 1<<31 | 1<<29 | 0x0a<<24 | 1<<21,
"ORNW": 0<<31 | 1<<29 | 0x0a<<24 | 1<<21,
"EOR": 1<<31 | 2<<29 | 0x0a<<24,
"EORW": 0<<31 | 2<<29 | 0x0a<<24,
"EON": 1<<31 | 2<<29 | 0x0a<<24 | 1<<21,
"EONW": 0<<31 | 2<<29 | 0x0a<<24 | 1<<21,
"ANDS": 1<<31 | 3<<29 | 0x0a<<24,
"ANDSW": 0<<31 | 3<<29 | 0x0a<<24,
"BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21,
"BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21,
// Shift
"LSL": 1<<31 | 0<<29 | 0x0a<<24, // alias of UBFM
"LSLW": 0<<31 | 0<<29 | 0x0a<<24,
"LSR": 1<<31 | 0<<29 | 0x0a<<24,
"LSRW": 0<<31 | 0<<29 | 0x0a<<24,
"ASR": 1<<31 | 0<<29 | 0x0a<<24,
"ASRW": 0<<31 | 0<<29 | 0x0a<<24,
"ROR": 1<<31 | 0<<29 | 0x0a<<24,
"RORW": 0<<31 | 0<<29 | 0x0a<<24,
// Multiply
"MADD": 1<<31 | 0<<29 | 0x1b<<24 | 0<<21,
"MADDW": 0<<31 | 0<<29 | 0x1b<<24 | 0<<21,
"MSUB": 1<<31 | 0<<29 | 0x1b<<24 | 1<<21,
"MSUBW": 0<<31 | 0<<29 | 0x1b<<24 | 1<<21,
// Divide
"SDIV": 1<<31 | 0<<29 | 0x0d<<24,
"SDIVW": 0<<31 | 0<<29 | 0x0d<<24,
"UDIV": 1<<31 | 0<<29 | 0x0d<<24 | 1<<10,
"UDIVW": 0<<31 | 0<<29 | 0x0d<<24 | 1<<10,
// CRC
"CRC32B": 0<<31 | 0<<29 | 0x1b<<24 | 4<<10,
"CRC32H": 0<<31 | 0<<29 | 0x1b<<24 | 5<<10,
"CRC32W": 0<<31 | 0<<29 | 0x1b<<24 | 6<<10,
"CRC32X": 1<<31 | 0<<29 | 0x1b<<24 | 7<<10,
// Conditional select
"CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10,
"CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10,
"CSINC": 1<<31 | 0<<29 | 0x1d<<24 | 1<<10,
"CSINCW": 0<<31 | 0<<29 | 0x1d<<24 | 1<<10,
"CSINV": 1<<31 | 0<<29 | 0x1d<<24 | 2<<10,
"CSINVW": 0<<31 | 0<<29 | 0x1d<<24 | 2<<10,
"CSNEG": 1<<31 | 0<<29 | 0x1d<<24 | 3<<10,
"CSNEGW": 0<<31 | 0<<29 | 0x1d<<24 | 3<<10,
}
for m, op := range dpsr {
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
}
// Aliases that map to the same encoding as their target.
a64InstrTable["CMP"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
a64InstrTable["CMPW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBSW"]}
a64InstrTable["CMN"] = a64Enc{format: a64FDPSR, op: dpsr["ADDS"]}
a64InstrTable["CMNW"] = a64Enc{format: a64FDPSR, op: dpsr["ADDSW"]}
a64InstrTable["TST"] = a64Enc{format: a64FDPSR, op: dpsr["ANDS"]}
a64InstrTable["TSTW"] = a64Enc{format: a64FDPSR, op: dpsr["ANDSW"]}
a64InstrTable["NEG"] = a64Enc{format: a64FDPSR, op: dpsr["SUB"]}
a64InstrTable["NEGW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBW"]}
a64InstrTable["NEGS"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
a64InstrTable["MVN"] = a64Enc{format: a64FDPSR, op: dpsr["ORN"]}
a64InstrTable["MVNW"] = a64Enc{format: a64FDPSR, op: dpsr["ORNW"]}
a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]}
a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]}
// ---- data-processing (immediate) ----
// ADD/SUB $imm, Rn, Rd
a64InstrTable["ADDImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 0<<29 | 0x11<<24}
a64InstrTable["ADDWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 0<<30 | 0<<29 | 0x11<<24}
a64InstrTable["SUBImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 0<<29 | 0x11<<24}
a64InstrTable["SUBWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 1<<30 | 0<<29 | 0x11<<24}
a64InstrTable["ADDSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 1<<29 | 0x11<<24}
a64InstrTable["SUBSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 1<<29 | 0x11<<24}
// ---- move wide ----
// MOVZ/MOVN/MOVK
a64InstrTable["MOVZ"] = a64Enc{format: a64FMovWide, op: 1<<31 | 2<<29 | 0x25<<23}
a64InstrTable["MOVZW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 2<<29 | 0x25<<23}
a64InstrTable["MOVN"] = a64Enc{format: a64FMovWide, op: 1<<31 | 0<<29 | 0x25<<23}
a64InstrTable["MOVNW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 0<<29 | 0x25<<23}
a64InstrTable["MOVK"] = a64Enc{format: a64FMovWide, op: 1<<31 | 3<<29 | 0x25<<23}
a64InstrTable["MOVKW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 3<<29 | 0x25<<23}
// ---- ADR/ADRP ----
a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0}
a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1}
// ---- load/store (unsigned immediate) ----
a64InstrTable["MOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<22} // LDR 64-bit
a64InstrTable["MOVWU"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<22} // LDR 32-bit unsigned
a64InstrTable["MOVHU"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 1<<22} // LDRH unsigned
a64InstrTable["MOVBU"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 1<<22} // LDRB unsigned
a64InstrTable["MOVW"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 2<<22} // LDRSW (signed 32→64)
a64InstrTable["MOVH"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 2<<22} // LDRSH (signed half)
a64InstrTable["MOVB"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 2<<22} // LDRSB (signed byte)
a64InstrTable["FMOVS"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 32-bit FP
a64InstrTable["FMOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 64-bit FP
// Store opcodes (load ^ (1<<22)):
// STR 64-bit: size=3, V=0, opc=00 → 3<<30 | 7<<27 | 0<<22
// STR 32-bit: size=2, V=0, opc=00 → 2<<30 | 7<<27 | 0<<22
// STRH: size=1, V=0, opc=00 → 1<<30 | 7<<27 | 0<<22
// STRB: size=0, V=0, opc=00 → 0<<30 | 7<<27 | 0<<22
// ---- branches ----
a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26}
a64InstrTable["BL"] = a64Enc{format: a64FBranch, op: 1<<31 | 5<<26}
// Conditional branches.
condBranches := map[string]uint32{
"BEQ": 0x0, "BNE": 0x1, "BCS": 0x2, "BHS": 0x2,
"BCC": 0x3, "BLO": 0x3, "BMI": 0x4, "BPL": 0x5,
"BVS": 0x6, "BVC": 0x7, "BHI": 0x8, "BLS": 0x9,
"BGE": 0xa, "BLT": 0xb, "BGT": 0xc, "BLE": 0xd,
}
for name, cond := range condBranches {
a64InstrTable[name] = a64Enc{format: a64FBranchCond, op: 0x2A<<25 | cond}
}
// Unconditional branch register (BR/BLR/RET).
a64InstrTable["BR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 0<<21}
a64InstrTable["BLR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 1<<21}
a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21}
// ---- system ----
a64InstrTable["NOP"] = a64Enc{format: a64FSystem, op: a64NOP}
a64InstrTable["NOOP"] = a64Enc{format: a64FSystem, op: a64NOP}
a64InstrTable["BRK"] = a64Enc{format: a64FSystem, op: 0xd4200000}
a64InstrTable["UNDEF"] = a64Enc{format: a64FSystem, op: a64BRK(0)}
// ---- EXTR ----
a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22}
a64InstrTable["EXTRW"] = a64Enc{format: a64FEXTR, op: 0<<31 | 0x27<<23 | 0<<22}
// ---- bitfield ----
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
a64InstrTable["UBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
a64InstrTable["BFI"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
// ---- FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL ----
fp3 := map[string]uint32{
"FADDS": 0x1e202800, "FADDD": 0x1e602800,
"FSUBS": 0x1e203800, "FSUBD": 0x1e603800,
"FMULS": 0x1e200800, "FMULD": 0x1e600800,
"FDIVS": 0x1e201800, "FDIVD": 0x1e601800,
"FMAXS": 0x1e204800, "FMAXD": 0x1e604800,
"FMINS": 0x1e205800, "FMIND": 0x1e605800,
"FMAXNMS": 0x1e206800, "FMAXNMD": 0x1e606800,
"FMINNMS": 0x1e207800, "FMINNMD": 0x1e607800,
"FNMULS": 0x1e208800, "FNMULD": 0x1e608800,
}
for m, op := range fp3 {
a64InstrTable[m] = a64Enc{format: a64FFP3, op: op}
}
// ---- FP unary (Rn, Rd): FMOV reg-reg, FABS, FNEG, FSQRT, FCVT, FRINT* ----
fp1 := map[string]uint32{
"FMOVS": 0x1e204000, "FMOVD": 0x1e604000,
"FABSS": 0x1e20c000, "FABSD": 0x1e60c000,
"FNEGS": 0x1e214000, "FNEGD": 0x1e614000,
"FSQRTS": 0x1e21c000, "FSQRTD": 0x1e61c000,
"FCVTSD": 0x1e22c000, "FCVTDS": 0x1e624000,
"FRINTNS": 0x1e244000, "FRINTND": 0x1e644000,
"FRINTPS": 0x1e24c000, "FRINTPD": 0x1e64c000,
"FRINTMS": 0x1e254000, "FRINTMD": 0x1e654000,
"FRINTZS": 0x1e25c000, "FRINTZD": 0x1e65c000,
"FRINTAS": 0x1e264000, "FRINTAD": 0x1e664000,
"FRINTXS": 0x1e274000, "FRINTXD": 0x1e674000,
"FRINTIS": 0x1e27c000, "FRINTID": 0x1e67c000,
}
for m, op := range fp1 {
a64InstrTable[m] = a64Enc{format: a64FFPUnary, op: op}
}
// ---- FP 4-operand FMA (Ra, Rm, Rn, Rd) ----
fp4 := map[string]uint32{
"FMADDS": 0x1f000000, "FMADDD": 0x1f400000,
"FMSUBS": 0x1f008000, "FMSUBD": 0x1f408000,
"FNMADDS": 0x1f200000, "FNMADDD": 0x1f600000,
"FNMSUBS": 0x1f208000, "FNMSUBD": 0x1f608000,
}
for m, op := range fp4 {
a64InstrTable[m] = a64Enc{format: a64FFP4, op: op}
}
// ---- FP compare (Rm, Rn or #0, Rn) ----
fpcmp := map[string]uint32{
"FCMPS": 0x1e202000, "FCMPD": 0x1e602000,
"FCMPES": 0x1e202010, "FCMPED": 0x1e602010,
}
for m, op := range fpcmp {
a64InstrTable[m] = a64Enc{format: a64FFPCmp, op: op}
}
// ---- FP conditional compare (Rm, Rn, #nzcv, cond) ----
fpccmp := map[string]uint32{
"FCCMPS": 0x1e200400, "FCCMPD": 0x1e600400,
"FCCMPES": 0x1e200410, "FCCMPED": 0x1e600410,
}
for m, op := range fpccmp {
a64InstrTable[m] = a64Enc{format: a64FFPCCmp, op: op}
}
// ---- FP conditional select (Rm, Rn, Rd, cond) ----
a64InstrTable["FCSELS"] = a64Enc{format: a64FFPSel, op: 0x1e200c00}
a64InstrTable["FCSELD"] = a64Enc{format: a64FFPSel, op: 0x1e600c00}
// ---- FP ↔ integer conversion ----
fpcvt := map[string]uint32{
"FCVTZSD": 0x9e780000, "FCVTZSDW": 0x1e780000,
"FCVTZSS": 0x9e380000, "FCVTZSSW": 0x1e380000,
"FCVTZUD": 0x9e790000, "FCVTZUDW": 0x1e790000,
"FCVTZUS": 0x9e390000, "FCVTZUSW": 0x1e390000,
"SCVTFD": 0x9e620000, "SCVTFS": 0x9e220000,
"SCVTFWD": 0x1e620000, "SCVTFWS": 0x1e220000,
"UCVTFD": 0x9e630000, "UCVTFS": 0x9e230000,
"UCVTFWD": 0x1e630000, "UCVTFWS": 0x1e230000,
}
for m, op := range fpcvt {
a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op}
}
// ---- FMOV between GP and FP registers ----
a64InstrTable["FMOVGR"] = a64Enc{format: a64FFMovGR, op: 0x1e260000} // placeholder, actual encoding depends on direction
// ---- conditional select: CSEL, CSINC, CSINV, CSNEG ----
csel := map[string]uint32{
"CSEL": 0x9a800000, "CSELW": 0x1a800000,
"CSINC": 0x9a800400, "CSINCW": 0x1a800400,
"CSINV": 0xda800000, "CSINVW": 0x5a800000,
"CSNEG": 0xda800400, "CSNEGW": 0x5a800400,
}
for m, op := range csel {
a64InstrTable[m] = a64Enc{format: a64FCSEL, op: op}
}
// Aliases
a64InstrTable["CSET"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
a64InstrTable["CSETW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
a64InstrTable["CSETM"] = a64Enc{format: a64FCSEL, op: 0xda800000}
a64InstrTable["CSETMW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
a64InstrTable["CINC"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
a64InstrTable["CINCW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
a64InstrTable["CINV"] = a64Enc{format: a64FCSEL, op: 0xda800000}
a64InstrTable["CINVW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
a64InstrTable["CNEG"] = a64Enc{format: a64FCSEL, op: 0xda800400}
a64InstrTable["CNEGW"] = a64Enc{format: a64FCSEL, op: 0x5a800400}
// ---- CRC32 ----
crc32 := map[string]uint32{
"CRC32B": 0x1ac04000, "CRC32H": 0x1ac04400,
"CRC32W": 0x1ac04800, "CRC32X": 0x9ac04c00,
"CRC32CB": 0x1ac05000, "CRC32CH": 0x1ac05400,
"CRC32CW": 0x1ac05800, "CRC32CX": 0x9ac05c00,
}
for m, op := range crc32 {
a64InstrTable[m] = a64Enc{format: a64FCRC32, op: op}
}
// ---- exclusive load/store ----
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
a64InstrTable["LDXRW"] = a64Enc{format: a64FExcl, op: 0x885f7c00}
a64InstrTable["LDAXR"] = a64Enc{format: a64FExcl, op: 0xc85ffc00}
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
a64InstrTable["STXRW"] = a64Enc{format: a64FExcl, op: 0x88007c00}
a64InstrTable["STLXR"] = a64Enc{format: a64FExcl, op: 0xc800fc00}
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
// ---- LSE atomics ----
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["LDADDW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["LDADDB"] = a64Enc{format: a64FLSE, op: 0<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["LDADDH"] = a64Enc{format: a64FLSE, op: 1<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["CASD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x45<<21 | 0x1f<<10}
a64InstrTable["CASW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x45<<21 | 0x1f<<10}
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
// ---- SIMD basics ----
a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400}
a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400}
a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00}
}
// ---- load/store helper tables ----
// a64LSType describes the load/store parameters for a MOV width mnemonic.
type a64LSType struct {
size int // 0=byte, 1=half, 2=word, 3=dword
V int // 0=integer, 1=FP
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
}
// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters.
// For loads, opc selects signed vs unsigned; for stores, we flip the opc.
var a64LoadTable = map[string]a64LSType{
"MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
}
// a64StoreOpc returns the store opc for a given load type.
// For integer: store opc = 00 (the load opc bits cleared).
// For FP: store opc = 00 (same pattern).
func a64StoreOpc(t a64LSType) int {
if t.V == 1 {
return 0 // FP store
}
return 0 // integer store
}
// a64MovRegTable maps register-to-register MOV mnemonic expansions.
// The Go toolchain encodes MOV Rn, Rd as ORR Rn, ZR, Rd.
var a64MovRegTable = map[string]uint32{
"MOVD": 1<<31 | 1<<29 | 0x0a<<24, // ORR 64-bit
"MOVW": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVB": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit (byte move)
"MOVBU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVH": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVHU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVWU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
}
// arm64RegClass discriminates integer (R), floating-point (F) registers for
// the MOV pseudo-instruction.
type arm64RegClass int
const (
arm64ClsNone arm64RegClass = iota
arm64ClsGR
arm64ClsFP
)
// arm64RegClassOf reports the register class of a register operand name.
func arm64RegClassOf(name string) arm64RegClass {
switch {
case name == "":
return arm64ClsNone
case len(name) >= 1 && name[0] == 'F':
return arm64ClsFP
default:
return arm64ClsGR
}
}
// arm64Movcon returns the shift (in units of 16 bits) at which a non-zero
// 16-bit chunk of v sits, or -1 if v cannot be represented as a single
// MOVZ/MOVN immediate. This is the Go toolchain's movcon function.
func arm64Movcon(v int64) int {
for s := 0; s < 64; s += 16 {
if (uint64(v) &^ (uint64(0xFFFF) << uint(s))) == 0 {
return s
}
}
return -1
}
+574
View File
@@ -0,0 +1,574 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
func TestArm64LDRSTREncoding(t *testing.T) {
tests := []struct {
name string
got uint32
want uint32
}{
{"LDR X4, [SP, #56]", a64LSU(3, 0, 1, 7, 31, 4), 0xf9401fe4},
{"STR X4, [SP, #64]", a64LSU(3, 0, 0, 8, 31, 4), 0xf90023e4},
{"STR X5, [SP, #32]", a64LSU(3, 0, 0, 4, 31, 5), 0xf90013e5},
{"LDR X6, [SP, #32]", a64LSU(3, 0, 1, 4, 31, 6), 0xf94013e6},
}
for _, tt := range tests {
if tt.got != tt.want {
t.Errorf("%s: got %08x, want %08x", tt.name, tt.got, tt.want)
}
}
}
func TestArm64PrologueEncoding(t *testing.T) {
fi := arm64FrameInfo{autosize: 48, frame: 32, leaf: false}
pro := arm64Prologue(fi)
if len(pro) != 12 {
t.Fatalf("prologue length: got %d, want 12", len(pro))
}
expected := []uint32{0xf81d0ffe, 0xf81f83fd, 0xd10023fd}
for i, w := range leWords(pro) {
if w != expected[i] {
t.Errorf("prologue word %d: got %08x, want %08x", i, w, expected[i])
}
}
}
func TestArm64EpilogueSmallEncoding(t *testing.T) {
fi := arm64FrameInfo{autosize: 48, frame: 32, leaf: false}
ret := arm64Return(fi)
if len(ret) != 12 {
t.Fatalf("epilogue length: got %d, want 12", len(ret))
}
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #48; RET
expected := []uint32{0xf85f83fd, 0xf84307fe, 0xd65f03c0}
for i, w := range leWords(ret) {
if w != expected[i] {
t.Errorf("epilogue word %d: got %08x, want %08x", i, w, expected[i])
}
}
}
func TestArm64LargeFrameEncoding(t *testing.T) {
fi := arm64FrameInfo{autosize: 272, frame: 256, leaf: false}
pro := arm64Prologue(fi)
if len(pro) != 16 {
t.Fatalf("prologue length: got %d, want 16", len(pro))
}
expected := []uint32{0xd10443f4, 0xa93ffa9d, 0x9100029f, 0xd10023fd}
for i, w := range leWords(pro) {
if w != expected[i] {
t.Errorf("prologue word %d: got %08x, want %08x", i, w, expected[i])
}
}
epi := arm64Return(fi)
if len(epi) != 12 {
t.Fatalf("epilogue length: got %d, want 12", len(epi))
}
eexpected := []uint32{0xa97ffbfd, 0x910443ff, 0xd65f03c0}
for i, w := range leWords(epi) {
if w != eexpected[i] {
t.Errorf("epilogue word %d: got %08x, want %08x", i, w, eexpected[i])
}
}
}
func TestArm64NoFrame(t *testing.T) {
fi := arm64FrameInfo{autosize: 0, frame: 0, leaf: true}
pro := arm64Prologue(fi)
if len(pro) != 0 {
t.Errorf("no-frame prologue: got %d bytes, want 0", len(pro))
}
ret := arm64Return(fi)
if len(ret) != 4 {
t.Fatalf("no-frame return: got %d bytes, want 4", len(ret))
}
if leWord(ret) != 0xd65f03c0 {
t.Errorf("no-frame RET: got %08x, want d65f03c0", leWord(ret))
}
}
func TestArm64RegNum(t *testing.T) {
tests := []struct {
name string
want int
}{
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
{"F0", 0}, {"F4", 4}, {"F31", 31},
{"INVALID", -1}, {"X0", -1}, {"", -1},
}
for _, tt := range tests {
got := arm64RegNum(tt.name)
if got != tt.want {
t.Errorf("arm64RegNum(%q) = %d, want %d", tt.name, got, tt.want)
}
}
}
func TestArm64ComputeFrame(t *testing.T) {
src := "TEXT ·f(SB), NOSPLIT, $32-0\n\tADD\tR4, R5\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
fi := arm64ComputeFrame(f.Decls[0].(*ast.Text))
if fi.frame != 32 {
t.Errorf("frame: got %d, want 32", fi.frame)
}
if fi.autosize != 48 { // 32+8=40, aligned to48
t.Errorf("autosize: got %d, want 48", fi.autosize)
}
// ADD + RET with no CALL/BL → leaf
if !fi.leaf {
t.Error("expected leaf")
}
}
func TestArm64IsLeaf(t *testing.T) {
src := "TEXT ·f(SB), NOSPLIT, $0-0\n\tADD\tR4, R5\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if !arm64IsLeaf(f.Decls[0].(*ast.Text)) {
t.Error("expected leaf")
}
src2 := "TEXT ·f(SB), NOSPLIT, $0-0\n\tBL\tother(SB)\n\tRET\n"
f2, errs := parser.Parse("test_arm64.s", src2)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if arm64IsLeaf(f2.Decls[0].(*ast.Text)) {
t.Error("expected non-leaf")
}
}
func TestArm64Bitmask(t *testing.T) {
tests := []struct {
v uint64
sf int
N, immr, imms uint32
ok bool
}{
{1, 1, 1, 0, 0, true}, // single bit at pos 0
{2, 1, 1, 63, 0, true}, // single bit at pos 1 (immr = esize-1)
{0, 1, 0, 0, 0, false}, // zero is not a bitmask
{0xFFFFFFFFFFFFFFFF, 1, 0, 0, 0, false}, // all ones is not a bitmask
{0x5555555555555555, 1, 0, 0, 0x3E, true}, // alternating bits (esize=2, ones=1)
{0xFFFFFFFF00000000, 1, 1, 32, 31, true}, // upper 32 bits set (esize=64, ones=32)
}
for _, tt := range tests {
N, immr, imms, ok := arm64Bitmask(tt.v, tt.sf)
if ok != tt.ok {
t.Errorf("arm64Bitmask(%#x, %d): ok=%v, want %v", tt.v, tt.sf, ok, tt.ok)
continue
}
if ok && (N != tt.N || immr != tt.immr || imms != tt.imms) {
t.Errorf("arm64Bitmask(%#x, %d): N=%d immr=%d imms=%d, want N=%d immr=%d imms=%d",
tt.v, tt.sf, N, immr, imms, tt.N, tt.immr, tt.imms)
}
}
}
func TestArm64AssembleFile(t *testing.T) {
src := `#include "textflag.h"
TEXT ·simple(SB), NOSPLIT, $0-0
MOV R4, R5
ADD R4, R5, R6
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if len(img.Funcs) != 1 {
t.Fatalf("got %d funcs, want 1", len(img.Funcs))
}
fn := img.Funcs[0]
if fn.Name != "simple" {
t.Errorf("func name: got %q, want %q", fn.Name, "simple")
}
//3 instructions ×4 bytes =12
if fn.Size != 12 {
t.Errorf("func size: got %d, want 12", fn.Size)
}
}
func TestArm64AssembleFileWithFrame(t *testing.T) {
src := `#include "textflag.h"
TEXT ·framed(SB), NOSPLIT, $16-8
MOVD arg+0(FP), R4
ADD $1, R4, R4
MOVD R4, ret+0(FP)
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if len(img.Funcs) != 1 {
t.Fatalf("got %d funcs, want 1", len(img.Funcs))
}
fn := img.Funcs[0]
if fn.Frame != 16 {
t.Errorf("frame: got %d, want 16", fn.Frame)
}
// Prologue (3×4=12) + body (3×4=12) + RET epilogue (3×4=12) = 36
if fn.Size != 36 {
t.Errorf("func size: got %d, want 36", fn.Size)
}
}
func TestArm64AssembleFileWithBranches(t *testing.T) {
src := `#include "textflag.h"
TEXT ·branch(SB), NOSPLIT, $0-0
BEQ done
BNE skip
skip:
ADD R4, R5
done:
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
fn := img.Funcs[0]
if fn.Size != 16 {
t.Errorf("func size: got %d, want 16", fn.Size)
}
}
func TestArm64AssembleFileWithJumpChain(t *testing.T) {
src := `#include "textflag.h"
TEXT ·chain(SB), NOSPLIT, $0-0
BNE skip
ADD R4, R5
RET
skip:
B target
target:
ADD R6, R7
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
// BNE should be redirected past skip→target to target directly.
if img.Funcs[0].Size != 24 {
t.Errorf("func size: got %d, want 24", img.Funcs[0].Size)
}
}
func TestArm64AssembleErrors(t *testing.T) {
tests := []struct {
name string
src string
}{
{"unsupported", "TEXT ·f(SB), NOSPLIT, $0-0\n\tINVALID\tR4, R5\n\tRET\n"},
{"undefined label", "TEXT ·f(SB), NOSPLIT, $0-0\n\tB\tnosuch\n\tRET\n"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
f, errs := parser.Parse("test_arm64.s", tt.src)
if len(errs) > 0 {
return // parse error, that's fine
}
_, err := AssembleFileARM64(f)
if err == nil {
t.Error("expected error, got nil")
}
})
}
}
func TestArm64Movcon(t *testing.T) {
tests := []struct {
v int64
want int
}{
{0, 0}, // 0 fits at shift 0
{1, 0}, // single bit at shift 0
{0x10000, 16}, // single bit at shift 16
{0x100000000, 32}, // single bit at shift 32
{0xFF, 0}, // 0xFF fits at shift 0
{0x12345, -1}, // multiple chunks, not movcon
}
for _, tt := range tests {
got := arm64Movcon(tt.v)
if got != tt.want {
t.Errorf("arm64Movcon(%#x) = %d, want %d", tt.v, got, tt.want)
}
}
}
func TestArm64RegClassOf(t *testing.T) {
if arm64RegClassOf("R4") != arm64ClsGR {
t.Error("R4 should be GR")
}
if arm64RegClassOf("F4") != arm64ClsFP {
t.Error("F4 should be FP")
}
if arm64RegClassOf("") != arm64ClsNone {
t.Error("empty should be None")
}
}
func TestArm64ResolvePseudo(t *testing.T) {
fi := arm64FrameInfo{autosize: 48, frame: 32}
// FP: offset = sym.Offset + autosize +8
base, off := arm64ResolvePseudo(&ast.Symbol{Pseudo: "FP", Offset: 0}, fi)
if base != 31 || off != 56 {
t.Errorf("FP: base=%d off=%d, want 31, 56", base, off)
}
// SP: offset = sym.Offset + frame +8
base, off = arm64ResolvePseudo(&ast.Symbol{Pseudo: "SP", Offset: -8}, fi)
if base != 31 || off != 32 {
t.Errorf("SP: base=%d off=%d, want 31, 32", base, off)
}
// SB: unresolved
base, _ = arm64ResolvePseudo(&ast.Symbol{Pseudo: "SB"}, fi)
if base != -1 {
t.Errorf("SB: base=%d, want -1", base)
}
}
// TestArm64FPSel tests FP conditional select encoding.
func TestArm64FPSel(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
FCSELD GE, F10, F11, F12
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
// FCSELD should be 4 bytes + RET 4 bytes = 8
if img.Funcs[0].Size != 8 {
t.Errorf("size: got %d, want 8", img.Funcs[0].Size)
}
}
// TestArm64FPCvt tests FP conversion encoding.
func TestArm64FPCvt(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
FCVTZSD F4, R0
SCVTFD R4, F8
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 12 {
t.Errorf("size: got %d, want 12", img.Funcs[0].Size)
}
}
// TestArm64CSEL tests conditional select encoding.
func TestArm64CSEL(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
CSEL EQ, R0, R1, R2
CSET NE, R3
CINC GE, R4, R5
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 16 {
t.Errorf("size: got %d, want 16", img.Funcs[0].Size)
}
}
// TestArm64CRC32 tests CRC32 encoding.
func TestArm64CRC32(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
CRC32B R0, R2
CRC32W R6, R8
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 12 {
t.Errorf("size: got %d, want 12", img.Funcs[0].Size)
}
}
// TestArm64Bitfield tests bitfield/shift encoding.
func TestArm64Bitfield(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
ASR $4, R0, R1
LSL $12, R4, R5
EXTR $8, R0, R1, R2
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 16 {
t.Errorf("size: got %d, want 16", img.Funcs[0].Size)
}
}
// TestArm64SIMD tests SIMD encoding (via the instruction table).
func TestArm64SIMD(t *testing.T) {
// Verify SIMD instructions are in the table.
for _, mnem := range []string{"VADD", "VSUB", "VMUL"} {
if _, ok := a64InstrTable[mnem]; !ok {
t.Errorf("%s not in instruction table", mnem)
}
}
}
// TestArm64LoadImm64 tests 64-bit immediate loading.
func TestArm64LoadImm64(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
MOVD $0x123456789ABCDEF0, R0
MOVD $0, R1
MOVD $1, R2
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
// $0x123456789ABCDEF0 needs 4 MOVZ/MOVK instructions (16 bytes)
// $0 is 1 instruction (4 bytes)
// $1 is 1 bitmask instruction (4 bytes)
// RET is 1 instruction (4 bytes)
if img.Funcs[0].Size != 28 {
t.Errorf("size: got %d, want 28", img.Funcs[0].Size)
}
}
// TestArm64BranchCond tests conditional branch encoding.
func TestArm64BranchCond(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
BEQ done
BNE done
BGE done
BLT done
ADD R4, R5
done:
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
// 4 branches + 1 ADD + 1 RET = 24 bytes
if img.Funcs[0].Size != 24 {
t.Errorf("size: got %d, want 24", img.Funcs[0].Size)
}
}
// TestArm64Errors tests error paths.
func TestArm64Errors(t *testing.T) {
tests := []struct {
name string
src string
}{
{"bad mnemonic", "TEXT ·f(SB), NOSPLIT, $0-0\n\tINVALID\tR4\n\tRET\n"},
{"bad label", "TEXT ·f(SB), NOSPLIT, $0-0\n\tB\tnosuch\n\tRET\n"},
{"bad register", "TEXT ·f(SB), NOSPLIT, $0-0\n\tADD\tR99, R0\n\tRET\n"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
f, errs := parser.Parse("test_arm64.s", tt.src)
if len(errs) > 0 {
return
}
_, err := AssembleFileARM64(f)
if err == nil {
t.Error("expected error, got nil")
}
})
}
}
// leWord reads a little-endian uint32 from b.
func leWord(b []byte) uint32 {
return uint32(b[0]) | uint32(b[1])<<8 | uint32(b[2])<<16 | uint32(b[3])<<24
}
// leWords reads all little-endian uint32s from b.
func leWords(b []byte) []uint32 {
n := len(b) / 4
w := make([]uint32, n)
for i := range w {
w[i] = leWord(b[i*4:])
}
return w
}
+237
View File
@@ -0,0 +1,237 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// arm64 frame mapping, matching the Go toolchain's arm64 backend.
//
// Go's arm64 functions use R29 as the frame pointer (FP) and R30 as the link
// register (LR). R31 is the stack pointer (SP). FP and SP in the source
// are synthetic pseudo-registers resolved against the hardware SP and the
// frame size.
//
// The autosize is the real stack adjustment: the declared local frame plus
// 8 bytes for the saved link register, rounded up to a 16-byte multiple.
// The toolchain adds an "extrasize" to align: if autosize%16 == 8, add 8;
// if autosize%16 == 0, add 16.
//
// Prologue (autosize > 0, small frame ≤ 0xf0):
//
// MOVD.W LR, -autosize(SP) // pre-index: SP -= autosize, store LR at SP
// MOVD FP, -8(SP) // store FP at SP-8
// SUB $8, SP, FP // FP = SP - 8
//
// Prologue (autosize > 0, large frame > 0xf0):
//
// SUB $autosize, SP, R20 // R20 = SP - autosize
// STP (FP, LR), -8(R20) // store FP,LR at R20-8
// MOVD R20, SP // SP = R20
// SUB $8, SP, FP // FP = SP - 8
//
// Epilogue (non-leaf, small frame):
//
// ADD $autosize-8, SP, FP // restore FP
// ADD $autosize, SP, SP // deallocate frame
// MOVD -8(SP), FP // (actually the reverse of prologue)
// Actually:
// MOVD -8(SP), FP // load FP from SP-8
// MOVD.P autosize(SP), LR // post-index: load LR, SP += autosize
//
// Epilogue (non-leaf, large frame):
// ADD $autosize-8, SP, FP
// ADD $autosize, SP, SP
// Actually:
// LDP -8(SP), (FP, LR) // load FP,LR
// ADD $autosize, SP, SP // deallocate frame
//
// Epilogue (leaf with frame):
// ADD $autosize-8, SP, FP
// ADD $autosize, SP, SP
//
// RET always emits as BR LR (0xd65f03c0).
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
type arm64FrameInfo struct {
autosize int // the real SP adjustment (locals + saved LR + alignment)
frame int // the declared $framesize
args int // the declared -argsize
noSplit bool // the NOSPLIT flag
leaf bool // no call instructions in the body
}
// arm64ComputeFrame derives the frame layout for a TEXT function.
func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
fi := arm64FrameInfo{
frame: frameSize(t),
args: argsSize(t),
}
for _, f := range t.Flags {
if f == "NOSPLIT" {
fi.noSplit = true
}
}
fi.leaf = arm64IsLeaf(t)
if fi.frame != 0 || !fi.leaf {
fi.autosize = fi.frame + 8 // space for the saved LR
if fi.autosize%16 != 0 {
// The toolchain aligns to 16: if autosize%16 == 8, add 8;
// otherwise add whatever is needed.
fi.autosize += 16 - (fi.autosize % 16)
}
}
return fi
}
// arm64IsLeaf reports whether a function contains no call instructions
// (BL/CALL), matching the toolchain's LEAF mark.
func arm64IsLeaf(t *ast.Text) bool {
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "BL", "CALL":
return false
}
}
return true
}
// arm64Prologue returns the prologue bytes for an arm64 function.
func arm64Prologue(fi arm64FrameInfo) []byte {
if fi.autosize == 0 {
return nil
}
if fi.autosize <= 0xf0 {
// Small frame: MOVD.W LR, -autosize(SP); MOVD FP, -8(SP); SUB $8, SP, FP
return a64WordsLE(
arm64PreStoreImm(3, 0, int32(-fi.autosize), 31, 30), // STR.W LR, -autosize(SP) (pre-index store)
arm64UnscaledStore(3, 0, -8, 31, 29), // STUR FP, [SP, #-8]
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
)
}
// Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP
return a64WordsLE(
a64AddSub(1, 1, 0, 0, uint32(fi.autosize), 31, 20), // SUB $autosize, SP, R20
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
)
}
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
// deallocate the frame when present) followed by RET (BR LR).
func arm64Return(fi arm64FrameInfo) []byte {
var ws []uint32
if fi.autosize != 0 {
if fi.leaf {
// Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP
ws = append(ws,
a64AddSub(1, 0, 0, 0, uint32(fi.autosize-8), 31, 29), // ADD $autosize-8, SP, FP
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
)
} else if fi.autosize <= 0xf0 {
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize
ws = append(ws,
arm64UnscaledLoad(3, 0, -8, 31, 29), // LDR FP, [SP, #-8]
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
)
} else {
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
ws = append(ws,
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
)
}
}
// RET: BR LR (0xd65f03c0)
ws = append(ws, a64UncondBranch(2, 30, 0)) // opc=2(RET), Rn=LR(30), Rd=0
return a64WordsLE(ws...)
}
// arm64PrologueSpadjPC returns the function-relative byte offset where the
// prologue has finished decrementing SP (the delta becomes autosize).
func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
if fi.autosize == 0 {
return 0
}
if fi.autosize <= 0xf0 {
return 4 // MOVD.W instruction decrements SP
}
return 8 // SUB + STP + MOVD (3 instructions, SP updated at the MOVD)
}
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final RET instruction.
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
if fi.autosize == 0 {
return 0
}
if fi.leaf {
return 8 // ADD + ADD
}
if fi.autosize <= 0xf0 {
return 8 // LDR + LDR.P
}
return 8 // LDP + ADD
}
// arm64ResolvePseudo translates a pseudo-register memory reference into a
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
// x+N(SP) → (N + frame + 8)(SP). Returns base = -1 for an unresolvable
// reference (SB: static data, handled by the relocation path).
//
// The Go toolchain resolves all pseudo-register references against the
// hardware stack pointer (R31/SP): FP references add autosize+8 (the
// distance from SP after the prologue to the caller's argument area),
// SP references add frame+8 (the distance to the local area).
func arm64ResolvePseudo(sym *ast.Symbol, fi arm64FrameInfo) (base int, off int32) {
if sym == nil {
return -1, 0
}
switch sym.Pseudo {
case "FP":
return 31, int32(sym.Offset) + int32(fi.autosize) + 8
case "SP":
return 31, int32(sym.Offset) + int32(fi.frame) + 8
case "SB":
return -1, int32(sym.Offset)
}
return -1, 0
}
// arm64PreStoreImm encodes a pre-index store (STR with writeback):
// size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func arm64PreStoreImm(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 |
3<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// arm64UnscaledStore encodes an unscaled store (STUR):
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt.
func arm64UnscaledStore(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 |
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// arm64UnscaledLoad encodes an unscaled load (LDUR):
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt.
func arm64UnscaledLoad(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// arm64PostLoad encodes a post-index load (LDR with post-increment):
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
+224
View File
@@ -0,0 +1,224 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// AArch64 ELF64 relocatable object emission.
const (
emAARCH64 = 183 // EM_AARCH64
// AArch64 relocation types (the ELF psABI).
rArm64PrelPgHi21 = 275 // R_AARCH64_ADR_PREL_PG_HI21 (ADRP page)
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD/STR/LDR page offset)
)
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
// optional .rela.text.
func (img *Image) ELFAARCH64Object() ([]byte, error) {
le := binary.LittleEndian
const (
secText = 1
secData = 2
)
// Build symbol table.
var locals, globals []elfSym
for _, fn := range img.Funcs {
s := elfSym{
name: objectName(fn.Pkg, fn.Name),
info: sttFunc,
shndx: secText,
value: uint64(fn.Offset),
size: uint64(fn.Size),
}
if fn.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := elfSym{
name: objectName(d.Pkg, d.Name),
info: sttObject,
shndx: secData,
value: uint64(d.Offset),
size: uint64(d.Size),
}
if d.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, name := range img.Externals {
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
}
syms := []elfSym{
{},
{name: ".text", info: sttSection, shndx: secText},
{name: ".data", info: sttSection, shndx: secData},
}
syms = append(syms, locals...)
shInfo := len(syms)
syms = append(syms, globals...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Build relocations. Each SB reference is an ADRP pair:
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21
// ADD/LDR/STR → R_AARCH64_ADD_ABS_LO12_NC
type elfRela struct {
off uint64
typ uint32
sym int
addend int64
}
var relas []elfRela
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
typ := uint32(rArm64PrelPgHi21)
if r.Kind == RelArm64Addr && r.Off%4 == 4 {
// The second instruction in an ADRP pair uses ADD_ABS_LO12_NC.
typ = rArm64AddAbsLo12NC
}
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
typ: typ,
sym: idx,
addend: r.Addend - int64(r.After-r.Off),
})
}
}
// String tables.
stNames := newElfStrtab()
for _, s := range syms {
stNames.add(s.name)
}
stSections := newElfStrtab()
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout.
var out []byte
out = append(out, make([]byte, 64)...)
align := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
align(16)
textOff := len(out)
out = append(out, img.Code...)
align(16)
dataOff := len(out)
out = append(out, img.Data...)
align(8)
symtabOff := len(out)
for _, s := range syms {
var b [24]byte
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
b[4] = s.info
b[5] = 0
le.PutUint16(b[6:], s.shndx)
le.PutUint64(b[8:], s.value)
le.PutUint64(b[16:], s.size)
out = append(out, b[:]...)
}
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
if hasRela {
align(8)
relaOff = len(out)
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
align(8)
shoff := len(out)
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
var b [64]byte
le.PutUint32(b[0:], uint32(stSections.at(name)))
le.PutUint32(b[4:], uint32(typ))
le.PutUint64(b[8:], flags)
le.PutUint64(b[16:], 0)
le.PutUint64(b[24:], uint64(off))
le.PutUint64(b[32:], uint64(size))
le.PutUint32(b[40:], uint32(link))
le.PutUint32(b[44:], uint32(info))
le.PutUint64(b[48:], alignV)
le.PutUint64(b[56:], entsize)
out = append(out, b[:]...)
}
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// ELF header.
hdr := out[:64]
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
le.PutUint16(hdr[16:], etREL)
le.PutUint16(hdr[18:], emAARCH64)
le.PutUint32(hdr[20:], elfVersion)
le.PutUint64(hdr[24:], 0)
le.PutUint64(hdr[32:], 0)
le.PutUint64(hdr[40:], uint64(shoff))
le.PutUint32(hdr[48:], 0)
le.PutUint16(hdr[52:], 64)
le.PutUint16(hdr[54:], 0)
le.PutUint16(hdr[56:], 0)
le.PutUint16(hdr[58:], 64)
le.PutUint16(hdr[60:], uint16(nSections))
le.PutUint16(hdr[62:], uint16(secShstr))
return out, nil
}
+142
View File
@@ -0,0 +1,142 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
// relocatable object: sections, the symbol table (bindings, types, values,
// sizes) and the .rela.text relocation pair for the static-symbol load,
// parsed back with debug/elf.
func TestELFAARCH64Object(t *testing.T) {
f, errs := parser.Parse("k_arm64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
TEXT ·getanswer(SB), NOSPLIT, $0-8
MOVD answer<>(SB), R4
MOVD R4, ret+0(FP)
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.ELFAARCH64Object()
if err != nil {
t.Fatalf("ELFAARCH64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_AARCH64 {
t.Errorf("type/machine = %v/%v, want ET_REL/EM_AARCH64", ef.Type, ef.Machine)
}
text := ef.Section(".text")
data := ef.Section(".data")
if text == nil || data == nil {
t.Fatal("missing .text or .data section")
}
if text.Size == 0 {
t.Error(".text section is empty")
}
syms, err := ef.Symbols()
if err != nil {
t.Fatalf("symbols: %v", err)
}
foundAdd, foundGetanswer, foundAnswer := false, false, false
for _, s := range syms {
switch s.Name {
case "add":
foundAdd = true
if elf.SymType(s.Info&0xf) != elf.STT_FUNC || elf.SymBind(s.Info>>4) != elf.STB_GLOBAL {
t.Errorf("add: info=0x%02x, want STT_FUNC|STB_GLOBAL", s.Info)
}
case "getanswer":
foundGetanswer = true
if elf.SymType(s.Info&0xf) != elf.STT_FUNC || elf.SymBind(s.Info>>4) != elf.STB_GLOBAL {
t.Errorf("getanswer: info=0x%02x, want STT_FUNC|STB_GLOBAL", s.Info)
}
case "answer":
foundAnswer = true
if elf.SymType(s.Info&0xf) != elf.STT_OBJECT || elf.SymBind(s.Info>>4) != elf.STB_LOCAL {
t.Errorf("answer: info=0x%02x, want STT_OBJECT|STB_LOCAL", s.Info)
}
}
}
if !foundAdd {
t.Error("symbol 'add' not found")
}
if !foundGetanswer {
t.Error("symbol 'getanswer' not found")
}
if !foundAnswer {
t.Error("symbol 'answer' not found")
}
// Check that .rela.text exists (getanswer has SB reference).
relaText := ef.Section(".rela.text")
if relaText == nil {
t.Error("missing .rela.text section")
}
}
// TestELFAARCH64ObjectNoRelocations checks the ELF output when there are no
// static-symbol references (no .rela.text section).
func TestELFAARCH64ObjectNoRelocations(t *testing.T) {
f, errs := parser.Parse("k_arm64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.ELFAARCH64Object()
if err != nil {
t.Fatalf("ELFAARCH64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Section(".rela.text") != nil {
t.Error("unexpected .rela.text section when there are no relocations")
}
}
+223
View File
@@ -0,0 +1,223 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// LoongArch ELF64 relocatable object emission.
const (
emLOONGARCH = 258 // EM_LOONGARCH
// LoongArch relocation types (the ELF psABI).
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
)
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
// optional .rela.text.
func (img *Image) ELFLOONG64Object() ([]byte, error) {
le := binary.LittleEndian
const (
secText = 1
secData = 2
)
// Build symbol table.
var locals, globals []elfSym
for _, fn := range img.Funcs {
s := elfSym{
name: objectName(fn.Pkg, fn.Name),
info: sttFunc,
shndx: secText,
value: uint64(fn.Offset),
size: uint64(fn.Size),
}
if fn.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := elfSym{
name: objectName(d.Pkg, d.Name),
info: sttObject,
shndx: secData,
value: uint64(d.Offset),
size: uint64(d.Size),
}
if d.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, name := range img.Externals {
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
}
syms := []elfSym{
{},
{name: ".text", info: sttSection, shndx: secText},
{name: ".data", info: sttSection, shndx: secData},
}
syms = append(syms, locals...)
shInfo := len(syms)
syms = append(syms, globals...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Build relocations. Each SB reference is a pcalau12i pair:
// pcalau12i rd, 0 → R_LARCH_PCALA_HI20
// addi.d/ld/st → R_LARCH_PCALA_LO12
type elfRela struct {
off uint64
typ uint32
sym int
addend int64
}
var relas []elfRela
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
typ := uint32(rLarchPCALAHI20)
if r.Kind == RelLoong64AddrLo {
typ = rLarchPCALALO12
}
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
typ: typ,
sym: idx,
addend: r.Addend - int64(r.After-r.Off),
})
}
}
// String tables.
stNames := newElfStrtab()
for _, s := range syms {
stNames.add(s.name)
}
stSections := newElfStrtab()
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout.
var out []byte
out = append(out, make([]byte, 64)...)
align := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
align(16)
textOff := len(out)
out = append(out, img.Code...)
align(16)
dataOff := len(out)
out = append(out, img.Data...)
align(8)
symtabOff := len(out)
for _, s := range syms {
var b [24]byte
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
b[4] = s.info
b[5] = 0
le.PutUint16(b[6:], s.shndx)
le.PutUint64(b[8:], s.value)
le.PutUint64(b[16:], s.size)
out = append(out, b[:]...)
}
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
if hasRela {
align(8)
relaOff = len(out)
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
align(8)
shoff := len(out)
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
var b [64]byte
le.PutUint32(b[0:], uint32(stSections.at(name)))
le.PutUint32(b[4:], uint32(typ))
le.PutUint64(b[8:], flags)
le.PutUint64(b[16:], 0)
le.PutUint64(b[24:], uint64(off))
le.PutUint64(b[32:], uint64(size))
le.PutUint32(b[40:], uint32(link))
le.PutUint32(b[44:], uint32(info))
le.PutUint64(b[48:], alignV)
le.PutUint64(b[56:], entsize)
out = append(out, b[:]...)
}
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// ELF header.
hdr := out[:64]
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
le.PutUint16(hdr[16:], etREL)
le.PutUint16(hdr[18:], emLOONGARCH)
le.PutUint32(hdr[20:], elfVersion)
le.PutUint64(hdr[24:], 0)
le.PutUint64(hdr[32:], 0)
le.PutUint64(hdr[40:], uint64(shoff))
le.PutUint32(hdr[48:], 0)
le.PutUint16(hdr[52:], 64)
le.PutUint16(hdr[54:], 0)
le.PutUint16(hdr[56:], 0)
le.PutUint16(hdr[58:], 64)
le.PutUint16(hdr[60:], uint16(nSections))
le.PutUint16(hdr[62:], uint16(secShstr))
return out, nil
}
+200
View File
@@ -0,0 +1,200 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
// relocatable object: sections, the symbol table (bindings, types, values,
// sizes) and the .rela.text relocation pair for the static-symbol load,
// parsed back with debug/elf.
func TestELFLOONG64Object(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
TEXT ·getanswer(SB), NOSPLIT, $0-8
MOVV answer<>(SB), R4
MOVV R4, ret+0(FP)
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.ELFLOONG64Object()
if err != nil {
t.Fatalf("ELFLOONG64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_LOONGARCH {
t.Errorf("type/machine = %v/%v, want ET_REL/EM_LOONGARCH", ef.Type, ef.Machine)
}
text := ef.Section(".text")
data := ef.Section(".data")
if text == nil || data == nil {
t.Fatal("missing .text or .data section")
}
if text.Flags&elf.SHF_EXECINSTR == 0 || text.Flags&elf.SHF_ALLOC == 0 {
t.Errorf(".text flags = %v", text.Flags)
}
if data.Flags&elf.SHF_WRITE == 0 {
t.Errorf(".data flags = %v", data.Flags)
}
textData, err := text.Data()
if err != nil {
t.Fatal(err)
}
if !bytes.Equal(textData, img.Code) {
t.Errorf(".text contents differ from the image code")
}
dataData, err := data.Data()
if err != nil {
t.Fatal(err)
}
syms, err := ef.Symbols()
if err != nil {
t.Fatalf("symbols: %v", err)
}
byName := map[string]elf.Symbol{}
for _, s := range syms {
byName[s.Name] = s
}
wantSym := func(name string, bind elf.SymBind, typ elf.SymType, section elf.SectionIndex, size uint64) {
t.Helper()
s, ok := byName[name]
if !ok {
t.Errorf("symbol %q not found", name)
return
}
if elf.ST_BIND(s.Info) != bind || elf.ST_TYPE(s.Info) != typ {
t.Errorf("%s: bind/type = %v/%v, want %v/%v", name, elf.ST_BIND(s.Info), elf.ST_TYPE(s.Info), bind, typ)
}
if s.Section != section {
t.Errorf("%s: section = %v, want %v", name, s.Section, section)
}
if s.Size != size {
t.Errorf("%s: size = %d, want %d", name, s.Size, size)
}
}
if ef.Sections[1].Name != ".text" || ef.Sections[2].Name != ".data" {
t.Fatalf("section layout = %s, %s; want .text, .data", ef.Sections[1].Name, ef.Sections[2].Name)
}
textIdx := elf.SectionIndex(1)
dataIdx := elf.SectionIndex(2)
wantSym("add", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 20)
wantSym("getanswer", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 16)
wantSym("answer", elf.STB_LOCAL, elf.STT_OBJECT, dataIdx, 8)
// The data section carries 16-byte alignment padding; the answer
// symbol sits at its padded offset.
ans := byName["answer"]
if ans.Value+8 > uint64(len(dataData)) {
t.Fatalf("answer value %d outside .data (%d bytes)", ans.Value, len(dataData))
}
if got := dataData[ans.Value : ans.Value+8]; !bytes.Equal(got, []byte{42, 0, 0, 0, 0, 0, 0, 0}) {
t.Errorf("answer data = % x, want $42", got)
}
// Relocations: the static-symbol load is a pcalau12i+ld.d pair, so one
// R_LARCH_PCALA_HI20 and one R_LARCH_PCALA_LO12, both against the local
// data symbol. debug/elf does not surface rela entries, so read the
// section directly.
relaSec := ef.Section(".rela.text")
if relaSec == nil {
t.Fatal("missing .rela.text")
}
raw, err := relaSec.Data()
if err != nil {
t.Fatal(err)
}
if len(raw)%24 != 0 || len(raw)/24 != 2 {
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
}
le := binary.LittleEndian
for i := 0; i < 2; i++ {
e := raw[i*24 : (i+1)*24]
off := le.Uint64(e[0:])
info := le.Uint64(e[8:])
typ := info & 0xffffffff
sym := int(info >> 32)
if i == 0 && (typ != uint64(elf.R_LARCH_PCALA_HI20) || off != 20) {
t.Errorf("reloc %d: type %d off %d, want R_LARCH_PCALA_HI20 at 20", i, typ, off)
}
if i == 1 && (typ != uint64(elf.R_LARCH_PCALA_LO12) || off != 24) {
t.Errorf("reloc %d: type %d off %d, want R_LARCH_PCALA_LO12 at 24", i, typ, off)
}
if sym != 3 { // NULL, .text, .data, then the first local: answer
t.Errorf("reloc %d: symbol index %d, want 3 (answer)", i, sym)
}
}
}
// TestELFLOONG64ObjectNoRelocations checks a file with no static-symbol
// references emits a valid object without a .rela.text section.
func TestELFLOONG64ObjectNoRelocations(t *testing.T) {
f, errs := parser.Parse("n_loong64.s", `
#include "textflag.h"
TEXT ·nop(SB), NOSPLIT, $0
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.ELFLOONG64Object()
if err != nil {
t.Fatalf("ELFLOONG64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Section(".rela.text") != nil {
t.Error("unexpected .rela.text section")
}
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
found := false
for _, s := range syms {
if s.Name == "nop" && elf.ST_TYPE(s.Info) == elf.STT_FUNC {
found = true
}
}
if !found {
t.Error("function symbol nop not found")
}
}
+22 -18
View File
@@ -15,6 +15,7 @@ const (
// RISC-V relocation types. // RISC-V relocation types.
rRISCV32 = 1 rRISCV32 = 1
rRISCVJAL = 17 // R_RISCV_JAL
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20 rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
@@ -79,11 +80,12 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
symIdx[s.name] = i symIdx[s.name] = i
} }
// Build relocations. Each SB reference produces a pair: // Build relocations. Each SB reference is an AUIPC + second-instruction
// AUIPC rd, 0 → R_RISCV_PCREL_HI20 // pair carrying a single relocation kind; the ELF writer expands it into
// ADDI/LD/SD → R_RISCV_PCREL_LO12_I or _S // the R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I/S pair the psABI expects.
// For now we record them as individual entries; at link time // The HI20 carries the symbol addend; the LO12 addend is zero, matching
// the linker must pair HI20 with its matching LO12. // cmd/link's own ELF conversion (the LO12 resolves against the HI20's
// AUIPC location).
type elfRela struct { type elfRela struct {
off uint64 off uint64
typ uint32 typ uint32
@@ -97,22 +99,24 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
if !ok { if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name) return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
} }
// Determine relocation type from the relocation kind.
typ := uint32(rRISCVPCRELHI20) // default: AUIPC
switch r.Kind { switch r.Kind {
case RelPCRelLO12: case RelRISCVPCRELIType:
typ = rRISCVPCRELLO12I relas = append(relas,
case RelPCRelLO12S: elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
typ = rRISCVPCRELLO12S elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: idx, addend: 0},
)
case RelRISCVPCRELSType:
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: idx, addend: 0},
)
case RelRISCVJal:
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVJAL, sym: idx, addend: r.Addend})
case RelPCRelAbs: case RelPCRelAbs:
typ = rRISCV32 relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCV32, sym: idx, addend: r.Addend})
default:
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
} }
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
typ: typ,
sym: idx,
addend: r.Addend - int64(r.After-r.Off),
})
} }
} }
+253 -86
View File
@@ -10,6 +10,7 @@ import (
"os" "os"
"os/exec" "os/exec"
"path/filepath" "path/filepath"
"strings"
"sync" "sync"
) )
@@ -22,9 +23,15 @@ import (
// //
// The object carries what the linker requires of an assembly object: the // The object carries what the linker requires of an assembly object: the
// functions (non-package symbols, as cmd/asm emits them), the GLOBL data, // functions (non-package symbols, as cmd/asm emits them), the GLOBL data,
// one FuncInfo per function, and the pc-value tables (pcsp, pcfile, // one FuncInfo per function, the per-function DWARF symbols (the
// pcline, pcinline). DWARF and the implicit funcdata symbols are omitted; // .debug_line program and the subprogram DIE, which the linker's DWARF
// the linker fills their defaults. // pass reads verbatim), and the pc-value tables (pcsp, pcfile, pcline,
// pcinline). The implicit funcdata symbols are omitted; the linker fills
// their defaults.
//
// emitGOObject is architecture-agnostic; the per-architecture GOObject*
// methods supply the toolchain preamble, the MinLC (pc-value delta unit)
// and the relocation-type mapping for code relocations.
// GOOBJ block indices (cmd/internal/goobj). // GOOBJ block indices (cmd/internal/goobj).
const ( const (
@@ -51,9 +58,11 @@ const (
// Symbol kinds used by assembly objects (cmd/internal/objabi). // Symbol kinds used by assembly objects (cmd/internal/objabi).
const ( const (
kindSTEXT = 1 kindSTEXT = 1
kindSRODATA = 3 kindSRODATA = 3
kindSDATA = 7 kindSDATA = 7
kindSDWARFFCN = 14
kindSDWARFLINES = 20
) )
// Symbol flags (cmd/internal/goobj). // Symbol flags (cmd/internal/goobj).
@@ -66,11 +75,13 @@ const (
// Aux entry types (cmd/internal/goobj). // Aux entry types (cmd/internal/goobj).
const ( const (
auxFuncInfo = 1 auxFuncInfo = 1
auxPcsp = 7 auxDwarfInfo = 3
auxPcfile = 8 auxDwarfLines = 6
auxPcline = 9 auxPcsp = 7
auxPcinline = 10 auxPcfile = 8
auxPcline = 9
auxPcinline = 10
) )
// FuncInfo flags (internal/abi). // FuncInfo flags (internal/abi).
@@ -80,7 +91,49 @@ const (
) )
// Relocation types (cmd/internal/objabi). // Relocation types (cmd/internal/objabi).
const relocPCRel = 14 // R_PCREL and R_ADDR are stable across Go versions.
const (
relocPCRel = 14 // R_PCREL
relocAddr = 1 // R_ADDR
)
// relocDWTXTADDRU4 returns the R_DWTXTADDR_U4 relocation type for the
// installed Go toolchain. The value shifted between Go 1.26 (103) and
// Go 1.27 (106) because new LoongArch relocations were inserted before it.
func relocDWTXTADDRU4() uint16 {
if isGo127OrLater() {
return 106
}
return 103
}
var (
goVersionOnce sync.Once
goVersionGT26 bool
)
// isGo127OrLater reports whether the installed Go toolchain is 1.27 or later.
func isGo127OrLater() bool {
goVersionOnce.Do(func() {
goBin, err := exec.LookPath("go")
if err != nil {
return
}
out, err := exec.Command(goBin, "version").Output()
if err != nil {
return
}
// "go version go1.27rc1 linux/amd64"
s := string(out)
for _, prefix := range []string{"go version go1.27", "go version go1.28", "go version go1.29", "go version go2."} {
if strings.Contains(s, prefix) {
goVersionGT26 = true
return
}
}
})
return goVersionGT26
}
// Special package indices for symbol references. // Special package indices for symbol references.
const ( const (
@@ -110,6 +163,13 @@ func (s goSym) append(b []byte, strOff map[string]uint32) []byte {
return binary.LittleEndian.AppendUint32(b, s.align) return binary.LittleEndian.AppendUint32(b, s.align)
} }
// dwarfRelocSet attaches emitter-generated relocations (the DWARF
// lines/info symbols' address references) to a definition index.
type dwarfRelocSet struct {
si int
relocs []goobjReloc
}
// GOObject returns the image as a GOOBJ object file for the given package // GOObject returns the image as a GOOBJ object file for the given package
// path (the linker qualifies the exported symbols with it, the way cmd/asm // path (the linker qualifies the exported symbols with it, the way cmd/asm
// does with its -p flag). srcPath names the source file recorded in the // does with its -p flag). srcPath names the source file recorded in the
@@ -117,19 +177,90 @@ func (s goSym) append(b []byte, strOff map[string]uint32) []byte {
// captured from the installed go tool asm, so the output links with the // captured from the installed go tool asm, so the output links with the
// toolchain it was produced on — exactly like a real assembly object. // toolchain it was produced on — exactly like a real assembly object.
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) { func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
if pkgPath == "" {
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
}
pre, err := toolchainObjectPreamble() pre, err := toolchainObjectPreamble()
if err != nil { if err != nil {
return nil, err return nil, err
} }
// amd64: MinLC 1, R_PCREL for the code relocations.
return img.emitGOObject(pkgPath, srcPath, pre, 1, func(Reloc) (uint16, uint8) { return relocPCRel, 4 })
}
// The symbol tables. Package definitions: the GLOBL symbols, then one // emitGOObject assembles the GOOBJ payload for any architecture. pre is
// anonymous FuncInfo symbol per function. Non-package definitions: the // the toolchain's object preamble; minLC is the architecture's minimum
// pc-value tables and the functions themselves, as cmd/asm lays them // instruction length, the unit of the pc-value table deltas; relocField
// out. defIdx maps a GLOBL's bare name to its definition index for the // maps a code relocation to its objabi relocation type and the width of
// relocations; fnNpIdx maps a function to its non-package index. // the instruction field the linker writes.
func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, relocField func(Reloc) (uint16, uint8)) ([]byte, error) {
if pkgPath == "" {
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
}
// The non-package definitions first — the DWARF symbols reference the
// functions by these indices: per function the four pc-value tables
// and the function itself, as cmd/asm lays them out.
type npSym struct {
sym goSym
data []byte
}
var nps []npSym
type pcRefs struct{ sp, file, line, inl int }
pcIdx := make([]pcRefs, len(img.Funcs))
fnNpIdx := make([]int, len(img.Funcs))
for i, fn := range img.Funcs {
tables := []struct {
data []byte
dst *int
}{
{pcspTable(fn, minLC), &pcIdx[i].sp},
{pcValueFlat(0, fn.Size, minLC), &pcIdx[i].file},
{pcValueFlat(int32(fn.Line), fn.Size, minLC), &pcIdx[i].line},
{pcValueFlat(-1, fn.Size, minLC), &pcIdx[i].inl},
}
for _, t := range tables {
*t.dst = len(nps)
nps = append(nps, npSym{
sym: goSym{typ: kindSRODATA, size: uint32(len(t.data)), align: 1},
data: t.data,
})
}
name := fn.Name
abi := uint16(0)
if fn.Static {
abi = symABIStatic
} else {
name = pkgPath + "." + name
}
flag := uint8(0)
if fn.NoSplit {
flag |= symFlagNoSplit
}
fnNpIdx[i] = len(nps)
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
// Only the amd64 encoder resolves file-local static symbols
// into a disp32 field at assemble time; GOOBJ must leave that
// field zero for the linker to fill. The RISC-V and LoongArch
// encoders emit zero immediates with a relocation instead, and
// their relocations cover whole AUIPC/pcalau12i pairs, so
// zeroing r.Off would erase the opcode/register bits the linker
// preserves when it patches only the immediate.
if r.Kind != RelPCRel32 {
continue
}
if r.Off >= 0 && r.Off+4 <= len(code) {
code[r.Off], code[r.Off+1], code[r.Off+2], code[r.Off+3] = 0, 0, 0, 0
}
}
nps = append(nps, npSym{
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
data: code,
})
}
// The package definitions: the GLOBL symbols, then, per function, the
// FuncInfo and the two DWARF symbols (the .debug_line program and the
// subprogram DIE). defIdx maps a GLOBL's bare name to its definition
// index for the code relocations.
var defs []goSym var defs []goSym
var defData [][]byte var defData [][]byte
defIdx := map[string]int{} defIdx := map[string]int{}
@@ -155,74 +286,82 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size]) defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
} }
fnFiIdx := make([]int, len(img.Funcs)) fnFiIdx := make([]int, len(img.Funcs))
for i := range img.Funcs { fnLinesIdx := make([]int, len(img.Funcs))
data := marshalFuncInfo(img.Funcs[i]) fnDIEIdx := make([]int, len(img.Funcs))
var dwarfRelocs []dwarfRelocSet
for i, fn := range img.Funcs {
data := marshalFuncInfo(fn)
fnFiIdx[i] = len(defs) fnFiIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))}) defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))})
defData = append(defData, data) defData = append(defData, data)
}
type npSym struct {
sym goSym
data []byte
}
var nps []npSym
type pcRefs struct{ sp, file, line, inl int }
pcIdx := make([]pcRefs, len(img.Funcs))
fnNpIdx := make([]int, len(img.Funcs))
for i, fn := range img.Funcs {
tables := []struct {
data []byte
dst *int
}{
{pcspTable(fn), &pcIdx[i].sp},
{pcValueFlat(0, fn.Size), &pcIdx[i].file},
{pcValueFlat(int32(fn.Line), fn.Size), &pcIdx[i].line},
{pcValueFlat(-1, fn.Size), &pcIdx[i].inl},
}
for _, t := range tables {
*t.dst = len(nps)
nps = append(nps, npSym{
sym: goSym{typ: kindSRODATA, size: uint32(len(t.data)), align: 1},
data: t.data,
})
}
name := fn.Name name := fn.Name
abi := uint16(0) if !fn.Static {
if fn.Static {
abi = symABIStatic
} else {
name = pkgPath + "." + name name = pkgPath + "." + name
} }
flag := uint8(0)
if fn.NoSplit { // The DWARF symbols: the .debug_line state-machine program and the
flag |= symFlagNoSplit // subprogram DIE, both referencing the function by its non-package
// index (package definitions, like cmd/asm's).
lines, lrel := goobjDwarfLines(fn, fnNpIdx[i])
fnLinesIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDWARFLINES, size: uint32(len(lines))})
defData = append(defData, lines)
die, drel := goobjDwarfInfo(fn, name, fnNpIdx[i])
fnDIEIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDWARFFCN, size: uint32(len(die))})
defData = append(defData, die)
dwarfRelocs = append(dwarfRelocs,
dwarfRelocSet{si: fnLinesIdx[i], relocs: lrel},
dwarfRelocSet{si: fnDIEIdx[i], relocs: drel},
)
}
// Resolve external symbol references (cross-package). Build the
// package index table and determine each external symbol's SymIdx
// by reading the target package's export data.
var extPkgTable []string
var extPkgIdx map[string]int
var extSymIdx map[string]int
if len(img.Externals) > 0 {
var err error
extPkgTable, extPkgIdx, extSymIdx, err = resolveExternalSymbols(img.Externals)
if err != nil {
return nil, fmt.Errorf("GOOBJ emission: resolving external symbols: %w", err)
} }
fnNpIdx[i] = len(nps)
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
// The linker writes the resolved displacement into the field;
// leave it zero, as cmd/asm's object does.
if r.Off >= 0 && r.Off+4 <= len(code) {
code[r.Off], code[r.Off+1], code[r.Off+2], code[r.Off+3] = 0, 0, 0, 0
}
}
nps = append(nps, npSym{
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
data: code,
})
} }
// Relocations, per defined symbol in definition order (package defs, // Relocations, per defined symbol in definition order (package defs,
// then non-package defs). Only file-local GLOBL references resolve; // then non-package defs).
// external symbols need the import machinery of a later increment.
nsyms := len(defs) + len(nps) nsyms := len(defs) + len(nps)
symRelocs := make([][]byte, nsyms) // flat 23-byte records symRelocs := make([][]byte, nsyms) // flat 23-byte records
for i, fn := range img.Funcs { for i, fn := range img.Funcs {
si := len(defs) + fnNpIdx[i] si := len(defs) + fnNpIdx[i]
for _, r := range fn.Relocs { for _, r := range fn.Relocs {
typ, size := relocField(r)
if r.External { if r.External {
return nil, fmt.Errorf("GOOBJ emission: external symbol %q is not supported yet", r.Name) // Split package-qualified name: "runtime·morestack" → runtime, morestack.
pkg, name := splitQualified(r.Name)
if pkg == "" {
return nil, fmt.Errorf("GOOBJ emission: external symbol %q has no package prefix", r.Name)
}
pIdx, ok := extPkgIdx[pkg]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: package %q not resolved", pkg)
}
sIdx, ok := extSymIdx[pkg+"·"+name]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: symbol %s·%s not resolved", pkg, name)
}
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size // field width
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], uint32(pIdx))
binary.LittleEndian.PutUint32(rec[19:], uint32(sIdx))
symRelocs[si] = append(symRelocs[si], rec[:]...)
continue
} }
di, ok := defIdx[r.Name] di, ok := defIdx[r.Name]
if !ok { if !ok {
@@ -230,17 +369,30 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
} }
var rec [23]byte var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off))) binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = 4 // field width rec[4] = size // field width
binary.LittleEndian.PutUint16(rec[5:], relocPCRel) binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend)) binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf) binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
binary.LittleEndian.PutUint32(rec[19:], uint32(di)) binary.LittleEndian.PutUint32(rec[19:], uint32(di))
symRelocs[si] = append(symRelocs[si], rec[:]...) symRelocs[si] = append(symRelocs[si], rec[:]...)
} }
} }
// The DWARF symbols' own relocations (the function address references).
for _, ds := range dwarfRelocs {
for _, r := range ds.relocs {
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(r.off))
rec[4] = r.siz
binary.LittleEndian.PutUint16(rec[5:], r.typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.add))
binary.LittleEndian.PutUint32(rec[15:], r.pkg)
binary.LittleEndian.PutUint32(rec[19:], r.sym)
symRelocs[ds.si] = append(symRelocs[ds.si], rec[:]...)
}
}
// Aux entries per function: FuncInfo, then the four pc tables. // Aux entries per function: FuncInfo, the DWARF symbols, then the four
// References into the non-package table use pkgIdxNone. // pc tables. References into the non-package table use pkgIdxNone.
symAux := make([][]byte, nsyms) symAux := make([][]byte, nsyms)
for i := range img.Funcs { for i := range img.Funcs {
si := len(defs) + fnNpIdx[i] si := len(defs) + fnNpIdx[i]
@@ -252,10 +404,14 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
symAux[si] = append(symAux[si], rec[:]...) symAux[si] = append(symAux[si], rec[:]...)
} }
aux(auxFuncInfo, pkgIdxSelf, uint32(fnFiIdx[i])) aux(auxFuncInfo, pkgIdxSelf, uint32(fnFiIdx[i]))
aux(auxPcsp, pkgIdxNone, uint32(len(defs)+pcIdx[i].sp)) aux(auxDwarfInfo, pkgIdxSelf, uint32(fnDIEIdx[i]))
aux(auxPcfile, pkgIdxNone, uint32(len(defs)+pcIdx[i].file)) aux(auxDwarfLines, pkgIdxSelf, uint32(fnLinesIdx[i]))
aux(auxPcline, pkgIdxNone, uint32(len(defs)+pcIdx[i].line)) // The pc-table references are 0-based within the non-package
aux(auxPcinline, pkgIdxNone, uint32(len(defs)+pcIdx[i].inl)) // definitions; the loader adds the package-definition count itself.
aux(auxPcsp, pkgIdxNone, uint32(pcIdx[i].sp))
aux(auxPcfile, pkgIdxNone, uint32(pcIdx[i].file))
aux(auxPcline, pkgIdxNone, uint32(pcIdx[i].line))
aux(auxPcinline, pkgIdxNone, uint32(pcIdx[i].inl))
} }
// The string table. Absolute offsets: it starts right after the // The string table. Absolute offsets: it starts right after the
@@ -291,7 +447,16 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
for _, s := range nps { for _, s := range nps {
npdefBlk = s.sym.append(npdefBlk, strOff) npdefBlk = s.sym.append(npdefBlk, strOff)
} }
pkgIdxBlk := stringRef(nil, "") // index 0: the dummy invalid package
// Package index table: index 0 is the dummy invalid package.
// External packages follow, in pkgIdx order.
for _, pkg := range extPkgTable {
addStr(pkg)
}
pkgIdxBlk := stringRef(nil, "") // index 0: dummy
for _, pkg := range extPkgTable {
pkgIdxBlk = stringRef(pkgIdxBlk, pkg)
}
fileBlk := stringRef(nil, srcPath) fileBlk := stringRef(nil, srcPath)
var relocBlk, auxBlk, dataBlk []byte var relocBlk, auxBlk, dataBlk []byte
@@ -374,19 +539,21 @@ func marshalFuncInfo(fn FuncLayout) []byte {
} }
// pcValueFlat encodes a pc-value table holding v over the whole function. // pcValueFlat encodes a pc-value table holding v over the whole function.
func pcValueFlat(v int32, size int) []byte { // The pc deltas are in MinLC units (the runtime scales them by the
// architecture's minimum instruction length).
func pcValueFlat(v int32, size, minLC int) []byte {
// The table is delta-encoded from an implicit value of -1: a varint // The table is delta-encoded from an implicit value of -1: a varint
// value delta, an unsigned pc delta to the end, and a zero terminator. // value delta, an unsigned pc delta to the end, and a zero terminator.
out := binary.AppendVarint(nil, int64(v)+1) out := binary.AppendVarint(nil, int64(v)+1)
out = binary.AppendUvarint(out, uint64(size)) out = binary.AppendUvarint(out, uint64(size/minLC))
return append(out, 0) return append(out, 0)
} }
// pcspTable encodes the stack-adjustment table: the SP delta in effect at // pcspTable encodes the stack-adjustment table: the SP delta in effect at
// every pc, from the function's prologue and epilogue boundaries. // every pc, from the function's prologue and epilogue boundaries.
func pcspTable(fn FuncLayout) []byte { func pcspTable(fn FuncLayout, minLC int) []byte {
if len(fn.Spadj) == 0 { if len(fn.Spadj) == 0 {
return pcValueFlat(0, fn.Size) return pcValueFlat(0, fn.Size, minLC)
} }
pts := make([]SpadjStep, 0, len(fn.Spadj)+1) pts := make([]SpadjStep, 0, len(fn.Spadj)+1)
pts = append(pts, SpadjStep{PC: 0, Value: 0}) pts = append(pts, SpadjStep{PC: 0, Value: 0})
@@ -394,11 +561,11 @@ func pcspTable(fn FuncLayout) []byte {
out := binary.AppendVarint(nil, int64(pts[0].Value)+1) out := binary.AppendVarint(nil, int64(pts[0].Value)+1)
cur, old := pts[0].PC, pts[0].Value cur, old := pts[0].PC, pts[0].Value
for _, p := range pts[1:] { for _, p := range pts[1:] {
out = binary.AppendUvarint(out, uint64(p.PC-cur)) out = binary.AppendUvarint(out, uint64((p.PC-cur)/minLC))
out = binary.AppendVarint(out, int64(p.Value-old)) out = binary.AppendVarint(out, int64(p.Value-old))
cur, old = p.PC, p.Value cur, old = p.PC, p.Value
} }
out = binary.AppendUvarint(out, uint64(fn.Size-cur)) out = binary.AppendUvarint(out, uint64((fn.Size-cur)/minLC))
return append(out, 0) return append(out, 0)
} }
+188
View File
@@ -0,0 +1,188 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
)
// This file generates the per-function DWARF symbols the linker's DWARF
// pass requires of an assembly object, byte-identical to what cmd/asm
// emits: the .debug_line state-machine program (SDWARFLINES) and the
// subprogram DIE (SDWARFFCN). The linker copies the DIE and line-program
// bytes verbatim into .debug_info and .debug_line, fixing up their
// relocations, so the formats here must match cmd/internal/dwarf's
// DW_ABRV_FUNCTION and generateDebugLinesSymbol exactly.
//
// DWARF5 is assumed throughout (the toolchain's default on Linux and the
// other non-Darwin targets gasm supports).
// Line-program parameters (cmd/internal/obj/dwarf.go).
const (
dwLineBase = -4
dwLineRange = 10
dwOpcodeBase = 11
dwPCRange = (255 - dwOpcodeBase) / dwLineRange
)
// goobjReloc is one relocation attached to an emitter-generated symbol
// (the DWARF lines/info symbols), in goobj's on-disk encoding fields.
type goobjReloc struct {
off int32
siz uint8
typ uint16
add int64
pkg uint32
sym uint32
}
// goobjDwarfLines builds the function's .debug_line state-machine program:
// an LNE_set_address extended opcode establishing the function's start
// address (carrying the R_ADDR relocation), one row per source line
// change across the function's instructions, an advance to the end of the
// function and an end-of-sequence opcode. The linker appends these bytes
// after the unit's line header, so they must start with the address and
// leave the state machine terminated.
func goobjDwarfLines(fn FuncLayout, fnNpIdx int) ([]byte, []goobjReloc) {
// Rows: the prologue, if any, then the body instructions (fn.Lines
// covers the body only). The first body offset > 0 means a prologue
// precedes it; the toolchain reports the prologue on the TEXT line.
pts := make([]LineEntry, 0, len(fn.Lines)+1)
if len(fn.Lines) == 0 || fn.Lines[0].Offset > 0 {
pts = append(pts, LineEntry{Offset: 0, Line: fn.Line})
}
pts = append(pts, fn.Lines...)
out := []byte{0, 9, 2, 0, 0, 0, 0, 0, 0, 0, 0} // LNE_set_address, address zeroed
relocs := []goobjReloc{{
off: 3, siz: 8, typ: relocAddr,
pkg: pkgIdxNone, sym: uint32(fnNpIdx),
}}
// The state machine starts at line 1, pc 0 (function-relative); the
// implicit initial pc is the function entry, so the first pc delta is
// against 0.
line := int64(1)
pc := uint64(0)
for _, p := range pts {
if p.Line == 0 || uint64(p.Offset) < pc {
continue
}
// Rows mark source-line changes only; the pc delta is measured from
// the previous row, not the previous instruction.
if int64(p.Line) == line {
continue
}
deltaPC := uint64(p.Offset) - pc
deltaLC := int64(p.Line) - line
out = dwPutPCLCDelta(out, deltaPC, deltaLC)
line, pc = int64(p.Line), uint64(p.Offset)
}
// Cover the rest of the function and close the sequence.
if end := uint64(fn.Size) - pc; end > 0 {
out = append(out, 2) // DW_LNS_advance_pc
out = binary.AppendUvarint(out, end)
}
out = append(out, 0, 1, 1) // LNE_end_sequence
return out, relocs
}
// dwPutPCLCDelta encodes one (pcDelta, lineDelta) step as the shortest
// special opcode plus any standard-opcode remainder, exactly like
// cmd/internal/obj's putpclcdelta.
func dwPutPCLCDelta(b []byte, deltaPC uint64, deltaLC int64) []byte {
opcode := dwSelectOpcode(deltaPC, deltaLC)
deltaPC -= uint64((opcode - dwOpcodeBase) / dwLineRange)
deltaLC -= (opcode-dwOpcodeBase)%dwLineRange + dwLineBase
// The remainder: standard opcodes first, then the special opcode
// (which emits the row).
if deltaPC != 0 {
switch {
case deltaPC <= uint64(dwPCRange):
opcode -= dwLineRange * int64(uint64(dwPCRange)-deltaPC)
b = append(b, 8) // DW_LNS_const_add_pc
case (1<<14) <= deltaPC && deltaPC < (1<<16):
b = append(b, 9) // DW_LNS_fixed_advance_pc
b = binary.LittleEndian.AppendUint16(b, uint16(deltaPC))
default:
b = append(b, 2) // DW_LNS_advance_pc
b = binary.AppendUvarint(b, deltaPC)
}
}
if deltaLC != 0 {
b = append(b, 3) // DW_LNS_advance_line
b = binary.AppendVarint(b, deltaLC)
}
return append(b, byte(opcode))
}
// dwSelectOpcode picks the special opcode for (deltaPC, deltaLC) per
// cmd/internal/obj's putpclcdelta selection logic.
func dwSelectOpcode(deltaPC uint64, deltaLC int64) int64 {
switch {
case deltaLC < dwLineBase:
if deltaPC >= uint64(dwPCRange) {
return dwOpcodeBase + dwLineRange*dwPCRange
}
return dwOpcodeBase + dwLineRange*int64(deltaPC)
case deltaLC < dwLineBase+dwLineRange:
if deltaPC >= uint64(dwPCRange) {
op := int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*dwPCRange
if op > 255 {
op -= dwLineRange
}
return op
}
return int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*int64(deltaPC)
default:
if deltaPC <= uint64(dwPCRange) {
op := int64(dwOpcodeBase) + (dwLineRange - 1) + dwLineRange*int64(deltaPC)
if op > 255 {
op = 255
}
return op
}
switch deltaPC - uint64(dwPCRange) {
case uint64(dwPCRange), (1 << 7) - 1, (1 << 16) - 1, (1 << 21) - 1,
(1 << 28) - 1, (1 << 35) - 1, (1 << 42) - 1, (1 << 49) - 1,
(1 << 56) - 1, (1 << 63) - 1:
return 255
default:
// 250: the toolchain's "249" comment is stale.
return dwOpcodeBase + dwLineRange*dwPCRange - 1
}
}
}
// goobjDwarfInfo builds the function's DWARF5 subprogram DIE (abbrev
// DW_ABRV_FUNCTION): name, low_pc as a .debug_addr index (the
// R_DWTXTADDR_U4 relocation), high_pc as the size, the call-frame-CFA
// frame base, the decl file/line and the external flag. name is the
// symbol's object name (package-qualified unless static).
func goobjDwarfInfo(fn FuncLayout, name string, fnNpIdx int) ([]byte, []goobjReloc) {
out := []byte{3} // DW_ABRV_FUNCTION
out = append(out, name...)
out = append(out, 0)
addrx := len(out)
out = append(out, 0, 0, 0, 0) // DW_AT_low_pc: addrx slot, zeroed
out = binary.AppendUvarint(out, uint64(fn.Size))
out = append(out, 1, 0x9c) // DW_AT_frame_base: block1, DW_OP_call_frame_cfa
out = binary.LittleEndian.AppendUint32(out, 1)
out = binary.AppendUvarint(out, uint64(fn.Line))
if fn.Static {
out = append(out, 0)
} else {
out = append(out, 1) // DW_AT_external
}
out = append(out, 0) // end of children
relocs := []goobjReloc{{
off: int32(addrx), siz: 4, typ: relocDWTXTADDRU4(),
pkg: pkgIdxNone, sym: uint32(fnNpIdx),
}}
return out, relocs
}
+214
View File
@@ -0,0 +1,214 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"testing"
)
// TestDWSelectOpcode checks the special-opcode selection against
// hand-computed values for the boundary cases: line deltas below, inside
// and above the line range, and pc deltas at and beyond PC_RANGE (24).
func TestDWSelectOpcode(t *testing.T) {
cases := []struct {
deltaPC uint64
deltaLC int64
want int64
}{
{0, 2, 17}, // the common single-instruction step
{0, -4, 11}, // deltaLC == LINE_BASE
{0, -5, 11}, // deltaLC below LINE_BASE: opcode adds nothing
{0, 6, 20}, // deltaLC == LINE_BASE+LINE_RANGE, remainder via advance_line
{4, 1, 56}, // the 4-byte loong64 instruction step
{23, 1, 246}, // deltaPC == PC_RANGE-1
{24, 1, 246}, // deltaPC == PC_RANGE: wraps past 255
{25, 1, 246}, // deltaPC past PC_RANGE (the const_add_pc remainder adjusts it later)
{100, 1, 246},
{151, 10, 255}, // deltaPC-PC_RANGE == (1<<7)-1, large line delta
{100, 10, 250}, // deltaPC-PC_RANGE not on a switch boundary
{23, 10, 250}, // large line delta inside PC_RANGE
}
for _, c := range cases {
if got := dwSelectOpcode(c.deltaPC, c.deltaLC); got != c.want {
t.Errorf("dwSelectOpcode(%d, %d) = %d, want %d", c.deltaPC, c.deltaLC, got, c.want)
}
}
}
// decodeDWLineProgram decodes a .debug_line state-machine program (as
// emitted by goobjDwarfLines) into (pc, line) rows.
func decodeDWLineProgram(t *testing.T, b []byte) (pcs []uint64, lines []int64) {
t.Helper()
pc, line := uint64(0), int64(1)
emit := func() {
if len(pcs) == 0 || pcs[len(pcs)-1] != pc || lines[len(lines)-1] != line {
pcs = append(pcs, pc)
lines = append(lines, line)
}
}
advancePC := func(delta uint64) { pc += delta }
advanceLine := func(delta int64) { line += delta }
for i := 0; i < len(b); {
op := b[i]
i++
switch {
case op == 0: // extended opcode
ln, n := binary.Uvarint(b[i:])
i += n
sub := b[i]
i++
_ = ln
switch sub {
case 2: // DW_LNE_set_address: 8-byte address
pc = binary.LittleEndian.Uint64(b[i:])
i += 8
case 1: // DW_LNE_end_sequence
// terminates the sequence; no new row
}
case op == 2: // DW_LNS_advance_pc
v, n := binary.Uvarint(b[i:])
i += n
advancePC(v)
case op == 3: // DW_LNS_advance_line
v, n := binary.Varint(b[i:])
i += n
advanceLine(v)
case op == 8: // DW_LNS_const_add_pc
advancePC(uint64(dwPCRange))
case op == 9: // DW_LNS_fixed_advance_pc
advancePC(uint64(binary.LittleEndian.Uint16(b[i:])))
i += 2
case op >= dwOpcodeBase: // special opcode
advancePC(uint64((int64(op) - dwOpcodeBase) / dwLineRange))
advanceLine((int64(op)-dwOpcodeBase)%dwLineRange + dwLineBase)
emit()
}
}
return pcs, lines
}
// TestGoobjDwarfLinesRows checks the emitted line program's rows for
// synthetic functions: a zero-frame function with one instruction per
// line, a framed function (the prologue row is prepended on the TEXT
// line), instructions sharing a line, and a function with a large pc gap
// (the const_add_pc remainder path).
func TestGoobjDwarfLinesRows(t *testing.T) {
cases := []struct {
name string
fn FuncLayout
want [][2]int64 // (pc, line)
}{
{
"one instruction per line",
FuncLayout{Size: 20, Line: 2, Lines: []LineEntry{
{0, 3}, {4, 4}, {8, 5}, {12, 6}, {16, 7},
}},
[][2]int64{{0, 3}, {4, 4}, {8, 5}, {12, 6}, {16, 7}},
},
{
"framed: prologue row on the TEXT line",
FuncLayout{Size: 24, Line: 2, Lines: []LineEntry{
{12, 3}, {16, 4},
}},
[][2]int64{{0, 2}, {12, 3}, {16, 4}},
},
{
"instructions sharing a line fold into one row",
FuncLayout{Size: 16, Line: 2, Lines: []LineEntry{
{0, 3}, {4, 3}, {8, 4}, {12, 4},
}},
[][2]int64{{0, 3}, {8, 4}},
},
{
"large gap crosses PC_RANGE",
FuncLayout{Size: 60, Line: 2, Lines: []LineEntry{
{0, 3}, {40, 4},
}},
[][2]int64{{0, 3}, {40, 4}},
},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
prog, relocs := goobjDwarfLines(c.fn, 0)
if len(relocs) != 1 || relocs[0].off != 3 || relocs[0].siz != 8 || relocs[0].typ != relocAddr || relocs[0].sym != 0 {
t.Fatalf("relocs = %+v", relocs)
}
pcs, lines := decodeDWLineProgram(t, prog)
if len(pcs) != len(c.want) {
t.Fatalf("rows = %d (%v / %v), want %d", len(pcs), pcs, lines, len(c.want))
}
for i, w := range c.want {
if pcs[i] != uint64(w[0]) || lines[i] != w[1] {
t.Errorf("row %d = (%d, %d), want (%d, %d)", i, pcs[i], lines[i], w[0], w[1])
}
}
})
}
}
// TestGoobjDwarfInfo checks the subprogram DIE for an exported and a
// static function: the abbrev, name, high_pc, frame base, decl file/line,
// the external flag and the addrx relocation position.
func TestGoobjDwarfInfo(t *testing.T) {
fn := FuncLayout{Size: 20, Line: 2}
die, relocs := goobjDwarfInfo(fn, "pkg.f", 3)
want := []byte{
0x03,
'p', 'k', 'g', '.', 'f', 0,
0, 0, 0, 0, // addrx slot at offset 7
0x14, // high_pc: 20
0x01, 0x9c, // frame_base
0x01, 0, 0, 0, // decl_file 1
0x02, // decl_line 2
0x01, // external
0x00, // end of children
}
if !bytes.Equal(die, want) {
t.Errorf("DIE = %x, want %x", die, want)
}
if len(relocs) != 1 || relocs[0].off != 7 || relocs[0].siz != 4 || relocs[0].typ != relocDWTXTADDRU4() || relocs[0].sym != 3 {
t.Errorf("relocs = %+v", relocs)
}
// A static function carries no external flag and no package prefix.
fn.Static = true
die, _ = goobjDwarfInfo(fn, "f", 1)
if die[len(die)-2] != 0 {
t.Errorf("static external flag = %d, want 0", die[len(die)-2])
}
}
// TestDwPutPCLCDeltaRemainders checks the standard-opcode remainders:
// const_add_pc and fixed_advance_pc after a special opcode.
func TestDwPutPCLCDeltaRemainders(t *testing.T) {
// deltaPC 25 past PC_RANGE: opcode 26 covers (1, 1), const_add_pc
// covers the remaining 23 pc and 0 line.
got := dwPutPCLCDelta(nil, 25, 1)
if !bytes.Equal(got, []byte{8, 26}) {
t.Errorf("25/1 = %x, want [8 1a]", got)
}
// deltaPC 20000: opcode 246 covers 23, fixed_advance_pc covers the
// remaining 19977.
got = dwPutPCLCDelta(nil, 20000, 1)
if got[0] != 9 || binary.LittleEndian.Uint16(got[1:]) != 19977 || got[3] != 246 {
t.Errorf("20000/1 = %x, want fixed_advance_pc 19977 then 246", got)
}
// Line remainder: deltaLC 10 leaves 5 past the opcode's reach, encoded
// as advance_line 5 (zigzag 0x0a) before opcode 250.
got = dwPutPCLCDelta(nil, 23, 10)
if !bytes.Equal(got, []byte{3, 0x0a, 250}) {
t.Errorf("23/10 = %x, want [03 0a fa]", got)
}
// Negative line remainder: deltaLC -5 leaves advance_line -1 (zigzag
// 0x01) after opcode 11.
got = dwPutPCLCDelta(nil, 0, -5)
if !bytes.Equal(got, []byte{3, 1, 11}) {
t.Errorf("0/-5 = %x, want [03 01 0b]", got)
}
}
+348
View File
@@ -0,0 +1,348 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"fmt"
"os"
"os/exec"
"strings"
)
// readGOOBJSymbols reads the GOOBJ symbol definitions from a compiled Go
// package's export file. The file is an ar archive containing a __.PKGDEF
// member whose payload is the "go object ...\n!\n" preamble followed by the
// GOOBJ data. The function returns the symbol names in definition order
// (the order they appear in blkSymdef), which matches the SymIdx the linker
// expects for cross-package references.
func readGOOBJSymbols(exportPath string) ([]string, error) {
data, err := os.ReadFile(exportPath)
if err != nil {
return nil, err
}
goobj, err := extractGOOBJ(data)
if err != nil {
return nil, fmt.Errorf("%s: %w", exportPath, err)
}
return goobj.symbols(), nil
}
// exportPath returns the export file path for a given import path by running
// "go list -export". The result is cached so repeated calls for the same
// package are fast.
func exportPath(importPath string) (string, error) {
cmd := exec.Command("go", "list", "-json", "-export", importPath)
out, err := cmd.Output()
if err != nil {
return "", fmt.Errorf("go list %s: %w", importPath, err)
}
// Quick JSON extraction: find "Export": "…"
const key = `"Export": "`
i := bytes.Index(out, []byte(key))
if i < 0 {
return "", fmt.Errorf("go list %s: no Export field", importPath)
}
start := i + len(key)
end := bytes.IndexByte(out[start:], '"')
if end < 0 {
return "", fmt.Errorf("go list %s: malformed Export field", importPath)
}
return string(out[start : start+end]), nil
}
// resolveExternalGOOBJ resolves a set of external symbol references into
// (package index, symbol index) pairs suitable for GOOBJ emission.
//
// refs maps package import paths to the symbol names referenced from that
// package. The returned pkgIdx maps each import path to its position in
// the blkPkgIdx table (0-based), and symIdx gives each symbol's index within
// its package.
func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symIdx map[string]int, err error) {
pkgIdx = make(map[string]int, len(refs))
symIdx = make(map[string]int)
// Assign package indices in sorted order for determinism.
packages := sortedPkgRefs(refs)
for i, pkg := range packages {
pkgIdx[pkg.path] = i
exp, err := exportPath(pkg.path)
if err != nil {
return nil, nil, err
}
data, err := os.ReadFile(exp)
if err != nil {
return nil, nil, err
}
gobj, err := extractGOOBJ(data)
if err != nil {
return nil, nil, fmt.Errorf("%s: %w", pkg.path, err)
}
for _, name := range pkg.syms {
idx := gobj.findSymbol(pkg.path, name)
if idx < 0 {
return nil, nil, fmt.Errorf("symbol %s·%s not found in export data of %s", pkg.path, name, pkg.path)
}
symIdx[pkg.path+"·"+name] = idx
}
}
return pkgIdx, symIdx, nil
}
type pkgRef struct {
path string
syms []string
}
func sortedPkgRefs(refs map[string][]string) []pkgRef {
var pkgs []pkgRef
for pkg, syms := range refs {
pkgs = append(pkgs, pkgRef{pkg, syms})
}
// Simple insertion sort — the list is tiny (usually 1–3 packages).
for i := 1; i < len(pkgs); i++ {
for j := i; j > 0 && pkgs[j-1].path > pkgs[j].path; j-- {
pkgs[j-1], pkgs[j] = pkgs[j], pkgs[j-1]
}
}
return pkgs
}
// extractGOOBJ finds the GOOBJ data in an ar archive and returns a parsed
// goobjFile. The archive member _go_.o contains the "go object …\n!\n"
// preamble followed by the GOOBJ payload; __.PKGDEF is the compiler export
// data (type information) and is not the GOOBJ object.
func extractGOOBJ(data []byte) (*goobjFile, error) {
if len(data) < 8 || string(data[:8]) != "!<arch>\n" {
return nil, fmt.Errorf("not an ar archive")
}
pos := 8
for pos+60 <= len(data) {
hdr := data[pos : pos+60]
pos += 60
// Parse ar header fields.
name := strings.TrimRight(string(hdr[:16]), " /")
size := parseArDecimal(hdr[48:58])
if size < 0 {
return nil, fmt.Errorf("invalid ar header: bad size")
}
if pos+size > len(data) {
return nil, fmt.Errorf("ar entry %q extends past end of file", name)
}
body := data[pos : pos+size]
pos += size
// ar pads to even bytes.
if pos%2 != 0 {
pos++
}
if name == "_go_.o" {
return parseGOOBJ(body)
}
}
return nil, fmt.Errorf("archive contains no _go_.o member")
}
// parseArDecimal parses a decimal number from a space-padded field.
func parseArDecimal(b []byte) int {
v := 0
for _, c := range b {
if c == ' ' {
continue
}
if c < '0' || c > '9' {
return -1
}
v = v*10 + int(c-'0')
}
return v
}
// goobjFile is a parsed GOOBJ file: the string table and the symbol-definition
// block.
type goobjFile struct {
strTab []byte // string table, at headerSize + n
symdef []byte // blkSymdef raw block
npdef []byte // blkNonpkgdef raw block
}
// symbols returns all symbol names in definition order by scanning the
// symdef and nonpkgdef blocks and resolving each name through the string
// table. Package definitions (blkSymdef) use fully-qualified names like
// "runtime.morestack"; non-package definitions (blkNonpkgdef) use bare
// names like "morestack". This combined list matches the index the
// linker expects for cross-package references.
func (f *goobjFile) symbols() []string {
return append(f.defNames(), f.npdefNames()...)
}
// findSymbol returns the index of a symbol within the combined symbol list,
// or -1 if not found. It first tries the fully-qualified name (pkg.name),
// then the bare name.
func (f *goobjFile) findSymbol(pkg, name string) int {
qualified := pkg + "." + name
syms := f.symbols()
for i, s := range syms {
if s == qualified {
return i
}
}
// Try bare name (for non-package definitions).
for i, s := range syms {
if s == name {
return i
}
}
return -1
}
// defNames returns names from blkSymdef only.
func (f *goobjFile) defNames() []string {
return f.readSymNames(f.symdef)
}
// npdefNames returns names from blkNonpkgdef.
func (f *goobjFile) npdefNames() []string {
return f.readSymNames(f.npdef)
}
// readSymNames reads symbol names from a symdef/nonpkgdef block. Each record
// is 21 bytes: nameLen (u32), nameOff (u32), abi (u16), typ, flag, flag2,
// size (u32), align (u32). nameOff is an absolute offset into the string
// table.
func (f *goobjFile) readSymNames(block []byte) []string {
const recSize = 21
if len(block) < recSize {
return nil
}
n := len(block) / recSize
names := make([]string, 0, n)
for i := 0; i < n; i++ {
rec := block[i*recSize : (i+1)*recSize]
nameLen := binary.LittleEndian.Uint32(rec[0:4])
nameOff := binary.LittleEndian.Uint32(rec[4:8])
// nameOff is an absolute offset into the GOOBJ payload. The string
// table we have starts at goobjHeaderSize, so we subtract that.
if nameOff < goobjHeaderSize {
continue
}
relOff := nameOff - goobjHeaderSize
if relOff >= uint32(len(f.strTab)) || relOff+nameLen > uint32(len(f.strTab)) {
continue
}
names = append(names, string(f.strTab[relOff:relOff+nameLen]))
}
return names
}
const goobjHeaderSize = 8 + 8 + 4 + 4*(blkEnd+1) // magic + fingerprint + flags + 19 block offsets
// parseGOOBJ parses a raw GOOBJ payload (the data after the "\n!\n" preamble).
func parseGOOBJ(data []byte) (*goobjFile, error) {
// Find the "\n!\n" separator.
sep := []byte("\n!\n")
i := bytes.Index(data, sep)
if i < 0 {
// Maybe the data has no preamble (e.g. a raw .o file).
i = -3 // treat as if preamble starts before the data
}
payload := data[i+len(sep):]
if len(payload) < goobjHeaderSize {
return nil, fmt.Errorf("GOOBJ payload too short (%d bytes)", len(payload))
}
if string(payload[:8]) != goobjMagic {
return nil, fmt.Errorf("bad GOOBJ magic: %q", payload[:8])
}
// Read block offsets. The header layout is:
// [0:8] magic
// [8:16] fingerprint
// [16:20] flags
// [20:96] 19 × uint32 offsets
var offs [blkEnd + 1]uint32
for i := 0; i <= blkEnd; i++ {
offs[i] = binary.LittleEndian.Uint32(payload[20+4*i:])
}
// The string table lives at headerSize.
strTabStart := uint32(goobjHeaderSize)
f := &goobjFile{
strTab: payload[strTabStart:offs[0]],
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
}
return f, nil
}
// blockSlice extracts a block from the payload using its offset pair.
func blockSlice(payload []byte, offs [blkEnd + 1]uint32, start, end int) []byte {
if start < 0 || end > blkEnd || offs[end] < offs[start] {
return nil
}
beg := offs[start]
fin := offs[end]
if int(fin) > len(payload) || int(beg) > int(fin) {
return nil
}
return payload[beg:fin]
}
// resolveExternalSymbols is the high-level entry point for GOOBJ emission.
// Given a list of external symbol names (e.g. ["runtime·morestack",
// "runtime·g0"]), it returns the package-index table entries and a map from
// full symbol name to GOOBJ {pkgIdx, symIdx}.
//
// The package table entries should be written into blkPkgIdx, and the
// returned indices should replace pkgIdxSelf / placeholder values in the
// relocation records.
func resolveExternalSymbols(externals []string) (pkgTable []string, pkgIdxMap map[string]int, symIdxMap map[string]int, err error) {
// Group references by package.
refs := make(map[string]map[string]bool)
for _, full := range externals {
pkg, name := splitQualified(full)
if refs[pkg] == nil {
refs[pkg] = make(map[string]bool)
}
refs[pkg][name] = true
}
// Convert maps to slices.
r := make(map[string][]string, len(refs))
for pkg, names := range refs {
for name := range names {
r[pkg] = append(r[pkg], name)
}
}
pkgIdx1, symIdx1, err := resolveExternalGOOBJ(r)
if err != nil {
return nil, nil, nil, err
}
// Build the package table in pkgIdx order.
pkgTable = make([]string, len(pkgIdx1))
for pkg, idx := range pkgIdx1 {
pkgTable[idx] = pkg
}
return pkgTable, pkgIdx1, symIdx1, nil
}
// splitQualified splits a qualified Go symbol name (pkgpath·name) into its
// package path and local name. The separator is the middle dot (U+00B7).
// If no separator is found, the symbol is assumed to be in the current
// package (empty pkg).
func splitQualified(full string) (pkg, name string) {
if idx := strings.IndexByte(full, '\u00b7'); idx >= 0 {
return full[:idx], full[idx+len("\u00b7"):]
}
if idx := strings.IndexByte(full, '.'); idx >= 0 {
return full[:idx], full[idx+1:]
}
return "", full
}
+66
View File
@@ -0,0 +1,66 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"os"
"os/exec"
"testing"
)
// TestReadRuntimeSymbols verifies the GOOBJ reader can extract and find
// symbols from the runtime package's compiled archive.
func TestReadRuntimeSymbols(t *testing.T) {
exp, err := exportPath("runtime")
if err != nil {
t.Skipf("cannot find runtime export: %v (need Go toolchain)", err)
}
data, err := os.ReadFile(exp)
if err != nil {
t.Skipf("cannot read runtime export: %v", err)
}
gobj, err := extractGOOBJ(data)
if err != nil {
t.Fatalf("extractGOOBJ: %v", err)
}
t.Logf("runtime: %d symbols", len(gobj.symbols()))
// Verify we can find well-known runtime symbols.
for _, tc := range []struct{ pkg, name string }{
{"runtime", "g0"},
{"runtime", "morestack"},
{"runtime", "newstack"},
} {
idx := gobj.findSymbol(tc.pkg, tc.name)
if idx < 0 {
t.Errorf("findSymbol(%q, %q) = -1", tc.pkg, tc.name)
} else {
t.Logf("findSymbol(%q, %q) = %d", tc.pkg, tc.name, idx)
}
}
}
// TestResolveExternalSymbols verifies end-to-end resolution of external
// symbol references.
func TestResolveExternalSymbols(t *testing.T) {
if _, err := exec.LookPath("go"); err != nil {
t.Skip("go toolchain not available")
}
refs := map[string][]string{
"runtime": {"g0"},
}
pkgIdx, symIdx, err := resolveExternalGOOBJ(refs)
if err != nil {
t.Fatalf("resolveExternalGOOBJ: %v", err)
}
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 0 {
t.Errorf("pkgIdx = %v, want runtime→0", pkgIdx)
}
if _, ok := symIdx["runtime·g0"]; !ok {
t.Errorf("symIdx missing runtime·g0, got %v", symIdx)
}
t.Logf("runtime·g0 → SymIdx=%d", symIdx["runtime·g0"])
}
+75 -39
View File
@@ -111,17 +111,26 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
t.Errorf("flags = %#x, want ObjFlagFromAssembly (4)", flags) t.Errorf("flags = %#x, want ObjFlagFromAssembly (4)", flags)
} }
// Package defs: the static GLOBL, then one anonymous FuncInfo per // Package defs: the static GLOBL, then per function the FuncInfo and the
// function. // two DWARF symbols (debug_line program, subprogram DIE).
defs := v.syms(blkSymdef) defs := v.syms(blkSymdef)
if len(defs) != 3 { if len(defs) != 7 {
t.Fatalf("symdefs = %d, want 3", len(defs)) t.Fatalf("symdefs = %d, want 7", len(defs))
} }
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link { if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
t.Errorf("mask symbol = %+v", defs[0]) t.Errorf("mask symbol = %+v", defs[0])
} }
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 { if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
t.Errorf("funcinfo symbol = %+v", defs[1]) t.Errorf("addq funcinfo symbol = %+v", defs[1])
}
if defs[2].name != "" || defs[2].typ != kindSDWARFLINES || defs[2].size == 0 {
t.Errorf("addq lines symbol = %+v", defs[2])
}
if defs[3].name != "" || defs[3].typ != kindSDWARFFCN || defs[3].size == 0 {
t.Errorf("addq DIE symbol = %+v", defs[3])
}
if defs[4].name != "" || defs[4].typ != kindSDATA || defs[4].size != 28 {
t.Errorf("loadmask funcinfo symbol = %+v", defs[4])
} }
// Non-package defs: four pc tables and the function, per function. // Non-package defs: four pc tables and the function, per function.
@@ -142,45 +151,62 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
// FuncInfo: args 24, FuncFlag Asm, one file, no inline tree. // FuncInfo: args 24, FuncFlag Asm, one file, no inline tree.
le := binary.LittleEndian le := binary.LittleEndian
data := v.blk(blkData) data := v.blk(blkData)
didx := v.blk(blkDataIdx)
fi := data[16:44] fi := data[16:44]
if le.Uint32(fi[0:]) != 24 || le.Uint32(fi[4:]) != 0 || fi[8] != 0 || fi[9] != funcFlagAsm || if le.Uint32(fi[0:]) != 24 || le.Uint32(fi[4:]) != 0 || fi[8] != 0 || fi[9] != funcFlagAsm ||
le.Uint32(fi[16:]) != 1 || le.Uint32(fi[20:]) != 0 || le.Uint32(fi[24:]) != 0 { le.Uint32(fi[16:]) != 1 || le.Uint32(fi[20:]) != 0 || le.Uint32(fi[24:]) != 0 {
t.Errorf("funcinfo bytes %x", fi) t.Errorf("funcinfo bytes %x", fi)
} }
// pcsp: a flat zero over the whole function (zero-frame NOSPLIT). // The pc-value tables of addq (non-package indices 0–3, so global
if got := data[72:75]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) { // indices 7–10): pcsp a flat zero over the whole function, pcinline a
// flat -1, both with the pc delta in MinLC (1) units.
pcsp := data[le.Uint32(didx[4*7:]):]
if got := pcsp[:3]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
t.Errorf("pcsp = %x, want 021300", got) t.Errorf("pcsp = %x, want 021300", got)
} }
// pcinline: a flat -1. pcinl := data[le.Uint32(didx[4*10:]):]
if got := data[81:84]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) { if got := pcinl[:3]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) {
t.Errorf("pcinline = %x, want 001300", got) t.Errorf("pcinline = %x, want 001300", got)
} }
// The one relocation: R_PCREL, four bytes wide, against the GLOBL, // Relocations: the four DWARF address references (two per function, in
// with the field in the function code left zero. The loadmask code's // definition order), then the loadmask code's R_PCREL against the
// offset comes from the data index (symbol 3 defs + 9 non-package). // GLOBL, with the field in the function code left zero. The loadmask
// code's offset comes from the data index (7 defs + 9 non-package).
relocs := v.blk(blkReloc) relocs := v.blk(blkReloc)
if len(relocs) != 23 { if len(relocs) != 5*23 {
t.Fatalf("relocs = %d bytes, want one 23-byte entry", len(relocs)) t.Fatalf("relocs = %d bytes, want 5 entries", len(relocs))
} }
off := int32(le.Uint32(relocs[0:])) // addq's DWARF references (defs 2 and 3) against the function, which
if off != 4 || relocs[4] != 4 || le.Uint16(relocs[5:]) != relocPCRel || // is non-package index 4.
le.Uint64(relocs[7:]) != 0 || le.Uint32(relocs[15:]) != pkgIdxSelf || le.Uint32(relocs[19:]) != 0 { lr := relocs[:23]
t.Errorf("reloc = %x", relocs) if int32(le.Uint32(lr[0:])) != 3 || lr[4] != 8 || le.Uint16(lr[5:]) != relocAddr ||
le.Uint32(lr[15:]) != pkgIdxNone || le.Uint32(lr[19:]) != 4 {
t.Errorf("addq lines reloc = %x", lr)
} }
didx := v.blk(blkDataIdx) dr := relocs[23:46]
lm := le.Uint32(didx[4*(3+9):]) if dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4() ||
le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 {
t.Errorf("addq DIE reloc = %x", dr)
}
cr := relocs[4*23:]
off := int32(le.Uint32(cr[0:]))
if off != 4 || cr[4] != 4 || le.Uint16(cr[5:]) != relocPCRel ||
le.Uint64(cr[7:]) != 0 || le.Uint32(cr[15:]) != pkgIdxSelf || le.Uint32(cr[19:]) != 0 {
t.Errorf("loadmask reloc = %x", cr)
}
lm := le.Uint32(didx[4*16:])
code := data[lm : lm+18] code := data[lm : lm+18]
if !bytes.Equal(code[4:8], []byte{0, 0, 0, 0}) { if !bytes.Equal(code[4:8], []byte{0, 0, 0, 0}) {
t.Errorf("relocated field = %x, want zeroed", code[4:8]) t.Errorf("relocated field = %x, want zeroed", code[4:8])
} }
// Aux wiring: FuncInfo (package symbol), then the four pc tables // Aux wiring: FuncInfo, the two DWARF symbols (package symbols), then
// (non-package symbols). // the four pc tables (non-package symbols).
auxs := v.blk(blkAux) auxs := v.blk(blkAux)
if len(auxs) != 2*5*9 { if len(auxs) != 2*7*9 {
t.Fatalf("aux = %d bytes, want 10 entries", len(auxs)) t.Fatalf("aux = %d bytes, want 14 entries", len(auxs))
} }
wantAux := []struct { wantAux := []struct {
typ uint8 typ uint8
@@ -188,15 +214,19 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
idx uint32 idx uint32
}{ }{
{auxFuncInfo, pkgIdxSelf, 1}, {auxFuncInfo, pkgIdxSelf, 1},
{auxPcsp, pkgIdxNone, uint32(len(defs) + 0)}, {auxDwarfInfo, pkgIdxSelf, 3},
{auxPcfile, pkgIdxNone, uint32(len(defs) + 1)}, {auxDwarfLines, pkgIdxSelf, 2},
{auxPcline, pkgIdxNone, uint32(len(defs) + 2)}, {auxPcsp, pkgIdxNone, 0},
{auxPcinline, pkgIdxNone, uint32(len(defs) + 3)}, {auxPcfile, pkgIdxNone, 1},
{auxFuncInfo, pkgIdxSelf, 2}, {auxPcline, pkgIdxNone, 2},
{auxPcsp, pkgIdxNone, uint32(len(defs) + 5)}, {auxPcinline, pkgIdxNone, 3},
{auxPcfile, pkgIdxNone, uint32(len(defs) + 6)}, {auxFuncInfo, pkgIdxSelf, 4},
{auxPcline, pkgIdxNone, uint32(len(defs) + 7)}, {auxDwarfInfo, pkgIdxSelf, 6},
{auxPcinline, pkgIdxNone, uint32(len(defs) + 8)}, {auxDwarfLines, pkgIdxSelf, 5},
{auxPcsp, pkgIdxNone, 5},
{auxPcfile, pkgIdxNone, 6},
{auxPcline, pkgIdxNone, 7},
{auxPcinline, pkgIdxNone, 8},
} }
for i, w := range wantAux { for i, w := range wantAux {
e := auxs[i*9:] e := auxs[i*9:]
@@ -253,7 +283,7 @@ TEXT ·framed(SB), NOSPLIT, $8-0
t.Fatalf("AssembleFile: %v", err) t.Fatalf("AssembleFile: %v", err)
} }
fn := img.Funcs[0] fn := img.Funcs[0]
pcs, vals := decodePCValues(pcspTable(fn)) pcs, vals := decodePCValues(pcspTable(fn, 1))
// Prologue: PUSHQ BP (1 byte, +8), MOVQ SP, BP (3 bytes, no change), // Prologue: PUSHQ BP (1 byte, +8), MOVQ SP, BP (3 bytes, no change),
// SUBQ $8, SP (4 bytes, +16 in total); the RET's epilogue unwinds // SUBQ $8, SP (4 bytes, +16 in total); the RET's epilogue unwinds
// ADDQ $8, SP (+8) then POPQ BP (0). // ADDQ $8, SP (+8) then POPQ BP (0).
@@ -349,7 +379,7 @@ func main() {
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil { if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err) t.Fatal(err)
} }
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module goobjtest\n\ngo 1.26\n"), 0o644); err != nil { if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module goobjtest\n\ngo 1.27\n"), 0o644); err != nil {
t.Fatal(err) t.Fatal(err)
} }
@@ -401,13 +431,12 @@ func main() {
if err != nil { if err != nil {
t.Fatalf("GOObject: %v", err) t.Fatalf("GOObject: %v", err)
} }
if err := os.WriteFile(asmObj, obj, 0o644); err != nil {
t.Fatal(err)
}
// Rebuild the package archive with our object in place of the // Rebuild the package archive with our object in place of the
// toolchain's (go tool pack has no replace-in-place that dedupes, so // toolchain's (go tool pack has no replace-in-place that dedupes, so
// extract, substitute and repack). // extract, substitute and repack). The archive member holding the
// assembler's output is named after the asm object file, e.g.
// main_amd64.o.
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch) extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
membersDir := filepath.Join(dir, "members") membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil { if err := os.MkdirAll(membersDir, 0o755); err != nil {
@@ -417,6 +446,13 @@ func main() {
if out, err := extract.CombinedOutput(); err != nil { if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out) t.Fatalf("pack x: %v\n%s", err, out)
} }
member := filepath.Join(membersDir, filepath.Base(asmObj))
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, obj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch) listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listOut, err := listCmd.CombinedOutput() listOut, err := listCmd.CombinedOutput()
if err != nil { if err != nil {
+84
View File
@@ -0,0 +1,84 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"fmt"
"os"
"os/exec"
"path/filepath"
"sync"
)
// GOObjectAARCH64 emits a GOOBJ object file for AArch64. The layout is
// the shared one in goobj.go — the toolchain preamble, the go120ld header
// with its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the arm64 preamble, the MinLC of 4
// for the pc-value deltas, and R_ADDRARM64 relocation types for the
// ADRP+ADD/LDR/STR address pairs.
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreambleAARCH64()
if err != nil {
return nil, err
}
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
return relocArm64Addr, 4
})
}
// arm64 relocation types (cmd/internal/objabi). R_ADDRARM64 resolves an
// ADRP+ADD/LDR/STR pair to a symbol's address.
const (
relocArm64Addr = 9 // R_ADDRARM64
)
// toolchainObjectPreambleAARCH64 returns the "go object ...\n!\n" header
// the installed go tool asm writes for arm64, captured by assembling a
// one-instruction probe.
var (
preambleAARCH64Once sync.Once
preambleAARCH64 []byte
preambleAARCH64Err error
)
func toolchainObjectPreambleAARCH64() ([]byte, error) {
preambleAARCH64Once.Do(func() {
goBin, err := exec.LookPath("go")
if err != nil {
preambleAARCH64Err = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
return
}
dir, err := os.MkdirTemp("", "gasm-preamble-arm64")
if err != nil {
preambleAARCH64Err = err
return
}
defer os.RemoveAll(dir)
src := filepath.Join(dir, "probe_arm64.s")
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
preambleAARCH64Err = err
return
}
obj := filepath.Join(dir, "probe.o")
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
cmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := cmd.CombinedOutput(); err != nil {
preambleAARCH64Err = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
return
}
data, err := os.ReadFile(obj)
if err != nil {
preambleAARCH64Err = err
return
}
i := bytes.Index(data, []byte("\n!\n"))
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
preambleAARCH64Err = fmt.Errorf("unrecognised assembler object layout")
return
}
preambleAARCH64 = data[:i+3]
})
return preambleAARCH64, preambleAARCH64Err
}
+91
View File
@@ -0,0 +1,91 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"fmt"
"os"
"os/exec"
"path/filepath"
"sync"
)
// GOObjectLOONG64 emits a GOOBJ object file for LoongArch. The layout is
// the shared one in goobj.go — the toolchain preamble, the go120ld header
// with its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the loong64 preamble, the MinLC of 4
// for the pc-value deltas, and R_LOONG64_ADDR_HI/LO relocation types for
// the pcalau12i+addi.d address pairs.
func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreambleLOONG64()
if err != nil {
return nil, err
}
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
// A pcalau12i+addi.d pair: the high part carries
// R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO.
if r.Kind == RelLoong64AddrLo {
return relocLoong64AddrLo, 4
}
return relocLoong64AddrHi, 4
})
}
// Loong64 relocation types (cmd/internal/objabi). R_LOONG64_ADDR_HI
// resolves the high 20 bits of a PC-relative address into pcalau12i;
// R_LOONG64_ADDR_LO the low 12 bits into addi.d/ld/st.
const (
relocLoong64AddrHi = 77 // R_LOONG64_ADDR_HI
relocLoong64AddrLo = 78 // R_LOONG64_ADDR_LO
)
// toolchainObjectPreambleLOONG64 returns the "go object ...\n!\n" header
// the installed go tool asm writes for loong64, captured by assembling a
// one-instruction probe (see toolchainObjectPreamble).
var (
preambleLOONG64Once sync.Once
preambleLOONG64 []byte
preambleLOONG64Err error
)
func toolchainObjectPreambleLOONG64() ([]byte, error) {
preambleLOONG64Once.Do(func() {
goBin, err := exec.LookPath("go")
if err != nil {
preambleLOONG64Err = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
return
}
dir, err := os.MkdirTemp("", "gasm-preamble-loong64")
if err != nil {
preambleLOONG64Err = err
return
}
defer os.RemoveAll(dir)
src := filepath.Join(dir, "probe_loong64.s")
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
preambleLOONG64Err = err
return
}
obj := filepath.Join(dir, "probe.o")
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
cmd.Env = append(os.Environ(), "GOARCH=loong64")
if out, err := cmd.CombinedOutput(); err != nil {
preambleLOONG64Err = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
return
}
data, err := os.ReadFile(obj)
if err != nil {
preambleLOONG64Err = err
return
}
i := bytes.Index(data, []byte("\n!\n"))
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
preambleLOONG64Err = fmt.Errorf("unrecognised assembler object layout")
return
}
preambleLOONG64 = data[:i+3]
})
return preambleLOONG64, preambleLOONG64Err
}
+25 -282
View File
@@ -5,7 +5,6 @@ package asm
import ( import (
"bytes" "bytes"
"encoding/binary"
"fmt" "fmt"
"os" "os"
"os/exec" "os/exec"
@@ -13,298 +12,42 @@ import (
"sync" "sync"
) )
// GOObjectRISCV emits a GOOBJ object file for RISC-V. // GOObjectRISCV emits a GOOBJ object file for RISC-V. The layout is the
// The format is the same as amd64 GOOBJ, but with the RISC-V architecture // shared one in goobj.go — the toolchain preamble, the go120ld header with
// marker in the preamble and RISC-V relocation types. // its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the RISC-V preamble, the MinLC of 2 for
// the pc-value deltas, and the single R_RISCV_PCREL_ITYPE/STYPE relocation
// per AUIPC pair, matching `go tool asm`'s model (each pair is one 8-byte
// relocation, not the ELF HI20/LO12 pair).
func (img *Image) GOObjectRISCV(pkgPath, srcPath string) ([]byte, error) { func (img *Image) GOObjectRISCV(pkgPath, srcPath string) ([]byte, error) {
if pkgPath == "" {
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
}
pre, err := toolchainObjectPreambleRISCV() pre, err := toolchainObjectPreambleRISCV()
if err != nil { if err != nil {
return nil, err return nil, err
} }
return img.emitGOObject(pkgPath, srcPath, pre, 2, func(r Reloc) (uint16, uint8) {
// The symbol tables. Package definitions: the GLOBL symbols, then one switch r.Kind {
// anonymous FuncInfo symbol per function. Non-package definitions: the case RelRISCVPCRELSType:
// pc-value tables and the functions themselves, as cmd/asm lays them return relocRISCVPcrelStype, 8
// out. defIdx maps a GLOBL's bare name to its definition index for the case RelRISCVJal:
// relocations; fnNpIdx maps a function to its non-package index. return relocRISCVJal, 4
var defs []goSym default:
var defData [][]byte return relocRISCVPcrelItype, 8
defIdx := map[string]int{}
for _, d := range img.DataSyms {
name := d.Name
if !d.Static {
name = pkgPath + "." + name
} }
typ := uint8(kindSDATA) })
if d.Rodata {
typ = kindSRODATA
}
flag := uint8(0)
if d.Dupok {
flag = symFlagDupok
}
abi := uint16(0)
if d.Static {
abi = symABIStatic
}
defIdx[d.Name] = len(defs)
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)})
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
}
fnFiIdx := make([]int, len(img.Funcs))
for i := range img.Funcs {
data := marshalFuncInfo(img.Funcs[i])
fnFiIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))})
defData = append(defData, data)
}
type npSym struct {
sym goSym
data []byte
}
var nps []npSym
type pcRefs struct{ sp, file, line, inl int }
pcIdx := make([]pcRefs, len(img.Funcs))
fnNpIdx := make([]int, len(img.Funcs))
for i, fn := range img.Funcs {
tables := []struct {
data []byte
dst *int
}{
{pcspTable(fn), &pcIdx[i].sp},
{pcValueFlat(0, fn.Size), &pcIdx[i].file},
{pcValueFlat(int32(fn.Line), fn.Size), &pcIdx[i].line},
{pcValueFlat(-1, fn.Size), &pcIdx[i].inl},
}
for _, t := range tables {
*t.dst = len(nps)
nps = append(nps, npSym{
sym: goSym{typ: kindSRODATA, size: uint32(len(t.data)), align: 1},
data: t.data,
})
}
name := fn.Name
abi := uint16(0)
if fn.Static {
abi = symABIStatic
} else {
name = pkgPath + "." + name
}
flag := uint8(0)
if fn.NoSplit {
flag |= symFlagNoSplit
}
fnNpIdx[i] = len(nps)
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
// The linker writes the resolved displacement into the field;
// leave it zero, as cmd/asm's object does.
if r.Off >= 0 && r.Off+4 <= len(code) {
code[r.Off], code[r.Off+1], code[r.Off+2], code[r.Off+3] = 0, 0, 0, 0
}
}
nps = append(nps, npSym{
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
data: code,
})
}
// Relocations, per defined symbol in definition order (package defs,
// then non-package defs). Only file-local GLOBL references resolve;
// external symbols need the import machinery of a later increment.
nsyms := len(defs) + len(nps)
symRelocs := make([][]byte, nsyms) // flat 23-byte records
for i, fn := range img.Funcs {
si := len(defs) + fnNpIdx[i]
for _, r := range fn.Relocs {
if r.External {
return nil, fmt.Errorf("GOOBJ emission: external symbol %q is not supported yet", r.Name)
}
di, ok := defIdx[r.Name]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
}
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = 4 // field width
binary.LittleEndian.PutUint16(rec[5:], relocRISCVPcrelHi20)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
symRelocs[si] = append(symRelocs[si], rec[:]...)
}
}
// Aux entries per function: FuncInfo, then the four pc tables.
// References into the non-package table use pkgIdxNone.
symAux := make([][]byte, nsyms)
for i := range img.Funcs {
si := len(defs) + fnNpIdx[i]
aux := func(typ uint8, pkg, idx uint32) {
var rec [9]byte
rec[0] = typ
binary.LittleEndian.PutUint32(rec[1:], pkg)
binary.LittleEndian.PutUint32(rec[5:], idx)
symAux[si] = append(symAux[si], rec[:]...)
}
aux(auxFuncInfo, pkgIdxSelf, uint32(fnFiIdx[i]))
aux(auxPcsp, pkgIdxNone, uint32(len(defs)+pcIdx[i].sp))
aux(auxPcfile, pkgIdxNone, uint32(len(defs)+pcIdx[i].file))
aux(auxPcline, pkgIdxNone, uint32(len(defs)+pcIdx[i].line))
aux(auxPcinline, pkgIdxNone, uint32(len(defs)+pcIdx[i].inl))
}
// --- Serialise ---
// String table: all symbol names, NUL-terminated.
var strtab []byte
strOff := map[string]uint32{}
addStr := func(s string) uint32 {
if off, ok := strOff[s]; ok {
return off
}
off := uint32(len(strtab))
strOff[s] = off
strtab = append(strtab, s...)
strtab = append(strtab, 0)
return off
}
for _, s := range defs {
addStr(s.name)
}
for _, s := range nps {
addStr(s.sym.name)
}
// Symbol definition records (21 bytes each).
var symdef, nonpkgdef []byte
for _, s := range defs {
symdef = s.append(symdef, strOff)
}
for _, s := range nps {
nonpkgdef = s.sym.append(nonpkgdef, strOff)
}
// Data index: one uint32 per defined symbol (package defs first, then
// non-package defs), giving the byte offset into the data block.
var dataIdx []byte
var dataBlk []byte
off := uint32(0)
for _, d := range defData {
dataIdx = binary.LittleEndian.AppendUint32(dataIdx, off)
dataBlk = append(dataBlk, d...)
off += uint32(len(d))
}
for _, s := range nps {
dataIdx = binary.LittleEndian.AppendUint32(dataIdx, off)
dataBlk = append(dataBlk, s.data...)
off += uint32(len(s.data))
}
dataIdx = binary.LittleEndian.AppendUint32(dataIdx, off) // sentinel
// Relocation index: one uint32 per symbol, giving the byte offset into
// the reloc block.
var relocIdx []byte
roff := uint32(0)
for i := 0; i < nsyms; i++ {
relocIdx = binary.LittleEndian.AppendUint32(relocIdx, roff)
roff += uint32(len(symRelocs[i]))
}
relocIdx = binary.LittleEndian.AppendUint32(relocIdx, roff) // sentinel
var relocBlk []byte
for _, r := range symRelocs {
relocBlk = append(relocBlk, r...)
}
// Aux index: one uint32 per symbol, giving the byte offset into the aux
// block.
var auxIdx []byte
aoff := uint32(0)
for i := 0; i < nsyms; i++ {
auxIdx = binary.LittleEndian.AppendUint32(auxIdx, aoff)
aoff += uint32(len(symAux[i]))
}
auxIdx = binary.LittleEndian.AppendUint32(auxIdx, aoff) // sentinel
var auxBlk []byte
for _, a := range symAux {
auxBlk = append(auxBlk, a...)
}
// File table: one entry, the source file.
var fileBlk []byte
fileOff := addStr(srcPath)
fileBlk = binary.LittleEndian.AppendUint32(fileBlk, uint32(len(srcPath)))
fileBlk = binary.LittleEndian.AppendUint32(fileBlk, fileOff)
// Assemble the object.
var out bytes.Buffer
out.Write(pre)
out.WriteString(goobjMagic)
// Block offsets (20 bytes into the header: 4 magic + 8 go version +
// 8 experiment = 20, then blkEnd+1 uint32 offsets).
// We'll fill these in after we know the sizes.
hdrStart := out.Len()
out.Write(make([]byte, 4*(blkEnd+1)))
writeBlock := func(data []byte) {
out.Write(data)
}
// Blocks in order: autolib, pkgidx, file, symdef, hashed64def, hasheddef,
// nonpkgdef, nonpkgref, refflags, hash64, hash, relocidx, auxidx, dataidx,
// reloc, aux, data, refname.
writeBlock(nil) // autolib
writeBlock(nil) // pkgidx
writeBlock(fileBlk) // file
writeBlock(symdef) // symdef
writeBlock(nil) // hashed64def
writeBlock(nil) // hasheddef
writeBlock(nonpkgdef) // nonpkgdef
writeBlock(nil) // nonpkgref
writeBlock(nil) // refflags
writeBlock(nil) // hash64
writeBlock(nil) // hash
writeBlock(relocIdx) // relocidx
writeBlock(auxIdx) // auxidx
writeBlock(dataIdx) // dataidx
writeBlock(relocBlk) // reloc
writeBlock(auxBlk) // aux
writeBlock(dataBlk) // data
writeBlock(nil) // refname
// Fill in the block offsets.
le := binary.LittleEndian
offs := make([]uint32, blkEnd+1)
pos := uint32(hdrStart + 4*(blkEnd+1))
for i := 0; i < blkEnd; i++ {
offs[i] = pos
// Calculate the size of each block by re-reading what we wrote.
// This is a simplification; a real implementation would track sizes.
}
offs[blkEnd] = uint32(out.Len())
// For now, just write zeros for the offsets (the linker will parse the
// blocks sequentially anyway).
for i := 0; i <= blkEnd; i++ {
le.PutUint32(out.Bytes()[hdrStart+4*i:], offs[i])
}
return out.Bytes(), nil
} }
// RISC-V relocation types (cmd/internal/objabi). // RISC-V relocation types (cmd/internal/objabi). The Go linker applies
// R_RISCV_PCREL_ITYPE/STYPE to an AUIPC + I/S-type instruction pair as a
// single 8-byte field; R_RISCV_JAL covers a single 4-byte J-type instruction.
const ( const (
relocRISCVPcrelHi20 = 23 relocRISCVJal = 59 // R_RISCV_JAL
relocRISCVPcrelLo12I = 24 relocRISCVPcrelItype = 62 // R_RISCV_PCREL_ITYPE
relocRISCVPcrelLo12S = 25 relocRISCVPcrelStype = 63 // R_RISCV_PCREL_STYPE
) )
// toolchainObjectPreambleRISCV returns the RISC-V object preamble. // toolchainObjectPreambleRISCV returns the "go object ...\n!\n" header
// the installed go tool asm writes for riscv64, captured by assembling a
// one-instruction probe (see toolchainObjectPreamble).
var ( var (
preambleRISCVOnce sync.Once preambleRISCVOnce sync.Once
preambleRISCV []byte preambleRISCV []byte
+326
View File
@@ -0,0 +1,326 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
// the symbol tables, the function code bytes and the relocation wiring.
func TestGOObjectLOONG64Structure(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
GLOBL ·table<>(SB), RODATA, $8
DATA ·table<>+0(SB)/8, $0x1122334455667788
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.GOObjectLOONG64("testpkg", "k_loong64.s")
if err != nil {
t.Fatalf("GOObjectLOONG64: %v", err)
}
v := openGoobj(t, obj)
// Package defs: the static GLOBL, then the FuncInfo and the two DWARF
// symbols (debug_line program, subprogram DIE).
defs := v.syms(blkSymdef)
if len(defs) != 4 {
t.Fatalf("symdefs = %d, want 4", len(defs))
}
if defs[0].name != "table" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 8 {
t.Errorf("table symbol = %+v", defs[0])
}
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
t.Errorf("funcinfo symbol = %+v", defs[1])
}
if defs[2].name != "" || defs[2].typ != kindSDWARFLINES || defs[2].size == 0 {
t.Errorf("lines symbol = %+v", defs[2])
}
if defs[3].name != "" || defs[3].typ != kindSDWARFFCN || defs[3].size == 0 {
t.Errorf("DIE symbol = %+v", defs[3])
}
// Non-package defs: four pc tables and the function.
nps := v.syms(blkNonpkgdef)
if len(nps) != 5 {
t.Fatalf("nonpkgdefs = %d, want 5", len(nps))
}
fn := nps[4]
if fn.name != "testpkg.add" || fn.typ != kindSTEXT || fn.flag != symFlagNoSplit || fn.size != 20 {
t.Errorf("add symbol = %+v", fn)
}
// The function code: 20 bytes, the ground-truth encoding. It sits
// after the GLOBL, FuncInfo, two DWARF symbols and four pc tables.
dataIdx := v.blk(blkDataIdx)
dataBlk := v.blk(blkData)
le := binary.LittleEndian
dOff := le.Uint32(dataIdx[8*4:])
code := dataBlk[dOff : dOff+20]
want := []byte{
0x64, 0x20, 0xc0, 0x28, // ld.d r4, 8(r3)
0x65, 0x40, 0xc0, 0x28, // ld.d r5, 16(r3)
0x84, 0x94, 0x10, 0x00, // add.d r4, r4, r5
0x64, 0x60, 0xc0, 0x29, // st.d r4, 24(r3)
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
}
for i := range want {
if code[i] != want[i] {
t.Fatalf("code byte %d = %02x, want %02x", i, code[i], want[i])
}
}
// The debug_line program: LNE_set_address (the R_ADDR relocation
// carries the function address), then one row per line change — the
// TEXT is on line 4 (a leading blank line precedes the include), the
// instructions on lines 5–9 — an advance to the 20-byte end and an
// end-of-sequence.
linesOff := le.Uint32(dataIdx[4*2:])
lines := dataBlk[linesOff : linesOff+21]
wantLines := []byte{
0x00, 0x09, 0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, // LNE_set_address
0x13, // pc 0, line 5
0x38, // pc 4, line 6
0x38, // pc 8, line 7
0x38, // pc 12, line 8
0x38, // pc 16, line 9
0x02, 0x04, // advance_pc to 20
0x00, 0x01, 0x01, // end_sequence
}
for i := range wantLines {
if lines[i] != wantLines[i] {
t.Fatalf("lines byte %d = %02x, want %02x", i, lines[i], wantLines[i])
}
}
// The subprogram DIE: abbrev 3 (FUNCTION), the qualified name, the
// addrx low_pc slot (R_DWTXTADDR_U4), the size as high_pc, the
// call-frame-CFA frame base, decl file/line and the external flag.
dieOff := le.Uint32(dataIdx[4*3:])
die := dataBlk[dieOff : dieOff+27]
wantDie := []byte{
0x03,
't', 'e', 's', 't', 'p', 'k', 'g', '.', 'a', 'd', 'd', 0,
0x00, 0x00, 0x00, 0x00, // low_pc: addrx slot
0x14, // high_pc: 20
0x01, 0x9c, // frame_base: DW_OP_call_frame_cfa
0x01, 0x00, 0x00, 0x00, // decl_file: 1
0x04, // decl_line: 4
0x01, // external
0x00, // end of children
}
for i := range wantDie {
if die[i] != wantDie[i] {
t.Fatalf("DIE byte %d = %02x, want %02x", i, die[i], wantDie[i])
}
}
// The DWARF symbols carry the function-address references: R_ADDR for
// the line program's set_address, R_DWTXTADDR_U4 for the DIE's addrx
// slot, both against the function's non-package index. The reloc
// index counts relocations, not bytes.
relocIdx := v.blk(blkRelocIdx)
relocs := v.blk(blkReloc)
if le.Uint32(relocIdx[4*2:]) != 0 || le.Uint32(relocIdx[4*3:]) != 1 || le.Uint32(relocIdx[4*4:]) != 2 {
t.Fatalf("dwarf reloc index ranges: %d %d %d", le.Uint32(relocIdx[4*2:]), le.Uint32(relocIdx[4*3:]), le.Uint32(relocIdx[4*4:]))
}
lr := relocs[:23]
if int32(le.Uint32(lr[0:])) != 3 || lr[4] != 8 || le.Uint16(lr[5:]) != relocAddr ||
le.Uint32(lr[15:]) != pkgIdxNone || le.Uint32(lr[19:]) != 4 {
t.Errorf("lines reloc = %x", lr)
}
dr := relocs[23:46]
if int32(le.Uint32(dr[0:])) != 13 || dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4() ||
le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 {
t.Errorf("die reloc = %x", dr)
}
// The pc-value deltas are in MinLC (4) units: the flat pcsp covers
// the whole 20-byte function with a delta of 5.
pcspOff := le.Uint32(dataIdx[4*4:])
if got := dataBlk[pcspOff : pcspOff+3]; !bytes.Equal(got, []byte{0x02, 0x05, 0x00}) {
t.Errorf("pcsp = %x, want 020500", got)
}
}
// TestGOObjectLOONG64Link cross-compiles a Go program with the gasm-produced
// object substituted into the package archive, proving cmd/link accepts the
// emitted GOOBJ. The binary is not executed (no LoongArch host or qemu).
// Skipped when no Go toolchain is available.
func TestGOObjectLOONG64Link(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
asmSrc := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
`
if err := os.WriteFile(filepath.Join(dir, "main_loong64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
mainSrc := `package main
func add(a, b int64) int64
func main() {
if add(20, 22) != 42 {
panic("bad add")
}
}
`
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module l64link\n\ngo 1.21\n"), 0o644); err != nil {
t.Fatal(err)
}
// Capture the cross build (GOARCH=loong64): the package archive and the
// link line.
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
build.Env = append(os.Environ(), "GOARCH=loong64")
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var pkgArch, work, linkLine, asmObj string
for _, line := range strings.Split(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_loong64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if pkgArch == "" || linkLine == "" || asmObj == "" {
t.Skip("could not locate the archive, asm output or link line in the build log")
}
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
// The archive member holding the assembler's output is named after the
// asm object file (main_loong64.o), as cmd/go packs it with `pack r`.
asmMember := filepath.Base(strings.ReplaceAll(asmObj, "$WORK", work))
// Assemble the same source with gasm and swap the object in.
pf, perrs := parser.Parse(filepath.Join(dir, "main_loong64.s"), asmSrc)
if len(perrs) > 0 {
t.Fatalf("parse: %v", perrs)
}
pimg, err := AssembleFileLOONG64(pf)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := pimg.GOObjectLOONG64("main", filepath.Join(dir, "main_loong64.s"))
if err != nil {
t.Fatalf("GOObjectLOONG64: %v", err)
}
// Extract the archive, substitute the object member, repack.
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
extract.Dir = membersDir
extract.Env = append(os.Environ(), "GOARCH=loong64")
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
// Substitute the gasm object for the assembler's archive member (pack
// extracts members read-only).
member := filepath.Join(membersDir, asmMember)
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, obj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listCmd.Env = append(os.Environ(), "GOARCH=loong64")
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for _, m := range strings.Fields(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, filepath.Join(membersDir, m))
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
pack.Env = append(os.Environ(), "GOARCH=loong64")
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
// Re-link with our archive in place of the toolchain's. The link line
// carries a GOROOT assignment and $WORK placeholders; run it through the
// shell with the GOEXPERIMENT and GOARCH the toolchain expects (the
// linker compares the object header against its own, experiments
// included).
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
link := exec.Command("sh", "-c", linkLine)
link.Dir = dir
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)), "GOARCH=loong64")
if out, err := link.CombinedOutput(); err != nil {
t.Fatalf("link with gasm object: %v\n%s", err, out)
}
// The binary is not executed: there is no LoongArch host or qemu here.
// The link itself and the symbol table prove cmd/link accepted the gasm
// object and laid out the function.
nm := exec.Command(goBin, "tool", "nm", filepath.Join(dir, "app2"))
nm.Env = append(os.Environ(), "GOARCH=loong64")
nmOut, err := nm.CombinedOutput()
if err != nil {
t.Fatalf("nm gasm-linked binary: %v\n%s", err, nmOut)
}
if !strings.Contains(string(nmOut), "main.add") {
t.Errorf("main.add not found in linked binary:\n%s", nmOut)
}
}
+83 -7
View File
@@ -89,11 +89,14 @@ func (fl *FuncLayout) LineAt(offset int) int {
type RelocKind int type RelocKind int
const ( const (
RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64) RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64)
RelPCRelHI20 // R_RISCV_PCREL_HI20 (AUIPC) RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
RelPCRelLO12 // R_RISCV_PCREL_LO12_I (ADDI, LD) RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
RelPCRelLO12S // R_RISCV_PCREL_LO12_S (SD) RelRISCVJal // R_RISCV_JAL (J-type call)
RelPCRelAbs // 32-bit absolute (R_RISCV_32) RelPCRelAbs // 32-bit absolute (R_RISCV_32)
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
RelArm64Addr // R_ADDRARM64 (ADRP + ADD/LDR/STR pair)
) )
type Reloc struct { type Reloc struct {
@@ -248,7 +251,7 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
if !ok { if !ok {
continue continue
} }
code, labels, relocs, err := assembleRISCV(t) code, labels, relocs, lines, spadj, err := assembleRISCV(t)
if err != nil { if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err) return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
} }
@@ -262,6 +265,8 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
Args: argsSize(t), Args: argsSize(t),
Line: t.Pos().Line, Line: t.Pos().Line,
Labels: labels, Labels: labels,
Lines: lines,
Spadj: spadj,
Relocs: relocs, Relocs: relocs,
} }
for _, f := range t.Flags { for _, f := range t.Flags {
@@ -289,7 +294,78 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
img.DataSyms = append(img.DataSyms, DataSymbol{ img.DataSyms = append(img.DataSyms, DataSymbol{
Name: d.name, Name: d.name,
Pkg: d.pkg, Pkg: d.pkg,
Offset: pos, Offset: len(img.Data) - len(d.buf), // relative to the data section
Size: d.size,
Static: d.static,
Rodata: d.rodata,
Dupok: d.dupok,
})
}
return img, nil
}
// AssembleFileLOONG64 assembles every TEXT function of a parsed loong64 file
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
// code. SB references in the code are encoded as pcalau12i pairs with zero
// immediates; the object-file emitters record R_LOONG64_ADDR_HI/LO
// relocations for the linker.
func AssembleFileLOONG64(f *ast.File) (*Image, error) {
dataSyms, err := collectData(f)
if err != nil {
return nil, err
}
img := &Image{Symbols: map[string]int{}}
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
continue
}
code, labels, relocs, lines, spadj, err := assembleLOONG64(t)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
fl := FuncLayout{
Name: t.Name.Name,
Pkg: t.Name.Pkg,
Static: t.Name.Static,
Offset: len(img.Code),
Size: len(code),
Frame: frameSize(t),
Args: argsSize(t),
Line: t.Pos().Line,
Labels: labels,
Lines: lines,
Spadj: spadj,
Relocs: relocs,
}
for _, f := range t.Flags {
switch f {
case "NOSPLIT":
fl.NoSplit = true
case "SPWRITE":
fl.SPWrite = true
}
}
img.Funcs = append(img.Funcs, fl)
img.Code = append(img.Code, code...)
}
// Lay out the data section behind the code, 16-aligned.
dataStart := len(img.Code)
for _, d := range dataSyms {
pos := dataStart + len(img.Data)
for pos%16 != 0 {
img.Data = append(img.Data, 0)
pos++
}
img.Symbols[d.name] = pos
img.Data = append(img.Data, d.buf...)
img.DataSyms = append(img.DataSyms, DataSymbol{
Name: d.name,
Pkg: d.pkg,
Offset: len(img.Data) - len(d.buf), // relative to the data section
Size: d.size, Size: d.size,
Static: d.static, Static: d.static,
Rodata: d.rodata, Rodata: d.rodata,
File diff suppressed because it is too large Load Diff
+595
View File
@@ -0,0 +1,595 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// loong64 (LoongArch) instruction encoding.
//
// The encoder is data-driven: each mnemonic maps to an instruction format and
// an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own loong64
// backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm`
// exactly — the ground-truth oracle for the verify suite.
//
// All LoongArch instructions are 32 bits, little-endian. The formats used
// here (per the LoongArch Volume I specification):
//
// 3R opcode[31:15] | rk[4:0] | rj[4:0] | rd[4:0]
// 2R opcode[31:15] | rj[4:0] | rd[4:0]
// 2RI12 opcode[31:22] | si12[11:0] | rj[4:0] | rd[4:0]
// 2RI14 opcode[31:18] | si14[13:0] | rj[4:0] | rd[4:0]
// 2RI16 opcode[31:22] | si16[15:0] | rj[4:0] | rd[4:0]
// 2RI20 opcode[31:25] | si20[19:0] | rd[4:0]
// 1RI21 opcode[31:26] | si21[20:0] | rj[4:0] (BEQZ/BNEZ, B*Z, BC*Z)
// B/BL opcode[31:26] | offs[25:0]
// 4R opcode[31:20] | r1[4:0] | r2[4:0] | r3[4:0] | r4[4:0]
// IRIR opcode[31:22] | msb[4:0] | rj[4:0] | lsb[4:0] | rd[4:0]
// 3RI2 opcode[31:17] | sa2[1:0] | rk[4:0] | rj[4:0] | rd[4:0]
//
// The opcode constants are pre-positioned (they include the zero bit ranges
// of the immediate and register fields), mirroring the toolchain's OP_*
// helpers, so each l64* function only ORs its fields in.
// loong64RegNum returns the 5-bit register number for a LoongArch register
// name: R0–R31 (integer), F0–F31 (floating point), FCC0–FCC7 (condition
// flags), FCSR0–FCSR31 (control/status) and the ABI aliases the runtime's
// assembly uses. Returns -1 for an unrecognised name.
func loong64RegNum(name string) int {
switch name {
case "R0", "ZERO":
return 0
case "R1", "RA", "LINK":
return 1
case "R2", "TP":
return 2
case "R3", "SP":
return 3
case "R4", "A0":
return 4
case "R5", "A1":
return 5
case "R6", "A2":
return 6
case "R7", "A3":
return 7
case "R8", "A4":
return 8
case "R9", "A5":
return 9
case "R10", "A6":
return 10
case "R11", "A7":
return 11
case "R12", "T0":
return 12
case "R13", "T1":
return 13
case "R14", "T2":
return 14
case "R15", "T3":
return 15
case "R16", "T4":
return 16
case "R17", "T5":
return 17
case "R18", "T6":
return 18
case "R19", "T7":
return 19
case "R20", "T8":
return 20
case "R21":
return 21
case "R22", "G", "g", "FP":
return 22
case "R23", "S0":
return 23
case "R24", "S1":
return 24
case "R25", "S2":
return 25
case "R26", "S3":
return 26
case "R27", "S4":
return 27
case "R28", "S5":
return 28
case "R29", "S6", "CTXT":
return 29
case "R30", "S7", "TMP":
return 30
case "R31", "S8":
return 31
}
// F0–F31, FCC0–FCC7, FCSR0–FCSR31.
if len(name) >= 4 && name[:4] == "FCSR" {
return loong64RegSpecial(name[4:], "FCSR", 31)
}
if len(name) >= 3 && name[:3] == "FCC" {
return loong64RegSpecial(name[3:], "FCC", 7)
}
if len(name) < 2 {
return -1
}
prefix, digits := name[:1], name[1:]
if digits[0] < '0' || digits[0] > '9' {
return -1
}
n := 0
for i := 0; i < len(digits); i++ {
if digits[i] < '0' || digits[i] > '9' {
return -1
}
n = n*10 + int(digits[i]-'0')
}
if prefix == "F" && n <= 31 {
return n
}
return -1
}
// loong64RegSpecial parses a numbered FCC/FCSR register.
func loong64RegSpecial(digits, prefix string, max int) int {
if digits == "" {
return -1
}
n := 0
for i := 0; i < len(digits); i++ {
if digits[i] < '0' || digits[i] > '9' {
return -1
}
n = n*10 + int(digits[i]-'0')
}
if n <= max {
return n
}
return -1
}
// ---- format helpers ----
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
func l64rrr(op uint32, rk, rj, rd int) uint32 {
return op | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64rr encodes a 2R instruction: op | rj<<5 | rd.
func l64rr(op uint32, rj, rd int) uint32 {
return op | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64irr encodes a 2RI12 instruction: op | si12<<10 | rj<<5 | rd.
func l64irr(op uint32, imm, rj, rd int) uint32 {
return op | (uint32(imm)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64irr14 encodes a 2RI14 instruction: op | si14<<10 | rj<<5 | rd.
func l64irr14(op uint32, imm, rj, rd int) uint32 {
return op | (uint32(imm)&0x3FFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64irr16 encodes a 2RI16 instruction: op | si16<<10 | rj<<5 | rd.
func l64irr16(op uint32, imm, rj, rd int) uint32 {
return op | (uint32(imm)&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64ir encodes a 2RI20 instruction: op | si20<<5 | rd.
func l64ir(op uint32, imm, rd int) uint32 {
return op | (uint32(imm)&0xFFFFF)<<5 | uint32(rd&0x1f)
}
// l64bbl encodes a B/BL instruction: op | offs[25:0], where offs is the
// 4-byte-aligned word distance (the toolchain stores the shifted value).
func l64bbl(op uint32, offs int) uint32 {
return op | (uint32(offs)&0xFFFF)<<10 | (uint32(offs)>>16)&0x3FF
}
// l64ir21 encodes a 1RI21 branch (BEQZ/BNEZ, BLTZ/BGEZ/BLEZ/BGTZ, BFPT/BFPF):
// op | si21[15:0]<<10 | rj<<5 | si21[20:16].
func l64ir21(op uint32, offs, rj int) uint32 {
v := uint32(offs)
return op | (v&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | (v>>16)&0x1F
}
// l64rrrr encodes a 4R instruction: op | r1<<15 | r2<<10 | r3<<5 | r4.
func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
return op | uint32(r1&0x1f)<<15 | uint32(r2&0x1f)<<10 | uint32(r3&0x1f)<<5 | uint32(r4&0x1f)
}
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
// The msb/lsb fields are 6 bits wide (0–63) and are validated by the caller.
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
}
// l64irrr encodes a 3RI2 instruction (ALSL): op | sa<<15 | rk<<10 | rj<<5 | rd.
func l64irrr(op uint32, sa, rk, rj, rd int) uint32 {
return op | uint32(sa&0x3)<<15 | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64i15 encodes a no-operand system instruction with a 15-bit code field
// (SYSCALL, BREAK, DBAR): op | code[14:0].
func l64i15(op uint32, code int) uint32 {
return op | uint32(code)&0x7FFF
}
// l64irr5i encodes PRELD: op | offs<<10 | rj<<5 | hint.
func l64irr5i(op uint32, offs, rj, hint int) uint32 {
return op | (uint32(offs)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(hint&0x1f)
}
// l64wordLE encodes a uint32 as 4 little-endian bytes.
func l64wordLE(w uint32) []byte {
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
}
// l64WordsLE concatenates one or more instruction words as little-endian bytes.
func l64WordsLE(ws ...uint32) []byte {
var out []byte
for _, w := range ws {
out = append(out, l64wordLE(w)...)
}
return out
}
// ---- instruction formats ----
type l64Format uint8
const (
l64Frrr l64Format = iota // 3R (integer and FP arithmetic)
l64Frr // 2R
l64Firr // 2RI12 (arithmetic with 12-bit immediate)
l64Firr14 // 2RI14 (ldptr/stptr)
l64Firr16 // 2RI16 (addu16i.d)
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub)
l64Firir // bstrins/bstrpick
l64Firrr // alsl
l64Fi15 // syscall/break/dbar
l64Fam // atomic (3R with the AM field order)
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
l64Fshift // 2RI12 with a 5/6-bit shift immediate
l64Fpreld // preld (2RI12 + 5-bit hint)
)
// l64Enc is one instruction's encoding: its bit layout (format) and the
// opcode constant, positioned at its exact bit range.
type l64Enc struct {
format l64Format
op uint32
}
// l64DualEnc holds both forms of a dual-form mnemonic: the 3R register form
// and the 2RI12 immediate form (which is a shift for the shift mnemonics).
type l64DualEnc struct {
rrr uint32 // 3R register form
imm uint32 // 2RI12 immediate form
shift bool // the immediate form is a 5/6-bit shift amount
}
// l64DualTable maps the dual-form arithmetic/logic mnemonics to both
// encodings; the assembler picks by operand kind.
var l64DualTable = map[string]l64DualEnc{}
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered
// yet; the base integer, memory and floating-point ISA is complete.
var l64InstrTable = map[string]l64Enc{}
func init() {
// 3R — integer.
rrr := map[string]uint32{
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
"SGT": 0x24 << 15, "SGTU": 0x25 << 15,
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15,
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
"ORN": 0x2c << 15, "ANDN": 0x2d << 15,
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
"SLLV": 0x31 << 15, "SRLV": 0x32 << 15, "SRAV": 0x33 << 15,
"ROTR": 0x36 << 15, "ROTRV": 0x37 << 15,
"MUL": 0x38 << 15, "MULW": 0x38 << 15, "MULH": 0x39 << 15, "MULHU": 0x3a << 15,
"MULV": 0x3b << 15, "MULVU": 0x3b << 15, "MULHV": 0x3c << 15, "MULHVU": 0x3d << 15,
"MULWVW": 0x3e << 15, "MULWVWU": 0x3f << 15,
"DIV": 0x40 << 15, "DIVW": 0x40 << 15, "REM": 0x41 << 15, "REMW": 0x41 << 15,
"DIVU": 0x42 << 15, "DIVWU": 0x42 << 15, "REMU": 0x43 << 15, "REMWU": 0x43 << 15,
"DIVV": 0x44 << 15, "REMV": 0x45 << 15, "DIVVU": 0x46 << 15, "REMVU": 0x47 << 15,
"CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15,
"CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15,
}
// 3R — floating point.
rrr["MULF"] = 0x209 << 15
rrr["MULD"] = 0x20a << 15
rrr["DIVF"] = 0x20d << 15
rrr["DIVD"] = 0x20e << 15
rrr["SUBF"] = 0x205 << 15
rrr["SUBD"] = 0x206 << 15
rrr["ADDF"] = 0x201 << 15
rrr["ADDD"] = 0x202 << 15
rrr["CMPEQF"] = 0x0c1<<20 | 0x4<<15
rrr["CMPEQD"] = 0x0c2<<20 | 0x4<<15
rrr["CMPGED"] = 0x0c2<<20 | 0x7<<15
rrr["CMPGEF"] = 0x0c1<<20 | 0x7<<15
rrr["CMPGTD"] = 0x0c2<<20 | 0x3<<15
rrr["CMPGTF"] = 0x0c1<<20 | 0x3<<15
rrr["FMINF"] = 0x215 << 15
rrr["FMIND"] = 0x216 << 15
rrr["FMAXF"] = 0x211 << 15
rrr["FMAXD"] = 0x212 << 15
rrr["FMAXAF"] = 0x219 << 15
rrr["FMAXAD"] = 0x21a << 15
rrr["FMINAF"] = 0x21d << 15
rrr["FMINAD"] = 0x21e << 15
rrr["FSCALEBF"] = 0x221 << 15
rrr["FSCALEBD"] = 0x222 << 15
rrr["FCOPYSGF"] = 0x225 << 15
rrr["FCOPYSGD"] = 0x226 << 15
for m, op := range rrr {
l64InstrTable[m] = l64Enc{format: l64Frrr, op: op}
}
// 2R.
rr := map[string]uint32{
"CLOW": 0x4 << 10, "CLZW": 0x5 << 10, "CTOW": 0x6 << 10, "CTZW": 0x7 << 10,
"CLOV": 0x8 << 10, "CLZV": 0x9 << 10, "CTOV": 0xa << 10, "CTZV": 0xb << 10,
"REVB2H": 0xc << 10, "REVB4H": 0xd << 10, "REVB2W": 0xe << 10, "REVBV": 0xf << 10,
"REVH2W": 0x10 << 10, "REVHV": 0x11 << 10,
"BITREV4B": 0x12 << 10, "BITREV8B": 0x13 << 10, "BITREVW": 0x14 << 10, "BITREVV": 0x15 << 10,
"EXTWH": 0x16 << 10, "EXTWB": 0x17 << 10, "CPUCFG": 0x1b << 10,
"TRUNCFV": 0x46a9 << 10, "TRUNCDV": 0x46aa << 10, "TRUNCFW": 0x46a1 << 10, "TRUNCDW": 0x46a2 << 10,
"MOVWF": 0x4744 << 10, "MOVVF": 0x4746 << 10, "MOVWD": 0x4748 << 10, "MOVVD": 0x474a << 10,
"MOVFW": 0x46c1 << 10, "MOVDW": 0x46c2 << 10, "MOVFV": 0x46c9 << 10, "MOVDV": 0x46ca << 10,
"FRINTF": 0x4791 << 10, "FRINTD": 0x4792 << 10,
"MOVDF": 0x4646 << 10, "MOVFD": 0x4649 << 10,
"ABSF": 0x4501 << 10, "ABSD": 0x4502 << 10,
"MOVF": 0x4525 << 10, "MOVD": 0x4526 << 10,
"NEGF": 0x4505 << 10, "NEGD": 0x4506 << 10,
"SQRTF": 0x4511 << 10, "SQRTD": 0x4512 << 10,
"FLOGBF": 0x4509 << 10, "FLOGBD": 0x450a << 10,
"FCLASSF": 0x450d << 10, "FCLASSD": 0x450e << 10,
"FTINTRMWF": 0x4681 << 10, "FTINTRMWD": 0x4682 << 10,
"FTINTRMVF": 0x4689 << 10, "FTINTRMVD": 0x468a << 10,
"FTINTRPWF": 0x4691 << 10, "FTINTRPWD": 0x4692 << 10,
"FTINTRPVF": 0x4699 << 10, "FTINTRPVD": 0x469a << 10,
"FTINTRZWF": 0x46a1 << 10, "FTINTRZWD": 0x46a2 << 10,
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
}
for m, op := range rr {
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
}
// RDTIME is a 2R instruction with rd and rj in swapped positions.
l64InstrTable["RDTIMELW"] = l64Enc{format: l64Frdtime, op: 0x18 << 10}
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
// selected by the operand kind; the shift mnemonics pair the 3R form
// with a 5/6-bit shift immediate.
for m, e := range map[string]l64DualEnc{
"ADD": {rrr: 0x20 << 15, imm: 0x00a << 22},
"ADDW": {rrr: 0x20 << 15, imm: 0x00a << 22},
"ADDV": {rrr: 0x21 << 15, imm: 0x00b << 22},
"ADDVU": {rrr: 0x21 << 15, imm: 0x00b << 22},
"AND": {rrr: 0x29 << 15, imm: 0x00d << 22},
"OR": {rrr: 0x2a << 15, imm: 0x00e << 22},
"XOR": {rrr: 0x2b << 15, imm: 0x00f << 22},
"SGT": {rrr: 0x24 << 15, imm: 0x008 << 22},
"SGTU": {rrr: 0x25 << 15, imm: 0x009 << 22},
"SLL": {rrr: 0x2e << 15, imm: 0x00081 << 15, shift: true},
"SRL": {rrr: 0x2f << 15, imm: 0x00089 << 15, shift: true},
"SRA": {rrr: 0x30 << 15, imm: 0x00091 << 15, shift: true},
"ROTR": {rrr: 0x36 << 15, imm: 0x00099 << 15, shift: true},
"SLLV": {rrr: 0x31 << 15, imm: 0x0041 << 16, shift: true},
"SRLV": {rrr: 0x32 << 15, imm: 0x0045 << 16, shift: true},
"SRAV": {rrr: 0x33 << 15, imm: 0x0049 << 16, shift: true},
"ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true},
} {
l64DualTable[m] = e
}
// 2RI12 — pure immediate arithmetic (LU52ID has no register form).
l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22}
// ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16.
l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26}
// 2RI14 — LL/SC are aliased by the Go assembler to the pointer loads and
// stores (ldptr/stptr), with the offset scaled by 4.
l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w
l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d
l64InstrTable["SC"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w
l64InstrTable["SCW"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w
l64InstrTable["SCV"] = l64Enc{format: l64Firr14, op: 0x23 << 24} // sc.d
l64InstrTable["LL"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w)
l64InstrTable["LLW"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w)
l64InstrTable["LLV"] = l64Enc{format: l64Firr14, op: 0x22 << 24} // ldptr.d (ll.d)
// 2RI20.
l64InstrTable["LU12IW"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
l64InstrTable["LU32ID"] = l64Enc{format: l64Fir20, op: 0x0b << 25}
l64InstrTable["PCALAU12I"] = l64Enc{format: l64Fir20, op: 0x0d << 25}
l64InstrTable["PCADDU12I"] = l64Enc{format: l64Fir20, op: 0x0e << 25}
// LUI is the Plan 9 spelling of lu12i.w.
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
// 4R — fused multiply-add.
rrrr := map[string]uint32{
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
}
for m, op := range rrrr {
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
}
// IRIR — bit-field insert/extract.
irir := map[string]uint32{
"BSTRINSW": 0x3<<21 | 0x0<<15,
"BSTRINSV": 0x2 << 22,
"BSTRPICKW": 0x3<<21 | 0x1<<15,
"BSTRPICKV": 0x3 << 22,
}
for m, op := range irir {
l64InstrTable[m] = l64Enc{format: l64Firir, op: op}
}
// 3RI2 — ALSL.
irrr := map[string]uint32{
"ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17,
}
for m, op := range irrr {
l64InstrTable[m] = l64Enc{format: l64Firrr, op: op}
}
// 0-operand system instructions.
l64InstrTable["SYSCALL"] = l64Enc{format: l64Fi15, op: 0x56 << 15}
l64InstrTable["BREAK"] = l64Enc{format: l64Fi15, op: 0x54 << 15}
l64InstrTable["DBAR"] = l64Enc{format: l64Fi15, op: 0x70e4 << 15}
// PRELD.
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
// Atomics — 3R with the AM field order (rk=value, rj=address, rd=result).
am := map[string]uint32{
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
"AMCASB": 0x070B0 << 15, "AMCASH": 0x070B1 << 15,
"AMCASW": 0x070B2 << 15, "AMCASV": 0x070B3 << 15,
"AMADDW": 0x070C2 << 15, "AMADDV": 0x070C3 << 15,
"AMANDW": 0x070C4 << 15, "AMANDV": 0x070C5 << 15,
"AMORW": 0x070C6 << 15, "AMORV": 0x070C7 << 15,
"AMXORW": 0x070C8 << 15, "AMXORV": 0x070C9 << 15,
"AMMAXW": 0x070CA << 15, "AMMAXV": 0x070CB << 15,
"AMMINW": 0x070CC << 15, "AMMINV": 0x070CD << 15,
"AMMAXWU": 0x070CE << 15, "AMMAXVU": 0x070CF << 15,
"AMMINWU": 0x070D0 << 15, "AMMINVU": 0x070D1 << 15,
"AMSWAPDBB": 0x070BC << 15, "AMSWAPDBH": 0x070BD << 15,
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
}
for m, op := range am {
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
}
}
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
// register move between the integer and floating-point register banks — the
// MOVW/MOVV specials the Go assembler accepts.
var l64FpMovTable = map[string]uint32{
"MOVV.R.F": 0x452a << 10, // movgr2fr.d
"MOVV.R.FCC": 0x4536 << 10, // movgr2cf
"MOVV.R.FCSR": 0x4530 << 10, // movgr2fcsr
"MOVV.F.R": 0x452e << 10, // movfr2gr.d
"MOVV.F.FCC": 0x4534 << 10, // movfr2cf
"MOVV.FCC.R": 0x4537 << 10, // movcf2gr
"MOVV.FCC.F": 0x4535 << 10, // movcf2fr
"MOVV.FCSR.R": 0x4532 << 10, // movfcsr2gr
"MOVW.R.F": 0x4529 << 10, // movgr2fr.w
"MOVW.F.R": 0x452d << 10, // movfr2gr.s
}
// l64branchTable holds the 16-bit branch and jump encodings (2RI16).
var l64branchTable = map[string]uint32{
"BEQ": 0x16 << 26,
"BNE": 0x17 << 26,
"BLT": 0x18 << 26,
"BGE": 0x19 << 26,
"BLTU": 0x1a << 26,
"BGEU": 0x1b << 26,
"JIRL": 0x13 << 26,
}
// l64branch21Table holds the single-register branches with 21-bit offsets:
// the negative opcode constants the toolchain uses for the short forms.
var l64branch21Table = map[string]uint32{
"BEQZ": 0x10 << 26, // beq r0, rj → beqz
"BNEZ": 0x11 << 26, // bne r0, rj → bnez
"BLTZ": 0x18 << 26, // blt rj, r0 → bltz
"BGEZ": 0x19 << 26, // bge rj, r0 → bgez
"BGTZ": 0x18 << 26, // blt r0, rj → bgtz
"BLEZ": 0x19 << 26, // bge r0, rj → blez
"BFPT": 0x12<<26 | 0x1<<8,
"BFPF": 0x12<<26 | 0x0<<8,
}
// l64jumpTable maps the jump pseudo-instructions and their aliases to the
// B/BL opcode constants.
var l64jumpTable = map[string]uint32{
"JMP": 0x14 << 26, // b
"B": 0x14 << 26, // b
"JAL": 0x15 << 26, // bl
"CALL": 0x15 << 26, // bl
"BL": 0x15 << 26, // bl
}
// l64loadStoreTable maps the MOV width mnemonics to their load and store
// 2RI12 opcodes. The load opcode is the negated store opcode, exactly as
// the toolchain derives it.
var l64loadStoreTable = map[string]struct{ ld, st uint32 }{
"MOVB": {0x0a0 << 22, 0x0a4 << 22},
"MOVH": {0x0a1 << 22, 0x0a5 << 22},
"MOVW": {0x0a2 << 22, 0x0a6 << 22},
"MOVV": {0x0a3 << 22, 0x0a7 << 22},
"MOVBU": {0x0a8 << 22, 0x0a4 << 22},
"MOVHU": {0x0a9 << 22, 0x0a5 << 22},
"MOVWU": {0x0aa << 22, 0x0a6 << 22},
"MOVF": {0x0ac << 22, 0x0ad << 22},
"MOVD": {0x0ae << 22, 0x0af << 22},
}
// l64movRegTable maps a register-to-register MOV mnemonic to its expansion,
// matching the toolchain's case-1 encoding: MOVB → ext.w.b, MOVH → ext.w.h,
// MOVW → sll.w, MOVV → or, MOVBU → andi. MOVHU/MOVWU expand to bstrpick.d
// and are handled separately in the assembler.
type l64MovRegEnc struct {
rr bool // 2R format (ext.w.b/ext.w.h)
op uint32 // opcode constant (rr forms) or 3R/2RI12 opcode
imm int // 2RI12 immediate for MOVBU's andi
}
var l64movRegTable = map[string]l64MovRegEnc{
"MOVB": {true, 0x17 << 10, 0}, // ext.w.b rd, rj
"MOVH": {true, 0x16 << 10, 0}, // ext.w.h rd, rj
"MOVW": {false, 0x2e << 15, 0}, // sll.w rd, rj, r0
"MOVV": {false, 0x2a << 15, 0}, // or rd, rj, r0
"MOVBU": {false, 0x00d << 22, 0xff}, // andi rd, rj, $0xff
}
// l64movFpRegTable maps a floating-point register move mnemonic to its 2R
// opcode (fmov.s / fmov.d), used when both operands are F registers.
var l64movFpRegTable = map[string]uint32{
"MOVF": 0x4525 << 10,
"MOVD": 0x4526 << 10,
}
// l64RegClass discriminates integer (R), floating-point (F) and condition
// (FCC) registers for the MOV pseudo-instruction's register-move encoding.
type l64RegClass int
const (
l64ClsNone l64RegClass = iota
l64ClsGR
l64ClsFP
l64ClsFCC
l64ClsFCSR
)
// loong64RegClass reports the register class of a register operand name.
func loong64RegClass(name string) l64RegClass {
switch {
case name == "":
return l64ClsNone
case len(name) >= 3 && name[:3] == "FCC":
return l64ClsFCC
case len(name) >= 4 && name[:4] == "FCSR":
return l64ClsFCSR
case name[0] == 'F':
return l64ClsFP
default:
return l64ClsGR
}
}
+293
View File
@@ -0,0 +1,293 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// firstTextLOONG64 parses assembly source and returns the first TEXT body.
func firstTextLOONG64(t *testing.T, src string) *ast.Text {
t.Helper()
f, errs := parser.Parse("f_loong64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
for _, d := range f.Decls {
if fn, ok := d.(*ast.Text); ok {
return fn
}
}
t.Fatal("no TEXT found")
return nil
}
// assembleLOONG64Helper assembles one TEXT function and returns its bytes.
func assembleLOONG64Helper(t *testing.T, fn *ast.Text) []byte {
t.Helper()
code, _, _, _, _, err := assembleLOONG64(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
return code
}
// wantWords checks that code matches the expected little-endian words.
func wantWords(t *testing.T, code []byte, want ...uint32) {
t.Helper()
got := make([]uint32, 0, len(code)/4)
for i := 0; i+4 <= len(code); i += 4 {
got = append(got, binary.LittleEndian.Uint32(code[i:]))
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
func TestLOONG64_add(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
`)
code := assembleLOONG64Helper(t, fn)
// 5 instructions: two ld.d, add.d, st.d, jirl r0, r1, 0.
wantWords(t, code,
0x28C02064, // ld.d r4, 8(r3)
0x28C04065, // ld.d r5, 16(r3)
0x00109484, // add.d r4, r4, r5
0x29C06064, // st.d r4, 24(r3)
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_arithmetic(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·arith(SB), NOSPLIT, $0
ADDV R4, R5, R6
SUBV R7, R8, R9
MULV R10, R11, R12
DIVV R13, R14, R15
AND R16, R17, R18
OR R18, R19, R20
XOR R20, R21, R2
SLLV R2, R23, R24
SRLV R24, R25, R26
SRAV R26, R27, R28
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x001090A6, // add.d r6, r5, r4
0x00119D09, // sub.d r9, r8, r7
0x001DA96C, // mul.d r12, r11, r10
0x002235CF, // div.d r15, r14, r13
0x0014C232, // and r18, r17, r16
0x00154A74, // or r20, r19, r18
0x0015D2A2, // xor r2, r21, r20
0x00188AF8, // sll.d r24, r23, r2
0x0019633A, // srl.d r26, r25, r24
0x0019EB7C, // sra.d r28, r27, r26
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_immediates(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·imm(SB), NOSPLIT, $0
ADDV $42, R4, R5
ADDV $-8, R6
AND $0xff, R7, R8
OR $1, R9, R10
SGT $100, R13, R14
SLLV $4, R15, R16
MOVV $0x12345, R17
MOVV $0, R18
MOVW $0, R19
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x02C0A885, // addi.d r5, r4, 42
0x02FFE0C6, // addi.d r6, r6, -8
0x0343FCE8, // andi r8, r7, 0xff
0x0380052A, // ori r10, r9, 1
0x020191AE, // slti r14, r13, 100
0x004111F0, // slli.d r16, r15, 4
0x14000251, // lu12i.w r17, 0x12
0x038D1631, // ori r17, r17, 0x345
0x00150012, // or r18, r0, r0
0x00170013, // sll.w r19, r0, r0
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_loadStore(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·mem(SB), NOSPLIT, $0
MOVV (R4), R5
MOVV R5, (R6)
MOVW 8(R7), R8
MOVB R9, -4(R10)
MOVV (R11)(R12), R13
MOVV R14, (R15)(R16)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x28C00085, // ld.d r5, 0(r4)
0x29C000C5, // st.d r5, 0(r6)
0x288020E8, // ld.w r8, 8(r7)
0x293FF149, // st.b r9, -4(r10)
0x380C316D, // ldx.d r13, r11, r12
0x381C41EE, // stx.d r14, r15, r16
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_branches(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·br(SB), NOSPLIT, $0
BEQ R4, R5, done
BNE R6, R7, skip
BLT R8, R9, done
BGE R10, R11, done
BLTU R12, R13, done
BGEU R14, R15, done
skip:
JMP done
done:
RET
`)
code := assembleLOONG64Helper(t, fn)
// skip is at 0x18 (6 words), done at 0x1c.
wantWords(t, code,
0x58001C85, // beq r5, r4, +7
0x5C0018C7, // bne r7, r6, +6
0x60001509, // blt r9, r8, +5
0x6400114B, // bge r11, r10, +4
0x68000D8D, // bltu r13, r12, +3
0x6C0009CF, // bgeu r15, r14, +2
0x50000400, // b done (+1, chain-folded through skip)
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_frame(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $32-8
MOVV R4, R5
MOVV arg+0(FP), R6
MOVV R7, local-8(SP)
MOVV local-8(SP), R8
MOVV R9, ret+0(FP)
RET
`)
code := assembleLOONG64Helper(t, fn)
// autosize = align8(32+8) = 40; prologue stores LR at -40(SP),
// opens the frame, stores LR again at 0(SP). The function is a leaf
// (no calls), so the epilogue skips the LR restore. FP args are at
// autosize+8; SP locals at autosize+offset.
wantWords(t, code,
0x29FF6061, // st.d r1, -40(r3)
0x02FF6063, // addi.d r3, r3, -40
0x29C00061, // st.d r1, 0(r3)
0x00150085, // or r5, r4, r0
0x28C0C066, // ld.d r6, 48(r3) arg+0(FP) → 0+40+8
0x29C08067, // st.d r7, 32(r3) local-8(SP) → 40-8
0x28C08068, // ld.d r8, 32(r3)
0x29C0C069, // st.d r9, 48(r3) ret+0(FP) → 0+40+8
0x02C0A063, // addi.d r3, r3, 40
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_jumpChain(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·jc(SB), NOSPLIT, $0
JMP a
a:
JMP b
b:
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x50000800, // b +2 (a, chain-folded to b)
0x50000400, // b +1 (b)
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_dconClasses(t *testing.T) {
cases := []struct {
v int64
word int // expected word count
}{
{0x123456789, 3}, // lu12i.w + ori + lu32i.d
{-1, 2}, // addi.d + lu52i.d (the MOV path handles -1 earlier)
{0x1000000000000, 2}, // addi.w + lu32i.d
{0x123456789abcdef0, 4}, // full sequence
{0xFFFFFFFFF, 2}, // lu12i.w + ori
{0x1234567800000000, 3}, // addi.w + lu32i.d + lu52i.d
}
for _, c := range cases {
if n := len(l64DconMovWords(0, c.v)); n != c.word {
t.Errorf("0x%x: %d words, want %d", c.v, n, c.word)
}
}
}
func TestLOONG64_regNames(t *testing.T) {
cases := map[string]int{
"R0": 0, "R31": 31, "F0": 0, "F31": 31, "FCC0": 0, "FCC7": 7,
"FCSR0": 0, "FCSR3": 3, "ZERO": 0, "RA": 1, "SP": 3, "g": 22, "G": 22,
"R32": -1, "FCC8": -1, "X0": -1, "R": -1, "TMP": 30, "CTXT": 29,
}
for name, want := range cases {
if got := loong64RegNum(name); got != want {
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
}
}
}
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
// A spot-check that assembleLOONG64 emits the same bytes the Go
// toolchain does for a small kernel (the full comparison lives in
// verify's TestGroundTruthLOONG64).
src := `#include "textflag.h"
TEXT ·k(SB), NOSPLIT, $0-0
ADDV R4, R5, R6
MOVV $0x100000, R7
BEQ R6, R7, done
JMP done
done:
RET
`
fn := firstTextLOONG64(t, src)
code := assembleLOONG64Helper(t, fn)
want := []byte{
0xa6, 0x90, 0x10, 0x00, // add.d r6, r5, r4
0x07, 0x20, 0x00, 0x14, // lu12i.w r7, 0x100
0xc7, 0x08, 0x00, 0x58, // beq r7, r6, +2 (done)
0x00, 0x04, 0x00, 0x50, // b +1 (done)
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
+129
View File
@@ -0,0 +1,129 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// Loong64 frame mapping, matching the Go toolchain's loong64 backend.
//
// Go's loong64 functions have no frame pointer: FP and SP are synthetic
// registers resolved against the hardware stack pointer (R3) and the frame
// size. The return address lives in R1 (the link register).
//
// The autosize is the real stack adjustment: the declared local frame plus
// the 8 bytes for the saved link register, rounded up to a multiple of 8
// (the toolchain aligns frames with `if autosize&4 != 0 { autosize += 4 }`).
// A leaf function (no calls) with a zero frame gets no prologue at all.
//
// Prologue (autosize > 0), byte-identical to the toolchain:
//
// MOVV R1, -autosize(R3) // save LR below the new SP (traceback-safe)
// ADDV $-autosize, R3 // open the frame
// MOVV R1, 0(R3) // save LR again at SP (signal-safety)
//
// Epilogue: MOVV 0(R3), R1; ADDV $autosize, R3 (non-leaf only for the LR
// restore); the RET's jirl r0, r1, 0 follows.
// loong64FrameInfo holds the frame layout derived from a TEXT directive.
type loong64FrameInfo struct {
autosize int // the real SP adjustment (locals + saved LR, aligned)
frame int // the declared $framesize
args int // the declared -argsize
noSplit bool // the NOSPLIT flag
leaf bool // no call instructions in the body
}
// loong64ComputeFrame derives the frame layout for a TEXT function.
func loong64ComputeFrame(t *ast.Text) loong64FrameInfo {
fi := loong64FrameInfo{
frame: frameSize(t),
args: argsSize(t),
}
for _, f := range t.Flags {
if f == "NOSPLIT" {
fi.noSplit = true
}
}
fi.leaf = loong64IsLeaf(t)
if fi.frame != 0 {
fi.autosize = fi.frame + 8 // space for the saved LR
if fi.autosize&4 != 0 {
fi.autosize += 4
}
} else if !fi.leaf {
// A zero-frame non-leaf function still opens an 8-byte frame for LR.
fi.autosize = 8
}
return fi
}
// loong64IsLeaf reports whether a function contains no call instructions
// (JAL/BL/CALL), matching the toolchain's LEAF mark, which drives the frame
// and the epilogue shape.
func loong64IsLeaf(t *ast.Text) bool {
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "JAL", "CALL", "BL":
return false
}
}
return true
}
// loong64Prologue returns the prologue bytes for a loong64 function.
func loong64Prologue(fi loong64FrameInfo) []byte {
if fi.autosize == 0 {
return nil
}
addiD := l64DualTable["ADDV"].imm
return l64WordsLE(
l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, 3, 1), // MOVV R1, -autosize(R3)
l64irr(addiD, -fi.autosize, 3, 3), // ADDV $-autosize, R3
l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1), // MOVV R1, 0(R3)
)
}
// loong64Return returns the bytes for a RET: the epilogue (restore LR and
// deallocate the frame when present) followed by jirl r0, r1, 0.
func loong64Return(fi loong64FrameInfo) []byte {
var ws []uint32
if fi.autosize != 0 {
if !fi.leaf {
// MOVV 0(R3), R1 — restore the link register.
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1))
}
// ADDV $autosize, R3 — close the frame.
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
}
// jirl r0, r1, 0 — return.
ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0))
return l64WordsLE(ws...)
}
// loong64ResolvePseudo translates a pseudo-register memory reference into a
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
// x-N(SP) → (autosize - N)(SP). Returns base = -1 for an unresolvable
// reference (SB: static data, handled by the relocation path).
func loong64ResolvePseudo(sym *ast.Symbol, fi loong64FrameInfo) (base int, off int32) {
if sym == nil {
return -1, 0
}
switch sym.Pseudo {
case "FP":
return 3, int32(sym.Offset) + int32(fi.autosize) + 8
case "SP":
return 3, int32(fi.autosize) + int32(sym.Offset)
case "SB":
return -1, int32(sym.Offset)
}
return -1, 0
}
+367
View File
@@ -0,0 +1,367 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestLOONG64_sys exercises the no-operand system instructions and the
// bare-data pseudo-instructions. The words match `go tool asm`
// (GOARCH=loong64) for the same source.
func TestLOONG64_sys(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·sys(SB), NOSPLIT, $0
NOOP
UNDEF
WORD $0x12345678
SYSCALL $0x10
BREAK $0x20
DBAR $1
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x03400000, // andi r0, r0, 0 (NOOP)
0x002A0000, // break 0 (UNDEF)
0x12345678, // WORD
0x002B0010, // syscall 0x10
0x002A0020, // break 0x20
0x38720001, // dbar 1
0x4C000020, // jirl r0, r1, 0
)
}
// TestLOONG64_branches21 exercises the single-register branch forms: the
// 21-bit BEQZ/BNEZ/BLTZ/BGEZ and the rd-field BGTZ/BLEZ.
func TestLOONG64_branches21(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·b21(SB), NOSPLIT, $0
BEQZ R4, done
BNEZ R5, done
BLTZ R6, done
BGEZ R7, done
BGTZ R8, done
BLEZ R9, done
done:
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x40001880, // beqz r4, +6
0x440014A0, // bnez r5, +5
0x600010C0, // bltz r6, +4
0x64000CE0, // bgez r7, +3
0x60000808, // bgtz r8, +2 (register in the rd field)
0x64000409, // blez r9, +1
0x4C000020, // jirl r0, r1, 0
)
}
// TestLOONG64_fma exercises the four fused multiply-add forms (4 and 3
// operand spellings).
func TestLOONG64_fma(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·fma(SB), NOSPLIT, $0
FMADDD F0, F1, F2, F3
FMSUBD F4, F5, F6
FNMADDD F7, F8, F9, F10
FNMSUBD F11, F12, F13
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x08200443, // fmadd.d f3, f2, f1, f0
0x086214C6, // fmsub.d f6, f5, f5, f4
0x08A3A12A, // fnmadd.d f10, f9, f8, f7
0x08E5B1AD, // fnmsub.d f13, f12, f12, f11
0x4C000020,
)
}
// TestLOONG64_bitops exercises BSTRINS/BSTRPICK (the 6-bit msb/lsb fields)
// and ALSL (the sa−1 shift field).
func TestLOONG64_bitops(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·bits(SB), NOSPLIT, $0
BSTRINSW $3, R4, $0, R5
BSTRINSV $3, R4, $1, R6
BSTRPICKW $3, R4, $0, R5
BSTRPICKV $6, R7, $0, R8
ALSLW $1, R4, R5, R6
ALSLW $4, R7, R8, R9
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x00630085, // bstrins.w r5, r4, $3, $0
0x00830486, // bstrins.d r6, r4, $3, $1
0x00638085, // bstrpick.w r5, r4, $3, $0
0x00C600E8, // bstrpick.d r8, r7, $6, $0
0x00041486, // alsl.w r6, r5, r4, $1 (sa-1)
0x0005A0E9, // alsl.w r9, r8, r7, $4
0x4C000020,
)
}
// TestLOONG64_ptr exercises the 14-bit-offset memory forms (LL/SC/MOVWP/
// MOVVP with the offset scaled by 4) and PRELD.
func TestLOONG64_ptr(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·ptr(SB), NOSPLIT, $0
LLW 8(R14), R15
SCW R16, -4(R17)
MOVWP 16(R18), R19
MOVVP R20, 24(R21)
PRELD 32(R22), $0
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x200009CF, // ll.w r15, 8(r14)
0x21FFFE30, // sc.w r16, -4(r17)
0x24001253, // ldptr.w r19, 16(r18)
0x27001AB4, // stptr.d r20, 24(r21)
0x2AC082C0, // preld 32(r22), 0
0x4C000020,
)
}
// TestLOONG64_atomics exercises the AM* read-modify-write forms and
// RDTIME, plus the MOVV FP→GP move.
func TestLOONG64_atomics(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·atoms(SB), NOSPLIT, $0
AMADDW R4, (R5), R6
RDTIMED R7, R8
MOVV F1, R2
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x386110A6, // amadd.w r6, r5, r4
0x000068E8, // rdtime.d r8, r7
0x0114B822, // movfr2gr.d r2, f1
0x4C000020,
)
}
// TestLOONG64_lu52 exercises the LU52I.D immediate form (a gasm extension
// the toolchain reaches only through its MOVV expansion).
func TestLOONG64_lu52(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·lu52(SB), NOSPLIT, $0
LU52ID $0x345, R10
LU52ID $0x123, R11, R12
ADDV16 $0x10000, R13
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x030D154A, // lu52i.d r10, r10, 0x345
0x03048D6C, // lu52i.d r12, r11, 0x123
0x100005AD, // addu16i.d r13, r13, 0x10000>>16
0x4C000020,
)
}
// TestLOONG64_sbRefs checks the static-symbol reference forms through the
// full file assembly: each pcalau12i+addi.d/ld/st pair carries the
// R_LOONG64_ADDR_HI/LO relocation pair, and the immediate fields are left
// zero for the linker.
func TestLOONG64_sbRefs(t *testing.T) {
f, errs := parser.Parse("sb_loong64.s", `#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0
MOVV $·table(SB), R4
MOVV ·table+8(SB), R5
MOVV R6, ·table(SB)
RET
GLOBL ·table(SB), RODATA, $8
DATA ·table+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
fn := img.Funcs[0]
if fn.Size != 28 {
t.Fatalf("function size = %d, want 28", fn.Size)
}
var hi, lo int
// The three references: $·table (0), ·table+8 (8), ·table (0).
wantAdd := []int64{0, 0, 8, 8, 0, 0}
for i, r := range fn.Relocs {
wantKind := RelLoong64AddrHi
wantOff := (i / 2) * 8
if i%2 == 1 {
wantKind = RelLoong64AddrLo
wantOff += 4
}
if r.Kind != wantKind || r.Off != wantOff || r.Name != "table" || r.Addend != wantAdd[i] {
t.Errorf("reloc %d = {kind %v off %d name %q addend %d}", i, r.Kind, r.Off, r.Name, r.Addend)
}
if r.Kind == RelLoong64AddrHi {
hi++
} else {
lo++
}
}
if hi != 3 || lo != 3 {
t.Errorf("relocs = %d hi + %d lo, want 3 + 3", hi, lo)
}
// The image carries the zero-immediate pair encodings (the linker
// fills the immediate fields from the relocations).
code := img.Code[fn.Offset : fn.Offset+fn.Size]
wantWords(t, code,
0x1A000004, // pcalau12i r4, 0
0x02C00084, // addi.d r4, r4, 0
0x1A00001E, // pcalau12i r30, 0
0x28C003C5, // ld.d r5, 0(r30)
0x1A00001E, // pcalau12i r30, 0
0x29C003C6, // st.d r6, 0(r30)
0x4C000020, // jirl r0, r1, 0
)
}
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
// invalid register operands and operand-count mismatches.
func TestLOONG64_errors(t *testing.T) {
cases := []string{
`TEXT ·e(SB), NOSPLIT, $0
JMP nowhere
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
BEQZ X0, done
done:
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
ADDV R4
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
FMADDD F0, F1
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
AMADDW R4, R5
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
WORD
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
PRELD 32(R4)
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
ALSLW $5, R4, R5, R6
RET
`,
}
for i, src := range cases {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_pcsp checks the stack-adjustment table of a framed function:
// the prologue raises the SP delta by autosize (in effect from the third
// instruction) and the RET's epilogue restores it to zero, with the pc deltas
// in MinLC (4) units — byte-identical to `go tool asm`.
func TestLOONG64_pcsp(t *testing.T) {
cases := []struct {
name string
src string
want []byte
}{
{
"leaf",
`#include "textflag.h"
TEXT ·leaf(SB), NOSPLIT, $8-0
MOVV R4, R5
RET
`,
[]byte{0x02, 0x02, 0x20, 0x03, 0x1f, 0x01, 0x00},
},
{
"nonleaf",
`#include "textflag.h"
TEXT ·nonleaf(SB), NOSPLIT, $8-0
MOVV R4, R5
JAL (R12)
RET
`,
[]byte{0x02, 0x02, 0x20, 0x05, 0x1f, 0x01, 0x00},
},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
f, errs := parser.Parse("pcsp_loong64.s", c.src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
if got := pcspTable(img.Funcs[0], 4); !bytes.Equal(got, c.want) {
t.Errorf("pcsp = % x, want % x", got, c.want)
}
})
}
}
// TestLOONG64_sbRefsUndefined checks that a reference to a symbol no GLOBL
// defines assembles into a relocation and is rejected at object emission.
func TestLOONG64_sbRefsUndefined(t *testing.T) {
f, errs := parser.Parse("sb_loong64.s", `#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0
MOVV missing(SB), R4
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
if len(img.Funcs[0].Relocs) != 2 {
t.Fatalf("relocs = %d, want the HI/LO pair", len(img.Funcs[0].Relocs))
}
if _, err := img.GOObjectLOONG64("p", "sb_loong64.s"); err == nil {
t.Error("expected an unknown-symbol error at emission")
}
}
// TestLOONG64_movImmToFp checks the immediate-to-FP move forms.
func TestLOONG64_movImmToFp(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·fpmov(SB), NOSPLIT, $0
MOVV $0x1, F0
MOVW $0x2, F4
RET
`)
code := assembleLOONG64Helper(t, fn)
want := []byte{
0x00, 0x04, 0x80, 0x03, // ori f0, r0, 1
0x04, 0x08, 0x80, 0x03, // ori f4, r0, 2
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
-258
View File
@@ -1,258 +0,0 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// This file emits Mach-O x86-64 objects (MH_OBJECT) from an assembled
// Image, in the shape the Darwin assembler produces: one unnamed segment
// carrying a __TEXT,__text and a __DATA,__data section laid out back to
// back at addresses zero and len(code), a symbol table (locals first, then
// exported definitions, then undefined externals) and one relocation entry
// per static-symbol reference, of type X86_64_RELOC_SIGNED.
//
// The image's own address space carries straight over — the data section
// starts immediately after the code, and the layout padding already lives
// inside Image.Data — so every symbol keeps its image address as its
// n_value, and a local (non-external) relocation leaves the displacement
// the assembler resolved in place: the linker only adjusts it by the
// section's final movement.
// Mach-O constants.
const (
machoMagic64 = 0xfeedfacf
machoCPUamd64 = 0x01000007 // CPU_TYPE_X86_64
machoCPUSubAll = 3 // CPU_SUBTYPE_X86_64_ALL
machoObj = 1 // MH_OBJECT
machoSegment64 = 0x19 // LC_SEGMENT_64
machoSymtab = 0x2 // LC_SYMTAB
machoSectTextFlags = 0x80000400 // S_ATTR_PURE_INSTRUCTIONS | S_ATTR_SOME_INSTRUCTIONS
nUndf = 0x00 // undefined symbol
nSect = 0x0e // defined in section number n_sect
nExt = 0x01 // external (exported or undefined-global) bit
x8664RelocSigned = 1
)
// MachOObject returns the image as a Mach-O x86-64 relocatable object
// (MH_OBJECT), the shape the Darwin toolchain links. Symbol names follow
// the same rules as the ELF output. Every static-symbol reference becomes
// an X86_64_RELOC_SIGNED relocation: external references against their
// undefined symbol, file-local ones against the __DATA section with the
// resolved displacement carried in the instruction bytes.
func (img *Image) MachOObject() ([]byte, error) {
le := binary.LittleEndian
// Section ordinals (1-based, as Mach-O numbers them).
const (
sectText = 1
sectData = 2
)
// Object address space: code at 0, data immediately after (the layout
// padding is already part of img.Data, so image addresses are object
// addresses).
textAddr := uint64(0)
dataAddr := uint64(len(img.Code))
vmsize := dataAddr + uint64(len(img.Data))
// The code, with external displacements primed to addend − 4: the
// linker adds the symbol's address to the field as it stands. Local
// displacements stay as the assembler resolved them.
code := append([]byte(nil), img.Code...)
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.External {
// Prime the field to the addend measured from the patch
// site: the assembler records it from the instruction end,
// After − Off bytes past the field.
copy(code[fn.Offset+r.Off:], le32(r.Addend-int64(r.After-r.Off)))
}
}
}
// Symbols: locals first, then exported definitions, then undefined
// externals — the order the classic link editor expects.
type machoSym struct {
name string
typ byte
sect byte
value uint64
}
var locals, globals, undefs []machoSym
for _, fn := range img.Funcs {
s := machoSym{name: objectName(fn.Pkg, fn.Name), typ: nSect, sect: sectText, value: textAddr + uint64(fn.Offset)}
if fn.Static {
locals = append(locals, s)
} else {
s.typ |= nExt
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := machoSym{name: objectName(d.Pkg, d.Name), typ: nSect, sect: sectData, value: dataAddr + uint64(d.Offset)}
if d.Static {
locals = append(locals, s)
} else {
s.typ |= nExt
globals = append(globals, s)
}
}
for _, name := range img.Externals {
undefs = append(undefs, machoSym{name: name, typ: nUndf | nExt})
}
syms := append(append(locals, globals...), undefs...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Relocations, attached to the __text section.
type machoReloc struct {
addr uint32
symnum uint32
extern bool
}
var relocs []machoReloc
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
rel := machoReloc{addr: uint32(fn.Offset + r.Off)}
if r.External {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
rel.symnum = uint32(idx)
rel.extern = true
} else {
// Section-relative: r_symbolnum carries the section number
// and the resolved displacement stays in the bytes.
rel.symnum = sectData
}
relocs = append(relocs, rel)
}
}
// The string table opens with the conventional " \0".
strtab := []byte{' ', 0}
strOff := map[string]int{}
for _, s := range syms {
if _, ok := strOff[s.name]; ok {
continue
}
strOff[s.name] = len(strtab)
strtab = append(strtab, s.name...)
strtab = append(strtab, 0)
}
// File layout: header, the two load commands, section data (code,
// data), the relocation table, the symbol table, the string table.
const (
hdrSize = 32
segCmdSize = 72 + 2*80 // segment command with two sections
symCmdSize = 24
)
sizeofcmds := segCmdSize + symCmdSize
dataOff := hdrSize + sizeofcmds
reloff := dataOff + len(code) + len(img.Data)
symoff := reloff + 8*len(relocs)
stroff := symoff + 16*len(syms)
out := make([]byte, stroff+len(strtab))
// mach_header_64.
le.PutUint32(out[0:], machoMagic64)
le.PutUint32(out[4:], machoCPUamd64)
le.PutUint32(out[8:], machoCPUSubAll)
le.PutUint32(out[12:], machoObj)
le.PutUint32(out[16:], 2) // ncmds
le.PutUint32(out[20:], uint32(sizeofcmds))
le.PutUint32(out[24:], 0) // flags
le.PutUint32(out[28:], 0) // reserved
// LC_SEGMENT_64 with the two sections.
p := hdrSize
le.PutUint32(out[p:], machoSegment64)
le.PutUint32(out[p+4:], segCmdSize)
// segname: the empty string, zero-padded to 16 bytes.
le.PutUint64(out[p+8:], 0)
le.PutUint64(out[p+16:], 0)
le.PutUint64(out[p+24:], 0) // vmaddr
le.PutUint64(out[p+32:], vmsize)
le.PutUint64(out[p+40:], uint64(dataOff))
le.PutUint64(out[p+48:], vmsize)
le.PutUint32(out[p+56:], 7) // maxprot rwx
le.PutUint32(out[p+60:], 7) // initprot rwx
le.PutUint32(out[p+64:], 2) // nsects
le.PutUint32(out[p+68:], 0) // flags
// __TEXT,__text
s := p + 72
copy(out[s:], "__text")
copy(out[s+16:], "__TEXT")
le.PutUint64(out[s+32:], textAddr)
le.PutUint64(out[s+40:], uint64(len(code)))
le.PutUint32(out[s+48:], uint32(dataOff))
le.PutUint32(out[s+52:], 4) // align 2^4
le.PutUint32(out[s+56:], uint32(reloff))
le.PutUint32(out[s+60:], uint32(len(relocs)))
le.PutUint32(out[s+64:], machoSectTextFlags)
// __DATA,__data
s += 80
copy(out[s:], "__data")
copy(out[s+16:], "__DATA")
le.PutUint64(out[s+32:], dataAddr)
le.PutUint64(out[s+40:], uint64(len(img.Data)))
le.PutUint32(out[s+48:], uint32(dataOff+len(code)))
le.PutUint32(out[s+52:], 4) // align 2^4
// LC_SYMTAB.
p = hdrSize + segCmdSize
le.PutUint32(out[p:], machoSymtab)
le.PutUint32(out[p+4:], symCmdSize)
le.PutUint32(out[p+8:], uint32(symoff))
le.PutUint32(out[p+12:], uint32(len(syms)))
le.PutUint32(out[p+16:], uint32(stroff))
le.PutUint32(out[p+20:], uint32(len(strtab)))
// Section data.
copy(out[dataOff:], code)
copy(out[dataOff+len(code):], img.Data)
// Relocation entries.
for i, r := range relocs {
e := out[reloff+i*8:]
le.PutUint32(e[0:], r.addr)
bits := r.symnum & 0x00ffffff
bits |= 1 << 24 // r_pcrel
bits |= 2 << 25 // r_length = 4 bytes
if r.extern {
bits |= 1 << 27 // r_extern
}
bits |= x8664RelocSigned << 28
le.PutUint32(e[4:], bits)
}
// nlist_64 entries.
for i, s := range syms {
e := out[symoff+i*16:]
le.PutUint32(e[0:], uint32(strOff[s.name]))
e[4] = s.typ
e[5] = s.sect
le.PutUint16(e[6:], 0) // n_desc
le.PutUint64(e[8:], s.value)
}
// String table.
copy(out[stroff:], strtab)
return out, nil
}
-127
View File
@@ -1,127 +0,0 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/macho"
"encoding/binary"
"testing"
)
// TestMachOObject checks the structure of the emitted MH_OBJECT: the two
// sections and their addresses, the symbol table (types, sections, values)
// and the __text relocation entries, parsed back with debug/macho. No
// Darwin toolchain is available on the test hosts, so the check is
// structural — the ELF output carries the end-to-end link-and-run proof of
// the shared symbol and relocation model.
func TestMachOObject(t *testing.T) {
img := elfTestImage(t)
obj, err := img.MachOObject()
if err != nil {
t.Fatalf("MachOObject: %v", err)
}
f, err := macho.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer f.Close()
if f.Type != macho.TypeObj {
t.Errorf("file type = %v, want MH_OBJECT", f.Type)
}
if f.Cpu != macho.CpuAmd64 {
t.Errorf("cpu = %v, want CpuAmd64", f.Cpu)
}
text := f.Section("__text")
data := f.Section("__data")
if text == nil || data == nil {
t.Fatal("missing __text or __data section")
}
if text.Addr != 0 || text.Size != uint64(len(img.Code)) {
t.Errorf("__text addr/size = %#x/%d, want 0/%d", text.Addr, text.Size, len(img.Code))
}
if data.Addr != uint64(len(img.Code)) {
t.Errorf("__data addr = %#x, want %#x", data.Addr, len(img.Code))
}
// Symbol table: locals, exported definitions, undefined externals.
syms := f.Symtab.Syms
byName := map[string]macho.Symbol{}
for _, s := range syms {
byName[s.Name] = s
}
wantSym := func(name string, typ, sect uint8, value uint64) {
t.Helper()
s, ok := byName[name]
if !ok {
t.Errorf("symbol %q not found", name)
return
}
if s.Type != typ || s.Sect != sect || s.Value != value {
t.Errorf("%s: type/sect/value = %#x/%d/%#x, want %#x/%d/%#x",
name, s.Type, s.Sect, s.Value, typ, sect, value)
}
}
const (
defined = nSect | nExt
local = nSect
undefined = nUndf | nExt
)
wantSym("addq", defined, 1, 0)
wantSym("getanswer", defined, 1, 5)
wantSym("useextern", defined, 1, 13)
answer := byName["answer"]
if answer.Type != local || answer.Sect != 2 {
t.Errorf("answer: type/sect = %#x/%d, want %#x/2", answer.Type, answer.Sect, local)
}
wantSym("extvar", undefined, 0, 0)
// Relocations: both X86_64_RELOC_SIGNED, PC-relative, 4 bytes wide.
// The local one carries its section number in Value, the external one
// its symbol number.
if len(text.Relocs) != 2 {
t.Fatalf("__text relocs = %d, want 2", len(text.Relocs))
}
var sawLocal, sawExternal bool
for _, r := range text.Relocs {
if !r.Pcrel || r.Len != 2 || r.Type != x8664RelocSigned {
t.Errorf("reloc at %#x: pcrel/len/type = %v/%d/%d", r.Addr, r.Pcrel, r.Len, r.Type)
}
switch {
case r.Extern:
if name := syms[r.Value].Name; name != "extvar" {
t.Errorf("external reloc at %#x names %q, want extvar", r.Addr, name)
}
sawExternal = true
default:
if r.Value != 2 { // __data, the second section
t.Errorf("local reloc at %#x: section %d, want 2 (__data)", r.Addr, r.Value)
}
sawLocal = true
}
}
if !sawLocal || !sawExternal {
t.Errorf("relocs seen: local=%v external=%v, want both", sawLocal, sawExternal)
}
// The __text bytes are the image code, with the external displacement
// primed to addend − 4 and the local one left resolved.
textData, err := text.Data()
if err != nil {
t.Fatal(err)
}
want := append([]byte(nil), img.Code...)
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.Name == "extvar" {
binary.LittleEndian.PutUint32(want[fn.Offset+r.Off:], 0xfffffffc) // −4
}
}
}
if !bytes.Equal(textData, want) {
t.Errorf("__text bytes %x, want %x", textData, want)
}
}
+386 -149
View File
@@ -5,17 +5,25 @@ package asm
import ( import (
"fmt" "fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-devkit/ast"
) )
// assembleRISCV assembles a RISC-V TEXT function body into machine code. // assembleRISCV assembles a RISC-V TEXT function body into machine code.
// It handles the full RV64IMAFDC instruction set including RVC compression. // It handles the full RV64IMAFDC instruction set including RVC compression.
func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, error) { func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
fi := riscvComputeFrame(t) fi := riscvComputeFrame(t)
prologue := riscvPrologue(fi) prologue := riscvPrologue(fi)
var relocs []Reloc var relocs []Reloc
var spadj []SpadjStep
// The prologue raises the SP delta by autosize; the boundary is reported
// at the pc just past its ADDI, exactly as the toolchain's pctospadj does.
if fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: riscvPrologueSpadjPC(fi), Value: fi.autosize})
}
// Pass 1: collect instructions and compute label offsets assuming 4 bytes // Pass 1: collect instructions and compute label offsets assuming 4 bytes
// per instruction (or 8 for MOV $large-imm). No encoding yet. // per instruction (or 8 for MOV $large-imm). No encoding yet.
@@ -33,7 +41,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, error) {
offsets[s.Name.Text] = pos offsets[s.Name.Text] = pos
case *ast.Instr: case *ast.Instr:
recs = append(recs, instrRec{instr: s}) recs = append(recs, instrRec{instr: s})
pos += riscvInstrSize(s) pos += riscvInstrSize(s, fi)
} }
} }
@@ -42,7 +50,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, error) {
for i := range recs { for i := range recs {
code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil) // no relocs in Pass 2 code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil) // no relocs in Pass 2
if err != nil { if err != nil {
return nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err) return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err)
} }
recs[i].code = code recs[i].code = code
pc += len(code) pc += len(code)
@@ -78,36 +86,52 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, error) {
out := append([]byte(nil), prologue...) out := append([]byte(nil), prologue...)
pc = len(prologue) pc = len(prologue)
preCount := len(relocs) preCount := len(relocs)
var lines []LineEntry
for _, r := range recs { for _, r := range recs {
lines = append(lines, LineEntry{Offset: pc, Line: r.instr.Pos().Line})
if r.compressed && !isBranchLike(r.instr.Mnemonic.Text) { if r.compressed && !isBranchLike(r.instr.Mnemonic.Text) {
out = append(out, r.code...) out = append(out, r.code...)
pc += len(r.code) pc += len(r.code)
} else { } else {
code, err := encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs) code, err := encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs)
if err != nil { if err != nil {
return nil, nil, nil, err return nil, nil, nil, nil, nil, err
} }
if c16, ok := tryCompressRVC(r.instr, fi); ok { if c16, ok := tryCompressRVC(r.instr, fi); ok {
code = []byte{byte(c16), byte(c16 >> 8)} code = []byte{byte(c16), byte(c16 >> 8)}
} }
// Make newly added relocation offsets absolute (subtract prologue to make // Make newly added relocation offsets function-relative. Each
// them function-relative, then the caller adds fn.Offset). // instruction records its reloc offset relative to its own start;
// the current pc is that instruction's offset from the function
// start (which includes the prologue). After is the address just
// past the relocated field, shifted by the same amount.
for j := preCount; j < len(relocs); j++ { for j := preCount; j < len(relocs); j++ {
relocs[j].Off += pc - len(prologue) relocs[j].Off += pc
relocs[j].After += pc
} }
preCount = len(relocs) preCount = len(relocs)
// The RET's epilogue closes the frame: the SP delta returns to zero
// after its ADDI (restore LR + ADDI).
if strings.ToUpper(r.instr.Mnemonic.Text) == "RET" && fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: pc + riscvReturnEpilogueLen(fi), Value: 0})
}
out = append(out, code...) out = append(out, code...)
pc += len(code) pc += len(code)
} }
} }
return out, offsets, relocs, nil return out, offsets, relocs, lines, spadj, nil
} }
// riscvInstrSize returns the encoded size in bytes of a RISC-V instruction. // riscvInstrSize returns the encoded size in bytes of a RISC-V instruction.
// Most instructions are 4 bytes; MOV with a large immediate is 8 (LUI+ADDIW). // Most instructions are 4 bytes; MOV with a large immediate and I-type
func riscvInstrSize(instr *ast.Instr) int { // arithmetic with a large immediate expand to several (possibly compressed)
// instructions.
func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
mnem := instr.Mnemonic.Text mnem := instr.Mnemonic.Text
ops := instr.Operands ops := instr.Operands
if mnem == "RET" {
return len(riscvReturn(fi))
}
if mnem == "MOV" && len(ops) == 2 { if mnem == "MOV" && len(ops) == 2 {
// MOV $sym(SB), rd → 8 bytes (AUIPC + ADDI). // MOV $sym(SB), rd → 8 bytes (AUIPC + ADDI).
if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil && ops[0].Imm.Sym.Pseudo == "SB" { if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil && ops[0].Imm.Sym.Pseudo == "SB" {
@@ -121,14 +145,15 @@ func riscvInstrSize(instr *ast.Instr) int {
if isMemOperand(ops[1]) && ops[1].Addr.Sym != nil && ops[1].Addr.Sym.Pseudo == "SB" { if isMemOperand(ops[1]) && ops[1].Addr.Sym != nil && ops[1].Addr.Sym.Pseudo == "SB" {
return 8 return 8
} }
// MOV $imm, rd → large immediate needs LUI+ADDIW. // MOV $imm, rd → size depends on the immediate and RVC compression.
if isImmOperand(ops[0]) { if isImmOperand(ops[0]) && ops[0].Imm.Sym == nil {
imm := immFromOperand(ops[0]) return riscvMovImmSize(regFromOperand(ops[1]), immFromOperand(ops[0]))
if imm < -2048 || imm > 2047 {
return 8
}
} }
} }
// I-type arithmetic with a large immediate expands to several instructions.
if (mnem == "ADDI" || mnem == "ANDI" || mnem == "ORI" || mnem == "XORI") && len(ops) >= 1 && isImmOperand(ops[0]) {
return riscvItypeImmediateSize(mnem, immFromOperand(ops[0]))
}
return 4 return 4
} }
@@ -151,36 +176,27 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
// Handle pseudo-instructions and special cases first. // Handle pseudo-instructions and special cases first.
switch mnem { switch mnem {
case "RET": case "RET":
// RET = JALR X0, 0(X1) // RET = epilogue (restore LR and close the frame when present) +
word = riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0) // uncompressed JALR X0, 0(X1) (the toolchain never compresses RET).
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return riscvReturn(fi), nil
case "CALL": case "CALL":
// CALL target → AUIPC X1, %pcrel_hi + JALR X1, %pcrel_lo(X1). // CALL sym(SB) → JAL X1, sym(SB) with a single R_RISCV_JAL
// For now, emit AUIPC X1, 0 + JALR X1, 0(X1) with zero offsets. // relocation. The Go assembler rejects CALL to a local branch label.
// The relocation system will fill the actual offsets. if len(ops) != 1 {
if len(ops) >= 1 { return nil, fmt.Errorf("CALL expects 1 operand, got %d", len(ops))
target := labelFromOperand(ops[0])
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
offset := int32(targetOff - pc)
// AUIPC X1, upper 20 bits
hi := (offset + 0x800) >> 12
word1 := riscvUType(riscvEnc{0x17, 0x0, 0x00}, 1, hi<<12)
// JALR X1, lower 12 bits(X1)
lo := offset - (hi << 12)
word2 := riscvIType(riscvEnc{0x67, 0x0, 0x00}, 1, 1, lo)
var out []byte
out = append(out, byte(word1), byte(word1>>8), byte(word1>>16), byte(word1>>24))
out = append(out, byte(word2), byte(word2>>8), byte(word2>>16), byte(word2>>24))
return out, nil
} }
// CALL with no target: encode as NOP (unsupported). op := ops[0]
word = riscvIType(riscvEnc{0x13, 0x0, 0x00}, 0, 0, 0) if op.Addr.Sym == nil || op.Addr.Sym.Pseudo != "SB" {
return nil, fmt.Errorf("CALL: local branch target is not supported (use CALL sym(SB))")
}
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelRISCVJal, Addend: op.Addr.Sym.Offset})
}
word = riscvJType(1, 0) // JAL X1, 0 — the linker fills the offset
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
case "JMP": case "JMP":
// JMP = JAL X0, target. Try C.J compression. // JMP = JAL X0, target. The Go assembler never compresses this to
// C.J, so always emit the 32-bit JAL.
var target string var target string
if len(ops) >= 1 { if len(ops) >= 1 {
target = labelFromOperand(ops[0]) target = labelFromOperand(ops[0])
@@ -190,11 +206,6 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
} }
offset := int32(targetOff - pc) offset := int32(targetOff - pc)
// C.J: funct3=0x5, offset in ±2 KB, bit 0 must be 0.
if offset >= -2048 && offset <= 2046 && offset%2 == 0 {
c16 := rvcCJ(0x5, offset)
return []byte{byte(c16), byte(c16 >> 8)}, nil
}
word = riscvJType(0, offset) word = riscvJType(0, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
case "JAL": case "JAL":
@@ -211,11 +222,6 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
} }
offset := int32(targetOff - pc) offset := int32(targetOff - pc)
// JAL X0, target → C.J when offset fits.
if rd == 0 && offset >= -2048 && offset <= 2046 && offset%2 == 0 {
c16 := rvcCJ(0x5, offset)
return []byte{byte(c16), byte(c16 >> 8)}, nil
}
word = riscvJType(rd, offset) word = riscvJType(rd, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
@@ -304,26 +310,44 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
} }
switch { switch {
// R-type: Plan 9 order is INSTR src1, src2, dst (destination last). // R-type: Go reverses the ISA order, writing rs2, rs1, rd (destination
// last); the two-operand form INSTR rs2, rd uses rd as rs1.
case len(ops) == 3 && isRTypeInstr(mnem): case len(ops) == 3 && isRTypeInstr(mnem):
rs1 := regFromOperand(ops[0]) // source 1 (first operand) rs2 := regFromOperand(ops[0]) // first operand = rs2
rs2 := regFromOperand(ops[1]) // source 2 (second operand) rs1 := regFromOperand(ops[1]) // second operand = rs1
rd := regFromOperand(ops[2]) // destination (last operand) rd := regFromOperand(ops[2]) // destination (last operand)
if rd < 0 || rs1 < 0 || rs2 < 0 { if rd < 0 || rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem) return nil, fmt.Errorf("invalid register in %s", mnem)
} }
word = riscvRType(enc, rd, rs1, rs2) word = riscvRType(enc, rd, rs1, rs2)
// I-type shift (SLLI, SRLI, SRAI): INSTR rs, $shamt, rd. case len(ops) == 2 && isRTypeInstr(mnem):
rs2 := regFromOperand(ops[0]) // source (first operand)
rd := regFromOperand(ops[1]) // destination (second operand)
if rd < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
word = riscvRType(enc, rd, rd, rs2)
// I-type shift (SLLI, SRLI, SRAI): INSTR $shamt, rs1, rd; the two-operand
// form INSTR $shamt, rd uses rd as the source.
case len(ops) == 3 && isShiftImmInstr(mnem): case len(ops) == 3 && isShiftImmInstr(mnem):
rs1 := regFromOperand(ops[0]) shamt := int(immFromOperand(ops[0]))
shamt := int(immFromOperand(ops[1])) rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2]) rd := regFromOperand(ops[2])
if rd < 0 || rs1 < 0 { if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem) return nil, fmt.Errorf("invalid register in %s", mnem)
} }
word = riscvRType(enc, rd, rs1, shamt) word = riscvRType(enc, rd, rs1, shamt)
case len(ops) == 2 && isShiftImmInstr(mnem):
shamt := int(immFromOperand(ops[0]))
rd := regFromOperand(ops[1])
if rd < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
word = riscvRType(enc, rd, rd, shamt)
// AMO atomics: Plan 9 order is INSTR src, (addr), dst. // AMO atomics: Plan 9 order is INSTR src, (addr), dst.
case len(ops) == 3 && isAMOInstr(mnem): case len(ops) == 3 && isAMOInstr(mnem):
rs2 := regFromOperand(ops[0]) // source value rs2 := regFromOperand(ops[0]) // source value
@@ -334,10 +358,10 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
} }
word = riscvAMOType(enc, rd, rs1, rs2) word = riscvAMOType(enc, rd, rs1, rs2)
// FP arithmetic: Plan 9 order is INSTR src1, src2, dst. // FP arithmetic: Go reverses the ISA order, writing rs2, rs1, rd.
case len(ops) == 3 && isFPArithInstr(mnem): case len(ops) == 3 && isFPArithInstr(mnem):
rs1 := regFromOperand(ops[0]) rs2 := regFromOperand(ops[0])
rs2 := regFromOperand(ops[1]) rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2]) rd := regFromOperand(ops[2])
if rd < 0 || rs1 < 0 || rs2 < 0 { if rd < 0 || rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid FP register in %s", mnem) return nil, fmt.Errorf("invalid FP register in %s", mnem)
@@ -390,25 +414,34 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
} }
word = riscvAMOType(enc, rd, rs1, rs2) word = riscvAMOType(enc, rd, rs1, rs2)
// FP compare: INSTR src1, src2, dst(int) — result in integer register. // FP compare: Go reverses the ISA order, writing rs2, rs1, rd.
case len(ops) == 3 && isFPCmpInstr(mnem): case len(ops) == 3 && isFPCmpInstr(mnem):
rs1 := regFromOperand(ops[0]) rs2 := regFromOperand(ops[0])
rs2 := regFromOperand(ops[1]) rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2]) rd := regFromOperand(ops[2])
if rd < 0 || rs1 < 0 || rs2 < 0 { if rd < 0 || rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem) return nil, fmt.Errorf("invalid operand in %s", mnem)
} }
word = riscvRType(enc, rd, rs1, rs2) word = riscvRType(enc, rd, rs1, rs2)
// I-type with immediate: Plan 9 order is INSTR src, imm, dst. // I-type with immediate: Plan 9 order is INSTR $imm, rs1, rd; the
// two-operand form INSTR $imm, rd uses rd as the source.
case len(ops) == 3 && isITypeInstr(mnem): case len(ops) == 3 && isITypeInstr(mnem):
rs1 := regFromOperand(ops[0]) // source register imm := immFromOperand(ops[0]) // immediate
imm := immFromOperand(ops[1]) // immediate rs1 := regFromOperand(ops[1]) // source register
rd := regFromOperand(ops[2]) // destination rd := regFromOperand(ops[2]) // destination
if rd < 0 || rs1 < 0 { if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem) return nil, fmt.Errorf("invalid register in %s", mnem)
} }
word = riscvIType(enc, rd, rs1, imm) return encodeRISCVItypeImmediate(mnem, enc, rd, rs1, imm)
case len(ops) == 2 && isITypeInstr(mnem):
imm := immFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rd < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
return encodeRISCVItypeImmediate(mnem, enc, rd, rd, imm)
// Loads: rd, offset(rs1) — Plan 9 order is LD src, dst. // Loads: rd, offset(rs1) — Plan 9 order is LD src, dst.
case len(ops) == 2 && isLoadInstr(mnem): case len(ops) == 2 && isLoadInstr(mnem):
@@ -442,18 +475,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return nil, fmt.Errorf("invalid register in %s", mnem) return nil, fmt.Errorf("invalid register in %s", mnem)
} }
// Try C.BEQZ / C.BNEZ compression. // The Go assembler never compresses branches to C.BEQZ/C.BNEZ.
if (mnem == "BEQ" || mnem == "BNE") && rs2 == 0 && isRVCIntReg(rs1) {
if cOff := offset; cOff >= -256 && cOff <= 254 && cOff%2 == 0 {
funct3 := uint32(0x6) // C.BEQZ
if mnem == "BNE" {
funct3 = 0x7 // C.BNEZ
}
c16 := rvcCB(funct3, rvcReg3(rs1), offset)
return []byte{byte(c16), byte(c16 >> 8)}, nil
}
}
word = riscvBType(enc, rs1, rs2, offset) word = riscvBType(enc, rs1, rs2, offset)
// U-type: rd, imm. // U-type: rd, imm.
@@ -585,62 +607,200 @@ func encodeRISCVMov(instr *ast.Instr, offsets map[string]int, fi riscvFrameInfo,
} }
} }
// encodeRISCVLoadImm encodes loading an immediate into a register. // encodeRISCVLoadImm encodes loading an immediate into a register (MOV $imm,
// For 12-bit immediates: ADDI $imm, ZERO, rd. // rd), matching the toolchain's instructionsForMOVConst. For 12-bit
// For larger: LUI $hi, rd + ADDIW $lo, rd, rd. // immediates it emits ADDI $imm, ZERO, rd (compressed to C.LI when it fits
// six signed bits); for larger immediates it emits LUI + [ADDIW], with the LUI
// and ADDIW compressed to C.LUI / C.ADDIW when their immediate fits.
func encodeRISCVLoadImm(rd int, imm int32) []byte { func encodeRISCVLoadImm(rd int, imm int32) []byte {
if imm >= -2048 && imm <= 2047 { if imm >= -2048 && imm <= 2047 {
word := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, 0, imm) if rd != 0 && imm >= -32 && imm <= 31 {
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)} return word16(rvcCI(0x2, uint32(rd), uint32(imm)&0x3F)) // C.LI
}
return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, 0, imm))
} }
// LUI + ADDIW for larger constants. low, high := splitRISCV32Imm(imm)
var out []byte var out []byte
hi := int32((uint32(imm)+0x800)>>12) << 12 // LUI loads upper 20 bits if rd != 0 && rd != 2 && high >= -32 && high <= 31 {
lo := imm - hi out = append(out, word16(rvcCI(0x3, uint32(rd), uint32(high)&0x3F))...) // C.LUI
wordLUI := riscvUType(riscvEnc{0x37, 0x0, 0x00}, rd, hi) } else {
out = append(out, byte(wordLUI), byte(wordLUI>>8), byte(wordLUI>>16), byte(wordLUI>>24)) out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, rd, high<<12))...)
if lo != 0 { }
wordADDIW := riscvIType(riscvEnc{0x1B, 0x0, 0x00}, rd, rd, lo) if low != 0 {
out = append(out, byte(wordADDIW), byte(wordADDIW>>8), byte(wordADDIW>>16), byte(wordADDIW>>24)) if low >= -32 && low <= 31 {
out = append(out, word16(rvcCI(0x1, uint32(rd), uint32(low)&0x3F))...) // C.ADDIW
} else {
out = append(out, wordLE(riscvIType(riscvEnc{0x1B, 0x0, 0x00}, rd, rd, low))...)
}
} }
return out return out
} }
// riscvMovImmSize returns the encoded byte length of MOV $imm, rd, mirroring
// encodeRISCVLoadImm's expansion and compression.
func riscvMovImmSize(rd int, imm int32) int {
if imm >= -2048 && imm <= 2047 {
if rd != 0 && imm >= -32 && imm <= 31 {
return 2 // C.LI
}
return 4 // ADDI
}
low, high := splitRISCV32Imm(imm)
size := 0
if rd != 0 && rd != 2 && high >= -32 && high <= 31 {
size += 2 // C.LUI
} else {
size += 4 // LUI
}
if low != 0 {
if low >= -32 && low <= 31 {
size += 2 // C.ADDIW
} else {
size += 4 // ADDIW
}
}
return size
}
// splitRISCV32Imm splits a signed 32-bit immediate into a signed 12-bit low
// part and a signed 20-bit high part, mirroring cmd/internal/obj/riscv's
// Split32BitImmediate. The high part is returned unshifted; callers place it
// in the upper bits of LUI (or its compressed C.LUI form).
func splitRISCV32Imm(imm int32) (low, high int32) {
if imm >= -2048 && imm <= 2047 {
return imm, 0
}
h := int64(imm) >> 12
if imm&(1<<11) != 0 {
h++
}
low = int32((int64(imm) << 52) >> 52) // sign extend 12 bits
high = int32((h << 44) >> 44) // sign extend 20 bits
return low, high
}
// encodeRISCVItypeImmediate encodes an I-type arithmetic instruction, expanding
// large immediates for ADDI/ANDI/ORI/XORI into LUI+ADDIW+op (or two ADDIs for
// ADDI), matching the Go assembler.
func encodeRISCVItypeImmediate(mnem string, enc riscvEnc, rd, rs1 int, imm int32) ([]byte, error) {
if imm >= -2048 && imm <= 2047 {
return wordLE(riscvIType(enc, rd, rs1, imm)), nil
}
var opMn string
switch mnem {
case "ADDI":
opMn = "ADD"
case "ANDI":
opMn = "AND"
case "ORI":
opMn = "OR"
case "XORI":
opMn = "XOR"
default:
return nil, fmt.Errorf("%s: immediate %d does not fit 12 bits", mnem, imm)
}
// ADDI with a small-ish immediate splits into two ADDIs.
if mnem == "ADDI" && imm >= -4096 && imm < 4095 {
imm0 := imm / 2
imm1 := imm - imm0
var out []byte
out = append(out, wordLE(riscvIType(enc, rd, rs1, imm0))...)
out = append(out, wordLE(riscvIType(enc, rd, rd, imm1))...)
return out, nil
}
// LUI $high, TMP; [ADDIW $low, TMP, TMP]; op TMP, rs1, rd. The LUI and
// ADDIW compress to their RVC forms (C.LUI / C.ADDIW) when the immediate
// fits 6 signed bits, matching the toolchain's compress pass.
low, high := splitRISCV32Imm(imm)
tmp := 31 // X31 = T6 = TMP
var out []byte
if high != 0 && high >= -32 && high <= 31 {
out = append(out, word16(rvcCI(0x3, uint32(tmp), uint32(high)&0x3F))...)
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, tmp, high<<12))...)
}
if low != 0 {
if low >= -32 && low <= 31 {
out = append(out, word16(rvcCI(0x1, uint32(tmp), uint32(low)&0x3F))...)
} else {
out = append(out, wordLE(riscvIType(riscvEnc{0x1B, 0x0, 0x00}, tmp, tmp, low))...)
}
}
opEnc, ok := riscvInstrTable[opMn]
if !ok {
return nil, fmt.Errorf("%s: unsupported operation %q", mnem, opMn)
}
out = append(out, wordLE(riscvRType(opEnc, rd, rs1, tmp))...)
return out, nil
}
// riscvItypeImmediateSize returns the encoded byte length of an I-type
// immediate instruction, accounting for the large-immediate expansion.
func riscvItypeImmediateSize(mnem string, imm int32) int {
if imm >= -2048 && imm <= 2047 {
return 4
}
switch mnem {
case "ADDI", "ANDI", "ORI", "XORI":
default:
return 4
}
if mnem == "ADDI" && imm >= -4096 && imm < 4095 {
return 8
}
low, high := splitRISCV32Imm(imm)
size := 4 // the R-type op (TMP is X31, never compressed)
if high != 0 && high >= -32 && high <= 31 {
size += 2 // C.LUI
} else {
size += 4 // LUI
}
if low != 0 {
if low >= -32 && low <= 31 {
size += 2 // C.ADDIW
} else {
size += 4 // ADDIW
}
}
return size
}
// encodeRISCVSBAddr emits AUIPC + ADDI to load the address of a static // encodeRISCVSBAddr emits AUIPC + ADDI to load the address of a static
// symbol into rd. Records R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I relocs. // symbol into rd, recording the single R_RISCV_PCREL_ITYPE relocation the Go
// toolchain uses for the pair (the object-file emitters expand or map it).
func encodeRISCVSBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte { func encodeRISCVSBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
name := sym.Name name := sym.Name
if relocs != nil { if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: name, Kind: RelPCRelHI20}) *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELIType, Addend: sym.Offset})
*relocs = append(*relocs, Reloc{Off: 4, After: 4, Name: name, Kind: RelPCRelLO12})
} }
auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, rd, 0) auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, rd, 0)
addi := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rd, 0) addi := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rd, 0)
return append(wordLE(auipc), wordLE(addi)...) return append(wordLE(auipc), wordLE(addi)...)
} }
// encodeRISCVSBLoad emits AUIPC + LD to load from a static symbol into rd. // encodeRISCVSBLoad emits AUIPC + LD to load from a static symbol into rd,
// Records R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I relocs. // recording the single R_RISCV_PCREL_ITYPE relocation for the pair.
func encodeRISCVSBLoad(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte { func encodeRISCVSBLoad(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
name := sym.Name name := sym.Name
if relocs != nil { if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: name, Kind: RelPCRelHI20}) *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELIType, Addend: sym.Offset})
*relocs = append(*relocs, Reloc{Off: 4, After: 4, Name: name, Kind: RelPCRelLO12})
} }
auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, rd, 0) auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, rd, 0)
ld := riscvIType(riscvEnc{0x03, 0x3, 0x00}, rd, rd, 0) ld := riscvIType(riscvEnc{0x03, 0x3, 0x00}, rd, rd, 0)
return append(wordLE(auipc), wordLE(ld)...) return append(wordLE(auipc), wordLE(ld)...)
} }
// encodeRISCVSBStore emits AUIPC + SD to store a register into a static symbol. // encodeRISCVSBStore emits AUIPC + SD to store a register into a static symbol,
// Records R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_S relocs. // recording the single R_RISCV_PCREL_STYPE relocation for the pair.
func encodeRISCVSBStore(sym *ast.Symbol, rs2 int, relocs *[]Reloc) []byte { func encodeRISCVSBStore(sym *ast.Symbol, rs2 int, relocs *[]Reloc) []byte {
tmp := 31 // X31 = T6 tmp := 31 // X31 = T6
name := sym.Name name := sym.Name
if relocs != nil { if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: name, Kind: RelPCRelHI20}) *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELSType, Addend: sym.Offset})
*relocs = append(*relocs, Reloc{Off: 4, After: 4, Name: name, Kind: RelPCRelLO12S})
} }
auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, tmp, 0) auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, tmp, 0)
sd := riscvSType(riscvEnc{0x23, 0x3, 0x00}, tmp, rs2, 0) sd := riscvSType(riscvEnc{0x23, 0x3, 0x00}, tmp, rs2, 0)
@@ -655,6 +815,11 @@ func wordLE(w uint32) []byte {
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)} return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
} }
// word16 encodes a uint16 as 2 little-endian bytes.
func word16(w uint16) []byte {
return []byte{byte(w), byte(w >> 8)}
}
// encodeRISCVJALR encodes the JALR indirect jump/call instruction. // encodeRISCVJALR encodes the JALR indirect jump/call instruction.
// Plan 9: JALR rs1, rd (2 regs) or JALR offset(rs1) (memory → rd=X1). // Plan 9: JALR rs1, rd (2 regs) or JALR offset(rs1) (memory → rd=X1).
func encodeRISCVJALR(instr *ast.Instr, fi riscvFrameInfo) ([]byte, error) { func encodeRISCVJALR(instr *ast.Instr, fi riscvFrameInfo) ([]byte, error) {
@@ -686,10 +851,6 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
ops := instr.Operands ops := instr.Operands
switch mnem { switch mnem {
case "RET":
// RET = JALR X0, 0(X1) → C.JR RA (CR-type: funct4=0x8, rd=0, rs2=1)
return rvcCR(0x8, 0, 1), true
case "LD", "MOV": case "LD", "MOV":
// LD rd, offset(SP) → C.LDSP when rd≠0 and uimm[8:3] fits. // LD rd, offset(SP) → C.LDSP when rd≠0 and uimm[8:3] fits.
// MOV name+off(FP), rd → load, same compression. // MOV name+off(FP), rd → load, same compression.
@@ -708,6 +869,10 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcLSP(0x3, uint32(rd), uint32(imm)), true return rvcLSP(0x3, uint32(rd), uint32(imm)), true
} }
// Register-relative C.LD: both in prime regs, 8-byte scaled offset.
if rs1 != -1 && rd != -1 && isRVCIntReg(rd) && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCL(0x3, rvcReg3(rd), rvcReg3(rs1), uint32(imm)), true
}
// MOV reg, mem → store, try C.SDSP. // MOV reg, mem → store, try C.SDSP.
if mnem == "MOV" && len(ops) == 2 && !isMemOperand(ops[0]) && isMemOperand(ops[1]) { if mnem == "MOV" && len(ops) == 2 && !isMemOperand(ops[0]) && isMemOperand(ops[1]) {
rs2, rs1, imm := extractSDParams(instr, fi) rs2, rs1, imm := extractSDParams(instr, fi)
@@ -722,16 +887,46 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcSSP(0x7, uint32(rs2), uint32(imm)), true return rvcSSP(0x7, uint32(rs2), uint32(imm)), true
} }
// Register-relative C.SD: base and source in prime regs.
if rs1 != -1 && rs2 != -1 && isRVCIntReg(rs1) && isRVCIntReg(rs2) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCS(0x7, rvcReg3(rs2), rvcReg3(rs1), uint32(imm)), true
}
case "LW":
rd, rs1, imm := extractLDParams(instr, fi)
if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 256 && imm%4 == 0 {
return rvcLSP(0x2, uint32(rd), uint32(imm)), true
}
if rs1 != -1 && rd != -1 && isRVCIntReg(rd) && isRVCIntReg(rs1) && imm >= 0 && imm < 128 && imm%4 == 0 {
return rvcCL(0x2, rvcReg3(rd), rvcReg3(rs1), uint32(imm)), true
}
case "SW":
rs2, rs1, imm := extractSDParams(instr, fi)
if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 256 && imm%4 == 0 {
return rvcSSP(0x6, uint32(rs2), uint32(imm)), true
}
if rs1 != -1 && rs2 != -1 && isRVCIntReg(rs1) && isRVCIntReg(rs2) && imm >= 0 && imm < 128 && imm%4 == 0 {
return rvcCS(0x6, rvcReg3(rs2), rvcReg3(rs1), uint32(imm)), true
}
case "ADDI": case "ADDI":
rd, rs1, imm := extractITypeParams(instr, fi) rd, rs1, imm := extractITypeParams(instr, fi)
if rd == -1 || rs1 == -1 { if rd == -1 || rs1 == -1 {
return 0, false return 0, false
} }
if rd == 2 && rs1 == 2 && imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
// C.ADDI16SP: ADDI to SP by a nonzero 16-byte multiple.
return rvcADDI16SP(2, imm), true
}
if rd == rs1 && rd != 0 && imm != 0 && imm >= -32 && imm <= 31 { if rd == rs1 && rd != 0 && imm != 0 && imm >= -32 && imm <= 31 {
// C.ADDI: funct3=0x0, rs1/rd, nzimm[5:0] // C.ADDI: funct3=0x0, rs1/rd, nzimm[5:0]
return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true
} }
if isRVCIntReg(rd) && rs1 == 2 && imm != 0 && imm >= 0 && imm < 1024 && imm%4 == 0 {
// C.ADDI4SPN: ADDI $imm, SP, rd for a prime rd.
return rvcCIW(0x0, rvcReg3(rd), uint32(imm)), true
}
if rs1 == 0 && rd != 0 && imm >= -32 && imm <= 31 { if rs1 == 0 && rd != 0 && imm >= -32 && imm <= 31 {
// C.LI: funct3=0x2, rd, imm[5:0] // C.LI: funct3=0x2, rd, imm[5:0]
return rvcCI(0x2, uint32(rd), uint32(imm)&0x3F), true return rvcCI(0x2, uint32(rd), uint32(imm)&0x3F), true
@@ -740,43 +935,44 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
// C.MV: funct4=0x8, rd, rs1 (CR-type) // C.MV: funct4=0x8, rd, rs1 (CR-type)
return rvcCR(0x8, uint32(rd), uint32(rs1)), true return rvcCR(0x8, uint32(rd), uint32(rs1)), true
} }
if rd == 0 && rs1 == 0 && imm == 0 {
case "JAL": // C.NOP
// JAL X0, target → C.J when offset fits in ±2KB. return 0x0001, true
if len(ops) >= 1 {
// For JAL with implicit rd=0 (JMP alias), check target.
// C.J: funct3=0x5
// Offset is computed at encode time — we can't check it here.
return 0, false
} }
case "JAL":
// JAL/JMP are never compressed to C.J by the Go assembler.
return 0, false
case "JMP": case "JMP":
// C.J — handled in encodeRISCVInstr with actual offset. // JAL/JMP are never compressed to C.J by the Go assembler.
return 0, false return 0, false
case "BEQ": case "BEQ":
// C.BEQZ — handled in encodeRISCVInstr with actual offset. // Branches are never compressed to C.BEQZ/C.BNEZ.
return 0, false return 0, false
case "BNE": case "BNE":
// C.BNEZ — handled in encodeRISCVInstr with actual offset. // Branches are never compressed to C.BEQZ/C.BNEZ.
return 0, false return 0, false
case "ADD": case "ADD":
// ADD rd, rs2 → C.ADD when rd == rs1 and both in prime regs (rd ≠ 0). // ADD rs2, rs1, rd → C.ADD (CR-type, funct4=0x9) when rd == rs1; ADD
// ADD is commutative: if rd == rs2, swap. // is commutative, so if rd == rs2, swap. ADD rs2, X0, rd is C.MV.
if len(ops) == 3 { if len(ops) == 3 {
rs1 := regFromOperand(ops[0]) rs2 := regFromOperand(ops[0])
rs2 := regFromOperand(ops[1]) rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2]) rd := regFromOperand(ops[2])
if rd != -1 && rs1 != -1 && rs2 != -1 && rd != 0 { if rd != -1 && rs1 != -1 && rs2 != -1 && rd != 0 {
if rd == rs1 && isRVCIntReg(rd) && isRVCIntReg(rs2) && rs2 != 0 { if rd == rs1 && rs2 != 0 {
// C.ADD: funct6=0x27, funct2=0x0 (CA-type) return rvcCR(0x9, uint32(rd), uint32(rs2)), true
return rvcCA(0x27, 0x0, rvcReg3(rd), rvcReg3(rs2)), true
} }
if rd == rs2 && isRVCIntReg(rd) && isRVCIntReg(rs1) && rs1 != 0 { if rd == rs2 && rs1 != 0 {
// Swap: C.ADD rd, rs1 return rvcCR(0x9, uint32(rd), uint32(rs1)), true
return rvcCA(0x27, 0x0, rvcReg3(rd), rvcReg3(rs1)), true }
if rs1 == 0 && rs2 != 0 {
// ADD rs2, X0, rd → C.MV rd, rs2.
return rvcCR(0x8, uint32(rd), uint32(rs2)), true
} }
} }
} }
@@ -795,13 +991,38 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
case "AND": case "AND":
funct2 = 0x3 funct2 = 0x3
} }
rs1 := regFromOperand(ops[0]) rs2 := regFromOperand(ops[0])
rs2 := regFromOperand(ops[1]) rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2]) rd := regFromOperand(ops[2])
if rd != -1 && rs1 != -1 && rs2 != -1 && rd != 0 { if rd != -1 && rs1 != -1 && rs2 != -1 && rd != 0 {
if rd == rs1 && isRVCIntReg(rd) && isRVCIntReg(rs2) && rs2 != 0 { if rd == rs1 && isRVCIntReg(rd) && isRVCIntReg(rs2) && rs2 != 0 {
return rvcCA(0x23, funct2, rvcReg3(rd), rvcReg3(rs2)), true return rvcCA(0x23, funct2, rvcReg3(rd), rvcReg3(rs2)), true
} }
// AND/OR/XOR are commutative; SUB is not.
if mnem != "SUB" && rd == rs2 && isRVCIntReg(rd) && isRVCIntReg(rs1) && rs1 != 0 {
return rvcCA(0x23, funct2, rvcReg3(rd), rvcReg3(rs1)), true
}
}
}
case "ADDW", "SUBW":
// C.ADDW (0x27,1) / C.SUBW (0x27,0) — CA-type, prime regs.
if len(ops) == 3 {
funct2 := uint32(0x0)
if mnem == "ADDW" {
funct2 = 0x1
}
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd != -1 && rs1 != -1 && rs2 != -1 && isRVCIntReg(rd) {
if rd == rs1 && isRVCIntReg(rs2) {
return rvcCA(0x27, funct2, rvcReg3(rd), rvcReg3(rs2)), true
}
// ADDW is commutative; SUBW is not.
if mnem == "ADDW" && isRVCIntReg(rs1) && rd == rs2 {
return rvcCA(0x27, funct2, rvcReg3(rd), rvcReg3(rs1)), true
}
} }
} }
@@ -811,6 +1032,10 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
if rs1 == 2 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { if rs1 == 2 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcLSP(0x1, uint32(rd), uint32(imm)), true return rvcLSP(0x1, uint32(rd), uint32(imm)), true
} }
// Register-relative C.FLD: rd in F8-F15, base in X8-X15.
if rs1 != -1 && rd != -1 && rd >= 8 && rd <= 15 && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCL(0x1, uint32(rd-8), rvcReg3(rs1), uint32(imm)), true
}
case "FSD": case "FSD":
// FSD rs2, imm(SP) → C.FSDSP (CSS-type, funct3=0x5). // FSD rs2, imm(SP) → C.FSDSP (CSS-type, funct3=0x5).
@@ -818,13 +1043,18 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcSSP(0x5, uint32(rs2), uint32(imm)), true return rvcSSP(0x5, uint32(rs2), uint32(imm)), true
} }
// Register-relative C.FSD: source in F8-F15, base in X8-X15.
if rs1 != -1 && rs2 != -1 && rs2 >= 8 && rs2 <= 15 && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCS(0x5, uint32(rs2-8), rvcReg3(rs1), uint32(imm)), true
}
case "LUI": case "LUI":
// LUI rd, imm → C.LUI when rd≠0, rd≠SP, imm nonzero and fits in 6 bits. // LUI rd, imm → C.LUI when rd≠0, rd≠SP, imm nonzero and fits in six
// signed bits (matching the toolchain's compress pass).
if len(ops) == 2 { if len(ops) == 2 {
rd := regFromOperand(ops[0]) rd := regFromOperand(ops[0])
imm := immFromOperand(ops[1]) imm := immFromOperand(ops[1])
if rd != -1 && rd != 0 && rd != 2 && imm != 0 && imm >= 1 && imm <= 63 { if rd != -1 && rd != 0 && rd != 2 && imm != 0 && imm >= -32 && imm <= 31 {
return rvcCI(0x3, uint32(rd), uint32(imm)&0x3F), true return rvcCI(0x3, uint32(rd), uint32(imm)&0x3F), true
} }
} }
@@ -836,33 +1066,32 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
} }
case "SLLI", "SRLI", "SRAI": case "SLLI", "SRLI", "SRAI":
// C.SLLI (funct3=0x0), C.SRLI (funct3=0x4, funct2=0), C.SRAI (funct3=0x4, funct2=1).
rd, rs1, imm := extractITypeParams(instr, fi) rd, rs1, imm := extractITypeParams(instr, fi)
if rd == rs1 && rd != 0 && imm != 0 && imm >= 1 && imm <= 63 { if rd == rs1 && rd != 0 && imm != 0 && imm >= 1 && imm <= 63 {
if mnem == "SLLI" { if mnem == "SLLI" {
// C.SLLI: funct3=0, CI-type with shamt in bits [12|6:2]. // C.SLLI: funct3=0, op=10 quadrant, shamt in bits [12|6:2].
// For simplicity, use the standard CI format — the shamt is in imm[5:0]. return rvcSLLI(uint32(rd), uint32(imm)&0x3F), true
return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true
} }
if isRVCIntReg(rd) { if isRVCIntReg(rd) {
funct2 := uint32(0x0) funct2 := uint32(0x0)
if mnem == "SRAI" { if mnem == "SRAI" {
funct2 = 0x1 funct2 = 0x1
} }
// CB-format shift: funct3=0x4, shamt in bits [12|6:2]. // C.SRLI/C.SRAI: CB-type, funct3=0x4.
// Use simplified encoding for now. return rvcCBShift(funct2, rvcReg3(rd), uint32(imm)&0x3F), true
_ = funct2
return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true
} }
} }
case "ANDI": case "ANDI":
rd, rs1, imm := extractITypeParams(instr, fi) rd, rs1, imm := extractITypeParams(instr, fi)
if isRVCIntReg(rd) && rd == rs1 && imm >= -32 && imm <= 31 { if isRVCIntReg(rd) && rd == rs1 && imm >= -32 && imm <= 31 {
// C.ANDI: funct3=0x4, funct2=0x2 (CB-type). // C.ANDI: CB-type, funct3=0x4, funct2=0x2.
// Simplified encoding for now. return rvcCBShift(0x2, rvcReg3(rd), uint32(imm)&0x3F), true
return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true
} }
case "EBREAK":
// C.EBREAK: CR-type, funct4=0x9, rd=0, rs2=0.
return rvcCR(0x9, 0, 0), true
} }
return 0, false return 0, false
@@ -899,15 +1128,23 @@ func extractSDParams(instr *ast.Instr, fi riscvFrameInfo) (rs2, rs1 int, imm int
return return
} }
// extractITypeParams extracts rd, rs1, and immediate for an I-type instruction. // extractITypeParams extracts rd, rs1, and immediate for an I-type
// instruction. The Plan 9 order is INSTR $imm, rs1, rd (3 operands) or
// INSTR $imm, rd (2 operands, rd is also the source).
func extractITypeParams(instr *ast.Instr, fi riscvFrameInfo) (rd, rs1 int, imm int32) { func extractITypeParams(instr *ast.Instr, fi riscvFrameInfo) (rd, rs1 int, imm int32) {
ops := instr.Operands ops := instr.Operands
if len(ops) != 3 { switch len(ops) {
case 3:
imm = immFromOperand(ops[0])
rs1 = regFromOperand(ops[1])
rd = regFromOperand(ops[2])
case 2:
imm = immFromOperand(ops[0])
rd = regFromOperand(ops[1])
rs1 = rd
default:
return -1, -1, 0 return -1, -1, 0
} }
rs1 = regFromOperand(ops[0])
imm = immFromOperand(ops[1])
rd = regFromOperand(ops[2])
return return
} }
+79 -52
View File
@@ -13,7 +13,7 @@ func riscvRegNum(name string) int {
// Numbered integer registers. // Numbered integer registers.
case "X0", "ZERO": case "X0", "ZERO":
return 0 return 0
case "X1", "RA": case "X1", "RA", "LR":
return 1 return 1
case "X2", "SP": case "X2", "SP":
return 2 return 2
@@ -21,9 +21,9 @@ func riscvRegNum(name string) int {
return 3 return 3
case "X4", "TP": case "X4", "TP":
return 4 return 4
case "X5", "T0", "LR": case "X5", "T0":
return 5 return 5
case "X6", "T1", "TMP": case "X6", "T1":
return 6 return 6
case "X7", "T2": case "X7", "T2":
return 7 return 7
@@ -73,7 +73,7 @@ func riscvRegNum(name string) int {
return 29 return 29
case "X30", "T5": case "X30", "T5":
return 30 return 30
case "X31", "T6": case "X31", "T6", "TMP":
return 31 return 31
// Floating-point registers (F0-F31). // Floating-point registers (F0-F31).
case "F0", "FT0": case "F0", "FT0":
@@ -457,63 +457,84 @@ func rvcCR(funct4, rd, rs2 uint32) uint16 {
// rvcCI encodes a CI-type (immediate) compressed instruction. // rvcCI encodes a CI-type (immediate) compressed instruction.
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW — linear 6-bit immediate. // Used for C.ADDI, C.LI, C.LUI, C.ADDIW — linear 6-bit immediate.
func rvcCI(funct3, rd uint32, imm uint32) uint16 { func rvcCI(funct3, rd uint32, imm uint32) uint16 {
return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x2) return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x1)
} }
// rvcLSP encodes a CI-type stack-relative load: C.LDSP (funct3=3) or // rvcSLLI encodes C.SLLI, which shares funct3=0 with C.ADDI but lives in the
// C.FLDSP (funct3=1). offset is the full byte offset; the immediate bits // op=10 quadrant (unlike C.ADDI's op=01).
// are interleaved per the RISC-V spec: [5:3|8:6]. func rvcSLLI(rd, shamt uint32) uint16 {
func rvcLSP(funct3, rd uint32, offset uint32) uint16 { return uint16(((shamt>>5)&1)<<12 | (rd << 7) | (shamt&0x1F)<<2 | 0x2)
// Bit interleave offset bits [5,4,3,8,7,6] → packed value. }
// encodeRVCPattern extracts the bits listed in pattern (MSB first) from imm
// into a packed value, matching cmd/internal/obj/riscv's encodeBitPattern.
func encodeRVCPattern(imm uint32, pattern []int) uint32 {
packed := uint32(0) packed := uint32(0)
for i, b := range []int{5, 4, 3, 8, 7, 6} { for _, bit := range pattern {
packed = packed<<1 | (imm>>bit)&1
}
return packed
}
// rvcLSP encodes a stack-relative compressed load (op=10 quadrant): C.LWSP
// (funct3=2, 4-byte scale), C.LDSP (funct3=3) or C.FLDSP (funct3=1, 8-byte
// scale). offset is the full byte offset.
func rvcLSP(funct3, rd uint32, offset uint32) uint16 {
pattern := []int{5, 4, 3, 8, 7, 6}
if funct3 == 0x2 {
pattern = []int{5, 4, 3, 2, 7, 6}
}
packed := uint32(0)
for i, b := range pattern {
packed |= ((offset >> b) & 1) << (5 - i) packed |= ((offset >> b) & 1) << (5 - i)
} }
return uint16((funct3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x2) return uint16((funct3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x2)
} }
// rvcSSP encodes a CSS-type stack-relative store: C.SDSP (funct3=7) or // rvcSSP encodes a stack-relative compressed store (op=10 quadrant): C.SWSP
// C.FSDSP (funct3=5). offset is the full byte offset; the immediate bits // (funct3=6, 4-byte scale), C.SDSP (funct3=7) or C.FSDSP (funct3=5, 8-byte
// are interleaved per the RISC-V spec: [5:3|8:6]. // scale). offset is the full byte offset.
func rvcSSP(funct3, rs2 uint32, offset uint32) uint16 { func rvcSSP(funct3, rs2 uint32, offset uint32) uint16 {
// Bit interleave offset bits [5,4,3,8,7,6] → packed value. pattern := []int{5, 4, 3, 8, 7, 6}
if funct3 == 0x6 {
pattern = []int{5, 4, 3, 2, 7, 6}
}
packed := uint32(0) packed := uint32(0)
for i, b := range []int{5, 4, 3, 8, 7, 6} { for i, b := range pattern {
packed |= ((offset >> b) & 1) << (5 - i) packed |= ((offset >> b) & 1) << (5 - i)
} }
return uint16((funct3 << 13) | (packed << 7) | (rs2 << 2) | 0x2) return uint16((funct3 << 13) | (packed << 7) | (rs2 << 2) | 0x2)
} }
// rvcCSS encodes a CSS-type (stack store) compressed instruction. // rvcCL encodes a register-relative compressed load (op=00 quadrant): C.LW
func rvcCSS(funct3, rs2 uint32, imm uint32) uint16 { // (funct3=2), C.LD (funct3=3) or C.FLD (funct3=1). imm is the full byte
return uint16((funct3 << 13) | (imm << 7) | (rs2 << 2) | 0x2) // offset; the immediate bits are extracted per the RISC-V CL format.
}
// rvcCL encodes a CL-type (load) compressed instruction.
// imm layout: [5:3] in bits [12:10], [2|6] in bits [6:5].
func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 { func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 {
bits := uint16((funct3 << 13) | ((imm>>3)&0x7)<<10 | (rs1 << 7) | ((imm & 0x7) << 5) | (rd << 2) | 0x0) pattern := []int{5, 4, 3, 7, 6}
return bits if funct3 == 0x2 {
pattern = []int{5, 4, 3, 2, 6}
}
packed := encodeRVCPattern(imm, pattern)
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rd << 2))
} }
// rvcCS encodes a CS-type (store) compressed instruction. // rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
// offset; the immediate bits are extracted per the RISC-V CS format.
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 { func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
return uint16((funct3 << 13) | ((imm>>3)&0x7)<<10 | (rs1 << 7) | ((imm & 0x7) << 5) | (rs2 << 2) | 0x0) pattern := []int{5, 3, 7, 6}
if funct3 == 0x6 {
pattern = []int{5, 3, 2, 6}
}
packed := encodeRVCPattern(imm, pattern)
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
} }
// rvcCJ encodes a CJ-type (jump) compressed instruction. // rvcCIW encodes a CIW-type compressed immediate wide instruction: C.ADDI4SPN
// offset is a 12-bit signed offset (bit 0 is always 0). // (funct3=0). imm is the raw byte offset.
func rvcCJ(funct3 uint32, offset int32) uint16 { func rvcCIW(funct3, rd uint32, imm uint32) uint16 {
uoff := uint32(offset) & 0xFFE packed := encodeRVCPattern(imm, []int{5, 4, 9, 8, 7, 6, 2, 3})
bits := ((uoff >> 11) & 1) << 10 return uint16((funct3 << 13) | (packed << 5) | (rd << 2))
bits |= ((uoff >> 4) & 1) << 9
bits |= ((uoff >> 9) & 0x3) << 7
bits |= ((uoff >> 10) & 1) << 6
bits |= ((uoff >> 6) & 1) << 5
bits |= ((uoff >> 7) & 1) << 4
bits |= ((uoff >> 1) & 0x7) << 1
bits |= ((uoff >> 5) & 1)
return uint16((funct3 << 13) | (bits << 2) | 0x1)
} }
// rvcCA encodes a CA-type (arithmetic) compressed instruction. // rvcCA encodes a CA-type (arithmetic) compressed instruction.
@@ -522,16 +543,22 @@ func rvcCA(funct6, funct2, rd, rs2 uint32) uint16 {
return uint16((funct6 << 10) | (rd << 7) | (funct2 << 5) | (rs2 << 2) | 0x1) return uint16((funct6 << 10) | (rd << 7) | (funct2 << 5) | (rs2 << 2) | 0x1)
} }
// rvcCB encodes a CB-type (branch) compressed instruction. // rvcCBShift encodes a CB-type shift/immediate compressed instruction
// Format: funct3[15:13] | offset[8|4:3] | rs1'[9:7] | offset[7:6|2:1|5] | op=01. // (C.SRLI, C.SRAI, C.ANDI). rd is the 3-bit prime-register index; imm is
// Bit pattern for offset: [8|4:3|7:6|2:1|5] // the 6-bit shamt/immediate; funct2 selects the operation (0=SRLI, 1=SRAI,
func rvcCB(funct3, rs1 uint32, offset int32) uint16 { // 2=ANDI).
uoff := uint32(offset) & 0x1FE // bits [8:1] func rvcCBShift(funct2, rd, imm uint32) uint16 {
offBits := uint32(0) return uint16((0x4 << 13) | ((imm>>5)&1)<<12 | (funct2 << 10) | (rd << 7) | (imm&0x1F)<<2 | 0x1)
offBits |= ((uoff >> 8) & 1) << 10 // bit 10 = offset[8] }
offBits |= ((uoff >> 3) & 0x3) << 8 // bits 9:8 = offset[4:3]
offBits |= ((uoff >> 6) & 0x3) << 6 // bits 7:6 = offset[7:6] // rvcADDI16SP encodes C.ADDI16SP: ADDI rd, imm, rd for the stack pointer
offBits |= ((uoff >> 1) & 0x3) << 3 // bits 4:3 = offset[2:1] // with a 10-bit signed, 16-byte-scaled immediate. imm is the raw byte
offBits |= ((uoff >> 5) & 1) << 2 // bit 2 = offset[5] // offset; the immediate bits are extracted in the order [9|4|6|8:7|5].
return uint16((funct3 << 13) | offBits | (rs1 << 7) | 0x1) func rvcADDI16SP(rd uint32, imm int32) uint16 {
u := uint32(imm)
packed := uint32(0)
for _, bit := range []uint{9, 4, 6, 8, 7, 5} {
packed = packed<<1 | (u>>bit)&1
}
return uint16((0x3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x1)
} }
+171 -157
View File
@@ -4,6 +4,7 @@
package asm package asm
import ( import (
"bytes"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -29,7 +30,7 @@ func firstTextRISCV(t *testing.T, src string) *ast.Text {
// assembleRISCVHelper assembles one TEXT function and returns its code bytes. // assembleRISCVHelper assembles one TEXT function and returns its code bytes.
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte { func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
t.Helper() t.Helper()
code, _, _, err := assembleRISCV(fn) code, _, _, _, _, err := assembleRISCV(fn)
if err != nil { if err != nil {
t.Fatalf("assemble: %v", err) t.Fatalf("assemble: %v", err)
} }
@@ -65,9 +66,9 @@ TEXT ·arith(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// 5 R-type instructions + RET compressed = 5*4 + 2 = 22 // 5 R-type instructions + RET = 5*4 + 4 = 24
if len(code) != 22 { if len(code) != 24 {
t.Errorf("expected 22 bytes, got %d", len(code)) t.Errorf("expected 24 bytes, got %d", len(code))
} }
} }
@@ -81,35 +82,35 @@ TEXT ·mem(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// 4 loads/stores (4B each) + C.JR RET (2B) = 18 // Four register-relative loads/stores compress (2B each) + JALR (4B) = 12.
if len(code) != 18 { if len(code) != 12 {
t.Errorf("expected 18 bytes, got %d", len(code)) t.Errorf("expected 12 bytes, got %d", len(code))
} }
} }
func TestRISCV_immediate(t *testing.T) { func TestRISCV_immediate(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h" fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·imm(SB), NOSPLIT, $0 TEXT ·imm(SB), NOSPLIT, $0
ADDI X10, $42, X11 ADDI $42, X10, X11
ANDI X11, $0xFF, X12 ANDI $0xFF, X11, X12
ORI X12, $1, X13 ORI $1, X12, X13
XORI X13, $0, X14 XORI $0, X13, X14
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// 4 I-type + C.JR = 4*4 + 2 = 18 // 4 I-type + JALR = 4*4 + 4 = 20
if len(code) != 18 { if len(code) != 20 {
t.Errorf("expected 18 bytes, got %d", len(code)) t.Errorf("expected 20 bytes, got %d", len(code))
} }
} }
func TestRISCV_branches(t *testing.T) { func TestRISCV_branches(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h" fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·br(SB), NOSPLIT, $0 TEXT ·br(SB), NOSPLIT, $0
ADDI X10, $1, X10 ADDI $1, X10, X10
loop: loop:
BEQ X10, X11, done BEQ X10, X11, done
ADDI X10, $1, X10 ADDI $1, X10, X10
JMP loop JMP loop
done: done:
RET RET
@@ -122,28 +123,28 @@ done:
} }
func TestRISCV_MOV_imm_small(t *testing.T) { func TestRISCV_MOV_imm_small(t *testing.T) {
// MOV $42, rd → ADDI (fits in 12 bits). Not RVC-compressed (treated as MOV, not ADDI). // MOV $42, rd → ADDI (fits in 12 bits, but not C.LI's 6-bit immediate).
fn := firstTextRISCV(t, `#include "textflag.h" fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·small(SB), NOSPLIT, $0 TEXT ·small(SB), NOSPLIT, $0
MOV $42, X10 MOV $42, X10
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// ADDI (4B) + C.JR (2B) = 6 // ADDI (4B) + JALR (4B) = 8
if len(code) != 6 { if len(code) != 8 {
t.Errorf("expected 6 bytes, got %d", len(code)) t.Errorf("expected 8 bytes, got %d", len(code))
} }
} }
func TestRISCV_MOV_imm_large(t *testing.T) { func TestRISCV_MOV_imm_large(t *testing.T) {
// MOV $0x12345, rd → LUI + ADDIW (8 bytes total) // MOV $0x12345, rd → C.LUI $18 (2B) + ADDIW $837 (4B).
fn := firstTextRISCV(t, `#include "textflag.h" fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·large(SB), NOSPLIT, $0 TEXT ·large(SB), NOSPLIT, $0
MOV $0x12345, X10 MOV $0x12345, X10
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// LUI (4B) + ADDIW (4B) + C.JR (2B) = 10 // C.LUI (2B) + ADDIW (4B) + JALR (4B) = 10
if len(code) != 10 { if len(code) != 10 {
t.Errorf("expected 10 bytes, got %d", len(code)) t.Errorf("expected 10 bytes, got %d", len(code))
} }
@@ -157,9 +158,9 @@ TEXT ·reg(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.MV (2B) + C.JR (2B) = 4 // C.MV (2B) + JALR (4B) = 6
if len(code) != 4 { if len(code) != 6 {
t.Errorf("expected 4 bytes, got %d (% x)", len(code), code) t.Errorf("expected 6 bytes, got %d (% x)", len(code), code)
} }
} }
@@ -172,9 +173,9 @@ TEXT ·frame(SB), NOSPLIT, $0-8
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.LDSP (2B) + C.SDSP (2B) + C.JR (2B) = 6 // C.LDSP (2B) + C.SDSP (2B) + JALR (4B) = 8
if len(code) != 6 { if len(code) != 8 {
t.Errorf("expected 6 bytes, got %d", len(code)) t.Errorf("expected 8 bytes, got %d", len(code))
} }
} }
@@ -187,9 +188,9 @@ TEXT ·rvcstore(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.LDSP (2B) + C.SDSP (2B) + C.JR (2B) = 6 // C.LDSP (2B) + C.SDSP (2B) + JALR (4B) = 8
if len(code) != 6 { if len(code) != 8 {
t.Errorf("expected 6 bytes, got %d (% x)", len(code), code) t.Errorf("expected 8 bytes, got %d (% x)", len(code), code)
} }
} }
@@ -202,9 +203,9 @@ TEXT ·amo(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// 3 AMO instructions (4B each) + C.JR (2B) = 14 // 3 AMO instructions (4B each) + JALR (4B) = 16
if len(code) != 14 { if len(code) != 16 {
t.Errorf("expected 14 bytes, got %d", len(code)) t.Errorf("expected 16 bytes, got %d", len(code))
} }
} }
@@ -219,9 +220,9 @@ TEXT ·fpadd(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// 5 FP instructions (4B each) + C.JR (2B) = 22 // 5 FP instructions (4B each) + JALR (4B) = 24
if len(code) != 22 { if len(code) != 24 {
t.Errorf("expected 22 bytes, got %d (%d)", len(code), len(code)) t.Errorf("expected 24 bytes, got %d (%d)", len(code), len(code))
} }
} }
@@ -234,9 +235,9 @@ TEXT ·csrtest(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// 3 CSR instructions (4B each) + C.JR (2B) = 14 // 3 CSR instructions (4B each) + JALR (4B) = 16
if len(code) != 14 { if len(code) != 16 {
t.Errorf("expected 14 bytes, got %d", len(code)) t.Errorf("expected 16 bytes, got %d", len(code))
} }
} }
@@ -250,9 +251,9 @@ TEXT ·fmatest(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// 4 FMA instructions (4B each) + C.JR (2B) = 18 // 4 FMA instructions (4B each) + JALR (4B) = 20
if len(code) != 18 { if len(code) != 20 {
t.Errorf("expected 18 bytes, got %d", len(code)) t.Errorf("expected 20 bytes, got %d", len(code))
} }
} }
@@ -266,9 +267,9 @@ TEXT ·cvt(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// 4 conversion instructions (4B each) + C.JR (2B) = 18 // 4 conversion instructions (4B each) + JALR (4B) = 20
if len(code) != 18 { if len(code) != 20 {
t.Errorf("expected 18 bytes, got %d", len(code)) t.Errorf("expected 20 bytes, got %d", len(code))
} }
} }
@@ -281,9 +282,9 @@ TEXT ·cmp(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// 3 FP compare (4B each) + C.JR (2B) = 14 // 3 FP compare (4B each) + JALR (4B) = 16
if len(code) != 14 { if len(code) != 16 {
t.Errorf("expected 14 bytes, got %d", len(code)) t.Errorf("expected 16 bytes, got %d", len(code))
} }
} }
@@ -291,9 +292,9 @@ func TestRISCV_forwardBranch(t *testing.T) {
// Forward label reference — must not fail. // Forward label reference — must not fail.
fn := firstTextRISCV(t, `#include "textflag.h" fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·fwd(SB), NOSPLIT, $0 TEXT ·fwd(SB), NOSPLIT, $0
ADDI X10, $1, X10 ADDI $1, X10, X10
BEQ X10, X11, done BEQ X10, X11, done
ADDI X10, $1, X10 ADDI $1, X10, X10
done: done:
RET RET
`) `)
@@ -308,13 +309,13 @@ func TestRISCV_RVC_ADDI(t *testing.T) {
// ADDI where rd=rs1 and small imm → C.ADDI // ADDI where rd=rs1 and small imm → C.ADDI
fn := firstTextRISCV(t, `#include "textflag.h" fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·caddi(SB), NOSPLIT, $0 TEXT ·caddi(SB), NOSPLIT, $0
ADDI X10, $5, X10 ADDI $5, X10, X10
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.ADDI (2B) + C.JR (2B) = 4 // C.ADDI (2B) + JALR (4B) = 6
if len(code) != 4 { if len(code) != 6 {
t.Errorf("expected 4 bytes, got %d", len(code)) t.Errorf("expected 6 bytes, got %d", len(code))
} }
} }
@@ -322,13 +323,13 @@ func TestRISCV_RVC_LI(t *testing.T) {
// ADDI X0, $imm, rd → C.LI // ADDI X0, $imm, rd → C.LI
fn := firstTextRISCV(t, `#include "textflag.h" fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cli(SB), NOSPLIT, $0 TEXT ·cli(SB), NOSPLIT, $0
ADDI X0, $7, X10 ADDI $7, X0, X10
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.LI (2B) + C.JR (2B) = 4 // C.LI (2B) + JALR (4B) = 6
if len(code) != 4 { if len(code) != 6 {
t.Errorf("expected 4 bytes, got %d", len(code)) t.Errorf("expected 6 bytes, got %d", len(code))
} }
} }
@@ -340,9 +341,9 @@ TEXT ·clui(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.LUI (2B) + C.JR (2B) = 4 // C.LUI (2B) + JALR (4B) = 6
if len(code) != 4 { if len(code) != 6 {
t.Errorf("expected 4 bytes, got %d", len(code)) t.Errorf("expected 6 bytes, got %d", len(code))
} }
} }
@@ -368,13 +369,13 @@ TEXT ·sub(SB), NOSPLIT, $0
if len(img.Funcs) != 2 { if len(img.Funcs) != 2 {
t.Fatalf("expected 2 functions, got %d", len(img.Funcs)) t.Fatalf("expected 2 functions, got %d", len(img.Funcs))
} }
// func add: C.LDSP(2) + C.JR(2) = 4 // func add: C.LDSP(2) + JALR(4) = 6
if img.Funcs[0].Size != 4 { if img.Funcs[0].Size != 6 {
t.Errorf("add: expected 4 bytes, got %d", img.Funcs[0].Size) t.Errorf("add: expected 6 bytes, got %d", img.Funcs[0].Size)
} }
// func sub: SUB(4) + C.JR(2) = 6 // func sub: SUB(4) + JALR(4) = 8
if img.Funcs[1].Size != 6 { if img.Funcs[1].Size != 8 {
t.Errorf("sub: expected 6 bytes, got %d", img.Funcs[1].Size) t.Errorf("sub: expected 8 bytes, got %d", img.Funcs[1].Size)
} }
} }
@@ -384,34 +385,34 @@ func TestRISCV_encodings(t *testing.T) {
name, src string name, src string
wantBytes int wantBytes int
}{ }{
{"ADD", "ADD X10, X11, X12\nRET\n", 6}, {"ADD", "ADD X10, X11, X12\nRET\n", 8},
{"SUBW", "SUBW X10, X11, X12\nRET\n", 6}, {"SUBW", "SUBW X10, X11, X12\nRET\n", 8},
{"MUL", "MUL X10, X11, X12\nRET\n", 6}, {"MUL", "MUL X10, X11, X12\nRET\n", 8},
{"DIVW", "DIVW X10, X11, X12\nRET\n", 6}, {"DIVW", "DIVW X10, X11, X12\nRET\n", 8},
{"REMUW", "REMUW X10, X11, X12\nRET\n", 6}, {"REMUW", "REMUW X10, X11, X12\nRET\n", 8},
{"ADDIW", "ADDIW X10, $5, X11\nRET\n", 6}, {"ADDIW", "ADDIW $5, X10, X11\nRET\n", 8},
{"SLLI", "SLLI X10, $3, X11\nRET\n", 6}, // ADDI+SLLI? No, SLLI uses I-type {"SLLI", "SLLI $3, X10, X11\nRET\n", 8}, // ADDI+SLLI? No, SLLI uses I-type
{"SRLI", "SRLI X10, $2, X11\nRET\n", 6}, {"SRLI", "SRLI $2, X10, X11\nRET\n", 8},
{"SRAI", "SRAI X10, $1, X11\nRET\n", 6}, {"SRAI", "SRAI $1, X10, X11\nRET\n", 8},
{"LB", "LB (X10), X11\nRET\n", 6}, {"LB", "LB (X10), X11\nRET\n", 8},
{"LBU", "LBU (X10), X11\nRET\n", 6}, {"LBU", "LBU (X10), X11\nRET\n", 8},
{"LH", "LH (X10), X11\nRET\n", 6}, {"LH", "LH (X10), X11\nRET\n", 8},
{"LHU", "LHU (X10), X11\nRET\n", 6}, {"LHU", "LHU (X10), X11\nRET\n", 8},
{"LWU", "LWU (X10), X11\nRET\n", 6}, {"LWU", "LWU (X10), X11\nRET\n", 8},
{"SB", "SB X10, (X11)\nRET\n", 6}, {"SB", "SB X10, (X11)\nRET\n", 8},
{"SH", "SH X10, (X11)\nRET\n", 6}, {"SH", "SH X10, (X11)\nRET\n", 8},
{"SW", "SW X10, (X11)\nRET\n", 6}, {"SW", "SW X10, (X11)\nRET\n", 6},
{"LUI", "LUI X10, $0x12345\nRET\n", 6}, {"LUI", "LUI X10, $0x12345\nRET\n", 8},
{"AUIPC", "AUIPC X10, $0\nRET\n", 6}, {"AUIPC", "AUIPC X10, $0\nRET\n", 8},
{"FLW", "FLW (X10), F10\nRET\n", 6}, {"FLW", "FLW (X10), F10\nRET\n", 8},
{"FSW", "FSW F10, (X11)\nRET\n", 6}, {"FSW", "FSW F10, (X11)\nRET\n", 8},
{"FADDS", "FADDS F10, F11, F12\nRET\n", 6}, {"FADDS", "FADDS F10, F11, F12\nRET\n", 8},
{"FMINS", "FMINS F10, F11, F12\nRET\n", 6}, {"FMINS", "FMINS F10, F11, F12\nRET\n", 8},
{"FMAXD", "FMAXD F10, F11, F12\nRET\n", 6}, {"FMAXD", "FMAXD F10, F11, F12\nRET\n", 8},
{"FCVTSD", "FCVTSD F10, F11\nRET\n", 6}, {"FCVTSD", "FCVTSD F10, F11\nRET\n", 8},
{"FCVTDS", "FCVTDS F10, F11\nRET\n", 6}, {"FCVTDS", "FCVTDS F10, F11\nRET\n", 8},
{"FMVXW", "FMVXW F10, X10\nRET\n", 6}, {"FMVXW", "FMVXW F10, X10\nRET\n", 8},
{"FMADD_S", "FMADDS F10, F11, F12, F13\nRET\n", 6}, {"FMADD_S", "FMADDS F10, F11, F12, F13\nRET\n", 8},
} }
for _, tt := range tests { for _, tt := range tests {
@@ -428,24 +429,24 @@ TEXT ·`+tt.name+`(SB), NOSPLIT, $0
} }
func TestRISCV_RVC_branch(t *testing.T) { func TestRISCV_RVC_branch(t *testing.T) {
// BEQ rs, X0, target → C.BEQZ when rs is in prime regs and offset fits. // Branches are never RVC-compressed (no C.BEQZ/C.BNEZ), matching go tool asm.
fn := firstTextRISCV(t, `#include "textflag.h" fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cbeqz(SB), NOSPLIT, $0 TEXT ·cbeqz(SB), NOSPLIT, $0
ADDI X10, $1, X10 ADDI $1, X10, X10
BEQ X10, X0, done BEQ X10, X0, done
ADDI X10, $1, X10 ADDI $1, X10, X10
done: done:
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.ADDI(2) + C.BEQZ(2) + C.ADDI(2) + C.JR(2) = 8 (all compress) // C.ADDI(2) + BEQ(4) + C.ADDI(2) + JALR(4) = 12
if len(code) != 8 { if len(code) != 12 {
t.Errorf("expected 8 bytes with C.BEQZ, got %d", len(code)) t.Errorf("expected 12 bytes with uncompressed BEQ, got %d", len(code))
} }
} }
func TestRISCV_RVC_CJ(t *testing.T) { func TestRISCV_RVC_CJ(t *testing.T) {
// JMP target → C.J when offset fits. // JMP target → JAL X0 (never compressed to C.J), matching go tool asm.
fn := firstTextRISCV(t, `#include "textflag.h" fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cj(SB), NOSPLIT, $0 TEXT ·cj(SB), NOSPLIT, $0
JMP done JMP done
@@ -453,9 +454,9 @@ func TestRISCV_RVC_CJ(t *testing.T) {
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.J(2) + C.JR(2) = 4 // JAL(4) + JALR(4) = 8
if len(code) != 4 { if len(code) != 8 {
t.Errorf("expected 4 bytes with C.J, got %d", len(code)) t.Errorf("expected 8 bytes with uncompressed JMP, got %d", len(code))
} }
} }
@@ -467,9 +468,9 @@ func TestRISCV_RVC_CADD(t *testing.T) {
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.ADD(2) + C.JR(2) = 4 // C.ADD(2) + JALR(4) = 6
if len(code) != 4 { if len(code) != 6 {
t.Errorf("expected 4 bytes with C.ADD, got %d", len(code)) t.Errorf("expected 6 bytes with C.ADD, got %d", len(code))
} }
} }
@@ -481,35 +482,23 @@ func TestRISCV_RVC_CADD_commute(t *testing.T) {
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.ADD(2) + C.JR(2) = 4 // C.ADD(2) + JALR(4) = 6
if len(code) != 4 { if len(code) != 6 {
t.Errorf("expected 4 bytes with C.ADD (commuted), got %d", len(code)) t.Errorf("expected 6 bytes with C.ADD (commuted), got %d", len(code))
} }
} }
func TestRISCV_RVC_CSUB(t *testing.T) { func TestRISCV_RVC_CSUB(t *testing.T) {
// SUB where rd==rs1 and both in prime regs → C.SUB. // SUB rs2, rs1, rd → C.SUB when rd == rs1 and both in prime regs.
fn := firstTextRISCV(t, `#include "textflag.h" fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·csub(SB), NOSPLIT, $0 TEXT ·csub(SB), NOSPLIT, $0
SUB X11, X10, X10 SUB X11, X10, X10
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// SUB X11,X10,X10 → rd=X10, rs1=X11 ≠ rd → no C.SUB. // SUB X11, X10, X10 → rs2=X11, rs1=X10, rd=X10; rd==rs1 → C.SUB (2B) + JALR (4B) = 6.
// Plan9: INSTR src1, src2, dst. For C.SUB: rd must equal rs1. if len(code) != 6 {
// So: SUB X10, X11, X10 → rd=10, rs1=10, rs2=11 ✓ t.Errorf("expected 6 bytes with C.SUB, got %d (% x)", len(code), code)
if len(code) == 4 {
return // compressed
}
// Try with correct operand order.
fn2 := firstTextRISCV(t, `#include "textflag.h"
TEXT ·csub2(SB), NOSPLIT, $0
SUB X10, X11, X10
RET
`)
code2 := assembleRISCVHelper(t, fn2)
if len(code2) != 4 {
t.Errorf("expected 4 bytes with C.SUB, got %d (% x)", len(code2), code2)
} }
} }
@@ -520,8 +509,8 @@ TEXT ·cxor(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
if len(code) != 4 { if len(code) != 6 {
t.Errorf("expected 4 bytes with C.XOR, got %d", len(code)) t.Errorf("expected 6 bytes with C.XOR, got %d", len(code))
} }
} }
@@ -532,8 +521,8 @@ TEXT ·cor(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
if len(code) != 4 { if len(code) != 6 {
t.Errorf("expected 4 bytes with C.OR, got %d", len(code)) t.Errorf("expected 6 bytes with C.OR, got %d", len(code))
} }
} }
@@ -544,8 +533,8 @@ TEXT ·cand(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
if len(code) != 4 { if len(code) != 6 {
t.Errorf("expected 4 bytes with C.AND, got %d", len(code)) t.Errorf("expected 6 bytes with C.AND, got %d", len(code))
} }
} }
@@ -556,9 +545,9 @@ TEXT ·cfldsp(SB), NOSPLIT, $0-8
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.FLDSP(2) + C.JR(2) = 4 // C.FLDSP(2) + JALR(4) = 6
if len(code) != 4 { if len(code) != 6 {
t.Errorf("expected 4 bytes with C.FLDSP, got %d", len(code)) t.Errorf("expected 6 bytes with C.FLDSP, got %d", len(code))
} }
} }
@@ -569,9 +558,9 @@ TEXT ·cfsdsp(SB), NOSPLIT, $0-8
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.FSDSP(2) + C.JR(2) = 4 // C.FSDSP(2) + JALR(4) = 6
if len(code) != 4 { if len(code) != 6 {
t.Errorf("expected 4 bytes with C.FSDSP, got %d", len(code)) t.Errorf("expected 6 bytes with C.FSDSP, got %d", len(code))
} }
} }
@@ -592,9 +581,9 @@ DATA answer<>+0(SB)/8, $42
if err != nil { if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err) t.Fatalf("AssembleFileRISCV: %v", err)
} }
// AUIPC(4) + ADDI(4) + C.JR(2) = 10 // AUIPC(4) + ADDI(4) + JALR(4) = 12
if img.Funcs[0].Size != 10 { if img.Funcs[0].Size != 12 {
t.Errorf("expected 10 bytes, got %d", img.Funcs[0].Size) t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
} }
} }
@@ -614,9 +603,9 @@ GLOBL result<>(SB), NOPTR, $8
if err != nil { if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err) t.Fatalf("AssembleFileRISCV: %v", err)
} }
// AUIPC X31(4) + SD X10,0(X31)(4) + C.JR(2) = 10 // AUIPC X31(4) + SD X10,0(X31)(4) + JALR(4) = 12
if img.Funcs[0].Size != 10 { if img.Funcs[0].Size != 12 {
t.Errorf("expected 10 bytes, got %d", img.Funcs[0].Size) t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
} }
} }
@@ -696,9 +685,9 @@ DATA answer<>+0(SB)/8, $42
if err != nil { if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err) t.Fatalf("AssembleFileRISCV: %v", err)
} }
// AUIPC(4) + LD(4) + C.JR(2) = 10 // AUIPC(4) + LD(4) + JALR(4) = 12
if img.Funcs[0].Size != 10 { if img.Funcs[0].Size != 12 {
t.Errorf("expected 10 bytes, got %d", img.Funcs[0].Size) t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
} }
} }
@@ -712,7 +701,7 @@ TEXT ·sys(SB), NOSPLIT, $0
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// 3 system instructions × 4 bytes + C.JR(2) = 14 // FENCE(4) + ECALL(4) + C.EBREAK(2) + JALR(4) = 14
if len(code) != 14 { if len(code) != 14 {
t.Errorf("expected 14 bytes, got %d (% x)", len(code), code) t.Errorf("expected 14 bytes, got %d (% x)", len(code), code)
} }
@@ -725,25 +714,50 @@ TEXT ·badfp(SB), NOSPLIT, $0
MOV $arg(FP), X10 MOV $arg(FP), X10
RET RET
`) `)
_, _, _, err := assembleRISCV(fn) _, _, _, _, _, err := assembleRISCV(fn)
if err == nil { if err == nil {
t.Error("expected error for MOV $arg(FP), got nil") t.Error("expected error for MOV $arg(FP), got nil")
} }
} }
func TestRISCV_CALL(t *testing.T) { func TestRISCV_CALL(t *testing.T) {
// CALL target → AUIPC + JALR (8 bytes). // CALL sym(SB) → JAL X1, sym(SB) with a single R_RISCV_JAL relocation.
fn := firstTextRISCV(t, `#include "textflag.h" fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·calltest(SB), NOSPLIT, $0 TEXT ·calltest(SB), NOSPLIT, $0
CALL sub CALL ext(SB)
done:
RET RET
`)
code, _, relocs, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
// prologue (8) + JAL (4) + epilogue+JALR (8) = 20
if len(code) != 20 {
t.Fatalf("expected 20 bytes with CALL sym(SB), got %d", len(code))
}
if len(relocs) != 1 {
t.Fatalf("relocs = %d, want 1", len(relocs))
}
r := relocs[0]
if r.Kind != RelRISCVJal || r.Name != "ext" || r.Off != 8 || r.After != 12 || r.Addend != 0 {
t.Errorf("reloc = {kind %v off %d after %d name %q addend %d}", r.Kind, r.Off, r.After, r.Name, r.Addend)
}
// The JAL instruction itself is JAL X1, 0 at function offset 8.
wantJAL := wordLE(riscvJType(1, 0))
if !bytes.Equal(code[8:12], wantJAL) {
t.Errorf("JAL = % x, want % x", code[8:12], wantJAL)
}
}
func TestRISCV_CALL_local_error(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·calllocal(SB), NOSPLIT, $0
CALL sub
sub: sub:
RET RET
`) `)
code := assembleRISCVHelper(t, fn) _, _, _, _, _, err := assembleRISCV(fn)
// CALL(8) + C.JR(2) + C.JR(2) = 12 if err == nil {
if len(code) != 12 { t.Error("expected error for CALL to local label, got nil")
t.Errorf("expected 12 bytes with CALL, got %d", len(code))
} }
} }
+149 -84
View File
@@ -3,115 +3,180 @@
package asm package asm
import "sourcedock.dev/petrbalvin/gasm-devkit/ast" import (
"strings"
// RISC-V frame mapping: translates Go's FP/SP pseudo-register addressing "sourcedock.dev/petrbalvin/gasm-devkit/ast"
// into real RISC-V memory accesses. )
//
// In Go's ABI0 (used by assembly functions), arguments are passed on the
// stack. At function entry the return address sits at SP, so the frame
// pointer FP == SP+8 and the first argument is at FP+0 == SP+8.
//
// On RISC-V the hardware registers are:
// SP = X2 (stack pointer)
// FP = S0 = X8 (frame pointer, by convention)
//
// For NOSPLIT $0 functions the prologue is omitted and arguments are read
// directly from SP+8+offset.
// riscvFrameInfo holds the frame parameters computed from a TEXT directive. // RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
//
// Go's riscv64 functions have no hardware frame pointer: FP and SP are
// synthetic registers resolved against the hardware stack pointer (X2) and
// the frame size. The return address lives in the link register (X1, RA/LR).
//
// The autosize is the real stack adjustment: the declared local frame plus
// the 8 bytes for the saved link register (the toolchain's FixedFrameSize).
// A leaf function with a zero frame gets no prologue at all.
//
// Prologue (autosize > 0), byte-identical to the toolchain:
//
// MOV LR, -autosize(SP) // save LR below the new SP (traceback-safe)
// ADDI $-autosize, SP, SP // open the frame
// MOV LR, 0(SP) // save LR again at SP (signal-safety)
//
// Epilogue (autosize > 0): MOV 0(SP), LR; ADDI $autosize, SP, SP; the RET's
// uncompressed JALR X0, 0(X1) follows. The toolchain restores LR on every
// frame, leaf or not.
// riscvFrameInfo holds the frame layout derived from a TEXT directive.
type riscvFrameInfo struct { type riscvFrameInfo struct {
frameSize int // the $framesize from TEXT autosize int // the real SP adjustment (locals + saved LR)
argsSize int // the -argsize from TEXT
noSplit bool // the NOSPLIT flag
} }
// riscvComputeFrame extracts frame information from a TEXT directive. // riscvComputeFrame derives the frame layout for a TEXT function.
func riscvComputeFrame(t *ast.Text) riscvFrameInfo { func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
fi := riscvFrameInfo{} frame := frameSize(t)
fi.frameSize = frameSize(t) if frame != 0 || !riscvIsLeaf(t) {
fi.argsSize = argsSize(t) // FixedFrameSize = 8: space for the saved link register. A
for _, f := range t.Flags { // zero-frame non-leaf function still opens an 8-byte frame for LR.
if f == "NOSPLIT" { return riscvFrameInfo{autosize: frame + 8}
fi.noSplit = true }
return riscvFrameInfo{}
}
// riscvIsLeaf reports whether a function contains no call instructions.
// CALL always links; JAL/JALR link only when their destination register is
// the link register (X1), matching cmd/internal/obj/riscv's containsCall.
func riscvIsLeaf(t *ast.Text) bool {
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "CALL":
return false
case "JAL":
// JAL rd, target — a call only when rd is the link register.
if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 {
return false
}
case "JALR":
// JALR rs1, rd — a call when rd is X1; JALR offset(rs1) always
// links to X1.
if len(in.Operands) == 1 {
return false
}
if len(in.Operands) >= 2 && regFromOperand(in.Operands[1]) == 1 {
return false
}
} }
} }
return fi return true
} }
// riscvPrologue returns the prologue bytes for a RISC-V function. // riscvPrologue returns the prologue bytes for a RISC-V function, matching
// For NOSPLIT $0 functions there is no prologue. For functions with a // the toolchain's compression: the SP adjustment compresses to C.ADDI when
// frame, we emit: ADDI SP, SP, -framesize; SD S0, (framesize-8)(SP); ... // the immediate fits, and the second LR save compresses to C.SDSP.
func riscvPrologue(fi riscvFrameInfo) []byte { func riscvPrologue(fi riscvFrameInfo) []byte {
if fi.noSplit && fi.frameSize == 0 { if fi.autosize == 0 {
return nil // no prologue for NOSPLIT $0
}
var out []byte
if fi.frameSize > 0 {
// ADDI SP, SP, -framesize
out = append(out, riscvITypeLE(0x13, 0x0, 2, 2, int32(-fi.frameSize))...)
// Save the frame pointer (S0 = X8) at the top of the new frame.
// SD S0, (framesize-8)(SP)
out = append(out, riscvSTypeLE(0x23, 0x3, 2, 8, int32(fi.frameSize-8))...)
}
return out
}
// riscvEpilogue returns the epilogue bytes for a RISC-V function.
func riscvEpilogue(fi riscvFrameInfo) []byte {
if fi.noSplit && fi.frameSize == 0 {
return nil return nil
} }
var out []byte var out []byte
if fi.frameSize > 0 { // MOV LR, -autosize(SP) — SD X1, -autosize(X2). The negative offset is
// Restore the frame pointer: LD S0, (framesize-8)(SP) // not compressible to C.SDSP (unsigned), so it stays 4 bytes.
out = append(out, riscvITypeLE(0x03, 0x3, 8, 2, int32(fi.frameSize-8))...) out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
// ADDI SP, SP, framesize // ADDI $-autosize, SP, SP — open the frame (C.ADDI when it fits).
out = append(out, riscvITypeLE(0x13, 0x0, 2, 2, int32(fi.frameSize))...) out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
} // MOV LR, 0(SP) — SD X1, 0(X2) → C.SDSP X1, 0.
c := rvcSSP(0x7, 1, 0)
out = append(out, byte(c), byte(c>>8))
return out return out
} }
// riscvReturn returns the bytes for a RET: the epilogue (restore LR and
// deallocate the frame when present) followed by the uncompressed JALR X0,
// 0(X1) the toolchain emits for RET (it never compresses RET to C.JR).
func riscvReturn(fi riscvFrameInfo) []byte {
var out []byte
if fi.autosize != 0 {
// MOV 0(SP), LR — LD X1, 0(X2) → C.LDSP X1, 0.
c := rvcLSP(0x3, 1, 0)
out = append(out, byte(c), byte(c>>8))
// ADDI $autosize, SP, SP — close the frame (C.ADDI when it fits).
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
}
// JALR X0, 0(X1).
return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...)
}
// riscvSPAdjust emits an ADDI rd, imm, rd for the stack pointer (rd = rs1 =
// X2), compressed to C.ADDI16SP when the immediate is a nonzero 16-byte
// multiple, else C.ADDI when it fits 6-bit signed.
func riscvSPAdjust(imm int32) []byte {
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
c := rvcADDI16SP(2, imm)
return []byte{byte(c), byte(c >> 8)}
}
if riscvFitsCAddi(imm) {
c := rvcCI(0x0, 2, uint32(imm)&0x3F)
return []byte{byte(c), byte(c >> 8)}
}
return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 2, 2, imm))
}
// riscvFitsCAddi reports whether imm compresses to C.ADDI (a nonzero 6-bit
// signed immediate).
func riscvFitsCAddi(imm int32) bool {
return imm != 0 && imm >= -32 && imm <= 31
}
// riscvPrologueSpadjPC returns the function-relative byte offset where the
// prologue has finished decrementing SP (the delta becomes autosize).
func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
if fi.autosize == 0 {
return 0
}
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 4 + riscvSPAdjustLen(int32(-fi.autosize))
}
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final JALR — the point where SP is restored.
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
if fi.autosize == 0 {
return 0
}
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 2 + riscvSPAdjustLen(int32(fi.autosize))
}
func riscvSPAdjustLen(imm int32) int {
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
return 2
}
if riscvFitsCAddi(imm) {
return 2
}
return 4
}
// riscvResolvePseudo translates a pseudo-register memory reference into a // riscvResolvePseudo translates a pseudo-register memory reference into a
// real base register and offset. It handles name+offset(FP) and // hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
// name+offset(SP). // x+N(SP) → (N + autosize)(SP). Returns base = -1 for an unresolvable
// // reference (SB: static data, handled by the relocation path).
// Returns the base register number and the adjusted offset.
func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32) { func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32) {
if sym == nil { if sym == nil {
return -1, 0 return -1, 0
} }
offset := int32(sym.Offset)
switch sym.Pseudo { switch sym.Pseudo {
case "FP": case "FP":
// FP == SP+8 for NOSPLIT $0; arguments are at SP+8+offset. return 2, int32(sym.Offset) + int32(fi.autosize) + 8
if fi.noSplit && fi.frameSize == 0 {
return 2, 8 + offset // SP + 8 + argOffset
}
// With a frame, FP points to the saved frame; args are at FP+offset.
return 8, offset // S0 + argOffset
case "SP": case "SP":
// SP-relative; the offset is from the current SP. return 2, int32(fi.autosize) + int32(sym.Offset)
return 2, offset
case "SB": case "SB":
// Static data reference — needs a relocation (not yet supported). return -1, int32(sym.Offset)
return -1, offset
default:
return -1, offset
} }
} return -1, 0
// riscvITypeLE encodes an I-type instruction and returns little-endian bytes.
func riscvITypeLE(opcode, funct3 uint32, rd, rs1 int, imm int32) []byte {
word := (uint32(imm&0xFFF) << 20) | (uint32(rs1) << 15) |
(funct3 << 12) | (uint32(rd) << 7) | opcode
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
}
// riscvSTypeLE encodes an S-type instruction and returns little-endian bytes.
func riscvSTypeLE(opcode, funct3 uint32, rs1, rs2 int, imm int32) []byte {
immU := uint32(imm) & 0xFFF
word := ((immU >> 5) << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
(funct3 << 12) | ((immU & 0x1F) << 7) | opcode
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
} }
+81
View File
@@ -0,0 +1,81 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestRISCVFrameSpadjAndLines checks that a framed function records its
// stack-adjustment boundaries and source-line table, the inputs the GOOBJ
// emitter turns into the pcsp/pcfile/pcline tables.
func TestRISCVFrameSpadjAndLines(t *testing.T) {
f, errs := parser.Parse("frame_riscv64.s", `#include "textflag.h"
TEXT ·framed(SB), NOSPLIT, $16-16
MOV a+0(FP), X10
MOV b+8(FP), X11
ADD X11, X10, X10
MOV X10, ret+16(FP)
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
fn := img.Funcs[0]
if fn.Size != 24 {
t.Fatalf("size = %d, want 24", fn.Size)
}
// autosize = 16 + 8 = 24; the prologue boundary is just past its C.ADDI
// (SD 4 + C.ADDI 2 = 6), and the RET restores SP just past its C.ADDI
// (RET starts at 16; C.LDSP 2 + C.ADDI 2 = 20).
wantSpadj := []SpadjStep{{PC: 6, Value: 24}, {PC: 20, Value: 0}}
if len(fn.Spadj) != len(wantSpadj) {
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
}
for i := range wantSpadj {
if fn.Spadj[i] != wantSpadj[i] {
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
}
}
// One line entry per instruction, in emission order.
wantLines := []LineEntry{
{Offset: 8, Line: 4},
{Offset: 10, Line: 5},
{Offset: 12, Line: 6},
{Offset: 14, Line: 7},
{Offset: 16, Line: 8},
}
if len(fn.Lines) != len(wantLines) {
t.Fatalf("lines = %v, want %v", fn.Lines, wantLines)
}
for i := range wantLines {
if fn.Lines[i] != wantLines[i] {
t.Errorf("lines[%d] = %v, want %v", i, fn.Lines[i], wantLines[i])
}
}
}
// TestRISCVRegAliases checks the Go ABI register aliases that the toolchain
// defines: LR is the link register (X1) and TMP is the assembler scratch
// register (X31/T6).
func TestRISCVRegAliases(t *testing.T) {
for name, want := range map[string]int{
"X1": 1, "RA": 1, "LR": 1,
"X31": 31, "T6": 31, "TMP": 31,
"X2": 2, "SP": 2,
} {
if got := riscvRegNum(name); got != want {
t.Errorf("riscvRegNum(%q) = %d, want %d", name, got, want)
}
}
}
+346
View File
@@ -0,0 +1,346 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
// instruction carrying an R_RISCV_JAL relocation (4-byte field) in both the
// GOOBJ and ELF object emitters.
func TestGOObjectRISCVCallReloc(t *testing.T) {
f, errs := parser.Parse("k_riscv64.s", `
#include "textflag.h"
TEXT ·c(SB), NOSPLIT, $0-0
CALL callee<>(SB)
RET
GLOBL callee<>(SB), RODATA, $8
DATA callee<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
fn := img.Funcs[0]
if len(fn.Relocs) != 1 {
t.Fatalf("relocs = %d, want 1", len(fn.Relocs))
}
r := fn.Relocs[0]
if r.Kind != RelRISCVJal || r.Off != 8 || r.After != 12 || r.Name != "callee" || r.Addend != 0 || r.External {
t.Errorf("reloc = {kind %v off %d after %d name %q addend %d external %v}", r.Kind, r.Off, r.After, r.Name, r.Addend, r.External)
}
obj, err := img.GOObjectRISCV("testpkg", "k_riscv64.s")
if err != nil {
t.Fatalf("GOObjectRISCV: %v", err)
}
v := openGoobj(t, obj)
relocIdx := v.blk(blkRelocIdx)
relocs := v.blk(blkReloc)
// The function is the last non-package symbol: 4 package defs, then the
// 4 pc tables and the function.
first := int(binary.LittleEndian.Uint32(relocIdx[(4+4)*4:]))
if (first+1)*23 > len(relocs) {
t.Fatalf("reloc block too short: first=%d len=%d", first, len(relocs))
}
e := relocs[first*23:]
le := binary.LittleEndian
if int32(le.Uint32(e[0:])) != 8 || e[4] != 4 || le.Uint16(e[5:]) != relocRISCVJal || le.Uint32(e[15:]) != pkgIdxSelf || le.Uint32(e[19:]) != 0 {
t.Errorf("GOOBJ reloc = off %d size %d type %d pkg %d sym %d", int32(le.Uint32(e[0:])), e[4], le.Uint16(e[5:]), le.Uint32(e[15:]), le.Uint32(e[19:]))
}
// The ELF object must carry a single R_RISCV_JAL relocation in .rela.text.
elfObj, err := img.ELFRISCVObject()
if err != nil {
t.Fatalf("ELFRISCVObject: %v", err)
}
if !hasELFRISCVJAL(t, elfObj) {
t.Error("ELF object missing R_RISCV_JAL relocation")
}
}
func TestGOObjectRISCVStructure(t *testing.T) {
f, errs := parser.Parse("k_riscv64.s", `
#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0-0
MOV $answer<>(SB), X10
MOV answer<>(SB), X11
MOV X12, answer<>(SB)
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
fn := img.Funcs[0]
if fn.Size != 28 {
t.Fatalf("function size = %d, want 28", fn.Size)
}
if len(fn.Relocs) != 3 {
t.Fatalf("relocs = %d, want 3", len(fn.Relocs))
}
wantKind := []RelocKind{RelRISCVPCRELIType, RelRISCVPCRELIType, RelRISCVPCRELSType}
wantOff := []int{0, 8, 16}
for i, r := range fn.Relocs {
if r.Kind != wantKind[i] || r.Off != wantOff[i] || r.After != r.Off+8 || r.Name != "answer" || r.Addend != 0 {
t.Errorf("reloc %d = {kind %v off %d after %d name %q addend %d}", i, r.Kind, r.Off, r.After, r.Name, r.Addend)
}
}
obj, err := img.GOObjectRISCV("testpkg", "k_riscv64.s")
if err != nil {
t.Fatalf("GOObjectRISCV: %v", err)
}
v := openGoobj(t, obj)
// Package defs: the static GLOBL, the FuncInfo, then the two DWARF
// symbols.
defs := v.syms(blkSymdef)
if len(defs) != 4 {
t.Fatalf("symdefs = %d, want 4", len(defs))
}
if defs[0].name != "answer" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 8 {
t.Errorf("answer symbol = %+v", defs[0])
}
if defs[2].typ != kindSDWARFLINES || defs[3].typ != kindSDWARFFCN {
t.Errorf("dwarf symbols = %+v, %+v", defs[2], defs[3])
}
// The three code relocations, in definition order: ITYPE, ITYPE, STYPE,
// each 8 bytes wide against the GLOBL (package symbol 0).
relocIdx := v.blk(blkRelocIdx)
relocs := v.blk(blkReloc)
if len(relocs) != 5*23 {
t.Fatalf("relocs = %d bytes, want 5 entries", len(relocs))
}
// The function is the last non-package symbol; its relocs start after
// the DWARF symbols' (defs 2 and 3 each carry one).
le := binary.LittleEndian
first := int(le.Uint32(relocIdx[4*(4+4):]))
wantType := []uint16{relocRISCVPcrelItype, relocRISCVPcrelItype, relocRISCVPcrelStype}
wantOffAbs := []int{0, 8, 16}
for i := 0; i < 3; i++ {
e := relocs[(first+i)*23:]
if int32(le.Uint32(e[0:])) != int32(wantOffAbs[i]) || e[4] != 8 || le.Uint16(e[5:]) != wantType[i] ||
le.Uint32(e[15:]) != pkgIdxSelf || le.Uint32(e[19:]) != 0 {
t.Errorf("reloc %d = off %d size %d type %d pkg %d sym %d", i, int32(le.Uint32(e[0:])), e[4], le.Uint16(e[5:]), le.Uint32(e[15:]), le.Uint32(e[19:]))
}
}
// The function code: three AUIPC+second-instruction pairs with zero
// immediates, then the uncompressed JALR X0, 0(X1) the toolchain emits
// for RET.
code := img.Code[fn.Offset : fn.Offset+fn.Size]
want := append(wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 10, 0)), wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 10, 10, 0))...)
want = append(want, wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 11, 0))...)
want = append(want, wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 11, 11, 0))...)
want = append(want, wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 31, 0))...)
want = append(want, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 12, 0))...)
want = append(want, 0x67, 0x80, 0x00, 0x00) // JALR X0, 0(X1)
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
// The same bytes must survive into the object's data block intact: the
// linker patches only the immediate fields of the AUIPC pairs, so the
// opcode/register bits of every instruction must not be zeroed.
dataIdx := v.blk(blkDataIdx)
dataBlk := v.blk(blkData)
dOff := int(le.Uint32(dataIdx[8*4:])) // the function is the last symbol
emitted := dataBlk[dOff : dOff+fn.Size]
if !bytes.Equal(emitted, want) {
t.Errorf("emitted data = % x\nwant % x", emitted, want)
}
}
// TestGOObjectRISCVLink cross-compiles a Go program with the gasm-produced
// object substituted into the package archive, proving cmd/link accepts the
// emitted RISC-V GOOBJ. The binary is not executed (no riscv64 host or
// qemu). Skipped when no Go toolchain is available.
func TestGOObjectRISCVLink(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
asmSrc := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOV a+0(FP), X10
MOV b+8(FP), X11
ADD X11, X10, X10
MOV X10, ret+16(FP)
RET
`
if err := os.WriteFile(filepath.Join(dir, "main_riscv64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
mainSrc := `package main
func add(a, b int64) int64
func main() {
if add(20, 22) != 42 {
panic("bad add")
}
}
`
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module rvlink\n\ngo 1.21\n"), 0o644); err != nil {
t.Fatal(err)
}
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
build.Env = append(os.Environ(), "GOARCH=riscv64")
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var pkgArch, work, linkLine, asmObj string
for _, line := range strings.Split(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_riscv64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if pkgArch == "" || linkLine == "" || asmObj == "" {
t.Skip("could not locate the archive, asm output or link line in the build log")
}
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
asmMember := filepath.Base(strings.ReplaceAll(asmObj, "$WORK", work))
pf, perrs := parser.Parse(filepath.Join(dir, "main_riscv64.s"), asmSrc)
if len(perrs) > 0 {
t.Fatalf("parse: %v", perrs)
}
pimg, err := AssembleFileRISCV(pf)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
obj, err := pimg.GOObjectRISCV("main", filepath.Join(dir, "main_riscv64.s"))
if err != nil {
t.Fatalf("GOObjectRISCV: %v", err)
}
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
extract.Dir = membersDir
extract.Env = append(os.Environ(), "GOARCH=riscv64")
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
member := filepath.Join(membersDir, asmMember)
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, obj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listCmd.Env = append(os.Environ(), "GOARCH=riscv64")
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for _, m := range strings.Fields(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, filepath.Join(membersDir, m))
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
pack.Env = append(os.Environ(), "GOARCH=riscv64")
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
link := exec.Command("sh", "-c", linkLine)
link.Dir = dir
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)), "GOARCH=riscv64")
if out, err := link.CombinedOutput(); err != nil {
t.Fatalf("link with gasm object: %v\n%s", err, out)
}
nm := exec.Command(goBin, "tool", "nm", filepath.Join(dir, "app2"))
nm.Env = append(os.Environ(), "GOARCH=riscv64")
nmOut, err := nm.CombinedOutput()
if err != nil {
t.Fatalf("nm gasm-linked binary: %v\n%s", err, nmOut)
}
if !strings.Contains(string(nmOut), "main.add") {
t.Errorf("main.add not found in linked binary:\n%s", nmOut)
}
}
// hasELFRISCVJAL reports whether the ELF object carries an R_RISCV_JAL
// relocation in its .rela.text section.
func hasELFRISCVJAL(t *testing.T, data []byte) bool {
t.Helper()
f, err := elf.NewFile(bytes.NewReader(data))
if err != nil {
t.Fatalf("parse ELF: %v", err)
}
defer f.Close()
rela := f.Section(".rela.text")
if rela == nil {
return false
}
b, err := rela.Data()
if err != nil {
t.Fatalf(".rela.text data: %v", err)
}
const rRISCVJAL = 17
for i := 0; i+24 <= len(b); i += 24 {
info := binary.LittleEndian.Uint64(b[i+8:])
if uint32(info) == rRISCVJAL {
return true
}
}
return false
}
+244 -45
View File
@@ -34,7 +34,7 @@ import (
// version is the release version, stamped at build time via // version is the release version, stamped at build time via
// -ldflags "-X main.version=…" (defaulting to the current release). // -ldflags "-X main.version=…" (defaulting to the current release).
var version = "0.29.0" var version = "0.30.0"
func main() { func main() {
if len(os.Args) < 2 { if len(os.Args) < 2 {
@@ -395,34 +395,30 @@ hover, document symbols, diagnostics and semantic-token highlighting.
} }
func cmdAsm(args []string) int { func cmdAsm(args []string) int {
fs := newCommand("asm", "gasm asm [--format raw|elf|macho|goobj] [-p pkg] [-o out] <file>", ` fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>", `
Assemble FILE (amd64 or riscv64) without the Go toolchain: every TEXT function is Assemble FILE without the Go toolchain: every TEXT function is encoded to
encoded to machine code — scalar, VEX/AVX2 and EVEX/AVX-512 instructions, machine code and printed as a hex dump. Supported architectures: amd64
FP/SP frame mapping, local labels and file-local static symbols (GLOBL/DATA) (including VEX/AVX2 and EVEX/AVX-512), riscv64 (RV64IMAFDC + RVC) and
resolved RIP-relative — and printed as a hex dump. loong64 (LoongArch base ISA); arm64 encoding is not yet implemented.
With -o the output is written to a file instead. The --format flag selects With -o the output is written to a file instead. The --format flag selects
what is written: raw (the default) concatenates the functions and the data what is written: raw (the default) concatenates the functions and the data
section into one self-consistent image; elf and macho emit a relocatable section into one self-consistent image; elf emits a relocatable object
object (.text/.data sections, a symbol table and one PC32 relocation per (.text/.data sections, a symbol table and one PC32 relocation per
static-symbol reference) that links with the system toolchain; goobj emits static-symbol reference) that links with the system toolchain; goobj emits
the Go toolchain's own object format, which cmd/link consumes directly (it the Go toolchain's own object format, which cmd/link consumes directly (it
requires -p, the package path, and the installed Go toolchain). requires -p, the package path, and the installed Go toolchain).
`) `)
out := fs.String("o", "", "write the output to this file") out := fs.String("o", "", "write the output to this file")
format := fs.String("format", "raw", "output format: raw (concatenated image), elf, macho or goobj (Go object)") format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)") pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
fs.Parse(args) fs.Parse(args)
if fs.NArg() != 1 { if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|macho|goobj] [-p pkg] [-o out] <file>") fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>")
return 2 return 2
} }
path := fs.Arg(0) path := fs.Arg(0)
targetArch := arch.FromFilename(path) targetArch := arch.FromFilename(path)
if targetArch != arch.AMD64 && targetArch != arch.RISCV {
fmt.Fprintln(os.Stderr, "gasm asm: only amd64 and riscv64 are supported")
return 1
}
src, err := readSource(path) src, err := readSource(path)
if err != nil { if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err) fmt.Fprintln(os.Stderr, "gasm:", err)
@@ -436,12 +432,7 @@ requires -p, the package path, and the installed Go toolchain).
return 1 return 1
} }
var img *asm.Image img, err := assembleFile(path, targetArch, f)
if targetArch == arch.RISCV {
img, err = asm.AssembleFileRISCV(f)
} else {
img, err = asm.AssembleFile(f)
}
if err != nil { if err != nil {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err) fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
return 1 return 1
@@ -497,29 +488,36 @@ requires -p, the package path, and the installed Go toolchain).
switch *format { switch *format {
case "raw": case "raw":
if len(img.Externals) > 0 { if len(img.Externals) > 0 {
fmt.Fprintf(os.Stderr, "gasm asm: external symbol %q needs an object file (use --format elf or --format macho)\n", img.Externals[0]) fmt.Fprintf(os.Stderr, "gasm asm: external symbol %q needs an object file (use --format elf)\n", img.Externals[0])
return 1 return 1
} }
obj, kind = img.Bytes(), "raw image" obj, kind = img.Bytes(), "raw image"
case "elf": case "elf":
if targetArch == arch.RISCV { switch targetArch {
case arch.RISCV:
obj, err = img.ELFRISCVObject() obj, err = img.ELFRISCVObject()
} else { case arch.LOONG64:
obj, err = img.ELFLOONG64Object()
case arch.ARM64:
obj, err = img.ELFAARCH64Object()
default:
obj, err = img.ELFObject() obj, err = img.ELFObject()
} }
kind = "ELF object" kind = "ELF object"
case "macho":
obj, err = img.MachOObject()
kind = "Mach-O object"
case "goobj": case "goobj":
if targetArch == arch.RISCV { switch targetArch {
case arch.RISCV:
obj, err = img.GOObjectRISCV(*pkg, path) obj, err = img.GOObjectRISCV(*pkg, path)
} else { case arch.LOONG64:
obj, err = img.GOObjectLOONG64(*pkg, path)
case arch.ARM64:
obj, err = img.GOObjectAARCH64(*pkg, path)
default:
obj, err = img.GOObject(*pkg, path) obj, err = img.GOObject(*pkg, path)
} }
kind = "Go object" kind = "Go object"
default: default:
fmt.Fprintf(os.Stderr, "gasm asm: unknown format %q (want raw, elf, macho or goobj)\n", *format) fmt.Fprintf(os.Stderr, "gasm asm: unknown format %q (want raw, elf or goobj)\n", *format)
return 2 return 2
} }
if err != nil { if err != nil {
@@ -568,12 +566,12 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
} }
// Assemble both files. // Assemble both files.
img1, err := assembleFile(path1) img1, err := assemblePath(path1)
if err != nil { if err != nil {
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err) fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err)
return 1 return 1
} }
img2, err := assembleFile(path2) img2, err := assemblePath(path2)
if err != nil { if err != nil {
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err) fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err)
return 1 return 1
@@ -632,8 +630,24 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
return 1 return 1
} }
// assembleFile assembles a file and returns the image. // assembleFile assembles a parsed file for the given architecture and returns the image.
func assembleFile(path string) (*asm.Image, error) { func assembleFile(path string, targetArch arch.Arch, f *ast.File) (*asm.Image, error) {
switch targetArch {
case arch.AMD64:
return asm.AssembleFile(f)
case arch.RISCV:
return asm.AssembleFileRISCV(f)
case arch.ARM64:
return asm.AssembleFileARM64(f)
case arch.LOONG64:
return asm.AssembleFileLOONG64(f)
default:
return nil, fmt.Errorf("unsupported architecture %q", targetArch)
}
}
// assemblePath reads, parses and assembles a file (used by cmdDiff).
func assemblePath(path string) (*asm.Image, error) {
src, err := readSource(path) src, err := readSource(path)
if err != nil { if err != nil {
return nil, err return nil, err
@@ -645,11 +659,7 @@ func assembleFile(path string) (*asm.Image, error) {
if len(errs) > 0 { if len(errs) > 0 {
return nil, fmt.Errorf("parse errors") return nil, fmt.Errorf("parse errors")
} }
targetArch := arch.FromFilename(path) return assembleFile(path, arch.FromFilename(path), f)
if targetArch == arch.RISCV {
return asm.AssembleFileRISCV(f)
}
return asm.AssembleFile(f)
} }
// printByteDiff shows the first few byte differences between two code blocks. // printByteDiff shows the first few byte differences between two code blocks.
@@ -822,6 +832,188 @@ func cmdVerifyRISCV(path string, groundTruth, profile bool) int {
return 0 return 0
} }
// cmdVerifyLOONG64 verifies a loong64 source file against `go tool asm`
// (GOARCH=loong64) — the ground-truth oracle — since gasm cannot JIT-load
// LoongArch code on an amd64 host. Relocation sites are masked before the
// byte comparison, as the toolchain leaves them zero for the linker.
func cmdVerifyLOONG64(path string, groundTruth, profile bool) int {
src, err := readSource(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
f, errs := parser.Parse(path, src)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return 1
}
img, err := asm.AssembleFileLOONG64(f)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
if groundTruth {
gt, err := verify.GroundTruthLOONG64(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: ground truth: %v\n", err)
return 1
}
matched, total := 0, 0
for _, fn := range img.Funcs {
gasmCode := img.Code[fn.Offset : fn.Offset+fn.Size]
goCode, ok := gt[fn.Name]
if !ok {
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", fn.Name)
continue
}
total++
gasmCmp := make([]byte, len(gasmCode))
goCmp := make([]byte, len(goCode))
copy(gasmCmp, gasmCode)
copy(goCmp, goCode)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ {
gasmCmp[j] = 0
}
for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ {
goCmp[j] = 0
}
}
if bytes.Equal(gasmCmp, goCmp) {
matched++
if len(fn.Relocs) > 0 {
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
} else {
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
}
} else {
fmt.Printf(" %s: MISMATCH (%d vs %d bytes)\n", fn.Name, fn.Size, len(goCode))
for i := 0; i < len(gasmCode) || i < len(goCode); i += 16 {
var gb, gs string
for j := i; j < i+16 && j < len(gasmCode); j++ {
gb += fmt.Sprintf(" %02x", gasmCode[j])
}
for j := i; j < i+16 && j < len(goCode); j++ {
gs += fmt.Sprintf(" %02x", goCode[j])
}
fmt.Printf(" %04x: gasm:%s\n", i, gb)
fmt.Printf(" %04x: gt: %s\n", i, gs)
}
}
}
fmt.Printf("%s: %d/%d matched\n", path, matched, total)
if matched < total {
return 1
}
return 0
}
if profile {
for _, fn := range img.Funcs {
fmt.Printf("%s: %d bytes, labels: %v\n", fn.Name, fn.Size, fn.Labels)
}
return 0
}
fmt.Printf("%s: %d functions assembled\n", path, len(img.Funcs))
for _, fn := range img.Funcs {
fmt.Printf(" %s: %d bytes\n", fn.Name, fn.Size)
}
return 0
}
func cmdVerifyARM64(path string, groundTruth, profile bool) int {
src, err := readSource(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
f, errs := parser.Parse(path, src)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return 1
}
img, err := asm.AssembleFileARM64(f)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
if groundTruth {
gt, err := verify.GroundTruthARM64(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: ground truth: %v\n", err)
return 1
}
matched, total := 0, 0
for _, fn := range img.Funcs {
gasmCode := img.Code[fn.Offset : fn.Offset+fn.Size]
goCode, ok := gt[fn.Name]
if !ok {
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", fn.Name)
continue
}
total++
gasmCmp := make([]byte, len(gasmCode))
goCmp := make([]byte, len(goCode))
copy(gasmCmp, gasmCode)
copy(goCmp, goCode)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ {
gasmCmp[j] = 0
}
for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ {
goCmp[j] = 0
}
}
if bytes.Equal(gasmCmp, goCmp) {
matched++
if len(fn.Relocs) > 0 {
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
} else {
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
}
} else {
fmt.Printf(" %s: MISMATCH (%d vs %d bytes)\n", fn.Name, fn.Size, len(goCode))
for i := 0; i < len(gasmCode) || i < len(goCode); i += 16 {
var gb, gs string
for j := i; j < i+16 && j < len(gasmCode); j++ {
gb += fmt.Sprintf(" %02x", gasmCode[j])
}
for j := i; j < i+16 && j < len(goCode); j++ {
gs += fmt.Sprintf(" %02x", goCode[j])
}
fmt.Printf(" %04x: gasm:%s\n", i, gb)
fmt.Printf(" %04x: gt: %s\n", i, gs)
}
}
}
fmt.Printf("%s: %d/%d matched\n", path, matched, total)
if matched < total {
return 1
}
return 0
}
if profile {
for _, fn := range img.Funcs {
fmt.Printf("%s: %d bytes, labels: %v\n", fn.Name, fn.Size, fn.Labels)
}
return 0
}
fmt.Printf("%s: %d functions assembled\n", path, len(img.Funcs))
for _, fn := range img.Funcs {
fmt.Printf(" %s: %d bytes\n", fn.Name, fn.Size)
}
return 0
}
func cmdVerify(args []string) int { func cmdVerify(args []string) int {
fs := newCommand("verify", "gasm verify [-smoke] [-abi] [-fuzz] [-ground-truth] [-profile] [-call] <file.s>", ` fs := newCommand("verify", "gasm verify [-smoke] [-abi] [-fuzz] [-ground-truth] [-profile] [-call] <file.s>", `
Assemble FILE (amd64), map it into executable memory and report the available Assemble FILE (amd64), map it into executable memory and report the available
@@ -867,14 +1059,21 @@ decoders) that crash on random input but should succeed on valid data.
} }
path := fs.Arg(0) path := fs.Arg(0)
targetArch := arch.FromFilename(path) targetArch := arch.FromFilename(path)
if targetArch != arch.AMD64 && targetArch != arch.RISCV { switch targetArch {
fmt.Fprintln(os.Stderr, "gasm verify: only amd64 and riscv64 are supported") case arch.AMD64:
return 1 // JIT-based verification below.
} case arch.RISCV:
// RISC-V: ground-truth only (no JIT on non-RISC-V hosts).
// RISC-V: ground-truth only (no JIT on non-RISC-V hosts).
if targetArch == arch.RISCV {
return cmdVerifyRISCV(path, *groundTruth, *profile) return cmdVerifyRISCV(path, *groundTruth, *profile)
case arch.LOONG64:
// LoongArch: ground-truth only (no JIT on non-LoongArch hosts).
return cmdVerifyLOONG64(path, *groundTruth, *profile)
case arch.ARM64:
// AArch64: ground-truth only (no JIT on non-ARM64 hosts).
return cmdVerifyARM64(path, *groundTruth, *profile)
default:
fmt.Fprintln(os.Stderr, "gasm verify: only amd64, riscv64 and loong64 are supported")
return 1
} }
k, err := verify.Load(path) k, err := verify.Load(path)
+50
View File
@@ -271,3 +271,53 @@ func TestBreakpointInfo(t *testing.T) {
t.Errorf("Info %q does not contain label", info) t.Errorf("Info %q does not contain label", info)
} }
} }
func TestWatchpointSlotTracking(t *testing.T) {
s := &Session{}
// All four slots are free initially.
for i := 0; i < 4; i++ {
if s.IsWatchpointSlotUsed(i) {
t.Errorf("slot %d should be free initially", i)
}
}
if got := s.FindFreeWatchpointSlot(); got != 0 {
t.Errorf("FindFreeWatchpointSlot() = %d, want 0", got)
}
// Manually mark slots 0 and 2 as used (simulating successful SetWatchpoint).
s.wpSlots[0] = true
s.wpSlots[2] = true
if !s.IsWatchpointSlotUsed(0) {
t.Error("slot 0 should be in use")
}
if s.IsWatchpointSlotUsed(1) {
t.Error("slot 1 should be free")
}
if !s.IsWatchpointSlotUsed(2) {
t.Error("slot 2 should be in use")
}
if s.IsWatchpointSlotUsed(3) {
t.Error("slot 3 should be free")
}
if got := s.FindFreeWatchpointSlot(); got != 1 {
t.Errorf("FindFreeWatchpointSlot() = %d, want 1", got)
}
// Out-of-range slot queries return false.
if s.IsWatchpointSlotUsed(-1) {
t.Error("slot -1 should be reported as free (out of range)")
}
if s.IsWatchpointSlotUsed(4) {
t.Error("slot 4 should be reported as free (out of range)")
}
// Mark all slots used: FindFreeWatchpointSlot returns -1.
for i := 0; i < 4; i++ {
s.wpSlots[i] = true
}
if got := s.FindFreeWatchpointSlot(); got != -1 {
t.Errorf("FindFreeWatchpointSlot() with all slots used = %d, want -1", got)
}
}
+2 -1
View File
@@ -25,7 +25,8 @@ type Session struct {
cmd *exec.Cmd cmd *exec.Cmd
stopped bool stopped bool
exited bool exited bool
codeBase uint64 // base address of the JIT code in the debuggee codeBase uint64 // base address of the JIT code in the debuggee
wpSlots [4]bool // watchpoint slot occupancy (DR0-DR3)
} }
// Launch starts the debuggee subprocess (gasm debug --target ...) and // Launch starts the debuggee subprocess (gasm debug --target ...) and
+25 -14
View File
@@ -384,8 +384,8 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Println(` break <label|addr> [if <reg> <op> <val>] set a breakpoint fmt.Println(` break <label|addr> [if <reg> <op> <val>] set a breakpoint
delete <label|addr> remove a breakpoint delete <label|addr> remove a breakpoint
info break list all breakpoints info break list all breakpoints
watch <addr> [r|w] set a hardware watchpoint (write by default) watch <addr> [r|w] [size] set a hardware watchpoint (write by default)
unwatch clear all watchpoints unwatch [<slot>] clear one or all watchpoints
step [n], s single-step n instructions step [n], s single-step n instructions
next, n step over CALL next, n step over CALL
continue, c run until breakpoint or exit continue, c run until breakpoint or exit
@@ -457,28 +457,39 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
if len(parts) > 3 { if len(parts) > 3 {
size, _ = strconv.Atoi(parts[3]) size, _ = strconv.Atoi(parts[3])
} }
// Find a free slot (0-3). slot := s.FindFreeWatchpointSlot()
slot := -1
for i := 0; i < 4; i++ {
// Simple: use slot 0 for now.
slot = i
break
}
if slot < 0 { if slot < 0 {
fmt.Println("no free watchpoint slots") fmt.Println("no free watchpoint slots (use 'unwatch <slot>' to clear one)")
continue continue
} }
if err := s.SetWatchpoint(slot, addr, typ, size); err != nil { if err := s.SetWatchpoint(slot, addr, typ, size); err != nil {
fmt.Printf("watch: %v\n", err) fmt.Printf("watch: %v\n", err)
} else { } else {
fmt.Printf("watchpoint %d set: %#x (%s, %d bytes)\n", slot, addr, parts[2], size) typStr := "w"
if typ == WatchRead {
typStr = "r"
}
fmt.Printf("watchpoint %d set: %#x (%s, %d bytes)\n", slot, addr, typStr, size)
} }
case "unwatch": case "unwatch":
if err := s.ClearAllWatchpoints(); err != nil { if len(parts) >= 2 {
fmt.Printf("unwatch: %v\n", err) slot, err := strconv.Atoi(parts[1])
if err != nil || slot < 0 || slot > 3 {
fmt.Println("usage: unwatch [<slot>]")
continue
}
if err := s.ClearWatchpoint(slot); err != nil {
fmt.Printf("unwatch: %v\n", err)
} else {
fmt.Printf("watchpoint %d cleared\n", slot)
}
} else { } else {
fmt.Println("all watchpoints cleared") if err := s.ClearAllWatchpoints(); err != nil {
fmt.Printf("unwatch: %v\n", err)
} else {
fmt.Println("all watchpoints cleared")
}
} }
default: default:
+36 -4
View File
@@ -25,12 +25,34 @@ const (
WatchRead WatchpointType = 3 // trigger on read or write WatchRead WatchpointType = 3 // trigger on read or write
) )
// FindFreeWatchpointSlot returns the index of the first free watchpoint slot
// (0-3), or -1 if all four hardware watchpoints are in use.
func (s *Session) FindFreeWatchpointSlot() int {
for i := 0; i < 4; i++ {
if !s.wpSlots[i] {
return i
}
}
return -1
}
// IsWatchpointSlotUsed reports whether slot (0-3) currently holds a watchpoint.
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot > 3 {
return false
}
return s.wpSlots[slot]
}
// SetWatchpoint installs a hardware watchpoint on the given address. // SetWatchpoint installs a hardware watchpoint on the given address.
// slot is 0-3 (four hardware watchpoints available). // slot is 0-3 (four hardware watchpoints available); the slot must be free.
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error { func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
if slot < 0 || slot > 3 { if slot < 0 || slot > 3 {
return fmt.Errorf("debug: watchpoint slot must be 0-3") return fmt.Errorf("debug: watchpoint slot must be 0-3")
} }
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
// Determine the length encoding. // Determine the length encoding.
var lenBits uint64 var lenBits uint64
@@ -82,6 +104,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil { if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
return fmt.Errorf("debug: set DR7: %w", err) return fmt.Errorf("debug: set DR7: %w", err)
} }
s.wpSlots[slot] = true
return nil return nil
} }
@@ -90,20 +113,29 @@ func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot > 3 { if slot < 0 || slot > 3 {
return fmt.Errorf("debug: watchpoint slot must be 0-3") return fmt.Errorf("debug: watchpoint slot must be 0-3")
} }
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
// Read DR7, clear the enable bit for this slot. // Read DR7, clear the enable bit for this slot.
dr7, err := ptracePeekUser(s.pid, 0x38) dr7, err := ptracePeekUser(s.pid, 0x38)
if err != nil { if err != nil {
return err return err
} }
dr7 &^= uint64(1) << (2 * slot) // disable dr7 &^= uint64(1) << (2 * slot) // disable
return ptracePokeUser(s.pid, 0x38, dr7) if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
return err
}
s.wpSlots[slot] = false
return nil
} }
// ClearAllWatchpoints removes all hardware watchpoints. // ClearAllWatchpoints removes all hardware watchpoints.
func (s *Session) ClearAllWatchpoints() error { func (s *Session) ClearAllWatchpoints() error {
for slot := 0; slot < 4; slot++ { for slot := 0; slot < 4; slot++ {
if err := s.ClearWatchpoint(slot); err != nil { if s.wpSlots[slot] {
return err if err := s.ClearWatchpoint(slot); err != nil {
return err
}
} }
} }
return nil return nil
+40 -8
View File
@@ -201,6 +201,28 @@ R_RISCV_PCREL_HI20/LO12 relocations). The encoder compresses eligible
instructions to 16-bit RVC forms and is validated byte-for-byte against instructions to 16-bit RVC forms and is validated byte-for-byte against
`GOARCH=riscv64 go tool asm`. `GOARCH=riscv64 go tool asm`.
A **LoongArch encoder** (Phase 5, LoongArch64) encodes the integer and
floating-point instruction sets with the dual-form arithmetic mnemonics (3R
vs 2RI12), the 16/21-bit branch families, the MOV pseudo-instruction and its
constant materialisation (the dcon classification driving lu12i.w/ori/lu32i.d/
lu52i.d expansions), the FP/SP frame mapping (autosize = align8(frame+8),
prologue storing the link register before and after the SP decrement) and
SB/global symbol references (pcalau12i pairs with R_LOONG64_ADDR_HI/LO
relocations). Like the RISC-V encoder it is validated byte-for-byte against
`GOARCH=loong64 go tool asm`, and its GOOBJ output is proven end-to-end by
substituting it into a cross-compiled `go build` and linking with `cmd/link`.
An **AArch64 encoder** (Phase 5, arm64) encodes the integer instruction set
with the data-processing (shifted register and immediate forms), load/store
(scaled unsigned immediate and unscaled9-bit immediate), conditional and
unconditional branches, the MOV pseudo-instruction and its constant
materialisation (MOVZ/MOVN/MOVK for wide immediates, ORR with logical bitmask
encoding for values like `$1`), the FP/SP frame mapping (autosize =
align16(frame+8), prologue using pre-index store for small frames and
STP+SUB for large frames) and SB/global symbol references (ADRP+ADD pairs with
R_ADDRARM64 relocations). Like the other encoders it is validated
byte-for-byte against `GOARCH=arm64 go tool asm`.
On top of the encoder, `Assemble` walks a parsed `TEXT` body, converts each On top of the encoder, `Assemble` walks a parsed `TEXT` body, converts each
operand to an encoder operand, and lays the instructions out so local labels operand to an encoder operand, and lays the instructions out so local labels
resolve to relative jump offsets: jumps start in the short (rel8) form and resolve to relative jump offsets: jumps start in the short (rel8) form and
@@ -275,8 +297,8 @@ RIP-relative loads whose displacements point inside the resulting image, so
the bytes are self-consistent at any base address. References to symbols no the bytes are self-consistent at any base address. References to symbols no
`GLOBL` defines are kept as relocations on the function layout, and the `GLOBL` defines are kept as relocations on the function layout, and the
object-file emitters turn the whole image into a linkable object: the ELF object-file emitters turn the whole image into a linkable object: the ELF
and Mach-O writers (`gasm asm --format elf|macho`) lay the code and data out writer (`gasm asm --format elf`) lays the code and data out as `.text`/`.data`
as `.text`/`.data` (or `__text`/`__data`) sections, export a symbol per sections, exports a symbol per
`TEXT` and `GLOBL` (the `<>` ones local, the rest global) and emit one `TEXT` and `GLOBL` (the `<>` ones local, the rest global) and emit one
PC-relative relocation per static-symbol reference — undefined external PC-relative relocation per static-symbol reference — undefined external
symbols included, so the output links with the system toolchain. The GOOBJ symbols included, so the output links with the system toolchain. The GOOBJ
@@ -288,12 +310,22 @@ boundaries, plus flat `pcfile`, `pcline` and `pcinline` tables — so a
gasm-assembled object drops into a `go build` in place of the toolchain's. gasm-assembled object drops into a `go build` in place of the toolchain's.
The object preamble (the version-and-experiment header the linker compares The object preamble (the version-and-experiment header the linker compares
verbatim) is captured from the installed `go tool asm`, so the output is verbatim) is captured from the installed `go tool asm`, so the output is
always consistent with the toolchain that links it. RISC-V GOOBJ emission always consistent with the toolchain that links it. RISC-V and LoongArch
uses the same format with the RISC-V architecture marker and RISC-V relocation GOOBJ emission share this emitter: the loong64 marker with
types. External cross-package R_LOONG64_ADDR_HI/LO relocation types, and the riscv64 marker with a single
references and the implicit funcdata/DWARF symbols remain future work (the R_RISCV_PCREL_ITYPE/STYPE relocation per AUIPC pair (plus `R_RISCV_JAL` for
linker fills the latter's defaults); the rest of Phase 2 is those, the `CALL sym(SB)`) — the model `cmd/asm`
remaining EVEX forms and the other architectures. writes, not the ELF HI20/LO12 pair — and both link into a real `go build` for
their `GOARCH`. Per function, the emitter also writes the two DWARF
symbols the linker's DWARF pass reads verbatim — the subprogram DIE
(`SDWARFFCN`) and the `.debug_line` state-machine program (`SDWARFLINES`),
both built the way `cmd/asm` builds them (the DIE carries the
R_DWTXTADDR_U4 address reference; the line program one row per source-line
change, in the same special-opcode encoding) — and the pc-value deltas are
in the architecture's MinLC units, as the runtime's `pcvalue` expects.
External cross-package references remain future work (the amd64 and RISC-V
paths resolve them; LoongArch does not yet); the rest of Phase 2 is those
and the remaining EVEX forms.
### `verify` ### `verify`
+20 -7
View File
@@ -50,13 +50,13 @@ Rules: `unknown-instruction`, `operand-count`, `undefined-label`,
`abi-argsize`, `unreachable-code`, `register-clobber`, `abi-argsize`, `unreachable-code`, `register-clobber`,
`funcdata-pcdata`. `funcdata-pcdata`.
## `gasm asm [--format raw|elf|macho|goobj] [-p pkg] [-o out] <file>` ## `gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>`
Assemble FILE (amd64) to machine code. Assemble FILE (amd64) to machine code.
| Flag | Description | | Flag | Description |
|------|-------------| |------|-------------|
| `--format` | Output format: `raw` (default), `elf`, `macho`, `goobj` | | `--format` | Output format: `raw` (default), `elf`, `goobj` |
| `-p` | Package path (required for `--format goobj`) | | `-p` | Package path (required for `--format goobj`) |
| `-o` | Write output to file (default: hex dump to stdout) | | `-o` | Write output to file (default: hex dump to stdout) |
@@ -102,13 +102,26 @@ REPL commands:
| Command | Description | | Command | Description |
|---------|-------------| |---------|-------------|
| `break <label\|addr>` | Set a breakpoint | | `break <label\|addr> [if <reg> <op> <val>]` | Set a breakpoint, optionally conditional |
| `step [n]` | Single-step n instructions | | `delete <label\|addr>` | Remove a breakpoint |
| `continue` | Run until next breakpoint or exit | | `info break` | List all breakpoints |
| `step [n]`, `s` | Single-step n instructions |
| `next`, `n` | Step over CALL |
| `finish`, `fin` | Run until the function returns |
| `continue`, `c` | Run until breakpoint, watchpoint or exit |
| `disas [n]`, `u` | Disassemble n instructions at PC |
| `regs` | Print general-purpose + YMM/XMM vector registers | | `regs` | Print general-purpose + YMM/XMM vector registers |
| `where` | Show source line and nearest label at PC |
| `stack` | Show stack near RSP (return address + ABI0 args) |
| `bt`, `backtrace` | Backtrace (current frame + return address) |
| `x [addr] [len]` | Hex-dump memory | | `x [addr] [len]` | Hex-dump memory |
| `labels` | List function labels and offsets | | `w <addr> <val...>` | Write bytes to memory |
| `quit` | Kill the debuggee and exit | | `set <reg> <value>` | Set a register |
| `watch <addr> [r\|w] [size]` | Set a hardware watchpoint (write by default) |
| `unwatch [<slot>]` | Clear one or all watchpoints |
| `labels`, `l` | List function labels and offsets |
| `help`, `h`, `?` | Show command help |
| `quit`, `q` | Kill the debuggee and exit |
## `gasm diff [--map old=new,...] <file1.s> <file2.s>` ## `gasm diff [--map old=new,...] <file1.s> <file2.s>`
+34
View File
@@ -0,0 +1,34 @@
# Deferred decisions
Design decisions deliberately postponed, with enough context to pick them up
again without re-deriving the analysis. Each entry records what is deferred,
why, the options on the table, and the trigger that should reopen it.
---
## GOOBJ external (cross-package) symbol references
**Status:** resolved (v0.29.0+, 2026-08-07).
**Approach taken.** Instead of parsing the compiler's iexport data (which
would have required either `golang.org/x/tools` or an in-house parser), the
resolver reads the **GOOBJ data directly** from the target package's `.a`
archive. The `.a` file contains a `_go_.o` member whose GOOBJ format is the
same one gasm writes — the parser reuses the same layout (`blkSymdef`,
`blkNonpkgdef`, the string table), so no new dependency was needed.
**How it works.**
1. `go list -json -export <pkg>` finds the target package's `.a` file.
2. `extractGOOBJ` reads the ar archive, finds the `_go_.o` member, skips
the `"go object …\n!\n"` preamble and parses the GOOBJ header.
3. `goobjFile.symbols()` walks `blkSymdef` and `blkNonpkgdef` in definition
order — the same order the linker uses — to build the symbol → index
mapping.
4. `resolveExternalSymbols` wires the resolved `{PkgIdx, SymIdx}` into the
GOOBJ emission.
The resolver is invoked automatically when `img.Externals` is non-empty; it
runs `go list` as a subprocess (consistent with `toolchainObjectPreamble`
which already calls `go tool asm`). All symbol data is cached per package
for the lifetime of the GOOBJ emission.
-56
View File
@@ -1,56 +0,0 @@
# Deferred decisions
Design decisions deliberately postponed, with enough context to pick them up
again without re-deriving the analysis. Each entry records what is deferred,
why, the options on the table, and the trigger that should reopen it.
---
## GOOBJ external (cross-package) symbol references
**Status:** deferred (v0.15.0, 2026-08-02). The GOOBJ emitter resolves only
symbols defined in the file being assembled; a reference to any other symbol
is rejected.
**Why it is deferred.** GOOBJ symbol references are *positional*: a
reference is a `{PkgIdx, SymIdx}` pair, where `SymIdx` is the index of the
symbol in the *referenced package's* symbol-definition table. That ordering
is not derivable from the reference site — it lives in the referenced
package's gc export data (the iexport binary format, which evolves with the
toolchain). `cmd/asm` reads it with `cmd/internal` readers gasm cannot
import, so emitting external references means either parsing export data
ourselves or taking a dependency that does.
**What works today.** Single-package objects: every symbol the file defines
(as `TEXT` or `GLOBL`, static or exported) and every reference to them.
This covers the production use case — the go-flac / go-lz4 kernels carry no
`FUNCDATA`/`PCDATA`, hence no references into `runtime`, and the Go side
references the assembly symbols, never the reverse. Such a package builds
with its assembly object replaced by a gasm-emitted one.
**The options, when we return.**
1. **`golang.org/x/tools/go/gcexportdata` as a production dependency.**
The straightforward path: read each imported package's export file
(paths from `-importcfg` or `go list -export`), assign symbol indices in
its symbol order, write `PkgIndex`/`Autolib` entries (fingerprints from
the export files' build IDs) and positional references. Robust across
toolchain versions — `x/tools` tracks the format. **Cost:** the first
production dependency beyond the standard library, an explicit deviation
from the "production code depends only on the standard library"
principle in the README. Requires the user's explicit agreement.
2. **A minimal iexport parser of our own.** Preserves self-containment.
Substantial effort and inherently fragile: the format is an internal
contract that changes with Go releases, so the parser needs a
version-gated fallback and regression tests against several toolchains.
3. **Shell out to the toolchain for symbol metadata.** Consistent with the
existing GOOBJ preamble probe (which already runs `go tool asm`), but no
toolchain command exposes a package's symbols *in definition-index
order* — `go tool nm` sorts differently — so this does not solve the
core problem on its own; it would only feed option 1 or 2.
**Trigger to reopen.** An assembly file that needs a cross-package
reference — in practice `FUNCDATA $…, runtime·…(SB)` (stack maps / GC
metadata written in assembly), or any kernel that calls into another
package directly. Until then, option 3's limitation is moot and the
single-package emitter suffices.
+1 -1
View File
@@ -4,7 +4,7 @@ Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrb
## Prerequisites ## Prerequisites
- **Go** 1.26+ with `toolchain go1.26.5` - **Go** 1.27+ with `toolchain go1.27.0`
- **just** — the command runner; every task below is a just recipe - **just** — the command runner; every task below is a just recipe
- No external dependencies beyond the Go toolchain - No external dependencies beyond the Go toolchain
-79
View File
@@ -1,79 +0,0 @@
# Using gasm-devkit with Zed
This document is deliberately blunt, because the situation is a genuine
conflict between two of the project's own commitments, and papering over it
would be dishonest.
## The conflict
gasm-devkit is **pure Go, no C, no cgo, no JavaScript runtimes, no native
binaries, no vendor lock-in, no platform-specific IDE internals.**
Zed's extension model, as verified against Zed's own documentation, is:
- Extensions are written in **Rust** and compiled to **WebAssembly**
(`wasm32-wasip2`).
- Syntax highlighting is provided by **Tree-sitter** grammars, which are
**C** compiled to WebAssembly with the wasi-sdk, from a grammar written in a
**JavaScript** DSL.
- A *new* language cannot be registered through configuration alone. Defining
a language requires an extension, and every language extension must name a
Tree-sitter grammar. (Zed's `lsp` settings section configures
already-registered servers; it does not register an arbitrary external binary
for a brand-new language.)
There is therefore **no pure-Go path into Zed's extension host.** This is a
property of Zed, not of gasm-devkit: no language tooling author can feed Zed a
pure-Go highlighting grammar, because Zed's highlighting engine is Tree-sitter
and its plugin runtime is Rust/WASM.
## What gasm-devkit gives Zed regardless
The toolkit's integration surface is the **Language Server Protocol**, an open
standard. Through `gasm lsp` it provides, with zero editor-specific code:
- autocomplete (instructions, registers, pseudo-registers, labels),
- hover documentation,
- diagnostics (the linter, pushed as you type),
- document outline (functions and labels),
- **syntax highlighting, delivered as LSP semantic tokens.**
That last point matters: Zed can render highlighting entirely from LSP semantic
tokens (`"semantic_tokens": "full"` replaces Tree-sitter highlighting for a
language). So the highlighting *capability* exists in pure Go; what Zed needs
is merely to be told that `.s` files are a language served by `gasm lsp`.
## The honest options
1. **Use an editor that registers an external LSP by configuration.**
Neovim, Helix, VS Code and Sublime all let you associate `.s` with the
`gasm lsp` binary and use its semantic tokens — no Rust, no C, no lock-in.
This is the option that satisfies every stated constraint with no
exception.
2. **Treat a Zed adapter as one quarantined exception.** A minimal Zed
extension — a few lines of Rust that register the language and launch
`gasm lsp` — plus either a Tree-sitter grammar or `"full"` semantic tokens
for highlighting. Crucially, this adapter is the *editor's plugin format*;
it is sandboxed inside Zed and never linked into, compiled into, or shipped
with the Go toolkit. gasm-devkit itself stays pure Go. But producing it
uses the Rust/wasi-sdk/Tree-sitter toolchain, which the project constraints
forbid — so it must be a conscious, explicit decision, not a silent one.
The author's philosophy — digital sovereignty, no dependency on toolchains he
does not control — is the tie-breaker, and it is a value judgement rather than
a technical one. gasm-devkit is built so that **either** choice keeps the
toolkit itself clean: the pure-Go core and the LSP are the product; a Zed
adapter, if ever wanted, is a thin, separable leaf.
## Wiring the LSP (editor-agnostic)
Run the server and point an LSP client at it:
```sh
go run ./cmd/gasm lsp # or: go install ./cmd/gasm && gasm lsp
```
Associate the command with `*.s` (and `*_amd64.s` / `*_arm64.s`) in whichever
editor you use. The server infers the target architecture from the file-name
suffix and selects the amd64 or arm64 instruction tables accordingly.
+2 -2
View File
@@ -1,7 +1,7 @@
module sourcedock.dev/petrbalvin/gasm-devkit module sourcedock.dev/petrbalvin/gasm-devkit
go 1.26 go 1.27
toolchain go1.26.5 toolchain go1.27.0
require golang.org/x/arch v0.29.0 require golang.org/x/arch v0.29.0
+1 -1
View File
@@ -3,7 +3,7 @@
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm). # gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
version := "0.29.0" version := "0.30.0"
default: default:
@just --list @just --list
+5 -1
View File
@@ -298,10 +298,14 @@ func parseSymbolPrefix(g []token.Token) (*ast.Symbol, int) {
sym.Static = true sym.Static = true
i += 2 i += 2
} }
if i < len(g) && g[i].Kind == token.Plus { if i < len(g) && (g[i].Kind == token.Plus || g[i].Kind == token.Minus) {
neg := g[i].Kind == token.Minus
i++ i++
if i < len(g) && g[i].Kind == token.Number { if i < len(g) && g[i].Kind == token.Number {
sym.Offset, sym.HasOff = parseInt(g[i].Text), true sym.Offset, sym.HasOff = parseInt(g[i].Text), true
if neg {
sym.Offset = -sym.Offset
}
i++ i++
} }
} }
+57
View File
@@ -0,0 +1,57 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// add returns a + b.
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
// arith exercises the register-register integer set.
TEXT ·arith(SB), NOSPLIT, $0-0
ADD R4, R5, R6
SUB R7, R8, R9
AND R10, R11, R12
ORR R12, R13, R14
EOR R14, R15, R16
CMP R16, R17
ADD R4, R5
SUB R6, R7
RET
// branch exercises conditional and unconditional control flow.
TEXT ·branch(SB), NOSPLIT, $0-0
BEQ done
BNE skip
BGE done
BLT done
BGT done
BLE done
skip:
B loop
loop:
ADD R4, R5
RET
done:
RET
// mov exercises the MOV pseudo-instruction.
TEXT ·mov(SB), NOSPLIT, $0-16
MOVD $0, R4
MOVD $1, R5
MOVD $42, R6
MOVD a+0(FP), R7
MOVD R7, ret+0(FP)
MOVW $100, R8
RET
// frame exercises the prologue/epilogue of a function with a real frame.
TEXT ·frame(SB), NOSPLIT, $32-8
MOVD arg+0(FP), R4
ADD $1, R4, R4
MOVD R4, ret+0(FP)
RET
+54
View File
@@ -0,0 +1,54 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// add returns a + b.
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
// arith exercises the 3R integer and FP set.
TEXT ·arith(SB), NOSPLIT, $0-0
ADDV R4, R5, R6
SUBV R7, R8, R9
MULV R10, R11, R12
DIVV R13, R14, R15
AND R16, R17, R18
OR R18, R19, R20
XOR R20, R21, R2
SLLV R2, R23, R24
SRLV R24, R25, R26
SRAV R26, R27, R28
RET
// imm exercises the immediate forms.
TEXT ·imm(SB), NOSPLIT, $0-0
ADDV $42, R4, R5
ADDV $-8, R6
AND $0xff, R7, R8
OR $1, R9, R10
XOR $0, R11, R12
SGT $100, R13, R14
SLLV $4, R15, R16
MOVV $0x12345, R17
RET
// branch exercises conditional and unconditional control flow.
TEXT ·branch(SB), NOSPLIT, $0-0
BEQ R4, R5, done
BNE R6, R7, skip
BLT R8, R9, done
BGE R10, R11, done
BLTU R12, R13, done
BGEU R14, R15, done
skip:
JMP loop
loop:
JAL skip
RET
done:
RET
+18
View File
@@ -0,0 +1,18 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·framed(SB), NOSPLIT, $16-16
MOV a+0(FP), X10
MOV b+8(FP), X11
ADD X11, X10, X10
MOV X10, ret+16(FP)
RET
TEXT ·leaf(SB), NOSPLIT, $0-16
MOV a+0(FP), X10
MOV b+8(FP), X11
ADD X11, X10, X10
MOV X10, ret+16(FP)
RET
+27
View File
@@ -0,0 +1,27 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// branch exercises all conditional branch forms and jump chain folding.
TEXT ·branch(SB), NOSPLIT, $0-0
BEQ done
BNE skip
BGE done
BLT done
BGT done
BLE done
BCS done
BCC done
BMI done
BPL done
BVS done
BVC done
BHI done
BLS done
skip:
B loop
loop:
ADD R4, R5
done:
RET
+35
View File
@@ -0,0 +1,35 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·branches(SB), NOSPLIT, $0
ADDI $1, X10, X10
BEQ X10, X11, beq_done
ADDI $2, X10, X10
beq_done:
BNE X10, X11, bne_done
ADDI $3, X10, X10
bne_done:
BLT X10, X11, blt_done
ADDI $4, X10, X10
blt_done:
BGE X10, X11, bge_done
ADDI $5, X10, X10
bge_done:
BLTU X10, X11, bltu_done
ADDI $6, X10, X10
bltu_done:
BGEU X10, X11, bgeu_done
ADDI $7, X10, X10
bgeu_done:
RET
TEXT ·jumps(SB), NOSPLIT, $0
JMP done
ADDI $1, X10, X10
done:
JAL X11, skip
ADDI $2, X10, X10
skip:
RET
+10
View File
@@ -0,0 +1,10 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// caller exercises BL to an external symbol (produces a relocation).
TEXT ·caller(SB), NOSPLIT, $0-0
BL other(SB)
ADD R4, R5
RET
+8
View File
@@ -0,0 +1,8 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·call(SB), NOSPLIT, $0
CALL callee(SB)
RET
+119
View File
@@ -0,0 +1,119 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// fparith exercises the FP arithmetic set.
TEXT ·fparith(SB), NOSPLIT, $0-0
FADDD F0, F1, F2
FSUBD F3, F4, F5
FMULD F6, F7, F8
FDIVD F9, F10, F11
FADDS F12, F13, F14
FSUBS F15, F16, F17
FMULS F18, F19, F20
FDIVS F21, F22, F23
FSQRTD F24, F25
FSQRTS F26, F27
FNEGD F28, F29
FNEGS F30, F31
FABSD F0, F1
FABSS F2, F3
FNMULD F4, F5, F6
FNMULS F7, F8, F9
FMIND F10, F11, F12
FMAXD F13, F14, F15
FMINS F16, F17, F18
FMAXS F19, F20, F21
RET
// fpfma exercises fused multiply-add.
TEXT ·fpfma(SB), NOSPLIT, $0-0
FMADDD F0, F1, F2, F3
FMSUBD F4, F5, F6, F7
FNMADDD F8, F9, F10, F11
FNMSUBD F12, F13, F14, F15
FMADDS F16, F17, F18, F19
FMSUBS F20, F21, F22, F23
FNMADDS F24, F25, F26, F27
FNMSUBS F28, F29, F30, F0
RET
// fpconv exercises FP↔integer conversion and cross-precision.
// Syntax: FCVTZSD Fd, Rn (float→int: FP source first, int dest second)
// SCVTFD Rn, Fd (int→float: int source first, FP dest second)
TEXT ·fpconv(SB), NOSPLIT, $0-0
FCVTSD F0, F1
FCVTDS F2, F3
FCVTZSD F4, R0
FCVTZSS F5, R1
FCVTZUD F6, R2
FCVTZUS F7, R3
SCVTFD R4, F8
SCVTFS R5, F9
UCVTFD R6, F10
UCVTFS R7, F11
SCVTFWD R0, F12
SCVTFWS R1, F13
UCVTFWD R2, F14
UCVTFWS R3, F15
FMOVS F14, R20
FMOVS R21, F15
FMOVD F16, R22
FMOVD R23, F17
RET
// fpcmp exercises FP compare and conditional compare.
// FCCMP syntax: FCCMP cond, Fn, Fm, $nzcv
// FCSEL syntax: FCSEL cond, Fn, Fm, Fd
TEXT ·fpcmp(SB), NOSPLIT, $0-0
FCMPS F0, F1
FCMPD F2, F3
FCMPS $0.0, F4
FCMPD $0.0, F5
FCCMPS EQ, F6, F7, $0
FCCMPD NE, F8, F9, $0
FCSELS GE, F10, F11, F12
FCSELD LT, F13, F14, F15
RET
// frint exercises FP rounding.
TEXT ·frint(SB), NOSPLIT, $0-0
FRINTND F0, F1
FRINTNS F2, F3
FRINTPD F4, F5
FRINTPS F6, F7
FRINTMD F8, F9
FRINTMS F10, F11
FRINTZD F12, F13
FRINTZS F14, F15
FRINTAD F16, F17
FRINTAS F18, F19
FRINTXD F20, F21
FRINTXS F22, F23
FRINTID F24, F25
FRINTIS F26, F27
FMOVD F0, F1
FMOVS F2, F3
RET
// condsel exercises conditional select and CRC32.
TEXT ·condsel(SB), NOSPLIT, $0-0
CSEL EQ, R0, R1, R2
CSINC NE, R3, R4, R5
CSINV GE, R6, R7, R8
CSNEG LT, R9, R10, R11
CSET EQ, R12
CSETM NE, R13
CINC EQ, R14, R15
CINV NE, R16, R17
CNEG GE, R19, R20
CRC32B R0, R2
CRC32H R3, R5
CRC32W R6, R8
CRC32X R9, R11
CRC32CB R12, R14
CRC32CH R15, R0
CRC32CW R1, R3
CRC32CX R4, R6
RET
+143
View File
@@ -0,0 +1,143 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// fp exercises the floating-point set: 3R arithmetic, 2R unary, compares
// into FCC, fused multiply-add and the register moves.
TEXT ·fp(SB), NOSPLIT, $0-0
ADDD F4, F5, F6
SUBD F7, F8, F9
MULD F9, F10, F11
DIVD F11, F12, F13
MULF F13, F14, F15
ADDF F15, F16, F17
SQRTD F17, F18
SQRTF F18, F19
ABSD F19, F20
NEGD F20, F21
MOVD F21, F22
CMPEQD F22, F23, FCC0
CMPGTF F23, F24, FCC1
CMPGED F24, F25, FCC2
FMADDD F0, F1, F2, F3
FMSUBF F3, F4, F5, F6
FNMADDD F6, F7, F8, F9
FNMSUBF F9, F10, F11, F12
FMAXD F12, F13, F14
FMINF F14, F15, F16
FMAXAD F16, F17, F18
FMINAF F18, F19, F20
FSCALEBF F20, F21, F22
FCOPYSGD F22, F23, F24
MOVV F25, R25
MOVV R26, F27
MOVW R28, F29
MOVW F30, R31
RET
// mov forms: register moves, immediates (12/32/64-bit), memory with FP/SP
// pseudo-registers and the register-indexed forms.
TEXT ·mov(SB), NOSPLIT, $0-16
MOVV R4, R5
MOVW R6, R7
MOVB R8, R9
MOVBU R10, R11
MOVHU R12, R13
MOVWU R14, R15
MOVV $42, R16
MOVV $0x12345, R17
MOVV $0x100000, R18
MOVW $-100, R19
MOVV $0x123456789, R20
MOVV a+0(FP), R21
MOVV R23, b+8(FP)
MOVW c+16(FP), R24
MOVV (R24)(R25), R26
MOVV R27, (R28)(R29)
RET
// frame exercises the prologue/epilogue of a function with a real frame.
TEXT ·frame(SB), NOSPLIT, $32-8
MOVV R4, R5
MOVV arg+0(FP), R6
MOVV R7, local-8(SP)
MOVV local-8(SP), R8
MOVV R9, ret+0(FP)
RET
// branches21 exercises the single-register and zero-register branch forms
// with 21-bit offsets.
TEXT ·branches21(SB), NOSPLIT, $0-0
BEQ R0, R4, l1
BEQ R5, R0, l2
BNE R0, R6, l3
BNE R7, R0, l4
BLTZ R8, l5
BGEZ R9, l6
BLEZ R10, l7
BGTZ R11, l8
JMP l9
l1:
JMP l10
l2:
JMP l11
l3:
JMP l12
l4:
JMP l13
l5:
JMP l14
l6:
JMP l15
l7:
JMP l16
l8:
JMP l16
l9:
MOVV R1, R2
l10:
LL (R12), R13
LLV (R14), R15
SC R16, (R17)
SCV R18, (R19)
RDTIMED R20, R21
SYSCALL
DBAR
RET
l11:
JAL (R30)
RET
l12:
BSTRINSV $7, R4, $0, R5
BSTRPICKV $63, R6, $32, R7
ALSLV $2, R8, R9, R10
ADDV16 $65536, R11, R12
RET
l13:
MOVV $0xffffffffffffffff, R13
RET
l14:
CPUCFG R14, R14
RET
l15:
NOR R15, R16, R17
ORN R18, R19, R20
ANDN R21, R24, R25
RET
l16:
MOVB R26, (R27)
MOVB (R28), R29
RET
// sbdata loads and stores a static symbol with relocations (the relocation
// fields are masked before comparison).
GLOBL ·table(SB), RODATA, $16
DATA ·table+0(SB)/8, $0x1122334455667788
DATA ·table+8(SB)/8, $0x8877665544332211
TEXT ·sbdata(SB), NOSPLIT, $0-0
MOVV $·table(SB), R4
MOVV ·table(SB), R5
MOVV R6, ·table+8(SB)
RET
+12
View File
@@ -0,0 +1,12 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·largeimm(SB), NOSPLIT, $0
ADDI $2048, X5
ADDI $4095, X5, X6
ANDI $4095, X5, X6
ORI $-4096, X5, X6
XORI $0x12345, X5, X6
RET
+24
View File
@@ -0,0 +1,24 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·ldst(SB), NOSPLIT, $0
LD (X8), X9
SD X9, (X8)
LW (X8), X9
SW X9, (X8)
LD 8(X2), X10
SD X10, 16(X2)
LW 4(X2), X11
SW X11, 8(X2)
RET
TEXT ·addi4spn(SB), NOSPLIT, $0
ADDI $16, X2, X8
RET
TEXT ·wordarith(SB), NOSPLIT, $0
ADDW X9, X8, X8
SUBW X9, X8, X8
RET
+21
View File
@@ -0,0 +1,21 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// movimm exercises MOV with various immediate values.
TEXT ·movimm(SB), NOSPLIT, $0-0
MOVD $0, R0
MOVD $1, R1
MOVD $42, R2
MOVD $255, R3
MOVD $256, R4
MOVD $0xFFFF, R5
MOVD $0x12345678, R6
MOVD $0x123456789ABCDEF0, R7
MOVD $-1, R8
MOVD $-2, R9
MOVW $0, R10
MOVW $100, R11
MOVW $0x12345, R12
RET
+19
View File
@@ -0,0 +1,19 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·movimm(SB), NOSPLIT, $0
MOV $0, X10
MOV $5, X10
MOV $42, X10
MOV $-1, X10
MOV $-2048, X10
MOV $-2049, X10
MOV $2047, X10
MOV $2048, X10
MOV $4095, X10
MOV $-4096, X10
MOV $0x12345, X10
MOV $2147483647, X10
RET
+32
View File
@@ -0,0 +1,32 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·shifts(SB), NOSPLIT, $0
SLLI $3, X10, X10
SRLI $2, X10, X10
SRAI $1, X10, X10
RET
TEXT ·logic(SB), NOSPLIT, $0
AND X11, X10, X10
OR X11, X10, X10
XOR X11, X10, X10
ANDI $7, X10, X10
RET
TEXT ·mv(SB), NOSPLIT, $0
ADDI $0, X11, X10
RET
TEXT ·nop(SB), NOSPLIT, $0
ADDI $0, X0
RET
TEXT ·bigframe(SB), NOSPLIT, $24-0
RET
TEXT ·ebreak(SB), NOSPLIT, $0
EBREAK
RET
+87
View File
@@ -0,0 +1,87 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"os"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGroundTruthARM64 assembles the arm64 test kernels with gasm and
// compares them byte-for-byte against `go tool asm` (GOARCH=arm64). The
// relocation fields of static-symbol references are masked before the
// comparison, since the toolchain leaves them zero for the linker.
func TestGroundTruthARM64(t *testing.T) {
for _, path := range []string{
"../testdata/verify/basic_arm64.s",
"../testdata/verify/fp_arm64.s",
"../testdata/verify/movimm_arm64.s",
"../testdata/verify/branch_arm64.s",
"../testdata/verify/call_arm64.s",
} {
t.Run(path, func(t *testing.T) {
src, err := os.ReadFile(path)
if err != nil {
t.Fatalf("read: %v", err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := asm.AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
gt, err := GroundTruthARM64(path)
if err != nil {
t.Fatalf("GroundTruthARM64: %v", err)
}
matched := 0
for _, fn := range img.Funcs {
gasmCode := maskRelocs(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
goCode, ok := gt[fn.Name]
if !ok {
t.Errorf("%s: not in ground truth (%d functions)", fn.Name, len(gt))
continue
}
goCode = maskRelocs(goCode, fn.Relocs)
// The Go toolchain may add zero padding at the end of
// functions. Compare up to the shorter length, then
// verify any trailing bytes are zero.
cmpLen := len(gasmCode)
if len(goCode) < cmpLen {
cmpLen = len(goCode)
}
if !bytes.Equal(gasmCode[:cmpLen], goCode[:cmpLen]) {
t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\n%s", fn.Name, len(gasmCode), len(goCode), diffHex(gasmCode, goCode))
continue
}
// Check trailing padding is zero.
trailingOK := true
if len(goCode) > len(gasmCode) {
for _, b := range goCode[len(gasmCode):] {
if b != 0 {
trailingOK = false
break
}
}
}
if !trailingOK {
t.Errorf("%s: non-zero trailing bytes in go tool asm output", fn.Name)
continue
}
matched++
t.Logf("%s: MATCH (%d bytes, go=%d)", fn.Name, fn.Size, len(goCode))
}
if matched == 0 {
t.Fatal("no functions matched")
}
})
}
}
+14
View File
@@ -32,6 +32,18 @@ func GroundTruthRISCV(path string) (map[string][]byte, error) {
return groundTruthArch(path, "riscv64") return groundTruthArch(path, "riscv64")
} }
// GroundTruthLOONG64 assembles the given .s file with the Go toolchain in
// LoongArch cross-assembly mode (GOARCH=loong64).
func GroundTruthLOONG64(path string) (map[string][]byte, error) {
return groundTruthArch(path, "loong64")
}
// GroundTruthARM64 assembles the given .s file with the Go toolchain in
// AArch64 cross-assembly mode (GOARCH=arm64).
func GroundTruthARM64(path string) (map[string][]byte, error) {
return groundTruthArch(path, "arm64")
}
func groundTruthArch(path, goarch string) (map[string][]byte, error) { func groundTruthArch(path, goarch string) (map[string][]byte, error) {
goroot := runtime.GOROOT() goroot := runtime.GOROOT()
asmBin := filepath.Join(goroot, "pkg", "tool", runtime.GOOS+"_"+runtime.GOARCH, "asm") asmBin := filepath.Join(goroot, "pkg", "tool", runtime.GOOS+"_"+runtime.GOARCH, "asm")
@@ -51,6 +63,8 @@ func groundTruthArch(path, goarch string) (map[string][]byte, error) {
pkg := strings.TrimSuffix(base, ".s") pkg := strings.TrimSuffix(base, ".s")
pkg = strings.TrimSuffix(pkg, "_amd64") pkg = strings.TrimSuffix(pkg, "_amd64")
pkg = strings.TrimSuffix(pkg, "_riscv64") pkg = strings.TrimSuffix(pkg, "_riscv64")
pkg = strings.TrimSuffix(pkg, "_loong64")
pkg = strings.TrimSuffix(pkg, "_arm64")
cmd := exec.Command(asmBin, "-I", includeDir, "-p", pkg, "-o", objPath, path) cmd := exec.Command(asmBin, "-I", includeDir, "-p", pkg, "-o", objPath, path)
if goarch != "" { if goarch != "" {
+110
View File
@@ -0,0 +1,110 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"fmt"
"os"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGroundTruthLOONG64 assembles the loong64 test kernels with gasm and
// compares them byte-for-byte against `go tool asm` (GOARCH=loong64). The
// relocation fields of static-symbol references are masked before the
// comparison, since the toolchain leaves them zero for the linker.
func TestGroundTruthLOONG64(t *testing.T) {
for _, path := range []string{
"../testdata/verify/basic_loong64.s",
"../testdata/verify/fp_loong64.s",
} {
t.Run(path, func(t *testing.T) {
src, err := os.ReadFile(path)
if err != nil {
t.Fatalf("read: %v", err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := asm.AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
gt, err := GroundTruthLOONG64(path)
if err != nil {
t.Fatalf("GroundTruthLOONG64: %v", err)
}
matched := 0
for _, fn := range img.Funcs {
gasmCode := maskRelocs(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
goCode, ok := gt[fn.Name]
if !ok {
t.Errorf("%s: not in ground truth (%d functions)", fn.Name, len(gt))
continue
}
goCode = maskRelocs(goCode, fn.Relocs)
if !bytes.Equal(gasmCode, goCode) {
t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\n%s", fn.Name, len(gasmCode), len(goCode), diffHex(gasmCode, goCode))
continue
}
matched++
t.Logf("%s: MATCH (%d bytes)", fn.Name, fn.Size)
}
if matched == 0 {
t.Fatal("no functions matched")
}
})
}
}
// maskRelocs zeroes the 4-byte immediate fields of the relocation sites.
func maskRelocs(code []byte, relocs []asm.Reloc) []byte {
for _, r := range relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
return code
}
func diffHex(a, b []byte) string {
var out bytes.Buffer
n := len(a)
if len(b) > n {
n = len(b)
}
for i := 0; i < n; i += 4 {
ab, bb := "??", "??"
if i+4 <= len(a) {
ab = fmt.Sprintf("%02x%02x%02x%02x", a[i], a[i+1], a[i+2], a[i+3])
} else if i < len(a) {
var sb strings.Builder
for j := i; j < len(a); j++ {
fmt.Fprintf(&sb, "%02x", a[j])
}
ab = sb.String()
}
if i+4 <= len(b) {
bb = fmt.Sprintf("%02x%02x%02x%02x", b[i], b[i+1], b[i+2], b[i+3])
} else if i < len(b) {
var sb strings.Builder
for j := i; j < len(b); j++ {
fmt.Fprintf(&sb, "%02x", b[j])
}
bb = sb.String()
}
mark := " "
if ab != bb {
mark = "!"
}
fmt.Fprintf(&out, "%04x: %s %s %s\n", i, ab, bb, mark)
}
return out.String()
}
+72
View File
@@ -0,0 +1,72 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"os"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGroundTruthRISCV assembles the riscv64 test kernels with gasm and
// compares them byte-for-byte against `go tool asm` (GOARCH=riscv64). The
// relocation fields of static-symbol references are masked before the
// comparison, since the toolchain leaves them zero for the linker.
func TestGroundTruthRISCV(t *testing.T) {
for _, path := range []string{
"../testdata/verify/basic_riscv64.s",
"../testdata/verify/rvc_riscv64.s",
"../testdata/verify/loadstore_riscv64.s",
"../testdata/verify/largeimm_riscv64.s",
"../testdata/verify/movimm_riscv64.s",
"../testdata/verify/branch_riscv64.s",
"../testdata/verify/call_riscv64.s",
} {
t.Run(path, func(t *testing.T) {
testGroundTruthRISCVFile(t, path)
})
}
}
func testGroundTruthRISCVFile(t *testing.T, path string) {
src, err := os.ReadFile(path)
if err != nil {
t.Fatalf("read: %v", err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := asm.AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
gt, err := GroundTruthRISCV(path)
if err != nil {
t.Fatalf("GroundTruthRISCV: %v", err)
}
matched := 0
for _, fn := range img.Funcs {
gasmCode := maskRelocs(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
goCode, ok := gt[fn.Name]
if !ok {
t.Errorf("%s: not in ground truth (%d functions)", fn.Name, len(gt))
continue
}
goCode = maskRelocs(goCode, fn.Relocs)
if !bytes.Equal(gasmCode, goCode) {
t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\n%s", fn.Name, len(gasmCode), len(goCode), diffHex(gasmCode, goCode))
continue
}
matched++
t.Logf("%s: MATCH (%d bytes)", fn.Name, fn.Size)
}
if matched == 0 {
t.Fatal("no functions matched")
}
}