Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6f4f2096e9 | ||
|
|
56630f8624 | ||
|
|
459f4a2b6e | ||
|
|
5d66343488 | ||
|
|
48334c4d5a | ||
|
|
7629963cab | ||
|
|
97951cbeb6 | ||
|
|
6e73f59e78 | ||
|
|
4221ec5741 |
@@ -25,7 +25,7 @@ jobs:
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.26"
|
||||
go-version: "1.27"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
|
||||
@@ -15,7 +15,7 @@ jobs:
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.26"
|
||||
go-version: "1.27"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
@@ -41,7 +41,7 @@ jobs:
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.26"
|
||||
go-version: "1.27"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
@@ -84,7 +84,7 @@ jobs:
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.26"
|
||||
go-version: "1.27"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
|
||||
+3
-4
@@ -7,9 +7,8 @@
|
||||
coverage.out
|
||||
*.test
|
||||
|
||||
# Editor detritus
|
||||
*.swp
|
||||
.DS_Store
|
||||
|
||||
# Scratch / temporary work
|
||||
_scratch/
|
||||
|
||||
# ZCode workspace
|
||||
.zcode
|
||||
|
||||
@@ -55,7 +55,7 @@ No body, no footers, no trailing period on the subject.
|
||||
|
||||
## Code Style
|
||||
|
||||
Language: Go 1.26 (`toolchain go1.26.5`).
|
||||
Language: Go 1.27 (`toolchain go1.27.0`).
|
||||
|
||||
### Formatter
|
||||
|
||||
|
||||
+35
-55
@@ -9,6 +9,37 @@ and this project adheres to [Conventional Commits](https://www.conventionalcommi
|
||||
|
||||
Unreleased changes on the `development` branch.
|
||||
|
||||
### Added
|
||||
|
||||
-
|
||||
|
||||
## [0.31.0] — 2026-08-20
|
||||
|
||||
The arm64 encoder (Phase 5 — complete) ships with ELF64 and GOOBJ emission,
|
||||
verified byte-for-byte against `GOARCH=arm64 go tool asm` and linked into a
|
||||
real `go build`. The encoder covers the full integer instruction set, FP
|
||||
arithmetic, conditional select, CRC32, and the MOV pseudo-instruction with
|
||||
bitmask immediate encoding. The project now requires Go 1.27.
|
||||
|
||||
### Added
|
||||
|
||||
- **arm64 encoder (Phase 5 — complete).** `gasm asm` can now assemble `_arm64.s`
|
||||
files: the AArch64 integer instruction set with the MOV pseudo-instruction and
|
||||
its immediate-constant expansions (MOVZ/MOVN/MOVK for wide immediates, ORR with
|
||||
logical bitmask encoding for values like `$1`), data-processing (shifted
|
||||
register and immediate forms), load/store (scaled unsigned and unscaled9-bit
|
||||
immediate), conditional and unconditional branches, FP/SP frame mapping,
|
||||
SB/global symbol references (ADRP+ADD pairs with `R_ADDRARM64` relocations),
|
||||
jump chain folding, and ELF64 emission (`gasm asm --format elf`). Ground-truth
|
||||
verification against `GOARCH=arm64 go tool asm` matches byte-for-byte. Phase 5
|
||||
(the other architectures — RISC-V, LoongArch, arm64) is now complete.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Go 1.27 required.** The project now requires Go 1.27 (`toolchain go1.27.0`).
|
||||
The `R_DWTXTADDR_U4` relocation type is detected at runtime for backward
|
||||
compatibility.
|
||||
|
||||
## [0.30.0] — 2026-08-13
|
||||
|
||||
The LoongArch encoder (Phase 5) ships with ELF64 and GOOBJ emission, verified
|
||||
@@ -166,20 +197,13 @@ new CLI commands. A signature-parser fix corrects grouped Go parameters.
|
||||
prevent GC from collecting heap objects whose addresses were passed to JIT
|
||||
code via `unsafe.Pointer`; all verify tests pass 100/100 under `-race`.
|
||||
|
||||
### Cleaned up
|
||||
### Changed
|
||||
|
||||
- **Removed external kernel test dependencies** — the verify test suite no
|
||||
longer references production kernels from the separate go-libraries project.
|
||||
The remaining test suite uses only `testdata/verify/*.s` kernels, which are
|
||||
part of this repository. Coverage is identical locally and in CI (80.3 %).
|
||||
|
||||
### Verified
|
||||
|
||||
- `gasm diff` detects byte-level differences; `--map` pairs differently-named
|
||||
functions for comparison.
|
||||
- `gasm verify --call` invokes functions with user-supplied buffers; the arg
|
||||
block is printed before and after the call, showing return values.
|
||||
- LSP go-to-definition resolves labels across functions and files.
|
||||
|
||||
## [0.28.0] — 2026-08-03
|
||||
|
||||
@@ -211,10 +235,6 @@ ground-truth verification against `GOARCH=riscv64 go tool asm`.
|
||||
- RVC: C.LDSP/C.SDSP/FLDSP/FSDSP immediate encoding now matches Go toolchain
|
||||
(bit-interleaved format).
|
||||
|
||||
### Verified
|
||||
|
||||
- 118 RISC-V tests, asm coverage 83.3%.
|
||||
- Ground-truth: C.LDSP, C.SDSP, C.FLDSP, C.FSDSP byte-exact vs Go toolchain.
|
||||
|
||||
## [0.27.0] — 2026-08-01
|
||||
|
||||
@@ -247,7 +267,7 @@ area bit-for-bit.
|
||||
analyze, autocorr) pass; partial functions (decoders that fault on malformed
|
||||
input) should use `--ground-truth` instead.
|
||||
|
||||
### Known limitation
|
||||
### Fixed
|
||||
|
||||
`--fuzz` crashes the process for partial functions (e.g. LZ4 decoders) whose
|
||||
over-copy paths read past the buffer on random garbage input. Subprocess
|
||||
@@ -310,11 +330,6 @@ The remaining go-flac encoder kernels join the differential suite.
|
||||
frames: the four zigzag-fold entropy sums compared against the scalar
|
||||
loop).
|
||||
|
||||
### Verified
|
||||
|
||||
- `gasm fmt` doc-comment indentation confirmed correct: comments before
|
||||
every TEXT are at column 0 (the RET-detection logic handles multi-exit
|
||||
functions).
|
||||
|
||||
## [0.21.0] — 2026-07-26
|
||||
|
||||
@@ -352,7 +367,7 @@ test corpus exercises.
|
||||
argument blocks and collect distinct output fingerprints (the result
|
||||
words); reports path diversity as a lower bound on code coverage.
|
||||
|
||||
### Note
|
||||
### Changed
|
||||
|
||||
INT3-based per-block hit counting was prototyped but deferred: Go's runtime
|
||||
signal management (sigaltstack, handler re-installation) makes raw
|
||||
@@ -417,11 +432,6 @@ toolchain.
|
||||
known-answer LZ4 blocks decode bit-for-bit, wide copies of 0–1024 bytes
|
||||
match, malformed input returns the correct error codes.
|
||||
|
||||
### Verified
|
||||
|
||||
- `just test` (race, 84.6 % total coverage, verify 82.2 %).
|
||||
- `gasm verify` on both go-lz4 kernels: all functions JIT-load and
|
||||
smoke-test clean.
|
||||
|
||||
## [0.16.0] — 2026-07-21
|
||||
|
||||
@@ -547,13 +557,6 @@ Go assembler.
|
||||
(`DATA mask<>+8(SB)/8, $0x800f…`) parse as unsigned and keep their bit
|
||||
pattern, instead of being rejected as non-integer.
|
||||
|
||||
### Verified
|
||||
|
||||
- End-to-end: a gasm-emitted GOOBJ swapped into a `go build` in place of
|
||||
the toolchain's assembly object links and runs with output identical to
|
||||
the baseline binary (stack-argument calls and a `GLOBL` relocation
|
||||
resolved by the Go linker). All 17 go-flac AVX2 kernel functions emit as
|
||||
a GOOBJ that `go tool nm` reads back with every symbol intact.
|
||||
|
||||
## [0.11.0] — 2026-07-16
|
||||
|
||||
@@ -612,7 +615,7 @@ verified byte for byte against the Go assembler.
|
||||
arithmetic, the unpacks, VMOVDDUP and the conversions all accept the
|
||||
explicit K1–K7 operand and the `.Z` suffix the way Go writes them.
|
||||
|
||||
### Documented
|
||||
### Changed
|
||||
|
||||
- VCVTPS2PD follows the Go assembler's encoding, which omits the F3
|
||||
mandatory prefix (VEX.pp / EVEX.pp = 00) that Intel's maps prescribe; the
|
||||
@@ -620,15 +623,6 @@ verified byte for byte against the Go assembler.
|
||||
gasm reproduces it exactly (and round-trips through the x86 decoder, which
|
||||
shares the convention).
|
||||
|
||||
### Verified
|
||||
|
||||
- 58 new ground-truth cases — every instruction extracted from the Go
|
||||
toolchain's own assembly (go build + an executable-segment dump), checked
|
||||
byte for byte and round-tripped through the decoder, covering disp8×N for
|
||||
the scalar (×8/×4), duplication (×8/×32/×64) and conversion (×8/×16/×32)
|
||||
memory operands, the 5-bit register fields and the masked/zeroing P2
|
||||
byte. All four go-flac/go-lz4 kernels still assemble byte-identically
|
||||
and lint clean.
|
||||
|
||||
## [0.9.0] — 2026-07-14
|
||||
|
||||
@@ -764,13 +758,6 @@ the Go toolchain, completing the production-kernel coverage.
|
||||
- `asm`: the VEX encoder now rejects vector register indices 16–31 instead of
|
||||
encoding a truncated (wrong) register.
|
||||
|
||||
### Verified
|
||||
|
||||
- All 10 functions of the go-flac `avx512_amd64.s` kernel assemble
|
||||
byte-identically to the Go toolchain's machine code (the disp32 of the one
|
||||
`VMOVDQU32 idx16(SB), Z13` load is linker-filled in Go and resolved within
|
||||
gasm's own image — checked to reach the right constant bytes). The AVX2
|
||||
kernel's 17 functions remain byte-identical.
|
||||
|
||||
## [0.4.0] — 2026-07-09
|
||||
|
||||
@@ -791,13 +778,6 @@ machine code byte for byte.
|
||||
- `gasm asm` prints the data section and symbol map alongside the functions
|
||||
and writes the whole image (code + data) with `-o`.
|
||||
|
||||
### Verified
|
||||
|
||||
- All 17 functions of the go-flac `avx2_amd64.s` kernel assemble
|
||||
byte-identically to the Go toolchain's machine code; the only differing
|
||||
bytes are the displacements of the two `VMOVDQU mask24<>(SB), X15` loads,
|
||||
which the Go linker fills at link time and gasm resolves within its own
|
||||
image (checked to reach the right constant bytes).
|
||||
|
||||
## [0.3.0] — 2026-07-08
|
||||
|
||||
|
||||
+1
-1
@@ -2,7 +2,7 @@
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Go 1.26 or later (`toolchain go1.26.5`)
|
||||
- Go 1.27 or later (`toolchain go1.27.0`)
|
||||
- `just` command runner
|
||||
- A Linux host on amd64, arm64, riscv64 or loong64
|
||||
|
||||
|
||||
@@ -2,8 +2,6 @@
|
||||
|
||||
Developer tooling for **GAsm** — Go's built-in Plan 9 assembler.
|
||||
|
||||
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
|
||||
|
||||
Go ships an assembler but no tooling for it. There is no syntax highlighting,
|
||||
no autocomplete, no linter, no static analyser, no formatter, no standalone
|
||||
assembler and no debugger for `.s` files. Developers write assembly blind,
|
||||
@@ -18,25 +16,13 @@ gasm parse parse and report syntax errors
|
||||
gasm fmt canonicalise formatting (gofmt for assembly)
|
||||
gasm lint static checks
|
||||
gasm lsp language server (completion, hover, symbols, diagnostics, highlighting)
|
||||
gasm asm standalone assembler (Phase 2)
|
||||
gasm verify dynamic analysis & verification (Phase 3)
|
||||
gasm debug source-level debugger (Phase 4)
|
||||
gasm asm standalone assembler
|
||||
gasm verify dynamic analysis & verification
|
||||
gasm debug source-level debugger
|
||||
gasm diff compare machine code of two .s files
|
||||
gasm profile show basic-block structure of functions
|
||||
```
|
||||
|
||||
> **Status: Phase 4 — done, Phase 5 underway.** Phase 1 (the language
|
||||
> foundation, linter, formatter and language server) shipped in v0.1.0;
|
||||
> Phase 2 (the standalone assembler — the full amd64 instruction set plus
|
||||
> ELF and GOOBJ object emission) in v0.12.0; Phase 3 (dynamic
|
||||
> analysis — JIT execution, differential testing, ABI checks and coverage
|
||||
> profiling) in v0.25.0; Phase 4 (interactive debugger — ptrace-based,
|
||||
> breakpoints, watchpoints, stepping, vector register display, named buffer
|
||||
> allocation) in v0.27.0; RISC-V encoder (RV64IMAFDC + RVC, ELF emission,
|
||||
> ground-truth, GOOBJ) in v0.28.0–v0.29.0; LoongArch encoder (the full
|
||||
> instruction set with the MOV expansions, ELF and GOOBJ emission, and
|
||||
> ground-truth verification) after v0.29.0. See [Roadmap](#roadmap).
|
||||
|
||||
## Architecture support
|
||||
|
||||
gasm-devkit targets every architecture Go's assembler speaks. The instruction
|
||||
@@ -66,219 +52,6 @@ cross-compiles the same four targets.
|
||||
|
||||
**FreeBSD support is planned for a future release.**
|
||||
|
||||
## Roadmap
|
||||
|
||||
The work is delivered in four phases. Each phase is completed and hardened
|
||||
before the next begins. The ordering follows a dependency chain: understand
|
||||
the code statically (Phase 1), make it runnable (Phase 2), then run it and
|
||||
observe or control it (Phases 3–4).
|
||||
|
||||
### Phase 1 — language foundation, editor tooling and static analysis · *done*
|
||||
|
||||
Everything needed to read, understand, check, format and highlight GAsm —
|
||||
without executing it.
|
||||
|
||||
| Capability | Status |
|
||||
|------------|--------|
|
||||
| Lexer — permissive, position-aware scanner for all four architectures | done |
|
||||
| Parser — line-oriented, error-tolerant, full AST with source positions | done |
|
||||
| Instruction + register tables for amd64, arm64, riscv64, loong64 (generated, complete) | done |
|
||||
| Linter — `unknown-instruction`, `operand-count`, `undefined-label`, `duplicate-label`, `missing-ret`, `missing-textflag-include`, `abi-argsize`, `unreachable-code`, `register-clobber`, `funcdata-pcdata` | done |
|
||||
| Formatter — idempotent, comment-preserving, per-function alignment; a `RET` terminates the body for indentation, so the next function's doc comment stays at column 0; exactly one blank line before every block (label, `TEXT`, `GLOBL`) and runs of blanks collapsed; directory / no-argument mode reformats every `.s` in place, `go fmt`-style | done |
|
||||
| Language server — completion, hover, document symbols, diagnostics, semantic-token highlighting | done |
|
||||
| CLI — `gasm tokens / parse / fmt / lint / lsp` | done |
|
||||
| Real-world validation against production AVX2 / AVX-512 kernels | done |
|
||||
| Lint hardening — zero false positives across the Go runtime corpus (90 files, all four architectures): macro-invocation handling, branch aliases (`B`/`BL`/`JAL`), addressing suffixes (`.P`/`.W`), terminal `UNDEF` | done |
|
||||
| Static analysis — `abi-argsize` (argument/result area computed from the `// func` signature under Go's ABI0 layout and checked against the TEXT declaration) and `unreachable-code` (dead code after `RET`, suppressed where reachability is undecidable: PC-relative jumps, register-indirect branches, `#ifdef`) | done |
|
||||
| Static analysis — register liveness (CFG construction + per-instruction def/use + iterative backward dataflow) driving `register-clobber`, calibrated to the **Go ABI** (not System V): flags writes to the registers Go fixes across calls — the frame pointer and the goroutine pointer (`R14` on amd64, `R28`/`R29` on arm64, `X27` on riscv64, `R22` on loong64, plus the OS-reserved `R18` on arm64) — that are never saved/restored; the goroutine pointer is reported only when the function can reach the runtime (not `NOSPLIT`, or makes calls), matching how the runtime's own assembly uses it. `funcdata-pcdata` structural validation of `FUNCDATA`/`PCDATA` operands and indices | done |
|
||||
|
||||
> **Limitation — macros.** gasm-devkit reads `.s` source as written; it does
|
||||
> **not** run the C preprocessor, so `#define` macros are not expanded. Files
|
||||
> that use macros (the runtime's `asm_*.s`, `race_*.s`, `sys_*.s`, …) parse
|
||||
> cleanly, and macro *invocations* are recognised and never flagged, but the
|
||||
> `undefined-label` and `missing-ret` heuristics are suppressed in macro-using
|
||||
> files because labels a macro defines are invisible without expansion. Full
|
||||
> macro expansion is future work (it pairs naturally with the Phase 2
|
||||
> assembler). Hand-written, macro-free kernels — such as everything in
|
||||
> `go-libraries` — are analysed in full.
|
||||
|
||||
### Phase 2 — standalone assembler · *done*
|
||||
|
||||
Assembly without the Go toolchain in the loop.
|
||||
|
||||
- **`gasm asm`:** a standalone assembler that turns a `.s` file into machine
|
||||
code directly — pure Go, no `go build`, no external toolchain. Useful for
|
||||
fast iteration, for environments without a full Go installation, and as the
|
||||
execution substrate that Phases 3 and 4 build on.
|
||||
|
||||
Done so far:
|
||||
|
||||
- An amd64 (x86-64) **instruction encoder** — REX/ModR-M/SIB/displacement/
|
||||
immediate machinery and the scalar instruction set (MOV, the ALU group, TEST,
|
||||
LEA, INC/DEC/NEG/NOT, shifts, IMUL and IMUL3, PUSH/POP, JMP/CALL/Jcc,
|
||||
CMOVcc, SETcc, LZCNT/TZCNT, the sign/zero-extending moves — MOVBLZX and
|
||||
friends, MOVLQSX — and CVTSL2SD/CVTSQ2SD), validated by round-tripping
|
||||
every encoding through `golang.org/x/arch`'s decoder and byte-for-byte
|
||||
against the Go assembler.
|
||||
- An **assembler** that drives the parser's AST into the encoder with local-
|
||||
label resolution — jumps start in the short (rel8) form and expand to rel32
|
||||
when the displacement does not fit, and jump-to-jump chains are folded the
|
||||
way the Go toolchain folds them — so `gasm asm <file>` emits machine code
|
||||
for each `TEXT` function.
|
||||
- **File-level assembly with static data** — `GLOBL`/`DATA` symbols are laid
|
||||
out in a data section behind the code and references to them (`mask<>(SB)`)
|
||||
are encoded RIP-relative with the displacement resolved within the image,
|
||||
so the output is self-consistent and position-independent. References to
|
||||
symbols no `GLOBL` in the file defines are recorded as relocations and
|
||||
carried into the object-file output.
|
||||
- **GOOBJ emission** — `gasm asm --format goobj -p <pkgpath>` writes the Go
|
||||
toolchain's own object format (the one `cmd/link` consumes directly), so
|
||||
gasm-assembled kernels drop into a `go build` without the Go assembler:
|
||||
the functions as non-package symbols, `GLOBL` data, one `FuncInfo` per
|
||||
function and the pc-value tables (`pcsp` with the real prologue/epilogue
|
||||
stack deltas, `pcfile`, `pcline`, `pcinline`). Verified end-to-end by
|
||||
swapping a gasm-emitted object into a `go build` in place of the
|
||||
toolchain's, linking and running — bit-identical behaviour.
|
||||
- **Object-file emission** — `gasm asm --format elf` writes a relocatable
|
||||
object (a `.text` and a `.data` section, a symbol table — file-local `<>`
|
||||
symbols local, the rest global — and one `R_X86_64_PC32` relocation per
|
||||
static-symbol reference) that links with the system toolchain: external
|
||||
references resolve against undefined symbols, file-local ones against the
|
||||
data section. Verified end-to-end by linking a gasm-emitted object with
|
||||
a C driver and running it. RISC-V uses the equivalent `R_RISCV_PCREL_HI20`
|
||||
/ `R_RISCV_PCREL_LO12_I` pair for AUIPC+JAL/JALR sequences.
|
||||
- **`FP`/`SP` frame mapping** — the pseudo-registers are translated onto the
|
||||
hardware stack pointer (`x+N(FP)` → `(N+8)(SP)` for a zero frame, `(N+frame+
|
||||
16)(SP)` with a frame pointer; locals via `x-N(SP)`), and the Go-style
|
||||
prologue/epilogue is generated for functions with a frame. The output is
|
||||
**byte-identical to the Go assembler** for these cases (verified against
|
||||
`go tool objdump`).
|
||||
- **SIMD (VEX / AVX2)** — the VEX prefix machinery (2-byte C5 and 3-byte C4)
|
||||
with XMM/YMM vector registers, validated by round-trip decoding **and**
|
||||
byte-for-byte against the Go assembler's machine code, across eight operand
|
||||
forms: the three-operand NDS form (VPADDD/Q, VPSUBD/Q, VPXOR, VPOR, VPAND/N,
|
||||
VPCMPEQD, VPCMPGTQ, VPUNPCK*, VPMULLD, VPMULDQ, VPSHUFB, VPACKSSDW,
|
||||
VPERMD), the two-operand reg/rm form (VPMOVSXWD/DQ, VPMOVZXDQ,
|
||||
VPBROADCASTD/Q, VPMOVMSKB, VMOVMSKPS, VCVTDQ2PD), the immediate-shift and
|
||||
variable-count shifts (VPSLLD/Q, VPSRAD, VPSRLD/Q with an immediate or an
|
||||
XMM/memory count), the immediate shuffle (VPSHUFD, VPERMQ), the
|
||||
three-operand-plus-immediate form (VSHUFPD, VPERM2I128, VINSERTI128), the
|
||||
lane extract (VEXTRACTI128, VEXTRACTF128), the direction-sensitive moves
|
||||
(VMOVDQU, VMOVUPD, VMOVD, VMOVQ, VMOVSD), the no-operand VZEROUPPER, and
|
||||
the floating-point set: the packed double arithmetic
|
||||
(VADDPD/VSUBPD/VMULPD/VDIVPD/VMINPD/VMAXPD), the unpacks
|
||||
(VUNPCKHPD/VUNPCKLPD), the scalar SD and SS operations, VMOVDDUP, the
|
||||
width-changing conversions (VCVTDQ2PS, VCVTPS2PD, VCVTDQ2PD and the
|
||||
VCVTPD2DQX/Y / VCVTTPD2DQX/Y spellings, whose VEX.L follows the wider
|
||||
source) and VFMADD231PD.
|
||||
- **SIMD (EVEX / AVX-512)** — the four-byte EVEX prefix with the 5-bit
|
||||
register fields (Z0–Z31, X/Y 16–31), opmask registers (K0–K7 as operands
|
||||
and mask destinations, KMOVW, KTESTW) and the compressed disp8×N
|
||||
displacement, covering every AVX-512 instruction the go-flac kernels use:
|
||||
VPXORD/Q, VPADDD, VPSUBD/Q, VPUNPCK*DQ, VPMULLD/Q, VPERMD, VPSLLD/VPSRAD/
|
||||
VPSRAQ, VALIGND, VPCMPEQD (with a K destination), VMOVDQU32, VMOVUPD,
|
||||
VCVTQQ2PD, VPMOVSXDQ, the narrowing stores VPMOVDW/VPMOVQD, the lane
|
||||
extracts VEXTRACTI64X4/VEXTRACTF64X4, VFMADD231PD, VADDPD, VMULPD,
|
||||
VMOVDQU64 and the broadcasts VPBROADCASTD/Q from a GPR or memory, plus the
|
||||
wider AVX-512 F/BW integer set (VPADDB/W, VPSUBB/W, VPANDD/Q/ND/NQ, VPMULLW,
|
||||
VPMIN*/VPMAX* for B/W/D/Q elements, signed and unsigned, VPAVGB/W, the variable
|
||||
shifts VPSLLV*/VPSRLV*/VPSRAV*, VMOVDQU8/16), the common floating-point
|
||||
and conversion set (the packed double and single arithmetic
|
||||
VADD/VSUB/VMUL/VDIV/VMIN/VMAX PD and PS, the scalar SD/SS operations —
|
||||
whose EVEX forms exist for masked and zeroing use — the VUNPCK{L,H}PD
|
||||
unpacks, VMOVDDUP, VMOVSLDUP/VMOVSHDUP and the VCVT* conversions), and
|
||||
the wider AVX-512 set: ternary logic (VPTERNLOGD/Q), lane shuffles,
|
||||
inserts and extracts (VSHUF{F,I}{32,64}X{2,4}, the VINSERT*/VEXTRACT*
|
||||
{F,I}{32,64}X{2,4,8} family, VPALIGNR), compares with an opmask
|
||||
destination (VCMPPD/PS/SD/SS), the permutes (VPERMB/W, VPERMI2/T2
|
||||
D/Q/PD), the wider integer families (VPMADDWD/UBSW, VPMULHUW, VPACK*,
|
||||
VPABS*, the VPROL*/VPROR* rotates and the word shifts), expand/compress
|
||||
(VEXPAND*/VCOMPRESS*, VPEXPAND*/VPCOMPRESS*), the broadcasts
|
||||
(VPBROADCASTB/W, VBROADCASTSS/SD), the opmask instructions (KAND/KOR/
|
||||
KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST, KMOVQ), the aligned moves
|
||||
(VMOVAPS/APD, VMOVDQA32/64, VMOVSS) and the remaining extending and
|
||||
narrowing moves, the floating-point helper and conversion tail
|
||||
(VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*, VSCALEF*, VRNDSCALE*,
|
||||
VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with a K destination, and the
|
||||
VCVT* conversions VCVTQQ2PS, VCVTPD2QQ/UQQ, VCVTPS2QQ, VCVTUDQ2PD/PS,
|
||||
VCVTPH2PS, VCVTPS2PH), and gather/scatter with VSIB addressing
|
||||
(VGATHER*/VPGATHER* in both the VEX mask-register spelling and the EVEX
|
||||
K-mask spelling — where the L'L field follows the VSIB index — plus
|
||||
VSCATTER*/VPSCATTER*). The EVEX mnemonic suffixes the Go assembler
|
||||
accepts are honoured: rounding modes (.RN_SAE, .RD_SAE, .RU_SAE,
|
||||
.RZ_SAE), suppress-all-exceptions (.SAE) and memory broadcast (.BCST,
|
||||
with the element-sized disp8×N), each combinable with the .Z zeroing
|
||||
suffix. Masking is supported the way
|
||||
Go writes it — an explicit K1–K7 operand placed among the operands, and a
|
||||
`.Z` mnemonic suffix for zeroing.
|
||||
- **Legacy SSE moves** — `MOVOU`/`MOVO` (the Plan 9 names for MOVDQU/MOVDQA),
|
||||
`MOVUPS`/`MOVAPS`/`MOVUPD`/`MOVAPD` and the scalar `MOVSD`/`MOVSS`.
|
||||
- **Both go-flac kernels — all 17 AVX2 and all 10 AVX-512 functions —
|
||||
assemble byte-identically to the Go toolchain's machine code**; the only
|
||||
differing bytes are the displacements of the static-constant loads, which
|
||||
the Go linker fills at link time and gasm resolves within its own image
|
||||
(verified to reach the right constant bytes).
|
||||
|
||||
Remaining for Phase 2:
|
||||
|
||||
- External (cross-package) symbol references in the GOOBJ output —
|
||||
**deferred** with a recorded decision and three options; see
|
||||
[`docs/DEFERRED.md`](docs/DEFERRED.md). Single-package objects (no
|
||||
cross-package references) work today, which covers the production
|
||||
kernels. With that item deferred, the amd64 instruction set — scalar,
|
||||
VEX/AVX2 and the full EVEX/AVX-512 set including GPR-interchanging
|
||||
conversions — is complete, and RISC-V encoding (RV64IMAFDC + RVC)
|
||||
including ELF and GOOBJ emission is complete.
|
||||
|
||||
### Phase 3 — dynamic analysis · *done*
|
||||
|
||||
Run the code and check what static analysis cannot. The oracle is the
|
||||
portable Go implementation every kernel is derived from.
|
||||
|
||||
- **`gasm verify`:**
|
||||
- **JIT execution substrate** — *done.* Assemble the kernel, map it into
|
||||
executable memory (`syscall.Mmap`, W^X) and call it through an ABI0
|
||||
trampoline; pure Go, no cgo, no external toolchain.
|
||||
- **Differential testing** — *done.* The JIT-assembled kernel is fuzzed
|
||||
against a portable Go reference, comparing the result bit-for-bit;
|
||||
the automated form of the project's bit-identical contract.
|
||||
- **Runtime ABI checks** — *done.* The ABI-checking trampoline sets
|
||||
sentinels in BP and R14, verifies they survive the call, and fills a
|
||||
128-byte red-zone canary below SP.
|
||||
- **Coverage / basic-block profiling** — *done.* Static block enumeration
|
||||
from the assembler's label map plus multi-input path-diversity
|
||||
measurement: how many observationally distinct execution paths a
|
||||
test corpus exercises.
|
||||
|
||||
### Phase 4 — debugger · *done*
|
||||
|
||||
- **`gasm debug`:** single-step a GAsm function, inspect registers (including
|
||||
YMM vector registers), set breakpoints and watchpoints on addresses, write
|
||||
memory, allocate and fill named buffers, disassemble at PC, and trace the
|
||||
source-line mapping — the interactive counterpart to Phase 3's execution
|
||||
substrate.
|
||||
- ptrace-based debuggee subprocess (PTRACE_TRACEME + LockOSThread), entry
|
||||
breakpoint (auto-run to function start), single-step, register inspection
|
||||
(GPR + YMM/XMM via PTRACE_GETFPREGS), label resolution, breakpoint
|
||||
management via `/proc/pid/mem`, named buffer allocation with pattern
|
||||
filling (`--buf`), interactive REPL with conditional breakpoints, four
|
||||
hardware watchpoints (DR0–DR3), step-over-CALL, run-to-return, backtrace,
|
||||
memory read/write, disassembly at PC (x86asm), and source-line ↔ offset
|
||||
mapping.
|
||||
|
||||
### Phase 5 — the other architectures · *in progress*
|
||||
|
||||
- **RISC-V encoding — done.** RV64IMAFDC instruction set, RVC compression,
|
||||
MOV pseudo-instruction, SB/global symbols (AUIPC pairs), ELF64 and GOOBJ
|
||||
emission, and ground-truth verification against `go tool asm`.
|
||||
- **LoongArch encoding — done.** The LoongArch64 instruction set with the
|
||||
MOV pseudo-instruction and its immediate-constant expansions, FP/SP frame
|
||||
handling, SB/global symbol references (pcalau12i pairs), ELF64 and GOOBJ
|
||||
emission, and ground-truth verification against `go tool asm` — the emitted
|
||||
GOOBJ links into a real `go build` for `GOARCH=loong64`.
|
||||
- **Remaining:** arm64 encoding, plus the same encode-and-verify treatment
|
||||
(instruction tables already generated from the toolchain).
|
||||
|
||||
## Principles
|
||||
|
||||
- **Pure Go and GAsm only.** No C, no cgo, no external toolchains, no native
|
||||
@@ -311,14 +84,14 @@ portable Go implementation every kernel is derived from.
|
||||
| `lint` | Conservative static checks. |
|
||||
| `format` | A canonical formatter — `gofmt` for assembly. |
|
||||
| `asm` | The standalone assembler: amd64, RISC-V and LoongArch encoders, linker, object-file emitters (ELF, GOOBJ). |
|
||||
| `verify` | JIT execution substrate for dynamic analysis, combined ABI+fuzz differential testing (Phase 3). |
|
||||
| `debug` | Interactive ptrace debugger with GPR/YMM register display and named buffer allocation (Phase 4). |
|
||||
| `verify` | JIT execution substrate for dynamic analysis, combined ABI+fuzz differential testing. |
|
||||
| `debug` | Interactive ptrace debugger with GPR/YMM register display and named buffer allocation. |
|
||||
| `lsp` | Language Server Protocol server. |
|
||||
| `cmd/gasm` | The `gasm` binary tying it all together. |
|
||||
| `_gen` | The generator that rebuilds the instruction tables from the Go toolchain. |
|
||||
|
||||
See [`docs/ARCHITECTURE.md`](docs/ARCHITECTURE.md) for the design rationale and
|
||||
data flow, and [`docs/DEFERRED.md`](docs/DEFERRED.md) for design decisions
|
||||
data flow, and [`docs/DECISIONS.md`](docs/DECISIONS.md) for design decisions
|
||||
deliberately postponed (with the analysis needed to pick them up again).
|
||||
|
||||
## Quick start
|
||||
@@ -353,8 +126,8 @@ gasm profile k.s # show basic-block structure
|
||||
```
|
||||
|
||||
See [CONTRIBUTING.md](CONTRIBUTING.md) for the full development workflow,
|
||||
[docs/cli.md](docs/cli.md) for the command reference, and
|
||||
[docs/development.md](docs/development.md) for setup and recipes.
|
||||
[docs/CLI.md](docs/CLI.md) for the command reference, and
|
||||
[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for setup and recipes.
|
||||
|
||||
## Editor integration
|
||||
|
||||
@@ -365,6 +138,7 @@ binary and associate it with `.s` files. Syntax highlighting is delivered as
|
||||
infers the target architecture from the file-name suffix
|
||||
(`_amd64.s` / `_arm64.s` / `_riscv64.s` / `_loong64.s`).
|
||||
|
||||
## Licence
|
||||
## License
|
||||
|
||||
BSD-3-Clause — the same licence as Go itself. See [`LICENSE`](LICENSE).
|
||||
BSD-3-Clause — see [LICENSE](LICENSE).
|
||||
Copyright © 2026 [Petr Balvín](https://petrbalvin.org)
|
||||
|
||||
@@ -0,0 +1,181 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestGOObjectAARCH64Structure checks the basic structure of the emitted
|
||||
// AArch64 GOOBJ: the preamble, the magic, the block offsets and the
|
||||
// non-package symbol definitions.
|
||||
func TestGOObjectAARCH64Structure(t *testing.T) {
|
||||
f, errs := parser.Parse("k_arm64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVD a+0(FP), R4
|
||||
MOVD b+8(FP), R5
|
||||
ADD R5, R4, R4
|
||||
MOVD R4, ret+16(FP)
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
obj, err := img.GOObjectAARCH64("testpkg", "k_arm64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObjectAARCH64: %v", err)
|
||||
}
|
||||
|
||||
// Check preamble.
|
||||
idx := strings.Index(string(obj), "\n!\n")
|
||||
if idx < 0 {
|
||||
t.Fatal("missing preamble separator")
|
||||
}
|
||||
preamble := string(obj[:idx])
|
||||
if !strings.HasPrefix(preamble, "go object") {
|
||||
t.Errorf("preamble = %q, want 'go object ...'", preamble)
|
||||
}
|
||||
|
||||
// Check GOOBJ magic.
|
||||
magicIdx := idx + 3
|
||||
if magicIdx+8 > len(obj) || string(obj[magicIdx:magicIdx+8]) != "\x00go120ld" {
|
||||
t.Error("missing GOOBJ magic")
|
||||
}
|
||||
|
||||
// The object should contain the function's code.
|
||||
if len(img.Code) == 0 {
|
||||
t.Error("no code generated")
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectAARCH64Link does an end-to-end link test: it cross-compiles a
|
||||
// Go program for arm64, substitutes the gasm-produced object into the package
|
||||
// archive, re-links with cmd/link, and verifies the symbol appears in the
|
||||
// resulting binary. The binary is not executed (no arm64 host or qemu).
|
||||
// Skipped when no Go toolchain is available.
|
||||
func TestGOObjectAARCH64Link(t *testing.T) {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
t.Skip("no Go toolchain available")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
asmSrc := `#include "textflag.h"
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVD a+0(FP), R4
|
||||
MOVD b+8(FP), R5
|
||||
ADD R5, R4, R4
|
||||
MOVD R4, ret+16(FP)
|
||||
RET
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
mainSrc := `package main
|
||||
|
||||
func add(a, b int64) int64
|
||||
|
||||
func main() {
|
||||
if add(20, 22) != 42 {
|
||||
panic("bad add")
|
||||
}
|
||||
}
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module a64link\n\ngo 1.21\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Capture the cross build (GOARCH=arm64): the package archive and the
|
||||
// link line.
|
||||
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||
build.Dir = dir
|
||||
build.Env = append(os.Environ(), "GOARCH=arm64")
|
||||
buildLog, err := build.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||
}
|
||||
var pkgArch, work, linkLine, asmObj string
|
||||
for _, line := range strings.Split(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
|
||||
asmObj = fieldAfter(line, "-o")
|
||||
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
|
||||
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
|
||||
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
|
||||
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||
linkLine = line
|
||||
}
|
||||
}
|
||||
if work == "" || asmObj == "" {
|
||||
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
|
||||
}
|
||||
defer os.RemoveAll(work)
|
||||
|
||||
// Expand $WORK in the object path.
|
||||
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
|
||||
|
||||
// Read the toolchain-produced object and assemble the same source with gasm.
|
||||
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f, errs := parser.Parse("main_arm64.s", string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
gasmObj, err := img.GOObjectAARCH64("a64link", "main_arm64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObjectAARCH64: %v", err)
|
||||
}
|
||||
|
||||
// Replace the toolchain-produced object with gasm's.
|
||||
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
|
||||
t.Fatalf("write gasm object: %v", err)
|
||||
}
|
||||
|
||||
// Re-link.
|
||||
if linkLine == "" {
|
||||
t.Skip("could not find link command in build log")
|
||||
}
|
||||
// Expand $WORK in the link command.
|
||||
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
|
||||
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
|
||||
if out, err := linkCmd.CombinedOutput(); err != nil {
|
||||
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
|
||||
}
|
||||
|
||||
// Verify the binary exists and contains the symbol.
|
||||
binPath := filepath.Join(dir, "prog")
|
||||
if _, err := os.Stat(binPath); err != nil {
|
||||
t.Fatalf("binary not found: %v", err)
|
||||
}
|
||||
binData, err := os.ReadFile(binPath)
|
||||
if err != nil {
|
||||
t.Fatalf("read binary: %v", err)
|
||||
}
|
||||
if !strings.Contains(string(binData), "add") && !strings.Contains(string(binData), "a64link") {
|
||||
t.Error("binary does not contain expected symbol")
|
||||
}
|
||||
}
|
||||
+1313
-8
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,764 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
// arm64 (AArch64) instruction encoding.
|
||||
//
|
||||
// The encoder is data-driven: each mnemonic maps to an instruction format and
|
||||
// an opcode constant, and the format selects the bit layout. The opcode
|
||||
// constants and formats are transcribed from the Go toolchain's own arm64
|
||||
// backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm`
|
||||
// exactly — the ground-truth oracle for the verify suite.
|
||||
//
|
||||
// All AArch64 instructions are 32 bits, little-endian. The formats used here
|
||||
// (per the ARM Architecture Reference Manual):
|
||||
//
|
||||
// DP-shifted-reg sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
|
||||
// DP-immediate sf<<31 | op<<30 | S<<29 | 0x11<<24 | imm12<<10 | Rn<<5 | Rd
|
||||
// Logical-imm sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd
|
||||
// Move-wide sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd
|
||||
// Load/store size<<30 | 0x7<<27 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt
|
||||
// LDST-unscaled size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<5 | Rt (actually imm9<<12 | Rn<<5 | Rt)
|
||||
// LDST-pair opc<<30 | 0x5<<27 | V<<26 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt
|
||||
// Branch-imm 0<<31 | 0x5<<26 | imm26 (B)
|
||||
// Branch-imm 1<<31 | 0x5<<26 | imm26 (BL)
|
||||
// Branch-cond 0x2A<<25 | imm19<<5 | cond (B.cond)
|
||||
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
|
||||
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
|
||||
|
||||
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
|
||||
// R0–R30 (integer), F0–F31 (floating point), and the ABI aliases the
|
||||
// runtime's assembly uses. Returns -1 for an unrecognised name.
|
||||
func arm64RegNum(name string) int {
|
||||
switch name {
|
||||
case "R0":
|
||||
return 0
|
||||
case "R1":
|
||||
return 1
|
||||
case "R2":
|
||||
return 2
|
||||
case "R3":
|
||||
return 3
|
||||
case "R4":
|
||||
return 4
|
||||
case "R5":
|
||||
return 5
|
||||
case "R6":
|
||||
return 6
|
||||
case "R7":
|
||||
return 7
|
||||
case "R8":
|
||||
return 8
|
||||
case "R9":
|
||||
return 9
|
||||
case "R10":
|
||||
return 10
|
||||
case "R11":
|
||||
return 11
|
||||
case "R12":
|
||||
return 12
|
||||
case "R13":
|
||||
return 13
|
||||
case "R14":
|
||||
return 14
|
||||
case "R15":
|
||||
return 15
|
||||
case "R16":
|
||||
return 16
|
||||
case "R17":
|
||||
return 17
|
||||
case "R18":
|
||||
return 18
|
||||
case "R19":
|
||||
return 19
|
||||
case "R20":
|
||||
return 20
|
||||
case "R21":
|
||||
return 21
|
||||
case "R22":
|
||||
return 22
|
||||
case "R23":
|
||||
return 23
|
||||
case "R24":
|
||||
return 24
|
||||
case "R25":
|
||||
return 25
|
||||
case "R26", "REGCTXT", "CTXT":
|
||||
return 26
|
||||
case "R27", "REGTMP", "TMP":
|
||||
return 27
|
||||
case "R28", "REGG", "g":
|
||||
return 28
|
||||
case "R29", "FP":
|
||||
return 29
|
||||
case "R30", "LR", "LINK":
|
||||
return 30
|
||||
case "R31", "ZR":
|
||||
return 31
|
||||
case "SP":
|
||||
return 31 // SP and ZR share encoding 31; context determines meaning
|
||||
}
|
||||
// F0–F31.
|
||||
if len(name) >= 1 && name[0] == 'F' {
|
||||
n := 0
|
||||
for i := 1; i < len(name); i++ {
|
||||
if name[i] < '0' || name[i] > '9' {
|
||||
return -1
|
||||
}
|
||||
n = n*10 + int(name[i]-'0')
|
||||
}
|
||||
if n <= 31 {
|
||||
return n
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// arm64IsSP reports whether a register operand is the stack pointer (R31/SP),
|
||||
// which uses a different encoding path for some instructions.
|
||||
func arm64IsSP(name string) bool {
|
||||
return name == "SP"
|
||||
}
|
||||
|
||||
// ---- format helpers ----
|
||||
|
||||
// a64wordLE encodes a uint32 as 4 little-endian bytes.
|
||||
func a64wordLE(w uint32) []byte {
|
||||
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
|
||||
}
|
||||
|
||||
// a64WordsLE concatenates one or more instruction words as little-endian bytes.
|
||||
func a64WordsLE(ws ...uint32) []byte {
|
||||
var out []byte
|
||||
for _, w := range ws {
|
||||
out = append(out, a64wordLE(w)...)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// ---- data-processing (shifted register) ----
|
||||
|
||||
// a64DPSR encodes a data-processing (shifted register) instruction:
|
||||
// sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd.
|
||||
func a64DPSR(sf, op, S, shift, rm, imm6, rn, rd uint32) uint32 {
|
||||
return sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | rm<<16 | imm6<<10 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// ---- data-processing (immediate) ----
|
||||
|
||||
// a64AddSub encodes an ADD/SUB (immediate) instruction:
|
||||
// sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | Rn<<5 | Rd.
|
||||
func a64AddSub(sf, op, S, sh, imm12, rn, rd uint32) uint32 {
|
||||
return sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// ---- logical (immediate) ----
|
||||
|
||||
// a64LogicalImm encodes a logical (immediate) instruction:
|
||||
// sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd.
|
||||
func a64LogicalImm(sf, opc, N, immr, imms, rn, rd uint32) uint32 {
|
||||
return sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// ---- move wide ----
|
||||
|
||||
// a64MoveWide encodes a MOVZ/MOVK/MOVN instruction:
|
||||
// sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd.
|
||||
func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 {
|
||||
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
|
||||
}
|
||||
|
||||
// ---- load/store (unsigned immediate, scaled) ----
|
||||
|
||||
// a64LSU encodes a load/store register (unsigned immediate, scaled):
|
||||
// size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt.
|
||||
// (0x39<<24 encodes bits 29:24 = 111001, the scaled unsigned offset form.)
|
||||
func a64LSU(size, V, opc, imm12, rn, rt uint32) uint32 {
|
||||
return size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | rn<<5 | rt
|
||||
}
|
||||
|
||||
// ---- load/store (unscaled immediate) ----
|
||||
|
||||
// a64LSUnscaled encodes a load/store register (unscaled immediate, 9-bit signed):
|
||||
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<12 | Rn<<5 | Rt.
|
||||
// Note: the 0<<24 distinguishes unscaled from the pre/post-index forms.
|
||||
func a64LSUnscaled(size, V, opc int, imm9 int32, rn, rt int) uint32 {
|
||||
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | uint32(opc)<<22 |
|
||||
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
||||
}
|
||||
|
||||
// ---- load/store pair ----
|
||||
|
||||
// a64LSP encodes a load/store pair instruction (signed offset):
|
||||
// opc<<30 | 0x5<<27 | V<<26 | 2<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
|
||||
// opc: 0=32-bit, 1=reserved, 2=64-bit. V: 0=integer, 1=FP/SIMD.
|
||||
// L: 0=store, 1=load. imm7 is the signed scaled offset (÷8 for 64-bit pairs).
|
||||
func a64LSP(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
|
||||
return opc<<30 | 5<<27 | V<<26 | 2<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
|
||||
}
|
||||
|
||||
// ---- load/store pair (pre-index) ----
|
||||
|
||||
// a64LSPPre encodes a load/store pair (pre-index):
|
||||
// opc<<30 | 0x5<<27 | V<<26 | 0b11<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
|
||||
func a64LSPPre(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
|
||||
return opc<<30 | 5<<27 | V<<26 | 3<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
|
||||
}
|
||||
|
||||
// ---- load/store pair (post-index) ----
|
||||
|
||||
// a64LSPPost encodes a load/store pair (post-index):
|
||||
// opc<<30 | 0x5<<27 | V<<26 | 0b01<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
|
||||
func a64LSPPost(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
|
||||
return opc<<30 | 5<<27 | V<<26 | 1<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
|
||||
}
|
||||
|
||||
// ---- pre-index load/store ----
|
||||
|
||||
// a64LSPreIndex encodes a load/store register (pre-index):
|
||||
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
|
||||
func a64LSPreIndex(size, V, opc uint32, imm9 int32, rn, rt uint32) uint32 {
|
||||
return size<<30 | 7<<27 | V<<26 | opc<<22 | 3<<10 | (uint32(imm9)&0x1FF)<<12 | rn<<5 | rt
|
||||
}
|
||||
|
||||
// ---- post-index load/store ----
|
||||
|
||||
// a64LSPostIndex encodes a load/store register (post-index):
|
||||
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
|
||||
func a64LSPostIndex(size, V, opc uint32, imm9 int32, rn, rt uint32) uint32 {
|
||||
return size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<10 | (uint32(imm9)&0x1FF)<<12 | rn<<5 | rt
|
||||
}
|
||||
|
||||
// ---- branches ----
|
||||
|
||||
// a64Branch encodes an unconditional branch (B/BL):
|
||||
// op<<31 | 0x5<<26 | imm26.
|
||||
func a64Branch(op uint32, imm26 int32) uint32 {
|
||||
return op<<31 | 5<<26 | (uint32(imm26) & 0x03FFFFFF)
|
||||
}
|
||||
|
||||
// a64BranchCond encodes a conditional branch (B.cond):
|
||||
// 0x2A<<25 | imm19<<5 | cond.
|
||||
func a64BranchCond(imm19 int32, cond uint32) uint32 {
|
||||
return 0x2A<<25 | (uint32(imm19)&0x7FFFF)<<5 | cond&0xF
|
||||
}
|
||||
|
||||
// a64UncondBranch encodes an unconditional branch register (BR/BLR/RET):
|
||||
// 0x6B<<25 | opc<<21 | 0x1F<<16 | Rn<<5 | Rd.
|
||||
// opc: 0=BR, 1=BLR, 2=RET. For RET, Rn defaults to LR(30).
|
||||
func a64UncondBranch(opc, rn, rd uint32) uint32 {
|
||||
return 0x6B<<25 | opc<<21 | 0x1F<<16 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// ---- ADR/ADRP ----
|
||||
|
||||
// a64ADR encodes an ADR instruction (p=0) or ADRP instruction (p=1):
|
||||
// p<<31 | immlo<<29 | 0x10<<24 | immhi<<5 | Rd.
|
||||
func a64ADR(p uint32, immhi int32, immlo uint32, rd uint32) uint32 {
|
||||
return p<<31 | immlo<<29 | 0x10<<24 | (uint32(immhi)&0x7FFFF)<<5 | rd
|
||||
}
|
||||
|
||||
// ---- EXTR ----
|
||||
|
||||
// a64EXTR encodes an EXTR instruction:
|
||||
// sf<<31 | 0<<29 | 0x27<<23 | N<<22 | 0<<21 | Rm<<16 | imms<<10 | Rn<<5 | Rd.
|
||||
func a64EXTR(sf, N, rm, imms, rn, rd uint32) uint32 {
|
||||
return sf<<31 | 0x27<<23 | N<<22 | rm<<16 | imms<<10 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// ---- system ----
|
||||
|
||||
// a64NOP encodes a NOP: 0xd503201f.
|
||||
const a64NOP uint32 = 0xd503201f
|
||||
|
||||
// a64BRK encodes a BRK instruction: 0xd4200000 | imm16<<5.
|
||||
func a64BRK(imm16 uint32) uint32 {
|
||||
return 0xd4200000 | imm16<<5
|
||||
}
|
||||
|
||||
// ---- condition codes ----
|
||||
|
||||
const (
|
||||
a64CondEQ = 0x0
|
||||
a64CondNE = 0x1
|
||||
a64CondCS = 0x2
|
||||
a64CondHS = 0x2
|
||||
a64CondCC = 0x3
|
||||
a64CondLO = 0x3
|
||||
a64CondMI = 0x4
|
||||
a64CondPL = 0x5
|
||||
a64CondVS = 0x6
|
||||
a64CondVC = 0x7
|
||||
a64CondHI = 0x8
|
||||
a64CondLS = 0x9
|
||||
a64CondGE = 0xa
|
||||
a64CondLT = 0xb
|
||||
a64CondGT = 0xc
|
||||
a64CondLE = 0xd
|
||||
a64CondAL = 0xe
|
||||
a64CondNV = 0xf
|
||||
)
|
||||
|
||||
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
|
||||
var arm64CondMap = map[string]uint32{
|
||||
"EQ": a64CondEQ,
|
||||
"NE": a64CondNE,
|
||||
"CS": a64CondCS,
|
||||
"HS": a64CondHS,
|
||||
"CC": a64CondCC,
|
||||
"LO": a64CondLO,
|
||||
"MI": a64CondMI,
|
||||
"PL": a64CondPL,
|
||||
"VS": a64CondVS,
|
||||
"VC": a64CondVC,
|
||||
"HI": a64CondHI,
|
||||
"LS": a64CondLS,
|
||||
"GE": a64CondGE,
|
||||
"LT": a64CondLT,
|
||||
"GT": a64CondGT,
|
||||
"LE": a64CondLE,
|
||||
}
|
||||
|
||||
// ---- instruction format tags ----
|
||||
|
||||
type a64Format uint8
|
||||
|
||||
const (
|
||||
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
|
||||
a64FDPIR // data-processing (immediate): ADD/SUB $imm
|
||||
a64FLogImm // logical (immediate): AND/ORR/EOR $imm
|
||||
a64FMovWide // move wide: MOVZ, MOVN, MOVK
|
||||
a64FLSU // load/store (unsigned immediate, scaled)
|
||||
a64FLSUnscaled // load/store (unscaled immediate)
|
||||
a64FLSPair // load/store pair
|
||||
a64FBranch // unconditional branch (B/BL)
|
||||
a64FBranchCond // conditional branch (B.cond)
|
||||
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
|
||||
a64FADR // ADR/ADRP
|
||||
a64FEXTR // EXTR
|
||||
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
||||
a64FSystem // system: NOP, BRK, etc.
|
||||
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
||||
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
|
||||
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
|
||||
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
|
||||
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
|
||||
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
|
||||
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
||||
a64FFMovGR // FMOV between GP and FP registers
|
||||
a64FCRC32 // CRC32
|
||||
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
||||
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR
|
||||
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
||||
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
|
||||
)
|
||||
|
||||
// a64Enc is one instruction's encoding: its bit layout (format) and the
|
||||
// opcode constant, positioned at its exact bit range.
|
||||
type a64Enc struct {
|
||||
format a64Format
|
||||
op uint32 // the pre-positioned opcode bits
|
||||
size int // 4 for most, 8 for DP-imm with shift, etc.
|
||||
}
|
||||
|
||||
// a64InstrTable maps AArch64 mnemonics (as the Go assembler spells them) to
|
||||
// their encoding. The base integer, memory, floating-point and SIMD
|
||||
// instruction sets are covered.
|
||||
var a64InstrTable = map[string]a64Enc{}
|
||||
|
||||
func init() {
|
||||
// ---- data-processing (shifted register) ----
|
||||
// Format: sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
|
||||
dpsr := map[string]uint32{
|
||||
// Add/Sub
|
||||
"ADD": 1<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=1, op=0, S=0 (64-bit default)
|
||||
"ADDW": 0<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=0
|
||||
"ADDS": 1<<31 | 0<<30 | 1<<29 | 0x0b<<24,
|
||||
"ADDSW": 0<<31 | 0<<30 | 1<<29 | 0x0b<<24,
|
||||
"SUB": 1<<31 | 1<<30 | 0<<29 | 0x0b<<24,
|
||||
"SUBW": 0<<31 | 1<<30 | 0<<29 | 0x0b<<24,
|
||||
"SUBS": 1<<31 | 1<<30 | 1<<29 | 0x0b<<24,
|
||||
"SUBSW": 0<<31 | 1<<30 | 1<<29 | 0x0b<<24,
|
||||
// Logical (shifted register)
|
||||
"AND": 1<<31 | 0<<29 | 0x0a<<24,
|
||||
"ANDW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
"BIC": 1<<31 | 0<<29 | 0x0a<<24 | 1<<21,
|
||||
"BICW": 0<<31 | 0<<29 | 0x0a<<24 | 1<<21,
|
||||
"ORR": 1<<31 | 1<<29 | 0x0a<<24,
|
||||
"ORRW": 0<<31 | 1<<29 | 0x0a<<24,
|
||||
"ORN": 1<<31 | 1<<29 | 0x0a<<24 | 1<<21,
|
||||
"ORNW": 0<<31 | 1<<29 | 0x0a<<24 | 1<<21,
|
||||
"EOR": 1<<31 | 2<<29 | 0x0a<<24,
|
||||
"EORW": 0<<31 | 2<<29 | 0x0a<<24,
|
||||
"EON": 1<<31 | 2<<29 | 0x0a<<24 | 1<<21,
|
||||
"EONW": 0<<31 | 2<<29 | 0x0a<<24 | 1<<21,
|
||||
"ANDS": 1<<31 | 3<<29 | 0x0a<<24,
|
||||
"ANDSW": 0<<31 | 3<<29 | 0x0a<<24,
|
||||
"BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21,
|
||||
"BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21,
|
||||
// Shift
|
||||
"LSL": 1<<31 | 0<<29 | 0x0a<<24, // alias of UBFM
|
||||
"LSLW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
"LSR": 1<<31 | 0<<29 | 0x0a<<24,
|
||||
"LSRW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
"ASR": 1<<31 | 0<<29 | 0x0a<<24,
|
||||
"ASRW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
"ROR": 1<<31 | 0<<29 | 0x0a<<24,
|
||||
"RORW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
// Multiply
|
||||
"MADD": 1<<31 | 0<<29 | 0x1b<<24 | 0<<21,
|
||||
"MADDW": 0<<31 | 0<<29 | 0x1b<<24 | 0<<21,
|
||||
"MSUB": 1<<31 | 0<<29 | 0x1b<<24 | 1<<21,
|
||||
"MSUBW": 0<<31 | 0<<29 | 0x1b<<24 | 1<<21,
|
||||
// Divide
|
||||
"SDIV": 1<<31 | 0<<29 | 0x0d<<24,
|
||||
"SDIVW": 0<<31 | 0<<29 | 0x0d<<24,
|
||||
"UDIV": 1<<31 | 0<<29 | 0x0d<<24 | 1<<10,
|
||||
"UDIVW": 0<<31 | 0<<29 | 0x0d<<24 | 1<<10,
|
||||
// CRC
|
||||
"CRC32B": 0<<31 | 0<<29 | 0x1b<<24 | 4<<10,
|
||||
"CRC32H": 0<<31 | 0<<29 | 0x1b<<24 | 5<<10,
|
||||
"CRC32W": 0<<31 | 0<<29 | 0x1b<<24 | 6<<10,
|
||||
"CRC32X": 1<<31 | 0<<29 | 0x1b<<24 | 7<<10,
|
||||
// Conditional select
|
||||
"CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10,
|
||||
"CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10,
|
||||
"CSINC": 1<<31 | 0<<29 | 0x1d<<24 | 1<<10,
|
||||
"CSINCW": 0<<31 | 0<<29 | 0x1d<<24 | 1<<10,
|
||||
"CSINV": 1<<31 | 0<<29 | 0x1d<<24 | 2<<10,
|
||||
"CSINVW": 0<<31 | 0<<29 | 0x1d<<24 | 2<<10,
|
||||
"CSNEG": 1<<31 | 0<<29 | 0x1d<<24 | 3<<10,
|
||||
"CSNEGW": 0<<31 | 0<<29 | 0x1d<<24 | 3<<10,
|
||||
}
|
||||
for m, op := range dpsr {
|
||||
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
|
||||
}
|
||||
|
||||
// Aliases that map to the same encoding as their target.
|
||||
a64InstrTable["CMP"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
|
||||
a64InstrTable["CMPW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBSW"]}
|
||||
a64InstrTable["CMN"] = a64Enc{format: a64FDPSR, op: dpsr["ADDS"]}
|
||||
a64InstrTable["CMNW"] = a64Enc{format: a64FDPSR, op: dpsr["ADDSW"]}
|
||||
a64InstrTable["TST"] = a64Enc{format: a64FDPSR, op: dpsr["ANDS"]}
|
||||
a64InstrTable["TSTW"] = a64Enc{format: a64FDPSR, op: dpsr["ANDSW"]}
|
||||
a64InstrTable["NEG"] = a64Enc{format: a64FDPSR, op: dpsr["SUB"]}
|
||||
a64InstrTable["NEGW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBW"]}
|
||||
a64InstrTable["NEGS"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
|
||||
a64InstrTable["MVN"] = a64Enc{format: a64FDPSR, op: dpsr["ORN"]}
|
||||
a64InstrTable["MVNW"] = a64Enc{format: a64FDPSR, op: dpsr["ORNW"]}
|
||||
a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]}
|
||||
a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]}
|
||||
|
||||
// ---- data-processing (immediate) ----
|
||||
// ADD/SUB $imm, Rn, Rd
|
||||
a64InstrTable["ADDImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 0<<29 | 0x11<<24}
|
||||
a64InstrTable["ADDWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 0<<30 | 0<<29 | 0x11<<24}
|
||||
a64InstrTable["SUBImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 0<<29 | 0x11<<24}
|
||||
a64InstrTable["SUBWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 1<<30 | 0<<29 | 0x11<<24}
|
||||
a64InstrTable["ADDSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 1<<29 | 0x11<<24}
|
||||
a64InstrTable["SUBSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 1<<29 | 0x11<<24}
|
||||
|
||||
// ---- move wide ----
|
||||
// MOVZ/MOVN/MOVK
|
||||
a64InstrTable["MOVZ"] = a64Enc{format: a64FMovWide, op: 1<<31 | 2<<29 | 0x25<<23}
|
||||
a64InstrTable["MOVZW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 2<<29 | 0x25<<23}
|
||||
a64InstrTable["MOVN"] = a64Enc{format: a64FMovWide, op: 1<<31 | 0<<29 | 0x25<<23}
|
||||
a64InstrTable["MOVNW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 0<<29 | 0x25<<23}
|
||||
a64InstrTable["MOVK"] = a64Enc{format: a64FMovWide, op: 1<<31 | 3<<29 | 0x25<<23}
|
||||
a64InstrTable["MOVKW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 3<<29 | 0x25<<23}
|
||||
|
||||
// ---- ADR/ADRP ----
|
||||
a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0}
|
||||
a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1}
|
||||
|
||||
// ---- load/store (unsigned immediate) ----
|
||||
a64InstrTable["MOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<22} // LDR 64-bit
|
||||
a64InstrTable["MOVWU"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<22} // LDR 32-bit unsigned
|
||||
a64InstrTable["MOVHU"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 1<<22} // LDRH unsigned
|
||||
a64InstrTable["MOVBU"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 1<<22} // LDRB unsigned
|
||||
a64InstrTable["MOVW"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 2<<22} // LDRSW (signed 32→64)
|
||||
a64InstrTable["MOVH"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 2<<22} // LDRSH (signed half)
|
||||
a64InstrTable["MOVB"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 2<<22} // LDRSB (signed byte)
|
||||
a64InstrTable["FMOVS"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 32-bit FP
|
||||
a64InstrTable["FMOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 64-bit FP
|
||||
|
||||
// Store opcodes (load ^ (1<<22)):
|
||||
// STR 64-bit: size=3, V=0, opc=00 → 3<<30 | 7<<27 | 0<<22
|
||||
// STR 32-bit: size=2, V=0, opc=00 → 2<<30 | 7<<27 | 0<<22
|
||||
// STRH: size=1, V=0, opc=00 → 1<<30 | 7<<27 | 0<<22
|
||||
// STRB: size=0, V=0, opc=00 → 0<<30 | 7<<27 | 0<<22
|
||||
|
||||
// ---- branches ----
|
||||
a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26}
|
||||
a64InstrTable["BL"] = a64Enc{format: a64FBranch, op: 1<<31 | 5<<26}
|
||||
|
||||
// Conditional branches.
|
||||
condBranches := map[string]uint32{
|
||||
"BEQ": 0x0, "BNE": 0x1, "BCS": 0x2, "BHS": 0x2,
|
||||
"BCC": 0x3, "BLO": 0x3, "BMI": 0x4, "BPL": 0x5,
|
||||
"BVS": 0x6, "BVC": 0x7, "BHI": 0x8, "BLS": 0x9,
|
||||
"BGE": 0xa, "BLT": 0xb, "BGT": 0xc, "BLE": 0xd,
|
||||
}
|
||||
for name, cond := range condBranches {
|
||||
a64InstrTable[name] = a64Enc{format: a64FBranchCond, op: 0x2A<<25 | cond}
|
||||
}
|
||||
|
||||
// Unconditional branch register (BR/BLR/RET).
|
||||
a64InstrTable["BR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 0<<21}
|
||||
a64InstrTable["BLR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 1<<21}
|
||||
a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21}
|
||||
|
||||
// ---- system ----
|
||||
a64InstrTable["NOP"] = a64Enc{format: a64FSystem, op: a64NOP}
|
||||
a64InstrTable["NOOP"] = a64Enc{format: a64FSystem, op: a64NOP}
|
||||
a64InstrTable["BRK"] = a64Enc{format: a64FSystem, op: 0xd4200000}
|
||||
a64InstrTable["UNDEF"] = a64Enc{format: a64FSystem, op: a64BRK(0)}
|
||||
|
||||
// ---- EXTR ----
|
||||
a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22}
|
||||
a64InstrTable["EXTRW"] = a64Enc{format: a64FEXTR, op: 0<<31 | 0x27<<23 | 0<<22}
|
||||
|
||||
// ---- bitfield ----
|
||||
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
|
||||
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
|
||||
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["UBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
|
||||
a64InstrTable["BFI"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
|
||||
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
|
||||
|
||||
// ---- FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL ----
|
||||
fp3 := map[string]uint32{
|
||||
"FADDS": 0x1e202800, "FADDD": 0x1e602800,
|
||||
"FSUBS": 0x1e203800, "FSUBD": 0x1e603800,
|
||||
"FMULS": 0x1e200800, "FMULD": 0x1e600800,
|
||||
"FDIVS": 0x1e201800, "FDIVD": 0x1e601800,
|
||||
"FMAXS": 0x1e204800, "FMAXD": 0x1e604800,
|
||||
"FMINS": 0x1e205800, "FMIND": 0x1e605800,
|
||||
"FMAXNMS": 0x1e206800, "FMAXNMD": 0x1e606800,
|
||||
"FMINNMS": 0x1e207800, "FMINNMD": 0x1e607800,
|
||||
"FNMULS": 0x1e208800, "FNMULD": 0x1e608800,
|
||||
}
|
||||
for m, op := range fp3 {
|
||||
a64InstrTable[m] = a64Enc{format: a64FFP3, op: op}
|
||||
}
|
||||
|
||||
// ---- FP unary (Rn, Rd): FMOV reg-reg, FABS, FNEG, FSQRT, FCVT, FRINT* ----
|
||||
fp1 := map[string]uint32{
|
||||
"FMOVS": 0x1e204000, "FMOVD": 0x1e604000,
|
||||
"FABSS": 0x1e20c000, "FABSD": 0x1e60c000,
|
||||
"FNEGS": 0x1e214000, "FNEGD": 0x1e614000,
|
||||
"FSQRTS": 0x1e21c000, "FSQRTD": 0x1e61c000,
|
||||
"FCVTSD": 0x1e22c000, "FCVTDS": 0x1e624000,
|
||||
"FRINTNS": 0x1e244000, "FRINTND": 0x1e644000,
|
||||
"FRINTPS": 0x1e24c000, "FRINTPD": 0x1e64c000,
|
||||
"FRINTMS": 0x1e254000, "FRINTMD": 0x1e654000,
|
||||
"FRINTZS": 0x1e25c000, "FRINTZD": 0x1e65c000,
|
||||
"FRINTAS": 0x1e264000, "FRINTAD": 0x1e664000,
|
||||
"FRINTXS": 0x1e274000, "FRINTXD": 0x1e674000,
|
||||
"FRINTIS": 0x1e27c000, "FRINTID": 0x1e67c000,
|
||||
}
|
||||
for m, op := range fp1 {
|
||||
a64InstrTable[m] = a64Enc{format: a64FFPUnary, op: op}
|
||||
}
|
||||
|
||||
// ---- FP 4-operand FMA (Ra, Rm, Rn, Rd) ----
|
||||
fp4 := map[string]uint32{
|
||||
"FMADDS": 0x1f000000, "FMADDD": 0x1f400000,
|
||||
"FMSUBS": 0x1f008000, "FMSUBD": 0x1f408000,
|
||||
"FNMADDS": 0x1f200000, "FNMADDD": 0x1f600000,
|
||||
"FNMSUBS": 0x1f208000, "FNMSUBD": 0x1f608000,
|
||||
}
|
||||
for m, op := range fp4 {
|
||||
a64InstrTable[m] = a64Enc{format: a64FFP4, op: op}
|
||||
}
|
||||
|
||||
// ---- FP compare (Rm, Rn or #0, Rn) ----
|
||||
fpcmp := map[string]uint32{
|
||||
"FCMPS": 0x1e202000, "FCMPD": 0x1e602000,
|
||||
"FCMPES": 0x1e202010, "FCMPED": 0x1e602010,
|
||||
}
|
||||
for m, op := range fpcmp {
|
||||
a64InstrTable[m] = a64Enc{format: a64FFPCmp, op: op}
|
||||
}
|
||||
|
||||
// ---- FP conditional compare (Rm, Rn, #nzcv, cond) ----
|
||||
fpccmp := map[string]uint32{
|
||||
"FCCMPS": 0x1e200400, "FCCMPD": 0x1e600400,
|
||||
"FCCMPES": 0x1e200410, "FCCMPED": 0x1e600410,
|
||||
}
|
||||
for m, op := range fpccmp {
|
||||
a64InstrTable[m] = a64Enc{format: a64FFPCCmp, op: op}
|
||||
}
|
||||
|
||||
// ---- FP conditional select (Rm, Rn, Rd, cond) ----
|
||||
a64InstrTable["FCSELS"] = a64Enc{format: a64FFPSel, op: 0x1e200c00}
|
||||
a64InstrTable["FCSELD"] = a64Enc{format: a64FFPSel, op: 0x1e600c00}
|
||||
|
||||
// ---- FP ↔ integer conversion ----
|
||||
fpcvt := map[string]uint32{
|
||||
"FCVTZSD": 0x9e780000, "FCVTZSDW": 0x1e780000,
|
||||
"FCVTZSS": 0x9e380000, "FCVTZSSW": 0x1e380000,
|
||||
"FCVTZUD": 0x9e790000, "FCVTZUDW": 0x1e790000,
|
||||
"FCVTZUS": 0x9e390000, "FCVTZUSW": 0x1e390000,
|
||||
"SCVTFD": 0x9e620000, "SCVTFS": 0x9e220000,
|
||||
"SCVTFWD": 0x1e620000, "SCVTFWS": 0x1e220000,
|
||||
"UCVTFD": 0x9e630000, "UCVTFS": 0x9e230000,
|
||||
"UCVTFWD": 0x1e630000, "UCVTFWS": 0x1e230000,
|
||||
}
|
||||
for m, op := range fpcvt {
|
||||
a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op}
|
||||
}
|
||||
|
||||
// ---- FMOV between GP and FP registers ----
|
||||
a64InstrTable["FMOVGR"] = a64Enc{format: a64FFMovGR, op: 0x1e260000} // placeholder, actual encoding depends on direction
|
||||
|
||||
// ---- conditional select: CSEL, CSINC, CSINV, CSNEG ----
|
||||
csel := map[string]uint32{
|
||||
"CSEL": 0x9a800000, "CSELW": 0x1a800000,
|
||||
"CSINC": 0x9a800400, "CSINCW": 0x1a800400,
|
||||
"CSINV": 0xda800000, "CSINVW": 0x5a800000,
|
||||
"CSNEG": 0xda800400, "CSNEGW": 0x5a800400,
|
||||
}
|
||||
for m, op := range csel {
|
||||
a64InstrTable[m] = a64Enc{format: a64FCSEL, op: op}
|
||||
}
|
||||
// Aliases
|
||||
a64InstrTable["CSET"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
|
||||
a64InstrTable["CSETW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
|
||||
a64InstrTable["CSETM"] = a64Enc{format: a64FCSEL, op: 0xda800000}
|
||||
a64InstrTable["CSETMW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
|
||||
a64InstrTable["CINC"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
|
||||
a64InstrTable["CINCW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
|
||||
a64InstrTable["CINV"] = a64Enc{format: a64FCSEL, op: 0xda800000}
|
||||
a64InstrTable["CINVW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
|
||||
a64InstrTable["CNEG"] = a64Enc{format: a64FCSEL, op: 0xda800400}
|
||||
a64InstrTable["CNEGW"] = a64Enc{format: a64FCSEL, op: 0x5a800400}
|
||||
|
||||
// ---- CRC32 ----
|
||||
crc32 := map[string]uint32{
|
||||
"CRC32B": 0x1ac04000, "CRC32H": 0x1ac04400,
|
||||
"CRC32W": 0x1ac04800, "CRC32X": 0x9ac04c00,
|
||||
"CRC32CB": 0x1ac05000, "CRC32CH": 0x1ac05400,
|
||||
"CRC32CW": 0x1ac05800, "CRC32CX": 0x9ac05c00,
|
||||
}
|
||||
for m, op := range crc32 {
|
||||
a64InstrTable[m] = a64Enc{format: a64FCRC32, op: op}
|
||||
}
|
||||
|
||||
// ---- exclusive load/store ----
|
||||
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
|
||||
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
|
||||
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
|
||||
a64InstrTable["LDXRW"] = a64Enc{format: a64FExcl, op: 0x885f7c00}
|
||||
a64InstrTable["LDAXR"] = a64Enc{format: a64FExcl, op: 0xc85ffc00}
|
||||
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
|
||||
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
|
||||
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
|
||||
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
|
||||
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
|
||||
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
|
||||
a64InstrTable["STXRW"] = a64Enc{format: a64FExcl, op: 0x88007c00}
|
||||
a64InstrTable["STLXR"] = a64Enc{format: a64FExcl, op: 0xc800fc00}
|
||||
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
|
||||
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
|
||||
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
|
||||
|
||||
// ---- LSE atomics ----
|
||||
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
|
||||
a64InstrTable["LDADDW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x00<<10}
|
||||
a64InstrTable["LDADDB"] = a64Enc{format: a64FLSE, op: 0<<30 | 0x1c1<<21 | 0x00<<10}
|
||||
a64InstrTable["LDADDH"] = a64Enc{format: a64FLSE, op: 1<<30 | 0x1c1<<21 | 0x00<<10}
|
||||
a64InstrTable["CASD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x45<<21 | 0x1f<<10}
|
||||
a64InstrTable["CASW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x45<<21 | 0x1f<<10}
|
||||
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
|
||||
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
|
||||
|
||||
// ---- SIMD basics ----
|
||||
a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400}
|
||||
a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400}
|
||||
a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00}
|
||||
}
|
||||
|
||||
// ---- load/store helper tables ----
|
||||
|
||||
// a64LSType describes the load/store parameters for a MOV width mnemonic.
|
||||
type a64LSType struct {
|
||||
size int // 0=byte, 1=half, 2=word, 3=dword
|
||||
V int // 0=integer, 1=FP
|
||||
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
|
||||
}
|
||||
|
||||
// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters.
|
||||
// For loads, opc selects signed vs unsigned; for stores, we flip the opc.
|
||||
var a64LoadTable = map[string]a64LSType{
|
||||
"MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset)
|
||||
"MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned)
|
||||
"MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit)
|
||||
"MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned)
|
||||
"MOVH": {1, 0, 2}, // LDRSH (16-bit signed)
|
||||
"MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned)
|
||||
"MOVB": {0, 0, 2}, // LDRSB (8-bit signed)
|
||||
"FMOVS": {2, 1, 1}, // LDR S (32-bit FP)
|
||||
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
|
||||
}
|
||||
|
||||
// a64StoreOpc returns the store opc for a given load type.
|
||||
// For integer: store opc = 00 (the load opc bits cleared).
|
||||
// For FP: store opc = 00 (same pattern).
|
||||
func a64StoreOpc(t a64LSType) int {
|
||||
if t.V == 1 {
|
||||
return 0 // FP store
|
||||
}
|
||||
return 0 // integer store
|
||||
}
|
||||
|
||||
// a64MovRegTable maps register-to-register MOV mnemonic expansions.
|
||||
// The Go toolchain encodes MOV Rn, Rd as ORR Rn, ZR, Rd.
|
||||
var a64MovRegTable = map[string]uint32{
|
||||
"MOVD": 1<<31 | 1<<29 | 0x0a<<24, // ORR 64-bit
|
||||
"MOVW": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
|
||||
"MOVB": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit (byte move)
|
||||
"MOVBU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
|
||||
"MOVH": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
|
||||
"MOVHU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
|
||||
"MOVWU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
|
||||
}
|
||||
|
||||
// arm64RegClass discriminates integer (R), floating-point (F) registers for
|
||||
// the MOV pseudo-instruction.
|
||||
type arm64RegClass int
|
||||
|
||||
const (
|
||||
arm64ClsNone arm64RegClass = iota
|
||||
arm64ClsGR
|
||||
arm64ClsFP
|
||||
)
|
||||
|
||||
// arm64RegClassOf reports the register class of a register operand name.
|
||||
func arm64RegClassOf(name string) arm64RegClass {
|
||||
switch {
|
||||
case name == "":
|
||||
return arm64ClsNone
|
||||
case len(name) >= 1 && name[0] == 'F':
|
||||
return arm64ClsFP
|
||||
default:
|
||||
return arm64ClsGR
|
||||
}
|
||||
}
|
||||
|
||||
// arm64Movcon returns the shift (in units of 16 bits) at which a non-zero
|
||||
// 16-bit chunk of v sits, or -1 if v cannot be represented as a single
|
||||
// MOVZ/MOVN immediate. This is the Go toolchain's movcon function.
|
||||
func arm64Movcon(v int64) int {
|
||||
for s := 0; s < 64; s += 16 {
|
||||
if (uint64(v) &^ (uint64(0xFFFF) << uint(s))) == 0 {
|
||||
return s
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
@@ -0,0 +1,574 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
func TestArm64LDRSTREncoding(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
got uint32
|
||||
want uint32
|
||||
}{
|
||||
{"LDR X4, [SP, #56]", a64LSU(3, 0, 1, 7, 31, 4), 0xf9401fe4},
|
||||
{"STR X4, [SP, #64]", a64LSU(3, 0, 0, 8, 31, 4), 0xf90023e4},
|
||||
{"STR X5, [SP, #32]", a64LSU(3, 0, 0, 4, 31, 5), 0xf90013e5},
|
||||
{"LDR X6, [SP, #32]", a64LSU(3, 0, 1, 4, 31, 6), 0xf94013e6},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
if tt.got != tt.want {
|
||||
t.Errorf("%s: got %08x, want %08x", tt.name, tt.got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64PrologueEncoding(t *testing.T) {
|
||||
fi := arm64FrameInfo{autosize: 48, frame: 32, leaf: false}
|
||||
pro := arm64Prologue(fi)
|
||||
if len(pro) != 12 {
|
||||
t.Fatalf("prologue length: got %d, want 12", len(pro))
|
||||
}
|
||||
expected := []uint32{0xf81d0ffe, 0xf81f83fd, 0xd10023fd}
|
||||
for i, w := range leWords(pro) {
|
||||
if w != expected[i] {
|
||||
t.Errorf("prologue word %d: got %08x, want %08x", i, w, expected[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64EpilogueSmallEncoding(t *testing.T) {
|
||||
fi := arm64FrameInfo{autosize: 48, frame: 32, leaf: false}
|
||||
ret := arm64Return(fi)
|
||||
if len(ret) != 12 {
|
||||
t.Fatalf("epilogue length: got %d, want 12", len(ret))
|
||||
}
|
||||
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #48; RET
|
||||
expected := []uint32{0xf85f83fd, 0xf84307fe, 0xd65f03c0}
|
||||
for i, w := range leWords(ret) {
|
||||
if w != expected[i] {
|
||||
t.Errorf("epilogue word %d: got %08x, want %08x", i, w, expected[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64LargeFrameEncoding(t *testing.T) {
|
||||
fi := arm64FrameInfo{autosize: 272, frame: 256, leaf: false}
|
||||
pro := arm64Prologue(fi)
|
||||
if len(pro) != 16 {
|
||||
t.Fatalf("prologue length: got %d, want 16", len(pro))
|
||||
}
|
||||
expected := []uint32{0xd10443f4, 0xa93ffa9d, 0x9100029f, 0xd10023fd}
|
||||
for i, w := range leWords(pro) {
|
||||
if w != expected[i] {
|
||||
t.Errorf("prologue word %d: got %08x, want %08x", i, w, expected[i])
|
||||
}
|
||||
}
|
||||
|
||||
epi := arm64Return(fi)
|
||||
if len(epi) != 12 {
|
||||
t.Fatalf("epilogue length: got %d, want 12", len(epi))
|
||||
}
|
||||
eexpected := []uint32{0xa97ffbfd, 0x910443ff, 0xd65f03c0}
|
||||
for i, w := range leWords(epi) {
|
||||
if w != eexpected[i] {
|
||||
t.Errorf("epilogue word %d: got %08x, want %08x", i, w, eexpected[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64NoFrame(t *testing.T) {
|
||||
fi := arm64FrameInfo{autosize: 0, frame: 0, leaf: true}
|
||||
pro := arm64Prologue(fi)
|
||||
if len(pro) != 0 {
|
||||
t.Errorf("no-frame prologue: got %d bytes, want 0", len(pro))
|
||||
}
|
||||
ret := arm64Return(fi)
|
||||
if len(ret) != 4 {
|
||||
t.Fatalf("no-frame return: got %d bytes, want 4", len(ret))
|
||||
}
|
||||
if leWord(ret) != 0xd65f03c0 {
|
||||
t.Errorf("no-frame RET: got %08x, want d65f03c0", leWord(ret))
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64RegNum(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
want int
|
||||
}{
|
||||
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
|
||||
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
|
||||
{"F0", 0}, {"F4", 4}, {"F31", 31},
|
||||
{"INVALID", -1}, {"X0", -1}, {"", -1},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
got := arm64RegNum(tt.name)
|
||||
if got != tt.want {
|
||||
t.Errorf("arm64RegNum(%q) = %d, want %d", tt.name, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64ComputeFrame(t *testing.T) {
|
||||
src := "TEXT ·f(SB), NOSPLIT, $32-0\n\tADD\tR4, R5\n\tRET\n"
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
fi := arm64ComputeFrame(f.Decls[0].(*ast.Text))
|
||||
if fi.frame != 32 {
|
||||
t.Errorf("frame: got %d, want 32", fi.frame)
|
||||
}
|
||||
if fi.autosize != 48 { // 32+8=40, aligned to48
|
||||
t.Errorf("autosize: got %d, want 48", fi.autosize)
|
||||
}
|
||||
// ADD + RET with no CALL/BL → leaf
|
||||
if !fi.leaf {
|
||||
t.Error("expected leaf")
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64IsLeaf(t *testing.T) {
|
||||
src := "TEXT ·f(SB), NOSPLIT, $0-0\n\tADD\tR4, R5\n\tRET\n"
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
if !arm64IsLeaf(f.Decls[0].(*ast.Text)) {
|
||||
t.Error("expected leaf")
|
||||
}
|
||||
|
||||
src2 := "TEXT ·f(SB), NOSPLIT, $0-0\n\tBL\tother(SB)\n\tRET\n"
|
||||
f2, errs := parser.Parse("test_arm64.s", src2)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
if arm64IsLeaf(f2.Decls[0].(*ast.Text)) {
|
||||
t.Error("expected non-leaf")
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64Bitmask(t *testing.T) {
|
||||
tests := []struct {
|
||||
v uint64
|
||||
sf int
|
||||
N, immr, imms uint32
|
||||
ok bool
|
||||
}{
|
||||
{1, 1, 1, 0, 0, true}, // single bit at pos 0
|
||||
{2, 1, 1, 63, 0, true}, // single bit at pos 1 (immr = esize-1)
|
||||
{0, 1, 0, 0, 0, false}, // zero is not a bitmask
|
||||
{0xFFFFFFFFFFFFFFFF, 1, 0, 0, 0, false}, // all ones is not a bitmask
|
||||
{0x5555555555555555, 1, 0, 0, 0x3E, true}, // alternating bits (esize=2, ones=1)
|
||||
{0xFFFFFFFF00000000, 1, 1, 32, 31, true}, // upper 32 bits set (esize=64, ones=32)
|
||||
}
|
||||
for _, tt := range tests {
|
||||
N, immr, imms, ok := arm64Bitmask(tt.v, tt.sf)
|
||||
if ok != tt.ok {
|
||||
t.Errorf("arm64Bitmask(%#x, %d): ok=%v, want %v", tt.v, tt.sf, ok, tt.ok)
|
||||
continue
|
||||
}
|
||||
if ok && (N != tt.N || immr != tt.immr || imms != tt.imms) {
|
||||
t.Errorf("arm64Bitmask(%#x, %d): N=%d immr=%d imms=%d, want N=%d immr=%d imms=%d",
|
||||
tt.v, tt.sf, N, immr, imms, tt.N, tt.immr, tt.imms)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64AssembleFile(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
|
||||
TEXT ·simple(SB), NOSPLIT, $0-0
|
||||
MOV R4, R5
|
||||
ADD R4, R5, R6
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
if len(img.Funcs) != 1 {
|
||||
t.Fatalf("got %d funcs, want 1", len(img.Funcs))
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
if fn.Name != "simple" {
|
||||
t.Errorf("func name: got %q, want %q", fn.Name, "simple")
|
||||
}
|
||||
//3 instructions ×4 bytes =12
|
||||
if fn.Size != 12 {
|
||||
t.Errorf("func size: got %d, want 12", fn.Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64AssembleFileWithFrame(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
|
||||
TEXT ·framed(SB), NOSPLIT, $16-8
|
||||
MOVD arg+0(FP), R4
|
||||
ADD $1, R4, R4
|
||||
MOVD R4, ret+0(FP)
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
if len(img.Funcs) != 1 {
|
||||
t.Fatalf("got %d funcs, want 1", len(img.Funcs))
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
if fn.Frame != 16 {
|
||||
t.Errorf("frame: got %d, want 16", fn.Frame)
|
||||
}
|
||||
// Prologue (3×4=12) + body (3×4=12) + RET epilogue (3×4=12) = 36
|
||||
if fn.Size != 36 {
|
||||
t.Errorf("func size: got %d, want 36", fn.Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64AssembleFileWithBranches(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
|
||||
TEXT ·branch(SB), NOSPLIT, $0-0
|
||||
BEQ done
|
||||
BNE skip
|
||||
skip:
|
||||
ADD R4, R5
|
||||
done:
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
if fn.Size != 16 {
|
||||
t.Errorf("func size: got %d, want 16", fn.Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64AssembleFileWithJumpChain(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
|
||||
TEXT ·chain(SB), NOSPLIT, $0-0
|
||||
BNE skip
|
||||
ADD R4, R5
|
||||
RET
|
||||
skip:
|
||||
B target
|
||||
target:
|
||||
ADD R6, R7
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
// BNE should be redirected past skip→target to target directly.
|
||||
if img.Funcs[0].Size != 24 {
|
||||
t.Errorf("func size: got %d, want 24", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64AssembleErrors(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
src string
|
||||
}{
|
||||
{"unsupported", "TEXT ·f(SB), NOSPLIT, $0-0\n\tINVALID\tR4, R5\n\tRET\n"},
|
||||
{"undefined label", "TEXT ·f(SB), NOSPLIT, $0-0\n\tB\tnosuch\n\tRET\n"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
f, errs := parser.Parse("test_arm64.s", tt.src)
|
||||
if len(errs) > 0 {
|
||||
return // parse error, that's fine
|
||||
}
|
||||
_, err := AssembleFileARM64(f)
|
||||
if err == nil {
|
||||
t.Error("expected error, got nil")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64Movcon(t *testing.T) {
|
||||
tests := []struct {
|
||||
v int64
|
||||
want int
|
||||
}{
|
||||
{0, 0}, // 0 fits at shift 0
|
||||
{1, 0}, // single bit at shift 0
|
||||
{0x10000, 16}, // single bit at shift 16
|
||||
{0x100000000, 32}, // single bit at shift 32
|
||||
{0xFF, 0}, // 0xFF fits at shift 0
|
||||
{0x12345, -1}, // multiple chunks, not movcon
|
||||
}
|
||||
for _, tt := range tests {
|
||||
got := arm64Movcon(tt.v)
|
||||
if got != tt.want {
|
||||
t.Errorf("arm64Movcon(%#x) = %d, want %d", tt.v, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64RegClassOf(t *testing.T) {
|
||||
if arm64RegClassOf("R4") != arm64ClsGR {
|
||||
t.Error("R4 should be GR")
|
||||
}
|
||||
if arm64RegClassOf("F4") != arm64ClsFP {
|
||||
t.Error("F4 should be FP")
|
||||
}
|
||||
if arm64RegClassOf("") != arm64ClsNone {
|
||||
t.Error("empty should be None")
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64ResolvePseudo(t *testing.T) {
|
||||
fi := arm64FrameInfo{autosize: 48, frame: 32}
|
||||
// FP: offset = sym.Offset + autosize +8
|
||||
base, off := arm64ResolvePseudo(&ast.Symbol{Pseudo: "FP", Offset: 0}, fi)
|
||||
if base != 31 || off != 56 {
|
||||
t.Errorf("FP: base=%d off=%d, want 31, 56", base, off)
|
||||
}
|
||||
// SP: offset = sym.Offset + frame +8
|
||||
base, off = arm64ResolvePseudo(&ast.Symbol{Pseudo: "SP", Offset: -8}, fi)
|
||||
if base != 31 || off != 32 {
|
||||
t.Errorf("SP: base=%d off=%d, want 31, 32", base, off)
|
||||
}
|
||||
// SB: unresolved
|
||||
base, _ = arm64ResolvePseudo(&ast.Symbol{Pseudo: "SB"}, fi)
|
||||
if base != -1 {
|
||||
t.Errorf("SB: base=%d, want -1", base)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64FPSel tests FP conditional select encoding.
|
||||
func TestArm64FPSel(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
FCSELD GE, F10, F11, F12
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
// FCSELD should be 4 bytes + RET 4 bytes = 8
|
||||
if img.Funcs[0].Size != 8 {
|
||||
t.Errorf("size: got %d, want 8", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64FPCvt tests FP conversion encoding.
|
||||
func TestArm64FPCvt(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
FCVTZSD F4, R0
|
||||
SCVTFD R4, F8
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
if img.Funcs[0].Size != 12 {
|
||||
t.Errorf("size: got %d, want 12", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64CSEL tests conditional select encoding.
|
||||
func TestArm64CSEL(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
CSEL EQ, R0, R1, R2
|
||||
CSET NE, R3
|
||||
CINC GE, R4, R5
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
if img.Funcs[0].Size != 16 {
|
||||
t.Errorf("size: got %d, want 16", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64CRC32 tests CRC32 encoding.
|
||||
func TestArm64CRC32(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
CRC32B R0, R2
|
||||
CRC32W R6, R8
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
if img.Funcs[0].Size != 12 {
|
||||
t.Errorf("size: got %d, want 12", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64Bitfield tests bitfield/shift encoding.
|
||||
func TestArm64Bitfield(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
ASR $4, R0, R1
|
||||
LSL $12, R4, R5
|
||||
EXTR $8, R0, R1, R2
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
if img.Funcs[0].Size != 16 {
|
||||
t.Errorf("size: got %d, want 16", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64SIMD tests SIMD encoding (via the instruction table).
|
||||
func TestArm64SIMD(t *testing.T) {
|
||||
// Verify SIMD instructions are in the table.
|
||||
for _, mnem := range []string{"VADD", "VSUB", "VMUL"} {
|
||||
if _, ok := a64InstrTable[mnem]; !ok {
|
||||
t.Errorf("%s not in instruction table", mnem)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64LoadImm64 tests 64-bit immediate loading.
|
||||
func TestArm64LoadImm64(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
MOVD $0x123456789ABCDEF0, R0
|
||||
MOVD $0, R1
|
||||
MOVD $1, R2
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
// $0x123456789ABCDEF0 needs 4 MOVZ/MOVK instructions (16 bytes)
|
||||
// $0 is 1 instruction (4 bytes)
|
||||
// $1 is 1 bitmask instruction (4 bytes)
|
||||
// RET is 1 instruction (4 bytes)
|
||||
if img.Funcs[0].Size != 28 {
|
||||
t.Errorf("size: got %d, want 28", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64BranchCond tests conditional branch encoding.
|
||||
func TestArm64BranchCond(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
BEQ done
|
||||
BNE done
|
||||
BGE done
|
||||
BLT done
|
||||
ADD R4, R5
|
||||
done:
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
// 4 branches + 1 ADD + 1 RET = 24 bytes
|
||||
if img.Funcs[0].Size != 24 {
|
||||
t.Errorf("size: got %d, want 24", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64Errors tests error paths.
|
||||
func TestArm64Errors(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
src string
|
||||
}{
|
||||
{"bad mnemonic", "TEXT ·f(SB), NOSPLIT, $0-0\n\tINVALID\tR4\n\tRET\n"},
|
||||
{"bad label", "TEXT ·f(SB), NOSPLIT, $0-0\n\tB\tnosuch\n\tRET\n"},
|
||||
{"bad register", "TEXT ·f(SB), NOSPLIT, $0-0\n\tADD\tR99, R0\n\tRET\n"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
f, errs := parser.Parse("test_arm64.s", tt.src)
|
||||
if len(errs) > 0 {
|
||||
return
|
||||
}
|
||||
_, err := AssembleFileARM64(f)
|
||||
if err == nil {
|
||||
t.Error("expected error, got nil")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// leWord reads a little-endian uint32 from b.
|
||||
func leWord(b []byte) uint32 {
|
||||
return uint32(b[0]) | uint32(b[1])<<8 | uint32(b[2])<<16 | uint32(b[3])<<24
|
||||
}
|
||||
|
||||
// leWords reads all little-endian uint32s from b.
|
||||
func leWords(b []byte) []uint32 {
|
||||
n := len(b) / 4
|
||||
w := make([]uint32, n)
|
||||
for i := range w {
|
||||
w[i] = leWord(b[i*4:])
|
||||
}
|
||||
return w
|
||||
}
|
||||
@@ -0,0 +1,237 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
// arm64 frame mapping, matching the Go toolchain's arm64 backend.
|
||||
//
|
||||
// Go's arm64 functions use R29 as the frame pointer (FP) and R30 as the link
|
||||
// register (LR). R31 is the stack pointer (SP). FP and SP in the source
|
||||
// are synthetic pseudo-registers resolved against the hardware SP and the
|
||||
// frame size.
|
||||
//
|
||||
// The autosize is the real stack adjustment: the declared local frame plus
|
||||
// 8 bytes for the saved link register, rounded up to a 16-byte multiple.
|
||||
// The toolchain adds an "extrasize" to align: if autosize%16 == 8, add 8;
|
||||
// if autosize%16 == 0, add 16.
|
||||
//
|
||||
// Prologue (autosize > 0, small frame ≤ 0xf0):
|
||||
//
|
||||
// MOVD.W LR, -autosize(SP) // pre-index: SP -= autosize, store LR at SP
|
||||
// MOVD FP, -8(SP) // store FP at SP-8
|
||||
// SUB $8, SP, FP // FP = SP - 8
|
||||
//
|
||||
// Prologue (autosize > 0, large frame > 0xf0):
|
||||
//
|
||||
// SUB $autosize, SP, R20 // R20 = SP - autosize
|
||||
// STP (FP, LR), -8(R20) // store FP,LR at R20-8
|
||||
// MOVD R20, SP // SP = R20
|
||||
// SUB $8, SP, FP // FP = SP - 8
|
||||
//
|
||||
// Epilogue (non-leaf, small frame):
|
||||
//
|
||||
// ADD $autosize-8, SP, FP // restore FP
|
||||
// ADD $autosize, SP, SP // deallocate frame
|
||||
// MOVD -8(SP), FP // (actually the reverse of prologue)
|
||||
// Actually:
|
||||
// MOVD -8(SP), FP // load FP from SP-8
|
||||
// MOVD.P autosize(SP), LR // post-index: load LR, SP += autosize
|
||||
//
|
||||
// Epilogue (non-leaf, large frame):
|
||||
// ADD $autosize-8, SP, FP
|
||||
// ADD $autosize, SP, SP
|
||||
// Actually:
|
||||
// LDP -8(SP), (FP, LR) // load FP,LR
|
||||
// ADD $autosize, SP, SP // deallocate frame
|
||||
//
|
||||
// Epilogue (leaf with frame):
|
||||
// ADD $autosize-8, SP, FP
|
||||
// ADD $autosize, SP, SP
|
||||
//
|
||||
// RET always emits as BR LR (0xd65f03c0).
|
||||
|
||||
import (
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
)
|
||||
|
||||
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
|
||||
type arm64FrameInfo struct {
|
||||
autosize int // the real SP adjustment (locals + saved LR + alignment)
|
||||
frame int // the declared $framesize
|
||||
args int // the declared -argsize
|
||||
noSplit bool // the NOSPLIT flag
|
||||
leaf bool // no call instructions in the body
|
||||
}
|
||||
|
||||
// arm64ComputeFrame derives the frame layout for a TEXT function.
|
||||
func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
|
||||
fi := arm64FrameInfo{
|
||||
frame: frameSize(t),
|
||||
args: argsSize(t),
|
||||
}
|
||||
for _, f := range t.Flags {
|
||||
if f == "NOSPLIT" {
|
||||
fi.noSplit = true
|
||||
}
|
||||
}
|
||||
fi.leaf = arm64IsLeaf(t)
|
||||
|
||||
if fi.frame != 0 || !fi.leaf {
|
||||
fi.autosize = fi.frame + 8 // space for the saved LR
|
||||
if fi.autosize%16 != 0 {
|
||||
// The toolchain aligns to 16: if autosize%16 == 8, add 8;
|
||||
// otherwise add whatever is needed.
|
||||
fi.autosize += 16 - (fi.autosize % 16)
|
||||
}
|
||||
}
|
||||
return fi
|
||||
}
|
||||
|
||||
// arm64IsLeaf reports whether a function contains no call instructions
|
||||
// (BL/CALL), matching the toolchain's LEAF mark.
|
||||
func arm64IsLeaf(t *ast.Text) bool {
|
||||
for _, stmt := range t.Body {
|
||||
in, ok := stmt.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
switch strings.ToUpper(in.Mnemonic.Text) {
|
||||
case "BL", "CALL":
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// arm64Prologue returns the prologue bytes for an arm64 function.
|
||||
func arm64Prologue(fi arm64FrameInfo) []byte {
|
||||
if fi.autosize == 0 {
|
||||
return nil
|
||||
}
|
||||
if fi.autosize <= 0xf0 {
|
||||
// Small frame: MOVD.W LR, -autosize(SP); MOVD FP, -8(SP); SUB $8, SP, FP
|
||||
return a64WordsLE(
|
||||
arm64PreStoreImm(3, 0, int32(-fi.autosize), 31, 30), // STR.W LR, -autosize(SP) (pre-index store)
|
||||
arm64UnscaledStore(3, 0, -8, 31, 29), // STUR FP, [SP, #-8]
|
||||
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
|
||||
)
|
||||
}
|
||||
// Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP
|
||||
return a64WordsLE(
|
||||
a64AddSub(1, 1, 0, 0, uint32(fi.autosize), 31, 20), // SUB $autosize, SP, R20
|
||||
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
|
||||
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
|
||||
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
|
||||
)
|
||||
}
|
||||
|
||||
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
|
||||
// deallocate the frame when present) followed by RET (BR LR).
|
||||
func arm64Return(fi arm64FrameInfo) []byte {
|
||||
var ws []uint32
|
||||
if fi.autosize != 0 {
|
||||
if fi.leaf {
|
||||
// Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP
|
||||
ws = append(ws,
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize-8), 31, 29), // ADD $autosize-8, SP, FP
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
|
||||
)
|
||||
} else if fi.autosize <= 0xf0 {
|
||||
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize
|
||||
ws = append(ws,
|
||||
arm64UnscaledLoad(3, 0, -8, 31, 29), // LDR FP, [SP, #-8]
|
||||
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
|
||||
)
|
||||
} else {
|
||||
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
|
||||
ws = append(ws,
|
||||
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
|
||||
)
|
||||
}
|
||||
}
|
||||
// RET: BR LR (0xd65f03c0)
|
||||
ws = append(ws, a64UncondBranch(2, 30, 0)) // opc=2(RET), Rn=LR(30), Rd=0
|
||||
return a64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// arm64PrologueSpadjPC returns the function-relative byte offset where the
|
||||
// prologue has finished decrementing SP (the delta becomes autosize).
|
||||
func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
|
||||
if fi.autosize == 0 {
|
||||
return 0
|
||||
}
|
||||
if fi.autosize <= 0xf0 {
|
||||
return 4 // MOVD.W instruction decrements SP
|
||||
}
|
||||
return 8 // SUB + STP + MOVD (3 instructions, SP updated at the MOVD)
|
||||
}
|
||||
|
||||
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
||||
// (but not including) the final RET instruction.
|
||||
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
|
||||
if fi.autosize == 0 {
|
||||
return 0
|
||||
}
|
||||
if fi.leaf {
|
||||
return 8 // ADD + ADD
|
||||
}
|
||||
if fi.autosize <= 0xf0 {
|
||||
return 8 // LDR + LDR.P
|
||||
}
|
||||
return 8 // LDP + ADD
|
||||
}
|
||||
|
||||
// arm64ResolvePseudo translates a pseudo-register memory reference into a
|
||||
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
|
||||
// x+N(SP) → (N + frame + 8)(SP). Returns base = -1 for an unresolvable
|
||||
// reference (SB: static data, handled by the relocation path).
|
||||
//
|
||||
// The Go toolchain resolves all pseudo-register references against the
|
||||
// hardware stack pointer (R31/SP): FP references add autosize+8 (the
|
||||
// distance from SP after the prologue to the caller's argument area),
|
||||
// SP references add frame+8 (the distance to the local area).
|
||||
func arm64ResolvePseudo(sym *ast.Symbol, fi arm64FrameInfo) (base int, off int32) {
|
||||
if sym == nil {
|
||||
return -1, 0
|
||||
}
|
||||
switch sym.Pseudo {
|
||||
case "FP":
|
||||
return 31, int32(sym.Offset) + int32(fi.autosize) + 8
|
||||
case "SP":
|
||||
return 31, int32(sym.Offset) + int32(fi.frame) + 8
|
||||
case "SB":
|
||||
return -1, int32(sym.Offset)
|
||||
}
|
||||
return -1, 0
|
||||
}
|
||||
|
||||
// arm64PreStoreImm encodes a pre-index store (STR with writeback):
|
||||
// size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
|
||||
func arm64PreStoreImm(size, V int, imm9 int32, rn, rt int) uint32 {
|
||||
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 |
|
||||
3<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
||||
}
|
||||
|
||||
// arm64UnscaledStore encodes an unscaled store (STUR):
|
||||
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt.
|
||||
func arm64UnscaledStore(size, V int, imm9 int32, rn, rt int) uint32 {
|
||||
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 |
|
||||
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
||||
}
|
||||
|
||||
// arm64UnscaledLoad encodes an unscaled load (LDUR):
|
||||
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt.
|
||||
func arm64UnscaledLoad(size, V int, imm9 int32, rn, rt int) uint32 {
|
||||
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
|
||||
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
||||
}
|
||||
|
||||
// arm64PostLoad encodes a post-index load (LDR with post-increment):
|
||||
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
|
||||
func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 {
|
||||
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
|
||||
1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
||||
}
|
||||
+224
@@ -0,0 +1,224 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
)
|
||||
|
||||
// AArch64 ELF64 relocatable object emission.
|
||||
|
||||
const (
|
||||
emAARCH64 = 183 // EM_AARCH64
|
||||
|
||||
// AArch64 relocation types (the ELF psABI).
|
||||
rArm64PrelPgHi21 = 275 // R_AARCH64_ADR_PREL_PG_HI21 (ADRP page)
|
||||
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD/STR/LDR page offset)
|
||||
)
|
||||
|
||||
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
|
||||
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
|
||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
|
||||
// optional .rela.text.
|
||||
func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
const (
|
||||
secText = 1
|
||||
secData = 2
|
||||
)
|
||||
|
||||
// Build symbol table.
|
||||
var locals, globals []elfSym
|
||||
for _, fn := range img.Funcs {
|
||||
s := elfSym{
|
||||
name: objectName(fn.Pkg, fn.Name),
|
||||
info: sttFunc,
|
||||
shndx: secText,
|
||||
value: uint64(fn.Offset),
|
||||
size: uint64(fn.Size),
|
||||
}
|
||||
if fn.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, d := range img.DataSyms {
|
||||
s := elfSym{
|
||||
name: objectName(d.Pkg, d.Name),
|
||||
info: sttObject,
|
||||
shndx: secData,
|
||||
value: uint64(d.Offset),
|
||||
size: uint64(d.Size),
|
||||
}
|
||||
if d.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, name := range img.Externals {
|
||||
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
|
||||
}
|
||||
syms := []elfSym{
|
||||
{},
|
||||
{name: ".text", info: sttSection, shndx: secText},
|
||||
{name: ".data", info: sttSection, shndx: secData},
|
||||
}
|
||||
syms = append(syms, locals...)
|
||||
shInfo := len(syms)
|
||||
syms = append(syms, globals...)
|
||||
symIdx := map[string]int{}
|
||||
for i, s := range syms {
|
||||
symIdx[s.name] = i
|
||||
}
|
||||
|
||||
// Build relocations. Each SB reference is an ADRP pair:
|
||||
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21
|
||||
// ADD/LDR/STR → R_AARCH64_ADD_ABS_LO12_NC
|
||||
type elfRela struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
sym int
|
||||
addend int64
|
||||
}
|
||||
var relas []elfRela
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
typ := uint32(rArm64PrelPgHi21)
|
||||
if r.Kind == RelArm64Addr && r.Off%4 == 4 {
|
||||
// The second instruction in an ADRP pair uses ADD_ABS_LO12_NC.
|
||||
typ = rArm64AddAbsLo12NC
|
||||
}
|
||||
relas = append(relas, elfRela{
|
||||
off: uint64(fn.Offset + r.Off),
|
||||
typ: typ,
|
||||
sym: idx,
|
||||
addend: r.Addend - int64(r.After-r.Off),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// String tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
stNames.add(s.name)
|
||||
}
|
||||
stSections := newElfStrtab()
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Layout.
|
||||
var out []byte
|
||||
out = append(out, make([]byte, 64)...)
|
||||
|
||||
align := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
|
||||
align(16)
|
||||
textOff := len(out)
|
||||
out = append(out, img.Code...)
|
||||
|
||||
align(16)
|
||||
dataOff := len(out)
|
||||
out = append(out, img.Data...)
|
||||
|
||||
align(8)
|
||||
symtabOff := len(out)
|
||||
for _, s := range syms {
|
||||
var b [24]byte
|
||||
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
|
||||
b[4] = s.info
|
||||
b[5] = 0
|
||||
le.PutUint16(b[6:], s.shndx)
|
||||
le.PutUint64(b[8:], s.value)
|
||||
le.PutUint64(b[16:], s.size)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
for _, r := range relas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
|
||||
align(8)
|
||||
shoff := len(out)
|
||||
|
||||
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
|
||||
var b [64]byte
|
||||
le.PutUint32(b[0:], uint32(stSections.at(name)))
|
||||
le.PutUint32(b[4:], uint32(typ))
|
||||
le.PutUint64(b[8:], flags)
|
||||
le.PutUint64(b[16:], 0)
|
||||
le.PutUint64(b[24:], uint64(off))
|
||||
le.PutUint64(b[32:], uint64(size))
|
||||
le.PutUint32(b[40:], uint32(link))
|
||||
le.PutUint32(b[44:], uint32(info))
|
||||
le.PutUint64(b[48:], alignV)
|
||||
le.PutUint64(b[56:], entsize)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
|
||||
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
|
||||
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
|
||||
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
|
||||
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
|
||||
// ELF header.
|
||||
hdr := out[:64]
|
||||
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
|
||||
le.PutUint16(hdr[16:], etREL)
|
||||
le.PutUint16(hdr[18:], emAARCH64)
|
||||
le.PutUint32(hdr[20:], elfVersion)
|
||||
le.PutUint64(hdr[24:], 0)
|
||||
le.PutUint64(hdr[32:], 0)
|
||||
le.PutUint64(hdr[40:], uint64(shoff))
|
||||
le.PutUint32(hdr[48:], 0)
|
||||
le.PutUint16(hdr[52:], 64)
|
||||
le.PutUint16(hdr[54:], 0)
|
||||
le.PutUint16(hdr[56:], 0)
|
||||
le.PutUint16(hdr[58:], 64)
|
||||
le.PutUint16(hdr[60:], uint16(nSections))
|
||||
le.PutUint16(hdr[62:], uint16(secShstr))
|
||||
|
||||
return out, nil
|
||||
}
|
||||
@@ -0,0 +1,142 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"debug/elf"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
|
||||
// relocatable object: sections, the symbol table (bindings, types, values,
|
||||
// sizes) and the .rela.text relocation pair for the static-symbol load,
|
||||
// parsed back with debug/elf.
|
||||
func TestELFAARCH64Object(t *testing.T) {
|
||||
f, errs := parser.Parse("k_arm64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVD a+0(FP), R4
|
||||
MOVD b+8(FP), R5
|
||||
ADD R5, R4, R4
|
||||
MOVD R4, ret+16(FP)
|
||||
RET
|
||||
|
||||
TEXT ·getanswer(SB), NOSPLIT, $0-8
|
||||
MOVD answer<>(SB), R4
|
||||
MOVD R4, ret+0(FP)
|
||||
RET
|
||||
|
||||
GLOBL answer<>(SB), RODATA, $8
|
||||
DATA answer<>+0(SB)/8, $42
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
obj, err := img.ELFAARCH64Object()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFAARCH64Object: %v", err)
|
||||
}
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
|
||||
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_AARCH64 {
|
||||
t.Errorf("type/machine = %v/%v, want ET_REL/EM_AARCH64", ef.Type, ef.Machine)
|
||||
}
|
||||
|
||||
text := ef.Section(".text")
|
||||
data := ef.Section(".data")
|
||||
if text == nil || data == nil {
|
||||
t.Fatal("missing .text or .data section")
|
||||
}
|
||||
if text.Size == 0 {
|
||||
t.Error(".text section is empty")
|
||||
}
|
||||
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatalf("symbols: %v", err)
|
||||
}
|
||||
|
||||
foundAdd, foundGetanswer, foundAnswer := false, false, false
|
||||
for _, s := range syms {
|
||||
switch s.Name {
|
||||
case "add":
|
||||
foundAdd = true
|
||||
if elf.SymType(s.Info&0xf) != elf.STT_FUNC || elf.SymBind(s.Info>>4) != elf.STB_GLOBAL {
|
||||
t.Errorf("add: info=0x%02x, want STT_FUNC|STB_GLOBAL", s.Info)
|
||||
}
|
||||
case "getanswer":
|
||||
foundGetanswer = true
|
||||
if elf.SymType(s.Info&0xf) != elf.STT_FUNC || elf.SymBind(s.Info>>4) != elf.STB_GLOBAL {
|
||||
t.Errorf("getanswer: info=0x%02x, want STT_FUNC|STB_GLOBAL", s.Info)
|
||||
}
|
||||
case "answer":
|
||||
foundAnswer = true
|
||||
if elf.SymType(s.Info&0xf) != elf.STT_OBJECT || elf.SymBind(s.Info>>4) != elf.STB_LOCAL {
|
||||
t.Errorf("answer: info=0x%02x, want STT_OBJECT|STB_LOCAL", s.Info)
|
||||
}
|
||||
}
|
||||
}
|
||||
if !foundAdd {
|
||||
t.Error("symbol 'add' not found")
|
||||
}
|
||||
if !foundGetanswer {
|
||||
t.Error("symbol 'getanswer' not found")
|
||||
}
|
||||
if !foundAnswer {
|
||||
t.Error("symbol 'answer' not found")
|
||||
}
|
||||
|
||||
// Check that .rela.text exists (getanswer has SB reference).
|
||||
relaText := ef.Section(".rela.text")
|
||||
if relaText == nil {
|
||||
t.Error("missing .rela.text section")
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFAARCH64ObjectNoRelocations checks the ELF output when there are no
|
||||
// static-symbol references (no .rela.text section).
|
||||
func TestELFAARCH64ObjectNoRelocations(t *testing.T) {
|
||||
f, errs := parser.Parse("k_arm64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVD a+0(FP), R4
|
||||
MOVD b+8(FP), R5
|
||||
ADD R5, R4, R4
|
||||
MOVD R4, ret+16(FP)
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
obj, err := img.ELFAARCH64Object()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFAARCH64Object: %v", err)
|
||||
}
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
|
||||
if ef.Section(".rela.text") != nil {
|
||||
t.Error("unexpected .rela.text section when there are no relocations")
|
||||
}
|
||||
}
|
||||
+42
-3
@@ -10,6 +10,7 @@ import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
)
|
||||
|
||||
@@ -90,12 +91,50 @@ const (
|
||||
)
|
||||
|
||||
// Relocation types (cmd/internal/objabi).
|
||||
// R_PCREL and R_ADDR are stable across Go versions.
|
||||
const (
|
||||
relocPCRel = 14 // R_PCREL
|
||||
relocAddr = 1 // R_ADDR
|
||||
relocDWTXTADDRU4 = 106 // R_DWTXTADDR_U4
|
||||
relocPCRel = 14 // R_PCREL
|
||||
relocAddr = 1 // R_ADDR
|
||||
)
|
||||
|
||||
// relocDWTXTADDRU4 returns the R_DWTXTADDR_U4 relocation type for the
|
||||
// installed Go toolchain. The value shifted between Go 1.26 (103) and
|
||||
// Go 1.27 (106) because new LoongArch relocations were inserted before it.
|
||||
func relocDWTXTADDRU4() uint16 {
|
||||
if isGo127OrLater() {
|
||||
return 106
|
||||
}
|
||||
return 103
|
||||
}
|
||||
|
||||
var (
|
||||
goVersionOnce sync.Once
|
||||
goVersionGT26 bool
|
||||
)
|
||||
|
||||
// isGo127OrLater reports whether the installed Go toolchain is 1.27 or later.
|
||||
func isGo127OrLater() bool {
|
||||
goVersionOnce.Do(func() {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
out, err := exec.Command(goBin, "version").Output()
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
// "go version go1.27rc1 linux/amd64"
|
||||
s := string(out)
|
||||
for _, prefix := range []string{"go version go1.27", "go version go1.28", "go version go1.29", "go version go2."} {
|
||||
if strings.Contains(s, prefix) {
|
||||
goVersionGT26 = true
|
||||
return
|
||||
}
|
||||
}
|
||||
})
|
||||
return goVersionGT26
|
||||
}
|
||||
|
||||
// Special package indices for symbol references.
|
||||
const (
|
||||
pkgIdxNone = 0x7fffffff
|
||||
|
||||
+1
-1
@@ -181,7 +181,7 @@ func goobjDwarfInfo(fn FuncLayout, name string, fnNpIdx int) ([]byte, []goobjRel
|
||||
out = append(out, 0) // end of children
|
||||
|
||||
relocs := []goobjReloc{{
|
||||
off: int32(addrx), siz: 4, typ: relocDWTXTADDRU4,
|
||||
off: int32(addrx), siz: 4, typ: relocDWTXTADDRU4(),
|
||||
pkg: pkgIdxNone, sym: uint32(fnNpIdx),
|
||||
}}
|
||||
return out, relocs
|
||||
|
||||
@@ -169,7 +169,7 @@ func TestGoobjDwarfInfo(t *testing.T) {
|
||||
if !bytes.Equal(die, want) {
|
||||
t.Errorf("DIE = %x, want %x", die, want)
|
||||
}
|
||||
if len(relocs) != 1 || relocs[0].off != 7 || relocs[0].siz != 4 || relocs[0].typ != relocDWTXTADDRU4 || relocs[0].sym != 3 {
|
||||
if len(relocs) != 1 || relocs[0].off != 7 || relocs[0].siz != 4 || relocs[0].typ != relocDWTXTADDRU4() || relocs[0].sym != 3 {
|
||||
t.Errorf("relocs = %+v", relocs)
|
||||
}
|
||||
|
||||
|
||||
+2
-2
@@ -186,7 +186,7 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
||||
t.Errorf("addq lines reloc = %x", lr)
|
||||
}
|
||||
dr := relocs[23:46]
|
||||
if dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4 ||
|
||||
if dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4() ||
|
||||
le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 {
|
||||
t.Errorf("addq DIE reloc = %x", dr)
|
||||
}
|
||||
@@ -379,7 +379,7 @@ func main() {
|
||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module goobjtest\n\ngo 1.26\n"), 0o644); err != nil {
|
||||
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module goobjtest\n\ngo 1.27\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"sync"
|
||||
)
|
||||
|
||||
// GOObjectAARCH64 emits a GOOBJ object file for AArch64. The layout is
|
||||
// the shared one in goobj.go — the toolchain preamble, the go120ld header
|
||||
// with its block offsets, the string table, the symbol definitions and the
|
||||
// reloc/aux/data index arrays — with the arm64 preamble, the MinLC of 4
|
||||
// for the pc-value deltas, and R_ADDRARM64 relocation types for the
|
||||
// ADRP+ADD/LDR/STR address pairs.
|
||||
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
|
||||
pre, err := toolchainObjectPreambleAARCH64()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
|
||||
return relocArm64Addr, 4
|
||||
})
|
||||
}
|
||||
|
||||
// arm64 relocation types (cmd/internal/objabi). R_ADDRARM64 resolves an
|
||||
// ADRP+ADD/LDR/STR pair to a symbol's address.
|
||||
const (
|
||||
relocArm64Addr = 9 // R_ADDRARM64
|
||||
)
|
||||
|
||||
// toolchainObjectPreambleAARCH64 returns the "go object ...\n!\n" header
|
||||
// the installed go tool asm writes for arm64, captured by assembling a
|
||||
// one-instruction probe.
|
||||
var (
|
||||
preambleAARCH64Once sync.Once
|
||||
preambleAARCH64 []byte
|
||||
preambleAARCH64Err error
|
||||
)
|
||||
|
||||
func toolchainObjectPreambleAARCH64() ([]byte, error) {
|
||||
preambleAARCH64Once.Do(func() {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
preambleAARCH64Err = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
|
||||
return
|
||||
}
|
||||
dir, err := os.MkdirTemp("", "gasm-preamble-arm64")
|
||||
if err != nil {
|
||||
preambleAARCH64Err = err
|
||||
return
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
src := filepath.Join(dir, "probe_arm64.s")
|
||||
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
|
||||
preambleAARCH64Err = err
|
||||
return
|
||||
}
|
||||
obj := filepath.Join(dir, "probe.o")
|
||||
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
|
||||
cmd.Env = append(os.Environ(), "GOARCH=arm64")
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
preambleAARCH64Err = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
|
||||
return
|
||||
}
|
||||
data, err := os.ReadFile(obj)
|
||||
if err != nil {
|
||||
preambleAARCH64Err = err
|
||||
return
|
||||
}
|
||||
i := bytes.Index(data, []byte("\n!\n"))
|
||||
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
|
||||
preambleAARCH64Err = fmt.Errorf("unrecognised assembler object layout")
|
||||
return
|
||||
}
|
||||
preambleAARCH64 = data[:i+3]
|
||||
})
|
||||
return preambleAARCH64, preambleAARCH64Err
|
||||
}
|
||||
@@ -153,7 +153,7 @@ DATA ·table<>+0(SB)/8, $0x1122334455667788
|
||||
t.Errorf("lines reloc = %x", lr)
|
||||
}
|
||||
dr := relocs[23:46]
|
||||
if int32(le.Uint32(dr[0:])) != 13 || dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4 ||
|
||||
if int32(le.Uint32(dr[0:])) != 13 || dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4() ||
|
||||
le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 {
|
||||
t.Errorf("die reloc = %x", dr)
|
||||
}
|
||||
|
||||
@@ -96,6 +96,7 @@ const (
|
||||
RelPCRelAbs // 32-bit absolute (R_RISCV_32)
|
||||
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
|
||||
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
|
||||
RelArm64Addr // R_ADDRARM64 (ADRP + ADD/LDR/STR pair)
|
||||
)
|
||||
|
||||
type Reloc struct {
|
||||
|
||||
@@ -498,6 +498,8 @@ requires -p, the package path, and the installed Go toolchain).
|
||||
obj, err = img.ELFRISCVObject()
|
||||
case arch.LOONG64:
|
||||
obj, err = img.ELFLOONG64Object()
|
||||
case arch.ARM64:
|
||||
obj, err = img.ELFAARCH64Object()
|
||||
default:
|
||||
obj, err = img.ELFObject()
|
||||
}
|
||||
@@ -508,6 +510,8 @@ requires -p, the package path, and the installed Go toolchain).
|
||||
obj, err = img.GOObjectRISCV(*pkg, path)
|
||||
case arch.LOONG64:
|
||||
obj, err = img.GOObjectLOONG64(*pkg, path)
|
||||
case arch.ARM64:
|
||||
obj, err = img.GOObjectAARCH64(*pkg, path)
|
||||
default:
|
||||
obj, err = img.GOObject(*pkg, path)
|
||||
}
|
||||
@@ -921,6 +925,95 @@ func cmdVerifyLOONG64(path string, groundTruth, profile bool) int {
|
||||
return 0
|
||||
}
|
||||
|
||||
func cmdVerifyARM64(path string, groundTruth, profile bool) int {
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
f, errs := parser.Parse(path, src)
|
||||
for _, e := range errs {
|
||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
||||
}
|
||||
if len(errs) > 0 {
|
||||
return 1
|
||||
}
|
||||
img, err := asm.AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
if groundTruth {
|
||||
gt, err := verify.GroundTruthARM64(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm verify: ground truth: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
matched, total := 0, 0
|
||||
for _, fn := range img.Funcs {
|
||||
gasmCode := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
goCode, ok := gt[fn.Name]
|
||||
if !ok {
|
||||
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", fn.Name)
|
||||
continue
|
||||
}
|
||||
total++
|
||||
gasmCmp := make([]byte, len(gasmCode))
|
||||
goCmp := make([]byte, len(goCode))
|
||||
copy(gasmCmp, gasmCode)
|
||||
copy(goCmp, goCode)
|
||||
for _, r := range fn.Relocs {
|
||||
for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ {
|
||||
gasmCmp[j] = 0
|
||||
}
|
||||
for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ {
|
||||
goCmp[j] = 0
|
||||
}
|
||||
}
|
||||
if bytes.Equal(gasmCmp, goCmp) {
|
||||
matched++
|
||||
if len(fn.Relocs) > 0 {
|
||||
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
|
||||
} else {
|
||||
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
|
||||
}
|
||||
} else {
|
||||
fmt.Printf(" %s: MISMATCH (%d vs %d bytes)\n", fn.Name, fn.Size, len(goCode))
|
||||
for i := 0; i < len(gasmCode) || i < len(goCode); i += 16 {
|
||||
var gb, gs string
|
||||
for j := i; j < i+16 && j < len(gasmCode); j++ {
|
||||
gb += fmt.Sprintf(" %02x", gasmCode[j])
|
||||
}
|
||||
for j := i; j < i+16 && j < len(goCode); j++ {
|
||||
gs += fmt.Sprintf(" %02x", goCode[j])
|
||||
}
|
||||
fmt.Printf(" %04x: gasm:%s\n", i, gb)
|
||||
fmt.Printf(" %04x: gt: %s\n", i, gs)
|
||||
}
|
||||
}
|
||||
}
|
||||
fmt.Printf("%s: %d/%d matched\n", path, matched, total)
|
||||
if matched < total {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
if profile {
|
||||
for _, fn := range img.Funcs {
|
||||
fmt.Printf("%s: %d bytes, labels: %v\n", fn.Name, fn.Size, fn.Labels)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
fmt.Printf("%s: %d functions assembled\n", path, len(img.Funcs))
|
||||
for _, fn := range img.Funcs {
|
||||
fmt.Printf(" %s: %d bytes\n", fn.Name, fn.Size)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func cmdVerify(args []string) int {
|
||||
fs := newCommand("verify", "gasm verify [-smoke] [-abi] [-fuzz] [-ground-truth] [-profile] [-call] <file.s>", `
|
||||
Assemble FILE (amd64), map it into executable memory and report the available
|
||||
@@ -975,6 +1068,9 @@ decoders) that crash on random input but should succeed on valid data.
|
||||
case arch.LOONG64:
|
||||
// LoongArch: ground-truth only (no JIT on non-LoongArch hosts).
|
||||
return cmdVerifyLOONG64(path, *groundTruth, *profile)
|
||||
case arch.ARM64:
|
||||
// AArch64: ground-truth only (no JIT on non-ARM64 hosts).
|
||||
return cmdVerifyARM64(path, *groundTruth, *profile)
|
||||
default:
|
||||
fmt.Fprintln(os.Stderr, "gasm verify: only amd64, riscv64 and loong64 are supported")
|
||||
return 1
|
||||
|
||||
@@ -212,6 +212,17 @@ relocations). Like the RISC-V encoder it is validated byte-for-byte against
|
||||
`GOARCH=loong64 go tool asm`, and its GOOBJ output is proven end-to-end by
|
||||
substituting it into a cross-compiled `go build` and linking with `cmd/link`.
|
||||
|
||||
An **AArch64 encoder** (Phase 5, arm64) encodes the integer instruction set
|
||||
with the data-processing (shifted register and immediate forms), load/store
|
||||
(scaled unsigned immediate and unscaled9-bit immediate), conditional and
|
||||
unconditional branches, the MOV pseudo-instruction and its constant
|
||||
materialisation (MOVZ/MOVN/MOVK for wide immediates, ORR with logical bitmask
|
||||
encoding for values like `$1`), the FP/SP frame mapping (autosize =
|
||||
align16(frame+8), prologue using pre-index store for small frames and
|
||||
STP+SUB for large frames) and SB/global symbol references (ADRP+ADD pairs with
|
||||
R_ADDRARM64 relocations). Like the other encoders it is validated
|
||||
byte-for-byte against `GOARCH=arm64 go tool asm`.
|
||||
|
||||
On top of the encoder, `Assemble` walks a parsed `TEXT` body, converts each
|
||||
operand to an encoder operand, and lays the instructions out so local labels
|
||||
resolve to relative jump offsets: jumps start in the short (rel8) form and
|
||||
|
||||
@@ -4,7 +4,7 @@ Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrb
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- **Go** 1.26+ with `toolchain go1.26.5`
|
||||
- **Go** 1.27+ with `toolchain go1.27.0`
|
||||
- **just** — the command runner; every task below is a just recipe
|
||||
- No external dependencies beyond the Go toolchain
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
module sourcedock.dev/petrbalvin/gasm-devkit
|
||||
|
||||
go 1.26
|
||||
go 1.27
|
||||
|
||||
toolchain go1.26.5
|
||||
toolchain go1.27.0
|
||||
|
||||
require golang.org/x/arch v0.29.0
|
||||
|
||||
Vendored
+57
@@ -0,0 +1,57 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// add returns a + b.
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVD a+0(FP), R4
|
||||
MOVD b+8(FP), R5
|
||||
ADD R5, R4, R4
|
||||
MOVD R4, ret+16(FP)
|
||||
RET
|
||||
|
||||
// arith exercises the register-register integer set.
|
||||
TEXT ·arith(SB), NOSPLIT, $0-0
|
||||
ADD R4, R5, R6
|
||||
SUB R7, R8, R9
|
||||
AND R10, R11, R12
|
||||
ORR R12, R13, R14
|
||||
EOR R14, R15, R16
|
||||
CMP R16, R17
|
||||
ADD R4, R5
|
||||
SUB R6, R7
|
||||
RET
|
||||
|
||||
// branch exercises conditional and unconditional control flow.
|
||||
TEXT ·branch(SB), NOSPLIT, $0-0
|
||||
BEQ done
|
||||
BNE skip
|
||||
BGE done
|
||||
BLT done
|
||||
BGT done
|
||||
BLE done
|
||||
skip:
|
||||
B loop
|
||||
loop:
|
||||
ADD R4, R5
|
||||
RET
|
||||
done:
|
||||
RET
|
||||
|
||||
// mov exercises the MOV pseudo-instruction.
|
||||
TEXT ·mov(SB), NOSPLIT, $0-16
|
||||
MOVD $0, R4
|
||||
MOVD $1, R5
|
||||
MOVD $42, R6
|
||||
MOVD a+0(FP), R7
|
||||
MOVD R7, ret+0(FP)
|
||||
MOVW $100, R8
|
||||
RET
|
||||
|
||||
// frame exercises the prologue/epilogue of a function with a real frame.
|
||||
TEXT ·frame(SB), NOSPLIT, $32-8
|
||||
MOVD arg+0(FP), R4
|
||||
ADD $1, R4, R4
|
||||
MOVD R4, ret+0(FP)
|
||||
RET
|
||||
Vendored
+27
@@ -0,0 +1,27 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// branch exercises all conditional branch forms and jump chain folding.
|
||||
TEXT ·branch(SB), NOSPLIT, $0-0
|
||||
BEQ done
|
||||
BNE skip
|
||||
BGE done
|
||||
BLT done
|
||||
BGT done
|
||||
BLE done
|
||||
BCS done
|
||||
BCC done
|
||||
BMI done
|
||||
BPL done
|
||||
BVS done
|
||||
BVC done
|
||||
BHI done
|
||||
BLS done
|
||||
skip:
|
||||
B loop
|
||||
loop:
|
||||
ADD R4, R5
|
||||
done:
|
||||
RET
|
||||
Vendored
+10
@@ -0,0 +1,10 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// caller exercises BL to an external symbol (produces a relocation).
|
||||
TEXT ·caller(SB), NOSPLIT, $0-0
|
||||
BL other(SB)
|
||||
ADD R4, R5
|
||||
RET
|
||||
Vendored
+119
@@ -0,0 +1,119 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// fparith exercises the FP arithmetic set.
|
||||
TEXT ·fparith(SB), NOSPLIT, $0-0
|
||||
FADDD F0, F1, F2
|
||||
FSUBD F3, F4, F5
|
||||
FMULD F6, F7, F8
|
||||
FDIVD F9, F10, F11
|
||||
FADDS F12, F13, F14
|
||||
FSUBS F15, F16, F17
|
||||
FMULS F18, F19, F20
|
||||
FDIVS F21, F22, F23
|
||||
FSQRTD F24, F25
|
||||
FSQRTS F26, F27
|
||||
FNEGD F28, F29
|
||||
FNEGS F30, F31
|
||||
FABSD F0, F1
|
||||
FABSS F2, F3
|
||||
FNMULD F4, F5, F6
|
||||
FNMULS F7, F8, F9
|
||||
FMIND F10, F11, F12
|
||||
FMAXD F13, F14, F15
|
||||
FMINS F16, F17, F18
|
||||
FMAXS F19, F20, F21
|
||||
RET
|
||||
|
||||
// fpfma exercises fused multiply-add.
|
||||
TEXT ·fpfma(SB), NOSPLIT, $0-0
|
||||
FMADDD F0, F1, F2, F3
|
||||
FMSUBD F4, F5, F6, F7
|
||||
FNMADDD F8, F9, F10, F11
|
||||
FNMSUBD F12, F13, F14, F15
|
||||
FMADDS F16, F17, F18, F19
|
||||
FMSUBS F20, F21, F22, F23
|
||||
FNMADDS F24, F25, F26, F27
|
||||
FNMSUBS F28, F29, F30, F0
|
||||
RET
|
||||
|
||||
// fpconv exercises FP↔integer conversion and cross-precision.
|
||||
// Syntax: FCVTZSD Fd, Rn (float→int: FP source first, int dest second)
|
||||
// SCVTFD Rn, Fd (int→float: int source first, FP dest second)
|
||||
TEXT ·fpconv(SB), NOSPLIT, $0-0
|
||||
FCVTSD F0, F1
|
||||
FCVTDS F2, F3
|
||||
FCVTZSD F4, R0
|
||||
FCVTZSS F5, R1
|
||||
FCVTZUD F6, R2
|
||||
FCVTZUS F7, R3
|
||||
SCVTFD R4, F8
|
||||
SCVTFS R5, F9
|
||||
UCVTFD R6, F10
|
||||
UCVTFS R7, F11
|
||||
SCVTFWD R0, F12
|
||||
SCVTFWS R1, F13
|
||||
UCVTFWD R2, F14
|
||||
UCVTFWS R3, F15
|
||||
FMOVS F14, R20
|
||||
FMOVS R21, F15
|
||||
FMOVD F16, R22
|
||||
FMOVD R23, F17
|
||||
RET
|
||||
|
||||
// fpcmp exercises FP compare and conditional compare.
|
||||
// FCCMP syntax: FCCMP cond, Fn, Fm, $nzcv
|
||||
// FCSEL syntax: FCSEL cond, Fn, Fm, Fd
|
||||
TEXT ·fpcmp(SB), NOSPLIT, $0-0
|
||||
FCMPS F0, F1
|
||||
FCMPD F2, F3
|
||||
FCMPS $0.0, F4
|
||||
FCMPD $0.0, F5
|
||||
FCCMPS EQ, F6, F7, $0
|
||||
FCCMPD NE, F8, F9, $0
|
||||
FCSELS GE, F10, F11, F12
|
||||
FCSELD LT, F13, F14, F15
|
||||
RET
|
||||
|
||||
// frint exercises FP rounding.
|
||||
TEXT ·frint(SB), NOSPLIT, $0-0
|
||||
FRINTND F0, F1
|
||||
FRINTNS F2, F3
|
||||
FRINTPD F4, F5
|
||||
FRINTPS F6, F7
|
||||
FRINTMD F8, F9
|
||||
FRINTMS F10, F11
|
||||
FRINTZD F12, F13
|
||||
FRINTZS F14, F15
|
||||
FRINTAD F16, F17
|
||||
FRINTAS F18, F19
|
||||
FRINTXD F20, F21
|
||||
FRINTXS F22, F23
|
||||
FRINTID F24, F25
|
||||
FRINTIS F26, F27
|
||||
FMOVD F0, F1
|
||||
FMOVS F2, F3
|
||||
RET
|
||||
|
||||
// condsel exercises conditional select and CRC32.
|
||||
TEXT ·condsel(SB), NOSPLIT, $0-0
|
||||
CSEL EQ, R0, R1, R2
|
||||
CSINC NE, R3, R4, R5
|
||||
CSINV GE, R6, R7, R8
|
||||
CSNEG LT, R9, R10, R11
|
||||
CSET EQ, R12
|
||||
CSETM NE, R13
|
||||
CINC EQ, R14, R15
|
||||
CINV NE, R16, R17
|
||||
CNEG GE, R19, R20
|
||||
CRC32B R0, R2
|
||||
CRC32H R3, R5
|
||||
CRC32W R6, R8
|
||||
CRC32X R9, R11
|
||||
CRC32CB R12, R14
|
||||
CRC32CH R15, R0
|
||||
CRC32CW R1, R3
|
||||
CRC32CX R4, R6
|
||||
RET
|
||||
Vendored
+21
@@ -0,0 +1,21 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// movimm exercises MOV with various immediate values.
|
||||
TEXT ·movimm(SB), NOSPLIT, $0-0
|
||||
MOVD $0, R0
|
||||
MOVD $1, R1
|
||||
MOVD $42, R2
|
||||
MOVD $255, R3
|
||||
MOVD $256, R4
|
||||
MOVD $0xFFFF, R5
|
||||
MOVD $0x12345678, R6
|
||||
MOVD $0x123456789ABCDEF0, R7
|
||||
MOVD $-1, R8
|
||||
MOVD $-2, R9
|
||||
MOVW $0, R10
|
||||
MOVW $100, R11
|
||||
MOVW $0x12345, R12
|
||||
RET
|
||||
@@ -0,0 +1,87 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"os"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestGroundTruthARM64 assembles the arm64 test kernels with gasm and
|
||||
// compares them byte-for-byte against `go tool asm` (GOARCH=arm64). The
|
||||
// relocation fields of static-symbol references are masked before the
|
||||
// comparison, since the toolchain leaves them zero for the linker.
|
||||
func TestGroundTruthARM64(t *testing.T) {
|
||||
for _, path := range []string{
|
||||
"../testdata/verify/basic_arm64.s",
|
||||
"../testdata/verify/fp_arm64.s",
|
||||
"../testdata/verify/movimm_arm64.s",
|
||||
"../testdata/verify/branch_arm64.s",
|
||||
"../testdata/verify/call_arm64.s",
|
||||
} {
|
||||
t.Run(path, func(t *testing.T) {
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("read: %v", err)
|
||||
}
|
||||
f, errs := parser.Parse(path, string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := asm.AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
gt, err := GroundTruthARM64(path)
|
||||
if err != nil {
|
||||
t.Fatalf("GroundTruthARM64: %v", err)
|
||||
}
|
||||
|
||||
matched := 0
|
||||
for _, fn := range img.Funcs {
|
||||
gasmCode := maskRelocs(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
|
||||
goCode, ok := gt[fn.Name]
|
||||
if !ok {
|
||||
t.Errorf("%s: not in ground truth (%d functions)", fn.Name, len(gt))
|
||||
continue
|
||||
}
|
||||
goCode = maskRelocs(goCode, fn.Relocs)
|
||||
// The Go toolchain may add zero padding at the end of
|
||||
// functions. Compare up to the shorter length, then
|
||||
// verify any trailing bytes are zero.
|
||||
cmpLen := len(gasmCode)
|
||||
if len(goCode) < cmpLen {
|
||||
cmpLen = len(goCode)
|
||||
}
|
||||
if !bytes.Equal(gasmCode[:cmpLen], goCode[:cmpLen]) {
|
||||
t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\n%s", fn.Name, len(gasmCode), len(goCode), diffHex(gasmCode, goCode))
|
||||
continue
|
||||
}
|
||||
// Check trailing padding is zero.
|
||||
trailingOK := true
|
||||
if len(goCode) > len(gasmCode) {
|
||||
for _, b := range goCode[len(gasmCode):] {
|
||||
if b != 0 {
|
||||
trailingOK = false
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
if !trailingOK {
|
||||
t.Errorf("%s: non-zero trailing bytes in go tool asm output", fn.Name)
|
||||
continue
|
||||
}
|
||||
matched++
|
||||
t.Logf("%s: MATCH (%d bytes, go=%d)", fn.Name, fn.Size, len(goCode))
|
||||
}
|
||||
if matched == 0 {
|
||||
t.Fatal("no functions matched")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -38,6 +38,12 @@ func GroundTruthLOONG64(path string) (map[string][]byte, error) {
|
||||
return groundTruthArch(path, "loong64")
|
||||
}
|
||||
|
||||
// GroundTruthARM64 assembles the given .s file with the Go toolchain in
|
||||
// AArch64 cross-assembly mode (GOARCH=arm64).
|
||||
func GroundTruthARM64(path string) (map[string][]byte, error) {
|
||||
return groundTruthArch(path, "arm64")
|
||||
}
|
||||
|
||||
func groundTruthArch(path, goarch string) (map[string][]byte, error) {
|
||||
goroot := runtime.GOROOT()
|
||||
asmBin := filepath.Join(goroot, "pkg", "tool", runtime.GOOS+"_"+runtime.GOARCH, "asm")
|
||||
@@ -58,6 +64,7 @@ func groundTruthArch(path, goarch string) (map[string][]byte, error) {
|
||||
pkg = strings.TrimSuffix(pkg, "_amd64")
|
||||
pkg = strings.TrimSuffix(pkg, "_riscv64")
|
||||
pkg = strings.TrimSuffix(pkg, "_loong64")
|
||||
pkg = strings.TrimSuffix(pkg, "_arm64")
|
||||
|
||||
cmd := exec.Command(asmBin, "-I", includeDir, "-p", pkg, "-o", objPath, path)
|
||||
if goarch != "" {
|
||||
|
||||
Reference in New Issue
Block a user