Compare commits

...
85 Commits
Author SHA1 Message Date
petrbalvin fff9f75595 chore: prepare release v0.33.0
Release / build (amd64, linux) (push) Successful in 49s
Release / build (arm64, linux) (push) Successful in 43s
Release / build (loong64, linux) (push) Successful in 46s
Release / build (riscv64, linux) (push) Successful in 45s
Test / vet (push) Successful in 47s
Release / release (push) Successful in 18s
Test / test (push) Successful in 2m39s
Test / build (push) Successful in 43s
2026-09-14 23:36:19 +02:00
petrbalvin 40476546df fix(asm): close the oracle parity gaps in frame addressing and calls 2026-09-14 23:25:14 +02:00
petrbalvin 70218e84ba feat(asm): emit the loong64 stack-split guard for big frames 2026-09-14 22:38:58 +02:00
petrbalvin db50b98179 feat(asm): emit the loong64 stack-split guard for small and medium frames 2026-09-14 21:21:58 +02:00
petrbalvin 2e2c0b82a0 feat(asm): emit the riscv64 stack-split guard and fix large-frame addressing 2026-09-14 21:09:13 +02:00
petrbalvin 8dc1e98ca1 feat(asm): emit the arm64 stack-split guard and morestack block 2026-09-14 20:49:03 +02:00
petrbalvin 1d8e68c574 feat(asm): emit the amd64 stack-split guard and morestack block 2026-09-14 20:35:55 +02:00
petrbalvin 89fa6ea15e feat(lsp): resolve definition and references across open documents 2026-09-14 18:50:55 +02:00
petrbalvin edc20ffa97 feat(cmd): add gasm dis and share the decoder with the debugger 2026-09-14 18:47:08 +02:00
petrbalvin 50db6615b2 feat(cmd): add gofmt-style -l and -d modes to gasm fmt 2026-09-14 18:47:08 +02:00
petrbalvin 95f1d6f083 style: replace em dashes in the scaffold comments 2026-09-14 18:22:25 +02:00
petrbalvin f43e791e5a chore: add .qwen to the gitignore metadata block 2026-09-14 18:22:18 +02:00
petrbalvin 1691c81095 style: replace em and en dashes across sources 2026-09-14 18:22:18 +02:00
petrbalvin 2db563be07 refactor(cmd): consolidate cross-arch verify and drop dead code 2026-09-14 18:22:00 +02:00
petrbalvin 4f190ee1a2 refactor(debug): move watchpoint slot state into the session 2026-09-14 18:22:00 +02:00
petrbalvin 909f874797 fix(lsp): recover from handler panics and decode client uris 2026-09-14 18:22:00 +02:00
petrbalvin e307bf830f fix(lint): guard unnamed TEXT and refresh the textflag table 2026-09-14 18:22:00 +02:00
petrbalvin 953c258d6a fix(asm): make arm64 and loong64 relocations match the toolchain 2026-09-14 18:22:00 +02:00
petrbalvin c6f0286732 fix(asm): encode amd64 frame adjustments above 127 bytes with imm32 2026-09-14 18:22:00 +02:00
petrbalvin f5fc22d390 fix(parser): reject malformed TEXT frames and parse signed frame sizes 2026-09-14 18:22:00 +02:00
petrbalvin 22226d59a5 chore: prepare release v0.32.0
Release / build (amd64, linux) (push) Successful in 43s
Release / build (arm64, linux) (push) Successful in 42s
Release / build (loong64, linux) (push) Successful in 43s
Release / build (riscv64, linux) (push) Successful in 43s
Test / vet (push) Successful in 49s
Release / release (push) Successful in 18s
Test / test (push) Successful in 2m40s
Test / build (push) Successful in 42s
Assisted-by: GLM 5.3 Flash
2026-08-31 13:10:37 +02:00
petrbalvin a0ae0e4e37 docs: audit the development changelog against the release delta
Assisted-by: GLM 5.3 Flash
2026-08-31 12:53:36 +02:00
petrbalvin 94c4756d47 fix(verify): fix non-amd64 JIT trampolines and validate under qemu 2026-08-31 12:34:50 +02:00
petrbalvin a5a59d6503 fix(verify): gate JIT verification to amd64 until trampolines are hardened
Test / vet (push) Successful in 48s
Test / test (push) Successful in 2m34s
Test / build (push) Successful in 41s
2026-08-30 22:48:48 +02:00
petrbalvin 9cb1666b35 fix(debug): make ptrace sessions reliable on Go tracees 2026-08-30 22:35:22 +02:00
petrbalvin 96cc70731f docs: sync rule counts and feature lists with the new capabilities
Assisted-by: GLM 5.3 Flash
2026-08-30 21:42:12 +02:00
petrbalvin b4da13d0f6 feat(lsp): pull diagnostics, include links and folding ranges
Assisted-by: GLM 5.3 Flash
2026-08-30 21:42:12 +02:00
petrbalvin 41b387e54d feat(debug): instruction-level coverage and FP register display
Assisted-by: GLM 5.3 Flash
2026-08-30 21:42:12 +02:00
petrbalvin 4171e412b5 feat(verify): save and replay fuzz corpora
Assisted-by: GLM 5.3 Flash
2026-08-30 21:42:12 +02:00
petrbalvin 57c0ca8b09 feat(gasm): audit-instructions for arm64, riscv64 and loong64
Assisted-by: GLM 5.3 Flash
2026-08-30 21:42:12 +02:00
petrbalvin 75bd83fd52 feat(lint): flag writes to the platform-reserved register 2026-08-30 21:42:12 +02:00
petrbalvin 62f6fb4faf fix(debug): cross-compile for arm64, riscv64 and loong64
Test / vet (push) Successful in 47s
Test / test (push) Successful in 2m35s
Test / build (push) Successful in 40s
Assisted-by: GLM 5.3 Flash
2026-08-30 11:27:54 +02:00
petrbalvin 8f84dac10b feat(verify): ABI checks on arm64, riscv64 and loong64
Assisted-by: GLM 5.3 Flash
2026-08-30 11:27:54 +02:00
petrbalvin 6d7f10f13e refactor(cmd): re-enter child modes via environment instead of hidden flags
Assisted-by: GLM 5.3 Flash
2026-08-30 11:00:40 +02:00
petrbalvin 56f8babbce docs: sync README, CHANGELOG and docs with the current state 2026-08-30 10:44:18 +02:00
petrbalvin 6c1c8d9d96 fix: point the coverage gate at the format package 2026-08-30 10:43:55 +02:00
petrbalvin 93794c02f7 chore: untrack .idea files 2026-08-30 10:43:45 +02:00
petrbalvin 56ad158772 fix: restore iota blocks, asm --format flag and prose after the syntax pass
Test / vet (push) Successful in 48s
Test / test (push) Successful in 2m35s
Test / build (push) Successful in 40s
2026-08-29 17:12:53 +02:00
petrbalvin 15e8b88d32 style: modernize the new tooling code to match the repo conventions 2026-08-29 16:15:14 +02:00
petrbalvin 9beff4ae85 style: modernize to splitseq, cut, min, maps.copy and range-over-int 2026-08-29 16:04:32 +02:00
petrbalvin eacf33d0f7 fix: staticcheck and deadcode findings repo-wide, modernize counting loops 2026-08-29 15:25:15 +02:00
petrbalvin 9a34733615 style(parser): cutprefix, default case, comma and tokens rename 2026-08-29 15:14:49 +02:00
petrbalvin 970df7c32a style(lint): drop duplicate rule code declaration 2026-08-29 15:14:49 +02:00
petrbalvin 49572efe16 style: gofmt the audit command 2026-08-29 14:25:02 +02:00
petrbalvin 568c553986 docs: document the audit, scaffold, scalar-args and headless debug additions 2026-08-29 14:24:52 +02:00
petrbalvin 7a30e902fc feat(debug): label coverage report for headless runs 2026-08-29 14:17:26 +02:00
petrbalvin cb398d9498 feat(debug): headless script mode with timeout watchdog 2026-08-29 14:16:04 +02:00
petrbalvin e44162a749 feat(verify): scalar arguments for -call invocations 2026-08-29 14:03:45 +02:00
petrbalvin 6fb9629ab6 fix(gasm): scaffold shared param names and two-sided seed sets 2026-08-29 13:57:04 +02:00
petrbalvin 8bda4066e3 feat(gasm): audit-instructions command and scaffold generator 2026-08-29 13:52:32 +02:00
petrbalvin 685b150ecf feat(lint): flag table-known instructions the encoder cannot emit 2026-08-29 13:42:47 +02:00
petrbalvin c92e6bed3a feat(lint): nonportable amd64 register name rule 2026-08-29 13:39:22 +02:00
petrbalvin d75e6bcae6 feat(lint): abi0 register-args rule and go-asm width model 2026-08-29 13:36:28 +02:00
petrbalvin 6699ebd34f feat(asm): add prefetch hint encoding
Test / vet (push) Successful in 51s
Test / test (push) Successful in 2m37s
Test / build (push) Successful in 41s
2026-08-29 10:42:15 +02:00
petrbalvin ba4d961b20 fix(asm): ADDQ imm8 frame adjust for 128-255 byte frames 2026-08-28 22:11:24 +02:00
petrbalvin 78b12dd427 fix(asm): match go tool asm encodings and strictness 2026-08-28 19:55:25 +02:00
petrbalvin 19a26e049b feat(asm): add vpcmp compare, full opmask set, legacy sse integers and bswap
Test / vet (push) Successful in 47s
Test / test (push) Successful in 2m35s
Test / build (push) Successful in 41s
Assisted-by: GLM 5.3
2026-08-27 22:41:03 +02:00
petrbalvin 163480e283 feat(amd64): encode scalar/double conversion ops (CVTSS2SD/CVTSD2SS/CVTPS2PD/CVTPD2PS) 2026-08-27 22:02:11 +02:00
petrbalvin 0d62818db7 feat(amd64): encode legacy SSE binaries, imm8 shuffles and MOVQ xmm moves 2026-08-27 22:02:11 +02:00
petrbalvin be2ceaafb9 feat(amd64): encode legacy SSE packed binaries and imm8 shuffles
Test / vet (push) Successful in 47s
Test / test (push) Successful in 2m34s
Test / build (push) Successful in 41s
2026-08-27 17:15:10 +02:00
petrbalvin 941f7fa990 chore: set development version to 0.32.0-dev
Test / vet (push) Successful in 48s
Test / test (push) Successful in 3m2s
Test / build (push) Successful in 47s
Assisted-by: GLM 5.3
2026-08-24 21:03:30 +02:00
petrbalvin eb3c79c3c8 fix(verify): isolate smoke and abi sweeps in a child process
Assisted-by: GLM 5.3
2026-08-24 21:01:32 +02:00
petrbalvin eade875b53 fix(asm): compress movq immediates to the go-tool-asm imm32 forms
Assisted-by: GLM 5.3
2026-08-24 20:23:39 +02:00
petrbalvin b08005753e fix(asm): encode BSF, BSR and POPCNT
Assisted-by: GLM 5.3
2026-08-24 20:19:34 +02:00
petrbalvin 5e06d6a6aa feat(lint): add register-width-mismatch rule
Test / vet (push) Successful in 48s
Test / test (push) Successful in 2m31s
Test / build (push) Successful in 44s
Assisted-by: MiMo V2.5 Pro
2026-08-21 01:20:38 +02:00
petrbalvin e4c9d78968 feat(lsp): add workspace symbol search
Assisted-by: MiMo V2.5 Pro
2026-08-21 01:20:35 +02:00
petrbalvin f5fcaf9fa6 perf(verify): parallelize smoke and ABI checks
Assisted-by: MiMo V2.5 Pro
2026-08-21 01:20:31 +02:00
petrbalvin 94b98468bb feat(debug): support register-register conditional breakpoints
Assisted-by: MiMo V2.5 Pro
2026-08-21 01:20:25 +02:00
petrbalvin 746cef8100 feat(asm): add .debug_frame CFI section for stack unwinding
Assisted-by: MiMo V2.5 Pro
2026-08-21 01:20:21 +02:00
petrbalvin f153be8158 feat(lsp): add code actions, signature help, and document highlights
Assisted-by: MiMo V2.5 Pro
2026-08-21 01:20:14 +02:00
petrbalvin 1bf95a169e fix: resolve audit findings — stale text, dead code, build tags, docs
Test / vet (push) Successful in 47s
Test / test (push) Successful in 2m39s
Test / build (push) Successful in 42s
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:50:31 +02:00
petrbalvin f9eb4021d6 docs: update CHANGELOG and README for development changes
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:39:38 +02:00
petrbalvin c4930438fd refactor(lsp): use format package for document formatting
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin f372e2db75 test(lsp): add tests for references, rename, formatting, inlay hints
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin ac02c83a86 feat(lint): add stack-imbalance rule
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin ce5ec24fa8 feat(asm): integrate DWARF5 sections into all ELF emitters
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin 181d8e508c feat(verify): add Call trampolines for arm64, riscv64, loong64
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin 1160c96427 chore: remove stale debug_linux_amd64.go
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin de9e211ff1 feat(debug): multi-architecture debugger support for arm64, riscv64, loong64
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin 3acbdd6533 feat(asm): add DWARF5 debug info generation for ELF output
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin ba502c9b79 refactor(debug): make Regs and breakpoint arch-neutral for arm64
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin 874e054ecb feat(lsp): add references, rename, formatting, and inlay hints
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin f1960febdc feat(lint): add unused-label and invalid-textflag rules
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin ae550cc05a fix(asm): add cross-package GOOBJ resolution for riscv64, loong64, arm64
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin cc50035375 fix(cmd): update asm help text to list arm64 as supported
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
177 changed files with 13623 additions and 2805 deletions
+11
View File
@@ -1,3 +1,9 @@
# Metadata (always first, per repo convention)
.idea/
.zcode/
.qwen/
.mimocode/
# Binaries # Binaries
/gasm /gasm
/bin/ /bin/
@@ -7,6 +13,11 @@
coverage.out coverage.out
*.test *.test
# Crash dumps
core
core.*
*.core
# Scratch / temporary work # Scratch / temporary work
_scratch/ _scratch/
-119
View File
@@ -1,119 +0,0 @@
# AGENTS.md — gasm-devkit
Repository rules for AI agents and contributors. Read before modifying any
code in this repository.
## AI Contribution Policy
AI agents may assist with code, documentation, tests, and review in this
repository. All AI-assisted changes must:
- Follow the code style and conventions in this file.
- Include the trailer `Assisted-by: <model-name>` in every commit message.
- Not commit directly to `main` — work on `development`.
- Pass the full Definition of Done before any commit.
## Workflow
- **Branching.** `development` is the working branch. `main` is
release-only: merge from `development`, then tag. Never commit directly
to `main`.
- **Release procedure.**
1. Bump `version` in `justfile` and `cmd/gasm/main.go`.
2. Update `CHANGELOG.md` with a new `## [X.Y.Z] — YYYY-MM-DD` section.
3. Update `README.md` and `docs/ARCHITECTURE.md` if user-visible
behaviour changed.
4. Run the Definition of Done (below).
5. Commit on `development`.
6. `git checkout main && git merge --ff-only development`.
7. `git tag vX.Y.Z`.
8. `git checkout development`.
9. `GOBIN=~/.local/bin just install-bin`.
## Commit Messages
Conventional Commits, subject line only, imperative mood, lowercase after
the colon:
```
feat(asm): add EVEX gather and scatter with VSIB addressing
```
Allowed types: `feat`, `fix`, `docs`, `style`, `refactor`, `perf`, `test`,
`chore`, `ci`, `build`, `revert`.
Every commit ends with exactly one trailer, using the model that
assisted with the change:
```
Assisted-by: <model-name>
```
Replace `<model-name>` with the actual model (e.g. `DeepSeek V4 Pro`).
No body, no footers, no trailing period on the subject.
## Code Style
Language: Go 1.27 (`toolchain go1.27.0`).
### Formatter
`gofmt` — zero diff. Run `just fmt` before committing.
### Linter
`go vet` — zero warnings. Run `just build` before committing.
### Tests
`go test -race -count=1 ./...` — all green, coverage ≥ 80 % (hard gate,
enforced by `just test`).
### Dependencies
- **Production code:** standard library only. No third-party imports in
shipped code.
- **Test code:** `golang.org/x/arch` is the sole test dependency (decode
oracle for round-trip validation). It is never linked into the binary.
- **No cgo, no C, no external toolchains, no JavaScript.**
### Error Handling
Explicit `if err != nil`. Wrap with `fmt.Errorf("context: %w", err)`.
No panics outside `main`. The one exception: the JIT trampoline's
`recover`-guarded decoder hot path, which converts bounds panics to
sentinel errors.
### Assembly
Plan 9 syntax (Go's assembler dialect). Hand-written — no code generators
except `_gen/gen.go` for instruction tables (which parses the Go
toolchain source). Every instruction table is committed; no runtime
dependency on the Go toolchain.
### File Naming
- `_amd64.s`, `_arm64.s`, `_riscv64.s`, `_loong64.s` for
architecture-specific assembly.
- `_linux_amd64.go` for platform-specific Go files.
- `_test.go` suffix for test files.
## Definition of Done
A task is not complete until all of these pass:
1. `just build` — `go vet` + `gofmt` check, zero errors, zero warnings.
2. `just test` — full suite with `-race`, coverage ≥ 80 %.
3. `just fmt` — produces no diff.
4. Diagnostics — zero warnings across the project.
5. Non-trivial changes reviewed.
## Licence
BSD-3-Clause. Every source file carries the SPDX header:
```
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
```
+211 -6
View File
@@ -7,12 +7,217 @@ and this project adheres to [Conventional Commits](https://www.conventionalcommi
## [development] ## [development]
Unreleased changes on the `development` branch.
### Added ### Added
- -
## [0.33.0] — 2026-09-14
### Added
- **Stack-split guards in `gasm asm`.** Every framed function now gets
the morestack prologue check and the trailing morestack block
(`CALL runtime.morestack_noctxt`), byte-identical to the toolchain's
`stacksplit` output on all four architectures: the small, medium and
big frame classes, auto-NOSPLIT leaves, the materialised constants of
large frames (arm64 `R27`, riscv64 `X31`, loong64 `R30`) and the
arm64 extrasize rule. Assembled objects are therefore linkable for
split functions, not only `NOSPLIT` leaves.
- **`gasm dis`.** Standalone disassembly through `golang.org/x/arch`:
a `.s` file is assembled and listed per `TEXT` function with local
labels at their real offsets, or raw bytes from a file or stdin are
disassembled linearly (`-a` selects the architecture). The debugger
shares the same decoder instead of carrying its own.
- **`gasm fmt -l` and `-d`.** Check mode lists files whose formatting
differs; diff mode prints a unified diff from the project's own
LCS-based differ, with GNU header semantics.
- **LSP cross-file navigation.** Go-to-definition and find references
fall back from local labels to the `TEXT` functions of every open
document, and rename follows the same cross-file matching.
- **Large frame offsets on riscv64 and arm64.** Frame-relative loads
and stores beyond the signed 12-bit immediate range materialise the
address through the toolchain temp register (riscv64 `X31`, arm64
`R27`) instead of silently truncating the offset (riscv64) or
rejecting the instruction (arm64); arm64 frame sizes now add the
toolchain's extrasize exactly (+8 when the frame leaves an alignment
gap, +16 when it is already aligned).
- **Tail calls `JMP sym(SB)`** on all four architectures (amd64 `E9`,
arm64 `B`, riscv64 `JAL X0`, loong64 `B`) with the call relocation.
- **Live oracle-parity tests.** Kernel files covering every guard
class, large-offset pattern and tail call are assembled by gasm and
by the installed `go tool asm` and compared byte-for-byte on all four
architectures, alongside the existing pinned-byte tests.
### Fixed
- The v0.32.0 review findings: the parser rejects malformed `TEXT`
frames and parses signed frame sizes; amd64 frame adjustments above
127 bytes encode with imm32; arm64 and loong64 relocation encodings
match the toolchain; the linter guards unnamed `TEXT` directives and
refreshes its textflag table; the LSP recovers from handler panics
and decodes client URIs; watchpoint slot state moved into the debug
session; dead verify code removed; em and en dashes replaced across
sources.
- `gasm asm --format goobj`: internal calls to `TEXT` symbols of the
same file resolve on every architecture (the reference check accepted
only the amd64 call kind).
- `gasm asm --format elf` on loong64: branch relocations now map to
`R_LARCH_B26` instead of falling into `R_LARCH_PCALA_HI20`.
- arm64 large-prologue `ADD`/`SUB` use the extended-register encoding
the toolchain picks, and the morestack block saves the link register
with the toolchain's `OR` form on loong64.
## [0.32.0] — 2026-08-31
### Added
- **Multi-architecture debugger.** `gasm debug` carries
per-architecture ptrace register access, disassemblers
(`golang.org/x/arch`), register display, FP register views and
stop-info handlers for arm64, riscv64 and loong64, and the REPL is
arch-neutral. Sessions are runtime-validated on amd64; the other
hosts execute through the now-working JIT trampolines, but their
ptrace loops have not seen hardware yet.
- **Headless debugging.** `gasm debug --script` runs REPL commands from a
file (or stdin) and exits; `--timeout` kills the debuggee when a run
hangs, with the watchdog armed before the ptrace attach.
- **Conditional breakpoints.** `break <label> if <reg> <op> <val>` now also
compares two registers (`break loop if RAX > RBX`), not only a register
against an immediate.
- **Instruction-level coverage.** `gasm debug --cover` now sets a
breakpoint on every instruction (walked by disassembly length), counts
the hits per instruction and reports the executed instructions with
their hit counts, with the label coverage derived from the same run.
Expect the run to slow to ptrace speed.
- **FP register display on riscv64 and loong64.** The debugger `regs`
command shows the 32 FP registers plus fcsr (and fcc on loong64) via
`PTRACE_GETREGSET`.
- **JIT execution trampolines.** `verify.Call` now works on all four
architectures via hand-written assembly trampolines
(`trampoline_{arm64,riscv64,loong64}.s`) that save the Go stack, switch
to a prepared stack, and branch to the JIT function.
- **ABI checks on arm64, riscv64 and loong64.** `gasm verify -abi` and
the ABI half of `-fuzz` now cover the non-amd64 architectures via
per-architecture checked trampolines: sentinels planted in the
registers the Go ABI fixes across calls (arm64 `R29`/`R28`, riscv64
`X27`, loong64 `R22`; amd64 keeps `BP`/`R14`) are verified on return
and the saved registers restored before Go code resumes, with the
below-SP canary on every architecture. loong64 kernels are verified
through the toolchain-comparison path only, pending hardware
validation of their trampoline. `verify.Load` assembles each file
with the encoder its name suffix calls for. The `ABIReport` fields
are the architecture-neutral `FPClobbered` and `GClobbered`.
- **Hardware watchpoints on all architectures.** arm64 uses DBGWVR/DBGWCR
via `PTRACE_SETREGSET` with `NT_ARM_HW_BREAK`; riscv64 and loong64 use
`PTRACE_POKEUSER` to access trigger/debug registers.
- **Cross-package GOOBJ resolution on arm64 and loong64.**
`AssembleFileARM64` and `AssembleFileLOONG64` mark external
relocations and populate `img.Externals`, so GOOBJ output from those
architectures resolves cross-package symbols like amd64 and riscv64
already do.
- **`gasm verify --args`.** Scalar arguments (`name=value`, decimal or
`0x` hex) can now be supplied to a `--call` invocation alongside `--buf`
buffers, closing the gap where only buffers could be supplied.
- **Fuzz corpus save and replay.** `gasm verify --fuzz --save-corpus dir`
records every input that crashes or mismatches as replayable JSON (buffer
contents and scalars, not raw pointers), and `gasm verify --replay dir`
re-runs the saved entries against the kernel in isolated child processes,
reporting whether each one reproduces.
- **`gasm audit-instructions`.** Black-box diff of a gasm encoder
against the installed `go tool asm`, for amd64, arm64, riscv64 and
loong64 (`gasm audit-instructions <arch>`): superset encodings
(gasm-only, shippable via `gasm asm --format goobj`),
known-but-unencodable names (the backlog) and go-only names
(feature gaps).
- **`gasm scaffold differential`.** Prints a differential test skeleton
for every `// func` signature in a kernel file: random seed states, the
kernel call and a portable reference (`<name>Portable`), compared
byte-for-byte.
- **LSP: find references** (`textDocument/references`).
- **LSP: rename symbol** (`textDocument/rename`).
- **LSP: document formatting** (`textDocument/formatting`) using the
`format` package for canonical gofmt-style output.
- **LSP: inlay hints** (`textDocument/inlayHint`): frame size hints after
the TEXT directive's argument area.
- **LSP: workspace symbol search** (`workspace/symbol`): substring search
over the TEXT functions and GLOBL/DATA symbols of every open document.
- **LSP: code actions.** Quick fixes for `missing-ret` (insert the RET)
and `unused-label` (remove the label) diagnostics.
- **LSP: signature help** (`textDocument/signatureHelp`): the callee's
`// func` signature while the cursor is on a `CALL`.
- **LSP: document highlights**: every reference to the function or label
under the cursor is highlighted.
- **LSP: pull diagnostics** (`textDocument/diagnostic`), **#include
document links** (resolved against the document directory, then
`$GOROOT/pkg/include`, so `textflag.h` opens) and **folding ranges**
(one collapsible region per TEXT function body).
- **Lint: `unused-label` rule.** Flags labels that are defined but never
referenced by any jump (Hint severity).
- **Lint: `invalid-textflag` rule.** Flags TEXT/GLOBL flags not in the
known set from `textflag.h` (Warning severity).
- **Lint: `stack-imbalance` rule.** Tracks SP changes and flags if the
net delta at RET does not match the declared frame size.
- **Lint: `register-width-mismatch` rule.** Flags amd64 operands whose
register width does not match the width the mnemonic suffix prescribes
(for example a 32-bit register in a `MOVQ`).
- **Lint: `abi0-register-args` rule.** Flags kernels whose `// func`
parameters are never read from the FP frame (for example arguments read
from registers instead), which pass every test today and break on a
toolchain upgrade. Shipped with the Go assembler's operand-width model.
- **Lint: `nonportable-register-name` rule.** Flags the `RAX`/`EAX`-style
register aliases gasm accepts but `go tool asm` rejects, so files using
them only link through the gasm GOOBJ path.
- **Lint: `unencodable-instruction` rule.** Flags mnemonics the
architecture table knows but the encoder cannot yet emit, at edit time
instead of at assembly time.
- **Lint: `reserved-register-write` rule.** Flags writes to arm64 R18,
the platform-reserved register the ABI checks cannot observe at runtime
and the Go assembler cannot even spell. Reads and macro-using files
are exempt.
- **DWARF5 debug sections in ELF output.** All four ELF emitters now emit
`.debug_abbrev`, `.debug_info`, `.debug_line`, and `.debug_line_str`
sections, enabling `addr2line` and GDB/LLDB source-level debugging, and
the amd64 emitter adds a `.debug_frame` CFI section for stack unwinding.
- **amd64: legacy SSE and conversion coverage.** The encoder now handles
the legacy (non-VEX) SSE packed binaries and immediate shuffles, the
legacy SSE integer instructions and `BSWAP`, the scalar and packed
double conversions (`CVTSS2SD`, `CVTSD2SS`, `CVTPS2PD`, `CVTPD2PS`) and
the prefetch hints.
- **amd64: `VPCMP` and the full opmask set.** The EVEX compare with an
opmask destination and the remaining opmask-register instructions are
encoded, byte for byte against the Go assembler.
### Changed
- **Parallel verify sweeps.** The `-smoke` and `-abi` per-function checks
now run in parallel instead of sequentially.
### Fixed
- **Reliable ptrace sessions.** The debugger no longer mistakes runtime
signal-delivery-stops (a Go tracee reports SIGURG preemption to the
tracer) for its launch barrier, runs every ptrace request on the thread
that forked the debuggee (requests from another thread fail with
ESRCH), prefers the debuggee's reported code base over an RWX scan, and
single-steps over a hit breakpoint so resuming cannot re-trap on the
same instruction. A ptrace integration test
(`debug/ptrace_integration_test.go`) drives a real session end to end.
- **arm64 BL relocation.** `BL sym(SB)` now records a `RelArm64Branch`
relocation instead of emitting a bare instruction with no relocation.
- **GOOBJ R_ADDRARM64 constant.** Corrected from 9 (`R_CALLARM64`) to 3
(`R_ADDRARM64`).
- **Subprocess-isolated smoke and ABI sweeps.** A function that faults
during the `-smoke` or `-abi` sweep is reported without killing the
parent; each sweep runs in a child process.
- **`BSF`, `BSR` and `POPCNT` encodings.** Corrected to the bytes
`go tool asm` emits.
- **`MOVQ` immediates.** Compressed to the toolchain's imm32 forms.
- **Frame adjustments of 128 to 255 bytes.** Now use the imm8 `ADDQ`
stack adjustment.
- **amd64 encoding parity.** A broad pass aligned the remaining encoder
outputs and operand strictness with `go tool asm`.
- **asm help text.** Updated to list arm64 as a supported architecture.
## [0.31.1] — 2026-08-20 ## [0.31.1] — 2026-08-20
### Fixed ### Fixed
@@ -588,7 +793,7 @@ objects.
`GLOBL` in the file defines no longer aborts assembly — it is recorded `GLOBL` in the file defines no longer aborts assembly — it is recorded
as an external relocation (`Image.Externals`, `FuncLayout.Relocs`) and as an external relocation (`Image.Externals`, `FuncLayout.Relocs`) and
becomes an undefined global symbol in the object output. The raw image becomes an undefined global symbol in the object output. The raw image
format (`--format raw`, the default) still reports them: only an object s (`--format raw`, the default) still reports them: only an object
file can represent a reference the linker must resolve. file can represent a reference the linker must resolve.
### Changed ### Changed
@@ -689,7 +894,7 @@ The formatter behaves like `go fmt` and canonicalises block separation.
### Changed ### Changed
- `format`: canonical blank-line layout — a new block (a label, `TEXT` or - `s`: canonical blank-line layout — a new block (a label, `TEXT` or
`GLOBL`) is preceded by exactly one blank line, neither more nor less. `GLOBL`) is preceded by exactly one blank line, neither more nor less.
Comments leading a block stay with it (the blank line goes before them), Comments leading a block stay with it (the blank line goes before them),
stacked labels share their block, the function's first label keeps hugging stacked labels share their block, the function's first label keeps hugging
@@ -724,7 +929,7 @@ the encoder learns the legacy SSE moves.
*first* operand on arm64, riscv64 and loong64; Plan 9 spelling puts it last *first* operand on arm64, riscv64 and loong64; Plan 9 spelling puts it last
on every architecture Go supports. The def/use and save/restore on every architecture Go supports. The def/use and save/restore
classification on those architectures was inverted. classification on those architectures was inverted.
- `format`: a comment that follows a `RET` (typically the next function's doc - `s`: a comment that follows a `RET` (typically the next function's doc
comment) is no longer indented as if it were still inside the finished comment) is no longer indented as if it were still inside the finished
function body. function body.
@@ -897,7 +1102,7 @@ Initial release — the Phase 1 foundation.
Zero error-severity diagnostics across the 90-file Go runtime corpus and the Zero error-severity diagnostics across the 90-file Go runtime corpus and the
production go-flac kernels (the `register-clobber` audit additionally reports production go-flac kernels (the `register-clobber` audit additionally reports
the go-flac kernels' unsaved callee-saved register use for review). the go-flac kernels' unsaved callee-saved register use for review).
- `format`: an idempotent canonical formatter (operand spacing and per-function - `s`: an idempotent canonical formatter (operand spacing and per-function
mnemonic alignment) that preserves comments and round-trips through the mnemonic alignment) that preserves comments and round-trips through the
parser. parser.
- `lsp`: a Language Server Protocol server over stdio providing completion, - `lsp`: a Language Server Protocol server over stdio providing completion,
+77 -71
View File
@@ -1,101 +1,107 @@
# Contributing to gasm-devkit # Contributing to gasm-devkit
## Prerequisites Thanks for contributing to gasm-devkit.
- Go 1.27 or later (`toolchain go1.27.0`) ## Development setup
- `just` command runner
- A Linux host on amd64, arm64, riscv64 or loong64
## Development Setup Requirements: Go 1.27 or later, the [just](https://github.com/casey/just)
command runner, and a Linux host on amd64, arm64, riscv64 or loong64.
```sh ```sh
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
cd gasm-devkit cd gasm-devkit
just install # download module dependencies just install # download module dependencies
just build # go vet + gofmt check just build # go vet + gofmt check
just test # full test suite with race detector just test # full suite, race detector, 80 % coverage gate
``` ```
## Commands ## Workflow
Every just recipe: 1. Branch from `development`; never commit directly to `main` (`main` is
release-only: merge from `development`, then tag).
2. Commit with [Conventional Commits](https://www.conventionalcommits.org/):
`type(scope): description`: subject line only, imperative mood,
lowercase after the colon, no trailing dot. Allowed types: `feat`,
`fix`, `docs`, `style`, `refactor`, `perf`, `test`, `chore`, `ci`,
`build`, `revert`. The only line after the subject is the trailer:
`Assisted-by: <model-name>`. No `Co-Authored-By`, no `Signed-off-by`,
no other trailers.
3. Record every user-visible change in `CHANGELOG.md` under
`## [development]` (categories: Added, Changed, Fixed, Removed,
Security).
4. Add or update tests; coverage must stay **at or above 80 %** (hard
gate, enforced by CI).
5. Update the documentation when behaviour, flags or the public surface
change.
6. Open a pull request against `development`.
| Recipe | What it does | Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`;
|--------|-------------| CI builds and publishes the binaries for all four architectures.
| `just` | List all recipes |
| `just install` | `go mod download` |
| `just build` | `go vet ./...` + `gofmt -l .` check — zero errors required |
| `just test` | `go test -race -count=1 -coverprofile=coverage.out ./...` + 80 % coverage gate |
| `just fmt` | `gofmt -w .` |
| `just run -- lint file.s` | Run the CLI with `go run` (args after `--`) |
| `just install-bin` | Install `gasm` into `$GOBIN` with the release version stamped |
| `just gen` | Regenerate `arch/*_gen.go` instruction tables from the Go toolchain |
| `just uninstall` | Remove build artefacts (`coverage.out`, `gasm`, `*.test`) |
## Running a Single Test ## Code style
`gofmt` and `go vet` via `just fmt` / `just build`; both must pass with
zero output; `go fix -diff ./...` must report nothing on touched packages.
- Standard library only in production code; `golang.org/x/arch` is used
in tests only (round-trip decoding) and is never linked into the `gasm`
binary.
- No cgo, no C, no external toolchains at runtime.
- Explicit `if err != nil`; errors wrapped with
`fmt.Errorf("context: %w", err)`; no panics outside `main`.
- The parser, lexer and formatter are hand-written; the `arch` instruction
tables are generated only via `_gen/gen.go` (`just gen`), never edited.
## Running a single test
```sh ```sh
go test -run TestVexGroundTruth ./asm/ go test -run TestVexGroundTruth ./asm/
go test -run TestDifferentialLZ4Fuzz ./verify/ go test -run TestGroundTruthBasic ./verify/
go test -run TestGOObjectLinkAndRun ./asm/
go test -run TestFuzzWideCopy ./verify/
``` ```
## Testing the Debugger The interactive debugger (`gasm debug`) requires a compiled binary on
`$PATH`; `go run` does not work for the traced child process. Install
first with `just install-bin`.
The interactive debugger (`gasm debug`) requires a compiled binary — ## CI (Gitea Actions)
`go run` does not work for the child process. Install first:
```sh Workflows live in `.gitea/workflows/` and run on self-hosted runners:
just install-bin
gasm debug --func add testdata/verify/basic_amd64.s
```
## Code Style | Workflow | Trigger | What it does |
|----------|---------|--------------|
| Test | push / PR to `development` | gofmt check, `go vet`, `go test -race`, 80 % coverage gate |
| Release | tag `v*` | cross-compiles binaries for linux/{amd64,arm64,riscv64,loong64} and publishes the Gitea release |
See [AGENTS.md](AGENTS.md) for the full style guide. Key points: The Definition of Done (`just build` + `just test` + `just fmt`) must
still pass locally before pushing.
- `gofmt` — zero diff. ## AI Contribution Policy
- `go vet` — zero warnings.
- Standard library only in production code; `golang.org/x/arch` in tests.
- No cgo, no C, no JavaScript.
- Hand-written Plan 9 assembly; tables generated only via `_gen/gen.go`.
## Branches and Releases AI tools are welcome as productivity aids. What matters is that
contributions remain understandable, reviewable, and genuinely useful.
- `development` is the working branch. - **Disclose AI use.** If you used AI to draft or generate any part of a
- `main` is release-only: `git merge --ff-only development`, then `git tag vX.Y.Z`. commit, issue, pull request, or code review, say so clearly.
- Conventional Commits: `feat(asm): add EVEX gather and scatter`. - **Commit messages:** end every commit with exactly one trailer:
- Every commit ends with `Assisted-by: <model-name>`. `Assisted-by: <model-name>` (e.g. `Assisted-by: GLM 5.3`).
- **Pull requests and issues:** attribute AI assistance in one trailing
line, e.g. `_Assisted-by: GLM 5.3_`. Do not paste it into the PR
description as a section.
- **Take responsibility.** You remain accountable for the accuracy,
completeness, and intent of everything you submit.
- **Review before marking ready.** Read AI-generated diffs carefully, run
them locally, and add or update tests where appropriate.
- **Preferred models.** Prefer open-weight models with transparent
training data: **GLM**, **DeepSeek**, and **MiMo**.
## CI ## Reporting bugs
CI runs on every push to `development` and on pull requests:
- **Test** (`test.yml`) — `gofmt` check, `go vet`, `go test -race` and the
80 % coverage gate.
- **Release** (`release.yml`) — cross-compiles release binaries for
linux/{amd64,arm64,riscv64,loong64} on version tags and publishes them.
The Definition of Done (`just build` + `just test` + `just fmt`) must still
pass locally before pushing.
## AI-Assisted Contributions
AI agents may assist with code, documentation, tests, and review. All
AI-assisted changes must:
- Include the trailer `Assisted-by: <model-name>` in the commit message
(e.g. `Assisted-by: DeepSeek V4 Pro`).
- Follow the [AGENTS.md](AGENTS.md) rules.
- Pass the Definition of Done before committing.
Attribute agent authorship in issues and pull requests on one trailing
line:
```
_Assisted-by: Qwen 3.8 Max_
```
## Questions
Open an issue at Open an issue at
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit/issues). [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit/issues)
with the version (`gasm --version`), OS and architecture, the exact
command, the full output, and the expected versus actual behaviour.
**Security issues:** email **opensource@petrbalvin.org** instead of opening
a public issue.
+131 -107
View File
@@ -1,144 +1,168 @@
# gasm-devkit # gasm-devkit
Developer tooling for **GAsm** — Go's built-in Plan 9 assembler. Developer tooling for **GAsm**, Go's built-in Plan 9 assembler.
Go ships an assembler but no tooling for it. There is no syntax highlighting, Go ships an assembler but no tooling for it: there is no syntax highlighting,
no autocomplete, no linter, no static analyser, no formatter, no standalone no autocomplete, no linter, no static analyser, no formatter, no standalone
assembler and no debugger for `.s` files. Developers write assembly blind, assembler and no debugger for `.s` files. Developers write assembly blind,
validate it by benchmark, and debug it by print statement. validate it by benchmark, and debug it by print statement. gasm-devkit is the
missing toolkit: a single, self-contained binary, `gasm`, that brings proper
developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
gasm-devkit is the missing toolkit. It is a single, self-contained binary — ## Features
`gasm` — that brings proper developer tooling to Plan 9 assembly:
``` - **Front end.** A hand-written lexer and an error-tolerant parser produce a
gasm tokens dump the lexical token stream typed AST with source positions; `gasm tokens` and `gasm parse` expose them
gasm parse parse and report syntax errors directly.
gasm fmt canonicalise formatting (gofmt for assembly) - **Formatter.** `gasm fmt` canonicalises indentation, operand spacing,
gasm lint static checks per-function mnemonic alignment and blank-line layout: `gofmt` for assembly,
gasm lsp language server (completion, hover, symbols, diagnostics, highlighting) operating recursively on directories the way `go fmt` does. `-l` lists
gasm asm standalone assembler files whose formatting differs and `-d` prints a unified diff.
gasm verify dynamic analysis & verification - **Linter.** `gasm lint` runs 18 conservative static checks, among them
gasm debug source-level debugger `undefined-label`, `abi-argsize` (declared frame vs the `// func` signature),
gasm diff compare machine code of two .s files `register-clobber` (Go ABI register liveness over the control-flow graph),
gasm profile show basic-block structure of functions `stack-imbalance`, `abi0-register-args` and `unencodable-instruction`.
``` - **Standalone assembler.** `gasm asm` encodes all four architectures without
the Go toolchain and writes raw images, linkable ELF objects (with DWARF5
debug sections) or the Go toolchain's own GOOBJ format, which `go build`
consumes in place of the toolchain's output. Framed functions get the
stack-split guard and the morestack block, byte-identical to the
toolchain's, so split functions link too.
- **Disassembler.** `gasm dis` lists a `.s` file's functions at their real
offsets after assembling, or disassembles raw bytes from a file or stdin.
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
executable memory: smoke calls, ABI checks (sentinel registers, red-zone
canary), differential fuzzing against the `go tool asm` build, and
byte-for-byte ground-truth comparison of the machine code.
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
breakpoints (optionally conditional), hardware watchpoints, register and
memory inspection, and headless script runs with label-level coverage.
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
push and pull diagnostics, semantic-token highlighting, go-to-definition,
find references, rename, formatting, inlay hints, code actions, signature
help, document highlights, workspace symbol search, #include document
links and folding ranges over stdio; definition, references and rename
work across every open document.
- **Comparators and audits.** `gasm diff` compares the machine code of two
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
`gasm audit-instructions` diffs the encoder against the installed toolchain,
and `gasm scaffold` generates a differential test skeleton for a kernel.
- **Complete instruction coverage.** The instruction tables are generated
from the Go toolchain's own assembler source, so the toolkit recognises
every mnemonic the real assembler accepts; `just gen` refreshes them.
## Architecture support ### Architecture support
gasm-devkit targets every architecture Go's assembler speaks. The instruction | Architecture | GOARCH | File suffix | Instructions recognised |
tables are **generated from the Go toolchain's own assembler source** |--------------|-------------|--------------|---------------------------------------------|
(`cmd/internal/obj/<arch>`), so gasm-devkit recognises *every* mnemonic the | AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
real assembler accepts — not a hand-maintained subset that drifts and rots. | ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes |
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes |
| Architecture | GOARCH | File suffix | Instructions recognised | | LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes |
|--------------|-------------|----------------|------------------------------------|
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
| ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes |
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes |
| LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes |
"Common opcodes" are the instructions shared by every architecture (`RET`, "Common opcodes" are the instructions shared by every architecture (`RET`,
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, …). AMD64 additionally `JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`, carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
…) that the assembler accepts as aliases. Regenerating the tables is one ...) that the assembler accepts as aliases. Regenerating the tables is one
command — `just gen` — and requires only a Go installation; the committed command (`just gen`) and requires only a Go installation; the committed output
output has no runtime dependency on the toolchain. has no runtime dependency on the toolchain.
## Supported Platforms ## Install
The toolkit runs on Linux. All four Linux architectures are supported as Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
hosts — amd64, arm64, riscv64 and loong64 — and the release matrix linux/loong64 are on the
cross-compiles the same four targets. [releases page](https://sourcedock.dev/petrbalvin/gasm-devkit/releases).
From source (Go 1.27 or later):
**FreeBSD support is planned for a future release.** ```sh
go install sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm@latest
```
## Principles Or from a repository checkout, with the development version stamped:
- **Pure Go and GAsm only.** No C, no cgo, no external toolchains, no native ```sh
binaries, no JavaScript runtimes. The parser is hand-written; there is no just install-bin
parser generator. ```
- **Self-contained.** The toolkit's production code depends only on the
standard library; one binary, no runtime data files. The single module
dependency, `golang.org/x/arch`, is used **only in tests** to validate the
instruction encoder by round-trip decoding — it is never linked into the
`gasm` binary.
- **Linux-only.** Runs natively on amd64, arm64, riscv64 and loong64 Linux
hosts; the release matrix cross-compiles the same four targets. Latest
stable Go only.
- **No vendor lock-in.** The integration surface is the Language Server
Protocol and a command-line interface — both open standards. No cloud
service, no proprietary API, no dependence on any one editor's internals.
- **Complete and verifiable.** Instruction coverage is generated from the
assembler's own source and regenerated on demand, so it cannot silently fall
behind the toolchain.
## Components
| Package | Purpose |
|---------|---------|
| `token` | Lexical token kinds and source positions. |
| `lexer` | Hand-written scanner for Plan 9 assembly. |
| `ast` | The abstract syntax tree. |
| `parser` | Line-oriented, error-tolerant parser producing the AST. |
| `arch` | amd64, arm64, riscv64 and loong64 register files and instruction tables. |
| `lint` | Conservative static checks. |
| `format` | A canonical formatter — `gofmt` for assembly. |
| `asm` | The standalone assembler: amd64, RISC-V and LoongArch encoders, linker, object-file emitters (ELF, GOOBJ). |
| `verify` | JIT execution substrate for dynamic analysis, combined ABI+fuzz differential testing. |
| `debug` | Interactive ptrace debugger with GPR/YMM register display and named buffer allocation. |
| `lsp` | Language Server Protocol server. |
| `cmd/gasm` | The `gasm` binary tying it all together. |
| `_gen` | The generator that rebuilds the instruction tables from the Go toolchain. |
See [`docs/ARCHITECTURE.md`](docs/ARCHITECTURE.md) for the design rationale and
data flow, and [`docs/DECISIONS.md`](docs/DECISIONS.md) for design decisions
deliberately postponed (with the analysis needed to pick them up again).
## Quick start ## Quick start
```sh ```sh
just install # download dependencies (there are none) cat > hello_amd64.s <<'EOF'
just build # go vet + gofmt check — zero errors, zero warnings #include "textflag.h"
just test # full suite, race detector, 80 % coverage gate
just fmt # gofmt the tree // func add(a, b int) int
just gen # regenerate the instruction tables from the Go toolchain TEXT ·add(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
ADDQ b+8(FP), AX
MOVQ AX, ret+16(FP)
RET
EOF
gasm lint hello_amd64.s # static checks
gasm asm -o hello.bin hello_amd64.s # assemble to a raw image
gasm verify --call add --args a=2,b=3 hello_amd64.s # JIT-call it with arguments
``` ```
Install the binary and use it: ## Usage
```sh ```sh
just install-bin # installs gasm into $GOBIN gasm fmt # reformat every .s below here, like go fmt
gasm fmt -w kernel_amd64.s # canonicalise one file in place
gasm --help # overview of commands and flags gasm fmt -l *.s # list files whose formatting differs
gasm tokens kernel_amd64.s # dump the token stream gasm fmt -d kernel_amd64.s # print a unified diff instead
gasm parse kernel_amd64.s # parse, report syntax errors gasm lint *.s # static checks
gasm fmt -w kernel_amd64.s # canonicalise in place gasm asm --format elf -o k.o k.s # assemble to a linkable ELF object
gasm fmt # reformat every .s below here, like go fmt gasm asm --format goobj -p pkg/path -o k.o k.s # Go object, consumed by go build
gasm lint *.s # static checks gasm dis k.s # assemble, then list each function
gasm asm --format elf -o k.o k.s # assemble to a linkable ELF object gasm dis -a amd64 - < dump.bin # disassemble raw bytes from stdin
gasm verify kernel_amd64.s # JIT-load and report functions gasm verify --ground-truth k.s # byte-for-byte vs go tool asm
gasm verify --ground-truth k.s # byte-for-byte vs go tool asm gasm verify --fuzz k.s # differential fuzz vs the go tool asm build
gasm verify --call decodeBlockAVX2 --buf src:64:hex...,dst:256:zero k.s gasm debug --func name k.s # interactive debugger
gasm debug --func name k.s # interactive debugger gasm debug --func name --script cmds.txt --timeout 30s k.s # headless run
gasm diff a.s b.s # compare machine code byte-for-byte gasm debug --func name --cover k.s # which labels did execution reach?
gasm diff a.s b.s # compare machine code byte-for-byte
gasm diff --map wideCopyAVX2=wideCopyAVX512 avx2.s avx512.s gasm diff --map wideCopyAVX2=wideCopyAVX512 avx2.s avx512.s
gasm profile k.s # show basic-block structure gasm profile k.s # show basic-block structure
gasm audit-instructions # encoder vs go tool asm name diff
gasm scaffold differential k.s # generate a differential test skeleton
``` ```
See [CONTRIBUTING.md](CONTRIBUTING.md) for the full development workflow, Run `gasm --help` for the command overview and `gasm <command> -h` for a
[docs/CLI.md](docs/CLI.md) for the command reference, and command's flags. [docs/CLI.md](docs/CLI.md) is the full reference.
[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for setup and recipes.
## Editor integration ### Editor integration
`gasm lsp` speaks the Language Server Protocol over standard input/output, so `gasm lsp` speaks the Language Server Protocol over standard input/output, so
any LSP-capable editor can use it — point your editor's LSP client at the any LSP-capable editor can use it: point your editor's LSP client at the
binary and associate it with `.s` files. Syntax highlighting is delivered as binary and associate it with `.s` files. Syntax highlighting is delivered as
**LSP semantic tokens**, so no editor-specific grammar is required. The server LSP semantic tokens, so no editor-specific grammar is required. The server
infers the target architecture from the file-name suffix infers the target architecture from the file-name suffix
(`_amd64.s` / `_arm64.s` / `_riscv64.s` / `_loong64.s`). (`_amd64.s` / `_arm64.s` / `_riscv64.s` / `_loong64.s`).
## License ## Development
```sh
just install # download module dependencies
just build # go vet + gofmt check, zero errors and zero warnings
just test # full suite, race detector, 80 % coverage gate
just fmt # gofmt the tree
just gen # regenerate the instruction tables from the Go toolchain
```
See [CONTRIBUTING.md](CONTRIBUTING.md) for the development workflow and
[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for setup details and every
recipe.
## Documentation
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/CLI.md](docs/CLI.md): full command reference
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
- [docs/DECISIONS.md](docs/DECISIONS.md): deferred design decisions
- [CHANGELOG.md](CHANGELOG.md): release history
## Licence
BSD-3-Clause — see [LICENSE](LICENSE).
BSD-3-Clause — see [LICENSE](LICENSE).
Copyright © 2026 [Petr Balvín](https://petrbalvin.org) Copyright © 2026 [Petr Balvín](https://petrbalvin.org)
+2
View File
@@ -270,6 +270,8 @@ func amd64Curated() []Instr {
"VMINPD", "VMINPS", "VMINSD", "VMINSS", "VMAXPD", "VMAXPS", "VMAXSD", "VMAXSS", "VMINPD", "VMINPS", "VMINSD", "VMINSS", "VMAXPD", "VMAXPS", "VMAXSD", "VMAXSS",
"VXORPD", "VXORPS", "VANDPD", "VANDPS", "VANDNPD", "VANDNPS", "VORPD", "VORPS", "VXORPD", "VXORPS", "VANDPD", "VANDPS", "VANDNPD", "VANDNPS", "VORPD", "VORPS",
"VUNPCKHPD", "VUNPCKLPD", "VUNPCKHPS", "VUNPCKLPS", "VUNPCKHPD", "VUNPCKLPD", "VUNPCKHPS", "VUNPCKLPS",
"PSHUFD", "PSHUFHW", "PSHUFLW", "SHUFPS", "SHUFPD",
"UNPCKLPS", "UNPCKHPS", "UNPCKLPD", "UNPCKHPD",
"VSQRTPD", "VSQRTPS", "VSQRTSD", "VSQRTSS", "VRSQRTPS", "VRCPPS", "VSQRTPD", "VSQRTPS", "VSQRTSD", "VSQRTSS", "VRSQRTPS", "VRCPPS",
"VCMPPD", "VCMPPS", "VCMPSD", "VCMPSS", "VCMPPD", "VCMPPS", "VCMPSD", "VCMPSS",
} { } {
+1 -1
View File
@@ -29,7 +29,7 @@ func arm64Registers() []Register {
regs = append(regs, Register{Name: name, Class: class, Desc: desc}) regs = append(regs, Register{Name: name, Class: class, Desc: desc})
} }
// General-purpose integer registers R0–R30. // General-purpose integer registers R0-R30.
for i := 0; i <= 30; i++ { for i := 0; i <= 30; i++ {
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register") add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
} }
+2 -5
View File
@@ -109,16 +109,13 @@ func main() {
if err != nil { if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog) t.Fatalf("baseline build: %v\n%s", err, buildLog)
} }
var pkgArch, work, linkLine, asmObj string var work, linkLine, asmObj string
for _, line := range strings.Split(string(buildLog), "\n") { for line := range strings.SplitSeq(string(buildLog), "\n") {
switch { switch {
case strings.HasPrefix(line, "WORK="): case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=") work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"): case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o") asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"): case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line linkLine = line
} }
+130 -65
View File
@@ -12,17 +12,21 @@ import (
// assembleARM64 assembles an AArch64 (arm64) TEXT function body into machine // assembleARM64 assembles an AArch64 (arm64) TEXT function body into machine
// code. Every instruction is 4 bytes; the MOV pseudo-instruction and the // code. Every instruction is 4 bytes; the MOV pseudo-instruction and the
// immediate-arithmetic forms expand to 2–4 instructions when the immediate // immediate-arithmetic forms expand to 2-4 instructions when the immediate
// does not fit, so the layout is computed in two passes (sizes, then encoding // does not fit, so the layout is computed in two passes (sizes, then encoding
// with resolved branch targets). // with resolved branch targets).
// //
// The emitted bytes match the Go toolchain's arm64 assembler, which is the // The emitted bytes match the Go toolchain's arm64 assembler, which is the
// ground-truth oracle: prologue/epilogue, FP/SP frame mapping, branch // ground-truth oracle: prologue/epilogue, FP/SP frame mapping, branch
// encodings and the MOV immediate expansions all follow cmd/internal/obj/ // encodings and the MOV immediate expansions all follow cmd/internal/obj/
// arm64's asmout cases. // arm64's asmout cases. One deliberate difference: the stack-growth guard
// (the morestack check in the prologue and the call back into the runtime in
// the epilogue) is not emitted, so the bytes match only for NOSPLIT functions
// or zero-frame leaves, where the toolchain emits no guard either.
func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) { func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
fi := arm64ComputeFrame(t) fi := arm64ComputeFrame(t)
prologue := arm64Prologue(fi) prologue := arm64Prologue(fi)
guardLen := arm64GuardLen(fi)
chain := arm64JumpChain(t) chain := arm64JumpChain(t)
resolve := func(name string) string { resolve := func(name string) string {
if r, ok := chain[name]; ok { if r, ok := chain[name]; ok {
@@ -35,14 +39,14 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
var spadj []SpadjStep var spadj []SpadjStep
// The prologue (3 instructions when a small frame, 4 for large) // The prologue (3 instructions when a small frame, 4 for large)
// raises the SP delta by autosize. // raises the SP delta by autosize. The guard prefix shifts its PC.
if fi.autosize != 0 { if fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: arm64PrologueSpadjPC(fi), Value: fi.autosize}) spadj = append(spadj, SpadjStep{PC: guardLen + arm64PrologueSpadjPC(fi), Value: fi.autosize})
} }
// Pass 1: label offsets from the instruction sizes. // Pass 1: label offsets from the instruction sizes.
offsets := map[string]int{} offsets := map[string]int{}
pos := len(prologue) pos := guardLen + len(prologue)
for _, stmt := range t.Body { for _, stmt := range t.Body {
switch s := stmt.(type) { switch s := stmt.(type) {
case *ast.Label: case *ast.Label:
@@ -52,9 +56,25 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
} }
} }
// Pass 2: encode. Relocation offsets are recorded function-relative. // Pass 2: encode. The guard prefix precedes the prologue; its branches
out := append([]byte(nil), prologue...) // target the morestack block at the end of the function, whose position
pc := len(prologue) // the first pass has settled.
bodyLen := 0
{
p := guardLen + len(prologue)
for _, stmt := range t.Body {
if in, ok := stmt.(*ast.Instr); ok {
p += arm64InstrSize(in, fi)
}
}
bodyLen = p - (guardLen + len(prologue))
}
var out []byte
if fi.needSplit {
out = append(out, arm64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...)
}
out = append(out, prologue...)
pc := guardLen + len(prologue)
preCount := len(relocs) preCount := len(relocs)
var lines []LineEntry var lines []LineEntry
for _, stmt := range t.Body { for _, stmt := range t.Body {
@@ -67,7 +87,12 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err) return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
} }
for j := preCount; j < len(relocs); j++ { for j := preCount; j < len(relocs); j++ {
relocs[j].Off += pc - len(prologue) // Make the relocation offsets function-relative: each instruction
// records its reloc offset relative to its own start, and pc is
// that instruction's offset from the function start (prologue
// included). After shifts by the same amount.
relocs[j].Off += pc
relocs[j].After += pc
} }
preCount = len(relocs) preCount = len(relocs)
lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line}) lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line})
@@ -79,6 +104,12 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
out = append(out, code...) out = append(out, code...)
pc += len(code) pc += len(code)
} }
if fi.needSplit {
block, blReloc := arm64MoreStackBlock(pc)
out = append(out, block...)
relocs = append(relocs, blReloc)
pc += len(block)
}
return out, offsets, relocs, lines, spadj, nil return out, offsets, relocs, lines, spadj, nil
} }
@@ -191,10 +222,10 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops)) return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
} }
return a64wordLE(uint32(immFromOperand(ops[0]))), nil return a64wordLE(uint32(immFromOperand(ops[0]))), nil
case "B": case "B", "JMP":
return encodeARM64Branch(mnem, ops, pc, offsets, false, resolve) return encodeARM64Branch(mnem, ops, pc, offsets, false, relocs, resolve)
case "BL", "CALL": case "BL", "CALL":
return encodeARM64Branch(mnem, ops, pc, offsets, true, resolve) return encodeARM64Branch(mnem, ops, pc, offsets, true, relocs, resolve)
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
"FMOVS", "FMOVD": "FMOVS", "FMOVD":
return encodeARM64Mov(instr, mnem, fi, relocs) return encodeARM64Mov(instr, mnem, fi, relocs)
@@ -300,16 +331,30 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// ---- branch encoding ---- // ---- branch encoding ----
// encodeARM64Branch encodes an unconditional branch (B/BL) to a label. // encodeARM64Branch encodes an unconditional branch (B/BL) to a label.
func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[string]int, link bool, resolve func(string) string) ([]byte, error) { func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[string]int, link bool, relocs *[]Reloc, resolve func(string) string) ([]byte, error) {
if len(ops) != 1 { if len(ops) != 1 {
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
} }
op := ops[0] op := ops[0]
// External symbol reference: BL sym(SB). // Symbol reference: BL sym(SB), or B sym(SB) for a tail call, against a
if link && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" { // relocation (R_CALLARM64 either way).
// Emit BL with zero offset; the linker fills in the target. if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
return a64wordLE(a64Branch(1, 0)), nil if relocs != nil {
*relocs = append(*relocs, Reloc{
Off: 0,
After: 4,
Name: op.Addr.Sym.Name,
Addend: op.Addr.Sym.Offset,
Kind: RelArm64Branch,
})
}
// Emit B/BL with zero offset; the linker fills in the target.
bop := uint32(0) // B
if link {
bop = 1 // BL
}
return a64wordLE(a64Branch(bop, 0)), nil
} }
target := resolve(arm64Label(op)) target := resolve(arm64Label(op))
@@ -404,7 +449,7 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 { if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
} }
v := int32(immFromOperand(ops[0])) v := immFromOperand(ops[0])
rd := arm64RegNum(operandRegName(ops[len(ops)-1])) rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
rn := rd rn := rd
if len(ops) == 3 { if len(ops) == 3 {
@@ -453,7 +498,7 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
// ---- MOV pseudo-instruction ---- // ---- MOV pseudo-instruction ----
// encodeARM64Mov encodes the MOV family — the load/store/immediate workhorse // encodeARM64Mov encodes the MOV family, the load/store/immediate workhorse
// of Go's arm64 assembly. MOV is an alias of MOVD (the width mnemonics // of Go's arm64 assembly. MOV is an alias of MOVD (the width mnemonics
// select the access width). The forms, mirroring the toolchain: // select the access width). The forms, mirroring the toolchain:
// //
@@ -553,9 +598,9 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
} }
_, off := arm64MemWithFrame(mem, fi) _, off := arm64MemWithFrame(mem, fi)
// Scaled unsigned offset fits if aligned and in range. // Scaled unsigned offset fits if aligned and in range.
lt := a64LoadTable[mnem] lt, ok := a64LoadTable[mnem]
if lt.size == 0 { if !ok {
lt.size = 3 // default to64-bit for MOV lt = a64LoadTable["MOVD"] // the MOV pseudo is a 64-bit access
} }
scale := int32(1) << uint(lt.size) scale := int32(1) << uint(lt.size)
if off >= 0 && off%scale == 0 && off/scale < 4096 { if off >= 0 && off%scale == 0 && off/scale < 4096 {
@@ -564,7 +609,10 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
if off >= -256 && off <= 255 { if off >= -256 && off <= 255 {
return 4 // unscaled return 4 // unscaled
} }
return 12 // materialise offset + LDR/STR if _, _, _, ok := arm64SplitOffset(off, scale); ok {
return 8 // ADD base, REGTMP + access
}
return 12 // literal pool range: encoding reports it as unsupported
default: default:
return 4 // register move return 4 // register move
} }
@@ -597,7 +645,7 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
// - C_ABCON0 (0 < v ≤ 4095): bitmask first for positive values // - C_ABCON0 (0 < v ≤ 4095): bitmask first for positive values
// - Negative values: MOVN first, then bitmask // - Negative values: MOVN first, then bitmask
// - C_MOVCON (movcon-eligible, outside ABCON range): MOVZ/MOVN first // - C_MOVCON (movcon-eligible, outside ABCON range): MOVZ/MOVN first
tryBitmaskFirst := (d > 0 && d <= 0xFFF) tryBitmaskFirst := d > 0 && d <= 0xFFF
if tryBitmaskFirst { if tryBitmaskFirst {
// Small immediate: try bitmask first (Go uses ORR for values like $1, $256). // Small immediate: try bitmask first (Go uses ORR for values like $1, $256).
@@ -629,7 +677,7 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
// Multi-instruction: MOVZ + MOVK for each non-zero16-bit chunk. // Multi-instruction: MOVZ + MOVK for each non-zero16-bit chunk.
var ws []uint32 var ws []uint32
first := true first := true
for i := 0; i < 4; i++ { for i := range 4 {
chunk := (d >> uint(i*16)) & 0xFFFF chunk := (d >> uint(i*16)) & 0xFFFF
if chunk == 0 { if chunk == 0 {
continue continue
@@ -670,7 +718,7 @@ func arm64Bitmask(v uint64, sf int) (N, immr, imms uint32, ok bool) {
} }
// Check each rotation: is the rotated pattern a contiguous block of 1s at the LSB? // Check each rotation: is the rotated pattern a contiguous block of 1s at the LSB?
for r := uint(0); r < esize; r++ { for r := range esize {
rotated := (pattern >> r) | ((pattern << (esize - r)) & emask) rotated := (pattern >> r) | ((pattern << (esize - r)) & emask)
if rotated == 0 { if rotated == 0 {
continue continue
@@ -784,33 +832,53 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6
} }
scale := int32(1) << uint(lt.size) scale := int32(1) << uint(lt.size)
if load {
// Try scaled unsigned offset first.
if off >= 0 && off%scale == 0 {
imm12 := uint32(off / scale)
if imm12 < 4096 {
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), imm12, uint32(rn), uint32(reg))), nil
}
}
// Try unscaled (9-bit signed).
if off >= -256 && off <= 255 {
return a64wordLE(a64LSUnscaled(lt.size, lt.V, lt.opc, off, rn, reg)), nil
}
// Large offset: materialise in R20 (TMP) and use register-offset.
return nil, fmt.Errorf("%s: offset %d out of range", mnem, off)
}
// Store: same encoding but opc bits indicate store.
storeOpc := a64StoreOpc(lt) storeOpc := a64StoreOpc(lt)
if off >= 0 && off%scale == 0 { var opc int
imm12 := uint32(off / scale) if load {
if imm12 < 4096 { opc = lt.opc
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), imm12, uint32(rn), uint32(reg))), nil } else {
} opc = storeOpc
}
// Scaled unsigned offset first, then the unscaled ±255 form.
if off >= 0 && off%scale == 0 && off/scale < 4096 {
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(off/scale), uint32(rn), uint32(reg))), nil
} }
if off >= -256 && off <= 255 { if off >= -256 && off <= 255 {
return a64wordLE(a64LSUnscaled(lt.size, lt.V, storeOpc, off, rn, reg)), nil return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, off, rn, reg)), nil
} }
return nil, fmt.Errorf("%s: offset %d out of range", mnem, off) // Large offset: materialise the base in REGTMP (R27) the way the
// toolchain does and access what remains.
addImm, addShift, access, ok := arm64SplitOffset(off, scale)
if !ok {
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
}
return a64WordsLE(
a64AddSub(1, 0, 0, addShift, uint32(addImm), 31, 27), // ADD $addImm<<shift, SP, R27
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(access/scale), 27, uint32(reg)),
), nil
}
// arm64SplitOffset decomposes an out-of-range frame offset for a REGTMP
// base: an ADD (plain, or shifted left by 12) brings SP near the target and
// the access covers what remains. ok is false when no decomposition exists
// (offsets at or beyond 16 MiB, where the toolchain falls back to a literal
// pool).
func arm64SplitOffset(off int32, scale int32) (addImm, addShift uint32, access int32, ok bool) {
if off < 0 {
return 0, 0, 0, false
}
// Plain ADD: bring SP to within the largest scaled access.
l := min(off, 4095*scale)
l -= l % scale
if a := off - l; a <= 4095 {
return uint32(a), 0, l, true
}
// Shifted ADD: cover everything but the bits the access imm12 carries.
rest := off &^ (0xFFF * scale)
if rest >= 0 && rest>>12 <= 4095 {
return uint32(rest >> 12), 1, off - rest, true
}
return 0, 0, 0, false
} }
// ---- static symbol references (ADRP + offset) ---- // ---- static symbol references (ADRP + offset) ----
@@ -830,7 +898,9 @@ func encodeARM64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
) )
} }
// encodeARM64SBLoad emits ADRP R20, 0; LDR Rd, [R20, 0] with relocations. // encodeARM64SBLoad emits ADRP R27, 0; LDR Rd, [R27, 0] with relocations,
// matching the toolchain: the scratch register is REGTMP (R27) and the pair
// carries R_ARM64_PCREL_LDST64.
func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([]byte, error) { func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([]byte, error) {
lt, ok := a64LoadTable[mnem] lt, ok := a64LoadTable[mnem]
if !ok { if !ok {
@@ -838,17 +908,17 @@ func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([
} }
if relocs != nil { if relocs != nil {
*relocs = append(*relocs, *relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, Reloc{Off: 0, After: 8, Name: sym.Name, Kind: RelArm64LDST64, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
) )
} }
return a64WordsLE( return a64WordsLE(
a64ADR(1, 0, 0, 20), // ADRP R20, 0 a64ADR(1, 0, 0, 27), // ADRP R27, 0
a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), 0, 20, uint32(rd)), // LDR Rd, [R20, #0] a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), 0, 27, uint32(rd)), // LDR Rd, [R27, #0]
), nil ), nil
} }
// encodeARM64SBStore emits ADRP R20, 0; STR Rs, [R20, 0] with relocations. // encodeARM64SBStore emits ADRP R27, 0; STR Rs, [R27, 0] with relocations,
// matching the toolchain's R27 scratch and R_ARM64_PCREL_LDST64 pair.
func encodeARM64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) ([]byte, error) { func encodeARM64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) ([]byte, error) {
lt, ok := a64LoadTable[mnem] lt, ok := a64LoadTable[mnem]
if !ok { if !ok {
@@ -857,23 +927,17 @@ func encodeARM64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) (
storeOpc := a64StoreOpc(lt) storeOpc := a64StoreOpc(lt)
if relocs != nil { if relocs != nil {
*relocs = append(*relocs, *relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, Reloc{Off: 0, After: 8, Name: sym.Name, Kind: RelArm64LDST64, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
) )
} }
return a64WordsLE( return a64WordsLE(
a64ADR(1, 0, 0, 20), // ADRP R20, 0 a64ADR(1, 0, 0, 27), // ADRP R27, 0
a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), 0, 20, uint32(rs)), // STR Rs, [R20, #0] a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), 0, 27, uint32(rs)), // STR Rs, [R27, #0]
), nil ), nil
} }
// ---- operand helpers ---- // ---- operand helpers ----
// arm64Reg returns the register number of an operand, or -1.
func arm64Reg(op *ast.Operand) int {
return arm64RegNum(operandRegName(op))
}
// arm64Imm64 returns the full 64-bit immediate value of an operand. // arm64Imm64 returns the full 64-bit immediate value of an operand.
func arm64Imm64(op *ast.Operand) int64 { func arm64Imm64(op *ast.Operand) int64 {
if op.Imm.HasVal { if op.Imm.HasVal {
@@ -1088,7 +1152,7 @@ func encodeARM64CSEL(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
return a64wordLE(baseOp | uint32(rn)<<16 | invCond<<12 | uint32(rn)<<5 | uint32(rd)), nil return a64wordLE(baseOp | uint32(rn)<<16 | invCond<<12 | uint32(rn)<<5 | uint32(rd)), nil
} }
// CSEL cond, Rn, Rm, Rd (4 operands) — condition first. // CSEL cond, Rn, Rm, Rd (4 operands), condition first.
// Go assembler syntax: CSEL cond, Rn, Rm, Rd // Go assembler syntax: CSEL cond, Rn, Rm, Rd
// ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5], Rd in bits[4:0]. // ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5], Rd in bits[4:0].
if len(ops) != 4 { if len(ops) != 4 {
@@ -1324,5 +1388,6 @@ func AssembleFileARM64(f *ast.File) (*Image, error) {
}) })
} }
markExternals(img, dataSyms)
return img, nil return img, nil
} }
+3 -80
View File
@@ -9,7 +9,7 @@ package asm
// an opcode constant, and the format selects the bit layout. The opcode // an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own arm64 // constants and formats are transcribed from the Go toolchain's own arm64
// backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm` // backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm`
// exactly — the ground-truth oracle for the verify suite. // exactly, the ground-truth oracle for the verify suite.
// //
// All AArch64 instructions are 32 bits, little-endian. The formats used here // All AArch64 instructions are 32 bits, little-endian. The formats used here
// (per the ARM Architecture Reference Manual): // (per the ARM Architecture Reference Manual):
@@ -28,7 +28,7 @@ package asm
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd // ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
// arm64RegNum returns the 5-bit register number for an AArch64 register name: // arm64RegNum returns the 5-bit register number for an AArch64 register name:
// R0–R30 (integer), F0–F31 (floating point), and the ABI aliases the // R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
// runtime's assembly uses. Returns -1 for an unrecognised name. // runtime's assembly uses. Returns -1 for an unrecognised name.
func arm64RegNum(name string) int { func arm64RegNum(name string) int {
switch name { switch name {
@@ -99,7 +99,7 @@ func arm64RegNum(name string) int {
case "SP": case "SP":
return 31 // SP and ZR share encoding 31; context determines meaning return 31 // SP and ZR share encoding 31; context determines meaning
} }
// F0–F31. // F0-F31.
if len(name) >= 1 && name[0] == 'F' { if len(name) >= 1 && name[0] == 'F' {
n := 0 n := 0
for i := 1; i < len(name); i++ { for i := 1; i < len(name); i++ {
@@ -115,12 +115,6 @@ func arm64RegNum(name string) int {
return -1 return -1
} }
// arm64IsSP reports whether a register operand is the stack pointer (R31/SP),
// which uses a different encoding path for some instructions.
func arm64IsSP(name string) bool {
return name == "SP"
}
// ---- format helpers ---- // ---- format helpers ----
// a64wordLE encodes a uint32 as 4 little-endian bytes. // a64wordLE encodes a uint32 as 4 little-endian bytes.
@@ -137,14 +131,6 @@ func a64WordsLE(ws ...uint32) []byte {
return out return out
} }
// ---- data-processing (shifted register) ----
// a64DPSR encodes a data-processing (shifted register) instruction:
// sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd.
func a64DPSR(sf, op, S, shift, rm, imm6, rn, rd uint32) uint32 {
return sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | rm<<16 | imm6<<10 | rn<<5 | rd
}
// ---- data-processing (immediate) ---- // ---- data-processing (immediate) ----
// a64AddSub encodes an ADD/SUB (immediate) instruction: // a64AddSub encodes an ADD/SUB (immediate) instruction:
@@ -153,14 +139,6 @@ func a64AddSub(sf, op, S, sh, imm12, rn, rd uint32) uint32 {
return sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | rn<<5 | rd return sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | rn<<5 | rd
} }
// ---- logical (immediate) ----
// a64LogicalImm encodes a logical (immediate) instruction:
// sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd.
func a64LogicalImm(sf, opc, N, immr, imms, rn, rd uint32) uint32 {
return sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | rn<<5 | rd
}
// ---- move wide ---- // ---- move wide ----
// a64MoveWide encodes a MOVZ/MOVK/MOVN instruction: // a64MoveWide encodes a MOVZ/MOVK/MOVN instruction:
@@ -198,38 +176,6 @@ func a64LSP(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
return opc<<30 | 5<<27 | V<<26 | 2<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt return opc<<30 | 5<<27 | V<<26 | 2<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
} }
// ---- load/store pair (pre-index) ----
// a64LSPPre encodes a load/store pair (pre-index):
// opc<<30 | 0x5<<27 | V<<26 | 0b11<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
func a64LSPPre(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
return opc<<30 | 5<<27 | V<<26 | 3<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
}
// ---- load/store pair (post-index) ----
// a64LSPPost encodes a load/store pair (post-index):
// opc<<30 | 0x5<<27 | V<<26 | 0b01<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
func a64LSPPost(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
return opc<<30 | 5<<27 | V<<26 | 1<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
}
// ---- pre-index load/store ----
// a64LSPreIndex encodes a load/store register (pre-index):
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func a64LSPreIndex(size, V, opc uint32, imm9 int32, rn, rt uint32) uint32 {
return size<<30 | 7<<27 | V<<26 | opc<<22 | 3<<10 | (uint32(imm9)&0x1FF)<<12 | rn<<5 | rt
}
// ---- post-index load/store ----
// a64LSPostIndex encodes a load/store register (post-index):
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func a64LSPostIndex(size, V, opc uint32, imm9 int32, rn, rt uint32) uint32 {
return size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<10 | (uint32(imm9)&0x1FF)<<12 | rn<<5 | rt
}
// ---- branches ---- // ---- branches ----
// a64Branch encodes an unconditional branch (B/BL): // a64Branch encodes an unconditional branch (B/BL):
@@ -259,14 +205,6 @@ func a64ADR(p uint32, immhi int32, immlo uint32, rd uint32) uint32 {
return p<<31 | immlo<<29 | 0x10<<24 | (uint32(immhi)&0x7FFFF)<<5 | rd return p<<31 | immlo<<29 | 0x10<<24 | (uint32(immhi)&0x7FFFF)<<5 | rd
} }
// ---- EXTR ----
// a64EXTR encodes an EXTR instruction:
// sf<<31 | 0<<29 | 0x27<<23 | N<<22 | 0<<21 | Rm<<16 | imms<<10 | Rn<<5 | Rd.
func a64EXTR(sf, N, rm, imms, rn, rd uint32) uint32 {
return sf<<31 | 0x27<<23 | N<<22 | rm<<16 | imms<<10 | rn<<5 | rd
}
// ---- system ---- // ---- system ----
// a64NOP encodes a NOP: 0xd503201f. // a64NOP encodes a NOP: 0xd503201f.
@@ -296,8 +234,6 @@ const (
a64CondLT = 0xb a64CondLT = 0xb
a64CondGT = 0xc a64CondGT = 0xc
a64CondLE = 0xd a64CondLE = 0xd
a64CondAL = 0xe
a64CondNV = 0xf
) )
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes. // arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
@@ -359,7 +295,6 @@ const (
type a64Enc struct { type a64Enc struct {
format a64Format format a64Format
op uint32 // the pre-positioned opcode bits op uint32 // the pre-positioned opcode bits
size int // 4 for most, 8 for DP-imm with shift, etc.
} }
// a64InstrTable maps AArch64 mnemonics (as the Go assembler spells them) to // a64InstrTable maps AArch64 mnemonics (as the Go assembler spells them) to
@@ -717,18 +652,6 @@ func a64StoreOpc(t a64LSType) int {
return 0 // integer store return 0 // integer store
} }
// a64MovRegTable maps register-to-register MOV mnemonic expansions.
// The Go toolchain encodes MOV Rn, Rd as ORR Rn, ZR, Rd.
var a64MovRegTable = map[string]uint32{
"MOVD": 1<<31 | 1<<29 | 0x0a<<24, // ORR 64-bit
"MOVW": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVB": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit (byte move)
"MOVBU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVH": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVHU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVWU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
}
// arm64RegClass discriminates integer (R), floating-point (F) registers for // arm64RegClass discriminates integer (R), floating-point (F) registers for
// the MOV pseudo-instruction. // the MOV pseudo-instruction.
type arm64RegClass int type arm64RegClass int
+186 -14
View File
@@ -63,6 +63,12 @@ type arm64FrameInfo struct {
args int // the declared -argsize args int // the declared -argsize
noSplit bool // the NOSPLIT flag noSplit bool // the NOSPLIT flag
leaf bool // no call instructions in the body leaf bool // no call instructions in the body
// Stack-split guard state: needSplit mirrors the toolchain, which skips
// the check for NOSPLIT functions and auto-marks leaf functions with an
// autosize below StackSmall as NOSPLIT.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
} }
// arm64ComputeFrame derives the frame layout for a TEXT function. // arm64ComputeFrame derives the frame layout for a TEXT function.
@@ -80,15 +86,68 @@ func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
if fi.frame != 0 || !fi.leaf { if fi.frame != 0 || !fi.leaf {
fi.autosize = fi.frame + 8 // space for the saved LR fi.autosize = fi.frame + 8 // space for the saved LR
if fi.autosize%16 != 0 { // The toolchain always adds an extrasize: 8 when the total leaves a
// The toolchain aligns to 16: if autosize%16 == 8, add 8; // 16-byte alignment gap, another 16 when already aligned.
// otherwise add whatever is needed. switch fi.autosize % 16 {
case 8:
fi.autosize += 8
case 0:
fi.autosize += 16
default:
// The toolchain rejects unaligned frames; round up so such
// sources still assemble.
fi.autosize += 16 - (fi.autosize % 16) fi.autosize += 16 - (fi.autosize % 16)
} }
} }
switch {
case fi.noSplit:
case fi.autosize < stackSmall && fi.leaf:
// Auto-NOSPLIT, as the toolchain's leaf mark concludes.
default:
fi.needSplit = true
switch {
case fi.autosize <= stackSmall:
fi.splitClass = 0
case fi.autosize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi return fi
} }
// arm64GuardLen returns the byte length of the stack-split guard prefix
// (zero when the function needs no guard). The big class materialises
// framesize-StackSmall into REGTMP, whose MOVZ/MOVK sequence length varies.
func arm64GuardLen(fi arm64FrameInfo) int {
if !fi.needSplit {
return 0
}
switch fi.splitClass {
case 0:
return 12
case 1:
return 16
default:
n, err := arm64LoadImmLen(int64(fi.autosize - stackSmall))
if err != nil {
return 0
}
return 4 + n + 4 + 4 + 4 + 4
}
}
// arm64LoadImmLen returns the byte length of the MOVZ/MOVK sequence that
// loads v into a register.
func arm64LoadImmLen(v int64) (int, error) {
b, err := encodeARM64LoadImm(27, v, "MOVD")
if err != nil {
return 0, err
}
return len(b), nil
}
// arm64IsLeaf reports whether a function contains no call instructions // arm64IsLeaf reports whether a function contains no call instructions
// (BL/CALL), matching the toolchain's LEAF mark. // (BL/CALL), matching the toolchain's LEAF mark.
func arm64IsLeaf(t *ast.Text) bool { func arm64IsLeaf(t *ast.Text) bool {
@@ -119,12 +178,47 @@ func arm64Prologue(fi arm64FrameInfo) []byte {
) )
} }
// Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP // Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP
return a64WordsLE( ws := arm64SubImmWords(uint32(fi.autosize), 20)
a64AddSub(1, 1, 0, 0, uint32(fi.autosize), 31, 20), // SUB $autosize, SP, R20 ws = append(ws,
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair) a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP) a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB) a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
) )
return a64WordsLE(ws...)
}
// arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value
// fits the imm12 field (plain, or shifted left by 12 when it is a multiple
// of 4096); otherwise the toolchain materialises it into REGTMP (R27) and
// subtracts the register in the extended-register form.
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 0xFFF {
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
}
if imm <= 4095<<12 && imm&0xFFF == 0 {
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
}
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
}
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12
// and REGTMP fallback ladder.
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 0xFFF {
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
}
if imm <= 4095<<12 && imm&0xFFF == 0 {
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
}
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
} }
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and // arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
@@ -134,10 +228,8 @@ func arm64Return(fi arm64FrameInfo) []byte {
if fi.autosize != 0 { if fi.autosize != 0 {
if fi.leaf { if fi.leaf {
// Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP // Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP
ws = append(ws, ws = append(ws, arm64AddImmWords(uint32(fi.autosize-8), 29)...)
a64AddSub(1, 0, 0, 0, uint32(fi.autosize-8), 31, 29), // ADD $autosize-8, SP, FP ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
)
} else if fi.autosize <= 0xf0 { } else if fi.autosize <= 0xf0 {
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize // Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize
ws = append(ws, ws = append(ws,
@@ -147,9 +239,9 @@ func arm64Return(fi arm64FrameInfo) []byte {
} else { } else {
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP // Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
ws = append(ws, ws = append(ws,
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair) a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
) )
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
} }
} }
// RET: BR LR (0xd65f03c0) // RET: BR LR (0xd65f03c0)
@@ -235,3 +327,83 @@ func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 | return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31) 1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
} }
// Data-processing (shifted register) base opcodes for the guard blocks.
const (
arm64OpAdd = 1<<31 | 0<<30 | 0<<29 | 0x0b<<24
arm64OpSub = 1<<31 | 1<<30 | 0<<29 | 0x0b<<24
arm64OpSubs = 1<<31 | 1<<30 | 1<<29 | 0x0b<<24
)
// arm64DPSRWords builds one data-processing (shifted register) word:
// OP Rm, Rn, Rd in the Go assembler's operand order.
func arm64DPSRWords(base uint32, rm, rn, rd uint32) uint32 {
return base | rm<<16 | rn<<5 | rd
}
// arm64DPExtWords builds one data-processing (extended register) word, the
// form the toolchain picks when a large immediate was materialised into
// REGTMP before the operation: base | 1<<21 | Rm<<16 | UXTX<<13 | Rn<<5 | Rd.
func arm64DPExtWords(base, rm, rn, rd uint32) uint32 {
return base | 1<<21 | rm<<16 | 3<<13 | rn<<5 | rd
}
// wordsOf converts little-endian instruction bytes back to words.
func wordsOf(b []byte) []uint32 {
ws := make([]uint32, 0, len(b)/4)
for i := 0; i+4 <= len(b); i += 4 {
ws = append(ws, uint32(b[i])|uint32(b[i+1])<<8|uint32(b[i+2])<<16|uint32(b[i+3])<<24)
}
return ws
}
// arm64GuardBytes emits the stack-split guard prefix; blockStart is the
// function-relative byte address of the morestack block the branches target.
func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte {
// MOVD 16(R28), R16 (g.stackguard0)
ws := []uint32{a64LSU(3, 0, 1, 2, 28, 16)}
br := func(from int, cond uint32) uint32 {
return a64BranchCond(int32((blockStart-from)>>2), cond)
}
switch fi.splitClass {
case 0:
// CMP R16, RSP in the exact encoding go tool asm emits for it.
ws = append(ws, 0xeb3063ff)
ws = append(ws, br(8, a64CondLS))
case 1:
ws = append(ws, a64AddSub(1, 1, 0, 0, uint32(fi.autosize-stackSmall), 31, 17))
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
ws = append(ws, br(12, a64CondLS))
default:
mov, err := encodeARM64LoadImm(27, int64(fi.autosize-stackSmall), "MOVD")
if err != nil {
mov = nil
}
ws = append(ws, wordsOf(mov)...)
ml := len(mov) / 4
ws = append(ws, arm64DPExtWords(arm64OpSubs, 27, 31, 17)) // SUBS R17, RSP, R27
ws = append(ws, br(8+ml, a64CondLO))
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
ws = append(ws, br(8+ml+8, a64CondLS))
}
return a64WordsLE(ws...)
}
// arm64MoreStackBlock emits the trailing block: MOVD R30, R3 (save LR),
// BL runtime.morestack_noctxt, B back to the function start. The BL carries
// the R_CALLARM64 relocation.
func arm64MoreStackBlock(blockStart int) ([]byte, Reloc) {
ws := []uint32{
1<<31 | 1<<29 | 0x0a<<24 | 30<<16 | 31<<5 | 3, // MOVD R30, R3
a64Branch(1, 0), // BL, patched by the linker
}
bPC := blockStart + 8
ws = append(ws, a64Branch(0, int32(-bPC>>2))) // B back to the entry
reloc := Reloc{
Off: blockStart + 4,
After: blockStart + 8,
Name: "runtime\u00b7morestack_noctxt",
Kind: RelArm64Branch,
}
return a64WordsLE(ws...), reloc
}
+126
View File
@@ -0,0 +1,126 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// parseArm64File is a helper assembling one arm64 source file.
func parseArm64File(t *testing.T, src string) *Image {
t.Helper()
f, errs := parser.Parse("k_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
return img
}
// TestArm64RelocOffsetsIncludePrologue pins the function-relative relocation
// offsets of a framed function: the offsets used to exclude the prologue, so
// every relocation landed on a prologue instruction in the GOOBJ/ELF output.
// The function calls an external, so it is a non-leaf and carries the
// stack-split guard (12 bytes, small class) before the prologue.
func TestArm64RelocOffsetsIncludePrologue(t *testing.T) {
img := parseArm64File(t, "TEXT \u00b7f(SB), $16-0\n"+
"\tBL ext\u00b7foo(SB)\n"+
"\tMOVD $gdata(SB), R5\n"+
"\tMOVD $extsym(SB), R6\n"+
"\tRET\n"+
"GLOBL gdata(SB), $8\n")
fn := img.Funcs[0]
// Layout: 12-byte guard, 12-byte prologue, BL (24), ADRP+ADD (28, 32),
// ADRP+ADD (36, 40), 12-byte epilogue with RET, 12-byte morestack block.
want := []struct {
off int
after int
name string
kind RelocKind
external bool
}{
{24, 28, "foo", RelArm64Branch, true},
{28, 28, "gdata", RelArm64Addr, false},
{32, 32, "gdata", RelArm64Addr, false},
{36, 36, "extsym", RelArm64Addr, true},
{40, 40, "extsym", RelArm64Addr, true},
{60, 64, "runtime\u00b7morestack_noctxt", RelArm64Branch, true},
}
if len(fn.Relocs) != len(want) {
t.Fatalf("relocs = %d, want %d", len(fn.Relocs), len(want))
}
for i, w := range want {
r := fn.Relocs[i]
if r.Off != w.off || r.After != w.after || r.Name != w.name || r.Kind != w.kind || r.External != w.external {
t.Errorf("reloc %d = {off %d after %d name %q kind %d ext %v}, want {off %d after %d name %q kind %d ext %v}",
i, r.Off, r.After, r.Name, r.Kind, r.External, w.off, w.after, w.name, w.kind, w.external)
}
}
// The BL with a zero offset sits exactly at the first reloc site.
code := img.Code[fn.Offset : fn.Offset+fn.Size]
if w := binary.LittleEndian.Uint32(code[24:28]); w != 0x94000000 {
t.Errorf("BL word = %08x, want 94000000", w)
}
}
// TestArm64SBLoadStoreMatchesToolchain pins the ADRP scratch register
// (REGTMP, R27) and the LDST64 relocation kind for sym loads and stores,
// against the bytes go tool asm emits for MOVD sym(SB), R5.
func TestArm64SBLoadStoreMatchesToolchain(t *testing.T) {
img := parseArm64File(t, "TEXT \u00b7ld(SB), NOSPLIT, $0\n"+
"\tMOVD sym(SB), R5\n"+
"\tMOVD R5, sym(SB)\n"+
"\tRET\n"+
"GLOBL sym(SB), $8\n")
fn := img.Funcs[0]
code := img.Code[fn.Offset : fn.Offset+fn.Size]
// go tool asm: ADRP 0(PC), R27 (9000001b); MOVD (R27), R5 (f9400365);
// ADRP 0(PC), R27; MOVD R5, (R27) (f9000365).
for off, want := range map[int]uint32{0: 0x9000001b, 4: 0xf9400365, 8: 0x9000001b, 12: 0xf9000365} {
if got := binary.LittleEndian.Uint32(code[off : off+4]); got != want {
t.Errorf("word at %d = %08x, want %08x", off, got, want)
}
}
if len(fn.Relocs) != 2 {
t.Fatalf("relocs = %d, want 2", len(fn.Relocs))
}
for i, w := range []struct{ off, after int }{{0, 8}, {8, 16}} {
r := fn.Relocs[i]
if r.Kind != RelArm64LDST64 {
t.Errorf("reloc %d kind = %d, want RelArm64LDST64 (%d)", i, r.Kind, RelArm64LDST64)
}
if r.Off != w.off || r.After != w.after {
t.Errorf("reloc %d = {off %d after %d}, want {off %d after %d}", i, r.Off, r.After, w.off, w.after)
}
}
}
// TestArm64GOObjRelocTypes checks that GOOBJ emission succeeds with the new
// relocation kinds in play; the detailed layout is covered by the goobj tests.
func TestArm64GOObjRelocTypes(t *testing.T) {
img := parseArm64File(t, "TEXT \u00b7ld(SB), NOSPLIT, $0\n"+
"\tMOVD sym(SB), R5\n"+
"\tMOVD R5, sym(SB)\n"+
"\tRET\n"+
"GLOBL sym(SB), $8\n")
obj, err := img.GOObjectAARCH64("testpkg", "k_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
if len(obj) == 0 {
t.Fatal("empty object")
}
// The detailed layout is covered by the goobj tests; here we only pin
// that emission succeeds with the new relocation kinds in play.
}
+299 -9
View File
@@ -22,6 +22,11 @@ import (
// FP/SP frame-relative operands, and local-label jumps. SB (global symbol) // FP/SP frame-relative operands, and local-label jumps. SB (global symbol)
// operands require relocations and are not yet supported; the SIMD (VEX/AVX2) // operands require relocations and are not yet supported; the SIMD (VEX/AVX2)
// integer and shuffle/extract/permute/move set is in. // integer and shuffle/extract/permute/move set is in.
//
// Like the other architectures, the stack-growth guard (the morestack check
// in the prologue and the call back into the runtime in the epilogue) is not
// emitted: the bytes match go tool asm only for NOSPLIT functions or
// zero-frame leaves, where the toolchain emits no guard either.
func Assemble(t *ast.Text) ([]byte, map[string]int, error) { func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
code, _, labels, _, _, err := assemble(t, nil) code, _, labels, _, _, err := assemble(t, nil)
return code, labels, err return code, labels, err
@@ -31,7 +36,7 @@ func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
// the set of static symbols a GLOBL in the same file defines. A nil link // the set of static symbols a GLOBL in the same file defines. A nil link
// rejects SB operands outright (single-function assembly cannot resolve // rejects SB operands outright (single-function assembly cannot resolve
// them). When allowExternal is set, a reference to a symbol no GLOBL in the // them). When allowExternal is set, a reference to a symbol no GLOBL in the
// file defines is recorded as an external relocation instead of failing — // file defines is recorded as an external relocation instead of failing
// the object-file emitters resolve it at link time. // the object-file emitters resolve it at link time.
type linkInfo struct { type linkInfo struct {
symbols map[string]bool symbols map[string]bool
@@ -46,6 +51,7 @@ type sbPatch struct {
after int after int
name string name string
addend int64 addend int64
kind RelocKind
} }
// spadjStep is one stack-adjustment boundary within a function: Value is the // spadjStep is one stack-adjustment boundary within a function: Value is the
@@ -70,13 +76,18 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
return name return name
} }
// Layout: iterate jump sizes to a fixed point. // Layout: iterate jump sizes to a fixed point. The stack-split guard
// prefix and the trailing morestack block participate in the iteration:
// their conditional branches relax from rel8 to rel32 when the body
// outgrows the short form.
long := make([]bool, len(t.Body)) long := make([]bool, len(t.Body))
sizes := make([]int, len(t.Body)) sizes := make([]int, len(t.Body))
offsets := map[string]int{} offsets := map[string]int{}
pcs := make([]int, len(t.Body)) pcs := make([]int, len(t.Body))
var guardJBlong, guardJBElong, moreJMPlong bool
for { for {
pos := len(fi.prologue) guard := fi.guardLen(guardJBlong, guardJBElong)
pos := guard + len(fi.prologue)
for i, stmt := range t.Body { for i, stmt := range t.Body {
switch s := stmt.(type) { switch s := stmt.(type) {
case *ast.Label: case *ast.Label:
@@ -91,6 +102,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
pos += sz pos += sz
} }
} }
bodyLen := pos - (guard + len(fi.prologue))
// Expand any short jump whose displacement no longer fits rel8. // Expand any short jump whose displacement no longer fits rel8.
changed := false changed := false
for i, stmt := range t.Body { for i, stmt := range t.Body {
@@ -116,25 +128,75 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
changed = true changed = true
} }
} }
// The guard's conditional branches target the morestack block, which
// starts right after the body: the JBE measures from the end of the
// guard, so its displacement is the prologue plus the body.
if !guardJBElong && !fits8(int64(len(fi.prologue)+bodyLen)) {
guardJBElong = true
changed = true
}
if fi.splitClass == 2 && !guardJBlong {
// The underflow JB sits before the CMPQ; its displacement spans
// the rest of the guard plus the prologue and the body.
jbLen := 2
if guardJBlong {
jbLen = 6
}
rest := fi.guardLen(guardJBlong, guardJBElong) - (9 + 3 + 7 + jbLen)
if !fits8(int64(rest + len(fi.prologue) + bodyLen)) {
guardJBlong = true
changed = true
}
}
// The morestack JMP returns to the function start, so its
// displacement is the negated distance from its own end.
if !moreJMPlong {
jmpLen := 2
if moreJMPlong {
jmpLen = 5
}
if !fits8(-int64(guard + len(fi.prologue) + bodyLen + 5 + jmpLen)) {
moreJMPlong = true
changed = true
}
}
if !changed { if !changed {
break break
} }
} }
// Pass 2: emit. // Pass 2: emit. The guard comes first, then the prologue, the body and
out := append([]byte(nil), fi.prologue...) // the morestack block.
guardLen := fi.guardLen(guardJBlong, guardJBElong)
bodyLen := 0
{
pos := guardLen + len(fi.prologue)
for i, stmt := range t.Body {
if _, ok := stmt.(*ast.Instr); ok {
pos += sizes[i]
}
}
bodyLen = pos - (guardLen + len(fi.prologue))
}
var out []byte
var patches []sbPatch var patches []sbPatch
if fi.needSplit {
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+2)+len(fi.prologue)+bodyLen))
out = append(out, guard...)
patches = append(patches, tlsPatch)
}
out = append(out, fi.prologue...)
var steps []spadjStep var steps []spadjStep
var lines []LineEntry var lines []LineEntry
if fi.useFP { if fi.useFP {
// PUSHQ BP saves the return-address-relative base (+8); the MOVQ // PUSHQ BP saves the return-address-relative base (+8); the MOVQ
// changes nothing; SUBQ $size, SP completes the frame. // changes nothing; SUBQ $size, SP completes the frame.
steps = append(steps, steps = append(steps,
spadjStep{1, 8}, spadjStep{guardLen + 1, 8},
spadjStep{len(fi.prologue), 8 + fi.size}, spadjStep{guardLen + len(fi.prologue), 8 + fi.size},
) )
} }
pos := len(fi.prologue) pos := guardLen + len(fi.prologue)
for i, stmt := range t.Body { for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr) s, ok := stmt.(*ast.Instr)
if !ok { if !ok {
@@ -156,11 +218,32 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
if len(code) != sizes[i] { if len(code) != sizes[i] {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i]) return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
} }
if strings.ToUpper(s.Mnemonic.Text) == "CALL" {
for k := range ps {
ps[k].kind = RelCall
}
}
patches = append(patches, ps...) patches = append(patches, ps...)
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line}) lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
out = append(out, code...) out = append(out, code...)
pos += len(code) pos += len(code)
} }
if fi.needSplit {
// The morestack block: CALL runtime.morestack_noctxt, then a JMP
// back to the function entry.
jmpLen := 2
if moreJMPlong {
jmpLen = 5
}
jmpDisp := -int64(pos + 5 + jmpLen)
suffix, callPatch := buildMoreStack(int32(jmpDisp))
callPatch.off += pos
callPatch.after = pos + 5
patches = append(patches, callPatch)
out = append(out, suffix...)
pos += len(suffix)
}
_ = pos
return out, patches, offsets, steps, lines, nil return out, patches, offsets, steps, lines, nil
} }
@@ -224,10 +307,31 @@ type frameInfo struct {
spAdjust int64 // x-N(SP) becomes (spAdjust - N)(SP) spAdjust int64 // x-N(SP) becomes (spAdjust - N)(SP)
prologue []byte prologue []byte
epilogue []byte epilogue []byte
// Stack-split guard state (matching the toolchain's stacksplit): needSplit
// is false for NOSPLIT functions and for leaf functions whose frame is
// below StackSmall, which the toolchain auto-marks NOSPLIT.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
framesize int // the size the guard checks: frame+8 for framed functions
} }
// Stack-frame size classes from runtime/stack.go.
const (
stackSmall = 128
stackBig = 4096
)
// sbPatch gains a kind so the emitters can tell CALL and TLS patches from
// plain PC-relative displacements.
// computeFrame derives the frame layout, matching the Go assembler's default // computeFrame derives the frame layout, matching the Go assembler's default
// (a frame pointer is used whenever the function has a non-zero frame). // (a frame pointer is used whenever the function has a non-zero frame). It
// also decides whether the function needs the stack-split guard, mirroring
// obj6: a NOSPLIT function never splits, and a leaf function whose frame is
// below StackSmall is auto-marked NOSPLIT. One deliberate deviation: the
// toolchain treats zero-argument runtime calls (duffcopy and friends) as
// leaf-compatible; here any CALL makes the function a non-leaf.
func computeFrame(t *ast.Text) frameInfo { func computeFrame(t *ast.Text) frameInfo {
fi := frameInfo{} fi := frameInfo{}
if t.Frame != nil && t.Frame.Imm.HasVal { if t.Frame != nil && t.Frame.Imm.HasVal {
@@ -242,9 +346,141 @@ func computeFrame(t *ast.Text) frameInfo {
} else { } else {
fi.fpAdjust = 8 // return address only fi.fpAdjust = 8 // return address only
} }
noSplit := false
for _, f := range t.Flags {
if strings.EqualFold(f, "NOSPLIT") {
noSplit = true
}
}
// The toolchain's autoffset: the frame plus the saved base pointer.
framesize := fi.size
if framesize > 0 {
framesize += 8
}
switch {
case noSplit:
case framesize < stackSmall && !hasCall(t):
// Auto-NOSPLIT, as the toolchain's leaf search concludes.
default:
fi.needSplit = true
fi.framesize = framesize
switch {
case framesize <= stackSmall:
fi.splitClass = 0
case framesize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi return fi
} }
// hasCall reports whether the function body contains a CALL instruction.
func hasCall(t *ast.Text) bool {
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
if strings.ToUpper(in.Mnemonic.Text) == "CALL" {
return true
}
}
return false
}
// guardLen returns the byte length of the stack-split guard prefix. The
// final conditional branch (JBE, and JB in the big class) is 2 bytes in the
// short form and 6 in the long form.
func (fi frameInfo) guardLen(jbLong, jbeLong bool) int {
if !fi.needSplit {
return 0
}
jb, jbe := 2, 2
if jbLong {
jb = 6
}
if jbeLong {
jbe = 6
}
switch fi.splitClass {
case 0:
return 9 + 4 + jbe
case 1:
return 9 + 8 + 4 + jbe
default:
return 9 + 3 + 7 + jb + 4 + jbe
}
}
// moreLen returns the byte length of the trailing morestack block: the CALL
// (always rel32) plus the JMP back to the function start.
func moreLen(jmpLong bool) int {
jmp := 2
if jmpLong {
jmp = 5
}
return 5 + jmp
}
// buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the
// already-computed displacements of the conditional branches that jump to the
// morestack block (unused in classes without them). The TLS load carries a
// R_TLS_LE patch site at offset 5.
func buildGuard(fi frameInfo, jbeDisp, jbDisp int32) ([]byte, sbPatch) {
out := []byte{
0x64, 0x4c, 0x8b, 0x34, 0x25, // MOVQ FS:0, R14
0, 0, 0, 0, // TLS slot offset, filled by the linker
}
tls := sbPatch{off: 5, after: 9, kind: RelTLSLE}
jmp := func(op8, op32 byte, disp int32) []byte {
if disp >= -128 && disp <= 127 {
return []byte{op8, byte(disp)}
}
return append([]byte{0x0F, op32}, le32(int64(disp))...)
}
switch fi.splitClass {
case 0:
// CMPQ SP, 16(R14)
out = append(out, 0x49, 0x3b, 0x66, 0x10)
out = append(out, jmp(0x76, 0x86, jbeDisp)...)
case 1:
// LEAQ -(framesize-StackSmall)(SP), R12; CMPQ R12, 16(R14)
out = append(out, 0x4c, 0x8d, 0xa4, 0x24)
out = append(out, le32(-int64(fi.framesize-stackSmall))...)
out = append(out, 0x4d, 0x3b, 0x66, 0x10)
out = append(out, jmp(0x76, 0x86, jbeDisp)...)
default:
// MOVQ SP, R12; SUBQ $(framesize-StackSmall), R12; JB; CMPQ R12, 16(R14)
out = append(out, 0x49, 0x89, 0xe4)
out = append(out, 0x49, 0x81, 0xec)
out = append(out, le32(int64(fi.framesize-stackSmall))...)
out = append(out, jmp(0x72, 0x82, jbDisp)...)
out = append(out, 0x4d, 0x3b, 0x66, 0x10)
out = append(out, jmp(0x76, 0x86, jbeDisp)...)
}
return out, tls
}
// buildMoreStack emits the trailing block: CALL runtime.morestack_noctxt
// (patched by the linker) and a JMP back to the function start.
func buildMoreStack(jmpDisp int32) ([]byte, sbPatch) {
out := []byte{0xE8, 0, 0, 0, 0}
call := sbPatch{off: 1, after: 5, name: "runtime\u00b7morestack_noctxt", kind: RelCall}
out = append(out, jmpBytes(jmpDisp)...)
return out, call
}
// jmpBytes encodes a near JMP in the short or long form.
func jmpBytes(disp int32) []byte {
if disp >= -128 && disp <= 127 {
return []byte{0xEB, byte(disp)}
}
return append([]byte{0xE9}, le32(int64(disp))...)
}
// prologueBytes emits: PUSHQ BP; MOVQ SP, BP; SUBQ $size, SP. // prologueBytes emits: PUSHQ BP; MOVQ SP, BP; SUBQ $size, SP.
func prologueBytes(size int) []byte { func prologueBytes(size int) []byte {
out := []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP out := []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
@@ -258,6 +494,8 @@ func epilogueBytes(size int) []byte {
} }
func subSP(size int) []byte { // SUBQ $size, SP func subSP(size int) []byte { // SUBQ $size, SP
// imm8 holds -128..127; anything larger takes the imm32 form, exactly as
// the Go assembler encodes it (verified for 8, 128, 200 and 255).
if size >= -128 && size <= 127 { if size >= -128 && size <= 127 {
return []byte{0x48, 0x83, 0xEC, byte(int8(size))} return []byte{0x48, 0x83, 0xEC, byte(int8(size))}
} }
@@ -277,6 +515,9 @@ func addSP(size int) []byte { // ADDQ $size, SP
func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, error) { func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, error) {
mnem := strings.ToUpper(s.Mnemonic.Text) mnem := strings.ToUpper(s.Mnemonic.Text)
if isJumpMnemonic(mnem) { if isJumpMnemonic(mnem) {
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
return 5, nil // opcode + rel32, always the long form
}
return jumpSize(mnem, long), nil return jumpSize(mnem, long), nil
} }
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link) code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
@@ -326,6 +567,24 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
var ps []sbPatch var ps []sbPatch
var err error var err error
if isJumpMnemonic(mnem) { if isJumpMnemonic(mnem) {
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
// CALL/JMP sym(SB): a rel32 call (or tail call) against a
// static or external symbol, resolved by the file-level layout
// or the linker.
code, ps, err = encodeSBCall(s, link)
if err != nil {
return nil, nil, err
}
for i := range ps {
ps[i].kind = RelCall
}
body := pc + len(prefix)
for i := range ps {
ps[i].off += body
ps[i].after = body + len(code)
}
return append(prefix, code...), ps, nil
}
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve) code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
} else { } else {
code, ps, err = encodeNormal(s, fi, link) code, ps, err = encodeNormal(s, fi, link)
@@ -407,6 +666,37 @@ func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long
} }
} }
// isSBCall reports whether the CALL operand is a symbol reference.
func isSBCall(s *ast.Instr) bool {
return len(s.Operands) == 1 && s.Operands[0].Kind == ast.OpAddr &&
s.Operands[0].Addr.Sym != nil && s.Operands[0].Addr.Sym.Pseudo == "SB"
}
// encodeSBCall encodes CALL sym(SB) as E8 rel32 with a patch site.
func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) {
o, err := operandFromAST(s.Operands[0], 8, frameInfo{}, link)
if err != nil {
return nil, nil, err
}
m, ok := o.(sbMem)
if !ok {
return nil, nil, fmt.Errorf("CALL: unsupported operand")
}
opcode := []byte{0xE8}
if strings.ToUpper(s.Mnemonic.Text) == "JMP" {
opcode = []byte{0xE9} // a tail call, no return address pushed
}
e := &enc{}
if err := e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}}); err != nil {
return nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend, kind: RelCall}
}
return e.out, ps, nil
}
// labelName extracts a local-label name from a jump operand. // labelName extracts a local-label name from a jump operand.
func labelName(op *ast.Operand) (string, bool) { func labelName(op *ast.Operand) (string, bool) {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
+59 -1
View File
@@ -4,6 +4,7 @@
package asm package asm
import ( import (
"bytes"
"strings" "strings"
"testing" "testing"
@@ -62,7 +63,7 @@ TEXT ·f(SB), NOSPLIT, $0
XORQ AX, AX XORQ AX, AX
loop: loop:
ADDQ $1, AX ADDQ $1, AX
CMPQ $10, AX CMPQ AX, $10
JLT loop JLT loop
RET RET
`) `)
@@ -317,3 +318,60 @@ end:
t.Errorf("jump-folding mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) t.Errorf("jump-folding mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
} }
} }
func TestAssemblePrefetch(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·pf(SB), NOSPLIT, $0
PREFETCHNTA (AX)
PREFETCHT0 (BX)
PREFETCHT1 8(CX)
PREFETCHT2 -1(AX)(R12*1)
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
got := strings.Join(disasm(t, code), "\n")
want := strings.Join([]string{
"prefetchnta zmmword ptr [rax]",
"prefetcht0 zmmword ptr [rbx]",
"prefetcht1 zmmword ptr [rcx+0x8]",
"prefetcht2 zmmword ptr [rax+r12-0x1]",
"ret",
}, "\n")
if got != want {
t.Errorf("prefetch disassembly mismatch:\n got:\n%s\n want:\n%s", got, want)
}
// Byte-level expectations: 0F 18 with the variant in the reg field.
if hex := hexBytes(code[:3]); hex != "0f 18 00" {
t.Errorf("PREFETCHNTA bytes: got %s, want 0f 18 00", hex)
}
if hex := hexBytes(code[3:6]); hex != "0f 18 0b" {
t.Errorf("PREFETCHT0 bytes: got %s, want 0f 18 0b", hex)
}
}
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
// larger. The intermediate 129..255 range used to encode an ADD with a
// truncated immediate, moving SP the wrong way.
func TestSubSPEncodings(t *testing.T) {
for _, tt := range []struct {
size int
want []byte
}{
{8, []byte{0x48, 0x83, 0xEC, 0x08}},
{127, []byte{0x48, 0x83, 0xEC, 0x7F}},
{128, []byte{0x48, 0x81, 0xEC, 0x80, 0x00, 0x00, 0x00}},
{200, []byte{0x48, 0x81, 0xEC, 0xC8, 0x00, 0x00, 0x00}},
{255, []byte{0x48, 0x81, 0xEC, 0xFF, 0x00, 0x00, 0x00}},
{4096, []byte{0x48, 0x81, 0xEC, 0x00, 0x10, 0x00, 0x00}},
} {
got := subSP(tt.size)
if !bytes.Equal(got, tt.want) {
t.Errorf("subSP(%d) = %x, want %x", tt.size, got, tt.want)
}
}
}
+38 -8
View File
@@ -12,12 +12,11 @@ import (
// Image: a .text section holding the function bodies, a .data section // Image: a .text section holding the function bodies, a .data section
// holding the GLOBL initialisers, a symbol table with one symbol per TEXT // holding the GLOBL initialisers, a symbol table with one symbol per TEXT
// and GLOBL (file-local <> symbols are STB_LOCAL, the rest STB_GLOBAL), and // and GLOBL (file-local <> symbols are STB_LOCAL, the rest STB_GLOBAL), and
// a .rela.text relocation table — one R_X86_64_PC32 entry per static-symbol // a .rela.text relocation table, one R_X86_64_PC32 entry per static-symbol
// reference, internal references resolving against the local data symbols // reference, internal references resolving against the local data symbols
// and external ones against undefined globals. The output links with the // and external ones against undefined globals. The output links with the
// system toolchain (cc/ld) the way a hand-assembled .o would. // system toolchain (cc/ld) the way a hand-assembled .o would.
// ELF constants (ELF64, little-endian, System V).
const ( const (
elfClass64 = 2 elfClass64 = 2
elfDataLSB = 1 elfDataLSB = 1
@@ -36,18 +35,15 @@ const (
shfAlloc = 2 shfAlloc = 2
shfExecInstr = 4 shfExecInstr = 4
stbLocal = 0
stbGlobal = 1 stbGlobal = 1
sttNotype = 0
sttObject = 1 sttObject = 1
sttFunc = 2 sttFunc = 2
sttSection = 3 sttSection = 3
stInfoShift = 4 stInfoShift = 4
shnUndef = 0 rX8664PC32 = 2
rX8664TPOFF32 = 20
rX8664PC32 = 2
) )
// elfSym is one symbol-table entry in construction. // elfSym is one symbol-table entry in construction.
@@ -76,7 +72,7 @@ func (img *Image) ELFObject() ([]byte, error) {
// Build the symbol table: the null entry and the two section symbols // Build the symbol table: the null entry and the two section symbols
// come first, then the local symbols (static TEXT and GLOBL), then the // come first, then the local symbols (static TEXT and GLOBL), then the
// globals (exported TEXT and GLOBL, and the undefined externals) — ELF // globals (exported TEXT and GLOBL, and the undefined externals), ELF
// requires every local to precede every global, and sh_info records the // requires every local to precede every global, and sh_info records the
// boundary. symIdx maps a symbol name to its index for the relocations. // boundary. symIdx maps a symbol name to its index for the relocations.
var locals, globals []elfSym var locals, globals []elfSym
@@ -130,11 +126,19 @@ func (img *Image) ELFObject() ([]byte, error) {
type elfRela struct { type elfRela struct {
off uint64 off uint64
sym int sym int
typ uint32
addend int64 addend int64
} }
var relas []elfRela var relas []elfRela
for _, fn := range img.Funcs { for _, fn := range img.Funcs {
for _, r := range fn.Relocs { for _, r := range fn.Relocs {
var typ uint32 = rX8664PC32
if r.Kind == RelTLSLE {
// R_X86_64_TPOFF32 resolves to the local-exec TLS offset and
// carries no symbol.
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), sym: 0, typ: rX8664TPOFF32})
continue
}
idx, ok := symIdx[r.Name] idx, ok := symIdx[r.Name]
if !ok { if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name) return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
@@ -142,6 +146,7 @@ func (img *Image) ELFObject() ([]byte, error) {
relas = append(relas, elfRela{ relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off), off: uint64(fn.Offset + r.Off),
sym: idx, sym: idx,
typ: typ,
// R_X86_64_PC32 computes S + A − P with P the patch site; the // R_X86_64_PC32 computes S + A − P with P the patch site; the
// assembler measures the symbol from the instruction end, // assembler measures the symbol from the instruction end,
// After − Off bytes past the field, so the addend carries // After − Off bytes past the field, so the addend carries
@@ -160,6 +165,9 @@ func (img *Image) ELFObject() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} { for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n) stSections.add(n)
} }
for _, n := range dwarfSectionNames {
stSections.add(n)
}
// Section presence: .rela.text only when there are relocations. // Section presence: .rela.text only when there are relocations.
hasRela := len(relas) > 0 hasRela := len(relas) > 0
@@ -220,6 +228,17 @@ func (img *Image) ELFObject() ([]byte, error) {
shstrOff := len(out) shstrOff := len(out)
out = append(out, stSections.bytes()...) out = append(out, stSections.bytes()...)
// DWARF debug sections (no relocations, the linker resolves DWARF fixups).
dwAlign := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
if dw != nil {
nSections += 4 // .debug_abbrev, .debug_info, .debug_line, .debug_line_str
}
align(8) align(8)
shoff := len(out) shoff := len(out)
@@ -248,6 +267,17 @@ func (img *Image) ELFObject() ([]byte, error) {
} }
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0) putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers.
if dw != nil {
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
}
}
// The ELF header. // The ELF header.
hdr := out[:64] hdr := out[:64]
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0}) copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
+319
View File
@@ -0,0 +1,319 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
)
// DWARF5 section generation for ELF output. Unlike the GOOBJ path (where
// the linker assembles the final DWARF), the ELF path must emit complete,
// self-contained sections because the system linker only performs fixup
// relocations, not assembly.
// dwarfAbbrevTable returns the .debug_abbrev content: a single compilation
// unit with DW_TAG_compile_unit and DW_TAG_subprogram entries.
func dwarfAbbrevTable() []byte {
var b []byte
// Abbrev 1: DW_TAG_compile_unit
b = append(b, 1) // abbreviation code
b = append(b, 0x11) // DW_TAG_compile_unit
b = append(b, 1) // DW_CHILDREN_yes
b = appendUleb(b, 0x1b) // DW_AT_low_pc
b = appendUleb(b, 0x01) // DW_FORM_addr
b = appendUleb(b, 0x29) // DW_AT_high_pc
b = appendUleb(b, 0x07) // DW_FORM_data8
b = appendUleb(b, 0x10) // DW_AT_stmt_list
b = appendUleb(b, 0x25) // DW_FORM_sec_offset
b = appendUleb(b, 0x01) // DW_AT_name
b = appendUleb(b, 0x08) // DW_FORM_string
b = appendUleb(b, 0) // end of attributes
// Abbrev 2: DW_TAG_subprogram
b = append(b, 2) // abbreviation code
b = append(b, 0x2e) // DW_TAG_subprogram
b = append(b, 0) // DW_CHILDREN_no
b = appendUleb(b, 0x03) // DW_AT_name
b = appendUleb(b, 0x08) // DW_FORM_string
b = appendUleb(b, 0x11) // DW_AT_low_pc
b = appendUleb(b, 0x01) // DW_FORM_addr
b = appendUleb(b, 0x29) // DW_AT_high_pc
b = appendUleb(b, 0x07) // DW_FORM_data8
b = appendUleb(b, 0x3f) // DW_AT_frame_base
b = appendUleb(b, 0x18) // DW_FORM_exprloc
b = appendUleb(b, 0x3b) // DW_AT_decl_file
b = appendUleb(b, 0x0b) // DW_FORM_data1
b = appendUleb(b, 0x37) // DW_AT_decl_line
b = appendUleb(b, 0x0b) // DW_FORM_data1
b = appendUleb(b, 0x63) // DW_AT_external
b = appendUleb(b, 0x0b) // DW_FORM_flag
b = appendUleb(b, 0) // end of attributes
// End of table.
b = append(b, 0)
return b
}
// dwarfSections holds the generated DWARF section payloads and their
// relocations (byte offsets within .debug_info and .debug_line that need
// fixup against .text symbols).
type dwarfSections struct {
debugAbbrev []byte
debugInfo []byte
debugLine []byte
debugLineStr []byte
debugFrame []byte
// Relocations for .debug_info: (offset, symbol name, addend).
infoRelocs []dwarfReloc
// Relocations for .debug_line: (offset, symbol name, addend).
lineRelocs []dwarfReloc
}
type dwarfReloc struct {
off uint64
name string
addend int64
}
// emitDWARF generates complete DWARF5 sections for the image.
func emitDWARF(img *Image, srcFile string) *dwarfSections {
ds := &dwarfSections{}
ds.debugAbbrev = dwarfAbbrevTable()
// Build the string table for .debug_line_str.
lineStr := newElfStrtab()
lineStr.add(srcFile)
ds.debugLineStr = lineStr.bytes()
// Build .debug_line.
ds.debugLine = dwarfBuildLineSection(img, ds)
// Build .debug_info.
ds.debugInfo = dwarfBuildInfoSection(img, srcFile, ds)
// Build .debug_frame.
ds.debugFrame = dwarfBuildFrameSection(img)
return ds
}
// dwarfBuildLineSection builds a complete .debug_line section.
func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
var b []byte
le := binary.LittleEndian
// We'll build the header first, then the programs, then patch the length.
headerStart := len(b)
b = append(b, 0, 0, 0, 0) // unit_length (placeholder)
b = le.AppendUint16(b, 5) // version (DWARF5)
b = append(b, 8) // address_size
b = append(b, 0) // segment_selector_size
b = append(b, 0, 0, 0, 0) // header_length (placeholder)
// Line program parameters.
b = append(b, 1) // minimum_instruction_length
b = append(b, 1) // maximum_ops_per_instruction
b = append(b, 1) // default_is_stmt
b = append(b, byte(dwLineBase&0xFF)) // line_base (-4 as unsigned)
b = append(b, uint8(dwLineRange)) // line_range
b = append(b, uint8(dwOpcodeBase)) // opcode_base
// Standard opcode lengths (opcode 1..opcode_base-1).
b = append(b, 0, 1, 1, 1, 1, 0, 0, 0, 1, 0)
// Directory table (DWARF5 format).
b = append(b, 0) // one directory entry (index 0 = empty)
// File table.
b = appendUleb(b, 1) // file count
// File 1: name index into .debug_line_str, dir index, time, size.
b = appendUleb(b, 0) // name (index 0 in line_str)
b = appendUleb(b, 0) // directory index
b = appendUleb(b, 0) // last modification time
b = appendUleb(b, 0) // file size
headerEnd := len(b)
// Per-function line programs.
for _, fn := range img.Funcs {
// LNE_set_address with the function's offset in .text.
b = append(b, 0, 9, 2) // extended opcode, length 9, DW_LNE_set_address
addrOff := len(b)
b = le.AppendUint64(b, 0) // placeholder for address
ds.lineRelocs = append(ds.lineRelocs, dwarfReloc{
off: uint64(addrOff),
name: fn.Name,
addend: 0,
})
// Build the line entries.
pts := make([]LineEntry, 0, len(fn.Lines)+1)
if len(fn.Lines) == 0 || fn.Lines[0].Offset > 0 {
pts = append(pts, LineEntry{Offset: 0, Line: fn.Line})
}
pts = append(pts, fn.Lines...)
line := int64(1)
pc := uint64(0)
for _, p := range pts {
if p.Line == 0 || uint64(p.Offset) < pc {
continue
}
if int64(p.Line) == line {
continue
}
deltaPC := uint64(p.Offset) - pc
deltaLC := int64(p.Line) - line
b = dwPutPCLCDelta(b, deltaPC, deltaLC)
line, pc = int64(p.Line), uint64(p.Offset)
}
// Advance to end of function.
if end := uint64(fn.Size) - pc; end > 0 {
b = append(b, 2) // DW_LNS_advance_pc
b = appendUleb(b, end)
}
b = append(b, 0, 1, 1) // LNE_end_sequence
}
// Patch unit_length.
le.PutUint32(b[headerStart:], uint32(len(b)-headerStart-4))
// Patch header_length.
le.PutUint32(b[headerStart+6:], uint32(headerEnd-headerStart-10))
return b
}
// dwarfBuildInfoSection builds a complete .debug_info section.
func dwarfBuildInfoSection(img *Image, srcFile string, ds *dwarfSections) []byte {
var b []byte
le := binary.LittleEndian
cuStart := len(b)
b = append(b, 0, 0, 0, 0) // unit_length (placeholder)
b = le.AppendUint16(b, 5) // version (DWARF5)
b = append(b, 0x01) // unit_type (DW_UT_compile)
b = append(b, 8) // address_size
b = le.AppendUint32(b, 0) // debug_abbrev_offset (0 since single CU)
// DW_TAG_compile_unit (abbrev 1).
b = append(b, 1) // abbreviation code
// DW_AT_low_pc: address of .text start.
infoRelocBase := len(b)
b = le.AppendUint64(b, 0) // placeholder
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
off: uint64(infoRelocBase),
name: img.Funcs[0].Name,
addend: 0,
})
// DW_AT_high_pc: size of .text.
b = le.AppendUint64(b, uint64(len(img.Code)))
// DW_AT_stmt_list: offset into .debug_line (0).
b = le.AppendUint32(b, 0)
// DW_AT_name: source file name.
b = append(b, srcFile...)
b = append(b, 0)
// DW_TAG_subprogram entries (abbrev 2).
for _, fn := range img.Funcs {
b = append(b, 2) // abbreviation code
// DW_AT_name.
b = append(b, fn.Name...)
b = append(b, 0)
// DW_AT_low_pc.
addrOff := len(b)
b = le.AppendUint64(b, 0) // placeholder
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
off: uint64(addrOff),
name: fn.Name,
addend: 0,
})
// DW_AT_high_pc: function size.
b = le.AppendUint64(b, uint64(fn.Size))
// DW_AT_frame_base: DW_OP_call_frame_cfa.
b = append(b, 1, 0x9c)
// DW_AT_decl_file: file index 1.
b = append(b, 1)
// DW_AT_decl_line.
b = append(b, uint8(fn.Line))
// DW_AT_external.
if fn.Static {
b = append(b, 0)
} else {
b = append(b, 1)
}
}
// End of compile unit children.
b = append(b, 0)
// Patch unit_length.
le.PutUint32(b[cuStart:], uint32(len(b)-cuStart-4))
return b
}
func appendUleb(b []byte, v uint64) []byte {
return binary.AppendUvarint(b, v)
}
func appendSleb(b []byte, v int64) []byte {
return binary.AppendVarint(b, v)
}
// dwarfBuildFrameSection builds a .debug_frame section with CFI for stack
// unwinding. It emits one CIE and one FDE per function, encoding the
// CFA (Canonical Frame Address) rule changes at each stack-adjustment
// boundary recorded in FuncLayout.Spadj.
func dwarfBuildFrameSection(img *Image) []byte {
var b []byte
le := binary.LittleEndian
// CIE (Common Information Entry).
cieStart := len(b)
b = append(b, 0, 0, 0, 0) // length (placeholder)
b = le.AppendUint32(b, 0xFFFFFFFF) // CIE marker
b = append(b, 3) // version (DWARF3, widely supported)
b = append(b, 0) // augmentation (empty)
b = appendUleb(b, 1) // code alignment
b = appendSleb(b, -8) // data alignment (-8 for 64-bit)
b = appendUleb(b, 16) // return address register (LR on arm64, RIP on amd64)
// Initial CFA rule: DW_CFA_def_cfa (SP, 0)
b = append(b, 0x0c) // DW_CFA_def_cfa
b = appendUleb(b, 31) // register: SP (RSP=7 on amd64, SP=31 on arm64)
b = appendUleb(b, 0) // offset: 0
b = append(b, 0) // DW_CFA_nop (padding)
// Patch CIE length.
le.PutUint32(b[cieStart:], uint32(len(b)-cieStart-4))
// FDEs (Frame Description Entries), one per function.
for _, fn := range img.Funcs {
fdeStart := len(b)
b = append(b, 0, 0, 0, 0) // length (placeholder)
b = le.AppendUint32(b, uint32(cieStart)) // CIE pointer (offset from start)
// Initial location: function offset in .text (relocated by linker).
b = le.AppendUint64(b, uint64(fn.Offset))
// Address range: function size.
b = le.AppendUint64(b, uint64(fn.Size))
// Emit CFA rule changes at each Spadj boundary.
for _, step := range fn.Spadj {
if step.Value == 0 {
continue
}
// DW_CFA_def_cfa_offset: set CFA = SP + |delta|.
// The delta is negative (stack grows down), so CFA offset = -delta.
offset := -step.Value
if offset > 0 {
b = append(b, 0x0e) // DW_CFA_def_cfa_offset
b = appendUleb(b, uint64(offset))
}
}
// Pad to alignment.
for len(b)%4 != 0 {
b = append(b, 0) // DW_CFA_nop
}
// Patch FDE length.
le.PutUint32(b[fdeStart:], uint32(len(b)-fdeStart-4))
}
return b
}
+119
View File
@@ -0,0 +1,119 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// dwarfELFSections holds the laid-out DWARF sections ready for inclusion
// in an ELF file.
type dwarfELFSections struct {
abbrevOff, abbrevSize int
infoOff, infoSize int
lineOff, lineSize int
lineStrOff, lineStrSize int
frameOff, frameSize int
// Relocations for .debug_info address references.
infoRelocs []elfDwarfReloc
// Relocations for .debug_line address references.
lineRelocs []elfDwarfReloc
}
type elfDwarfReloc struct {
off uint64
sym int // symbol index in .symtab
addend int64
}
// appendDWARFSections generates and appends DWARF5 debug sections to the ELF
// output. It returns the section offsets/sizes and relocations for the caller
// to emit section headers and relocation records.
//
// symIdx maps function names to their .symtab indices (needed for relocations
// against .text symbols). The map uses objectName format (pkg.name); the
// DWARF code uses bare function names, so we build a reverse lookup.
func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[string]int, align func(int)) *dwarfELFSections {
// Build a lookup from bare function name to symbol index.
nameToIdx := make(map[string]int, len(symIdx))
for name, idx := range symIdx {
// Strip package prefix: "pkg.name" → "name".
if i := len(name) - 1; i >= 0 {
for j := len(name) - 1; j >= 0; j-- {
if name[j] == '.' {
nameToIdx[name[j+1:]] = idx
break
}
}
}
nameToIdx[name] = idx
}
ds := emitDWARF(img, srcFile)
if ds == nil || len(ds.debugAbbrev) == 0 {
return nil
}
result := &dwarfELFSections{}
// .debug_abbrev
align(1)
result.abbrevOff = len(*out)
result.abbrevSize = len(ds.debugAbbrev)
*out = append(*out, ds.debugAbbrev...)
// .debug_line_str
align(1)
result.lineStrOff = len(*out)
result.lineStrSize = len(ds.debugLineStr)
*out = append(*out, ds.debugLineStr...)
// .debug_line
align(1)
result.lineOff = len(*out)
result.lineSize = len(ds.debugLine)
lineBase := len(*out)
*out = append(*out, ds.debugLine...)
// Patch .debug_line relocations: replace placeholder addresses with
// actual .text offsets via symbol lookup.
for _, dr := range ds.lineRelocs {
if idx, ok := nameToIdx[dr.name]; ok {
result.lineRelocs = append(result.lineRelocs, elfDwarfReloc{
off: uint64(lineBase) + dr.off,
sym: idx,
addend: dr.addend,
})
}
}
// .debug_info
align(1)
result.infoOff = len(*out)
result.infoSize = len(ds.debugInfo)
infoBase := len(*out)
*out = append(*out, ds.debugInfo...)
// .debug_frame
if len(ds.debugFrame) > 0 {
align(1)
result.frameOff = len(*out)
result.frameSize = len(ds.debugFrame)
*out = append(*out, ds.debugFrame...)
}
// Patch .debug_info relocations.
for _, dr := range ds.infoRelocs {
if idx, ok := nameToIdx[dr.name]; ok {
result.infoRelocs = append(result.infoRelocs, elfDwarfReloc{
off: uint64(infoBase) + dr.off,
sym: idx,
addend: dr.addend,
})
}
}
return result
}
// dwarfSectionNames returns the DWARF section names for the string table.
var dwarfSectionNames = []string{
".debug_abbrev", ".debug_info", ".debug_line", ".debug_line_str",
".debug_frame", ".rela.debug_info", ".rela.debug_line",
}
+81
View File
@@ -0,0 +1,81 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
func TestEmitDWARF(t *testing.T) {
src := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
ADDQ BX, AX
MOVQ AX, ret+16(FP)
RET
`
f, errs := parser.Parse("test_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
ds := emitDWARF(img, "test_amd64.s")
// .debug_abbrev must not be empty and must start with abbrev code 1.
if len(ds.debugAbbrev) == 0 {
t.Fatal("empty .debug_abbrev")
}
if ds.debugAbbrev[0] != 1 {
t.Fatalf(".debug_abbrev first byte = %d, want 1", ds.debugAbbrev[0])
}
// .debug_info must have a compile unit header (DWARF5 version 5).
if len(ds.debugInfo) < 12 {
t.Fatalf(".debug_info too short: %d bytes", len(ds.debugInfo))
}
// Version field at offset 4 (after unit_length).
if ds.debugInfo[4] != 5 || ds.debugInfo[5] != 0 {
t.Fatalf(".debug_info version = %d, want 5", uint16(ds.debugInfo[4])|uint16(ds.debugInfo[5])<<8)
}
// .debug_line must have a header.
if len(ds.debugLine) < 20 {
t.Fatalf(".debug_line too short: %d bytes", len(ds.debugLine))
}
// Version at offset 4.
if ds.debugLine[4] != 5 || ds.debugLine[5] != 0 {
t.Fatalf(".debug_line version = %d, want 5", uint16(ds.debugLine[4])|uint16(ds.debugLine[5])<<8)
}
// .debug_line_str must contain the source file name.
if len(ds.debugLineStr) == 0 {
t.Fatal("empty .debug_line_str")
}
// Relocations must reference the function.
if len(ds.lineRelocs) == 0 {
t.Fatal("no .debug_line relocations")
}
if len(ds.infoRelocs) == 0 {
t.Fatal("no .debug_info relocations")
}
}
func TestDwarfAbbrevTable(t *testing.T) {
abbrev := dwarfAbbrevTable()
if len(abbrev) == 0 {
t.Fatal("empty abbrev table")
}
// Must end with a zero byte (end of table).
if abbrev[len(abbrev)-1] != 0 {
t.Fatalf("abbrev table last byte = %d, want 0", abbrev[len(abbrev)-1])
}
}
+1 -1
View File
@@ -187,7 +187,7 @@ func TestELFObject(t *testing.T) {
end := bytes.IndexByte(strtabRaw[stName:], 0) end := bytes.IndexByte(strtabRaw[stName:], 0)
return string(strtabRaw[stName : int(stName)+end]) return string(strtabRaw[stName : int(stName)+end])
} }
for i := 0; i < 2; i++ { for i := range 2 {
e := raw[i*24 : (i+1)*24] e := raw[i*24 : (i+1)*24]
off := binary.LittleEndian.Uint64(e[0:]) off := binary.LittleEndian.Uint64(e[0:])
info := binary.LittleEndian.Uint64(e[8:]) info := binary.LittleEndian.Uint64(e[8:])
+43 -6
View File
@@ -15,7 +15,9 @@ const (
// AArch64 relocation types (the ELF psABI). // AArch64 relocation types (the ELF psABI).
rArm64PrelPgHi21 = 275 // R_AARCH64_ADR_PREL_PG_HI21 (ADRP page) rArm64PrelPgHi21 = 275 // R_AARCH64_ADR_PREL_PG_HI21 (ADRP page)
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD/STR/LDR page offset) rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset)
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset)
) )
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for // ELFAARCH64Object returns the image as an ELF64 relocatable object file for
@@ -80,7 +82,13 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
// Build relocations. Each SB reference is an ADRP pair: // Build relocations. Each SB reference is an ADRP pair:
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21 // ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21
// ADD/LDR/STR → R_AARCH64_ADD_ABS_LO12_NC // ADD → R_AARCH64_ADD_ABS_LO12_NC
// LDR/STR X → R_AARCH64_LDST64_ABS_LO12_NC
// BL → R_AARCH64_CALL26
// Addends stay raw: ADR_PREL_PG_HI21 and the ABS_LO12_NC forms resolve
// against S+A, and CALL26 branches take the branch instruction's own
// place as the PC-relative base, so subtracting the field width (the
// amd64 R_PCREL convention) would misplace every branch by 4 bytes.
type elfRela struct { type elfRela struct {
off uint64 off uint64
typ uint32 typ uint32
@@ -94,16 +102,22 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
if !ok { if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name) return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
} }
typ := uint32(rArm64PrelPgHi21) var typ uint32
if r.Kind == RelArm64Addr && r.Off%4 == 4 { switch {
// The second instruction in an ADRP pair uses ADD_ABS_LO12_NC. case r.Kind == RelArm64Branch:
typ = rArm64Call26
case r.Kind == RelArm64LDST64 && r.Off%4 == 4:
typ = rArm64Ldst64Lo12NC
case r.Kind == RelArm64Addr && r.Off%4 == 4:
typ = rArm64AddAbsLo12NC typ = rArm64AddAbsLo12NC
default:
typ = rArm64PrelPgHi21
} }
relas = append(relas, elfRela{ relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off), off: uint64(fn.Offset + r.Off),
typ: typ, typ: typ,
sym: idx, sym: idx,
addend: r.Addend - int64(r.After-r.Off), addend: r.Addend,
}) })
} }
} }
@@ -117,6 +131,9 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} { for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n) stSections.add(n)
} }
for _, n := range dwarfSectionNames {
stSections.add(n)
}
hasRela := len(relas) > 0 hasRela := len(relas) > 0
nSections := 6 nSections := 6
@@ -176,6 +193,17 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
shstrOff := len(out) shstrOff := len(out)
out = append(out, stSections.bytes()...) out = append(out, stSections.bytes()...)
// DWARF debug sections.
dwAlign := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
if dw != nil {
nSections += 4
}
align(8) align(8)
shoff := len(out) shoff := len(out)
@@ -202,6 +230,15 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24) putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
} }
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0) putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
if dw != nil {
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
}
}
// ELF header. // ELF header.
hdr := out[:64] hdr := out[:64]
+28 -2
View File
@@ -16,6 +16,7 @@ const (
// LoongArch relocation types (the ELF psABI). // LoongArch relocation types (the ELF psABI).
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i) rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st) rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
rLarchB26 = 66 // R_LARCH_B26 (b/bl, matches the Go linker's mapping)
) )
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for // ELFLOONG64Object returns the image as an ELF64 relocatable object file for
@@ -95,14 +96,17 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name) return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
} }
typ := uint32(rLarchPCALAHI20) typ := uint32(rLarchPCALAHI20)
if r.Kind == RelLoong64AddrLo { switch r.Kind {
case RelLoong64AddrLo:
typ = rLarchPCALALO12 typ = rLarchPCALALO12
case RelLoong64Branch:
typ = rLarchB26
} }
relas = append(relas, elfRela{ relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off), off: uint64(fn.Offset + r.Off),
typ: typ, typ: typ,
sym: idx, sym: idx,
addend: r.Addend - int64(r.After-r.Off), addend: r.Addend,
}) })
} }
} }
@@ -116,6 +120,9 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} { for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n) stSections.add(n)
} }
for _, n := range dwarfSectionNames {
stSections.add(n)
}
hasRela := len(relas) > 0 hasRela := len(relas) > 0
nSections := 6 nSections := 6
@@ -175,6 +182,16 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
shstrOff := len(out) shstrOff := len(out)
out = append(out, stSections.bytes()...) out = append(out, stSections.bytes()...)
dwAlign := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
if dw != nil {
nSections += 4
}
align(8) align(8)
shoff := len(out) shoff := len(out)
@@ -201,6 +218,15 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24) putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
} }
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0) putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
if dw != nil {
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
}
}
// ELF header. // ELF header.
hdr := out[:64] hdr := out[:64]
+43 -1
View File
@@ -139,7 +139,7 @@ DATA answer<>+0(SB)/8, $42
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw)) t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
} }
le := binary.LittleEndian le := binary.LittleEndian
for i := 0; i < 2; i++ { for i := range 2 {
e := raw[i*24 : (i+1)*24] e := raw[i*24 : (i+1)*24]
off := le.Uint64(e[0:]) off := le.Uint64(e[0:])
info := le.Uint64(e[8:]) info := le.Uint64(e[8:])
@@ -198,3 +198,45 @@ TEXT ·nop(SB), NOSPLIT, $0
t.Error("function symbol nop not found") t.Error("function symbol nop not found")
} }
} }
// TestELFLOONG64BranchRelocation checks that the morestack call and an
// internal CALL both carry R_LARCH_B26 in the emitted object, matching the
// Go linker's mapping of its call relocation.
func TestELFLOONG64BranchRelocation(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", "TEXT \u00b7callbig(SB), $8192-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.ELFLOONG64Object()
if err != nil {
t.Fatalf("ELFLOONG64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaSec := ef.Section(".rela.text")
if relaSec == nil {
t.Fatal("missing .rela.text")
}
raw, err := relaSec.Data()
if err != nil {
t.Fatal(err)
}
// The guard's morestack call plus the body's CALL to other.
if len(raw)%24 != 0 || len(raw)/24 != 2 {
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
}
le := binary.LittleEndian
for i := range 2 {
info := le.Uint64(raw[i*24+8:])
if elf.R_LARCH(info&0xffffffff) != elf.R_LARCH_B26 {
t.Errorf("relocation %d type = %v, want R_LARCH_B26", i, elf.R_LARCH(info&0xffffffff))
}
}
}
+22
View File
@@ -129,6 +129,9 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} { for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n) stSections.add(n)
} }
for _, n := range dwarfSectionNames {
stSections.add(n)
}
hasRela := len(relas) > 0 hasRela := len(relas) > 0
nSections := 6 nSections := 6
@@ -188,6 +191,16 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
shstrOff := len(out) shstrOff := len(out)
out = append(out, stSections.bytes()...) out = append(out, stSections.bytes()...)
dwAlign := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
if dw != nil {
nSections += 4
}
align(8) align(8)
shoff := len(out) shoff := len(out)
@@ -214,6 +227,15 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24) putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
} }
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0) putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
if dw != nil {
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
}
}
// ELF header. // ELF header.
hdr := out[:64] hdr := out[:64]
+89
View File
@@ -0,0 +1,89 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "strings"
// Encodable reports whether the amd64 encoder knows how to encode the
// mnemonic. It mirrors the dispatch in (*enc).encode: the fixed-name
// instructions, conditional jumps, the CMOV/SET condition families, the
// VEX/EVEX/opmask/gather/scatter vector paths, the legacy SSE tables and the
// explicit scalar cases. A mnemonic that parses (is in the architecture
// table) but is not encodable would otherwise surface only at assembly time,
// deep inside a build; the linter uses this predicate to flag it at edit
// time.
func Encodable(mnemonic string) bool {
upper := strings.ToUpper(mnemonic)
// Fixed-name instructions (no size suffix).
switch upper {
case "RET", "NOP", "CALL", "JMP":
return true
}
if _, ok := condCode(upper); ok {
return true
}
// VEX/EVEX and friends: the trailing B/W/L/Q/D is part of the mnemonic.
base, _, err := parseEvexSuffix(upper)
if err != nil {
return false
}
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
base == "KMOVW" || base == "KMOVQ" {
return true
}
// CMOV carries size then condition (CMOVLGT); SET carries the condition
// alone (SETNE).
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
if _, ok := jccMap[rest[1:]]; ok {
return true
}
}
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
if _, ok := jccMap[rest]; ok {
return true
}
}
// Legacy SSE shuffles and packed binaries dispatch on the full name.
if _, ok := sseShufTable[upper]; ok {
return true
}
if _, ok := sseBinTable[upper]; ok {
return true
}
// The size-suffix split: retry the tables and the scalar switch on the
// base.
base2, size := splitSize(upper)
if size == 0 {
size = 8
}
_ = size
if base2 != upper {
if _, ok := sseBinTable[base2]; ok {
return true
}
}
switch base2 {
case "MOV",
"ADD", "SUB", "AND", "OR", "XOR", "CMP",
"TEST",
"LEA",
"INC", "DEC", "NEG", "NOT",
"SHL", "SHR", "SAR",
"IMUL", "IMUL3",
"PUSH", "POP",
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
"BSWAP",
"PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2",
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
"CVTSL2SD", "CVTSQ2SD",
"MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
return true
}
return false
}
+53 -3
View File
@@ -76,6 +76,21 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if size == 0 { if size == 0 {
size = 8 // default operand size in 64-bit mode (e.g. PUSHQ) size = 8 // default operand size in 64-bit mode (e.g. PUSHQ)
} }
// Legacy SSE imm8 shuffles whose names end in W/H (PSHUFLW,
// PSHUFHW) must dispatch BEFORE the size-suffix split, and the
// others ride along.
if m, ok := sseShufTable[upper]; ok {
return e.encodeSSEShuf(m, ops)
}
// Legacy SSE packed binaries dispatch on the full name: the packed
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
// which the size split must not eat.
if m, ok := sseBinTable[upper]; ok {
return e.encodeSSEBin(m, ops)
}
if m, ok := sseBinTable[base]; ok {
return e.encodeSSEBin(m, ops)
}
switch base { switch base {
case "MOV": case "MOV":
return e.encodeMov(ops, size) return e.encodeMov(ops, size)
@@ -95,8 +110,12 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodePushPop(ops, true) return e.encodePushPop(ops, true)
case "POP": case "POP":
return e.encodePushPop(ops, false) return e.encodePushPop(ops, false)
case "LZCNT", "TZCNT": case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
return e.encodeCount(base, ops, size) return e.encodeCount(base, ops, size)
case "BSWAP":
return e.encodeBswap(ops, size)
case "PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2":
return e.encodePrefetch(base, ops)
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX": case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX":
return e.encodeMovExtend(base, ops) return e.encodeMovExtend(base, ops)
case "CVTSL2SD", "CVTSQ2SD": case "CVTSL2SD", "CVTSQ2SD":
@@ -107,6 +126,30 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return fmt.Errorf("unsupported instruction %q", mnem) return fmt.Errorf("unsupported instruction %q", mnem)
} }
// encodePrefetch emits the 0F 18 /r prefetch hints: the reg field selects
// the locality (NTA=0, T0=1, T1=2, T2=3) and the single operand is memory.
func (e *enc) encodePrefetch(base string, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects one memory operand", base)
}
m, ok := ops[0].(Mem)
if !ok {
return fmt.Errorf("%s requires a memory operand", base)
}
i := newInstr(0, []byte{0x0F, 0x18})
if err := setMem(i, prefetchVariant[base], m); err != nil {
return err
}
return e.emit(i)
}
var prefetchVariant = map[string]int{
"PREFETCHNTA": 0,
"PREFETCHT0": 1,
"PREFETCHT1": 2,
"PREFETCHT2": 3,
}
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic. // splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
func splitSize(upper string) (base string, size int) { func splitSize(upper string) (base string, size int) {
if upper == "" { if upper == "" {
@@ -241,7 +284,7 @@ func setRM(i *instr, reg Reg, rm Operand, opSize int) error {
} }
// setRMDigit fills in the ModR/M for an instruction whose reg field is an // setRMDigit fills in the ModR/M for an instruction whose reg field is an
// opcode /digit extension (0–7), which carries none of the register REX rules. // opcode /digit extension (0-7), which carries none of the register REX rules.
func setRMDigit(i *instr, digit int, rm Operand, opSize int) error { func setRMDigit(i *instr, digit int, rm Operand, opSize int) error {
return setRMReg(i, digit, false, false, rm, opSize) return setRMReg(i, digit, false, false, rm, opSize)
} }
@@ -297,6 +340,13 @@ func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101 return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
} }
// The SIB scale field only encodes 1/2/4/8; the Go assembler rejects
// anything else ("bad scale: 16"), so a silent fallback to scale 1 here
// would mis-assemble the operand instead of reporting it.
if m.HasIndex && m.Scale != 1 && m.Scale != 2 && m.Scale != 4 && m.Scale != 8 {
return 0, -1, nil, 0, 0, fmt.Errorf("bad scale: %d", m.Scale)
}
needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4) needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4)
var mod int var mod int
@@ -369,7 +419,7 @@ func le16(v int64) []byte {
func le64(v int64) []byte { func le64(v int64) []byte {
u := uint64(v) u := uint64(v)
b := make([]byte, 8) b := make([]byte, 8)
for i := 0; i < 8; i++ { for i := range 8 {
b[i] = byte(u >> (8 * i)) b[i] = byte(u >> (8 * i))
} }
return b return b
+154 -3
View File
@@ -4,6 +4,7 @@
package asm package asm
import ( import (
"fmt"
"strings" "strings"
"testing" "testing"
@@ -57,8 +58,11 @@ func TestMov(t *testing.T) {
checkSyntax(t, "mov qword ptr [rbx], rax", "MOVQ", AX, Ptr(BX, 0, 8)) checkSyntax(t, "mov qword ptr [rbx], rax", "MOVQ", AX, Ptr(BX, 0, 8))
checkSyntax(t, "mov rbx, qword ptr [rax+0x10]", "MOVQ", Ptr(AX, 0x10, 8), BX) checkSyntax(t, "mov rbx, qword ptr [rax+0x10]", "MOVQ", Ptr(AX, 0x10, 8), BX)
checkSyntax(t, "mov rbx, qword ptr [rsi+4*rbx]", "MOVQ", Idx(SI, BX, 4, 0, 8), BX) checkSyntax(t, "mov rbx, qword ptr [rsi+4*rbx]", "MOVQ", Idx(SI, BX, 4, 0, 8), BX)
checkSyntax(t, "mov rax, 0x5", "MOVQ", Imm(5), AX) // A small positive immediate compresses to the 32-bit zero-extending
checkSyntax(t, "mov r8, 0x5", "MOVQ", Imm(5), Reg{idx: 8, size: 8}) // form (matching go tool asm), so the disassembler renders the 32-bit
// register name even for MOVQ.
checkSyntax(t, "mov eax, 0x5", "MOVQ", Imm(5), AX)
checkSyntax(t, "mov r8d, 0x5", "MOVQ", Imm(5), Reg{idx: 8, size: 8})
checkSyntax(t, "mov qword ptr [rax], 0x5", "MOVQ", Imm(5), Ptr(AX, 0, 8)) checkSyntax(t, "mov qword ptr [rax], 0x5", "MOVQ", Imm(5), Ptr(AX, 0, 8))
checkSyntax(t, "mov r12, r13", "MOVQ", Reg{idx: 13, size: 8}, Reg{idx: 12, size: 8}) checkSyntax(t, "mov r12, r13", "MOVQ", Reg{idx: 13, size: 8}, Reg{idx: 12, size: 8})
} }
@@ -74,13 +78,63 @@ func TestALU(t *testing.T) {
checkSyntax(t, "cmp rsi, r10", "CMPQ", SI, Reg{idx: 10, size: 8}) checkSyntax(t, "cmp rsi, r10", "CMPQ", SI, Reg{idx: 10, size: 8})
checkSyntax(t, "add rbx, qword ptr [rax]", "ADDQ", Ptr(AX, 0, 8), BX) checkSyntax(t, "add rbx, qword ptr [rax]", "ADDQ", Ptr(AX, 0, 8), BX)
checkSyntax(t, "add qword ptr [rax], rbx", "ADDQ", BX, Ptr(AX, 0, 8)) checkSyntax(t, "add qword ptr [rax], rbx", "ADDQ", BX, Ptr(AX, 0, 8))
checkSyntax(t, "cmp rbx, -0x20", "CMPQ", Imm(-32), BX) // The Go assembler rejects the immediate-first CMP spelling outright,
// so Encode errors instead of silently emitting the swapped form.
if _, err := Encode("CMPQ", Imm(-32), BX); err == nil {
t.Errorf("Encode(CMPQ imm-first) should error, got success")
}
// The Go assembler's own spelling: immediate second. // The Go assembler's own spelling: immediate second.
checkSyntax(t, "cmp ecx, 0x1f", "CMPL", CX, Imm(31)) checkSyntax(t, "cmp ecx, 0x1f", "CMPL", CX, Imm(31))
checkSyntax(t, "cmp ecx, -0x80000000", "CMPL", CX, Imm(-2147483648)) checkSyntax(t, "cmp ecx, -0x80000000", "CMPL", CX, Imm(-2147483648))
checkSyntax(t, "cmp r9, -0x80000000", "CMPQ", Reg{idx: 9, size: 8}, Imm(-2147483648)) checkSyntax(t, "cmp r9, -0x80000000", "CMPQ", Reg{idx: 9, size: 8}, Imm(-2147483648))
} }
// TestScalarXmmRegMoves pins the Go-assembler byte forms of scalar
// MOVQ/MOVL between GPRs and XMM registers (66 REX.W 0F 6E/0F 7E) and the
// memory forms (F3 0F 7E load, 66 0F D6 store), all byte-for-byte.
func TestScalarXmmRegMoves(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"MOVQ AX,X1", "MOVQ", []Operand{AX, vreg(t, "X1")}, "66480f6ec8"},
{"MOVQ DX,X2", "MOVQ", []Operand{DX, vreg(t, "X2")}, "66480f6ed2"},
{"MOVQ X1,AX", "MOVQ", []Operand{vreg(t, "X1"), AX}, "66480f7ec8"},
{"MOVQ X0,DX", "MOVQ", []Operand{vreg(t, "X0"), DX}, "66480f7ec2"},
{"MOVL AX,X1", "MOVL", []Operand{AX, vreg(t, "X1")}, "660f6ec8"},
{"MOVL X1,AX", "MOVL", []Operand{vreg(t, "X1"), AX}, "660f7ec8"},
{"MOVQ (SI),X1", "MOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X1")}, "f30f7e0e"},
{"MOVQ X3,(DI)", "MOVQ", []Operand{vreg(t, "X3"), Ptr(DI, 0, 8)}, "660fd61f"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
}
// TestBadScale pins the go-tool-asm parity of rejecting SIB scales the
// hardware cannot encode.
func TestBadScale(t *testing.T) {
for _, sc := range []int{3, 5, 16, 32} {
if _, err := Encode("LEAQ", Idx(SI, BX, sc, 0, 8), AX); err == nil {
t.Errorf("LEAQ scale %d: expected error, got success", sc)
}
}
for _, sc := range []int{1, 2, 4, 8} {
if _, err := Encode("LEAQ", Idx(SI, BX, sc, 0, 8), AX); err != nil {
t.Errorf("LEAQ scale %d: %v", sc, err)
}
}
}
func TestLea(t *testing.T) { func TestLea(t *testing.T) {
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8}) checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
checkSyntax(t, "lea rax, ptr [rbx+0x8]", "LEAQ", Ptr(BX, 0x8, 8), AX) checkSyntax(t, "lea rax, ptr [rbx+0x8]", "LEAQ", Ptr(BX, 0x8, 8), AX)
@@ -203,6 +257,21 @@ func TestScalarGroundTruth(t *testing.T) {
{"LZCNTQ R8,R9", "LZCNTQ", []Operand{r8, r9}, "f34d0fbdc8", "LZCNT"}, {"LZCNTQ R8,R9", "LZCNTQ", []Operand{r8, r9}, "f34d0fbdc8", "LZCNT"},
{"LZCNTW AX,CX", "LZCNTW", []Operand{AX, CX}, "66f30fbdc8", "LZCNT"}, {"LZCNTW AX,CX", "LZCNTW", []Operand{AX, CX}, "66f30fbdc8", "LZCNT"},
{"TZCNTL AX,CX", "TZCNTL", []Operand{AX, CX}, "f30fbcc8", "TZCNT"}, {"TZCNTL AX,CX", "TZCNTL", []Operand{AX, CX}, "f30fbcc8", "TZCNT"},
// Bit scan: BSF/BSR are the unprefixed forms of TZCNT/LZCNT's map.
{"BSFL AX,CX", "BSFL", []Operand{AX, CX}, "0fbcc8", "BSF"},
{"BSFQ R8,R9", "BSFQ", []Operand{r8, r9}, "4d0fbcc8", "BSF"},
{"BSFW AX,CX", "BSFW", []Operand{AX, CX}, "660fbcc8", "BSF"},
{"BSRL AX,CX", "BSRL", []Operand{AX, CX}, "0fbdc8", "BSR"},
{"BSRQ AX,CX", "BSRQ", []Operand{AX, CX}, "480fbdc8", "BSR"},
{"POPCNTL AX,CX", "POPCNTL", []Operand{AX, CX}, "f30fb8c8", "POPCNT"},
{"POPCNTQ R8,R9", "POPCNTQ", []Operand{r8, r9}, "f34d0fb8c8", "POPCNT"},
// A 64-bit immediate that fits a signed int32 is compressed exactly
// as the Go assembler does: positive via B8+rd without REX.W
// (zero-extended), negative via REX.W C7 /0 (sign-extended).
{"MOVQ $4,BX", "MOVQ", []Operand{Imm(4), BX}, "bb04000000", "MOV"},
{"MOVQ $4,R8", "MOVQ", []Operand{Imm(4), r8}, "41b804000000", "MOV"},
{"MOVQ $-1,BX", "MOVQ", []Operand{Imm(-1), BX}, "48c7c3ffffffff", "MOV"},
{"MOVQ big,BX", "MOVQ", []Operand{Imm(0x1122334455667788), BX}, "48bb8877665544332211", "MOV"},
{"CMOVLGT CX,AX", "CMOVLGT", []Operand{CX, AX}, "0f4fc1", "CMOVG"}, {"CMOVLGT CX,AX", "CMOVLGT", []Operand{CX, AX}, "0f4fc1", "CMOVG"},
{"CMOVLEQ CX,AX", "CMOVLEQ", []Operand{CX, AX}, "0f44c1", "CMOVE"}, {"CMOVLEQ CX,AX", "CMOVLEQ", []Operand{CX, AX}, "0f44c1", "CMOVE"},
{"CMOVQGT R9,R8", "CMOVQGT", []Operand{r9, r8}, "4d0f4fc1", "CMOVG"}, {"CMOVQGT R9,R8", "CMOVQGT", []Operand{r9, r8}, "4d0f4fc1", "CMOVG"},
@@ -294,3 +363,85 @@ func TestScalarErrors(t *testing.T) {
} }
} }
} }
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
// byte for byte (no prefix / 66 / F2 / F3 variants).
func TestSSEBinGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"MULPS X0,X1", "MULPS", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f59c8"},
{"MULPS (DI),X1", "MULPS", []Operand{Ptr(DI, 0, 16), vreg(t, "X1")}, "0f590f"},
{"ADDPD X1,X2", "ADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "660f58d1"},
{"XORPS X0,X0", "XORPS", []Operand{vreg(t, "X0"), vreg(t, "X0")}, "0f57c0"},
{"UNPCKLPS X0,X0", "UNPCKLPS", []Operand{vreg(t, "X0"), vreg(t, "X0")}, "0f14c0"},
{"MULSD X1,X2", "MULSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f20f59d1"},
{"ADDSS (DI),X0", "ADDSS", []Operand{Ptr(DI, 0, 4), vreg(t, "X0")}, "f30f5807"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
}
// TestSSEShuffleGroundTruth checks the imm8 shuffle family: immediate
// first in Plan 9 order, encoded last on the wire.
func TestSSEShuffleGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"SHUFPS $0,X0,X0", "SHUFPS", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X0")}, "0fc6c000"},
{"SHUFPS $27,X1,X2", "SHUFPS", []Operand{Imm(27), vreg(t, "X1"), vreg(t, "X2")}, "0fc6d11b"},
{"PSHUFD $0,X0,X0", "PSHUFD", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X0")}, "660f70c000"},
{"PSHUFLW $3,(DI),X1", "PSHUFLW", []Operand{Imm(3), Ptr(DI, 0, 8), vreg(t, "X1")}, "f20f700f03"},
{"PSHUFHW $2,X1,X2", "PSHUFHW", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2")}, "f30f70d102"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
}
// TestMOVQXMMGroundTruth pins the SSE2 packed-quadword move encodings:
// loads and register moves on F3 0F 7E, stores on 66 0F D6 — the forms
// the GPR-move fallback silently corrupted.
func TestMOVQXMMGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"MOVQ (DI),X0", "MOVQ", []Operand{Ptr(DI, 0, 8), vreg(t, "X0")}, "f30f7e07"},
{"MOVQ X1,X2", "MOVQ", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f30f7ed1"},
{"MOVQ X0,(DI)", "MOVQ", []Operand{vreg(t, "X0"), Ptr(DI, 0, 8)}, "660fd607"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
}
+141 -92
View File
@@ -9,10 +9,10 @@ import (
) )
// This file implements EVEX (AVX-512) instruction encoding: the four-byte // This file implements EVEX (AVX-512) instruction encoding: the four-byte
// EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the // EVEX prefix with 5-bit vector register fields (Z0-Z31, X/Y 16-31), the
// compressed disp8×N displacement, and the operand shapes the go-flac // compressed disp8×N displacement, and the operand shapes the go-flac
// AVX-512 kernels use plus the common floating-point and conversion set. // AVX-512 kernels use plus the common floating-point and conversion set.
// Masking follows the Go assembler's spelling: an explicit K1–K7 operand // Masking follows the Go assembler's spelling: an explicit K1-K7 operand
// anywhere among the operands (merging) plus a ".Z" mnemonic suffix for // anywhere among the operands (merging) plus a ".Z" mnemonic suffix for
// zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are // zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are
// supported too. // supported too.
@@ -36,7 +36,7 @@ type evexSpec struct {
// are taken from the Go assembler's opcode tables, which are authoritative // are taken from the Go assembler's opcode tables, which are authoritative
// for byte-for-byte agreement. // for byte-for-byte agreement.
var evexTable = map[string]evexSpec{ var evexTable = map[string]evexSpec{
// EVEX.128/256/512.66.0F — integer arithmetic / logic, NDS form. // EVEX.128/256/512.66.0F, integer arithmetic / logic, NDS form.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -48,25 +48,25 @@ var evexTable = map[string]evexSpec{
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — packed double arithmetic. // EVEX.128/256/512.66.0F.W1, packed double arithmetic.
"VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0 — packed single arithmetic. // EVEX.128/256/512.0F.W0, packed single arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, "VADDPS": {1, 0x58, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, "VMULPS": {1, 0x59, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, "VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, "VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, "VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, "VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — packed double unpack. // EVEX.128/256/512.66.0F.W1, packed double unpack.
"VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128.F2.0F.W1 — scalar double arithmetic (the packed opcodes with // EVEX.128.F2.0F.W1, scalar double arithmetic (the packed opcodes with
// an F2 pp; the EVEX forms exist for masked and zeroing use). The // an F2 pp; the EVEX forms exist for masked and zeroing use). The
// memory operand is a single double, so disp8×N = 8. // memory operand is a single double, so disp8×N = 8.
"VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, "VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
@@ -76,7 +76,7 @@ var evexTable = map[string]evexSpec{
"VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, "VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, "VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128.F3.0F.W0 — scalar single arithmetic (disp8×N = 4). // EVEX.128.F3.0F.W0, scalar single arithmetic (disp8×N = 4).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
@@ -84,38 +84,38 @@ var evexTable = map[string]evexSpec{
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.512.66.0F3A — align (NDS + imm8). // EVEX.512.66.0F3A, align (NDS + imm8).
"VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F — immediate shift (VPSRAD /4). // EVEX.128/256/512.66.0F, immediate shift (VPSRAD /4).
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}}, "VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — variable shift with an XMM count (VPSRAQ; // EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
// the W bit distinguishes it from VPSRAD's E2 form). // the W bit distinguishes it from VPSRAD's E2 form).
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst, // EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
// rm=src, no vvvv). // rm=src, no vvvv).
"VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, "VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1 — duplicate the low double (reg=dst, // EVEX.128/256/512.F2.0F.W1, duplicate the low double (reg=dst,
// rm=src, no vvvv): a 128-bit destination reads a single double from // rm=src, no vvvv): a 128-bit destination reads a single double from
// memory (disp8×8), the wider ones read the full operand. // memory (disp8×8), the wider ones read the full operand.
"VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}}, "VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}},
// EVEX.128/256/512.0F.W0 — signed dword to packed single (reg=dst, // EVEX.128/256/512.0F.W0, signed dword to packed single (reg=dst,
// rm=src, no vvvv, no mandatory prefix — as in the VEX form). // rm=src, no vvvv, no mandatory prefix, as in the VEX form).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}}, "VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0 — packed single to packed double: the // EVEX.128/256/512.0F.W0, packed single to packed double: the
// destination is twice the source width and sets the length; disp8×N // destination is twice the source width and sets the length; disp8×N
// follows the narrow memory source. No F3 prefix: the Go assembler // follows the narrow memory source. No F3 prefix: the Go assembler
// emits this instruction with pp = 00 (Intel's maps would call that // emits this instruction with pp = 00 (Intel's maps would call that
// undefined) and gasm reproduces the Go assembler's bytes — its machine // undefined) and gasm reproduces the Go assembler's bytes, its machine
// code is the oracle, not the manual. // code is the oracle, not the manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}}, "VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.128/256/512.F3.0F.W0 — signed dword to packed double (the EVEX // EVEX.128/256/512.F3.0F.W0, signed dword to packed double (the EVEX
// form of the VEX instruction; the destination sets the length, disp8×N // form of the VEX instruction; the destination sets the length, disp8×N
// follows the narrow memory source). // follows the narrow memory source).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}}, "VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
// EVEX packed double → dword conversions: the source is the wide // EVEX packed double → dword conversions: the source is the wide
// operand and the mnemonic fixes the length — the bare names are // operand and the mnemonic fixes the length, the bare names are
// 512-bit only (ZMM source, XMM destination), the X/Y spellings are // 512-bit only (ZMM source, XMM destination), the X/Y spellings are
// EVEX-128/256. Exactly one slot of n is valid; it names the vector // EVEX-128/256. Exactly one slot of n is valid; it names the vector
// length (and the disp8×N multiplier) a register or memory source // length (and the disp8×N multiplier) a register or memory source
@@ -127,7 +127,7 @@ var evexTable = map[string]evexSpec{
"VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}}, "VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}}, "VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.66.0F3A — ternary logic and lane shuffles (NDS + imm8). // EVEX.66.0F3A, ternary logic and lane shuffles (NDS + imm8).
"VPTERNLOGD": {3, 0x25, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VPTERNLOGD": {3, 0x25, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPTERNLOGQ": {3, 0x25, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VPTERNLOGQ": {3, 0x25, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFI32X4": {3, 0x43, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VSHUFI32X4": {3, 0x43, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
@@ -136,11 +136,11 @@ var evexTable = map[string]evexSpec{
"VSHUFF64X2": {3, 0x23, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VSHUFF64X2": {3, 0x23, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F — the EVEX forms of the VEX two-source shuffle. // EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F3A — lane insert ($imm, xsrc, zsrc1, zdst). // EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}}, "VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTF32X8": {3, 0x1A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}}, "VINSERTF32X8": {3, 0x1A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
"VINSERTF64X2": {3, 0x18, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}}, "VINSERTF64X2": {3, 0x18, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
@@ -150,7 +150,7 @@ var evexTable = map[string]evexSpec{
"VINSERTI64X2": {3, 0x38, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}}, "VINSERTI64X2": {3, 0x38, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTI64X4": {3, 0x3A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}}, "VINSERTI64X4": {3, 0x3A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
// EVEX.66.0F3A — lane extract (reg=source, rm=XMM/YMM destination, // EVEX.66.0F3A, lane extract (reg=source, rm=XMM/YMM destination,
// imm8). // imm8).
"VEXTRACTF32X4": {3, 0x19, 0, 1, -1, vexExtract, [3]int{0, 16, 16}}, "VEXTRACTF32X4": {3, 0x19, 0, 1, -1, vexExtract, [3]int{0, 16, 16}},
"VEXTRACTF32X8": {3, 0x1B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}}, "VEXTRACTF32X8": {3, 0x1B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
@@ -159,14 +159,27 @@ var evexTable = map[string]evexSpec{
"VEXTRACTI32X8": {3, 0x3B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}}, "VEXTRACTI32X8": {3, 0x3B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTI64X2": {3, 0x39, 1, 1, -1, vexExtract, [3]int{0, 16, 16}}, "VEXTRACTI64X2": {3, 0x39, 1, 1, -1, vexExtract, [3]int{0, 16, 16}},
// EVEX.66.0F — compare with an opmask destination ($imm, src2, src1, // EVEX.66.0F, compare with an opmask destination ($imm, src2, src1,
// kdst): NDS3Imm with the K register in the reg field. // kdst): NDS3Imm with the K register in the reg field.
"VCMPPD": {1, 0xC2, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VCMPPD": {1, 0xC2, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VCMPPS": {1, 0xC2, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VCMPPS": {1, 0xC2, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}}, "VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}}, "VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}},
// EVEX.66.0F38 — permutes (NDS form). // EVEX.66.0F3A, integer compares with an opmask destination, the same
// NDS3Imm-with-k-reg shape as the floating-point compares; W selects the
// operand width (byte/word vs dword/qword), the opcode the signedness.
// The memory form takes a full vector, so disp8×N is 16/32/64.
"VPCMPB": {3, 0x3F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUB": {3, 0x3E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPW": {3, 0x3F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUW": {3, 0x3E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPD": {3, 0x1F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUD": {3, 0x1E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPQ": {3, 0x1F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F38, permutes (NDS form).
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -175,7 +188,7 @@ var evexTable = map[string]evexSpec{
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F — the wider integer set (NDS form). // EVEX.66.0F, the wider integer set (NDS form).
"VPMADDWD": {1, 0xF5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMADDWD": {1, 0xF5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -186,34 +199,34 @@ var evexTable = map[string]evexSpec{
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKUSDW": {2, 0x2B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPACKUSDW": {2, 0x2B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F38 — absolute values and replicating moves (reg=dst, // EVEX.66.0F38, absolute values and replicating moves (reg=dst,
// rm=src). // rm=src).
"VPABSB": {2, 0x1C, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VPABSB": {2, 0x1C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSW": {2, 0x1D, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VPABSW": {2, 0x1D, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSD": {2, 0x1E, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VPABSD": {2, 0x1E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSQ": {2, 0x1F, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, "VPABSQ": {2, 0x1F, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.F3.0F — replicate even/odd singles. // EVEX.F3.0F, replicate even/odd singles.
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, "VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, "VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — sign/zero-extending moves; the memory source is the // EVEX.66.0F38, sign/zero-extending moves; the memory source is the
// narrow half (here byte to word). // narrow half (here byte to word).
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F — packed single conversions (reg=dst, rm=src). // EVEX.66.0F, packed single conversions (reg=dst, rm=src).
"VCVTPS2DQ": {1, 0x5B, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VCVTPS2DQ": {1, 0x5B, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTTPS2DQ": {1, 0x5B, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, "VCVTTPS2DQ": {1, 0x5B, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — broadcast a single/double to all lanes (reg=dst, // EVEX.66.0F38, broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; disp8×N is the element size). // rm=scalar memory; disp8×N is the element size).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM, [3]int{4, 4, 4}}, "VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VBROADCASTSD": {2, 0x19, 1, 1, -1, vexRM, [3]int{0, 8, 8}}, "VBROADCASTSD": {2, 0x19, 1, 1, -1, vexRM, [3]int{0, 8, 8}},
// EVEX.66.0F38 — expand loads (rm → vector register destination). // EVEX.66.0F38, expand loads (rm → vector register destination).
"VEXPANDPD": {2, 0x88, 1, 1, -1, vexRM, [3]int{8, 8, 8}}, "VEXPANDPD": {2, 0x88, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
"VEXPANDPS": {2, 0x88, 0, 1, -1, vexRM, [3]int{4, 4, 4}}, "VEXPANDPS": {2, 0x88, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VPEXPANDD": {2, 0x89, 0, 1, -1, vexRM, [3]int{4, 4, 4}}, "VPEXPANDD": {2, 0x89, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VPEXPANDQ": {2, 0x89, 1, 1, -1, vexRM, [3]int{8, 8, 8}}, "VPEXPANDQ": {2, 0x89, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
// EVEX.66.0F38 — compress stores (vector register source → rm), and the // EVEX.66.0F38, compress stores (vector register source → rm), and the
// remaining narrowing stores. // remaining narrowing stores.
"VCOMPRESSPD": {2, 0x8A, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}}, "VCOMPRESSPD": {2, 0x8A, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}},
"VCOMPRESSPS": {2, 0x8A, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}}, "VCOMPRESSPS": {2, 0x8A, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}},
@@ -222,7 +235,7 @@ var evexTable = map[string]evexSpec{
"VPMOVWB": {2, 0x30, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, "VPMOVWB": {2, 0x30, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQB": {2, 0x32, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}}, "VPMOVQB": {2, 0x32, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
// EVEX.66.0F — rotates (immediate form: /0 right, /1 left). // EVEX.66.0F, rotates (immediate form: /0 right, /1 left).
"VPRORD": {1, 0x72, 0, 1, 0, vexShiftImm, [3]int{16, 32, 64}}, "VPRORD": {1, 0x72, 0, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
"VPRORQ": {1, 0x72, 1, 1, 0, vexShiftImm, [3]int{16, 32, 64}}, "VPRORQ": {1, 0x72, 1, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
"VPROLD": {1, 0x72, 0, 1, 1, vexShiftImm, [3]int{16, 32, 64}}, "VPROLD": {1, 0x72, 0, 1, 1, vexShiftImm, [3]int{16, 32, 64}},
@@ -235,14 +248,14 @@ var evexTable = map[string]evexSpec{
"VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}}, "VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
"VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}}, "VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.66.0F38 — floating-point helpers, packed (reg=dst, rm=src). // EVEX.66.0F38, floating-point helpers, packed (reg=dst, rm=src).
"VRCP14PD": {2, 0x4C, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, "VRCP14PD": {2, 0x4C, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRCP14PS": {2, 0x4C, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VRCP14PS": {2, 0x4C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRSQRT14PD": {2, 0x4E, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, "VRSQRT14PD": {2, 0x4E, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRSQRT14PS": {2, 0x4E, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VRSQRT14PS": {2, 0x4E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VGETEXPPD": {2, 0x42, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, "VGETEXPPD": {2, 0x42, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VGETEXPPS": {2, 0x42, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, "VGETEXPPS": {2, 0x42, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — floating-point helpers, scalar (NDS form: src2 is // EVEX.66.0F38, floating-point helpers, scalar (NDS form: src2 is
// rm, src1 is vvvv, the XMM destination is reg). Like the scalar 0F3A // rm, src1 is vvvv, the XMM destination is reg). Like the scalar 0F3A
// forms, these take the 66 prefix; W selects double/single. // forms, these take the 66 prefix; W selects double/single.
"VRCP14SD": {2, 0x4D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}}, "VRCP14SD": {2, 0x4D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
@@ -251,13 +264,13 @@ var evexTable = map[string]evexSpec{
"VRSQRT14SS": {2, 0x4F, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}}, "VRSQRT14SS": {2, 0x4F, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
"VGETEXPSD": {2, 0x43, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}}, "VGETEXPSD": {2, 0x43, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VGETEXPSS": {2, 0x43, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}}, "VGETEXPSS": {2, 0x43, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.66.0F38 — scale by a power of two (NDS form). // EVEX.66.0F38, scale by a power of two (NDS form).
"VSCALEFPD": {2, 0x2C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VSCALEFPD": {2, 0x2C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSCALEFPS": {2, 0x2C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VSCALEFPS": {2, 0x2C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSCALEFSD": {2, 0x2D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}}, "VSCALEFSD": {2, 0x2D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VSCALEFSS": {2, 0x2D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}}, "VSCALEFSS": {2, 0x2D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.66.0F3A — packed round/getmant/reduce ($imm, src, dst: reg=dst, // EVEX.66.0F3A, packed round/getmant/reduce ($imm, src, dst: reg=dst,
// rm=src, imm8). // rm=src, imm8).
"VRNDSCALEPD": {3, 0x09, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}}, "VRNDSCALEPD": {3, 0x09, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VRNDSCALEPS": {3, 0x08, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}}, "VRNDSCALEPS": {3, 0x08, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
@@ -265,7 +278,7 @@ var evexTable = map[string]evexSpec{
"VGETMANTPS": {3, 0x26, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}}, "VGETMANTPS": {3, 0x26, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VREDUCEPD": {3, 0x56, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}}, "VREDUCEPD": {3, 0x56, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VREDUCEPS": {3, 0x56, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}}, "VREDUCEPS": {3, 0x56, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.66.0F3A — scalar round/getmant/reduce and fixup/range (NDS + // EVEX.66.0F3A, scalar round/getmant/reduce and fixup/range (NDS +
// imm8: $imm, src2, src1, dst). The scalar 0F3A forms all take the 66 // imm8: $imm, src2, src1, dst). The scalar 0F3A forms all take the 66
// prefix; W selects double/single. // prefix; W selects double/single.
"VRNDSCALESD": {3, 0x0B, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}}, "VRNDSCALESD": {3, 0x0B, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
@@ -283,7 +296,7 @@ var evexTable = map[string]evexSpec{
"VRANGESD": {3, 0x51, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}}, "VRANGESD": {3, 0x51, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VRANGESS": {3, 0x51, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}}, "VRANGESS": {3, 0x51, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
// EVEX.66.0F3A — floating-point class test ($imm, src, kdst): the // EVEX.66.0F3A, floating-point class test ($imm, src, kdst): the
// reg field carries the opmask destination. The packed forms carry an // reg field carries the opmask destination. The packed forms carry an
// explicit length in the mnemonic (X/Y/Z). // explicit length in the mnemonic (X/Y/Z).
"VFPCLASSPDX": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{16, 0, 0}}, "VFPCLASSPDX": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{16, 0, 0}},
@@ -295,7 +308,7 @@ var evexTable = map[string]evexSpec{
"VFPCLASSSD": {3, 0x67, 1, 1, -1, vexImmRM, [3]int{8, 0, 0}}, "VFPCLASSSD": {3, 0x67, 1, 1, -1, vexImmRM, [3]int{8, 0, 0}},
"VFPCLASSSS": {3, 0x67, 0, 1, -1, vexImmRM, [3]int{4, 0, 0}}, "VFPCLASSSS": {3, 0x67, 0, 1, -1, vexImmRM, [3]int{4, 0, 0}},
// EVEX — the remaining conversions. VCVTQQ2PS narrows (the 512-bit // EVEX, the remaining conversions. VCVTQQ2PS narrows (the 512-bit
// source sets the length); the rest follow the destination. // source sets the length); the rest follow the destination.
"VCVTQQ2PS": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}}, "VCVTQQ2PS": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2QQ": {1, 0x7B, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, "VCVTPD2QQ": {1, 0x7B, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
@@ -303,13 +316,13 @@ var evexTable = map[string]evexSpec{
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}}, "VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}}, "VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F38 — half-precision convert (half-width source). // EVEX.66.0F38, half-precision convert (half-width source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src, // EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
// rm=dst, imm8 — the extract layout). // rm=dst, imm8, the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}}, "VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
// EVEX — unsigned and truncating conversions. The PD sources are the // EVEX, unsigned and truncating conversions. The PD sources are the
// wide operand (the bare names are 512-bit only, the X/Y spellings fix // wide operand (the bare names are 512-bit only, the X/Y spellings fix
// the length); the PS/UQQ destinations are wide and follow the // the length); the PS/UQQ destinations are wide and follow the
// destination. // destination.
@@ -336,7 +349,7 @@ var evexTable = map[string]evexSpec{
"VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}}, "VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}}, "VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.66.0F38 — the remaining sign/zero-extending moves (narrow // EVEX.66.0F38, the remaining sign/zero-extending moves (narrow
// source; disp8×N follows its size). // source; disp8×N follows its size).
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}}, "VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}}, "VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
@@ -348,7 +361,7 @@ var evexTable = map[string]evexSpec{
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}}, "VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.F3.0F38 — the remaining narrowing stores (vector source in reg, // EVEX.F3.0F38, the remaining narrowing stores (vector source in reg,
// narrow destination in r/m): signed, unsigned and the D/Q truncations. // narrow destination in r/m): signed, unsigned and the D/Q truncations.
"VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}}, "VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}}, "VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
@@ -365,7 +378,7 @@ var evexTable = map[string]evexSpec{
"VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}}, "VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}}, "VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
// EVEX.F3.0F38 — mask/vector conversions: M2* moves an opmask register // EVEX.F3.0F38, mask/vector conversions: M2* moves an opmask register
// into a vector (rm = K source, reg = vector destination), *2M does the // into a vector (rm = K source, reg = vector destination), *2M does the
// reverse (reg = K destination, rm = vector source, the length follows // reverse (reg = K destination, rm = vector source, the length follows
// the vector). // the vector).
@@ -378,7 +391,7 @@ var evexTable = map[string]evexSpec{
"VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, "VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, "VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX — scalar conversions between vector and general-purpose // EVEX, scalar conversions between vector and general-purpose
// registers. Vector to GPR (two operands: vec/mem source, GPR // registers. Vector to GPR (two operands: vec/mem source, GPR
// destination, vvvv unused): the signed and truncated pair, and the // destination, vvvv unused): the signed and truncated pair, and the
// unsigned forms (EVEX only). // unsigned forms (EVEX only).
@@ -408,22 +421,22 @@ var evexTable = map[string]evexSpec{
"VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, "VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}}, "VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory // EVEX.128/256/512.66.0F38.W0, sign-extend dwords to qwords; the memory
// operand is the narrow source, so disp8×N follows its size (8/16/32 for // operand is the narrow source, so disp8×N follows its size (8/16/32 for
// the xmm/ymm/zmm destination lengths). // the xmm/ymm/zmm destination lengths).
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.512.66.0F3A.W1 — lane extract (reg=ZMM source, rm=YMM/memory // EVEX.512.66.0F3A.W1, lane extract (reg=ZMM source, rm=YMM/memory
// destination, imm8). // destination, imm8).
"VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}}, "VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}}, "VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
// EVEX.66.0F38 — more integer NDS forms (W distinguishes D/Q). // EVEX.66.0F38, more integer NDS forms (W distinguishes D/Q).
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}}, "VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
// EVEX.128/256/512 — the wider integer set (AVX-512 F/BW): byte/word // EVEX.128/256/512, the wider integer set (AVX-512 F/BW): byte/word
// arithmetic, the bitwise ops with D/Q suffixes, min/max, averages and // arithmetic, the bitwise ops with D/Q suffixes, min/max, averages and
// variable shifts. All NDS form; W distinguishes element size. // variable shifts. All NDS form; W distinguishes element size.
"VPADDB": {1, 0xFC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPADDB": {1, 0xFC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -461,21 +474,21 @@ var evexTable = map[string]evexSpec{
"VPSRAVQ": {2, 0x46, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSRAVQ": {2, 0x46, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX forms of instructions that also exist in VEX (selected when a ZMM // EVEX forms of instructions that also exist in VEX (selected when a ZMM
// or K register, or indices 16–31, demand EVEX). // or K register, or indices 16-31, demand EVEX).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}}, "VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F — immediate shift (VPSLLD /6). // EVEX.66.0F, immediate shift (VPSLLD /6).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}}, "VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.F3.0F38.W0 — narrowing stores: reg = wide source, rm = narrow // EVEX.F3.0F38.W0, narrowing stores: reg = wide source, rm = narrow
// destination (VPMOVDW dword→word, VPMOVQD qword→dword). // destination (VPMOVDW dword→word, VPMOVQD qword→dword).
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, "VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, "VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
} }
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode // evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
// depends on the source kind — a GPR source uses opReg, a memory source uses // depends on the source kind, a GPR source uses opReg, a memory source uses
// opMem with a disp8×N of n. // opMem with a disp8×N of n.
type evexBcastSpec struct { type evexBcastSpec struct {
mapSel int mapSel int
@@ -486,10 +499,10 @@ type evexBcastSpec struct {
} }
var evexBcastTable = map[string]evexBcastSpec{ var evexBcastTable = map[string]evexBcastSpec{
// EVEX.128/256/512.66.0F38 — broadcast a dword/qword to all lanes. // EVEX.128/256/512.66.0F38, broadcast a dword/qword to all lanes.
"VPBROADCASTD": {2, 0x7C, 0x58, 0, 4}, "VPBROADCASTD": {2, 0x7C, 0x58, 0, 4},
"VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8}, "VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8},
// EVEX.128/256/512.66.0F38 — broadcast a byte/word (GPR or memory // EVEX.128/256/512.66.0F38, broadcast a byte/word (GPR or memory
// source) to all lanes. // source) to all lanes.
"VPBROADCASTB": {2, 0x7A, 0x78, 0, 1}, "VPBROADCASTB": {2, 0x7A, 0x78, 0, 1},
"VPBROADCASTW": {2, 0x7B, 0x79, 0, 2}, "VPBROADCASTW": {2, 0x7B, 0x79, 0, 2},
@@ -508,26 +521,26 @@ type evexMoveSpec struct {
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding. // evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{ var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0 — unaligned integer move. // EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1 — unaligned qword move. // EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, "VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W0 — unaligned byte move (byte/word moves use the // EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple // F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics). // semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, "VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1 — unaligned word move (shares the qword // EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding). // encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — unaligned packed double move. // EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}}, "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512 — aligned packed moves. // EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}}, "VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}}, "VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F — aligned integer moves. // EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, "VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, "VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128.F3.0F.W0 — scalar single move, memory operands (the // EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported). // three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}}, "VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}},
} }
@@ -546,7 +559,7 @@ func isEvex(mnemUpper string) bool {
// evexRequired reports whether the operands force the EVEX encoding of a // evexRequired reports whether the operands force the EVEX encoding of a
// mnemonic that also has a VEX form: ZMM and K registers do, and so do // mnemonic that also has a VEX form: ZMM and K registers do, and so do
// register indices 16–31, which only EVEX can represent (X16–Y31 exist // register indices 16-31, which only EVEX can represent (X16-Y31 exist
// solely under AVX-512). // solely under AVX-512).
func evexRequired(upper string, ops []Operand) bool { func evexRequired(upper string, ops []Operand) bool {
_, inVex := vexTable[upper] _, inVex := vexTable[upper]
@@ -565,7 +578,7 @@ func evexRequired(upper string, ops []Operand) bool {
// evexSuffix carries the EVEX mnemonic suffixes the Go assembler accepts: // evexSuffix carries the EVEX mnemonic suffixes the Go assembler accepts:
// zeroing (.Z), a rounding mode (.RN_SAE, .RD_SAE, .RU_SAE, .RZ_SAE), // zeroing (.Z), a rounding mode (.RN_SAE, .RD_SAE, .RU_SAE, .RZ_SAE),
// suppress-all-exceptions (.SAE) and memory broadcast (.BCST). Masking is // suppress-all-exceptions (.SAE) and memory broadcast (.BCST). Masking is
// not a suffix — Go writes it as an explicit K operand. // not a suffix, Go writes it as an explicit K operand.
type evexSuffix struct { type evexSuffix struct {
zeroing bool zeroing bool
sae bool sae bool
@@ -590,12 +603,12 @@ func (s evexSuffix) evexOnly() bool {
// broadcast together with rounding/SAE. // broadcast together with rounding/SAE.
func parseEvexSuffix(mnem string) (string, evexSuffix, error) { func parseEvexSuffix(mnem string) (string, evexSuffix, error) {
sfx := evexSuffix{rounding: -1} sfx := evexSuffix{rounding: -1}
i := strings.IndexByte(mnem, '.') before, after, ok := strings.Cut(mnem, ".")
if i < 0 { if !ok {
return mnem, sfx, nil return mnem, sfx, nil
} }
base := mnem[:i] base := before
parts := strings.Split(mnem[i+1:], ".") parts := strings.Split(after, ".")
seen := map[string]bool{} seen := map[string]bool{}
for j, p := range parts { for j, p := range parts {
if seen[p] { if seen[p] {
@@ -605,7 +618,7 @@ func parseEvexSuffix(mnem string) (string, evexSuffix, error) {
switch p { switch p {
case "Z": case "Z":
if j != len(parts)-1 { if j != len(parts)-1 {
return "", sfx, fmt.Errorf("the .Z suffix must come last in %q", mnem[i+1:]) return "", sfx, fmt.Errorf("the .Z suffix must come last in %q", after)
} }
sfx.zeroing = true sfx.zeroing = true
case "SAE": case "SAE":
@@ -625,7 +638,7 @@ func parseEvexSuffix(mnem string) (string, evexSuffix, error) {
} }
} }
if sfx.bcst && (sfx.sae || sfx.rounding >= 0) { if sfx.bcst && (sfx.sae || sfx.rounding >= 0) {
return "", sfx, fmt.Errorf("cannot combine .BCST with rounding or SAE in %q", mnem[i+1:]) return "", sfx, fmt.Errorf("cannot combine .BCST with rounding or SAE in %q", after)
} }
return base, sfx, nil return base, sfx, nil
} }
@@ -655,7 +668,7 @@ var evexRound = map[string]bool{
} }
// evexBcstN maps an instruction accepting .BCST to the broadcast element // evexBcstN maps an instruction accepting .BCST to the broadcast element
// size — the disp8×N multiplier for its memory operand. // size, the disp8×N multiplier for its memory operand.
var evexBcstN = map[string]int{ var evexBcstN = map[string]int{
"VADDPD": 8, "VSUBPD": 8, "VMULPD": 8, "VDIVPD": 8, "VADDPD": 8, "VSUBPD": 8, "VMULPD": 8, "VDIVPD": 8,
"VMINPD": 8, "VMAXPD": 8, "VMINPD": 8, "VMAXPD": 8,
@@ -676,7 +689,7 @@ var evexBcstN = map[string]int{
"VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8, "VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8,
} }
// splitMask extracts an explicit mask register (K1–K7) from the operand list, // splitMask extracts an explicit mask register (K1-K7) from the operand list,
// returning the remaining operands and the mask index. K0 is not a usable // returning the remaining operands and the mask index. K0 is not a usable
// mask (aaa = 0 means "no mask"), matching the assembler. // mask (aaa = 0 means "no mask"), matching the assembler.
func splitMask(ops []Operand) ([]Operand, int, error) { func splitMask(ops []Operand) ([]Operand, int, error) {
@@ -688,7 +701,7 @@ func splitMask(ops []Operand) ([]Operand, int, error) {
return nil, 0, fmt.Errorf("at most one mask register operand") return nil, 0, fmt.Errorf("at most one mask register operand")
} }
if r.idx == 0 { if r.idx == 0 {
return nil, 0, fmt.Errorf("K0 is not a usable mask register") return nil, 0, fmt.Errorf("k0 is not a usable mask register")
} }
mask = r.idx mask = r.idx
continue continue
@@ -699,7 +712,7 @@ func splitMask(ops []Operand) ([]Operand, int, error) {
} }
// encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The // encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The
// mask, when present, is an explicit K1–K7 operand anywhere among the // mask, when present, is an explicit K1-K7 operand anywhere among the
// operands; the mnemonic suffix carries zeroing, rounding/SAE and // operands; the mnemonic suffix carries zeroing, rounding/SAE and
// broadcast. // broadcast.
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error { func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
@@ -1029,7 +1042,7 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
} }
// encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with // encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the length fixed by the mnemonic — the // the destination always XMM and the length fixed by the mnemonic, the
// single valid slot of spec.n names the vector length (and the disp8×N // single valid slot of spec.n names the vector length (and the disp8×N
// multiplier) a register or memory source encodes. // multiplier) a register or memory source encodes.
func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
@@ -1048,7 +1061,7 @@ func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx eve
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx) return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
} }
// soleLen returns the vector-length index of the single valid slot of n — // soleLen returns the vector-length index of the single valid slot of n
// the length a length-fixed mnemonic (the EVEX conversion spellings) encodes // the length a length-fixed mnemonic (the EVEX conversion spellings) encodes
// regardless of its operands. // regardless of its operands.
func soleLen(n [3]int) (int, error) { func soleLen(n [3]int) (int, error) {
@@ -1118,8 +1131,8 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx eve
// emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement // emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement
// (disp8×N compressed) for the given precomputed fields. regIdx is the // (disp8×N compressed) for the given precomputed fields. regIdx is the
// unextended reg-field register index, or a /digit (0–7); vvvvIdx is the // unextended reg-field register index, or a /digit (0-7); vvvvIdx is the
// vvvv register index, or -1 when unused. mask (K1–K7, 0 = unmasked) and // vvvv register index, or -1 when unused. mask (K1-K7, 0 = unmasked) and
// zeroing fill the aaa and z bits of the P2 byte. // zeroing fill the aaa and z bits of the P2 byte.
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, sfx evexSuffix) error { func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, sfx evexSuffix) error {
if ll > 2 { if ll > 2 {
@@ -1190,7 +1203,7 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
// The b bit and the L'L field carry the rounding/SAE/broadcast mode: // The b bit and the L'L field carry the rounding/SAE/broadcast mode:
// a rounding mode replaces L'L with the rc value, plain SAE and // a rounding mode replaces L'L with the rc value, plain SAE and
// broadcast keep the vector length. // broadcast keep the vector length.
b, ll := 0, ll b := 0
switch { switch {
case sfx.rounding >= 0: case sfx.rounding >= 0:
b, ll = 1, sfx.rounding b, ll = 1, sfx.rounding
@@ -1311,7 +1324,7 @@ func isScatter(upper string) bool {
} }
// vsibLen validates a VSIB memory operand (the index must be a vector // vsibLen validates a VSIB memory operand (the index must be a vector
// register) and returns it with the vector length the index selects — the // register) and returns it with the vector length the index selects, the
// EVEX L'L field follows the index register, not the data register. // EVEX L'L field follows the index register, not the data register.
func vsibLen(op Operand, what string) (Mem, int, error) { func vsibLen(op Operand, what string) (Mem, int, error) {
m, ok := op.(Mem) m, ok := op.(Mem)
@@ -1370,7 +1383,7 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib) return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib)
} }
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib — reg = src, // encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src,
// rm = the VSIB memory operand, the K mask in aaa and L following the VSIB // rm = the VSIB memory operand, the K mask in aaa and L following the VSIB
// index. // index.
func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evexSuffix) error { func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evexSuffix) error {
@@ -1398,15 +1411,15 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
// evexKOperand lists the instructions whose K register is a genuine operand // evexKOperand lists the instructions whose K register is a genuine operand
// (the source or destination of a mask/vector conversion) rather than a // (the source or destination of a mask/vector conversion) rather than a
// mask modifier — the M2 and 2M conversions. They take no masking. // mask modifier, the M2 and 2M conversions. They take no masking.
var evexKOperand = map[string]bool{ var evexKOperand = map[string]bool{
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true, "VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true, "VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
} }
// kmovSpec describes a KMOV width: the opcode depends on the operand // kmovSpec describes a KMOV width: the opcode depends on the operand
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem), // direction, kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory // gprk (GPR/mem → K), kgpr (K → GPR), and the GPR forms carry a mandatory
// prefix and W for the wider widths. // prefix and W for the wider widths.
type kmovSpec struct { type kmovSpec struct {
kk, kmem, gprk, kgpr byte kk, kmem, gprk, kgpr byte
@@ -1471,24 +1484,60 @@ type kOpSpec struct {
var kOpsTable = map[string]kOpSpec{ var kOpsTable = map[string]kOpSpec{
// k ← k OP k: reg = dst, vvvv = src1, rm = src2 (three opmask // k ← k OP k: reg = dst, vvvv = src1, rm = src2 (three opmask
// registers). // registers). Byte/word widths share W0 and differ by the 66 prefix;
// dword/qword share the W bit selection the Go assembler emits.
"KANDB": {1, 0x41, 0, 1, 1, vexNDS3}, "KANDB": {1, 0x41, 0, 1, 1, vexNDS3},
"KANDW": {1, 0x41, 0, 0, 1, vexNDS3}, "KANDW": {1, 0x41, 0, 0, 1, vexNDS3},
"KANDD": {1, 0x41, 1, 1, 1, vexNDS3},
"KANDQ": {1, 0x41, 1, 0, 1, vexNDS3}, "KANDQ": {1, 0x41, 1, 0, 1, vexNDS3},
"KANDNB": {1, 0x42, 0, 1, 1, vexNDS3},
"KANDNW": {1, 0x42, 0, 0, 1, vexNDS3},
"KANDND": {1, 0x42, 1, 1, 1, vexNDS3},
"KANDNQ": {1, 0x42, 1, 0, 1, vexNDS3},
"KORB": {1, 0x45, 0, 1, 1, vexNDS3}, "KORB": {1, 0x45, 0, 1, 1, vexNDS3},
"KORW": {1, 0x45, 0, 0, 1, vexNDS3},
"KORD": {1, 0x45, 1, 1, 1, vexNDS3}, "KORD": {1, 0x45, 1, 1, 1, vexNDS3},
"KORQ": {1, 0x45, 1, 0, 1, vexNDS3},
"KXNORB": {1, 0x46, 0, 1, 1, vexNDS3},
"KXNORW": {1, 0x46, 0, 0, 1, vexNDS3}, "KXNORW": {1, 0x46, 0, 0, 1, vexNDS3},
"KXNORD": {1, 0x46, 1, 1, 1, vexNDS3},
"KXNORQ": {1, 0x46, 1, 0, 1, vexNDS3}, "KXNORQ": {1, 0x46, 1, 0, 1, vexNDS3},
"KXORB": {1, 0x47, 0, 1, 1, vexNDS3},
"KXORW": {1, 0x47, 0, 0, 1, vexNDS3},
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3}, "KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3}, "KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3}, "KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3}, "KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
"KADDD": {1, 0x4A, 1, 1, 1, vexNDS3},
"KADDQ": {1, 0x4A, 1, 0, 1, vexNDS3}, "KADDQ": {1, 0x4A, 1, 0, 1, vexNDS3},
// k ← OP k (KNOT) and flags ← k OP k (KORTEST): reg = dst, rm = src. // k ← OP k (KNOT), k ← k AND~ k (KTEST-style RM) and flags ← k OP k
// (KORTEST): reg = dst, rm = src.
"KNOTB": {1, 0x44, 0, 1, 0, vexRM}, "KNOTB": {1, 0x44, 0, 1, 0, vexRM},
"KNOTW": {1, 0x44, 0, 0, 0, vexRM},
"KNOTD": {1, 0x44, 1, 1, 0, vexRM},
"KNOTQ": {1, 0x44, 1, 0, 0, vexRM},
"KORTESTB": {1, 0x98, 0, 1, 0, vexRM},
"KORTESTW": {1, 0x98, 0, 0, 0, vexRM},
"KORTESTD": {1, 0x98, 1, 1, 0, vexRM}, "KORTESTD": {1, 0x98, 1, 1, 0, vexRM},
// OP $imm, src, dst: reg = dst, rm = src, imm8. "KORTESTQ": {1, 0x98, 1, 0, 0, vexRM},
"KTESTB": {1, 0x99, 0, 1, 0, vexRM},
"KTESTW": {1, 0x99, 0, 0, 0, vexRM},
"KTESTD": {1, 0x99, 1, 1, 0, vexRM},
"KTESTQ": {1, 0x99, 1, 0, 0, vexRM},
// OP $imm, src, dst: reg = dst, rm = src, imm8. The opcodes split by
// direction (0x32/0x33 left, 0x30/0x31 right) and within each by
// element half (0x32 byte/word, 0x33 dword/qword); W picks byte/dword
// (W0) against word/qword (W1).
"KSHIFTLB": {3, 0x32, 0, 1, 0, vexImmRM},
"KSHIFTLW": {3, 0x32, 1, 1, 0, vexImmRM}, "KSHIFTLW": {3, 0x32, 1, 1, 0, vexImmRM},
"KSHIFTLD": {3, 0x33, 0, 1, 0, vexImmRM},
"KSHIFTLQ": {3, 0x33, 1, 1, 0, vexImmRM},
"KSHIFTRB": {3, 0x30, 0, 1, 0, vexImmRM},
"KSHIFTRW": {3, 0x30, 1, 1, 0, vexImmRM},
"KSHIFTRD": {3, 0x31, 0, 1, 0, vexImmRM},
"KSHIFTRQ": {3, 0x31, 1, 1, 0, vexImmRM},
} }
// isKOp reports whether the mnemonic is an opmask-register instruction. // isKOp reports whether the mnemonic is an opmask-register instruction.
+37
View File
@@ -319,6 +319,43 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
{"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"}, {"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"},
{"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"}, {"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"},
{"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"}, {"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"},
// Completed opmask families (ANDN, NOT, OR/XOR word+qword, TEST,
// word-width shifts; byte-exact against go tool asm).
{"KANDNW", "KANDNW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec42d9"},
{"KANDNB", "KANDNB", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c5d542f4"},
{"KANDND", "KANDND", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ed42d9"},
{"KANDNQ", "KANDNQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d442f4"},
{"KANDD", "KANDD", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ed41d9"},
{"KADDD", "KADDD", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d54af4"},
{"KNOTW", "KNOTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f844d1"},
{"KNOTD", "KNOTD", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c4e1f944e3"},
{"KNOTQ", "KNOTQ", []Operand{vreg(t, "K5"), vreg(t, "K6")}, "c4e1f844f5"},
{"KORW", "KORW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec45d9"},
{"KORQ", "KORQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d445f4"},
{"KXNORB", "KXNORB", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ed46d9"},
{"KXORW", "KXORW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec47d9"},
{"KXORQ", "KXORQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d447f4"},
{"KORTESTW", "KORTESTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f898d1"},
{"KORTESTB", "KORTESTB", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c5f998e3"},
{"KORTESTQ", "KORTESTQ", []Operand{vreg(t, "K5"), vreg(t, "K6")}, "c4e1f898f5"},
{"KTESTW", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f899d1"},
{"KTESTD", "KTESTD", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c4e1f999e3"},
{"KSHIFTLB", "KSHIFTLB", []Operand{Imm(1), vreg(t, "K1"), vreg(t, "K2")}, "c4e37932d101"},
{"KSHIFTLD", "KSHIFTLD", []Operand{Imm(2), vreg(t, "K3"), vreg(t, "K4")}, "c4e37933e302"},
{"KSHIFTLQ", "KSHIFTLQ", []Operand{Imm(3), vreg(t, "K5"), vreg(t, "K6")}, "c4e3f933f503"},
{"KSHIFTRB", "KSHIFTRB", []Operand{Imm(4), vreg(t, "K1"), vreg(t, "K2")}, "c4e37930d104"},
{"KSHIFTRW", "KSHIFTRW", []Operand{Imm(5), vreg(t, "K3"), vreg(t, "K4")}, "c4e3f930e305"},
{"KSHIFTRQ", "KSHIFTRQ", []Operand{Imm(6), vreg(t, "K5"), vreg(t, "K6")}, "c4e3f931f506"},
// Integer compares with an opmask destination (0F3A map, the
// go-bzip2 partition kernel's classify instructions).
{"VPCMPUB", "VPCMPUB", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X0"), vreg(t, "K1")}, "62f37d083ec901"},
{"VPCMPB", "VPCMPB", []Operand{Imm(2), vreg(t, "Y2"), vreg(t, "Y3"), vreg(t, "K2")}, "62f365283fd202"},
{"VPCMPUW", "VPCMPUW", []Operand{Imm(5), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3")}, "62f3ed483ed905"},
{"VPCMPW", "VPCMPW", []Operand{Imm(6), vreg(t, "X3"), vreg(t, "X4"), vreg(t, "K4")}, "62f3dd083fe306"},
{"VPCMPD", "VPCMPD", []Operand{Imm(0), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "K1")}, "62f36d281fc900"},
{"VPCMPUD", "VPCMPUD", []Operand{Imm(1), vreg(t, "Z2"), vreg(t, "Z3"), vreg(t, "K2")}, "62f365481ed201"},
{"VPCMPQ", "VPCMPQ", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K3")}, "62f3ed081fd902"},
{"VPCMPUQ", "VPCMPUQ", []Operand{Imm(3), vreg(t, "Y3"), vreg(t, "Y4"), vreg(t, "K4")}, "62f3dd281ee303"},
// Lane extract / insert. // Lane extract / insert.
{"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"}, {"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"},
{"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"}, {"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"},
+106 -21
View File
@@ -14,8 +14,8 @@ import (
"sync" "sync"
) )
// This file emits GOOBJ — the Go toolchain's object format, which cmd/link // This file emits GOOBJ, the Go toolchain's object format, which cmd/link
// consumes directly — so gasm-assembled functions drop into a go build // consumes directly, so gasm-assembled functions drop into a go build
// without the Go assembler. The layout follows cmd/internal/goobj: a // without the Go assembler. The layout follows cmd/internal/goobj: a
// toolchain preamble ("go object ...\n!\n"), the go120ld header with its // toolchain preamble ("go object ...\n!\n"), the go120ld header with its
// block offsets, a string table, symbol definitions, the relocation / // block offsets, a string table, symbol definitions, the relocation /
@@ -33,7 +33,10 @@ import (
// methods supply the toolchain preamble, the MinLC (pc-value delta unit) // methods supply the toolchain preamble, the MinLC (pc-value delta unit)
// and the relocation-type mapping for code relocations. // and the relocation-type mapping for code relocations.
// GOOBJ block indices (cmd/internal/goobj). // GOOBJ block indices (cmd/internal/goobj). These MUST match the real
// archive layout: the emitter writes the header offsets per index and the
// reader (groundtruth, goobj_resolve) parses real Go archives with them.
// blkAutolib is unused by the emitter but still defines index 0.
const ( const (
blkAutolib = iota blkAutolib = iota
blkPkgIdx blkPkgIdx
@@ -91,10 +94,12 @@ const (
) )
// Relocation types (cmd/internal/objabi). // Relocation types (cmd/internal/objabi).
// R_PCREL and R_ADDR are stable across Go versions. // R_ADDR, R_CALL, R_PCREL and R_TLS_LE are stable across Go versions.
const ( const (
relocPCRel = 14 // R_PCREL
relocAddr = 1 // R_ADDR relocAddr = 1 // R_ADDR
relocCall = 7 // R_CALL
relocPCRel = 14 // R_PCREL
relocTLSLE = 15 // R_TLS_LE
) )
// relocDWTXTADDRU4 returns the R_DWTXTADDR_U4 relocation type for the // relocDWTXTADDRU4 returns the R_DWTXTADDR_U4 relocation type for the
@@ -137,10 +142,30 @@ func isGo127OrLater() bool {
// Special package indices for symbol references. // Special package indices for symbol references.
const ( const (
pkgIdxNone = 0x7fffffff pkgIdxNone = 0x7fffffff
pkgIdxSelf = 0x7ffffffb pkgIdxSelf = 0x7ffffffb
pkgIdxBuiltin = 0x7ffffffc
) )
// goobjBuiltinMorestackNoctxt is the index of runtime.morestack_noctxt in
// cmd/internal/goobj/builtinlist.go of the toolchain the object targets
// (246 since Go 1.25; the list is append-only).
const goobjBuiltinMorestackNoctxt = 246
// goobjBuiltinMorestack is the builtin reference the toolchain emits for the
// stack-guard call.
var goobjBuiltinMorestack = "runtime\u00b7morestack_noctxt"
// isCallReloc reports whether k is one of the per-arch call relocations a
// direct branch to a TEXT symbol carries.
func isCallReloc(k RelocKind) bool {
switch k {
case RelCall, RelRISCVJal, RelArm64Branch, RelLoong64Branch:
return true
}
return false
}
const goobjMagic = "\x00go120ld" const goobjMagic = "\x00go120ld"
// goSym is one symbol definition under construction. // goSym is one symbol definition under construction.
@@ -175,14 +200,24 @@ type dwarfRelocSet struct {
// does with its -p flag). srcPath names the source file recorded in the // does with its -p flag). srcPath names the source file recorded in the
// object's file table and line tables. The toolchain's object preamble is // object's file table and line tables. The toolchain's object preamble is
// captured from the installed go tool asm, so the output links with the // captured from the installed go tool asm, so the output links with the
// toolchain it was produced on — exactly like a real assembly object. // toolchain it was produced on, exactly like a real assembly object.
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) { func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreamble() pre, err := toolchainObjectPreamble()
if err != nil { if err != nil {
return nil, err return nil, err
} }
// amd64: MinLC 1, R_PCREL for the code relocations. // amd64: MinLC 1, R_PCREL for displacements, R_CALL for calls and
return img.emitGOObject(pkgPath, srcPath, pre, 1, func(Reloc) (uint16, uint8) { return relocPCRel, 4 }) // R_TLS_LE for the stack-guard TLS load.
return img.emitGOObject(pkgPath, srcPath, pre, 1, func(r Reloc) (uint16, uint8) {
switch r.Kind {
case RelCall:
return relocCall, 4
case RelTLSLE:
return relocTLSLE, 4
default:
return relocPCRel, 4
}
})
} }
// emitGOObject assembles the GOOBJ payload for any architecture. pre is // emitGOObject assembles the GOOBJ payload for any architecture. pre is
@@ -195,7 +230,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)") return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
} }
// The non-package definitions first — the DWARF symbols reference the // The non-package definitions first, the DWARF symbols reference the
// functions by these indices: per function the four pc-value tables // functions by these indices: per function the four pc-value tables
// and the function itself, as cmd/asm lays them out. // and the function itself, as cmd/asm lays them out.
type npSym struct { type npSym struct {
@@ -244,7 +279,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
// their relocations cover whole AUIPC/pcalau12i pairs, so // their relocations cover whole AUIPC/pcalau12i pairs, so
// zeroing r.Off would erase the opcode/register bits the linker // zeroing r.Off would erase the opcode/register bits the linker
// preserves when it patches only the immediate. // preserves when it patches only the immediate.
if r.Kind != RelPCRel32 { if r.Kind != RelPCRel32 && r.Kind != RelCall {
continue continue
} }
if r.Off >= 0 && r.Off+4 <= len(code) { if r.Off >= 0 && r.Off+4 <= len(code) {
@@ -317,6 +352,13 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
) )
} }
// Index the non-package TEXT definitions by short name for the internal
// call references.
textNpIdx := map[string]int{}
for i, fn := range img.Funcs {
textNpIdx[fn.Name] = fnNpIdx[i]
}
// Resolve external symbol references (cross-package). Build the // Resolve external symbol references (cross-package). Build the
// package index table and determine each external symbol's SymIdx // package index table and determine each external symbol's SymIdx
// by reading the target package's export data. // by reading the target package's export data.
@@ -324,10 +366,20 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
var extPkgIdx map[string]int var extPkgIdx map[string]int
var extSymIdx map[string]int var extSymIdx map[string]int
if len(img.Externals) > 0 { if len(img.Externals) > 0 {
var err error // The morestack call is a builtin reference, not a resolved external.
extPkgTable, extPkgIdx, extSymIdx, err = resolveExternalSymbols(img.Externals) var need []string
if err != nil { for _, n := range img.Externals {
return nil, fmt.Errorf("GOOBJ emission: resolving external symbols: %w", err) if n == goobjBuiltinMorestack {
continue
}
need = append(need, n)
}
if len(need) > 0 {
var err error
extPkgTable, extPkgIdx, extSymIdx, err = resolveExternalSymbols(need)
if err != nil {
return nil, fmt.Errorf("GOOBJ emission: resolving external symbols: %w", err)
}
} }
} }
@@ -339,6 +391,31 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
si := len(defs) + fnNpIdx[i] si := len(defs) + fnNpIdx[i]
for _, r := range fn.Relocs { for _, r := range fn.Relocs {
typ, size := relocField(r) typ, size := relocField(r)
if r.Kind == RelTLSLE {
// The TLS load has no symbol: {0, 0} is the nil ref.
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], 0)
binary.LittleEndian.PutUint32(rec[19:], 0)
symRelocs[si] = append(symRelocs[si], rec[:]...)
continue
}
if r.External && r.Name == goobjBuiltinMorestack {
// The stack-guard morestack call uses the toolchain's
// builtin reference.
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxBuiltin)
binary.LittleEndian.PutUint32(rec[19:], goobjBuiltinMorestackNoctxt)
symRelocs[si] = append(symRelocs[si], rec[:]...)
continue
}
if r.External { if r.External {
// Split package-qualified name: "runtime·morestack" → runtime, morestack. // Split package-qualified name: "runtime·morestack" → runtime, morestack.
pkg, name := splitQualified(r.Name) pkg, name := splitQualified(r.Name)
@@ -363,16 +440,24 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
symRelocs[si] = append(symRelocs[si], rec[:]...) symRelocs[si] = append(symRelocs[si], rec[:]...)
continue continue
} }
pkg := uint32(pkgIdxSelf)
di, ok := defIdx[r.Name] di, ok := defIdx[r.Name]
if !ok { if !ok {
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name) // A call to a TEXT function of the same file references the
// non-package definition table.
ni, isText := textNpIdx[r.Name]
if !isText || !isCallReloc(r.Kind) {
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
}
pkg = pkgIdxNone
di = ni
} }
var rec [23]byte var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off))) binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size // field width rec[4] = size // field width
binary.LittleEndian.PutUint16(rec[5:], typ) binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend)) binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf) binary.LittleEndian.PutUint32(rec[15:], pkg)
binary.LittleEndian.PutUint32(rec[19:], uint32(di)) binary.LittleEndian.PutUint32(rec[19:], uint32(di))
symRelocs[si] = append(symRelocs[si], rec[:]...) symRelocs[si] = append(symRelocs[si], rec[:]...)
} }
@@ -417,7 +502,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
// The string table. Absolute offsets: it starts right after the // The string table. Absolute offsets: it starts right after the
// 96-byte header (magic, fingerprint, flags, the 19 block offsets). // 96-byte header (magic, fingerprint, flags, the 19 block offsets).
const headerSize = 8 + 8 + 4 + 4*(blkEnd+1) const headerSize = 8 + 8 + 4 + 4*(blkEnd+1)
strTab := []byte{} var strTab []byte
strOff := map[string]uint32{} strOff := map[string]uint32{}
addStr := func(s string) { addStr := func(s string) {
if _, ok := strOff[s]; ok { if _, ok := strOff[s]; ok {
@@ -464,7 +549,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
auxIdxBlk := make([]byte, 0, 4*(nsyms+1)) auxIdxBlk := make([]byte, 0, 4*(nsyms+1))
dataIdxBlk := make([]byte, 0, 4*(nsyms+1)) dataIdxBlk := make([]byte, 0, 4*(nsyms+1))
var nr, na, nd uint32 var nr, na, nd uint32
for si := 0; si < nsyms; si++ { for si := range nsyms {
relocIdxBlk = binary.LittleEndian.AppendUint32(relocIdxBlk, nr) relocIdxBlk = binary.LittleEndian.AppendUint32(relocIdxBlk, nr)
auxIdxBlk = binary.LittleEndian.AppendUint32(auxIdxBlk, na) auxIdxBlk = binary.LittleEndian.AppendUint32(auxIdxBlk, na)
dataIdxBlk = binary.LittleEndian.AppendUint32(dataIdxBlk, nd) dataIdxBlk = binary.LittleEndian.AppendUint32(dataIdxBlk, nd)
@@ -505,7 +590,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
// The fingerprint stays zero, as cmd/asm leaves it. // The fingerprint stays zero, as cmd/asm leaves it.
binary.LittleEndian.PutUint32(payload[16:], 4) // ObjFlagFromAssembly binary.LittleEndian.PutUint32(payload[16:], 4) // ObjFlagFromAssembly
off := uint32(headerSize + len(strTab)) off := uint32(headerSize + len(strTab))
for i := 0; i < blkEnd; i++ { for i := range blkEnd {
binary.LittleEndian.PutUint32(payload[20+4*i:], off) binary.LittleEndian.PutUint32(payload[20+4*i:], off)
off += uint32(len(blocks[i])) off += uint32(len(blocks[i]))
} }
+1 -4
View File
@@ -139,10 +139,7 @@ func dwSelectOpcode(deltaPC uint64, deltaLC int64) int64 {
return int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*int64(deltaPC) return int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*int64(deltaPC)
default: default:
if deltaPC <= uint64(dwPCRange) { if deltaPC <= uint64(dwPCRange) {
op := int64(dwOpcodeBase) + (dwLineRange - 1) + dwLineRange*int64(deltaPC) op := min(int64(dwOpcodeBase)+(dwLineRange-1)+dwLineRange*int64(deltaPC), 255)
if op > 255 {
op = 255
}
return op return op
} }
switch deltaPC - uint64(dwPCRange) { switch deltaPC - uint64(dwPCRange) {
+4 -22
View File
@@ -12,24 +12,6 @@ import (
"strings" "strings"
) )
// readGOOBJSymbols reads the GOOBJ symbol definitions from a compiled Go
// package's export file. The file is an ar archive containing a __.PKGDEF
// member whose payload is the "go object ...\n!\n" preamble followed by the
// GOOBJ data. The function returns the symbol names in definition order
// (the order they appear in blkSymdef), which matches the SymIdx the linker
// expects for cross-package references.
func readGOOBJSymbols(exportPath string) ([]string, error) {
data, err := os.ReadFile(exportPath)
if err != nil {
return nil, err
}
goobj, err := extractGOOBJ(data)
if err != nil {
return nil, fmt.Errorf("%s: %w", exportPath, err)
}
return goobj.symbols(), nil
}
// exportPath returns the export file path for a given import path by running // exportPath returns the export file path for a given import path by running
// "go list -export". The result is cached so repeated calls for the same // "go list -export". The result is cached so repeated calls for the same
// package are fast. // package are fast.
@@ -102,7 +84,7 @@ func sortedPkgRefs(refs map[string][]string) []pkgRef {
for pkg, syms := range refs { for pkg, syms := range refs {
pkgs = append(pkgs, pkgRef{pkg, syms}) pkgs = append(pkgs, pkgRef{pkg, syms})
} }
// Simple insertion sort — the list is tiny (usually 1–3 packages). // Simple insertion sort, the list is tiny (usually 1-3 packages).
for i := 1; i < len(pkgs); i++ { for i := 1; i < len(pkgs); i++ {
for j := i; j > 0 && pkgs[j-1].path > pkgs[j].path; j-- { for j := i; j > 0 && pkgs[j-1].path > pkgs[j].path; j-- {
pkgs[j-1], pkgs[j] = pkgs[j], pkgs[j-1] pkgs[j-1], pkgs[j] = pkgs[j], pkgs[j-1]
@@ -221,7 +203,7 @@ func (f *goobjFile) readSymNames(block []byte) []string {
} }
n := len(block) / recSize n := len(block) / recSize
names := make([]string, 0, n) names := make([]string, 0, n)
for i := 0; i < n; i++ { for i := range n {
rec := block[i*recSize : (i+1)*recSize] rec := block[i*recSize : (i+1)*recSize]
nameLen := binary.LittleEndian.Uint32(rec[0:4]) nameLen := binary.LittleEndian.Uint32(rec[0:4])
nameOff := binary.LittleEndian.Uint32(rec[4:8]) nameOff := binary.LittleEndian.Uint32(rec[4:8])
@@ -341,8 +323,8 @@ func splitQualified(full string) (pkg, name string) {
if idx := strings.IndexByte(full, '\u00b7'); idx >= 0 { if idx := strings.IndexByte(full, '\u00b7'); idx >= 0 {
return full[:idx], full[idx+len("\u00b7"):] return full[:idx], full[idx+len("\u00b7"):]
} }
if idx := strings.IndexByte(full, '.'); idx >= 0 { if before, after, ok := strings.Cut(full, "."); ok {
return full[:idx], full[idx+1:] return before, after
} }
return "", full return "", full
} }
+2 -2
View File
@@ -393,7 +393,7 @@ func main() {
} }
var work string var work string
var asmObj, pkgArch, linkLine string var asmObj, pkgArch, linkLine string
for _, line := range strings.Split(string(buildLog), "\n") { for line := range strings.SplitSeq(string(buildLog), "\n") {
switch { switch {
case strings.HasPrefix(line, "WORK="): case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=") work = strings.TrimPrefix(line, "WORK=")
@@ -461,7 +461,7 @@ func main() {
newArch := filepath.Join(dir, "pkg.a") newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch} args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{} seen := map[string]bool{}
for _, m := range strings.Fields(string(listOut)) { for m := range strings.FieldsSeq(string(listOut)) {
if seen[m] { if seen[m] {
continue continue
} }
+16 -8
View File
@@ -13,25 +13,33 @@ import (
) )
// GOObjectAARCH64 emits a GOOBJ object file for AArch64. The layout is // GOObjectAARCH64 emits a GOOBJ object file for AArch64. The layout is
// the shared one in goobj.go — the toolchain preamble, the go120ld header // the shared one in goobj.go, the toolchain preamble, the go120ld header
// with its block offsets, the string table, the symbol definitions and the // with its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the arm64 preamble, the MinLC of 4 // reloc/aux/data index arrays, with the arm64 preamble, the MinLC of 4
// for the pc-value deltas, and R_ADDRARM64 relocation types for the // for the pc-value deltas, and the arm64 relocation types for the ADRP
// ADRP+ADD/LDR/STR address pairs. // pairs and BL calls.
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) { func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreambleAARCH64() pre, err := toolchainObjectPreambleAARCH64()
if err != nil { if err != nil {
return nil, err return nil, err
} }
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) { return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
return relocArm64Addr, 4 switch r.Kind {
case RelArm64Branch:
return relocArm64Branch, 4
case RelArm64LDST64:
return relocArm64LDST64, 4
default:
return relocArm64Addr, 4
}
}) })
} }
// arm64 relocation types (cmd/internal/objabi). R_ADDRARM64 resolves an // arm64 relocation types (cmd/internal/objabi).
// ADRP+ADD/LDR/STR pair to a symbol's address.
const ( const (
relocArm64Addr = 9 // R_ADDRARM64 relocArm64Addr = 3 // R_ADDRARM64, ADRP+ADD pair
relocArm64Branch = 9 // R_CALLARM64, BL instruction
relocArm64LDST64 = 40 // R_ARM64_PCREL_LDST64, ADRP+LDR/STR pair
) )
// toolchainObjectPreambleAARCH64 returns the "go object ...\n!\n" header // toolchainObjectPreambleAARCH64 returns the "go object ...\n!\n" header
+11 -5
View File
@@ -13,9 +13,9 @@ import (
) )
// GOObjectLOONG64 emits a GOOBJ object file for LoongArch. The layout is // GOObjectLOONG64 emits a GOOBJ object file for LoongArch. The layout is
// the shared one in goobj.go — the toolchain preamble, the go120ld header // the shared one in goobj.go, the toolchain preamble, the go120ld header
// with its block offsets, the string table, the symbol definitions and the // with its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the loong64 preamble, the MinLC of 4 // reloc/aux/data index arrays, with the loong64 preamble, the MinLC of 4
// for the pc-value deltas, and R_LOONG64_ADDR_HI/LO relocation types for // for the pc-value deltas, and R_LOONG64_ADDR_HI/LO relocation types for
// the pcalau12i+addi.d address pairs. // the pcalau12i+addi.d address pairs.
func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) { func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
@@ -25,11 +25,16 @@ func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
} }
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) { return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
// A pcalau12i+addi.d pair: the high part carries // A pcalau12i+addi.d pair: the high part carries
// R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO. // R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO; the guard's
if r.Kind == RelLoong64AddrLo { // morestack call carries R_CALLLOONG64.
switch {
case r.Kind == RelLoong64AddrLo:
return relocLoong64AddrLo, 4 return relocLoong64AddrLo, 4
case r.Kind == RelLoong64Branch:
return relocCallLoong64, 4
default:
return relocLoong64AddrHi, 4
} }
return relocLoong64AddrHi, 4
}) })
} }
@@ -39,6 +44,7 @@ func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
const ( const (
relocLoong64AddrHi = 77 // R_LOONG64_ADDR_HI relocLoong64AddrHi = 77 // R_LOONG64_ADDR_HI
relocLoong64AddrLo = 78 // R_LOONG64_ADDR_LO relocLoong64AddrLo = 78 // R_LOONG64_ADDR_LO
relocCallLoong64 = 84 // R_CALLLOONG64
) )
// toolchainObjectPreambleLOONG64 returns the "go object ...\n!\n" header // toolchainObjectPreambleLOONG64 returns the "go object ...\n!\n" header
+2 -2
View File
@@ -13,9 +13,9 @@ import (
) )
// GOObjectRISCV emits a GOOBJ object file for RISC-V. The layout is the // GOObjectRISCV emits a GOOBJ object file for RISC-V. The layout is the
// shared one in goobj.go — the toolchain preamble, the go120ld header with // shared one in goobj.go, the toolchain preamble, the go120ld header with
// its block offsets, the string table, the symbol definitions and the // its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the RISC-V preamble, the MinLC of 2 for // reloc/aux/data index arrays, with the RISC-V preamble, the MinLC of 2 for
// the pc-value deltas, and the single R_RISCV_PCREL_ITYPE/STYPE relocation // the pc-value deltas, and the single R_RISCV_PCREL_ITYPE/STYPE relocation
// per AUIPC pair, matching `go tool asm`'s model (each pair is one 8-byte // per AUIPC pair, matching `go tool asm`'s model (each pair is one 8-byte
// relocation, not the ELF HI20/LO12 pair). // relocation, not the ELF HI20/LO12 pair).
+283
View File
@@ -0,0 +1,283 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/hex"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64,
// verified with go tool objdump): the stack-split guard classes, the morestack
// block and the auto-NOSPLIT leaf behaviour.
func TestStackGuardBytes(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"554889e54883ec104883c4105dc3"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"644c8b3425000000004c8da42478ffffff4d3b66107614554889e54881ec000100004881c4000100005dc3e800000000ebce"},
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"644c8b3425000000004989e44981ec881f0000721a4d3b66107614554889e54881ec002000004881c4002000005dc3e800000000ebca"},
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"644c8b342500000000493b66107613554889e54883ec10e8000000004883c4105dc3e800000000ebd7"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"554889e54883ec104883c4105dc3"},
} {
f, errs := parser.Parse("g_amd64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
// The toolchain's object leaves every relocation field zero for the
// linker, while the gasm image resolves file-internal references, so
// the comparison masks the patch sites the way verify's ground truth
// does.
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// TestStackGuardRelocs checks the guard's patch sites: the TLS slot and the
// morestack call.
func TestStackGuardRelocs(t *testing.T) {
f, errs := parser.Parse("g_amd64.s", "TEXT \u00b7f(SB), $256-0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
relocs := img.Funcs[0].Relocs
if len(relocs) != 2 {
t.Fatalf("relocs = %d, want 2", len(relocs))
}
tls, call := relocs[0], relocs[1]
if tls.Kind != RelTLSLE || tls.Off != 5 || tls.Name != "" || tls.External {
t.Errorf("tls reloc = %+v, want RelTLSLE at 5 with no symbol", tls)
}
if call.Kind != RelCall || call.Name != "runtime\u00b7morestack_noctxt" || !call.External {
t.Errorf("call reloc = %+v, want RelCall to runtime.morestack_noctxt", call)
}
}
// TestStackGuardGOObj emissions succeed with the guard's TLS and builtin
// references in play.
func TestStackGuardGOObj(t *testing.T) {
f, errs := parser.Parse("g_amd64.s", "TEXT \u00b7f(SB), $256-0\n\tCALL \u00b7helper(SB)\n\tRET\nTEXT \u00b7helper(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
obj, err := img.GOObject("testpkg", "g_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
if !bytes.Contains(obj, []byte("go120ld")) {
t.Fatal("object lacks the GOOBJ magic")
}
}
// The arm64 stack-split guard, pinned from `go tool asm` (Go 1.27, arm64):
// the guard classes, the auto-NOSPLIT leaf behaviour and the morestack
// block. Relocation fields are masked: the toolchain's object leaves them
// zero for the linker, the gasm image resolves file-internal references.
func TestStackGuardBytesARM64(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"900b40f9f14302d13f0210eb09010054f44304d19dfa3fa99f020091fd2300d1fd230491ff430491c0035fd6e3031eaa00000000f3ffff17"},
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"900b40f91bf283d2f1633beba30100543f0210eb690100541b0284d2f4633bcb9dfa3fa99f020091fd2300d11b0184d2fd633b8b1b0284d2ff633b8bc0035fd6e3031eaa00000000eeffff17"},
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"900b40f9ff6330eb09010054fe0f1ef8fd831ff8fd2300d100000000fd835ff8fe0742f8c0035fd6e3031eaa00000000f4ffff17"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"},
} {
f, errs := parser.Parse("g_arm64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// The riscv64 stack-split guard, pinned from `go tool asm` (Go 1.27,
// riscv64): the morestack call sits between the guard and the body, and the
// guard branches forward over it. Relocation fields are masked.
func TestStackGuardBytesRISCV64(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"03b30d0163662300000000006ff05fff233411fe211106e08260610167800000"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"03b30d01930381f763667300000000006ff01fff233c11ee130181ef06e082601301811067800000"},
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"03b30d0189639b8383f863697100f97f9b8f8f07b303f10163667300000000006ff01ffef97f8a9f23bc1ffef97fe13f7e9106e08260896fa12f7e9167800000"},
{"frameless", "TEXT \u00b7frameless(SB), $0-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"03b30d0163662300000000006ff05fff233c11fe611106e0000000008260210167800000"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"233411fe211106e08260610167800000"},
} {
f, errs := parser.Parse("g_riscv64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// The loong64 stack-split guard, pinned from `go tool asm` (Go 1.27,
// loong64): every guard class (including the medium class with the
// materialised constant and the big class with the ORI-less constants), the
// auto-NOSPLIT leaf behaviour, the large-frame R30 prologue/epilogue forms
// and the morestack block. Relocation fields are masked.
func TestStackGuardBytesLOONG64(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"61a0ff2963a0ff026100c0296360c0022000004c"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"d442c02878e0fd0294e21200801a004061e0fb2963e0fb026100c0296320c4022000004c3f00150000000000ffd7ff53"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"61a0ff2963a0ff026100c0296360c0022000004c"},
// The LR store leaves the 12-bit store-offset range while the SP
// adjust immediate still fits, and the epilogue adjusts through a
// single ORI.
{"fit2048", "TEXT \u00b7fit2048(SB), $2040-0\n\tRET\n",
"d442c0287800e20294e21200802600401e000014de8f1000c103e0296300e0026100c0291e00a00363f810002000004c3f00150000000000ffcbff53"},
// Medium class at the materialisation boundary (off = 2048 still
// immediate, 2049+ goes through R30).
{"med2048off", "TEXT \u00b7med2048off(SB), $2168-0\n\tRET\n",
"d442c0287800e00294e21200802e0040feffff15de8f1000c103de29feffff15de039e0363f810006100c0291e00a20363f810002000004c3f00150000000000ffc3ff53"},
{"medmat", "TEXT \u00b7medmat(SB), $2176-0\n\tRET\n",
"d442c028feffff15dee39f0378f8100094e21200802e0040feffff15de8f1000c1e3dd29feffff15dee39d0363f810006100c0291e20a20363f810002000004c3f00150000000000ffbbff53"},
// Big class with the rounding-split store and the floor-split adjust.
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"d442c0283e000014de23be0378f8120000470044deffff15dee3810378f8100094e2120080320040deffff15de8f1000c1e3ff29beffff15dee3bf0363f810006100c0295e000014de23800363f810002000004c3f00150000000000ffa7ff53"},
// Zero low 12 bits drop the ORI from the store, the adjust and the
// epilogue materialisation.
{"bigzero", "TEXT \u00b7bigzero(SB), $4088-0\n\tRET\n",
"d442c028feffff15de03820378f8100094e21200802a0040feffff15de8f1000c103c029feffff1563f810006100c0293e00001463f810002000004c3f00150000000000ffbfff53"},
// Big class whose first constant has a zero high part: a single ORI.
{"big3976", "TEXT \u00b7big3976(SB), $4096-0\n\tRET\n",
"d442c0281e20be0378f8120000470044feffff15dee3810378f8100094e2120080320040feffff15de8f1000c1e3ff29deffff15dee3bf0363f810006100c0293e000014de23800363f810002000004c3f00150000000000ffabff53"},
// Big class at a multiple of 4096: both guard constants lose their
// ORI word.
{"giantlo0", "TEXT \u00b7giantlo0(SB), $4216-0\n\tRET\n",
"d442c0283e00001478f8120000430044feffff1578f8100094e2120080320040feffff15de8f1000c103fe29deffff15de03be0363f810006100c0293e000014de03820363f810002000004c3f00150000000000ffafff53"},
// Non-leaf big frame: the body call plus the LR restore epilogue.
{"callbig", "TEXT \u00b7callbig(SB), $8192-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"d442c0283e000014de23be0378f81200004f0044deffff15dee3810378f8100094e21200803a0040deffff15de8f1000c1e3ff29beffff15dee3bf0363f810006100c029000000006100c0285e000014de23800363f810002000004c3f00150000000000ff9fff53"},
} {
f, errs := parser.Parse("g_loong64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// TestStackGuardGOObjInternalCall checks that GOOBJ emission succeeds when a
// guarded function calls a TEXT symbol of the same file, for every arch's
// call relocation kind.
func TestStackGuardGOObjInternalCall(t *testing.T) {
for _, tt := range []struct {
src string
assemble func(*ast.File) (*Image, error)
}{
{"g_amd64.s", AssembleFile},
{"g_arm64.s", AssembleFileARM64},
{"g_riscv64.s", AssembleFileRISCV},
{"g_loong64.s", AssembleFileLOONG64},
} {
f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.src, errs)
}
img, err := tt.assemble(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.src, err)
}
if _, err := img.GOObject("testpkg", tt.src); err != nil {
t.Errorf("%s: GOObject: %v", tt.src, err)
}
}
}
+254 -21
View File
@@ -21,7 +21,7 @@ var aluOp = map[string]struct {
} }
// unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use // unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use
// the 0xFE/0xFF group (the short 0x40–0x4F forms are REX prefixes in 64-bit // the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes in 64-bit
// mode); NEG/NOT use the 0xF6/0xF7 group. // mode); NEG/NOT use the 0xF6/0xF7 group.
var unaryOp = map[string]struct { var unaryOp = map[string]struct {
digit int digit int
@@ -33,7 +33,7 @@ var unaryOp = map[string]struct {
"NEG": {3, 0xF7}, "NEG": {3, 0xF7},
} }
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0–0xD3 group. // shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0-0xD3 group.
var shiftOp = map[string]int{ var shiftOp = map[string]int{
"SHL": 4, "SHL": 4,
"SHR": 5, "SHR": 5,
@@ -48,11 +48,62 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
} }
src, dst := ops[0], ops[1] src, dst := ops[0], ops[1]
// Integer scalar XMM moves: MOVQ with an XMM operand is the SSE2
// packed-quadword move, NOT a GPR move: mem→xmm encodes as F3 0F 7E
// (reg = dst, no REX.W, the Go assembler's form), xmm→mem as
// 66 0F D6 (rm = xmm). Register forms against a GPR use the MOVD
// opcodes with REX.W instead: 66 REX.W 0F 6E (gpr→xmm) and
// 66 REX.W 0F 7E (xmm→gpr); the memory opcodes with a register r/m
// would be undefined forms. MOVL is the packed-dword move:
// 66 0F 6E load, 66 0F 7E store, no REX.W. A GPR-move fallback would
// silently emit REX.W 8B with the wrong operand meaning.
_, srcVec := vecReg(src)
dstReg, dstVec := vecReg(dst)
if srcVec || dstVec {
if dstVec {
if g, ok := src.(Reg); ok && !g.isVec() {
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x6E}, modrm: -1, sib: -1, rexW: size == 8}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
return e.emit(i)
}
i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1}
if size == 4 {
i.prefix = 0x66
i.opcode = []byte{0x0F, 0x6E}
}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
return e.emit(i)
}
srcXMM, srcIsXMM := src.(Reg)
if !srcIsXMM || !srcXMM.isVec() {
return fmt.Errorf("MOV: store needs an XMM source")
}
if g, ok := dst.(Reg); ok && !g.isVec() {
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1, rexW: size == 8}
if err := setRM(i, srcXMM, dst, 8); err != nil {
return err
}
return e.emit(i)
}
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
if size == 4 {
i.opcode = []byte{0x0F, 0x7E}
}
if err := setRM(i, srcXMM, dst, 8); err != nil {
return err
}
return e.emit(i)
}
dstReg, dstIsReg := dst.(Reg) dstReg, dstIsReg := dst.(Reg)
switch src := src.(type) { switch src := src.(type) {
case Reg: case Reg:
if dstIsReg { if dstIsReg {
// MOV r/m, r: 0x88/0x89, reg=src, rm=dst — the form the Go // MOV r/m, r: 0x88/0x89, reg=src, rm=dst, the form the Go
// assembler emits for register-to-register moves. // assembler emits for register-to-register moves.
i := newInstr(size, []byte{movRM(size)}) i := newInstr(size, []byte{movRM(size)})
if err := setRM(i, src, dst, size); err != nil { if err := setRM(i, src, dst, size); err != nil {
@@ -91,7 +142,28 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
case Imm: case Imm:
if dstIsReg { if dstIsReg {
// MOV r, imm: 0xB0+reg (8-bit) / 0xB8+reg (16/32/64, imm64 for Q). v := int64(src)
// The Go assembler compresses 64-bit moves whose immediate fits
// a signed int32, choosing per sign:
// v >= 0: B8+rd imm32 without REX.W (zero-extended by the
// hardware, REX.B still emitted for R8-R15);
// v < 0: REX.W C7 /0 imm32 (sign-extended, the plain B8+rd
// form would zero-extend and corrupt the value).
// Out-of-range immediates keep the B8+rd imm64 form.
if size == 8 && v >= 0 && v <= (1<<31)-1 {
i := newInstr(4, []byte{0xB8 + byte(dstReg.idx&7)})
i.rexB = dstReg.idx >= 8
i.imm = le32(v)
return e.emit(i)
}
if size == 8 && v < 0 && v >= -(1<<31) {
i := newInstr(8, []byte{0xC7})
if err := setRMDigit(i, 0, dstReg, 8); err != nil {
return err
}
i.imm = le32(v)
return e.emit(i)
}
opBase := byte(0xB8) opBase := byte(0xB8)
if size == 1 { if size == 1 {
opBase = 0xB0 opBase = 0xB0
@@ -101,7 +173,7 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
if dstReg.needsREX(size) { if dstReg.needsREX(size) {
i.rexForced = true i.rexForced = true
} }
i.imm = immediate(int64(src), size, true) i.imm = immediate(v, size, true)
return e.emit(i) return e.emit(i)
} }
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0. // MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
@@ -144,12 +216,18 @@ func (e *enc) encodeALU(op struct {
} }
src, dst := ops[0], ops[1] src, dst := ops[0], ops[1]
// CMP never takes its immediate first: the Go assembler rejects
// CMPL $0, AX outright (only CMPL AX, $0 is legal, unlike TEST and the
// writing ALU ops whose immediate is naturally the source).
if imm, ok := src.(Imm); ok { if imm, ok := src.(Imm); ok {
if op.digit == 7 {
return fmt.Errorf("CMP immediate must be the second operand (reg, $imm)")
}
return e.encodeALUImm(op.digit, dst, int64(imm), size) return e.encodeALUImm(op.digit, dst, int64(imm), size)
} }
// CMP accepts the immediate in the second position too — CMPL CX, $31 is // CMP accepts the immediate in the second position too, CMPL CX, $31 is
// the form the Go assembler itself accepts — and encodes it identically // the form the Go assembler itself accepts, and encodes it identically
// (CMP r/m, imm sets the flags as first − second). No other ALU op takes // (CMP r/m, imm sets the flags as first − second). No other ALU op takes
// an immediate destination. // an immediate destination.
if imm, ok := dst.(Imm); ok { if imm, ok := dst.(Imm); ok {
@@ -235,6 +313,15 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
i.imm = []byte{byte(int8(imm))} i.imm = []byte{byte(int8(imm))}
return e.emit(i) return e.emit(i)
} }
// 0x81 /digit, imm16/imm32, or the Go assembler's accumulator short
// form (opcode+5, no ModR/M) when the destination is AX/AL, which it
// prefers over the generic form exactly here.
if r, ok := dst.(Reg); ok && r.idx == 0 {
accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit]
i := newInstr(size, []byte{accOp})
i.imm = immediate(imm, size, false)
return e.emit(i)
}
// 0x81 /digit, imm16/imm32. // 0x81 /digit, imm16/imm32.
i := newInstr(size, []byte{0x81}) i := newInstr(size, []byte{0x81})
if err := setRMDigit(i, digit, dst, size); err != nil { if err := setRMDigit(i, digit, dst, size); err != nil {
@@ -252,7 +339,18 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
} }
src, dst := ops[0], ops[1] src, dst := ops[0], ops[1]
if imm, ok := src.(Imm); ok { if imm, ok := src.(Imm); ok {
// TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0. // TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0, but the Go assembler
// always uses the accumulator forms (A8/A9, no ModR/M) when the
// register operand is AL/AX, whatever the immediate's width.
if r, ok := dst.(Reg); ok && r.idx == 0 {
op := byte(0xA9)
if size == 1 {
op = 0xA8
}
i := newInstr(size, []byte{op})
i.imm = immediate(int64(imm), size, false)
return e.emit(i)
}
op := byte(0xF7) op := byte(0xF7)
if size == 1 { if size == 1 {
op = 0xF6 op = 0xF6
@@ -580,7 +678,7 @@ func (e *enc) encodeCmov(upper string, ops []Operand) error {
} }
// encodeSet encodes a conditional byte set: SET + condition (SETNE, SETEQ, …), // encodeSet encodes a conditional byte set: SET + condition (SETNE, SETEQ, …),
// always a byte write — 0F 90+cc /0 into a register or memory operand. // always a byte write, 0F 90+cc /0 into a register or memory operand.
func (e *enc) encodeSet(upper string, ops []Operand) error { func (e *enc) encodeSet(upper string, ops []Operand) error {
if len(ops) != 1 { if len(ops) != 1 {
return fmt.Errorf("SETcc expects 1 operand, got %d", len(ops)) return fmt.Errorf("SETcc expects 1 operand, got %d", len(ops))
@@ -597,31 +695,60 @@ func (e *enc) encodeSet(upper string, ops []Operand) error {
return e.emit(i) return e.emit(i)
} }
// --- LZCNT / TZCNT ---------------------------------------------------------- // --- bit scan / bit count ----------------------------------------------------
// encodeCount encodes LZCNT/TZCNT (leading / trailing zero count): F3 0F BD // countOp maps the bit-scan and bit-count mnemonics to their opcode byte and
// or F3 0F BC, with reg = dst and rm = src. The size suffix selects the // mandatory prefix. TZCNT/LZCNT/POPCNT are the F3-prefixed forms of the
// operand width (LZCNTW/LZCNTL/LZCNTQ). // same map as BSF/BSR's 0F BC/BD; POPCNT is F3 0F B8.
var countOp = map[string]struct {
op byte
prefix byte
}{
"BSF": {0xBC, 0},
"BSR": {0xBD, 0},
"TZCNT": {0xBC, 0xF3},
"LZCNT": {0xBD, 0xF3},
"POPCNT": {0xB8, 0xF3},
}
// encodeCount encodes the bit-scan and bit-count family, BSF (0F BC),
// BSR (0F BD), TZCNT (F3 0F BC), LZCNT (F3 0F BD) and POPCNT (F3 0F B8)
// with reg = dst and rm = src. The size suffix selects the operand width
// (BSFQ, TZCNTL, …). Note BSF/BSR leave the destination undefined when the
// source is zero (unlike their F3-prefixed counterparts); callers must
// guard non-zero inputs themselves.
func (e *enc) encodeCount(base string, ops []Operand, size int) error { func (e *enc) encodeCount(base string, ops []Operand, size int) error {
if len(ops) != 2 { if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops)) return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
} }
op := byte(0xBD) spec := countOp[base]
if base == "TZCNT" {
op = 0xBC
}
dstReg, ok := ops[1].(Reg) dstReg, ok := ops[1].(Reg)
if !ok { if !ok {
return fmt.Errorf("%s destination must be a register", base) return fmt.Errorf("%s destination must be a register", base)
} }
i := newInstr(size, []byte{0x0F, op}) i := newInstr(size, []byte{0x0F, spec.op})
i.prefix = 0xF3 i.prefix = spec.prefix
if err := setRM(i, dstReg, ops[0], size); err != nil { if err := setRM(i, dstReg, ops[0], size); err != nil {
return err return err
} }
return e.emit(i) return e.emit(i)
} }
// encodeBswap encodes BSWAP: the single register operand is encoded in the
// opcode byte (0F C8+r), with REX.B for R8-R15 and REX.W for the quad form.
func (e *enc) encodeBswap(ops []Operand, size int) error {
if len(ops) != 1 {
return fmt.Errorf("BSWAP expects 1 operand, got %d", len(ops))
}
reg, ok := ops[0].(Reg)
if !ok {
return fmt.Errorf("BSWAP operand must be a register")
}
i := newInstr(size, []byte{0x0F, 0xC8 + byte(reg.idx&7)})
i.rexB = reg.idx >= 8
return e.emit(i)
}
// --- mixed-width sign/zero-extending moves ----------------------------------- // --- mixed-width sign/zero-extending moves -----------------------------------
// movExtendOp maps Go's mixed-width move names to their opcode and destination // movExtendOp maps Go's mixed-width move names to their opcode and destination
@@ -674,8 +801,8 @@ type sseMove struct {
} }
var sseMoveTable = map[string]sseMove{ var sseMoveTable = map[string]sseMove{
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU — unaligned octa "MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA — aligned octa "MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single "MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single "MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double "MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
@@ -721,6 +848,112 @@ func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
return e.emit(i) return e.emit(i)
} }
// --- legacy SSE packed binary and shuffles -----------------------------------
// sseBin describes a legacy (non-VEX) SSE packed/scalar binary op: an
// optional mandatory prefix plus the 0F-prefixed opcode (0F38 for the
// SSSE3 integer shuffles). Plan 9 asm lists the source operand first, so
// MULPS X0, X1 computes X1 = X1 * X0.
type sseBin struct {
prefix byte // 0, 0x66, 0xF2 or 0xF3
op byte
map38 bool // opcode lives under 0F38 instead of 0F
}
var sseBinTable = map[string]sseBin{
"ADDPS": {0, 0x58, false}, "ADDPD": {0x66, 0x58, false},
"MULPS": {0, 0x59, false}, "MULPD": {0x66, 0x59, false},
"SUBPS": {0, 0x5C, false}, "SUBPD": {0x66, 0x5C, false},
"DIVPS": {0, 0x5E, false}, "DIVPD": {0x66, 0x5E, false},
"ANDPS": {0, 0x54, false}, "ANDPD": {0x66, 0x54, false},
"ORPS": {0, 0x56, false}, "ORPD": {0x66, 0x56, false},
"XORPS": {0, 0x57, false}, "XORPD": {0x66, 0x57, false},
"MINPS": {0, 0x5D, false}, "MINPD": {0x66, 0x5D, false},
"MAXPS": {0, 0x5F, false}, "MAXPD": {0x66, 0x5F, false},
"ADDSS": {0xF3, 0x58, false}, "ADDSD": {0xF2, 0x58, false},
"MULSS": {0xF3, 0x59, false}, "MULSD": {0xF2, 0x59, false},
"SUBSS": {0xF3, 0x5C, false}, "SUBSD": {0xF2, 0x5C, false},
"DIVSS": {0xF3, 0x5E, false}, "DIVSD": {0xF2, 0x5E, false},
"MINSS": {0xF3, 0x5D, false}, "MINSD": {0xF2, 0x5D, false},
"MAXSS": {0xF3, 0x5F, false}, "MAXSD": {0xF2, 0x5F, false},
"UNPCKLPS": {0, 0x14, false}, "UNPCKHPS": {0, 0x15, false},
"UNPCKLPD": {0x66, 0x14, false}, "UNPCKHPD": {0x66, 0x15, false},
"CVTSS2SD": {0xF3, 0x5A, false}, "CVTSD2SS": {0xF2, 0x5A, false},
"CVTPS2PD": {0, 0x5A, false}, "CVTPD2PS": {0x66, 0x5A, false},
// SSE2 packed integers (reg = reg op rm) and the SSSE3 byte shuffle.
"PXOR": {0x66, 0xEF, false},
"POR": {0x66, 0xEB, false},
"PAND": {0x66, 0xDB, false},
"PANDN": {0x66, 0xDF, false},
"PADDB": {0x66, 0xFC, false}, "PADDW": {0x66, 0xFD, false},
"PADDD": {0x66, 0xFE, false}, "PADDQ": {0x66, 0xD4, false},
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
"PCMPEQD": {0x66, 0x76, false},
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
"PCMPGTD": {0x66, 0x66, false},
"PSHUFB": {0x66, 0x00, true},
}
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
// (PSHUFD/PSHUFHW/PSHUFLW also carry the packed-int 0x66/F3/F2 prefixes).
type sseShuf struct {
prefix byte
op byte
}
var sseShufTable = map[string]sseShuf{
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
}
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("SSE binary expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("SSE binary destination must be a vector register")
}
opcode := []byte{0x0F, m.op}
if m.map38 {
opcode = []byte{0x0F, 0x38, m.op}
}
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
return e.emit(i)
}
// encodeSSEShuf encodes an imm8 shuffle: SHUFPS $imm, src, dst.
func (e *enc) encodeSSEShuf(m sseShuf, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("SSE shuffle expects 3 operands, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("SSE shuffle needs an imm8 first operand")
}
if imm < -128 || imm > 255 {
return fmt.Errorf("SSE shuffle imm8 %d out of range", imm)
}
src, dst := ops[1], ops[2]
dstReg, ok2 := dst.(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("SSE shuffle destination must be a vector register")
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
i.imm = []byte{byte(int8(imm))}
return e.emit(i)
}
// --- CVTSL2SD / CVTSQ2SD ----------------------------------------------------- // --- CVTSL2SD / CVTSQ2SD -----------------------------------------------------
// encodeCvtsi2sd encodes a signed integer to scalar double conversion // encodeCvtsi2sd encodes a signed integer to scalar double conversion
+2 -2
View File
@@ -214,7 +214,7 @@ func main() {
t.Fatalf("baseline build: %v\n%s", err, buildLog) t.Fatalf("baseline build: %v\n%s", err, buildLog)
} }
var pkgArch, work, linkLine, asmObj string var pkgArch, work, linkLine, asmObj string
for _, line := range strings.Split(string(buildLog), "\n") { for line := range strings.SplitSeq(string(buildLog), "\n") {
switch { switch {
case strings.HasPrefix(line, "WORK="): case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=") work = strings.TrimPrefix(line, "WORK=")
@@ -278,7 +278,7 @@ func main() {
newArch := filepath.Join(dir, "pkg.a") newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch} args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{} seen := map[string]bool{}
for _, m := range strings.Fields(string(listOut)) { for m := range strings.FieldsSeq(string(listOut)) {
if seen[m] { if seen[m] {
continue continue
} }
+70 -13
View File
@@ -6,6 +6,7 @@ package asm
import ( import (
"fmt" "fmt"
"sort" "sort"
"strconv"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-devkit/ast"
) )
@@ -15,7 +16,7 @@ import (
// file-local static symbols are encoded RIP-relative and resolved within the // file-local static symbols are encoded RIP-relative and resolved within the
// image, so the raw bytes are self-consistent and executable at any base // image, so the raw bytes are self-consistent and executable at any base
// address; references to external symbols are recorded as relocations // address; references to external symbols are recorded as relocations
// (Funcs[i].Relocs, Externals) and left unresolved — the object-file // (Funcs[i].Relocs, Externals) and left unresolved, the object-file
// emitters turn them into linker relocations. // emitters turn them into linker relocations.
type Image struct { type Image struct {
Code []byte // concatenated function bodies Code []byte // concatenated function bodies
@@ -80,7 +81,7 @@ func (fl *FuncLayout) LineAt(offset int) int {
return 0 return 0
} }
// Reloc is one static-symbol reference within a function body: the disp32 // RelocKind Reloc is one static-symbol reference within a function body: the disp32
// field at Off (function-relative) must reach the symbol plus Addend, // field at Off (function-relative) must reach the symbol plus Addend,
// measured from After, the address just past the instruction. An External // measured from After, the address just past the instruction. An External
// relocation names a symbol no GLOBL in the file defines; the object-file // relocation names a symbol no GLOBL in the file defines; the object-file
@@ -90,13 +91,18 @@ type RelocKind int
const ( const (
RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64) RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64)
RelCall // R_CALL: CALL to a function symbol (amd64)
RelTLSLE // R_TLS_LE: local-exec TLS load, no symbol (amd64 guard)
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair) RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair) RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
RelRISCVJal // R_RISCV_JAL (J-type call) RelRISCVJal // R_RISCV_JAL (J-type call)
RelPCRelAbs // 32-bit absolute (R_RISCV_32) RelPCRelAbs // 32-bit absolute (R_RISCV_32)
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i) RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st) RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
RelArm64Addr // R_ADDRARM64 (ADRP + ADD/LDR/STR pair) RelArm64Addr // R_ADDRARM64 (ADRP + ADD pair)
RelArm64Branch // R_CALLARM64 (BL instruction)
RelArm64LDST64 // R_ARM64_PCREL_LDST64 (ADRP + 64-bit LDR/STR pair)
RelLoong64Branch // R_CALLLOONG64 (BL instruction)
) )
type Reloc struct { type Reloc struct {
@@ -131,7 +137,7 @@ func (img *Image) Bytes() []byte {
// reference to a file-local static symbol becomes a RIP-relative load whose // reference to a file-local static symbol becomes a RIP-relative load whose
// displacement is resolved against that layout; a reference to a symbol no // displacement is resolved against that layout; a reference to a symbol no
// GLOBL defines is recorded as an external relocation (Externals) with its // GLOBL defines is recorded as an external relocation (Externals) with its
// displacement left zero — the object-file emitters resolve it at link // displacement left zero, the object-file emitters resolve it at link
// time, while the raw image (Bytes) cannot represent it. // time, while the raw image (Bytes) cannot represent it.
func AssembleFile(f *ast.File) (*Image, error) { func AssembleFile(f *ast.File) (*Image, error) {
dataSyms, err := collectData(f) dataSyms, err := collectData(f)
@@ -145,6 +151,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
link := &linkInfo{symbols: known, allowExternal: true} link := &linkInfo{symbols: known, allowExternal: true}
img := &Image{Symbols: map[string]int{}} img := &Image{Symbols: map[string]int{}}
textOff := map[string]int{}
type asmFunc struct { type asmFunc struct {
name string name string
patches []sbPatch patches []sbPatch
@@ -182,6 +189,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
for _, s := range steps { for _, s := range steps {
fl.Spadj = append(fl.Spadj, SpadjStep{PC: s.pc, Value: s.value}) fl.Spadj = append(fl.Spadj, SpadjStep{PC: s.pc, Value: s.value})
} }
textOff[t.Name.Name] = len(img.Code)
img.Funcs = append(img.Funcs, fl) img.Funcs = append(img.Funcs, fl)
img.Code = append(img.Code, code...) img.Code = append(img.Code, code...)
funcs = append(funcs, asmFunc{name: t.Name.Name, patches: patches}) funcs = append(funcs, asmFunc{name: t.Name.Name, patches: patches})
@@ -214,13 +222,27 @@ func AssembleFile(f *ast.File) (*Image, error) {
base := img.Funcs[i].Offset base := img.Funcs[i].Offset
code := img.Code[base : base+img.Funcs[i].Size] code := img.Code[base : base+img.Funcs[i].Size]
for _, p := range fn.patches { for _, p := range fn.patches {
reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend} reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend, Kind: p.kind}
if p.kind == RelTLSLE {
// The TLS slot has no symbol: the linker fills the offset
// from the runtime's TLS layout.
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
continue
}
if imgOff, ok := img.Symbols[p.name]; ok { if imgOff, ok := img.Symbols[p.name]; ok {
rel := int64(imgOff) + p.addend - int64(base+p.after) rel := int64(imgOff) + p.addend - int64(base+p.after)
if rel < -1<<31 || rel >= 1<<31 { if rel < -1<<31 || rel >= 1<<31 {
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name) return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
} }
copy(code[p.off:p.off+4], le32(rel)) copy(code[p.off:p.off+4], le32(rel))
} else if imgOff, ok := textOff[p.name]; ok {
// A CALL to a TEXT function of the same file: resolve the
// displacement against the function's layout position.
rel := int64(imgOff) + p.addend - int64(base+p.after)
if rel < -1<<31 || rel >= 1<<31 {
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
}
copy(code[p.off:p.off+4], le32(rel))
} else { } else {
reloc.External = true reloc.External = true
externals[p.name] = true externals[p.name] = true
@@ -302,6 +324,7 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
}) })
} }
markExternals(img, dataSyms)
return img, nil return img, nil
} }
@@ -373,9 +396,38 @@ func AssembleFileLOONG64(f *ast.File) (*Image, error) {
}) })
} }
markExternals(img, dataSyms)
return img, nil return img, nil
} }
// markExternals identifies relocations that reference symbols not defined in
// the file (neither a GLOBL/DATA symbol nor a TEXT function) and records them
// as external. The non-amd64 architectures emit relocations for every SB
// reference; this post-processing step distinguishes file-local from external.
func markExternals(img *Image, dataSyms []dataSym) {
known := make(map[string]bool, len(dataSyms)+len(img.Funcs))
for _, d := range dataSyms {
known[d.name] = true
}
for _, fn := range img.Funcs {
known[fn.Name] = true
}
externals := map[string]bool{}
for i := range img.Funcs {
for j := range img.Funcs[i].Relocs {
r := &img.Funcs[i].Relocs[j]
if !known[r.Name] {
r.External = true
externals[r.Name] = true
}
}
}
for name := range externals {
img.Externals = append(img.Externals, name)
}
sort.Strings(img.Externals)
}
// dataSym is one GLOBL symbol and its DATA initialiser. // dataSym is one GLOBL symbol and its DATA initialiser.
type dataSym struct { type dataSym struct {
name string name string
@@ -420,13 +472,18 @@ func collectData(f *ast.File) ([]dataSym, error) {
ds.rodata = true ds.rodata = true
case "DUPOK": case "DUPOK":
ds.dupok = true ds.dupok = true
case "1": default:
ds.dupok = true // Legacy numeric flag constants (runtime/textflag.h):
case "8": // DUPOK is 2, RODATA is 8; combinations arrive as one
ds.rodata = true // number (e.g. 10 = RODATA|DUPOK).
case "9": if n, err := strconv.Atoi(f); err == nil {
ds.dupok = true if n&2 != 0 {
ds.rodata = true ds.dupok = true
}
if n&8 != 0 {
ds.rodata = true
}
}
} }
} }
syms = append(syms, ds) syms = append(syms, ds)
@@ -457,7 +514,7 @@ func collectData(f *ast.File) ([]dataSym, error) {
if dd.Value.Imm.Neg { if dd.Value.Imm.Neg {
v = -v v = -v
} }
for j := 0; j < w; j++ { for j := range w {
buf[off+int64(j)] = byte(v >> (8 * j)) buf[off+int64(j)] = byte(v >> (8 * j))
} }
} }
+38
View File
@@ -128,3 +128,41 @@ DATA x<>+0(SB)/4, $1
t.Errorf("single-function SB: error %v, want a file-level-assembly error", err) t.Errorf("single-function SB: error %v, want a file-level-assembly error", err)
} }
} }
// TestCollectDataNumericFlags pins the numeric GLOBL flag constants from
// runtime/textflag.h: DUPOK is 2, RODATA is 8, and combinations arrive as
// one number (9 = NOPROF|RODATA, 10 = RODATA|DUPOK).
func TestCollectDataNumericFlags(t *testing.T) {
tests := []struct {
flags string
rodata bool
dupok bool
}{
{"2", false, true},
{"8", true, false},
{"9", true, false}, // NOPROF|RODATA, not DUPOK
{"10", true, true}, // RODATA|DUPOK
{"RODATA", true, false},
{"DUPOK", false, true},
{"RODATA|DUPOK", true, true},
}
for _, tt := range tests {
src := "TEXT \u00b7f(SB), NOSPLIT, $0\n\tRET\nGLOBL sym(SB), " + tt.flags + ", $8\n"
f, errs := parser.Parse("f_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse %q: %v", tt.flags, errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble %q: %v", tt.flags, err)
}
if len(img.DataSyms) != 1 {
t.Fatalf("%q: data syms = %d, want 1", tt.flags, len(img.DataSyms))
}
d := img.DataSyms[0]
if d.Rodata != tt.rodata || d.Dupok != tt.dupok {
t.Errorf("flags %q: rodata=%v dupok=%v, want rodata=%v dupok=%v",
tt.flags, d.Rodata, d.Dupok, tt.rodata, tt.dupok)
}
}
}
+123 -92
View File
@@ -13,17 +13,19 @@ import (
// assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into // assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into
// machine code. Every instruction is 4 bytes; the MOV pseudo-instruction and // machine code. Every instruction is 4 bytes; the MOV pseudo-instruction and
// the immediate-arithmetic forms expand to 2–5 instructions when the // the immediate-arithmetic forms expand to 2-5 instructions when the
// immediate does not fit, so the layout is computed in two passes (sizes, // immediate does not fit, so the layout is computed in two passes (sizes,
// then encoding with resolved branch targets). // then encoding with resolved branch targets).
// //
// The emitted bytes match the Go toolchain's loong64 assembler, which is the // The emitted bytes match the Go toolchain's loong64 assembler, which is the
// ground-truth oracle: prologue/epilogue, FP/SP frame mapping, branch // ground-truth oracle: prologue/epilogue (including the large-frame R30
// encodings and the MOV immediate expansions all follow cmd/internal/obj/ // materialisations), FP/SP frame mapping, the stack-split guard classes, and
// loong64's asmout cases. // branch encodings all follow cmd/internal/obj/loong64. The morestack block
// at the end of split functions carries the runtime.morestack_noctxt call.
func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) { func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
fi := loong64ComputeFrame(t) fi := loong64ComputeFrame(t)
prologue := loong64Prologue(fi) prologue := loong64Prologue(fi)
guardLen := loong64GuardLen(fi)
chain := loong64JumpChain(t) chain := loong64JumpChain(t)
resolve := func(name string) string { resolve := func(name string) string {
if r, ok := chain[name]; ok { if r, ok := chain[name]; ok {
@@ -35,16 +37,17 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
var relocs []Reloc var relocs []Reloc
var spadj []SpadjStep var spadj []SpadjStep
// The prologue (3 instructions when a frame is present) raises the SP // The prologue raises the SP delta by autosize; the boundary is reported
// delta by autosize; the boundary is reported at the third instruction's // after the SP adjust instruction, exactly as the toolchain's pctospadj
// pc, exactly as the toolchain's pctospadj does. // does. The prologue (3 instructions when a frame is present) may
// materialise its store or adjust through R30, which widens it.
if fi.autosize != 0 { if fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: 8, Value: fi.autosize}) spadj = append(spadj, SpadjStep{PC: guardLen + (loong64StoreWords(fi.autosize)+loong64AdjustWords(-int64(fi.autosize)))*4, Value: fi.autosize})
} }
// Pass 1: label offsets from the instruction sizes. // Pass 1: label offsets from the instruction sizes.
offsets := map[string]int{} offsets := map[string]int{}
pos := len(prologue) pos := guardLen + len(prologue)
for _, stmt := range t.Body { for _, stmt := range t.Body {
switch s := stmt.(type) { switch s := stmt.(type) {
case *ast.Label: case *ast.Label:
@@ -54,9 +57,25 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
} }
} }
// Pass 2: encode. Relocation offsets are recorded function-relative. // Pass 2: encode. The guard prefix precedes the prologue; its branches
out := append([]byte(nil), prologue...) // target the morestack block at the end of the function, which the first
pc := len(prologue) // pass has sized.
bodyLen := 0
{
p := guardLen + len(prologue)
for _, stmt := range t.Body {
if in, ok := stmt.(*ast.Instr); ok {
p += loong64InstrSize(in, fi)
}
}
bodyLen = p - (guardLen + len(prologue))
}
var out []byte
if fi.needSplit {
out = append(out, loong64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...)
}
out = append(out, prologue...)
pc := guardLen + len(prologue)
preCount := len(relocs) preCount := len(relocs)
var lines []LineEntry var lines []LineEntry
for _, stmt := range t.Body { for _, stmt := range t.Body {
@@ -69,23 +88,29 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err) return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
} }
for j := preCount; j < len(relocs); j++ { for j := preCount; j < len(relocs); j++ {
relocs[j].Off += pc - len(prologue) // Make the relocation offsets function-relative: each instruction
// records its reloc offset relative to its own start, and pc is
// that instruction's offset from the function start (prologue
// included). After shifts by the same amount.
relocs[j].Off += pc
relocs[j].After += pc
} }
preCount = len(relocs) preCount = len(relocs)
lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line}) lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line})
// The RET's epilogue closes the frame: the SP delta returns to zero // The RET's epilogue closes the frame: the SP delta returns to zero
// after the addi.d (one instruction for a leaf, two for a non-leaf // after the frame-deallocating ADDV.
// with the LR restore).
if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 { if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 {
epi := 4 spadj = append(spadj, SpadjStep{PC: pc + loong64EpilogueWords(fi)*4, Value: 0})
if !fi.leaf {
epi = 8
}
spadj = append(spadj, SpadjStep{PC: pc + epi, Value: 0})
} }
out = append(out, code...) out = append(out, code...)
pc += len(code) pc += len(code)
} }
if fi.needSplit {
block, blReloc := loong64MoreStackBlock(pc)
out = append(out, block...)
relocs = append(relocs, blReloc)
pc += len(block)
}
return out, offsets, relocs, lines, spadj, nil return out, offsets, relocs, lines, spadj, nil
} }
@@ -224,9 +249,9 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
} }
return l64wordLE(uint32(immFromOperand(ops[0]))), nil return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "JMP", "B": case "JMP", "B":
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve) return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs)
case "JAL", "CALL", "BL": case "JAL", "CALL", "BL":
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve) return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs)
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD": case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return encodeLOONG64Mov(instr, mnem, fi, relocs) return encodeLOONG64Mov(instr, mnem, fi, relocs)
} }
@@ -417,7 +442,7 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
case l64Firrr: case l64Firrr:
// ALSL: INSTR $sa, rj, rk, rd (the toolchain's optab places rj in // ALSL: INSTR $sa, rj, rk, rd (the toolchain's optab places rj in
// the second register position); the source amount is 1–4, encoded // the second register position); the source amount is 1-4, encoded
// as sa-1. // as sa-1.
if len(ops) != 4 { if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
@@ -485,7 +510,7 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
// //
// JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0 // JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0
// JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0 // JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string) ([]byte, error) { func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc) ([]byte, error) {
if len(instr.Operands) != 1 { if len(instr.Operands) != 1 {
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands)) return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands))
} }
@@ -502,6 +527,19 @@ func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[stri
} }
return l64wordLE(l64irr16(l64branchTable["JIRL"], 0, rj, rd)), nil return l64wordLE(l64irr16(l64branchTable["JIRL"], 0, rj, rd)), nil
} }
// Direct symbol: sym+off(SB) → b/bl with an R_CALLLOONG64 relocation
// (the linker fills the offset), as the toolchain does for CALL/BL/JAL
// and for tail-calling JMP.
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
opc := l64jumpTable["B"]
if link {
opc = l64jumpTable["BL"]
}
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelLoong64Branch, Addend: op.Addr.Sym.Offset})
}
return l64wordLE(l64bbl(opc, 0)), nil
}
// Direct: label → b/bl. // Direct: label → b/bl.
target := resolve(l64Label(op)) target := resolve(l64Label(op))
targetOff, ok := offsets[target] targetOff, ok := offsets[target]
@@ -578,8 +616,8 @@ func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, o
// encodeLOONG64Branch21 encodes a single-register branch: BLTZ/BGEZ and // encodeLOONG64Branch21 encodes a single-register branch: BLTZ/BGEZ and
// BFPT/BFPF use the 21-bit offset form (register in the rj field), while // BFPT/BFPF use the 21-bit offset form (register in the rj field), while
// BGTZ/BLEZ — which the toolchain encodes with the register in the rd field // BGTZ/BLEZ, which the toolchain encodes with the register in the rd field
// and a 16-bit offset — are handled separately. // and a 16-bit offset, are handled separately.
func encodeLOONG64Branch21(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { func encodeLOONG64Branch21(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
if len(ops) != 2 { if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
@@ -672,8 +710,6 @@ func encodeLOONG64ImmArith(mnem string, de l64DualEnc, ops []*ast.Operand) ([]by
const ( const (
lu12iw = 0x0a << 25 lu12iw = 0x0a << 25
ori = 0x00e << 22 ori = 0x00e << 22
lu32id = 0x0b << 25
lu52id = 0x00c << 22
) )
if v == int64(int32(v)) { if v == int64(int32(v)) {
if v&0xfff == 0 && (v < 0x800 || v > 0xfff) { if v&0xfff == 0 && (v < 0x800 || v > 0xfff) {
@@ -694,7 +730,7 @@ func encodeLOONG64ImmArith(mnem string, de l64DualEnc, ops []*ast.Operand) ([]by
} }
// isLoong64ShiftD reports whether a shift-immediate opcode constant is one of // isLoong64ShiftD reports whether a shift-immediate opcode constant is one of
// the 6-bit (.d) variants — the toolchain distinguishes them by the bit // the 6-bit (.d) variants, the toolchain distinguishes them by the bit
// position of the opcode field (bits [25:16]). // position of the opcode field (bits [25:16]).
func isLoong64ShiftD(op uint32) bool { func isLoong64ShiftD(op uint32) bool {
return op&0x03ff0000 != 0 && op>>25 == 0 return op&0x03ff0000 != 0 && op>>25 == 0
@@ -742,7 +778,7 @@ func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off in
// ---- the MOV pseudo-instruction ---- // ---- the MOV pseudo-instruction ----
// encodeLOONG64Mov encodes the MOV family — the load/store/immediate // encodeLOONG64Mov encodes the MOV family, the load/store/immediate
// workhorse of Go's loong64 assembly. MOV is an alias of MOVV (the width // workhorse of Go's loong64 assembly. MOV is an alias of MOVV (the width
// mnemonics MOVB/MOVH/MOVW/MOVV/MOVBU/MOVHU/MOVWU/MOVF/MOVD select the // mnemonics MOVB/MOVH/MOVW/MOVV/MOVBU/MOVHU/MOVWU/MOVF/MOVD select the
// access width). The forms, mirroring the toolchain: // access width). The forms, mirroring the toolchain:
@@ -771,7 +807,7 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs
if rd < 0 { if rd < 0 {
return nil, fmt.Errorf("%s $sym(SB): invalid destination register", mnem) return nil, fmt.Errorf("%s $sym(SB): invalid destination register", mnem)
} }
return encodeLOONG64SBAddr(src.Imm.Sym, rd, mnem, relocs), nil return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil
} }
rd := l64Reg(dst) rd := l64Reg(dst)
if rd < 0 { if rd < 0 {
@@ -832,14 +868,14 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs
if rd < 0 { if rd < 0 {
return nil, fmt.Errorf("%s: invalid destination register", mnem) return nil, fmt.Errorf("%s: invalid destination register", mnem)
} }
return encodeLOONG64MemOp(mnem, ops[0], rd, true, fi, relocs) return encodeLOONG64MemOp(mnem, ops[0], rd, true, fi)
} }
if !isMemOperand(src) && isMemOperand(dst) { if !isMemOperand(src) && isMemOperand(dst) {
rs := l64Reg(src) rs := l64Reg(src)
if rs < 0 { if rs < 0 {
return nil, fmt.Errorf("%s: invalid source register", mnem) return nil, fmt.Errorf("%s: invalid source register", mnem)
} }
return encodeLOONG64MemOp(mnem, ops[1], rs, false, fi, relocs) return encodeLOONG64MemOp(mnem, ops[1], rs, false, fi)
} }
// Register → register. // Register → register.
@@ -933,20 +969,20 @@ const (
l64St1 l64St1
l64St0 l64St0
l64Dcon12_0 l64dcon120
l64Dcon12_20S l64dcon1220s
l64Dcon20S_20 l64dcon20s20
l64Dcon12_12S l64dcon1212s
l64Dcon20S_12S l64dcon20s12s
l64Dcon20S_0 l64dcon20s0
l64Dcon12_12U l64dcon1212u
l64Dcon20S_12U l64dcon20s12u
l64Dcon32_12S l64dcon3212s
l64Dcon32_0 l64dcon320
l64Dcon32_20 l64dcon3220
l64Dcon12_32S l64dcon1232s
l64Dcon20S_32 l64dcon20s32
l64Dcon32_12U l64dcon3212u
l64Dcon l64Dcon
) )
@@ -987,91 +1023,91 @@ func l64DconClass(v int64) int {
lo20 := l64BitField(v, 12, 20) lo20 := l64BitField(v, 12, 20)
lo12 := l64BitField(v, 0, 12) lo12 := l64BitField(v, 0, 12)
if tzb >= 52 { if tzb >= 52 {
return l64Dcon12_0 return l64dcon120
} }
if tzb >= 32 { if tzb >= 32 {
if ((hi20 == l64All1 || hi20 == l64St1) && hi12 == l64All1) || ((hi20 == l64All0 || hi20 == l64St0) && hi12 == l64All0) { if ((hi20 == l64All1 || hi20 == l64St1) && hi12 == l64All1) || ((hi20 == l64All0 || hi20 == l64St0) && hi12 == l64All0) {
return l64Dcon20S_0 return l64dcon20s0
} }
return l64Dcon32_0 return l64dcon320
} }
if tzb >= 12 { if tzb >= 12 {
if lo20 == l64St1 || lo20 == l64All1 { if lo20 == l64St1 || lo20 == l64All1 {
if hi20 == l64All1 { if hi20 == l64All1 {
return l64Dcon12_20S return l64dcon1220s
} }
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) { if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64Dcon20S_20 return l64dcon20s20
} }
return l64Dcon32_20 return l64dcon3220
} }
if hi20 == l64All0 { if hi20 == l64All0 {
return l64Dcon12_20S return l64dcon1220s
} }
if (hi20 == l64St0 && hi12 == l64All0) || ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) { if (hi20 == l64St0 && hi12 == l64All0) || ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) {
return l64Dcon20S_20 return l64dcon20s20
} }
return l64Dcon32_20 return l64dcon3220
} }
if lo12 == l64St1 || lo12 == l64All1 { if lo12 == l64St1 || lo12 == l64All1 {
if lo20 == l64All1 { if lo20 == l64All1 {
if hi20 == l64All1 { if hi20 == l64All1 {
return l64Dcon12_12S return l64dcon1212s
} }
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) { if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64Dcon20S_12S return l64dcon20s12s
} }
return l64Dcon32_12S return l64dcon3212s
} }
if lo20 == l64St1 { if lo20 == l64St1 {
if hi20 == l64All1 { if hi20 == l64All1 {
return l64Dcon12_32S return l64dcon1232s
} }
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) { if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64Dcon20S_32 return l64dcon20s32
} }
return l64Dcon return l64Dcon
} }
if lo20 == l64All0 { if lo20 == l64All0 {
if hi20 == l64All0 { if hi20 == l64All0 {
return l64Dcon12_12U return l64dcon1212u
} }
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) { if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64Dcon20S_12U return l64dcon20s12u
} }
return l64Dcon32_12U return l64dcon3212u
} }
if hi20 == l64All0 { if hi20 == l64All0 {
return l64Dcon12_32S return l64dcon1232s
} }
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) { if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64Dcon20S_32 return l64dcon20s32
} }
return l64Dcon return l64Dcon
} }
if lo20 == l64All0 { if lo20 == l64All0 {
if hi20 == l64All0 { if hi20 == l64All0 {
return l64Dcon12_12U return l64dcon1212u
} }
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) { if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64Dcon20S_12U return l64dcon20s12u
} }
return l64Dcon32_12U return l64dcon3212u
} }
if lo20 == l64St1 || lo20 == l64All1 { if lo20 == l64St1 || lo20 == l64All1 {
if hi20 == l64All1 { if hi20 == l64All1 {
return l64Dcon12_32S return l64dcon1232s
} }
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) { if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64Dcon20S_32 return l64dcon20s32
} }
return l64Dcon return l64Dcon
} }
if hi20 == l64All0 { if hi20 == l64All0 {
return l64Dcon12_32S return l64dcon1232s
} }
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) { if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64Dcon20S_32 return l64dcon20s32
} }
return l64Dcon return l64Dcon
} }
@@ -1088,29 +1124,29 @@ func l64DconMovWords(rd int, v int64) []uint32 {
ori = 0x00e << 22 ori = 0x00e << 22
) )
switch l64DconClass(v) { switch l64DconClass(v) {
case l64Dcon12_0: case l64dcon120:
return []uint32{l64irr(lu52id, int(v>>52), 0, rd)} return []uint32{l64irr(lu52id, int(v>>52), 0, rd)}
case l64Dcon12_20S: case l64dcon1220s:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(lu52id, int(v>>52), rd, rd)} return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64Dcon20S_20: case l64dcon20s20:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd)} return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd)}
case l64Dcon12_12S: case l64dcon1212s:
return []uint32{l64irr(addid, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)} return []uint32{l64irr(addid, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64Dcon20S_12S, l64Dcon20S_0: case l64dcon20s12s, l64dcon20s0:
return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)} return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)}
case l64Dcon12_12U: case l64dcon1212u:
return []uint32{l64irr(ori, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)} return []uint32{l64irr(ori, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64Dcon20S_12U: case l64dcon20s12u:
return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)} return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)}
case l64Dcon32_12S, l64Dcon32_0: case l64dcon3212s, l64dcon320:
return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)} return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64Dcon32_20: case l64dcon3220:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)} return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64Dcon12_32S: case l64dcon1232s:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64irr(lu52id, int(v>>52), rd, rd)} return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64Dcon20S_32: case l64dcon20s32:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd)} return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd)}
case l64Dcon32_12U: case l64dcon3212u:
return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)} return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
default: default:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)} return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
@@ -1161,7 +1197,7 @@ func encodeLOONG64LoadImm(rd int, v int64, mnem string) []byte {
// encodeLOONG64MemOp encodes a memory load (load = true) or store with a // encodeLOONG64MemOp encodes a memory load (load = true) or store with a
// 12-bit offset, or the 3-instruction expansion for larger offsets: // 12-bit offset, or the 3-instruction expansion for larger offsets:
// lu12i.w r30, (off+0x800)>>12; add.d r30, rj, r30; ld/st rd, off(r30). // lu12i.w r30, (off+0x800)>>12; add.d r30, rj, r30; ld/st rd, off(r30).
func encodeLOONG64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi loong64FrameInfo, relocs *[]Reloc) ([]byte, error) { func encodeLOONG64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi loong64FrameInfo) ([]byte, error) {
rj, off := l64MemWithFrame(mem, fi) rj, off := l64MemWithFrame(mem, fi)
if rj < 0 { if rj < 0 {
return nil, fmt.Errorf("invalid memory operand") return nil, fmt.Errorf("invalid memory operand")
@@ -1266,7 +1302,7 @@ func l64FpMoveKey(mnem string, sc, dc l64RegClass) (string, bool) {
// encodeLOONG64SBAddr emits pcalau12i rd, 0; addi.d rd, rd, 0 with the // encodeLOONG64SBAddr emits pcalau12i rd, 0; addi.d rd, rd, 0 with the
// R_LOONG64_ADDR_HI/LO relocation pair, loading a symbol's address. // R_LOONG64_ADDR_HI/LO relocation pair, loading a symbol's address.
func encodeLOONG64SBAddr(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) []byte { func encodeLOONG64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
if relocs != nil { if relocs != nil {
*relocs = append(*relocs, *relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset}, Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset},
@@ -1342,11 +1378,6 @@ func l64Reg(op *ast.Operand) int {
return loong64RegNum(operandRegName(op)) return loong64RegNum(operandRegName(op))
} }
// l64Imm returns the immediate value of an operand.
func l64Imm(op *ast.Operand) int32 {
return immFromOperand(op)
}
// l64Imm64 returns the full 64-bit immediate value of an operand. // l64Imm64 returns the full 64-bit immediate value of an operand.
func l64Imm64(op *ast.Operand) int64 { func l64Imm64(op *ast.Operand) int64 {
if op.Imm.HasVal { if op.Imm.HasVal {
+21 -21
View File
@@ -9,7 +9,7 @@ package asm
// an opcode constant, and the format selects the bit layout. The opcode // an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own loong64 // constants and formats are transcribed from the Go toolchain's own loong64
// backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm` // backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm`
// exactly — the ground-truth oracle for the verify suite. // exactly, the ground-truth oracle for the verify suite.
// //
// All LoongArch instructions are 32 bits, little-endian. The formats used // All LoongArch instructions are 32 bits, little-endian. The formats used
// here (per the LoongArch Volume I specification): // here (per the LoongArch Volume I specification):
@@ -30,9 +30,11 @@ package asm
// of the immediate and register fields), mirroring the toolchain's OP_* // of the immediate and register fields), mirroring the toolchain's OP_*
// helpers, so each l64* function only ORs its fields in. // helpers, so each l64* function only ORs its fields in.
import "maps"
// loong64RegNum returns the 5-bit register number for a LoongArch register // loong64RegNum returns the 5-bit register number for a LoongArch register
// name: R0–R31 (integer), F0–F31 (floating point), FCC0–FCC7 (condition // name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
// flags), FCSR0–FCSR31 (control/status) and the ABI aliases the runtime's // flags), FCSR0-FCSR31 (control/status) and the ABI aliases the runtime's
// assembly uses. Returns -1 for an unrecognised name. // assembly uses. Returns -1 for an unrecognised name.
func loong64RegNum(name string) int { func loong64RegNum(name string) int {
switch name { switch name {
@@ -101,12 +103,12 @@ func loong64RegNum(name string) int {
case "R31", "S8": case "R31", "S8":
return 31 return 31
} }
// F0–F31, FCC0–FCC7, FCSR0–FCSR31. // F0-F31, FCC0-FCC7, FCSR0-FCSR31.
if len(name) >= 4 && name[:4] == "FCSR" { if len(name) >= 4 && name[:4] == "FCSR" {
return loong64RegSpecial(name[4:], "FCSR", 31) return loong64RegSpecial(name[4:], 31)
} }
if len(name) >= 3 && name[:3] == "FCC" { if len(name) >= 3 && name[:3] == "FCC" {
return loong64RegSpecial(name[3:], "FCC", 7) return loong64RegSpecial(name[3:], 7)
} }
if len(name) < 2 { if len(name) < 2 {
return -1 return -1
@@ -129,7 +131,7 @@ func loong64RegNum(name string) int {
} }
// loong64RegSpecial parses a numbered FCC/FCSR register. // loong64RegSpecial parses a numbered FCC/FCSR register.
func loong64RegSpecial(digits, prefix string, max int) int { func loong64RegSpecial(digits string, max int) int {
if digits == "" { if digits == "" {
return -1 return -1
} }
@@ -197,7 +199,7 @@ func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
} }
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd. // l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
// The msb/lsb fields are 6 bits wide (0–63) and are validated by the caller. // The msb/lsb fields are 6 bits wide (0-63) and are validated by the caller.
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 { func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f) return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
} }
@@ -278,7 +280,7 @@ var l64DualTable = map[string]l64DualEnc{}
var l64InstrTable = map[string]l64Enc{} var l64InstrTable = map[string]l64Enc{}
func init() { func init() {
// 3R — integer. // 3R, integer.
rrr := map[string]uint32{ rrr := map[string]uint32{
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15, "ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15, "SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
@@ -298,7 +300,7 @@ func init() {
"CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15, "CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15,
"CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15, "CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15,
} }
// 3R — floating point. // 3R, floating point.
rrr["MULF"] = 0x209 << 15 rrr["MULF"] = 0x209 << 15
rrr["MULD"] = 0x20a << 15 rrr["MULD"] = 0x20a << 15
rrr["DIVF"] = 0x20d << 15 rrr["DIVF"] = 0x20d << 15
@@ -368,7 +370,7 @@ func init() {
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12), // The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
// selected by the operand kind; the shift mnemonics pair the 3R form // selected by the operand kind; the shift mnemonics pair the 3R form
// with a 5/6-bit shift immediate. // with a 5/6-bit shift immediate.
for m, e := range map[string]l64DualEnc{ maps.Copy(l64DualTable, map[string]l64DualEnc{
"ADD": {rrr: 0x20 << 15, imm: 0x00a << 22}, "ADD": {rrr: 0x20 << 15, imm: 0x00a << 22},
"ADDW": {rrr: 0x20 << 15, imm: 0x00a << 22}, "ADDW": {rrr: 0x20 << 15, imm: 0x00a << 22},
"ADDV": {rrr: 0x21 << 15, imm: 0x00b << 22}, "ADDV": {rrr: 0x21 << 15, imm: 0x00b << 22},
@@ -386,16 +388,14 @@ func init() {
"SRLV": {rrr: 0x32 << 15, imm: 0x0045 << 16, shift: true}, "SRLV": {rrr: 0x32 << 15, imm: 0x0045 << 16, shift: true},
"SRAV": {rrr: 0x33 << 15, imm: 0x0049 << 16, shift: true}, "SRAV": {rrr: 0x33 << 15, imm: 0x0049 << 16, shift: true},
"ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true}, "ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true},
} { })
l64DualTable[m] = e
}
// 2RI12 — pure immediate arithmetic (LU52ID has no register form). // 2RI12, pure immediate arithmetic (LU52ID has no register form).
l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22} l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22}
// ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16. // ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16.
l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26} l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26}
// 2RI14 — LL/SC are aliased by the Go assembler to the pointer loads and // 2RI14, LL/SC are aliased by the Go assembler to the pointer loads and
// stores (ldptr/stptr), with the offset scaled by 4. // stores (ldptr/stptr), with the offset scaled by 4.
l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w
l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d
@@ -414,7 +414,7 @@ func init() {
// LUI is the Plan 9 spelling of lu12i.w. // LUI is the Plan 9 spelling of lu12i.w.
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25} l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
// 4R — fused multiply-add. // 4R, fused multiply-add.
rrrr := map[string]uint32{ rrrr := map[string]uint32{
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20, "FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20, "FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
@@ -425,7 +425,7 @@ func init() {
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op} l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
} }
// IRIR — bit-field insert/extract. // IRIR, bit-field insert/extract.
irir := map[string]uint32{ irir := map[string]uint32{
"BSTRINSW": 0x3<<21 | 0x0<<15, "BSTRINSW": 0x3<<21 | 0x0<<15,
"BSTRINSV": 0x2 << 22, "BSTRINSV": 0x2 << 22,
@@ -436,7 +436,7 @@ func init() {
l64InstrTable[m] = l64Enc{format: l64Firir, op: op} l64InstrTable[m] = l64Enc{format: l64Firir, op: op}
} }
// 3RI2 — ALSL. // 3RI2, ALSL.
irrr := map[string]uint32{ irrr := map[string]uint32{
"ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17, "ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17,
} }
@@ -452,7 +452,7 @@ func init() {
// PRELD. // PRELD.
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22} l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
// Atomics — 3R with the AM field order (rk=value, rj=address, rd=result). // Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
am := map[string]uint32{ am := map[string]uint32{
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15, "AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15, "AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
@@ -477,7 +477,7 @@ func init() {
} }
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the // l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
// register move between the integer and floating-point register banks — the // register move between the integer and floating-point register banks, the
// MOVW/MOVV specials the Go assembler accepts. // MOVW/MOVV specials the Go assembler accepts.
var l64FpMovTable = map[string]uint32{ var l64FpMovTable = map[string]uint32{
"MOVV.R.F": 0x452a << 10, // movgr2fr.d "MOVV.R.F": 0x452a << 10, // movgr2fr.d
+223 -12
View File
@@ -20,14 +20,20 @@ import (
// (the toolchain aligns frames with `if autosize&4 != 0 { autosize += 4 }`). // (the toolchain aligns frames with `if autosize&4 != 0 { autosize += 4 }`).
// A leaf function (no calls) with a zero frame gets no prologue at all. // A leaf function (no calls) with a zero frame gets no prologue at all.
// //
// Prologue (autosize > 0), byte-identical to the toolchain: // Prologue (autosize > 0, small), byte-identical to the toolchain:
// //
// MOVV R1, -autosize(R3) // save LR below the new SP (traceback-safe) // MOVV R1, -autosize(R3) // save LR below the new SP (traceback-safe)
// ADDV $-autosize, R3 // open the frame // ADDV $-autosize, R3 // open the frame
// MOVV R1, 0(R3) // save LR again at SP (signal-safety) // MOVV R1, 0(R3) // save LR again at SP (signal-safety)
// //
// Large frames (autosize past the 12-bit offset or immediate ranges) expand
// the store and the adjust through REGTMP (R30) exactly as the toolchain's
// assembler does: the store via the rounding LU12IW split, the adjust via
// the floor LU12IW/ORI split.
//
// Epilogue: MOVV 0(R3), R1; ADDV $autosize, R3 (non-leaf only for the LR // Epilogue: MOVV 0(R3), R1; ADDV $autosize, R3 (non-leaf only for the LR
// restore); the RET's jirl r0, r1, 0 follows. // restore; the adjust materialised when the immediate does not fit); the
// RET's jirl r0, r1, 0 follows.
// loong64FrameInfo holds the frame layout derived from a TEXT directive. // loong64FrameInfo holds the frame layout derived from a TEXT directive.
type loong64FrameInfo struct { type loong64FrameInfo struct {
@@ -36,6 +42,11 @@ type loong64FrameInfo struct {
args int // the declared -argsize args int // the declared -argsize
noSplit bool // the NOSPLIT flag noSplit bool // the NOSPLIT flag
leaf bool // no call instructions in the body leaf bool // no call instructions in the body
// Stack-split guard state: like amd64 and arm64, a leaf function with a
// small autosize is auto-marked NOSPLIT by the toolchain.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
} }
// loong64ComputeFrame derives the frame layout for a TEXT function. // loong64ComputeFrame derives the frame layout for a TEXT function.
@@ -59,9 +70,157 @@ func loong64ComputeFrame(t *ast.Text) loong64FrameInfo {
// A zero-frame non-leaf function still opens an 8-byte frame for LR. // A zero-frame non-leaf function still opens an 8-byte frame for LR.
fi.autosize = 8 fi.autosize = 8
} }
switch {
case fi.noSplit:
case fi.autosize < stackSmall && fi.leaf:
// Auto-NOSPLIT, as the toolchain's leaf mark concludes.
default:
fi.needSplit = true
switch {
case fi.autosize <= stackSmall:
fi.splitClass = 0
case fi.autosize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi return fi
} }
// loong64GuardLen returns the byte length of the stack-split guard prefix
// (zero when the function needs no guard). The big class materialises two
// constants through R30; each materialisation shrinks by one word when the
// constant's low 12 bits are zero.
func loong64GuardLen(fi loong64FrameInfo) int {
if !fi.needSplit {
return 0
}
off := int64(fi.autosize - stackSmall)
switch fi.splitClass {
case 0:
return 12
case 1:
if off <= 2048 {
return 16 // ADDV $-off fits the signed 12-bit immediate
}
return 24 // MOVV + LU12IW + ORI + ADDV + SGTU + BEQ
default:
// MOVV + [mat] + SGTU + BNE + [mat] + ADDV + SGTU + BEQ
return (6 + loong64MatLen(off) + loong64MatLen(-off)) * 4
}
}
// loong64MatLen reports the word count of materialising v in R30: a value
// with a zero high part needs only the ORI (the toolchain's MOVW $v, R30),
// one with a zero low part only the LU12IW.
func loong64MatLen(v int64) int {
if v>>12 == 0 || v&0xFFF == 0 {
return 1
}
return 2
}
// loong64MatWords appends the words that materialise v in R30, splitting it
// as v>>12 plus the zero-extended low 12 bits.
func loong64MatWords(ws []uint32, v int64) []uint32 {
hi := v >> 12
lo := v & 0xFFF
if hi == 0 {
return append(ws, l64irr(l64OriOp, int(v), 0, 30))
}
ws = append(ws, l64ir(l64Lu12iwOp, int(hi), 30))
if lo != 0 {
ws = append(ws, l64irr(l64OriOp, int(lo), 30, 30))
}
return ws
}
// The LU12IW and ORI opcode bases (2RI20 and 2RI12 formats); the ORI reads
// and writes rd itself.
const (
l64Lu12iwOp = 0x0a << 25
l64OriOp = 0x0e << 22
)
// loong64Imm12 reports whether v fits a signed 12-bit immediate.
func loong64Imm12(v int64) bool { return v >= -2048 && v <= 2047 }
// loong64GuardBytes emits the stack-split guard prefix. blockStart is the
// function-relative address of the morestack call at the end of the function;
// branch displacements are in instructions and are computed from each
// branch's own position.
func loong64GuardBytes(fi loong64FrameInfo, blockStart int) []byte {
// MOVV 16(g), R20 (g.stackguard0), g = R22.
ws := []uint32{l64irr(l64loadStoreTable["MOVV"].ld, 16, 22, 20)}
off := int64(fi.autosize - stackSmall)
// beq appends BEQ R20, blockStart from the branch's own position.
beq := func() {
ws = append(ws, loong64Beqz(20, int32((blockStart-len(ws)*4)>>2)))
}
switch fi.splitClass {
case 0:
// SGTU SP, R20, R20; BEQ R20, more
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 3, 20, 20))
beq()
case 1:
ws = append(ws, loong64MediumWords(off)...)
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 24, 20, 20))
beq()
default:
// SGTU $off, SP, R24 catches the SP underflow a huge frame would
// cause; BNE jumps to morestack in that case.
ws = append(ws, loong64MatWords(nil, off)...)
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 30, 3, 24))
ws = append(ws, loong64Bnez(24, int32((blockStart-len(ws)*4)>>2)))
ws = append(ws, loong64MatWords(nil, -off)...)
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 24))
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 24, 20, 20))
beq()
}
return l64WordsLE(ws...)
}
// loong64MediumWords emits the medium-class stack check for offset off: the
// ADDV immediate when it fits, otherwise the same sequence with the constant
// materialised in R30.
func loong64MediumWords(off int64) []uint32 {
if off <= 2048 {
return []uint32{l64irr(l64DualTable["ADDV"].imm, int(-off), 3, 24)}
}
ws := loong64MatWords(nil, -off)
return append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 24))
}
// loong64Beqz/loong64Bnez build the 21-bit conditional branches against R0
// that the toolchain emits for its guard compares.
func loong64Beqz(rj int, dispInstr int32) uint32 {
return l64ir21(l64branch21Table["BEQZ"], int(dispInstr), rj)
}
func loong64Bnez(rj int, dispInstr int32) uint32 {
return l64ir21(l64branch21Table["BNEZ"], int(dispInstr), rj)
}
// loong64MoreStackBlock emits the trailing block: MOVV R1, R31 (save LR, the
// toolchain's OR R1, R0, R31 expansion), BL runtime.morestack_noctxt, B back
// to the function entry.
func loong64MoreStackBlock(blockStart int) ([]byte, Reloc) {
ws := []uint32{
l64rrr(l64DualTable["OR"].rrr, 0, 1, 31), // MOVV R1, R31 (OR R1, R0, R31)
l64bbl(l64jumpTable["BL"], 0), // BL, patched by the linker
}
disp := (-(blockStart + 8)) >> 2
ws = append(ws, l64bbl(l64jumpTable["B"], int(disp)))
reloc := Reloc{
Off: blockStart + 4,
After: blockStart + 8,
Name: "runtime\u00b7morestack_noctxt",
Kind: RelLoong64Branch,
}
return l64WordsLE(ws...), reloc
}
// loong64IsLeaf reports whether a function contains no call instructions // loong64IsLeaf reports whether a function contains no call instructions
// (JAL/BL/CALL), matching the toolchain's LEAF mark, which drives the frame // (JAL/BL/CALL), matching the toolchain's LEAF mark, which drives the frame
// and the epilogue shape. // and the epilogue shape.
@@ -79,17 +238,37 @@ func loong64IsLeaf(t *ast.Text) bool {
return true return true
} }
// loong64Prologue returns the prologue bytes for a loong64 function. // loong64Prologue returns the prologue bytes for a loong64 function. When
// the LR store offset leaves the toolchain's 12-bit store range ([-2046,
// 2045], BIG_12 = 2046) or the SP adjust immediate its 12-bit immediate
// range, each switches to the R30 materialisation the assembler expands it
// to: the store uses the rounding %hi/%lo split (LU12IW of (v+2048)>>12,
// REGTMP += SP, store at the raw offset), the adjust the floor split
// (LU12IW, ORI when the low part is non-zero, REGTMP += SP).
func loong64Prologue(fi loong64FrameInfo) []byte { func loong64Prologue(fi loong64FrameInfo) []byte {
if fi.autosize == 0 { if fi.autosize == 0 {
return nil return nil
} }
addiD := l64DualTable["ADDV"].imm addiD := l64DualTable["ADDV"].imm
return l64WordsLE( var ws []uint32
l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, 3, 1), // MOVV R1, -autosize(R3) storeBase := 3
l64irr(addiD, -fi.autosize, 3, 3), // ADDV $-autosize, R3 if fi.autosize > 2046 {
l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1), // MOVV R1, 0(R3) // The store goes through REGTMP: LU12IW of the rounding split,
) // REGTMP += SP, then the store at REGTMP with the truncated offset.
v := -int64(fi.autosize)
ws = append(ws, l64ir(l64Lu12iwOp, int((v+2048)>>12), 30))
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 3, 30, 30))
storeBase = 30
}
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, storeBase, 1)) // MOVV R1, -autosize(base)
if loong64Imm12(-int64(fi.autosize)) {
ws = append(ws, l64irr(addiD, -fi.autosize, 3, 3)) // ADDV $-autosize, R3
} else {
ws = append(ws, loong64MatWords(nil, -int64(fi.autosize))...)
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 3))
}
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1)) // MOVV R1, 0(R3)
return l64WordsLE(ws...)
} }
// loong64Return returns the bytes for a RET: the epilogue (restore LR and // loong64Return returns the bytes for a RET: the epilogue (restore LR and
@@ -98,17 +277,49 @@ func loong64Return(fi loong64FrameInfo) []byte {
var ws []uint32 var ws []uint32
if fi.autosize != 0 { if fi.autosize != 0 {
if !fi.leaf { if !fi.leaf {
// MOVV 0(R3), R1 — restore the link register. // MOVV 0(R3), R1, restore the link register.
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1)) ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1))
} }
// ADDV $autosize, R3 — close the frame. // ADDV $autosize, R3, close the frame (materialised when the
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3)) // immediate does not fit).
if loong64Imm12(int64(fi.autosize)) {
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
} else {
ws = append(ws, loong64MatWords(nil, int64(fi.autosize))...)
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 3))
}
} }
// jirl r0, r1, 0 — return. // jirl r0, r1, 0, return.
ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0)) ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0))
return l64WordsLE(ws...) return l64WordsLE(ws...)
} }
// loong64StoreWords reports the prologue word count of the LR store, and
// loong64AdjustWords the word count of an SP adjust of v: the immediate
// forms when they fit, otherwise the R30 materialisation sequences.
func loong64StoreWords(autosize int) int {
if autosize > 2046 {
return 3
}
return 1
}
func loong64AdjustWords(v int64) int {
if loong64Imm12(v) {
return 1
}
return loong64MatLen(v) + 1
}
// loong64EpilogueWords reports the epilogue word count the RET expands to.
func loong64EpilogueWords(fi loong64FrameInfo) int {
n := loong64AdjustWords(int64(fi.autosize))
if !fi.leaf {
n++
}
return n
}
// loong64ResolvePseudo translates a pseudo-register memory reference into a // loong64ResolvePseudo translates a pseudo-register memory reference into a
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP); // hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
// x-N(SP) → (autosize - N)(SP). Returns base = -1 for an unresolvable // x-N(SP) → (autosize - N)(SP). Returns base = -1 for an unresolvable
+42
View File
@@ -0,0 +1,42 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestLOONG64RelocOffsetsIncludePrologue pins the function-relative
// relocation offsets of a framed loong64 function: the offsets used to
// exclude the prologue, so every relocation landed on a prologue
// instruction in the GOOBJ/ELF output.
func TestLOONG64RelocOffsetsIncludePrologue(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", "TEXT \u00b7f(SB), $16-0\n"+
"\tMOVV $gdata(SB), R4\n"+
"\tRET\n"+
"GLOBL gdata(SB), $8\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
fn := img.Funcs[0]
// Layout: 12-byte prologue (autosize 32), pcalau12i+addi.d (12, 16),
// epilogue with RET.
if len(fn.Relocs) != 2 {
t.Fatalf("relocs = %d, want 2", len(fn.Relocs))
}
hi, lo := fn.Relocs[0], fn.Relocs[1]
if hi.Kind != RelLoong64AddrHi || hi.Off != 12 || hi.After != 12 {
t.Errorf("hi reloc = {off %d after %d kind %d}, want {off 12 after 12 kind RelLoong64AddrHi}", hi.Off, hi.After, hi.Kind)
}
if lo.Kind != RelLoong64AddrLo || lo.Off != 16 || lo.After != 16 {
t.Errorf("lo reloc = {off %d after %d kind %d}, want {off 16 after 16 kind RelLoong64AddrLo}", lo.Off, lo.After, lo.Kind)
}
}
-5
View File
@@ -37,11 +37,6 @@ func Idx(base, index Reg, scale int, disp int64, size int) Mem {
return Mem{Base: base, Index: index, Scale: scale, Disp: disp, Size: size, HasBase: true, HasIndex: true} return Mem{Base: base, Index: index, Scale: scale, Disp: disp, Size: size, HasBase: true, HasIndex: true}
} }
// Rip builds a RIP-relative memory operand (RIP)+disp.
func Rip(disp int64, size int) Mem {
return Mem{Disp: disp, Size: size}
}
// sbMem is a memory operand that references a static (SB) symbol. It encodes // sbMem is a memory operand that references a static (SB) symbol. It encodes
// as a RIP-relative reference with a placeholder displacement; the encoder // as a RIP-relative reference with a placeholder displacement; the encoder
// records a patch site so the file-level layout can fill in the true rel32 // records a patch site so the file-level layout can fill in the true rel32
+31 -31
View File
@@ -7,36 +7,38 @@
// by round-tripping through golang.org/x/arch's decoder in the tests. // by round-tripping through golang.org/x/arch's decoder in the tests.
package asm package asm
import "maps"
import "strings" import "strings"
// Reg is an x86-64 register. In Plan 9 assembly the classic names (AX, BX, …) // Reg is an x86-64 register. In Plan 9 assembly the classic names (AX, BX, …)
// are size-agnostic — the instruction suffix (MOVQ vs MOVL) fixes the width — // are size-agnostic, the instruction suffix (MOVQ vs MOVL) fixes the width
// so the encoder keys off the register's index and lets the mnemonic supply the // so the encoder keys off the register's index and lets the mnemonic supply the
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which // size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4–7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share // occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one. The mask flag marks the AVX-512 opmask // those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0–K7. // registers K0-K7.
type Reg struct { type Reg struct {
idx int idx int
size int // informational width implied by the name; the mnemonic decides size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH high bool // AH/CH/DH/BH
mask bool // K0–K7 opmask register mask bool // K0-K7 opmask register
} }
// Index returns the register number (0–15 for GPRs, 0–31 for vectors). // Index returns the register number (0-15 for GPRs, 0-31 for vectors).
func (r Reg) Index() int { return r.idx } func (r Reg) Index() int { return r.idx }
// Size returns the width in bytes implied by the register's name. // Size returns the width in bytes implied by the register's name.
func (r Reg) Size() int { return r.size } func (r Reg) Size() int { return r.size }
// IsMask reports whether r is an AVX-512 opmask register (K0–K7). // IsMask reports whether r is an AVX-512 opmask register (K0-K7).
func (r Reg) IsMask() bool { return r.mask } func (r Reg) IsMask() bool { return r.mask }
func (r Reg) isOperand() {} func (r Reg) isOperand() {}
// needsREX reports whether this register forces a REX prefix at the given // needsREX reports whether this register forces a REX prefix at the given
// operand size: the extended registers R8–R15 always do, and at byte size the // operand size: the extended registers R8-R15 always do, and at byte size the
// low registers SPL/BPL/SIL/DIL (indices 4–7, not high) do as well. // low registers SPL/BPL/SIL/DIL (indices 4-7, not high) do as well.
func (r Reg) needsREX(opSize int) bool { func (r Reg) needsREX(opSize int) bool {
if r.idx >= 8 { if r.idx >= 8 {
return true return true
@@ -63,28 +65,28 @@ var (
CX = Reg{idx: 1, size: 2} CX = Reg{idx: 1, size: 2}
DX = Reg{idx: 2, size: 2} DX = Reg{idx: 2, size: 2}
BX = Reg{idx: 3, size: 2} BX = Reg{idx: 3, size: 2}
SP = Reg{idx: 4, size: 2} _ = Reg{idx: 4, size: 2}
BP = Reg{idx: 5, size: 2} _ = Reg{idx: 5, size: 2}
SI = Reg{idx: 6, size: 2} SI = Reg{idx: 6, size: 2}
DI = Reg{idx: 7, size: 2} DI = Reg{idx: 7, size: 2}
EAX = Reg{idx: 0, size: 4} _ = Reg{idx: 0, size: 4}
ECX = Reg{idx: 1, size: 4} _ = Reg{idx: 1, size: 4}
EDX = Reg{idx: 2, size: 4} _ = Reg{idx: 2, size: 4}
EBX = Reg{idx: 3, size: 4} _ = Reg{idx: 3, size: 4}
ESP = Reg{idx: 4, size: 4} _ = Reg{idx: 4, size: 4}
EBP = Reg{idx: 5, size: 4} _ = Reg{idx: 5, size: 4}
ESI = Reg{idx: 6, size: 4} _ = Reg{idx: 6, size: 4}
EDI = Reg{idx: 7, size: 4} _ = Reg{idx: 7, size: 4}
RAX = Reg{idx: 0, size: 8} _ = Reg{idx: 0, size: 8}
RCX = Reg{idx: 1, size: 8} _ = Reg{idx: 1, size: 8}
RDX = Reg{idx: 2, size: 8} _ = Reg{idx: 2, size: 8}
RBX = Reg{idx: 3, size: 8} _ = Reg{idx: 3, size: 8}
RSP = Reg{idx: 4, size: 8} _ = Reg{idx: 4, size: 8}
RBP = Reg{idx: 5, size: 8} _ = Reg{idx: 5, size: 8}
RSI = Reg{idx: 6, size: 8} _ = Reg{idx: 6, size: 8}
RDI = Reg{idx: 7, size: 8} _ = Reg{idx: 7, size: 8}
) )
// regByName maps an assembly register name (case-insensitive) to a Reg. // regByName maps an assembly register name (case-insensitive) to a Reg.
@@ -121,19 +123,17 @@ func buildRegByName() map[string]Reg {
} }
// 8-bit: AL..BH, SPL..DIL, R8B..R15B. // 8-bit: AL..BH, SPL..DIL, R8B..R15B.
for n, r := range map[string]Reg{ maps.Copy(m, map[string]Reg{
"AL": AL, "CL": CL, "DL": DL, "BL": BL, "AL": AL, "CL": CL, "DL": DL, "BL": BL,
"AH": AH, "CH": CH, "DH": DH, "BH": BH, "AH": AH, "CH": CH, "DH": DH, "BH": BH,
"SPL": SPL, "BPL": BPL, "SIL": SIL, "DIL": DIL, "SPL": SPL, "BPL": BPL, "SIL": SIL, "DIL": DIL,
} { })
m[n] = r
}
for i := 8; i <= 15; i++ { for i := 8; i <= 15; i++ {
m["R"+itoa(i)+"B"] = Reg{idx: i, size: 1} m["R"+itoa(i)+"B"] = Reg{idx: i, size: 1}
} }
// Vector: X0..X31 (128-bit, size 16), Y0..Y31 (256-bit, size 32), // Vector: X0..X31 (128-bit, size 16), Y0..Y31 (256-bit, size 32),
// Z0..Z31 (512-bit, size 64). Indices 16–31 are only encodable in EVEX // Z0..Z31 (512-bit, size 64). Indices 16-31 are only encodable in EVEX
// (AVX-512) instructions; the encoder validates that through its tables. // (AVX-512) instructions; the encoder validates that through its tables.
for i := 0; i <= 31; i++ { for i := 0; i <= 31; i++ {
m["X"+itoa(i)] = Reg{idx: i, size: 16} m["X"+itoa(i)] = Reg{idx: i, size: 16}
+92 -32
View File
@@ -15,14 +15,16 @@ import (
func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) { func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
fi := riscvComputeFrame(t) fi := riscvComputeFrame(t)
prologue := riscvPrologue(fi) prologue := riscvPrologue(fi)
guardLen := riscvGuardLen(fi)
var relocs []Reloc var relocs []Reloc
var spadj []SpadjStep var spadj []SpadjStep
// The prologue raises the SP delta by autosize; the boundary is reported // The prologue raises the SP delta by autosize; the boundary is reported
// at the pc just past its ADDI, exactly as the toolchain's pctospadj does. // at the pc just past its ADDI, exactly as the toolchain's pctospadj does.
// The guard prefix shifts its PC.
if fi.autosize != 0 { if fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: riscvPrologueSpadjPC(fi), Value: fi.autosize}) spadj = append(spadj, SpadjStep{PC: guardLen + riscvPrologueSpadjPC(fi), Value: fi.autosize})
} }
// Pass 1: collect instructions and compute label offsets assuming 4 bytes // Pass 1: collect instructions and compute label offsets assuming 4 bytes
@@ -34,7 +36,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
} }
var recs []instrRec var recs []instrRec
offsets := map[string]int{} offsets := map[string]int{}
pos := len(prologue) pos := guardLen + len(prologue)
for _, stmt := range t.Body { for _, stmt := range t.Body {
switch s := stmt.(type) { switch s := stmt.(type) {
case *ast.Label: case *ast.Label:
@@ -66,7 +68,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
// Pass 4: recompute offsets with actual sizes. // Pass 4: recompute offsets with actual sizes.
offsets = map[string]int{} offsets = map[string]int{}
pos = len(prologue) pos = guardLen + len(prologue)
for _, stmt := range t.Body { for _, stmt := range t.Body {
switch s := stmt.(type) { switch s := stmt.(type) {
case *ast.Label: case *ast.Label:
@@ -82,9 +84,17 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
} }
// Pass 5: re-encode branches with corrected offsets. Record relocations // Pass 5: re-encode branches with corrected offsets. Record relocations
// during this final pass (relocation offsets are relative to instruction start). // during this final pass (relocation offsets are relative to instruction
out := append([]byte(nil), prologue...) // start). The guard prefix precedes the prologue; its branches target
pc = len(prologue) // the morestack block at the end of the function, which the previous
// passes have sized.
var out []byte
guardBytes, guardReloc := riscvGuard(fi)
if fi.needSplit {
out = append(out, guardBytes...)
}
out = append(out, prologue...)
pc = guardLen + len(prologue)
preCount := len(relocs) preCount := len(relocs)
var lines []LineEntry var lines []LineEntry
for _, r := range recs { for _, r := range recs {
@@ -119,6 +129,9 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
pc += len(code) pc += len(code)
} }
} }
if fi.needSplit {
relocs = append(relocs, guardReloc)
}
return out, offsets, relocs, lines, spadj, nil return out, offsets, relocs, lines, spadj, nil
} }
@@ -149,6 +162,14 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
if isImmOperand(ops[0]) && ops[0].Imm.Sym == nil { if isImmOperand(ops[0]) && ops[0].Imm.Sym == nil {
return riscvMovImmSize(regFromOperand(ops[1]), immFromOperand(ops[0])) return riscvMovImmSize(regFromOperand(ops[1]), immFromOperand(ops[0]))
} }
// Frame-relative loads and stores: a frame offset beyond the signed
// 12-bit range materialises the address in X31 first.
if isMemOperand(ops[0]) && !isMemOperand(ops[1]) {
return riscvFrameMemSize(ops[0], fi)
}
if isMemOperand(ops[1]) && !isMemOperand(ops[0]) {
return riscvFrameMemSize(ops[1], fi)
}
} }
// I-type arithmetic with a large immediate expands to several instructions. // I-type arithmetic with a large immediate expands to several instructions.
if (mnem == "ADDI" || mnem == "ANDI" || mnem == "ORI" || mnem == "XORI") && len(ops) >= 1 && isImmOperand(ops[0]) { if (mnem == "ADDI" || mnem == "ANDI" || mnem == "ORI" || mnem == "XORI") && len(ops) >= 1 && isImmOperand(ops[0]) {
@@ -192,13 +213,21 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
if relocs != nil { if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelRISCVJal, Addend: op.Addr.Sym.Offset}) *relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelRISCVJal, Addend: op.Addr.Sym.Offset})
} }
word = riscvJType(1, 0) // JAL X1, 0 — the linker fills the offset word = riscvJType(1, 0) // JAL X1, 0, the linker fills the offset
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
case "JMP": case "JMP":
// JMP = JAL X0, target. The Go assembler never compresses this to // JMP = JAL X0, target. The Go assembler never compresses this to
// C.J, so always emit the 32-bit JAL. // C.J, so always emit the 32-bit JAL.
var target string var target string
if len(ops) >= 1 { if len(ops) >= 1 {
// JMP sym(SB): a tail call, JAL X0 against a symbol relocation.
if ops[0].Addr.Sym != nil && ops[0].Addr.Sym.Pseudo == "SB" {
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: ops[0].Addr.Sym.Name, Kind: RelRISCVJal, Addend: ops[0].Addr.Sym.Offset})
}
word = riscvJType(0, 0)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
target = labelFromOperand(ops[0]) target = labelFromOperand(ops[0])
} }
targetOff, ok := offsets[target] targetOff, ok := offsets[target]
@@ -228,7 +257,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
// MOV is a pseudo-instruction that the Go assembler uses for loads, // MOV is a pseudo-instruction that the Go assembler uses for loads,
// stores, register moves and immediate loads. // stores, register moves and immediate loads.
case "MOV": case "MOV":
return encodeRISCVMov(instr, offsets, fi, relocs) return encodeRISCVMov(instr, fi, relocs)
// JALR: indirect jump/call. Plan 9: JALR rs1, rd or JALR offset(rs1). // JALR: indirect jump/call. Plan 9: JALR rs1, rd or JALR offset(rs1).
case "JALR": case "JALR":
@@ -395,7 +424,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
} }
word = riscvSType(enc, rs1, rs2, imm) word = riscvSType(enc, rs1, rs2, imm)
// LR (load-reserved): INSTR (addr), dst — 2 operands. // LR (load-reserved): INSTR (addr), dst, 2 operands.
case len(ops) == 2 && isLRInstr(mnem): case len(ops) == 2 && isLRInstr(mnem):
rs1, _ := memFromOperandWithFrame(ops[0], fi) rs1, _ := memFromOperandWithFrame(ops[0], fi)
rd := regFromOperand(ops[1]) rd := regFromOperand(ops[1])
@@ -404,7 +433,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
} }
word = riscvAMOType(enc, rd, rs1, 0) // rs2=0 for LR word = riscvAMOType(enc, rd, rs1, 0) // rs2=0 for LR
// SC (store-conditional): INSTR src, (addr), dst — 3 operands. // SC (store-conditional): INSTR src, (addr), dst, 3 operands.
case len(ops) == 3 && isSCInstr(mnem): case len(ops) == 3 && isSCInstr(mnem):
rs2 := regFromOperand(ops[0]) rs2 := regFromOperand(ops[0])
rs1, _ := memFromOperandWithFrame(ops[1], fi) rs1, _ := memFromOperandWithFrame(ops[1], fi)
@@ -443,7 +472,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
} }
return encodeRISCVItypeImmediate(mnem, enc, rd, rd, imm) return encodeRISCVItypeImmediate(mnem, enc, rd, rd, imm)
// Loads: rd, offset(rs1) — Plan 9 order is LD src, dst. // Loads: rd, offset(rs1), Plan 9 order is LD src, dst.
case len(ops) == 2 && isLoadInstr(mnem): case len(ops) == 2 && isLoadInstr(mnem):
rd := regFromOperand(ops[1]) // destination (last operand) rd := regFromOperand(ops[1]) // destination (last operand)
rs1, imm := memFromOperandWithFrame(ops[0], fi) // memory source (first operand) rs1, imm := memFromOperandWithFrame(ops[0], fi) // memory source (first operand)
@@ -527,7 +556,7 @@ func isImmOperand(op *ast.Operand) bool {
// - MOV Rs, (Rd) register-relative store // - MOV Rs, (Rd) register-relative store
// - MOV Rs, Rd register-to-register move (ADDI $0) // - MOV Rs, Rd register-to-register move (ADDI $0)
// - MOV $imm, Rd load immediate (ADDI or LUI+ADDIW) // - MOV $imm, Rd load immediate (ADDI or LUI+ADDIW)
func encodeRISCVMov(instr *ast.Instr, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) { func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) {
ops := instr.Operands ops := instr.Operands
if len(ops) != 2 { if len(ops) != 2 {
return nil, fmt.Errorf("MOV expects 2 operands, got %d", len(ops)) return nil, fmt.Errorf("MOV expects 2 operands, got %d", len(ops))
@@ -538,7 +567,7 @@ func encodeRISCVMov(instr *ast.Instr, offsets map[string]int, fi riscvFrameInfo,
// Immediate → register. // Immediate → register.
if isImmOperand(src) { if isImmOperand(src) {
// MOV $sym(SB), rd — load address of a static symbol or external. // MOV $sym(SB), rd, load address of a static symbol or external.
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
rd := regFromOperand(dst) rd := regFromOperand(dst)
if rd < 0 { if rd < 0 {
@@ -546,7 +575,7 @@ func encodeRISCVMov(instr *ast.Instr, offsets map[string]int, fi riscvFrameInfo,
} }
return encodeRISCVSBAddr(src.Imm.Sym, rd, relocs), nil return encodeRISCVSBAddr(src.Imm.Sym, rd, relocs), nil
} }
// MOV $sym(FP/SP), rd — not supported: immediate symbol references // MOV $sym(FP/SP), rd, not supported: immediate symbol references
// other than SB cannot be encoded as a simple immediate. // other than SB cannot be encoded as a simple immediate.
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo != "" { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo != "" {
return nil, fmt.Errorf("MOV $%s(%s): unsupported immediate symbol reference (only SB is supported)", src.Imm.Sym.Name, src.Imm.Sym.Pseudo) return nil, fmt.Errorf("MOV $%s(%s): unsupported immediate symbol reference (only SB is supported)", src.Imm.Sym.Name, src.Imm.Sym.Pseudo)
@@ -562,7 +591,7 @@ func encodeRISCVMov(instr *ast.Instr, offsets map[string]int, fi riscvFrameInfo,
// Memory → register (load). // Memory → register (load).
if isMemOperand(src) && !isMemOperand(dst) { if isMemOperand(src) && !isMemOperand(dst) {
rd := regFromOperand(dst) rd := regFromOperand(dst)
// MOV sym(SB), rd — load from static data. // MOV sym(SB), rd, load from static data.
if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" { if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" {
if rd < 0 { if rd < 0 {
return nil, fmt.Errorf("MOV sym(SB): invalid destination register") return nil, fmt.Errorf("MOV sym(SB): invalid destination register")
@@ -573,14 +602,13 @@ func encodeRISCVMov(instr *ast.Instr, offsets map[string]int, fi riscvFrameInfo,
if rd < 0 || rs1 < 0 { if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("MOV load: invalid operand") return nil, fmt.Errorf("MOV load: invalid operand")
} }
word := riscvIType(riscvEnc{0x03, 0x3, 0x00}, rd, rs1, off) return riscvFrameMemOp(riscvEnc{0x03, 0x3, 0x00}, false, rd, rs1, off), nil
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
} }
// Register → memory (store). // Register → memory (store).
if !isMemOperand(src) && isMemOperand(dst) { if !isMemOperand(src) && isMemOperand(dst) {
rs2 := regFromOperand(src) rs2 := regFromOperand(src)
// MOV rd, sym(SB) — store to static data. // MOV rd, sym(SB), store to static data.
if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" { if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" {
if rs2 < 0 { if rs2 < 0 {
return nil, fmt.Errorf("MOV rd, sym(SB): invalid source register") return nil, fmt.Errorf("MOV rd, sym(SB): invalid source register")
@@ -591,8 +619,7 @@ func encodeRISCVMov(instr *ast.Instr, offsets map[string]int, fi riscvFrameInfo,
if rs2 < 0 || rs1 < 0 { if rs2 < 0 || rs1 < 0 {
return nil, fmt.Errorf("MOV store: invalid operand") return nil, fmt.Errorf("MOV store: invalid operand")
} }
word := riscvSType(riscvEnc{0x23, 0x3, 0x00}, rs1, rs2, off) return riscvFrameMemOp(riscvEnc{0x23, 0x3, 0x00}, true, rs2, rs1, off), nil
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
} }
// Register → register (ADDI $0, src, dst). // Register → register (ADDI $0, src, dst).
@@ -607,6 +634,44 @@ func encodeRISCVMov(instr *ast.Instr, offsets map[string]int, fi riscvFrameInfo,
} }
} }
// riscvFrameMemOp encodes a register-relative load (store=false, I-type
// width 0x03) or store (store=true, S-type width 0x23) of the 64-bit width
// at off(rs1). Offsets beyond the signed 12-bit range materialise the
// address in X31 first: LUI hi (the rounding split), then ADD X31, rs1,
// matching the toolchain's large-frame addressing; the access uses the
// sign-extended low part, which always fits.
func riscvFrameMemOp(enc riscvEnc, store bool, reg, rs1 int, off int32) []byte {
if fits12(off) {
var word uint32
if store {
word = riscvSType(enc, rs1, reg, off)
} else {
word = riscvIType(enc, reg, rs1, off)
}
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
}
lo := off - (splitHi(off) << 12)
out := riscvAddressInX31WithBase(off, rs1)
var word uint32
if store {
word = riscvSType(enc, 31, reg, lo)
} else {
word = riscvIType(enc, reg, 31, lo)
}
return append(out, wordLE(word)...)
}
// riscvFrameMemSize returns the encoded size of a frame-relative MOV for the
// layout pass: 4 bytes when the offset fits, otherwise the X31
// materialisation plus the access.
func riscvFrameMemSize(op *ast.Operand, fi riscvFrameInfo) int {
rs1, off := memFromOperandWithFrame(op, fi)
if fits12(off) {
return 4
}
return len(riscvAddressInX31WithBase(off, rs1)) + 4
}
// encodeRISCVLoadImm encodes loading an immediate into a register (MOV $imm, // encodeRISCVLoadImm encodes loading an immediate into a register (MOV $imm,
// rd), matching the toolchain's instructionsForMOVConst. For 12-bit // rd), matching the toolchain's instructionsForMOVConst. For 12-bit
// immediates it emits ADDI $imm, ZERO, rd (compressed to C.LI when it fits // immediates it emits ADDI $imm, ZERO, rd (compressed to C.LI when it fits
@@ -911,7 +976,7 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
} }
case "ADDI": case "ADDI":
rd, rs1, imm := extractITypeParams(instr, fi) rd, rs1, imm := extractITypeParams(instr)
if rd == -1 || rs1 == -1 { if rd == -1 || rs1 == -1 {
return 0, false return 0, false
} }
@@ -1006,7 +1071,7 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
} }
case "ADDW", "SUBW": case "ADDW", "SUBW":
// C.ADDW (0x27,1) / C.SUBW (0x27,0) — CA-type, prime regs. // C.ADDW (0x27,1) / C.SUBW (0x27,0), CA-type, prime regs.
if len(ops) == 3 { if len(ops) == 3 {
funct2 := uint32(0x0) funct2 := uint32(0x0)
if mnem == "ADDW" { if mnem == "ADDW" {
@@ -1060,13 +1125,13 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
} }
case "ADDIW": case "ADDIW":
rd, rs1, imm := extractITypeParams(instr, fi) rd, rs1, imm := extractITypeParams(instr)
if rd == rs1 && rd != 0 && imm >= -32 && imm <= 31 { if rd == rs1 && rd != 0 && imm >= -32 && imm <= 31 {
return rvcCI(0x1, uint32(rd), uint32(imm)&0x3F), true return rvcCI(0x1, uint32(rd), uint32(imm)&0x3F), true
} }
case "SLLI", "SRLI", "SRAI": case "SLLI", "SRLI", "SRAI":
rd, rs1, imm := extractITypeParams(instr, fi) rd, rs1, imm := extractITypeParams(instr)
if rd == rs1 && rd != 0 && imm != 0 && imm >= 1 && imm <= 63 { if rd == rs1 && rd != 0 && imm != 0 && imm >= 1 && imm <= 63 {
if mnem == "SLLI" { if mnem == "SLLI" {
// C.SLLI: funct3=0, op=10 quadrant, shamt in bits [12|6:2]. // C.SLLI: funct3=0, op=10 quadrant, shamt in bits [12|6:2].
@@ -1083,7 +1148,7 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
} }
case "ANDI": case "ANDI":
rd, rs1, imm := extractITypeParams(instr, fi) rd, rs1, imm := extractITypeParams(instr)
if isRVCIntReg(rd) && rd == rs1 && imm >= -32 && imm <= 31 { if isRVCIntReg(rd) && rd == rs1 && imm >= -32 && imm <= 31 {
// C.ANDI: CB-type, funct3=0x4, funct2=0x2. // C.ANDI: CB-type, funct3=0x4, funct2=0x2.
return rvcCBShift(0x2, rvcReg3(rd), uint32(imm)&0x3F), true return rvcCBShift(0x2, rvcReg3(rd), uint32(imm)&0x3F), true
@@ -1131,7 +1196,7 @@ func extractSDParams(instr *ast.Instr, fi riscvFrameInfo) (rs2, rs1 int, imm int
// extractITypeParams extracts rd, rs1, and immediate for an I-type // extractITypeParams extracts rd, rs1, and immediate for an I-type
// instruction. The Plan 9 order is INSTR $imm, rs1, rd (3 operands) or // instruction. The Plan 9 order is INSTR $imm, rs1, rd (3 operands) or
// INSTR $imm, rd (2 operands, rd is also the source). // INSTR $imm, rd (2 operands, rd is also the source).
func extractITypeParams(instr *ast.Instr, fi riscvFrameInfo) (rd, rs1 int, imm int32) { func extractITypeParams(instr *ast.Instr) (rd, rs1 int, imm int32) {
ops := instr.Operands ops := instr.Operands
switch len(ops) { switch len(ops) {
case 3: case 3:
@@ -1250,11 +1315,6 @@ func isFPCmpInstr(m string) bool {
return false return false
} }
func isFPCvtInstr(m string) bool {
_, ok := riscvCvtTable[m]
return ok
}
// Operand helpers. // Operand helpers.
func regFromOperand(op *ast.Operand) int { func regFromOperand(op *ast.Operand) int {
// Register is in Addr.Base (from (base) syntax) or Addr.Sym.Name (bare ident). // Register is in Addr.Base (from (base) syntax) or Addr.Sym.Name (bare ident).
@@ -1319,7 +1379,7 @@ func suggestLabel(target string, offsets map[string]int) string {
} }
// Only suggest if the distance is small enough. // Only suggest if the distance is small enough.
if bestDist <= 3 && bestDist < len(target)/2+1 { if bestDist <= 3 && bestDist < len(target)/2+1 {
return fmt.Sprintf(" — did you mean %q?", best) return fmt.Sprintf("; did you mean %q?", best)
} }
return "" return ""
} }
+14 -14
View File
@@ -154,7 +154,7 @@ type riscvEnc struct {
// riscvInstrTable maps RISC-V mnemonics to their encoding. // riscvInstrTable maps RISC-V mnemonics to their encoding.
var riscvInstrTable = map[string]riscvEnc{ var riscvInstrTable = map[string]riscvEnc{
// RV64I — R-type arithmetic/logic. // RV64I, R-type arithmetic/logic.
"ADD": {0x33, 0x0, 0x00}, "ADD": {0x33, 0x0, 0x00},
"SUB": {0x33, 0x0, 0x20}, "SUB": {0x33, 0x0, 0x20},
"SLL": {0x33, 0x1, 0x00}, "SLL": {0x33, 0x1, 0x00},
@@ -165,20 +165,20 @@ var riscvInstrTable = map[string]riscvEnc{
"SRA": {0x33, 0x5, 0x20}, "SRA": {0x33, 0x5, 0x20},
"OR": {0x33, 0x6, 0x00}, "OR": {0x33, 0x6, 0x00},
"AND": {0x33, 0x7, 0x00}, "AND": {0x33, 0x7, 0x00},
// RV64I — 32-bit variants (W suffix). // RV64I, 32-bit variants (W suffix).
"ADDW": {0x3B, 0x0, 0x00}, "ADDW": {0x3B, 0x0, 0x00},
"SUBW": {0x3B, 0x0, 0x20}, "SUBW": {0x3B, 0x0, 0x20},
"SLLW": {0x3B, 0x1, 0x00}, "SLLW": {0x3B, 0x1, 0x00},
"SRLW": {0x3B, 0x5, 0x00}, "SRLW": {0x3B, 0x5, 0x00},
"SRAW": {0x3B, 0x5, 0x20}, "SRAW": {0x3B, 0x5, 0x20},
// RV64I — I-type shift-immediate (shamt in rs2 field). // RV64I, I-type shift-immediate (shamt in rs2 field).
"SLLI": {0x13, 0x1, 0x00}, "SLLI": {0x13, 0x1, 0x00},
"SRLI": {0x13, 0x5, 0x00}, "SRLI": {0x13, 0x5, 0x00},
"SRAI": {0x13, 0x5, 0x20}, "SRAI": {0x13, 0x5, 0x20},
"SLLIW": {0x1B, 0x1, 0x00}, "SLLIW": {0x1B, 0x1, 0x00},
"SRLIW": {0x1B, 0x5, 0x00}, "SRLIW": {0x1B, 0x5, 0x00},
"SRAIW": {0x1B, 0x5, 0x20}, "SRAIW": {0x1B, 0x5, 0x20},
// RV64M — multiply/divide. // RV64M, multiply/divide.
"MUL": {0x33, 0x0, 0x01}, "MUL": {0x33, 0x0, 0x01},
"MULH": {0x33, 0x1, 0x01}, "MULH": {0x33, 0x1, 0x01},
"MULHSU": {0x33, 0x2, 0x01}, "MULHSU": {0x33, 0x2, 0x01},
@@ -187,13 +187,13 @@ var riscvInstrTable = map[string]riscvEnc{
"DIVU": {0x33, 0x5, 0x01}, "DIVU": {0x33, 0x5, 0x01},
"REM": {0x33, 0x6, 0x01}, "REM": {0x33, 0x6, 0x01},
"REMU": {0x33, 0x7, 0x01}, "REMU": {0x33, 0x7, 0x01},
// RV64M — 32-bit variants. // RV64M, 32-bit variants.
"MULW": {0x3B, 0x0, 0x01}, "MULW": {0x3B, 0x0, 0x01},
"DIVW": {0x3B, 0x4, 0x01}, "DIVW": {0x3B, 0x4, 0x01},
"DIVUW": {0x3B, 0x5, 0x01}, "DIVUW": {0x3B, 0x5, 0x01},
"REMW": {0x3B, 0x6, 0x01}, "REMW": {0x3B, 0x6, 0x01},
"REMUW": {0x3B, 0x7, 0x01}, "REMUW": {0x3B, 0x7, 0x01},
// RV64I — I-type arithmetic. // RV64I, I-type arithmetic.
"ADDI": {0x13, 0x0, 0x00}, "ADDI": {0x13, 0x0, 0x00},
"ADDIW": {0x1B, 0x0, 0x00}, "ADDIW": {0x1B, 0x0, 0x00},
"SLTI": {0x13, 0x2, 0x00}, "SLTI": {0x13, 0x2, 0x00},
@@ -228,10 +228,10 @@ var riscvInstrTable = map[string]riscvEnc{
"ECALL": {0x73, 0x0, 0x00}, "ECALL": {0x73, 0x0, 0x00},
"EBREAK": {0x73, 0x0, 0x00}, "EBREAK": {0x73, 0x0, 0x00},
"FENCE": {0x0F, 0x0, 0x00}, "FENCE": {0x0F, 0x0, 0x00},
// JALR — indirect jump/call (I-type). // JALR, indirect jump/call (I-type).
"JALR": {0x67, 0x0, 0x00}, "JALR": {0x67, 0x0, 0x00},
// RV64A — atomics (AMO opcode 0x2F). // RV64A, atomics (AMO opcode 0x2F).
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27]. // funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2}, "AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2}, "AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
@@ -252,7 +252,7 @@ var riscvInstrTable = map[string]riscvEnc{
"AMOMINUW": {0x2F, 0x2, 0x18 << 2}, "AMOMINUW": {0x2F, 0x2, 0x18 << 2},
"AMOMINUD": {0x2F, 0x3, 0x18 << 2}, "AMOMINUD": {0x2F, 0x3, 0x18 << 2},
// RV64F/D — floating-point arithmetic. // RV64F/D, floating-point arithmetic.
"FADDS": {0x53, 0x0, 0x00}, "FADDS": {0x53, 0x0, 0x00},
"FSUBS": {0x53, 0x0, 0x04}, "FSUBS": {0x53, 0x0, 0x04},
"FMULS": {0x53, 0x0, 0x08}, "FMULS": {0x53, 0x0, 0x08},
@@ -274,13 +274,13 @@ var riscvInstrTable = map[string]riscvEnc{
"FMIND": {0x53, 0x0, 0x15}, "FMIND": {0x53, 0x0, 0x15},
"FMAXD": {0x53, 0x1, 0x15}, "FMAXD": {0x53, 0x1, 0x15},
// RV64A — load-reserved / store-conditional (funct5 0x02 / 0x03). // RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
"LRW": {0x2F, 0x2, 0x02 << 2}, "LRW": {0x2F, 0x2, 0x02 << 2},
"LRD": {0x2F, 0x3, 0x02 << 2}, "LRD": {0x2F, 0x3, 0x02 << 2},
"SCW": {0x2F, 0x2, 0x03 << 2}, "SCW": {0x2F, 0x2, 0x03 << 2},
"SCD": {0x2F, 0x3, 0x03 << 2}, "SCD": {0x2F, 0x3, 0x03 << 2},
// FP compare — result in integer register (funct7 0x50/0x51). // FP compare, result in integer register (funct7 0x50/0x51).
"FEQS": {0x53, 0x2, 0x50}, "FEQS": {0x53, 0x2, 0x50},
"FLTS": {0x53, 0x1, 0x50}, "FLTS": {0x53, 0x1, 0x50},
"FLES": {0x53, 0x0, 0x50}, "FLES": {0x53, 0x0, 0x50},
@@ -442,10 +442,10 @@ func riscvJType(rd int, offset int32) uint32 {
// ---- RVC (compressed) encoding helpers ---- // ---- RVC (compressed) encoding helpers ----
// isRVCIntReg reports whether a register number can be encoded in the 3-bit // isRVCIntReg reports whether a register number can be encoded in the 3-bit
// prime register field used by compressed instructions (x8–x15). // prime register field used by compressed instructions (x8-x15).
func isRVCIntReg(r int) bool { return r >= 8 && r <= 15 } func isRVCIntReg(r int) bool { return r >= 8 && r <= 15 }
// rvcReg3 returns the 3-bit encoding for registers x8–x15 (0–7). // rvcReg3 returns the 3-bit encoding for registers x8-x15 (0-7).
func rvcReg3(r int) uint32 { return uint32(r - 8) } func rvcReg3(r int) uint32 { return uint32(r - 8) }
// rvcCR encodes a CR-type (register) compressed instruction. // rvcCR encodes a CR-type (register) compressed instruction.
@@ -455,7 +455,7 @@ func rvcCR(funct4, rd, rs2 uint32) uint16 {
} }
// rvcCI encodes a CI-type (immediate) compressed instruction. // rvcCI encodes a CI-type (immediate) compressed instruction.
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW — linear 6-bit immediate. // Used for C.ADDI, C.LI, C.LUI, C.ADDIW, linear 6-bit immediate.
func rvcCI(funct3, rd uint32, imm uint32) uint16 { func rvcCI(funct3, rd uint32, imm uint32) uint16 {
return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x1) return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x1)
} }
+185 -13
View File
@@ -32,6 +32,13 @@ import (
// riscvFrameInfo holds the frame layout derived from a TEXT directive. // riscvFrameInfo holds the frame layout derived from a TEXT directive.
type riscvFrameInfo struct { type riscvFrameInfo struct {
autosize int // the real SP adjustment (locals + saved LR) autosize int // the real SP adjustment (locals + saved LR)
// Stack-split guard state: the toolchain emits the check for every
// non-NOSPLIT function whose autosize is nonzero (a zero autosize is
// "effectively NOSPLIT"); unlike amd64 and arm64 there is no leaf
// auto-NOSPLIT.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
} }
// riscvComputeFrame derives the frame layout for a TEXT function. // riscvComputeFrame derives the frame layout for a TEXT function.
@@ -40,11 +47,34 @@ func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
if frame != 0 || !riscvIsLeaf(t) { if frame != 0 || !riscvIsLeaf(t) {
// FixedFrameSize = 8: space for the saved link register. A // FixedFrameSize = 8: space for the saved link register. A
// zero-frame non-leaf function still opens an 8-byte frame for LR. // zero-frame non-leaf function still opens an 8-byte frame for LR.
return riscvFrameInfo{autosize: frame + 8} autosize := frame + 8
fi := riscvFrameInfo{autosize: autosize}
if !hasNoSplitFlag(t) {
fi.needSplit = true
switch {
case autosize <= stackSmall:
fi.splitClass = 0
case autosize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi
} }
return riscvFrameInfo{} return riscvFrameInfo{}
} }
// hasNoSplitFlag reports whether the TEXT directive carries NOSPLIT.
func hasNoSplitFlag(t *ast.Text) bool {
for _, f := range t.Flags {
if strings.EqualFold(f, "NOSPLIT") {
return true
}
}
return false
}
// riscvIsLeaf reports whether a function contains no call instructions. // riscvIsLeaf reports whether a function contains no call instructions.
// CALL always links; JAL/JALR link only when their destination register is // CALL always links; JAL/JALR link only when their destination register is
// the link register (X1), matching cmd/internal/obj/riscv's containsCall. // the link register (X1), matching cmd/internal/obj/riscv's containsCall.
@@ -58,12 +88,12 @@ func riscvIsLeaf(t *ast.Text) bool {
case "CALL": case "CALL":
return false return false
case "JAL": case "JAL":
// JAL rd, target — a call only when rd is the link register. // JAL rd, target, a call only when rd is the link register.
if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 { if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 {
return false return false
} }
case "JALR": case "JALR":
// JALR rs1, rd — a call when rd is X1; JALR offset(rs1) always // JALR rs1, rd, a call when rd is X1; JALR offset(rs1) always
// links to X1. // links to X1.
if len(in.Operands) == 1 { if len(in.Operands) == 1 {
return false return false
@@ -84,28 +114,99 @@ func riscvPrologue(fi riscvFrameInfo) []byte {
return nil return nil
} }
var out []byte var out []byte
// MOV LR, -autosize(SP) — SD X1, -autosize(X2). The negative offset is // MOV LR, -autosize(SP), SD X1, -autosize(X2). The negative offset is
// not compressible to C.SDSP (unsigned), so it stays 4 bytes. // not compressible to C.SDSP (unsigned), so it stays 4 bytes. Beyond
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...) // the imm12 range the toolchain materialises the address in X31.
// ADDI $-autosize, SP, SP — open the frame (C.ADDI when it fits). if fits12(int32(-fi.autosize)) {
out = append(out, riscvSPAdjust(int32(-fi.autosize))...) out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
// MOV LR, 0(SP) — SD X1, 0(X2) → C.SDSP X1, 0. } else {
out = append(out, riscvAddressInX31(int32(-fi.autosize))...)
lo := int32(-fi.autosize) - (splitHi(int32(-fi.autosize)) << 12)
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 1, lo))...)
}
// ADDI $-autosize, SP, SP, open the frame (C.ADDI when it fits; X31
// materialisation beyond imm12).
if fits12(int32(-fi.autosize)) {
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
} else {
out = append(out, riscvAddToSP(int32(-fi.autosize))...)
}
// MOV LR, 0(SP), SD X1, 0(X2) → C.SDSP X1, 0.
c := rvcSSP(0x7, 1, 0) c := rvcSSP(0x7, 1, 0)
out = append(out, byte(c), byte(c>>8)) out = append(out, byte(c), byte(c>>8))
return out return out
} }
func fits12(v int32) bool { return v >= -2048 && v <= 2047 }
// splitHi returns the LUI half of the hi/lo split of v (what remains is the
// sign-extended 12-bit low part).
func splitHi(v int32) int32 {
_, high := splitRISCV32Imm(v)
return high
}
// riscvAddressInX31 materialises hi(v) into X31 against the stack pointer,
// matching the toolchain's large-frame addressing: C.LUI (or LUI) X31, hi;
// C.ADD (or ADD) X31, SP.
func riscvAddressInX31(v int32) []byte {
return riscvAddressInX31WithBase(v, 2)
}
// riscvAddressInX31WithBase materialises hi(v) into X31 against an arbitrary
// base register: LUI (or C.LUI) X31, hi; C.ADD X31, rs1. The CR rs2 field
// carries the full 5-bit register, so the compressed form is always
// available.
func riscvAddressInX31WithBase(v int32, rs1 int) []byte {
hi := splitHi(v)
var out []byte
if hi >= -32 && hi <= 31 {
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
}
c := rvcCR(0x9, 31, uint32(rs1))
return append(out, byte(c), byte(c>>8))
}
// riscvAddToSP adds v to SP through X31 for the values imm12 cannot carry:
// C.LUI X31, hi; C.ADDIW X31, lo; C.ADD SP, X31 (the toolchain's form).
func riscvAddToSP(v int32) []byte {
hi := splitHi(v)
lo := v - (hi << 12)
var out []byte
if hi >= -32 && hi <= 31 {
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
}
if lo >= -32 && lo <= 31 {
c := rvcCI(0x1, 31, uint32(lo)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvIType(riscvEnc{0x1b, 0x0, 0x00}, 31, 31, lo))...)
}
c := rvcCR(0x9, 2, 31)
return append(out, byte(c), byte(c>>8))
}
// riscvReturn returns the bytes for a RET: the epilogue (restore LR and // riscvReturn returns the bytes for a RET: the epilogue (restore LR and
// deallocate the frame when present) followed by the uncompressed JALR X0, // deallocate the frame when present) followed by the uncompressed JALR X0,
// 0(X1) the toolchain emits for RET (it never compresses RET to C.JR). // 0(X1) the toolchain emits for RET (it never compresses RET to C.JR).
func riscvReturn(fi riscvFrameInfo) []byte { func riscvReturn(fi riscvFrameInfo) []byte {
var out []byte var out []byte
if fi.autosize != 0 { if fi.autosize != 0 {
// MOV 0(SP), LR — LD X1, 0(X2) → C.LDSP X1, 0. // MOV 0(SP), LR, LD X1, 0(X2) → C.LDSP X1, 0.
c := rvcLSP(0x3, 1, 0) c := rvcLSP(0x3, 1, 0)
out = append(out, byte(c), byte(c>>8)) out = append(out, byte(c), byte(c>>8))
// ADDI $autosize, SP, SP — close the frame (C.ADDI when it fits). // ADDI $autosize, SP, SP, close the frame (C.ADDI when it fits).
out = append(out, riscvSPAdjust(int32(fi.autosize))...) if fits12(int32(fi.autosize)) {
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
} else {
out = append(out, riscvAddToSP(int32(fi.autosize))...)
}
} }
// JALR X0, 0(X1). // JALR X0, 0(X1).
return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...) return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...)
@@ -143,7 +244,7 @@ func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
} }
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to // riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final JALR — the point where SP is restored. // (but not including) the final JALR, the point where SP is restored.
func riscvReturnEpilogueLen(fi riscvFrameInfo) int { func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
if fi.autosize == 0 { if fi.autosize == 0 {
return 0 return 0
@@ -180,3 +281,74 @@ func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32
} }
return -1, 0 return -1, 0
} }
// riscvGuardLen returns the byte length of the stack-split guard prefix
// including the inline morestack call (zero when the function needs no
// guard). Unlike amd64 and arm64, the toolchain places the morestack call
// between the guard and the body: the guard branches forward over it.
func riscvGuardLen(fi riscvFrameInfo) int {
_, reloc := riscvGuard(fi)
_ = reloc
return len(riscvGuardBytes(fi))
}
// riscvGuard emits the stack-split guard prefix with the inline morestack
// call: the branch skips forward over JAL X5 and JAL X0 straight into the
// body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are
// relative to the guard itself, which sits at function offset 0.
func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
if !fi.needSplit {
return nil, Reloc{}
}
// MOV 16(g), X6 (g.stackguard0), g = X27.
out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16))
jalBack := func() []byte {
// JAL X0 back to the function start: it sits right after the JAL X5,
// so its displacement is minus the current offset.
return wordLE(riscvJType(0, int32(-len(out))))
}
var reloc Reloc
switch fi.splitClass {
case 0:
// BLTU X6, SP, done (+8: over the CALL and the JMP back)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
case 1:
// ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+8)
off := int32(fi.autosize - stackSmall)
out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
default:
// MOV $(framesize-StackSmall), X7; BLTU SP, X7, call;
// ADD $-(framesize-StackSmall), SP, X7; BLTU X6, X7, call
off := int32(fi.autosize - stackSmall)
mov := encodeRISCVLoadImm(7, off)
out = append(out, mov...)
addiLen := riscvItypeImmediateSize("ADDI", -off)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...)
addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off)
if err != nil {
addi = nil
}
out = append(out, addi...)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
}
return out, reloc
}
// riscvGuardBytes emits the guard prefix bytes alone (sizing helper).
func riscvGuardBytes(fi riscvFrameInfo) []byte {
g, _ := riscvGuard(fi)
return g
}
+3 -3
View File
@@ -141,7 +141,7 @@ DATA answer<>+0(SB)/8, $42
first := int(le.Uint32(relocIdx[4*(4+4):])) first := int(le.Uint32(relocIdx[4*(4+4):]))
wantType := []uint16{relocRISCVPcrelItype, relocRISCVPcrelItype, relocRISCVPcrelStype} wantType := []uint16{relocRISCVPcrelItype, relocRISCVPcrelItype, relocRISCVPcrelStype}
wantOffAbs := []int{0, 8, 16} wantOffAbs := []int{0, 8, 16}
for i := 0; i < 3; i++ { for i := range 3 {
e := relocs[(first+i)*23:] e := relocs[(first+i)*23:]
if int32(le.Uint32(e[0:])) != int32(wantOffAbs[i]) || e[4] != 8 || le.Uint16(e[5:]) != wantType[i] || if int32(le.Uint32(e[0:])) != int32(wantOffAbs[i]) || e[4] != 8 || le.Uint16(e[5:]) != wantType[i] ||
le.Uint32(e[15:]) != pkgIdxSelf || le.Uint32(e[19:]) != 0 { le.Uint32(e[15:]) != pkgIdxSelf || le.Uint32(e[19:]) != 0 {
@@ -221,7 +221,7 @@ func main() {
t.Fatalf("baseline build: %v\n%s", err, buildLog) t.Fatalf("baseline build: %v\n%s", err, buildLog)
} }
var pkgArch, work, linkLine, asmObj string var pkgArch, work, linkLine, asmObj string
for _, line := range strings.Split(string(buildLog), "\n") { for line := range strings.SplitSeq(string(buildLog), "\n") {
switch { switch {
case strings.HasPrefix(line, "WORK="): case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=") work = strings.TrimPrefix(line, "WORK=")
@@ -279,7 +279,7 @@ func main() {
newArch := filepath.Join(dir, "pkg.a") newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch} args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{} seen := map[string]bool{}
for _, m := range strings.Fields(string(listOut)) { for m := range strings.FieldsSeq(string(listOut)) {
if seen[m] { if seen[m] {
continue continue
} }
+45 -43
View File
@@ -36,11 +36,11 @@ const (
vexNDS3Imm vexNDS3Imm
// vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg = // vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg =
// ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM // ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM
// source lives in the reg field, the destination in r/m — the PEXTR-style // source lives in the reg field, the destination in r/m, the PEXTR-style
// layout. VEXTRACTI128 and VEXTRACTF128 use this shape. // layout. VEXTRACTI128 and VEXTRACTF128 use this shape.
vexExtract vexExtract
// vexRMRev is the reversed two-operand form `OP src, dst` with the source // vexRMRev is the reversed two-operand form `OP src, dst` with the source
// in ModRM.reg and the destination in r/m — the layout of the EVEX // in ModRM.reg and the destination in r/m, the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD). // narrowing stores (VPMOVDW, VPMOVQD).
vexRMRev vexRMRev
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose // vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
@@ -68,7 +68,7 @@ type vexSpec struct {
// incrementally; every entry is covered by a byte-for-byte ground-truth test // incrementally; every entry is covered by a byte-for-byte ground-truth test
// against the Go assembler. // against the Go assembler.
var vexTable = map[string]vexSpec{ var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG — integer arithmetic / logic / compare. // VEX.128/256.66.0F.WIG, integer arithmetic / logic / compare.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3}, "VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3},
"VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3}, "VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3}, "VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3},
@@ -82,7 +82,7 @@ var vexTable = map[string]vexSpec{
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3}, "VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3},
"VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3}, "VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3},
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3}, "VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3},
// VEX.256.66.0F38.W0 — dword permute (three-operand NDS form). // VEX.256.66.0F38.W0, dword permute (three-operand NDS form).
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3}, "VPERMD": {2, 0x36, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG. // VEX.128/256.66.0F38.WIG.
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3}, "VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3},
@@ -90,14 +90,14 @@ var vexTable = map[string]vexSpec{
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3}, "VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3},
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3}, "VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic. // VEX.128/256.66.0F.WIG, packed double-precision arithmetic / logic.
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3}, "VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3}, "VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3}, "VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3}, "VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3}, "VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3}, "VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG — packed single-precision arithmetic. // VEX.128/256.0F.WIG, packed single-precision arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3}, "VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3}, "VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3}, "VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
@@ -107,7 +107,7 @@ var vexTable = map[string]vexSpec{
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3}, "VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3}, "VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3}, "VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
// VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed // VEX.128.F2.0F.WIG, scalar double-precision arithmetic (the packed
// opcodes with an F2 pp). // opcodes with an F2 pp).
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3}, "VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3}, "VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
@@ -115,7 +115,7 @@ var vexTable = map[string]vexSpec{
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3}, "VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3}, "VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3}, "VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
// VEX.128.F3.0F.WIG — scalar single-precision arithmetic (the packed // VEX.128.F3.0F.WIG, scalar single-precision arithmetic (the packed
// opcodes with an F3 pp). // opcodes with an F3 pp).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3}, "VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3}, "VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
@@ -123,10 +123,10 @@ var vexTable = map[string]vexSpec{
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3}, "VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3}, "VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3}, "VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
// VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form). // VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3}, "VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG — sign/zero extend and broadcast (reg=dst, rm=src, // VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
// no vvvv). // no vvvv).
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM}, "VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM}, "VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
@@ -141,70 +141,72 @@ var vexTable = map[string]vexSpec{
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM}, "VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM}, "VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM}, "VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion "VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM},
"VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM},
// VEX.128/256.F3.0F.WIG, signed dword to packed double conversion
// (reg=dst, rm=src, no vvvv; the length follows the destination). // (reg=dst, rm=src, no vvvv; the length follows the destination).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM}, "VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
// VEX.128/256.0F.WIG — signed dword to packed single conversion // VEX.128/256.0F.WIG, signed dword to packed single conversion
// (reg=dst, rm=src, no vvvv, no mandatory prefix). // (reg=dst, rm=src, no vvvv, no mandatory prefix).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM}, "VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
// VEX.128/256.0F.WIG — packed single to packed double conversion // VEX.128/256.0F.WIG, packed single to packed double conversion
// (reg=dst, rm=src; the destination is the wide operand and sets the // (reg=dst, rm=src; the destination is the wide operand and sets the
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but // length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
// the Go assembler emits the instruction with pp = 00, and gasm follows // the Go assembler emits the instruction with pp = 00, and gasm follows
// the Go assembler's bytes — its machine code is the oracle, not the // the Go assembler's bytes, its machine code is the oracle, not the
// manual. // manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM}, "VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
// VEX.128.F2.0F.WIG — duplicate the low double of each 128-bit lane // VEX.128.F2.0F.WIG, duplicate the low double of each 128-bit lane
// (reg=dst, rm=src, no vvvv; the length follows the destination). // (reg=dst, rm=src, no vvvv; the length follows the destination).
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM}, "VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
// VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src). // VEX.128/256.66.0F.WIG, move mask to a GPR (reg=gpr dst, rm=vec src).
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM}, "VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD) "VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
// VEX.128/256.66.0F.WIG — immediate shifts (opdigit selects the shift). // VEX.128/256.66.0F.WIG, immediate shifts (opdigit selects the shift).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm}, "VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm},
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm}, "VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm},
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm}, "VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm},
"VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm}, "VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm},
"VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm}, "VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm},
// VEX.128/256.66.0F.WIG — immediate shuffle (reg=dst, rm=src, imm8). // VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM}, "VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
// VEX.256.66.0F3A.W1 — qword permute (reg=dst, rm=src, imm8). // VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8).
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM}, "VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
// VEX.128/256.66.0F.WIG — two-source shuffle (reg=dst, vvvv=src1, rm=src2, // VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// imm8). // imm8).
"VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm}, "VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm},
// VEX.256.66.0F3A.W0 — permute / insert (same shape; VINSERTI128's rm is // VEX.256.66.0F3A.W0, permute / insert (same shape; VINSERTI128's rm is
// the XMM or memory source). // the XMM or memory source).
"VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm}, "VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm},
"VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm}, "VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm},
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8). // VEX.256.66.0F3A.W0, lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract}, "VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract}, "VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
// VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst: // VEX.128/256.66.0F3A.W0, half-precision convert back ($imm, src, dst:
// reg=src, rm=XMM/memory dst, imm8 — the extract layout). // reg=src, rm=XMM/memory dst, imm8, the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract}, "VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
// VEX.128.0F.W0 — no operands. // VEX.128.0F.W0, no operands.
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero}, "VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
// VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src). // VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM}, "KTESTW": {1, 0x99, 0, 0, -1, vexRM},
// VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst, // VEX.66.0F38.W0, broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; SD is 256-bit only). // rm=scalar memory; SD is 256-bit only).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM}, "VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM}, "VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
// VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width // VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
// source). // source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM}, "VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src). // VEX.F3.0F.WIG, replicate even/odd singles (reg=dst, rm=src).
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM}, "VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM}, "VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y // VEX.66.0F.WIG, packed double to packed single conversion, the X/Y
// spellings: the destination is always XMM and the spelling fixes the // spellings: the destination is always XMM and the spelling fixes the
// source length (X = 128, Y = 256). // source length (X = 128, Y = 256).
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen}, "VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
@@ -228,14 +230,14 @@ var vexTable = map[string]vexSpec{
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3}, "VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3}, "VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift). // VEX.128/256.66.0F.WIG, word shifts (opdigit selects the shift).
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm}, "VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm}, "VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm}, "VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
// VEX.F2.0F — packed double to packed dword conversions, truncating and // VEX.F2.0F, packed double to packed dword conversions, truncating and
// non-truncating. The destination is always XMM; the X/Y spellings fix // non-truncating. The destination is always XMM; the X/Y spellings fix
// the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen. // the source length (XMM/YMM), and VEX.L follows it, see vexSrcLen.
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen}, "VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen}, "VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen}, "VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
@@ -255,7 +257,7 @@ var vexSrcLen = map[string]int{
"VCVTPD2PSY": 1, "VCVTPD2PSY": 1,
} }
// vexVarShift maps the shift mnemonics to their variable-count opcode — the // vexVarShift maps the shift mnemonics to their variable-count opcode, the
// form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8), // form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8),
// an ordinary NDS encoding rather than the /digit immediate form above. // an ordinary NDS encoding rather than the /digit immediate form above.
var vexVarShift = map[string]byte{ var vexVarShift = map[string]byte{
@@ -286,20 +288,20 @@ type vexMoveSpec struct {
// vexMoveTable maps an upper-case move mnemonic to its encoding. // vexMoveTable maps an upper-case move mnemonic to its encoding.
var vexMoveTable = map[string]vexMoveSpec{ var vexMoveTable = map[string]vexMoveSpec{
// VEX.128/256.F3.0F.WIG — unaligned integer move. // VEX.128/256.F3.0F.WIG, unaligned integer move.
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false}, "VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG — unaligned packed double move. // VEX.128/256.66.0F.WIG, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false}, "VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
// VEX.128.66.0F.W0 — 32-bit GPR/memory ↔ XMM. // VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
"VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true}, "VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true},
// VMOVQ — 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm). // VMOVQ, 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true}, "VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true},
// VEX.128.F2.0F.WIG — scalar double move, memory operands only (the // VEX.128.F2.0F.WIG, scalar double move, memory operands only (the
// register form takes three operands and is not supported yet). // register form takes three operands and is not supported yet).
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true}, "VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128.F3.0F.WIG — scalar single move, memory operands only. // VEX.128.F3.0F.WIG, scalar single move, memory operands only.
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true}, "VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128/256 — aligned packed moves. // VEX.128/256, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false}, "VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false}, "VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
} }
@@ -315,7 +317,7 @@ func isVex(mnemUpper string) bool {
// encodeVex encodes a VEX instruction with operands in Plan 9 order. // encodeVex encodes a VEX instruction with operands in Plan 9 order.
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error { func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
// Vector register indices 16–31 exist only in EVEX encodings; fail // Vector register indices 16-31 exist only in EVEX encodings; fail
// loudly rather than silently truncating the index. // loudly rather than silently truncating the index.
for _, op := range ops { for _, op := range ops {
if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 { if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 {
@@ -418,7 +420,7 @@ func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
} }
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with // encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the VEX.L bit following the source — fixed // the destination always XMM and the VEX.L bit following the source, fixed
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when // by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
// the source is memory. // the source is memory.
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error { func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
+2 -4
View File
@@ -17,7 +17,7 @@ type File struct {
Orphans []Stmt // labels/instructions seen before any TEXT directive Orphans []Stmt // labels/instructions seen before any TEXT directive
// Macros holds the names introduced by #define directives in this file. // Macros holds the names introduced by #define directives in this file.
// The linter uses it to avoid flagging macro invocations as unknown // The linter uses it to avoid flagging macro invocations as unknown
// instructions (macro expansion itself is out of scope — see the docs). // instructions (macro expansion itself is out of scope, see the docs).
Macros map[string]bool Macros map[string]bool
} }
@@ -123,10 +123,8 @@ type Symbol struct {
// OpKind classifies an operand syntactically. // OpKind classifies an operand syntactically.
type OpKind int type OpKind int
// Operand kinds.
const ( const (
OpInvalid OpKind = iota OpImmediate = iota // $value
OpImmediate // $value
OpAddr // register, memory reference, symbol or label OpAddr // register, memory reference, symbol or label
) )
+2 -2
View File
@@ -54,11 +54,11 @@ func TestStmtPositions(t *testing.T) {
// TestInterfaces confirms the node types satisfy their interfaces, so callers // TestInterfaces confirms the node types satisfy their interfaces, so callers
// can range over Decls and Stmts. // can range over Decls and Stmts.
func TestInterfaces(t *testing.T) { func TestInterfaces(t *testing.T) {
var decls []Decl = []Decl{&Include{}, &Preproc{}, &Text{}, &Globl{}, &Data{}} var decls = []Decl{&Include{}, &Preproc{}, &Text{}, &Globl{}, &Data{}}
if len(decls) != 5 { if len(decls) != 5 {
t.Fatal("decl interface set") t.Fatal("decl interface set")
} }
var stmts []Stmt = []Stmt{&Label{}, &Instr{}} var stmts = []Stmt{&Label{}, &Instr{}}
if len(stmts) != 2 { if len(stmts) != 2 {
t.Fatal("stmt interface set") t.Fatal("stmt interface set")
} }
+307
View File
@@ -0,0 +1,307 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"fmt"
"os"
"os/exec"
"path/filepath"
"regexp"
"runtime"
"slices"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// cmdAuditInstructions cross-checks a gasm encoder against the Go toolchain's
// own assembler, probed black-box: every mnemonic in the gasm table is offered
// to go tool asm in its bare form, and a mnemonic counts as known to Go when
// the error is anything but "unrecognized instruction" (a wrong-shape error
// still proves the mnemonic exists in Go's tables). The audit answers three
// questions at a glance:
//
// - which mnemonics gasm can encode that go tool asm does not know
// (superset encodings, usable only through the gasm goobj path);
// - which mnemonics the architecture table knows but the encoder cannot
// emit yet (the implementation backlog);
// - which mnemonics go tool asm knows that gasm cannot encode (feature
// gaps).
//
// The amd64 derived families (Jcc, CMOVcc, SETcc) exist on both sides by
// construction and are excluded from the diff; the other architectures list
// their conditional branches outright.
func cmdAuditInstructions(args []string) error {
fs := newCommand("audit-instructions", "gasm audit-instructions [amd64|arm64|riscv64|loong64]", `
Compare the gasm encoder for the given architecture (default amd64) against
go tool asm and print the diff: superset encodings (gasm-only, shippable via
gasm asm --format goobj), known-but-unencodable names (the backlog) and go-
only names (feature gaps). The Go side is probed black-box with a battery
of bare mnemonics, so the audit tracks whatever toolchain `+"`go env GOROOT`"+`
provides.
`)
if err := fs.Parse(args); err != nil {
return err
}
archName := "amd64"
switch n := len(fs.Args()); {
case n > 1:
return fmt.Errorf("audit-instructions takes at most one architecture argument")
case n == 1:
archName = strings.ToLower(fs.Arg(0))
}
a, err := auditArch(archName)
if err != nil {
return err
}
tab := arch.ForArch(a)
var names []string
seen := map[string]bool{}
for _, in := range tab.Instructions() {
name := strings.ToUpper(in.Name)
if a == arch.AMD64 && derivedFamily(name) || seen[name] {
continue
}
seen[name] = true
names = append(names, name)
}
goKnown, err := probeGoAsm(goarchName(a), names)
if err != nil {
return err
}
var superset, backlog, shared []string
for _, name := range names {
switch {
case !gasmEncodable(a, name):
backlog = append(backlog, name)
case !goKnown[name]:
superset = append(superset, name)
default:
shared = append(shared, name)
}
}
// GO-ONLY is not enumerable by probing: Go's table is only visible
// through names we already know, so nothing can be reported there.
slices.Sort(superset)
slices.Sort(backlog)
slices.Sort(shared)
w := os.Stdout
fmt.Fprintf(w, "gasm table (%s, families excluded): %d mnemonics\n", archName, len(names))
fmt.Fprintf(w, "gasm encodable: %d go tool asm recognized: %d\n", len(shared)+len(superset), countTrue(goKnown))
fmt.Fprintf(w, "shared: %d\n", len(shared))
fmt.Fprintf(w, "\nSuperset encodings (gasm-only; ship via gasm asm --format goobj):\n")
for _, n := range superset {
fmt.Fprintf(w, " %s\n", n)
}
fmt.Fprintf(w, "\nKnown but not encodable (backlog):\n")
for _, n := range backlog {
fmt.Fprintf(w, " %s\n", n)
}
fmt.Fprintf(w, "\nGo-only names cannot be enumerated by probing; extend the gasm\n")
fmt.Fprintf(w, "table from the Go release notes when a new instruction family ships.\n")
return nil
}
// auditArch resolves the audit's architecture argument.
func auditArch(name string) (arch.Arch, error) {
switch strings.ToLower(name) {
case "amd64":
return arch.AMD64, nil
case "arm64":
return arch.ARM64, nil
case "riscv64", "riscv":
return arch.RISCV, nil
case "loong64", "loong":
return arch.LOONG64, nil
}
return arch.Unknown, fmt.Errorf("unknown architecture %q: want amd64, arm64, riscv64 or loong64", name)
}
// goarchName maps an arch identifier onto its GOARCH spelling.
func goarchName(a arch.Arch) string {
switch a {
case arch.ARM64:
return "arm64"
case arch.RISCV:
return "riscv64"
case arch.LOONG64:
return "loong64"
}
return "amd64"
}
func countTrue(m map[string]bool) int {
n := 0
for _, v := range m {
if v {
n++
}
}
return n
}
// derivedFamily reports whether a mnemonic belongs to a family both
// assemblers construct from condition codes rather than list exhaustively
// (JEQ/CMOVLGT/SETNE and friends). Such names never probe cleanly, so
// including them in the diff would be noise. amd64 only: the other
// architectures list their conditional branches outright.
func derivedFamily(name string) bool {
if strings.HasPrefix(name, "J") && name != "JMP" && name != "JMPQ" {
return true
}
if strings.HasPrefix(name, "CMOV") || strings.HasPrefix(name, "SET") {
return true
}
return false
}
var unrecognizedRe = regexp.MustCompile(`unrecognized instruction`)
// probeGoAsm feeds every mnemonic to go tool asm in one generated file and
// classifies the diagnostics. "Unrecognized instruction" is a parse-stage
// verdict on the mnemonic alone, so a single bare-instruction probe per
// mnemonic decides recognition; the combined file still reports every line's
// error even when others fail.
func probeGoAsm(goarch string, names []string) (map[string]bool, error) {
dir, err := os.MkdirTemp("", "gasm-audit")
if err != nil {
return nil, err
}
defer os.RemoveAll(dir)
var sb strings.Builder
sb.WriteString("TEXT ·probe(SB), 4, $0\n\tRET\n")
lineMnemonic := map[int]string{}
line := 3
for _, name := range names {
fmt.Fprintf(&sb, "TEXT ·p%s%d(SB), 4, $0\n", sanitize(name), line)
sb.WriteString("\t" + name + "\n\tRET\n")
lineMnemonic[line+1] = name // the instruction line, after TEXT
line += 3
}
probePath := filepath.Join(dir, "probe.s")
if err := os.WriteFile(probePath, []byte(sb.String()), 0o644); err != nil {
return nil, err
}
toolDir, err := exec.Command("go", "env", "GOTOOLDIR").Output()
if err != nil {
return nil, fmt.Errorf("go env GOTOOLDIR: %w", err)
}
asmBin := filepath.Join(strings.TrimSpace(string(toolDir)), "asm")
if _, err := os.Stat(asmBin); err != nil {
return nil, fmt.Errorf("go tool asm not found at %s", asmBin)
}
cmd := exec.Command(asmBin, "-p", "probe", "-o", filepath.Join(dir, "probe.o"), probePath)
cmd.Env = append(os.Environ(), "GOARCH="+goarch, "GOOS="+runtime.GOOS)
out, _ := cmd.CombinedOutput()
result := map[string]bool{}
for _, name := range names {
result[name] = true // no news = the name parsed fine
}
reParse := regexp.MustCompile(`probe\.s:(\d+):`)
for l := range strings.SplitSeq(string(out), "\n") {
m := reParse.FindStringSubmatch(l)
if m == nil {
continue
}
lineNo, err := strconv.Atoi(m[1])
if err != nil {
continue
}
if name, ok := lineMnemonic[lineNo]; ok && unrecognizedRe.MatchString(l) {
result[name] = false
}
}
return result, nil
}
// probeShapes lists representative operand shapes for the encodability
// probe. The assemblers report an unknown mnemonic and a known mnemonic
// with no supported form alike ("unsupported <arch> instruction"), so only
// a shape that assembles cleanly counts, and the backlog over-approximates:
// a name whose real forms the battery misses lands there. amd64 keeps its
// exact table-driven check.
func probeShapes(a arch.Arch) []string {
switch a {
case arch.ARM64:
return []string{
"X0, X1, X2", "X0, X1", "X0", "$1, X0", "X0, (X1)", "(X0), X1",
"X0, (X1, 8)", "(SP), X0", "F0, F1, F2", "F0, F1", "F0",
"V0.B16, V1.B16, V2.B16", "p2", "X0, p2", "X0, X1, p2",
// The conditional select family spells the condition first
// and takes R register spellings.
"EQ, R0, R1, R2", "EQ, R0, R1", "EQ, R0",
"GE, F0, F1, F2", "NE, F0, F1, $0",
}
case arch.RISCV:
return []string{
"X5, X6, X7", "X5, X6", "X5", "$1, X5", "X5, (X6)", "$1, X5, X6",
"(X5), X6", "F0, F1, F2", "F0, F1", "p2", "X1, p2", "X0, p2",
"X5, X6, p2", "p2(SB)",
}
case arch.LOONG64:
return []string{
"R4, R5, R6", "R4, R5", "R4", "$1, R4", "R4, (R5)", "(R4), R5",
"F0, F1, F2", "F0, F1", "p2", "R1, p2", "R4, p2",
"$1, R4, R5, R6", "$65536, R4", "R4, R5, p2", "p2(SB)",
}
}
return nil
}
// gasmEncodable reports whether the gasm encoder for a can emit the
// mnemonic, decided by trial assembly over the shape battery.
func gasmEncodable(a arch.Arch, name string) bool {
switch a {
case arch.ARM64, arch.RISCV, arch.LOONG64:
default:
return asm.Encodable(name)
}
for _, shape := range probeShapes(a) {
if gasmAssembles(a, name, shape) {
return true
}
}
return false
}
// gasmAssembles reports whether a one-instruction probe file containing name
// with the given operand shape assembles without error.
func gasmAssembles(a arch.Arch, name, shape string) bool {
src := "TEXT ·p(SB), NOSPLIT, $0\n\t" + name
if shape != "" {
src += " " + shape
}
src += "\n\tRET\np2:\n\tRET\n"
f, errs := parser.Parse("probe.s", src)
if len(errs) > 0 {
return false
}
var err error
switch a {
case arch.ARM64:
_, err = asm.AssembleFileARM64(f)
case arch.RISCV:
_, err = asm.AssembleFileRISCV(f)
case arch.LOONG64:
_, err = asm.AssembleFileLOONG64(f)
}
return err == nil
}
// sanitize makes a mnemonic safe for use in a Go symbol name.
func sanitize(name string) string {
return strings.NewReplacer(".", "_", "$", "_").Replace(name)
}
+66
View File
@@ -0,0 +1,66 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
)
func TestDerivedFamily(t *testing.T) {
for _, n := range []string{"JEQ", "JLT", "JCC", "CMOVLGT", "SETNE", "SETA"} {
if !derivedFamily(n) {
t.Errorf("derivedFamily(%q) = false, want true", n)
}
}
for _, n := range []string{"JMP", "ADDQ", "VPGATHERDD", "MOVBE", "PSHUFB"} {
if derivedFamily(n) {
t.Errorf("derivedFamily(%q) = true, want false", n)
}
}
}
func TestSanitize(t *testing.T) {
if got := sanitize("VPCMP.UB"); got != "VPCMP_UB" {
t.Errorf("sanitize: got %q", got)
}
}
func TestAuditArch(t *testing.T) {
for in, want := range map[string]arch.Arch{
"amd64": arch.AMD64, "arm64": arch.ARM64,
"riscv64": arch.RISCV, "riscv": arch.RISCV,
"loong64": arch.LOONG64, "LOONG": arch.LOONG64,
} {
got, err := auditArch(in)
if err != nil || got != want {
t.Errorf("auditArch(%q) = %v, %v; want %v", in, got, err, want)
}
}
if _, err := auditArch("mips"); err == nil {
t.Error("auditArch(mips) must fail")
}
}
func TestGasmEncodable(t *testing.T) {
cases := []struct {
a arch.Arch
yes string
no string
}{
{arch.AMD64, "ADDQ", "NOSUCHMNEMONIC"},
{arch.ARM64, "ADD", "NOSUCHMNEMONIC"},
{arch.RISCV, "ADD", "NOSUCHMNEMONIC"},
{arch.LOONG64, "ADDV", "NOSUCHMNEMONIC"},
}
for _, c := range cases {
if !gasmEncodable(c.a, c.yes) {
t.Errorf("%s: %s should be encodable", c.a, c.yes)
}
if gasmEncodable(c.a, c.no) {
t.Errorf("%s: %s should not be encodable", c.a, c.no)
}
}
}
+328
View File
@@ -0,0 +1,328 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
package main
import (
"fmt"
"io"
"os"
"sort"
"strings"
"time"
"sourcedock.dev/petrbalvin/gasm-devkit/debug"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
)
func cmdDebug(args []string) int {
fs := newCommand("debug", "gasm debug <file.s> --func <name>", `
Interactive debugger for JIT-assembled functions. Launches the
function in a traced subprocess (ptrace), then provides a REPL for
single-stepping, breakpoints, register and memory inspection.
REPL commands:
break <label|addr> [if <reg> <op> <val>]
set a breakpoint, optionally conditional on a
register comparison (reg-reg or reg-immediate)
delete <label|addr> remove a breakpoint
info break list all breakpoints
step [n], s single-step n instructions (default 1)
next, n step over a CALL
finish, fin run until the function returns
continue, c run until a breakpoint, watchpoint or exit
disas [n], u disassemble n instructions at PC
regs print general-purpose and vector registers
where show source line and nearest label at PC
stack show stack near RSP (return address + ABI0 args)
bt, backtrace backtrace (current frame + return address)
x [addr] [len] hex-dump memory (default: current PC, 64 bytes)
w <addr> <val...> write bytes to memory
set <reg> <value> set a register
watch <addr> [r|w] [size]
set a hardware watchpoint (write by default)
unwatch [<slot>] clear one watchpoint, or all without an argument
labels, l list function labels and offsets
help, h, ? show command help
quit, q kill the debuggee and exit
`)
funcName := fs.String("func", "", "function to debug")
argsFile := fs.String("args", "", "file containing the ABI0 argument block")
bufSpec := fs.String("buf", "", "buffer specification: name:size:pattern[,name:size:pattern...] where pattern is zero, ones, seq, or hex")
script := fs.String("script", "", "run REPL commands from a file (one per line) and exit; '-' reads stdin")
cover := fs.Bool("cover", false, "run to completion with a breakpoint on every instruction and report which executed and how often")
timeout := fs.Duration("timeout", 0, "kill the debuggee after this duration (e.g. 30s); for headless --script runs")
fs.Parse(args)
// --- Debuggee mode (internal, spawned by the debugger) ---
if os.Getenv("GASM_DEBUG_TARGET") != "" {
tmpDir := os.Getenv("GASM_DEBUG_TMP")
if tmpDir == "" || fs.NArg() < 1 || *funcName == "" || *argsFile == "" {
fmt.Fprintln(os.Stderr, "gasm debug: internal debuggee mode")
return 2
}
if err := debug.RunTarget(fs.Arg(0), *funcName, *argsFile, tmpDir); err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
return 0
}
// --- Debugger mode (interactive REPL) ---
if fs.NArg() < 1 || *funcName == "" {
fmt.Fprintln(os.Stderr, "usage: gasm debug <file.s> --func <name>")
return 2
}
path := fs.Arg(0)
// The watchdog is armed before anything can block: ptrace attach and a
// continued kernel loop both hang the run when the environment forbids
// tracing or the kernel loops forever, and neither is interruptible from
// the inside.
if *timeout > 0 {
go func() {
time.Sleep(*timeout)
fmt.Fprintf(os.Stderr, "gasm debug: timeout (%s), killing the debuggee\n", *timeout)
os.Exit(3)
}()
}
// Load the kernel to extract function metadata and labels.
k, err := verify.Load(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
defer k.Close()
fl, err := k.Func(*funcName)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
// Build the label list for the REPL.
var labels []debug.Label
for name, off := range fl.Labels {
labels = append(labels, debug.Label{Name: name, Offset: off})
}
sort.Slice(labels, func(i, j int) bool { return labels[i].Offset < labels[j].Offset })
// Launch the debuggee with the argument block.
var argBlock []byte
var bufAddrs []uint64
var sess *debug.Session
if *bufSpec != "" {
// Parse the function signature to determine argument layout.
src, err := readSource(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
sig, ok := verify.ExtractFuncSig(src, *funcName)
if !ok {
fmt.Fprintf(os.Stderr, "gasm debug: no // func signature found for %s\n", *funcName)
return 1
}
layout := verify.ArgLayout(sig)
// Parse the buffer spec to get buffer names.
bufNames := parseBufNames(*bufSpec)
// Allocate buffers in the debuggee.
argBlock = make([]byte, fl.Args)
sess, bufAddrs, err = debug.LaunchWithBuffers("", path, *funcName, argBlock, *bufSpec)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
// Construct the argument block with buffer pointers at the correct positions.
for _, arg := range layout {
if !arg.IsPtr {
continue
}
// Find the buffer that matches this argument.
for i, name := range bufNames {
if i < len(bufAddrs) && (name == arg.Name || strings.HasPrefix(arg.Name, name)) {
addr := bufAddrs[i]
off := arg.Offset
if off+8 <= len(argBlock) {
argBlock[off] = byte(addr)
argBlock[off+1] = byte(addr >> 8)
argBlock[off+2] = byte(addr >> 16)
argBlock[off+3] = byte(addr >> 24)
argBlock[off+4] = byte(addr >> 32)
argBlock[off+5] = byte(addr >> 40)
argBlock[off+6] = byte(addr >> 48)
argBlock[off+7] = byte(addr >> 56)
}
// For slices, also set the length and capacity.
if strings.HasPrefix(arg.Typ, "[]") && off+24 <= len(argBlock) {
// Find the buffer size from the spec.
size := parseBufSize(*bufSpec, name)
// Length at offset+8, capacity at offset+16.
for j := range 8 {
argBlock[off+8+j] = byte(size >> (j * 8))
argBlock[off+16+j] = byte(size >> (j * 8))
}
}
break
}
}
}
} else {
argBlock = make([]byte, fl.Args)
sess, err = debug.Launch("", path, *funcName, argBlock)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
}
defer sess.Kill()
bm := debug.NewBreakpoints(sess)
fmt.Printf("gasm debug: %s in %s (pid %d)\n", *funcName, path, sess.Pid())
// Convert the line table for the command loop.
var srcLines []debug.SourceLine
for _, le := range fl.Lines {
srcLines = append(srcLines, debug.SourceLine{Offset: le.Offset, Line: le.Line})
}
// Coverage mode: pre-register a breakpoint on every instruction (walked
// by length through the function body while the debuggee is stopped) and
// let the kernel run to completion. Each trap counts a hit for that
// instruction, so the final report shows exactly which instructions
// executed and how often, with the label-level view derived from it.
// Expect the run to slow to ptrace speed: one trap per executed
// instruction.
if *cover {
base := sess.CodeBase() + uint64(fl.Offset)
type coverInstr struct {
off uint64
text string
}
var instrs []coverInstr
for off := uint64(0); off < uint64(fl.Size); {
text, ln, err := sess.Disassemble(base + off)
if err != nil || ln == 0 {
break
}
instrs = append(instrs, coverInstr{off: off, text: text})
off += uint64(ln)
}
for _, in := range instrs {
if _, err := bm.SetWithCond(base+in.off, fmt.Sprintf("func+%#x", in.off), nil); err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: cover: %v\n", err)
return 1
}
}
fmt.Printf("gasm debug: coverage run over %d instructions\n", len(instrs))
for {
for _, bp := range bm.All() {
bm.Reinsert(bp.Addr)
}
if err := sess.Continue(); err != nil {
break // debuggee finished or died
}
if sess.Exited() {
break
}
regs, rerr := sess.GetRegs()
if rerr != nil {
break
}
// HandleTrap restores the original byte, rewinds PC and counts
// the hit on the breakpoint itself. Single-step over the
// restored instruction so the reinsertion at the top of the
// loop cannot re-trap on the same breakpoint.
if bp := bm.HandleTrap(&regs); bp != nil {
if err := sess.Step(); err != nil {
break
}
}
}
hits := map[uint64]int{}
traps := 0
for _, bp := range bm.All() {
if n := bp.Hits(); n > 0 {
hits[bp.Addr-base] = n
traps += n
}
}
var hit []string
var missed []string
for _, l := range labels {
if hits[uint64(l.Offset)] > 0 {
hit = append(hit, l.Name)
} else {
missed = append(missed, l.Name)
}
}
sort.Strings(hit)
sort.Strings(missed)
fmt.Printf("coverage: %d/%d instructions executed (%d traps)\n", len(hits), len(instrs), traps)
fmt.Printf("coverage: %d/%d labels reached\n", len(hit), len(labels))
for _, l := range hit {
fmt.Printf(" covered %s\n", l)
}
for _, l := range missed {
fmt.Printf(" MISSED %s\n", l)
}
fmt.Println("executed instructions:")
for _, in := range instrs {
if n := hits[in.off]; n > 0 {
fmt.Printf(" func+%#04x %4dx %s\n", in.off, n, in.text)
}
}
return 0
}
// Headless mode: run the script through the normal command loop and
// exit. The watchdog armed above covers launch, continue and step.
var in io.Reader = os.Stdin
if *script != "" {
if *script == "-" {
in = os.Stdin
} else {
f, err := os.Open(*script)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
defer f.Close()
in = f
}
}
debug.REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, labels, srcLines, in)
return 0
}
// parseBufNames extracts buffer names from a buffer specification.
// Format: name:size:pattern[,name:size:pattern...]
func parseBufNames(spec string) []string {
var names []string
for part := range strings.SplitSeq(spec, ",") {
fields := strings.SplitN(part, ":", 3)
if len(fields) >= 1 && fields[0] != "" {
names = append(names, fields[0])
}
}
return names
}
// parseBufSize extracts the size of a named buffer from a buffer specification.
func parseBufSize(spec, name string) int {
for part := range strings.SplitSeq(spec, ",") {
fields := strings.SplitN(part, ":", 3)
if len(fields) >= 2 && fields[0] == name {
var size int
fmt.Sscanf(fields[1], "%d", &size)
return size
}
}
return 0
}
-193
View File
@@ -1,193 +0,0 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package main
import (
"fmt"
"os"
"sort"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/debug"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
)
func cmdDebug(args []string) int {
fs := newCommand("debug", "gasm debug <file.s> --func <name>", `
Interactive debugger for JIT-assembled amd64 functions. Launches the
function in a traced subprocess (ptrace), then provides a REPL for
single-stepping, breakpoints, register and memory inspection.
REPL commands:
break <label|addr> set a breakpoint at a label or absolute address
step [n] single-step n instructions (default 1)
continue run until next breakpoint or exit
regs print general-purpose registers
x [addr] [len] hex-dump memory (default: current PC, 64 bytes)
labels list function labels and offsets
quit kill the debuggee and exit
`)
target := fs.Bool("target", false, "") // hidden: debuggee subprocess mode
funcName := fs.String("func", "", "function to debug")
argsFile := fs.String("args", "", "file containing the ABI0 argument block")
bufSpec := fs.String("buf", "", "buffer specification: name:size:pattern[,name:size:pattern...] where pattern is zero, ones, seq, or hex")
fs.Parse(args)
// --- Debuggee mode (internal, spawned by the debugger) ---
if *target {
tmpDir := os.Getenv("GASM_DEBUG_TMP")
if tmpDir == "" || fs.NArg() < 1 || *funcName == "" || *argsFile == "" {
fmt.Fprintln(os.Stderr, "gasm debug --target: internal mode")
return 2
}
if err := debug.RunTarget(fs.Arg(0), *funcName, *argsFile, tmpDir); err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
return 0
}
// --- Debugger mode (interactive REPL) ---
if fs.NArg() < 1 || *funcName == "" {
fmt.Fprintln(os.Stderr, "usage: gasm debug <file.s> --func <name>")
return 2
}
path := fs.Arg(0)
// Load the kernel to extract function metadata and labels.
k, err := verify.Load(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
defer k.Close()
fl, err := k.Func(*funcName)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
// Build the label list for the REPL.
var labels []debug.Label
for name, off := range fl.Labels {
labels = append(labels, debug.Label{Name: name, Offset: off})
}
sort.Slice(labels, func(i, j int) bool { return labels[i].Offset < labels[j].Offset })
// Launch the debuggee with the argument block.
var argBlock []byte
var bufAddrs []uint64
var sess *debug.Session
if *bufSpec != "" {
// Parse the function signature to determine argument layout.
src, err := readSource(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
sig, ok := verify.ExtractFuncSig(src, *funcName)
if !ok {
fmt.Fprintf(os.Stderr, "gasm debug: no // func signature found for %s\n", *funcName)
return 1
}
layout := verify.ArgLayout(sig)
// Parse the buffer spec to get buffer names.
bufNames := parseBufNames(*bufSpec)
// Allocate buffers in the debuggee.
argBlock = make([]byte, fl.Args)
sess, bufAddrs, err = debug.LaunchWithBuffers("", path, *funcName, argBlock, *bufSpec)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
// Construct the argument block with buffer pointers at the correct positions.
bufIdx := 0
for _, arg := range layout {
if !arg.IsPtr {
continue
}
// Find the buffer that matches this argument.
for i, name := range bufNames {
if i < len(bufAddrs) && (name == arg.Name || strings.HasPrefix(arg.Name, name)) {
addr := bufAddrs[i]
off := arg.Offset
if off+8 <= len(argBlock) {
argBlock[off] = byte(addr)
argBlock[off+1] = byte(addr >> 8)
argBlock[off+2] = byte(addr >> 16)
argBlock[off+3] = byte(addr >> 24)
argBlock[off+4] = byte(addr >> 32)
argBlock[off+5] = byte(addr >> 40)
argBlock[off+6] = byte(addr >> 48)
argBlock[off+7] = byte(addr >> 56)
}
// For slices, also set the length and capacity.
if strings.HasPrefix(arg.Typ, "[]") && off+24 <= len(argBlock) {
// Find the buffer size from the spec.
size := parseBufSize(*bufSpec, name)
// Length at offset+8, capacity at offset+16.
for j := 0; j < 8; j++ {
argBlock[off+8+j] = byte(size >> (j * 8))
argBlock[off+16+j] = byte(size >> (j * 8))
}
}
bufIdx++
break
}
}
}
_ = bufIdx
} else {
argBlock = make([]byte, fl.Args)
sess, err = debug.Launch("", path, *funcName, argBlock)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
}
defer sess.Kill()
bm := debug.NewBreakpoints(sess)
fmt.Printf("gasm debug: %s in %s (pid %d)\n", *funcName, path, sess.Pid())
// Convert the line table for the REPL.
var srcLines []debug.SourceLine
for _, le := range fl.Lines {
srcLines = append(srcLines, debug.SourceLine{Offset: le.Offset, Line: le.Line})
}
debug.REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, labels, srcLines)
return 0
}
// parseBufNames extracts buffer names from a buffer specification.
// Format: name:size:pattern[,name:size:pattern...]
func parseBufNames(spec string) []string {
var names []string
for _, part := range strings.Split(spec, ",") {
fields := strings.SplitN(part, ":", 3)
if len(fields) >= 1 && fields[0] != "" {
names = append(names, fields[0])
}
}
return names
}
// parseBufSize extracts the size of a named buffer from a buffer specification.
func parseBufSize(spec, name string) int {
for _, part := range strings.Split(spec, ",") {
fields := strings.SplitN(part, ":", 3)
if len(fields) >= 2 && fields[0] == name {
var size int
fmt.Sscanf(fields[1], "%d", &size)
return size
}
}
return 0
}
+2 -2
View File
@@ -1,7 +1,7 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause // SPDX-License-Identifier: BSD-3-Clause
//go:build !(linux && amd64) //go:build !linux
package main package main
@@ -11,6 +11,6 @@ import (
) )
func cmdDebug(args []string) int { func cmdDebug(args []string) int {
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires linux/amd64 (ptrace)") fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux (ptrace)")
return 1 return 1
} }
+148
View File
@@ -0,0 +1,148 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"fmt"
"os"
"sort"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// cmdDis disassembles machine code: either a raw binary (standard input with
// "-") whose architecture is given with -a, or a .s file, which is assembled
// first so the listing shows the real function and label layout.
func cmdDis(args []string) int {
fs := newCommand("dis", "gasm dis [-a arch] <file>", `
Disassemble machine code to instruction text (via golang.org/x/arch).
With a .s file, the file is assembled first and the listing follows the
real layout: one block per TEXT function, local labels printed at their
offsets. The architecture comes from the file name suffix, or from -a.
With any other file, or "-" for standard input, the bytes are disassembled
linearly and -a selects the architecture (amd64, arm64, riscv64 or
loong64).
`)
archName := fs.String("a", "", "architecture for raw input: amd64, arm64, riscv64 or loong64")
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm dis [-a arch] <file>")
return 2
}
path := fs.Arg(0)
var target arch.Arch
if *archName != "" {
var err error
target, err = auditArch(*archName)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
return 2
}
}
if strings.HasSuffix(path, ".s") {
if target == arch.Unknown {
target = arch.FromFilename(path)
}
if target == arch.Unknown {
fmt.Fprintln(os.Stderr, "gasm dis: cannot infer the architecture from the file name; use -a")
return 2
}
return disSource(path, target)
}
if target == arch.Unknown {
fmt.Fprintln(os.Stderr, "gasm dis: raw input needs -a (amd64, arm64, riscv64 or loong64)")
return 2
}
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm dis:", err)
return 1
}
printListing(target, []byte(src), 0, nil)
return 0
}
// disSource assembles a .s file and prints one listing block per function.
func disSource(path string, target arch.Arch) int {
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm dis:", err)
return 1
}
f, errs := parser.Parse(path, src)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return 1
}
img, err := assembleFile(target, f)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
return 1
}
if len(img.Funcs) == 0 {
fmt.Fprintln(os.Stderr, "gasm dis: no assemblable TEXT functions found")
return 1
}
for _, fn := range img.Funcs {
code := img.Code[fn.Offset : fn.Offset+fn.Size]
fmt.Printf("%s: %d bytes\n", fn.Name, fn.Size)
labels := make(map[int][]string, len(fn.Labels))
for name, off := range fn.Labels {
labels[off] = append(labels[off], name)
}
for off := range labels {
sort.Strings(labels[off])
}
printListing(target, code, uint64(fn.Offset), labels)
}
if len(img.Data) > 0 {
fmt.Printf("data: %d bytes at 0x%x\n", len(img.Data), len(img.Code))
}
return 0
}
// printListing decodes code linearly from offset base, printing label lines
// (label name to offset within the block) as they are reached.
func printListing(a arch.Arch, code []byte, base uint64, labels map[int][]string) {
pc := 0
for pc < len(code) {
for _, name := range labels[pc] {
fmt.Printf("%s:\n", name)
}
ins, err := disasm.Decode(a, code[pc:], base+uint64(pc))
if err != nil {
break
}
end := min(pc+ins.Len, len(code))
fmt.Printf(" %04x: %-16s %s\n", base+uint64(pc), hexBytes(code[pc:end]), ins.Text)
if ins.Len <= 0 {
break
}
pc += ins.Len
}
}
// hexBytes renders up to 8 bytes as contiguous hex.
func hexBytes(b []byte) string {
var sb strings.Builder
for i, c := range b {
if i == 8 {
break
}
if i > 0 {
sb.WriteByte(' ')
}
fmt.Fprintf(&sb, "%02x", c)
}
return sb.String()
}
+530 -402
View File
File diff suppressed because it is too large Load Diff
+55
View File
@@ -7,8 +7,11 @@ import (
"bytes" "bytes"
"io" "io"
"os" "os"
"os/exec"
"path/filepath" "path/filepath"
"runtime"
"strings" "strings"
"syscall"
"testing" "testing"
) )
@@ -237,3 +240,55 @@ func TestCmdArgErrors(t *testing.T) {
t.Errorf("cmdParse() code = %d, want 2", code) t.Errorf("cmdParse() code = %d, want 2", code)
} }
} }
// TestVerifySmokeCrashIsolation checks that a function faulting on its
// zeroed smoke arguments is reported as CRASH by a child process instead of
// killing `gasm verify` itself.
func TestVerifySmokeCrashIsolation(t *testing.T) {
if testing.Short() {
t.Skip("builds the gasm binary")
}
if runtime.GOARCH != "amd64" {
t.Skip("amd64 JIT only")
}
bin := filepath.Join(t.TempDir(), "gasm")
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
t.Fatalf("build gasm: %v\n%s", err, out)
}
src := filepath.Join(t.TempDir(), "crash_amd64.s")
kernel := "#include \"textflag.h\"\n" +
"\n" +
"// func Fault(x []byte) int\n" +
"TEXT ·Fault(SB), NOSPLIT, $0-32\n" +
"\tMOVQ\tx+0(FP), AX\n" +
"\tMOVQ\t(AX), AX // faults on the zeroed nil pointer\n" +
"\tMOVQ\tAX, ret+24(FP)\n" +
"\tRET\n"
if err := os.WriteFile(src, []byte(kernel), 0o644); err != nil {
t.Fatal(err)
}
cmd := exec.Command(bin, "verify", "-smoke", src)
out, err := cmd.CombinedOutput()
if err == nil {
t.Fatalf("expected a failure report, got success:\n%s", out)
}
if exitErr, ok := err.(*exec.ExitError); ok {
if ws, ok := exitErr.Sys().(syscall.WaitStatus); ok && ws.Signaled() {
t.Fatalf("verify died from %v — the crash was not isolated:\n%s", ws.Signal(), out)
}
}
if !strings.Contains(string(out), "CRASH") {
t.Errorf("output does not report CRASH:\n%s", out)
}
}
func TestSweepCheckLines(t *testing.T) {
out := []byte("crash_amd64.s: 1 functions JIT-loaded\n" +
" Fault: 21 bytes, args=32, frame=0 NOSPLIT\n" +
" smoke: OK\n" +
" abi: clean (10 varied inputs)\n")
want := " smoke: OK\n abi: clean (10 varied inputs)"
if got := sweepCheckLines(out); got != want {
t.Errorf("sweepCheckLines = %q, want %q", got, want)
}
}
+345
View File
@@ -0,0 +1,345 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"fmt"
"go/ast"
"go/parser"
"go/token"
"os"
"strings"
gasmast "sourcedock.dev/petrbalvin/gasm-devkit/ast"
gasmparser "sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// cmdScaffold generates a differential test skeleton for every kernel in a
// file: a Go test that seeds random states, drives both the assembly kernel
// and a caller-provided portable reference, and compares the outputs
// byte-for-byte. The lesson this encodes: a pipeline-level fuzz cannot see
// an unwired kernel, only a direct-call differential against the portable
// specification can, so every kernel ships with one.
//
// The generated file follows two conventions the caller fills in:
// - the assembly symbols resolve because the test lives in the kernel's
// own package (the //go:noescape declarations reference them);
// - each kernel gets a <name>Portable Go function the author implements as
// the specification, and the test fails on the first divergent byte.
func cmdScaffold(args []string) error {
fs := newCommand("scaffold", "gasm scaffold differential <file.s>", `
Print a differential test skeleton for every // func signature in FILE.
The test seeds random states, drives the kernel and a portable reference
(<name>Portable), and compares outputs byte-for-byte. Write the reference
bodies, place the file in the kernel's package, and run it in CI.
`)
if err := fs.Parse(args); err != nil {
return err
}
rest := fs.Args()
// The first positional word is the scaffold style; "differential" is the
// only one today.
if len(rest) > 0 && rest[0] == "differential" {
rest = rest[1:]
}
if len(rest) != 1 {
return fmt.Errorf("usage: gasm scaffold differential <file.s>")
}
path := rest[0]
src, err := os.ReadFile(path)
if err != nil {
return err
}
f, errs := gasmparser.Parse(path, string(src))
if len(errs) > 0 {
return fmt.Errorf("parse: %v", errs[0])
}
var out strings.Builder
out.WriteString(headerComment)
out.WriteString("package " + packageName + "\n\n")
out.WriteString("import (\n\t\"bytes\"\n\t\"math/rand\"\n\t\"testing\"\n)\n\n")
out.WriteString(generatedHelpers)
kernels := 0
for _, d := range f.Decls {
txt, ok := d.(*gasmast.Text)
if !ok {
continue
}
params, results, ok := parseSig(txt.Doc)
if !ok || len(params) == 0 {
continue
}
kernels++
name := txt.Name.Name
fmt.Fprintf(&out, "// %sPortable is the specification %s is pinned against:\n", name, name)
fmt.Fprintf(&out, "// fill in a straightforward implementation of the same contract.\n")
fmt.Fprintf(&out, "func %sPortable(%s) (%s) {\n\tpanic(\"implement the portable specification\")\n}\n\n", name, paramDecl(params), resultDecl(results))
fmt.Fprintf(&out, "func Test%sDifferential(t *testing.T) {\n", strings.ToUpper(name[:1])+name[1:])
fmt.Fprintf(&out, "\trng := rand.New(rand.NewSource(1))\n")
fmt.Fprintf(&out, "\tfor range 1000 {\n")
// Seed two independent argument sets per iteration: the kernel runs
// on set A, the portable reference on set B, so in-place writes
// through pointer/slice arguments cannot contaminate the other side.
var sliceNames []string
seen := map[string]bool{}
aArgs := make([]string, 0, len(params))
bArgs := make([]string, 0, len(params))
for _, p := range params {
a, b, slices := genParamSeed(&out, p, seen)
aArgs = append(aArgs, a)
bArgs = append(bArgs, b)
sliceNames = append(sliceNames, slices...)
}
fmt.Fprintf(&out, "\t\tgot := %s(%s)\n", name, strings.Join(aArgs, ", "))
fmt.Fprintf(&out, "\t\twant := %sPortable(%s)\n", name, strings.Join(bArgs, ", "))
fmt.Fprintf(&out, "\t\tif !bytes.Equal(outputBytes(got), outputBytes(want)) {\n")
fmt.Fprintf(&out, "\t\t\tt.Fatalf(\"kernel diverges from the portable spec (seed 1, deterministic)\")\n")
fmt.Fprintf(&out, "\t\t}\n")
for _, s := range sliceNames {
fmt.Fprintf(&out, "\t\tif !bytes.Equal(outputBytes(%sA), outputBytes(%sB)) {\n", s, s)
fmt.Fprintf(&out, "\t\t\tt.Fatalf(\"kernel mutated %%q differently (seed 1, deterministic)\", %q)\n", s)
fmt.Fprintf(&out, "\t\t}\n")
}
fmt.Fprintf(&out, "\t}\n}\n\n")
}
if kernels == 0 {
return fmt.Errorf("%s: no // func signatures found; add one doc comment per kernel", path)
}
os.Stdout.WriteString(out.String())
return nil
}
const packageName = "yourpkg"
const headerComment = `// Code generated by gasm scaffold differential; EDIT THE PANICS.
// Each Test*Differential drives the assembly kernel and its portable
// reference over the same random states and compares the outputs.
// Place this file in the kernel's own package so the symbols resolve.
`
// sigParam is one parsed // func parameter.
type sigParam struct {
Names []string
Type string
}
type sigResult struct {
Names []string
Type string
}
// parseSig parses the // func signature of a doc comment.
func parseSig(doc string) ([]sigParam, []sigResult, bool) {
var line string
for l := range strings.SplitSeq(doc, "\n") {
if t := strings.TrimSpace(l); strings.HasPrefix(t, "func ") {
line = t
break
}
}
if line == "" {
return nil, nil, false
}
fset := token.NewFileSet()
f, err := parser.ParseFile(fset, "sig.go", "package p\n"+line+" {}\n", 0)
if err != nil {
return nil, nil, false
}
fd, ok := f.Decls[0].(*ast.FuncDecl)
if !ok || fd.Type == nil {
return nil, nil, false
}
var params []sigParam
for _, field := range fd.Type.Params.List {
typ := exprString(field.Type)
if len(field.Names) == 0 {
params = append(params, sigParam{Names: []string{""}, Type: typ})
continue
}
// Shared names (`L, result *byte`) expand to one entry per name:
// every name is a separate argument at the call site.
for _, n := range field.Names {
params = append(params, sigParam{Names: []string{n.Name}, Type: typ})
}
}
var results []sigResult
if fd.Type.Results != nil {
for _, field := range fd.Type.Results.List {
results = append(results, sigResult{Names: identNames(field.Names), Type: exprString(field.Type)})
}
}
return params, results, true
}
func identNames(idents []*ast.Ident) []string {
var out []string
for _, id := range idents {
out = append(out, id.Name)
}
return out
}
func exprString(e ast.Expr) string {
switch t := e.(type) {
case *ast.Ident:
return t.Name
case *ast.StarExpr:
return "*" + exprString(t.X)
case *ast.SelectorExpr:
return exprString(t.X) + "." + t.Sel.Name
case *ast.ArrayType:
if t.Len == nil {
return "[]" + exprString(t.Elt)
}
return "[N]" + exprString(t.Elt)
}
return "interface{}"
}
// paramDecl renders a parameter list for the portable reference signature.
func paramDecl(params []sigParam) string {
var parts []string
for _, p := range params {
if len(p.Names) == 0 {
parts = append(parts, p.Type)
continue
}
for _, n := range p.Names {
parts = append(parts, n+" "+p.Type)
}
}
return strings.Join(parts, ", ")
}
// resultDecl renders a result list; unnamed results keep bare types.
func resultDecl(results []sigResult) string {
if len(results) == 0 {
return ""
}
var parts []string
for _, r := range results {
parts = append(parts, r.Type)
}
return strings.Join(parts, ", ")
}
// genParamSeed emits the seeding statements for one parameter and returns
// the kernel-side (A) and reference-side (B) argument expressions, plus the
// names of any slice variables written in place (compared after the calls).
func genParamSeed(out *strings.Builder, p sigParam, seen map[string]bool) (aArg, bArg string, slices []string) {
name := p.Names[0]
elem := strings.TrimPrefix(p.Type, "*")
isSlice := strings.HasPrefix(p.Type, "[]")
if isSlice {
elem = strings.TrimPrefix(p.Type, "[]")
}
switch {
case isSlice:
v := uniqueName(seen, name)
fmt.Fprintf(out, "\t\t%sA := make([]%s, 1+rng.Intn(512))\n", v, elem)
fmt.Fprintf(out, "\t\t%sB := make([]%s, len(%sA))\n", v, elem, v)
fmt.Fprintf(out, "\t\tfor i := range %sA {\n", v)
fmt.Fprintf(out, "\t\t\tw%s := %s(rng.Intn(256))\n", v, goCast(elem))
fmt.Fprintf(out, "\t\t\t%sA[i] = w%s\n", v, v)
fmt.Fprintf(out, "\t\t\t%sB[i] = w%s\n", v, v)
fmt.Fprintf(out, "\t\t}\n")
return v, v, []string{v}
case strings.HasPrefix(p.Type, "*"):
v := uniqueName(seen, name)
fmt.Fprintf(out, "\t\tvar %sA, %sB %s\n", v, v, elem)
fmt.Fprintf(out, "\t\tw%s := %s(rng.Intn(256))\n", v, goCast(elem))
fmt.Fprintf(out, "\t\t%sA = w%s\n", v, v)
fmt.Fprintf(out, "\t\t%sB = w%s\n", v, v)
return "&" + v + "A", "&" + v + "B", nil
default:
v := uniqueName(seen, name)
fmt.Fprintf(out, "\t\tw%s := %s(rng.Intn(512))\n", v, goCast(""))
fmt.Fprintf(out, "\t\tvar %sA, %sB %s = w%s, w%s\n", v, v, p.Type, v, v)
return v + "A", v + "B", nil
}
}
// uniqueName de-duplicates seeded variable names when one kernel takes two
// parameters of the same name (impossible in Go) or a name repeats across
// kernels in one file.
func uniqueName(seen map[string]bool, base string) string {
if base == "" {
base = "arg"
}
if !seen[base] {
seen[base] = true
return base
}
for i := 2; ; i++ {
cand := fmt.Sprintf("%s%d", base, i)
if !seen[cand] {
seen[cand] = true
return cand
}
}
}
// goCast returns the conversion turning rng.Intn into the element type.
func goCast(elem string) string {
switch elem {
case "byte", "uint8":
return "byte"
case "int8":
return "int8"
case "uint16":
return "uint16"
case "int16":
return "int16"
case "uint32":
return "uint32"
case "int32":
return "int32"
case "uint64":
return "uint64"
default:
return "int"
}
}
// generatedHelpers is emitted into every generated test file: outputBytes
// narrows returned slices and scalars to a byte form for the comparison.
// It lives in the template, not in this binary, because only the generated
// file ever calls it.
const generatedHelpers = `// outputBytes narrows a returned slice or scalar to bytes for the
// comparison; extend the switch when a kernel returns a wider type.
func outputBytes(v any) []byte {
switch t := v.(type) {
case []byte:
return t
case []int32:
b := make([]byte, 4*len(t))
for i, x := range t {
b[i*4] = byte(x)
b[i*4+1] = byte(x >> 8)
b[i*4+2] = byte(x >> 16)
b[i*4+3] = byte(x >> 24)
}
return b
case []uint16:
b := make([]byte, 2*len(t))
for i, x := range t {
b[i*2] = byte(x)
b[i*2+1] = byte(x >> 8)
}
return b
case int:
b := make([]byte, 8)
for i := range 8 {
b[i] = byte(uint64(t) >> (8 * i))
}
return b
default:
return nil
}
}
`
+140
View File
@@ -0,0 +1,140 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"fmt"
"slices"
"strings"
)
// unifiedDiff renders a unified diff with three lines of context between the
// two line slices, in the form `gofmt -d` prints. An empty result means the
// inputs are identical.
func unifiedDiff(name string, a, b []string) string {
if slices.Equal(a, b) {
return ""
}
var out strings.Builder
fmt.Fprintf(&out, "--- %s\n+++ %s\n", name, name)
// Longest common subsequence over the lines (assembly files are small
// enough for the quadratic table).
n, m := len(a), len(b)
lcs := make([][]int, n+1)
for i := range lcs {
lcs[i] = make([]int, m+1)
}
for i := n - 1; i >= 0; i-- {
for j := m - 1; j >= 0; j-- {
if a[i] == b[j] {
lcs[i][j] = lcs[i+1][j+1] + 1
} else if lcs[i+1][j] >= lcs[i][j+1] {
lcs[i][j] = lcs[i+1][j]
} else {
lcs[i][j] = lcs[i][j+1]
}
}
}
// Walk the LCS once, assigning every op its absolute position in both
// files (1-based, the position an insertion sits before).
type op struct {
kind byte // ' ', '-' or '+'
aLine, bLine int
text string
}
var ops []op
aPos, bPos := 0, 0
emit := func(kind byte, text string) {
ops = append(ops, op{kind: kind, aLine: aPos + 1, bLine: bPos + 1, text: text})
switch kind {
case ' ':
aPos++
bPos++
case '-':
aPos++
case '+':
bPos++
}
}
i, j := 0, 0
for i < n && j < m {
switch {
case a[i] == b[j]:
emit(' ', a[i])
i++
j++
case lcs[i+1][j] >= lcs[i][j+1]:
emit('-', a[i])
i++
default:
emit('+', b[j])
j++
}
}
for ; i < n; i++ {
emit('-', a[i])
}
for ; j < m; j++ {
emit('+', b[j])
}
// Group the edits into hunks: consecutive changes separated by more than
// twice the context lines start a new hunk.
const context = 3
var changes []int
for k, o := range ops {
if o.kind != ' ' {
changes = append(changes, k)
}
}
for g := 0; g < len(changes); {
last := g
for last+1 < len(changes) && changes[last+1]-changes[last]-1 <= 2*context {
last++
}
lo := max(0, changes[g]-context)
hi := min(len(ops), changes[last]+1+context)
// The header numbers are the first line of each side actually shown:
// the first context, deletion or insertion line. A hunk that shows
// no old lines is a pure insertion and reports the position it sits
// before (0 at the top of the file); the mirror rule holds for a
// pure deletion.
aStart := ops[lo].aLine - 1
bStart := ops[lo].bLine - 1
countA, countB := 0, 0
for _, o := range ops[lo:hi] {
switch o.kind {
case ' ':
countA++
countB++
case '-':
countA++
case '+':
countB++
}
}
for _, o := range ops[lo:hi] {
if o.kind != '+' {
aStart = o.aLine
break
}
}
for _, o := range ops[lo:hi] {
if o.kind != '-' {
bStart = o.bLine
break
}
}
fmt.Fprintf(&out, "@@ -%d,%d +%d,%d @@\n", aStart, countA, bStart, countB)
for _, o := range ops[lo:hi] {
out.WriteByte(o.kind)
out.WriteString(o.text)
out.WriteByte('\n')
}
g = last + 1
}
return out.String()
}
+94
View File
@@ -0,0 +1,94 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"slices"
"strings"
"testing"
)
func lines(ss ...string) []string { return ss }
func TestUnifiedDiffIdentical(t *testing.T) {
if got := unifiedDiff("f", lines("a", "b"), lines("a", "b")); got != "" {
t.Errorf("identical inputs produced %q, want empty", got)
}
}
func TestUnifiedDiffSingleChange(t *testing.T) {
a := lines("1", "2", "3", "4", "5", "6", "7", "8")
b := lines("1", "2", "3!", "4", "5", "6", "7", "8")
want := "--- f\n+++ f\n" +
"@@ -1,6 +1,6 @@\n" +
" 1\n 2\n-3\n+3!\n 4\n 5\n 6\n"
if got := unifiedDiff("f", a, b); got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func TestUnifiedDiffInsertAtStart(t *testing.T) {
got := unifiedDiff("f", lines("x"), lines("new", "x"))
// The single existing line is shown as trailing context, so the hunk
// covers it.
want := "--- f\n+++ f\n@@ -1,1 +1,2 @@\n+new\n x\n"
if got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func TestUnifiedDiffDeleteAtEnd(t *testing.T) {
got := unifiedDiff("f", lines("x", "y"), lines("x"))
want := "--- f\n+++ f\n@@ -1,2 +1,1 @@\n x\n-y\n"
if got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func TestUnifiedDiffTwoHunks(t *testing.T) {
var a, b []string
for i := 1; i <= 20; i++ {
a = append(a, itoa(i))
b = append(b, itoa(i))
}
b[1] = "2!"
b[17] = "18!"
got := unifiedDiff("f", a, b)
if !strings.Contains(got, "@@ -1,5 +1,5 @@\n 1\n-2\n+2!\n 3\n 4\n 5\n") {
t.Errorf("first hunk wrong:\n%s", got)
}
if !strings.Contains(got, "@@ -15,6 +15,6 @@\n 15\n 16\n 17\n-18\n+18!\n 19\n 20\n") {
t.Errorf("second hunk wrong:\n%s", got)
}
}
// TestUnifiedDiffAdjacentHunks merges changes separated by exactly twice the
// context into one hunk.
func TestUnifiedDiffAdjacentHunks(t *testing.T) {
a := lines("1", "2", "3", "4", "5", "6", "7", "8")
b := slices.Clone(a)
b[0] = "1!"
b[7] = "8!"
got := unifiedDiff("f", a, b)
want := "--- f\n+++ f\n" +
"@@ -1,8 +1,8 @@\n" +
"-1\n+1!\n 2\n 3\n 4\n 5\n 6\n 7\n-8\n+8!\n"
if got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func itoa(n int) string {
if n == 0 {
return "0"
}
var buf [4]byte
i := len(buf)
for n > 0 {
i--
buf[i] = byte('0' + n%10)
n /= 10
}
return string(buf[i:])
}
+73 -69
View File
@@ -1,8 +1,12 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause // SPDX-License-Identifier: BSD-3-Clause
//go:build linux
package debug package debug
import "strings"
import "fmt" import "fmt"
// Breakpoint is one INT3 breakpoint in the debuggee. // Breakpoint is one INT3 breakpoint in the debuggee.
@@ -15,75 +19,60 @@ type Breakpoint struct {
hits int hits int
} }
// Condition is a simple register-comparison condition evaluated when a // Condition is a register-comparison condition evaluated when a breakpoint
// breakpoint is hit. Format: <reg> <op> <value>. // is hit. Supports three forms:
// - register vs constant: <reg> <op> <value>
// - register vs register: <reg> <op> <reg2>
// - register vs memory: <reg> <op> *<addr>
type Condition struct { type Condition struct {
Reg string // register name (rax, rbx, rip, rsp, ...) Reg string // register name (rax, rbx, rip, rsp, ...)
Op string // comparison operator: ==, !=, <, >, <=, >= Op string // comparison operator: ==, !=, <, >, <=, >=
Value uint64 Value uint64 // constant value (when Reg2 == "" and MemAddr == 0)
Reg2 string // second register name (for register-register comparison)
MemAddr uint64 // memory address (for register-memory comparison, prefixed with *)
} }
// Eval checks the condition against the current registers. // Eval checks the condition against the current registers.
func (c *Condition) Eval(regs *Regs) bool { func (c *Condition) Eval(regs *Regs) bool {
var actual uint64 actual, ok := regs.RegValue(c.Reg)
switch c.Reg { if !ok {
case "rax", "eax", "ax", "al": return true // unknown register, don't block
actual = regs.RAX }
case "rbx", "ebx", "bx", "bl": var expected uint64
actual = regs.RBX switch {
case "rcx", "ecx", "cx", "cl": case c.Reg2 != "":
actual = regs.RCX // Register-register comparison.
case "rdx", "edx", "dx", "dl": v, ok := regs.RegValue(c.Reg2)
actual = regs.RDX if !ok {
case "rsi", "esi", "si": return true
actual = regs.RSI }
case "rdi", "edi", "di": expected = v
actual = regs.RDI case c.MemAddr != 0:
case "rbp", "ebp", "bp": // Register-memory comparison, requires a Session, not available here.
actual = regs.RBP // Fall back to treating as constant (the caller should resolve).
case "rsp", "esp", "sp": expected = c.Value
actual = regs.RSP
case "r8":
actual = regs.R8
case "r9":
actual = regs.R9
case "r10":
actual = regs.R10
case "r11":
actual = regs.R11
case "r12":
actual = regs.R12
case "r13":
actual = regs.R13
case "r14":
actual = regs.R14
case "r15":
actual = regs.R15
case "rip", "eip":
actual = regs.RIP
default: default:
return true // unknown register — don't block expected = c.Value
} }
switch c.Op { switch c.Op {
case "==", "=": case "==", "=":
return actual == c.Value return actual == expected
case "!=": case "!=":
return actual != c.Value return actual != expected
case "<": case "<":
return actual < c.Value return actual < expected
case ">": case ">":
return actual > c.Value return actual > expected
case "<=": case "<=":
return actual <= c.Value return actual <= expected
case ">=": case ">=":
return actual >= c.Value return actual >= expected
default: default:
return true return true
} }
} }
// Breakpoints manages the set of breakpoints for a Session. // Breakpoints manages the software breakpoints of one Session.
// Breakpoints manages software breakpoints for a debuggee.
type Breakpoints struct { type Breakpoints struct {
t tracer t tracer
bps map[uint64]*Breakpoint bps map[uint64]*Breakpoint
@@ -106,14 +95,18 @@ func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (
bp.Cond = cond bp.Cond = cond
return bp, nil return bp, nil
} }
// Read the original byte. // Read the original bytes.
word, err := bm.t.Peek(addr) word, err := bm.t.Peek(addr)
if err != nil { if err != nil {
return nil, err return nil, err
} }
orig := byte(word) orig := byte(word)
// Patch with INT3 (0xCC), preserving the rest of the word. // Patch with the breakpoint instruction, preserving the rest of the word.
patched := (word &^ 0xFF) | 0xCC mask := uint64(0)
for range breakpointInsn {
mask = (mask << 8) | 0xFF
}
patched := (word &^ mask) | breakpointWord(breakpointInsn)
if err := bm.t.Poke(addr, patched); err != nil { if err := bm.t.Poke(addr, patched); err != nil {
return nil, err return nil, err
} }
@@ -122,17 +115,12 @@ func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (
return bp, nil return bp, nil
} }
// Hits returns the number of times the breakpoint has been hit.
func (bp *Breakpoint) Hits() int {
return bp.hits
}
// Info returns a formatted list of all breakpoints. // Info returns a formatted list of all breakpoints.
func (bm *Breakpoints) Info() string { func (bm *Breakpoints) Info() string {
if len(bm.bps) == 0 { if len(bm.bps) == 0 {
return "no breakpoints set\n" return "no breakpoints set\n"
} }
result := "" var result strings.Builder
i := 0 i := 0
for _, bp := range bm.bps { for _, bp := range bm.bps {
i++ i++
@@ -148,9 +136,9 @@ func (bm *Breakpoints) Info() string {
if bp.Cond != nil { if bp.Cond != nil {
cond = fmt.Sprintf(" if %s %s %#x", bp.Cond.Reg, bp.Cond.Op, bp.Cond.Value) cond = fmt.Sprintf(" if %s %s %#x", bp.Cond.Reg, bp.Cond.Op, bp.Cond.Value)
} }
result += fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond) result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond))
} }
return result return result.String()
} }
// Clear removes the breakpoint at addr, restoring the original byte. // Clear removes the breakpoint at addr, restoring the original byte.
@@ -196,19 +184,22 @@ func (bm *Breakpoints) All() []*Breakpoint {
} }
// HandleTrap is called after the debuggee stops on SIGTRAP. It checks // HandleTrap is called after the debuggee stops on SIGTRAP. It checks
// whether the trap was caused by one of our breakpoints (RIP-1 matches // whether the trap was caused by one of our breakpoints (PC-adjust matches
// a breakpoint address), restores the original byte, rewinds RIP, and // a breakpoint address), restores the original byte, rewinds PC, and
// returns the breakpoint that was hit (or nil if it was a single-step). // returns the breakpoint that was hit (or nil if it was a single-step).
// Hits returns how many times the breakpoint has been hit.
func (bp *Breakpoint) Hits() int { return bp.hits }
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint { func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
// After INT3, RIP points to the byte AFTER the 0xCC. // After a breakpoint trap, PC points past the breakpoint instruction.
trapAddr := regs.RIP - 1 trapAddr := regs.GetPC() - uint64(breakpointPCAdjust)
bp, ok := bm.bps[trapAddr] bp, ok := bm.bps[trapAddr]
if !ok || !bp.Enabled { if !ok || !bp.Enabled {
return nil // single-step trap or unknown return nil // single-step trap or unknown
} }
// Check the condition (if any). // Check the condition (if any).
if bp.Cond != nil && !bp.Cond.Eval(regs) { if bp.Cond != nil && !bp.Cond.Eval(regs) {
// Condition not met — restore the byte but do NOT rewind RIP. // Condition not met, restore the byte but do NOT rewind RIP.
// The process continues from the next instruction (past the INT3). // The process continues from the next instruction (past the INT3).
word, err := bm.t.Peek(trapAddr) word, err := bm.t.Peek(trapAddr)
if err == nil { if err == nil {
@@ -225,8 +216,8 @@ func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
restored := (word &^ 0xFF) | uint64(bp.Orig) restored := (word &^ 0xFF) | uint64(bp.Orig)
bm.t.Poke(trapAddr, restored) bm.t.Poke(trapAddr, restored)
} }
// Rewind RIP to re-execute the original instruction. // Rewind PC to re-execute the original instruction.
regs.RIP = trapAddr regs.SetPC(trapAddr)
bm.t.SetRegs(regs) bm.t.SetRegs(regs)
return bp return bp
} }
@@ -243,6 +234,19 @@ func (bm *Breakpoints) Reinsert(addr uint64) error {
if err != nil { if err != nil {
return err return err
} }
patched := (word &^ 0xFF) | 0xCC mask := uint64(0)
for range breakpointInsn {
mask = (mask << 8) | 0xFF
}
patched := (word &^ mask) | breakpointWord(breakpointInsn)
return bm.t.Poke(addr, patched) return bm.t.Poke(addr, patched)
} }
// breakpointWord converts the breakpoint instruction bytes to a uint64.
func breakpointWord(insn []byte) uint64 {
var w uint64
for i, b := range insn {
w |= uint64(b) << (i * 8)
}
return w
}
+5 -3
View File
@@ -1,6 +1,8 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause // SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug package debug
import ( import (
@@ -273,10 +275,10 @@ func TestBreakpointInfo(t *testing.T) {
} }
func TestWatchpointSlotTracking(t *testing.T) { func TestWatchpointSlotTracking(t *testing.T) {
s := &Session{} s := &Session{} // per-session slots start free
// All four slots are free initially. // All four slots are free initially.
for i := 0; i < 4; i++ { for i := range 4 {
if s.IsWatchpointSlotUsed(i) { if s.IsWatchpointSlotUsed(i) {
t.Errorf("slot %d should be free initially", i) t.Errorf("slot %d should be free initially", i)
} }
@@ -314,7 +316,7 @@ func TestWatchpointSlotTracking(t *testing.T) {
} }
// Mark all slots used: FindFreeWatchpointSlot returns -1. // Mark all slots used: FindFreeWatchpointSlot returns -1.
for i := 0; i < 4; i++ { for i := range 4 {
s.wpSlots[i] = true s.wpSlots[i] = true
} }
if got := s.FindFreeWatchpointSlot(); got != -1 { if got := s.FindFreeWatchpointSlot(); got != -1 {
+12 -16
View File
@@ -7,46 +7,42 @@ package debug
import ( import (
"fmt" "fmt"
"strings"
"golang.org/x/arch/x86/x86asm" "sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
) )
// Disassemble decodes the instruction at the given address in the debuggee's // Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes. // memory and returns its text representation and length in bytes.
func (s *Session) Disassemble(addr uint64) (string, int, error) { func (s *Session) Disassemble(addr uint64) (string, int, error) {
// Read up to 15 bytes (max x86 instruction length).
mem, err := s.ReadMemory(addr, 15) mem, err := s.ReadMemory(addr, 15)
if err != nil { if err != nil {
// Try a shorter read if we're near a page boundary. return "", 0, err
mem, err = s.ReadMemory(addr, 1)
if err != nil {
return "", 0, err
}
} }
inst, err := x86asm.Decode(mem, 64) ins, err := disasm.Decode(arch.AMD64, mem, addr)
if err != nil { if err != nil {
return "???", 1, nil return "", 0, err
} }
text := x86asm.IntelSyntax(inst, addr, nil) return ins.Text, ins.Len, nil
return text, inst.Len, nil
} }
// DisassembleN decodes up to n instructions starting at addr and returns // DisassembleN decodes up to n instructions starting at addr and returns
// them as a formatted string with addresses and byte offsets. // them as a formatted string with addresses and byte offsets.
func (s *Session) DisassembleN(addr uint64, n int) string { func (s *Session) DisassembleN(addr uint64, n int) string {
var result string var result strings.Builder
pc := addr pc := addr
for i := 0; i < n; i++ { for range n {
text, length, err := s.Disassemble(pc) text, length, err := s.Disassemble(pc)
if err != nil { if err != nil {
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err) result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
break break
} }
result += fmt.Sprintf(" %#08x: %s\n", pc, text) result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
if length == 0 { if length == 0 {
length = 1 length = 1
} }
pc += uint64(length) pc += uint64(length)
} }
return result return result.String()
} }
+46
View File
@@ -0,0 +1,46 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && arm64
package debug
import (
"fmt"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 4)
if err != nil {
return "", 0, err
}
ins, err := disasm.Decode(arch.ARM64, mem, addr)
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result string
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
break
}
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
if length == 0 {
length = 4
}
pc += uint64(length)
}
return result
}
+46
View File
@@ -0,0 +1,46 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && loong64
package debug
import (
"fmt"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 4)
if err != nil {
return "", 0, err
}
ins, err := disasm.Decode(arch.LOONG64, mem, addr)
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result string
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
break
}
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
if length == 0 {
length = 4
}
pc += uint64(length)
}
return result
}
+46
View File
@@ -0,0 +1,46 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && riscv64
package debug
import (
"fmt"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 4)
if err != nil {
return "", 0, err
}
ins, err := disasm.Decode(arch.RISCV, mem, addr)
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result string
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
break
}
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
if length == 0 {
length = 4
}
pc += uint64(length)
}
return result
}
+82
View File
@@ -0,0 +1,82 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug
import "fmt"
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" RIP = %#016x (func+%#x)\n", regs.RIP, regs.RIP-codeBase-funcOff)
fmt.Printf(" RSP = %#016x RBP = %#016x\n", regs.RSP, regs.RBP)
fmt.Printf(" RAX = %#016x RBX = %#016x\n", regs.RAX, regs.RBX)
fmt.Printf(" RCX = %#016x RDX = %#016x\n", regs.RCX, regs.RDX)
fmt.Printf(" RSI = %#016x RDI = %#016x\n", regs.RSI, regs.RDI)
fmt.Printf(" R8 = %#016x R9 = %#016x\n", regs.R8, regs.R9)
fmt.Printf(" R10 = %#016x R11 = %#016x\n", regs.R10, regs.R11)
fmt.Printf(" R12 = %#016x R13 = %#016x\n", regs.R12, regs.R13)
fmt.Printf(" R14 = %#016x R15 = %#016x\n", regs.R14, regs.R15)
fmt.Printf(" RFLAGS = %#x [%s]\n", regs.RFLAGS, decodeRflags(regs.RFLAGS))
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n Vector registers (YMM):")
for i := 0; i < 16; i += 2 {
fmt.Printf(" YMM%-2d = ", i)
printYMM(v.YMM[i][:])
fmt.Printf(" YMM%-2d = ", i+1)
printYMM(v.YMM[i+1][:])
fmt.Println()
}
}
func printYMM(b []byte) {
for j := 0; j < 32; j += 4 {
v := uint32(b[j]) | uint32(b[j+1])<<8 | uint32(b[j+2])<<16 | uint32(b[j+3])<<24
fmt.Printf("%08x ", v)
}
}
func decodeRflags(f uint64) string {
var flags string
if f&1 != 0 {
flags += "CF "
}
if f&(1<<2) != 0 {
flags += "PF "
}
if f&(1<<4) != 0 {
flags += "AF "
}
if f&(1<<6) != 0 {
flags += "ZF "
}
if f&(1<<7) != 0 {
flags += "SF "
}
if f&(1<<8) != 0 {
flags += "TF "
}
if f&(1<<9) != 0 {
flags += "IF "
}
if f&(1<<10) != 0 {
flags += "DF "
}
if f&(1<<11) != 0 {
flags += "OF "
}
if flags == "" {
return "none"
}
return flags[:len(flags)-1]
}
// archReturnAddr reads the return address from the stack (amd64 ABI0 convention).
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return s.Peek(regs.GetSP())
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "RSP" }
+45
View File
@@ -0,0 +1,45 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && arm64
package debug
import "fmt"
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.SP, regs.X29)
fmt.Printf(" LR = %#016x\n", regs.X30)
fmt.Printf(" X0 = %#016x X1 = %#016x\n", regs.X0, regs.X1)
fmt.Printf(" X2 = %#016x X3 = %#016x\n", regs.X2, regs.X3)
fmt.Printf(" X4 = %#016x X5 = %#016x\n", regs.X4, regs.X5)
fmt.Printf(" X6 = %#016x X7 = %#016x\n", regs.X6, regs.X7)
fmt.Printf(" X8 = %#016x X9 = %#016x\n", regs.X8, regs.X9)
fmt.Printf(" X10 = %#016x X11 = %#016x\n", regs.X10, regs.X11)
fmt.Printf(" X12 = %#016x X13 = %#016x\n", regs.X12, regs.X13)
fmt.Printf(" X14 = %#016x X15 = %#016x\n", regs.X14, regs.X15)
fmt.Printf(" X16 = %#016x X17 = %#016x\n", regs.X16, regs.X17)
fmt.Printf(" X18 = %#016x X19 = %#016x\n", regs.X18, regs.X19)
fmt.Printf(" X20 = %#016x X21 = %#016x\n", regs.X20, regs.X21)
fmt.Printf(" X22 = %#016x X23 = %#016x\n", regs.X22, regs.X23)
fmt.Printf(" X24 = %#016x X25 = %#016x\n", regs.X24, regs.X25)
fmt.Printf(" X26 = %#016x X27 = %#016x\n", regs.X26, regs.X27)
fmt.Printf(" X28 = %#016x PSTATE = %#x\n", regs.X28, regs.PSTATE)
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n Vector registers (V0-V31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" V%-2d = %016x%016x\n", i, v.V[i][8], v.V[i][0])
fmt.Printf(" V%-2d = %016x%016x\n", i+1, v.V[i+1][8], v.V[i+1][0])
}
}
// archReturnAddr reads the return address from LR (arm64 convention).
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return regs.X30, nil
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "SP" }
+44
View File
@@ -0,0 +1,44 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && loong64
package debug
import "fmt"
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.R31, regs.R31-codeBase-funcOff)
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.R3, regs.R21)
fmt.Printf(" RA = %#016x\n", regs.R1)
fmt.Printf(" A0 = %#016x A1 = %#016x\n", regs.R4, regs.R5)
fmt.Printf(" A2 = %#016x A3 = %#016x\n", regs.R6, regs.R7)
fmt.Printf(" A4 = %#016x A5 = %#016x\n", regs.R8, regs.R9)
fmt.Printf(" A6 = %#016x A7 = %#016x\n", regs.R10, regs.R11)
fmt.Printf(" T0 = %#016x T1 = %#016x\n", regs.R12, regs.R13)
fmt.Printf(" T2 = %#016x T3 = %#016x\n", regs.R14, regs.R15)
fmt.Printf(" T4 = %#016x T5 = %#016x\n", regs.R16, regs.R17)
fmt.Printf(" T6 = %#016x T7 = %#016x\n", regs.R18, regs.R19)
fmt.Printf(" T8 = %#016x\n", regs.R20)
fmt.Printf(" S0 = %#016x S1 = %#016x\n", regs.R22, regs.R23)
fmt.Printf(" S2 = %#016x S3 = %#016x\n", regs.R24, regs.R25)
fmt.Printf(" S4 = %#016x S5 = %#016x\n", regs.R26, regs.R27)
fmt.Printf(" S6 = %#016x S7 = %#016x\n", regs.R28, regs.R29)
fmt.Printf(" S8 = %#016x\n", regs.R30)
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n FP registers (F0-F31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" F%-2d = %#018x F%-2d = %#018x\n", i, v.F[i], i+1, v.F[i+1])
}
fmt.Printf(" FCC = %#016x FCSR = %#x\n", v.FCC, v.FCSR)
}
// archReturnAddr reads the return address from RA (loong64 convention).
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return regs.R1, nil
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "SP" }
+44
View File
@@ -0,0 +1,44 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && riscv64
package debug
import "fmt"
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.Sp, regs.S0)
fmt.Printf(" RA = %#016x\n", regs.Ra)
fmt.Printf(" A0 = %#016x A1 = %#016x\n", regs.A0, regs.A1)
fmt.Printf(" A2 = %#016x A3 = %#016x\n", regs.A2, regs.A3)
fmt.Printf(" A4 = %#016x A5 = %#016x\n", regs.A4, regs.A5)
fmt.Printf(" A6 = %#016x A7 = %#016x\n", regs.A6, regs.A7)
fmt.Printf(" T0 = %#016x T1 = %#016x\n", regs.T0, regs.T1)
fmt.Printf(" T2 = %#016x T3 = %#016x\n", regs.T2, regs.T3)
fmt.Printf(" T4 = %#016x T5 = %#016x\n", regs.T4, regs.T5)
fmt.Printf(" T6 = %#016x\n", regs.T6)
fmt.Printf(" S1 = %#016x S2 = %#016x\n", regs.S1, regs.S2)
fmt.Printf(" S3 = %#016x S4 = %#016x\n", regs.S3, regs.S4)
fmt.Printf(" S5 = %#016x S6 = %#016x\n", regs.S5, regs.S6)
fmt.Printf(" S7 = %#016x S8 = %#016x\n", regs.S7, regs.S8)
fmt.Printf(" S9 = %#016x S10 = %#016x\n", regs.S9, regs.S10)
fmt.Printf(" S11 = %#016x\n", regs.S11)
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n FP registers (F0-F31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" F%-2d = %#018x F%-2d = %#018x\n", i, v.F[i], i+1, v.F[i+1])
}
fmt.Printf(" FCSR = %#x\n", v.FCSR)
}
// archReturnAddr reads the return address from RA (riscv64 convention).
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return regs.Ra, nil
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "SP" }
+126
View File
@@ -0,0 +1,126 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug
import (
"fmt"
"os"
"os/exec"
"path/filepath"
"runtime"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
)
// buildGasm produces the gasm binary the debugger spawns as its debuggee.
func buildGasm(t *testing.T) string {
t.Helper()
if p := os.Getenv("GASM_TEST_BIN"); p != "" {
return p
}
bin := filepath.Join(t.TempDir(), "gasm")
cmd := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm")
out, err := cmd.CombinedOutput()
if err != nil {
t.Fatalf("build gasm: %v: %s", err, out)
}
return bin
}
// TestLaunchAndBreakpoint drives a real ptrace session end to end: launch the
// debuggee, break on the first instruction of the function and expect a
// breakpoint trap instead of a clean exit.
func TestLaunchAndBreakpoint(t *testing.T) {
if runtime.GOARCH != "amd64" {
t.Skip("runs only on amd64 hosts")
}
// The tracer is the OS thread that forked the debuggee (PTRACE_TRACEME
// binds the relation to that thread); every ptrace request must come
// from the same thread, so pin the test goroutine to one thread.
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernelPath = "../testdata/verify/basic_amd64.s"
k, err := verify.Load(kernelPath)
if err != nil {
t.Fatalf("Load: %v", err)
}
t.Cleanup(k.Close)
fl, err := k.Func("wideCopy")
if err != nil {
t.Fatalf("Func: %v", err)
}
sess, err := Launch(bin, kernelPath, "wideCopy", make([]byte, fl.Args))
if err != nil {
t.Fatalf("Launch: %v", err)
}
t.Cleanup(sess.Kill)
bm := NewBreakpoints(sess)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
// The INT3 must be visible in the debuggee's memory.
word, err := sess.Peek(entry)
if err != nil {
t.Fatalf("Peek: %v", err)
}
if b := word & 0xFF; b != 0xCC {
t.Fatalf("int3 not patched: first byte %#02x at %#x", b, entry)
}
// The debuggee raises a second SIGSTOP after the launch barrier (the
// child's RunTarget marks its entry), so like the REPL and the cover
// mode the test keeps resuming until the breakpoint trap arrives.
for range 10 {
if err := sess.Continue(); err != nil {
st, _ := os.ReadFile(fmt.Sprintf("/proc/%d/stat", sess.Pid()))
status, _ := os.ReadFile(fmt.Sprintf("/proc/%d/status", sess.Pid()))
t.Fatalf("Continue: %v\nstate: %s\n%s", err, fieldName(st), statusDump(status))
}
if sess.Exited() {
t.Fatal("debuggee exited instead of trapping on the breakpoint")
}
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if bp := bm.HandleTrap(&regs); bp != nil {
if bp.Addr != entry {
t.Fatalf("trap at %#x, want %#x", bp.Addr, entry)
}
return // trap on the entry breakpoint: the whole flow works
}
}
t.Fatal("no breakpoint trap after 10 resumes")
}
func fieldName(stat []byte) string {
f := strings.Split(string(stat), " ")
if len(f) > 2 {
return "state=" + f[2]
}
return "no stat"
}
func statusDump(b []byte) string {
var out []string
for l := range strings.SplitSeq(string(b), "\n") {
if strings.HasPrefix(l, "State") || strings.HasPrefix(l, "Pid") ||
strings.HasPrefix(l, "PPid") || strings.HasPrefix(l, "TracerPid") ||
strings.HasPrefix(l, "Threads") || strings.HasPrefix(l, "SigPnd") ||
strings.HasPrefix(l, "SigBlk") || strings.HasPrefix(l, "SigIgn") {
out = append(out, l)
}
}
return strings.Join(out, "\n")
}
+329
View File
@@ -0,0 +1,329 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
package debug
import (
"fmt"
"os"
"os/exec"
"path/filepath"
"runtime"
"strings"
"syscall"
"time"
)
// Session is a ptrace debugging session controlling one debuggee process.
type Session struct {
pid int
cmd *exec.Cmd
stopped bool
exited bool
codeBase uint64 // base address of the JIT code in the debuggee
wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 BADVR0-15)
}
// Launch starts the debuggee subprocess (gasm debug --target ...) and
// attaches to it via ptrace.
func Launch(gasmBin, asmPath, funcName string, args []byte) (*Session, error) {
sess, _, err := LaunchWithBuffers(gasmBin, asmPath, funcName, args, "")
return sess, err
}
// LaunchWithBuffers is like Launch but also allocates buffers in the debuggee.
//
// It pins the calling goroutine to its OS thread and leaves it pinned: the
// debuggee's PTRACE_TRACEME binds the tracer relation to the forking thread,
// and every ptrace request on the session must come from that same thread.
// All Session methods must therefore be called from the goroutine that
// launched the session (the REPL and coverage loops do exactly that).
func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec string) (*Session, []uint64, error) {
runtime.LockOSThread() // ptrace requests must stay on the forking thread
self, err := os.Executable()
if err != nil {
return nil, nil, fmt.Errorf("debug: cannot find gasm binary: %w", err)
}
if gasmBin != "" {
self = gasmBin
}
tmpDir, err := os.MkdirTemp("", "gasm-debug-*")
if err != nil {
return nil, nil, fmt.Errorf("debug: tempdir: %w", err)
}
argsFile := filepath.Join(tmpDir, "args.bin")
if err := os.WriteFile(argsFile, args, 0o644); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: write args: %w", err)
}
if bufSpec != "" {
if err := os.WriteFile(filepath.Join(tmpDir, "bufspec"), []byte(bufSpec), 0o644); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: write bufspec: %w", err)
}
}
cmd := exec.Command(self, "debug", "--func", funcName, "--args", argsFile, asmPath)
cmd.Env = append(os.Environ(), "GASM_DEBUG_TARGET=1", "GASM_DEBUG_TMP="+tmpDir)
cmd.Stdout = nil
cmd.Stderr = os.Stderr
cmd.SysProcAttr = &syscall.SysProcAttr{}
if err := cmd.Start(); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: start debuggee: %w", err)
}
s := &Session{pid: cmd.Process.Pid, cmd: cmd}
readyFile := filepath.Join(tmpDir, "ready")
for range 500 {
if _, err := os.Stat(readyFile); err == nil {
break
}
time.Sleep(5 * time.Millisecond)
}
// The debuggee parks itself with SIGSTOP once the JIT code is mapped.
// A Go tracee also reports SIGURG preemption as signal-delivery-stops,
// so the wait loops until a stop the debugger cares about instead of
// assuming the first event is the SIGSTOP.
if _, err := s.waitStopped(); err != nil {
cmd.Process.Kill()
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: wait for debuggee: %w", err)
}
s.stopped = true
// The debuggee reports its JIT mapping in the codebase file; that is the
// exact region the kernel was written to. Scanning /proc/pid/maps for
// any RWX region is only the fallback.
if data, err := os.ReadFile(filepath.Join(tmpDir, "codebase")); err == nil {
fmt.Sscanf(string(data), "%d", &s.codeBase)
}
if s.codeBase == 0 {
s.codeBase = findRWXMapping(s.pid)
}
var bufAddrs []uint64
if bufSpec != "" {
addrFile := filepath.Join(tmpDir, "bufaddrs")
if data, err := os.ReadFile(addrFile); err == nil {
for line := range strings.SplitSeq(strings.TrimSpace(string(data)), "\n") {
var addr uint64
if _, err := fmt.Sscanf(line, "%d", &addr); err == nil {
bufAddrs = append(bufAddrs, addr)
}
}
}
}
return s, bufAddrs, nil
}
// wait waits for the debuggee to stop and returns the wait status.
func (s *Session) wait() error {
var ws syscall.WaitStatus
_, err := syscall.Wait4(s.pid, &ws, 0, nil)
if err != nil {
return err
}
if ws.Exited() {
s.exited = true
return fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
}
s.stopped = true
return nil
}
// waitStopped consumes ptrace-stop events until one the debugger cares
// about arrives: SIGTRAP (a breakpoint or a completed single-step) or the
// debuggee's own SIGSTOP. A Go tracee's runtime raises SIGURG for
// asynchronous preemption, and every signal on a traced thread surfaces as
// a signal-delivery-stop, so those are suppressed and the tracee resumed
// without them. Runtime noise is why a single wait can return in the
// middle of runtime code and a resume can then fail: the event stream must
// be drained by the tracer.
func (s *Session) waitStopped() (syscall.Signal, error) {
for {
var ws syscall.WaitStatus
if _, err := syscall.Wait4(s.pid, &ws, syscall.WUNTRACED, nil); err != nil {
return 0, err
}
if ws.Exited() {
s.exited = true
return 0, fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
}
if ws.Signaled() {
s.exited = true
return 0, fmt.Errorf("debuggee killed by signal %v", ws.Signal())
}
switch sig := ws.StopSignal(); sig {
case syscall.SIGTRAP, syscall.SIGSTOP:
s.stopped = true
return sig, nil
default:
// Runtime noise (SIGURG preemption and friends): resume the
// tracee without delivering the signal.
if _, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_CONT),
uintptr(s.pid),
0, 0, 0, 0,
); errno != 0 {
return 0, fmt.Errorf("debug: PTRACE_CONT: %w", errno)
}
}
}
}
// Peek reads a word (8 bytes) from the debuggee's memory at addr.
func (s *Session) Peek(addr uint64) (uint64, error) {
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
if err != nil {
return 0, fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
}
defer mem.Close()
buf := make([]byte, 8)
if _, err := mem.ReadAt(buf, int64(addr)); err != nil {
return 0, fmt.Errorf("debug: read mem %#x: %w", addr, err)
}
return uint64(buf[0]) | uint64(buf[1])<<8 | uint64(buf[2])<<16 | uint64(buf[3])<<24 |
uint64(buf[4])<<32 | uint64(buf[5])<<40 | uint64(buf[6])<<48 | uint64(buf[7])<<56, nil
}
// Poke writes a word (8 bytes) to the debuggee's memory at addr.
func (s *Session) Poke(addr, val uint64) error {
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_WRONLY, 0)
if err != nil {
return fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
}
defer mem.Close()
buf := []byte{byte(val), byte(val >> 8), byte(val >> 16), byte(val >> 24),
byte(val >> 32), byte(val >> 40), byte(val >> 48), byte(val >> 56)}
if _, err := mem.WriteAt(buf, int64(addr)); err != nil {
return fmt.Errorf("debug: write mem %#x: %w", addr, err)
}
return nil
}
// ReadMemory reads len bytes from the debuggee's memory at addr.
func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) {
out := make([]byte, length)
for i := 0; i < length; i += 8 {
word, err := s.Peek(addr + uint64(i))
if err != nil {
return out[:i], err
}
for j := 0; j < 8 && i+j < length; j++ {
out[i+j] = byte(word >> (8 * j))
}
}
return out, nil
}
// WriteMemory writes bytes to the debuggee's memory at addr.
func (s *Session) WriteMemory(addr uint64, data []byte) error {
for i := 0; i < len(data); i += 8 {
end := min(i+8, len(data))
var word uint64
for j := 0; j < end-i; j++ {
word |= uint64(data[i+j]) << (8 * j)
}
if end-i < 8 {
existing, err := s.Peek(addr + uint64(i))
if err != nil {
return err
}
mask := ^((uint64(1) << (8 * (end - i))) - 1)
word = (existing & mask) | word
}
if err := s.Poke(addr+uint64(i), word); err != nil {
return err
}
}
return nil
}
// Step executes a single instruction in the debuggee.
func (s *Session) Step() error {
if s.exited {
return fmt.Errorf("debug: debuggee has exited")
}
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_SINGLESTEP),
uintptr(s.pid),
0, 0, 0, 0,
)
if errno != 0 {
return fmt.Errorf("debug: PTRACE_SINGLESTEP: %w", errno)
}
_, err := s.waitStopped()
return err
}
// Continue resumes execution until the next breakpoint or exit.
func (s *Session) Continue() error {
if s.exited {
return fmt.Errorf("debug: debuggee has exited")
}
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_CONT),
uintptr(s.pid),
0, 0, 0, 0,
)
if errno != 0 {
return fmt.Errorf("debug: PTRACE_CONT: %w", errno)
}
_, err := s.waitStopped()
return err
}
// Exited returns true if the debuggee has terminated.
func (s *Session) Exited() bool { return s.exited }
// Pid returns the debuggee's process ID.
func (s *Session) Pid() int { return s.pid }
// CodeBase returns the base address of the JIT code in the debuggee.
func (s *Session) CodeBase() uint64 { return s.codeBase }
// Kill terminates the debuggee.
func (s *Session) Kill() {
if !s.exited {
syscall.Kill(s.pid, syscall.SIGKILL)
syscall.Wait4(s.pid, nil, 0, nil)
s.exited = true
}
if s.cmd != nil && s.cmd.Process != nil {
s.cmd.Wait()
}
}
// findRWXMapping reads /proc/pid/maps and returns the base address of the
// first read-write-execute mapping (the JIT code region).
func findRWXMapping(pid int) uint64 {
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
if err != nil {
return 0
}
for line := range strings.SplitSeq(string(data), "\n") {
fields := strings.Fields(line)
if len(fields) < 2 {
continue
}
perms := fields[1]
if len(perms) >= 3 && perms[0] == 'r' && perms[1] == 'w' && perms[2] == 'x' {
var start uint64
fmt.Sscanf(fields[0], "%x-", &start)
return start
}
}
return 0
}
+2 -316
View File
@@ -3,165 +3,14 @@
//go:build linux && amd64 //go:build linux && amd64
// Package debug implements the interactive debugger for gasm (Phase 4):
// single-stepping, breakpoints, register and memory inspection for
// JIT-assembled Plan 9 amd64 functions, controlled via ptrace.
package debug package debug
import ( import (
"fmt" "fmt"
"os"
"os/exec"
"path/filepath"
"strings"
"syscall" "syscall"
"time"
"unsafe" "unsafe"
) )
// Session is a ptrace debugging session controlling one debuggee process.
type Session struct {
pid int
cmd *exec.Cmd
stopped bool
exited bool
codeBase uint64 // base address of the JIT code in the debuggee
wpSlots [4]bool // watchpoint slot occupancy (DR0-DR3)
}
// Launch starts the debuggee subprocess (gasm debug --target ...) and
// attaches to it via ptrace. The debuggee assembles the file, maps the
// JIT code, calls PTRACE_TRACEME and raises SIGSTOP; Launch waits for
// that initial stop and returns a ready Session.
func Launch(gasmBin, asmPath, funcName string, args []byte) (*Session, error) {
sess, _, err := LaunchWithBuffers(gasmBin, asmPath, funcName, args, "")
return sess, err
}
// LaunchWithBuffers is like Launch but also allocates buffers in the debuggee
// based on the buffer specification. Returns the Session and the buffer
// addresses (in the order they appear in the spec).
func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec string) (*Session, []uint64, error) {
self, err := os.Executable()
if err != nil {
return nil, nil, fmt.Errorf("debug: cannot find gasm binary: %w", err)
}
if gasmBin != "" {
self = gasmBin
}
// Write the arg block to a temp file (the child reads it).
tmpDir, err := os.MkdirTemp("", "gasm-debug-*")
if err != nil {
return nil, nil, fmt.Errorf("debug: tempdir: %w", err)
}
argsFile := filepath.Join(tmpDir, "args.bin")
if err := os.WriteFile(argsFile, args, 0o644); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: write args: %w", err)
}
// Write the buffer spec if present.
if bufSpec != "" {
if err := os.WriteFile(filepath.Join(tmpDir, "bufspec"), []byte(bufSpec), 0o644); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: write bufspec: %w", err)
}
}
cmd := exec.Command(self, "debug", "--target", "--func", funcName, "--args", argsFile, asmPath)
cmd.Env = append(os.Environ(), "GASM_DEBUG_TMP="+tmpDir)
cmd.Stdout = nil // output goes to the debugger, not the terminal
cmd.Stderr = os.Stderr
cmd.SysProcAttr = &syscall.SysProcAttr{}
if err := cmd.Start(); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: start debuggee: %w", err)
}
s := &Session{pid: cmd.Process.Pid, cmd: cmd}
// Wait for the child to signal readiness and stop. The child calls
// PTRACE_TRACEME then SIGSTOP, so Wait4 with WUNTRACED observes the
// ptrace-stop directly (no PTRACE_ATTACH needed).
readyFile := filepath.Join(tmpDir, "ready")
for i := 0; i < 500; i++ {
if _, err := os.Stat(readyFile); err == nil {
break
}
time.Sleep(5 * time.Millisecond)
}
var ws syscall.WaitStatus
if _, err := syscall.Wait4(s.pid, &ws, syscall.WUNTRACED, nil); err != nil {
cmd.Process.Kill()
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: wait for stop: %w", err)
}
// Wait for the debuggee to reach the function entry point.
entryFile := filepath.Join(tmpDir, "entry")
for i := 0; i < 500; i++ {
if _, err := os.Stat(entryFile); err == nil {
break
}
time.Sleep(5 * time.Millisecond)
}
// Continue the debuggee to the entry point.
if err := s.Continue(); err != nil {
return nil, nil, fmt.Errorf("debug: continue to entry: %w", err)
}
// Wait for the entry stop.
if _, err := syscall.Wait4(s.pid, &ws, syscall.WUNTRACED, nil); err != nil {
return nil, nil, fmt.Errorf("debug: wait for entry: %w", err)
}
s.stopped = true
// Read the code base from /proc/pid/maps (find the RWX mapping).
s.codeBase = findRWXMapping(s.pid)
if s.codeBase == 0 {
// Fallback: try the file the child wrote.
baseFile := filepath.Join(tmpDir, "codebase")
if data, err := os.ReadFile(baseFile); err == nil {
fmt.Sscanf(string(data), "%d", &s.codeBase)
}
}
// Read buffer addresses if buffers were allocated.
var bufAddrs []uint64
if bufSpec != "" {
addrFile := filepath.Join(tmpDir, "bufaddrs")
if data, err := os.ReadFile(addrFile); err == nil {
for _, line := range strings.Split(strings.TrimSpace(string(data)), "\n") {
var addr uint64
if _, err := fmt.Sscanf(line, "%d", &addr); err == nil {
bufAddrs = append(bufAddrs, addr)
}
}
}
}
return s, bufAddrs, nil
}
// wait waits for the debuggee to stop and returns the wait status.
func (s *Session) wait() error {
var ws syscall.WaitStatus
_, err := syscall.Wait4(s.pid, &ws, 0, nil)
if err != nil {
return err
}
if ws.Exited() {
s.exited = true
return fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
}
s.stopped = true
return nil
}
// GetRegs reads the general-purpose registers of the stopped debuggee. // GetRegs reads the general-purpose registers of the stopped debuggee.
func (s *Session) GetRegs() (Regs, error) { func (s *Session) GetRegs() (Regs, error) {
var regs Regs var regs Regs
@@ -234,179 +83,16 @@ type VectorRegs struct {
} }
// GetVectorRegs retrieves the YMM registers via PTRACE_GETREGSET + XSAVE. // GetVectorRegs retrieves the YMM registers via PTRACE_GETREGSET + XSAVE.
// Falls back to XMM if XSAVE is unavailable.
func (s *Session) GetVectorRegs() (VectorRegs, error) { func (s *Session) GetVectorRegs() (VectorRegs, error) {
var v VectorRegs var v VectorRegs
fp, err := s.GetFPRegs() fp, err := s.GetFPRegs()
if err != nil { if err != nil {
return v, err return v, err
} }
// PTRACE_GETFPREGS gives XMM registers (lower 128 bits). for i := range 16 {
// For YMM we'd need XSAVE; for now, copy XMM and zero the upper half. for j := range 16 {
for i := 0; i < 16; i++ {
for j := 0; j < 16; j++ {
v.YMM[i][j] = fp.XMM[i][j] v.YMM[i][j] = fp.XMM[i][j]
} }
// Upper 128 bits would come from XSAVE, not available via GETFPREGS.
} }
return v, nil return v, nil
} }
// Peek reads a word (8 bytes) from the debuggee's memory at addr.
// Uses /proc/pid/mem which works reliably with Go's multi-threaded runtime.
func (s *Session) Peek(addr uint64) (uint64, error) {
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
if err != nil {
return 0, fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
}
defer mem.Close()
buf := make([]byte, 8)
if _, err := mem.ReadAt(buf, int64(addr)); err != nil {
return 0, fmt.Errorf("debug: read mem %#x: %w", addr, err)
}
return uint64(buf[0]) | uint64(buf[1])<<8 | uint64(buf[2])<<16 | uint64(buf[3])<<24 |
uint64(buf[4])<<32 | uint64(buf[5])<<40 | uint64(buf[6])<<48 | uint64(buf[7])<<56, nil
}
// Poke writes a word (8 bytes) to the debuggee's memory at addr.
func (s *Session) Poke(addr, val uint64) error {
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_WRONLY, 0)
if err != nil {
return fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
}
defer mem.Close()
buf := []byte{byte(val), byte(val >> 8), byte(val >> 16), byte(val >> 24),
byte(val >> 32), byte(val >> 40), byte(val >> 48), byte(val >> 56)}
if _, err := mem.WriteAt(buf, int64(addr)); err != nil {
return fmt.Errorf("debug: write mem %#x: %w", addr, err)
}
return nil
}
// ReadMemory reads len bytes from the debuggee's memory at addr.
func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) {
out := make([]byte, length)
for i := 0; i < length; i += 8 {
word, err := s.Peek(addr + uint64(i))
if err != nil {
return out[:i], err
}
for j := 0; j < 8 && i+j < length; j++ {
out[i+j] = byte(word >> (8 * j))
}
}
return out, nil
}
// WriteMemory writes bytes to the debuggee's memory at addr.
func (s *Session) WriteMemory(addr uint64, data []byte) error {
for i := 0; i < len(data); i += 8 {
end := i + 8
if end > len(data) {
end = len(data)
}
var word uint64
for j := 0; j < end-i; j++ {
word |= uint64(data[i+j]) << (8 * j)
}
// For partial writes, read-modify-write the existing word.
if end-i < 8 {
existing, err := s.Peek(addr + uint64(i))
if err != nil {
return err
}
// Clear the bytes we're overwriting and merge.
mask := ^((uint64(1) << (8 * (end - i))) - 1)
word = (existing & mask) | word
}
if err := s.Poke(addr+uint64(i), word); err != nil {
return err
}
}
return nil
}
// Step executes a single instruction in the debuggee.
func (s *Session) Step() error {
if s.exited {
return fmt.Errorf("debug: debuggee has exited")
}
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_SINGLESTEP),
uintptr(s.pid),
0, 0, 0, 0,
)
if errno != 0 {
return fmt.Errorf("debug: PTRACE_SINGLESTEP: %w", errno)
}
return s.wait()
}
// Continue resumes execution until the next breakpoint or exit.
func (s *Session) Continue() error {
if s.exited {
return fmt.Errorf("debug: debuggee has exited")
}
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_CONT),
uintptr(s.pid),
0, 0, 0, 0,
)
if errno != 0 {
return fmt.Errorf("debug: PTRACE_CONT: %w", errno)
}
return s.wait()
}
// Exited returns true if the debuggee has terminated.
func (s *Session) Exited() bool {
return s.exited
}
// Pid returns the debuggee's process ID.
func (s *Session) Pid() int {
return s.pid
}
// CodeBase returns the base address of the JIT code in the debuggee.
func (s *Session) CodeBase() uint64 {
return s.codeBase
}
// Kill terminates the debuggee.
func (s *Session) Kill() {
if !s.exited {
syscall.Kill(s.pid, syscall.SIGKILL)
syscall.Wait4(s.pid, nil, 0, nil)
s.exited = true
}
if s.cmd != nil && s.cmd.Process != nil {
s.cmd.Wait()
}
}
// findRWXMapping reads /proc/pid/maps and returns the base address of the
// first read-write-execute mapping (the JIT code region).
func findRWXMapping(pid int) uint64 {
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
if err != nil {
return 0
}
for _, line := range strings.Split(string(data), "\n") {
// Format: addr-addr perms offset dev inode pathname
fields := strings.Fields(line)
if len(fields) < 2 {
continue
}
perms := fields[1]
if len(perms) >= 3 && perms[0] == 'r' && perms[1] == 'w' && perms[2] == 'x' {
// Parse the start address.
var start uint64
fmt.Sscanf(fields[0], "%x-", &start)
return start
}
}
return 0
}
+94
View File
@@ -0,0 +1,94 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && arm64
package debug
import (
"fmt"
"syscall"
"unsafe"
)
// GetRegs reads the general-purpose registers of the stopped debuggee.
func (s *Session) GetRegs() (Regs, error) {
var regs Regs
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_GETREGS),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(&regs)),
0, 0,
)
if errno != 0 {
return regs, fmt.Errorf("debug: PTRACE_GETREGS: %w", errno)
}
return regs, nil
}
// SetRegs writes the general-purpose registers of the stopped debuggee.
func (s *Session) SetRegs(regs *Regs) error {
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_SETREGS),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(regs)),
0, 0,
)
if errno != 0 {
return fmt.Errorf("debug: PTRACE_SETREGS: %w", errno)
}
return nil
}
// ntPrFPREG is the NT_PRFPREG note type (ELF NT ver): the FP register set.
const ntPrFPREG = 0x2
// FPRegs holds the arm64 FP/NEON register state, matching the kernel's
// user_fpsimd_struct layout (32 128-bit V registers, then FPSR and FPCR).
type FPRegs struct {
V [32][16]byte // V0-V31 (128-bit NEON/FP registers)
FPSR uint32
FPCR uint32
}
// GetFPRegs retrieves the FP register state via PTRACE_GETREGSET with
// NT_PRFPREG (this architecture has no PTRACE_GETFPREGS request).
func (s *Session) GetFPRegs() (FPRegs, error) {
var fp FPRegs
iovec := syscall.Iovec{
Base: (*byte)(unsafe.Pointer(&fp)),
Len: uint64(unsafe.Sizeof(fp)),
}
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_GETREGSET),
uintptr(s.pid),
uintptr(ntPrFPREG),
uintptr(unsafe.Pointer(&iovec)),
0, 0,
)
if errno != 0 {
return fp, fmt.Errorf("debug: PTRACE_GETREGSET (NT_PRFPREG): %w", errno)
}
return fp, nil
}
// VectorRegs holds the full SIMD register state.
type VectorRegs struct {
V [32][16]byte // V0-V31 (128-bit)
}
// GetVectorRegs retrieves the SIMD registers.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
var v VectorRegs
fp, err := s.GetFPRegs()
if err != nil {
return v, err
}
copy(v.V[:][:], fp.V[:][:])
return v, nil
}
+97
View File
@@ -0,0 +1,97 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && loong64
package debug
import (
"fmt"
"syscall"
"unsafe"
)
// GetRegs reads the general-purpose registers of the stopped debuggee.
func (s *Session) GetRegs() (Regs, error) {
var regs Regs
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_GETREGS),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(&regs)),
0, 0,
)
if errno != 0 {
return regs, fmt.Errorf("debug: PTRACE_GETREGS: %w", errno)
}
return regs, nil
}
// SetRegs writes the general-purpose registers of the stopped debuggee.
func (s *Session) SetRegs(regs *Regs) error {
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_SETREGS),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(regs)),
0, 0,
)
if errno != 0 {
return fmt.Errorf("debug: PTRACE_SETREGS: %w", errno)
}
return nil
}
// ntPrFPREG is the NT_PRFPREG note type (ELF NT ver): the FP register set.
const ntPrFPREG = 0x2
// FPRegs holds the LoongArch FP register state, matching the kernel's
// user_fp_struct layout (32 64-bit FP registers, the fcc condition flags,
// and fcsr).
type FPRegs struct {
F [32]uint64 // F0-F31 (64-bit FP registers)
FCC uint64 // eight per-register condition flags, packed
FCSR uint32
}
// GetFPRegs retrieves the FP register state via PTRACE_GETREGSET with
// NT_PRFPREG (this architecture has no PTRACE_GETFPREGS request).
func (s *Session) GetFPRegs() (FPRegs, error) {
var fp FPRegs
iovec := syscall.Iovec{
Base: (*byte)(unsafe.Pointer(&fp)),
Len: uint64(unsafe.Sizeof(fp)),
}
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_GETREGSET),
uintptr(s.pid),
uintptr(ntPrFPREG),
uintptr(unsafe.Pointer(&iovec)),
0, 0,
)
if errno != 0 {
return fp, fmt.Errorf("debug: PTRACE_GETREGSET (NT_PRFPREG): %w", errno)
}
return fp, nil
}
// VectorRegs holds the FP register state shown by the regs command
// (the scalar FP subset: 32 64-bit registers, fcc and fcsr; the LSX/LASX
// vector files are not read yet).
type VectorRegs struct {
F [32]uint64
FCC uint64
FCSR uint32
}
// GetVectorRegs retrieves the FP registers.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
fp, err := s.GetFPRegs()
if err != nil {
return VectorRegs{}, err
}
return VectorRegs{F: fp.F, FCC: fp.FCC, FCSR: fp.FCSR}, nil
}
+93
View File
@@ -0,0 +1,93 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && riscv64
package debug
import (
"fmt"
"syscall"
"unsafe"
)
// GetRegs reads the general-purpose registers of the stopped debuggee.
func (s *Session) GetRegs() (Regs, error) {
var regs Regs
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_GETREGS),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(&regs)),
0, 0,
)
if errno != 0 {
return regs, fmt.Errorf("debug: PTRACE_GETREGS: %w", errno)
}
return regs, nil
}
// SetRegs writes the general-purpose registers of the stopped debuggee.
func (s *Session) SetRegs(regs *Regs) error {
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_SETREGS),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(regs)),
0, 0,
)
if errno != 0 {
return fmt.Errorf("debug: PTRACE_SETREGS: %w", errno)
}
return nil
}
// ntPrFPREG is the NT_PRFPREG note type (ELF NT ver): the FP register set.
const ntPrFPREG = 0x2
// FPRegs holds the RISC-V FP register state, matching the kernel's
// user_fp_struct layout (32 64-bit FP registers plus fcsr).
type FPRegs struct {
F [32]uint64 // F0-F31 (64-bit FP registers)
FCSR uint32
}
// GetFPRegs retrieves the FP register state via PTRACE_GETREGSET with
// NT_PRFPREG (this architecture has no PTRACE_GETFPREGS request).
func (s *Session) GetFPRegs() (FPRegs, error) {
var fp FPRegs
iovec := syscall.Iovec{
Base: (*byte)(unsafe.Pointer(&fp)),
Len: uint64(unsafe.Sizeof(fp)),
}
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_GETREGSET),
uintptr(s.pid),
uintptr(ntPrFPREG),
uintptr(unsafe.Pointer(&iovec)),
0, 0,
)
if errno != 0 {
return fp, fmt.Errorf("debug: PTRACE_GETREGSET (NT_PRFPREG): %w", errno)
}
return fp, nil
}
// VectorRegs holds the FP register state shown by the regs command
// (riscv64 has 32 64-bit FP registers and fcsr).
type VectorRegs struct {
F [32]uint64
FCSR uint32
}
// GetVectorRegs retrieves the FP registers.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
fp, err := s.GetFPRegs()
if err != nil {
return VectorRegs{}, err
}
return VectorRegs{F: fp.F, FCSR: fp.FCSR}, nil
}
+95
View File
@@ -0,0 +1,95 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug
// Regs holds the full general-purpose register set of a traced process
// (the Linux amd64 user_regs_struct layout).
type Regs struct {
R15 uint64
R14 uint64
R13 uint64
R12 uint64
RBP uint64
RBX uint64
R11 uint64
R10 uint64
R9 uint64
R8 uint64
RAX uint64
RCX uint64
RDX uint64
RSI uint64
RDI uint64
OrigRAX uint64
RIP uint64
CS uint64
RFLAGS uint64
RSP uint64
SS uint64
FSBase uint64
GSBase uint64
DS uint64
ES uint64
FS uint64
GS uint64
}
// GetPC returns the program counter.
func (r *Regs) GetPC() uint64 { return r.RIP }
// SetPC sets the program counter.
func (r *Regs) SetPC(pc uint64) { r.RIP = pc }
// GetSP returns the stack pointer.
func (r *Regs) GetSP() uint64 { return r.RSP }
// RegValue returns the value of the named register, or false if unknown.
func (r *Regs) RegValue(name string) (uint64, bool) {
switch name {
case "rax", "eax", "ax", "al":
return r.RAX, true
case "rbx", "ebx", "bx", "bl":
return r.RBX, true
case "rcx", "ecx", "cx", "cl":
return r.RCX, true
case "rdx", "edx", "dx", "dl":
return r.RDX, true
case "rsi", "esi", "si":
return r.RSI, true
case "rdi", "edi", "di":
return r.RDI, true
case "rbp", "ebp", "bp":
return r.RBP, true
case "rsp", "esp", "sp":
return r.RSP, true
case "r8":
return r.R8, true
case "r9":
return r.R9, true
case "r10":
return r.R10, true
case "r11":
return r.R11, true
case "r12":
return r.R12, true
case "r13":
return r.R13, true
case "r14":
return r.R14, true
case "r15":
return r.R15, true
case "rip", "eip":
return r.RIP, true
default:
return 0, false
}
}
// breakpointInsn is the software breakpoint instruction.
var breakpointInsn = []byte{0xCC} // INT3
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
const breakpointPCAdjust = 1
+134
View File
@@ -0,0 +1,134 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && arm64
package debug
// Regs holds the full general-purpose register set of a traced process
// (the Linux arm64 user_pt_regs layout).
type Regs struct {
X0 uint64
X1 uint64
X2 uint64
X3 uint64
X4 uint64
X5 uint64
X6 uint64
X7 uint64
X8 uint64
X9 uint64
X10 uint64
X11 uint64
X12 uint64
X13 uint64
X14 uint64
X15 uint64
X16 uint64
X17 uint64
X18 uint64
X19 uint64
X20 uint64
X21 uint64
X22 uint64
X23 uint64
X24 uint64
X25 uint64
X26 uint64
X27 uint64
X28 uint64
X29 uint64 // FP (frame pointer)
X30 uint64 // LR (link register)
SP uint64
PC uint64
PSTATE uint64
}
// PC returns the program counter.
func (r *Regs) GetPC() uint64 { return r.PC }
// SetPC sets the program counter.
func (r *Regs) SetPC(pc uint64) { r.PC = pc }
// GetSP returns the stack pointer.
func (r *Regs) GetSP() uint64 { return r.SP }
// RegValue returns the value of the named register, or false if unknown.
func (r *Regs) RegValue(name string) (uint64, bool) {
switch name {
case "x0":
return r.X0, true
case "x1":
return r.X1, true
case "x2":
return r.X2, true
case "x3":
return r.X3, true
case "x4":
return r.X4, true
case "x5":
return r.X5, true
case "x6":
return r.X6, true
case "x7":
return r.X7, true
case "x8":
return r.X8, true
case "x9":
return r.X9, true
case "x10":
return r.X10, true
case "x11":
return r.X11, true
case "x12":
return r.X12, true
case "x13":
return r.X13, true
case "x14":
return r.X14, true
case "x15":
return r.X15, true
case "x16":
return r.X16, true
case "x17":
return r.X17, true
case "x18":
return r.X18, true
case "x19":
return r.X19, true
case "x20":
return r.X20, true
case "x21":
return r.X21, true
case "x22":
return r.X22, true
case "x23":
return r.X23, true
case "x24":
return r.X24, true
case "x25":
return r.X25, true
case "x26":
return r.X26, true
case "x27":
return r.X27, true
case "x28":
return r.X28, true
case "x29", "fp":
return r.X29, true
case "x30", "lr":
return r.X30, true
case "sp":
return r.SP, true
case "pc":
return r.PC, true
default:
return 0, false
}
}
// breakpointInsn is the software breakpoint instruction (BRK #0).
var breakpointInsn = []byte{0x00, 0x00, 0x20, 0xD4} // BRK #0
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
const breakpointPCAdjust = 4
+130
View File
@@ -0,0 +1,130 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && loong64
package debug
// Regs holds the full general-purpose register set of a traced process
// (the Linux loong64 user_pt_regs layout).
type Regs struct {
R0 uint64 // zero
R1 uint64 // RA (return address)
R2 uint64 // TP (thread pointer)
R3 uint64 // SP (stack pointer)
R4 uint64 // A0
R5 uint64 // A1
R6 uint64 // A2
R7 uint64 // A3
R8 uint64 // A4
R9 uint64 // A5
R10 uint64 // A6
R11 uint64 // A7
R12 uint64 // T0
R13 uint64 // T1
R14 uint64 // T2
R15 uint64 // T3
R16 uint64 // T4
R17 uint64 // T5
R18 uint64 // T6
R19 uint64 // T7
R20 uint64 // T8
R21 uint64 // FP (frame pointer)
R22 uint64 // S0
R23 uint64 // S1
R24 uint64 // S2
R25 uint64 // S3
R26 uint64 // S4
R27 uint64 // S5
R28 uint64 // S6
R29 uint64 // S7
R30 uint64 // S8
R31 uint64 // PC
}
// GetPC returns the program counter.
func (r *Regs) GetPC() uint64 { return r.R31 }
// SetPC sets the program counter.
func (r *Regs) SetPC(pc uint64) { r.R31 = pc }
// GetSP returns the stack pointer.
func (r *Regs) GetSP() uint64 { return r.R3 }
// RegValue returns the value of the named register, or false if unknown.
func (r *Regs) RegValue(name string) (uint64, bool) {
switch name {
case "r0", "zero":
return r.R0, true
case "r1", "ra":
return r.R1, true
case "r2", "tp":
return r.R2, true
case "r3", "sp":
return r.R3, true
case "r4", "a0":
return r.R4, true
case "r5", "a1":
return r.R5, true
case "r6", "a2":
return r.R6, true
case "r7", "a3":
return r.R7, true
case "r8", "a4":
return r.R8, true
case "r9", "a5":
return r.R9, true
case "r10", "a6":
return r.R10, true
case "r11", "a7":
return r.R11, true
case "r12", "t0":
return r.R12, true
case "r13", "t1":
return r.R13, true
case "r14", "t2":
return r.R14, true
case "r15", "t3":
return r.R15, true
case "r16", "t4":
return r.R16, true
case "r17", "t5":
return r.R17, true
case "r18", "t6":
return r.R18, true
case "r19", "t7":
return r.R19, true
case "r20", "t8":
return r.R20, true
case "r21", "fp":
return r.R21, true
case "r22", "s0":
return r.R22, true
case "r23", "s1":
return r.R23, true
case "r24", "s2":
return r.R24, true
case "r25", "s3":
return r.R25, true
case "r26", "s4":
return r.R26, true
case "r27", "s5":
return r.R27, true
case "r28", "s6":
return r.R28, true
case "r29", "s7":
return r.R29, true
case "r30", "s8":
return r.R30, true
case "r31", "pc":
return r.R31, true
default:
return 0, false
}
}
// breakpointInsn is the software breakpoint instruction (BRK $0).
var breakpointInsn = []byte{0x05, 0x00, 0x2a, 0x00} // break 0
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
const breakpointPCAdjust = 4
+130
View File
@@ -0,0 +1,130 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && riscv64
package debug
// Regs holds the full general-purpose register set of a traced process
// (the Linux riscv64 user_regs_struct layout).
type Regs struct {
PC uint64
Ra uint64 // x1 (return address)
Sp uint64 // x2
Gp uint64 // x3
Tp uint64 // x4
T0 uint64 // x5
T1 uint64 // x6
T2 uint64 // x7
S0 uint64 // x8 (frame pointer)
S1 uint64 // x9
A0 uint64 // x10
A1 uint64 // x11
A2 uint64 // x12
A3 uint64 // x13
A4 uint64 // x14
A5 uint64 // x15
A6 uint64 // x16
A7 uint64 // x17
S2 uint64 // x18
S3 uint64 // x19
S4 uint64 // x20
S5 uint64 // x21
S6 uint64 // x22
S7 uint64 // x23
S8 uint64 // x24
S9 uint64 // x25
S10 uint64 // x26
S11 uint64 // x27
T3 uint64 // x28
T4 uint64 // x29
T5 uint64 // x30
T6 uint64 // x31
}
// GetPC returns the program counter.
func (r *Regs) GetPC() uint64 { return r.PC }
// SetPC sets the program counter.
func (r *Regs) SetPC(pc uint64) { r.PC = pc }
// GetSP returns the stack pointer.
func (r *Regs) GetSP() uint64 { return r.Sp }
// RegValue returns the value of the named register, or false if unknown.
func (r *Regs) RegValue(name string) (uint64, bool) {
switch name {
case "pc":
return r.PC, true
case "ra", "x1":
return r.Ra, true
case "sp", "x2":
return r.Sp, true
case "gp", "x3":
return r.Gp, true
case "tp", "x4":
return r.Tp, true
case "t0", "x5":
return r.T0, true
case "t1", "x6":
return r.T1, true
case "t2", "x7":
return r.T2, true
case "s0", "fp", "x8":
return r.S0, true
case "s1", "x9":
return r.S1, true
case "a0", "x10":
return r.A0, true
case "a1", "x11":
return r.A1, true
case "a2", "x12":
return r.A2, true
case "a3", "x13":
return r.A3, true
case "a4", "x14":
return r.A4, true
case "a5", "x15":
return r.A5, true
case "a6", "x16":
return r.A6, true
case "a7", "x17":
return r.A7, true
case "s2", "x18":
return r.S2, true
case "s3", "x19":
return r.S3, true
case "s4", "x20":
return r.S4, true
case "s5", "x21":
return r.S5, true
case "s6", "x22":
return r.S6, true
case "s7", "x23":
return r.S7, true
case "s8", "x24":
return r.S8, true
case "s9", "x25":
return r.S9, true
case "s10", "x26":
return r.S10, true
case "s11", "x27":
return r.S11, true
case "t3", "x28":
return r.T3, true
case "t4", "x29":
return r.T4, true
case "t5", "x30":
return r.T5, true
case "t6", "x31":
return r.T6, true
default:
return 0, false
}
}
// breakpointInsn is the software breakpoint instruction (EBREAK).
var breakpointInsn = []byte{0x73, 0x00, 0x10, 0x00} // ebreak
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
const breakpointPCAdjust = 4
+53 -143
View File
@@ -1,14 +1,14 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause // SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64 //go:build linux
package debug package debug
import ( import (
"bufio" "bufio"
"fmt" "fmt"
"os" "io"
"sort" "sort"
"strconv" "strconv"
"strings" "strings"
@@ -26,19 +26,15 @@ type SourceLine struct {
Line int Line int
} }
// REPL runs the interactive debugger loop. On entry, the debuggee is // REPL runs the interactive debugger loop, reading commands from in (pass
// stopped in the Go runtime (after PTRACE_TRACEME + SIGSTOP). The REPL // os.Stdin interactively, or a bytes.Reader/script file for headless runs).
// sets a temporary breakpoint at the function entry, continues to it, and func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, argsSize int, labels []Label, lines []SourceLine, in io.Reader) {
// then presents the prompt — so the user starts debugging at the first
// instruction of the assembled function.
func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, argsSize int, labels []Label, lines []SourceLine) {
entryAddr := codeBase + uint64(funcOffset) entryAddr := codeBase + uint64(funcOffset)
// The debuggee is already stopped at the function entry point.
fmt.Printf("stopped at function entry: %#x (%d bytes)\n", entryAddr, funcSize) fmt.Printf("stopped at function entry: %#x (%d bytes)\n", entryAddr, funcSize)
fmt.Println("commands: break <label|addr> | step [n] | continue | disas [n] | regs | where | x <addr> [len] | w <addr> <val...> | labels | quit") fmt.Println("commands: break <label|addr> | step [n] | continue | disas [n] | regs | where | x <addr> [len] | w <addr> <val...> | labels | quit")
scanner := bufio.NewScanner(os.Stdin) scanner := bufio.NewScanner(in)
for { for {
fmt.Print("(gasm) ") fmt.Print("(gasm) ")
@@ -64,7 +60,6 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
continue continue
} }
printRegs(&regs, codeBase, uint64(funcOffset)) printRegs(&regs, codeBase, uint64(funcOffset))
// Also show vector registers.
vregs, err := s.GetVectorRegs() vregs, err := s.GetVectorRegs()
if err != nil { if err != nil {
fmt.Printf(" (vector regs unavailable: %v)\n", err) fmt.Printf(" (vector regs unavailable: %v)\n", err)
@@ -77,7 +72,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
if len(parts) > 1 { if len(parts) > 1 {
n, _ = strconv.Atoi(parts[1]) n, _ = strconv.Atoi(parts[1])
} }
for i := 0; i < n; i++ { for range n {
if s.Exited() { if s.Exited() {
fmt.Println("debuggee exited") fmt.Println("debuggee exited")
break break
@@ -89,24 +84,22 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
} }
if !s.Exited() { if !s.Exited() {
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
text, _, _ := s.Disassemble(regs.RIP) pc := regs.GetPC()
fmt.Printf("=> %#x (func+%#x): %s\n", regs.RIP, regs.RIP-codeBase-uint64(funcOffset), text) text, _, _ := s.Disassemble(pc)
fmt.Printf("=> %#x (func+%#x): %s\n", pc, pc-codeBase-uint64(funcOffset), text)
} }
case "next", "n": case "next", "n":
// Step over: if the current instruction is a CALL, set a
// breakpoint after it and continue; otherwise single-step.
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
text, instLen, _ := s.Disassemble(regs.RIP) pc := regs.GetPC()
if strings.HasPrefix(strings.ToLower(text), "call") { text, instLen, _ := s.Disassemble(pc)
// Set a temporary breakpoint after the CALL. if strings.HasPrefix(strings.ToLower(text), "call") || strings.HasPrefix(strings.ToLower(text), "bl") {
afterAddr := regs.RIP + uint64(instLen) afterAddr := pc + uint64(instLen)
bp, err := bm.Set(afterAddr, "(next)") _, err := bm.Set(afterAddr, "(next)")
if err != nil { if err != nil {
fmt.Printf("cannot set next breakpoint: %v\n", err) fmt.Printf("cannot set next breakpoint: %v\n", err)
continue continue
} }
// Continue until the breakpoint.
for _, b := range bm.All() { for _, b := range bm.All() {
bm.Reinsert(b.Addr) bm.Reinsert(b.Addr)
} }
@@ -117,9 +110,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
} }
bm.HandleTrap(&regs) bm.HandleTrap(&regs)
bm.Clear(afterAddr) bm.Clear(afterAddr)
_ = bp
} else { } else {
// Not a CALL — just single-step.
if err := s.Step(); err != nil { if err := s.Step(); err != nil {
fmt.Println(err) fmt.Println(err)
continue continue
@@ -127,26 +118,23 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
} }
if !s.Exited() { if !s.Exited() {
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
text, _, _ := s.Disassemble(regs.RIP) pc := regs.GetPC()
fmt.Printf("=> %#x (func+%#x): %s\n", regs.RIP, regs.RIP-codeBase-uint64(funcOffset), text) text, _, _ := s.Disassemble(pc)
fmt.Printf("=> %#x (func+%#x): %s\n", pc, pc-codeBase-uint64(funcOffset), text)
} }
case "finish", "fin": case "finish", "fin":
// Run until the current function returns.
// For NOSPLIT frame=0: return address is at [RSP].
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
retAddr, err := s.Peek(regs.RSP) retAddr, err := archReturnAddr(s, &regs)
if err != nil { if err != nil {
fmt.Printf("cannot read return address: %v\n", err) fmt.Printf("cannot read return address: %v\n", err)
continue continue
} }
// Set a temporary breakpoint at the return address. _, err = bm.Set(retAddr, "(finish)")
bp, err := bm.Set(retAddr, "(finish)")
if err != nil { if err != nil {
fmt.Printf("cannot set finish breakpoint: %v\n", err) fmt.Printf("cannot set finish breakpoint: %v\n", err)
continue continue
} }
// Continue until the breakpoint.
for _, b := range bm.All() { for _, b := range bm.All() {
bm.Reinsert(b.Addr) bm.Reinsert(b.Addr)
} }
@@ -159,12 +147,11 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
bm.HandleTrap(&regs) bm.HandleTrap(&regs)
} }
bm.Clear(retAddr) bm.Clear(retAddr)
_ = bp
if s.Exited() { if s.Exited() {
fmt.Println("debuggee exited") fmt.Println("debuggee exited")
} else { } else {
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
fmt.Printf("finished, now at %#x\n", regs.RIP) fmt.Printf("finished, now at %#x\n", regs.GetPC())
} }
case "continue", "c": case "continue", "c":
@@ -172,9 +159,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Println("debuggee exited") fmt.Println("debuggee exited")
continue continue
} }
// Loop: continue until a breakpoint fires (condition met) or exit.
for { for {
// Re-insert all breakpoints before continuing.
for _, bp := range bm.All() { for _, bp := range bm.All() {
bm.Reinsert(bp.Addr) bm.Reinsert(bp.Addr)
} }
@@ -186,7 +171,6 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Println("debuggee exited") fmt.Println("debuggee exited")
break break
} }
// Check for watchpoint hits.
reason, wpAddr := s.StopInfo() reason, wpAddr := s.StopInfo()
if reason == StopWatchpoint { if reason == StopWatchpoint {
fmt.Printf("watchpoint hit at %#x\n", wpAddr) fmt.Printf("watchpoint hit at %#x\n", wpAddr)
@@ -194,6 +178,13 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
} }
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
if bp := bm.HandleTrap(&regs); bp != nil { if bp := bm.HandleTrap(&regs); bp != nil {
// Execute the instruction under the restored breakpoint
// so the next continue cannot re-trap on the same
// breakpoint; the process parks right after it.
if err := s.Step(); err != nil {
fmt.Println(err)
break
}
name := bp.Label name := bp.Label
if name == "" { if name == "" {
name = fmt.Sprintf("%#x", bp.Addr) name = fmt.Sprintf("%#x", bp.Addr)
@@ -201,7 +192,6 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Printf("breakpoint hit: %s (func+%#x)\n", name, bp.Addr-codeBase-uint64(funcOffset)) fmt.Printf("breakpoint hit: %s (func+%#x)\n", name, bp.Addr-codeBase-uint64(funcOffset))
break break
} }
// Condition not met (or single-step trap) — re-insert and continue.
} }
case "break", "b": case "break", "b":
@@ -209,11 +199,9 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Println("usage: break <label|addr|line> [if <reg> <op> <val>]") fmt.Println("usage: break <label|addr|line> [if <reg> <op> <val>]")
continue continue
} }
// Try as a line number first.
var addr uint64 var addr uint64
var label string var label string
if lineNum, err := strconv.Atoi(parts[1]); err == nil && lineNum > 0 { if lineNum, err := strconv.Atoi(parts[1]); err == nil && lineNum > 0 {
// Find the byte offset for this line.
off := offsetForLine(lines, lineNum) off := offsetForLine(lines, lineNum)
if off < 0 { if off < 0 {
fmt.Printf("no instruction at line %d\n", lineNum) fmt.Printf("no instruction at line %d\n", lineNum)
@@ -228,17 +216,18 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Printf("unknown label, address, or line: %s\n", parts[1]) fmt.Printf("unknown label, address, or line: %s\n", parts[1])
continue continue
} }
// Parse optional condition: "if <reg> <op> <value>"
var cond *Condition var cond *Condition
if len(parts) >= 6 && parts[2] == "if" { if len(parts) >= 6 && parts[2] == "if" {
val, err := strconv.ParseUint(parts[5], 0, 64) reg := strings.ToLower(parts[3])
if err != nil { op := parts[4]
fmt.Printf("invalid condition value: %s\n", parts[5]) operand := parts[5]
continue if val, err := strconv.ParseUint(operand, 0, 64); err == nil {
cond = &Condition{Reg: reg, Op: op, Value: val}
} else {
cond = &Condition{Reg: reg, Op: op, Reg2: strings.ToLower(operand)}
} }
cond = &Condition{Reg: strings.ToLower(parts[3]), Op: parts[4], Value: val}
} else if len(parts) >= 4 && parts[2] == "if" { } else if len(parts) >= 4 && parts[2] == "if" {
fmt.Println("usage: break <label|addr> if <reg> <op> <value>") fmt.Println("usage: break <label|addr> if <reg> <op> <value|reg>")
continue continue
} }
bp, err := bm.SetWithCond(addr, label, cond) bp, err := bm.SetWithCond(addr, label, cond)
@@ -282,7 +271,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
case "x": case "x":
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
addr := regs.RIP // default: current PC addr := regs.GetPC()
length := 64 length := 64
if len(parts) > 1 { if len(parts) > 1 {
addr, _ = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels) addr, _ = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
@@ -314,9 +303,8 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Printf("invalid value: %s\n", arg) fmt.Printf("invalid value: %s\n", arg)
continue continue
} }
// Write as 8-byte word if it looks like a large value, else single byte.
if v > 255 { if v > 255 {
for j := 0; j < 8; j++ { for j := range 8 {
bytes = append(bytes, byte(v>>(8*j))) bytes = append(bytes, byte(v>>(8*j)))
} }
} else { } else {
@@ -364,11 +352,11 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
} }
} }
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
fmt.Print(s.DisassembleN(regs.RIP, n)) fmt.Print(s.DisassembleN(regs.GetPC(), n))
case "where": case "where":
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
funcOff := int(regs.RIP - codeBase - uint64(funcOffset)) funcOff := int(regs.GetPC() - codeBase - uint64(funcOffset))
line := lineAt(lines, funcOff) line := lineAt(lines, funcOff)
label := nearestLabel(labels, funcOff) label := nearestLabel(labels, funcOff)
fmt.Printf(" func+%#x", funcOff) fmt.Printf(" func+%#x", funcOff)
@@ -381,32 +369,32 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Println() fmt.Println()
case "help", "h", "?": case "help", "h", "?":
fmt.Println(` break <label|addr> [if <reg> <op> <val>] set a breakpoint fmt.Printf(` break <label|addr> [if <reg> <op> <val>] set a breakpoint
delete <label|addr> remove a breakpoint delete <label|addr> remove a breakpoint
info break list all breakpoints info break list all breakpoints
watch <addr> [r|w] [size] set a hardware watchpoint (write by default) watch <addr> [r|w] [size] set a hardware watchpoint (write by default)
unwatch [<slot>] clear one or all watchpoints unwatch [<slot>] clear one or all watchpoints
step [n], s single-step n instructions step [n], s single-step n instructions
next, n step over CALL next, n step over CALL/BL
continue, c run until breakpoint or exit continue, c run until breakpoint or exit
disas [n], u disassemble n instructions at PC disas [n], u disassemble n instructions at PC
regs print registers and RFLAGS regs print registers
where show source line and nearest label where show source line and nearest label
stack show stack near RSP (args + return address) stack show stack near %s (args + return address)
x [addr] [len] hex-dump memory x [addr] [len] hex-dump memory
w <addr> <val...> write bytes to memory w <addr> <val...> write bytes to memory
labels, l list function labels labels, l list function labels
help, h, ? this help help, h, ? this help
quit, q kill debuggee and exit`) quit, q kill debuggee and exit`, archSPLabel())
case "stack": case "stack":
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
// For NOSPLIT frame=0: [RSP] = return address, [RSP+8..] = args. sp := regs.GetSP()
retAddr, _ := s.Peek(regs.RSP) retAddr, _ := archReturnAddr(s, &regs)
fmt.Printf(" [RSP] return addr = %#x\n", retAddr) fmt.Printf(" [%s] return addr = %#x\n", archSPLabel(), retAddr)
if argsSize > 0 { if argsSize > 0 {
fmt.Printf(" args (%d bytes at RSP+8):\n", argsSize) fmt.Printf(" args (%d bytes at %s+8):\n", argsSize, archSPLabel())
argBytes, err := s.ReadMemory(regs.RSP+8, argsSize) argBytes, err := s.ReadMemory(sp+8, argsSize)
if err == nil { if err == nil {
for i := 0; i < argsSize; i += 8 { for i := 0; i < argsSize; i += 8 {
var v uint64 var v uint64
@@ -420,7 +408,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
case "bt", "backtrace": case "bt", "backtrace":
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
funcOff := int(regs.RIP - codeBase - uint64(funcOffset)) funcOff := int(regs.GetPC() - codeBase - uint64(funcOffset))
line := lineAt(lines, funcOff) line := lineAt(lines, funcOff)
label := nearestLabel(labels, funcOff) label := nearestLabel(labels, funcOff)
fmt.Printf(" #0 func+%#x", funcOff) fmt.Printf(" #0 func+%#x", funcOff)
@@ -431,7 +419,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Printf(" [line %d]", line) fmt.Printf(" [line %d]", line)
} }
fmt.Println() fmt.Println()
retAddr, _ := s.Peek(regs.RSP) retAddr, _ := archReturnAddr(s, &regs)
fmt.Printf(" #1 return to %#x\n", retAddr) fmt.Printf(" #1 return to %#x\n", retAddr)
case "watch": case "watch":
@@ -499,80 +487,9 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
s.Kill() s.Kill()
} }
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" RIP = %#016x (func+%#x)\n", regs.RIP, regs.RIP-codeBase-funcOff)
fmt.Printf(" RSP = %#016x RBP = %#016x\n", regs.RSP, regs.RBP)
fmt.Printf(" RAX = %#016x RBX = %#016x\n", regs.RAX, regs.RBX)
fmt.Printf(" RCX = %#016x RDX = %#016x\n", regs.RCX, regs.RDX)
fmt.Printf(" RSI = %#016x RDI = %#016x\n", regs.RSI, regs.RDI)
fmt.Printf(" R8 = %#016x R9 = %#016x\n", regs.R8, regs.R9)
fmt.Printf(" R10 = %#016x R11 = %#016x\n", regs.R10, regs.R11)
fmt.Printf(" R12 = %#016x R13 = %#016x\n", regs.R12, regs.R13)
fmt.Printf(" R14 = %#016x R15 = %#016x\n", regs.R14, regs.R15)
fmt.Printf(" RFLAGS = %#x [%s]\n", regs.RFLAGS, decodeRflags(regs.RFLAGS))
}
// printVectorRegs displays the YMM registers.
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n Vector registers (YMM):")
for i := 0; i < 16; i += 2 {
fmt.Printf(" YMM%-2d = ", i)
printYMM(v.YMM[i][:])
fmt.Printf(" YMM%-2d = ", i+1)
printYMM(v.YMM[i+1][:])
fmt.Println()
}
}
func printYMM(b []byte) {
// Show as 8 32-bit values.
for j := 0; j < 32; j += 4 {
v := uint32(b[j]) | uint32(b[j+1])<<8 | uint32(b[j+2])<<16 | uint32(b[j+3])<<24
fmt.Printf("%08x ", v)
}
}
func decodeRflags(f uint64) string {
var flags string
if f&1 != 0 {
flags += "CF "
}
if f&(1<<2) != 0 {
flags += "PF "
}
if f&(1<<4) != 0 {
flags += "AF "
}
if f&(1<<6) != 0 {
flags += "ZF "
}
if f&(1<<7) != 0 {
flags += "SF "
}
if f&(1<<8) != 0 {
flags += "TF "
}
if f&(1<<9) != 0 {
flags += "IF "
}
if f&(1<<10) != 0 {
flags += "DF "
}
if f&(1<<11) != 0 {
flags += "OF "
}
if flags == "" {
return "none"
}
return flags[:len(flags)-1] // trim trailing space
}
func hexDump(addr uint64, data []byte) { func hexDump(addr uint64, data []byte) {
for i := 0; i < len(data); i += 16 { for i := 0; i < len(data); i += 16 {
end := i + 16 end := min(i+16, len(data))
if end > len(data) {
end = len(data)
}
fmt.Printf(" %#08x:", addr+uint64(i)) fmt.Printf(" %#08x:", addr+uint64(i))
for j := i; j < i+16; j++ { for j := i; j < i+16; j++ {
if j < end { if j < end {
@@ -594,21 +511,18 @@ func hexDump(addr uint64, data []byte) {
} }
func resolveAddr(s string, codeBase, funcOff uint64, labels []Label) (uint64, string) { func resolveAddr(s string, codeBase, funcOff uint64, labels []Label) (uint64, string) {
// Try as a hex address.
if strings.HasPrefix(s, "0x") || strings.HasPrefix(s, "0X") { if strings.HasPrefix(s, "0x") || strings.HasPrefix(s, "0X") {
v, err := strconv.ParseUint(s, 0, 64) v, err := strconv.ParseUint(s, 0, 64)
if err == nil { if err == nil {
return v, "" return v, ""
} }
} }
// Try as func+offset.
if strings.HasPrefix(s, "+") { if strings.HasPrefix(s, "+") {
off, err := strconv.ParseUint(s[1:], 0, 64) off, err := strconv.ParseUint(s[1:], 0, 64)
if err == nil { if err == nil {
return codeBase + funcOff + off, fmt.Sprintf("func+%#x", off) return codeBase + funcOff + off, fmt.Sprintf("func+%#x", off)
} }
} }
// Try as a label name.
for _, l := range labels { for _, l := range labels {
if l.Name == s { if l.Name == s {
return codeBase + funcOff + uint64(l.Offset), l.Name return codeBase + funcOff + uint64(l.Offset), l.Name
@@ -617,7 +531,6 @@ func resolveAddr(s string, codeBase, funcOff uint64, labels []Label) (uint64, st
return 0, "" return 0, ""
} }
// lineAt returns the source line for a given function-relative offset.
func lineAt(lines []SourceLine, offset int) int { func lineAt(lines []SourceLine, offset int) int {
if len(lines) == 0 { if len(lines) == 0 {
return 0 return 0
@@ -637,8 +550,6 @@ func lineAt(lines []SourceLine, offset int) int {
return 0 return 0
} }
// offsetForLine returns the byte offset for a given source line number.
// Returns -1 if no instruction is at that line.
func offsetForLine(lines []SourceLine, line int) int { func offsetForLine(lines []SourceLine, line int) int {
for _, le := range lines { for _, le := range lines {
if le.Line == line { if le.Line == line {
@@ -648,7 +559,6 @@ func offsetForLine(lines []SourceLine, line int) int {
return -1 return -1
} }
// nearestLabel returns the name of the label at or just before the offset.
func nearestLabel(labels []Label, offset int) string { func nearestLabel(labels []Label, offset int) string {
best := "" best := ""
bestOff := -1 bestOff := -1
+69
View File
@@ -0,0 +1,69 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
package debug
import (
"syscall"
"unsafe"
)
// StopReason describes why the debuggee stopped.
type StopReason int
const (
StopNone StopReason = iota
StopBreakpoint // software breakpoint hit
StopWatchpoint // hardware watchpoint triggered
StopSingleStep // single-step completed
StopSignal // stopped by a signal
StopExited // process exited
)
// siginfo_t layout (Linux): si_signo, si_errno, si_code, then union.
// The si_addr field is at offset 16 on all supported architectures.
type siginfoT struct {
SiSigno int32
SiErrno int32
SiCode int32
_pad [125]byte
}
const (
trapBRKPT = 1 // software breakpoint
trapHWBRKPT = 4 // hardware watchpoint
)
// StopInfo returns the reason the debuggee stopped and the faulting address
// (for watchpoints, the watched address that was accessed).
func (s *Session) StopInfo() (StopReason, uint64) {
if s.exited {
return StopExited, 0
}
var info siginfoT
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_GETSIGINFO),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(&info)),
0, 0,
)
if errno != 0 {
return StopNone, 0
}
if info.SiSigno != int32(syscall.SIGTRAP) {
return StopSignal, uint64(info.SiCode)
}
switch info.SiCode {
case trapBRKPT:
return StopBreakpoint, 0
case trapHWBRKPT:
addr := *(*uint64)(unsafe.Add(unsafe.Pointer(&info), 16))
return StopWatchpoint, addr
default:
return StopSingleStep, 0
}
}
+1 -63
View File
@@ -5,69 +5,7 @@
package debug package debug
import ( import "fmt"
"fmt"
"syscall"
"unsafe"
)
// StopReason describes why the debuggee stopped.
type StopReason int
const (
StopNone StopReason = iota
StopBreakpoint // INT3 breakpoint hit
StopWatchpoint // hardware watchpoint triggered
StopSingleStep // single-step completed
StopSignal // stopped by a signal
StopExited // process exited
)
// siginfo_t layout (Linux amd64): si_signo, si_errno, si_code, then union.
type siginfoT struct {
SiSigno int32
SiErrno int32
SiCode int32
_pad [125]byte
}
const (
trapBRKPT = 1 // INT3 breakpoint
trapHWBRKPT = 4 // hardware watchpoint
)
// StopInfo returns the reason the debuggee stopped and the faulting address
// (for watchpoints, the watched address that was accessed).
func (s *Session) StopInfo() (StopReason, uint64) {
if s.exited {
return StopExited, 0
}
var info siginfoT
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_GETSIGINFO),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(&info)),
0, 0,
)
if errno != 0 {
return StopNone, 0
}
if info.SiSigno != int32(syscall.SIGTRAP) {
return StopSignal, uint64(info.SiCode)
}
switch info.SiCode {
case trapBRKPT:
return StopBreakpoint, 0
case trapHWBRKPT:
// The faulting address is in si_addr (offset 16 in siginfo_t on amd64).
addr := *(*uint64)(unsafe.Pointer(uintptr(unsafe.Pointer(&info)) + 16))
return StopWatchpoint, addr
default:
return StopSingleStep, 0
}
}
// SetReg modifies a register value in the debuggee. // SetReg modifies a register value in the debuggee.
func (s *Session) SetReg(name string, value uint64) error { func (s *Session) SetReg(name string, value uint64) error {
+87
View File
@@ -0,0 +1,87 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && arm64
package debug
import "fmt"
// SetReg modifies a register value in the debuggee.
func (s *Session) SetReg(name string, value uint64) error {
regs, err := s.GetRegs()
if err != nil {
return err
}
switch name {
case "x0":
regs.X0 = value
case "x1":
regs.X1 = value
case "x2":
regs.X2 = value
case "x3":
regs.X3 = value
case "x4":
regs.X4 = value
case "x5":
regs.X5 = value
case "x6":
regs.X6 = value
case "x7":
regs.X7 = value
case "x8":
regs.X8 = value
case "x9":
regs.X9 = value
case "x10":
regs.X10 = value
case "x11":
regs.X11 = value
case "x12":
regs.X12 = value
case "x13":
regs.X13 = value
case "x14":
regs.X14 = value
case "x15":
regs.X15 = value
case "x16":
regs.X16 = value
case "x17":
regs.X17 = value
case "x18":
regs.X18 = value
case "x19":
regs.X19 = value
case "x20":
regs.X20 = value
case "x21":
regs.X21 = value
case "x22":
regs.X22 = value
case "x23":
regs.X23 = value
case "x24":
regs.X24 = value
case "x25":
regs.X25 = value
case "x26":
regs.X26 = value
case "x27":
regs.X27 = value
case "x28":
regs.X28 = value
case "x29", "fp":
regs.X29 = value
case "x30", "lr":
regs.X30 = value
case "sp":
regs.SP = value
case "pc":
regs.PC = value
default:
return fmt.Errorf("debug: unknown register %q", name)
}
return s.SetRegs(&regs)
}
+85
View File
@@ -0,0 +1,85 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && loong64
package debug
import "fmt"
// SetReg modifies a register value in the debuggee.
func (s *Session) SetReg(name string, value uint64) error {
regs, err := s.GetRegs()
if err != nil {
return err
}
switch name {
case "r0", "zero":
regs.R0 = value
case "r1", "ra":
regs.R1 = value
case "r2", "tp":
regs.R2 = value
case "r3", "sp":
regs.R3 = value
case "r4", "a0":
regs.R4 = value
case "r5", "a1":
regs.R5 = value
case "r6", "a2":
regs.R6 = value
case "r7", "a3":
regs.R7 = value
case "r8", "a4":
regs.R8 = value
case "r9", "a5":
regs.R9 = value
case "r10", "a6":
regs.R10 = value
case "r11", "a7":
regs.R11 = value
case "r12", "t0":
regs.R12 = value
case "r13", "t1":
regs.R13 = value
case "r14", "t2":
regs.R14 = value
case "r15", "t3":
regs.R15 = value
case "r16", "t4":
regs.R16 = value
case "r17", "t5":
regs.R17 = value
case "r18", "t6":
regs.R18 = value
case "r19", "t7":
regs.R19 = value
case "r20", "t8":
regs.R20 = value
case "r21", "fp":
regs.R21 = value
case "r22", "s0":
regs.R22 = value
case "r23", "s1":
regs.R23 = value
case "r24", "s2":
regs.R24 = value
case "r25", "s3":
regs.R25 = value
case "r26", "s4":
regs.R26 = value
case "r27", "s5":
regs.R27 = value
case "r28", "s6":
regs.R28 = value
case "r29", "s7":
regs.R29 = value
case "r30", "s8":
regs.R30 = value
case "r31", "pc":
regs.R31 = value
default:
return fmt.Errorf("debug: unknown register %q", name)
}
return s.SetRegs(&regs)
}
+85
View File
@@ -0,0 +1,85 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && riscv64
package debug
import "fmt"
// SetReg modifies a register value in the debuggee.
func (s *Session) SetReg(name string, value uint64) error {
regs, err := s.GetRegs()
if err != nil {
return err
}
switch name {
case "pc":
regs.PC = value
case "ra", "x1":
regs.Ra = value
case "sp", "x2":
regs.Sp = value
case "gp", "x3":
regs.Gp = value
case "tp", "x4":
regs.Tp = value
case "t0", "x5":
regs.T0 = value
case "t1", "x6":
regs.T1 = value
case "t2", "x7":
regs.T2 = value
case "s0", "fp", "x8":
regs.S0 = value
case "s1", "x9":
regs.S1 = value
case "a0", "x10":
regs.A0 = value
case "a1", "x11":
regs.A1 = value
case "a2", "x12":
regs.A2 = value
case "a3", "x13":
regs.A3 = value
case "a4", "x14":
regs.A4 = value
case "a5", "x15":
regs.A5 = value
case "a6", "x16":
regs.A6 = value
case "a7", "x17":
regs.A7 = value
case "s2", "x18":
regs.S2 = value
case "s3", "x19":
regs.S3 = value
case "s4", "x20":
regs.S4 = value
case "s5", "x21":
regs.S5 = value
case "s6", "x22":
regs.S6 = value
case "s7", "x23":
regs.S7 = value
case "s8", "x24":
regs.S8 = value
case "s9", "x25":
regs.S9 = value
case "s10", "x26":
regs.S10 = value
case "s11", "x27":
regs.S11 = value
case "t3", "x28":
regs.T3 = value
case "t4", "x29":
regs.T4 = value
case "t5", "x30":
regs.T5 = value
case "t6", "x31":
regs.T6 = value
default:
return fmt.Errorf("debug: unknown register %q", name)
}
return s.SetRegs(&regs)
}
+99
View File
@@ -0,0 +1,99 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
package debug
import (
"encoding/hex"
"fmt"
"os"
"strconv"
"strings"
"syscall"
"unsafe"
)
// mapRWX maps code into a read-write-execute region.
func mapRWX(code []byte) ([]byte, error) {
const pageSize = 4096
size := (len(code) + pageSize - 1) &^ (pageSize - 1)
mem, err := syscall.Mmap(-1, 0, size,
syscall.PROT_READ|syscall.PROT_WRITE|syscall.PROT_EXEC,
syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, err
}
copy(mem, code)
return mem, nil
}
// setupBuffers allocates buffers in the debuggee's memory.
func setupBuffers(spec string, args []byte, tmpDir string) ([]byte, error) {
type bufSpec struct {
name string
size int
pattern string
}
var specs []bufSpec
for part := range strings.SplitSeq(spec, ",") {
fields := strings.SplitN(part, ":", 3)
if len(fields) != 3 {
continue
}
size, err := strconv.Atoi(fields[1])
if err != nil || size <= 0 {
continue
}
specs = append(specs, bufSpec{name: fields[0], size: size, pattern: fields[2]})
}
if len(specs) == 0 {
return args, nil
}
var bufAddrs []uint64
for _, s := range specs {
buf, err := syscall.Mmap(-1, 0, s.size,
syscall.PROT_READ|syscall.PROT_WRITE,
syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, fmt.Errorf("mmap buffer %s: %w", s.name, err)
}
fillBuffer(buf, s.pattern)
bufAddrs = append(bufAddrs, uint64(uintptr(unsafe.Pointer(&buf[0]))))
}
addrFile, err := os.Create(tmpDir + "/bufaddrs")
if err != nil {
return nil, err
}
for _, addr := range bufAddrs {
fmt.Fprintf(addrFile, "%d\n", addr)
}
addrFile.Close()
return args, nil
}
// fillBuffer fills a buffer with the specified pattern.
func fillBuffer(buf []byte, pattern string) {
switch pattern {
case "zero":
case "ones":
for i := range buf {
buf[i] = 0xFF
}
case "seq":
for i := range buf {
buf[i] = byte(i)
}
default:
if data, err := hex.DecodeString(pattern); err == nil && len(data) > 0 {
for i := range buf {
buf[i] = data[i%len(data)]
}
}
}
}
+5 -125
View File
@@ -6,12 +6,9 @@
package debug package debug
import ( import (
"encoding/hex"
"fmt" "fmt"
"os" "os"
"runtime" "runtime"
"strconv"
"strings"
"syscall" "syscall"
"unsafe" "unsafe"
@@ -20,12 +17,8 @@ import (
"sourcedock.dev/petrbalvin/gasm-devkit/verify" "sourcedock.dev/petrbalvin/gasm-devkit/verify"
) )
// RunTarget is the debuggee entry point (gasm debug --target). It // RunTarget is the debuggee entry point (gasm debug --target).
// assembles the file, maps the JIT code, registers itself for ptrace,
// stops, and then executes the named function. The parent debugger
// controls execution from there.
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error { func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
// Parse and assemble.
src, err := os.ReadFile(asmPath) src, err := os.ReadFile(asmPath)
if err != nil { if err != nil {
return fmt.Errorf("debug target: %w", err) return fmt.Errorf("debug target: %w", err)
@@ -39,7 +32,6 @@ func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
return fmt.Errorf("debug target: assemble: %w", err) return fmt.Errorf("debug target: assemble: %w", err)
} }
// Find the function.
var fl *asm.FuncLayout var fl *asm.FuncLayout
for i := range img.Funcs { for i := range img.Funcs {
if img.Funcs[i].Name == funcName { if img.Funcs[i].Name == funcName {
@@ -51,24 +43,20 @@ func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
return fmt.Errorf("debug target: function %q not found", funcName) return fmt.Errorf("debug target: function %q not found", funcName)
} }
// Map the entire image RWX (we need write access for breakpoints).
code := img.Bytes() code := img.Bytes()
exec, err := mapRWX(code) exec, err := mapRWX(code)
if err != nil { if err != nil {
return fmt.Errorf("debug target: mmap: %w", err) return fmt.Errorf("debug target: mmap: %w", err)
} }
// Write the code base address for the parent.
codeBase := uintptr(unsafe.Pointer(&exec[0])) codeBase := uintptr(unsafe.Pointer(&exec[0]))
if err := os.WriteFile(tmpDir+"/codebase", []byte(fmt.Sprintf("%d", codeBase)), 0o644); err != nil { if err := os.WriteFile(tmpDir+"/codebase", []byte(fmt.Sprintf("%d", codeBase)), 0o644); err != nil {
return fmt.Errorf("debug target: write codebase: %w", err) return fmt.Errorf("debug target: write codebase: %w", err)
} }
// Write function metadata (offset, size, args) for the parent.
meta := fmt.Sprintf("%d %d %d", fl.Offset, fl.Size, fl.Args) meta := fmt.Sprintf("%d %d %d", fl.Offset, fl.Size, fl.Args)
os.WriteFile(tmpDir+"/funcmeta", []byte(meta), 0o644) os.WriteFile(tmpDir+"/funcmeta", []byte(meta), 0o644)
// Write label table for breakpoint resolution.
labelsFile, _ := os.Create(tmpDir + "/labels") labelsFile, _ := os.Create(tmpDir + "/labels")
if labelsFile != nil { if labelsFile != nil {
for label, off := range fl.Labels { for label, off := range fl.Labels {
@@ -77,7 +65,6 @@ func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
labelsFile.Close() labelsFile.Close()
} }
// Read the argument block.
args, err := os.ReadFile(argsFile) args, err := os.ReadFile(argsFile)
if err != nil { if err != nil {
return fmt.Errorf("debug target: read args: %w", err) return fmt.Errorf("debug target: read args: %w", err)
@@ -88,140 +75,33 @@ func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
args = padded args = padded
} }
// Read buffer specification if present.
bufSpecFile := tmpDir + "/bufspec" bufSpecFile := tmpDir + "/bufspec"
if bufSpec, err := os.ReadFile(bufSpecFile); err == nil && len(bufSpec) > 0 { if bufSpec, err := os.ReadFile(bufSpecFile); err == nil && len(bufSpec) > 0 {
args, err = setupBuffers(string(bufSpec), args, fl.Args, tmpDir) args, err = setupBuffers(string(bufSpec), args, tmpDir)
if err != nil { if err != nil {
return fmt.Errorf("debug target: setup buffers: %w", err) return fmt.Errorf("debug target: setup buffers: %w", err)
} }
} }
// Lock this goroutine to the current OS thread so the parent's
// ptrace (attached to this thread) controls the JIT execution.
runtime.LockOSThread() runtime.LockOSThread()
// Request tracing by the parent, then stop. PTRACE_TRACEME makes
// the subsequent SIGSTOP a ptrace-stop (not a group-stop), giving
// the parent full control from the start.
if _, _, errno := syscall.Syscall(syscall.SYS_PTRACE, uintptr(syscall.PTRACE_TRACEME), 0, 0); errno != 0 { if _, _, errno := syscall.Syscall(syscall.SYS_PTRACE, uintptr(syscall.PTRACE_TRACEME), 0, 0); errno != 0 {
return fmt.Errorf("debug target: PTRACE_TRACEME: %v", errno) return fmt.Errorf("debug target: PTRACE_TRACEME: %v", errno)
} }
os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644) os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644)
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP) syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
// --- Execution resumes here after the parent continues us ---
// Stop at the function entry point so the debugger can set breakpoints.
// The parent will continue us when ready.
os.WriteFile(tmpDir+"/entry", []byte("ok"), 0o644) os.WriteFile(tmpDir+"/entry", []byte("ok"), 0o644)
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP) syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
// Prepare the ABI0 stack and call the function.
fnAddr := codeBase + uintptr(fl.Offset) fnAddr := codeBase + uintptr(fl.Offset)
stackArgs := make([]byte, fl.Args) stackArgs := make([]byte, fl.Args)
copy(stackArgs, args) copy(stackArgs, args)
_, callErr := verify.Call(fnAddr, stackArgs) if _, callErr := verify.Call(fnAddr, stackArgs); callErr != nil {
if callErr != nil {
// The function returned an error (shouldn't happen for valid code).
os.Exit(1) os.Exit(1)
} }
os.Exit(0) // Success returns to the caller, which exits with status 0; the JIT
// code has already run to its own trampoline by the time Call returns.
return nil return nil
} }
// mapRWX maps code into a read-write-execute region (needed for
// breakpoint patching via ptrace POKETEXT, though ptrace can write
// to any mapping regardless of permissions).
func mapRWX(code []byte) ([]byte, error) {
const pageSize = 4096
size := (len(code) + pageSize - 1) &^ (pageSize - 1)
mem, err := syscall.Mmap(-1, 0, size,
syscall.PROT_READ|syscall.PROT_WRITE|syscall.PROT_EXEC,
syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, err
}
copy(mem, code)
return mem, nil
}
// setupBuffers allocates buffers in the debuggee's memory and updates the
// argument block with pointers to them.
// Format: name:size:pattern[,name:size:pattern...]
// Patterns: zero, ones, seq, or hex (e.g. "deadbeef").
func setupBuffers(spec string, args []byte, argSize int, tmpDir string) ([]byte, error) {
// Parse the buffer spec.
type bufSpec struct {
name string
size int
pattern string
}
var specs []bufSpec
for _, part := range strings.Split(spec, ",") {
fields := strings.SplitN(part, ":", 3)
if len(fields) != 3 {
continue
}
size, err := strconv.Atoi(fields[1])
if err != nil || size <= 0 {
continue
}
specs = append(specs, bufSpec{name: fields[0], size: size, pattern: fields[2]})
}
if len(specs) == 0 {
return args, nil
}
// Allocate buffers and write their addresses to a file for the parent.
var bufAddrs []uint64
for _, s := range specs {
buf, err := syscall.Mmap(-1, 0, s.size,
syscall.PROT_READ|syscall.PROT_WRITE,
syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, fmt.Errorf("mmap buffer %s: %w", s.name, err)
}
fillBuffer(buf, s.pattern)
bufAddrs = append(bufAddrs, uint64(uintptr(unsafe.Pointer(&buf[0]))))
}
// Write buffer addresses to a file for the parent to read.
addrFile, err := os.Create(tmpDir + "/bufaddrs")
if err != nil {
return nil, err
}
for _, addr := range bufAddrs {
fmt.Fprintf(addrFile, "%d\n", addr)
}
addrFile.Close()
// For now, return the args unchanged. The parent will read bufaddrs
// and construct the final argument block with the correct pointers.
return args, nil
}
// fillBuffer fills a buffer with the specified pattern.
func fillBuffer(buf []byte, pattern string) {
switch pattern {
case "zero":
// Already zeroed by mmap.
case "ones":
for i := range buf {
buf[i] = 0xFF
}
case "seq":
for i := range buf {
buf[i] = byte(i)
}
default:
// Try to parse as hex.
if data, err := hex.DecodeString(pattern); err == nil && len(data) > 0 {
for i := range buf {
buf[i] = data[i%len(data)]
}
}
}
}
+116
View File
@@ -0,0 +1,116 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && arm64
package debug
import (
"fmt"
"os"
"runtime"
"syscall"
"unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
)
// RunTarget is the debuggee entry point (gasm debug --target).
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
src, err := os.ReadFile(asmPath)
if err != nil {
return fmt.Errorf("debug target: %w", err)
}
file, errs := parser.Parse(asmPath, string(src))
if len(errs) > 0 {
return fmt.Errorf("debug target: parse: %v", errs[0])
}
var img *asm.Image
switch "arm64" {
case "arm64":
img, err = asm.AssembleFileARM64(file)
case "riscv64":
img, err = asm.AssembleFileRISCV(file)
case "loong64":
img, err = asm.AssembleFileLOONG64(file)
}
if err != nil {
return fmt.Errorf("debug target: assemble: %w", err)
}
var fl *asm.FuncLayout
for i := range img.Funcs {
if img.Funcs[i].Name == funcName {
fl = &img.Funcs[i]
break
}
}
if fl == nil {
return fmt.Errorf("debug target: function %q not found", funcName)
}
code := img.Bytes()
exec, err := mapRWX(code)
if err != nil {
return fmt.Errorf("debug target: mmap: %w", err)
}
codeBase := uintptr(unsafe.Pointer(&exec[0]))
if err := os.WriteFile(tmpDir+"/codebase", []byte(fmt.Sprintf("%d", codeBase)), 0o644); err != nil {
return fmt.Errorf("debug target: write codebase: %w", err)
}
meta := fmt.Sprintf("%d %d %d", fl.Offset, fl.Size, fl.Args)
os.WriteFile(tmpDir+"/funcmeta", []byte(meta), 0o644)
labelsFile, _ := os.Create(tmpDir + "/labels")
if labelsFile != nil {
for label, off := range fl.Labels {
fmt.Fprintf(labelsFile, "%s %d\n", label, off)
}
labelsFile.Close()
}
args, err := os.ReadFile(argsFile)
if err != nil {
return fmt.Errorf("debug target: read args: %w", err)
}
if len(args) < fl.Args {
padded := make([]byte, fl.Args)
copy(padded, args)
args = padded
}
bufSpecFile := tmpDir + "/bufspec"
if bufSpec, err := os.ReadFile(bufSpecFile); err == nil && len(bufSpec) > 0 {
args, err = setupBuffers(string(bufSpec), args, tmpDir)
if err != nil {
return fmt.Errorf("debug target: setup buffers: %w", err)
}
}
runtime.LockOSThread()
if _, _, errno := syscall.Syscall(syscall.SYS_PTRACE, uintptr(syscall.PTRACE_TRACEME), 0, 0); errno != 0 {
return fmt.Errorf("debug target: PTRACE_TRACEME: %v", errno)
}
os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644)
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
os.WriteFile(tmpDir+"/entry", []byte("ok"), 0o644)
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
fnAddr := codeBase + uintptr(fl.Offset)
stackArgs := make([]byte, fl.Args)
copy(stackArgs, args)
_, callErr := verify.Call(fnAddr, stackArgs)
if callErr != nil {
os.Exit(1)
}
os.Exit(0)
return nil
}
+116
View File
@@ -0,0 +1,116 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && loong64
package debug
import (
"fmt"
"os"
"runtime"
"syscall"
"unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
)
// RunTarget is the debuggee entry point (gasm debug --target).
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
src, err := os.ReadFile(asmPath)
if err != nil {
return fmt.Errorf("debug target: %w", err)
}
file, errs := parser.Parse(asmPath, string(src))
if len(errs) > 0 {
return fmt.Errorf("debug target: parse: %v", errs[0])
}
var img *asm.Image
switch "loong64" {
case "arm64":
img, err = asm.AssembleFileARM64(file)
case "riscv64":
img, err = asm.AssembleFileRISCV(file)
case "loong64":
img, err = asm.AssembleFileLOONG64(file)
}
if err != nil {
return fmt.Errorf("debug target: assemble: %w", err)
}
var fl *asm.FuncLayout
for i := range img.Funcs {
if img.Funcs[i].Name == funcName {
fl = &img.Funcs[i]
break
}
}
if fl == nil {
return fmt.Errorf("debug target: function %q not found", funcName)
}
code := img.Bytes()
exec, err := mapRWX(code)
if err != nil {
return fmt.Errorf("debug target: mmap: %w", err)
}
codeBase := uintptr(unsafe.Pointer(&exec[0]))
if err := os.WriteFile(tmpDir+"/codebase", []byte(fmt.Sprintf("%d", codeBase)), 0o644); err != nil {
return fmt.Errorf("debug target: write codebase: %w", err)
}
meta := fmt.Sprintf("%d %d %d", fl.Offset, fl.Size, fl.Args)
os.WriteFile(tmpDir+"/funcmeta", []byte(meta), 0o644)
labelsFile, _ := os.Create(tmpDir + "/labels")
if labelsFile != nil {
for label, off := range fl.Labels {
fmt.Fprintf(labelsFile, "%s %d\n", label, off)
}
labelsFile.Close()
}
args, err := os.ReadFile(argsFile)
if err != nil {
return fmt.Errorf("debug target: read args: %w", err)
}
if len(args) < fl.Args {
padded := make([]byte, fl.Args)
copy(padded, args)
args = padded
}
bufSpecFile := tmpDir + "/bufspec"
if bufSpec, err := os.ReadFile(bufSpecFile); err == nil && len(bufSpec) > 0 {
args, err = setupBuffers(string(bufSpec), args, tmpDir)
if err != nil {
return fmt.Errorf("debug target: setup buffers: %w", err)
}
}
runtime.LockOSThread()
if _, _, errno := syscall.Syscall(syscall.SYS_PTRACE, uintptr(syscall.PTRACE_TRACEME), 0, 0); errno != 0 {
return fmt.Errorf("debug target: PTRACE_TRACEME: %v", errno)
}
os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644)
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
os.WriteFile(tmpDir+"/entry", []byte("ok"), 0o644)
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
fnAddr := codeBase + uintptr(fl.Offset)
stackArgs := make([]byte, fl.Args)
copy(stackArgs, args)
_, callErr := verify.Call(fnAddr, stackArgs)
if callErr != nil {
os.Exit(1)
}
os.Exit(0)
return nil
}
+108
View File
@@ -0,0 +1,108 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && riscv64
package debug
import (
"fmt"
"os"
"runtime"
"syscall"
"unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
)
// RunTarget is the debuggee entry point (gasm debug --target).
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
src, err := os.ReadFile(asmPath)
if err != nil {
return fmt.Errorf("debug target: %w", err)
}
file, errs := parser.Parse(asmPath, string(src))
if len(errs) > 0 {
return fmt.Errorf("debug target: parse: %v", errs[0])
}
img, err := asm.AssembleFileRISCV(file)
if err != nil {
return fmt.Errorf("debug target: assemble: %w", err)
}
var fl *asm.FuncLayout
for i := range img.Funcs {
if img.Funcs[i].Name == funcName {
fl = &img.Funcs[i]
break
}
}
if fl == nil {
return fmt.Errorf("debug target: function %q not found", funcName)
}
code := img.Bytes()
exec, err := mapRWX(code)
if err != nil {
return fmt.Errorf("debug target: mmap: %w", err)
}
codeBase := uintptr(unsafe.Pointer(&exec[0]))
if err := os.WriteFile(tmpDir+"/codebase", []byte(fmt.Sprintf("%d", codeBase)), 0o644); err != nil {
return fmt.Errorf("debug target: write codebase: %w", err)
}
meta := fmt.Sprintf("%d %d %d", fl.Offset, fl.Size, fl.Args)
os.WriteFile(tmpDir+"/funcmeta", []byte(meta), 0o644)
labelsFile, _ := os.Create(tmpDir + "/labels")
if labelsFile != nil {
for label, off := range fl.Labels {
fmt.Fprintf(labelsFile, "%s %d\n", label, off)
}
labelsFile.Close()
}
args, err := os.ReadFile(argsFile)
if err != nil {
return fmt.Errorf("debug target: read args: %w", err)
}
if len(args) < fl.Args {
padded := make([]byte, fl.Args)
copy(padded, args)
args = padded
}
bufSpecFile := tmpDir + "/bufspec"
if bufSpec, err := os.ReadFile(bufSpecFile); err == nil && len(bufSpec) > 0 {
args, err = setupBuffers(string(bufSpec), args, tmpDir)
if err != nil {
return fmt.Errorf("debug target: setup buffers: %w", err)
}
}
runtime.LockOSThread()
if _, _, errno := syscall.Syscall(syscall.SYS_PTRACE, uintptr(syscall.PTRACE_TRACEME), 0, 0); errno != 0 {
return fmt.Errorf("debug target: PTRACE_TRACEME: %v", errno)
}
os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644)
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
os.WriteFile(tmpDir+"/entry", []byte("ok"), 0o644)
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
fnAddr := codeBase + uintptr(fl.Offset)
stackArgs := make([]byte, fl.Args)
copy(stackArgs, args)
_, callErr := verify.Call(fnAddr, stackArgs)
if callErr != nil {
os.Exit(1)
}
os.Exit(0)
return nil
}
+4 -34
View File
@@ -1,39 +1,9 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause // SPDX-License-Identifier: BSD-3-Clause
package debug //go:build linux
// Regs holds the full general-purpose register set of a traced process package debug
// (the Linux amd64 user_regs_struct layout).
type Regs struct {
R15 uint64
R14 uint64
R13 uint64
R12 uint64
RBP uint64
RBX uint64
R11 uint64
R10 uint64
R9 uint64
R8 uint64
RAX uint64
RCX uint64
RDX uint64
RSI uint64
RDI uint64
OrigRAX uint64
RIP uint64
CS uint64
RFLAGS uint64
RSP uint64
SS uint64
FSBase uint64
GSBase uint64
DS uint64
ES uint64
FS uint64
GS uint64
}
// tracer abstracts the minimal ptrace operations needed by the breakpoint // tracer abstracts the minimal ptrace operations needed by the breakpoint
// manager and the stop-information helpers. The live implementation is // manager and the stop-information helpers. The live implementation is
@@ -66,7 +36,7 @@ func newMockTracer() *mockTracer {
func (m *mockTracer) Peek(addr uint64) (uint64, error) { func (m *mockTracer) Peek(addr uint64) (uint64, error) {
m.peeks = append(m.peeks, addr) m.peeks = append(m.peeks, addr)
var val uint64 var val uint64
for i := uint64(0); i < 8; i++ { for i := range uint64(8) {
val |= uint64(m.mem[addr+i]) << (i * 8) val |= uint64(m.mem[addr+i]) << (i * 8)
} }
return val, nil return val, nil
@@ -77,7 +47,7 @@ func (m *mockTracer) Poke(addr uint64, val uint64) error {
addr uint64 addr uint64
val uint64 val uint64
}{addr, val}) }{addr, val})
for i := uint64(0); i < 8; i++ { for i := range uint64(8) {
m.mem[addr+i] = byte(val >> (i * 8)) m.mem[addr+i] = byte(val >> (i * 8))
} }
return nil return nil
+13 -27
View File
@@ -11,11 +11,6 @@ import (
) )
// Hardware watchpoint support via x86-64 debug registers (DR0-DR3, DR7). // Hardware watchpoint support via x86-64 debug registers (DR0-DR3, DR7).
//
// DR0-DR3 hold the watched addresses. DR7 is the control register:
// bits 0,2,4,6: local enable for DR0-DR3
// bits 16-17,20-21,24-25,28-29: R/W type (00=exec, 01=write, 11=read/write)
// bits 18-19,22-23,26-27,30-31: length (00=1, 01=2, 10=8, 11=4)
// WatchpointType selects what triggers the watchpoint. // WatchpointType selects what triggers the watchpoint.
type WatchpointType int type WatchpointType int
@@ -28,7 +23,7 @@ const (
// FindFreeWatchpointSlot returns the index of the first free watchpoint slot // FindFreeWatchpointSlot returns the index of the first free watchpoint slot
// (0-3), or -1 if all four hardware watchpoints are in use. // (0-3), or -1 if all four hardware watchpoints are in use.
func (s *Session) FindFreeWatchpointSlot() int { func (s *Session) FindFreeWatchpointSlot() int {
for i := 0; i < 4; i++ { for i := range 4 {
if !s.wpSlots[i] { if !s.wpSlots[i] {
return i return i
} }
@@ -45,7 +40,6 @@ func (s *Session) IsWatchpointSlotUsed(slot int) bool {
} }
// SetWatchpoint installs a hardware watchpoint on the given address. // SetWatchpoint installs a hardware watchpoint on the given address.
// slot is 0-3 (four hardware watchpoints available); the slot must be free.
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error { func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
if slot < 0 || slot > 3 { if slot < 0 || slot > 3 {
return fmt.Errorf("debug: watchpoint slot must be 0-3") return fmt.Errorf("debug: watchpoint slot must be 0-3")
@@ -54,7 +48,6 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
return fmt.Errorf("debug: watchpoint slot %d already in use", slot) return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
} }
// Determine the length encoding.
var lenBits uint64 var lenBits uint64
switch size { switch size {
case 1: case 1:
@@ -69,35 +62,31 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8") return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
} }
// Write the watched address to DR0-DR3.
var drAddr uintptr var drAddr uintptr
switch slot { switch slot {
case 0: case 0:
drAddr = 0x0 // DR0 offset in user_regs_struct drAddr = 0x0
case 1: case 1:
drAddr = 0x8 // DR1 drAddr = 0x8
case 2: case 2:
drAddr = 0x10 // DR2 drAddr = 0x10
case 3: case 3:
drAddr = 0x18 // DR3 drAddr = 0x18
} }
// PTRACE_POKEUSER writes to the debuggee's user area (includes debug regs).
if err := ptracePokeUser(s.pid, drAddr, addr); err != nil { if err := ptracePokeUser(s.pid, drAddr, addr); err != nil {
return fmt.Errorf("debug: set DR%d: %w", slot, err) return fmt.Errorf("debug: set DR%d: %w", slot, err)
} }
// Read the current DR7, set the enable and type bits, write it back. dr7, err := ptracePeekUser(s.pid, 0x38)
dr7, err := ptracePeekUser(s.pid, 0x38) // DR7 offset
if err != nil { if err != nil {
return fmt.Errorf("debug: read DR7: %w", err) return fmt.Errorf("debug: read DR7: %w", err)
} }
enableBit := uint64(1) << (2 * slot) // local enable enableBit := uint64(1) << (2 * slot)
rwBits := uint64(typ) << (16 + 4*slot) // R/W type rwBits := uint64(typ) << (16 + 4*slot)
lenField := lenBits << (18 + 4*slot) // length lenField := lenBits << (18 + 4*slot)
// Clear the existing bits for this slot, then set the new ones.
mask := ^((uint64(1) << (2 * slot)) | (uint64(3) << (16 + 4*slot)) | (uint64(3) << (18 + 4*slot))) mask := ^((uint64(1) << (2 * slot)) | (uint64(3) << (16 + 4*slot)) | (uint64(3) << (18 + 4*slot)))
dr7 = (dr7 & mask) | enableBit | rwBits | lenField dr7 = (dr7 & mask) | enableBit | rwBits | lenField
@@ -116,12 +105,11 @@ func (s *Session) ClearWatchpoint(slot int) error {
if !s.wpSlots[slot] { if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot) return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
} }
// Read DR7, clear the enable bit for this slot.
dr7, err := ptracePeekUser(s.pid, 0x38) dr7, err := ptracePeekUser(s.pid, 0x38)
if err != nil { if err != nil {
return err return err
} }
dr7 &^= uint64(1) << (2 * slot) // disable dr7 &^= uint64(1) << (2 * slot)
if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil { if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
return err return err
} }
@@ -131,7 +119,7 @@ func (s *Session) ClearWatchpoint(slot int) error {
// ClearAllWatchpoints removes all hardware watchpoints. // ClearAllWatchpoints removes all hardware watchpoints.
func (s *Session) ClearAllWatchpoints() error { func (s *Session) ClearAllWatchpoints() error {
for slot := 0; slot < 4; slot++ { for slot := range 4 {
if s.wpSlots[slot] { if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil { if err := s.ClearWatchpoint(slot); err != nil {
return err return err
@@ -141,9 +129,8 @@ func (s *Session) ClearAllWatchpoints() error {
return nil return nil
} }
// ptracePokeUser writes a value to the debuggee's user area at the given offset.
func ptracePokeUser(pid int, offset uintptr, val uint64) error { func ptracePokeUser(pid int, offset uintptr, val uint64) error {
const ptracePokeuser = 6 // PTRACE_POKEUSER const ptracePokeuser = 6
_, _, errno := syscall.Syscall6( _, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE, syscall.SYS_PTRACE,
uintptr(ptracePokeuser), uintptr(ptracePokeuser),
@@ -158,9 +145,8 @@ func ptracePokeUser(pid int, offset uintptr, val uint64) error {
return nil return nil
} }
// ptracePeekUser reads a value from the debuggee's user area at the given offset.
func ptracePeekUser(pid int, offset uintptr) (uint64, error) { func ptracePeekUser(pid int, offset uintptr) (uint64, error) {
const ptracePeekuser = 3 // PTRACE_PEEKUSER const ptracePeekuser = 3
val, _, errno := syscall.Syscall6( val, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE, syscall.SYS_PTRACE,
uintptr(ptracePeekuser), uintptr(ptracePeekuser),
+178
View File
@@ -0,0 +1,178 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && arm64
package debug
import (
"fmt"
"syscall"
"unsafe"
)
// Hardware watchpoint support via arm64 debug registers (DBGWVR/DBGWCR).
// Accessed via PTRACE_SETREGSET with NT_ARM_HW_BREAK.
// WatchpointType selects what triggers the watchpoint.
type WatchpointType int
const (
WatchWrite WatchpointType = 1
WatchRead WatchpointType = 3
)
const maxWatchpoints = 16
// hwBreakState mirrors the kernel's struct user_hwdebug_state.
type hwBreakState struct {
DbgInfo uint32
_pad [4]byte
DbgRegs [16]hwBreakReg
}
type hwBreakReg struct {
Addr uint64
Ctrl uint64
}
const (
ntArmHWBreak = 0x403 // NT_ARM_HW_BREAK
)
func (s *Session) FindFreeWatchpointSlot() int {
for i := range maxWatchpoints {
if !s.wpSlots[i] {
return i
}
}
return -1
}
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot >= maxWatchpoints {
return false
}
return s.wpSlots[slot]
}
// SetWatchpoint installs a hardware watchpoint on the given address.
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
}
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
state, err := s.getHWBreakState()
if err != nil {
return fmt.Errorf("debug: read watchpoint state: %w", err)
}
if uint32(slot) >= state.DbgInfo {
return fmt.Errorf("debug: slot %d exceeds available watchpoints (%d)", slot, state.DbgInfo)
}
state.DbgRegs[slot].Addr = addr
ctrl := uint64(1) // enable
switch typ {
case WatchWrite:
ctrl |= 1 << 3 // store only
case WatchRead:
ctrl |= 3 << 3 // load+store
}
var bas uint64
switch size {
case 1:
bas = 0x01
case 2:
bas = 0x03
case 4:
bas = 0x0F
case 8:
bas = 0xFF
default:
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
}
ctrl |= bas << 5
state.DbgRegs[slot].Ctrl = ctrl
if err := s.setHWBreakState(state); err != nil {
return fmt.Errorf("debug: set watchpoint: %w", err)
}
s.wpSlots[slot] = true
return nil
}
func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
}
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
state, err := s.getHWBreakState()
if err != nil {
return err
}
state.DbgRegs[slot].Addr = 0
state.DbgRegs[slot].Ctrl = 0
if err := s.setHWBreakState(state); err != nil {
return err
}
s.wpSlots[slot] = false
return nil
}
func (s *Session) ClearAllWatchpoints() error {
for slot := 0; slot < maxWatchpoints; slot++ {
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
}
}
}
return nil
}
func (s *Session) getHWBreakState() (*hwBreakState, error) {
var state hwBreakState
iovec := syscall.Iovec{
Base: (*byte)(unsafe.Pointer(&state)),
Len: uint64(unsafe.Sizeof(state)),
}
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_GETREGSET),
uintptr(s.pid),
uintptr(ntArmHWBreak),
uintptr(unsafe.Pointer(&iovec)),
0, 0,
)
if errno != 0 {
return nil, errno
}
return &state, nil
}
func (s *Session) setHWBreakState(state *hwBreakState) error {
iovec := syscall.Iovec{
Base: (*byte)(unsafe.Pointer(state)),
Len: uint64(unsafe.Sizeof(*state)),
}
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_SETREGSET),
uintptr(s.pid),
uintptr(ntArmHWBreak),
uintptr(unsafe.Pointer(&iovec)),
0, 0,
)
if errno != 0 {
return errno
}
return nil
}
+144
View File
@@ -0,0 +1,144 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && loong64
package debug
import (
"fmt"
"syscall"
)
// Hardware watchpoint support for LoongArch via debug registers.
// Uses PTRACE_POKEUSER/PEEKUSER to access HW watchpoint registers.
// WatchpointType selects what triggers the watchpoint.
type WatchpointType int
const (
WatchWrite WatchpointType = 1
WatchRead WatchpointType = 3
)
const maxWatchpoints = 4
func (s *Session) FindFreeWatchpointSlot() int {
for i := range maxWatchpoints {
if !s.wpSlots[i] {
return i
}
}
return -1
}
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot >= maxWatchpoints {
return false
}
return s.wpSlots[slot]
}
// SetWatchpoint installs a hardware watchpoint.
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
}
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
if size != 1 && size != 2 && size != 4 && size != 8 {
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
}
// LoongArch debug registers: DBGWVR (watchpoint value) and DBGWCR (watchpoint control).
// Accessed via PTRACE_POKEUSER at architecture-specific offsets.
if err := ptracePokeUser(s.pid, uintptr(0x1000+slot*8), addr); err != nil {
return fmt.Errorf("debug: set watchpoint address: %w", err)
}
// DBGWCR: enable + type + size.
var wcr uint64 = 1 // enable
switch typ {
case WatchWrite:
wcr |= 1 << 3 // store
case WatchRead:
wcr |= 3 << 3 // load+store
}
var sizeBits uint64
switch size {
case 1:
sizeBits = 0
case 2:
sizeBits = 1
case 4:
sizeBits = 2
case 8:
sizeBits = 3
}
wcr |= sizeBits << 5
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), wcr); err != nil {
return fmt.Errorf("debug: set watchpoint control: %w", err)
}
s.wpSlots[slot] = true
return nil
}
func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
}
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), 0); err != nil {
return err
}
s.wpSlots[slot] = false
return nil
}
func (s *Session) ClearAllWatchpoints() error {
for slot := 0; slot < maxWatchpoints; slot++ {
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
}
}
}
return nil
}
func ptracePokeUser(pid int, offset uintptr, val uint64) error {
const ptracePokeuser = 6
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(ptracePokeuser),
uintptr(pid),
offset,
uintptr(val),
0, 0,
)
if errno != 0 {
return errno
}
return nil
}
func ptracePeekUser(pid int, offset uintptr) (uint64, error) {
const ptracePeekuser = 3
val, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(ptracePeekuser),
uintptr(pid),
offset,
0, 0, 0,
)
if errno != 0 {
return 0, errno
}
return uint64(val), nil
}
+147
View File
@@ -0,0 +1,147 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && riscv64
package debug
import (
"fmt"
"syscall"
)
// Hardware watchpoint support for RISC-V via Sdtrig trigger registers.
// Uses PTRACE_POKEUSER/PEEKUSER to access debug registers.
// WatchpointType selects what triggers the watchpoint.
type WatchpointType int
const (
WatchWrite WatchpointType = 1
WatchRead WatchpointType = 3
)
const maxWatchpoints = 4
func (s *Session) FindFreeWatchpointSlot() int {
for i := range maxWatchpoints {
if !s.wpSlots[i] {
return i
}
}
return -1
}
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot >= maxWatchpoints {
return false
}
return s.wpSlots[slot]
}
// SetWatchpoint installs a hardware watchpoint.
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
}
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
if size != 1 && size != 2 && size != 4 && size != 8 {
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
}
// RISC-V trigger registers: tdata1 encodes type/control, tdata2 holds address.
// The exact encoding depends on the trigger implementation (Sdtrig).
// Use PTRACE_POKEUSER to write to the trigger CSRs via the kernel's
// debug register interface.
if err := ptracePokeUser(s.pid, uintptr(0x1000+slot*8), addr); err != nil {
return fmt.Errorf("debug: set watchpoint address: %w", err)
}
// tdata1: set match control. Mode=2 (data match), select=0, action=1 (debug exception).
var tdata1 uint64 = 2 << 60 // type = match (2)
tdata1 |= 1 << 0 // action = enter debug mode
tdata1 |= 1 << 7 // store (write) trigger
if typ == WatchRead {
tdata1 |= 1 << 6 // load trigger
}
// Size encoding: 0=1byte, 1=2byte, 2=4byte, 3=8byte.
var sizeBits uint64
switch size {
case 1:
sizeBits = 0
case 2:
sizeBits = 1
case 4:
sizeBits = 2
case 8:
sizeBits = 3
}
tdata1 |= sizeBits << 16 // size field
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), tdata1); err != nil {
return fmt.Errorf("debug: set watchpoint control: %w", err)
}
s.wpSlots[slot] = true
return nil
}
func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
}
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
// Disable by clearing tdata1.
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), 0); err != nil {
return err
}
s.wpSlots[slot] = false
return nil
}
func (s *Session) ClearAllWatchpoints() error {
for slot := 0; slot < maxWatchpoints; slot++ {
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
}
}
}
return nil
}
func ptracePokeUser(pid int, offset uintptr, val uint64) error {
const ptracePokeuser = 6
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(ptracePokeuser),
uintptr(pid),
offset,
uintptr(val),
0, 0,
)
if errno != 0 {
return errno
}
return nil
}
func ptracePeekUser(pid int, offset uintptr) (uint64, error) {
const ptracePeekuser = 3
val, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(ptracePeekuser),
uintptr(pid),
offset,
0, 0, 0,
)
if errno != 0 {
return 0, errno
}
return uint64(val), nil
}

Some files were not shown because too many files have changed in this diff Show More