Compare commits

..
31 Commits
Author SHA1 Message Date
petrbalvin 9f4f949c1f chore: prepare release v0.34.0
Test / test (push) Successful in 2m11s
Release / gates (push) Successful in 2m11s
Release / build (amd64, linux) (push) Successful in 1m13s
Release / build (arm64, linux) (push) Successful in 1m10s
Release / build (loong64, linux) (push) Successful in 1m12s
Release / build (riscv64, linux) (push) Successful in 1m33s
Release / release (push) Successful in 58s
2026-09-20 01:44:23 +02:00
petrbalvin f0d5238c47 docs: state the validation status and correct claims the material contradicts
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 01:40:51 +02:00
petrbalvin 2931bbd6b2 ci(release): refuse a tag the security policy does not name
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 01:40:51 +02:00
petrbalvin 63562a503a test(justfile): run the CLI and debugger tests outside the coverage set
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 01:40:51 +02:00
petrbalvin e836d6150d docs: changelog entry for the loong64 JIT enablement
Test / test (push) Successful in 2m7s
Assisted-by: GLM 5.3
2026-09-20 00:57:02 +02:00
petrbalvin 8a51b060da feat(cmd): enable loong64 JIT execution, all trampolines qemu-validated
Assisted-by: GLM 5.3
2026-09-20 00:57:02 +02:00
petrbalvin d3d47db727 test(verify): seed the arm64 ABI kernel arguments
Assisted-by: GLM 5.3
2026-09-20 00:57:02 +02:00
petrbalvin 0758556b7d docs: changelog entries for the parity round and corpus number
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin ddb8440340 fix(cmd): padding-aware ground-truth comparison
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin f15ff66fb1 fix(riscv64): accept the g spelling of the goroutine register
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin 187e4856d3 feat(amd64): encode the mixed-width extend family and PMOVMSKB
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin d315a998ce fix(arm64): store-exclusive operand order and large-frame parity
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin a6f3828c02 docs: changelog entries for the review fixes
Test / test (push) Successful in 2m4s
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin e3b35bb817 style(testdata): canonical gasm formatting for the verify kernels
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin eb0a89e58d ci(release): state the version contract inline
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin dd32d9e66e chore(justfile): one-line install-man comment and long flag forms
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin 3a73acb20a docs: drop process labels and refresh the architecture and manual pages
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin 7604a9443f fix(cmd): usage exit codes, asm output file and cross-arch ground truth
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin b3908fc43d fix(lsp): parse-error survival, symbol ranges and UTF-16 positions
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin eb8b0cd316 fix(lint): trailing-label CFG guard and the goroutine alias
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin a8bfd54ed2 fix(debug): hardware watchpoints, signal stops and breakpoint restore
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin 375182ef1f fix(verify): arm64 stack save, adaptive canary and host gating
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin 87b1081c53 fix(goobj): external package and symbol indices and arm64 pair relocations
Assisted-by: GLM 5.3
2026-09-19 23:49:13 +02:00
petrbalvin f3c8510a58 fix(elf): relocation records, DWARF tables and per-architecture frame data
Assisted-by: GLM 5.3
2026-09-19 23:49:13 +02:00
petrbalvin ebdf14939f fix(loong64): FP immediates through R30 and unsigned branch forms
Assisted-by: GLM 5.3
2026-09-19 23:49:13 +02:00
petrbalvin 79a2c16bac fix(riscv64): compressed store offsets, FENCE and branch range checks
Assisted-by: GLM 5.3
2026-09-19 23:49:13 +02:00
petrbalvin 401386956c fix(arm64): encode shifts, divides and multiplies and align sizes with emission
Assisted-by: GLM 5.3
2026-09-19 23:49:07 +02:00
petrbalvin 4258131a3a fix(amd64): correct guard displacements, frameless FP offsets and immediate ranges
Assisted-by: GLM 5.3
2026-09-19 23:49:07 +02:00
petrbalvin 94e09e8070 fix(format): preserve flag separators and normalise CRLF input
Assisted-by: GLM 5.3
2026-09-19 23:48:47 +02:00
petrbalvin 7aefe6a42d fix(parser): parse ABI markers and keep TEXT decls usable on errors
Assisted-by: GLM 5.3
2026-09-19 23:48:47 +02:00
petrbalvin ac1c05c793 fix(lexer): tokenise the flag separator and handle NUL and invalid UTF-8
Assisted-by: GLM 5.3
2026-09-19 23:48:47 +02:00
158 changed files with 8027 additions and 1608 deletions
+27 -3
View File
@@ -5,9 +5,9 @@
# even at its own <module>/vX.Y.Z tag and this workflow's smoke test can never pass for
# it. A Go repository is one module at the root.
#
# The version contract these steps implement is in the `release` skill, and its point is
# that nothing is injected: the toolchain records the tag into the binary's build
# information, so the build simply has to happen at the tag, which the trigger guarantees.
# The version contract these steps implement: nothing is injected. The toolchain records
# the tag into the binary's build information, so the build simply has to happen at the
# tag, which the trigger guarantees.
#
# The gates run in their own job, once, before the matrix, minus the race detector: race
# never runs on a push path or a tag, and the local gate raced this tree before the tag
@@ -55,6 +55,25 @@ jobs:
print qq{tag $v\n};
'
- name: Security policy names this release
# The supported-versions table is the one part of SECURITY.md that
# carries a version, so it goes stale the moment a tag is cut. Fail
# here rather than publish a policy naming the previous release.
env:
VERSION: ${{ gitea.ref_name }}
run: |
perl -e '
my $v = $ENV{VERSION} // q{};
(my $nv = $v) =~ s/^v//;
open(my $f, q{<}, q{SECURITY.md}) or die qq{SECURITY.md: $!\n};
local $/;
my $t = <$f>;
close $f;
$t =~ m{^\|\s*\Q$nv\E\s*\|\s*yes\s*\|}m
or die qq{ERROR: SECURITY.md does not name $nv as supported; update the table before releasing.\n};
print qq{SECURITY.md names $nv\n};
'
- name: Build
run: go build ./...
@@ -78,6 +97,11 @@ jobs:
# The same command as in test.yml, so the floor is the same number everywhere.
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Tests outside the coverage set
# The same command as in test.yml: the CLI's exit codes and manual-page guard,
# and the debugger's architecture-neutral units, run outside the floor.
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
- name: Coverage floor
run: |
perl -e '
+8
View File
@@ -84,6 +84,14 @@ jobs:
# suites); the runner's Go setup provides both the tool and GOROOT.
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Tests outside the coverage set
# The CLI and the debugger sit outside `packages` because a thin main and a
# ptrace-bound package pull the total under the floor, but their tests guard
# shipped surfaces: the command exit codes, the manual pages against the
# binary's own help, and the debugger's architecture-neutral units. They run
# here so the floor stays a product measure and nothing is left untested.
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
- name: Oracle parity
# Re-run the live go-tool-asm comparison as its own step so that a parity
# regression names the gate that failed instead of hiding inside the suite.
+248 -40
View File
@@ -9,6 +9,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added
-
## [0.34.0] - 2026-09-20
### Added
- **Indirect JMP and CALL on all four architectures.** `JMP AX`,
`CALL AX`, `JMP (BX)` and the memory forms encode at byte parity with
the toolchain (FF /2 and FF /4 on amd64); arm64 lowers `JMP (R0)` to
@@ -16,8 +22,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
JALR; loong64 accepts the raw `JIRL rd, rj, off` spelling the Go
assembler cannot express. A frameless amd64 function containing a
CALL now receives the toolchain's forced base-pointer frame. The
verify trampolines join the ground-truth lists, and a lint check for
control flow through registers and memory extends to the new forms.
riscv64 and loong64 verify trampolines join their ground-truth lists,
and a lint check for control flow through registers and memory extends
to the new forms.
- **`gasm asm -GOARCH` and `gasm diff -GOARCH`.** The target
architecture can be named explicitly instead of inferred from the
file-name suffix, which is how the suffix-less majority of GOROOT's
@@ -25,38 +32,40 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- **`gasm audit-instructions --corpus [dir]`.** Assembles every `.s`
file under a directory (default GOROOT/src) with the gasm encoder
only: suffixed files for their architecture, suffix-less files for
all four, as a GOARCH build would. Reports the headline number (108
of 627 GOROOT files, 17.2 %, assemble for every target architecture,
against 23 in the previous release), the per-architecture pass rates
and the most common failure reasons with a representative file each,
all four, as a GOARCH build would. Reports the headline number (127
of 627 GOROOT files, 20.3 %, assemble for every target architecture),
the per-architecture pass rates and the most common failure reasons
with a representative file each,
which drive the encodability backlog by frequency.
- **Fuzz targets for the parser and the formatter.** FuzzParse (no
panic, always a usable file) and FuzzFormatIdempotency (formatting
twice equals formatting once; clean input stays clean) seed
themselves from the repository's kernels, so the plain test suite
replays every seed in CI and `just fuzz` runs the mutation engine on
demand.
- **Oracle parity as its own CI step.** The push pipeline already ran
the live go-tool-asm comparison inside the suite; a dedicated step
now names that gate when it fails.
- **Man pages.** docs/man carries gasm(1) and one page per command,
written in roff: synopsis, description, every flag with its default,
exit status, worked examples and cross-references.
- **The parser and the formatter are fuzzed.** Two targets carry the
guarantee: no input makes the parser panic, and every input yields a
file the rest of the toolkit can work on; formatting twice equals
formatting once, and clean input stays clean. They seed from the
repository's own kernels, and `just fuzz` drives the mutation engine
on demand.
- **Man pages.** docs/man carries gasm(1) and a page for every command
except `version`, which gasm(1) documents itself, written in roff:
synopsis, description, every flag with its default, exit status,
worked examples and cross-references.
`just install-man` compresses them into ~/.local/share/man (MANDIR
overrides) and `just uninstall-man` removes them. A test builds the
binary and compares every command's `-h` output with its page, so the
pages cannot drift from the CLI.
binary and compares each page's flags and synopsis with its own `-h`
output, so the pages cannot drift from the CLI.
### Changed
- **Go 1.27.1 required.** The module declares `go 1.27.1`, so building
from source needs that patch release or newer.
- **Canonical just recipes.** `just gates` is the definition of done
(build, fmt-check, vet, test, race). `install` now builds and copies
the binary into `~/.local/bin` (`BINDIR` overrides) instead of
downloading module dependencies, and `install-bin` is gone. The test
gate sweeps the logic packages (arch through verify; the hardware-bound
`debug` and the thin `cmd/gasm` sit outside it), so the coverage floor
is computed over the product code and the number is identical locally
and in CI. `fuzz` requires its target package.
gate sweeps the logic packages (arch through verify; the ptrace-bound
`debug` and the thin `cmd/gasm` sit outside the coverage profile), so
the coverage floor is computed over the product code and the number is
identical locally and in CI; the two excluded packages' own tests run
in the gate and in the pipelines, outside the floor. `fuzz` requires
its target package.
- **The reported version comes from the build.** `gasm --version`
prints the version the toolchain recorded: the tag on a tagged
checkout, a pseudo-version naming the commit below one, `+dirty` on a
@@ -78,8 +87,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
synopsis, the commands, every flag with its default, the exit codes and
worked examples; `CONTRIBUTING.md` carries the Contributor terms and
states the commit trailer form, the one-logical-change rule and the
licence header rule. The repository's own assembly (the `verify`
trampolines and the test kernels) is in `gasm fmt` canonical form.
licence header rule; `SECURITY.md` states how a vulnerability is
reported and what to expect. The repository's own assembly (the
`verify` trampolines and the test kernels) is in `gasm fmt` canonical
form.
- **The README states the project's purpose and status.** It opens with
a warning that the tool is an experiment under active development,
version 0.x.x, free to change without warning, with 1.0.0 far off,
@@ -88,7 +99,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
Go toolchain), argues the case for the syntax in a new Why Plan 9
assembly section, and carries a Direction section: extended
instruction support, full GOOBJ and ELF compilation, Linux and
FreeBSD, and the four architectures.
FreeBSD, and the four architectures. A Validation status section
states what has been executed where: amd64 on real hardware, the other
three architectures under qemu-user emulation, the encoding parity on
the host for all four, and the debugger's ptrace path on amd64 only.
### Fixed
@@ -109,6 +123,199 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
lines entered the alignment width computation, and rendered `/ *`,
`> >` sequences re-lexed as comments and shifts. The label, width and
spacing rules now agree between passes.
- **`gasm fmt` deleted the `|` separators from TEXT and GLOBL flag
lists.** The lexer had no token for `|`, the formatter dropped the
resulting illegal token, and an in-place format silently rewrote
`NOSPLIT|DUPOK` as `NOSPLIT DUPOK`, which the Go assembler rejects.
The bars now round-trip byte-identically, and `·foo<ABIInternal>(SB)`
parses its ABI marker instead of swallowing `ABIInternal` and `SB`
into the flags, which produced false lint warnings on the standard
runtime spelling.
- **A malformed TEXT declaration crashed `gasm lint` and the language
server.** A TEXT line without a symbol left a nil name that lint and
the LSP dereferenced; both now carry on with a diagnostic. A branch
to a label at the end of a function body panicked the liveness
analysis the same way. A real NUL byte truncated the token stream
(everything after it was dropped); it is an illegal token now, invalid
UTF-8 no longer inflates byte offsets, and CRLF files format to
uniform LF.
- **The class-2 stack guard branched four bytes past its target.** When
the underflow branch relaxed to its 32-bit form, its displacement was
still computed as if the branch were two bytes long, so it landed
inside the morestack CALL instead of the compare that decides it.
The long form is reachable once a large frame carries a body of roughly
a hundred bytes.
- **Immediate operands wrapped silently on amd64.** Shift counts,
immediates beyond the operand's width and displacements beyond int32
truncated without a diagnostic (`SHLQ $300` assembled as `$44`); they
are range-checked now, matching `go tool asm`. EVEX scalar moves
(`VMOVSS Z1, Z2`) accepted forms the toolchain rejects and emitted
invalid encodings; `PUSHW`/`POPW` emit the 0x66-prefixed forms the
toolchain emits; `PUSHL` is rejected as illegal in 64-bit mode; a bare
zero-operand `JE` reports a diagnostic instead of panicking.
- **The arm64 shift and divide instructions encoded entirely different
operations.** `LSL`, `LSR`, `ASR` and `ROR`, immediate and register
forms, all encoded as `ORR`; `SDIV`/`UDIV` sat in the wrong opcode
space; `MADD`/`MSUB` never encoded the accumulate operand and silently
read X0 for it. All now match the toolchain byte for byte (new
differential kernels cover shifts, divides and multiplies), MADD
takes its four operands in the toolchain's order, and shift amounts at
or above the operand width are rejected.
- **arm64 multi-chunk immediates corrupted every branch that followed
them.** The size pass and the emitter disagreed on the expansion of
constants with three or more non-zero 16-bit chunks and of `MOVW $-1`,
so later label displacements were computed against the wrong offsets.
The size now comes from the encoder itself. Large-frame stack guards
(frames from roughly 64 KiB) branched to the wrong morestack entry,
and the pcsp and DWARF CFA boundaries for materialised large frames
are computed from the real prologue word counts.
- **arm64 immediates and addressing wrapped instead of erroring.**
Constants beyond the encodable range (`ADD $0x100000000`) wrapped to
zero, memory offsets wrapped at 2^31, exclusive and atomic accesses
silently ignored their offsets (`LDXR 8(R1)` read `[R1]`), and large
register-based offsets were routed through SP instead of the operand's
base. All four now either encode correctly or produce diagnostics.
- **riscv64 compressed stores with certain offsets wrote to the wrong
address.** The C.SD/C.SW/C.FSD immediate pattern dropped one bit, so
any register-relative store with offset bit 4 or 5 set targeted a
different address than the same-index load beside it. `FENCE`
assembled as `fence 0,0` instead of `fence iorw, iorw`. Branch and
jump displacements beyond ±4 KiB / ±1 MiB wrapped silently; they are
diagnostics now. The GOROOT width spellings (`MOVW 4(SP), X9`)
compress to their C.LW/C.SW forms exactly as the toolchain lowers
them, restoring byte parity for those shapes.
- **loong64 `MOVW $c, Fd` wrote a general register.** The immediate was
routed to the GPR of the F register's number (`MOVW $2, F4` clobbered
argument register R4), and the correct R30 + `movgr2fr.w` sequence was
unreachable. Two-operand `BLTU R4, label` encoded as `beqz`
(sometimes-taken where the toolchain's form is never-taken), and
out-of-range FP constants and BSTRINS/BSTRPICK bit numbers wrapped
silently; all are corrected or diagnosed.
- **ELF objects carried wrong relocation records.** The amd64
stack-guard TLS load relocated as R_X86_64_PC32 against the null
symbol (every non-NOSPLIT object mislinked); arm64 SB references
applied HI21 twice instead of the HI21/LO12 pair; riscv64 PCREL_LO12
referenced the target instead of its AUIPC site, which the system
linker rejects; riscv64 and loong64 e_flags declared the soft-float
ABI, so standard linkers refused the merge.
- **ELF DWARF was unparseable.** Eight abbrev-table constants were
wrong, the version-5 line header carried DWARF2-shaped tables, the
section count omitted `.debug_frame` (it sat past the section table,
invisible to every tool), the CIE hardcoded one architecture's
CFA and return-address registers for all four, and no DWARF address
was ever relocated: the `.rela.debug_info` and `.rela.debug_line`
records were computed and then discarded, so every address stayed
zero after linking. The tables parse in readelf, the registers are
per-architecture, `.rela.debug_info`, `.rela.debug_line` and
`.rela.debug_frame` are emitted, and addresses resolve after the
link; a data-only file emits a valid object instead of panicking, and
the DWARF records the real source path.
- **GOOBJ cross-package references resolved against the wrong object.**
External package indices were zero-based against a table that
reserves zero for the dummy invalid package, and symbol indices
ignored the hashed definition blocks between the sections, so a
reference into the first external package could bind to whatever
object the loader saw first. An end-to-end cross-package link pins
the chain. arm64 ADRP pairs now emit the toolchain's single 8-byte
relocation (the previous twin 4-byte records were a hard link error),
and symbols no longer claim the linkname flag the toolchain reserves
for `//go:linkname` declarations.
- **The arm64 JIT trampolines saved a scratch register as the stack
pointer.** `enterJIT` and its checked twin stored R3, a plain
caller-saved register on arm64, and restored RSP from it, so the
first JIT call would have returned to a garbage stack. The loong64
trampoline hands its leave address through the raw-symbol pattern the
arm64 one uses, avoiding the ABI wrapper's prologue.
- **The checked-ABI report flagged legal frames.** The red-zone canary
sat 64 bytes below the entry stack, so any kernel with a larger
declared frame reported "stack below SP written"; the guard now sizes
itself from the kernel's frame. The amd64 JIT tests are gated to
amd64 hosts (the suite previously SIGILL-crashed on the other three
architectures), fuzz signatures wider than the TEXT frame report
instead of panicking, `--buf` specifications are validated strictly
(a typo no longer verifies against a zeroed buffer), and ABI0
parameter sizes cover `string` and `complex` correctly.
- **amd64 hardware watchpoints never armed.** The debug registers were
poked at offsets inside `user_regs_struct`, corrupting five general
registers while the REPL reported success; they now use the real
u_debugreg window and stop on the watched address. loong64 watch
goes through the kernel's HW_WATCH regset (riscv64 reports the
kernel's interface as unsupported instead of failing obscurely).
- **`gasm debug` hung on the first faulting kernel.** Genuine
SIGSEGV/SIGBUS/SIGFPE/SIGILL stops were discarded as runtime noise
and the faulting instruction restarted forever; faults now surface as
reported stops. Conditional breakpoints with a false condition
resumed mid-instruction, `next` and `finish` evaluated traps with
stale registers and landed off instruction boundaries, and the
breakpoint restore covered one byte of the four-byte traps (arm64
silently skipped the instruction under it); the trap PCs follow the
kernel's reporting on every architecture. `regs` reports YMM from
the xstate (a struct-size overrun crashed FP register reads before),
V register halves print correctly on arm64, `unwatch` accepts the
architecture's slot range, `x <addr> -8` no longer crashes, break
conditions accept memory operands, and session scratch directories
are cleaned up.
- **The language server died on one malformed frame and corrupted
sources on rename.** A bad `Content-Length` or an unparsable JSON
body terminated the process instead of answering `-32700` and
continuing; rename and references covered the stripped name instead
of the full `·name` token, so renaming produced `·helper` minus its
last letter; documentHighlight never matched middle-dot symbols.
Positions are UTF-16 code units in both directions (astral characters
no longer shift columns) and responses always carry an explicit
`result` member.
- **The linter now recognises the `g` spelling of the goroutine
register.** `MOVD R0, g` clobbered R28 on arm64 (and the equivalents
on the other architectures) unflagged, and the numeric spellings the
linter did track are rejected by the toolchain there, so `g` was the
one spelling that escaped the audit. FUNCDATA and PCDATA literal
indices are validated against the ranges the runtime defines.
- **Usage errors exit 2 uniformly.** `audit-instructions` and
`scaffold` argument errors and an unknown `asm --format` exited 1 (or,
for `--format` without `-o`, exited 0 silently); the documented
exit-2 contract now holds, `asm -o` no longer prints the hex dump it
claimed to replace, and `verify --ground-truth` works for amd64
kernels on non-amd64 hosts instead of refusing with JIT advice.
- **arm64 store-exclusive instructions read their operands in the
toolchain's order.** `STXR` treated the first register as the status
register where `go tool asm` reads it as the data register, so the
same source assembled to different code in the two assemblers; the
pair forms (`STXP`, `LDXP` and their acquire/release variants) are
accepted now, in the toolchain spelling.
- **Large arm64 frames matched the toolchain's sequences.** A frame
beyond the immediate range that is not a movcon constant (roughly
64 KiB and up) made `gasm verify` report a false mismatch: the
toolchain splits the prologue subtraction into two 12-bit immediates
and materialises the non-leaf epilogue addition through the temporary
register; gasm emits the same sequences and the spadj boundaries
follow the real word counts.
- **The width spellings GOROOT uses assemble.** `MOVLQZX` (four uses in
`runtime/asm_amd64.s`), `MOVBQSX`, `MOVWQSX`, `MOVBLSX`, `MOVBWSX`,
`MOVBWZX` and `PMOVMSKB` (the bytealg kernels) encode byte-identically
with `go tool asm`, and the linter reports them encodable; a
`MOVLQZX` is the plain 32-bit move, exactly as the toolchain lowers
it.
- **`verify --ground-truth` no longer reports a mismatch for functions
whose size is not a multiple of 16.** The toolchain pads text symbols
to 16-byte boundaries; the comparison now checks the padding is zero
instead of comparing it, the same rule the test suite applies.
- **riscv64 accepts the `g` spelling of the goroutine register**, like
the other architectures, and the abi kernels use it; every verify
kernel is now ground-truth checkable (the numeric `X27` spelling the
kernels used is one `go tool asm` rejects).
- **Two more spellings GOROOT uses now assemble.** riscv64 `FCLASSD`
(classify a float64 into an integer mask) is encodable, and a
displacement written as a product (`0*8(X5)`, the toolchain's own
spelling in several kernels) parses instead of being rejected.
- **loong64 JIT execution enabled.** The loong64 trampoline is now
validated end to end under qemu-user emulation (plain and ABI-checked
calls, goroutine-clobber detection), so `gasm verify` runs the JIT
checks on loong64 hosts instead of forcing every loong64 kernel down
the ground-truth path. The arm64 and riscv64 trampolines carry the
same validation; the arm64 ABI test now seeds its kernel arguments
(a zeroed block made the passthrough check meaningless), and the
loong64 basic kernel's branch maze terminates on every path so the
smoke sweep cannot spin on leftover register values.
## [0.33.0] - 2026-09-14
@@ -327,7 +534,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
## [0.31.0] - 2026-08-20
The arm64 encoder (Phase 5; complete) ships with ELF64 and GOOBJ emission,
The arm64 encoder ships with ELF64 and GOOBJ emission,
verified byte-for-byte against `GOARCH=arm64 go tool asm` and linked into a
real `go build`. The encoder covers the full integer instruction set, FP
arithmetic, conditional select, CRC32, and the MOV pseudo-instruction with
@@ -335,7 +542,7 @@ bitmask immediate encoding. The project now requires Go 1.27.
### Added
- **arm64 encoder (Phase 5; complete).** `gasm asm` can now assemble `_arm64.s`
- **arm64 encoder.** `gasm asm` can now assemble `_arm64.s`
files: the AArch64 integer instruction set with the MOV pseudo-instruction and
its immediate-constant expansions (MOVZ/MOVN/MOVK for wide immediates, ORR with
logical bitmask encoding for values like `$1`), data-processing (shifted
@@ -343,8 +550,9 @@ bitmask immediate encoding. The project now requires Go 1.27.
immediate), conditional and unconditional branches, FP/SP frame mapping,
SB/global symbol references (ADRP+ADD pairs with `R_ADDRARM64` relocations),
jump chain folding, and ELF64 emission (`gasm asm --format elf`). Ground-truth
verification against `GOARCH=arm64 go tool asm` matches byte-for-byte. Phase 5
(the other architectures; RISC-V, LoongArch, arm64) is now complete.
verification against `GOARCH=arm64 go tool asm` matches byte-for-byte. The
encoder set for the remaining architectures (RISC-V, LoongArch, arm64) is
complete.
### Changed
@@ -354,7 +562,7 @@ bitmask immediate encoding. The project now requires Go 1.27.
## [0.30.0] - 2026-08-13
The LoongArch encoder (Phase 5) ships with ELF64 and GOOBJ emission, verified
The LoongArch encoder ships with ELF64 and GOOBJ emission, verified
byte-for-byte against `GOARCH=loong64 go tool asm` and linked into a real
`go build`; the shared GOOBJ emitter now writes the per-function DWARF symbols
the linker's DWARF pass reads. The RISC-V encoder reaches byte-for-byte parity
@@ -365,7 +573,7 @@ tracks four hardware watchpoint slots, and the toolkit is Linux-only.
### Added
- **LoongArch encoder (Phase 5).** `gasm asm` can now assemble `_loong64.s`
- **LoongArch encoder.** `gasm asm` can now assemble `_loong64.s`
files: the full LoongArch64 instruction set with the dual-form arithmetic
mnemonics, the 16/21-bit branch families, the MOV pseudo-instruction and
its immediate-constant expansions, FP/SP frame mapping, SB/global symbol
@@ -454,7 +662,7 @@ tracks four hardware watchpoint slots, and the toolkit is Linux-only.
- **Linux only.** The toolkit, its CI and the released binaries are now
Linux-only; cross-compiled to linux/{amd64,arm64,riscv64,loong64}.
- **Phase 4 closed.** README's "Remaining" list for the debugger is gone;
- **Debugger complete.** README's "Remaining" list for the debugger is gone;
disassembly at PC, memory-write, watchpoints, and source-line mapping are
all shipped.
@@ -664,7 +872,7 @@ exposes the full dynamic-analysis toolkit.
## [0.20.0] - 2026-07-25
Coverage profiling: the third pillar of Phase 3. Static basic-block
Coverage profiling. Static basic-block
enumeration from the assembler's label map, combined with multi-input path
diversity measurement; how many observationally distinct execution paths a
test corpus exercises.
@@ -689,7 +897,7 @@ execute) without fighting the runtime.
## [0.19.0] - 2026-07-24
Runtime ABI checks: the second pillar of Phase 3. The JIT trampoline now
Runtime ABI checks. The JIT trampoline now
has an ABI-checking variant that sets sentinels in the callee-saved registers
(BP, R14) before entering the assembled function and verifies they survive on
return, plus a red-zone canary (128 bytes below SP filled with 0xA5) that
@@ -724,7 +932,7 @@ codes. This is the automated form of the project's bit-identical contract.
## [0.17.0] - 2026-07-22
Phase 3 begins: dynamic analysis. A JIT execution substrate that assembles
Dynamic analysis. A JIT execution substrate that assembles
Plan 9 amd64 kernels into executable memory and calls them directly; pure Go
(stdlib only, `syscall.Mmap` + an assembly trampoline), no cgo, no external
toolchain.
@@ -1131,8 +1339,8 @@ support).
## [0.2.0] - 2026-07-07
The Phase 2 assembler grows the SIMD set: shuffles, extract/insert, permute
and the moves, on top of the Phase 1 VEX forms.
The assembler grows the SIMD set: shuffles, extract/insert, permute
and the moves, on top of the VEX forms of the first release.
### Added
@@ -1168,7 +1376,7 @@ and the moves, on top of the Phase 1 VEX forms.
## [0.1.0] - 2026-07-06
Initial release; the Phase 1 foundation.
Initial release: the foundation.
### Added
+15 -8
View File
@@ -24,7 +24,9 @@ below; submitting one means you accept them.
## Development setup
Requirements: Go 1.27.1, the exact version the `go` directive in `go.mod`
declares, and [just](https://github.com/casey/just) for the recipes.
declares, [just](https://github.com/casey/just) for the recipes, and a C
compiler (gcc), because `just gates` includes `just race` and the race
detector needs cgo.
```sh
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
@@ -54,15 +56,20 @@ workflow builds the assets and publishes the release and its notes.
## Code style
`gofmt` and `go vet` run through `just fmt` and `just vet`, with zero diff and zero
warnings tolerated. `just gates` is the definition of done in one command, and the recipe
file names what it contains. Errors are checked explicitly, wrapped as
`fmt.Errorf("context: %w", err)`, and nothing panics outside `main`. The `golang`
skill holds the rules the project follows; the recipe file holds the commands.
warnings tolerated. `just vet` is two gates, `go vet ./...` and `go fix -diff ./...`,
so the modernisation rewrites are enforced too. `just gates` is the definition of done in
one command, and the recipe file names what it contains. Errors are checked explicitly,
wrapped as `fmt.Errorf("context: %w", err)`, and nothing panics outside `main`. The
recipe file holds the commands, and the language and standard-library surface is the one
the `go` directive in `go.mod` pins.
- `golang.org/x/arch` is the one module dependency, and it is linked into the binary:
`gasm dis` and the debugger's listings decode through it. Everything else is the
standard library.
- No cgo, no C, no external toolchain at runtime.
- No cgo and no C. The standalone encoder paths (`gasm asm --format raw` and `--format
elf`) need no Go installation; `gasm verify --ground-truth`, `gasm verify --fuzz`,
`gasm audit-instructions` and `gasm asm --format goobj` resolve through the installed
Go toolchain.
- The parser, lexer and formatter are hand-written; the `arch` instruction tables are
generated only by `_gen/gen.go` (`just gen`) and never edited by hand.
- Assembly committed to the repository goes through `gasm fmt` and `gasm lint`, so a
@@ -110,8 +117,8 @@ Workflows live in `.gitea/workflows/` and run on the project's own runners:
| Workflow | Trigger | What it does |
|---|---|---|
| Test | push or pull request to `development` | build, format check, vet, modernisation, the test suite with the coverage floor |
| Release | a `v*` tag | the same gates as Test, then the matrix build, the proven version and the release itself; the race detector runs locally in `just gates` before the tag is cut |
| Test | push or pull request to `development` | build, format check, vet, modernisation, the test suite with the coverage floor, the CLI and debugger tests outside the profile, then the oracle-parity rerun against `go tool asm` |
| Release | a `v*` tag | the same gates as Test minus the oracle-parity step, then the matrix build, the version smoke test and the release itself; the race detector runs locally in `just gates` before the tag is cut |
The local equivalent is `just gates`, which is the same set plus the race detector. The
race detector also has its own workflow, dispatched by hand; it never runs on a push or a
+54 -19
View File
@@ -5,23 +5,25 @@
> output formats and behaviour can change without warning at any time.
> A 1.0.0 release is light years away. Nothing in this document is a
> stability promise. For all of that, this is not a paper project: gasm
> is already in active use and is tested on real assembly work.
> is already in active use and is tested on real assembly work. Only
> amd64 is validated on real hardware; the other three architectures run
> under emulation ([Validation status](#validation-status)).
**GAsm** is Go's Plan 9 assembler, and Go ships it without tooling:
there is no formatter, no linter, no static analyser, no standalone
assembler and no debugger for `.s` files. Developers write assembly
blind, validate it by benchmark, and debug it by print statement.
gasm-devkit is the missing toolkit: a single, self-contained binary,
`gasm`, that serves both purposes.
there is no formatter, no linter and no debugger for `.s` files, and no
assembler that works without a Go installation. Developers write
assembly blind, validate it by benchmark, and debug it by print
statement. gasm-devkit is the missing toolkit: a single, self-contained
binary, `gasm`, that serves both purposes.
- **Help develop Plan 9 assembly.** Formatting, linting, disassembly,
dynamic verification, a source-level debugger and a language server,
for `.s` files in Go programs.
- **Use Plan 9 assembly outside the Go toolchain.** `gasm asm` encodes
on its own, with no Go installation in the loop, and writes raw
images, linkable ELF objects with DWARF5 debug sections, or the Go
on its own and writes raw images or linkable ELF objects with DWARF5
debug sections, with no Go installation in the loop; the Go
toolchain's own GOOBJ format, which `go build` consumes in place of
the toolchain's output.
the toolchain's output, needs the installed toolchain.
## Why Plan 9 assembly
@@ -69,12 +71,14 @@ to give that syntax the tooling it deserves.
operating recursively on directories the way `go fmt` does. `-l` lists
files whose formatting differs and `-d` prints a unified diff.
- **Linter.** `gasm lint` runs 18 conservative static checks, among them
`undefined-label`, `abi-argsize` (declared frame vs the `// func` signature),
`register-clobber` (Go ABI register liveness over the control-flow graph),
`stack-imbalance`, `abi0-register-args` and `unencodable-instruction`.
`undefined-label`, `abi-argsize` (declared argument area vs the `// func`
signature), `register-clobber` (Go ABI register liveness over the
control-flow graph), `stack-imbalance`, `abi0-register-args` and
`unencodable-instruction`.
- **Standalone assembler.** `gasm asm` encodes all four architectures without
the Go toolchain and writes raw images, linkable ELF objects (with DWARF5
debug sections) or the Go toolchain's own GOOBJ format, which `go build`
the Go toolchain and writes raw images or linkable ELF objects (with DWARF5
debug sections) with no Go installation needed, or the Go toolchain's own
GOOBJ format, which needs the installed toolchain and which `go build`
consumes in place of the toolchain's output. Framed functions get the
stack-split guard and the morestack block, byte-identical to the
toolchain's, so split functions link too.
@@ -86,7 +90,8 @@ to give that syntax the tooling it deserves.
byte-for-byte ground-truth comparison of the machine code.
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
breakpoints (optionally conditional), hardware watchpoints, register and
memory inspection, and headless script runs with label-level coverage.
memory inspection, and headless script runs that report instruction and
label coverage.
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
push and pull diagnostics, semantic-token highlighting, go-to-definition,
find references, rename, formatting, inlay hints, code actions, signature
@@ -119,10 +124,38 @@ can emit today is narrower, and a recognised but unencodable instruction is
reported as an explicit error, never as a wrong byte.
The same measurement runs over GOROOT's whole assembly corpus:
`gasm audit-instructions --corpus` reports 108 of 627 files (17.2 %)
`gasm audit-instructions --corpus` reports 127 of 627 files (20.3 %)
assembling for every target architecture today, with the top failure
reasons per architecture; the number moves with every release.
### Validation status
**Only amd64 is validated on real hardware.** The other three
architectures are validated under qemu-user emulation, because the
project owns no arm64, riscv64 or loong64 machine, and emulation is the
only substitute available for the hardware. The distinction matters and
is stated rather than implied: everything below is a claim about what has
actually been executed.
| Layer | amd64 | arm64, riscv64, loong64 |
|---|---|---|
| Encoding: byte-for-byte against `go tool asm` | native hardware | native hardware (the toolchain cross-assembles any GOARCH on any host) |
| Execution: JIT calls, ABI checks, differential fuzzing | native hardware | qemu-user emulation |
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more |
Consequences, stated plainly. An emulator is a model of a CPU, not the
CPU: instruction semantics are implemented in software and can differ
from silicon in ways a test suite does not reveal. A kernel that passes
under qemu-user is therefore not proven correct on real hardware, and a
discrepancy found on real hardware is a defect in gasm, reported like any
other. Encoding parity is the exception: the byte comparison against the
toolchain runs on the host for every architecture, so no emulator stands
between the claim and the evidence. The debugger is the weakest case: on
the three emulated architectures its per-architecture ptrace code has
been compiled and read, never executed. Its architecture-neutral units
run under `go test ./...`, which the race workflow and a manual run
perform; the default `just test` gate does not sweep `./debug/...`.
## Direction
The plan, in the order it is being worked:
@@ -134,7 +167,8 @@ The plan, in the order it is being worked:
an extended instruction set the toolchain does not know at all. The
toolchain-derived tables stay generated and untouched; only the
extended instructions are hand-maintained, with their own spellings
and encoders, verified by execution on real hardware because the
and encoders, verified by execution (on real hardware for amd64, under
emulation for the rest, per the validation status above) because the
toolchain offers no ground truth to compare against. The gaps exist
on every architecture, amd64 included.
- **Full GOOBJ and ELF compilation.** The destination is a complete,
@@ -205,7 +239,7 @@ gasm verify --ground-truth k.s # byte-for-byte vs go tool asm
gasm verify --fuzz k.s # differential fuzz vs the go tool asm build
gasm debug --func name k.s # interactive debugger
gasm debug --func name --script cmds.txt --timeout 30s k.s # headless run
gasm debug --func name --cover k.s # which labels did execution reach?
gasm debug --func name --cover k.s # instruction and label coverage
gasm diff a.s b.s # compare machine code byte-for-byte
gasm diff --map wideCopyAVX2=wideCopyAVX512 avx2.s avx512.s
gasm profile k.s # show basic-block structure
@@ -243,7 +277,8 @@ recipe.
- [docs/CLI.md](docs/CLI.md): full command reference
- man pages: `just install-man` installs gasm(1) and one page per command
into ~/.local/share/man (MANDIR overrides); `just uninstall-man` removes
except `version`, which is documented inside gasm(1) instead, into
~/.local/share/man (MANDIR overrides); `just uninstall-man` removes
them
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
+3 -2
View File
@@ -7,7 +7,7 @@ releases do not receive them.
| Version | Supported |
|---|---|
| 0.33.0 | yes |
| 0.34.0 | yes |
| older releases | no |
## Reporting a vulnerability
@@ -29,7 +29,8 @@ Include:
- You are kept informed while the fix is being made, and told when it ships.
- The fix is released before the details are published, and the timing is agreed with
you.
- The reporter is credited in the release notes unless they ask otherwise.
- The fix ships without naming you: the project keeps no credits list, so the release
notes, the changelog and the commits name no reporter.
## Out of scope
+3
View File
@@ -55,6 +55,7 @@ const (
Mask // AVX-512 mask register (K)
Float // arm64 floating-point register (F)
VecARM // arm64 SIMD/vector register (V)
VecSIMD // architecture-neutral SIMD/vector register (LoongArch LSX/LASX)
Special // architecture-special register
)
@@ -73,6 +74,8 @@ func (c RegClass) String() string {
return "float"
case VecARM:
return "vector (arm64)"
case VecSIMD:
return "vector"
case Special:
return "special"
default:
+1 -1
View File
@@ -145,7 +145,7 @@ func arm64Curated() []Instr {
for _, op := range []string{
"LDAXR", "LDAXRB", "LDAXRH", "LDAXRW", "STXR", "STXRB", "STXRH", "STXRW",
"LDAR", "LDARB", "LDARH", "LDARW", "STLR", "STLRB", "STLRH", "STLRW",
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL", "CASAL",
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL",
} {
t = append(t, i(op, "Atomic memory operation"))
}
+2 -2
View File
@@ -31,10 +31,10 @@ func loong64Registers() []Register {
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
}
for i := 0; i <= 31; i++ {
add(fmt.Sprintf("V%d", i), VecARM, "LSX 128-bit vector register")
add(fmt.Sprintf("V%d", i), VecSIMD, "LSX 128-bit vector register")
}
for i := 0; i <= 31; i++ {
add(fmt.Sprintf("X%d", i), VecARM, "LASX 256-bit vector register")
add(fmt.Sprintf("X%d", i), VecSIMD, "LASX 256-bit vector register")
}
return regs
}
+69
View File
@@ -4,6 +4,7 @@
package asm
import (
"encoding/binary"
"os"
"os/exec"
"path/filepath"
@@ -61,6 +62,62 @@ TEXT ·add(SB), NOSPLIT, $0-24
}
}
// TestGOObjectAARCH64PairReloc pins the ADRP-pair relocation shape against
// the toolchain's own object for the same source: exactly one R_ADDRARM64
// of Siz 8 at the ADRP word (cmd/internal/obj/arm64/asm7.go adds a single
// Siz-8 relocation per pair and the linker patches both instructions from
// it). gasm's assembler records the ADRP+ADD form as two word relocs; the
// emitter must coalesce them, not emit two Siz-4 records.
func TestGOObjectAARCH64PairReloc(t *testing.T) {
f, errs := parser.Parse("gv_arm64.s", `
#include "textflag.h"
TEXT ·getv(SB), NOSPLIT, $0-8
MOVD $v<>(SB), R4
MOVD R4, ret+0(FP)
RET
GLOBL v<>(SB), RODATA, $8
DATA v<>+0(SB)/8, $7
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.GOObjectAARCH64("main", "gv_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
v := openGoobj(t, obj)
relocs := v.blk(blkReloc)
le := binary.LittleEndian
// Two DWARF relocs on the lines/DIE symbols, then the code's one pair
// relocation.
if len(relocs) != 3*23 {
t.Fatalf("relocs = %d bytes, want three entries", len(relocs))
}
cr := relocs[2*23:]
if off := int32(le.Uint32(cr[0:])); off != 0 {
t.Errorf("pair reloc off = %d, want 0 (the ADRP word)", off)
}
if siz := cr[4]; siz != 8 {
t.Errorf("pair reloc siz = %d, want 8", siz)
}
if typ := le.Uint16(cr[5:]); typ != relocArm64Addr {
t.Errorf("pair reloc type = %d, want %d (R_ADDRARM64)", typ, relocArm64Addr)
}
if pkg := le.Uint32(cr[15:]); pkg != pkgIdxSelf {
t.Errorf("pair reloc PkgIdx = %#x, want pkgIdxSelf", pkg)
}
// The GLOBL is the first package definition.
if sym := le.Uint32(cr[19:]); sym != 0 {
t.Errorf("pair reloc SymIdx = %d, want 0 (the GLOBL definition)", sym)
}
}
// TestGOObjectAARCH64Link does an end-to-end link test: it cross-compiles a
// Go program for arm64, substitutes the gasm-produced object into the package
// archive, re-links with cmd/link, and verifies the symbol appears in the
@@ -79,6 +136,14 @@ TEXT ·add(SB), NOSPLIT, $0-24
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
TEXT ·getv(SB), NOSPLIT, $0-8
MOVD $v<>(SB), R4
MOVD R4, ret+0(FP)
RET
GLOBL v<>(SB), RODATA, $8
DATA v<>+0(SB)/8, $7
`
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
@@ -86,11 +151,15 @@ TEXT ·add(SB), NOSPLIT, $0-24
mainSrc := `package main
func add(a, b int64) int64
func getv() *int64
func main() {
if add(20, 22) != 42 {
panic("bad add")
}
if getv() == nil {
panic("bad getv")
}
}
`
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
+265 -99
View File
@@ -189,7 +189,7 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo) int {
return arm64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "SUB", "SUBW", "AND", "ANDW", "ORR", "ORRW", "EOR", "EORW":
if len(ops) >= 2 && isImmOperand(ops[0]) {
v := immFromOperand(ops[0])
v := arm64Imm64(ops[0])
// Small immediate (0..4095 or -2048..-1) fits in one instruction.
if v >= 0 && v <= 0xFFF {
return 4
@@ -221,7 +221,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
if len(ops) != 1 {
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
return a64wordLE(uint32(immFromOperand(ops[0]))), nil
w := arm64Imm64(ops[0])
if w < 0 || w > 0xFFFFFFFF {
return nil, fmt.Errorf("WORD: immediate %d does not fit a 32-bit word", w)
}
return a64wordLE(uint32(w)), nil
case "B", "JMP":
return encodeARM64Branch(mnem, ops, pc, offsets, false, relocs, resolve)
case "BL", "CALL":
@@ -249,14 +253,19 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
}
}
// Shifts: immediate forms alias SBFM/UBFM/EXTR, register forms are the
// two-source LSLV/LSRV/ASRV/RORV.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FShift {
return encodeARM64Shift(mnem, enc.op, ops)
}
// Multiply-accumulate: MADD/MSUB Rm, Ra, Rn, Rd.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDPR4 {
return encodeARM64MAddSub(mnem, enc.op, ops)
}
// Register-register data processing.
// ASR/LSL/LSR/ROR with immediate operands use bitfield encoding (SBFM/UBFM).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDPSR {
isShift := mnem == "ASR" || mnem == "ASRW" || mnem == "LSL" || mnem == "LSLW" ||
mnem == "LSR" || mnem == "LSRW" || mnem == "ROR" || mnem == "RORW"
if isShift && len(ops) >= 2 && isImmOperand(ops[0]) {
return encodeARM64Bitfield(mnem, enc.op, ops)
}
return encodeARM64DPSR(mnem, enc.op, ops)
}
@@ -305,7 +314,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64CRC32(mnem, enc.op, ops)
}
// Exclusive load/store (LDXR, STXR, LDAXR, STLXR).
// Exclusive load/store (LDXR, STXR, LDAXR, STLXR and the register-pair
// forms LDXP, STXP).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FExcl {
return encodeARM64Excl(mnem, enc.op, ops)
}
@@ -478,6 +488,88 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
// encodeARM64Shift encodes LSL/LSR/ASR/ROR in both widths. The operand order
// is source first, destination last: OP $sh|Rm, Rn, Rd or OP $sh|Rm, Rd.
// With an immediate the shift is the SBFM/UBFM (ROR: EXTR) alias, with a
// register it is the data-processing (2 source) LSLV/LSRV/ASRV/RORV; the
// two-source opcode rides the same 0xd6<<21 field as SDIV/UDIV, with
// LSLV=0b001000, LSRV=0b001001, ASRV=0b001010, RORV=0b001011 at bits 15:10.
func encodeARM64Shift(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
if isImmOperand(ops[0]) {
width := uint32(64)
if strings.HasSuffix(mnem, "W") {
width = 32
}
sh := arm64Imm64(ops[0])
if sh < 0 || uint32(sh) >= width {
return nil, fmt.Errorf("%s: shift amount %d out of range for %d-bit form", mnem, sh, width)
}
switch mnem {
case "LSL", "LSLW":
// UBFM Rd, Rn, #(-sh) mod W, #(W-1)-sh
immr := (width - uint32(sh)) % width
return a64wordLE(baseOp | immr<<16 | (width-1-uint32(sh))<<10 | uint32(rn)<<5 | uint32(rd)), nil
case "LSR", "LSRW":
// UBFM Rd, Rn, #sh, #(W-1)
return a64wordLE(baseOp | uint32(sh)<<16 | (width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
case "ASR", "ASRW":
// SBFM Rd, Rn, #sh, #(W-1)
return a64wordLE(baseOp | uint32(sh)<<16 | (width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
default:
// ROR, RORW: EXTR Rd, Rn, Rn, #sh (Rm = Rn, imms = sh).
return a64wordLE(baseOp | uint32(rn)<<16 | uint32(sh)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
}
rm := arm64RegNum(operandRegName(ops[0]))
if rm < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
op2 := uint32(8) // LSLV
switch mnem {
case "LSR", "LSRW":
op2 = 9 // LSRV
case "ASR", "ASRW":
op2 = 10 // ASRV
case "ROR", "RORW":
op2 = 11 // RORV
}
sf := uint32(1)
if strings.HasSuffix(mnem, "W") {
sf = 0
}
return a64wordLE(sf<<31 | 0xd6<<21 | op2<<10 | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64MAddSub encodes MADD/MSUB/MADDW/MSUBW. The toolchain's operand
// order is Rm, Ra, Rn, Rd (its optab case 15 comment says exactly that), so
// the accumulate register is the SECOND operand: base | Rm<<16 | Ra<<10 |
// Rn<<5 | Rd. The optab has no shorter row for these mnemonics, so all four
// operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the
// MUL mnemonic, not to these.
func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops))
}
rm := arm64RegNum(operandRegName(ops[0]))
ra := arm64RegNum(operandRegName(ops[1]))
rn := arm64RegNum(operandRegName(ops[2]))
rd := arm64RegNum(operandRegName(ops[3]))
if rm < 0 || rn < 0 || ra < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(ra)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
// ---- ADD/SUB immediate ----
// encodeARM64AddSubImm encodes an ADD/SUB immediate instruction.
@@ -485,7 +577,7 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
v := immFromOperand(ops[0])
v := arm64Imm64(ops[0])
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
rn := rd
if len(ops) == 3 {
@@ -529,6 +621,8 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 {
return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil
}
// The imm12 field cannot carry the value; rejecting (rather than
// truncating) matches the toolchain, which reports the same shape.
return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v)
}
@@ -615,14 +709,15 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
return 8 // ADRP + ADD
}
v := arm64Imm64(src)
if v == 0 {
// Size the immediate exactly as the encoder will emit it: multi-chunk
// values expand to up to four words and the W forms truncate first.
// Anything else would desynchronise the label offsets of pass 1 from
// the bytes pass 2 lays down, corrupting every later branch.
b, err := encodeARM64LoadImm(31, arm64Imm64(src), mnem)
if err != nil {
return 4
}
if arm64Movcon(v) >= 0 || arm64Movcon(^v) >= 0 {
return 4
}
return 8 // MOVZ + MOVK
return len(b)
case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB":
return 8 // ADRP + LDR
case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB":
@@ -638,7 +733,7 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
if !ok {
lt = a64LoadTable["MOVD"] // the MOV pseudo is a 64-bit access
}
scale := int32(1) << uint(lt.size)
scale := int64(1) << uint(lt.size)
if off >= 0 && off%scale == 0 && off/scale < 4096 {
return 4
}
@@ -655,51 +750,60 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
}
// encodeARM64LoadImm loads an immediate into a register, matching the
// toolchain's MOVZ/MOVN/MOVK sequence.
// toolchain's MOVZ/MOVN/MOVK sequence. W forms truncate to 32 bits first and
// every classification (movcon, complement, chunk count) runs on the truncated
// value, so a 32-bit immediate never reaches the 64-bit halves: MOVW $-1
// truncates to 0xFFFFFFFF, whose complement is a single zero chunk, and encodes
// as MOVN W, #0.
func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
d := v
// For 32-bit MOVW, zero-extend.
sf := uint32(1) // 64-bit
if mnem == "MOVW" || mnem == "MOVWU" {
d = int64(uint32(v))
sf = 0
}
if d == 0 {
// ORR Rd, ZR, ZR (MOV $0, Rd)
op := uint32(1<<31 | 1<<29 | 0x0a<<24) // ORR 64-bit
if mnem == "MOVW" || mnem == "MOVWU" {
if sf == 0 {
op = 0<<31 | 1<<29 | 0x0a<<24 // ORR 32-bit
}
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
}
sf := uint32(1) // 64-bit
if mnem == "MOVW" || mnem == "MOVWU" {
sf = 0
}
// The Go toolchain classifies immediates:
// - C_ABCON0 (0 < v ≤ 4095): bitmask first for positive values
// - Negative values: MOVN first, then bitmask
// - C_MOVCON (movcon-eligible, outside ABCON range): MOVZ/MOVN first
tryBitmaskFirst := d > 0 && d <= 0xFFF
// The Go toolchain classifies immediates (asm7.go conclass):
// - inside the imm12/shifted-imm12 "addcon" band (C_ABCON0/C_ABCON,
// 0 < v ≤ 4095 or a 4096 multiple up to 0xFFF000): bitmask first, so
// `MOVD $4096, R27` is ORR $4096, not MOVZ $(1<<12)
// - outside that band: MOVZ/MOVN first (C_MOVCON before C_BITCON), and
// negative values reach MOVN before the bitmask test
tryBitmaskFirst := d > 0 && (d <= 0xFFF || (d&0xFFF == 0 && d <= 0xFFF000))
if tryBitmaskFirst {
// Small immediate: try bitmask first (Go uses ORR for values like $1, $256).
// Addcon-band immediate: try bitmask first (Go uses ORR for values
// like $1, $256 and $65536).
N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf))
if ok {
return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil
}
}
// Try MOVZ (single non-zero 16-bit chunk).
// Try MOVZ (single non-zero 16-bit chunk) and MOVN (single non-0xFFFF
// chunk of the complement). The W forms must look inside the 32-bit
// window only, so the complement is masked to the operand width; d is
// already truncated and needs no mask.
width := uint64(0xFFFFFFFF)
if sf == 1 {
width = 0xFFFFFFFFFFFFFFFF
}
s := arm64Movcon(d)
if s >= 0 {
return a64wordLE(a64MoveWide(sf, 2, uint32(s>>4), uint32((d>>uint(s))&0xFFFF), uint32(rd))), nil
}
// Try MOVN (single non-0xFFFF 16-bit chunk of ^d).
sn := arm64Movcon(^d)
sn := arm64Movcon(^d & int64(width))
if sn >= 0 {
return a64wordLE(a64MoveWide(sf, 0, uint32(sn>>4), uint32((^d>>uint(sn))&0xFFFF), uint32(rd))), nil
return a64wordLE(a64MoveWide(sf, 0, uint32(sn>>4), uint32(((^d)>>uint(sn))&0xFFFF), uint32(rd))), nil
}
// For values outside the bitmask-first range that are not movcon: try bitmask.
@@ -710,7 +814,7 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
}
}
// Multi-instruction: MOVZ + MOVK for each non-zero16-bit chunk.
// Multi-instruction: MOVZ + MOVK for each non-zero 16-bit chunk.
var ws []uint32
first := true
for i := range 4 {
@@ -805,7 +909,7 @@ func arm64Bitmask(v uint64, sf int) (N, immr, imms uint32, ok bool) {
// Integer → integer: ORR Rd, ZR, Rs.
// FP → FP: FMOV Fd, Fn (FP data processing).
// FP ↔ GP: FMOV general (FPCVTI encoding).
// Go Plan 9 syntax: MOV dst, src (first operand = destination).
// Go Plan 9 syntax is source first, destination last: MOV src, dst.
func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
rs := arm64RegNum(operandRegName(src))
rd := arm64RegNum(operandRegName(dst))
@@ -826,8 +930,7 @@ func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
}
// GP ↔ FP: FMOV general (FPCVTI encoding).
// Go syntax: FMOV FPdst, GPsrc or FMOV GPdst, FPsrc.
// First operand = destination, second = source.
// Go syntax: FMOV GPsrc, FPdst or FMOV FPsrc, GPdst, source first.
if sc == arm64ClsFP && dc == arm64ClsGR {
// FP → GP: FMOV Wd/Xd, Sn/Dn. opcode bits[20:16]=6.
sf, typ := uint32(0), uint32(0)
@@ -867,7 +970,7 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6
lt = a64LoadTable["MOVD"]
}
scale := int32(1) << uint(lt.size)
scale := int64(1) << uint(lt.size)
storeOpc := a64StoreOpc(lt)
var opc int
if load {
@@ -880,30 +983,31 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(off/scale), uint32(rn), uint32(reg))), nil
}
if off >= -256 && off <= 255 {
return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, off, rn, reg)), nil
return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, int32(off), rn, reg)), nil
}
// Large offset: materialise the base in REGTMP (R27) the way the
// toolchain does and access what remains.
// toolchain does and access what remains. The ADD offsets from the
// operand's own base register, [SP] and [Rn] alike.
addImm, addShift, access, ok := arm64SplitOffset(off, scale)
if !ok {
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
}
return a64WordsLE(
a64AddSub(1, 0, 0, addShift, uint32(addImm), 31, 27), // ADD $addImm<<shift, SP, R27
a64AddSub(1, 0, 0, addShift, addImm, uint32(rn), 27), // ADD $addImm<<shift, Rn, R27
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(access/scale), 27, uint32(reg)),
), nil
}
// arm64SplitOffset decomposes an out-of-range frame offset for a REGTMP
// base: an ADD (plain, or shifted left by 12) brings SP near the target and
// the access covers what remains. ok is false when no decomposition exists
// (offsets at or beyond 16 MiB, where the toolchain falls back to a literal
// pool).
func arm64SplitOffset(off int32, scale int32) (addImm, addShift uint32, access int32, ok bool) {
// arm64SplitOffset decomposes an out-of-range offset for a REGTMP base: an
// ADD (plain, or shifted left by 12) brings the base near the target and the
// access covers what remains. ok is false when no decomposition exists
// (negative offsets, or beyond 16 MiB, where the toolchain falls back to a
// literal pool).
func arm64SplitOffset(off int64, scale int64) (addImm, addShift uint32, access int64, ok bool) {
if off < 0 {
return 0, 0, 0, false
}
// Plain ADD: bring SP to within the largest scaled access.
// Plain ADD: bring the base to within the largest scaled access.
l := min(off, 4095*scale)
l -= l % scale
if a := off - l; a <= 4095 {
@@ -987,12 +1091,15 @@ func arm64Imm64(op *ast.Operand) int64 {
}
// arm64MemWithFrame resolves a memory operand, translating FP/SP pseudo-
// registers via the frame mapping.
func arm64MemWithFrame(op *ast.Operand, fi arm64FrameInfo) (rn int, off int32) {
// registers via the frame mapping. The offset stays 64-bit: the AST carries
// int64 displacements and truncating here would wrap offsets beyond 2^31
// silently.
func arm64MemWithFrame(op *ast.Operand, fi arm64FrameInfo) (rn int, off int64) {
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" {
return arm64ResolvePseudo(op.Addr.Sym, fi)
base, pseudo := arm64ResolvePseudo(op.Addr.Sym, fi)
return base, int64(pseudo)
}
return arm64RegNum(op.Addr.Base), int32(op.Addr.Offset)
return arm64RegNum(op.Addr.Base), op.Addr.Offset
}
// arm64Label returns the label name of an operand.
@@ -1068,7 +1175,7 @@ func encodeARM64FPCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, e
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
// Check if first operand is #0 (compare with zero): FCMP $0.0, Fn.
if isImmOperand(ops[0]) && immFromOperand(ops[0]) == 0 {
if isImmOperand(ops[0]) && arm64Imm64(ops[0]) == 0 {
rn := arm64RegNum(operandRegName(ops[1]))
if rn < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
@@ -1105,8 +1212,11 @@ func encodeARM64FPCCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
if rm < 0 || rn < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
nzcv := uint32(immFromOperand(ops[3]))
return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | nzcv&0xF), nil
nzcv := arm64Imm64(ops[3])
if nzcv < 0 || nzcv > 0xF {
return nil, fmt.Errorf("%s: nzcv %d out of range (0..15)", mnem, nzcv)
}
return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(nzcv)&0xF), nil
}
// encodeARM64FPSel encodes a FP conditional select.
@@ -1233,32 +1343,94 @@ func encodeARM64CRC32(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, e
// ---- Atomics encoding ----
// encodeARM64Excl encodes an exclusive load/store instruction.
// LDXR (Rn), Rt → LDXR Rt, [Rn] (2 operands: mem, reg or reg, mem)
// STXR Rs, (Rn), Rt → STXR Rs, Rt, [Rn] (3 operands: Rs, mem, Rt-status)
// arm64ExclMem resolves the memory operand of an exclusive or atomic
// instruction. These encodings have no immediate field: the toolchain
// rejects `LDXR 8(R1), R2` as an illegal combination, so a non-zero offset is
// reported rather than silently dropped (which would read the wrong address).
func arm64ExclMem(mnem string, op *ast.Operand) (int, error) {
rn, off := arm64MemWithFrame(op, arm64FrameInfo{})
if rn < 0 {
return 0, fmt.Errorf("invalid memory operand in %s", mnem)
}
if off != 0 {
return 0, fmt.Errorf("%s: offset %d not supported, exclusive and atomic accesses take a plain (Rn) operand", mnem, off)
}
return rn, nil
}
// arm64PairOf parses a register-pair operand `(R1, R2)`, reporting false
// when the operand is not a pair. The toolchain takes the second register of
// the pair from the operand's Offset (its C_PAIR class,
// cmd/internal/obj/arm64/asm7.go cases 58/59).
func arm64PairOf(op *ast.Operand) (int, int, bool) {
raw := strings.TrimSpace(op.Raw)
if !strings.HasPrefix(raw, "(") || !strings.HasSuffix(raw, ")") {
return -1, -1, false
}
parts := strings.Split(raw[1:len(raw)-1], ",")
if len(parts) != 2 {
return -1, -1, false
}
r1 := arm64RegNum(strings.TrimSpace(parts[0]))
r2 := arm64RegNum(strings.TrimSpace(parts[1]))
if r1 < 0 || r2 < 0 {
return -1, -1, false
}
return r1, r2, true
}
// encodeARM64Excl encodes the exclusive load/store family with the operand
// order the toolchain parses (cmd/internal/obj/arm64/asm7.go cases 58 and 59,
// and its own spellings in arm64enc.s):
//
// STXR Rt, (Rn), Rs store, single register
// STXP (Rt1, Rt2), (Rn), Rs store, register pair
// LDXR (Rn), Rt load, single register
// LDXP (Rn), (Rt1, Rt2) load, register pair
//
// Decoded toolchain evidence: `STXR R1, (R2), R3` assembles to 0xc8037c41,
// whose fields are Rs=3, Rn=2, Rt=1: the FIRST register operand is the data
// register and the LAST the status register.
func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
// LDXR/STXR have different operand forms.
isLoad := strings.HasPrefix(mnem, "LD")
if isLoad {
// LDXR (Rn), Rt → 2 operands: mem, reg
// LDXR (Rn), Rt / LDXP (Rn), (Rt1, Rt2): 2 operands.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rn, _ := arm64MemWithFrame(ops[0], arm64FrameInfo{})
rn, err := arm64ExclMem(mnem, ops[0])
if err != nil {
return nil, err
}
if rt1, rt2, ok := arm64PairOf(ops[1]); ok {
// The single-register opcodes pre-set the unused Rs (bits 20:16)
// and Rt2 (bits 14:10) fields to 31; the pair forms carry a real
// Rt2 and keep Rs at 31.
return a64wordLE(baseOp | 0x1F<<16 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
}
rt := arm64RegNum(operandRegName(ops[1]))
if rn < 0 || rt < 0 {
if rt < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil
}
// STXR Rs, (Rn), Rt → 3 operands: Rs, mem, Rt
// STXR Rt, (Rn), Rs / STXP (Rt1, Rt2), (Rn), Rs: 3 operands.
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rs := arm64RegNum(operandRegName(ops[0]))
rn, _ := arm64MemWithFrame(ops[1], arm64FrameInfo{})
rt := arm64RegNum(operandRegName(ops[2]))
if rs < 0 || rn < 0 || rt < 0 {
rn, err := arm64ExclMem(mnem, ops[1])
if err != nil {
return nil, err
}
rs := arm64RegNum(operandRegName(ops[2]))
if rs < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
if rt1, rt2, ok := arm64PairOf(ops[0]); ok {
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
}
rt := arm64RegNum(operandRegName(ops[0]))
if rt < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
@@ -1272,9 +1444,15 @@ func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rs := arm64RegNum(operandRegName(ops[0]))
rn, _ := arm64MemWithFrame(ops[1], arm64FrameInfo{})
if rs < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
rn, err := arm64ExclMem(mnem, ops[1])
if err != nil {
return nil, err
}
rt := arm64RegNum(operandRegName(ops[2]))
if rs < 0 || rn < 0 || rt < 0 {
if rt < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
@@ -1283,43 +1461,25 @@ func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
// ---- Bitfield/EXTR encoding ----
// encodeARM64Bitfield encodes a bitfield instruction.
// ASR/LSL/LSR/ROR $shamt, Rn, Rd → 3 operands: $imm, Rn, Rd
// BFI/BFXIL/SBFM/UBFM $immr, Rn, $imms, Rd → 4 operands
func encodeARM64Bitfield(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
isShift := mnem == "ASR" || mnem == "ASRW" || mnem == "LSL" || mnem == "LSLW" ||
mnem == "LSR" || mnem == "LSRW" || mnem == "ROR" || mnem == "RORW"
if isShift {
// ASR $shamt, Rn, Rd → SBFM with immr=shamt, imms=31/63
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
shamt := int(immFromOperand(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
// ASR: SBFM with immr=shamt, imms=31(32-bit) or 63(64-bit)
is64 := mnem == "ASR"
imms := 31
if is64 {
imms = 63
}
return a64wordLE(baseOp | uint32(shamt)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
// BFI/BFXIL/SBFM/UBFM: 4 operands ($immr, Rn, $imms, Rd)
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
}
immr := int(immFromOperand(ops[0]))
immr := arm64Imm64(ops[0])
rn := arm64RegNum(operandRegName(ops[1]))
imms := int(immFromOperand(ops[2]))
imms := arm64Imm64(ops[2])
rd := arm64RegNum(operandRegName(ops[3]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
// The toolchain rejects bit numbers at or above the operand width, which
// sf (bit 31 of the base) selects: 64 when set, 32 otherwise.
width := uint32(32) << (baseOp >> 31 & 1)
if immr < 0 || uint32(immr) >= width || imms < 0 || uint32(imms) >= width {
return nil, fmt.Errorf("%s: bit number out of range (immr=%d imms=%d, width=%d)", mnem, immr, imms, width)
}
return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
@@ -1329,13 +1489,19 @@ func encodeARM64Extr(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
}
lsb := int(immFromOperand(ops[0]))
lsb := arm64Imm64(ops[0])
rm := arm64RegNum(operandRegName(ops[1]))
rn := arm64RegNum(operandRegName(ops[2]))
rd := arm64RegNum(operandRegName(ops[3]))
if rm < 0 || rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
// The imms field is 6 bits and must stay below the operand width, which
// sf (bit 31 of the base) selects: 64 when set, 32 otherwise.
width := int64(32) << (baseOp >> 31 & 1)
if lsb < 0 || lsb >= width {
return nil, fmt.Errorf("%s: bit number %d out of range (width=%d)", mnem, lsb, width)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(lsb)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
@@ -1367,7 +1533,7 @@ func AssembleFileARM64(f *ast.File) (*Image, error) {
return nil, err
}
img := &Image{Symbols: map[string]int{}}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
+77 -71
View File
@@ -27,6 +27,8 @@ package asm
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
import "maps"
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
// R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
// runtime's assembly uses. Returns -1 for an unrecognised name.
@@ -96,8 +98,12 @@ func arm64RegNum(name string) int {
return 30
case "R31", "ZR":
return 31
case "SP":
return 31 // SP and ZR share encoding 31; context determines meaning
case "SP", "RSP":
// RSP is the toolchain's spelling for register 31 (it rejects
// R31 in an operand); SP stays for sources that spell it the
// amd64 way. SP and ZR share encoding 31; context determines
// the meaning.
return 31
}
// F0-F31.
if len(name) >= 1 && name[0] == 'F' {
@@ -262,19 +268,15 @@ type a64Format uint8
const (
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
a64FDPIR // data-processing (immediate): ADD/SUB $imm
a64FLogImm // logical (immediate): AND/ORR/EOR $imm
a64FMovWide // move wide: MOVZ, MOVN, MOVK
a64FLSU // load/store (unsigned immediate, scaled)
a64FLSUnscaled // load/store (unscaled immediate)
a64FLSPair // load/store pair
a64FBranch // unconditional branch (B/BL)
a64FBranchCond // conditional branch (B.cond)
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
a64FADR // ADR/ADRP
a64FEXTR // EXTR
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
a64FSystem // system: NOP, BRK, etc.
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
@@ -282,10 +284,9 @@ const (
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
a64FFMovGR // FMOV between GP and FP registers
a64FCRC32 // CRC32
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
)
@@ -332,30 +333,14 @@ func init() {
"ANDSW": 0<<31 | 3<<29 | 0x0a<<24,
"BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21,
"BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21,
// Shift
"LSL": 1<<31 | 0<<29 | 0x0a<<24, // alias of UBFM
"LSLW": 0<<31 | 0<<29 | 0x0a<<24,
"LSR": 1<<31 | 0<<29 | 0x0a<<24,
"LSRW": 0<<31 | 0<<29 | 0x0a<<24,
"ASR": 1<<31 | 0<<29 | 0x0a<<24,
"ASRW": 0<<31 | 0<<29 | 0x0a<<24,
"ROR": 1<<31 | 0<<29 | 0x0a<<24,
"RORW": 0<<31 | 0<<29 | 0x0a<<24,
// Multiply
"MADD": 1<<31 | 0<<29 | 0x1b<<24 | 0<<21,
"MADDW": 0<<31 | 0<<29 | 0x1b<<24 | 0<<21,
"MSUB": 1<<31 | 0<<29 | 0x1b<<24 | 1<<21,
"MSUBW": 0<<31 | 0<<29 | 0x1b<<24 | 1<<21,
// Divide
"SDIV": 1<<31 | 0<<29 | 0x0d<<24,
"SDIVW": 0<<31 | 0<<29 | 0x0d<<24,
"UDIV": 1<<31 | 0<<29 | 0x0d<<24 | 1<<10,
"UDIVW": 0<<31 | 0<<29 | 0x0d<<24 | 1<<10,
// CRC
"CRC32B": 0<<31 | 0<<29 | 0x1b<<24 | 4<<10,
"CRC32H": 0<<31 | 0<<29 | 0x1b<<24 | 5<<10,
"CRC32W": 0<<31 | 0<<29 | 0x1b<<24 | 6<<10,
"CRC32X": 1<<31 | 0<<29 | 0x1b<<24 | 7<<10,
// Divide (data-processing 2 source): the opcode occupies bits 15:10
// of the 0xd6<<21 fixed field, UDIV=0b0010 and SDIV=0b0011 (ARM ARM
// "Data-processing (2 source)"; the toolchain spells them OPDP2(2)
// and OPDP2(3)). sf=1 selects the X forms.
"SDIV": 1<<31 | 0xd6<<21 | 3<<10,
"SDIVW": 0<<31 | 0xd6<<21 | 3<<10,
"UDIV": 1<<31 | 0xd6<<21 | 2<<10,
"UDIVW": 0<<31 | 0xd6<<21 | 2<<10,
// Conditional select
"CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10,
"CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10,
@@ -385,14 +370,37 @@ func init() {
a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]}
a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]}
// ---- data-processing (immediate) ----
// ADD/SUB $imm, Rn, Rd
a64InstrTable["ADDImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 0<<29 | 0x11<<24}
a64InstrTable["ADDWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 0<<30 | 0<<29 | 0x11<<24}
a64InstrTable["SUBImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 0<<29 | 0x11<<24}
a64InstrTable["SUBWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 1<<30 | 0<<29 | 0x11<<24}
a64InstrTable["ADDSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 1<<29 | 0x11<<24}
a64InstrTable["SUBSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 1<<29 | 0x11<<24}
// ---- shifts ----
// The mnemonic serves both forms: with an immediate the aliases of the
// data-processing (immediate) group apply (ARM ARM "Shifts"), with a
// register the data-processing (2 source) LSLV/LSRV/ASRV/RORV. The op
// field carries the immediate-alias base; encodeARM64Shift derives both
// it and the two-source opcode. Identities, W = 64 (X) or 32 (W):
//
// LSL $sh, Rn, Rd = UBFM Rd, Rn, #(-sh) mod W, #(W-1)-sh
// LSR $sh, Rn, Rd = UBFM Rd, Rn, #sh, #(W-1)
// ASR $sh, Rn, Rd = SBFM Rd, Rn, #sh, #(W-1)
// ROR $sh, Rn, Rd = EXTR Rd, Rn, Rn, #sh
shifts := map[string]a64Enc{
"LSL": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X
"LSLW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W
"LSR": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X
"LSRW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W
"ASR": {format: a64FShift, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}, // SBFM X
"ASRW": {format: a64FShift, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}, // SBFM W
"ROR": {format: a64FShift, op: 1<<31 | 0x27<<23 | 1<<22}, // EXTR X
"RORW": {format: a64FShift, op: 0<<31 | 0x27<<23 | 0<<22}, // EXTR W
}
maps.Copy(a64InstrTable, shifts)
// ---- multiply accumulate ----
// MADD/MSUB Rm, Ra, Rn, Rd: sf 00 11011 o0(15) Rm Ra Rn Rd. The
// toolchain's optab has no shorter row, so all four operands are
// mandatory, and Ra is the SECOND operand.
a64InstrTable["MADD"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24}
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
// ---- move wide ----
// MOVZ/MOVN/MOVK
@@ -407,22 +415,9 @@ func init() {
a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0}
a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1}
// ---- load/store (unsigned immediate) ----
a64InstrTable["MOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<22} // LDR 64-bit
a64InstrTable["MOVWU"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<22} // LDR 32-bit unsigned
a64InstrTable["MOVHU"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 1<<22} // LDRH unsigned
a64InstrTable["MOVBU"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 1<<22} // LDRB unsigned
a64InstrTable["MOVW"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 2<<22} // LDRSW (signed 32→64)
a64InstrTable["MOVH"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 2<<22} // LDRSH (signed half)
a64InstrTable["MOVB"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 2<<22} // LDRSB (signed byte)
a64InstrTable["FMOVS"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 32-bit FP
a64InstrTable["FMOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 64-bit FP
// Store opcodes (load ^ (1<<22)):
// STR 64-bit: size=3, V=0, opc=00 → 3<<30 | 7<<27 | 0<<22
// STR 32-bit: size=2, V=0, opc=00 → 2<<30 | 7<<27 | 0<<22
// STRH: size=1, V=0, opc=00 → 1<<30 | 7<<27 | 0<<22
// STRB: size=0, V=0, opc=00 → 0<<30 | 7<<27 | 0<<22
// Load/store mnemonics never enter this table: the MOV pseudo-instruction
// dispatch handles them through a64LoadTable, which also carries the store
// opcode (integer and FP stores both use opc=00, differing only in V).
// ---- branches ----
a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26}
@@ -445,10 +440,8 @@ func init() {
a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21}
// ---- system ----
a64InstrTable["NOP"] = a64Enc{format: a64FSystem, op: a64NOP}
a64InstrTable["NOOP"] = a64Enc{format: a64FSystem, op: a64NOP}
a64InstrTable["BRK"] = a64Enc{format: a64FSystem, op: 0xd4200000}
a64InstrTable["UNDEF"] = a64Enc{format: a64FSystem, op: a64BRK(0)}
// NOP/NOOP/UNDEF are spelled out in encodeARM64Instr's pseudo switch,
// so they carry no table entry; a64NOP and a64BRK are the encoders.
// ---- EXTR ----
a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22}
@@ -549,8 +542,8 @@ func init() {
a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op}
}
// ---- FMOV between GP and FP registers ----
a64InstrTable["FMOVGR"] = a64Enc{format: a64FFMovGR, op: 0x1e260000} // placeholder, actual encoding depends on direction
// FMOV between GP and FP registers needs no table entry: the MOV
// pseudo-instruction dispatches it by operand class (encodeARM64RegMove).
// ---- conditional select: CSEL, CSINC, CSINV, CSNEG ----
csel := map[string]uint32{
@@ -586,6 +579,10 @@ func init() {
}
// ---- exclusive load/store ----
// Single-register forms pre-set the unused Rs and Rt2 fields to 31 (the
// 0x7c00/0x1f0000 halves of the constants below); the register-pair
// forms carry a real Rt2 in bits 14:10, so their opcodes pre-set
// neither field.
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
@@ -594,6 +591,12 @@ func init() {
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
// Pair loads, LDSTX(sz, 0, l=1, o1=1, o0) in asm7.go: LDXP/ LDXPW have
// o0=0, LDAXP/LDAXPW o0=1 (bit 15). Rs (bits 20:16) stays 31.
a64InstrTable["LDXP"] = a64Enc{format: a64FExcl, op: 0xc8600000}
a64InstrTable["LDXPW"] = a64Enc{format: a64FExcl, op: 0x88600000}
a64InstrTable["LDAXP"] = a64Enc{format: a64FExcl, op: 0xc8608000}
a64InstrTable["LDAXPW"] = a64Enc{format: a64FExcl, op: 0x88608000}
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
@@ -602,6 +605,12 @@ func init() {
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
// Pair stores, LDSTX(sz, 0, l=0, o1=1, o0): STXP/STXPW have o0=0,
// STLXP/STLXPW o0=1 (bit 15). Both Rs and Rt2 are real fields.
a64InstrTable["STXP"] = a64Enc{format: a64FExcl, op: 0xc8200000}
a64InstrTable["STXPW"] = a64Enc{format: a64FExcl, op: 0x88200000}
a64InstrTable["STLXP"] = a64Enc{format: a64FExcl, op: 0xc8208000}
a64InstrTable["STLXPW"] = a64Enc{format: a64FExcl, op: 0x88208000}
// ---- LSE atomics ----
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
@@ -642,14 +651,11 @@ var a64LoadTable = map[string]a64LSType{
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
}
// a64StoreOpc returns the store opc for a given load type.
// For integer: store opc = 00 (the load opc bits cleared).
// For FP: store opc = 00 (same pattern).
// a64StoreOpc returns the store opc for a given load type: integer and FP
// stores both encode opc=00 (the load's signedness bit sits in opc[1], which
// the store form clears; FP registers are selected by V, not opc).
func a64StoreOpc(t a64LSType) int {
if t.V == 1 {
return 0 // FP store
}
return 0 // integer store
return 0
}
// arm64RegClass discriminates integer (R), floating-point (F) registers for
+377
View File
@@ -611,3 +611,380 @@ TEXT ·f(SB), NOSPLIT, $0-0
}
}
}
// arm64Words assembles a single NOSPLIT leaf body and returns its words.
func arm64Words(t *testing.T, body string) []uint32 {
t.Helper()
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
return leWords(img.Code)
}
// TestArm64ShiftEncodings pins the shift words against `go tool asm -S`
// output (Go 1.27, arm64): immediate forms alias SBFM/UBFM with ROR as EXTR,
// register forms are the two-source LSLV/LSRV/ASRV/RORV.
func TestArm64ShiftEncodings(t *testing.T) {
got := arm64Words(t, "\tLSL $4, R0, R1\n\tLSR $8, R0, R2\n\tASR $4, R0, R3\n\tROR $12, R0, R4\n"+
"\tLSLW $4, R0, R5\n\tLSRW $8, R0, R6\n\tASRW $4, R0, R7\n\tRORW $12, R0, R8\n")
want := []uint32{
0xd37cec01, // LSL $4 = UBFM X1, X0, #60, #59
0xd348fc02, // LSR $8 = UBFM X2, X0, #8, #63
0x9344fc03, // ASR $4 = SBFM X3, X0, #4, #63
0x93c03004, // ROR $12 = EXTR X4, X0, X0, #12
0x531c6c05, // LSLW $4 = UBFM W5, W0, #28, #27
0x53087c06, // LSRW $8 = UBFM W6, W0, #8, #31
0x13047c07, // ASRW $4 = SBFM W7, W0, #4, #31
0x13803008, // RORW $12 = EXTR W8, W0, W0, #12
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("imm shift word %d = %08x, want %08x", i, got[i], want[i])
}
}
got = arm64Words(t, "\tLSL R9, R0, R10\n\tLSR R9, R0, R11\n\tASR R9, R0, R12\n\tROR R9, R0, R13\n"+
"\tLSLW R9, R0, R14\n\tLSRW R9, R0, R15\n\tASRW R9, R0, R16\n\tRORW R9, R0, R17\n")
want = []uint32{
0x9ac9200a, // LSLV X10, X0, X9
0x9ac9240b, // LSRV X11, X0, X9
0x9ac9280c, // ASRV X12, X0, X9
0x9ac92c0d, // RORV X13, X0, X9
0x1ac9200e, // LSLV W14, W0, W9
0x1ac9240f, // LSRV W15, W0, W9
0x1ac92810, // ASRV W16, W0, W9
0x1ac92c11, // RORV W17, W0, W9
0xd65f03c0, // RET
}
for i := range want {
if got[i] != want[i] {
t.Errorf("reg shift word %d = %08x, want %08x", i, got[i], want[i])
}
}
// Two-operand spellings fold to Rn = Rd.
got = arm64Words(t, "\tLSL $4, R1\n\tLSR R9, R1\n\tASR $4, R1\n\tROR R9, R1\n\tLSLW $4, R1\n\tRORW R9, R1\n")
want = []uint32{
0xd37cec21, // LSL $4, R1 = UBFM X1, X1, #60, #59
0x9ac92421, // LSRV X1, X1, X9
0x9344fc21, // ASR $4, R1 = SBFM X1, X1, #4, #63
0x9ac92c21, // RORV X1, X1, X9
0x531c6c21, // LSLW $4, R1 = UBFM W1, W1, #28, #27
0x1ac92c21, // RORV W1, W1, W9
0xd65f03c0, // RET
}
for i := range want {
if got[i] != want[i] {
t.Errorf("2op shift word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64ShiftRangeErrors: the toolchain reports "illegal bit number" for
// shift amounts at or above the operand width.
func TestArm64ShiftRangeErrors(t *testing.T) {
for _, src := range []string{
"\tLSL $64, R0, R1\n",
"\tLSRW $32, R0, R1\n",
"\tRORW $32, R0, R1\n",
"\tASR $-1, R0, R1\n",
} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+src+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", src)
}
}
}
// TestArm64DivEncodings pins SDIV/UDIV in both widths: the 2-source opcode
// field (bits 15:10 of the 0xd6<<21 fixed field) is UDIV=0b0010, SDIV=0b0011.
func TestArm64DivEncodings(t *testing.T) {
got := arm64Words(t, "\tSDIV R1, R2, R3\n\tUDIV R1, R2, R3\n\tSDIVW R1, R2, R3\n\tUDIVW R1, R2, R3\n")
want := []uint32{
0x9ac10c43, // SDIV X3, X2, X1
0x9ac10843, // UDIV X3, X2, X1
0x1ac10c43, // SDIV W3, W2, W1
0x1ac10843, // UDIV W3, W2, W1
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("div word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64MAddSub pins the four-operand MADD/MSUB words (Rm, Ra, Rn, Rd,
// with Ra in bits 14:10) and rejects the shorter spellings the toolchain
// also rejects.
func TestArm64MAddSub(t *testing.T) {
got := arm64Words(t, "\tMADD R1, R2, R3, R4\n\tMSUB R1, R2, R3, R4\n\tMADDW R1, R2, R3, R5\n\tMSUBW R1, R2, R3, R5\n")
want := []uint32{
0x9b010864, // MADD X4, X3, X1, X2 (Rm=1, Ra=2, Rn=3)
0x9b018864, // MSUB X4, X3, X1, X2
0x1b010865, // MADD W5, W3, W1, W2
0x1b018865, // MSUB W5, W3, W1, W2
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("madd word %d = %08x, want %08x", i, got[i], want[i])
}
}
// The accumulate operand is mandatory: 2- and 3-operand forms error
// rather than silently reading R0 or ZR as the accumulator.
for _, body := range []string{
"\tMADD R1, R2\n",
"\tMADD R1, R2, R3\n",
"\tMSUBW R1, R2, R3\n",
} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", body)
}
}
}
// TestArm64MovImmWidth pins the immediate classifications whose size pass
// once disagreed with the encoder: negative and 0xFFFFFFFF W values go
// through MOVN after 32-bit truncation, and 3- to 4-chunk constants expand
// to one word per non-zero chunk.
func TestArm64MovImmWidth(t *testing.T) {
got := arm64Words(t, "\tMOVW $-1, R0\n\tMOVW $0xFFFFFFFF, R3\n")
want := []uint32{
0x12800000, // MOVN W0, #0
0x12800003, // MOVN W3, #0
0xd65f03c0, // RET
}
for i := range want {
if got[i] != want[i] {
t.Errorf("movw word %d = %08x, want %08x", i, got[i], want[i])
}
}
for _, tt := range []struct {
body string
words int
}{
{"\tMOVD $0x0001000200030000, R2\n", 3}, // three chunks
{"\tMOVD $0x0001000200030004, R1\n", 4}, // four chunks
{"\tMOVW $-1, R0\n", 1}, // MOVN after truncation
} {
if got := arm64Words(t, tt.body); len(got) != tt.words+1 {
t.Errorf("%s: %d words, want %d (including RET)", tt.body, len(got), tt.words+1)
}
}
}
// TestArm64ExclOffsetErrors: exclusive and atomic encodings carry no
// immediate field, so a non-zero offset is rejected the way the toolchain
// reports "illegal combination" for it, never silently dropped.
func TestArm64ExclOffsetErrors(t *testing.T) {
for _, body := range []string{
"\tLDXR 8(R1), R2\n",
"\tLDAXR 8(R1), R2\n",
"\tSTXR R3, 8(R1), R4\n",
"\tSTLXR R3, 8(R1), R4\n",
"\tCASD R3, 8(R1), R4\n",
"\tLDADDD R3, 8(R1), R4\n",
} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", body)
}
}
}
// TestArm64ExclNoOffset pins the plain (Rn) forms, byte-for-byte against
// go tool asm. The toolchain parses the FIRST register of a store as the
// data register and the LAST as the status register (asm7.go case 59), and
// the pair forms as (Rt1, Rt2) (case 58/59):
//
// STXR R3, (R1), R4 → c8047c23 (Rt=3, Rn=1, Rs=4)
// STXP (R3, R4), (R1), R5 → c8251023 (Rt=3, Rt2=4, Rn=1, Rs=5)
// LDXP (R1), (R3, R4) → c87f1023 (Rn=1, Rt=3, Rt2=4)
func TestArm64ExclNoOffset(t *testing.T) {
got := arm64Words(t, "\tLDXR (R1), R2\n\tSTXR R3, (R1), R4\n"+
"\tSTXP (R3, R4), (R1), R5\n\tSTXPW (R3, R4), (R1), R5\n"+
"\tLDXP (R1), (R3, R4)\n\tLDXPW (R1), (R3, R4)\n"+
"\tSTXR R3, (RSP), R4\n\tLDXR (RSP), R2\n")
want := []uint32{
0xc85f7c22, // LDXR X2, [X1]
0xc8047c23, // STXR W3, [X1], W4 with Rt = R3, Rs = R4
0xc8251023, // STXP (R3, R4), [X1], R5
0x88251023, // STXPW (R3, R4), [X1], R5
0xc87f1023, // LDXP [X1], (R3, R4)
0x887f1023, // LDXPW [X1], (R3, R4)
0xc8047fe3, // STXR R3, [SP], R4
0xc85f7fe2, // LDXR [SP], R2
0xd65f03c0, // RET
}
for i := range want {
if got[i] != want[i] {
t.Errorf("excl word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64AddSubImmRange: immediates that cannot ride the imm12 field are
// rejected instead of wrapping through int32.
func TestArm64AddSubImmRange(t *testing.T) {
for _, body := range []string{
"\tADD $0x100000000, R0, R1\n",
"\tSUB $-0x100000000, R0, R1\n",
"\tCMP $0x100000000, R0\n",
} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", body)
}
}
}
// TestArm64LargeRegisterOffset pins the large-offset path for a register
// base: the ADD offsets from the operand's own base, not from SP, matching
// the toolchain's `ADD $(256<<12), R2, R27; MOVD (R27), R3`.
func TestArm64LargeRegisterOffset(t *testing.T) {
got := arm64Words(t, "\tMOVD 0x100000(R2), R3\n\tMOVD R3, 0x100000(R2)\n")
want := []uint32{
0x9144005b, // ADD $(256<<12), R2, R27
0xf9400363, // MOVD (R27), R3
0x9144005b, // ADD $(256<<12), R2, R27
0xf9000363, // MOVD R3, (R27)
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("large offset word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64LargeFrameSpadj checks the stack-adjustment boundaries of a frame
// whose autosize must be materialised into REGTMP: $5000 rounds the autosize
// to 5024, so the prologue is [MOVD $5024, R27][SUB R27, RSP, R20][STP][ADD
// R20, SP][SUB $8] and SP moves only at its fourth word, while the RET's
// epilogue is [LDP][MOVD $5024, R27][ADD R27, RSP, RSP] before the final
// RET. These PCs feed the DWARF CFA rules and the goobj stack maps.
func TestArm64LargeFrameSpadj(t *testing.T) {
f, errs := parser.Parse("frame_arm64.s", "#include \"textflag.h\"\n\nTEXT ·framed(SB), $5000-0\n\tCALL ·other(SB)\n\tRET\n\nTEXT ·other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
fn := img.Funcs[0]
// autosize 5024: class-2 guard of 6 words (24 bytes), a 5-word prologue
// whose ADD R20, SP sits at byte 8 inside it, a one-instruction body,
// then a 3-word epilogue before the final RET.
wantSpadj := []SpadjStep{{PC: 24 + 12, Value: 5024}, {PC: 24 + 20 + 4 + 12, Value: 0}}
if len(fn.Spadj) != len(wantSpadj) {
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
}
for i := range wantSpadj {
if fn.Spadj[i] != wantSpadj[i] {
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
}
}
// The words those PCs point between: the prologue's ADD R20, SP at byte
// 36, and the epilogue's materialised ADD R27, RSP, RSP right before the
// final RET at byte 60.
words := leWords(img.Code[fn.Offset : fn.Offset+fn.Size])
if got := words[(24+12)/4]; got != 0x9100029f {
t.Errorf("prologue word at byte 36 = %08x, want 9100029f (ADD R20, SP)", got)
}
if got := words[(24+20+4+8)/4]; got != 0x8b3b63ff {
t.Errorf("epilogue word at byte 56 = %08x, want 8b3b63ff (ADD R27, RSP, RSP)", got)
}
if got := words[(24+20+4+12)/4]; got != 0xd65f03c0 {
t.Errorf("final RET word at byte 60 = %08x, want d65f03c0", got)
}
}
// TestArm64SplitFrameSpadj pins the addcon2 band, where neither imm12 form
// nor a single MOVZ carries the autosize and the toolchain splits the
// prologue SUB into two imm12 instructions (asm7.go case 48) while the
// non-leaf RET still materialises the value into REGTMP (obj7.go ARET,
// issue 73259). $65664 rounds the autosize to 65680 = 144 + 16<<12:
//
// [SUB $144, RSP, R20][SUB $(16<<12), R20, R20][STP][MOVD R20, SP][SUB $8]
// [CALL]
// [LDP][MOVD $144, R27][MOVK $(1<<16), R27][ADD R27, RSP, RSP][RET]
//
// SP moves at the fourth word (byte 12) and returns to zero at the final
// RET (byte 40); the words are go tool asm's own for the same source.
func TestArm64SplitFrameSpadj(t *testing.T) {
f, errs := parser.Parse("frame_arm64.s", "#include \"textflag.h\"\n\nTEXT ·framed(SB), NOSPLIT, $65664-0\n\tCALL ·other(SB)\n\tRET\n\nTEXT ·other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
fn := img.Funcs[0]
wantSpadj := []SpadjStep{{PC: 12, Value: 65680}, {PC: 40, Value: 0}}
if len(fn.Spadj) != len(wantSpadj) {
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
}
for i := range wantSpadj {
if fn.Spadj[i] != wantSpadj[i] {
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
}
}
want := []uint32{
0xd10243f4, // SUB $144, RSP, R20
0xd1404294, // SUB $(16<<12), R20, R20
0xa93ffa9d, // STP (R29, R30), -8(R20)
0x9100029f, // MOVD R20, RSP
0xd10023fd, // SUB $8, RSP, R29
0x94000000, // CALL (relocation masked at link time)
0xa97ffbfd, // LDP -8(RSP), (R29, R30)
0xd280121b, // MOVD $144, R27
0xf2a0003b, // MOVK $(1<<16), R27
0x8b3b63ff, // ADD R27, RSP, RSP
0xd65f03c0, // RET
}
words := leWords(img.Code[fn.Offset : fn.Offset+fn.Size])
if len(words) != len(want) {
t.Fatalf("framed = %d words, want %d", len(words), len(want))
}
for i, w := range want {
if words[i] != w {
t.Errorf("word %d = %08x, want %08x", i, words[i], w)
}
}
}
+85 -20
View File
@@ -187,10 +187,32 @@ func arm64Prologue(fi arm64FrameInfo) []byte {
return a64WordsLE(ws...)
}
// arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value
// fits the imm12 field (plain, or shifted left by 12 when it is a multiple
// of 4096); otherwise the toolchain materialises it into REGTMP (R27) and
// subtracts the register in the extended-register form.
// arm64SplitImm12 reports whether the toolchain decomposes ADD/SUB $imm into
// two imm12 instructions instead of materialising it into REGTMP
// (asm7.go case 48, the C_ADDCON2 class): the value must fit 24 bits
// unsigned and be neither encodable as one imm12 (checked by the callers
// first), nor loadable into a register in a single MOVZ/MOVN word, nor a
// logical immediate, because conclass tests all three before C_ADDCON2.
func arm64SplitImm12(imm uint32) bool {
if imm > 0xFFFFFF {
return false
}
if _, _, _, ok := arm64Bitmask(uint64(imm), 1); ok {
return false
}
return arm64Movcon(int64(imm)) < 0 && arm64Movcon(^int64(imm)) < 0
}
// arm64SubImmWords emits SUB $imm, SP, Rd with the toolchain's ladder for an
// ADD/SUB constant (asm7.go conclass and cases 2, 48, 62 and 13): the
// immediate form when the value fits imm12 (plain, or shifted left by 12
// when it is a multiple of 4096); a value with a single 16-bit chunk, a
// logical immediate, or one wider than 24 bits is materialised into REGTMP
// (R27) and subtracted in the extended-register form; everything else up to
// 0xFFFFFF is split into two imm12 instructions:
//
// SUB $(imm&0xfff), SP, Rd
// SUB $((imm&0xfff000)>>12)<<12, Rd, Rd
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 0xFFF {
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
@@ -198,15 +220,21 @@ func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 4095<<12 && imm&0xFFF == 0 {
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
}
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
if err != nil {
mov = nil
if !arm64SplitImm12(imm) {
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
}
return []uint32{
a64AddSub(1, 1, 0, 0, imm&0xFFF, 31, rd),
a64AddSub(1, 1, 0, 1, (imm&0xFFF000)>>12, rd, rd),
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
}
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12
// and REGTMP fallback ladder.
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12,
// split and REGTMP ladder as arm64SubImmWords.
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 0xFFF {
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
@@ -214,11 +242,35 @@ func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 4095<<12 && imm&0xFFF == 0 {
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
}
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
if !arm64SplitImm12(imm) {
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
}
return []uint32{
a64AddSub(1, 0, 0, 0, imm&0xFFF, 31, rd),
a64AddSub(1, 0, 0, 1, (imm&0xFFF000)>>12, rd, rd),
}
}
// arm64RetAddWords emits the frame deallocation of a non-leaf RET with a
// large frame. The toolchain adds the frame back with a single instruction:
// a plain imm12 ADD when autosize fits 12 bits, otherwise the value is
// materialised into REGTMP and added as a register, so the epilogue never
// leaves a partially deallocated frame (obj7.go ARET, issue 73259). The
// shifted-imm12 and split-imm12 forms are therefore never used here, unlike
// the leaf epilogue's plain ADD instructions.
func arm64RetAddWords(autosize uint32) []uint32 {
if autosize < 1<<12 {
return []uint32{a64AddSub(1, 0, 0, 0, autosize, 31, 31)}
}
mov, err := encodeARM64LoadImm(27, int64(autosize), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, 31))
}
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
@@ -237,11 +289,11 @@ func arm64Return(fi arm64FrameInfo) []byte {
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
)
} else {
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
// Large frame: LDP -8(SP), (FP, LR), then deallocate.
ws = append(ws,
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
)
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
ws = append(ws, arm64RetAddWords(uint32(fi.autosize))...)
}
}
// RET: BR LR (0xd65f03c0)
@@ -258,22 +310,32 @@ func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
if fi.autosize <= 0xf0 {
return 4 // MOVD.W instruction decrements SP
}
return 8 // SUB + STP + MOVD (3 instructions, SP updated at the MOVD)
// Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose
// position depends on how many words the SUB itself took (immediate,
// shifted immediate, the two-word imm12 split, or a materialised REGTMP
// sequence).
return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1)
}
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final RET instruction.
// (but not including) the final RET instruction. The lengths are read from
// the same word-emitting helpers the epilogue uses rather than assumed: the
// leaf path shares the prologue's immediate ladder, and a materialised
// autosize costs its MOV words plus the ADD itself.
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
if fi.autosize == 0 {
return 0
}
if fi.leaf {
return 8 // ADD + ADD
return 4 * (len(arm64AddImmWords(uint32(fi.autosize-8), 29)) +
len(arm64AddImmWords(uint32(fi.autosize), 31)))
}
if fi.autosize <= 0xf0 {
return 8 // LDR + LDR.P
}
return 8 // LDP + ADD
// LDP + the deallocation emitted by arm64RetAddWords, so the length
// tracks whatever the MOVD ladder needs.
return 4 + 4*len(arm64RetAddWords(uint32(fi.autosize)))
}
// arm64ResolvePseudo translates a pseudo-register memory reference into a
@@ -382,9 +444,12 @@ func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte {
ws = append(ws, wordsOf(mov)...)
ml := len(mov) / 4
ws = append(ws, arm64DPExtWords(arm64OpSubs, 27, 31, 17)) // SUBS R17, RSP, R27
ws = append(ws, br(8+ml, a64CondLO))
// The branches sit at fixed byte offsets in the guard prefix: after
// the LDR (4), the ml MOV words (4*ml) and the SUBS (4) for B.LO,
// then a further B.LO word and the CMP for B.LS.
ws = append(ws, br(8+4*ml, a64CondLO))
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
ws = append(ws, br(8+ml+8, a64CondLS))
ws = append(ws, br(16+4*ml, a64CondLS))
}
return a64WordsLE(ws...)
}
+28 -28
View File
@@ -114,6 +114,11 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] {
continue
}
// A zero-operand jump parses; its arity is reported during
// emission (encodeJump), so the layout must not index Operands.
if len(s.Operands) != 1 {
continue
}
name, ok := labelName(s.Operands[0])
if !ok {
continue // reported during emission
@@ -137,28 +142,22 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
}
if fi.splitClass == 2 && !guardJBlong {
// The underflow JB sits before the CMPQ; its displacement spans
// the rest of the guard plus the prologue and the body.
jbLen := 2
if guardJBlong {
jbLen = 6
}
rest := fi.guardLen(guardJBlong, guardJBElong) - (9 + 3 + 7 + jbLen)
// the rest of the guard plus the prologue and the body. The JB
// is still the short form this branch tests (relaxing it is this
// branch's job), so guardLen is taken with a short JB and the
// subtraction drops the prefix and the JB's own 2 bytes.
rest := fi.guardLen(false, guardJBElong) - (9 + 3 + 7 + 2)
if !fits8(int64(rest + len(fi.prologue) + bodyLen)) {
guardJBlong = true
changed = true
}
}
// The morestack JMP returns to the function start, so its
// displacement is the negated distance from its own end.
if !moreJMPlong {
jmpLen := 2
if moreJMPlong {
jmpLen = 5
}
if !fits8(-int64(guard + len(fi.prologue) + bodyLen + 5 + jmpLen)) {
moreJMPlong = true
changed = true
}
// displacement is the negated distance from its own end; while it is
// still short, its own length is 2 bytes.
if !moreJMPlong && !fits8(-int64(guard+len(fi.prologue)+bodyLen+5+2)) {
moreJMPlong = true
changed = true
}
if !changed {
break
@@ -181,7 +180,15 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
var out []byte
var patches []sbPatch
if fi.needSplit {
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+2)+len(fi.prologue)+bodyLen))
// The JBE ends the guard, so its displacement is the prologue plus
// the body; the underflow JB additionally spans the trailing CMPQ and
// JBE, whose combined length is guardLen minus the prefix and the
// JB's own length (2 short, 6 long).
jbLen := 2
if guardJBlong {
jbLen = 6
}
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+jbLen)+len(fi.prologue)+bodyLen))
out = append(out, guard...)
patches = append(patches, tlsPatch)
}
@@ -344,7 +351,10 @@ func computeFrame(t *ast.Text) frameInfo {
// pass one extra slot, and the virtual SP is the hardware SP.
fi.size = 8
fi.useFP = true
fi.fpAdjust = int64(fi.size) + 16 // return address + saved BP + args base
// The push is the frame: the saved BP sits at SP+0 and the
// return address at SP+8, so arguments begin at SP+16. Unlike
// a SUBQ frame, the 8-byte size must not be added again.
fi.fpAdjust = 16
fi.spAdjust = 0
fi.prologue = []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
fi.epilogue = []byte{0x5D} // POPQ BP
@@ -426,16 +436,6 @@ func (fi frameInfo) guardLen(jbLong, jbeLong bool) int {
}
}
// moreLen returns the byte length of the trailing morestack block: the CALL
// (always rel32) plus the JMP back to the function start.
func moreLen(jmpLong bool) int {
jmp := 2
if jmpLong {
jmp = 5
}
return 5 + jmp
}
// buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the
// already-computed displacements of the conditional branches that jump to the
// morestack block (unused in classes without them). The TLS load carries a
+64
View File
@@ -160,6 +160,57 @@ TEXT ·loadarg(SB), NOSPLIT, $0-24
}
}
// TestAssembleFramelessCall verifies the forced base-pointer frame a $0-frame
// function containing a CALL receives: the PUSHQ BP prologue with no stack
// adjustment and the x+N(FP) → (N+16)(SP) translation, against the bytes the
// Go assembler produces. The push is the frame, so the offset must not count
// it twice.
func TestAssembleFramelessCall(t *testing.T) {
f, errs := parser.Parse("frameless_call_amd64.s", `
#include "textflag.h"
TEXT ·withcall(SB), NOSPLIT, $0-16
MOVQ x+0(FP), AX
CALL ·other(SB)
MOVQ AX, ret+8(FP)
RET
TEXT ·other(SB), NOSPLIT, $0-0
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
code := append([]byte(nil), img.Code[img.Funcs[0].Offset:img.Funcs[0].Offset+img.Funcs[0].Size]...)
for _, r := range img.Funcs[0].Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
// From `go tool objdump` of the Go-assembled function:
// PUSHQ BP 55
// MOVQ SP, BP 4889e5
// MOVQ 0x10(SP), AX 488b442410
// CALL other e800000000
// MOVQ AX, 0x18(SP) 4889442418
// POPQ BP 5d
// RET c3
want := []byte{
0x55,
0x48, 0x89, 0xe5,
0x48, 0x8b, 0x44, 0x24, 0x10,
0xe8, 0x00, 0x00, 0x00, 0x00,
0x48, 0x89, 0x44, 0x24, 0x18,
0x5d,
0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("frameless CALL FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
// the bytes the Go assembler produces.
@@ -353,6 +404,19 @@ TEXT ·pf(SB), NOSPLIT, $0
}
}
// TestAssembleBareJump checks that a zero-operand jump (which parses, because
// the parser does not arity-check mnemonics) is rejected with an error rather
// than panicking in the layout loop, which indexes Operands[0] before the
// emission pass gets a chance to diagnose the arity.
func TestAssembleBareJump(t *testing.T) {
for _, mnem := range []string{"JE", "JMP", "JLT", "CALL"} {
fn := firstText(t, "TEXT ·bare(SB), $16-0\n\t"+mnem+"\n")
if _, _, err := Assemble(fn); err == nil {
t.Errorf("%s with no operand: expected an error, got none", mnem)
}
}
}
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
// larger. The intermediate 129..255 range used to encode an ADD with a
+49 -7
View File
@@ -42,8 +42,10 @@ const (
sttSection = 3
stInfoShift = 4
rX8664PC32 = 2
rX8664TPOFF32 = 20
rX8664PC32 = 2
// R_X86_64_TPOFF32 (debug/elf): the local-exec TLS offset the stack
// guard loads from FS. 20 is R_X86_64_TLSLD, a different relocation.
rX8664TPOFF32 = 23
)
// elfSym is one symbol-table entry in construction.
@@ -219,7 +221,7 @@ func (img *Image) ELFObject() ([]byte, error) {
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|rX8664PC32)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
@@ -228,15 +230,32 @@ func (img *Image) ELFObject() ([]byte, error) {
shstrOff := len(out)
out = append(out, stSections.bytes()...)
// DWARF debug sections (no relocations, the linker resolves DWARF fixups).
// DWARF debug sections; the address placeholders they leave are carried
// as .rela.debug_info/.rela.debug_line entries the system linker applies.
dwAlign := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiAMD64)
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
if dw != nil {
nSections += 4 // .debug_abbrev, .debug_info, .debug_line, .debug_line_str
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
// .debug_line_str and .debug_frame (the CIE is unconditional, so
// the frame section is always present), plus the relocation
// sections below when they carry entries.
dwarfStart = nSections
nSections += 5
appendDWARFRelas(&out, dw, rX8664Abs64, dwAlign)
if dw.infoRelaCount > 0 {
nSections++
}
if dw.lineRelaCount > 0 {
nSections++
}
if dw.frameRelaCount > 0 {
nSections++
}
}
align(8)
@@ -267,14 +286,37 @@ func (img *Image) ELFObject() ([]byte, error) {
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers.
// DWARF section headers; their indices follow the write order.
if dw != nil {
// secIdx is a running section index: each putSh below emits the
// next header, and the sh_info of a .rela section names the index
// of the section it relocates.
secIdx := dwarfStart
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
secIdx++
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
secInfoIdx := secIdx
secIdx++
if dw.infoRelaCount > 0 {
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
secIdx++
}
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
secLineIdx := secIdx
secIdx++
if dw.lineRelaCount > 0 {
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
secIdx++
}
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
secIdx++
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
secFrameIdx := secIdx
secIdx++
if dw.frameRelaCount > 0 {
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
}
}
}
+161 -74
View File
@@ -12,43 +12,74 @@ import (
// self-contained sections because the system linker only performs fixup
// relocations, not assembly.
// DWARF5 attribute, form and line-table constants (the values the
// toolchain uses, cmd/internal/dwarf/dwarf_defs.go; the DIE streams below
// are written against these forms).
const (
dwAtName = 0x03 // DW_AT_name
dwAtStmtList = 0x10 // DW_AT_stmt_list
dwAtLowPC = 0x11 // DW_AT_low_pc
dwAtHighPC = 0x12 // DW_AT_high_pc
dwAtDeclFile = 0x3a // DW_AT_decl_file
dwAtDeclLine = 0x3b // DW_AT_decl_line
dwAtExternal = 0x3f // DW_AT_external
dwAtFrameBase = 0x40 // DW_AT_frame_base
dwTagSubprog = 0x2e // DW_TAG_subprogram
dwTagCompUnit = 0x11 // DW_TAG_compile_unit
dwFormAddr = 0x01 // DW_FORM_addr
dwFormData8 = 0x07 // DW_FORM_data8
dwFormString = 0x08 // DW_FORM_string
dwFormData1 = 0x0b // DW_FORM_data1
dwFormUdata = 0x0f // DW_FORM_udata
dwFormSecOff = 0x17 // DW_FORM_sec_offset
dwFormExprloc = 0x18 // DW_FORM_exprloc
dwFormLineStrp = 0x1f // DW_FORM_line_strp
dwLnctPath = 0x01 // DW_LNCT_path
dwLnctDirIndex = 0x02 // DW_LNCT_directory_index
)
// dwarfAbbrevTable returns the .debug_abbrev content: a single compilation
// unit with DW_TAG_compile_unit and DW_TAG_subprogram entries.
// unit with DW_TAG_compile_unit and DW_TAG_subprogram entries. The
// attribute/form pairs must match the DIE streams dwarfBuildInfoSection
// writes byte for byte, in the same order, or every consumer's parse of
// .debug_info desynchronises.
func dwarfAbbrevTable() []byte {
var b []byte
// Abbrev 1: DW_TAG_compile_unit
b = append(b, 1) // abbreviation code
b = append(b, 0x11) // DW_TAG_compile_unit
b = append(b, 1) // DW_CHILDREN_yes
b = appendUleb(b, 0x1b) // DW_AT_low_pc
b = appendUleb(b, 0x01) // DW_FORM_addr
b = appendUleb(b, 0x29) // DW_AT_high_pc
b = appendUleb(b, 0x07) // DW_FORM_data8
b = appendUleb(b, 0x10) // DW_AT_stmt_list
b = appendUleb(b, 0x25) // DW_FORM_sec_offset
b = appendUleb(b, 0x01) // DW_AT_name
b = appendUleb(b, 0x08) // DW_FORM_string
b = appendUleb(b, 0) // end of attributes
// Abbrev 1: DW_TAG_compile_unit.
b = append(b, 1) // abbreviation code
b = appendUleb(b, dwTagCompUnit) // DW_TAG_compile_unit
b = append(b, 1) // DW_CHILDREN_yes
b = appendUleb(b, dwAtLowPC) // DW_AT_low_pc
b = appendUleb(b, dwFormAddr) // DW_FORM_addr
b = appendUleb(b, dwAtHighPC) // DW_AT_high_pc
b = appendUleb(b, dwFormData8) // DW_FORM_data8
b = appendUleb(b, dwAtStmtList) // DW_AT_stmt_list
b = appendUleb(b, dwFormSecOff) // DW_FORM_sec_offset (4 bytes here)
b = appendUleb(b, dwAtName) // DW_AT_name
b = appendUleb(b, dwFormString) // DW_FORM_string
b = appendUleb(b, 0) // end of attributes: attr 0
b = appendUleb(b, 0) // ... paired with form 0
// Abbrev 2: DW_TAG_subprogram
b = append(b, 2) // abbreviation code
b = append(b, 0x2e) // DW_TAG_subprogram
b = append(b, 0) // DW_CHILDREN_no
b = appendUleb(b, 0x03) // DW_AT_name
b = appendUleb(b, 0x08) // DW_FORM_string
b = appendUleb(b, 0x11) // DW_AT_low_pc
b = appendUleb(b, 0x01) // DW_FORM_addr
b = appendUleb(b, 0x29) // DW_AT_high_pc
b = appendUleb(b, 0x07) // DW_FORM_data8
b = appendUleb(b, 0x3f) // DW_AT_frame_base
b = appendUleb(b, 0x18) // DW_FORM_exprloc
b = appendUleb(b, 0x3b) // DW_AT_decl_file
b = appendUleb(b, 0x0b) // DW_FORM_data1
b = appendUleb(b, 0x37) // DW_AT_decl_line
b = appendUleb(b, 0x0b) // DW_FORM_data1
b = appendUleb(b, 0x63) // DW_AT_external
b = appendUleb(b, 0x0b) // DW_FORM_flag
b = appendUleb(b, 0) // end of attributes
// Abbrev 2: DW_TAG_subprogram.
b = append(b, 2) // abbreviation code
b = appendUleb(b, dwTagSubprog) // DW_TAG_subprogram
b = append(b, 0) // DW_CHILDREN_no
b = appendUleb(b, dwAtName) // DW_AT_name
b = appendUleb(b, dwFormString) // DW_FORM_string
b = appendUleb(b, dwAtLowPC) // DW_AT_low_pc
b = appendUleb(b, dwFormAddr) // DW_FORM_addr
b = appendUleb(b, dwAtHighPC) // DW_AT_high_pc
b = appendUleb(b, dwFormData8) // DW_FORM_data8
b = appendUleb(b, dwAtFrameBase) // DW_AT_frame_base
b = appendUleb(b, dwFormExprloc) // DW_FORM_exprloc
b = appendUleb(b, dwAtDeclFile) // DW_AT_decl_file
b = appendUleb(b, dwFormData1) // DW_FORM_data1
b = appendUleb(b, dwAtDeclLine) // DW_AT_decl_line
b = appendUleb(b, dwFormData1) // DW_FORM_data1
b = appendUleb(b, dwAtExternal) // DW_AT_external
b = appendUleb(b, 0x0c) // DW_FORM_flag (one byte, 0 or 1)
b = appendUleb(b, 0) // end of attributes: attr 0
b = appendUleb(b, 0) // ... paired with form 0
// End of table.
b = append(b, 0)
@@ -68,6 +99,9 @@ type dwarfSections struct {
infoRelocs []dwarfReloc
// Relocations for .debug_line: (offset, symbol name, addend).
lineRelocs []dwarfReloc
// Relocations for .debug_frame: (offset, symbol name, addend), one per
// FDE initial_location.
frameRelocs []dwarfReloc
}
type dwarfReloc struct {
@@ -76,8 +110,9 @@ type dwarfReloc struct {
addend int64
}
// emitDWARF generates complete DWARF5 sections for the image.
func emitDWARF(img *Image, srcFile string) *dwarfSections {
// emitDWARF generates complete DWARF5 sections for the image. cfi carries
// the architecture's .debug_frame register conventions.
func emitDWARF(img *Image, srcFile string, cfi cfiArch) *dwarfSections {
ds := &dwarfSections{}
ds.debugAbbrev = dwarfAbbrevTable()
@@ -86,19 +121,21 @@ func emitDWARF(img *Image, srcFile string) *dwarfSections {
lineStr.add(srcFile)
ds.debugLineStr = lineStr.bytes()
// Build .debug_line.
ds.debugLine = dwarfBuildLineSection(img, ds)
// Build .debug_line; the file table references the source name through
// its offset in .debug_line_str.
ds.debugLine = dwarfBuildLineSection(img, uint32(lineStr.at(srcFile)), ds)
// Build .debug_info.
ds.debugInfo = dwarfBuildInfoSection(img, srcFile, ds)
// Build .debug_frame.
ds.debugFrame = dwarfBuildFrameSection(img)
ds.debugFrame = dwarfBuildFrameSection(img, cfi, ds)
return ds
}
// dwarfBuildLineSection builds a complete .debug_line section.
func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
// dwarfBuildLineSection builds a complete .debug_line section. srcStrOff is
// the source file name's offset in .debug_line_str.
func dwarfBuildLineSection(img *Image, srcStrOff uint32, ds *dwarfSections) []byte {
var b []byte
le := binary.LittleEndian
@@ -120,15 +157,25 @@ func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
// Standard opcode lengths (opcode 1..opcode_base-1).
b = append(b, 0, 1, 1, 1, 1, 0, 0, 0, 1, 0)
// Directory table (DWARF5 format).
b = append(b, 0) // one directory entry (index 0 = empty)
// File table.
b = appendUleb(b, 1) // file count
// File 1: name index into .debug_line_str, dir index, time, size.
b = appendUleb(b, 0) // name (index 0 in line_str)
b = appendUleb(b, 0) // directory index
b = appendUleb(b, 0) // last modification time
b = appendUleb(b, 0) // file size
// Directory table (DWARF5 §6.2.4): entry format descriptors followed by
// the entries. One directory, the compilation directory, whose path is
// the empty string at .debug_line_str offset 0.
b = append(b, 1) // directory_entry_format_count
b = appendUleb(b, dwLnctPath) // DW_LNCT_path
b = appendUleb(b, dwFormLineStrp) // DW_FORM_line_strp
b = appendUleb(b, 1) // directories_count
b = le.AppendUint32(b, 0) // .debug_line_str offset of ""
// File table (DWARF5 §6.2.5). v5 indexes files from 0, so the source
// file is entry 0, matching the DW_AT_decl_file value 0 the DIEs carry.
b = append(b, 2) // file_name_entry_format_count
b = appendUleb(b, dwLnctPath) // DW_LNCT_path
b = appendUleb(b, dwFormLineStrp) // DW_FORM_line_strp
b = appendUleb(b, dwLnctDirIndex) // DW_LNCT_directory_index
b = appendUleb(b, dwFormUdata) // DW_FORM_udata
b = appendUleb(b, 1) // file_names_count
b = le.AppendUint32(b, srcStrOff) // .debug_line_str offset of the source name
b = appendUleb(b, 0) // directory index 0 (the compilation directory)
headerEnd := len(b)
@@ -176,8 +223,11 @@ func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
// Patch unit_length.
le.PutUint32(b[headerStart:], uint32(len(b)-headerStart-4))
// Patch header_length.
le.PutUint32(b[headerStart+6:], uint32(headerEnd-headerStart-10))
// Patch header_length. In the v5 header it follows the one-byte
// address_size and segment_selector_size (offset 8, not the DWARF2-4
// offset 6), and counts from just past itself to the first program
// byte.
le.PutUint32(b[headerStart+8:], uint32(headerEnd-headerStart-12))
return b
}
@@ -195,14 +245,16 @@ func dwarfBuildInfoSection(img *Image, srcFile string, ds *dwarfSections) []byte
// DW_TAG_compile_unit (abbrev 1).
b = append(b, 1) // abbreviation code
// DW_AT_low_pc: address of .text start.
infoRelocBase := len(b)
// DW_AT_low_pc: address of .text start. A data-only image has no
// functions to relocate against; its CU covers no code, so the base
// stays zero (the DWARF "no base address" value) with no relocation.
b = le.AppendUint64(b, 0) // placeholder
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
off: uint64(infoRelocBase),
name: img.Funcs[0].Name,
addend: 0,
})
if len(img.Funcs) > 0 {
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
off: uint64(len(b) - 8),
name: img.Funcs[0].Name,
})
}
// DW_AT_high_pc: size of .text.
b = le.AppendUint64(b, uint64(len(img.Code)))
// DW_AT_stmt_list: offset into .debug_line (0).
@@ -229,8 +281,9 @@ func dwarfBuildInfoSection(img *Image, srcFile string, ds *dwarfSections) []byte
b = le.AppendUint64(b, uint64(fn.Size))
// DW_AT_frame_base: DW_OP_call_frame_cfa.
b = append(b, 1, 0x9c)
// DW_AT_decl_file: file index 1.
b = append(b, 1)
// DW_AT_decl_file: the single file-table entry, index 0 (v5 indexes
// files from 0).
b = append(b, 0)
// DW_AT_decl_line.
b = append(b, uint8(fn.Line))
// DW_AT_external.
@@ -253,32 +306,61 @@ func appendUleb(b []byte, v uint64) []byte {
return binary.AppendUvarint(b, v)
}
// appendSleb appends v in signed LEB128, the encoding DWARF specifies:
// two's-complement sign extension, which is NOT Go's zigzag varint
// (binary.AppendVarint(-8) encodes 15, where DWARF wants 0x78).
func appendSleb(b []byte, v int64) []byte {
return binary.AppendVarint(b, v)
for {
c := byte(v & 0x7f)
v >>= 7
if (v == 0 && c&0x40 == 0) || (v == -1 && c&0x40 != 0) {
return append(b, c)
}
b = append(b, c|0x80)
}
}
// cfiArch carries the .debug_frame CIE parameters that differ per
// architecture: the DWARF register numbers of the stack pointer the initial
// CFA rule names and of the return address. The values are the ones the Go
// linker writes into its own CIE (cmd/link/internal/ld/dwarf.go uses
// Dwarfregsp and Dwarfreglr; the per-architecture constants live in
// cmd/link/internal/<arch>/l.go).
type cfiArch struct {
name string
cfaReg byte // the stack-pointer register the initial CFA rule names
raReg byte // the return-address register
}
var (
cfiAMD64 = cfiArch{"amd64", 7, 16} // RSP, RIP
cfiARM64 = cfiArch{"arm64", 31, 30} // SP (X31), LR (X30)
cfiRISCV64 = cfiArch{"riscv64", 2, 1} // X2 (sp), X1 (ra)
cfiLOONG64 = cfiArch{"loong64", 3, 1} // $r3 (sp), $r1 (ra)
)
// dwarfBuildFrameSection builds a .debug_frame section with CFI for stack
// unwinding. It emits one CIE and one FDE per function, encoding the
// CFA (Canonical Frame Address) rule changes at each stack-adjustment
// boundary recorded in FuncLayout.Spadj.
func dwarfBuildFrameSection(img *Image) []byte {
func dwarfBuildFrameSection(img *Image, cfi cfiArch, ds *dwarfSections) []byte {
var b []byte
le := binary.LittleEndian
// CIE (Common Information Entry).
cieStart := len(b)
b = append(b, 0, 0, 0, 0) // length (placeholder)
b = le.AppendUint32(b, 0xFFFFFFFF) // CIE marker
b = append(b, 3) // version (DWARF3, widely supported)
b = append(b, 0) // augmentation (empty)
b = appendUleb(b, 1) // code alignment
b = appendSleb(b, -8) // data alignment (-8 for 64-bit)
b = appendUleb(b, 16) // return address register (LR on arm64, RIP on amd64)
b = append(b, 0, 0, 0, 0) // length (placeholder)
b = le.AppendUint32(b, 0xFFFFFFFF) // CIE marker
b = append(b, 3) // version (DWARF3, widely supported)
b = append(b, 0) // augmentation (empty)
b = appendUleb(b, 1) // code alignment
b = appendSleb(b, -8) // data alignment (-8 for 64-bit)
b = appendUleb(b, uint64(cfi.raReg)) // return address register
// Initial CFA rule: DW_CFA_def_cfa (SP, 0)
b = append(b, 0x0c) // DW_CFA_def_cfa
b = appendUleb(b, 31) // register: SP (RSP=7 on amd64, SP=31 on arm64)
b = appendUleb(b, 0) // offset: 0
b = append(b, 0) // DW_CFA_nop (padding)
b = append(b, 0x0c) // DW_CFA_def_cfa
b = appendUleb(b, uint64(cfi.cfaReg)) // the architecture's stack pointer
b = appendUleb(b, 0) // offset: 0
b = append(b, 0) // DW_CFA_nop (padding)
// Patch CIE length.
le.PutUint32(b[cieStart:], uint32(len(b)-cieStart-4))
@@ -287,7 +369,12 @@ func dwarfBuildFrameSection(img *Image) []byte {
fdeStart := len(b)
b = append(b, 0, 0, 0, 0) // length (placeholder)
b = le.AppendUint32(b, uint32(cieStart)) // CIE pointer (offset from start)
// Initial location: function offset in .text (relocated by linker).
// Initial location: function offset in .text, referenced through
// the function's symbol so the linker relocates it.
ds.frameRelocs = append(ds.frameRelocs, dwarfReloc{
off: uint64(fdeStart + 8),
name: fn.Name,
})
b = le.AppendUint64(b, uint64(fn.Offset))
// Address range: function size.
b = le.AppendUint64(b, uint64(fn.Size))
+84 -24
View File
@@ -3,6 +3,17 @@
package asm
import "encoding/binary"
// Absolute 64-bit relocation types for the DWARF address fixups, one per
// supported architecture (the numbers debug/elf carries).
const (
rX8664Abs64 = 1 // R_X86_64_64
rAARCH64Abs64 = 257 // R_AARCH64_ABS64
rRISCVAbs64 = 2 // R_RISCV_64
rLarchAbs64 = 2 // R_LARCH_64
)
// dwarfELFSections holds the laid-out DWARF sections ready for inclusion
// in an ELF file.
type dwarfELFSections struct {
@@ -11,15 +22,23 @@ type dwarfELFSections struct {
lineOff, lineSize int
lineStrOff, lineStrSize int
frameOff, frameSize int
// Relocations for .debug_info address references.
// .rela.debug_info and .rela.debug_line contents: file offsets and
// entry counts (zero count: the section is absent).
infoRelaOff, infoRelaCount int
lineRelaOff, lineRelaCount int
frameRelaOff, frameRelaCount int
// Relocations for .debug_info address references, offsets relative to
// the section start (what an r_offset in .rela.debug_info means).
infoRelocs []elfDwarfReloc
// Relocations for .debug_line address references.
// Relocations for .debug_line address references, section-relative.
lineRelocs []elfDwarfReloc
// Relocations for .debug_frame FDE initial locations, section-relative.
frameRelocs []elfDwarfReloc
}
type elfDwarfReloc struct {
off uint64
sym int // symbol index in .symtab
off uint64 // offset within the target section
sym int // symbol index in .symtab
addend int64
}
@@ -29,8 +48,9 @@ type elfDwarfReloc struct {
//
// symIdx maps function names to their .symtab indices (needed for relocations
// against .text symbols). The map uses objectName format (pkg.name); the
// DWARF code uses bare function names, so we build a reverse lookup.
func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[string]int, align func(int)) *dwarfELFSections {
// DWARF code uses bare function names, so we build a reverse lookup. cfi
// carries the architecture's .debug_frame register conventions.
func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[string]int, align func(int), cfi cfiArch) *dwarfELFSections {
// Build a lookup from bare function name to symbol index.
nameToIdx := make(map[string]int, len(symIdx))
for name, idx := range symIdx {
@@ -45,7 +65,7 @@ func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[str
}
nameToIdx[name] = idx
}
ds := emitDWARF(img, srcFile)
ds := emitDWARF(img, srcFile, cfi)
if ds == nil || len(ds.debugAbbrev) == 0 {
return nil
}
@@ -68,15 +88,11 @@ func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[str
align(1)
result.lineOff = len(*out)
result.lineSize = len(ds.debugLine)
lineBase := len(*out)
*out = append(*out, ds.debugLine...)
// Patch .debug_line relocations: replace placeholder addresses with
// actual .text offsets via symbol lookup.
for _, dr := range ds.lineRelocs {
if idx, ok := nameToIdx[dr.name]; ok {
result.lineRelocs = append(result.lineRelocs, elfDwarfReloc{
off: uint64(lineBase) + dr.off,
off: dr.off,
sym: idx,
addend: dr.addend,
})
@@ -87,33 +103,77 @@ func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[str
align(1)
result.infoOff = len(*out)
result.infoSize = len(ds.debugInfo)
infoBase := len(*out)
*out = append(*out, ds.debugInfo...)
// .debug_frame
if len(ds.debugFrame) > 0 {
align(1)
result.frameOff = len(*out)
result.frameSize = len(ds.debugFrame)
*out = append(*out, ds.debugFrame...)
}
// Patch .debug_info relocations.
for _, dr := range ds.infoRelocs {
if idx, ok := nameToIdx[dr.name]; ok {
result.infoRelocs = append(result.infoRelocs, elfDwarfReloc{
off: uint64(infoBase) + dr.off,
off: dr.off,
sym: idx,
addend: dr.addend,
})
}
}
// .debug_frame: the section header declares alignment 8, so the data is
// padded to 8, matching it.
if len(ds.debugFrame) > 0 {
align(8)
result.frameOff = len(*out)
result.frameSize = len(ds.debugFrame)
*out = append(*out, ds.debugFrame...)
for _, dr := range ds.frameRelocs {
if idx, ok := nameToIdx[dr.name]; ok {
result.frameRelocs = append(result.frameRelocs, elfDwarfReloc{
off: dr.off,
sym: idx,
addend: dr.addend,
})
}
}
}
return result
}
// appendDWARFRelas writes the .rela.debug_info and .rela.debug_line section
// bodies from the relocations appendDWARFSections recorded, with the
// architecture's absolute 64-bit relocation type, and records their file
// offsets and entry counts on dw. Called after the DWARF sections
// themselves so the r_offsets (section-relative) need no adjustment.
func appendDWARFRelas(out *[]byte, dw *dwarfELFSections, abs64 uint32, align func(int)) {
le := binary.LittleEndian
write := func(relas []elfDwarfReloc) (off, count int) {
if len(relas) == 0 {
return 0, 0
}
align(8)
off = len(*out)
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(abs64))
le.PutUint64(b[16:], uint64(r.addend))
*out = append(*out, b[:]...)
}
return off, len(relas)
}
dw.infoRelaOff, dw.infoRelaCount = write(dw.infoRelocs)
dw.lineRelaOff, dw.lineRelaCount = write(dw.lineRelocs)
dw.frameRelaOff, dw.frameRelaCount = write(dw.frameRelocs)
}
// dwarfSourceName returns the source name the DWARF sections record: the
// image's source path when the assembler captured one, "gasm.s" otherwise.
func dwarfSourceName(img *Image) string {
if img.SourcePath != "" {
return img.SourcePath
}
return "gasm.s"
}
// dwarfSectionNames returns the DWARF section names for the string table.
var dwarfSectionNames = []string{
".debug_abbrev", ".debug_info", ".debug_line", ".debug_line_str",
".debug_frame", ".rela.debug_info", ".rela.debug_line",
".rela.debug_frame",
}
+303 -12
View File
@@ -4,11 +4,313 @@
package asm
import (
"bytes"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// ulebIter reads ULEB128 values, the .debug_abbrev and line-header
// encoding.
type ulebIter struct {
b []byte
i int
}
func (r *ulebIter) uleb(t *testing.T) uint64 {
t.Helper()
v, n := binary.Uvarint(r.b[r.i:])
if n <= 0 {
t.Fatalf("bad ULEB at %d", r.i)
}
r.i += n
return v
}
func (r *ulebIter) byteAt(t *testing.T) byte {
t.Helper()
if r.i >= len(r.b) {
t.Fatalf("read past end at %d", r.i)
}
c := r.b[r.i]
r.i++
return c
}
func (r *ulebIter) uint32At(t *testing.T) uint32 {
t.Helper()
v := binary.LittleEndian.Uint32(r.b[r.i:])
r.i += 4
return v
}
// sleb reads a signed LEB128, the DWARF encoding (sign-extended two's
// complement, not Go's zigzag varint).
func (r *ulebIter) sleb(t *testing.T) int64 {
t.Helper()
var v int64
var shift uint
for {
c := r.byteAt(t)
v |= int64(c&0x7f) << shift
shift += 7
if c&0x80 == 0 {
if c&0x40 != 0 {
v |= -1 << shift
}
return v
}
}
}
// dwarfAttr is one attribute/form pair of an abbreviation.
type dwarfAttr struct{ attr, form uint64 }
// dwarfAbbrev is one parsed abbreviation declaration.
type dwarfAbbrev struct {
code uint64
tag uint64
children bool
attrs []dwarfAttr
}
// parseAbbrevs walks a .debug_abbrev table: abbreviation code, tag,
// children flag, then attr/form ULEB pairs terminated by a double zero.
func parseAbbrevs(t *testing.T, b []byte) map[uint64]dwarfAbbrev {
t.Helper()
out := map[uint64]dwarfAbbrev{}
r := &ulebIter{b: b}
for {
code := r.uleb(t)
if code == 0 {
return out
}
ab := dwarfAbbrev{code: code, tag: r.uleb(t)}
ab.children = r.byteAt(t) == 1
for {
attr := r.uleb(t)
form := r.uleb(t)
if attr == 0 && form == 0 {
break
}
if attr == 0 || form == 0 {
t.Fatalf("abbrev %d: half-terminated attr/form pair (%d, %d)", code, attr, form)
}
ab.attrs = append(ab.attrs, dwarfAttr{attr, form})
}
out[code] = ab
}
}
func eqAttrs(t *testing.T, ab dwarfAbbrev, want []dwarfAttr) {
t.Helper()
if len(ab.attrs) != len(want) {
t.Fatalf("abbrev %d attrs = %v, want %v", ab.code, ab.attrs, want)
}
for i, w := range want {
if ab.attrs[i] != w {
t.Fatalf("abbrev %d attr %d = (%#x, %#x), want (%#x, %#x)", ab.code, i, ab.attrs[i].attr, ab.attrs[i].form, w.attr, w.form)
}
}
}
// TestDwarfAbbrevTable walks the abbreviation table as a consumer does and
// checks the attribute/form sets against the constants the toolchain uses
// (cmd/internal/dwarf/dwarf_defs.go). A wrong constant here renames an
// attribute (0x1b is comp_dir, not low_pc; 0x29 and 0x37 are bounds and
// count) and a wrong form desynchronises the DIE parse: 0x25 is strx1, one
// byte, where the writer emits four for a section offset.
func TestDwarfAbbrevTable(t *testing.T) {
abbrev := dwarfAbbrevTable()
if len(abbrev) == 0 {
t.Fatal("empty abbrev table")
}
// Must end with a zero byte (end of table).
if abbrev[len(abbrev)-1] != 0 {
t.Fatalf("abbrev table last byte = %d, want 0", abbrev[len(abbrev)-1])
}
abs := parseAbbrevs(t, abbrev)
if len(abs) != 2 {
t.Fatalf("abbreviations = %d, want 2", len(abs))
}
cu, ok := abs[1]
if !ok {
t.Fatal("missing abbreviation 1 (compile unit)")
}
if cu.tag != dwTagCompUnit || !cu.children {
t.Errorf("abbrev 1: tag %#x children %v, want compile unit with children", cu.tag, cu.children)
}
eqAttrs(t, cu, []dwarfAttr{
{dwAtLowPC, dwFormAddr},
{dwAtHighPC, dwFormData8},
{dwAtStmtList, dwFormSecOff},
{dwAtName, dwFormString},
})
sp, ok := abs[2]
if !ok {
t.Fatal("missing abbreviation 2 (subprogram)")
}
if sp.tag != dwTagSubprog || sp.children {
t.Errorf("abbrev 2: tag %#x children %v, want subprogram without children", sp.tag, sp.children)
}
eqAttrs(t, sp, []dwarfAttr{
{dwAtName, dwFormString},
{dwAtLowPC, dwFormAddr},
{dwAtHighPC, dwFormData8},
{dwAtFrameBase, dwFormExprloc},
{dwAtDeclFile, dwFormData1},
{dwAtDeclLine, dwFormData1},
{dwAtExternal, 0x0c}, // DW_FORM_flag
})
}
// TestDwarfLineHeaderV5 parses the .debug_line header under DWARF5 rules:
// the directory and file tables are format-descriptor lists, not the
// DWARF2-4 shape of null-terminated strings, and the file entry references
// the source name through .debug_line_str.
func TestDwarfLineHeaderV5(t *testing.T) {
src := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
ADDQ BX, AX
MOVQ AX, ret+16(FP)
RET
`
f, errs := parser.Parse("test_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
r := &ulebIter{b: ds.debugLine}
r.uint32At(t) // unit_length
if v := binary.LittleEndian.Uint16(ds.debugLine[4:]); v != 5 {
t.Fatalf("version = %d, want 5", v)
}
r.i = 6
r.byteAt(t) // address_size
r.byteAt(t) // segment_selector_size
r.uint32At(t) // header_length
r.byteAt(t) // minimum_instruction_length
r.byteAt(t) // maximum_ops_per_instruction
r.byteAt(t) // default_is_stmt
r.byteAt(t) // line_base
r.byteAt(t) // line_range
opcodeBase := r.byteAt(t)
for range int(opcodeBase) - 1 {
r.byteAt(t) // standard opcode lengths
}
// Directory table (DWARF5 §6.2.4).
if n := r.byteAt(t); n != 1 {
t.Fatalf("directory_entry_format_count = %d, want 1", n)
}
if lnct := r.uleb(t); lnct != dwLnctPath {
t.Errorf("directory content type = %#x, want DW_LNCT_path", lnct)
}
if form := r.uleb(t); form != dwFormLineStrp {
t.Errorf("directory form = %#x, want DW_FORM_line_strp", form)
}
if n := r.uleb(t); n != 1 {
t.Fatalf("directories_count = %d, want 1", n)
}
if off := r.uint32At(t); off != 0 {
t.Errorf("compilation directory line_strp = %d, want 0 (the empty string)", off)
}
// File table (DWARF5 §6.2.5).
if n := r.byteAt(t); n != 2 {
t.Fatalf("file_name_entry_format_count = %d, want 2", n)
}
if lnct := r.uleb(t); lnct != dwLnctPath {
t.Errorf("file content type = %#x, want DW_LNCT_path", lnct)
}
if form := r.uleb(t); form != dwFormLineStrp {
t.Errorf("file path form = %#x, want DW_FORM_line_strp", form)
}
if lnct := r.uleb(t); lnct != dwLnctDirIndex {
t.Errorf("file content type = %#x, want DW_LNCT_directory_index", lnct)
}
if form := r.uleb(t); form != dwFormUdata {
t.Errorf("file dir-index form = %#x, want DW_FORM_udata", form)
}
if n := r.uleb(t); n != 1 {
t.Fatalf("file_names_count = %d, want 1", n)
}
strOff := r.uint32At(t)
if dirIdx := r.uleb(t); dirIdx != 0 {
t.Errorf("file directory index = %d, want 0", dirIdx)
}
// The file entry's line_strp must resolve to the source name.
end := int(strOff) + len("test_amd64.s")
if int(strOff) >= len(ds.debugLineStr) || !bytes.Equal(ds.debugLineStr[strOff:end], []byte("test_amd64.s")) {
t.Errorf("file entry line_strp %d does not name the source: %q", strOff, ds.debugLineStr)
}
// The fixed header fields: address_size 8 and a header_length that
// points just past the file table (the patch site is offset 8 in the
// v5 header, and the field counts from its own end).
if ds.debugLine[6] != 8 || ds.debugLine[7] != 0 {
t.Errorf("address_size/segment_selector = %d/%d, want 8/0", ds.debugLine[6], ds.debugLine[7])
}
if hl := binary.LittleEndian.Uint32(ds.debugLine[8:]); hl != uint32(r.i-12) {
t.Errorf("header_length = %d, want %d (the byte after the file table is %d)", hl, r.i-12, r.i)
}
}
// TestDwarfFrameCIEArch checks the shared CIE carries each architecture's
// stack-pointer and return-address registers: the values the Go linker
// writes (cmd/link/internal/<arch>/l.go dwarfRegSP/dwarfRegLR).
func TestDwarfFrameCIEArch(t *testing.T) {
for _, tc := range []struct {
name string
cfi cfiArch
}{
{"amd64", cfiAMD64},
{"arm64", cfiARM64},
{"riscv64", cfiRISCV64},
{"loong64", cfiLOONG64},
} {
frame := dwarfBuildFrameSection(&Image{}, tc.cfi, &dwarfSections{})
r := &ulebIter{b: frame}
r.uint32At(t) // length
if cid := r.uint32At(t); cid != 0xFFFFFFFF {
t.Errorf("%s: CIE id = %#x, want 0xffffffff", tc.name, cid)
}
if v := r.byteAt(t); v != 3 {
t.Errorf("%s: CIE version = %d, want 3", tc.name, v)
}
if aug := r.byteAt(t); aug != 0 {
t.Errorf("%s: CIE augmentation = %d, want 0", tc.name, aug)
}
if ca := r.uleb(t); ca != 1 {
t.Errorf("%s: code alignment = %d, want 1", tc.name, ca)
}
if da := r.sleb(t); da != -8 {
t.Errorf("%s: data alignment = %d, want -8 (signed LEB128, not zigzag)", tc.name, da)
}
if ra := r.uleb(t); ra != uint64(tc.cfi.raReg) {
t.Errorf("%s: return-address register = %d, want %d", tc.name, ra, tc.cfi.raReg)
}
if op := r.byteAt(t); op != 0x0c {
t.Errorf("%s: expected DW_CFA_def_cfa, got opcode %#x", tc.name, op)
}
if cfa := r.uleb(t); cfa != uint64(tc.cfi.cfaReg) {
t.Errorf("%s: CFA register = %d, want %d", tc.name, cfa, tc.cfi.cfaReg)
}
if off := r.uleb(t); off != 0 {
t.Errorf("%s: CFA offset = %d, want 0", tc.name, off)
}
}
}
func TestEmitDWARF(t *testing.T) {
src := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
@@ -27,7 +329,7 @@ TEXT ·add(SB), NOSPLIT, $0-24
t.Fatalf("assemble: %v", err)
}
ds := emitDWARF(img, "test_amd64.s")
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
// .debug_abbrev must not be empty and must start with abbrev code 1.
if len(ds.debugAbbrev) == 0 {
@@ -68,14 +370,3 @@ TEXT ·add(SB), NOSPLIT, $0-24
t.Fatal("no .debug_info relocations")
}
}
func TestDwarfAbbrevTable(t *testing.T) {
abbrev := dwarfAbbrevTable()
if len(abbrev) == 0 {
t.Fatal("empty abbrev table")
}
// Must end with a zero byte (end of table).
if abbrev[len(abbrev)-1] != 0 {
t.Fatalf("abbrev table last byte = %d, want 0", abbrev[len(abbrev)-1])
}
}
+442
View File
@@ -12,6 +12,7 @@ import (
"path/filepath"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
@@ -211,6 +212,75 @@ func TestELFObject(t *testing.T) {
}
}
// TestELFObjectTLSGuardReloc checks that a non-NOSPLIT function's stack
// guard carries an R_X86_64_TPOFF32 relocation against the null symbol in
// .rela.text. The serialisation must honour the record's type field: a
// hardcoded R_X86_64_PC32 mislinks the TLS load as an ordinary
// PC-relative reference.
func TestELFObjectTLSGuardReloc(t *testing.T) {
f, errs := parser.Parse("g_amd64.s", `
#include "textflag.h"
TEXT ·grow(SB), $0
CALL ·other(SB)
RET
TEXT ·other(SB), NOSPLIT, $0
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
var haveTLS bool
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.Kind == RelTLSLE {
haveTLS = true
}
}
}
if !haveTLS {
t.Fatal("test source produced no RelTLSLE relocation")
}
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaSec := ef.Section(".rela.text")
if relaSec == nil {
t.Fatal("missing .rela.text")
}
raw, err := relaSec.Data()
if err != nil {
t.Fatal(err)
}
found := false
for i := 0; i+24 <= len(raw); i += 24 {
e := raw[i:]
info := binary.LittleEndian.Uint64(e[8:])
typ := info & 0xffffffff
sym := int(info >> 32)
if typ == uint64(elf.R_X86_64_TPOFF32) {
found = true
if sym != 0 {
t.Errorf("TPOFF32 relocation against symbol %d, want 0 (the null symbol)", sym)
}
}
}
if !found {
t.Errorf("no R_X86_64_TPOFF32 relocation in .rela.text (%d bytes)", len(raw))
}
}
// TestELFObjectNoRelocations checks a file with no static-symbol references
// emits a valid object without a .rela.text section.
func TestELFObjectNoRelocations(t *testing.T) {
@@ -253,6 +323,238 @@ TEXT ·nop(SB), NOSPLIT, $0
}
}
// elfSectionHeaderCount returns the e_shnum the ELF header declares.
func elfSectionHeaderCount(t *testing.T, obj []byte) int {
t.Helper()
return int(binary.LittleEndian.Uint16(obj[60:]))
}
// checkELFSectionAccounting verifies the number of section headers the
// writer physically laid out equals e_shnum: every DWARF section written
// after .shstrtab must be counted, or the last ones (always .debug_frame)
// are invisible to every consumer, debug/elf included.
func checkELFSectionAccounting(t *testing.T, obj []byte) {
t.Helper()
shoff := int(binary.LittleEndian.Uint64(obj[40:]))
shentsize := int(binary.LittleEndian.Uint16(obj[58:]))
shnum := elfSectionHeaderCount(t, obj)
if shentsize != 64 {
t.Fatalf("e_shentsize = %d, want 64", shentsize)
}
if (len(obj)-shoff)%shentsize != 0 {
t.Fatalf("section header table is not a whole number of entries: shoff=%d len=%d", shoff, len(obj))
}
if present := (len(obj) - shoff) / shentsize; present != shnum {
t.Errorf("e_shnum = %d but %d section headers are laid out", shnum, present)
}
}
// TestELFDWARFSectionAccounting runs the header accounting check over all
// four architecture emitters, and additionally checks the .debug_frame
// section is visible (its data aligned as its header declares).
func TestELFDWARFSectionAccounting(t *testing.T) {
parse := func(name, src string) *ast.File {
f, errs := parser.Parse(name, src)
if len(errs) > 0 {
t.Fatalf("parse %s: %v", name, errs)
}
return f
}
cases := []struct {
name string
img *Image
emit func(*Image) ([]byte, error)
}{
{"amd64", elfTestImage(t), (*Image).ELFObject},
{"arm64", mustImage(t, func() (*Image, error) {
return AssembleFileARM64(parse("k_arm64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
`))
}), (*Image).ELFAARCH64Object},
{"riscv64", mustImage(t, func() (*Image, error) {
return AssembleFileRISCV(parse("k_riscv64.s", `
#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0-0
MOV $answer<>(SB), X10
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`))
}), (*Image).ELFRISCVObject},
{"loong64", mustImage(t, func() (*Image, error) {
return AssembleFileLOONG64(parse("k_loong64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
`))
}), (*Image).ELFLOONG64Object},
}
for _, tc := range cases {
obj, err := tc.emit(tc.img)
if err != nil {
t.Fatalf("%s: emit: %v", tc.name, err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("%s: parse emitted object: %v", tc.name, err)
}
frame := ef.Section(".debug_frame")
if frame == nil {
t.Errorf("%s: .debug_frame invisible to debug/elf (e_shnum too small?)", tc.name)
ef.Close()
continue
}
if frame.Offset%8 != 0 || frame.Addralign != 8 {
t.Errorf("%s: .debug_frame offset %d align %d, want offset%%8==0 align 8", tc.name, frame.Offset, frame.Addralign)
}
ef.Close()
}
}
func mustImage(t *testing.T, f func() (*Image, error)) *Image {
t.Helper()
img, err := f()
if err != nil {
t.Fatal(err)
}
return img
}
// TestELFDWARFRelocations checks the .rela.debug_info and .rela.debug_line
// sections exist and carry absolute 64-bit relocations against the
// function symbols, with r_offsets inside their target sections.
func TestELFDWARFRelocations(t *testing.T) {
img := elfTestImage(t)
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
// The DWARF must record the assembled file's path (threaded through
// Image.SourcePath), not a placeholder name.
info, err := ef.Section(".debug_info").Data()
if err != nil {
t.Fatal(err)
}
if img.SourcePath != "t_amd64.s" || !bytes.Contains(info, []byte(img.SourcePath)) {
t.Errorf("DWARF compilation unit does not name the source %q", img.SourcePath)
}
for _, tc := range []struct {
rela string
target string
want uint32
}{
{".rela.debug_info", ".debug_info", rX8664Abs64},
{".rela.debug_line", ".debug_line", rX8664Abs64},
{".rela.debug_frame", ".debug_frame", rX8664Abs64},
} {
rs := ef.Section(tc.rela)
if rs == nil {
t.Fatalf("missing %s", tc.rela)
}
if rs.Type != elf.SHT_RELA {
t.Errorf("%s: type %v, want SHT_RELA", tc.rela, rs.Type)
}
target := ef.Section(tc.target)
if target == nil {
t.Fatalf("missing %s", tc.target)
}
if rs.Link == 0 || ef.Sections[rs.Info] != target {
t.Errorf("%s: link %d info %d, want the symtab and %s", tc.rela, rs.Link, rs.Info, tc.target)
}
b, err := rs.Data()
if err != nil {
t.Fatal(err)
}
// .debug_line has one address per function; .debug_info adds the
// compile unit's own low_pc.
want := len(img.Funcs)
if tc.target == ".debug_info" {
want++
}
if len(b)/24 != want {
t.Errorf("%s: %d entries, want %d", tc.rela, len(b)/24, want)
}
for i := 0; i+24 <= len(b); i += 24 {
r_offset := binary.LittleEndian.Uint64(b[i:])
info := binary.LittleEndian.Uint64(b[i+8:])
typ := uint32(info)
sym := int(info >> 32)
if typ != tc.want {
t.Errorf("%s entry %d: type %d, want R_X86_64_64 (%d)", tc.rela, i/24, typ, tc.want)
}
if r_offset >= uint64(target.Size) {
t.Errorf("%s entry %d: r_offset %d outside %s (%d bytes)", tc.rela, i/24, r_offset, tc.target, target.Size)
}
if sym == 0 {
t.Errorf("%s entry %d: against the null symbol", tc.rela, i/24)
}
}
}
}
// TestELFDataOnly checks a source with GLOBL data and no TEXT emits a valid
// ELF object: the DWARF compilation unit of a code-less image has no
// function to relocate against and must not reach for one.
func TestELFDataOnly(t *testing.T) {
f, errs := parser.Parse("d0_amd64.s", `
GLOBL table<>(SB), RODATA, $8
DATA table<>+0(SB)/8, $12345
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
found := false
for _, s := range syms {
if s.Name == "table" && s.Size == 8 {
found = true
}
}
if !found {
t.Errorf("data symbol table missing: %v", syms)
}
if ef.Section(".rela.debug_info") != nil || ef.Section(".rela.debug_line") != nil {
t.Error("data-only image must not emit DWARF address relocations")
}
}
// TestELFLinkAndRun is the end-to-end check: assemble the test functions,
// link the emitted object with a C driver that defines the external symbol,
// and run the result. Skipped when no C compiler is available.
@@ -307,4 +609,144 @@ int main(void) {
if got := string(run); got != "42 42 7\n" {
t.Errorf("output %q, want \"42 42 7\\n\"", got)
}
// The DWARF addresses must have resolved at link time: the .debug_info
// placeholders were carried by .rela.debug_info, so every subprogram's
// low_pc must now equal its linked symbol address.
bin, err := os.ReadFile(appPath)
if err != nil {
t.Fatal(err)
}
lef, err := elf.NewFile(bytes.NewReader(bin))
if err != nil {
t.Fatalf("parse linked binary: %v", err)
}
defer lef.Close()
syms, err := lef.Symbols()
if err != nil {
t.Fatal(err)
}
addrByName := map[string]uint64{}
for _, s := range syms {
if elf.ST_TYPE(s.Info) == elf.STT_FUNC && s.Value != 0 {
addrByName[s.Name] = s.Value
}
}
lowPCs := dwarfSubprogramLowPCs(t, lef)
if len(lowPCs) == 0 {
t.Fatal("no subprogram DW_AT_low_pc parsed from the linked binary")
}
for name, pc := range lowPCs {
addr, ok := addrByName[name]
if !ok {
t.Errorf("subprogram %q not in the linked symbol table", name)
continue
}
if pc != addr {
t.Errorf("subprogram %q: DW_AT_low_pc = %#x, linked address %#x (DWARF relocation unresolved)", name, pc, addr)
}
}
}
// dwarfSubprogramLowPCs walks the linked binary's .debug_info with its own
// .debug_abbrev and returns each DW_TAG_subprogram's DW_AT_low_pc by name.
func dwarfSubprogramLowPCs(t *testing.T, ef *elf.File) map[string]uint64 {
t.Helper()
abbrevSec := ef.Section(".debug_abbrev")
infoSec := ef.Section(".debug_info")
if abbrevSec == nil || infoSec == nil {
t.Fatal("linked binary lacks .debug_abbrev or .debug_info")
}
abbrev, err := abbrevSec.Data()
if err != nil {
t.Fatal(err)
}
info, err := infoSec.Data()
if err != nil {
t.Fatal(err)
}
abs := parseAbbrevs(t, abbrev)
le := binary.LittleEndian
out := map[string]uint64{}
r := &ulebIter{b: info}
r.uint32At(t) // unit_length
if v := le.Uint16(info[4:]); v != 5 {
t.Fatalf(".debug_info version %d, want 5", v)
}
r.i = 6
r.byteAt(t) // unit_type
r.byteAt(t) // address_size
r.uint32At(t) // debug_abbrev_offset
var name string
var lowPC uint64
for r.i < len(r.b) {
code := r.uleb(t)
if code == 0 {
continue // end of the CU's children
}
ab, ok := abs[code]
if !ok {
t.Fatalf("unknown abbreviation code %d", code)
}
name, lowPC = "", 0
for _, a := range ab.attrs {
switch a.attr {
case dwAtName:
readFormKeep(t, r, a.form, &name, nil)
case dwAtLowPC:
readFormKeep(t, r, a.form, nil, &lowPC)
default:
readFormSkip(t, r, a.form)
}
}
if ab.tag == dwTagSubprog && name != "" {
out[name] = lowPC
}
}
return out
}
// readFormKeep reads one DIE attribute value, keeping a string or an
// address into the pointer it was given (nil keeps nothing).
func readFormKeep(t *testing.T, r *ulebIter, form uint64, name *string, addr *uint64) {
t.Helper()
switch form {
case dwFormString:
end := r.i
for end < len(r.b) && r.b[end] != 0 {
end++
}
if name != nil {
*name = string(r.b[r.i:end])
}
r.i = end + 1
case dwFormAddr:
if addr != nil {
*addr = binary.LittleEndian.Uint64(r.b[r.i:])
}
r.i += 8
default:
readFormSkip(t, r, form)
}
}
func readFormSkip(t *testing.T, r *ulebIter, form uint64) {
t.Helper()
switch form {
case dwFormString:
for r.i < len(r.b) && r.b[r.i] != 0 {
r.i++
}
r.i++
case dwFormAddr, dwFormData8:
r.i += 8
case dwFormSecOff:
r.i += 4
case dwFormExprloc:
r.i += int(r.uleb(t))
case dwFormData1, 0x0c:
r.i++
default:
t.Fatalf("unsupported form %#x", form)
}
}
+74 -22
View File
@@ -81,10 +81,15 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
}
// Build relocations. Each SB reference is an ADRP pair:
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21
// ADD → R_AARCH64_ADD_ABS_LO12_NC
// LDR/STR X → R_AARCH64_LDST64_ABS_LO12_NC
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21 at the ADRP
// ADD → R_AARCH64_ADD_ABS_LO12_NC at the ADD word
// LDR/STR X → R_AARCH64_LDST64_ABS_LO12_NC at the LDR/STR word
// BL → R_AARCH64_CALL26
// cmd/link's own conversion emits the HI21 at sectoff and the LO12 at
// sectoff+4 (cmd/link/internal/arm64/asm.go), so the ADD or load word
// carries the page-offset relocation, never a second HI21. The
// assembler records two RelArm64Addr relocs per ADRP+ADD pair (one per
// word), so the second of the pair is consumed here.
// Addends stay raw: ADR_PREL_PG_HI21 and the ABS_LO12_NC forms resolve
// against S+A, and CALL26 branches take the branch instruction's own
// place as the PC-relative base, so subtracting the field width (the
@@ -97,28 +102,34 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
}
var relas []elfRela
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
for i := 0; i < len(fn.Relocs); i++ {
r := fn.Relocs[i]
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
var typ uint32
switch {
case r.Kind == RelArm64Branch:
typ = rArm64Call26
case r.Kind == RelArm64LDST64 && r.Off%4 == 4:
typ = rArm64Ldst64Lo12NC
case r.Kind == RelArm64Addr && r.Off%4 == 4:
typ = rArm64AddAbsLo12NC
switch r.Kind {
case RelArm64Branch:
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off), typ: rArm64Call26, sym: idx, addend: r.Addend,
})
case RelArm64Addr:
// ADRP+ADD: the pair's second reloc (at Off+4) is the
// assembler's twin of the same pair; skip it.
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rArm64PrelPgHi21, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rArm64AddAbsLo12NC, sym: idx, addend: r.Addend},
)
i++
case RelArm64LDST64:
// ADRP+LDR/STR: one assembler reloc covers the pair.
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rArm64PrelPgHi21, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rArm64Ldst64Lo12NC, sym: idx, addend: r.Addend},
)
default:
typ = rArm64PrelPgHi21
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
}
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
typ: typ,
sym: idx,
addend: r.Addend,
})
}
}
@@ -193,15 +204,32 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
shstrOff := len(out)
out = append(out, stSections.bytes()...)
// DWARF debug sections.
// DWARF debug sections; the address placeholders they leave are carried
// as .rela.debug_info/.rela.debug_line entries the system linker applies.
dwAlign := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiARM64)
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
if dw != nil {
nSections += 4
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
// .debug_line_str and .debug_frame (the CIE is unconditional, so
// the frame section is always present), plus the relocation
// sections below when they carry entries.
dwarfStart = nSections
nSections += 5
appendDWARFRelas(&out, dw, rAARCH64Abs64, dwAlign)
if dw.infoRelaCount > 0 {
nSections++
}
if dw.lineRelaCount > 0 {
nSections++
}
if dw.frameRelaCount > 0 {
nSections++
}
}
align(8)
@@ -230,13 +258,37 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order.
if dw != nil {
// secIdx is a running section index: each putSh below emits the
// next header, and the sh_info of a .rela section names the index
// of the section it relocates.
secIdx := dwarfStart
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
secIdx++
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
secInfoIdx := secIdx
secIdx++
if dw.infoRelaCount > 0 {
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
secIdx++
}
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
secLineIdx := secIdx
secIdx++
if dw.lineRelaCount > 0 {
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
secIdx++
}
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
secIdx++
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
secFrameIdx := secIdx
secIdx++
if dw.frameRelaCount > 0 {
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
}
}
}
+58 -1
View File
@@ -6,6 +6,7 @@ package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
@@ -28,6 +29,7 @@ TEXT ·add(SB), NOSPLIT, $0-24
TEXT ·getanswer(SB), NOSPLIT, $0-8
MOVD answer<>(SB), R4
MOVD $answer<>(SB), R5
MOVD R4, ret+0(FP)
RET
@@ -102,8 +104,63 @@ DATA answer<>+0(SB)/8, $42
// Check that .rela.text exists (getanswer has SB reference).
relaText := ef.Section(".rela.text")
if relaText == nil {
t.Error("missing .rela.text section")
t.Fatal("missing .rela.text section")
}
// The SB references of getanswer form two ADRP pairs: the load
// (MOVD answer<>(SB), R4) is ADRP+LDR carrying HI21 at the ADRP and
// LDST64_ABS_LO12_NC at the LDR word, and the address-of
// (MOVD $answer<>(SB), R5) is ADRP+ADD carrying HI21 and
// ADD_ABS_LO12_NC. cmd/link's own conversion emits exactly this
// sectoff / sectoff+4 pairing; a second HI21 at the ADD or LDR word
// corrupts the pair.
raw, err := relaText.Data()
if err != nil {
t.Fatal(err)
}
if len(raw)%24 != 0 || len(raw)/24 != 4 {
t.Fatalf(".rela.text has %d bytes, want four 24-byte entries", len(raw))
}
wantRela := []struct {
typ elf.R_AARCH64
off uint64 // relative to the getanswer function start
}{
{elf.R_AARCH64_ADR_PREL_PG_HI21, 0},
{elf.R_AARCH64_LDST64_ABS_LO12_NC, 4},
{elf.R_AARCH64_ADR_PREL_PG_HI21, 8},
{elf.R_AARCH64_ADD_ABS_LO12_NC, 12},
}
getanswer := byNameElf(t, ef, "getanswer")
for i, w := range wantRela {
e := raw[i*24 : (i+1)*24]
off := binary.LittleEndian.Uint64(e[0:])
info := binary.LittleEndian.Uint64(e[8:])
typ := elf.R_AARCH64(info & 0xffffffff)
sym := int(info >> 32)
if typ != w.typ || off != getanswer.Value+w.off {
t.Errorf("reloc %d: type %v off %d, want %v at %d", i, typ, off, w.typ, getanswer.Value+w.off)
}
if sym != 3 { // NULL, .text, .data, then the first local: answer
t.Errorf("reloc %d: symbol index %d, want 3 (answer)", i, sym)
}
}
}
// byNameElf returns the symbol table entry for name from the raw .symtab,
// which carries every entry including the null and section symbols in order.
func byNameElf(t *testing.T, ef *elf.File, name string) elf.Symbol {
t.Helper()
syms, err := ef.Symbols()
if err != nil {
t.Fatalf("symbols: %v", err)
}
for _, s := range syms {
if s.Name == name {
return s
}
}
t.Fatalf("symbol %q not found", name)
return elf.Symbol{}
}
// TestELFAARCH64ObjectNoRelocations checks the ELF output when there are no
+49 -3
View File
@@ -13,6 +13,12 @@ import (
const (
emLOONGARCH = 258 // EM_LOONGARCH
// EF_LOONGARCH_ABI_DOUBLE_FLOAT | EF_LOONGARCH_OBJABI_V1: the flags the
// Go toolchain writes (cmd/link/internal/ld/elf.go: Flags = 0x43 for
// Loong64). System linkers refuse to merge ET_REL objects whose float
// ABI differs, so 0 (soft-float) would make the object unlinkable.
efLarchAbiDoubleObjV1 = 0x43
// LoongArch relocation types (the ELF psABI).
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
@@ -187,9 +193,25 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiLOONG64)
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
if dw != nil {
nSections += 4
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
// .debug_line_str and .debug_frame (the CIE is unconditional, so
// the frame section is always present), plus the relocation
// sections below when they carry entries.
dwarfStart = nSections
nSections += 5
appendDWARFRelas(&out, dw, rLarchAbs64, dwAlign)
if dw.infoRelaCount > 0 {
nSections++
}
if dw.lineRelaCount > 0 {
nSections++
}
if dw.frameRelaCount > 0 {
nSections++
}
}
align(8)
@@ -218,13 +240,37 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order.
if dw != nil {
// secIdx is a running section index: each putSh below emits the
// next header, and the sh_info of a .rela section names the index
// of the section it relocates.
secIdx := dwarfStart
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
secIdx++
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
secInfoIdx := secIdx
secIdx++
if dw.infoRelaCount > 0 {
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
secIdx++
}
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
secLineIdx := secIdx
secIdx++
if dw.lineRelaCount > 0 {
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
secIdx++
}
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
secIdx++
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
secFrameIdx := secIdx
secIdx++
if dw.frameRelaCount > 0 {
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
}
}
}
@@ -237,7 +283,7 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
le.PutUint64(hdr[24:], 0)
le.PutUint64(hdr[32:], 0)
le.PutUint64(hdr[40:], uint64(shoff))
le.PutUint32(hdr[48:], 0)
le.PutUint32(hdr[48:], efLarchAbiDoubleObjV1)
le.PutUint16(hdr[52:], 64)
le.PutUint16(hdr[54:], 0)
le.PutUint16(hdr[56:], 0)
+5
View File
@@ -55,6 +55,11 @@ DATA answer<>+0(SB)/8, $42
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_LOONGARCH {
t.Errorf("type/machine = %v/%v, want ET_REL/EM_LOONGARCH", ef.Type, ef.Machine)
}
// The double-float ABI plus OBJABI_V1 flags the Go toolchain writes;
// system linkers refuse ABI-mismatched merges.
if flags := binary.LittleEndian.Uint32(obj[48:]); flags != efLarchAbiDoubleObjV1 {
t.Errorf("e_flags = %#x, want %#x (double-float, OBJABI_V1)", flags, efLarchAbiDoubleObjV1)
}
text := ef.Section(".text")
data := ef.Section(".data")
+60 -11
View File
@@ -13,8 +13,13 @@ import (
const (
emRISCV = 243 // EM_RISCV
// EF_RISCV_FLOAT_ABI_DOUBLE: the double-precision float ABI the Go
// toolchain targets (cmd/link/internal/ld/elf.go writes Flags = 0x4 for
// RISCV64). System linkers refuse to merge ET_REL objects whose float
// ABI differs, so 0 (soft-float) would make the object unlinkable.
efRISCVFloatAbiDouble = 0x4
// RISC-V relocation types.
rRISCV32 = 1
rRISCVJAL = 17 // R_RISCV_JAL
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
@@ -83,9 +88,14 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
// Build relocations. Each SB reference is an AUIPC + second-instruction
// pair carrying a single relocation kind; the ELF writer expands it into
// the R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I/S pair the psABI expects.
// The HI20 carries the symbol addend; the LO12 addend is zero, matching
// cmd/link's own ELF conversion (the LO12 resolves against the HI20's
// AUIPC location).
// The HI20 carries the symbol and its addend. The LO12's symbol must
// denote the AUIPC site the HI20 relocates (psABI §8.4.9: the pair is
// resolved against the label of the AUIPC, not the target symbol;
// cmd/link generates one local text symbol per AUIPC for exactly this,
// cmd/link/internal/riscv64/asm.go). The .text section symbol with the
// AUIPC's section-relative offset as addend gives S + A = the AUIPC
// address, which is that label.
const secSymText = 1 // syms[1], the .text section symbol
type elfRela struct {
off uint64
typ uint32
@@ -99,21 +109,20 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
auipc := int64(fn.Offset + r.Off)
switch r.Kind {
case RelRISCVPCRELIType:
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: idx, addend: 0},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: secSymText, addend: auipc},
)
case RelRISCVPCRELSType:
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: idx, addend: 0},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: secSymText, addend: auipc},
)
case RelRISCVJal:
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVJAL, sym: idx, addend: r.Addend})
case RelPCRelAbs:
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCV32, sym: idx, addend: r.Addend})
default:
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
}
@@ -196,9 +205,25 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiRISCV64)
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
if dw != nil {
nSections += 4
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
// .debug_line_str and .debug_frame (the CIE is unconditional, so
// the frame section is always present), plus the relocation
// sections below when they carry entries.
dwarfStart = nSections
nSections += 5
appendDWARFRelas(&out, dw, rRISCVAbs64, dwAlign)
if dw.infoRelaCount > 0 {
nSections++
}
if dw.lineRelaCount > 0 {
nSections++
}
if dw.frameRelaCount > 0 {
nSections++
}
}
align(8)
@@ -227,13 +252,37 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order.
if dw != nil {
// secIdx is a running section index: each putSh below emits the
// next header, and the sh_info of a .rela section names the index
// of the section it relocates.
secIdx := dwarfStart
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
secIdx++
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
secInfoIdx := secIdx
secIdx++
if dw.infoRelaCount > 0 {
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
secIdx++
}
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
secLineIdx := secIdx
secIdx++
if dw.lineRelaCount > 0 {
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
secIdx++
}
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
secIdx++
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
secFrameIdx := secIdx
secIdx++
if dw.frameRelaCount > 0 {
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
}
}
}
@@ -246,7 +295,7 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
le.PutUint64(hdr[24:], 0)
le.PutUint64(hdr[32:], 0)
le.PutUint64(hdr[40:], uint64(shoff))
le.PutUint32(hdr[48:], 0)
le.PutUint32(hdr[48:], efRISCVFloatAbiDouble)
le.PutUint16(hdr[52:], 64)
le.PutUint16(hdr[54:], 0)
le.PutUint16(hdr[56:], 0)
+15 -3
View File
@@ -36,10 +36,15 @@ func Encodable(mnemonic string) bool {
}
// CMOV carries size then condition (CMOVLGT); SET carries the condition
// alone (SETNE).
// alone (SETNE). The size letter is checked exactly as encodeCmov does,
// so a spelling like CMOVBGT is not reported encodable when Encode
// would reject it.
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
if _, ok := jccMap[rest[1:]]; ok {
return true
switch rest[0] {
case 'W', 'L', 'Q':
if _, ok := jccMap[rest[1:]]; ok {
return true
}
}
}
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
@@ -81,9 +86,16 @@ func Encodable(mnemonic string) bool {
"BSWAP",
"PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2",
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX",
"CVTSL2SD", "CVTSQ2SD",
"MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
return true
}
// Full-name dispatches the size split would eat (a trailing width
// letter that is part of the mnemonic).
switch upper {
case "PMOVMSKB":
return true
}
return false
}
+15 -3
View File
@@ -101,6 +101,11 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if m, ok := sseBinTable[base]; ok {
return e.encodeSSEBin(m, ops)
}
// PMOVMSKB ends in a width letter the size split would eat, so it
// dispatches on the full name like the packed binaries above.
if upper == "PMOVMSKB" {
return e.encodePmovmskb(upper, ops)
}
switch base {
case "MOV":
return e.encodeMov(ops, size)
@@ -117,16 +122,17 @@ func (e *enc) encode(mnem string, ops []Operand) error {
case "IMUL", "IMUL3":
return e.encodeImul(ops, size)
case "PUSH":
return e.encodePushPop(ops, true)
return e.encodePushPop(ops, size, true)
case "POP":
return e.encodePushPop(ops, false)
return e.encodePushPop(ops, size, false)
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
return e.encodeCount(base, ops, size)
case "BSWAP":
return e.encodeBswap(ops, size)
case "PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2":
return e.encodePrefetch(base, ops)
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX":
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX":
return e.encodeMovExtend(base, ops)
case "CVTSL2SD", "CVTSQ2SD":
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
@@ -345,6 +351,12 @@ func setMem(i *instr, regField int, m Mem) error {
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
sib = -1
// A displacement wider than int32 fits no encoding form; truncating it
// would address a different location, and go tool asm reports "offset
// too large" for the same operand.
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
}
// RIP-relative: neither base nor index.
if !m.HasBase && !m.HasIndex {
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
+149
View File
@@ -149,6 +149,45 @@ func TestPushPop(t *testing.T) {
checkSyntax(t, "push rbx", "PUSHQ", BX)
checkSyntax(t, "pop r12", "POPQ", Reg{idx: 12, size: 8})
checkSyntax(t, "push 0x5", "PUSHQ", Imm(5))
// The W spelling carries the 0x66 operand-size prefix, byte for byte
// with go tool asm; the L and B spellings are illegal in 64-bit mode
// there and rejected here rather than silently widened.
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"PUSHW AX", "PUSHW", []Operand{AX}, "6650"},
{"POPW AX", "POPW", []Operand{AX}, "6658"},
{"PUSHW $5", "PUSHW", []Operand{Imm(5)}, "666a05"},
{"PUSHW (AX)", "PUSHW", []Operand{Ptr(AX, 0, 2)}, "66ff30"},
{"PUSHQ AX", "PUSHQ", []Operand{AX}, "50"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"PUSHL AX", "PUSHL", []Operand{AX}},
{"PUSHL R8", "PUSHL", []Operand{Reg{idx: 8, size: 8}}},
{"POPL BX", "POPL", []Operand{BX}},
{"PUSHB AX", "PUSHB", []Operand{AX}},
} {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
func TestUnary(t *testing.T) {
@@ -318,6 +357,18 @@ func TestScalarGroundTruth(t *testing.T) {
{"MOVBQZX AL,R8", "MOVBQZX", []Operand{AL, r8}, "4c0fb6c0", "MOVZX"},
{"MOVWLZX AX,CX", "MOVWLZX", []Operand{AX, CX}, "0fb7c8", "MOVZX"},
{"MOVWQZX AX,R8", "MOVWQZX", []Operand{AX, r8}, "4c0fb7c0", "MOVZX"},
// The width pairs the toolchain accepts and GOROOT uses; bytes
// pinned from go tool asm (see testdata/verify/widen_amd64.s).
{"MOVBWZX (BX),R11W", "MOVBWZX", []Operand{Ptr(BX, 0, 1), Reg{idx: 11, size: 2}}, "66440fb61b", "MOVZX"},
{"MOVBWSX (BX),R11W", "MOVBWSX", []Operand{Ptr(BX, 0, 1), Reg{idx: 11, size: 2}}, "66440fbe1b", "MOVSX"},
{"MOVBLSX (BX),AX", "MOVBLSX", []Operand{Ptr(BX, 0, 1), AX}, "0fbe03", "MOVSX"},
{"MOVBQSX (BX),R8", "MOVBQSX", []Operand{Ptr(BX, 0, 1), r8}, "4c0fbe03", "MOVSX"},
{"MOVWQSX (BX),R9", "MOVWQSX", []Operand{Ptr(BX, 0, 2), r9}, "4c0fbf0b", "MOVSX"},
// A long to quad zero-extend is a plain 32-bit move.
{"MOVLQZX (BX),DX", "MOVLQZX", []Operand{Ptr(BX, 0, 4), DX}, "8b13", "MOV"},
{"MOVLQZX AX,DX", "MOVLQZX", []Operand{AX, DX}, "8bd0", "MOV"},
{"PMOVMSKB X1,AX", "PMOVMSKB", []Operand{vreg(t, "X1"), AX}, "660fd7c1", "PMOVMSKB"},
{"PMOVMSKB X11,CX", "PMOVMSKB", []Operand{vreg(t, "X11"), CX}, "66410fd7cb", "PMOVMSKB"},
{"CVTSL2SD R8,X13", "CVTSL2SD", []Operand{r8, vreg(t, "X13")}, "f2450f2ae8", "CVTSI2SD"},
{"CVTSL2SD AX,X0", "CVTSL2SD", []Operand{AX, vreg(t, "X0")}, "f20f2ac0", "CVTSI2SD"},
{"CVTSQ2SD R8,X13", "CVTSQ2SD", []Operand{r8, vreg(t, "X13")}, "f24d0f2ae8", "CVTSI2SD"},
@@ -397,6 +448,104 @@ func TestScalarErrors(t *testing.T) {
}
}
// TestImmediateOutOfRange pins the go-tool-asm parity of the immediate and
// displacement spans: a scalar immediate must fit a signed or unsigned 32-bit
// word (only MOVQ reg, $imm takes the full int64), a scalar shift count must
// be an unsigned byte, and a displacement must fit int32. Every rejected
// shape here is rejected by `go tool asm` too; every accepted one encodes the
// same bytes.
func TestImmediateOutOfRange(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
}{
{"SHLQ count 300", "SHLQ", []Operand{Imm(300), AX}},
{"SHLQ count -1", "SHLQ", []Operand{Imm(-1), AX}},
{"SHLW count 256", "SHLW", []Operand{Imm(256), DX}},
{"SHLB count 300", "SHLB", []Operand{Imm(300), BL}},
{"MOVL imm32+", "MOVL", []Operand{Imm(4294967296), AX}},
{"MOVL imm32-", "MOVL", []Operand{Imm(-2147483649), AX}},
{"MOVW imm32+", "MOVW", []Operand{Imm(4294967296), AX}},
{"MOVB imm32+", "MOVB", []Operand{Imm(4294967296), AL}},
{"ADDB imm32+", "ADDB", []Operand{Imm(4294967296), AL}},
{"ADDL imm32+", "ADDL", []Operand{Imm(4294967296), AX}},
{"ADDQ imm32+", "ADDQ", []Operand{Imm(8589934592), AX}},
{"CMPQ imm32+", "CMPQ", []Operand{AX, Imm(4294967296)}},
{"CMPQ imm32-", "CMPQ", []Operand{AX, Imm(-2147483649)}},
{"TESTL imm32+", "TESTL", []Operand{Imm(4294967296), AX}},
{"IMUL3L imm32+", "IMUL3L", []Operand{Imm(4294967296), CX, DX}},
{"PUSHQ imm32+", "PUSHQ", []Operand{Imm(4294967296)}},
{"MOVQ mem imm32+", "MOVQ", []Operand{Imm(4294967296), Ptr(AX, 0, 8)}},
{"disp32+", "MOVQ", []Operand{Ptr(AX, 4294967296, 8), BX}},
{"disp32+ max", "MOVQ", []Operand{Ptr(AX, 2147483648, 8), BX}},
{"disp32-", "MOVQ", []Operand{Ptr(AX, -2147483649, 8), BX}},
{"VEX disp32+", "VMOVDQU", []Operand{Ptr(AX, 4294967296, 32), vreg(t, "Y1")}},
{"EVEX disp32+", "VMOVDQU32", []Operand{Ptr(AX, 4294967296, 64), vreg(t, "Z1")}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
// TestImmediateTruncation pins the toolchain-matching truncations inside the
// accepted 32-bit span: the narrower fields take the low bits silently, byte
// for byte with `go tool asm` (which rejects none of these).
func TestImmediateTruncation(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"ADDB $256,BL", "ADDB", []Operand{Imm(256), BL}, "80c300"},
{"ADDB $1000,BL", "ADDB", []Operand{Imm(1000), BL}, "80c3e8"},
{"MOVB $256,AL", "MOVB", []Operand{Imm(256), AL}, "b000"},
{"MOVB $-129,AL", "MOVB", []Operand{Imm(-129), AL}, "b07f"},
{"MOVW $65536,AX", "MOVW", []Operand{Imm(65536), AX}, "66b80000"},
{"MOVW $65535,AX", "MOVW", []Operand{Imm(65535), AX}, "66b8ffff"},
{"MOVW $-32769,AX", "MOVW", []Operand{Imm(-32769), AX}, "66b8ff7f"},
{"MOVL $4294967295,AX", "MOVL", []Operand{Imm(4294967295), AX}, "b8ffffffff"},
{"ADDQ $4294967295,AX", "ADDQ", []Operand{Imm(4294967295), AX}, "4805ffffffff"},
{"CMPB BL,$255", "CMPB", []Operand{BL, Imm(255)}, "80fbff"},
{"CMPQ AX,$4294967295", "CMPQ", []Operand{AX, Imm(4294967295)}, "483dffffffff"},
{"MOVQ $4294967295,0(AX)", "MOVQ", []Operand{Imm(4294967295), Ptr(AX, 0, 8)}, "48c700ffffffff"},
{"SHLQ $255,AX", "SHLQ", []Operand{Imm(255), AX}, "48c1e0ff"},
{"SHLQ $0,AX", "SHLQ", []Operand{Imm(0), AX}, "48c1e000"},
// The one form beyond the 32-bit span: the imm64 MOVQ register move.
{"MOVQ $4294967296,AX", "MOVQ", []Operand{Imm(4294967296), AX}, "48b80000000001000000"},
{"MOVQ disp32 max", "MOVQ", []Operand{Ptr(AX, 2147483647, 8), BX}, "488b98ffffff7f"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
}
// TestEncodableCmovSize pins the linter contract for CMOVcc: Encodable must
// reject the spellings Encode rejects, so a mnemonic like CMOVBGT (no size
// letter) is not reported as encodable.
func TestEncodableCmovSize(t *testing.T) {
for _, m := range []string{"CMOVBGT", "CMOVXEQ", "CMOVB", "CMOV", "CMOVWXX"} {
if Encodable(m) {
t.Errorf("Encodable(%q) = true, want false", m)
}
}
for _, m := range []string{"CMOVLGT", "CMOVQGT", "CMOVWLS", "CMOVLEQ"} {
if !Encodable(m) {
t.Errorf("Encodable(%q) = false, want true", m)
}
}
}
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
// byte for byte (no prefix / 66 / F2 / F3 variants).
func TestSSEBinGroundTruth(t *testing.T) {
+37 -20
View File
@@ -509,40 +509,43 @@ var evexBcastTable = map[string]evexBcastSpec{
}
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
// move table).
// move table). vecOK and xmmOnly mirror the VEX twin's operand rules: a
// scalar move (vecOK false, xmmOnly true) takes XMM↔memory operands only.
type evexMoveSpec struct {
mapSel int
pp int
load byte // r/m → vector
store byte // vector → r/m
w int
n [3]int
mapSel int
pp int
load byte // r/m → vector
store byte // vector → r/m
w int
n [3]int
vecOK bool // the non-memory operand may be a vector register
xmmOnly bool // wider than XMM registers are rejected
}
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}},
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}},
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true},
}
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
@@ -1022,6 +1025,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
var rm Operand
switch {
case srcIsVec && dstIsVec:
// A store-form reg-reg move, the layout the Go assembler uses; a
// scalar move has no two-register form at all (the register form
// takes three operands), matching the VEX twin's vecOK rule.
if !ms.vecOK {
return fmt.Errorf("%s does not take two vector registers", mnem)
}
reg, rm = srcReg, dst
case srcIsVec:
if !memOperand(dst) {
@@ -1037,6 +1046,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
default:
return fmt.Errorf("%s needs a vector register operand", mnem)
}
// The scalar move is 128-bit only, so the register the length follows
// must be an XMM (the VEX twin's xmmOnly rule; EVEX also reaches ZMM,
// hence the inequality rather than a YMM test).
if ms.xmmOnly && reg.size != 16 {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx)
}
@@ -1171,9 +1186,6 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
if r.idx&16 != 0 {
xBar = 0
}
if r.idx&16 != 0 {
xBar = 0
}
case Mem:
var err error
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
@@ -1232,6 +1244,11 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
sib = -1
xBar, bBar = 1, 1 // inverted bits: 1 = no extension
// The disp32 fallback bounds the displacement by int32, and the
// compressed disp8 form reaches at most ±127×64, well inside it.
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
}
if !m.HasBase && !m.HasIndex {
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
}
+9
View File
@@ -675,6 +675,15 @@ func TestEvexErrors(t *testing.T) {
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
// VEX-only mnemonics reject registers only EVEX can encode.
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
// The scalar EVEX move matches its VEX twin and the Go assembler:
// XMM↔memory only, never reg-reg and never a wider register (the
// toolchain rejects every one of these shapes).
{"VMOVSS X1,X2", "VMOVSS", []Operand{vreg(t, "X1"), vreg(t, "X2")}},
{"VMOVSS X16,X2", "VMOVSS", []Operand{vreg(t, "X16"), vreg(t, "X2")}},
{"VMOVSS Y1,(AX)", "VMOVSS", []Operand{vreg(t, "Y1"), Ptr(AX, 0, 4)}},
{"VMOVSS Z1,Z2", "VMOVSS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}},
{"VMOVSS Z1,(AX)", "VMOVSS", []Operand{vreg(t, "Z1"), Ptr(AX, 0, 4)}},
{"VMOVSS (AX),Z2", "VMOVSS", []Operand{Ptr(AX, 0, 4), vreg(t, "Z2")}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
+5 -4
View File
@@ -68,11 +68,12 @@ const (
kindSDWARFLINES = 20
)
// Symbol flags (cmd/internal/goobj).
// Symbol flags (cmd/internal/goobj). The linkname flag is set only for
// //go:linkname symbols (and main.main); ordinary assembly symbols carry
// none, matching cmd/asm's output.
const (
symFlagDupok = 0x01
symFlagNoSplit = 0x10
symFlag2Link = 0x10 // asm objects flag every named symbol as linkname
symABIStatic = 0xffff
)
@@ -287,7 +288,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
}
}
nps = append(nps, npSym{
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, size: uint32(fn.Size)},
data: code,
})
}
@@ -317,7 +318,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
abi = symABIStatic
}
defIdx[d.Name] = len(defs)
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)})
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, size: uint32(d.Size)})
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
}
fnFiIdx := make([]int, len(img.Funcs))
+72 -32
View File
@@ -40,8 +40,11 @@ func exportPath(importPath string) (string, error) {
//
// refs maps package import paths to the symbol names referenced from that
// package. The returned pkgIdx maps each import path to its position in
// the blkPkgIdx table (0-based), and symIdx gives each symbol's index within
// its package.
// the blkPkgIdx table, which reserves index 0 for the dummy invalid
// package (cmd/internal/obj/sym.go: "0 is invalid index"; the loader's
// reader loop starts at 1), so package i sits at block index i+1 and its
// relocations carry i+1. symIdx gives each symbol's index within its
// package.
func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symIdx map[string]int, err error) {
pkgIdx = make(map[string]int, len(refs))
symIdx = make(map[string]int)
@@ -50,7 +53,9 @@ func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symI
packages := sortedPkgRefs(refs)
for i, pkg := range packages {
pkgIdx[pkg.path] = i
// Block index 0 is the dummy invalid package; the first real
// package starts at 1.
pkgIdx[pkg.path] = i + 1
exp, err := exportPath(pkg.path)
if err != nil {
return nil, nil, err
@@ -145,40 +150,65 @@ func parseArDecimal(b []byte) int {
}
// goobjFile is a parsed GOOBJ file: the string table and the symbol-definition
// block.
// blocks. The hashed blocks are kept raw: their symbols carry no names, only
// the loader needs their counts.
type goobjFile struct {
strTab []byte // string table, at headerSize + n
symdef []byte // blkSymdef raw block
npdef []byte // blkNonpkgdef raw block
strTab []byte // string table, at headerSize + n
symdef []byte // blkSymdef raw block
hashed64 []byte // blkHashed64def raw block
hashed []byte // blkHasheddef raw block
npdef []byte // blkNonpkgdef raw block
}
// symbols returns all symbol names in definition order by scanning the
// symdef and nonpkgdef blocks and resolving each name through the string
// table. Package definitions (blkSymdef) use fully-qualified names like
// "runtime.morestack"; non-package definitions (blkNonpkgdef) use bare
// names like "morestack". This combined list matches the index the
// linker expects for cross-package references.
// loaderIndexBase returns the index the first nonpkgdef symbol occupies in the
// loader's per-object symbol array. cmd/link lays the definition blocks out as
// symdef, hashed64def, hasheddef, nonpkgdef, nonpkgref (loader.go: preloadSyms
// fills r.syms in exactly that order, and resolve() indexes PkgIdxNone and
// cross-package SymIdx into it), so a symbol found in blkNonpkgdef carries the
// three leading blocks' symbol counts as its base.
func (f *goobjFile) loaderIndexBase() int {
return len(f.symdef)/recSymSize + len(f.hashed64)/recSymSize + len(f.hashed)/recSymSize
}
// symbols returns the names of the symdef and nonpkgdef blocks in
// definition order. Package definitions (blkSymdef) use fully-qualified
// names like "runtime.morestack"; non-package definitions (blkNonpkgdef)
// use bare names like "morestack". For lookups by index prefer
// findSymbol: it adds the hashed blocks' count the loader's array
// interleaves between the two.
func (f *goobjFile) symbols() []string {
return append(f.defNames(), f.npdefNames()...)
}
// findSymbol returns the index of a symbol within the combined symbol list,
// or -1 if not found. It first tries the fully-qualified name (pkg.name),
// then the bare name.
// findSymbol returns the index of a symbol within the loader's per-object
// symbol array, or -1 if not found. It first tries the fully-qualified
// name (pkg.name), then the bare name (assembly objects store dotless
// names, e.g. runtime's "gogo", for symbols other packages reach through
// a linkname).
func (f *goobjFile) findSymbol(pkg, name string) int {
base := f.loaderIndexBase()
qualified := pkg + "." + name
syms := f.symbols()
for i, s := range syms {
for i, s := range f.defNames() {
if s == qualified {
return i
}
}
// Try bare name (for non-package definitions).
for i, s := range syms {
for i, s := range f.npdefNames() {
if s == qualified {
return base + i
}
}
// Try bare name (for dotless assembly definitions).
for i, s := range f.defNames() {
if s == name {
return i
}
}
for i, s := range f.npdefNames() {
if s == name {
return base + i
}
}
return -1
}
@@ -192,12 +222,16 @@ func (f *goobjFile) npdefNames() []string {
return f.readSymNames(f.npdef)
}
// recSymSize is the size of one Sym record in the definition blocks
// (goobj.SymSize: stringRefSize + 2 + 1 + 1 + 1 + 4 + 4).
const recSymSize = 21
// readSymNames reads symbol names from a symdef/nonpkgdef block. Each record
// is 21 bytes: nameLen (u32), nameOff (u32), abi (u16), typ, flag, flag2,
// size (u32), align (u32). nameOff is an absolute offset into the string
// table.
func (f *goobjFile) readSymNames(block []byte) []string {
const recSize = 21
const recSize = recSymSize
if len(block) < recSize {
return nil
}
@@ -247,16 +281,18 @@ func parseGOOBJ(data []byte) (*goobjFile, error) {
// [16:20] flags
// [20:96] 19 × uint32 offsets
var offs [blkEnd + 1]uint32
for i := 0; i <= blkEnd; i++ {
for i := range blkEnd + 1 {
offs[i] = binary.LittleEndian.Uint32(payload[20+4*i:])
}
// The string table lives at headerSize.
strTabStart := uint32(goobjHeaderSize)
f := &goobjFile{
strTab: payload[strTabStart:offs[0]],
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
strTab: payload[strTabStart:offs[0]],
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
hashed64: blockSlice(payload, offs, blkHashed64def, blkHashed64def+1),
hashed: blockSlice(payload, offs, blkHasheddef, blkHasheddef+1),
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
}
return f, nil
}
@@ -306,22 +342,26 @@ func resolveExternalSymbols(externals []string) (pkgTable []string, pkgIdxMap ma
return nil, nil, nil, err
}
// Build the package table in pkgIdx order.
// Build the package table in pkgIdx order. The indices are 1-based
// (0 is the dummy invalid package, written by the emitter itself), so
// the table without the dummy is indexed one below.
pkgTable = make([]string, len(pkgIdx1))
for pkg, idx := range pkgIdx1 {
pkgTable[idx] = pkg
pkgTable[idx-1] = pkg
}
return pkgTable, pkgIdx1, symIdx1, nil
}
// splitQualified splits a qualified Go symbol name (pkgpath·name) into its
// package path and local name. The separator is the middle dot (U+00B7).
// If no separator is found, the symbol is assumed to be in the current
// package (empty pkg).
// package path and local name. The separator is the middle dot (U+00B7),
// whose UTF-8 encoding is two bytes, so the search must be string-based:
// IndexByte would match only the second byte and leave the lead byte on
// the package path. If no separator is found, the symbol is assumed to be
// in the current package (empty pkg).
func splitQualified(full string) (pkg, name string) {
if idx := strings.IndexByte(full, '\u00b7'); idx >= 0 {
return full[:idx], full[idx+len("\u00b7"):]
if before, after, ok := strings.Cut(full, "\u00b7"); ok {
return before, after
}
if before, after, ok := strings.Cut(full, "."); ok {
return before, after
+6 -2
View File
@@ -56,8 +56,12 @@ func TestResolveExternalSymbols(t *testing.T) {
if err != nil {
t.Fatalf("resolveExternalGOOBJ: %v", err)
}
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 0 {
t.Errorf("pkgIdx = %v, want runtime→0", pkgIdx)
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 1 {
// Index 0 is the dummy invalid package in the blkPkgIdx table;
// the loader's reader loop starts at 1 (cmd/link/internal/
// loader/loader.go: "PkgIdx 0 is a dummy invalid package"), so
// the first real package must carry index 1.
t.Errorf("pkgIdx = %v, want runtime→1", pkgIdx)
}
if _, ok := symIdx["runtime·g0"]; !ok {
t.Errorf("symIdx missing runtime·g0, got %v", symIdx)
+188 -1
View File
@@ -117,7 +117,9 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
if len(defs) != 7 {
t.Fatalf("symdefs = %d, want 7", len(defs))
}
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
// The linkname flag stays clear: the toolchain sets it only for
// //go:linkname symbols, and an ordinary static GLOBL is not one.
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != 0 {
t.Errorf("mask symbol = %+v", defs[0])
}
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
@@ -511,3 +513,188 @@ func fieldAfter(line, flag string) string {
}
return ""
}
// TestGOObjectExternalPackageLink is the cross-package end-to-end check: a
// GOOBJ whose code references a real external package symbol (runtime's
// morestack, a plain reference rather than the builtin noctxt form) must
// carry a package index that points past the blkPkgIdx table's dummy entry
// 0, and the object must link against the real runtime. Pre-fix, the
// relocations carried block index 0, which the loader never fills, so the
// reference resolved against whatever object was loaded first and the link
// failed. The binary is not run: morestack returns to the call site's
// stack check, which a hand-written caller has none of.
func TestGOObjectExternalPackageLink(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
const asmSrc = `
#include "textflag.h"
TEXT ·fn(SB), NOSPLIT, $0-0
CALL ·helper(SB)
RET
TEXT ·helper(SB), NOSPLIT, $0-0
RET
`
const mainSrc = `package main
func fn()
func helper()
func main() {
fn()
helper()
}
`
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module extlink\n\ngo 1.27\n"), 0o644); err != nil {
t.Fatal(err)
}
// Capture the build the toolchain performs and re-run only its link
// step with our object swapped into the package archive, mirroring
// TestGOObjectLinkAndRun.
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj, pkgArch string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || pkgArch == "" || linkLine == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
}
defer os.RemoveAll(work)
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
// Assemble the source with gasm, then retarget fn's internal call at
// a real external package symbol: the reloc's qualified name drives
// the export-data resolution the way a source-level runtime·sym(SB)
// reference would.
f, errs := parser.Parse("main_amd64.s", asmSrc)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
fn := &img.Funcs[0]
for i := range fn.Relocs {
fn.Relocs[i].Name = "runtime\u00b7morestack"
fn.Relocs[i].External = true
}
img.Externals = []string{"runtime\u00b7morestack"}
obj, err := img.GOObject("main", "main_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
// Structural check: the blkPkgIdx block reserves entry 0 for the
// dummy invalid package and places runtime at entry 1, and fn's call
// relocation carries PkgIdx 1.
v := openGoobj(t, obj)
pkgBlk := v.blk(blkPkgIdx)
if len(pkgBlk) != 2*8 {
t.Fatalf("blkPkgIdx = %d bytes, want two entries", len(pkgBlk))
}
le := binary.LittleEndian
strEntry := func(i int) string {
e := pkgBlk[i*8 : (i+1)*8]
return v.str(le.Uint32(e[4:]), le.Uint32(e[0:]))
}
if s := strEntry(0); s != "" {
t.Errorf("blkPkgIdx[0] = %q, want the dummy empty package", s)
}
if s := strEntry(1); s != "runtime" {
t.Errorf("blkPkgIdx[1] = %q, want runtime", s)
}
relocs := v.blk(blkReloc)
// fn is the last non-package symbol (two functions, four pc tables
// each); its one reloc is the final record.
fnRec := relocs[len(relocs)-23:]
if pIdx := le.Uint32(fnRec[15:]); pIdx != 1 {
t.Errorf("external reloc PkgIdx = %d, want 1 (runtime)", pIdx)
}
// Swap the object into the package archive and link with cmd/link;
// the link line consumes the archive, not the loose object file.
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
extract.Dir = membersDir
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
member := filepath.Join(membersDir, filepath.Base(asmObj))
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, obj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for m := range strings.FieldsSeq(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, filepath.Join(membersDir, m))
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, pkgArch, newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "prog2"))
linkCmd := exec.Command("sh", "-c", "cd "+dir+" && "+linkLine)
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
// The call must have resolved to the real runtime symbol.
dump, err := exec.Command(goBin, "tool", "objdump", "-s", "main.fn", filepath.Join(dir, "prog2")).CombinedOutput()
if err != nil {
t.Fatalf("objdump main.fn: %v\n%s", err, dump)
}
if !bytes.Contains(dump, []byte("runtime.morestack")) {
t.Errorf("main.fn does not call runtime.morestack:\n%s", dump)
}
}
+24 -3
View File
@@ -18,19 +18,40 @@ import (
// reloc/aux/data index arrays, with the arm64 preamble, the MinLC of 4
// for the pc-value deltas, and the arm64 relocation types for the ADRP
// pairs and BL calls.
//
// The toolchain records one relocation per ADRP pair: a single R_ADDRARM64
// or R_ARM64_PCREL_LDST64 of Siz 8 at the ADRP word, from which the linker
// patches both instructions of the pair (cmd/internal/obj/arm64/asm7.go,
// the ADRP cases: one AddRel with Off at the pair's pc and Siz 8). gasm's
// assembler records the ADRP+ADD form as two word relocs, so the second
// word's twin is dropped here before emission.
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreambleAARCH64()
if err != nil {
return nil, err
}
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
coalesced := *img
coalesced.Funcs = append([]FuncLayout(nil), img.Funcs...)
for i := range coalesced.Funcs {
rs := coalesced.Funcs[i].Relocs
var keep []Reloc
for j := 0; j < len(rs); j++ {
keep = append(keep, rs[j])
if rs[j].Kind == RelArm64Addr && j+1 < len(rs) &&
rs[j+1].Kind == RelArm64Addr && rs[j+1].Off == rs[j].Off+4 {
j++ // the ADD word's twin: the Siz-8 pair reloc covers it
}
}
coalesced.Funcs[i].Relocs = keep
}
return coalesced.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
switch r.Kind {
case RelArm64Branch:
return relocArm64Branch, 4
case RelArm64LDST64:
return relocArm64LDST64, 4
return relocArm64LDST64, 8
default:
return relocArm64Addr, 4
return relocArm64Addr, 8
}
})
}
+46
View File
@@ -6,6 +6,7 @@ package asm
import (
"bytes"
"encoding/hex"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -27,6 +28,12 @@ func TestStackGuardBytes(t *testing.T) {
"644c8b3425000000004c8da42478ffffff4d3b66107614554889e54881ec000100004881c4000100005dc3e800000000ebce"},
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"644c8b3425000000004989e44981ec881f0000721a4d3b66107614554889e54881ec002000004881c4002000005dc3e800000000ebca"},
// Class 2 with a body long enough that the underflow JB relaxes to
// rel32: its displacement must span the real 6-byte JB, else the
// branch lands 4 bytes past the morestack block, inside the CALL
// displacement field.
{"leafbiglong", "TEXT \u00b7leafbiglong(SB), $8192-0\n" + strings.Repeat("\tMOVQ AX, BX\n", 40) + "\tRET\n",
"644c8b3425000000004989e44981ec881f00000f82960000004d3b66100f868c000000554889e54881ec00200000" + strings.Repeat("4889c3", 40) + "4881c4002000005dc3e800000000e947ffffff"},
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"644c8b342500000000493b66107613554889e54883ec10e8000000004883c4105dc3e800000000ebd7"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
@@ -145,6 +152,45 @@ func TestStackGuardBytesARM64(t *testing.T) {
}
}
// TestStackGuardBranchTargetsARM64 checks the class-2 guard's branch
// positions for a frame whose guard constant needs two MOV words: the
// displacements must be computed from byte offsets (8+4*ml and 16+4*ml), so
// both branches land on the morestack block rather than inside the body.
// The frame size makes the toolchain switch its own prologue decomposition,
// so the assertion is on the branch targets, not pinned bytes.
func TestStackGuardBranchTargetsARM64(t *testing.T) {
f, errs := parser.Parse("g_arm64.s", "TEXT \u00b7f(SB), $65664-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
fn := img.Funcs[0]
code := img.Code[fn.Offset : fn.Offset+fn.Size]
if len(code)%4 != 0 {
t.Fatalf("function size %d is not a word multiple", len(code))
}
// autosize = 65680, so the guard materialises 65552 = MOVZ+MOVK: ml = 2
// and the branches sit at bytes 16 and 24 of the guard prefix.
const morestackBlock = 12 // MOVD R30, R3; BL; B back
blockStart := len(code) - morestackBlock
check := func(name string, off int) {
t.Helper()
w := leWord(code[off:])
imm19 := int32(w>>5) & 0x7FFFF
if imm19&(1<<18) != 0 {
imm19 -= 1 << 19
}
if target := off + int(imm19)*4; target != blockStart {
t.Errorf("%s at byte %d targets byte %d, want the morestack block at %d", name, off, target, blockStart)
}
}
check("B.LO", 16)
check("B.LS", 24)
}
// The riscv64 stack-split guard, pinned from `go tool asm` (Go 1.27,
// riscv64): the morestack call sits between the guard and the body, and the
// guard branches forward over it. Relocation fields are masked.
+126 -34
View File
@@ -173,7 +173,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
if dstReg.needsREX(size) {
i.rexForced = true
}
i.imm = immediate(v, size, true)
imm, err := immediate(v, size, true)
if err != nil {
return err
}
i.imm = imm
return e.emit(i)
}
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
@@ -185,7 +189,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
if err := setRMDigit(i, 0, dst, size); err != nil {
return err
}
i.imm = immediate(int64(src), size, false)
imm, err := immediate(int64(src), size, false)
if err != nil {
return err
}
i.imm = imm
return e.emit(i)
}
return fmt.Errorf("MOV: invalid operands")
@@ -297,11 +305,15 @@ func (e *enc) encodeALU(op struct {
func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
if size == 1 {
immBytes, err := immediate(imm, 1, false)
if err != nil {
return err
}
i := newInstr(1, []byte{0x80})
if err := setRMDigit(i, digit, dst, 1); err != nil {
return err
}
i.imm = []byte{byte(int8(imm))}
i.imm = immBytes
return e.emit(i)
}
if fits8(imm) {
@@ -319,7 +331,11 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
if r, ok := dst.(Reg); ok && r.idx == 0 {
accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit]
i := newInstr(size, []byte{accOp})
i.imm = immediate(imm, size, false)
immBytes, err := immediate(imm, size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
// 0x81 /digit, imm16/imm32.
@@ -327,7 +343,11 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
if err := setRMDigit(i, digit, dst, size); err != nil {
return err
}
i.imm = immediate(imm, size, false)
immBytes, err := immediate(imm, size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
@@ -348,7 +368,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
op = 0xA8
}
i := newInstr(size, []byte{op})
i.imm = immediate(int64(imm), size, false)
immBytes, err := immediate(int64(imm), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
op := byte(0xF7)
@@ -359,7 +383,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
if err := setRMDigit(i, 0, dst, size); err != nil {
return err
}
i.imm = immediate(int64(imm), size, false)
immBytes, err := immediate(int64(imm), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
srcReg, ok := src.(Reg)
@@ -457,7 +485,13 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
}
return e.emit(i)
}
// 0xC0 (8-bit) / 0xC1, imm8.
// 0xC0 (8-bit) / 0xC1, imm8. The count is an unsigned byte: go tool asm
// rejects negative and ≥256 counts, and the hardware masks the count, so
// a silent truncation ($300 encoding 44) would shift by a different
// amount than the source states.
if imm < 0 || imm > 255 {
return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm))
}
op := byte(0xC1)
if size == 1 {
op = 0xC0
@@ -466,7 +500,7 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
if err := setRMDigit(i, digit, dst, size); err != nil {
return err
}
i.imm = []byte{byte(int8(imm))}
i.imm = []byte{byte(imm)}
return e.emit(i)
}
@@ -508,7 +542,11 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
if err := setRM(i, dstReg, ops[1], size); err != nil {
return err
}
i.imm = immediate(int64(imm), size, false)
immBytes, err := immediate(int64(imm), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
@@ -516,10 +554,21 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
// --- PUSH / POP -------------------------------------------------------------
func (e *enc) encodePushPop(ops []Operand, push bool) error {
func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
if len(ops) != 1 {
return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops))
}
// In 64-bit mode go tool asm knows the 64-bit push (the default, with or
// without the Q suffix) and the 16-bit W form with its 0x66 operand-size
// prefix, and rejects the B and L spellings outright ("illegal in 64-bit
// mode"); silently widening those would push a different width than the
// source states.
switch size {
case 0, 8, 2:
default:
return fmt.Errorf("PUSH/POP size suffix is illegal in 64-bit mode")
}
w16 := size == 2
switch op := ops[0].(type) {
case Reg:
base := byte(0x50) // PUSH r; POP is 0x58
@@ -527,7 +576,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
base = 0x58
}
// PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed.
i := &instr{opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
i := &instr{opSize16: w16, opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
i.rexB = op.idx >= 8
return e.emit(i)
case Mem:
@@ -537,7 +586,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
opc = 0x8F // POP r/m: /0
digit = 0
}
i := &instr{opcode: []byte{opc}, modrm: -1, sib: -1}
i := &instr{opSize16: w16, opcode: []byte{opc}, modrm: -1, sib: -1}
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
return err
}
@@ -547,10 +596,17 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
return fmt.Errorf("POP does not take an immediate")
}
if fits8(int64(op)) {
i := &instr{opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
i := &instr{opSize16: w16, opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
return e.emit(i)
}
i := &instr{opSize16: false, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: le32(int64(op))}
// PUSH imm32, sign-extended to 64 bits; go tool asm bounds the
// immediate by the same signed/unsigned 32-bit span as every other
// scalar immediate.
immBytes, err := immediate(int64(op), 8, false)
if err != nil {
return err
}
i := &instr{opSize16: w16, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: immBytes}
return e.emit(i)
}
return fmt.Errorf("PUSH/POP: invalid operand")
@@ -639,19 +695,28 @@ func (e *enc) encodeJcc(cc int, ops []Operand) error {
// immediate encodes an immediate of the given operand size. full64 selects the
// 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit
// sign-extended immediate is used for 64-bit operands.
func immediate(v int64, size int, full64 bool) []byte {
//
// The span mirrors go tool asm: every scalar immediate must fit a signed or
// unsigned 32-bit word, and the narrower fields then take the low bits
// silently (ADDB $256, AL encodes imm8 0, MOVW $65536, AX imm16 0). Only the
// imm64 form may exceed the span; anything wider elsewhere is an error rather
// than a truncation the source never asked for.
func immediate(v int64, size int, full64 bool) ([]byte, error) {
if !(size == 8 && full64) && (v < -(1<<31) || v > (1<<32)-1) {
return nil, fmt.Errorf("immediate $%d does not fit in 32 bits", v)
}
switch size {
case 1:
return []byte{byte(int8(v))}
return []byte{byte(int8(v))}, nil
case 2:
return le16(v)
return le16(v), nil
case 4:
return le32(v)
return le32(v), nil
default: // 8
if full64 {
return le64(v)
return le64(v), nil
}
return le32(v) // sign-extended imm32
return le32(v), nil // sign-extended imm32
}
}
@@ -773,15 +838,24 @@ func (e *enc) encodeBswap(ops []Operand, size int) error {
// width. The source is narrower than the destination, so the plain size-suffix
// convention does not apply to these names.
var movExtendOp = map[string]struct {
op []byte
dst64 bool
op []byte
dstSize int
}{
"MOVBLZX": {[]byte{0x0F, 0xB6}, false}, // byte → long, zero-extend
"MOVBQZX": {[]byte{0x0F, 0xB6}, true}, // byte → quad, zero-extend
"MOVWLZX": {[]byte{0x0F, 0xB7}, false}, // word → long, zero-extend
"MOVWQZX": {[]byte{0x0F, 0xB7}, true}, // word → quad, zero-extend
"MOVWLSX": {[]byte{0x0F, 0xBF}, false}, // word → long, sign-extend
"MOVLQSX": {[]byte{0x63}, true}, // long → quad, sign-extend (MOVSXD)
"MOVBLZX": {[]byte{0x0F, 0xB6}, 4}, // byte → long, zero-extend
"MOVBQZX": {[]byte{0x0F, 0xB6}, 8}, // byte → quad, zero-extend
"MOVWLZX": {[]byte{0x0F, 0xB7}, 4}, // word → long, zero-extend
"MOVWQZX": {[]byte{0x0F, 0xB7}, 8}, // word → quad, zero-extend
"MOVWLSX": {[]byte{0x0F, 0xBF}, 4}, // word → long, sign-extend
"MOVLQSX": {[]byte{0x63}, 8}, // long → quad, sign-extend (MOVSXD)
"MOVBWZX": {[]byte{0x0F, 0xB6}, 2}, // byte → word, zero-extend
"MOVBWSX": {[]byte{0x0F, 0xBE}, 2}, // byte → word, sign-extend
"MOVBLSX": {[]byte{0x0F, 0xBE}, 4}, // byte → long, sign-extend
"MOVBQSX": {[]byte{0x0F, 0xBE}, 8}, // byte → quad, sign-extend
"MOVWQSX": {[]byte{0x0F, 0xBF}, 8}, // word → quad, sign-extend
// A long → quad zero-extend is a plain 32-bit move: every 32-bit
// operation zero-extends its result into the full register, so the
// toolchain lowers MOVLQZX to the plain MOVL encoding.
"MOVLQZX": {[]byte{0x8B}, 4},
}
// encodeMovExtend encodes a mixed-width extending move: reg = dst (the wider
@@ -795,12 +869,30 @@ func (e *enc) encodeMovExtend(base string, ops []Operand) error {
if !ok {
return fmt.Errorf("%s destination must be a register", base)
}
size := 4
if spec.dst64 {
size = 8
i := newInstr(spec.dstSize, spec.op)
if err := setRM(i, dstReg, ops[0], spec.dstSize); err != nil {
return err
}
i := newInstr(size, spec.op)
if err := setRM(i, dstReg, ops[0], size); err != nil {
return e.emit(i)
}
// encodePmovmskb encodes PMOVMSKB, the legacy SSE2 byte mask extract: the
// XMM source's sign bytes pack into a GP destination, 66 0F D7 /r.
func (e *enc) encodePmovmskb(base string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
}
srcReg, srcVec := vecReg(ops[0])
if !srcVec {
return fmt.Errorf("%s source must be an XMM register", base)
}
dstReg, ok := ops[1].(Reg)
if !ok {
return fmt.Errorf("%s destination must be a register", base)
}
i := newInstr(4, []byte{0x0F, 0xD7})
i.prefix = 0x66
if err := setRM(i, dstReg, srcReg, 4); err != nil {
return err
}
return e.emit(i)
+17 -10
View File
@@ -25,6 +25,10 @@ type Image struct {
Symbols map[string]int // static symbol → byte offset within the image
DataSyms []DataSymbol // GLOBL symbols, in layout order
Externals []string // referenced but undefined symbols, sorted
// SourcePath is the assembled file's path, recorded in the DWARF
// sections in place of a placeholder name. Empty when the image was
// not built from a named file.
SourcePath string
}
// FuncLayout describes one assembled function within an Image.
@@ -81,12 +85,9 @@ func (fl *FuncLayout) LineAt(offset int) int {
return 0
}
// RelocKind Reloc is one static-symbol reference within a function body: the disp32
// field at Off (function-relative) must reach the symbol plus Addend,
// measured from After, the address just past the instruction. An External
// relocation names a symbol no GLOBL in the file defines; the object-file
// emitters carry it into the output's relocation table.
// RelocKind discriminates the type of relocation needed.
// RelocKind discriminates the relocation a static-symbol reference needs;
// the encoders record one per SB reference, and the object-file emitters map
// it to their format's relocation type.
type RelocKind int
const (
@@ -96,7 +97,6 @@ const (
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
RelRISCVJal // R_RISCV_JAL (J-type call)
RelPCRelAbs // 32-bit absolute (R_RISCV_32)
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
RelArm64Addr // R_ADDRARM64 (ADRP + ADD pair)
@@ -106,6 +106,13 @@ const (
)
type Reloc struct {
// Off is the function-relative offset of the field the linker patches
// and After the address just past the instruction, the base the
// assembler measures PC-relative displacements from. Name plus
// Addend select the target: the symbol plus the byte offset. An
// External relocation names a symbol no GLOBL in the file defines;
// the object-file emitters carry it into the output's relocation
// table.
Off int
After int
Name string
@@ -150,7 +157,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
}
link := &linkInfo{symbols: known, allowExternal: true}
img := &Image{Symbols: map[string]int{}}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
textOff := map[string]int{}
type asmFunc struct {
name string
@@ -267,7 +274,7 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
return nil, err
}
img := &Image{Symbols: map[string]int{}}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
@@ -339,7 +346,7 @@ func AssembleFileLOONG64(f *ast.File) (*Image, error) {
return nil, err
}
img := &Image{Symbols: map[string]int{}}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
+44 -17
View File
@@ -442,6 +442,15 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
// The toolchain validates the bit numbers ("illegal bit number"):
// 0..31 for the .w forms, 0..63 for the .d forms, lsb <= msb.
b := 64
if strings.HasSuffix(mnem, "W") {
b = 32
}
if msb < 0 || msb >= b || lsb < 0 || lsb >= b || lsb > msb {
return nil, fmt.Errorf("%s: illegal bit number (msb %d, lsb %d)", mnem, msb, lsb)
}
return l64wordLE(l64irir(enc.op, msb, rj, lsb, rd)), nil
case l64Firrr:
@@ -618,6 +627,15 @@ func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, o
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
if mnem == "BLTU" || mnem == "BGEU" {
// The unsigned compares have no single-register pseudo: the
// toolchain keeps the register-register form with rd = R0
// (bltu rj, r0 is never taken), not a sometimes-taken beqz.
if (v<<16)>>16 != v {
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
}
return l64wordLE(l64irr16(op, v, rj, 0)), nil
}
if (v<<11)>>11 != v {
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
}
@@ -857,9 +875,16 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs
if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
}
// MOVF/MOVD $imm, Fd → materialise in R30, then movgr2fr.{w,d}.
if (mnem == "MOVF" || mnem == "MOVD") && loong64RegClass(operandRegName(dst)) == l64ClsFP {
return encodeLOONG64ImmToFp(rd, l64Imm64(src), mnem), nil
// MOVW $imm, Fd is the only immediate-to-F form the toolchain's optab
// accepts (AMOVW's C_12CON against C_FREG): it materialises the
// constant in R30 and moves it across with movgr2fr.w. MOVV/MOVF/
// MOVD are illegal combinations there, and are diagnosed here rather
// than silently written into the GPR of the register's number.
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
if mnem != "MOVW" {
return nil, fmt.Errorf("%s $imm: illegal combination with an F register destination (only MOVW $c, Fd is supported)", mnem)
}
return encodeLOONG64ImmToFp(rd, l64Imm64(src))
}
return encodeLOONG64LoadImm(rd, l64Imm64(src), mnem), nil
}
@@ -940,8 +965,8 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
return 8 // pcalau12i + addi.d
}
if (mnem == "MOVF" || mnem == "MOVD") && loong64RegClass(operandRegName(dst)) == l64ClsFP {
return 8 // addi/ori r30 + movgr2fr
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid)
}
v := l64Imm64(src)
if v == 0 {
@@ -982,22 +1007,24 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
}
}
// encodeLOONG64ImmToFp materialises a 12-bit immediate in R30 and moves it to
// an F register (the toolchain's case 34: movgr2fr.w/movgr2fr.d).
func encodeLOONG64ImmToFp(fd int, v int64, mnem string) []byte {
// ori for positive constants, addi.d for zero/negative.
op := uint32(0x00b << 22)
if v > 0 {
op = 0x00e << 22
// encodeLOONG64ImmToFp materialises a 12-bit immediate in R30 and moves it
// to an F register, the toolchain's expansion of MOVW $c, Fd: ori (which
// zero-extends) for the positive span, addi.w for zero and the negative
// span, then movgr2fr.w. The toolchain's optab accepts no wider constant on
// this path (it never materialises one fully first), so values outside
// [-2048, 4095] are diagnosed rather than masked into si12.
func encodeLOONG64ImmToFp(fd int, v int64) ([]byte, error) {
if v < -2048 || v > 4095 {
return nil, fmt.Errorf("MOVW $%d: immediate out of the [-2048, 4095] range for an F register destination", v)
}
mov := uint32(0x452a << 10) // movgr2fr.d
if mnem == "MOVF" {
mov = 0x4529 << 10 // movgr2fr.w
op := uint32(0x00a << 22) // addi.w r30, r0, v (sign-extends)
if v > 0 {
op = 0x00e << 22 // ori r30, r0, v (zero-extends)
}
return l64WordsLE(
l64irr(op, int(v), 0, 30),
l64rr(mov, 30, fd),
)
l64rr(0x4529<<10, 30, fd), // movgr2fr.w fd, r30
), nil
}
// ---- 64-bit immediate classification ----
+3 -1
View File
@@ -199,7 +199,9 @@ func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
}
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
// The msb/lsb fields are 6 bits wide (0-63) and are validated by the caller.
// The msb/lsb fields are 6 bits wide and are inserted unmasked: the caller
// must have validated them (0..31 for the .w forms, 0..63 for the .d forms,
// lsb <= msb), the same rule the toolchain enforces as "illegal bit number".
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
}
+4 -4
View File
@@ -319,11 +319,11 @@ TEXT ·f(SB), NOSPLIT, $0-0
`)
code = assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x29FFE061, // addi.d r1, r2, -8 (prologue)
0x02FFE063, // addi.d r3, r3, -8
0x29C00061, // st.d r1, r2, 0 (prologue saves RA)
0x29FFE061, // st.d r1, -8(r3) (prologue saves RA below the new SP)
0x02FFE063, // addi.d r3, r3, -8 (prologue opens the frame)
0x29C00061, // st.d r1, 0(r3) (prologue saves RA at SP)
0x4C0000A1, // jirl r1, r5, 0
0x28C00061, // ld.d r1, r2, 0 (epilogue restores RA)
0x28C00061, // ld.d r1, 0(r3) (epilogue restores RA)
0x02C02063, // addi.d r3, r3, 8
0x4C000020, // jirl r0, r1, 0 (RET)
)
+77 -4
View File
@@ -347,21 +347,94 @@ TEXT ·sb(SB), NOSPLIT, $0
}
}
// TestLOONG64_movImmToFp checks the immediate-to-FP move forms.
// TestLOONG64_movImmToFp checks the immediate-to-FP move: MOVW $c, Fd is the
// only spelling the toolchain accepts, expanding to ori (or addi.w for the
// negative span) into R30 plus movgr2fr.w. The pinned words are the
// toolchain's own bytes; the other widths and out-of-range constants are
// illegal combinations there and are diagnosed here.
func TestLOONG64_movImmToFp(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·fpmov(SB), NOSPLIT, $0
MOVV $0x1, F0
MOVW $0x1, F0
MOVW $0x2, F4
MOVW $-1, F4
RET
`)
code := assembleLOONG64Helper(t, fn)
want := []byte{
0x00, 0x04, 0x80, 0x03, // ori f0, r0, 1
0x04, 0x08, 0x80, 0x03, // ori f4, r0, 2
0x1e, 0x04, 0x80, 0x03, // ori r30, r0, 1
0xc0, 0xa7, 0x14, 0x01, // movgr2fr.w f0, r30
0x1e, 0x08, 0x80, 0x03, // ori r30, r0, 2
0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30
0x1e, 0xfc, 0xbf, 0x02, // addi.w r30, r0, -1
0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
// TestLOONG64_movImmToFpErrors checks the immediate-to-FP diagnostics: the
// widths the toolchain rejects as illegal combinations, and constants beyond
// the 12-bit ori/addi.w span (the toolchain never materialises a wider
// constant on this path).
func TestLOONG64_movImmToFpErrors(t *testing.T) {
cases := []string{
"MOVV $1, F0",
"MOVF $2, F4",
"MOVD $2, F4",
"MOVW $100000, F1",
"MOVW $-2049, F1",
"MOVW $4096, F1",
}
for _, src := range cases {
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("%s: expected an error, got none", src)
}
}
}
// TestLOONG64_branch16Unsigned pins the unsigned two-operand branches: with
// one register BLTU/BGEU keep the register-register form against R0 (never
// taken), the toolchain's encoding, where a beqz would test the wrong
// condition; the three-operand forms are unchanged.
func TestLOONG64_branch16Unsigned(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·u(SB), NOSPLIT, $0
BLTU R4, done
BGEU R5, done
BLTU R6, R7, done
BGEU R8, R9, done
done:
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x68001080, // bltu r4, r0, +4
0x6C000CA0, // bgeu r5, r0, +3
0x680008C7, // bltu r6, r7, +2
0x6C000509, // bgeu r8, r9, +1
0x4C000020, // jirl r0, r1, 0
)
}
// TestLOONG64_bitFieldRange checks the BSTRINS/BSTRPICK bit-number
// validation, mirroring the toolchain's "illegal bit number" rule: 0..31 for
// the .w forms, 0..63 for the .d forms, and lsb <= msb.
func TestLOONG64_bitFieldRange(t *testing.T) {
cases := []string{
"BSTRINSW $32, R4, $0, R5",
"BSTRPICKW $31, R4, $32, R5",
"BSTRINSV $64, R4, $0, R5",
"BSTRPICKV $3, R4, $4, R5",
"BSTRINSW $-1, R4, $0, R5",
}
for _, src := range cases {
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("%s: expected an error, got none", src)
}
}
}
+135 -19
View File
@@ -15,7 +15,10 @@ import (
func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
fi := riscvComputeFrame(t)
prologue := riscvPrologue(fi)
guardLen := riscvGuardLen(fi)
guardLen, err := riscvGuardLen(fi)
if err != nil {
return nil, nil, nil, nil, nil, err
}
var relocs []Reloc
var spadj []SpadjStep
@@ -66,20 +69,20 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
}
}
// Pass 4: recompute offsets with actual sizes.
// Pass 4: recompute offsets with actual sizes. recs holds the
// instructions in emission order, so an index into it walks t.Body in
// lockstep (the same single pass Pass 1 uses) instead of rescanning the
// whole slice per statement.
offsets = map[string]int{}
pos = guardLen + len(prologue)
ri := 0
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
for _, r := range recs {
if r.instr == s {
pos += len(r.code)
break
}
}
pos += len(recs[ri].code)
ri++
}
}
@@ -89,7 +92,10 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
// the morestack block at the end of the function, which the previous
// passes have sized.
var out []byte
guardBytes, guardReloc := riscvGuard(fi)
guardBytes, guardReloc, err := riscvGuard(fi)
if err != nil {
return nil, nil, nil, nil, nil, err
}
if fi.needSplit {
out = append(out, guardBytes...)
}
@@ -231,6 +237,25 @@ func isBranchLike(mnem string) bool {
return false
}
// riscvCheckBranchOffset rejects a B-type displacement outside its signed
// 13-bit span [-4096, 4094]; the encoder masks to 13 bits, so an
// out-of-range offset would otherwise wrap to a wrong target.
func riscvCheckBranchOffset(target string, off int32) error {
if off < -4096 || off > 4094 {
return fmt.Errorf("branch to %q too far (13-bit range)", target)
}
return nil
}
// riscvCheckJumpOffset rejects a J-type displacement outside its signed
// 21-bit span [-1048576, 1048574].
func riscvCheckJumpOffset(target string, off int32) error {
if off < -1048576 || off > 1048574 {
return fmt.Errorf("jump to %q too far (21-bit range)", target)
}
return nil
}
// encodeRISCVInstr encodes a single RISC-V instruction.
func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) {
mnem := instr.Mnemonic.Text
@@ -304,6 +329,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
offset := int32(targetOff - pc)
if err := riscvCheckJumpOffset(target, offset); err != nil {
return nil, err
}
word = riscvJType(0, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
case "JAL":
@@ -320,6 +348,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
offset := int32(targetOff - pc)
if err := riscvCheckJumpOffset(target, offset); err != nil {
return nil, err
}
word = riscvJType(rd, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
@@ -365,6 +396,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
case "BGTZ":
enc, rs1, rs2 = riscvEnc{0x63, 0x4, 0x00}, 0, rs // blt x0, rs
}
if err := riscvCheckBranchOffset(target, int32(targetOff-pc)); err != nil {
return nil, err
}
word = riscvBType(enc, rs1, rs2, int32(targetOff-pc))
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
@@ -374,7 +408,14 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
if !ok {
return nil, fmt.Errorf("unsupported system instruction %q", mnem)
}
word = riscvIType(enc, 0, 0, 0)
// The bare FENCE expands to fence iorw, iorw: the predecessor and
// successor fields both carry 0xF in the I-type immediate
// (the toolchain's encodeFenceOperand TYPE_NONE default).
imm := int32(0)
if mnem == "FENCE" {
imm = 0x0FF
}
word = riscvIType(enc, 0, 0, imm)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
@@ -416,7 +457,10 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
csr := immFromOperand(ops[0]) // CSR address (12-bit)
rd := regFromOperand(ops[2]) // destination register
if csr < 0 || csr > 0xFFF {
return nil, fmt.Errorf("%s: CSR address %d out of range 0-0xFFF", mnem, csr)
}
rd := regFromOperand(ops[2]) // destination register
if rd < 0 {
return nil, fmt.Errorf("invalid destination register in %s", mnem)
}
@@ -561,9 +605,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
// I-type with immediate: Plan 9 order is INSTR $imm, rs1, rd; the
// two-operand form INSTR $imm, rd uses rd as the source.
case len(ops) == 3 && isITypeInstr(mnem):
imm := immFromOperand(ops[0]) // immediate
if immNeg {
imm = -imm // SUB $imm arrived through the ADDI alias
imm, err := riscvImm32FromOperand(ops[0], immNeg) // immediate
if err != nil {
return nil, err
}
rs1 := regFromOperand(ops[1]) // source register
rd := regFromOperand(ops[2]) // destination
@@ -573,9 +617,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return encodeRISCVItypeImmediate(mnem, enc, rd, rs1, imm)
case len(ops) == 2 && isITypeInstr(mnem):
imm := immFromOperand(ops[0])
if immNeg {
imm = -imm
imm, err := riscvImm32FromOperand(ops[0], immNeg)
if err != nil {
return nil, err
}
rd := regFromOperand(ops[1])
if rd < 0 {
@@ -614,6 +658,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
if rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
if err := riscvCheckBranchOffset(target, offset); err != nil {
return nil, err
}
// The Go assembler never compresses branches to C.BEQZ/C.BNEZ.
word = riscvBType(enc, rs1, rs2, offset)
@@ -695,7 +742,10 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
if rd < 0 {
return nil, fmt.Errorf("MOV $imm: invalid destination register")
}
imm := immFromOperand(src)
imm, err := riscvImm32FromOperand(src, false)
if err != nil {
return nil, err
}
return encodeRISCVLoadImm(rd, imm), nil
}
@@ -1081,7 +1131,7 @@ func encodeRISCVJALR(instr *ast.Instr, fi riscvFrameInfo) ([]byte, error) {
// tryCompressRVC attempts to compress a RISC-V instruction to its 16-bit
// RVC form. It returns the compressed instruction word and true on success.
func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
mnem := instr.Mnemonic.Text
mnem := riscvCompressMnem(instr)
ops := instr.Operands
switch mnem {
@@ -1331,6 +1381,49 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
return 0, false
}
// riscvCompressMnem maps a MOV-family load or store onto the base mnemonic
// the toolchain lowers it to (MOVW 4(SP), X9 is LW under another name), so
// the width spellings compress exactly like their base forms. Register and
// immediate forms keep their own mnemonic: the C.MV path matches "MOV" and
// nothing else in the switch has a width case.
func riscvCompressMnem(instr *ast.Instr) string {
mnem := instr.Mnemonic.Text
ops := instr.Operands
if !strings.HasPrefix(mnem, "MOV") || len(ops) != 2 {
return mnem
}
load := isMemOperand(ops[0]) && !isMemOperand(ops[1])
store := !isMemOperand(ops[0]) && isMemOperand(ops[1])
if !load && !store {
return mnem
}
switch mnem {
case "MOVW":
if load {
return "LW"
}
return "SW"
case "MOVF":
if load {
return "FLW"
}
return "FSW"
case "MOVD":
if load {
return "FLD"
}
return "FSD"
case "MOV":
if load {
return "LD"
}
return "SD"
}
// MOVB/MOVBU/MOVH/MOVHU/MOVWU have no compressed form; their base
// mnemonics (LB/LBU/LH/LHU/LWU, SB/SH) match no case either.
return mnem
}
// extractLDParams extracts rd, rs1, and immediate offset for a load instruction.
func extractLDParams(instr *ast.Instr, fi riscvFrameInfo) (rd, rs1 int, imm int32) {
ops := instr.Operands
@@ -1507,6 +1600,29 @@ func immFromOperand(op *ast.Operand) int32 {
return 0
}
// riscvImm32FromOperand reads an immediate for the MOV/I-type paths as a
// signed 32-bit value. The toolchain materialises wider constants through
// its SLLI expansion, which this assembler does not implement, so values
// outside the int32 span are diagnosed instead of silently truncated (MOV
// $0x123456789 must not assemble as $0x3456789). The neg flag carries the
// SUB $imm alias, whose negated value may fit when the written one does not.
func riscvImm32FromOperand(op *ast.Operand, neg bool) (int32, error) {
var v int64
if op.Imm.HasVal {
v = op.Imm.Val
if op.Imm.Neg {
v = -v
}
}
if neg {
v = -v
}
if int64(int32(v)) != v {
return 0, fmt.Errorf("immediate %d out of range; 64-bit materialisation not supported", v)
}
return int32(v), nil
}
func memFromOperand(op *ast.Operand) (rs1 int, imm int32) {
rs1 = riscvRegNum(op.Addr.Base)
imm = int32(op.Addr.Offset)
+7 -5
View File
@@ -65,7 +65,7 @@ func riscvRegNum(name string) int {
return 25
case "X26", "S10":
return 26
case "X27", "S11":
case "X27", "S11", "g":
return 27
case "X28", "T3":
return 28
@@ -373,7 +373,7 @@ var riscvFmaTable = map[string]riscvFmaEnc{
// riscvFmaType encodes an R4-type fused multiply-add instruction.
func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 {
return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) |
(uint32(rs1) << 15) | (0x0 << 12) /* rm=dynamic */ | (uint32(rd) << 7) | enc.opcode
(uint32(rs1) << 15) | (0x0 << 12) /* rm=RNE */ | (uint32(rd) << 7) | enc.opcode
}
// CSR (Control and Status Register) instructions.
@@ -522,11 +522,13 @@ func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 {
// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
// offset; the immediate bits are extracted per the RISC-V CS format.
// offset; the immediate bits are extracted per the RISC-V CS format, with the
// same five-bit patterns as the load side ({5,4,3,7,6} and {5,4,3,2,6},
// matching the toolchain's encodeCS).
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
pattern := []int{5, 3, 7, 6}
pattern := []int{5, 4, 3, 7, 6}
if funct3 == 0x6 {
pattern = []int{5, 3, 2, 6}
pattern = []int{5, 4, 3, 2, 6}
}
packed := encodeRVCPattern(imm, pattern)
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
+189
View File
@@ -5,6 +5,7 @@ package asm
import (
"bytes"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -691,6 +692,81 @@ DATA answer<>+0(SB)/8, $42
}
}
// TestRISCV_RVC_StorePatterns pins the register-relative compressed store
// encodings for offsets with immediate bits 4 and 5 set, byte-identical to
// the toolchain's encodeCS (patterns {5,4,3,7,6} and {5,4,3,2,6}).
// Regression: the store-side patterns dropped imm[4], so every such store
// silently encoded the wrong address while the loads stayed correct.
func TestRISCV_RVC_StorePatterns(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·csstores(SB), NOSPLIT, $0
SD X9, 24(X8)
SW X10, 16(X11)
FSD F8, 40(X12)
LD 24(X8), X9
LW 16(X11), X10
FLD 40(X12), F8
RET
`)
code := assembleRISCVHelper(t, fn)
want := []byte{
0x04, 0xec, // c.sd x9, 24(x8)
0x88, 0xc9, // c.sw x10, 16(x11)
0x00, 0xb6, // c.fsd f8, 40(x12)
0x04, 0x6c, // c.ld x9, 24(x8)
0x88, 0x49, // c.lw x10, 16(x11)
0x00, 0x36, // c.fld f8, 40(x12)
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
// TestRISCV_FENCE pins the FENCE encoding: the toolchain expands the bare
// mnemonic to fence iorw, iorw (0x0FF0000F), not fence 0,0.
func TestRISCV_FENCE(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·fence(SB), NOSPLIT, $0
FENCE
RET
`)
code := assembleRISCVHelper(t, fn)
want := []byte{
0x0f, 0x00, 0xf0, 0x0f, // fence iorw, iorw
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
// TestRISCV_RVC_WidthSpellings pins the compression of the GOROOT width
// spellings: MOVW and MOVD lower to their base load/store and compress
// exactly like LW/SW/FLD/FSD would (the toolchain compresses these shapes;
// before the normalisation they stayed 4 bytes).
func TestRISCV_RVC_WidthSpellings(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·widths(SB), NOSPLIT, $0-16
MOVW w+0(FP), X9
MOVW X9, v+4(FP)
MOVD d+0(FP), F8
MOVD F8, r+8(FP)
RET
`)
code := assembleRISCVHelper(t, fn)
want := []byte{
0xa2, 0x44, // c.lwsp x9, 8
0x26, 0xc6, // c.swsp x9, 12
0x22, 0x24, // c.fldsp f8, 8
0x22, 0xa8, // c.fsdsp f8, 16
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
func TestRISCV_system_instrs(t *testing.T) {
// Test FENCE, ECALL, EBREAK encoding.
fn := firstTextRISCV(t, `#include "textflag.h"
@@ -782,3 +858,116 @@ TEXT ·f(SB), NOSPLIT, $0-0
0x00008067, // jalr x0, 1(x0), 0 (RET)
)
}
// encodeOneInstrRISCV encodes a single parsed instruction against a synthetic
// offsets map, the smallest honest harness for the branch-range diagnostics:
// the spans are far larger than any source a test would want to spell out.
func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]int) ([]byte, error) {
t.Helper()
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
instr := fn.Body[0].(*ast.Instr)
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil)
}
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span
// [-4096, 4094] and the J-type span [-1048576, 1048574] are diagnosed instead
// of wrapping silently to a wrong target.
func TestRISCVBranchJumpRange(t *testing.T) {
cases := []struct {
name string
src string
off int // the target's function-relative offset (pc 0)
ok bool
}{
{"branch max", "BEQ X10, X11, tgt\nRET\n", 4094, true},
{"branch past max", "BEQ X10, X11, tgt\nRET\n", 4096, false},
{"branch back max", "BEQ X10, X11, tgt\nRET\n", -4096, true},
{"branch back past max", "BEQ X10, X11, tgt\nRET\n", -4098, false},
{"branchz past max", "BEQZ X10, tgt\nRET\n", 4096, false},
{"jump max", "JMP tgt\nRET\n", 1048574, true},
{"jump past max", "JMP tgt\nRET\n", 1048576, false},
{"jump back max", "JMP tgt\nRET\n", -1048576, true},
{"jump back past max", "JMP tgt\nRET\n", -1048578, false},
{"jal past max", "JAL tgt\nRET\n", 1048576, false},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
_, err := encodeOneInstrRISCV(t, "TEXT ·f(SB), NOSPLIT, $0\n\t"+c.src, 0, map[string]int{"tgt": c.off})
if c.ok && err != nil {
t.Fatalf("unexpected error: %v", err)
}
if !c.ok && err == nil {
t.Fatal("expected an out-of-range diagnostic, got none")
}
})
}
}
// TestRISCVBranchFarBody drives the range check through the full two-pass
// assembler: a forward branch over a body larger than the B-type span must
// error rather than wrap.
func TestRISCVBranchFarBody(t *testing.T) {
var sb strings.Builder
sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n")
for range 1100 {
sb.WriteString("\tADD X10, X11, X12\n")
}
sb.WriteString("done:\n\tRET\n")
fn := firstTextRISCV(t, sb.String())
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
t.Error("expected a branch-out-of-range error, got none")
}
}
// TestRISCV_CSRRange checks the CSR address range: the 12-bit field is
// diagnosed rather than masked, so CSRRW $4096 does not silently address
// CSR 0.
func TestRISCV_CSRRange(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·csrhi(SB), NOSPLIT, $0
CSRRW $4096, X10, X11
RET
`)
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
t.Error("expected an out-of-range error for CSR $4096, got none")
}
fn = firstTextRISCV(t, `#include "textflag.h"
TEXT ·csrmax(SB), NOSPLIT, $0
CSRRW $4095, X10, X11
RET
`)
if _, _, _, _, _, err := assembleRISCV(fn); err != nil {
t.Errorf("CSR $4095 must assemble: %v", err)
}
}
// TestRISCV_Imm64Rejected checks that immediates outside the signed 32-bit
// span are diagnosed instead of silently truncated to their low 32 bits (the
// toolchain materialises such constants via SLLI expansion, which this
// assembler does not implement).
func TestRISCV_Imm64Rejected(t *testing.T) {
cases := []string{
"MOV $0x123456789, X10",
"ADDI $0x100000000, X10, X11",
"ANDI $-0x800000001, X10, X11",
"SUB $0x100000000, X10, X11",
}
for _, src := range cases {
fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·wide(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
t.Errorf("%s: expected an out-of-range error, got none", src)
}
}
// The full signed 32-bit span still assembles, including the SUB form
// whose negated immediate only just fits.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·edge(SB), NOSPLIT, $0
MOV $2147483647, X10
MOV $-2147483648, X11
SUB $0x80000000, X12, X13
RET
`)
if _, _, _, _, _, err := assembleRISCV(fn); err != nil {
t.Errorf("int32-span immediates must assemble: %v", err)
}
}
+34 -31
View File
@@ -4,6 +4,7 @@
package asm
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -241,33 +242,36 @@ func riscvFitsCAddi(imm int32) bool {
}
// riscvPrologueSpadjPC returns the function-relative byte offset where the
// prologue has finished decrementing SP (the delta becomes autosize).
// prologue has finished decrementing SP (the delta becomes autosize). It is
// computed from the same expansion functions the prologue emits, so the
// large-frame X31 materialisations are counted: C.LUI + C.ADD before the SD,
// C.LUI + ADDIW + C.ADD for the SP adjust.
func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
if fi.autosize == 0 {
return 0
}
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 4 + riscvSPAdjustLen(int32(-fi.autosize))
adj := int32(-fi.autosize)
if fits12(adj) {
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 4 + len(riscvSPAdjust(adj))
}
return len(riscvAddressInX31(adj)) + 4 + len(riscvAddToSP(adj))
}
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final JALR, the point where SP is restored.
// (but not including) the final JALR, the point where SP is restored. The
// small frame closes with C.LDSP + ADDI/C.ADDI; the large frame materialises
// the adjustment through X31 (C.LUI + ADDIW + C.ADD).
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
if fi.autosize == 0 {
return 0
}
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 2 + riscvSPAdjustLen(int32(fi.autosize))
}
func riscvSPAdjustLen(imm int32) int {
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
return 2
adj := int32(fi.autosize)
if fits12(adj) {
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 2 + len(riscvSPAdjust(adj))
}
if riscvFitsCAddi(imm) {
return 2
}
return 4
return 2 + len(riscvAddToSP(adj))
}
// riscvResolvePseudo translates a pseudo-register memory reference into a
@@ -293,19 +297,21 @@ func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32
// including the inline morestack call (zero when the function needs no
// guard). Unlike amd64 and arm64, the toolchain places the morestack call
// between the guard and the body: the guard branches forward over it.
func riscvGuardLen(fi riscvFrameInfo) int {
_, reloc := riscvGuard(fi)
_ = reloc
return len(riscvGuardBytes(fi))
func riscvGuardLen(fi riscvFrameInfo) (int, error) {
g, _, err := riscvGuard(fi)
if err != nil {
return 0, err
}
return len(g), nil
}
// riscvGuard emits the stack-split guard prefix with the inline morestack
// call: the branch skips forward over JAL X5 and JAL X0 straight into the
// body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are
// relative to the guard itself, which sits at function offset 0.
func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc, error) {
if !fi.needSplit {
return nil, Reloc{}
return nil, Reloc{}, nil
}
// MOV 16(g), X6 (g.stackguard0), g = X27.
out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16))
@@ -317,14 +323,14 @@ func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
var reloc Reloc
switch fi.splitClass {
case 0:
// BLTU X6, SP, done (+8: over the CALL and the JMP back)
// BLTU X6, SP, done (+12: over the CALL and the JMP back)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
case 1:
// ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+8)
// ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+12)
off := int32(fi.autosize - stackSmall)
out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
@@ -342,7 +348,10 @@ func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...)
addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off)
if err != nil {
addi = nil
// The ADDI expansion failed: the SP adjustment this class
// depends on is not emittable, and silently dropping it would
// corrupt every stack reference in the body.
return nil, Reloc{}, fmt.Errorf("stack-split guard: %w", err)
}
out = append(out, addi...)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
@@ -351,11 +360,5 @@ func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
}
return out, reloc
}
// riscvGuardBytes emits the guard prefix bytes alone (sizing helper).
func riscvGuardBytes(fi riscvFrameInfo) []byte {
g, _ := riscvGuard(fi)
return g
return out, reloc, nil
}
+41
View File
@@ -65,6 +65,47 @@ TEXT ·framed(SB), NOSPLIT, $16-16
}
}
// TestRISCVFrameSpadjLargeFrame checks the stack-adjustment boundaries of a
// frame past the imm12 range: the prologue materialises the LR-store address
// and the SP adjustment through X31 (C.LUI + C.ADD + SD, then C.LUI + ADDIW +
// C.ADD), so the SP boundary lands at PC 16, and the RET closes with
// C.LDSP plus the same X31 adjustment, 10 bytes. Regression: both helpers
// assumed the small-frame prologue and reported 8 and 6.
func TestRISCVFrameSpadjLargeFrame(t *testing.T) {
f, errs := parser.Parse("bigframe_riscv64.s", `#include "textflag.h"
TEXT ·big(SB), NOSPLIT, $9000-8
MOV a+0(FP), X10
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
fn := img.Funcs[0]
// autosize = 9008. Prologue: C.LUI X31 + C.ADD X31,SP (4) + SD (4) +
// C.LUI X31 + ADDIW X31 + C.ADD SP,X31 (8) = 16 bytes to the SP boundary;
// C.SDSP X1 (2) follows, so the body starts at 18.
wantSpadj := []SpadjStep{{PC: 16, Value: 9008}, {PC: 36, Value: 0}}
if len(fn.Spadj) != len(wantSpadj) {
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
}
for i := range wantSpadj {
if fn.Spadj[i] != wantSpadj[i] {
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
}
}
// The FP load materialises its 9016-byte offset through X31 as well
// (8 bytes), then RET's epilogue (C.LDSP + X31 adjust = 10) plus JALR.
if fn.Size != 18+8+14 {
t.Errorf("size = %d, want %d", fn.Size, 18+8+14)
}
}
// TestRISCVRegAliases checks the Go ABI register aliases that the toolchain
// defines: LR is the link register (X1) and TMP is the assembler scratch
// register (X31/T6).
+83
View File
@@ -74,6 +74,89 @@ DATA callee<>+0(SB)/8, $42
t.Error("ELF object missing R_RISCV_JAL relocation")
}
}
// TestELFRISCVPCRELLO12Anchor checks the psABI's LO12 pairing rule: the
// R_RISCV_PCREL_LO12_I/S relocation must reference a symbol whose value is
// the AUIPC site of its HI20 partner (psABI §8.4.9; cmd/link generates one
// local text symbol per AUIPC for exactly this). The emitter pairs each
// HI20 (against the target symbol) with a LO12 against the .text section
// symbol whose addend is the AUIPC's section-relative offset, so S + A is
// the AUIPC address.
func TestELFRISCVPCRELLO12Anchor(t *testing.T) {
f, errs := parser.Parse("k_riscv64.s", `
#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0-0
MOV $answer<>(SB), X10
MOV answer<>(SB), X11
MOV X12, answer<>(SB)
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
obj, err := img.ELFRISCVObject()
if err != nil {
t.Fatalf("ELFRISCVObject: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse ELF: %v", err)
}
defer ef.Close()
if flags := binary.LittleEndian.Uint32(obj[48:]); flags != efRISCVFloatAbiDouble {
t.Errorf("e_flags = %#x, want %#x (EF_RISCV_FLOAT_ABI_DOUBLE)", flags, efRISCVFloatAbiDouble)
}
rela := ef.Section(".rela.text")
if rela == nil {
t.Fatal("missing .rela.text")
}
b, err := rela.Data()
if err != nil {
t.Fatal(err)
}
if len(b) != 6*24 {
t.Fatalf(".rela.text holds %d entries, want six (three HI20/LO12 pairs)", len(b)/24)
}
le := binary.LittleEndian
wantLo := []uint32{rRISCVPCRELLO12I, rRISCVPCRELLO12I, rRISCVPCRELLO12S}
for p := range 3 {
auipc := 8 * p
hi := b[p*2*24:]
lo := b[(p*2+1)*24:]
if off := le.Uint64(hi[0:]); off != uint64(auipc) {
t.Errorf("pair %d: HI20 r_offset = %d, want %d (the AUIPC)", p, off, auipc)
}
if typ := uint32(le.Uint64(hi[8:])); typ != rRISCVPCRELHI20 {
t.Errorf("pair %d: HI20 type = %d, want %d", p, typ, rRISCVPCRELHI20)
}
if sym := int(le.Uint64(hi[8:]) >> 32); sym == 0 || sym == 1 {
t.Errorf("pair %d: HI20 against symbol %d, want the target", p, sym)
}
if off := le.Uint64(lo[0:]); off != uint64(auipc+4) {
t.Errorf("pair %d: LO12 r_offset = %d, want %d", p, off, auipc+4)
}
if typ := uint32(le.Uint64(lo[8:])); typ != wantLo[p] {
t.Errorf("pair %d: LO12 type = %d, want %d", p, typ, wantLo[p])
}
// The LO12 must denote the AUIPC site: the .text section symbol
// (index 1) plus the AUIPC's section-relative offset as addend.
if sym := int(le.Uint64(lo[8:]) >> 32); sym != 1 {
t.Errorf("pair %d: LO12 against symbol %d, want 1 (the .text section symbol)", p, sym)
}
if add := int64(le.Uint64(lo[16:])); add != int64(auipc) {
t.Errorf("pair %d: LO12 addend = %d, want %d (S + A = the AUIPC address)", p, add, auipc)
}
}
}
func TestGOObjectRISCVStructure(t *testing.T) {
f, errs := parser.Parse("k_riscv64.s", `
#include "textflag.h"
+2 -1
View File
@@ -114,6 +114,7 @@ type Symbol struct {
Pkg string // package prefix before the middle dot ("" = current package)
Name string // identifier without the middle dot or <>
Static bool // the <> marker is present
ABI string // the <NAME> ABI marker, e.g. ABIInternal ("" when absent)
Pseudo string // FP, SP, SB or PC ("" for a bare name)
Offset int64
HasOff bool
@@ -156,5 +157,5 @@ type Address struct {
Scale int // index scale; 0 when absent
Offset int64 // leading displacement, from off(base)
HasOff bool // a leading displacement is present
Shift string // verbatim arm64 shift suffix, e.g. "<<2"
Shift string // verbatim arm64 shift suffix, e.g. "<< 2"
}
+18 -8
View File
@@ -40,10 +40,13 @@ func cmdAuditInstructions(args []string) error {
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64]", `
Compare the gasm encoder for the given architecture (default amd64) against
go tool asm and print the diff: superset encodings (gasm-only, shippable via
gasm asm --format goobj), known-but-unencodable names (the backlog) and go-
only names (feature gaps). The Go side is probed black-box with a battery
of bare mnemonics, so the audit tracks whatever toolchain `+"`go env GOROOT`"+`
provides.
gasm asm --format goobj) and known-but-unencodable names (the backlog). The
Go side is probed black-box one bare mnemonic at a time, so the audit tracks
whatever toolchain `+"`go env GOROOT`"+` provides; the gasm side answers from
the encoder table on amd64 and from trial assembly over a battery of operand
shapes elsewhere. Names go tool asm knows and gasm does not cannot be
enumerated by probing, because Go's table is visible only through names
already in the gasm table; the report closes with a note saying so.
With --corpus the audit changes shape: it assembles every .s file under the
given directory (default GOROOT/src) with the gasm encoder only, no
@@ -63,7 +66,7 @@ the encodability backlog by frequency rather than by table order.
archName := "amd64"
switch n := len(fs.Args()); {
case n > 1:
return fmt.Errorf("audit-instructions takes at most one architecture argument")
return &usageError{fmt.Errorf("audit-instructions takes at most one architecture argument")}
case n == 1:
archName = strings.ToLower(fs.Arg(0))
}
@@ -109,7 +112,7 @@ the encodability backlog by frequency rather than by table order.
w := os.Stdout
fmt.Fprintf(w, "gasm table (%s, families excluded): %d mnemonics\n", archName, len(names))
fmt.Fprintf(w, "gasm encodable: %d go tool asm recognized: %d\n", len(shared)+len(superset), countTrue(goKnown))
fmt.Fprintf(w, "gasm encodable: %d go tool asm recognised: %d\n", len(shared)+len(superset), countTrue(goKnown))
fmt.Fprintf(w, "shared: %d\n", len(shared))
fmt.Fprintf(w, "\nSuperset encodings (gasm-only; ship via gasm asm --format goobj):\n")
for _, n := range superset {
@@ -136,7 +139,7 @@ func auditArch(name string) (arch.Arch, error) {
case "loong64", "loong":
return arch.LOONG64, nil
}
return arch.Unknown, fmt.Errorf("unknown architecture %q: want amd64, arm64, riscv64 or loong64", name)
return arch.Unknown, &usageError{fmt.Errorf("unknown architecture %q: want amd64, arm64, riscv64 or loong64", name)}
}
// goarchName maps an arch identifier onto its GOARCH spelling.
@@ -217,6 +220,13 @@ func probeGoAsm(goarch string, names []string) (map[string]bool, error) {
cmd := exec.Command(asmBin, "-p", "probe", "-o", filepath.Join(dir, "probe.o"), probePath)
cmd.Env = append(os.Environ(), "GOARCH="+goarch, "GOOS="+runtime.GOOS)
out, _ := cmd.CombinedOutput()
// The expected failure mode is a non-zero exit with compiler diagnostics
// on stdout; empty output means the probe broke at the exec level (a
// killed child, a tool that would not start), and seeding every name as
// recognized on that silence would fake a clean audit.
if len(out) == 0 {
return nil, fmt.Errorf("go tool asm probe for GOARCH=%s produced no output", goarch)
}
result := map[string]bool{}
for _, name := range names {
@@ -344,7 +354,7 @@ func (t *corpusTally) fail(path, reason string) {
// cmdAuditCorpus implements audit-instructions --corpus.
func cmdAuditCorpus(args []string) error {
if len(args) > 1 {
return fmt.Errorf("audit-instructions --corpus takes at most one directory argument")
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
}
root := ""
if len(args) == 1 {
+11 -3
View File
@@ -24,9 +24,10 @@ function in a traced subprocess (ptrace), then provides a REPL for
single-stepping, breakpoints, register and memory inspection.
REPL commands:
break <label|addr> [if <reg> <op> <val>]
break <label|addr|line> [if <reg> <op> <val|reg|*addr>]
set a breakpoint, optionally conditional on a
register comparison (reg-reg or reg-immediate)
comparison of one register against a constant,
another register, or the 8-byte word at *addr
delete <label|addr> remove a breakpoint
info break list all breakpoints
step [n], s single-step n instructions (default 1)
@@ -53,7 +54,7 @@ REPL commands:
bufSpec := fs.String("buf", "", "buffer specification: name:size:pattern[,name:size:pattern...] where pattern is zero, ones, seq, or hex")
script := fs.String("script", "", "run REPL commands from a file (one per line) and exit; '-' reads stdin")
cover := fs.Bool("cover", false, "run to completion with a breakpoint on every instruction and report which executed and how often")
timeout := fs.Duration("timeout", 0, "kill the debuggee after this duration (e.g. 30s); for headless --script runs")
timeout := fs.Duration("timeout", 0, "kill the debuggee after this duration (e.g. 30s); for headless --script runs; a timeout exits 3")
fs.Parse(args)
// --- Debuggee mode (internal, spawned by the debugger) ---
@@ -231,6 +232,13 @@ REPL commands:
if sess.Exited() {
break
}
// A genuine signal-delivery-stop (a fault in the kernel): the
// run cannot make progress, because resuming would restart the
// faulting instruction and fault forever. Report and stop.
if sig := sess.LastSignal(); sig != 0 {
fmt.Printf("gasm debug: cover: stopped on signal %v\n", sig)
break
}
regs, rerr := sess.GetRegs()
if rerr != nil {
break
+123 -68
View File
@@ -10,6 +10,7 @@ package main
import (
"bytes"
"encoding/json"
"errors"
"flag"
"fmt"
"io"
@@ -48,6 +49,22 @@ func version() string {
return bi.Main.Version
}
// usageError marks an error the caller's arguments caused, which exits 2
// instead of the 1 a runtime failure gets.
type usageError struct{ err error }
func (e *usageError) Error() string { return e.err.Error() }
func (e *usageError) Unwrap() error { return e.err }
// exitCodeFor maps an error onto the process exit status: 2 for a usage
// error, 1 for anything else.
func exitCodeFor(err error) int {
if _, ok := errors.AsType[*usageError](err); ok {
return 2
}
return 1
}
func main() {
if len(os.Args) < 2 {
usage(os.Stderr)
@@ -77,12 +94,12 @@ func main() {
case "audit-instructions":
if err := cmdAuditInstructions(os.Args[2:]); err != nil {
fmt.Fprintln(os.Stderr, err)
os.Exit(1)
os.Exit(exitCodeFor(err))
}
case "scaffold":
if err := cmdScaffold(os.Args[2:]); err != nil {
fmt.Fprintln(os.Stderr, err)
os.Exit(1)
os.Exit(exitCodeFor(err))
}
case "lsp":
os.Exit(cmdLSP(os.Args[2:]))
@@ -104,11 +121,11 @@ func cmdVersion() int {
// ANSI colour helpers for terminal output.
const (
colorReset = "\033[0m"
colorBold = "\033[1m"
colorCyan = "\033[36m"
colorYellow = "\033[33m"
colorGray = "\033[90m"
colourReset = "\033[0m"
colourBold = "\033[1m"
colourCyan = "\033[36m"
colourYellow = "\033[33m"
colourGrey = "\033[90m"
)
// isTTY reports whether the writer is a terminal (for colour output).
@@ -122,9 +139,9 @@ func isTTY(w io.Writer) bool {
func usage(w io.Writer) {
useColor := isTTY(w)
bold, cyan, yellow, gray, reset := "", "", "", "", ""
bold, cyan, yellow, grey, reset := "", "", "", "", ""
if useColor {
bold, cyan, yellow, gray, reset = colorBold, colorCyan, colorYellow, colorGray, colorReset
bold, cyan, yellow, grey, reset = colourBold, colourCyan, colourYellow, colourGrey, colourReset
}
fmt.Fprintf(w, "%sgasm %s%s: developer tooling for Go's Plan 9 assembler (GAsm)%s\n\n", bold, version(), reset, reset)
@@ -153,12 +170,12 @@ func usage(w io.Writer) {
{"version", "print the version (same as --version)"},
}
for _, c := range commands {
fmt.Fprintf(w, " %s%-10s%s %s%s%s\n", cyan, c.name, reset, gray, c.desc, reset)
fmt.Fprintf(w, " %s%-10s%s %s%s%s\n", cyan, c.name, reset, grey, c.desc, reset)
}
fmt.Fprintf(w, "\n%sFlags:%s\n", yellow, reset)
fmt.Fprintf(w, " %s-h, --help%s %sshow this help%s\n", cyan, reset, gray, reset)
fmt.Fprintf(w, " %s-V, --version%s %sprint the version%s\n", cyan, reset, gray, reset)
fmt.Fprintf(w, " %s-h, --help%s %sshow this help%s\n", cyan, reset, grey, reset)
fmt.Fprintf(w, " %s-V, --version%s %sprint the version%s\n", cyan, reset, grey, reset)
fmt.Fprintf(w, "\nRun \"gasm <command> -h\" for a command's usage and flags.\n\n")
@@ -172,7 +189,7 @@ func usage(w io.Writer) {
}
for _, e := range examples {
if e.desc != "" {
fmt.Fprintf(w, " %s%s%s %s%s%s\n", cyan, e.cmd, reset, gray, e.desc, reset)
fmt.Fprintf(w, " %s%s%s %s%s%s\n", cyan, e.cmd, reset, grey, e.desc, reset)
} else {
fmt.Fprintf(w, " %s%s%s\n", cyan, e.cmd, reset)
}
@@ -469,10 +486,13 @@ func cmdAsm(args []string) int {
With -o the output is written to a file instead. The --format flag selects
what is written: raw (the default) concatenates the functions and the data
section into one self-consistent image; elf emits a relocatable object
(.text/.data sections, a symbol table and one PC32 relocation per
static-symbol reference) that links with the system toolchain; goobj emits
the Go toolchain's own object format, which cmd/link consumes directly (it
requires -p, the package path, and the installed Go toolchain).
(.text/.data sections, a symbol table and one relocation per static-symbol
reference, in the architecture's own form: R_X86_64_PC32 on amd64,
R_AARCH64_*, R_RISCV_* or R_LARCH_* on the others) that links with the
system toolchain; goobj emits the Go toolchain's own object format, which
cmd/link consumes directly (it requires -p, the package path, and the
installed Go toolchain: the object preamble is captured from go tool asm
and the format version from go version).
`)
out := fs.String("o", "", "write the output to this file")
format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
@@ -483,6 +503,14 @@ requires -p, the package path, and the installed Go toolchain).
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>")
return 2
}
// The format is validated before anything else, so a bogus value exits 2
// with or without -o instead of silently dumping the hex of a raw image.
switch *format {
case "raw", "elf", "goobj":
default:
fmt.Fprintf(os.Stderr, "gasm asm: unknown format %q (want raw, elf or goobj)\n", *format)
return 2
}
path := fs.Arg(0)
targetArch := arch.FromFilename(path)
if *archName != "" {
@@ -517,38 +545,42 @@ requires -p, the package path, and the installed Go toolchain).
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions or GLOBL data found")
return 1
}
for _, fn := range img.Funcs {
code := img.Code[fn.Offset : fn.Offset+fn.Size]
fmt.Printf("%s: %d bytes\n", fn.Name, fn.Size)
for i := 0; i < len(code); i += 16 {
end := min(i+16, len(code))
fmt.Printf(" %04x:", i)
for _, b := range code[i:end] {
fmt.Printf(" %02x", b)
// Without -o the hex dump on stdout is the output; with -o the file is,
// and the dump is skipped, as the -o help text promises.
if *out == "" {
for _, fn := range img.Funcs {
code := img.Code[fn.Offset : fn.Offset+fn.Size]
fmt.Printf("%s: %d bytes\n", fn.Name, fn.Size)
for i := 0; i < len(code); i += 16 {
end := min(i+16, len(code))
fmt.Printf(" %04x:", i)
for _, b := range code[i:end] {
fmt.Printf(" %02x", b)
}
fmt.Println()
}
fmt.Println()
}
}
if len(img.Data) > 0 {
fmt.Printf("data: %d bytes at 0x%x\n", len(img.Data), len(img.Code))
for _, d := range f.Decls {
g, ok := d.(*ast.Globl)
if !ok || g.Name == nil || g.Name.Pseudo != "SB" {
continue
if len(img.Data) > 0 {
fmt.Printf("data: %d bytes at 0x%x\n", len(img.Data), len(img.Code))
for _, d := range f.Decls {
g, ok := d.(*ast.Globl)
if !ok || g.Name == nil || g.Name.Pseudo != "SB" {
continue
}
size := 0
if g.Size != nil && g.Size.Imm.HasVal {
size = int(g.Size.Imm.Val)
}
fmt.Printf(" %s: %d bytes at 0x%x\n", g.Name.Name, size, img.Symbols[g.Name.Name])
}
size := 0
if g.Size != nil && g.Size.Imm.HasVal {
size = int(g.Size.Imm.Val)
for i := 0; i < len(img.Data); i += 16 {
end := min(i+16, len(img.Data))
fmt.Printf(" %04x:", len(img.Code)+i)
for _, b := range img.Data[i:end] {
fmt.Printf(" %02x", b)
}
fmt.Println()
}
fmt.Printf(" %s: %d bytes at 0x%x\n", g.Name.Name, size, img.Symbols[g.Name.Name])
}
for i := 0; i < len(img.Data); i += 16 {
end := min(i+16, len(img.Data))
fmt.Printf(" %04x:", len(img.Code)+i)
for _, b := range img.Data[i:end] {
fmt.Printf(" %02x", b)
}
fmt.Println()
}
}
if *out != "" {
@@ -586,9 +618,6 @@ requires -p, the package path, and the installed Go toolchain).
obj, err = img.GOObject(*pkg, path)
}
kind = "Go object"
default:
fmt.Fprintf(os.Stderr, "gasm asm: unknown format %q (want raw, elf or goobj)\n", *format)
return 2
}
if err != nil {
fmt.Fprintln(os.Stderr, "gasm asm:", err)
@@ -767,8 +796,8 @@ func cmdProfile(args []string) int {
flagSet := newCommand("profile", "gasm profile <file.s>", `
Show the basic-block structure of functions in an assembly file.
Lists each function's labels, their offsets, and the block boundaries.
This is the static structure; for runtime execution counts, use
gasm verify --fuzz which exercises the code paths.
This is the static structure; for runtime execution counts use
gasm debug --cover, and for input coverage gasm verify --fuzz.
`)
flagSet.Parse(args)
if flagSet.NArg() != 1 {
@@ -912,11 +941,40 @@ func compareGroundTruth(img *asm.Image, gt map[string][]byte) (matched, total, d
goCmp[j] = 0
}
}
if bytes.Equal(gasmCmp, goCmp) {
// The toolchain pads text symbols to 16-byte boundaries with
// zeros, so a function whose size is not a multiple of 16
// carries trailing zeros in the ground truth that are not part
// of the encoding. Compare up to the shorter side and require
// the remainder of whichever is longer to be zero, so padding
// never masks a real difference.
cmpLen := min(len(gasmCmp), len(goCmp))
equal := bytes.Equal(gasmCmp[:cmpLen], goCmp[:cmpLen])
if equal {
for _, b := range gasmCmp[cmpLen:] {
if b != 0 {
equal = false
break
}
}
}
if equal {
for _, b := range goCmp[cmpLen:] {
if b != 0 {
equal = false
break
}
}
}
if equal {
matched++
if len(fn.Relocs) > 0 {
switch {
case len(fn.Relocs) > 0 && len(goCmp) > cmpLen:
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked, %d padding)\n", fn.Name, fn.Size, len(fn.Relocs), len(goCmp)-cmpLen)
case len(fn.Relocs) > 0:
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
} else {
case len(goCmp) > cmpLen:
fmt.Printf(" %s: MATCH (%d bytes, %d padding)\n", fn.Name, fn.Size, len(goCmp)-cmpLen)
default:
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
}
} else {
@@ -964,8 +1022,8 @@ that tolerate nil pointers and zero lengths in their arguments.
With -abi, each function is called with sentinel values in the registers
the Go ABI fixes across calls (the frame pointer and the goroutine
pointer) plus a canary below SP; violations are reported. JIT-based
checks run when the host matches the file's architecture (all but
loong64, which is ground-truth only for now).
checks run when the host matches the file's architecture, on all four
architectures.
With -fuzz, each function with a // func signature is differentially fuzzed
against the go-tool-asm version in a subprocess (so a crash on a partial
@@ -978,7 +1036,8 @@ With -profile, the static basic-block structure is listed for each function.
With -call, a single function is invoked with user-supplied buffers (-buf)
instead of the smoke/abi/fuzz sweeps. Useful for partial functions (e.g.
decoders) that crash on random input but should succeed on valid data.
decoders) that crash on random input but should succeed on valid data. The
function named must be NOSPLIT: a function with a stack frame is refused.
With -save-corpus (and -fuzz), every input that crashes or mismatches is
written to the directory as replayable JSON. -replay re-runs saved
@@ -1007,20 +1066,16 @@ each entry reproduces.
path := set.Arg(0)
targetArch := arch.FromFilename(path)
// JIT execution runs when the host CPU matches the kernel's
// architecture, except loong64: its trampoline is implemented but not
// yet validated against real hardware (the Go runtime cannot start
// under the available loong64 emulators), so those kernels take the
// toolchain-comparison path.
if targetArch != hostArch() || targetArch == arch.LOONG64 {
// No JIT on this host: ground truth and profile remain available.
// (loong64 is ground-truth-only everywhere for now: its trampoline
// is implemented but not yet validated against real hardware.)
// architecture; every trampoline is validated end to end under
// qemu-user emulation (the loong64 one included, via the raw-address
// leave handoff).
if targetArch != hostArch() {
// No JIT on this host: ground truth and profile remain available for
// every architecture, because cmdVerifyNonJIT assembles and compares
// against the toolchain without executing anything.
switch targetArch {
case arch.RISCV, arch.LOONG64, arch.ARM64:
case arch.AMD64, arch.RISCV, arch.LOONG64, arch.ARM64:
return cmdVerifyNonJIT(path, targetArch, *groundTruth, *profile)
case arch.AMD64:
fmt.Fprintln(os.Stderr, "gasm verify: JIT-based checks need an amd64 host; use --ground-truth here")
return 1
default:
fmt.Fprintln(os.Stderr, "gasm verify: unsupported architecture")
return 1
+112
View File
@@ -13,6 +13,9 @@ import (
"strings"
"syscall"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
)
const clean = "#include \"textflag.h\"\n" +
@@ -242,6 +245,96 @@ func TestCmdArgErrors(t *testing.T) {
}
}
// TestUsageExitCodes pins the exit-code contract for the commands whose main
// dispatches on a returned error: a wrong argument set exits 2, the same as
// the commands that count their arguments themselves, while a runtime
// failure (an unreadable file) keeps exit 1.
func TestUsageExitCodes(t *testing.T) {
for name, err := range map[string]error{
"audit-instructions extra argument": cmdAuditInstructions([]string{"amd64", "extra"}),
"audit-instructions unknown arch": cmdAuditInstructions([]string{"mips"}),
"audit-instructions corpus extra": cmdAuditInstructions([]string{"--corpus", "a", "b"}),
"scaffold no arguments": cmdScaffold(nil),
"scaffold extra arguments": cmdScaffold([]string{"differential", "a.s", "b.s"}),
} {
if err == nil {
t.Errorf("%s: expected an error", name)
continue
}
if code := exitCodeFor(err); code != 2 {
t.Errorf("%s: exit code = %d, want 2 (err: %v)", name, code, err)
}
}
if err := cmdScaffold([]string{"differential", "/nonexistent/file.s"}); err == nil {
t.Error("scaffold on a missing file should fail")
} else if code := exitCodeFor(err); code != 1 {
t.Errorf("scaffold on a missing file: exit code = %d, want 1", code)
}
}
// TestCmdAsmFormatValidation checks that an unknown --format exits 2 with
// and without -o, instead of assembling and silently dumping a raw image.
func TestCmdAsmFormatValidation(t *testing.T) {
path := writeTemp(t, "f_amd64.s", clean)
out := filepath.Join(t.TempDir(), "f.bin")
if _, _, code := capture(func() int { return cmdAsm([]string{"--format", "bogus", path}) }); code != 2 {
t.Errorf("asm --format bogus without -o: code = %d, want 2", code)
}
if _, _, code := capture(func() int { return cmdAsm([]string{"--format", "bogus", "-o", out, path}) }); code != 2 {
t.Errorf("asm --format bogus with -o: code = %d, want 2", code)
}
}
// TestCmdAsmOutputFile pins the documented -o behaviour: the output goes to
// the file and stdout carries no hex dump; without -o the dump is the output.
func TestCmdAsmOutputFile(t *testing.T) {
path := writeTemp(t, "f_amd64.s", clean)
out := filepath.Join(t.TempDir(), "f.bin")
stdout, _, code := capture(func() int { return cmdAsm([]string{"-o", out, path}) })
if code != 0 {
t.Fatalf("code = %d", code)
}
if strings.Contains(stdout, "0000:") {
t.Errorf("stdout carries a hex dump despite -o:\n%s", stdout)
}
if !strings.Contains(stdout, "wrote ") {
t.Errorf("stdout misses the wrote line:\n%s", stdout)
}
b, err := os.ReadFile(out)
if err != nil {
t.Fatal(err)
}
if len(b) == 0 {
t.Error("the output file is empty")
}
stdout, _, code = capture(func() int { return cmdAsm([]string{path}) })
if code != 0 {
t.Fatalf("without -o: code = %d", code)
}
if !strings.Contains(stdout, "0000:") {
t.Errorf("without -o the hex dump is missing:\n%s", stdout)
}
}
// TestVerifyNonJITAMD64GroundTruth drives the cross-architecture
// ground-truth path for an amd64 kernel: the path a host of any other
// architecture takes, which must compare against the toolchain rather than
// refuse to run.
func TestVerifyNonJITAMD64GroundTruth(t *testing.T) {
if testing.Short() {
t.Skip("runs go tool asm")
}
path := writeTemp(t, "f_amd64.s", clean)
out, _, code := capture(func() int { return cmdVerifyNonJIT(path, arch.AMD64, true, false) })
if code != 0 {
t.Fatalf("code = %d (%s)", code, out)
}
if !strings.Contains(out, "1/1 matched") {
t.Errorf("output misses the matched report:\n%s", out)
}
}
// TestVerifySmokeCrashIsolation checks that a function faulting on its
// zeroed smoke arguments is reported as CRASH by a child process instead of
// killing `gasm verify` itself.
@@ -348,3 +441,22 @@ func TestRunCorpusAudit(t *testing.T) {
t.Errorf("arm64 unencodable reasons = %d, want 1", r)
}
}
// TestCompareGroundTruthPadding pins the padding-aware ground-truth
// comparison: the toolchain pads text symbols to 16-byte boundaries, so
// trailing zeros in the reference must not read as a mismatch, while any
// non-zero tail still must.
func TestCompareGroundTruthPadding(t *testing.T) {
code := []byte{0x48, 0x8b, 0x07, 0xc3} // 4 bytes, not a multiple of 16
img := &asm.Image{Code: code, Funcs: []asm.FuncLayout{{Name: "f", Offset: 0, Size: len(code)}}}
padded := append(append([]byte(nil), code...), 0, 0, 0)
matched, total, diffs := compareGroundTruth(img, map[string][]byte{"f": padded})
if matched != 1 || total != 1 || diffs != 0 {
t.Fatalf("zero padding should match: matched=%d total=%d diffs=%d", matched, total, diffs)
}
dirty := append(append([]byte(nil), code...), 0, 0x90, 0)
matched, _, diffs = compareGroundTruth(img, map[string][]byte{"f": dirty})
if matched != 0 || diffs != 1 {
t.Fatalf("non-zero padding must mismatch: matched=%d diffs=%d", matched, diffs)
}
}
+84
View File
@@ -62,6 +62,42 @@ func TestManPagesTrackTheCLI(t *testing.T) {
}
}
// TestManCommandsTrackHelp compares the gasm(1) COMMANDS list with the
// top-level help output, so a subcommand added to the binary cannot miss
// its man entry and a stale entry cannot outlive its command.
func TestManCommandsTrackHelp(t *testing.T) {
if testing.Short() {
t.Skip("builds the gasm binary")
}
bin := filepath.Join(t.TempDir(), "gasm")
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
t.Fatalf("build gasm: %v\n%s", err, out)
}
raw, err := os.ReadFile(filepath.Join("..", "..", "docs", "man", "gasm.1"))
if err != nil {
t.Fatalf("read man page: %v", err)
}
helpOut, err := exec.Command(bin, "--help").Output()
if err != nil {
t.Fatalf("gasm --help: %v", err)
}
binCmds := helpCommands(string(helpOut))
pageCmds := roffCommands(string(raw))
for c := range binCmds {
if !pageCmds[c] {
t.Errorf("command %q is in the binary's help but missing from gasm(1) COMMANDS", c)
}
}
for c := range pageCmds {
if !binCmds[c] {
t.Errorf("command %q is in gasm(1) COMMANDS but the binary does not list it", c)
}
}
}
// helpFlags extracts the flag names from a `gasm <cmd> -h` output.
func helpFlags(help string) map[string]bool {
m := map[string]bool{}
@@ -123,6 +159,54 @@ func roffFlags(page string) map[string]bool {
return m
}
// helpCommands extracts the command names from the top-level help output's
// Commands section.
func helpCommands(help string) map[string]bool {
m := map[string]bool{}
inCmds := false
for line := range strings.SplitSeq(help, "\n") {
if strings.TrimSpace(line) == "Commands:" {
inCmds = true
continue
}
if !inCmds {
continue
}
t := strings.TrimSpace(line)
if t == "" {
break
}
name, _, _ := strings.Cut(t, " ")
m[name] = true
}
return m
}
// roffCommands extracts the command names from gasm(1)'s COMMANDS section,
// where each entry is written as `.B gasm\-<name>(1)` or `.B gasm <name>`.
func roffCommands(page string) map[string]bool {
m := map[string]bool{}
inCmds := false
for line := range strings.SplitSeq(page, "\n") {
if strings.HasPrefix(line, ".SH ") {
inCmds = strings.HasPrefix(line, ".SH COMMANDS")
continue
}
if !inCmds || !strings.HasPrefix(line, ".B gasm") {
continue
}
entry := strings.ReplaceAll(strings.TrimPrefix(line, ".B "), `\-`, "-")
entry = strings.TrimSuffix(entry, "(1)")
switch {
case strings.HasPrefix(entry, "gasm-"):
m[strings.TrimPrefix(entry, "gasm-")] = true
case strings.HasPrefix(entry, "gasm "):
m[strings.TrimPrefix(entry, "gasm ")] = true
}
}
return m
}
// helpUsage returns the command's usage line without the "Usage: " prefix.
func helpUsage(help string) string {
for line := range strings.SplitSeq(help, "\n") {
+1 -1
View File
@@ -44,7 +44,7 @@ bodies, place the file in the kernel's package, and run it in CI.
rest = rest[1:]
}
if len(rest) != 1 {
return fmt.Errorf("usage: gasm scaffold differential <file.s>")
return &usageError{fmt.Errorf("usage: gasm scaffold differential <file.s>")}
}
path := rest[0]
src, err := os.ReadFile(path)
+99 -44
View File
@@ -9,11 +9,11 @@ import "strings"
import "fmt"
// Breakpoint is one INT3 breakpoint in the debuggee.
// Breakpoint is one software breakpoint in the debuggee.
type Breakpoint struct {
Addr uint64 // absolute address in the debuggee
Label string // source label ("" for raw addresses)
Orig byte // original byte at Addr (restored on removal)
Orig []byte // original bytes at Addr (restored on removal)
Enabled bool
Cond *Condition // optional condition (nil = unconditional)
hits int
@@ -32,8 +32,12 @@ type Condition struct {
MemAddr uint64 // memory address (for register-memory comparison, prefixed with *)
}
// Eval checks the condition against the current registers.
func (c *Condition) Eval(regs *Regs) bool {
// Eval checks the condition against the current registers. For the
// register-memory form, mem reads an 8-byte little-endian word from the
// debuggee; it may be nil when no reader is available. Anything that cannot
// be decided (unknown register or operator, unreadable memory) does not
// block the breakpoint.
func (c *Condition) Eval(regs *Regs, mem func(addr uint64) (uint64, bool)) bool {
actual, ok := regs.RegValue(c.Reg)
if !ok {
return true // unknown register, don't block
@@ -48,9 +52,16 @@ func (c *Condition) Eval(regs *Regs) bool {
}
expected = v
case c.MemAddr != 0:
// Register-memory comparison, requires a Session, not available here.
// Fall back to treating as constant (the caller should resolve).
expected = c.Value
// Register-memory comparison, resolved in the debuggee at
// evaluation time.
if mem == nil {
return true
}
v, ok := mem(c.MemAddr)
if !ok {
return true
}
expected = v
default:
expected = c.Value
}
@@ -72,6 +83,18 @@ func (c *Condition) Eval(regs *Regs) bool {
}
}
// String renders the condition for display.
func (c *Condition) String() string {
switch {
case c.Reg2 != "":
return fmt.Sprintf("%s %s %s", c.Reg, c.Op, c.Reg2)
case c.MemAddr != 0:
return fmt.Sprintf("%s %s *%#x", c.Reg, c.Op, c.MemAddr)
default:
return fmt.Sprintf("%s %s %#x", c.Reg, c.Op, c.Value)
}
}
// Breakpoints manages the software breakpoints of one Session.
type Breakpoints struct {
t tracer
@@ -83,6 +106,18 @@ func NewBreakpoints(t tracer) *Breakpoints {
return &Breakpoints{t: t, bps: make(map[uint64]*Breakpoint)}
}
// breakpointMask is the byte mask of the breakpoint instruction inside a
// peeked word: the low len(breakpointInsn) bytes, because every supported
// architecture is little-endian and patches the instruction at the lowest
// address of the word.
func breakpointMask() uint64 {
var mask uint64
for range breakpointInsn {
mask = (mask << 8) | 0xFF
}
return mask
}
// Set installs a breakpoint at addr (replaces any existing one).
func (bm *Breakpoints) Set(addr uint64, label string) (*Breakpoint, error) {
return bm.SetWithCond(addr, label, nil)
@@ -100,13 +135,12 @@ func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (
if err != nil {
return nil, err
}
orig := byte(word)
// Patch with the breakpoint instruction, preserving the rest of the word.
mask := uint64(0)
for range breakpointInsn {
mask = (mask << 8) | 0xFF
orig := make([]byte, len(breakpointInsn))
for i := range orig {
orig[i] = byte(word >> (8 * i))
}
patched := (word &^ mask) | breakpointWord(breakpointInsn)
// Patch with the breakpoint instruction, preserving the rest of the word.
patched := (word &^ breakpointMask()) | breakpointWord(breakpointInsn)
if err := bm.t.Poke(addr, patched); err != nil {
return nil, err
}
@@ -134,26 +168,40 @@ func (bm *Breakpoints) Info() string {
}
cond := ""
if bp.Cond != nil {
cond = fmt.Sprintf(" if %s %s %#x", bp.Cond.Reg, bp.Cond.Op, bp.Cond.Value)
cond = " if " + bp.Cond.String()
}
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond))
}
return result.String()
}
// Clear removes the breakpoint at addr, restoring the original byte.
// restore writes the saved original bytes back over the breakpoint
// instruction, preserving the rest of the peeked word. It reports whether
// both the peek and the poke succeeded.
func (bm *Breakpoints) restore(addr uint64, bp *Breakpoint) bool {
word, err := bm.t.Peek(addr)
if err != nil {
return false
}
orig := uint64(0)
for i, b := range bp.Orig {
orig |= uint64(b) << (8 * i)
}
return bm.t.Poke(addr, (word&^breakpointMask())|orig) == nil
}
// Clear removes the breakpoint at addr, restoring the original bytes.
func (bm *Breakpoints) Clear(addr uint64) error {
bp, ok := bm.bps[addr]
if !ok {
return fmt.Errorf("debug: no breakpoint at %#x", addr)
}
word, err := bm.t.Peek(addr)
if err != nil {
return err
}
restored := (word &^ 0xFF) | uint64(bp.Orig)
if err := bm.t.Poke(addr, restored); err != nil {
return err
if !bm.restore(addr, bp) {
word, err := bm.t.Peek(addr)
if err != nil {
return err
}
return fmt.Errorf("debug: restore breakpoint at %#x failed, word is %#x", addr, word)
}
delete(bm.bps, addr)
return nil
@@ -185,43 +233,54 @@ func (bm *Breakpoints) All() []*Breakpoint {
// HandleTrap is called after the debuggee stops on SIGTRAP. It checks
// whether the trap was caused by one of our breakpoints (PC-adjust matches
// a breakpoint address), restores the original byte, rewinds PC, and
// a breakpoint address), restores the original bytes, rewinds PC, and
// returns the breakpoint that was hit (or nil if it was a single-step).
// Hits returns how many times the breakpoint has been hit.
func (bp *Breakpoint) Hits() int { return bp.hits }
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
// After a breakpoint trap, PC points past the breakpoint instruction.
// On amd64 the kernel reports the trap with RIP past the INT3; on the
// other supported architectures the PC still stands on the trap
// instruction, which breakpointPCAdjust encodes per architecture.
trapAddr := regs.GetPC() - uint64(breakpointPCAdjust)
bp, ok := bm.bps[trapAddr]
if !ok || !bp.Enabled {
return nil // single-step trap or unknown
}
// Check the condition (if any).
if bp.Cond != nil && !bp.Cond.Eval(regs) {
// Condition not met, restore the byte but do NOT rewind RIP.
// The process continues from the next instruction (past the INT3).
word, err := bm.t.Peek(trapAddr)
if err == nil {
restored := (word &^ 0xFF) | uint64(bp.Orig)
bm.t.Poke(trapAddr, restored)
if bp.Cond != nil && !bp.Cond.Eval(regs, bm.peekValue) {
// Condition not met: step the original instruction and re-arm the
// breakpoint, leaving the debuggee stopped just past it, ready to
// resume silently. The PC must be rewound first: on architectures
// that report the trap past the instruction (amd64) it would
// otherwise sit on the second byte of the replaced instruction.
if !bm.restore(trapAddr, bp) {
return nil
}
// RIP is already past the INT3 (trapAddr + 1). Don't rewind.
regs.SetPC(trapAddr)
if err := bm.t.SetRegs(regs); err != nil {
return nil
}
if err := bm.t.Step(); err != nil {
return nil
}
bm.Reinsert(trapAddr)
return nil
}
bp.hits++
// Restore the original byte.
word, err := bm.t.Peek(trapAddr)
if err == nil {
restored := (word &^ 0xFF) | uint64(bp.Orig)
bm.t.Poke(trapAddr, restored)
}
// Rewind PC to re-execute the original instruction.
// Restore the original bytes and rewind PC to re-execute them.
bm.restore(trapAddr, bp)
regs.SetPC(trapAddr)
bm.t.SetRegs(regs)
return bp
}
// peekValue adapts tracer.Peek to the Condition value reader.
func (bm *Breakpoints) peekValue(addr uint64) (uint64, bool) {
v, err := bm.t.Peek(addr)
return v, err == nil
}
// Reinsert re-inserts the breakpoint at addr after a single-step past it.
// Called after Step() when we want the breakpoint to fire again on the
// next Continue().
@@ -234,11 +293,7 @@ func (bm *Breakpoints) Reinsert(addr uint64) error {
if err != nil {
return err
}
mask := uint64(0)
for range breakpointInsn {
mask = (mask << 8) | 0xFF
}
patched := (word &^ mask) | breakpointWord(breakpointInsn)
patched := (word &^ breakpointMask()) | breakpointWord(breakpointInsn)
return bm.t.Poke(addr, patched)
}
+265
View File
@@ -0,0 +1,265 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
package debug
// Architecture-neutral tests: label and line tables, and the breakpoint
// manager against the mock tracer. These do not launch a debuggee, so they
// build on every supported linux architecture.
import (
"strings"
"testing"
)
func TestLineAt(t *testing.T) {
lines := []SourceLine{
{Offset: 0, Line: 5},
{Offset: 5, Line: 6},
{Offset: 10, Line: 7},
{Offset: 15, Line: 8},
}
tests := []struct {
offset int
want int
}{
{0, 5},
{1, 5},
{4, 5},
{5, 6},
{7, 6},
{10, 7},
{12, 7},
{15, 8},
{20, 8},
}
for _, tt := range tests {
got := lineAt(lines, tt.offset)
if got != tt.want {
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
}
}
// Empty table.
if lineAt(nil, 5) != 0 {
t.Error("lineAt(nil, 5) should return 0")
}
}
func TestOffsetForLine(t *testing.T) {
lines := []SourceLine{
{Offset: 0, Line: 5},
{Offset: 5, Line: 6},
{Offset: 10, Line: 7},
}
tests := []struct {
line int
want int
}{
{5, 0},
{6, 5},
{7, 10},
{99, -1}, // not found
{0, -1}, // not found
}
for _, tt := range tests {
got := offsetForLine(lines, tt.line)
if got != tt.want {
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
}
}
}
func TestNearestLabel(t *testing.T) {
labels := []Label{
{Name: "start", Offset: 0},
{Name: "loop", Offset: 10},
{Name: "done", Offset: 20},
}
tests := []struct {
offset int
want string
}{
{0, "start"},
{5, "start"},
{10, "loop"},
{15, "loop"},
{20, "done"},
{25, "done"},
}
for _, tt := range tests {
got := nearestLabel(labels, tt.offset)
if got != tt.want {
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
}
}
}
func TestBreakpointsSetAndClear(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
// Set a breakpoint at address 0x1000.
bp, err := bm.Set(0x1000, "test")
if err != nil {
t.Fatalf("Set: %v", err)
}
if !bp.Enabled {
t.Error("breakpoint not enabled")
}
if bp.Label != "test" {
t.Errorf("label = %q, want test", bp.Label)
}
// Verify Peek was called.
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
}
// Verify Poke wrote the breakpoint instruction's bytes.
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
t.Errorf("pokes = %v", tr.pokes)
}
if got := tr.pokes[0].val & breakpointMask(); got != breakpointWord(breakpointInsn) {
t.Errorf("patched bytes %#x, want %#x", got, breakpointWord(breakpointInsn))
}
// At should find it.
if bm.At(0x1000) == nil {
t.Error("At(0x1000) returned nil")
}
// All should return it.
all := bm.All()
if len(all) != 1 {
t.Errorf("All() = %d breakpoints, want 1", len(all))
}
// Clear it.
if err := bm.Clear(0x1000); err != nil {
t.Fatalf("Clear: %v", err)
}
if bm.At(0x1000) != nil {
t.Error("At(0x1000) after Clear should be nil")
}
}
// TestBreakpointRestoreWidth proves the restore path writes back every
// byte of the breakpoint instruction's width, not just the first byte: on
// arm64, riscv64 and loong64 the instruction is four bytes, and restoring
// one byte would leave three bytes of the trap instruction in place.
func TestBreakpointRestoreWidth(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
tr.mem[0x3000] = 0x11
tr.mem[0x3001] = 0x22
tr.mem[0x3002] = 0x33
tr.mem[0x3003] = 0x44
if _, err := bm.Set(0x3000, "width"); err != nil {
t.Fatalf("Set: %v", err)
}
for i, b := range breakpointInsn {
if tr.mem[0x3000+uint64(i)] != b {
t.Fatalf("byte %d after Set = %#x, want the breakpoint byte %#x", i, tr.mem[0x3000+uint64(i)], b)
}
}
if len(bm.At(0x3000).Orig) != len(breakpointInsn) {
t.Fatalf("Orig holds %d bytes, want %d", len(bm.At(0x3000).Orig), len(breakpointInsn))
}
if err := bm.Clear(0x3000); err != nil {
t.Fatalf("Clear: %v", err)
}
want := []byte{0x11, 0x22, 0x33, 0x44}
for i, b := range want {
if tr.mem[0x3000+uint64(i)] != b {
t.Errorf("byte %d after Clear = %#x, want %#x (restore must cover the full instruction width)", i, tr.mem[0x3000+uint64(i)], b)
}
}
}
func TestBreakpointsSetWithCond(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
if err != nil {
t.Fatalf("SetWithCond: %v", err)
}
if bp.Cond == nil || bp.Cond.Value != 42 {
t.Error("condition not set")
}
// Re-setting the same address should update the condition.
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
if err != nil {
t.Fatalf("SetWithCond (update): %v", err)
}
if bp2.Cond.Value != 100 {
t.Error("condition not updated")
}
// Should have only 1 Peek (first Set), second is update (no Peek needed).
if len(tr.peeks) != 1 {
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
}
}
func TestBreakpointsClearAll(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
bm.Set(0x1000, "a")
bm.Set(0x2000, "b")
bm.Set(0x3000, "c")
if len(bm.All()) != 3 {
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
}
bm.ClearAll()
if len(bm.All()) != 0 {
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
}
}
func TestBreakpointInfo(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
bm.Set(0x4000, "info_test")
info := bm.Info()
if info == "" {
t.Error("Info returned empty string")
}
if !strings.Contains(info, "info_test") {
t.Errorf("Info %q does not contain label", info)
}
}
// TestConditionString covers the display of all three condition forms.
func TestConditionString(t *testing.T) {
tests := []struct {
cond Condition
want string
}{
{Condition{Reg: "rax", Op: "==", Value: 42}, "rax == 0x2a"},
{Condition{Reg: "rax", Op: "!=", Reg2: "rbx"}, "rax != rbx"},
{Condition{Reg: "rax", Op: "<", MemAddr: 0x5000}, "rax < *0x5000"},
}
for _, tt := range tests {
if got := tt.cond.String(); got != tt.want {
t.Errorf("Condition.String() = %q, want %q", got, tt.want)
}
}
}
+38 -188
View File
@@ -6,7 +6,6 @@
package debug
import (
"strings"
"testing"
)
@@ -48,7 +47,7 @@ func TestConditionEval(t *testing.T) {
}
for _, tt := range tests {
got := tt.cond.Eval(regs)
got := tt.cond.Eval(regs, nil)
if got != tt.want {
t.Errorf("Condition{%q %q %d}.Eval() = %v, want %v",
tt.cond.Reg, tt.cond.Op, tt.cond.Value, got, tt.want)
@@ -56,65 +55,33 @@ func TestConditionEval(t *testing.T) {
}
}
func TestLineAt(t *testing.T) {
lines := []SourceLine{
{Offset: 0, Line: 5},
{Offset: 5, Line: 6},
{Offset: 10, Line: 7},
{Offset: 15, Line: 8},
}
tests := []struct {
offset int
want int
}{
{0, 5},
{1, 5},
{4, 5},
{5, 6},
{7, 6},
{10, 7},
{12, 7},
{15, 8},
{20, 8},
}
for _, tt := range tests {
got := lineAt(lines, tt.offset)
if got != tt.want {
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
// TestConditionEvalMem covers the register-memory form: the value is read
// through the supplied reader, and a missing or failing reader must not
// block the breakpoint.
func TestConditionEvalMem(t *testing.T) {
regs := &Regs{RAX: 7}
mem := func(addr uint64) (uint64, bool) {
if addr == 0x5000 {
return 7, true
}
return 0, false
}
// Empty table.
if lineAt(nil, 5) != 0 {
t.Error("lineAt(nil, 5) should return 0")
eq := Condition{Reg: "rax", Op: "==", MemAddr: 0x5000}
if !eq.Eval(regs, mem) {
t.Error("register-memory comparison with matching word should hold")
}
}
func TestOffsetForLine(t *testing.T) {
lines := []SourceLine{
{Offset: 0, Line: 5},
{Offset: 5, Line: 6},
{Offset: 10, Line: 7},
ne := Condition{Reg: "rax", Op: "!=", MemAddr: 0x5000}
if ne.Eval(regs, mem) {
t.Error("register-memory comparison with mismatching word should not hold")
}
tests := []struct {
line int
want int
}{
{5, 0},
{6, 5},
{7, 10},
{99, -1}, // not found
{0, -1}, // not found
bad := Condition{Reg: "rax", Op: "==", MemAddr: 0x6000}
if !bad.Eval(regs, mem) {
t.Error("unreadable memory must not block the breakpoint")
}
for _, tt := range tests {
got := offsetForLine(lines, tt.line)
if got != tt.want {
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
}
noReader := Condition{Reg: "rax", Op: "==", MemAddr: 0x5000}
if !noReader.Eval(regs, nil) {
t.Error("missing memory reader must not block the breakpoint")
}
}
@@ -141,139 +108,6 @@ func TestDecodeRflags(t *testing.T) {
}
}
func TestNearestLabel(t *testing.T) {
labels := []Label{
{Name: "start", Offset: 0},
{Name: "loop", Offset: 10},
{Name: "done", Offset: 20},
}
tests := []struct {
offset int
want string
}{
{0, "start"},
{5, "start"},
{10, "loop"},
{15, "loop"},
{20, "done"},
{25, "done"},
}
for _, tt := range tests {
got := nearestLabel(labels, tt.offset)
if got != tt.want {
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
}
}
}
func TestBreakpointsSetAndClear(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
// Set a breakpoint at address 0x1000.
bp, err := bm.Set(0x1000, "test")
if err != nil {
t.Fatalf("Set: %v", err)
}
if !bp.Enabled {
t.Error("breakpoint not enabled")
}
if bp.Label != "test" {
t.Errorf("label = %q, want test", bp.Label)
}
// Verify Peek was called.
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
}
// Verify Poke wrote INT3.
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
t.Errorf("pokes = %v", tr.pokes)
}
// At should find it.
if bm.At(0x1000) == nil {
t.Error("At(0x1000) returned nil")
}
// All should return it.
all := bm.All()
if len(all) != 1 {
t.Errorf("All() = %d breakpoints, want 1", len(all))
}
// Clear it.
if err := bm.Clear(0x1000); err != nil {
t.Fatalf("Clear: %v", err)
}
if bm.At(0x1000) != nil {
t.Error("At(0x1000) after Clear should be nil")
}
}
func TestBreakpointsSetWithCond(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
if err != nil {
t.Fatalf("SetWithCond: %v", err)
}
if bp.Cond == nil || bp.Cond.Value != 42 {
t.Error("condition not set")
}
// Re-setting the same address should update the condition.
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
if err != nil {
t.Fatalf("SetWithCond (update): %v", err)
}
if bp2.Cond.Value != 100 {
t.Error("condition not updated")
}
// Should have only 1 Peek (first Set), second is update (no Peek needed).
if len(tr.peeks) != 1 {
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
}
}
func TestBreakpointsClearAll(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
bm.Set(0x1000, "a")
bm.Set(0x2000, "b")
bm.Set(0x3000, "c")
if len(bm.All()) != 3 {
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
}
bm.ClearAll()
if len(bm.All()) != 0 {
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
}
}
func TestBreakpointInfo(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
bm.Set(0x4000, "info_test")
info := bm.Info()
if info == "" {
t.Error("Info returned empty string")
}
if !strings.Contains(info, "info_test") {
t.Errorf("Info %q does not contain label", info)
}
}
func TestWatchpointSlotTracking(t *testing.T) {
s := &Session{} // per-session slots start free
@@ -323,3 +157,19 @@ func TestWatchpointSlotTracking(t *testing.T) {
t.Errorf("FindFreeWatchpointSlot() with all slots used = %d, want -1", got)
}
}
// TestUnwatchSlotBound checks the bound the REPL parses against: it must
// cover the architecture's whole slot range, not a hardcoded 0-3.
func TestUnwatchSlotBound(t *testing.T) {
max := maxWatchpoints()
if max < 4 {
t.Fatalf("maxWatchpoints() = %d, want at least 4", max)
}
s := &Session{}
if s.IsWatchpointSlotUsed(max - 1) {
t.Errorf("slot %d should be free initially", max-1)
}
if s.IsWatchpointSlotUsed(max) {
t.Errorf("slot %d must be out of range", max)
}
}
+8
View File
@@ -46,3 +46,11 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
}
return result.String()
}
// isCallInsn reports whether disassembled text (x86asm.IntelSyntax) is a
// call. The first token must match exactly: a prefix test would also catch
// unrelated mnemonics.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
return strings.ToLower(m) == "call"
}
+13
View File
@@ -7,6 +7,7 @@ package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
@@ -44,3 +45,15 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
}
return result
}
// isCallInsn reports whether disassembled text (arm64asm.GoSyntax) is a
// call. GoSyntax renders bl as CALL; the native mnemonic is accepted too.
// The first token must match exactly so branches never match.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "bl":
return true
}
return false
}
+14
View File
@@ -7,6 +7,7 @@ package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
@@ -44,3 +45,16 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
}
return result
}
// isCallInsn reports whether disassembled text (loong64asm.GoSyntax) is a
// call. GoSyntax renders bl and jirl calls as CALL (jirl returns print
// RET); the native mnemonics are accepted too. The first token must match
// exactly: a "bl" prefix would catch bltz and other branches.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "bl", "jirl":
return true
}
return false
}
+15
View File
@@ -7,6 +7,7 @@ package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
@@ -44,3 +45,17 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
}
return result
}
// isCallInsn reports whether disassembled text (riscv64asm.GoSyntax) is a
// call. GoSyntax renders jal and jalr calls as CALL; the native mnemonics
// are accepted too. The first token must match exactly: a prefix test on
// "bl" would catch branches on other architectures, and jalr as ret prints
// RET, which must not be stepped over.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "jal", "jalr":
return true
}
return false
}
+22 -2
View File
@@ -73,9 +73,29 @@ func decodeRflags(f uint64) string {
return flags[:len(flags)-1]
}
// archReturnAddr reads the return address from the stack (amd64 ABI0 convention).
// archReturnAddr reads the return address of the current frame (amd64
// ABI0 convention). A function that contains a CALL (or has a frame) is
// assembled with the prologue PUSHQ BP; MOVQ SP, BP, so mid-function the
// word at SP is the saved caller BP, a stack address, and the return
// address sits further up. Walk the stack from SP and take the first word
// that lies in an executable mapping: stack and data words never do, a
// return address always does.
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return s.Peek(regs.GetSP())
ranges := execRanges(s.pid)
for off := uint64(0); off < 512; off += 8 {
word, err := s.Peek(regs.RSP + off)
if err != nil {
break
}
for _, r := range ranges {
if word >= r.lo && word < r.hi {
return word, nil
}
}
}
// No mapping available or nothing code-like on the stack: fall back to
// the raw entry convention, [SP] before any push.
return s.Peek(regs.RSP)
}
// archSPLabel returns the SP register name for display.
+6 -3
View File
@@ -5,7 +5,10 @@
package debug
import "fmt"
import (
"encoding/binary"
"fmt"
)
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
@@ -31,8 +34,8 @@ func printRegs(regs *Regs, codeBase, funcOff uint64) {
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n Vector registers (V0-V31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" V%-2d = %016x%016x\n", i, v.V[i][8], v.V[i][0])
fmt.Printf(" V%-2d = %016x%016x\n", i+1, v.V[i+1][8], v.V[i+1][0])
fmt.Printf(" V%-2d = %016x%016x\n", i, binary.LittleEndian.Uint64(v.V[i][8:16]), binary.LittleEndian.Uint64(v.V[i][0:8]))
fmt.Printf(" V%-2d = %016x%016x\n", i+1, binary.LittleEndian.Uint64(v.V[i+1][8:16]), binary.LittleEndian.Uint64(v.V[i+1][0:8]))
}
}
+427
View File
@@ -0,0 +1,427 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug
import (
"bytes"
"fmt"
"io"
"os"
"path/filepath"
"runtime"
"strings"
"testing"
"time"
"unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
)
// Integration tests beyond the basic entry breakpoint: hardware watchpoints,
// conditional breakpoints, next/finish over a CALL, faulting kernels and the
// xstate vector-register readout. All drive a real ptrace session, so they
// run on amd64 hosts only.
// writeKernel writes an assembly source to a temporary file with the
// architecture suffix the assembler dispatcher expects.
func writeKernel(t *testing.T, src string) string {
t.Helper()
path := filepath.Join(t.TempDir(), "kernel_amd64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatalf("write kernel: %v", err)
}
return path
}
// launchKernel launches a session for the kernel source and returns the
// session, its breakpoint manager and the function layout.
func launchKernel(t *testing.T, bin, path, funcName string, args []byte) (*Session, *Breakpoints, asm.FuncLayout) {
t.Helper()
k, err := verify.Load(path)
if err != nil {
t.Fatalf("Load: %v", err)
}
t.Cleanup(k.Close)
fl, err := k.Func(funcName)
if err != nil {
t.Fatalf("Func: %v", err)
}
if len(args) < fl.Args {
padded := make([]byte, fl.Args)
copy(padded, args)
args = padded
}
sess, err := Launch(bin, path, funcName, args)
if err != nil {
t.Fatalf("Launch: %v", err)
}
t.Cleanup(sess.Kill)
bm := NewBreakpoints(sess)
return sess, bm, fl
}
// runToEntry resumes the freshly launched debuggee until the breakpoint at
// the function entry traps, mirroring the REPL continue loop: the debuggee
// SIGSTOPs twice (launch barrier and entry barrier) before entering the JIT
// call.
func runToEntry(t *testing.T, sess *Session, bm *Breakpoints, entry uint64) {
t.Helper()
for range 50 {
for _, bp := range bm.All() {
bm.Reinsert(bp.Addr)
}
if err := sess.Continue(); err != nil {
t.Fatalf("Continue: %v", err)
}
if sess.Exited() {
t.Fatal("debuggee exited before the entry breakpoint trapped")
}
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if bm.HandleTrap(&regs) != nil {
return
}
}
t.Fatal("no entry breakpoint trap after 50 resumes")
}
// captureStdout runs fn with os.Stdout redirected to a pipe and returns
// what it printed (the REPL writes its reports to stdout).
func captureStdout(t *testing.T, fn func()) string {
t.Helper()
r, w, err := os.Pipe()
if err != nil {
t.Fatalf("pipe: %v", err)
}
old := os.Stdout
os.Stdout = w
done := make(chan string, 1)
go func() {
b, _ := io.ReadAll(r)
done <- string(b)
}()
defer func() { os.Stdout = old }()
fn()
w.Close()
return <-done
}
// TestWatchpointArmRunHit proves the debug-register offsets: the watchpoint
// must fire on the store, with si_addr naming the watched address. The
// kernel writes its return value to ret+0(FP), which is the 8-byte word
// right above the stack pointer at entry.
func TestWatchpointArmRunHit(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func wpret() int64
TEXT ·wpret(SB), NOSPLIT, $0-8
MOVQ $0x5a5a5a5a5a5a5a5a, AX
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "wpret", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
watched := regs.RSP + 8 // ret+0(FP): the store target
slot := sess.FindFreeWatchpointSlot()
if slot < 0 {
t.Fatal("no free watchpoint slot")
}
if err := sess.SetWatchpoint(slot, watched, WatchWrite, 8); err != nil {
t.Fatalf("SetWatchpoint: %v (wrong debug-register offsets?)", err)
}
if err := sess.Continue(); err != nil {
t.Fatalf("Continue: %v", err)
}
reason, addr := sess.StopInfo()
if reason != StopWatchpoint {
t.Fatalf("stop reason = %v, want StopWatchpoint (DR0-DR3/DR7 offsets are wrong)", reason)
}
if addr != watched {
t.Fatalf("watchpoint address = %#x, want %#x", addr, watched)
}
// The watched word holds the stored value: x86 data breakpoints are
// reported with the access complete.
if word, err := sess.Peek(watched); err != nil || word != 0x5a5a5a5a5a5a5a5a {
t.Errorf("watched word = %#x (err %v), want 0x5a5a5a5a5a5a5a5a", word, err)
}
if err := sess.ClearWatchpoint(slot); err != nil {
t.Fatalf("ClearWatchpoint: %v", err)
}
}
// TestConditionalBreakpointFalseThenTrue proves the false-condition path:
// the breakpoint steps over the original instruction, re-arms itself and
// keeps running silently, and the true condition stops exactly once with the
// register in the expected state.
func TestConditionalBreakpointFalseThenTrue(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func countdown(n int64) int64
TEXT ·countdown(SB), NOSPLIT, $0-16
MOVQ n+0(FP), CX
loop:
DECQ CX
CMPQ CX, $0
JNE loop
MOVQ CX, ret+8(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "countdown", []byte{8})
loopAddr := sess.CodeBase() + uint64(fl.Offset) + uint64(fl.Labels["loop"])
// The length of the breakpointed instruction, from a disassembly taken
// before the INT3 is patched in.
_, insnLen, err := sess.Disassemble(loopAddr)
if err != nil || insnLen <= 0 {
t.Fatalf("Disassemble at %#x: len=%d err=%v", loopAddr, insnLen, err)
}
cond := &Condition{Reg: "rcx", Op: "==", Value: 1}
bp, err := bm.SetWithCond(loopAddr, "loop", cond)
if err != nil {
t.Fatalf("SetWithCond: %v", err)
}
hits := 0
exited := false
for range 200 {
for _, b := range bm.All() {
bm.Reinsert(b.Addr)
}
if err := sess.Continue(); err != nil {
exited = true
break // the debuggee finished
}
if sess.Exited() {
exited = true
break
}
if sig := sess.LastSignal(); sig != 0 {
t.Fatalf("unexpected signal stop %v", sig)
}
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if hit := bm.HandleTrap(&regs); hit != nil {
hits++
if regs.RCX != 1 {
t.Fatalf("hit with RCX=%d, want 1", regs.RCX)
}
// Park after the instruction, as the REPL does.
if err := sess.Step(); err != nil {
t.Fatalf("Step: %v", err)
}
} else {
// A false evaluation must leave the debuggee past the whole
// original instruction: a PC inside it (trapAddr+1 on amd64)
// means the resume happens mid-instruction.
fresh, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if fresh.RIP > loopAddr && fresh.RIP < loopAddr+uint64(insnLen) {
t.Fatalf("false evaluation left the PC at %#x, inside the %d-byte instruction at %#x",
fresh.RIP, insnLen, loopAddr)
}
}
}
if hits != 1 {
t.Fatalf("conditional breakpoint hit %d times, want exactly 1 (false evaluations must run through silently)", hits)
}
if bp.Hits() != 1 {
t.Errorf("bp.Hits() = %d, want 1", bp.Hits())
}
if !exited || !sess.Exited() {
t.Fatal("debuggee did not run to completion after the conditional hit")
}
}
// TestNextAndFinishOverCall proves next and finish evaluate the trap with
// registers fetched after the stop: next lands exactly on the instruction
// after the CALL, and finish stops exactly on the return address.
func TestNextAndFinishOverCall(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func caller(x int64) int64
// The argument travels in AX: FP argument slots of CALL-bearing functions
// are an assembler concern outside this test's scope.
TEXT ·caller(SB), NOSPLIT, $0-16
MOVQ $5, AX
CALL ·bump(SB)
aftercall:
MOVQ AX, ret+8(FP)
RET
// func bump(x int64) int64
TEXT ·bump(SB), NOSPLIT, $0-0
ADDQ $3, AX
RET
`
path := writeKernel(t, kernel)
// next: step the prologue and the constant load (3 instructions), then
// step over the CALL and check the landing address and RAX.
sess, bm, fl := launchKernel(t, bin, path, "caller", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
afterOff := uint64(fl.Labels["aftercall"])
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, nil, nil,
strings.NewReader("step 3\nnext\nregs\nquit\n"))
})
if !strings.Contains(out, fmt.Sprintf("func+%#x", afterOff)) {
t.Errorf("next did not land on the instruction after the CALL (func+%#x); output:\n%s", afterOff, out)
}
if !strings.Contains(out, "RAX = 0x0000000000000008") {
t.Errorf("callee did not run exactly once under next (want RAX=8); output:\n%s", out)
}
// finish: run to the return address read off the stack at entry.
sess2, bm2, fl2 := launchKernel(t, bin, path, "caller", nil)
entry2 := sess2.CodeBase() + uint64(fl2.Offset)
if _, err := bm2.Set(entry2, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess2, bm2, entry2)
regs, err := sess2.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
retAddr, err := sess2.Peek(regs.RSP)
if err != nil {
t.Fatalf("Peek return address: %v", err)
}
out2 := captureStdout(t, func() {
REPL(sess2, bm2, sess2.CodeBase(), fl2.Offset, fl2.Size, fl2.Args, nil, nil,
strings.NewReader("step 1\nfinish\nquit\n"))
})
want := fmt.Sprintf("finished, now at %#x\n", retAddr)
if !strings.Contains(out2, want) {
t.Errorf("finish stopped at the wrong PC; want %q in output:\n%s", want, out2)
}
}
// TestSignalStopSurfaced proves a faulting kernel surfaces as a reported
// stop instead of an infinite fault loop. A regression here hangs, so a
// watchdog fails the run rather than letting CI stall.
func TestSignalStopSurfaced(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func crash() int64
TEXT ·crash(SB), NOSPLIT, $0-8
XORQ AX, AX
MOVQ (AX), AX
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, _ := launchKernel(t, bin, path, "crash", nil)
timer := time.AfterFunc(time.Minute, func() {
panic("watchdog: the debugger hung on the faulting kernel instead of reporting the signal stop")
})
defer timer.Stop()
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), 0, 0, 0, nil, nil,
strings.NewReader("continue\nquit\n"))
})
if !strings.Contains(out, "stopped on signal") {
t.Errorf("SIGSEGV did not surface as a reported stop; output:\n%s", out)
}
if !sess.Exited() {
t.Error("debuggee should be killed by quit after the signal stop")
}
}
// TestGetVectorRegsXState proves the NT_X86_XSTATE readout: the request
// succeeds on a normal process and the XMM halves agree with
// PTRACE_GETFPREGS.
func TestGetVectorRegsXState(t *testing.T) {
// The FPRegs layout must mirror the kernel's user_fpregs_struct
// exactly: PTRACE_GETFPREGS fills all 512 bytes, so a short struct
// overflows the caller's memory.
if got := unsafe.Sizeof(FPRegs{}); got != 512 {
t.Fatalf("sizeof(FPRegs) = %d, want 512", got)
}
if got := unsafe.Offsetof(FPRegs{}.XMM); got != 160 {
t.Fatalf("offsetof(FPRegs.XMM) = %d, want 160", got)
}
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func vprobe() int64
TEXT ·vprobe(SB), NOSPLIT, $0-8
MOVQ $1, AX
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "vprobe", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
v, err := sess.GetVectorRegs()
if err != nil {
t.Fatalf("GetVectorRegs: %v", err)
}
fp, err := sess.GetFPRegs()
if err != nil {
t.Fatalf("GetFPRegs: %v", err)
}
for i := range 16 {
if !bytes.Equal(v.YMM[i][:16], fp.XMM[i][:]) {
t.Errorf("YMM%d low half %x, want the FPRegs XMM half %x", i, v.YMM[i][:16], fp.XMM[i][:])
}
}
}
+66 -28
View File
@@ -23,7 +23,13 @@ type Session struct {
stopped bool
exited bool
codeBase uint64 // base address of the JIT code in the debuggee
wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 BADVR0-15)
tmpDir string // scratch directory of the session, removed on Kill
wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 DBGWVR0-15)
// lastSignal holds the signal of the most recent stop when that stop
// was a genuine signal-delivery-stop the caller must see (a fault such
// as SIGSEGV, SIGBUS, SIGFPE or SIGILL); 0 for breakpoint traps,
// single-steps, SIGSTOP and suppressed runtime signals.
lastSignal syscall.Signal
}
// Launch starts the debuggee subprocess (gasm debug --target ...) and
@@ -78,7 +84,7 @@ func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec s
return nil, nil, fmt.Errorf("debug: start debuggee: %w", err)
}
s := &Session{pid: cmd.Process.Pid, cmd: cmd}
s := &Session{pid: cmd.Process.Pid, cmd: cmd, tmpDir: tmpDir}
readyFile := filepath.Join(tmpDir, "ready")
for range 500 {
@@ -125,29 +131,17 @@ func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec s
return s, bufAddrs, nil
}
// wait waits for the debuggee to stop and returns the wait status.
func (s *Session) wait() error {
var ws syscall.WaitStatus
_, err := syscall.Wait4(s.pid, &ws, 0, nil)
if err != nil {
return err
}
if ws.Exited() {
s.exited = true
return fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
}
s.stopped = true
return nil
}
// waitStopped consumes ptrace-stop events until one the debugger cares
// about arrives: SIGTRAP (a breakpoint or a completed single-step) or the
// debuggee's own SIGSTOP. A Go tracee's runtime raises SIGURG for
// asynchronous preemption, and every signal on a traced thread surfaces as
// a signal-delivery-stop, so those are suppressed and the tracee resumed
// without them. Runtime noise is why a single wait can return in the
// middle of runtime code and a resume can then fail: the event stream must
// be drained by the tracer.
// about arrives: SIGTRAP (a breakpoint or a completed single-step), the
// debuggee's own SIGSTOP, or a genuine signal-delivery-stop. A Go tracee's
// runtime raises SIGURG for asynchronous preemption, and every signal on a
// traced thread surfaces as a signal-delivery-stop, so SIGURG is suppressed
// and the tracee resumed without it. Every other signal (SIGSEGV, SIGBUS,
// SIGFPE, SIGILL, ...) is returned to the caller: resuming with signal 0
// would restart the faulting instruction and fault forever, so a faulting
// kernel must surface as a stop the caller reports. Runtime noise is also
// why a single wait can return in the middle of runtime code and a resume
// can then fail: the event stream must be drained by the tracer.
func (s *Session) waitStopped() (syscall.Signal, error) {
for {
var ws syscall.WaitStatus
@@ -165,10 +159,12 @@ func (s *Session) waitStopped() (syscall.Signal, error) {
switch sig := ws.StopSignal(); sig {
case syscall.SIGTRAP, syscall.SIGSTOP:
s.stopped = true
s.lastSignal = 0
return sig, nil
default:
// Runtime noise (SIGURG preemption and friends): resume the
// tracee without delivering the signal.
case syscall.SIGURG:
// Go runtime asynchronous preemption: resume the tracee
// without delivering the signal.
s.lastSignal = 0
if _, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_CONT),
@@ -177,10 +173,22 @@ func (s *Session) waitStopped() (syscall.Signal, error) {
); errno != 0 {
return 0, fmt.Errorf("debug: PTRACE_CONT: %w", errno)
}
default:
// A genuine signal-delivery-stop. Report it; the caller
// decides how to proceed.
s.stopped = true
s.lastSignal = sig
return sig, nil
}
}
}
// LastSignal returns the signal of the most recent stop when that stop was
// a genuine signal-delivery-stop (a fault such as SIGSEGV, SIGFPE, SIGILL
// or SIGBUS), and 0 for breakpoint traps, single-steps, SIGSTOP and
// suppressed runtime signals.
func (s *Session) LastSignal() syscall.Signal { return s.lastSignal }
// Peek reads a word (8 bytes) from the debuggee's memory at addr.
func (s *Session) Peek(addr uint64) (uint64, error) {
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
@@ -294,7 +302,8 @@ func (s *Session) Pid() int { return s.pid }
// CodeBase returns the base address of the JIT code in the debuggee.
func (s *Session) CodeBase() uint64 { return s.codeBase }
// Kill terminates the debuggee.
// Kill terminates the debuggee and removes the session's scratch
// directory, so a successful session leaves no gasm-debug-* debris behind.
func (s *Session) Kill() {
if !s.exited {
syscall.Kill(s.pid, syscall.SIGKILL)
@@ -304,6 +313,35 @@ func (s *Session) Kill() {
if s.cmd != nil && s.cmd.Process != nil {
s.cmd.Wait()
}
if s.tmpDir != "" {
os.RemoveAll(s.tmpDir)
s.tmpDir = ""
}
}
// execRange is one executable mapping of the debuggee.
type execRange struct {
lo, hi uint64
}
// execRanges parses the debuggee's executable mappings from /proc/pid/maps.
func execRanges(pid int) []execRange {
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
if err != nil {
return nil
}
var out []execRange
for line := range strings.SplitSeq(string(data), "\n") {
fields := strings.Fields(line)
if len(fields) < 2 || !strings.Contains(fields[1], "x") {
continue
}
var lo, hi uint64
if _, err := fmt.Sscanf(fields[0], "%x-%x", &lo, &hi); err == nil {
out = append(out, execRange{lo, hi})
}
}
return out
}
// findRWXMapping reads /proc/pid/maps and returns the base address of the
+68 -11
View File
@@ -6,6 +6,7 @@
package debug
import (
"encoding/binary"
"fmt"
"syscall"
"unsafe"
@@ -44,20 +45,24 @@ func (s *Session) SetRegs(regs *Regs) error {
return nil
}
// FPRegs holds the x87 FPU and SSE (XMM) register state from PTRACE_GETFPREGS.
// FPRegs holds the x87 FPU and SSE (XMM) register state from
// PTRACE_GETFPREGS. The layout is the kernel's struct user_fpregs_struct
// (sys/user.h), the FXSAVE image: 512 bytes with XMM0-15 at offset 160.
// The i387 fcs/ds segment fields do not exist in the 64-bit layout. The
// size matters: the copy fills all 512 bytes, so a short or misaligned
// struct makes PTRACE_GETFPREGS overflow the caller's memory.
type FPRegs struct {
FCW uint16
FSW uint16
FTW byte
FTW uint16
FOP uint16
FIP uint64
FCS uint16
FDP uint64
FDS uint16
MXCSR uint32
MXCSRMask uint32
ST [8][16]byte // x87 stack (10 bytes per reg, padded to 16)
XMM [16][16]byte // XMM0-15
XMM [16][16]byte // XMM0-15, struct offset 160
Reserved [96]byte // FXSAVE padding, to the full 512 bytes
}
// GetFPRegs retrieves the FPU/SSE register state of the stopped debuggee.
@@ -82,16 +87,68 @@ type VectorRegs struct {
YMM [16][32]byte // YMM0-15 (full 256-bit values)
}
// GetVectorRegs retrieves the YMM registers via PTRACE_GETREGSET + XSAVE.
// NT_X86_XSTATE (0x202), the xsave extended-state regset
// (include/uapi/linux/elf.h).
const ntX86XState = 0x202
// Layout of the buffer PTRACE_GETREGSET returns for NT_X86_XSTATE: the
// 512-byte legacy fxsave image (x87 state in 0-159, XMM0-15 in 160-511),
// then the 64-byte xsave header whose first 8 bytes are xstate_bv, then one
// component per set feature bit, each 64-byte aligned. The YMM high halves
// are the first extended component, at offset 576; that offset is fixed by
// the ISA on AVX-capable x86-64. XFEATURE_MASK_YMM is bit 2 of xstate_bv
// (arch/x86/include/asm/fpu/types.h); the high halves are zero when the bit
// is clear.
const (
xsaveXMMOffset = 160
xsaveXMMSize = 256
xsaveHeaderOffset = 512
xsaveBVOffset = xsaveHeaderOffset
ymmOffset = xsaveHeaderOffset + 64 // 576
ymmSize = 256 // 16 registers, 16 bytes each
xfeatureMaskYMM = 1 << 2
xstateMaxBuffer = 4096 // CPUID(0xD).xsave_size is far below this
)
// GetVectorRegs retrieves the YMM registers via PTRACE_GETREGSET on
// NT_X86_XSTATE. The low (XMM) halves always come from the legacy image;
// the high halves are copied only when xstate_bv reports the YMM feature,
// and read as zero otherwise. When the regset request fails the FP image
// still provides correct XMM halves, so that is the fallback.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
var v VectorRegs
fp, err := s.GetFPRegs()
if err != nil {
return v, err
buf := make([]byte, xstateMaxBuffer)
iovec := syscall.Iovec{
Base: &buf[0],
Len: uint64(len(buf)),
}
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_GETREGSET),
uintptr(s.pid),
uintptr(ntX86XState),
uintptr(unsafe.Pointer(&iovec)),
0, 0,
)
if errno != 0 {
fp, err := s.GetFPRegs()
if err != nil {
return v, err
}
for i := range 16 {
copy(v.YMM[i][:16], fp.XMM[i][:])
}
return v, nil
}
n := int(iovec.Len)
for i := range 16 {
for j := range 16 {
v.YMM[i][j] = fp.XMM[i][j]
copy(v.YMM[i][:16], buf[xsaveXMMOffset+16*i:xsaveXMMOffset+16*i+16])
}
if n >= ymmOffset+ymmSize {
if binary.LittleEndian.Uint64(buf[xsaveBVOffset:xsaveBVOffset+8])&xfeatureMaskYMM != 0 {
for i := range 16 {
copy(v.YMM[i][16:], buf[ymmOffset+16*i:ymmOffset+16*i+16])
}
}
}
return v, nil
+5 -1
View File
@@ -91,5 +91,9 @@ func (r *Regs) RegValue(name string) (uint64, bool) {
// breakpointInsn is the software breakpoint instruction.
var breakpointInsn = []byte{0xCC} // INT3
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap. x86-64 reports the #DB for INT3 with RIP on the byte after the
// INT3 (Intel SDM vol 3, "Debug Exceptions"), so the trap address is
// PC-1. The other supported architectures leave the PC on the trap
// instruction and use 0 there.
const breakpointPCAdjust = 1
+8 -2
View File
@@ -130,5 +130,11 @@ func (r *Regs) RegValue(name string) (uint64, bool) {
// breakpointInsn is the software breakpoint instruction (BRK #0).
var breakpointInsn = []byte{0x00, 0x00, 0x20, 0xD4} // BRK #0
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
const breakpointPCAdjust = 4
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap: 0, because the arm64 kernel delivers the BRK SIGTRAP with the PC
// still on the BRK. do_el0_brk64 calls send_user_sigtrap, which uses
// instruction_pointer(regs) unmodified (arch/arm64/kernel/debug-monitors.c);
// only the kernel-internal skip paths advance the PC. GDB history agrees:
// decr_pc_after_break on aarch64 Linux is 0 (the +4 variant was a QEMU bug,
// sourceware PR 17280).
const breakpointPCAdjust = 0
+6 -2
View File
@@ -126,5 +126,9 @@ func (r *Regs) RegValue(name string) (uint64, bool) {
// breakpointInsn is the software breakpoint instruction (BRK $0).
var breakpointInsn = []byte{0x05, 0x00, 0x2a, 0x00} // break 0
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
const breakpointPCAdjust = 4
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap: 0, because the kernel delivers the break SIGTRAP with csr_era
// still on the break instruction. do_bp passes regs->csr_era straight to
// force_sig_fault(SIGTRAP, TRAP_BRKPT, ...) and never adjusts era on the
// signal path (arch/loongarch/kernel/traps.c).
const breakpointPCAdjust = 0
+6 -2
View File
@@ -126,5 +126,9 @@ func (r *Regs) RegValue(name string) (uint64, bool) {
// breakpointInsn is the software breakpoint instruction (EBREAK).
var breakpointInsn = []byte{0x73, 0x00, 0x10, 0x00} // ebreak
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
const breakpointPCAdjust = 4
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap: 0, because the kernel delivers the EBREAK SIGTRAP with sepc still
// on the ebreak. handle_break passes regs->epc straight to
// force_sig_fault(SIGTRAP, TRAP_BRKPT, ...) and only the kernel-internal
// WARN/CFI paths advance epc (arch/riscv/kernel/traps.c).
const breakpointPCAdjust = 0
+76 -20
View File
@@ -7,9 +7,10 @@ package debug
import (
"bufio"
"cmp"
"fmt"
"io"
"sort"
"slices"
"strconv"
"strings"
)
@@ -32,7 +33,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
entryAddr := codeBase + uint64(funcOffset)
fmt.Printf("stopped at function entry: %#x (%d bytes)\n", entryAddr, funcSize)
fmt.Println("commands: break <label|addr> | step [n] | continue | disas [n] | regs | where | x <addr> [len] | w <addr> <val...> | labels | quit")
fmt.Println("commands: break <label|addr|line> | step [n] | continue | disas [n] | regs | where | x <addr> [len] | w <addr> <val...> | labels | quit")
scanner := bufio.NewScanner(in)
@@ -93,7 +94,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
regs, _ := s.GetRegs()
pc := regs.GetPC()
text, instLen, _ := s.Disassemble(pc)
if strings.HasPrefix(strings.ToLower(text), "call") || strings.HasPrefix(strings.ToLower(text), "bl") {
if isCallInsn(text) {
afterAddr := pc + uint64(instLen)
_, err := bm.Set(afterAddr, "(next)")
if err != nil {
@@ -108,6 +109,21 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
bm.Clear(afterAddr)
continue
}
if s.Exited() {
bm.Clear(afterAddr)
fmt.Println("debuggee exited")
continue
}
if sig := s.LastSignal(); sig != 0 {
bm.Clear(afterAddr)
regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
continue
}
// Fetch the registers after the stop: the trap must be
// evaluated against the real PC, not the pre-Continue
// snapshot, and a stale SetRegs would clobber live state.
regs, _ = s.GetRegs()
bm.HandleTrap(&regs)
bm.Clear(afterAddr)
} else {
@@ -143,9 +159,21 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
bm.Clear(retAddr)
continue
}
if !s.Exited() {
bm.HandleTrap(&regs)
if s.Exited() {
bm.Clear(retAddr)
fmt.Println("debuggee exited")
continue
}
if sig := s.LastSignal(); sig != 0 {
bm.Clear(retAddr)
regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
continue
}
// Fetch the registers after the stop, as the continue case
// does: HandleTrap must see the PC the trap left behind.
regs, _ = s.GetRegs()
bm.HandleTrap(&regs)
bm.Clear(retAddr)
if s.Exited() {
fmt.Println("debuggee exited")
@@ -171,6 +199,15 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Println("debuggee exited")
break
}
if sig := s.LastSignal(); sig != 0 {
// A genuine signal-delivery-stop (a fault): report it
// and return to the prompt. Continuing would restart
// the faulting instruction and fault forever.
regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x (func+%#x)\n",
sig, regs.GetPC(), regs.GetPC()-codeBase-uint64(funcOffset))
break
}
reason, wpAddr := s.StopInfo()
if reason == StopWatchpoint {
fmt.Printf("watchpoint hit at %#x\n", wpAddr)
@@ -196,7 +233,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
case "break", "b":
if len(parts) < 2 {
fmt.Println("usage: break <label|addr|line> [if <reg> <op> <val>]")
fmt.Println("usage: break <label|addr|line> [if <reg> <op> <val|reg|*addr>]")
continue
}
var addr uint64
@@ -221,13 +258,27 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
reg := strings.ToLower(parts[3])
op := parts[4]
operand := parts[5]
if val, err := strconv.ParseUint(operand, 0, 64); err == nil {
cond = &Condition{Reg: reg, Op: op, Value: val}
} else {
cond = &Condition{Reg: reg, Op: op, Reg2: strings.ToLower(operand)}
switch {
case strings.HasPrefix(operand, "*"):
// Memory operand: compare against the 8-byte word at
// the address, resolved in the debuggee when the
// breakpoint is evaluated.
addr, err := strconv.ParseUint(strings.TrimPrefix(operand, "*"), 0, 64)
if err != nil {
fmt.Printf("invalid memory operand: %s\n", operand)
continue
}
cond = &Condition{Reg: reg, Op: op, MemAddr: addr}
default:
val, err := strconv.ParseUint(operand, 0, 64)
if err == nil {
cond = &Condition{Reg: reg, Op: op, Value: val}
} else {
cond = &Condition{Reg: reg, Op: op, Reg2: strings.ToLower(operand)}
}
}
} else if len(parts) >= 4 && parts[2] == "if" {
fmt.Println("usage: break <label|addr> if <reg> <op> <value|reg>")
fmt.Println("usage: break <label|addr|line> if <reg> <op> <value|reg|*addr>")
continue
}
bp, err := bm.SetWithCond(addr, label, cond)
@@ -237,7 +288,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
}
condStr := ""
if cond != nil {
condStr = fmt.Sprintf(" if %s %s %#x", cond.Reg, cond.Op, cond.Value)
condStr = " if " + cond.String()
}
fmt.Printf("breakpoint set: %s at %#x (func+%#x)%s\n", bp.Label, bp.Addr, bp.Addr-codeBase-uint64(funcOffset), condStr)
@@ -277,7 +328,11 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
addr, _ = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
}
if len(parts) > 2 {
length, _ = strconv.Atoi(parts[2])
// A malformed or non-positive length would panic
// ReadMemory's make; fall back to the default instead.
if n, err := strconv.Atoi(parts[2]); err == nil && n > 0 {
length = n
}
}
mem, err := s.ReadMemory(addr, length)
if err != nil {
@@ -336,10 +391,8 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
}
case "labels", "l":
sorted := make([]Label, len(labels))
copy(sorted, labels)
sort.Slice(sorted, func(i, j int) bool { return sorted[i].Offset < sorted[j].Offset })
for _, l := range sorted {
slices.SortFunc(labels, func(a, b Label) int { return cmp.Compare(a.Offset, b.Offset) })
for _, l := range labels {
fmt.Printf(" func+%#04x %s\n", l.Offset, l.Name)
}
@@ -369,7 +422,10 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Println()
case "help", "h", "?":
fmt.Printf(` break <label|addr> [if <reg> <op> <val>] set a breakpoint
fmt.Printf(` break <label|addr|line> [if <reg> <op> <val|reg|*addr>]
set a breakpoint, optionally conditional on a
register compared to a constant, a register, or the
8-byte word at *addr
delete <label|addr> remove a breakpoint
info break list all breakpoints
watch <addr> [r|w] [size] set a hardware watchpoint (write by default)
@@ -463,8 +519,8 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
case "unwatch":
if len(parts) >= 2 {
slot, err := strconv.Atoi(parts[1])
if err != nil || slot < 0 || slot > 3 {
fmt.Println("usage: unwatch [<slot>]")
if err != nil || slot < 0 || slot >= maxWatchpoints() {
fmt.Printf("usage: unwatch [<slot 0-%d>]\n", maxWatchpoints()-1)
continue
}
if err := s.ClearWatchpoint(slot); err != nil {
+10 -2
View File
@@ -6,6 +6,7 @@
package debug
import (
"encoding/binary"
"syscall"
"unsafe"
)
@@ -61,8 +62,15 @@ func (s *Session) StopInfo() (StopReason, uint64) {
case trapBRKPT:
return StopBreakpoint, 0
case trapHWBRKPT:
addr := *(*uint64)(unsafe.Add(unsafe.Pointer(&info), 16))
return StopWatchpoint, addr
// si_addr sits at struct offset 16 (12 bytes of signo/errno/code
// plus 4 bytes of union alignment). The siginfo buffer is only
// 4-byte aligned, so the address is read byte-wise to keep the
// load aligned on riscv64 and loong64. What si_addr names is
// architecture-specific (the data address on arm64, the
// instruction pointer on x86), so the per-architecture
// archWatchpointAddr resolves it to the watched address.
addr := binary.LittleEndian.Uint64(info._pad[4:12])
return StopWatchpoint, archWatchpointAddr(s, addr)
default:
return StopSingleStep, 0
}
+1 -9
View File
@@ -28,15 +28,7 @@ func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
return fmt.Errorf("debug target: parse: %v", errs[0])
}
var img *asm.Image
switch "arm64" {
case "arm64":
img, err = asm.AssembleFileARM64(file)
case "riscv64":
img, err = asm.AssembleFileRISCV(file)
case "loong64":
img, err = asm.AssembleFileLOONG64(file)
}
img, err := asm.AssembleFileARM64(file)
if err != nil {
return fmt.Errorf("debug target: assemble: %w", err)
}
+1 -9
View File
@@ -28,15 +28,7 @@ func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
return fmt.Errorf("debug target: parse: %v", errs[0])
}
var img *asm.Image
switch "loong64" {
case "arm64":
img, err = asm.AssembleFileARM64(file)
case "riscv64":
img, err = asm.AssembleFileRISCV(file)
case "loong64":
img, err = asm.AssembleFileLOONG64(file)
}
img, err := asm.AssembleFileLOONG64(file)
if err != nil {
return fmt.Errorf("debug target: assemble: %w", err)
}
+8 -2
View File
@@ -12,10 +12,11 @@ type tracer interface {
Peek(addr uint64) (uint64, error)
Poke(addr uint64, val uint64) error
SetRegs(regs *Regs) error
Step() error
Pid() int
}
// mockTracer records Peek/Poke calls and provides fake register state.
// mockTracer records Peek/Poke/Step calls and provides fake register state.
type mockTracer struct {
mem map[uint64]byte
peeks []uint64
@@ -23,7 +24,8 @@ type mockTracer struct {
addr uint64
val uint64
}
regs *Regs
steps int
regs *Regs
}
func newMockTracer() *mockTracer {
@@ -57,4 +59,8 @@ func (m *mockTracer) SetRegs(regs *Regs) error {
m.regs = regs
return nil
}
func (m *mockTracer) Step() error {
m.steps++
return nil
}
func (m *mockTracer) Pid() int { return 42 }
+52 -21
View File
@@ -8,10 +8,48 @@ package debug
import (
"fmt"
"syscall"
"unsafe"
)
// Hardware watchpoint support via x86-64 debug registers (DR0-DR3, DR7).
// The kernel translates PTRACE_POKEUSER/PEEKUSER offsets inside
// [offsetof(struct user, u_debugreg[0]), u_debugreg[7]] to DR0-DR7
// (arch/x86/kernel/ptrace.c, arch_ptrace). sys/user.h places u_debugreg at
// 0x350: DR0-DR3 are 0x350/0x358/0x360/0x368, DR6 (status) is 0x380 and
// DR7 (control) is 0x388. Offsets below 0x350 write user_regs_struct
// fields (r15 at 0x0, r10 at 0x38), not debug registers.
const (
drOffset = 0x350 // offsetof(struct user, u_debugreg[0]), DR0
dr6Off = 0x380 // offsetof(struct user, u_debugreg[6]), DR6
dr7Off = 0x388 // offsetof(struct user, u_debugreg[7]), DR7
)
// archWatchpointAddr resolves the address of the watchpoint that fired.
// x86 delivers si_addr = the instruction pointer of the trapping access
// (arch/x86/kernel/ptrace.c send_sigtrap passes regs->ip), so the watched
// data address is recovered from DR6's slot bits (B0-B3, positive polarity
// through PEEKUSER) and the matching DR0-DR3.
func archWatchpointAddr(s *Session, siAddr uint64) uint64 {
dr6, err := ptracePeekUser(s.pid, dr6Off)
if err != nil {
return siAddr
}
for slot := range 4 {
if dr6&(1<<slot) != 0 {
addr, err := ptracePeekUser(s.pid, drOffset+uintptr(slot*8))
if err == nil && addr != 0 {
return addr
}
}
}
return siAddr
}
// maxWatchpoints reports the number of hardware watchpoint slots the
// architecture provides: four address registers, DR0-DR3.
func maxWatchpoints() int { return 4 }
// WatchpointType selects what triggers the watchpoint.
type WatchpointType int
@@ -62,23 +100,11 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
}
var drAddr uintptr
switch slot {
case 0:
drAddr = 0x0
case 1:
drAddr = 0x8
case 2:
drAddr = 0x10
case 3:
drAddr = 0x18
}
if err := ptracePokeUser(s.pid, drAddr, addr); err != nil {
if err := ptracePokeUser(s.pid, drOffset+uintptr(slot*8), addr); err != nil {
return fmt.Errorf("debug: set DR%d: %w", slot, err)
}
dr7, err := ptracePeekUser(s.pid, 0x38)
dr7, err := ptracePeekUser(s.pid, dr7Off)
if err != nil {
return fmt.Errorf("debug: read DR7: %w", err)
}
@@ -90,7 +116,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
mask := ^((uint64(1) << (2 * slot)) | (uint64(3) << (16 + 4*slot)) | (uint64(3) << (18 + 4*slot)))
dr7 = (dr7 & mask) | enableBit | rwBits | lenField
if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
if err := ptracePokeUser(s.pid, dr7Off, dr7); err != nil {
return fmt.Errorf("debug: set DR7: %w", err)
}
s.wpSlots[slot] = true
@@ -105,12 +131,12 @@ func (s *Session) ClearWatchpoint(slot int) error {
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
dr7, err := ptracePeekUser(s.pid, 0x38)
dr7, err := ptracePeekUser(s.pid, dr7Off)
if err != nil {
return err
}
dr7 &^= uint64(1) << (2 * slot)
if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
if err := ptracePokeUser(s.pid, dr7Off, dr7); err != nil {
return err
}
s.wpSlots[slot] = false
@@ -119,7 +145,7 @@ func (s *Session) ClearWatchpoint(slot int) error {
// ClearAllWatchpoints removes all hardware watchpoints.
func (s *Session) ClearAllWatchpoints() error {
for slot := range 4 {
for slot := range maxWatchpoints() {
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
@@ -146,16 +172,21 @@ func ptracePokeUser(pid int, offset uintptr, val uint64) error {
}
func ptracePeekUser(pid int, offset uintptr) (uint64, error) {
// x86 PEEKUSR writes the word to the user-space pointer in data
// (arch/x86/kernel/ptrace.c uses put_user); passing 0 there fails with
// EFAULT, so the word is read through a real address.
const ptracePeekuser = 3
val, _, errno := syscall.Syscall6(
var word uint64
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(ptracePeekuser),
uintptr(pid),
offset,
0, 0, 0,
uintptr(unsafe.Pointer(&word)),
0, 0,
)
if errno != 0 {
return 0, errno
}
return uint64(val), nil
return word, nil
}
+41 -28
View File
@@ -12,7 +12,7 @@ import (
)
// Hardware watchpoint support via arm64 debug registers (DBGWVR/DBGWCR).
// Accessed via PTRACE_SETREGSET with NT_ARM_HW_BREAK.
// Accessed via PTRACE_GETREGSET/SETREGSET with NT_ARM_HW_WATCH.
// WatchpointType selects what triggers the watchpoint.
type WatchpointType int
@@ -22,26 +22,35 @@ const (
WatchRead WatchpointType = 3
)
const maxWatchpoints = 16
// maxWatchpoints reports the number of hardware watchpoint slots the
// architecture provides: DBGWVR0-DBGWCR15.
func maxWatchpoints() int { return 16 }
// hwBreakState mirrors the kernel's struct user_hwdebug_state.
type hwBreakState struct {
// hwWatchState mirrors the kernel's struct user_hwdebug_state.
type hwWatchState struct {
DbgInfo uint32
_pad [4]byte
DbgRegs [16]hwBreakReg
DbgRegs [16]hwWatchReg
}
type hwBreakReg struct {
type hwWatchReg struct {
Addr uint64
Ctrl uint64
}
const (
ntArmHWBreak = 0x403 // NT_ARM_HW_BREAK
)
// ntArmHWWatch is NT_ARM_HW_WATCH (0x403), the watchpoint regset
// (include/uapi/linux/elf.h; 0x402 is NT_ARM_HW_BREAK). Watchpoints and
// breakpoints live in different regsets with the same struct shape, so the
// constant is named for what it arms to keep a future edit from arming
// breakpoints instead.
const ntArmHWWatch = 0x403
// archWatchpointAddr resolves the address of the watchpoint that fired:
// the arm64 kernel already reports the watched data address as si_addr.
func archWatchpointAddr(s *Session, siAddr uint64) uint64 { return siAddr }
func (s *Session) FindFreeWatchpointSlot() int {
for i := range maxWatchpoints {
for i := range maxWatchpoints() {
if !s.wpSlots[i] {
return i
}
@@ -50,7 +59,7 @@ func (s *Session) FindFreeWatchpointSlot() int {
}
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot >= maxWatchpoints {
if slot < 0 || slot >= maxWatchpoints() {
return false
}
return s.wpSlots[slot]
@@ -58,27 +67,31 @@ func (s *Session) IsWatchpointSlotUsed(slot int) bool {
// SetWatchpoint installs a hardware watchpoint on the given address.
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
if slot < 0 || slot >= maxWatchpoints() {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints()-1)
}
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
state, err := s.getHWBreakState()
state, err := s.getHWWatchState()
if err != nil {
return fmt.Errorf("debug: read watchpoint state: %w", err)
}
if uint32(slot) >= state.DbgInfo {
return fmt.Errorf("debug: slot %d exceeds available watchpoints (%d)", slot, state.DbgInfo)
// MDSCR_EL1 packs (debug_arch << 8) | num_slots into dbg_info, so only
// the low byte counts slots.
if uint32(slot) >= state.DbgInfo&0xff {
return fmt.Errorf("debug: slot %d exceeds available watchpoints (%d)", slot, state.DbgInfo&0xff)
}
state.DbgRegs[slot].Addr = addr
// DBGWCR bits 3-4 select the access type: 01 load, 10 store, 11 either
// (ARM DDI 0487, DBGWCR<n>_EL1 watchpoint type field).
ctrl := uint64(1) // enable
switch typ {
case WatchWrite:
ctrl |= 1 << 3 // store only
ctrl |= 2 << 3 // store only
case WatchRead:
ctrl |= 3 << 3 // load+store
}
@@ -98,7 +111,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
ctrl |= bas << 5
state.DbgRegs[slot].Ctrl = ctrl
if err := s.setHWBreakState(state); err != nil {
if err := s.setHWWatchState(state); err != nil {
return fmt.Errorf("debug: set watchpoint: %w", err)
}
@@ -107,20 +120,20 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
}
func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
if slot < 0 || slot >= maxWatchpoints() {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints()-1)
}
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
state, err := s.getHWBreakState()
state, err := s.getHWWatchState()
if err != nil {
return err
}
state.DbgRegs[slot].Addr = 0
state.DbgRegs[slot].Ctrl = 0
if err := s.setHWBreakState(state); err != nil {
if err := s.setHWWatchState(state); err != nil {
return err
}
s.wpSlots[slot] = false
@@ -128,7 +141,7 @@ func (s *Session) ClearWatchpoint(slot int) error {
}
func (s *Session) ClearAllWatchpoints() error {
for slot := 0; slot < maxWatchpoints; slot++ {
for slot := range maxWatchpoints() {
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
@@ -138,8 +151,8 @@ func (s *Session) ClearAllWatchpoints() error {
return nil
}
func (s *Session) getHWBreakState() (*hwBreakState, error) {
var state hwBreakState
func (s *Session) getHWWatchState() (*hwWatchState, error) {
var state hwWatchState
iovec := syscall.Iovec{
Base: (*byte)(unsafe.Pointer(&state)),
Len: uint64(unsafe.Sizeof(state)),
@@ -148,7 +161,7 @@ func (s *Session) getHWBreakState() (*hwBreakState, error) {
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_GETREGSET),
uintptr(s.pid),
uintptr(ntArmHWBreak),
uintptr(ntArmHWWatch),
uintptr(unsafe.Pointer(&iovec)),
0, 0,
)
@@ -158,7 +171,7 @@ func (s *Session) getHWBreakState() (*hwBreakState, error) {
return &state, nil
}
func (s *Session) setHWBreakState(state *hwBreakState) error {
func (s *Session) setHWWatchState(state *hwWatchState) error {
iovec := syscall.Iovec{
Base: (*byte)(unsafe.Pointer(state)),
Len: uint64(unsafe.Sizeof(*state)),
@@ -167,7 +180,7 @@ func (s *Session) setHWBreakState(state *hwBreakState) error {
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_SETREGSET),
uintptr(s.pid),
uintptr(ntArmHWBreak),
uintptr(ntArmHWWatch),
uintptr(unsafe.Pointer(&iovec)),
0, 0,
)
+121 -60
View File
@@ -8,10 +8,47 @@ package debug
import (
"fmt"
"syscall"
"unsafe"
)
// Hardware watchpoint support for LoongArch via debug registers.
// Uses PTRACE_POKEUSER/PEEKUSER to access HW watchpoint registers.
// Hardware watchpoint support via the NT_LOONGARCH_HW_WATCH regset.
//
// The kernel's PTRACE_POKEUSER on loong64 accepts only the user_pt_regs
// indices 0-34 (GPRs, orig_a0, era, badv, per
// arch/loongarch/include/uapi/asm/ptrace.h), so there is no debug-register
// window to poke. The real interface is PTRACE_GETREGSET/SETREGSET on
// NT_LOONGARCH_HW_WATCH (0xa06, include/uapi/linux/elf.h) with struct
// user_watch_state_v2 (arch/loongarch/include/uapi/asm/ptrace.h): a dbg_info
// word followed by 14 slots of {addr u64, mask u64, ctrl u32, pad u32}.
// hw_break_get puts the slot count in the low byte of dbg_info
// (arch/loongarch/kernel/ptrace.c, ptrace_hbp_get_resource_info) and
// hw_break_set ignores dbg_info, reading addr, mask and ctrl per slot.
const ntLoongHWWatch = 0xa06
// loongWatchState mirrors the kernel's struct user_watch_state_v2.
type loongWatchState struct {
DbgInfo uint64
DbgRegs [14]loongWatchReg
}
type loongWatchReg struct {
Addr uint64
Mask uint64
Ctrl uint32
Pad uint32
}
// Control word bit layout (arch/loongarch/include/asm/hw_breakpoint.h):
// bits 1-4 privilege enables (CTRL_PLV3_ENABLE, 0x10, covers user mode),
// bits 8-9 access type (LOAD 1<<0, STORE 1<<1), bits 10-11 length
// (0=8 bytes, 1=4, 2=2, 3=1, inverted like the hardware FWP cfg).
const (
loongCtrlPLV3Enable = 0x10
loongTypeLoad = 1 << 8
loongTypeStore = 2 << 8
loongLenShift = 10
)
// WatchpointType selects what triggers the watchpoint.
type WatchpointType int
@@ -21,10 +58,16 @@ const (
WatchRead WatchpointType = 3
)
const maxWatchpoints = 4
// maxWatchpoints reports the slot capacity of the regset struct; the number
// the hardware actually provides is read from dbg_info at arm time.
func maxWatchpoints() int { return len(loongWatchState{}.DbgRegs) }
// archWatchpointAddr resolves the address of the watchpoint that fired:
// the loongarch kernel already reports the accessed address as si_addr.
func archWatchpointAddr(s *Session, siAddr uint64) uint64 { return siAddr }
func (s *Session) FindFreeWatchpointSlot() int {
for i := range maxWatchpoints {
for i := range maxWatchpoints() {
if !s.wpSlots[i] {
return i
}
@@ -33,53 +76,56 @@ func (s *Session) FindFreeWatchpointSlot() int {
}
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot >= maxWatchpoints {
if slot < 0 || slot >= maxWatchpoints() {
return false
}
return s.wpSlots[slot]
}
// SetWatchpoint installs a hardware watchpoint.
// SetWatchpoint installs a hardware watchpoint on the given address.
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
if slot < 0 || slot >= maxWatchpoints() {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints()-1)
}
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
if size != 1 && size != 2 && size != 4 && size != 8 {
var ctrlType uint32
switch typ {
case WatchWrite:
ctrlType = loongTypeStore
case WatchRead:
ctrlType = loongTypeLoad | loongTypeStore
}
var lenBits uint32
switch size {
case 1:
lenBits = 3
case 2:
lenBits = 2
case 4:
lenBits = 1
case 8:
lenBits = 0
default:
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
}
// LoongArch debug registers: DBGWVR (watchpoint value) and DBGWCR (watchpoint control).
// Accessed via PTRACE_POKEUSER at architecture-specific offsets.
if err := ptracePokeUser(s.pid, uintptr(0x1000+slot*8), addr); err != nil {
return fmt.Errorf("debug: set watchpoint address: %w", err)
state, err := s.getLoongWatchState()
if err != nil {
return fmt.Errorf("debug: read watchpoint state: %w", err)
}
if uint64(slot) >= state.DbgInfo&0xff {
return fmt.Errorf("debug: slot %d exceeds available watchpoints (%d)", slot, state.DbgInfo&0xff)
}
// DBGWCR: enable + type + size.
var wcr uint64 = 1 // enable
switch typ {
case WatchWrite:
wcr |= 1 << 3 // store
case WatchRead:
wcr |= 3 << 3 // load+store
}
var sizeBits uint64
switch size {
case 1:
sizeBits = 0
case 2:
sizeBits = 1
case 4:
sizeBits = 2
case 8:
sizeBits = 3
}
wcr |= sizeBits << 5
state.DbgRegs[slot].Addr = addr
state.DbgRegs[slot].Mask = 0
state.DbgRegs[slot].Ctrl = loongCtrlPLV3Enable | ctrlType | lenBits<<loongLenShift
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), wcr); err != nil {
return fmt.Errorf("debug: set watchpoint control: %w", err)
if err := s.setLoongWatchState(state); err != nil {
return fmt.Errorf("debug: set watchpoint: %w", err)
}
s.wpSlots[slot] = true
@@ -87,14 +133,21 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
}
func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
if slot < 0 || slot >= maxWatchpoints() {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints()-1)
}
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), 0); err != nil {
state, err := s.getLoongWatchState()
if err != nil {
return err
}
state.DbgRegs[slot].Addr = 0
state.DbgRegs[slot].Mask = 0
state.DbgRegs[slot].Ctrl = 0
if err := s.setLoongWatchState(state); err != nil {
return err
}
s.wpSlots[slot] = false
@@ -102,7 +155,7 @@ func (s *Session) ClearWatchpoint(slot int) error {
}
func (s *Session) ClearAllWatchpoints() error {
for slot := 0; slot < maxWatchpoints; slot++ {
for slot := range maxWatchpoints() {
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
@@ -112,14 +165,37 @@ func (s *Session) ClearAllWatchpoints() error {
return nil
}
func ptracePokeUser(pid int, offset uintptr, val uint64) error {
const ptracePokeuser = 6
func (s *Session) getLoongWatchState() (*loongWatchState, error) {
var state loongWatchState
iovec := syscall.Iovec{
Base: (*byte)(unsafe.Pointer(&state)),
Len: uint64(unsafe.Sizeof(state)),
}
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(ptracePokeuser),
uintptr(pid),
offset,
uintptr(val),
uintptr(syscall.PTRACE_GETREGSET),
uintptr(s.pid),
uintptr(ntLoongHWWatch),
uintptr(unsafe.Pointer(&iovec)),
0, 0,
)
if errno != 0 {
return nil, errno
}
return &state, nil
}
func (s *Session) setLoongWatchState(state *loongWatchState) error {
iovec := syscall.Iovec{
Base: (*byte)(unsafe.Pointer(state)),
Len: uint64(unsafe.Sizeof(*state)),
}
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_SETREGSET),
uintptr(s.pid),
uintptr(ntLoongHWWatch),
uintptr(unsafe.Pointer(&iovec)),
0, 0,
)
if errno != 0 {
@@ -127,18 +203,3 @@ func ptracePokeUser(pid int, offset uintptr, val uint64) error {
}
return nil
}
func ptracePeekUser(pid int, offset uintptr) (uint64, error) {
const ptracePeekuser = 3
val, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(ptracePeekuser),
uintptr(pid),
offset,
0, 0, 0,
)
if errno != 0 {
return 0, errno
}
return uint64(val), nil
}
+26 -102
View File
@@ -7,11 +7,16 @@ package debug
import (
"fmt"
"syscall"
)
// Hardware watchpoint support for RISC-V via Sdtrig trigger registers.
// Uses PTRACE_POKEUSER/PEEKUSER to access debug registers.
// Hardware watchpoints are not reachable through the riscv64 kernel ptrace
// interface. arch/riscv/kernel/ptrace.c forwards every POKEUSER/PEEKUSER to
// the generic ptrace_request, and the riscv user_regset view contains only
// the GPR, FP and vector regsets: there is no debug-register or trigger
// regset, and offsets outside the view fail with EIO. The Sdtrig CSRs
// (tselect/tdata1/tdata2) are not exposed to ptrace either. Until the
// kernel grows a trigger regset, SetWatchpoint reports the fact instead of
// poking a window that does not exist.
// WatchpointType selects what triggers the watchpoint.
type WatchpointType int
@@ -21,10 +26,18 @@ const (
WatchRead WatchpointType = 3
)
const maxWatchpoints = 4
// maxWatchpoints reports the number of hardware watchpoint slots the
// architecture provides. riscv64 exposes none via ptrace; the bound exists
// so the slot bookkeeping stays consistent.
func maxWatchpoints() int { return 4 }
// archWatchpointAddr resolves the address of the watchpoint that fired.
// Unreachable in practice (watchpoints cannot be armed), but si_addr names
// the accessed address where the kernel does report one.
func archWatchpointAddr(s *Session, siAddr uint64) uint64 { return siAddr }
func (s *Session) FindFreeWatchpointSlot() int {
for i := range maxWatchpoints {
for i := range maxWatchpoints() {
if !s.wpSlots[i] {
return i
}
@@ -33,115 +46,26 @@ func (s *Session) FindFreeWatchpointSlot() int {
}
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot >= maxWatchpoints {
if slot < 0 || slot >= maxWatchpoints() {
return false
}
return s.wpSlots[slot]
}
// SetWatchpoint installs a hardware watchpoint.
// SetWatchpoint always fails: the riscv64 kernel ptrace interface has no
// hardware-watchpoint access.
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
}
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
if size != 1 && size != 2 && size != 4 && size != 8 {
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
}
// RISC-V trigger registers: tdata1 encodes type/control, tdata2 holds address.
// The exact encoding depends on the trigger implementation (Sdtrig).
// Use PTRACE_POKEUSER to write to the trigger CSRs via the kernel's
// debug register interface.
if err := ptracePokeUser(s.pid, uintptr(0x1000+slot*8), addr); err != nil {
return fmt.Errorf("debug: set watchpoint address: %w", err)
}
// tdata1: set match control. Mode=2 (data match), select=0, action=1 (debug exception).
var tdata1 uint64 = 2 << 60 // type = match (2)
tdata1 |= 1 << 0 // action = enter debug mode
tdata1 |= 1 << 7 // store (write) trigger
if typ == WatchRead {
tdata1 |= 1 << 6 // load trigger
}
// Size encoding: 0=1byte, 1=2byte, 2=4byte, 3=8byte.
var sizeBits uint64
switch size {
case 1:
sizeBits = 0
case 2:
sizeBits = 1
case 4:
sizeBits = 2
case 8:
sizeBits = 3
}
tdata1 |= sizeBits << 16 // size field
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), tdata1); err != nil {
return fmt.Errorf("debug: set watchpoint control: %w", err)
}
s.wpSlots[slot] = true
return nil
return fmt.Errorf("debug: hardware watchpoints are not supported by the riscv64 kernel ptrace interface")
}
// ClearWatchpoint always fails: no watchpoint can ever be armed.
func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
if slot < 0 || slot >= maxWatchpoints() {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints()-1)
}
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
// Disable by clearing tdata1.
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), 0); err != nil {
return err
}
s.wpSlots[slot] = false
return nil
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
func (s *Session) ClearAllWatchpoints() error {
for slot := 0; slot < maxWatchpoints; slot++ {
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
}
}
}
return nil
}
func ptracePokeUser(pid int, offset uintptr, val uint64) error {
const ptracePokeuser = 6
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(ptracePokeuser),
uintptr(pid),
offset,
uintptr(val),
0, 0,
)
if errno != 0 {
return errno
}
return nil
}
func ptracePeekUser(pid int, offset uintptr) (uint64, error) {
const ptracePeekuser = 3
val, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(ptracePeekuser),
uintptr(pid),
offset,
0, 0, 0,
)
if errno != 0 {
return 0, errno
}
return uint64(val), nil
}
+81 -52
View File
@@ -13,11 +13,13 @@ Three design goals shape everything below.
So the centre of the toolkit is a hand-written lexer and a parser that
produce a typed AST with source positions on every node.
2. **Architecture as data, not code.** Per-architecture differences (amd64,
arm64, riscv64, loong64) live in register and instruction *tables* (`arch`),
never in `if arch == …` branches scattered through the logic. The
instruction tables are generated from the Go toolchain's own assembler
source (`just gen`), so adding or refreshing an architecture is a data
operation, not a coding one.
arm64, riscv64, loong64) live in register and instruction *tables* (`arch`)
and per-architecture encoders, rather than in `if arch == …` branches
threaded through the analysis; the arch tests that remain are dispatch and
policy points, such as which encoder a file name selects and which
registers the liveness pass audits. The instruction tables are generated
from the Go toolchain's own assembler source (`just gen`), so refreshing an
architecture is a data operation, not a coding one.
3. **Open integration surface.** Everything the toolkit can do is reachable
through two vendor-neutral interfaces: a CLI and an LSP server. No editor
owns the toolkit; the toolkit is offered to editors on standard terms.
@@ -35,19 +37,28 @@ flowchart TD
ARCH["arch tables<br/>amd64 / arm64 / riscv64 / loong64"] --> LINT
ARCH --> LSP
LINT --> LSP
PAR --> ASM["asm<br/>encoders, image, object emitters"]
ASM --> VER["verify<br/>JIT mapping, ABI checks, fuzzing"]
ASM --> DBG["debug<br/>ptrace session"]
VER --> DBG
DIS["disasm<br/>golang.org/x/arch"] --> DBG
FMT --> CLI["gasm CLI"]
LINT --> CLI
PAR --> CLI
LEX --> CLI
ASM --> CLI
VER --> CLI
DBG --> CLI
DIS --> CLI
LSP --> EDITOR["any LSP editor"]
```
The lexer is the shared foundation: the parser builds the AST from it, the
formatter re-spaces its tokens directly, and the language server uses it for
semantic highlighting. The phases follow a dependency chain: Phase 1 (static
analysis) builds only on the AST, Phase 2 (the standalone assembler) emits
object code, and Phases 3 (dynamic analysis) and 4 (the debugger) both consume
the execution substrate that the assembler provides.
semantic highlighting. The packages follow a dependency chain: static analysis
builds only on the AST, the standalone assembler emits object code, and both
the dynamic analysis and the debugger consume the execution substrate the
assembler provides.
## Packages
@@ -62,6 +73,7 @@ the execution substrate that the assembler provides.
| `format` | canonical formatter over the token stream |
| `lsp` | the language server |
| `asm` | standalone assembler: encoders, image layout, object emitters |
| `disasm` | disassembly backend over golang.org/x/arch |
| `verify` | JIT execution, ABI checks, differential fuzzing |
| `debug` | interactive ptrace debugger |
| `cmd/gasm` | the CLI |
@@ -69,9 +81,11 @@ the execution substrate that the assembler provides.
The boundaries matter as much as the responsibilities: `ast` records syntax
only, and whether a name is a register or a label is left to `arch`, so the
parser stays architecture-agnostic. `asm` and `verify` are the only packages
that touch machine code and executable memory, and `cmd/gasm` owns no logic
beyond flags and output.
parser stays architecture-agnostic. `asm` produces the machine code, `verify`
and `debug` are the two packages that map it executable (read-execute in
`verify`, read-write-execute in the debuggee), and `cmd/gasm` is the CLI, with
the verify sweep orchestration and the audit, scaffold and unified-diff
helpers beside its flags and output.
### `token` and `lexer`
@@ -111,14 +125,17 @@ Register files are generated programmatically (the regular `R8`-`R15`,
`X0`-`X15`, `Y0`-`Y15`, `Z0`-`Z31`, `K0`-`K7` ranges) plus the irregularly
named registers listed explicitly. Instruction names are **generated from the
Go toolchain's own assembler source** (`cmd/internal/obj/<arch>/anames.go`,
plus the common opcodes and the per-architecture front-end aliases such as the
arm64 `B`/`BL` branches and the `.P`/`.W` load-store addressing suffixes) by
`just gen`, so the tables always match what the real assembler accepts. Each
mnemonic maps to a summary and an optional operand-count range; counts are
recorded only where unambiguous (`-1` disables the operand-count lint for that
instruction) so the linter stays silent rather than guess. For architectures
with highly variable operand forms (arm64, riscv64, loong64) only a few
fixed-arity instructions (`RET`, `NOP`, `JMP`, `CALL`) carry counts at all.
plus the common opcodes in `cmd/internal/obj/util.go`) by `just gen`, so the
tables always match what the real assembler accepts. The spellings the
toolchain's tables do not carry are hand-maintained instead: the front-end
alias lists in `arch/arm64.go`, `arch/amd64.go` and `arch/loong64.go` (the
arm64 `B`/`BL` branches among them), and the arm64 `.P`/`.W` load-store suffix
stripping in `arch/arch.go`. Each mnemonic maps to a summary and an optional
operand-count range; counts are recorded only where unambiguous (`-1`
disables the operand-count lint for that instruction) so the linter stays
silent rather than guess. For architectures with highly variable operand
forms (arm64, riscv64, loong64) `relaxCounts` clears those counts, leaving
`RET` and `NOP` with a range (`RET` alone on riscv64).
### `lint`
@@ -230,20 +247,20 @@ them from the standard LSP legend, so no editor-specific grammar is needed.
### `asm`
The standalone assembler (Phase 2). Its core is an amd64 instruction encoder:
The standalone assembler. Its core is an amd64 instruction encoder:
a REX/ModR-M/SIB/displacement/immediate engine plus the scalar instruction set,
with the Plan 9 operand order (source first) mapped onto the x86 encoding.
Every encoding is validated by decoding it again with `golang.org/x/arch`, the
one module dependency, which also backs the `gasm dis` listings.
A **RISC-V encoder** (Phase 5, RV64IMAFDC + RVC compression) encodes the full
A **RISC-V encoder** (RV64IMAFDC + RVC compression) encodes the full
integer, atomic, float/double, FMA and CSR instruction sets with the MOV
pseudo-instruction and SB/global symbol references (AUIPC pairs with
R_RISCV_PCREL_HI20/LO12 relocations). The encoder compresses eligible
instructions to 16-bit RVC forms and is validated byte-for-byte against
`GOARCH=riscv64 go tool asm`.
A **LoongArch encoder** (Phase 5, LoongArch64) encodes the integer and
A **LoongArch encoder** (LoongArch64) encodes the integer and
floating-point instruction sets with the dual-form arithmetic mnemonics (3R
vs 2RI12), the 16/21-bit branch families, the MOV pseudo-instruction and its
constant materialisation (the dcon classification driving lu12i.w/ori/lu32i.d/
@@ -254,7 +271,7 @@ relocations). Like the RISC-V encoder it is validated byte-for-byte against
`GOARCH=loong64 go tool asm`, and its GOOBJ output is proven end-to-end by
substituting it into a cross-compiled `go build` and linking with `cmd/link`.
An **AArch64 encoder** (Phase 5, arm64) encodes the integer instruction set
An **AArch64 encoder** (arm64) encodes the integer instruction set
with the data-processing (shifted register and immediate forms), load/store
(scaled unsigned immediate and unscaled9-bit immediate), conditional and
unconditional branches, the MOV pseudo-instruction and its constant
@@ -290,11 +307,11 @@ registers are translated onto the hardware stack pointer: `x+N(FP)` becomes
pointer is set up, with the matching Go prologue/epilogue generated, so the
output is byte-identical to the Go assembler for these cases. SIMD is handled
by a VEX (AVX/AVX2) encoder (the two- and three-byte VEX prefixes with XMM/YMM
registers) across eight operand forms: the three-operand NDS form, the
two-operand reg/rm form, the immediate-shift form (plus the variable-count
shifts, which share the NDS shape with the count in an XMM register or
memory), the immediate shuffle form (`VPSHUFD`, `VPERMQ`), the
three-operand-plus-immediate form (`VSHUFPD`,
registers) over nine operand forms plus a dedicated move encoder: the
three-operand NDS form, the two-operand reg/rm form, the immediate-shift form
(plus the variable-count shifts, which share the NDS shape with the count in
an XMM register or memory), the immediate shuffle form (`VPSHUFD`, `VPERMQ`),
the three-operand-plus-immediate form (`VSHUFPD`,
`VPERM2I128`, `VINSERTI128`), the lane-extract form (`VEXTRACTI128`,
`VEXTRACTF128`, where the YMM source occupies the reg field and the XMM or
memory destination r/m), the direction-sensitive moves (`VMOVDQU`, `VMOVUPD`,
@@ -339,10 +356,10 @@ b bit and the L'L rounding-control field (broadcast keeps the vector length
and scales disp8 by the element size), and combine with the .Z zeroing
suffix. Every encoding is validated two ways: by
round-trip decoding through `golang.org/x/arch`, and byte-for-byte against
the machine code the real Go assembler emits, a comparison that holds for
whole functions: all 27 functions of both kernels assemble to exactly the Go
toolchain's bytes, the lone exception being the displacements of the
static-constant loads, which the Go linker fills at link time.
the machine code the real Go assembler emits; the parity suites carry that
comparison over whole kernel files on all four architectures, with the
relocation fields masked because the Go linker fills those displacements at
link time.
File-level assembly (`AssembleFile`) goes beyond single functions: it
materialises the file's static symbols (`GLOBL`/`DATA`) in a data section
@@ -389,7 +406,7 @@ compiled packages it references.
### `verify`
The dynamic-analysis substrate (Phase 3). It JIT-loads assembled images into
The dynamic-analysis substrate. It JIT-loads assembled images into
executable memory and invokes them directly, enabling differential testing,
runtime ABI checks and coverage profiling.
@@ -409,9 +426,9 @@ every architecture too: `enterJITChecked` plants sentinels in the registers
the Go ABI fixes across calls (amd64 `BP`/`R14`, arm64 `R29`/`R28`, riscv64
`X27`, loong64 `R22`; the latter two keep no hardware frame pointer) and the
raw return trampoline `leaveJITCheckedRaw` verifies them, restoring the
saved registers before Go code resumes. riscv64 is validated end to
end under qemu-user emulation; arm64 shares the same stack convention and
fix; loong64 stays ground-truth-only until hardware validation.
saved registers before Go code resumes. All three non-amd64 trampolines
are validated end to end under qemu-user emulation, the loong64 one
through its raw-address leave handoff.
`gasm verify` runs the JIT checks when the host
matches the kernel's architecture and the toolchain comparisons
elsewhere.
@@ -425,10 +442,13 @@ The `gasm verify` CLI subcommand exposes this: it loads a file, reports the
available functions and (with `-smoke`) calls each NOSPLIT function with zeroed
arguments to confirm the trampoline round-trips. The `-smoke` and `-abi`
sweeps run in parallel and each inside a child process, so a function that
faults is reported without ending the sweep. `gasm verify --fuzz` combines
ABI checks (sentinel registers, canary, stack bounds) with differential fuzz
testing, comparing the JIT-assembled kernel against the portable Go reference
bit-for-bit while verifying the ABI contract on every iteration. When a fuzz
faults is reported without ending the sweep; `-abi` is where the ABI check
lives, fuzzing each function with sentinel values in the registers the Go ABI
fixes across calls and a canary below `SP`, and reporting a violation on any
iteration. `gasm verify --fuzz` is the differential campaign instead: it
JIT-loads the kernel and the `go tool asm` build of the same kernel and
compares the output argument areas bit-for-bit, one child process per function
so a crash on a partial function is reported rather than fatal. When a fuzz
iteration crashes or mismatches, `FuzzResult.CrashInput` stores the exact input
for reproducibility. `gasm verify --call <func> --buf name:size:pattern`
invokes a single function with user-supplied buffers (patterns: zero, ones,
@@ -448,21 +468,30 @@ masked), reporting any encoding drift.
The interactive debugger (all four architectures). It launches the target
function in a child process that maps the JIT code, calls
`PTRACE_TRACEME`, and stops; the parent attaches via ptrace and controls
execution. Breakpoints are patched as INT3 bytes through `/proc/pid/mem`
(PTRACE_PEEKTEXT is unreliable with Go's multi-threaded runtime).
execution. Breakpoints are patched through `/proc/pid/mem`: the one-byte
`INT3` on amd64, the four-byte break instruction on the other three (arm64
`BRK #0`, riscv64 `ebreak`, loong64 `break 0`).
The child pins its goroutine to the OS thread with `runtime.LockOSThread`
so the traced thread is the one executing JIT code. The REPL provides
single-step, register inspection (GPR + YMM/XMM via `PTRACE_GETFPREGS`),
single-step, register inspection (the GPRs on every architecture; on amd64 the
XMM set through `PTRACE_GETFPREGS` and the YMM set through `PTRACE_GETREGSET`
on `NT_X86_XSTATE`; on the other three the FP/SIMD regset through
`PTRACE_GETREGSET` on `NT_PRFPREG`),
label resolution, named buffer allocation with pattern filling
(`--buf name:size:pattern`: zero, ones, seq, or hex), and breakpoint
management. Breakpoints accept conditions
(`break <label> if <reg> <op> <val>`, including register-against-register
comparisons), and hardware watchpoints work on all four architectures.
comparisons), and hardware watchpoints work on amd64 (the DR0-DR3 debug
registers), arm64 (`NT_ARM_HW_WATCH`) and loong64 (`NT_LOONGARCH_HW_WATCH`);
riscv64 reports that its kernel ptrace interface exposes no trigger regset.
The ptrace path is validated at run time on amd64, where the session tests are
built; arm64, riscv64 and loong64 compile and are covered by the
architecture-neutral units (label and line tables, the breakpoint manager).
For non-interactive use, `--script` runs REPL commands from a file (or
stdin) and exits, `--timeout` kills the debuggee when a run hangs (the
watchdog is armed before the ptrace attach, so a sandboxed debuggee cannot
block it), and `--cover` runs to completion with a breakpoint on every
label and reports which blocks executed.
instruction and reports which instructions executed and how often.
### Extending the toolkit
@@ -498,9 +527,10 @@ sequenceDiagram
Errors are produced where the parse or the encoding fails and become values at
the CLI boundary: the parser returns a diagnostic list and never aborts a file,
`AssembleFile` returns an error, and `cmd/gasm` prints what it has to stderr
and returns a non-zero exit code. The formatter and the linter take the same
AST by a different route: `gasm fmt` re-spaces the token stream and `gasm lint`
walks the parsed file, so neither depends on an encoding.
and returns a non-zero exit code. The formatter and the linter take different
inputs from the assembler: `gasm fmt` re-spaces the token stream
(`format.Source` lexes the source text itself) and `gasm lint` walks the parsed
AST, so neither depends on an encoding.
## State and lifetime
@@ -521,9 +551,8 @@ walks the parsed file, so neither depends on an encoding.
## Dependencies
- **`golang.org/x/arch`** (v0.30.0) is the one module dependency: it is the
disassembler backend (`gasm dis` and the debugger's listings) and the source
of the register metadata the encoder consults (`asm/reg.go`, `asm/vex.go`).
The tests additionally decode through it to validate the encodings.
disassembler backend (`gasm dis` and the debugger's listings). The tests
additionally decode through it to validate the encodings.
- **The Go toolchain**, as an oracle and never as a library: `go tool asm`
supplies the object preamble and the ground truth for `gasm verify
--ground-truth`, `go list -json -export` locates the archives of the packages
+43 -28
View File
@@ -3,9 +3,11 @@
The reference below is taken from the program's own `--help`. If the two disagree, the
program is right and this file is a defect.
The same reference is installed as man pages: `just install-man` puts gasm(1) and one
page per command into ~/.local/share/man (`MANDIR` overrides), and a test compares each
page against the binary so the two cannot drift apart.
The same reference is installed as man pages: `just install-man` puts gasm(1) and a page
for every command except `version` (which gasm(1) itself documents) into
~/.local/share/man (`MANDIR` overrides). A test in `cmd/gasm` keeps the two from
drifting: it compares each page's flag set and SYNOPSIS line with the binary's own `-h`
output, and gasm(1)'s COMMANDS list with the top-level help. The prose is not compared.
## Synopsis
@@ -58,7 +60,8 @@ Usage: gasm parse <file>
```
Parse FILE and report syntax errors on stderr. On success, print how many
declarations and TEXT functions the file contains.
declarations and TEXT functions the file contains. FILE may be `-` to read
standard input.
```sh
gasm parse hello_amd64.s
@@ -155,7 +158,9 @@ recognisable suffix (most of GOROOT's, for example `cpu_x86.s`) are
assembled. `raw` concatenates the functions and the data section into one
self-consistent image; `elf` emits a relocatable object that links with the
system toolchain; `goobj` emits the Go toolchain's own object format, which
`cmd/link` consumes directly.
`cmd/link` consumes directly, and is the one format that needs the toolchain
installed: the object preamble is captured from `go tool asm` and the format
version from `go version`. `raw` and `elf` need no toolchain at all.
```sh
gasm asm hello_amd64.s
@@ -208,7 +213,7 @@ Usage: gasm verify [-smoke] [-abi] [-fuzz] [-ground-truth] [-profile] [-call] <f
| `--abi-n` | 100 | ABI check iterations with varied inputs |
| `--profile` | off | list the basic-block structure per function |
| `--smoke` | off | call each NOSPLIT function with zeroed arguments |
| `--call` | empty | invoke a single function with `--buf` instead of the sweeps |
| `--call` | empty | invoke a single NOSPLIT function with `--buf` instead of the sweeps |
| `--buf` | empty | buffer spec for `--call`: `name:size:pattern[,name:size:pattern]` |
| `--args` | empty | scalar args for `--call`: `name=value[,name=value]` (decimal or `0x` hex) |
| `--repeat` | 1 | number of times to repeat a `--call` invocation |
@@ -219,8 +224,8 @@ The JIT checks run when the host matches the file's architecture; the
toolchain comparison works everywhere. `--fuzz`, `--smoke` and `--abi` run each
function in its own child process, so a partial function that faults on random
input is reported as `CRASH` instead of ending the sweep; `--call` with `--buf`
invokes such a function with valid data. loong64 stays on the ground-truth path
until hardware validation.
invokes such a function with valid data. The function named by `--call` must be
NOSPLIT: a function with a stack frame is refused with a diagnostic and exits 1.
```sh
gasm verify --ground-truth hello_amd64.s
@@ -260,20 +265,21 @@ Usage: gasm debug <file.s> --func <name>
| `-buf` | empty | buffer spec: `name:size:pattern[,name:size:pattern]` (zero, ones, seq or hex) |
| `-args` | empty | file containing the ABI0 argument block |
| `-script` | empty | run REPL commands from a file, one per line, and exit; `-` reads stdin |
| `-timeout` | 0 | kill the debuggee after this duration, for headless `-script` runs |
| `-timeout` | 0 | kill the debuggee after this duration, for headless `-script` runs; a timeout exits 3 |
| `-cover` | off | run to completion with a breakpoint on every instruction and report which executed |
The debugger spawns the debuggee from the `gasm` binary on `$PATH`, so install
it first with `just install`; `go run` does not work for the traced child.
Requires Linux (ptrace) and all four architectures are supported.
The debugger re-executes the binary it is running as (`os.Executable()`) for the
traced child, so the child is the same `gasm`, whether it is installed on `$PATH`
or run with `go run ./cmd/gasm`; nothing has to be installed first. Requires
Linux (ptrace), and all four architectures are supported.
REPL commands:
| Command | Effect |
|---|---|
| `break <label\|addr> [if <reg> <op> <val>]` | set a breakpoint, optionally conditional |
| `delete <label\|addr>` | remove a breakpoint |
| `info break` | list the breakpoints |
| `break <label\|addr\|line> [if <reg> <op> <val\|reg\|*addr>]`, `b` | set a breakpoint; the condition compares a register with a constant, another register or the 8-byte word at `*addr` |
| `delete <label\|addr>`, `d` | remove a breakpoint |
| `info break`, `info breakpoints`, `info b` | list the breakpoints |
| `step [n]`, `s` | single-step n instructions |
| `next`, `n` | step over a CALL |
| `finish`, `fin` | run until the function returns |
@@ -347,13 +353,18 @@ Usage: gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64]
Compare the gasm encoder for the given architecture (default amd64) against the
installed `go tool asm` and print the diff: superset encodings (gasm-only
spellings, shippable via `gasm asm --format goobj`), known-but-unencodable
names (the encoder backlog) and go-only names (feature gaps). The Go side is
probed black-box with a battery of operand shapes per mnemonic, so the audit
tracks whatever toolchain `go env GOROOT` provides. On non-amd64
architectures the backlog is an over-approximation: a name counts as encodable
only when a probe shape assembles cleanly, so a name whose real forms the
battery misses lands in the backlog.
spellings, shippable via `gasm asm --format goobj`) and known-but-unencodable
names (the encoder backlog). The Go side is probed black-box one bare mnemonic
at a time, classified by the toolchain's diagnostic for an instruction it does
not know, so the audit tracks whatever toolchain `go env GOROOT` provides; the
gasm side answers from the encoder table on amd64 and from trial assembly over a
battery of operand shapes on the other architectures. On non-amd64
architectures the backlog is therefore an over-approximation: a name counts as
encodable only when a probe shape assembles cleanly, so a name whose real forms
the battery misses lands in the backlog. Names the toolchain knows and gasm does
not cannot be enumerated by probing at all, because Go's table is visible only
through names already in the gasm table; the report closes with a note saying
so rather than listing them.
```sh
gasm audit-instructions amd64
@@ -361,8 +372,9 @@ gasm audit-instructions amd64
```text
gasm table (amd64, families excluded): 1542 mnemonics
gasm encodable: 580 go tool asm recognized: 1542
shared: 580
gasm encodable: 587 go tool asm recognised: 1542
shared: 587
...
```
With `--corpus` the audit changes shape: it assembles every `.s` file under
@@ -382,10 +394,12 @@ gasm audit-instructions --corpus "$(go env GOROOT)/src/crypto"
```text
corpus /usr/local/go/src: 627 files (365 generic, attempted for all architectures)
assemble for every target architecture: 108 (17.2%)
amd64: 77/464 attempted
148 instruction not encodable
assemble for every target architecture: 127 (20.3%)
amd64: 82/464 attempted
165 unsupported operand form
e.g. /usr/local/go/src/cmd/asm/internal/asm/testdata/386enc.s
109 instruction not encodable
e.g. /usr/local/go/src/cmd/asm/internal/asm/testdata/386.s
...
```
@@ -447,6 +461,7 @@ version control.
| `0` | success |
| `1` | a failure the program detected: a parse or assembly error, an error-severity lint diagnostic, a mismatch in `verify`, a file that cannot be read |
| `2` | the arguments were wrong: a missing or extra argument, an unknown command or format, an invalid `--map` pair |
| `3` | `debug --timeout` killed the debuggee |
## Examples
@@ -468,5 +483,5 @@ gasm asm --format goobj -p example.com/kernel -o kernel.o kernel_amd64.s
Find which labels a failing kernel reaches, headlessly:
```sh
gasm debug --func decodeBlockAVX2 --cover --script cmds.txt --timeout 30s kernel_amd64.s
gasm debug --func decodeBlockAVX2 --cover --timeout 30s kernel_amd64.s
```
+26 -7
View File
@@ -6,9 +6,15 @@ Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrb
- **Go** 1.27.1, the exact version the `go` directive in `go.mod` declares
- **just**, the command runner; every task below is a just recipe
- **A C compiler** (`gcc`): `just race` runs the suite under the race detector,
which needs cgo
- **Perl**: the `test`, `fmt-check`, `install-man` and `uninstall-man` recipes
are Perl programs
- **`gzip`**: `install-man` compresses the man pages with it
- A Linux host on amd64, arm64, riscv64 or loong64: `gasm debug` needs ptrace
and the JIT checks of `gasm verify` need executable memory
- No external dependencies beyond the Go toolchain
- **`golang.org/x/arch`**, the one module dependency, which the Go toolchain
fetches; nothing else sits outside the standard library
## Setup
@@ -25,8 +31,9 @@ Every recipe in the `justfile`, and what it does.
| Recipe | What it does |
|---|---|
| `default` (bare `just`) | prints the recipe list (`@just --list`) |
| `just build` | compiles `bin/gasm` with `CGO_ENABLED=0` and stripped symbols; zero errors and zero warnings |
| `just test` | the test gate: the suite with `-count=1`, the coverage profile and the 80 % floor |
| `just test` | the test gate: the suite with `-count=1`, the coverage profile and the 80 % floor, then the CLI and debugger tests outside the profile |
| `just race` | the same suite under the race detector; the expensive one, so it runs once, inside `gates` |
| `just unit [packages] [run]` | fast, cached, scoped run for iterating: no race and no coverage, so an unchanged package reports instantly |
| `just fuzz <target> <pkg> [fuzztime]` | time-boxed fuzz of one target; the package is required, because `go test -fuzz` refuses more than one |
@@ -36,8 +43,10 @@ Every recipe in the `justfile`, and what it does.
| `just vet` | both static gates: `go vet` and `go fix -diff` |
| `just gates` | `build`, `fmt-check`, `vet`, `test` and `race`, in that order: the definition of done |
| `just clean` | removes the build artefacts, `bin/` and `coverage.out` |
| `just install` | builds, then copies the binary into `bindir` (`~/.local/bin`); `gasm debug` needs an installed binary, because it spawns the debuggee from `$PATH` |
| `just install` | builds, then copies the binary into `bindir` (`~/.local/bin`) |
| `just uninstall` | removes the installed binary from `bindir` |
| `just install-man` | installs the man pages under `docs/man` into `~/.local/share/man/man1` (`MANDIR` overrides), gzip-compressed; not a gate |
| `just uninstall-man` | removes the installed man pages |
| `just run` | runs the CLI with `go run -buildvcs=true`; the recipe takes no arguments, so flags go through the package instead |
| `just dev` | the same as `run`; the project has no watcher to add |
| `just gen` | regenerates the `arch` instruction tables from the Go toolchain source; not a gate |
@@ -53,9 +62,18 @@ go test -count=1 -timeout 10m -coverprofile=coverage.out \
The suite runs over the logic packages (`-count=1`, so no cached pass
counts): arch, asm, ast, disasm, format, lexer, lint, lsp, parser,
token, verify. `debug` traces a live process and `cmd/gasm` is thin CLI
glue, so both sit outside the sweep, and a thin `cmd/` in it would drag
the coverage total under the floor. The floor fails if the total is
below 80 %. CI runs the same command with the same ten-minute bound, so
glue, so both sit outside the profile sweep, and a thin `cmd/` in it
would drag the coverage total under the floor. Their tests still run, in
a second invocation without a profile:
```sh
go test -count=1 -timeout 10m ./cmd/... ./debug/...
```
That covers the CLI's exit codes and the guard that compares the manual
pages with the binary's own help, and the debugger's architecture-neutral
units. The floor fails if the total is below 80 %. CI runs the same two
commands with the same ten-minute bound, so
the number is the same everywhere.
### `just run`
@@ -129,7 +147,8 @@ therefore the fastest way to a green pipeline.
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`,
which triggers the release workflow: it builds the portable Linux targets,
takes the notes from the matching `CHANGELOG.md` section and uploads the
assets.
assets. `SECURITY.md` carries the supported-versions table, so that table
moves with the release; the pipeline refuses a tag the policy does not name.
The version is never injected. `gasm --version` prints what the
toolchain recorded in the build information: the tag on a tagged
+18 -7
View File
@@ -1,4 +1,4 @@
.TH GASM-ASM 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM-ASM 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm-asm \- assemble Plan 9 assembly without the Go toolchain
.SH SYNOPSIS
@@ -20,13 +20,24 @@ flag selects what is written:
self-consistent image;
.B elf
emits a relocatable object (.text/.data sections, a symbol table and
one PC32 relocation per static-symbol reference) that links with the
one relocation per static-symbol reference, in the architecture's own
form: R_X86_64_PC32 on amd64, R_AARCH64_*, R_RISCV_* or R_LARCH_* on the
others) that links with the
system toolchain;
.B goobj
emits the Go toolchain's own object format, which cmd/link consumes
directly (it requires
.BR \-p ,
the package path, and the installed Go toolchain).
the package path, and the installed Go toolchain: the object preamble is
captured from
.B go tool asm
and the format version from
.BR "go version" ).
.PP
.B raw
and
.B elf
need no toolchain at all.
.PP
Framed functions receive the stack-split guard and the trailing
morestack block, byte-identical to the toolchain's output, so split
@@ -51,10 +62,10 @@ Exits 0 on success, 1 when parsing or assembly fails, and 2 on a usage
error.
.SH EXAMPLES
.nf
gasm asm \-o hello.bin hello_amd64.s raw image
gasm asm \-\-format elf \-o k.o k.s linkable ELF object
gasm asm \-\-format goobj \-p pkg/path \-o k.o k.s Go object for go build
gasm asm \-GOARCH amd64 cpu_x86.s arch override
gasm asm \-o hello.bin hello_amd64.s raw image
gasm asm \-\-format elf \-o k.o k_amd64.s linkable ELF object
gasm asm \-\-format goobj \-p pkg/path \-o k.o k_amd64.s Go object for go build
gasm asm \-GOARCH amd64 cpu_x86.s arch override
.fi
.SH SEE ALSO
.BR gasm (1),
+11 -6
View File
@@ -1,4 +1,4 @@
.TH GASM-AUDIT-INSTRUCTIONS 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM-AUDIT-INSTRUCTIONS 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm-audit-instructions \- diff the encoder against the Go toolchain, or measure a corpus
.SH SYNOPSIS
@@ -8,12 +8,17 @@ Compare the gasm encoder for the given architecture (default amd64)
against
.B go tool asm
and print the diff: superset encodings (gasm-only, shippable via
.BR "gasm asm \-\-format goobj" ),
known-but-unencodable names (the backlog) and go-only names (feature
gaps). The Go side is probed black-box with a battery of bare
mnemonics, so the audit tracks whatever toolchain
.BR "gasm asm \-\-format goobj" )
and known-but-unencodable names (the backlog). The Go side is probed
black-box one bare mnemonic at a time, so the audit tracks whatever
toolchain
.B go env GOROOT
provides.
provides; the gasm side answers from the encoder table on amd64 and from
trial assembly over a battery of operand shapes elsewhere. Names
.B go tool asm
knows and gasm does not cannot be enumerated by probing, because Go's
table is visible only through names already in the gasm table; the report
closes with a note saying so rather than listing them.
.PP
With
.BR \-\-corpus ,
+13 -9
View File
@@ -1,4 +1,4 @@
.TH GASM-DEBUG 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM-DEBUG 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm-debug \- interactive source-level debugger for JIT-assembled functions
.SH SYNOPSIS
@@ -17,14 +17,16 @@ runs to completion with a breakpoint on every instruction and reports
which executed and how often, the label-level coverage view.
.SH REPL COMMANDS
.TP
.B break \fIlabel|addr\fR [\fBif \fIreg op val\fR]
Set a breakpoint, optionally conditional on a register comparison
(register against register or immediate).
.B break \fIlabel|addr|line\fR [\fBif \fIreg op val|reg|*addr\fR], b
Set a breakpoint at a label, an address or a source line number, optionally
conditional on a register comparison: against a constant, against another
register, or against the 8-byte word at
.BR *addr .
.TP
.B delete \fIlabel|addr\fR
.B delete \fIlabel|addr\fR, d
Remove a breakpoint.
.TP
.B info break
.B info break, info breakpoints, info b
List all breakpoints.
.TP
.BR step " [" n ], " s
@@ -98,10 +100,12 @@ Run REPL commands from a file (one per line) and exit; - reads stdin.
.TP
.B \-timeout \fIduration\fR
Kill the debuggee after this duration (e.g. 30s); for headless --script
runs.
runs; a timeout exits 3.
.SH EXIT STATUS
Exits 0 when the scripted session completes and 1 when the debuggee
crashes or a check fails; the debugger is Linux-only.
Exits 0 when the scripted session completes, 1 when the debuggee crashes
or a check fails, and 3 when
.B \-\-timeout
kills the debuggee; the debugger is Linux-only.
.SH EXAMPLES
.nf
gasm debug \-\-func name k.s
+2 -2
View File
@@ -1,4 +1,4 @@
.TH GASM-DIFF 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM-DIFF 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm-diff \- compare the machine code of two assembly files
.SH SYNOPSIS
@@ -28,7 +28,7 @@ differs; a usage error exits 2.
.SH EXAMPLES
.nf
gasm diff hello_amd64.s hello_amd64.s
gasm diff \-\-map wideCopyAVX2=wideCopyAVX512 avx2.s avx512.s
gasm diff \-\-map wideCopyAVX2=wideCopyAVX512 avx2_amd64.s avx512_amd64.s
.fi
.SH SEE ALSO
.BR gasm (1),
+2 -2
View File
@@ -1,4 +1,4 @@
.TH GASM-DIS 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM-DIS 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm-dis \- disassemble machine code to instruction text
.SH SYNOPSIS
@@ -28,7 +28,7 @@ Exits 0 on success, 1 when assembly or decoding fails, and 2 on a usage
error.
.SH EXAMPLES
.nf
gasm dis k.s assemble, then list each function
gasm dis k_amd64.s assemble, then list each function
gasm dis \-a amd64 \- < dump.bin disassemble raw bytes from stdin
.fi
.SH SEE ALSO
+8 -1
View File
@@ -1,4 +1,4 @@
.TH GASM-FMT 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM-FMT 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm-fmt \- canonicalise the formatting of Plan 9 assembly sources
.SH SYNOPSIS
@@ -41,6 +41,13 @@ List files whose formatting differs from gasm's.
.TP
.B \-w
Write the result to the source file.
.SH EXIT STATUS
Exits 0 on success, 1 when a path cannot be read or written, and 2 on a
usage error (combining
.B \-l
and
.BR \-d ,
or an unknown flag).
.SH EXAMPLES
.nf
gasm fmt reformat every .s below here
+18 -6
View File
@@ -1,4 +1,4 @@
.TH GASM-LINT 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM-LINT 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm-lint \- run the static checks over assembly files
.SH SYNOPSIS
@@ -33,7 +33,11 @@ The function can fall off its end without a terminator.
TEXT flags are used without including textflag.h.
.TP
.B abi-argsize
The declared frame or argument size disagrees with the
The declared argument area (the
.I \-args
part of
.IR $frame\-args )
disagrees with the
.B //\ function
signature.
.TP
@@ -52,7 +56,8 @@ FUNCDATA and PCDATA indices are malformed.
A label no jump reaches.
.TP
.B invalid-textflag
A TEXT flag combination the toolchain rejects.
An unknown TEXT or GLOBL flag, reported one flag at a time; numeric flags
are accepted as textflag.h constants.
.TP
.B stack-imbalance
The function does not restore the stack pointer on every path.
@@ -61,11 +66,18 @@ The function does not restore the stack pointer on every path.
An operand register has the wrong width for the instruction.
.TP
.B abi0-register-args
A call passes arguments in registers where ABI0 expects the stack
frame.
A function whose
.B //\ function
parameters are never read from their
.IR name+offset(FP)
frame slots, which usually means the body takes its arguments from
registers instead.
.TP
.B nonportable-register-name
A register spelling that does not exist on the target architecture.
An amd64 register alias gasm accepts but
.B go tool asm
rejects (the RAX/EAX family); the canonical spelling is named in the
diagnostic.
.TP
.B unencodable-instruction
The mnemonic is known to the table but the encoder cannot assemble it
+1 -1
View File
@@ -1,4 +1,4 @@
.TH GASM-LSP 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM-LSP 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm-lsp \- run the Plan 9 assembly language server
.SH SYNOPSIS
+1 -1
View File
@@ -1,4 +1,4 @@
.TH GASM-PARSE 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM-PARSE 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm-parse \- parse an assembly file and report syntax errors
.SH SYNOPSIS
+5 -4
View File
@@ -1,4 +1,4 @@
.TH GASM-PROFILE 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM-PROFILE 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm-profile \- show the basic-block structure of functions
.SH SYNOPSIS
@@ -6,9 +6,10 @@ gasm-profile \- show the basic-block structure of functions
.SH DESCRIPTION
Show the basic-block structure of functions in an assembly file: each
function's labels, their offsets, and the block boundaries. This is
the static structure; for runtime execution counts, use
.BR "gasm verify \-fuzz" ,
which exercises the code paths.
the static structure; for runtime execution counts use
.BR "gasm debug \-\-cover" ,
and for input coverage
.BR "gasm verify \-\-fuzz" .
.SH EXIT STATUS
Exits 0 on success and 1 when the file cannot be assembled.
.SH SEE ALSO
+1 -1
View File
@@ -1,4 +1,4 @@
.TH GASM-SCAFFOLD 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM-SCAFFOLD 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm-scaffold \- generate a differential test skeleton for a kernel file
.SH SYNOPSIS
+1 -1
View File
@@ -1,4 +1,4 @@
.TH GASM-TOKENS 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM-TOKENS 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm-tokens \- print the lexical token stream of an assembly file
.SH SYNOPSIS
+7 -7
View File
@@ -1,4 +1,4 @@
.TH GASM-VERIFY 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM-VERIFY 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm-verify \- JIT-assemble a file and run dynamic checks against it
.SH SYNOPSIS
@@ -20,8 +20,7 @@ With
each function is called with sentinel values in the registers the Go
ABI fixes across calls (the frame pointer and the goroutine pointer)
plus a canary below SP; violations are reported. JIT-based checks run
when the host matches the file's architecture (all but loong64, which
is ground-truth only for now).
when the host matches the file's architecture, on all four architectures.
.PP
With
.BR \-fuzz ,
@@ -46,7 +45,8 @@ a single function is invoked with user-supplied buffers
.RB ( \-buf )
instead of the smoke/abi/fuzz sweeps. Useful for partial functions
(e.g. decoders) that crash on random input but should succeed on valid
data.
data. The function named must be NOSPLIT: a function with a stack frame
is refused with a diagnostic and exits 1.
.PP
With
.B \-save\-corpus
@@ -74,7 +74,7 @@ Buffer spec for -call: name:size:pattern[,name:size:pattern] where
pattern is zero, ones, seq, or hex.
.TP
.B \-call \fIname\fR
Call a single function with -buf instead of the sweeps.
Call a single NOSPLIT function with -buf instead of the sweeps.
.TP
.B \-fuzz
Differential fuzz: JIT both the gasm and the go-tool-asm versions and
@@ -108,8 +108,8 @@ a file that cannot be assembled exits 1 and a usage error exits 2.
.SH EXAMPLES
.nf
gasm verify \-\-call add \-\-args a=2,b=3 hello_amd64.s
gasm verify \-\-ground\-truth k.s
gasm verify \-\-fuzz \-n 500 k.s
gasm verify \-\-ground\-truth k_amd64.s
gasm verify \-\-fuzz \-n 500 k_amd64.s
.fi
.SH SEE ALSO
.BR gasm (1),
+11 -6
View File
@@ -1,4 +1,4 @@
.TH GASM 1 "2026-09-19" "gasm 0.33.0" "User Commands"
.TH GASM 1 "2026-09-19" "gasm" "User Commands"
.SH NAME
gasm \- developer tooling for Go's Plan 9 assembler
.SH SYNOPSIS
@@ -17,11 +17,15 @@ bundles a lexer, parser, formatter, linter, standalone assembler and
language server for Plan 9 assembly into one self-contained binary. It
serves two purposes: it brings developer tooling to the
.I .s
files of Go programs, and it assembles Plan 9 assembly without the Go
toolchain at all, to raw images, linkable ELF objects with DWARF5 debug
sections, or the Go toolchain's own GOOBJ format, which
files of Go programs, and it assembles Plan 9 assembly to raw images or
linkable ELF objects with DWARF5 debug sections without the Go toolchain at
all, plus the Go toolchain's own GOOBJ format, which
.B go build
consumes directly.
consumes directly. GOOBJ is the one format that needs the toolchain
installed: the object preamble is captured from
.B go tool asm
and the format version from
.BR "go version" .
.PP
Four architectures are covered: amd64 (including VEX/AVX2 and
EVEX/AVX-512), arm64, riscv64 (RV64IMAFDC and RVC) and loong64. The
@@ -82,7 +86,8 @@ Show the command overview.
Print the version the toolchain recorded at build time.
.SH EXIT STATUS
Exits 0 on success, 1 when a command fails, and 2 on a usage error. An
unknown command exits 2.
unknown command exits 2; \fBgasm debug\fR exits 3 when \-\-timeout kills
the debuggee.
.SH SEE ALSO
.BR gasm\-asm (1),
.BR gasm\-fmt (1),
+2 -2
View File
@@ -312,7 +312,7 @@ func spaceBetween(prev, cur token.Token) bool {
return false
case token.Comma:
return false
case token.Star, token.Plus, token.Minus, token.Slash:
case token.Star, token.Plus, token.Minus, token.Slash, token.Pipe:
return false
case token.LShift, token.RShift, token.Arrow, token.At:
return false
@@ -328,7 +328,7 @@ func spaceBetween(prev, cur token.Token) bool {
}
}
switch prev.Kind {
case token.LParen, token.Star, token.Plus, token.Minus, token.Slash:
case token.LParen, token.Star, token.Plus, token.Minus, token.Slash, token.Pipe:
return false
case token.Dollar:
return false
+32 -1
View File
@@ -85,7 +85,7 @@ func TestBlankLines(t *testing.T) {
"TEXT ·f(SB), NOSPLIT, $0\n" +
"first:\n" + // first label: no blank after TEXT
"XORQ AX, AX\n" +
"JMP next\n" + // unlabeled glue: fmt inserts a blank before next:
"JMP next\n" + // unlabelled glue: fmt inserts a blank before next:
"next:\n" +
"stacked:\n" + // stacked labels share an address: no blank between
"INCQ AX\n" +
@@ -143,6 +143,7 @@ func TestOperandSpacing(t *testing.T) {
"swin_base+0(FP)": "swin_base+0(FP)",
"mask24<>(SB)": "mask24<>(SB)",
"·idx16+0(SB)/4": "·idx16+0(SB)/4",
"NOSPLIT|DUPOK": "NOSPLIT|DUPOK",
}
for in, want := range cases {
toks := lexOperands(in)
@@ -152,6 +153,36 @@ func TestOperandSpacing(t *testing.T) {
}
}
// TestFlagListRoundTrip pins the '|' flag separator and the <ABIInternal>
// marker through a full format pass: the bars the Go toolchain requires and
// the ABI bracket must survive byte for byte, on TEXT and GLOBL alike.
func TestFlagListRoundTrip(t *testing.T) {
for _, in := range []string{
"TEXT ·f(SB), NOSPLIT|NOFRAME|DUPOK, $0\n\tRET\n",
"TEXT ·foo<ABIInternal>(SB), NOSPLIT, $-0-24\n\tRET\n",
"GLOBL ·mask(SB), RODATA|NOPTR, $8\n",
} {
if got := Source(in); got != in {
t.Fatalf("flag list did not round-trip:\n--- got ---\n%q\n--- want ---\n%q", got, in)
}
}
}
// TestCRLFInputIsNormalisedToLF checks that a CRLF file comes out with
// uniform LF endings: a // comment must not carry its line's trailing \r
// into the output.
func TestCRLFInputIsNormalisedToLF(t *testing.T) {
in := "// func f()\r\nTEXT ·f(SB), NOSPLIT, $0\r\nRET\r\n"
want := "// func f()\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n"
got := Source(in)
if got != want {
t.Fatalf("CRLF formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
}
if strings.Contains(got, "\r") {
t.Fatalf("output still contains CR: %q", got)
}
}
// lexOperands lexes a single operand string and drops the EOF token.
func lexOperands(s string) []token.Token {
toks := lexer.Tokenize(s)

Some files were not shown because too many files have changed in this diff Show More