Compare commits

..
49 Commits
Author SHA1 Message Date
petrbalvin e02918c17b fix(ci): keep the push suite inside the runner's memory and time budget
Test / test (push) Successful in 2m55s
Assisted-by: GLM 5.3 Flash
2026-10-02 17:20:31 +02:00
petrbalvin 02a6359c1f ci: shrink the push pipeline to the affordable gate set
Test / test (push) Failing after 5m26s
Assisted-by: GLM 5.3 Flash
2026-10-02 16:47:52 +02:00
petrbalvin b4c1e133c0 ci: keep the GOOBJ link parity gate off the push pipeline
Test / test (push) Failing after 12m18s
Assisted-by: GLM 5.3 Flash
2026-10-02 16:15:02 +02:00
petrbalvin 1107928870 build(justfile): run the test recipes under the memory fence
Assisted-by: GLM 5.3 Flash
2026-10-02 16:15:02 +02:00
petrbalvin bc4ac93fd9 style(asm): reindent the evex comment gofmt asks for
Test / test (push) Failing after 21m29s
Assisted-by: GLM 5.3
2026-10-02 00:41:46 +02:00
petrbalvin c1bca7ce7e docs: record the development deltas in the changelog
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin fefb76beb9 docs(asm): correct the reference against the assemblers' behaviour
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin 42bc1669d7 feat(lsp): document directives on hover and widen completion
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin 69dcbec8ef feat(lint): eleven new rules over directives, data and addressing
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin f405cea5bc fix(cmd): stop the coverage run on a stray in-place trap
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin f57377abb9 fix(debug): handle mapping edges, stray traps and dying debuggees
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin dd1782c538 test(verify): gate GOOBJ link parity with cmd/link
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin 17cc49fee4 fix(asm): emit NOPTR data as its own symbol kind
Assisted-by: GLM 5.3
2026-10-02 00:40:43 +02:00
petrbalvin bafb2fd130 feat(asm): encode the amd64 and loong64 tails of the corpus testdata
Assisted-by: GLM 5.3
2026-10-02 00:40:43 +02:00
petrbalvin 2f679326c2 style(testdata): canonicalise forms_amd64.s
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 96f2dd65b4 test(lexer): fuzz the token stream invariants
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin ca887d3927 fix(parser): bound folding depth and macro expansion work
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 6570709226 fix(parser): peel stacked labels the way the formatter renders them
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 9a5217d9c1 fix(format): keep every token of a line in the canonical output
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 4d01bb3ecf build: rename the module to sourcedock.dev/petrbalvin/gasm-sdk
Test / test (push) Successful in 4m18s
2026-09-26 11:08:43 +02:00
petrbalvin 332c63e440 ci: compile-gate FreeBSD in the test pipeline
Test / test (push) Successful in 4m10s
Assisted-by: GLM 5.3 Flash
2026-09-25 21:46:49 +02:00
petrbalvin b306c210c6 feat(debug): port the debugger to FreeBSD
Assisted-by: GLM 5.3 Flash
2026-09-25 21:46:40 +02:00
petrbalvin b9015e1c2e fix(verify): make the executable mapping build on FreeBSD
Assisted-by: GLM 5.3 Flash
2026-09-25 21:46:31 +02:00
petrbalvin 26c5008136 fix(cmd): honour //go:build in the corpus audit
Test / test (push) Successful in 2m32s
Assisted-by: GLM 5.3 Flash
2026-09-23 21:03:16 +02:00
petrbalvin 74d6b90d69 fix(asm): read the arm64 move-wide immediate as an unsigned pattern
Assisted-by: GLM 5.3 Flash
2026-09-23 21:03:03 +02:00
petrbalvin 7b11c62f53 fix(asm): resolve negative numeric PC-relative jumps
Assisted-by: GLM 5.3 Flash
2026-09-23 21:02:50 +02:00
petrbalvin 8eed54b3da feat(lsp): quick fixes for the textflag include and the argument area
Test / test (push) Successful in 2m56s
Assisted-by: GLM 5.3 Flash
2026-09-23 20:23:35 +02:00
petrbalvin 4be16dcdf5 feat(lsp): resolve symbols across workspace files
Assisted-by: GLM 5.3 Flash
2026-09-23 20:21:58 +02:00
petrbalvin ded9cabdf4 fix(lsp): apply each rename edit to its own document
Assisted-by: GLM 5.3 Flash
2026-09-23 20:19:21 +02:00
petrbalvin cf6bc6987e fix(ci): pass the upload file to curl, not its interpolation
Test / test (push) Successful in 2m26s
Release / gates (push) Successful in 2m25s
Release / build (amd64, linux) (push) Successful in 1m15s
Release / build (arm64, linux) (push) Successful in 1m16s
Release / build (loong64, linux) (push) Successful in 1m16s
Release / build (riscv64, linux) (push) Successful in 1m16s
Release / release (push) Successful in 34s
2026-09-22 01:31:49 +02:00
petrbalvin ff7b1452b1 docs: name 0.35.0 as the supported release
Test / test (push) Successful in 2m33s
Release / gates (push) Successful in 2m29s
Release / build (amd64, linux) (push) Successful in 1m18s
Release / build (arm64, linux) (push) Successful in 1m20s
Release / build (loong64, linux) (push) Successful in 1m17s
Release / build (riscv64, linux) (push) Successful in 1m26s
Release / release (push) Failing after 35s
2026-09-22 00:52:56 +02:00
petrbalvin 517c1cea25 chore: prepare release v0.35.0
Test / test (push) Successful in 2m33s
Release / gates (push) Failing after 46s
Release / build (amd64, linux) (push) Skipped
Release / build (arm64, linux) (push) Skipped
Release / build (loong64, linux) (push) Skipped
Release / build (riscv64, linux) (push) Skipped
Release / release (push) Skipped
2026-09-22 00:44:10 +02:00
petrbalvin a3e3010e0f fix(cmd): resolve the runtime header test GOROOT from the go command
Test / test (push) Successful in 2m39s
2026-09-21 22:46:07 +02:00
petrbalvin 057c4eb545 docs: complete the release delta in the changelog and readme 2026-09-21 22:45:56 +02:00
petrbalvin f720381d43 feat(asm): the segment-absolute and crash-store forms GOROOT writes
Test / test (push) Failing after 2m28s
Assisted-by: GLM 5.3 Flash
2026-09-21 22:19:53 +02:00
petrbalvin 2c9042d62c feat(asm): PCALIGN alignment on amd64
Assisted-by: GLM 5.3 Flash
2026-09-21 22:00:30 +02:00
petrbalvin 82ef289d3a feat(asm): the immediate multiply and arm64 indirect branches GOROOT writes
Assisted-by: GLM 5.3 Flash
2026-09-21 21:50:11 +02:00
petrbalvin 7246b0e002 feat(asm): the TLS access pair in the toolchain's one-instruction form
Assisted-by: GLM 5.3 Flash
2026-09-21 21:35:15 +02:00
petrbalvin 8cfd40aac8 feat(asm): the operand forms and defines GOROOT writes
Assisted-by: GLM 5.3 Flash
2026-09-21 21:17:34 +02:00
petrbalvin 5382c9a8e4 feat(audit): list every corpus failure per architecture 2026-09-21 21:17:34 +02:00
petrbalvin 53de91b2df docs(asm): describe the four target architectures
Test / test (push) Failing after 2m23s
Assisted-by: GLM 5.3 Flash
2026-09-21 20:15:55 +02:00
petrbalvin 8a36af7c7d docs(asm): generate the instruction appendices
Assisted-by: GLM 5.3 Flash
2026-09-21 20:15:55 +02:00
petrbalvin e9789ce3f4 chore(arch): regenerate the instruction tables 2026-09-21 20:15:55 +02:00
petrbalvin 837231c068 docs(asm): open the assembly language reference
Assisted-by: GLM 5.3 Flash
2026-09-21 19:49:04 +02:00
petrbalvin 95025be1bc docs(changelog): describe the encoder entries by content
Test / test (push) Failing after 2m33s
2026-09-21 19:20:01 +02:00
petrbalvin 03a964bb2d docs(goobj): document the GOOBJ object file format 2026-09-21 19:19:53 +02:00
petrbalvin 123a16e346 docs(readme): state the documentation goal 2026-09-21 18:35:27 +02:00
petrbalvin 9701812bee docs: changelog for the completeness waves
Test / test (push) Failing after 3m6s
Assisted-by: GLM 5.3 Flash
2026-09-21 02:04:44 +02:00
petrbalvin 29ac03468e feat(amd64): floating-point immediates through a synthesised pool
Assisted-by: GLM 5.3 Flash
2026-09-21 02:04:44 +02:00
184 changed files with 16241 additions and 764 deletions
+42
View File
@@ -0,0 +1,42 @@
# FreeBSD compile gates. Dispatched by hand, never on a push.
#
# The debugger's ptrace surface and the JIT substrate are the two
# FreeBSD-portable layers the tree carries; the forge has no FreeBSD runner,
# so they can only be compile-gated, and three foreign-GOOS builds of the
# whole module are minutes of one-core work the push pipeline's budget cannot
# carry. The push pipeline stays fast and light; this workflow is the
# deliberate run, before a release or after touching the ported layers.
# Running the ptrace suite itself needs real FreeBSD hardware.
#
# A dispatched workflow takes no concurrency block: it is one deliberate run.
name: FreeBSD build
on:
workflow_dispatch:
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
build:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
# The module is the source of truth for the version, so it cannot drift.
go-version-file: go.mod
cache: true
- name: FreeBSD build (amd64)
run: GOOS=freebsd GOARCH=amd64 go build ./...
- name: FreeBSD build (arm64)
run: GOOS=freebsd GOARCH=arm64 go build ./...
- name: FreeBSD build (riscv64)
run: GOOS=freebsd GOARCH=riscv64 go build ./...
+4 -1
View File
@@ -342,7 +342,10 @@ jobs:
my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}}, my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}},
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}}, q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
q{-H}, q{Content-Type: application/octet-stream}, q{-H}, q{Content-Type: application/octet-stream},
q{-X}, q{POST}, q{--data-binary}, qq{@$path}, # The @ must not sit inside a qq{} string: there it starts an
# array interpolation and the upload body collapses to empty,
# which Gitea stores as a 201-created zero-byte attachment.
q{-X}, q{POST}, q{--data-binary}, q{@} . $path,
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name}); qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name});
open(my $curl, q{-|}, @cmd) or die qq{curl: $!}; open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
my $code = <$curl>; my $code = <$curl>;
+19 -11
View File
@@ -6,6 +6,11 @@
# everything runs in one job. Extra jobs would duplicate the checkout, the Go setup and # everything runs in one job. Extra jobs would duplicate the checkout, the Go setup and
# the dependency download three times without buying any parallelism. # the dependency download three times without buying any parallelism.
# #
# The budget is part of the contract: a push run is fast and light, about two minutes,
# and nothing that cannot run natively on the runner belongs here. The FreeBSD compile
# gates live in freebsd.yml behind workflow_dispatch for that reason; the GOOBJ link
# parity campaign is an opt-in local verification (just link-parity).
#
# Every step is one command, so the step that fails is the gate that failed, and no shell # Every step is one command, so the step that fails is the gate that failed, and no shell
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and # option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
# not Python: Perl behaves the same on both runner images, there is no bashism to trip over # not Python: Perl behaves the same on both runner images, there is no bashism to trip over
@@ -46,10 +51,9 @@ jobs:
go-version-file: go.mod go-version-file: go.mod
cache: true cache: true
- name: Install Perl # No Install Perl step: the fedora image carries perl (verified by run
# The runner images are minimal and Perl is not guaranteed. The install is a # 76: the install degraded into a package upgrade costing ~50 s), and a
# no-op where it is already present; drop this step once verified on the box. # dnf on the push path is network work the budget does not need.
run: dnf install -y perl
# The steps follow the `gates` order of the justfile contract: build, format, # The steps follow the `gates` order of the justfile contract: build, format,
# vet, test. The vet gate is go vet and go fix -diff, two steps here. # vet, test. The vet gate is go vet and go fix -diff, two steps here.
@@ -82,7 +86,11 @@ jobs:
# command, so the floor is the same number everywhere. ./verify/... carries the # command, so the floor is the same number everywhere. ./verify/... carries the
# live oracle-parity comparison against `go tool asm` (the TestGroundTruth # live oracle-parity comparison against `go tool asm` (the TestGroundTruth
# suites); the runner's Go setup provides both the tool and GOROOT. # suites); the runner's Go setup provides both the tool and GOROOT.
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/... # -short skips the deliberate-run categories inside the suites (the live
# ptrace sessions above all): they need the machine to themselves and a
# starved single-core runner turns each into a timeout the budget cannot
# carry. The local `just test` gate runs everything, in full.
run: go test -short -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Tests outside the coverage set - name: Tests outside the coverage set
# The CLI and the debugger sit outside `packages` because a thin main and a # The CLI and the debugger sit outside `packages` because a thin main and a
@@ -90,12 +98,12 @@ jobs:
# shipped surfaces: the command exit codes, the manual pages against the # shipped surfaces: the command exit codes, the manual pages against the
# binary's own help, and the debugger's architecture-neutral units. They run # binary's own help, and the debugger's architecture-neutral units. They run
# here so the floor stays a product measure and nothing is left untested. # here so the floor stays a product measure and nothing is left untested.
run: go test -count=1 -timeout 10m ./cmd/... ./debug/... # -short skips the debugger's live ptrace sessions, the deliberate-run
# category the runner cannot starve-proof. The live go-tool-asm oracle
- name: Oracle parity # comparison (TestGroundTruth in ./verify/...) runs inside the coverage
# Re-run the live go-tool-asm comparison as its own step so that a parity # sweep above; it is not re-run as its own step, because every second on
# regression names the gate that failed instead of hiding inside the suite. # this box is budget.
run: go test -count=1 -timeout 10m -run 'TestGroundTruth' ./verify/... run: go test -short -count=1 -timeout 10m ./cmd/... ./debug/...
- name: Coverage floor - name: Coverage floor
run: | run: |
+248 -18
View File
@@ -1,6 +1,6 @@
# Changelog # Changelog
All notable changes to gasm-devkit are documented here. All notable changes to gasm-sdk are documented here.
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
@@ -9,6 +9,178 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added ### Added
- **The FreeBSD port of the debugger.** `gasm debug` runs on FreeBSD on
amd64, arm64 and riscv64 with the same interactive surface as on Linux:
breakpoints, hardware watchpoints (x86 debug registers, the arm64 debug
register file), single-stepping, register and memory access, all behind
the kernel's own ptrace requests, with tracee memory through `PT_IO` and
stop reports through `PT_LWPINFO`. The JIT substrate maps executable
memory through `golang.org/x/sys/unix`, so `verify` builds on FreeBSD
too. The pipeline compile-gates all three architectures; live
validation awaits a FreeBSD machine.
- **Workspace-wide navigation in the language server.** `gasm lsp` indexes
the `.s` files under the workspace root beyond the documents the editor
has open, so go-to-definition, find references and workspace symbol search
reach files that were never opened. An open buffer always shadows its
disk copy, and watched-file events together with a per-query freshness
check keep the index current.
- **Quick fixes for the textflag include and the argument area.** The
`missing-textflag-include` warning offers to add the include after the
last one in the file, and the `abi-argsize` warning offers to set the
TEXT argument area to the size the `// func` signature implies, computed
by the new `lint.ExpectedArgSize`.
- **Eleven new lint rules over the directives, the data section and the
sharpest addressing edges.** `missing-argsize` flags a TEXT that
declares no argument area its `// func` signature implies;
`noframe-frame-size` a NOFRAME with a positive frame;
`unnamed-fp-reference` a nameless `0(FP)`, which both assemblers
reject; `hardware-sp-addressing` a negative offset off the hardware SP
rather than the virtual frame; `vex-sse-mixing` a kernel that mixes VEX
and legacy SSE spellings and pays the transition penalty;
`unnamed-result` a `ret+N(FP)` the signature names; `data-width`,
`data-value-overflow`, `data-string-width`, `data-without-globl` and
`data-exceeds-globl` police the DATA width against its value type and
the GLOBL size behind it. `missing-ret` now also flags a function
whose tail can fall off its end even though a RET sits somewhere in the
body, and `invalid-textflag` also reports a flag misplaced between TEXT
and GLOBL.
- **Hover documentation and wider completions in the language server.**
Hovering a directive or pseudo-operation (TEXT, DATA, GLOBL, PCALIGN,
FUNCDATA, PCDATA, the BYTE family) shows its grammar and rules in
preference to the empty instruction-table entry, and completion offers
those names beside the instruction set. The `missing-argsize` warning
carries a quick fix that declares the argument area the signature
implies (`$0` becomes `$0-16`).
- **The GOOBJ link parity gate.** A regression test assembles a kernel
per architecture through `gasm asm --format goobj`, substitutes the
object into a real `go build`'s package archive, proves the archive
carries it byte for byte and re-links with cmd/link on amd64, arm64,
riscv64 and loong64. The binaries run, natively on amd64 and under
qemu-user on arm64 and riscv64, and their output must match the
toolchain-built baseline; loong64 is link-only. The pipeline installs
qemu-user and runs the gate on every push.
- **The amd64 and loong64 encoders close four more corpus files.** The
whole-tree measure moves to 272 of 322 (84.5 %): amd64 gains the
one-operand IMUL, the SSE compare family (CMPPD, CMPPS, CMPSS), RETFL,
the LOOP family, the MMX register bank with its bank-crossing moves,
MOVNTDQ, the CR and DR register moves, PUSH and POP of FS and GS, the
`(TLS)` pseudo-base, the wait and cache controls (CLWB, CLDEMOTE,
TPAUSE, UMONITOR, UMWAIT, RDPID, ENDBR64), indirect branches with the
star spelling (`JMP *(R12)(R13*4)`), `RET sym(SB)` as the tail jump,
the colon shift spelling (`SHLL CX, R11:AX`) and the EVEX and VEX forms
of the rounds, AES key assist, string compares, extracts, blends and
permutes; loong64 gains the acquire and release pair (LLACQ, SCREL,
with the vector widths), the VMOVQ and XVMOVQ lane forms and the BYTE
literal-data escape hatch the other architectures already take.
### Changed
- **The corpus audit assembles like the build.** A file's `//go:build`
constraint decides which target architectures attempt it: cpu_x86.s is
an x86 build alone, and the msan and goexperiment.runtimesecret trees
are compiled by no supported build, so they leave the measured set
instead of failing it. The headline now reads "assemble for every
applicable target": every real-code GOROOT assembly file, the tree
without testdata, assembles for all four architectures (250 of 250,
100 %); over the whole tree including testdata the measure is 272 of
322 (84.5 %).
- **The module moves to `sourcedock.dev/petrbalvin/gasm-sdk`.** The
repository and the module rename together with the product, now the
GAsm Software Development Kit. Fresh installs become
`go install sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm@latest`, and
installs pinned to the old `gasm-sdk` path stop resolving once the
repository takes the new name: reinstall from the new path. The
binary stays `gasm`.
### Fixed
- **Rename edits land in their own documents.** A rename collected the
ranges of every reference across the open documents but applied them all
to the document that started it, so renaming a symbol used in a second
file moved that file's text into the first. Each edit now applies to the
document it was collected in.
- **Negative numeric PC-relative jumps.** `JMP -3(PC)`, the shape the
runtime's exit loops write (sys_linux_amd64.s, sys_netbsd_amd64.s),
resolved to nothing: only the forward forms counted. A negative count
now walks the same instruction statements backwards, labels excluded,
byte-identical with the toolchain.
- **The arm64 move-wide family reads its immediate as an unsigned
pattern.** `MOVK $(40000<<48)` folds to a negative int64 and was
rejected; the toolchain picks the 16-bit lane from the 64-bit bit
pattern, so the encoder now does the same, and a zero immediate is
rejected where the toolchain rejects it.
- **NOPTR data emits its own symbol kind.** A GLOBL with NOPTR was
emitted as plain SDATA, the kind the linker holds to its Go type
information requirement, so every gasm object carrying runtime-shaped
data (`GLOBL ·x(SB), NOPTR, ...`) died in cmd/link with "missing Go
type information". NOPTR data is now SNOPTRDATA, the toolchain's kind
for pointer-free globals, and RODATA still wins where both flags
appear, exactly as the toolchain chooses. The three end-to-end link
tests that should have caught this substituted the gasm object after
the package archive was already packed, so they passed vacuously; they
now substitute inside the archive, prove the substitution byte for
byte and re-link.
- **Latent encoder divergences against the toolchain, found by the
whole-file differential harness.** On loong64, `MOVx $off(reg)` lost
its base register, SCQ swapped its operand fields, the vector lane
inserts did not scale offsets by the element width and two immediate
opcodes were mistyped; on amd64, NOP with operands encoded 0x90 where
the toolchain emits nothing at all, the VEX gather length bit ignored
the VSIB index width, four permute and extract families were pushed
into EVEX where the toolchain stays VEX, and a reference to a static
symbol no GLOBL defines failed where the toolchain defers it to the
linker as an external relocation.
- **Seven front-end defects found by fuzzing.** The formatter swallowed
the statement after a leading block comment (`/* head */ MOVQ AX, BX`
formatted to the comment alone), dropped trailing tokens after an
`#include` header, and trimmed the trailing whitespace inside a string
literal; the parser peeled only one label of a stacked pair (`a: b:`
parsed differently than it formatted); deeply nested parentheses in an
immediate overflowed the stack through the constant folder, which now
stops at a bounded depth and falls back to the ordinary operand paths;
macro expansion is linear in the invocation count instead of
quadratic; and amplifying macros (the billion-laughs shape) stop at a
work budget sized by the line and the macro table, reported as a
diagnostic instead of running for hours.
- **The debugger handles the edges its tests now reach.** Memory reads
and writes cover exactly the requested bytes, so a request ending in
the last page of a mapping no longer fails on the unmapped page behind
it; disassembly shrinks its instruction window at a mapping's end
instead of failing; a debuggee that dies before signalling readiness
writes a failure notice the launcher reads, so launch fails fast with
the reason instead of after the whole poll (and the poll no longer
waits on the child, which could consume the SIGSTOP park and hang the
launch); amd64 watchpoints acknowledge the sticky DR6 hit bits and
clear the address register on release; the breakpoint listing is
ordered by address so its numbering is stable; the REPL rejects bad
counts, sizes and watchpoint types instead of silently guessing,
reports stops on signals during stepping, and a breakpoint-class trap
that matches no breakpoint and leaves the PC in place surfaces instead
of spinning the continue loop and the coverage run forever.
## [0.35.0] - 2026-09-22
### Added
- **The go_asm.h generator.** `gasm asm` generates the package's go_asm.h
itself when an assembly file includes it: the Go files beside the source
are type-checked for the target architecture and the constants and field
offsets become assembler defines, so package-context files assemble with
no compiler and no `go build` in the loop. `-GOOS` selects the
type-checking GOOS for GOOS-specific files, and the corpus audit derives
the GOOS from the file name.
- **ELF data relocations on arm64, riscv64 and loong64.** `gasm asm
--format elf` emits `.rela.data` for symbol-valued DATA initialisers on
every architecture (amd64 carried them already), so standalone ELF
objects link on all four targets.
- **Corpus failure listing.** `gasm audit-instructions --corpus --list`
prints every failing file with its failure reason, per architecture,
instead of one representative file per reason.
- **DATA with symbol values and relaxed symbol spellings.** DATA
initialisers accept `$symbol(SB)` values, laid down as an absolute
relocation at the data field (GOOBJ on all four architectures and ELF
on all four as of this release), and U+2215 is accepted inside symbol
package paths.
- **Macro expansion and include splicing.** `gasm asm`, `gasm diff` and - **Macro expansion and include splicing.** `gasm asm`, `gasm diff` and
`gasm audit-instructions` now preprocess assembly the way the `gasm audit-instructions` now preprocess assembly the way the
toolchain does: object and parameterised `#define` macros expand at toolchain does: object and parameterised `#define` macros expand at
@@ -19,7 +191,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
(`$(32-7)`, `$~63`, `(index*4)(base)`) fold at parse. Expansion (`$(32-7)`, `$~63`, `(index*4)(base)`) fold at parse. Expansion
happens only on the assembly path: `gasm lint`, `gasm fmt` and the happens only on the assembly path: `gasm lint`, `gasm fmt` and the
language server keep reading the raw file. language server keep reading the raw file.
- **The GOROOT instruction wave, part 1.** The encoder now covers the - **Encoder coverage: the instruction families GOROOT's real code
uses.** The encoder now covers the
instruction families GOROOT's real code uses that gasm lacked, instruction families GOROOT's real code uses that gasm lacked,
byte-verified against `go tool asm`: on amd64 the carry ALU, the byte-verified against `go tool asm`: on amd64 the carry ALU, the
atomics (CMPXCHG, XADD, XCHG), AES-NI, SHA-1/256, PCLMULQDQ, CRC32, atomics (CMPXCHG, XADD, XCHG), AES-NI, SHA-1/256, PCLMULQDQ, CRC32,
@@ -35,7 +208,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
Also fixed on the way: arm64 `CASD`/`CASW` lacked an opcode bit, and Also fixed on the way: arm64 `CASD`/`CASW` lacked an opcode bit, and
riscv64 `VSETVLI` with an immediate length now canonicalises to riscv64 `VSETVLI` with an immediate length now canonicalises to
`vsetivli` as the toolchain does. `vsetivli` as the toolchain does.
- **The GOROOT instruction wave, part 2.** The encoder gains the - **Encoder coverage: quad-register AVX-512 and floating-point
immediates.** The encoder gains the
quad-register AVX-512 families (4FMAPS, 4FNMADD, 4VNNIW, VP4DPWSSD, quad-register AVX-512 families (4FMAPS, 4FNMADD, 4VNNIW, VP4DPWSSD,
VP4DPWSSDS) with the register list riding the inverted V'VVVV field, VP4DPWSSDS) with the register list riding the inverted V'VVVV field,
floating-point immediates on the SSE scalar moves and arithmetic floating-point immediates on the SSE scalar moves and arithmetic
@@ -47,21 +221,77 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
ranges, index-only VSIB memory operands and bare trailing immediates; ranges, index-only VSIB memory operands and bare trailing immediates;
macro substitution reaches parameters used with element suffixes macro substitution reaches parameters used with element suffixes
(`A.S4`), and `;` separates statements in plain files. (`A.S4`), and `;` separates statements in plain files.
- **`gasm asm -GOOS`.** The go_asm.h generator type-checks per target - **Per-architecture reference pages.** [docs/asm/](docs/asm/README.md)
GOOS, so the darwin-only and windows-only runtime files assemble with gains AMD64, ARM64, RISCV64 and LOONG64: the register files and the
their own defines; the corpus audit derives the GOOS from the file roles the ABI fixes, addressing, operand order with every special form,
name. DATA initialisers accept `$symbol(SB)` values (an absolute constants and materialisation, alignment, fences and the relocations
relocation at the data field, GOOBJ on all four architectures and ELF each target emits. An instruction inventory appendix per architecture
on amd64, arm64, riscv64 and loong64), and U+2215 is accepted inside is generated from the toolchain's own tables by `just gen`, and the
symbol package paths. regenerated tables recognise 147 more mnemonics than the previous
- **The corpus audit measures honestly.** Files named for Go ports gasm release carried (arm64 107, riscv64 31, loong64 9).
does not target (arm, 386, s390x, ...) are no longer attempted for the - **The Plan 9 assembly language reference.** [docs/asm/](docs/asm/README.md)
four supported architectures (no supported build compiles them), and opens the complete language reference with its common core: the lexicon,
the headline rate is reported over attemptable files: 136 of 433 on statement structure and constant expressions, the operand grammar with
the full corpus (31.4 %), 135 of 383 on real code (35.2 %), from the the pseudo-registers and symbol naming, the directives and the function
127 that the previous release measured. The probe battery that flag vocabulary, preprocessing with `#define` and `#include`, and the
decides encodability gained the operand shapes the new families use. Go-embedded layer (ABI0, prototypes, `go_asm.h`, `funcdata.h` and the
- runtime contract). Every claim is verified against `go tool asm` of
Go 1.27.1 and gasm's differential tests; the per-architecture pages and
generated instruction appendices follow.
- **GOOBJ format specification.** [docs/GOOBJ.md](docs/GOOBJ.md)
documents the Go object file format in full: both containers, the 96
byte header and all 19 blocks, every structure with its byte
offsets, symbol kinds and flag bits, all 106 relocation types with
the weak variants, aux symbols, the FuncInfo payload, the pc-value
table encoding, the content hashes and the builtin table, all
verified byte for byte against objects produced by Go 1.27.1's own
tools.
### Changed
- **The corpus audit measures like a build.** Files named for a Go port
gasm does not target (arm, 386, s390x, ...) are never attempted, because
no supported build compiles them; the GOOS comes from the file name; and
each target's go_asm.h is generated on the fly. The headline is reported
over attemptable files: 291 of 353 on the full corpus (82.4 %) assemble
for every target architecture and 295 of 303 on real code (97.4 %),
against 108 of 627 over all files (17.2 %) that the previous release
measured.
### Fixed
- **The operand forms GOROOT writes.** Numeric PC-relative jumps
(`JEQ 2(PC)`, the park loop `JMP 0(PC)`) resolve with the toolchain's
own instruction counting and fold jump-to-jump chains exactly as its
branch optimiser does; symbol immediates (`MOVQ $sym(SB), AX`)
assemble to the toolchain's RIP-relative LEA with an R_PCREL
relocation; negated constant expressions in operands (`ADJSP
$-(REGS - 8)`, the shape the cgo ABI macros write) fold; the immediate
multiply (`IMULQ $1000000000, AX`) encodes with the toolchain's
0x69/0x6B selection; the TLS access pair assembles as the toolchain's
one-instruction form (the bare `MOVQ TLS, r` load nops out and
`off(r)(TLS*1)` folds to the segment-prefixed absolute whose disp32
carries the R_TLSLE relocation, per-GOOS); arm64 accepts the
bare-register indirect branch (`BL R9` beside `BL (R9)`, both BLR) and
the zero-immediate store (`MOVD $0, mem` through the zero register,
rejecting non-zero immediates as the toolchain does); `PCALIGN` now
aligns on amd64, padding with the toolchain's greedy
single-instruction NOPs; the segment-absolute forms (`MOVQ 0x30(GS),
AX` and the store direction) and the absolute crash-store
(`MOVL $0xf1, 0xf1`) encode; and `gasm asm` predefines the
`GOARCH_<arch>` and `GOOS_<goos>` macros the go command passes to
`go tool asm`, so GOROOT headers' `#ifdef GOARCH_amd64` platform
blocks (`go_tls.h`'s `get_tls` and friends) select as intended. The
GOROOT corpus measure moves to 291 of 353 files assembling for every
target architecture (82.4 %), 97.4 % of the real-code corpus, from
70.8 % and 82.2 %.
- **Tool corrections across the pipeline.** The formatter keeps square
brackets in SIMD operands, statement separators and canonical macro
bodies; the linter drops false positives on shift counts, SETcc
spellings and ABIInternal references; the lexer treats a trailing
carriage return as a line end so comment text stays idempotent; and
arm64 rejects bare BTI with a diagnostic while accepting the full
family.
## [0.34.0] - 2026-09-20 ## [0.34.0] - 2026-09-20
+4 -4
View File
@@ -1,6 +1,6 @@
# Contributing # Contributing
Contributions to **gasm-devkit** are governed by the Contributor terms Contributions to **gasm-sdk** are governed by the Contributor terms
below; submitting one means you accept them. below; submitting one means you accept them.
## Contributor terms ## Contributor terms
@@ -29,8 +29,8 @@ compiler (gcc), because `just gates` includes `just race` and the race
detector needs cgo. detector needs cgo.
```sh ```sh
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git git clone https://sourcedock.dev/petrbalvin/gasm-sdk.git
cd gasm-devkit cd gasm-sdk
just build just build
just gates just gates
``` ```
@@ -126,7 +126,7 @@ tag, where it would double the time and the memory a shared runner cannot spare.
## Reporting bugs ## Reporting bugs
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-devkit/issues` with the Open an issue at `https://sourcedock.dev/petrbalvin/gasm-sdk/issues` with the
version, the operating system and architecture, the exact command, the full output, version, the operating system and architecture, the exact command, the full output,
and the expected against the actual behaviour. and the expected against the actual behaviour.
+66 -20
View File
@@ -1,6 +1,6 @@
# Plan 9 assembly tooling, inside and outside Go # GAsm: Software Development Kit for Plan 9 Assembly
> **Warning: this is an experiment.** gasm-devkit is under active > **Warning: this is an experiment.** gasm-sdk is under active
> development and is not stable. The version is 0.x.x: commands, flags, > development and is not stable. The version is 0.x.x: commands, flags,
> output formats and behaviour can change without warning at any time. > output formats and behaviour can change without warning at any time.
> A 1.0.0 release is light years away. Nothing in this document is a > A 1.0.0 release is light years away. Nothing in this document is a
@@ -13,7 +13,7 @@
there is no formatter, no linter and no debugger for `.s` files, and no there is no formatter, no linter and no debugger for `.s` files, and no
assembler that works without a Go installation. Developers write assembler that works without a Go installation. Developers write
assembly blind, validate it by benchmark, and debug it by print assembly blind, validate it by benchmark, and debug it by print
statement. gasm-devkit is the missing toolkit: a single, self-contained statement. gasm-sdk is the missing toolkit: a single, self-contained
binary, `gasm`, that serves both purposes. binary, `gasm`, that serves both purposes.
- **Help develop Plan 9 assembly.** Formatting, linting, disassembly, - **Help develop Plan 9 assembly.** Formatting, linting, disassembly,
@@ -58,7 +58,7 @@ Plan 9 (Go): MOVQ AX, total-16(SP)
The same lines, but only one of them tells you what the number is for. The same lines, but only one of them tells you what the number is for.
The syntax is uppercase, regular and boring, which is the highest The syntax is uppercase, regular and boring, which is the highest
compliment a language for machine code can earn. gasm-devkit exists compliment a language for machine code can earn. gasm-sdk exists
to give that syntax the tooling it deserves. to give that syntax the tooling it deserves.
## Features ## Features
@@ -81,7 +81,10 @@ to give that syntax the tooling it deserves.
GOOBJ format, which needs the installed toolchain and which `go build` GOOBJ format, which needs the installed toolchain and which `go build`
consumes in place of the toolchain's output. Framed functions get the consumes in place of the toolchain's output. Framed functions get the
stack-split guard and the morestack block, byte-identical to the stack-split guard and the morestack block, byte-identical to the
toolchain's, so split functions link too. toolchain's, so split functions link too. The assembler preprocesses
like the toolchain (`#define`, `#include` with `-I`, `#ifdef`), generates
`go_asm.h` from the package's Go files, and carries `PCALIGN`, the
`LOCK`/`REP` prefixes and the literal-data pseudo-ops.
- **Disassembler.** `gasm dis` lists a `.s` file's functions at their real - **Disassembler.** `gasm dis` lists a `.s` file's functions at their real
offsets after assembling, or disassembles raw bytes from a file or stdin. offsets after assembling, or disassembles raw bytes from a file or stdin.
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into - **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
@@ -91,13 +94,16 @@ to give that syntax the tooling it deserves.
- **Debugger.** `gasm debug` is a source-level ptrace debugger with - **Debugger.** `gasm debug` is a source-level ptrace debugger with
breakpoints (optionally conditional), hardware watchpoints, register and breakpoints (optionally conditional), hardware watchpoints, register and
memory inspection, and headless script runs that report instruction and memory inspection, and headless script runs that report instruction and
label coverage. label coverage; it runs on Linux (all four architectures) and FreeBSD
(amd64, arm64, riscv64).
- **Language server.** `gasm lsp` serves completion, hover, document symbols, - **Language server.** `gasm lsp` serves completion, hover, document symbols,
push and pull diagnostics, semantic-token highlighting, go-to-definition, push and pull diagnostics, semantic-token highlighting, go-to-definition,
find references, rename, formatting, inlay hints, code actions, signature find references, rename, formatting, inlay hints, code actions, signature
help, document highlights, workspace symbol search, #include document help, document highlights, workspace symbol search, #include document
links and folding ranges over stdio; definition, references and rename links and folding ranges over stdio; definition, references and rename
work across every open document. work across every open document and the indexed workspace files beyond
them, and the quick fixes add a missing textflag.h include and set the
argument area from the // func signature.
- **Comparators and audits.** `gasm diff` compares the machine code of two - **Comparators and audits.** `gasm diff` compares the machine code of two
assembly files byte-for-byte, `gasm profile` shows basic-block structure, assembly files byte-for-byte, `gasm profile` shows basic-block structure,
`gasm audit-instructions` diffs the encoder against the installed toolchain, `gasm audit-instructions` diffs the encoder against the installed toolchain,
@@ -110,9 +116,9 @@ Four architectures, the four that matter in practice:
| Architecture | GOARCH | File suffix | Instructions recognised | | Architecture | GOARCH | File suffix | Instructions recognised |
|--------------|-------------|--------------|---------------------------------------------| |--------------|-------------|--------------|---------------------------------------------|
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases | | AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
| ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes | | ARM64 | `arm64` | `_arm64.s` | 645 + common opcodes |
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes | | RISC-V | `riscv64` | `_riscv64.s` | 992 + common opcodes |
| LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes | | LoongArch | `loong64` | `_loong64.s` | 808 + common opcodes |
"Common opcodes" are the instructions shared by every architecture (`RET`, "Common opcodes" are the instructions shared by every architecture (`RET`,
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally `JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
@@ -124,10 +130,13 @@ can emit today is narrower, and a recognised but unencodable instruction is
reported as an explicit error, never as a wrong byte. reported as an explicit error, never as a wrong byte.
The same measurement runs over GOROOT's whole assembly corpus: The same measurement runs over GOROOT's whole assembly corpus:
`gasm audit-instructions --corpus` reports 136 of 433 attemptable files `gasm audit-instructions --corpus` reports every real-code GOROOT assembly
(31.4 %) assembling for every target architecture today (files named for file (the tree without testdata) assembling for every target its build
other Go ports are counted but never attempted), with the top failure admits: 250 of 250, 100 %. Over the whole tree including testdata the
reasons per architecture; the number moves with every release. measure is 271 of 322 attemptable (84.2 %); files named for other Go ports
are counted but never attempted, and `//go:build` constraints decide which
targets attempt a file at all, exactly as the build does. The number moves
with every release.
### Validation status ### Validation status
@@ -142,7 +151,7 @@ actually been executed.
|---|---|---| |---|---|---|
| Encoding: byte-for-byte against `go tool asm` | native hardware | native hardware (the toolchain cross-assembles any GOARCH on any host) | | Encoding: byte-for-byte against `go tool asm` | native hardware | native hardware (the toolchain cross-assembles any GOARCH on any host) |
| Execution: JIT calls, ABI checks, differential fuzzing | native hardware | qemu-user emulation | | Execution: JIT calls, ABI checks, differential fuzzing | native hardware | qemu-user emulation |
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more | | Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more. FreeBSD (amd64, arm64, riscv64) is in the same position: the port compiles behind the cross-build gate and its integration test is ready, but no FreeBSD machine has executed it |
Consequences, stated plainly. An emulator is a model of a CPU, not the Consequences, stated plainly. An emulator is a model of a CPU, not the
CPU: instruction semantics are implemented in software and can differ CPU: instruction semantics are implemented in software and can differ
@@ -157,6 +166,39 @@ been compiled and read, never executed. Its architecture-neutral units
run under `go test ./...`, which the race workflow and a manual run run under `go test ./...`, which the race workflow and a manual run
perform; the default `just test` gate does not sweep `./debug/...`. perform; the default `just test` gate does not sweep `./debug/...`.
## The documentation goal
The toolkit is the primary goal. The secondary one is documentation: a
specification of the Plan 9 assembly language and of the GOOBJ object
format that is 100 % complete, detailed enough to implement against,
and written to a professional standard. These are the two subjects this
project works with every day, and they are the two for which no usable
documentation exists.
Go documents the language on a single page, "A Quick Guide to Go's
Assembler", which carries no section for loong64, one of the four
architectures gasm supports, and covers a fraction of what each
assembler accepts. What exists beyond it lives as comments inside the
toolchain's internal source: per-architecture reference manuals for
arm64, ppc64, riscv64 and loong64, written for the toolchain's own
maintainers rather than for an outside reader, and none at all for
amd64. GOOBJ fares worst of all. The format that `go build` consumes
has no specification anywhere: it is described by a comment in an
internal package, it is not a stable interface, and it can change with
any toolchain release.
The gap is therefore filled the only way it can be filled: by reverse
engineering the toolchain itself, the same work the encoders already
perform. Most of the documentation can come from nowhere else, and it
is written as that knowledge is produced during development. It is
verified the way the code is verified: an encoding documented here is
one that differential tests against `go tool asm` confirm
byte-for-byte, and a format field documented here is one the linker
demonstrably reads. The work has begun: [docs/GOOBJ.md](docs/GOOBJ.md)
specifies the object file format completely, and
[docs/asm/README.md](docs/asm/README.md) opens the language reference
with its common core. The per-architecture pages follow.
## Direction ## Direction
The plan, in the order it is being worked: The plan, in the order it is being worked:
@@ -179,9 +221,11 @@ The plan, in the order it is being worked:
toolchain itself does not support; through ELF, Plan 9 assembly becomes toolchain itself does not support; through ELF, Plan 9 assembly becomes
usable outside Go entirely. usable outside Go entirely.
- **Platforms: Linux and FreeBSD.** Linux is supported today on all four - **Platforms: Linux and FreeBSD.** Linux is supported today on all four
architectures and is where the binary builds. FreeBSD follows: the architectures and is where the binary builds. FreeBSD follows on amd64,
JIT's executable-memory mapping and the ptrace debugger layer are the arm64 and riscv64: the JIT's executable-memory mapping and the ptrace
two pieces of porting work. Other unix systems may follow those two. debugger layer are ported (the debugger's live validation awaits a
FreeBSD machine, as the validation status states). Other unix systems
may follow those two.
- **Four architectures, no more.** amd64, arm64, riscv64 and loong64. - **Four architectures, no more.** amd64, arm64, riscv64 and loong64.
No others are planned. No others are planned.
@@ -189,11 +233,11 @@ The plan, in the order it is being worked:
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
linux/loong64 are on the linux/loong64 are on the
[releases page](https://sourcedock.dev/petrbalvin/gasm-devkit/releases). [releases page](https://sourcedock.dev/petrbalvin/gasm-sdk/releases).
From source (Go 1.27.1): From source (Go 1.27.1):
```sh ```sh
go install sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm@latest go install sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm@latest
``` ```
Or from a repository checkout: Or from a repository checkout:
@@ -282,6 +326,8 @@ recipe.
~/.local/share/man (MANDIR overrides); `just uninstall-man` removes ~/.local/share/man (MANDIR overrides); `just uninstall-man` removes
them them
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow - [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/GOOBJ.md](docs/GOOBJ.md): the GOOBJ object file format specification
- [docs/asm/](docs/asm/README.md): the Plan 9 assembly language reference
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes - [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
- [CHANGELOG.md](CHANGELOG.md): release history - [CHANGELOG.md](CHANGELOG.md): release history
+1 -1
View File
@@ -7,7 +7,7 @@ releases do not receive them.
| Version | Supported | | Version | Supported |
|---|---| |---|---|
| 0.34.0 | yes | | 0.35.0 | yes |
| older releases | no | | older releases | no |
## Reporting a vulnerability ## Reporting a vulnerability
+105 -7
View File
@@ -5,9 +5,13 @@
// toolchain's own assembler source. Go's Plan 9 assembler defines the exact, // toolchain's own assembler source. Go's Plan 9 assembler defines the exact,
// complete set of mnemonics it accepts for each architecture in // complete set of mnemonics it accepts for each architecture in
// $GOROOT/src/cmd/internal/obj/<arch>/anames.go; this tool extracts those // $GOROOT/src/cmd/internal/obj/<arch>/anames.go; this tool extracts those
// names so gasm-devkit supports every instruction the real assembler does, // names so gasm-sdk supports every instruction the real assembler does,
// with no hand-maintained (and therefore inevitably incomplete) lists. // with no hand-maintained (and therefore inevitably incomplete) lists.
// //
// The same data feeds the generated instruction appendices of the assembly
// language reference, docs/asm/INSTRUCTIONS-<ARCH>.md, so that the reference
// cannot drift from the tables it documents.
//
// Usage (via the justfile): // Usage (via the justfile):
// //
// just gen // just gen
@@ -26,9 +30,12 @@ import (
"path/filepath" "path/filepath"
"sort" "sort"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
) )
// archDirs maps a gasm-devkit architecture name to its obj sub-directory. // archDirs maps a gasm-sdk architecture name to its obj sub-directory.
var archDirs = []struct { var archDirs = []struct {
arch string arch string
sub string sub string
@@ -39,11 +46,30 @@ var archDirs = []struct {
{"loong64", "loong64"}, {"loong64", "loong64"},
} }
// docPages maps an architecture to its generated appendix in the language
// reference. The amd64 page carries a per-mnemonic encodability column,
// decided by asm.Encodable, which mirrors the encoder's own dispatch; the
// other targets have no single cheap predicate, so their pages carry the
// inventory and point at the live measurement instead.
var docPages = []struct {
arch arch.Arch
title string
file string
anames string
encodable bool
}{
{arch.AMD64, "AMD64", "INSTRUCTIONS-AMD64.md", "cmd/internal/obj/x86/anames.go", true},
{arch.ARM64, "ARM64", "INSTRUCTIONS-ARM64.md", "cmd/internal/obj/arm64/anames.go", false},
{arch.RISCV, "RISC-V 64", "INSTRUCTIONS-RISCV64.md", "cmd/internal/obj/riscv/anames.go", false},
{arch.LOONG64, "LoongArch 64", "INSTRUCTIONS-LOONG64.md", "cmd/internal/obj/loong64/anames.go", false},
}
func main() { func main() {
goroot := strings.TrimSpace(runGoEnvGOROOT()) goroot := strings.TrimSpace(runGoEnvGOROOT())
if goroot == "" { if goroot == "" {
fatal("could not determine GOROOT") fatal("could not determine GOROOT")
} }
version := strings.TrimSpace(runGoEnv("GOVERSION"))
// The common opcodes shared by every architecture (RET, JMP, NOP, CALL, // The common opcodes shared by every architecture (RET, JMP, NOP, CALL,
// TEXT, FUNCDATA, …) live in cmd/internal/obj/util.go. // TEXT, FUNCDATA, …) live in cmd/internal/obj/util.go.
commonPath := filepath.Join(goroot, "src", "cmd", "internal", "obj", "util.go") commonPath := filepath.Join(goroot, "src", "cmd", "internal", "obj", "util.go")
@@ -57,16 +83,24 @@ func main() {
} }
fmt.Printf("%-8s %4d instructions -> arch/common_gen.go\n", "common", len(common)) fmt.Printf("%-8s %4d instructions -> arch/common_gen.go\n", "common", len(common))
names := map[string][]string{}
for _, a := range archDirs { for _, a := range archDirs {
path := filepath.Join(goroot, "src", "cmd", "internal", "obj", a.sub, "anames.go") path := filepath.Join(goroot, "src", "cmd", "internal", "obj", a.sub, "anames.go")
names, err := extractInstrs(path) names[a.arch], err = extractInstrs(path)
if err != nil { if err != nil {
fatal("extract %s: %v", a.arch, err) fatal("extract %s: %v", a.arch, err)
} }
if err := writeGen(a.arch, a.sub, names); err != nil { if err := writeGen(a.arch, a.sub, names[a.arch]); err != nil {
fatal("write %s: %v", a.arch, err) fatal("write %s: %v", a.arch, err)
} }
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names), a.arch) fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names[a.arch]), a.arch)
}
for _, p := range docPages {
if err := writeDocPage(p.arch, p.title, p.file, p.anames, version, p.encodable); err != nil {
fatal("write %s: %v", p.file, err)
}
fmt.Printf("%-8s -> docs/asm/%s\n", p.arch, p.file)
} }
} }
@@ -85,7 +119,7 @@ func filterCommon(names []string) []string {
// writeCommon emits arch/common_gen.go. // writeCommon emits arch/common_gen.go.
func writeCommon(names []string) error { func writeCommon(names []string) error {
var b strings.Builder var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n") b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n") b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
b.WriteString("//\n") b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n") b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
@@ -156,7 +190,7 @@ func stringLit(elt ast.Expr) string {
// writeGen emits arch/<arch>_gen.go. // writeGen emits arch/<arch>_gen.go.
func writeGen(arch, sub string, names []string) error { func writeGen(arch, sub string, names []string) error {
var b strings.Builder var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n") b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n") b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
b.WriteString("//\n") b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n") b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
@@ -172,6 +206,61 @@ func writeGen(arch, sub string, names []string) error {
return os.WriteFile(filepath.Join("arch", arch+"_gen.go"), []byte(b.String()), 0o644) return os.WriteFile(filepath.Join("arch", arch+"_gen.go"), []byte(b.String()), 0o644)
} }
// writeDocPage emits docs/asm/<file>, the generated instruction appendix of
// the language reference for one architecture: every mnemonic the toolchain
// accepts, with the curated summary where the architecture table carries one
// and, on amd64, a per-mnemonic encodability column.
func writeDocPage(a arch.Arch, title, file, anames, version string, encodable bool) error {
table := arch.ForArch(a)
instrs := table.Instructions()
var b strings.Builder
b.WriteString("# " + title + ": instruction inventory\n\n")
b.WriteString("Generated by gasm-sdk's `_gen` from the Go toolchain's instruction table\n")
b.WriteString("(`" + anames + "`, " + version + "); DO NOT EDIT. This page lists every mnemonic\n")
b.WriteString("`go tool asm` accepts on this target, which is the upper bound of the\n")
b.WriteString("language on it: a name absent here is not an instruction of the target,\n")
b.WriteString("and a name present here may still be one gasm's encoder cannot emit yet.\n\n")
encodableCount := 0
if encodable {
b.WriteString("The `gasm encodes` column reports whether gasm's encoder can emit the\n")
b.WriteString("mnemonic today; the gap is the encoder backlog, measured live by\n")
b.WriteString("`gasm audit-instructions`.\n\n")
b.WriteString("| Mnemonic | gasm encodes | Notes |\n")
b.WriteString("|---|---|---|\n")
for _, in := range instrs {
ok := asm.Encodable(in.Name)
if ok {
encodableCount++
}
b.WriteString("| `" + in.Name + "` | " + yesNo(ok) + " | " + in.Summary + " |\n")
}
b.WriteString("\n")
fmt.Fprintf(&b, "Recognised: %d mnemonics. gasm encodes: %d.\n", len(instrs), encodableCount)
} else {
b.WriteString("The inventory carries no per-mnemonic encoder column: on this target\n")
b.WriteString("encodability is decided per operand shape, and the live measured\n")
b.WriteString("coverage is reported by `gasm audit-instructions`.\n\n")
b.WriteString("| Mnemonic | Notes |\n")
b.WriteString("|---|---|\n")
for _, in := range instrs {
b.WriteString("| `" + in.Name + "` | " + in.Summary + " |\n")
}
b.WriteString("\n")
fmt.Fprintf(&b, "Recognised: %d mnemonics.\n", len(instrs))
}
return os.WriteFile(filepath.Join("docs", "asm", file), []byte(b.String()), 0o644)
}
// yesNo renders a boolean as the word the appendix tables use.
func yesNo(v bool) string {
if v {
return "yes"
}
return "no"
}
func runGoEnvGOROOT() string { func runGoEnvGOROOT() string {
out, err := exec.Command("go", "env", "GOROOT").Output() out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil { if err != nil {
@@ -180,6 +269,15 @@ func runGoEnvGOROOT() string {
return string(out) return string(out)
} }
// runGoEnv runs `go env` for a single variable.
func runGoEnv(name string) string {
out, err := exec.Command("go", "env", name).Output()
if err != nil {
return ""
}
return string(out)
}
func fatal(format string, args ...any) { func fatal(format string, args ...any) {
fmt.Fprintf(os.Stderr, "gen: "+format+"\n", args...) fmt.Fprintf(os.Stderr, "gen: "+format+"\n", args...)
os.Exit(1) os.Exit(1)
+1 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT. // Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain. // Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
// //
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
+108 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT. // Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain. // Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
// //
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
@@ -364,6 +364,8 @@ var arm64GeneratedInstrs = []string{
"REVW", "REVW",
"ROR", "ROR",
"RORW", "RORW",
"RPRFM",
"SB",
"SBC", "SBC",
"SBCS", "SBCS",
"SBCSW", "SBCSW",
@@ -477,23 +479,68 @@ var arm64GeneratedInstrs = []string{
"UXTH", "UXTH",
"UXTHW", "UXTHW",
"UXTW", "UXTW",
"VABS",
"VADD", "VADD",
"VADDP", "VADDP",
"VADDV", "VADDV",
"VAND", "VAND",
"VBCAX", "VBCAX",
"VBIC",
"VBIF", "VBIF",
"VBIT", "VBIT",
"VBSL", "VBSL",
"VCLS",
"VCLZ",
"VCMEQ", "VCMEQ",
"VCMGE",
"VCMGT",
"VCMHI",
"VCMHS",
"VCMLE",
"VCMLT",
"VCMTST", "VCMTST",
"VCNT", "VCNT",
"VDUP", "VDUP",
"VEOR", "VEOR",
"VEOR3", "VEOR3",
"VEXT", "VEXT",
"VFABS",
"VFADD",
"VFADDP",
"VFCMEQ",
"VFCMGE",
"VFCMGT",
"VFCMLE",
"VFCMLT",
"VFCVTL",
"VFCVTL2",
"VFCVTN",
"VFCVTN2",
"VFCVTZS",
"VFCVTZU",
"VFDIV",
"VFMAX",
"VFMAXNM",
"VFMAXNMP",
"VFMAXNMV",
"VFMAXP",
"VFMAXV",
"VFMIN",
"VFMINNM",
"VFMINNMP",
"VFMINNMV",
"VFMINP",
"VFMINV",
"VFMLA", "VFMLA",
"VFMLS", "VFMLS",
"VFMUL",
"VFNEG",
"VFRINTM",
"VFRINTN",
"VFRINTP",
"VFRINTZ",
"VFSQRT",
"VFSUB",
"VLD1", "VLD1",
"VLD1R", "VLD1R",
"VLD2", "VLD2",
@@ -502,11 +549,17 @@ var arm64GeneratedInstrs = []string{
"VLD3R", "VLD3R",
"VLD4", "VLD4",
"VLD4R", "VLD4R",
"VMLA",
"VMLS",
"VMOV", "VMOV",
"VMOVD", "VMOVD",
"VMOVI", "VMOVI",
"VMOVQ", "VMOVQ",
"VMOVS", "VMOVS",
"VMUL",
"VNEG",
"VNOT",
"VORN",
"VORR", "VORR",
"VPMULL", "VPMULL",
"VPMULL2", "VPMULL2",
@@ -515,14 +568,47 @@ var arm64GeneratedInstrs = []string{
"VREV16", "VREV16",
"VREV32", "VREV32",
"VREV64", "VREV64",
"VSCVTF",
"VSHADD",
"VSHL", "VSHL",
"VSHRN",
"VSHRN2",
"VSLI", "VSLI",
"VSMAX",
"VSMAXP",
"VSMAXV",
"VSMIN",
"VSMINP",
"VSMINV",
"VSMLAL",
"VSMLAL2",
"VSMLSL",
"VSMLSL2",
"VSMULL",
"VSMULL2",
"VSQABS",
"VSQADD",
"VSQNEG",
"VSQSHL",
"VSQSUB",
"VSQXTN",
"VSQXTN2",
"VSQXTUN",
"VSQXTUN2",
"VSRHADD",
"VSRI", "VSRI",
"VSRSHR",
"VSSHL",
"VSSHLL",
"VSSHLL2",
"VSSHR",
"VST1", "VST1",
"VST2", "VST2",
"VST3", "VST3",
"VST4", "VST4",
"VSUB", "VSUB",
"VSXTL",
"VSXTL2",
"VTBL", "VTBL",
"VTBX", "VTBX",
"VTRN1", "VTRN1",
@@ -530,8 +616,27 @@ var arm64GeneratedInstrs = []string{
"VUADDLV", "VUADDLV",
"VUADDW", "VUADDW",
"VUADDW2", "VUADDW2",
"VUCVTF",
"VUHADD",
"VUMAX", "VUMAX",
"VUMAXP",
"VUMAXV",
"VUMIN", "VUMIN",
"VUMINP",
"VUMINV",
"VUMLAL",
"VUMLAL2",
"VUMLSL",
"VUMLSL2",
"VUMULL",
"VUMULL2",
"VUQADD",
"VUQSHL",
"VUQSUB",
"VUQXTN",
"VUQXTN2",
"VURHADD",
"VUSHL",
"VUSHLL", "VUSHLL",
"VUSHLL2", "VUSHLL2",
"VUSHR", "VUSHR",
@@ -541,6 +646,8 @@ var arm64GeneratedInstrs = []string{
"VUZP1", "VUZP1",
"VUZP2", "VUZP2",
"VXAR", "VXAR",
"VXTN",
"VXTN2",
"VZIP1", "VZIP1",
"VZIP2", "VZIP2",
"WFE", "WFE",
+1 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT. // Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/util.go from the Go toolchain. // Source: cmd/internal/obj/util.go from the Go toolchain.
// //
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
+10 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT. // Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain. // Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
// //
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
@@ -152,6 +152,8 @@ var loong64GeneratedInstrs = []string{
"FNMADDF", "FNMADDF",
"FNMSUBD", "FNMSUBD",
"FNMSUBF", "FNMSUBF",
"FRINTD",
"FRINTF",
"FSCALEBD", "FSCALEBD",
"FSCALEBF", "FSCALEBF",
"FSEL", "FSEL",
@@ -177,7 +179,10 @@ var loong64GeneratedInstrs = []string{
"FTINTWF", "FTINTWF",
"JIRL", "JIRL",
"LL", "LL",
"LLACQV",
"LLACQW",
"LLV", "LLV",
"LLW",
"LU12IW", "LU12IW",
"LU32ID", "LU32ID",
"LU52ID", "LU52ID",
@@ -248,7 +253,11 @@ var loong64GeneratedInstrs = []string{
"ROTR", "ROTR",
"ROTRV", "ROTRV",
"SC", "SC",
"SCQ",
"SCRELV",
"SCRELW",
"SCV", "SCV",
"SCW",
"SGT", "SGT",
"SGTU", "SGTU",
"SLL", "SLL",
+32 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT. // Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain. // Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
// //
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
@@ -81,6 +81,9 @@ var riscvGeneratedInstrs = []string{
"CLD", "CLD",
"CLDSP", "CLDSP",
"CLI", "CLI",
"CLMUL",
"CLMULH",
"CLMULR",
"CLUI", "CLUI",
"CLW", "CLW",
"CLWSP", "CLWSP",
@@ -95,13 +98,20 @@ var riscvGeneratedInstrs = []string{
"CSDSP", "CSDSP",
"CSLLI", "CSLLI",
"CSRAI", "CSRAI",
"CSRC",
"CSRCI",
"CSRLI", "CSRLI",
"CSRR",
"CSRRC", "CSRRC",
"CSRRCI", "CSRRCI",
"CSRRS", "CSRRS",
"CSRRSI", "CSRRSI",
"CSRRW", "CSRRW",
"CSRRWI", "CSRRWI",
"CSRS",
"CSRSI",
"CSRW",
"CSRWI",
"CSUB", "CSUB",
"CSUBW", "CSUBW",
"CSW", "CSW",
@@ -259,6 +269,7 @@ var riscvGeneratedInstrs = []string{
"ORCB", "ORCB",
"ORI", "ORI",
"ORN", "ORN",
"PAUSE",
"RDCYCLE", "RDCYCLE",
"RDINSTRET", "RDINSTRET",
"RDTIME", "RDTIME",
@@ -322,6 +333,8 @@ var riscvGeneratedInstrs = []string{
"VADDVI", "VADDVI",
"VADDVV", "VADDVV",
"VADDVX", "VADDVX",
"VANDNVV",
"VANDNVX",
"VANDVI", "VANDVI",
"VANDVV", "VANDVV",
"VANDVX", "VANDVX",
@@ -329,8 +342,17 @@ var riscvGeneratedInstrs = []string{
"VASUBUVX", "VASUBUVX",
"VASUBVV", "VASUBVV",
"VASUBVX", "VASUBVX",
"VBREV8V",
"VBREVV",
"VCLMULHVV",
"VCLMULHVX",
"VCLMULVV",
"VCLMULVX",
"VCLZV",
"VCOMPRESSVM", "VCOMPRESSVM",
"VCPOPM", "VCPOPM",
"VCPOPV",
"VCTZV",
"VDIVUVV", "VDIVUVV",
"VDIVUVX", "VDIVUVX",
"VDIVVV", "VDIVVV",
@@ -743,10 +765,16 @@ var riscvGeneratedInstrs = []string{
"VREMUVX", "VREMUVX",
"VREMVV", "VREMVV",
"VREMVX", "VREMVX",
"VREV8V",
"VRGATHEREI16VV", "VRGATHEREI16VV",
"VRGATHERVI", "VRGATHERVI",
"VRGATHERVV", "VRGATHERVV",
"VRGATHERVX", "VRGATHERVX",
"VROLVV",
"VROLVX",
"VRORVI",
"VRORVV",
"VRORVX",
"VRSUBVI", "VRSUBVI",
"VRSUBVX", "VRSUBVX",
"VS1RV", "VS1RV",
@@ -950,6 +978,9 @@ var riscvGeneratedInstrs = []string{
"VWMULVX", "VWMULVX",
"VWREDSUMUVS", "VWREDSUMUVS",
"VWREDSUMVS", "VWREDSUMVS",
"VWSLLVI",
"VWSLLVV",
"VWSLLVX",
"VWSUBUVV", "VWSUBUVV",
"VWSUBUVX", "VWSUBUVX",
"VWSUBUWV", "VWSUBUWV",
+15 -66
View File
@@ -11,7 +11,7 @@ import (
"strings" "strings"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// TestGOObjectAARCH64Structure checks the basic structure of the emitted // TestGOObjectAARCH64Structure checks the basic structure of the emitted
@@ -178,26 +178,10 @@ func main() {
if err != nil { if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog) t.Fatalf("baseline build: %v\n%s", err, buildLog)
} }
var work, linkLine, asmObj string st := parseBuildLog(t, buildLog, "main_arm64.s")
for line := range strings.SplitSeq(string(buildLog), "\n") { defer os.RemoveAll(st.work)
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
}
defer os.RemoveAll(work)
// Expand $WORK in the object path. // Assemble the same source with gasm and substitute the object.
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
// Read the toolchain-produced object and assemble the same source with gasm.
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s")) src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
@@ -210,30 +194,16 @@ func main() {
if err != nil { if err != nil {
t.Fatalf("AssembleFileARM64: %v", err) t.Fatalf("AssembleFileARM64: %v", err)
} }
gasmObj, err := img.GOObjectAARCH64("a64link", "main_arm64.s") // The package path is "main", the prefix the Go code's references carry.
gasmObj, err := img.GOObjectAARCH64("main", "main_arm64.s")
if err != nil { if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err) t.Fatalf("GOObjectAARCH64: %v", err)
} }
substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"),
// Replace the toolchain-produced object with gasm's. gasmObj, "GOARCH=arm64")
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
// Re-link.
if linkLine == "" {
t.Skip("could not find link command in build log")
}
// Expand $WORK in the link command.
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
// Verify the binary exists and contains the symbol. // Verify the binary exists and contains the symbol.
binPath := filepath.Join(dir, "prog") binPath := filepath.Join(dir, "prog2")
if _, err := os.Stat(binPath); err != nil { if _, err := os.Stat(binPath); err != nil {
t.Fatalf("binary not found: %v", err) t.Fatalf("binary not found: %v", err)
} }
@@ -296,23 +266,8 @@ func main() {
if err != nil { if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog) t.Fatalf("baseline build: %v\n%s", err, buildLog)
} }
var work, linkLine, asmObj string st := parseBuildLog(t, buildLog, "main_arm64.s")
for line := range strings.SplitSeq(string(buildLog), "\n") { defer os.RemoveAll(st.work)
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || linkLine == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
}
defer os.RemoveAll(work)
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s")) src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
if err != nil { if err != nil {
@@ -326,19 +281,13 @@ func main() {
if err != nil { if err != nil {
t.Fatalf("AssembleFileARM64: %v", err) t.Fatalf("AssembleFileARM64: %v", err)
} }
gasmObj, err := img.GOObjectAARCH64("a64dlink", "main_arm64.s") gasmObj, err := img.GOObjectAARCH64("main", "main_arm64.s")
if err != nil { if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err) t.Fatalf("GOObjectAARCH64: %v", err)
} }
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil { substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"),
t.Fatalf("write gasm object: %v", err) gasmObj, "GOARCH=arm64")
} binData, err := os.ReadFile(filepath.Join(dir, "prog2"))
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
binData, err := os.ReadFile(filepath.Join(dir, "prog"))
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
+39 -15
View File
@@ -9,7 +9,7 @@ import (
"strconv" "strconv"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
) )
// assembleARM64 assembles an AArch64 (arm64) TEXT function body into machine // assembleARM64 assembles an AArch64 (arm64) TEXT function body into machine
@@ -637,6 +637,20 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
return a64wordLE(a64UncondBranch(opc, uint32(rn), 0)), nil return a64wordLE(a64UncondBranch(opc, uint32(rn), 0)), nil
} }
// The bare spelling BL R9 is the same indirect branch: the parser reads
// a bare identifier as a symbol, and one named for a register is an
// indirect branch through it, which the toolchain accepts alongside the
// parenthesised form (BL (R3) and BL R3 both encode BLR R3).
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" && op.Addr.Index == "" {
if rn := arm64RegNum(op.Addr.Sym.Name); rn >= 0 {
opc := uint32(0) // BR
if link {
opc = 1 // BLR
}
return a64wordLE(a64UncondBranch(opc, uint32(rn), 0)), nil
}
}
// Symbol reference: BL sym(SB), or B sym(SB) for a tail call, against a // Symbol reference: BL sym(SB), or B sym(SB) for a tail call, against a
// relocation (R_CALLARM64 either way). // relocation (R_CALLARM64 either way).
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" { if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
@@ -1454,6 +1468,15 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo,
} }
return encodeARM64SBAddr(src.Imm.Sym, rd, relocs), nil return encodeARM64SBAddr(src.Imm.Sym, rd, relocs), nil
} }
// Immediate → memory: only storing zero is encodable (the ZR
// register); the toolchain rejects any other immediate-to-memory
// combination ("illegal combination").
if isMemOperand(dst) {
if arm64Imm64(src) != 0 {
return nil, fmt.Errorf("%s: illegal combination: an immediate store must be zero", mnem)
}
return encodeARM64MemOp(mnem, dst, 31, false, fi, "")
}
rd := arm64RegNum(operandRegName(dst)) rd := arm64RegNum(operandRegName(dst))
if rd < 0 { if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem) return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
@@ -3046,29 +3069,29 @@ func encodeARM64MoveWide(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte
// base, so MOVZ and MOVN come along for free. // base, so MOVZ and MOVN come along for free.
opc := baseOp >> 29 & 3 opc := baseOp >> 29 & 3
sf := baseOp >> 31 & 1 sf := baseOp >> 31 & 1
v := arm64Imm64(ops[0]) // The toolchain's optab case 33, shared by the whole family in both
if v < 0 { // widths: the immediate is one unsigned 64-bit pattern (a high-lane
return nil, fmt.Errorf("%s: negative immediate %d", mnem, v) // constant such as $(40000<<48) arrives negative through int64
// folding), it must occupy exactly one 16-bit lane, zero is rejected,
// and the W forms cannot reach the top half.
u := uint64(arm64Imm64(ops[0]))
if u == 0 {
return nil, fmt.Errorf("%s: zero immediate cannot be handled", mnem)
} }
hw := -1 hw := -1
for i := range 4 { for lane := range 4 {
if v>>(uint(i)*16)&0xFFFF != 0 { if u&^(uint64(0xFFFF)<<(lane*16)) == 0 {
hw = i hw = lane
break break
} }
} }
if hw < 0 { if hw < 0 {
hw = 0 // zero: every chunk is zero, hw = 0 carries it return nil, fmt.Errorf("%s: immediate %#x does not fit one 16-bit chunk", mnem, u)
}
for i := hw + 1; i < 4; i++ {
if v>>(uint(i)*16)&0xFFFF != 0 {
return nil, fmt.Errorf("%s: immediate %d does not fit one 16-bit chunk", mnem, v)
}
} }
if sf == 0 && hw > 1 { if sf == 0 && hw > 1 {
return nil, fmt.Errorf("%s: immediate %d out of range for the 32-bit form", mnem, v) return nil, fmt.Errorf("%s: immediate %#x out of range for the 32-bit form", mnem, u)
} }
return a64wordLE(a64MoveWide(sf, opc, uint32(hw), uint32(v>>uint(hw*16)&0xFFFF), uint32(rd))), nil return a64wordLE(a64MoveWide(sf, opc, uint32(hw), uint32(u>>uint(hw*16)&0xFFFF), uint32(rd))), nil
} }
// ---- Bitfield/EXTR encoding ---- // ---- Bitfield/EXTR encoding ----
@@ -3980,6 +4003,7 @@ func AssembleFileARM64(f *ast.File) (*Image, error) {
Size: d.size, Size: d.size,
Static: d.static, Static: d.static,
Rodata: d.rodata, Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok, Dupok: d.dupok,
}) })
} }
+1 -1
View File
@@ -33,7 +33,7 @@ import (
"strconv" "strconv"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
) )
// arm64RegNum returns the 5-bit register number for an AArch64 register name: // arm64RegNum returns the 5-bit register number for an AArch64 register name:
+37 -2
View File
@@ -7,8 +7,8 @@ import (
"strings" "strings"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
func TestArm64LDRSTREncoding(t *testing.T) { func TestArm64LDRSTREncoding(t *testing.T) {
@@ -1051,6 +1051,41 @@ func TestArm64MOVK(t *testing.T) {
} }
} }
// TestArm64MOVKHighLane pins the shifted high-lane immediate the arm64 test
// kernels write: $(40000<<48) folds to a negative int64, and the toolchain
// reads the value as an unsigned 64-bit pattern when it picks the lane.
func TestArm64MOVKHighLane(t *testing.T) {
got := arm64Words(t, "\tMOVK $(40000<<48), R0\n\tMOVK $0x9c40000000000000, R1\n")
want := []uint32{
0xf2f38800, // MOVK $(40000<<48), R0 (go tool asm: f2f38800)
0xf2f38801, // MOVK hw=3
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64MoveWideZeroImmediate pins the toolchain's rejection of a zero
// immediate in the move-wide family (optab case 33: "zero shifts cannot be
// handled"): every lane is zero, so no hw field can carry it.
func TestArm64MoveWideZeroImmediate(t *testing.T) {
for _, mnem := range []string{"MOVK", "MOVZ", "MOVN"} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\t"+mnem+" $0, R0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", mnem, errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s $0: expected error, got nil", mnem)
}
}
}
// TestArm64LoadImm64 tests 64-bit immediate loading. // TestArm64LoadImm64 tests 64-bit immediate loading.
func TestArm64LoadImm64(t *testing.T) { func TestArm64LoadImm64(t *testing.T) {
src := `#include "textflag.h" src := `#include "textflag.h"
+1 -1
View File
@@ -53,7 +53,7 @@ package asm
import ( import (
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
) )
// arm64FrameInfo holds the frame layout derived from a TEXT directive. // arm64FrameInfo holds the frame layout derived from a TEXT directive.
+1 -1
View File
@@ -7,7 +7,7 @@ import (
"encoding/binary" "encoding/binary"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// parseArm64File is a helper assembling one arm64 source file. // parseArm64File is a helper assembling one arm64 source file.
+583 -37
View File
@@ -8,7 +8,7 @@ import (
"strconv" "strconv"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
) )
// Assemble encodes the body of a TEXT function into x86-64 machine code, // Assemble encodes the body of a TEXT function into x86-64 machine code,
@@ -38,10 +38,25 @@ func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
// rejects SB operands outright (single-function assembly cannot resolve // rejects SB operands outright (single-function assembly cannot resolve
// them). When allowExternal is set, a reference to a symbol no GLOBL in the // them). When allowExternal is set, a reference to a symbol no GLOBL in the
// file defines is recorded as an external relocation instead of failing // file defines is recorded as an external relocation instead of failing
// the object-file emitters resolve it at link time. // the object-file emitters resolve it at link time. goos selects the TLS
// access form: the empty default behaves as linux.
type linkInfo struct { type linkInfo struct {
symbols map[string]bool symbols map[string]bool
allowExternal bool allowExternal bool
goos string
}
// tlsOneInsn reports the one-instruction TLS form, obj6.go's
// CanUse1InsnTLS for the GOOS gasm supports: the bare TLS load nops out and
// the (TLS*1) index folds to a segment-absolute access. Windows and plan9
// keep the two-instruction form; shared linux does too, which gasm's raw
// path does not model and therefore does not select.
func (l *linkInfo) tlsOneInsn() bool {
switch l.goos {
case "", "linux", "freebsd":
return true
}
return false
} }
// sbPatch is a function-relative static-symbol relocation: the disp32 field // sbPatch is a function-relative static-symbol relocation: the disp32 field
@@ -86,6 +101,10 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
// outgrows the short form. // outgrows the short form.
long := make([]bool, len(t.Body)) long := make([]bool, len(t.Body))
sizes := make([]int, len(t.Body)) sizes := make([]int, len(t.Body))
numTargets := make([]int, len(t.Body))
for i := range numTargets {
numTargets[i] = -1
}
offsets := map[string]int{} offsets := map[string]int{}
pcs := make([]int, len(t.Body)) pcs := make([]int, len(t.Body))
var guardJBlong, guardJBElong, moreJMPlong bool var guardJBlong, guardJBElong, moreJMPlong bool
@@ -94,23 +113,106 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
for { for {
guard := fi.guardLen(guardJBlong, guardJBElong) guard := fi.guardLen(guardJBlong, guardJBElong)
pos := guard + len(fi.prologue) pos := guard + len(fi.prologue)
for i := range numTargets {
numTargets[i] = -1
}
idxAtPc := map[int]int{}
for i, stmt := range t.Body { for i, stmt := range t.Body {
switch s := stmt.(type) { switch s := stmt.(type) {
case *ast.Label: case *ast.Label:
offsets[s.Name.Text] = pos offsets[s.Name.Text] = pos
case *ast.Instr: case *ast.Instr:
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
// The alignment pseudo-statement: its size is the
// padding to the next boundary at this very position,
// filled with NOPs at emission.
pad, err := pcAlignPad(pcAlignValue(s), pos)
if err != nil {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("PCALIGN: %w", err)
}
sizes[i] = pad
pcs[i] = pos
pos += pad
continue
}
sz, err := instrSize(s, fi, long[i], link) sz, err := instrSize(s, fi, long[i], link)
if err != nil { if err != nil {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err) return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
} }
sizes[i] = sz sizes[i] = sz
pcs[i] = pos pcs[i] = pos
idxAtPc[pos] = i
pos += sz pos += sz
} }
} }
bodyLen := pos - (guard + len(fi.prologue)) bodyLen := pos - (guard + len(fi.prologue))
// Expand any short jump whose displacement no longer fits rel8. // Expand any short jump whose displacement no longer fits rel8.
changed := false changed := false
// Numeric ±N(PC) jumps resolve against this iteration's layout; the
// emission pass reads the same table after the loop converges. A
// target that is itself an unconditional local JMP is chased to the
// ultimate target: the toolchain's brloop pass collapses branch-to-
// branch chains before it encodes, so matching its bytes requires
// the same redirection.
for i := range numTargets {
numTargets[i] = -1
}
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
if !ok {
continue
}
if len(s.Operands) == 1 {
if n, isNum := pcJumpOffset(s.Operands[0]); isNum {
if target, okT := pcJumpTarget(t, i, n, pcs); okT {
numTargets[i] = target
}
}
}
}
for i := range numTargets {
if numTargets[i] < 0 {
continue
}
tgt := numTargets[i]
for hop := 0; hop < len(t.Body); hop++ {
idx, ok := idxAtPc[tgt]
if !ok {
break
}
in, ok := t.Body[idx].(*ast.Instr)
if !ok || strings.ToUpper(in.Mnemonic.Text) != "JMP" || len(in.Operands) != 1 {
break
}
if name, isLabel := labelName(in.Operands[0]); isLabel {
tgt = offsets[resolve(name)]
continue
}
if n, isNum := pcJumpOffset(in.Operands[0]); isNum {
next, okT := pcJumpTarget(t, idx, n, pcs)
if !okT {
break
}
tgt = next
continue
}
break // JMP through a register or memory: the chain ends
}
numTargets[i] = tgt
}
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
if !ok {
continue
}
if numTargets[i] >= 0 && !long[i] {
rel := int64(numTargets[i] - (pcs[i] + jumpSize(strings.ToUpper(s.Mnemonic.Text), false)))
if !fits8(rel) {
long[i] = true
changed = true
}
}
}
for i, stmt := range t.Body { for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr) s, ok := stmt.(*ast.Instr)
if !ok { if !ok {
@@ -232,7 +334,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
spadjStep{pos + epi, 0}, spadjStep{pos + epi, 0},
) )
} }
code, ps, pool, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link) code, ps, pool, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link, numTargets[i])
if err != nil { if err != nil {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err) return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
} }
@@ -428,6 +530,99 @@ func computeFrame(t *ast.Text) frameInfo {
return fi return fi
} }
// pcJumpOffset recognises the numeric relative jump operand ±N(PC) and
// returns N: the toolchain counts instructions, not bytes, so +2(PC) targets
// the second instruction boundary after the branch.
func pcJumpOffset(op *ast.Operand) (int, bool) {
if op.Kind != ast.OpAddr || op.Addr.Base != "PC" {
return 0, false
}
return int(op.Addr.Offset), true
}
// pcJumpTarget resolves a numeric jump at statement index j: N counts the
// instruction statements after the jump itself (N = 0 is the jump's own
// address, the classic park loop), and the target is the start of the Nth
// one. A negative N counts the same way backwards, before the jump: the
// exit loops write JMP -3(PC) to land three instructions earlier. Labels
// count not, in either direction. It reports false when the count runs
// past the end of the function, or before its first instruction.
func pcJumpTarget(t *ast.Text, j, n int, pcs []int) (int, bool) {
if n == 0 {
return pcs[j], true
}
if n < 0 {
seen := 0
for k := j - 1; k >= 0; k-- {
if _, ok := t.Body[k].(*ast.Instr); !ok {
continue
}
seen--
if seen == n {
return pcs[k], true
}
}
return 0, false
}
seen := 0
for k := j + 1; k < len(t.Body); k++ {
if _, ok := t.Body[k].(*ast.Instr); !ok {
continue
}
seen++
if seen == n {
return pcs[k], true
}
}
return 0, false
}
// x86 NOP encodings, single-instruction no-ops of lengths 1 to 9 (the
// toolchain's asm6.go nop table); longer padding repeats the largest that
// fits, greedy from the end.
var x86Nops = [][]byte{
{0x90},
{0x66, 0x90},
{0x0F, 0x1F, 0x00},
{0x0F, 0x1F, 0x40, 0x00},
{0x0F, 0x1F, 0x44, 0x00, 0x00},
{0x66, 0x0F, 0x1F, 0x44, 0x00, 0x00},
{0x0F, 0x1F, 0x80, 0x00, 0x00, 0x00, 0x00},
{0x0F, 0x1F, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00},
{0x66, 0x0F, 0x1F, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00},
}
// fillNOPs fills p with the greedy largest single-instruction NOPs, exactly
// the toolchain's fillnop.
func fillNOPs(p []byte) {
for len(p) > 0 {
m := min(len(p), len(x86Nops))
copy(p[:m], x86Nops[m-1])
p = p[m:]
}
}
// pcAlignPad computes the padding PCALIGN $align inserts at pos: the
// alignment must be a power of two in [8, 2048] and the padding runs to the
// next boundary (zero when the position is already aligned).
func pcAlignPad(align, pos int) (int, error) {
if align <= 0 || align&(align-1) != 0 || align < 8 || align > 2048 {
return 0, fmt.Errorf("alignment value of an instruction must be a power of two and in the range [8, 2048], got %d", align)
}
if lob := pos & (align - 1); lob != 0 {
return align - lob, nil
}
return 0, nil
}
// pcAlignValue reads a PCALIGN statement's alignment operand.
func pcAlignValue(s *ast.Instr) int {
if len(s.Operands) == 1 && s.Operands[0].Kind == ast.OpImmediate && s.Operands[0].Imm.HasVal {
return int(s.Operands[0].Imm.Val)
}
return 0 // rejected by pcAlignPad's range check
}
// hasCall reports whether the function body contains a CALL instruction. // hasCall reports whether the function body contains a CALL instruction.
func hasCall(t *ast.Text) bool { func hasCall(t *ast.Text) bool {
for _, stmt := range t.Body { for _, stmt := range t.Body {
@@ -602,7 +797,7 @@ func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, erro
} }
return jumpSize(mnem, long), nil return jumpSize(mnem, long), nil
} }
code, _, _, err := encodeInstr(s, 0, nil, fi, false, nil, link) code, _, _, err := encodeInstr(s, 0, nil, fi, false, nil, link, -1)
if err != nil { if err != nil {
return 0, err return 0, err
} }
@@ -613,17 +808,43 @@ func isJumpMnemonic(mnem string) bool {
if mnem == "JMP" || mnem == "CALL" { if mnem == "JMP" || mnem == "CALL" {
return true return true
} }
if isLoopMnemonic(mnem) {
return true
}
_, ok := condCode(mnem) _, ok := condCode(mnem)
return ok return ok
} }
// isLoopMnemonic reports the LOOP family, rel8 alone (E0-E2).
func isLoopMnemonic(mnem string) bool {
switch mnem {
case "LOOP", "LOOPE", "LOOPNE":
return true
}
return false
}
// loopOpcode maps the LOOP family to its E0-E2 opcode.
func loopOpcode(mnem string) byte {
switch mnem {
case "LOOPE":
return 0xE1
case "LOOPNE":
return 0xE0
}
return 0xE2
}
// jumpSize returns the length of a jump instruction in the requested form: // jumpSize returns the length of a jump instruction in the requested form:
// short (rel8) where available, otherwise the rel32 form. CALL is always // short (rel8) where available, otherwise the rel32 form. CALL is always
// rel32. // rel32; the LOOP family is rel8 alone.
func jumpSize(mnem string, long bool) int { func jumpSize(mnem string, long bool) int {
if mnem == "CALL" { if mnem == "CALL" {
return 5 // opcode + rel32 return 5 // opcode + rel32
} }
if isLoopMnemonic(mnem) {
return 2 // opcode + rel8, the only form
}
if !long { if !long {
return 2 // opcode + rel8 return 2 // opcode + rel8
} }
@@ -637,9 +858,21 @@ func jumpSize(mnem string, long bool) int {
// (relative to pc, the instruction's own offset). A RET in a frame-pointer // (relative to pc, the instruction's own offset). A RET in a frame-pointer
// function is prefixed with the epilogue. resolve, when non-nil, redirects a // function is prefixed with the epilogue. resolve, when non-nil, redirects a
// jump label through the jump-to-jump chain before the offset lookup. // jump label through the jump-to-jump chain before the offset lookup.
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) { func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo, numTarget int) ([]byte, []sbPatch, []floatPoolEntry, error) {
mnem := strings.ToUpper(s.Mnemonic.Text) mnem := strings.ToUpper(s.Mnemonic.Text)
if mnem == "PCALIGN" {
// The layout pass already accounted the padding; emit the same
// amount of NOP bytes for the statement's own position.
pad, err := pcAlignPad(pcAlignValue(s), pc)
if err != nil {
return nil, nil, nil, err
}
out := make([]byte, pad)
fillNOPs(out)
return out, nil, nil, nil
}
var prefix []byte var prefix []byte
if mnem == "RET" && fi.useFP { if mnem == "RET" && fi.useFP {
prefix = fi.epilogue prefix = fi.epilogue
@@ -677,7 +910,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
} }
return append(prefix, code...), nil, nil, nil return append(prefix, code...), nil, nil, nil
} }
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve) code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve, numTarget)
} else { } else {
code, ps, pool, err = encodeNormal(s, fi, link) code, ps, pool, err = encodeNormal(s, fi, link)
} }
@@ -703,6 +936,51 @@ func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch
} }
return code, nil, nil, nil return code, nil, nil, nil
} }
// MOVQ $sym±off(SB), r64: the toolchain assembles a symbol immediate as
// LEAQ disp32(RIP), r64 with an R_PCREL relocation at the disp32 field,
// never as a 64-bit absolute immediate (verified against go tool asm).
// MOVD is the MOVQ alias; the narrower widths reject the form outright.
if (mnemUpper == "MOVQ" || mnemUpper == "MOVD") && len(s.Operands) == 2 &&
s.Operands[0].Kind == ast.OpImmediate && s.Operands[0].Imm.Sym != nil &&
s.Operands[0].Imm.Sym.Pseudo == "SB" {
mem := &ast.Operand{Kind: ast.OpAddr, Addr: ast.Address{Sym: s.Operands[0].Imm.Sym}}
src, err := operandFromAST(mnemUpper, mem, 8, fi, link)
if err != nil {
return nil, nil, nil, err
}
dst, err := operandFromAST(mnemUpper, s.Operands[1], 8, fi, link)
if err != nil {
return nil, nil, nil, err
}
e := &enc{}
if err := e.encodeLea([]Operand{src, dst}, 8); err != nil {
return nil, nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
}
return e.out, ps, nil, nil
}
// MOVQ/MOVL TLS, r: the bare TLS load. The toolchain's progedit nops
// it out on the one-instruction TLS systems (linux and freebsd, not
// shared) and encodes the segment-prefixed load elsewhere; get_tls(r),
// the macro GOROOT's go_tls.h defines, expands to exactly this
// statement, and the toolchain's pairing pass removes it whenever the
// following instruction's (TLS*1) index folds.
if (mnemUpper == "MOVQ" || mnemUpper == "MOVL") && len(s.Operands) == 2 && isBareTLS(s.Operands[0]) {
return encodeTLSBaseLoad(s, fi, link)
}
// The old paired-register shift spelling, SHLL CX, R11:AX (a colon
// between the two registers), is the toolchain's SHLD family: SHLDL CL,
// AX, R11 with the count register first, the paired source in the reg
// field and the pair's head in r/m.
if code, ps, err := encodeColonShift(s, mnemUpper, fi, link); code != nil || err != nil {
if err != nil {
return nil, nil, nil, err
}
return code, ps, nil, nil
}
_, size := splitSize(mnemUpper) _, size := splitSize(mnemUpper)
if size == 0 { if size == 0 {
size = 8 size = 8
@@ -722,10 +1000,67 @@ func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch
ps := make([]sbPatch, len(e.patches)) ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches { for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend} ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
if p.tls {
ps[i].kind = RelTLSLE
}
} }
return e.out, ps, e.floatPoolList(), nil return e.out, ps, e.floatPoolList(), nil
} }
// isBareTLS reports whether the operand is the bare TLS pseudo-register
// load source, the expansion of go_tls.h's get_tls(r) macro.
func isBareTLS(op *ast.Operand) bool {
return op.Kind == ast.OpAddr && op.Addr.Sym != nil &&
op.Addr.Sym.Pseudo == "" && op.Addr.Sym.Name == "TLS" &&
op.Addr.Base == "" && op.Addr.Index == ""
}
// encodeTLSBaseLoad assembles MOVQ/MOVL TLS, r. On the one-instruction TLS
// systems (linux and freebsd outside -shared, obj6.go's CanUse1InsnTLS) the
// statement nops out: the following (TLS*1) access folds to a direct
// segment-absolute load. The two-instruction systems keep the segment load,
// nine bytes with the R_TLSLE patch site at the disp32.
func encodeTLSBaseLoad(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
if size == 0 {
size = 8
}
dst, err := operandFromAST("MOVQ", s.Operands[1], 8, fi, link)
if err != nil {
return nil, nil, nil, err
}
reg, ok := dst.(Reg)
if !ok || reg.isVec() {
return nil, nil, nil, fmt.Errorf("TLS: destination must be a general register")
}
if link == nil || link.tlsOneInsn() {
return nil, nil, nil, nil // noped out
}
seg := byte(0x64) // FS
if link.goos == "windows" {
seg = 0x65 // GS
}
e := &enc{}
i := &instr{
prefix: seg,
rexW: size == 8,
rexR: reg.idx >= 8,
opcode: []byte{0x8B},
modrm: 0x04 | (reg.idx&7)<<3,
sib: 0x25,
disp: le32(0),
tls: true,
}
if err := e.emit(i); err != nil {
return nil, nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend, kind: RelTLSLE}
}
return e.out, ps, nil, nil
}
// encodeBookkeeping accepts-and-ignores FUNCDATA and PCDATA at the statement // encodeBookkeeping accepts-and-ignores FUNCDATA and PCDATA at the statement
// level, before operand conversion: the toolchain's shapes are FUNCDATA // level, before operand conversion: the toolchain's shapes are FUNCDATA
// $n, sym(SB) and PCDATA $n, $m, and neither contributes a byte to the // $n, sym(SB) and PCDATA $n, $m, and neither contributes a byte to the
@@ -752,22 +1087,90 @@ func encodeBookkeeping(upper string, s *ast.Instr) ([]byte, error) {
return nil, nil return nil, nil
} }
// encodeColonShift encodes the paired-register shift spellings, SHLx CX,
// dst:src: the toolchain reads them as the SHLD family (double-precision
// shift by CL), reg = the paired source, r/m = the pair's head. The second
// operand's raw text carries the colon; ok reports the spelling was found.
func encodeColonShift(s *ast.Instr, mnemUpper string, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) {
base, _ := strings.CutPrefix(mnemUpper, "SHL")
if base == mnemUpper || len(s.Operands) != 2 {
return nil, nil, nil
}
_, size := splitSize(mnemUpper)
raw := strings.ReplaceAll(s.Operands[1].Raw, " ", "")
head, tail, ok := strings.Cut(raw, ":")
if !ok || head == "" || tail == "" {
return nil, nil, nil
}
headReg, ok1 := ParseReg(head)
srcReg, ok2 := ParseReg(tail)
if !ok1 || !ok2 {
return nil, nil, fmt.Errorf("%s: invalid paired register %q", mnemUpper, s.Operands[1].Raw)
}
cnt, err := operandFromAST(mnemUpper, s.Operands[0], size, fi, link)
if err != nil {
return nil, nil, err
}
cntReg, ok := cnt.(Reg)
if !ok || cntReg.idx != 1 {
return nil, nil, fmt.Errorf("%s: the paired-register form counts in CL", mnemUpper)
}
// SHLD r/m, reg, CL: 0F A5 (REX.W for the 64-bit width).
e := &enc{}
i := &instr{rexW: size == 8, opcode: []byte{0x0F, 0xA5}, modrm: -1, sib: -1}
if err := setRM(i, srcReg, headReg, size); err != nil {
return nil, nil, err
}
if err := e.emit(i); err != nil {
return nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for j, p := range e.patches {
ps[j] = sbPatch{off: p.off, name: p.name, addend: p.addend}
}
return e.out, ps, nil
}
// trailingIndexGroup recovers a trailing "(index*scale)" or "(index)" group
// from an operand's raw text: the symbol-pseudo parse returns before the
// index group, so foo(SP)(AX*1) keeps its index only in the spelling.
func trailingIndexGroup(raw string) (string, int, bool) {
compact := strings.ReplaceAll(raw, " ", "")
if !strings.HasSuffix(compact, ")") {
return "", 0, false
}
open := strings.LastIndex(compact, "(")
if open < 2 || !strings.Contains(compact[:open], ")") {
return "", 0, false // one group alone: no trailing index
}
name, scale, _, ok := cutParenGroup(compact[open:])
return name, scale, ok
}
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the // encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
// target label, in the short (rel8) or long (rel32) form. // target label or from a numeric ±N(PC) instruction count, in the short
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long bool, resolve func(string) string) ([]byte, error) { // (rel8) or long (rel32) form. numTarget is the resolved byte offset of a
// numeric operand, negative when the operand is not one.
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long bool, resolve func(string) string, numTarget int) ([]byte, error) {
if len(s.Operands) != 1 { if len(s.Operands) != 1 {
return nil, fmt.Errorf("jump expects 1 operand, got %d", len(s.Operands)) return nil, fmt.Errorf("jump expects 1 operand, got %d", len(s.Operands))
} }
name, ok := labelName(s.Operands[0]) name, isLabel := labelName(s.Operands[0])
if !ok { if !isLabel && numTarget < 0 {
return nil, fmt.Errorf("jump target must be a local label") return nil, fmt.Errorf("jump target must be a local label")
} }
if resolve != nil && mnem != "CALL" { var target int
name = resolve(name) if isLabel {
} if resolve != nil && mnem != "CALL" {
target, ok := offsets[name] name = resolve(name)
if !ok { }
return nil, fmt.Errorf("undefined label %q", name) t, ok := offsets[name]
if !ok {
return nil, fmt.Errorf("undefined label %q", name)
}
target = t
} else {
target = numTarget
} }
rel := int64(target - (pc + jumpSize(mnem, long))) rel := int64(target - (pc + jumpSize(mnem, long)))
@@ -778,9 +1181,15 @@ func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long
if mnem == "JMP" { if mnem == "JMP" {
return []byte{0xEB, byte(int8(rel))}, nil return []byte{0xEB, byte(int8(rel))}, nil
} }
if isLoopMnemonic(mnem) {
return []byte{loopOpcode(mnem), byte(int8(rel))}, nil
}
cc, _ := condCode(mnem) cc, _ := condCode(mnem)
return []byte{0x70 + byte(cc), byte(int8(rel))}, nil return []byte{0x70 + byte(cc), byte(int8(rel))}, nil
} }
if isLoopMnemonic(mnem) {
return nil, fmt.Errorf("%s has no long form", mnem)
}
switch mnem { switch mnem {
case "JMP": case "JMP":
return append([]byte{0xE9}, le32(rel)...), nil return append([]byte{0xE9}, le32(rel)...), nil
@@ -832,6 +1241,72 @@ func labelName(op *ast.Operand) (string, bool) {
return "", false return "", false
} }
// jumpOperand returns the branch-target operand of a JMP/CALL, rewriting the
// `*`-prefixed indirect spellings (JMP *(R12), JMP *4(SP)) into their plain
// memory form. The star marks an indirect target and changes no bytes; the
// address parser leaves the operand's address empty because of the leading
// star, so the fields are rebuilt from the raw text onto a copy of the
// operand, never on the shared syntax tree.
func jumpOperand(s *ast.Instr) *ast.Operand {
if len(s.Operands) != 1 {
return nil
}
op := s.Operands[0]
compact := strings.ReplaceAll(op.Raw, " ", "")
inner, ok := strings.CutPrefix(compact, "*")
if !ok {
return op
}
var addr ast.Address
if i := strings.IndexByte(inner, '('); i > 0 {
v, err := strconv.ParseInt(inner[:i], 0, 64)
if err != nil {
return op
}
addr.Offset, addr.HasOff = v, true
inner = inner[i:]
}
base, _, rest, ok := cutParenGroup(inner)
if !ok {
return op
}
if base != "" {
addr.Base = base
}
if rest != "" {
idx, scale, _, ok := cutParenGroup(rest)
if ok && idx != "" {
addr.Index = idx
addr.Scale = scale
}
}
c := *op
c.Addr = addr
return &c
}
// cutParenGroup splits a leading "(name)" or "(name*n)" off s, returning the
// inner text, the scale it names (1 when the group spells no multiplier) and
// the remainder.
func cutParenGroup(s string) (name string, scale int, rest string, ok bool) {
if !strings.HasPrefix(s, "(") {
return "", 0, "", false
}
i := strings.IndexByte(s, ')')
if i < 0 {
return "", 0, "", false
}
inner, rest := s[1:i], s[i+1:]
if before, after, ok := strings.Cut(inner, "*"); ok {
n, err := strconv.Atoi(after)
if err != nil {
return "", 0, "", false
}
return before, n, rest, true
}
return inner, 1, rest, true
}
// indirectJumpTarget reports whether the JMP/CALL operand addresses a // indirectJumpTarget reports whether the JMP/CALL operand addresses a
// register or a memory location rather than a label or a static symbol. // register or a memory location rather than a label or a static symbol.
// A bare identifier is a register when the register table knows the name and // A bare identifier is a register when the register table knows the name and
@@ -840,7 +1315,19 @@ func indirectJumpTarget(s *ast.Instr) bool {
if len(s.Operands) != 1 || s.Operands[0].Kind != ast.OpAddr { if len(s.Operands) != 1 || s.Operands[0].Kind != ast.OpAddr {
return false return false
} }
a := s.Operands[0].Addr op := jumpOperand(s)
if op == nil {
return false
}
if op != s.Operands[0] {
return true // the star marker spells an indirect target
}
a := op.Addr
// ±N(PC) is the numeric relative form, the PC counts instructions from
// the branch: relative, not indirect.
if a.Base == "PC" || a.Index == "PC" {
return false
}
if a.Base != "" || a.Index != "" { if a.Base != "" || a.Index != "" {
return true return true
} }
@@ -853,18 +1340,20 @@ func indirectJumpTarget(s *ast.Instr) bool {
} }
// encodeIndirectJump assembles a JMP/CALL through a register or memory // encodeIndirectJump assembles a JMP/CALL through a register or memory
// operand, which carries no relocation and no label to resolve. // operand, which carries no relocation and no label to resolve. The
// `*`-prefixed spellings go through jumpOperand first, their star rebuilt
// into a plain memory operand.
func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) { func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
ops := make([]Operand, len(s.Operands)) op := s.Operands[0]
for i, op := range s.Operands { if cleaned := jumpOperand(s); cleaned != nil {
o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil) op = cleaned
if err != nil { }
return nil, err o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil)
} if err != nil {
ops[i] = o return nil, err
} }
e := &enc{} e := &enc{}
if err := e.encodeIndirectBranch(mnem, ops); err != nil { if err := e.encodeIndirectBranch(mnem, []Operand{o}); err != nil {
return nil, err return nil, err
} }
return e.out, nil return e.out, nil
@@ -933,37 +1422,89 @@ func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, l
off := a.Sym.Offset + fi.fpAdjust off := a.Sym.Offset + fi.fpAdjust
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
} }
// SP-relative local: x-N(SP) → (spAdjust + offset)(SP). // SP-relative local: x-N(SP) → (spAdjust + offset)(SP), keeping a scaled
// index beside the virtual stack pointer (foo(SP)(AX*1)). The
// symbol-pseudo parse returns before the index group, so the index
// is recovered from the raw text when the address lacks it.
if a.Sym != nil && a.Sym.Pseudo == "SP" && a.Base == "" { if a.Sym != nil && a.Sym.Pseudo == "SP" && a.Base == "" {
off := fi.spAdjust + a.Sym.Offset off := fi.spAdjust + a.Sym.Offset
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil m := Mem{Base: spReg, Disp: off, HasBase: true, Size: size}
if name, scale, ok := trailingIndexGroup(op.Raw); ok {
idx, ok := ParseReg(name)
if !ok {
return nil, fmt.Errorf("unknown index register %q", name)
}
m.Index = idx
m.Scale = scale
m.HasIndex = true
}
return m, nil
} }
// SB (global symbol): a symbol defined in the same file (GLOBL) is // SB (global symbol): a symbol defined in the same file (GLOBL) is
// encoded RIP-relative and resolved by the file-level layout; // encoded RIP-relative and resolved by the file-level layout;
// anything not defined here needs object-file emission. // anything not defined here needs object-file emission. A static
// (file-local) spelling of an undefined symbol defers the same way
// the toolchain does: the relocation names it and the linker decides.
if a.Sym != nil && a.Sym.Pseudo == "SB" { if a.Sym != nil && a.Sym.Pseudo == "SB" {
if link == nil || link.symbols == nil { if link == nil || link.symbols == nil {
return nil, fmt.Errorf("symbol %q needs file-level assembly (AssembleFile)", a.Sym.Name) return nil, fmt.Errorf("symbol %q needs file-level assembly (AssembleFile)", a.Sym.Name)
} }
if !link.symbols[a.Sym.Name] { if !link.symbols[a.Sym.Name] && !link.allowExternal {
if a.Sym.Static { return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name)
return nil, fmt.Errorf("undefined symbol %q", a.Sym.Name)
}
if !link.allowExternal {
return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name)
}
} }
return sbMem{size: size, name: a.Sym.Name, addend: a.Sym.Offset}, nil return sbMem{size: size, name: a.Sym.Name, addend: a.Sym.Offset}, nil
} }
// Memory with a real base register: (base), off(base), (base)(index*scale). // Memory with a real base register: (base), off(base), (base)(index*scale).
if a.Base != "" { if a.Base != "" {
// The TLS pseudo-base, off(TLS): the segment-prefixed absolute
// the thread-local access lowers to, 64 8B 04 25 with its
// R_TLS_LE patch site on the disp32.
if a.Base == "TLS" {
seg := byte(0x64) // FS on linux, freebsd, plan9
if link != nil && link.goos == "windows" {
seg = 0x65 // GS
}
return TLSMem{Disp: a.Offset, Size: size, Seg: seg}, nil
}
// Segment-absolute: 0x30(GS) and 0x28(FS), the windows TLS
// spellings. The segment override prefixes a disp32 absolute
// reference with no relocation.
if a.Base == "GS" || a.Base == "FS" {
seg := byte(0x64)
if a.Base == "GS" {
seg = 0x65
}
return SegAbs{Disp: a.Offset, Size: size, Seg: seg}, nil
}
base, ok := ParseReg(a.Base) base, ok := ParseReg(a.Base)
if !ok { if !ok {
return nil, fmt.Errorf("unknown base register %q", a.Base) return nil, fmt.Errorf("unknown base register %q", a.Base)
} }
m := Mem{Base: base, Disp: a.Offset, HasBase: true, Size: size} m := Mem{Base: base, Disp: a.Offset, HasBase: true, Size: size}
if a.Index != "" { if a.Index != "" {
if a.Index == "TLS" {
// off(base)(TLS*1): the thread-local annotation. The
// one-instruction TLS form folds it to off(TLS), the
// segment-prefixed absolute whose disp32 carries an
// R_TLS_LE patch site; the base register disappears
// from the encoding, exactly as the toolchain's
// progedit rewrites the address.
seg := byte(0x64) // FS on linux, freebsd, plan9
if link != nil && link.goos == "windows" {
seg = 0x65 // GS
}
return TLSMem{Disp: a.Offset, Size: size, Seg: seg}, nil
}
if a.Index == "GS" || a.Index == "FS" {
// 0(CX)(GS): the segment annotation rides the base
// access as the override prefix.
m.Seg = 0x64
if a.Index == "GS" {
m.Seg = 0x65
}
return m, nil
}
idx, ok := ParseReg(a.Index) idx, ok := ParseReg(a.Index)
if !ok { if !ok {
return nil, fmt.Errorf("unknown index register %q", a.Index) return nil, fmt.Errorf("unknown index register %q", a.Index)
@@ -984,6 +1525,11 @@ func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, l
} }
return Mem{Index: idx, Scale: a.Scale, Disp: a.Offset, HasIndex: true, Size: size}, nil return Mem{Index: idx, Scale: a.Scale, Disp: a.Offset, HasIndex: true, Size: size}, nil
} }
// A bare displacement with no base: the absolute address form,
// MOVL $0xf1, 0xf1. No segment and no relocation.
if a.Sym == nil && a.Base == "" && a.Index == "" && a.HasOff {
return SegAbs{Disp: a.Offset, Size: size}, nil
}
// Bare register. // Bare register.
if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" { if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" {
if r, ok := ParseReg(a.Sym.Name); ok { if r, ok := ParseReg(a.Sym.Name); ok {
+54 -2
View File
@@ -10,8 +10,8 @@ import (
"golang.org/x/arch/x86/x86asm" "golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// firstText parses src and returns its first TEXT function. // firstText parses src and returns its first TEXT function.
@@ -370,6 +370,58 @@ end:
} }
} }
// TestAssembleNumericPCJumps pins the numeric ±N(PC) branch operands: N
// counts instruction statements, skipping labels, in both directions (the
// runtime's exit loops write JMP -3(PC)), N = 0 parks on the jump itself.
func TestAssembleNumericPCJumps(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·exit(SB), NOSPLIT, $0
MOVB $1, AL
lab:
MOVB $2, AL
MOVB $3, AL
JMP -3(PC)
MOVB $4, AL
park:
JMP 0(PC)
MOVB $5, AL
JMP 2(PC)
MOVB $6, AL
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
// From the Go-assembled function:
// MOVB $1, AL b001
// MOVB $2, AL b002
// MOVB $3, AL b003
// JMP -3(PC) ebf8 (three instructions back, past lab:)
// MOVB $4, AL b004
// JMP 0(PC) ebfe (the park loop)
// MOVB $5, AL b005
// JMP 2(PC) eb02 (over MOVB $6 to the RET)
// MOVB $6, AL b006
// RET c3
want := []byte{
0xb0, 0x01,
0xb0, 0x02,
0xb0, 0x03,
0xeb, 0xf8,
0xb0, 0x04,
0xeb, 0xfe,
0xb0, 0x05,
0xeb, 0x02,
0xb0, 0x06,
0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("numeric-PC mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
func TestAssemblePrefetch(t *testing.T) { func TestAssemblePrefetch(t *testing.T) {
fn := firstText(t, ` fn := firstText(t, `
#include "textflag.h" #include "textflag.h"
+1 -1
View File
@@ -8,7 +8,7 @@ import (
"encoding/binary" "encoding/binary"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// ulebIter reads ULEB128 values, the .debug_abbrev and line-header // ulebIter reads ULEB128 values, the .debug_abbrev and line-header
+2 -2
View File
@@ -12,8 +12,8 @@ import (
"path/filepath" "path/filepath"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// The object-file tests share one source: two exported functions, one // The object-file tests share one source: two exported functions, one
+1 -1
View File
@@ -9,7 +9,7 @@ import (
"encoding/binary" "encoding/binary"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64 // TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
+1 -1
View File
@@ -9,7 +9,7 @@ import (
"encoding/binary" "encoding/binary"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64 // TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
+1 -1
View File
@@ -9,7 +9,7 @@ import (
"encoding/binary" "encoding/binary"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// TestELFRISCVObjectDataRelocation checks that a symbol-valued DATA field // TestELFRISCVObjectDataRelocation checks that a symbol-valued DATA field
+13
View File
@@ -20,11 +20,24 @@ func Encodable(mnemonic string) bool {
switch upper { switch upper {
case "RET", "NOP", "CALL", "JMP", case "RET", "NOP", "CALL", "JMP",
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2", "POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
// The SSE compare family sharing CMPSD's predicate-last shape, the
// far return with its stack pop, the loop family, the bank-crossing
// MMX moves and the one-operand system controls.
"CMPSS", "CMPPS", "CMPPD", "RETFL",
"LOOP", "LOOPE", "LOOPNE",
"MOVDQ2Q", "MOVQ2DQ",
"ENDBR64", "CLWB", "TPAUSE", "UMONITOR", "UMWAIT", "RDPID", "CLDEMOTE",
// The literal-data pseudo-ops, the accepted-and-ignored END and // The literal-data pseudo-ops, the accepted-and-ignored END and
// bookkeeping statements, and the SP adjust. // bookkeeping statements, and the SP adjust.
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA": "BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA":
return true return true
} }
if _, ok := sysUnaryTable[upper]; ok {
return true
}
if _, ok := sseStoreOnly[upper]; ok {
return true
}
if _, ok := noOperandTable[upper]; ok { if _, ok := noOperandTable[upper]; ok {
return true return true
} }
+108 -3
View File
@@ -64,6 +64,7 @@ type encPatch struct {
off int off int
name string name string
addend int64 addend int64
tls bool // a TLS slot offset: the patch is R_TLSLE with no symbol
} }
func (e *enc) encode(mnem string, ops []Operand) error { func (e *enc) encode(mnem string, ops []Operand) error {
@@ -72,9 +73,23 @@ func (e *enc) encode(mnem string, ops []Operand) error {
// Fixed-name instructions (no size suffix). // Fixed-name instructions (no size suffix).
switch { switch {
case upper == "RET": case upper == "RET":
// RET sym(SB), the absolute return: the toolchain encodes it as a
// tail jump, E9 rel32 with a call relocation against the symbol.
if len(ops) == 1 {
if m, ok := ops[0].(sbMem); ok {
return e.emit(&instr{opcode: []byte{0xE9}, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}})
}
return fmt.Errorf("RET: unsupported operand")
}
if len(ops) != 0 {
return fmt.Errorf("RET expects no operands, got %d", len(ops))
}
return e.encodeRet() return e.encodeRet()
case upper == "NOP": case upper == "NOP":
return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1}) // The toolchain consumes every NOP statement as a pseudo and emits
// nothing for it, operands included (a bare NOP, NOP AX and
// NOP sym(SB) all vanish from the object).
return nil
case upper == "CALL" || upper == "JMP": case upper == "CALL" || upper == "JMP":
// Through a register or memory: FF /2 (CALL) or FF /4 (JMP). // Through a register or memory: FF /2 (CALL) or FF /4 (JMP).
// Anything else is a rel32 against a label resolved by the assembler. // Anything else is a rel32 against a label resolved by the assembler.
@@ -101,6 +116,15 @@ func (e *enc) encode(mnem string, ops []Operand) error {
} }
return e.emit(&instr{opcode: op, modrm: -1, sib: -1}) return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
} }
// One-operand system instructions whose reg field is a fixed digit:
// the cache and wait controls under 0F AE/0F 1C and the RDPID read.
if m, ok := sysUnaryTable[upper]; ok {
return e.encodeSysUnary(upper, m, ops)
}
// The store-only SSE moves (the non-temporal store).
if m, ok := sseStoreOnly[upper]; ok {
return e.encodeSSEStoreOnly(upper, m, ops)
}
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings // POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
// are rejected by go tool asm in 64-bit mode, so they stay unsupported. // are rejected by go tool asm in 64-bit mode, so they stay unsupported.
switch upper { switch upper {
@@ -116,14 +140,63 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1}) return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
case "INT": case "INT":
return e.encodeInt(ops) return e.encodeInt(ops)
// The LOOP family outside the assembler's label settlement: the operand
// is the already-computed rel8 (E0-E2).
case "LOOP", "LOOPE", "LOOPNE":
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 operand, got %d", upper, len(ops))
}
imm, ok := ops[0].(Imm)
if !ok || !fits8(int64(imm)) {
return fmt.Errorf("%s: relative offset must be a signed byte", upper)
}
return e.emit(&instr{opcode: []byte{loopOpcode(upper)}, modrm: -1, sib: -1, imm: []byte{byte(int8(imm))}})
case "LDMXCSR": case "LDMXCSR":
return e.encodeMxcsr(2, ops) return e.encodeMxcsr(2, ops)
case "STMXCSR": case "STMXCSR":
return e.encodeMxcsr(3, ops) return e.encodeMxcsr(3, ops)
// CMPSD is the scalar double compare, whose predicate immediate comes // CMPSD is the scalar double compare, whose predicate immediate comes
// LAST in Plan 9 order (src, dst, $imm). // LAST in Plan 9 order (src, dst, $imm); the family shares the shape.
case "CMPSD": case "CMPSD":
return e.encodeCmpsd(ops) return e.encodeSSECmp("CMPSD", 0xF2, ops)
case "CMPSS":
return e.encodeSSECmp("CMPSS", 0xF3, ops)
case "CMPPS":
return e.encodeSSECmp("CMPPS", 0x00, ops)
case "CMPPD":
return e.encodeSSECmp("CMPPD", 0x66, ops)
// RETFL pops the immediate's worth of bytes after the far return
// (LRET iw: CA imm16), the toolchain's RETF spelling with a stack
// adjustment.
case "RETFL":
if len(ops) != 1 {
return fmt.Errorf("RETFL expects 1 operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("RETFL expects an immediate")
}
return e.emit(&instr{opcode: []byte{0xCA}, modrm: -1, sib: -1, imm: le16(int64(imm))})
// MOVDQ2Q/MOVQ2DQ cross the MMX and XMM banks (F2 0F D6), the register
// in the reg field, the other bank's in r/m.
case "MOVDQ2Q", "MOVQ2DQ":
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
srcReg, ok1 := ops[0].(Reg)
dstReg, ok2 := ops[1].(Reg)
if !ok1 || !ok2 {
return fmt.Errorf("%s takes register operands alone", upper)
}
if upper == "MOVDQ2Q" && (!srcReg.isVec() || !dstReg.mmx) ||
upper == "MOVQ2DQ" && (!srcReg.mmx || !dstReg.isVec()) {
return fmt.Errorf("%s crosses the XMM and MMX banks in that order", upper)
}
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, srcReg, 8); err != nil {
return err
}
return e.emit(i)
// SHA256RNDS2 carries the round constant in a literal X0 first operand. // SHA256RNDS2 carries the round constant in a literal X0 first operand.
case "SHA256RNDS2": case "SHA256RNDS2":
return e.encodeSha256rnds2(ops) return e.encodeSha256rnds2(ops)
@@ -578,6 +651,7 @@ type instr struct {
disp []byte disp []byte
imm []byte imm []byte
sb *sbRef // static-symbol displacement in disp, awaiting resolution sb *sbRef // static-symbol displacement in disp, awaiting resolution
tls bool // the displacement is a TLS slot offset, patched R_TLSLE
} }
// sbRef records that an instruction's displacement refers to a static symbol // sbRef records that an instruction's displacement refers to a static symbol
@@ -620,6 +694,9 @@ func (e *enc) emit(i *instr) error {
if i.sb != nil { if i.sb != nil {
e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend}) e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend})
} }
if i.tls {
e.patches = append(e.patches, encPatch{off: len(e.out), tls: true})
}
e.out = append(e.out, i.disp...) e.out = append(e.out, i.disp...)
e.out = append(e.out, i.imm...) e.out = append(e.out, i.imm...)
return nil return nil
@@ -674,12 +751,30 @@ func setRMReg(i *instr, regField int, rexR, regForced bool, rm Operand, opSize i
i.disp = le32(0) i.disp = le32(0)
i.sb = &sbRef{name: r.name, addend: r.addend} i.sb = &sbRef{name: r.name, addend: r.addend}
return nil return nil
case TLSMem:
// off(TLS): the segment-prefixed absolute access, mod=00 with the
// SIB escape's disp32 absolute form. The displacement is the TLS
// slot offset, patched by the linker's TLS relocation.
i.prefix = r.Seg
i.modrm = 0x04 | regField<<3
i.sib = 0x25
i.disp = le32(r.Disp)
i.tls = true
return nil
case SegAbs:
// 0x30(GS): the segment override with the SIB escape's disp32
// absolute form, no relocation.
setSegAbs(i, regField, r)
return nil
default: default:
return fmt.Errorf("invalid r/m operand %T", rm) return fmt.Errorf("invalid r/m operand %T", rm)
} }
} }
func setMem(i *instr, regField int, m Mem) error { func setMem(i *instr, regField int, m Mem) error {
if m.Seg != 0 {
i.prefix = m.Seg
}
modrm, sib, disp, xBit, bBit, err := memComponents(regField, m) modrm, sib, disp, xBit, bBit, err := memComponents(regField, m)
if err != nil { if err != nil {
return err return err
@@ -692,6 +787,16 @@ func setMem(i *instr, regField int, m Mem) error {
return nil return nil
} }
// setSegAbs assembles a segment-absolute operand, 0x30(GS): the segment
// override with the mod=00 SIB escape's disp32 absolute form and no
// relocation.
func setSegAbs(i *instr, regField int, m SegAbs) {
i.prefix = m.Seg
i.modrm = 0x04 | regField<<3
i.sib = 0x25
i.disp = le32(m.Disp)
}
// memComponents computes the ModR/M byte (with the given reg field), the SIB // memComponents computes the ModR/M byte (with the given reg field), the SIB
// byte (-1 if none), the displacement bytes, and the high index/base bits, for // byte (-1 if none), the displacement bytes, and the high index/base bits, for
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths. // a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
+158 -4
View File
@@ -10,8 +10,8 @@ import (
"golang.org/x/arch/x86/x86asm" "golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// decode encodes an instruction and decodes it back, returning the decoded // decode encodes an instruction and decodes it back, returning the decoded
@@ -265,7 +265,11 @@ func TestImul(t *testing.T) {
func TestControl(t *testing.T) { func TestControl(t *testing.T) {
checkSyntax(t, "ret", "RET") checkSyntax(t, "ret", "RET")
checkSyntax(t, "nop", "NOP") // NOP contributes nothing on amd64, consumed whole by the toolchain as
// a pseudo; only the bytes pin it (no decodable instruction remains).
if code, err := Encode("NOP"); err != nil || len(code) != 0 {
t.Errorf("NOP: bytes %x (err %v), want empty", code, err)
}
checkOp(t, x86asm.JMP, "JMP", Imm(0)) checkOp(t, x86asm.JMP, "JMP", Imm(0))
checkOp(t, x86asm.CALL, "CALL", Imm(0)) checkOp(t, x86asm.CALL, "CALL", Imm(0))
checkOp(t, x86asm.JGE, "JGE", Imm(0)) checkOp(t, x86asm.JGE, "JGE", Imm(0))
@@ -1193,7 +1197,7 @@ func TestBookkeepingGroundTruth(t *testing.T) {
if err != nil { if err != nil {
t.Fatalf("assemble: %v", err) t.Fatalf("assemble: %v", err)
} }
want := "90c3" want := "c3"
if got := hexCompact(img.Code); got != want { if got := hexCompact(img.Code); got != want {
t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want) t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want)
} }
@@ -1221,3 +1225,153 @@ func mustParse(t *testing.T, src string) *ast.File {
} }
return f return f
} }
// TestCorpusTailSystem pins the system and control forms the toolchain's own
// amd64 testdata carries, byte for byte: the one-operand IMUL, the compare
// family, the far return, the loop, the MMX moves, the CR/DR and segment
// register moves, the TLS pseudo-base and the 0F AE/1C/C7 controls.
func TestCorpusTailSystem(t *testing.T) {
regBPT := Reg{idx: 5, size: 8}
X0, X1, X2 := vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")
Y1, Y2, Y7 := vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y7")
X5, X20 := vreg(t, "X5"), vreg(t, "X20")
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"IMUL one-op byte", "IMULB", []Operand{DX}, "f6ea"},
{"IMUL one-op long", "IMULL", []Operand{AX}, "f7e8"},
{"CMPPD", "CMPPD", []Operand{X1, X2, Imm(4)}, "660fc2d104"},
{"CMPSS", "CMPSS", []Operand{X1, X2, Imm(4)}, "f30fc2d104"},
{"CMPPS", "CMPPS", []Operand{X1, X2, Imm(4)}, "0fc2d104"},
{"RETFL", "RETFL", []Operand{Imm(4)}, "ca0400"},
{"LOOP", "LOOP", []Operand{Imm(-2)}, "e2fe"},
{"LOOPE", "LOOPE", []Operand{Imm(-2)}, "e1fe"},
{"LOOPNE", "LOOPNE", []Operand{Imm(-2)}, "e0fe"},
{"PADDD MMX", "PADDD", []Operand{Reg{idx: 2, size: 8, mmx: true}, Reg{idx: 1, size: 8, mmx: true}}, "0ffeca"},
{"MOVDQ2Q", "MOVDQ2Q", []Operand{X1, Reg{idx: 1, size: 8, mmx: true}}, "f20fd6c9"},
{"MOVNTDQ", "MOVNTDQ", []Operand{X1, Ptr(AX, 0, 16)}, "660fe708"},
{"MOVQ mmx load", "MOVQ", []Operand{Ptr(AX, 0, 8), Reg{idx: 0, size: 8, mmx: true}}, "0f6f00"},
{"MOVQ mmx store", "MOVQ", []Operand{Reg{idx: 0, size: 8, mmx: true}, Ptr(SI, 0, 8)}, "0f7f06"},
{"MOVQ CR0 load", "MOVQ", []Operand{Reg{idx: 0, size: 8, ctl: 1}, AX}, "0f20c0"},
{"MOVQ CR4 load", "MOVQ", []Operand{Reg{idx: 4, size: 8, ctl: 1}, DI}, "0f20e7"},
{"MOVQ CR0 store", "MOVQ", []Operand{AX, Reg{idx: 0, size: 8, ctl: 1}}, "0f22c0"},
{"MOVQ DR0 load", "MOVQ", []Operand{Reg{idx: 0, size: 8, ctl: 2}, AX}, "0f21c0"},
{"MOVQ DR7 load", "MOVQ", []Operand{Reg{idx: 7, size: 8, ctl: 2}, SI}, "0f21fe"},
{"PUSHQ FS", "PUSHQ", []Operand{Reg{idx: 4, size: 2, seg: 5}}, "0fa0"},
{"PUSHQ GS", "PUSHQ", []Operand{Reg{idx: 5, size: 2, seg: 6}}, "0fa8"},
{"POPQ FS", "POPQ", []Operand{Reg{idx: 4, size: 2, seg: 5}}, "0fa1"},
{"POPQ GS", "POPQ", []Operand{Reg{idx: 5, size: 2, seg: 6}}, "0fa9"},
{"ENDBR64", "ENDBR64", nil, "f30f1efa"},
{"CLWB", "CLWB", []Operand{Ptr(BX, 0, 8)}, "660fae33"},
{"CLDEMOTE", "CLDEMOTE", []Operand{Ptr(BX, 0, 8)}, "0f1c03"},
{"TPAUSE", "TPAUSE", []Operand{BX}, "660faef3"},
{"UMONITOR", "UMONITOR", []Operand{BX}, "f30faef3"},
{"UMWAIT", "UMWAIT", []Operand{BX}, "f20faef3"},
{"RDPID", "RDPID", []Operand{DX}, "f30fc7fa"},
{"RDPID r11", "RDPID", []Operand{Reg{idx: 11, size: 8}}, "f3410fc7fb"},
{"LEAL wide disp", "LEAL", []Operand{Idx(regBPT, Reg{idx: 10, size: 8}, 1, 0x8f1bbcdc, 8), regBPT}, "428dac15dcbc1b8f"},
{"VPERMPD", "VPERMPD", []Operand{Imm(0xd8), Y7, Y7}, "c4e3fd01ffd8"},
{"VPERMILPD", "VPERMILPD", []Operand{Imm(0xff), X1, X2}, "c4e37905d1ff"},
{"VPERMILPS", "VPERMILPS", []Operand{Imm(0xff), X1, X2}, "c4e37904d1ff"},
{"VROUNDPD", "VROUNDPD", []Operand{Imm(-1), X1, X2}, "c4e37909d1ff"},
{"VROUNDPS", "VROUNDPS", []Operand{Imm(-1), Y1, Y2}, "c4e37d08d1ff"},
{"VAESKEYGENASSIST", "VAESKEYGENASSIST", []Operand{Imm(-1), X1, X2}, "c4e379dfd1ff"},
{"VPCMPESTRI", "VPCMPESTRI", []Operand{Imm(-1), X1, X2}, "c4e37961d1ff"},
{"VPCMPESTRM", "VPCMPESTRM", []Operand{Imm(-1), X1, X2}, "c4e37960d1ff"},
{"VPCMPISTRI", "VPCMPISTRI", []Operand{Imm(-1), X1, X2}, "c4e37963d1ff"},
{"VPCMPISTRM", "VPCMPISTRM", []Operand{Imm(-1), X1, X2}, "c4e37962d1ff"},
{"VEXTRACTPS", "VEXTRACTPS", []Operand{Imm(-1), X1, AX}, "c4e37917c8ff"},
{"VPEXTRW", "VPEXTRW", []Operand{Imm(0xff), X1, AX}, "c4e37915c8ff"},
{"VPBLENDVB", "VPBLENDVB", []Operand{X0, Ptr(BX, 0, 16), X1, X2}, "c4e3714c1300"},
{"VMOVHPD load", "VMOVHPD", []Operand{Ptr(AX, 0, 8), X5, X5}, "c5d11628"},
{"VMOVHPD load disp", "VMOVHPD", []Operand{Ptr(DX, 7, 8), X5, X5}, "c5d1166a07"},
{"VMOVHPD store", "VMOVHPD", []Operand{X5, Ptr(AX, 0, 8)}, "c5f91728"},
{"VMOVLPD load", "VMOVLPD", []Operand{Ptr(AX, 0, 8), X5, X5}, "c5d11228"},
{"VMOVLPD store", "VMOVLPD", []Operand{X5, Ptr(AX, 0, 8)}, "c5f91328"},
{"VMOVQ EVEX gpr load", "VMOVQ", []Operand{Reg{idx: 4, size: 8}, X20}, "62e1fd086ee4"},
{"VMOVQ EVEX mem store", "VMOVQ", []Operand{X20, Ptr(AX, 0, 8)}, "62e1fd087e20"},
{"VMOVQ EVEX mem load", "VMOVQ", []Operand{Ptr(AX, 0, 8), X20}, "62e1fd086e20"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
}
// TestCorpusTailFileForms pins the file-level forms the toolchain's amd64
// testdata carries: the star-marked indirect jumps, the TLS pseudo-base, the
// paired-register shift spelling, the absolute RET and the jump to an
// undefined static symbol (its displacement and the TLS slot offsets are
// relocation sites, zeroed here as the kernel parity suites do).
func TestCorpusTailFileForms(t *testing.T) {
mask32 := func(b []byte, at int) { b[at], b[at+1], b[at+2], b[at+3] = 0, 0, 0, 0 }
cases := []struct {
name string
src string
want string // hex, with X marking a masked 32-bit relocation site
}{
{"star reg jump", "\tJMP *(R12)\n\tRET\n", "41ff2424c3"},
{"star sp jump", "\tJMP *4(SP)\n\tRET\n", "ff642404c3"},
{"star indexed jump", "\tJMP *(R12)(R13*4)\n\tRET\n", "43ff24acc3"},
{"TLS load", "\tMOVQ (TLS), AX\n\tRET\n", "64488b0425XXXXXXXXc3"},
{"TLS load offset", "\tMOVQ 8(TLS), DX\n\tRET\n", "64488b1425XXXXXXXXc3"},
{"colon shift", "\tSHLL CX, R11:AX\n\tRET\n", "410fa5c3c3"},
{"SP indexed local", "\tMOVQ foo(SP)(AX*1), BX\n\tRET\n", "488b1c04c3"},
}
for _, c := range cases {
f, errs := parser.Parse("t_amd64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n"+c.src)
if len(errs) > 0 {
t.Errorf("%s: parse: %v", c.name, errs)
continue
}
img, err := AssembleFile(f)
if err != nil {
t.Errorf("%s: assemble: %v", c.name, err)
continue
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
if at := strings.Index(c.want, "XXXXXXXX"); at >= 0 {
mask32(code, at/2) // the masked relocation site
}
if got := hexCompact(code); got != strings.ReplaceAll(c.want, "X", "0") {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
// JMP to an undefined static symbol and the absolute RET: their rel32
// carries a call relocation against the symbol, masked to zero here.
for _, c := range []struct{ name, src string }{
{"static external jump", "\tJMP bar<>+4(SB)\n\tRET\n"},
{"static external indexed jump", "\tJMP bar<>+4(SB)(R11*4)\n\tRET\n"},
{"absolute ret", "\tRET\n\tRET foo(SB)\n"},
} {
f, errs := parser.Parse("t_amd64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n"+c.src)
if len(errs) > 0 {
t.Errorf("%s: parse: %v", c.name, errs)
continue
}
img, err := AssembleFile(f)
if err != nil {
t.Errorf("%s: assemble: %v", c.name, err)
continue
}
fn := img.Funcs[0]
code := maskCode(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
want := "e900000000c3"
if c.name == "absolute ret" {
want = "c3e900000000"
}
if got := hexCompact(code); got != want {
t.Errorf("%s: bytes %s, want %s", c.name, got, want)
}
}
}
+67 -18
View File
@@ -856,37 +856,41 @@ type evexMoveSpec struct {
vecOK bool // the non-memory operand may be a vector register vecOK bool // the non-memory operand may be a vector register
xmmOnly bool // wider than XMM registers are rejected xmmOnly bool // wider than XMM registers are rejected
nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS) nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS)
gprOK bool // the r/m side may be a general-purpose register (VMOVQ)
} }
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding. // evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{ var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0, unaligned integer move. // EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false}, "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F3.0F.W1, unaligned qword move. // EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false}, "VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the // EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple // F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics). // semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false}, "VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword // EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding). // encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false}, "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.66.0F.W1, unaligned packed double move. // EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false}, "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512, aligned packed moves. // EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false}, "VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false}, "VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.66.0F, aligned integer moves. // EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false}, "VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false}, "VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the // EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported). // three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true}, "VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true, false},
// EVEX.128.F2.0F.W1, scalar double move: memory operands and the // EVEX.128.F2.0F.W1, scalar double move: memory operands and the
// three-operand register form (VMOVSD dst, src1, src2). // three-operand register form (VMOVSD dst, src1, src2).
"VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true}, "VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true, false},
// EVEX.128/256/512.0F.W0, unaligned packed single move. // EVEX.128/256/512.0F.W0, unaligned packed single move.
"VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false}, "VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128.66.0F.W1, the 64-bit GPR/memory ↔ XMM move (VMOVQ RSP, X20
// and friends, the EVEX spelling the high registers demand).
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, [3]int{8, 8, 8}, true, true, false, true},
} }
// isEvex reports whether the mnemonic has an EVEX encoding we handle. // isEvex reports whether the mnemonic has an EVEX encoding we handle.
@@ -900,6 +904,9 @@ func isEvex(mnemUpper string) bool {
if _, ok := evexMoveTable[mnemUpper]; ok { if _, ok := evexMoveTable[mnemUpper]; ok {
return true return true
} }
if _, ok := evexHptrTable[mnemUpper]; ok {
return true
}
return isEvexQuad(mnemUpper) return isEvexQuad(mnemUpper)
} }
@@ -911,7 +918,13 @@ func evexRequired(upper string, ops []Operand) bool {
_, inVex := vexTable[upper] _, inVex := vexTable[upper]
_, inVexMove := vexMoveTable[upper] _, inVexMove := vexMoveTable[upper]
if !inVex && !inVexMove { if !inVex && !inVexMove {
return true // EVEX-only mnemonic // The dual-shape moves pick their VEX form by operand count, so
// they are not EVEX-only either.
switch upper {
case "VMOVHPD", "VMOVLPD":
default:
return true // EVEX-only mnemonic
}
} }
// The byte-quad shifts have VEX register forms but EVEX-only memory // The byte-quad shifts have VEX register forms but EVEX-only memory
// forms: a memory count source forces the EVEX encoding. // forms: a memory count source forces the EVEX encoding.
@@ -1018,6 +1031,8 @@ var evexRound = map[string]bool{
"VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true, "VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true,
"VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true, "VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true,
"VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true, "VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true,
// The scalar compares suppress exceptions on their LIG encoding.
"VCMPSD": true, "VCMPSS": true,
} }
// evexBcstN maps an instruction accepting .BCST to the broadcast element // evexBcstN maps an instruction accepting .BCST to the broadcast element
@@ -1082,6 +1097,14 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
return e.encodeEvexRM(spec, ops, 0, sfx) return e.encodeEvexRM(spec, ops, 0, sfx)
} }
spec, inTable := evexTable[mnemUpper] spec, inTable := evexTable[mnemUpper]
// A high/low half move that lives in the hptr table alone (the packed
// double twins) reaches the same inTable block below, which completes
// its spec from the hptr entry.
if !inTable {
if _, ok := evexHptrTable[mnemUpper]; ok {
inTable = true
}
}
if q, ok := evexQuadTable[mnemUpper]; ok { if q, ok := evexQuadTable[mnemUpper]; ok {
// The quad-register family carries no rounding, SAE or broadcast; // The quad-register family carries no rounding, SAE or broadcast;
// only masking and zeroing apply. // only masking and zeroing apply.
@@ -1446,6 +1469,19 @@ func (e *enc) encodeEvexExtractGPR(spec evexSpec, ops []Operand, mask int, sfx e
// assembler. The scalar moves also carry a three-operand register form // assembler. The scalar moves also carry a three-operand register form
// (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3 // (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3
// opens. // opens.
// validEvexMoveOther reports whether the non-vector side of an EVEX move may
// take the operand: memory always, a general-purpose register when gprOK.
func validEvexMoveOther(ms evexMoveSpec, op Operand) bool {
if memOperand(op) {
return true
}
if !ms.gprOK {
return false
}
r, ok := op.(Reg)
return ok && !r.isVec() && !r.mask && r.ctl == 0 && !r.mmx && !r.fp
}
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error { func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) == 3 { if len(ops) == 3 {
if !ms.nds3 { if !ms.nds3 {
@@ -1453,7 +1489,7 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
} }
// The masked scalar register form keeps the Go assembler's own // The masked scalar register form keeps the Go assembler's own
// layout: the store opcode with reg = op0, vvvv = op1 and the // layout: the store opcode with reg = op0, vvvv = op1 and the
// destination in r/m (op2) — the bytes go tool asm emits, not // destination in r/m (op2), the bytes go tool asm emits, not
// the manual's NDS reading. // the manual's NDS reading.
src, src1, dst := ops[0], ops[1], ops[2] src, src1, dst := ops[0], ops[1], ops[2]
reg, ok := src.(Reg) reg, ok := src.(Reg)
@@ -1494,12 +1530,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
} }
reg, rm = srcReg, dst reg, rm = srcReg, dst
case srcIsVec: case srcIsVec:
if !memOperand(dst) { if !validEvexMoveOther(ms, dst) {
return fmt.Errorf("%s: invalid destination operand", mnem) return fmt.Errorf("%s: invalid destination operand", mnem)
} }
reg, rm = srcReg, dst reg, rm = srcReg, dst
case dstIsVec: case dstIsVec:
if !memOperand(src) { if !validEvexMoveOther(ms, src) {
return fmt.Errorf("%s: invalid source operand", mnem) return fmt.Errorf("%s: invalid source operand", mnem)
} }
op = ms.load op = ms.load
@@ -1890,6 +1926,15 @@ var evexHptrTable = map[string]evexHptrSpec{
"VMOVLHPS": { "VMOVLHPS": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}}, insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
}, },
// The packed-double twins, 66-prefixed.
"VMOVHPD": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 1, pp: 1, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x17, w: 1, pp: 1, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
"VMOVLPD": {
insert: evexSpec{mapSel: 1, opcode: 0x12, w: 1, pp: 1, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x13, w: 1, pp: 1, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
} }
// encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib. // encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib.
@@ -1954,7 +1999,7 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
if !ok || !maskReg.isVec() { if !ok || !maskReg.isVec() {
return fmt.Errorf("%s: mask must be a vector register", upper) return fmt.Errorf("%s: mask must be a vector register", upper)
} }
vsib, _, err := vsibLen(rest[1], upper) vsib, idxLen, err := vsibLen(rest[1], upper)
if err != nil { if err != nil {
return err return err
} }
@@ -1962,12 +2007,16 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
if !ok || !dst.isVec() { if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", upper) return fmt.Errorf("%s: destination must be a vector register", upper)
} }
// The L bit is the wider of the data register and the VSIB index
// lengths (a YMM index under an XMM destination selects 256-bit, the
// bytes go tool asm emits).
ll := max(idxLen, dst.vecLenBit())
spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1} spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1}
rBit := 0 rBit := 0
if dst.idx >= 8 { if dst.idx >= 8 {
rBit = 1 rBit = 1
} }
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib) return e.emitVexFields(spec, ll, dst.idx&7, rBit, 15-maskReg.idx, vsib)
} }
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src, // encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src,
+8
View File
@@ -63,6 +63,7 @@ const (
const ( const (
kindSTEXT = 1 kindSTEXT = 1
kindSRODATA = 3 kindSRODATA = 3
kindSNOPTRDATA = 5
kindSDATA = 7 kindSDATA = 7
kindSDWARFFCN = 14 kindSDWARFFCN = 14
kindSDWARFLINES = 20 kindSDWARFLINES = 20
@@ -305,9 +306,16 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
if !d.Static { if !d.Static {
name = pkgPath + "." + name name = pkgPath + "." + name
} }
// RODATA implies no pointers, so it wins over NOPTR: the kind is
// SRODATA either way, exactly as the toolchain chooses it. Plain
// NOPTR data is SNOPTRDATA, which the linker keeps out of the GC's
// type scan; a plain SDATA symbol would demand Go type information
// no assembly file can supply, and the link would fail.
typ := uint8(kindSDATA) typ := uint8(kindSDATA)
if d.Rodata { if d.Rodata {
typ = kindSRODATA typ = kindSRODATA
} else if d.Noptr {
typ = kindSNOPTRDATA
} }
flag := uint8(0) flag := uint8(0)
if d.Dupok { if d.Dupok {
+3
View File
@@ -45,6 +45,9 @@ func TestReadRuntimeSymbols(t *testing.T) {
// TestResolveExternalSymbols verifies end-to-end resolution of external // TestResolveExternalSymbols verifies end-to-end resolution of external
// symbol references. // symbol references.
func TestResolveExternalSymbols(t *testing.T) { func TestResolveExternalSymbols(t *testing.T) {
if testing.Short() {
t.Skip("resolves through a live go list -export: skipped in -short mode")
}
if _, err := exec.LookPath("go"); err != nil { if _, err := exec.LookPath("go"); err != nil {
t.Skip("go toolchain not available") t.Skip("go toolchain not available")
} }
+143 -1
View File
@@ -12,7 +12,7 @@ import (
"strings" "strings"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// goobjView is a minimal parsed view of a GOOBJ payload, enough to check // goobjView is a minimal parsed view of a GOOBJ payload, enough to check
@@ -514,6 +514,145 @@ func fieldAfter(line, flag string) string {
return "" return ""
} }
// buildLogSteps is what a substitution test needs from a `go build -x -work`
// log: the work directory, the assembler's object, the package archive and
// the link command line.
type buildLogSteps struct {
work string
asmObj string // $WORK expanded
pkgArch string // $WORK expanded
linkLine string // still carries $WORK placeholders
}
// parseBuildLog extracts the build steps from a `go build -x -work` log.
// asmFile names the assembly file whose object the test substitutes. A
// missing step is a failure, not a skip: the toolchain changed shape and the
// substitution would silently test nothing.
func parseBuildLog(t *testing.T, log []byte, asmFile string) buildLogSteps {
t.Helper()
var st buildLogSteps
for line := range strings.SplitSeq(string(log), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
st.work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, asmFile) && !strings.Contains(line, "-gensymabis"):
st.asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
rest := strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
st.pkgArch = strings.Fields(strings.SplitN(rest, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
st.linkLine = line
}
}
if st.work == "" || st.asmObj == "" || st.pkgArch == "" || st.linkLine == "" {
t.Fatalf("could not locate the build steps (work=%q asmObj=%q pkgArch=%q link=%q):\n%s",
st.work, st.asmObj, st.pkgArch, st.linkLine, log)
}
st.asmObj = strings.ReplaceAll(st.asmObj, "$WORK", st.work)
st.pkgArch = strings.ReplaceAll(st.pkgArch, "$WORK", st.work)
return st
}
// substituteAndRelink swaps the gasm object into the package archive the
// baseline build produced and re-runs the captured link line against the
// rebuilt archive, writing the binary to outBin (the -x log's link step
// always targets the action graph's internal a.out, which the helper
// redirects; the copy to the -o target is a separate build action the helper
// does not need). The archive handed to the linker is proven to carry the
// gasm object byte for byte, so a build-layout change that skipped the
// substitution fails here instead of passing vacuously.
func substituteAndRelink(t *testing.T, goBin, dir string, st buildLogSteps, outBin string, gasmObj []byte, extraEnv ...string) {
t.Helper()
// The deliberate-run boundary: this path drives a real `go build` and
// cmd/link per invocation, minutes-scale work on the small single-core
// CI runner. Under -short (the push pipeline's mode) it skips; the
// local test gate and the dispatched workflows run it in full.
if testing.Short() {
t.Skip("end-to-end go build and link: skipped in -short mode")
}
// Extract the archive, overwrite the assembler's member with the gasm
// object and repack (go tool pack has no replace-in-place).
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract := exec.Command(goBin, "tool", "pack", "x", st.pkgArch)
extract.Dir = membersDir
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
member := filepath.Join(membersDir, filepath.Base(st.asmObj))
if _, err := os.Stat(member); err != nil {
t.Fatalf("the assembler's archive member was not extracted: %v", err)
}
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, gasmObj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", st.pkgArch)
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for m := range strings.FieldsSeq(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, m)
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
// Prove the substitution: the archive the linker is about to consume
// holds the gasm object, byte for byte.
checkDir := filepath.Join(dir, "check")
if err := os.MkdirAll(checkDir, 0o755); err != nil {
t.Fatal(err)
}
check := exec.Command(goBin, "tool", "pack", "x", newArch)
check.Dir = checkDir
if out, err := check.CombinedOutput(); err != nil {
t.Fatalf("pack x (verification): %v\n%s", err, out)
}
got, err := os.ReadFile(filepath.Join(checkDir, filepath.Base(st.asmObj)))
if err != nil {
t.Fatalf("read the substituted member back: %v", err)
}
if !bytes.Equal(got, gasmObj) {
t.Fatal("the repacked archive does not carry the gasm object")
}
// Re-link. The line carries a GOROOT assignment and $WORK placeholders;
// GOEXPERIMENT must match the toolchain's own, because the linker
// compares the object header against its configuration.
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
linkLine := strings.ReplaceAll(st.linkLine, "$WORK", st.work)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(st.work, "b001", "_pkg_.a"), newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(st.work, "b001", "exe", "a.out"), outBin)
env := append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)))
env = append(env, extraEnv...)
link := exec.Command("sh", "-c", linkLine)
link.Dir = dir
link.Env = env
if out, err := link.CombinedOutput(); err != nil {
t.Fatalf("link with the gasm object: %v\n%s", err, out)
}
}
// TestGOObjectExternalPackageLink is the cross-package end-to-end check: a // TestGOObjectExternalPackageLink is the cross-package end-to-end check: a
// GOOBJ whose code references a real external package symbol (runtime's // GOOBJ whose code references a real external package symbol (runtime's
// morestack, a plain reference rather than the builtin noctxt form) must // morestack, a plain reference rather than the builtin noctxt form) must
@@ -524,6 +663,9 @@ func fieldAfter(line, flag string) string {
// failed. The binary is not run: morestack returns to the call site's // failed. The binary is not run: morestack returns to the call site's
// stack check, which a hand-written caller has none of. // stack check, which a hand-written caller has none of.
func TestGOObjectExternalPackageLink(t *testing.T) { func TestGOObjectExternalPackageLink(t *testing.T) {
if testing.Short() {
t.Skip("end-to-end go build and link: skipped in -short mode")
}
goBin, err := exec.LookPath("go") goBin, err := exec.LookPath("go")
if err != nil { if err != nil {
t.Skip("no Go toolchain available") t.Skip("no Go toolchain available")
+6 -6
View File
@@ -9,8 +9,8 @@ import (
"strings" "strings"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64, // The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64,
@@ -307,12 +307,12 @@ func TestStackGuardBytesLOONG64(t *testing.T) {
func TestStackGuardGOObjInternalCall(t *testing.T) { func TestStackGuardGOObjInternalCall(t *testing.T) {
for _, tt := range []struct { for _, tt := range []struct {
src string src string
assemble func(*ast.File) (*Image, error) assemble func(*ast.File, ...AssembleOption) (*Image, error)
}{ }{
{"g_amd64.s", AssembleFile}, {"g_amd64.s", AssembleFile},
{"g_arm64.s", AssembleFileARM64}, {"g_arm64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileARM64(f) }},
{"g_riscv64.s", AssembleFileRISCV}, {"g_riscv64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileRISCV(f) }},
{"g_loong64.s", AssembleFileLOONG64}, {"g_loong64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileLOONG64(f) }},
} { } {
f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n") f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 { if len(errs) > 0 {
+271 -25
View File
@@ -90,6 +90,40 @@ var noOperandTable = map[string][]byte{
"LOCK": {0xF0}, "LOCK": {0xF0},
"REP": {0xF3}, "REP": {0xF3},
"REPN": {0xF2}, "REPN": {0xF2},
"ENDBR64": {0xF3, 0x0F, 0x1E, 0xFA},
}
// sysUnaryTable maps the one-operand system instructions to their bytes:
// the prefix, the opcode and the /digit the reg field carries. The operand
// is a register or memory in r/m.
var sysUnaryTable = map[string]struct {
prefix byte
opcode []byte
digit int
}{
"CLWB": {0x66, []byte{0x0F, 0xAE}, 6},
"TPAUSE": {0x66, []byte{0x0F, 0xAE}, 6},
"UMONITOR": {0xF3, []byte{0x0F, 0xAE}, 6},
"UMWAIT": {0xF2, []byte{0x0F, 0xAE}, 6},
"RDPID": {0xF3, []byte{0x0F, 0xC7}, 7},
"CLDEMOTE": {0x00, []byte{0x0F, 0x1C}, 0},
}
// encodeSysUnary emits a one-operand system instruction: the operand in r/m
// under the fixed /digit, no REX.W.
func (e *enc) encodeSysUnary(mnem string, m struct {
prefix byte
opcode []byte
digit int
}, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
}
i := &instr{prefix: m.prefix, opcode: m.opcode, modrm: -1, sib: -1}
if err := setRMDigit(i, m.digit, ops[0], 8); err != nil {
return err
}
return e.emit(i)
} }
// --- MOV -------------------------------------------------------------------- // --- MOV --------------------------------------------------------------------
@@ -111,6 +145,76 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
// silently emit REX.W 8B with the wrong operand meaning. // silently emit REX.W 8B with the wrong operand meaning.
_, srcVec := vecReg(src) _, srcVec := vecReg(src)
dstReg, dstVec := vecReg(dst) dstReg, dstVec := vecReg(dst)
// Control and debug register moves: 0F 20 (CRn→r64), 0F 22 (r64→CRn),
// 0F 21 (DRn→r64) and 0F 23 (r64→DRn). The CR/DR number rides the reg
// field, the general register r/m; CR8+/DR8+ take REX.R.
if c, ok := src.(Reg); ok && c.ctl != 0 {
g, ok := dst.(Reg)
if !ok || g.isVec() || g.ctl != 0 {
return fmt.Errorf("MOV: control/debug register load needs a general register destination")
}
opc := byte(0x20)
if c.ctl == 2 {
opc = 0x21
}
return e.emit(&instr{
opcode: []byte{0x0F, opc},
modrm: 0xC0 | (c.idx&7)<<3 | (g.idx & 7),
sib: -1, rexR: c.idx >= 8, rexB: g.idx >= 8,
})
}
if c, ok := dst.(Reg); ok && c.ctl != 0 {
g, ok := src.(Reg)
if !ok || g.isVec() || g.ctl != 0 {
return fmt.Errorf("MOV: control/debug register store needs a general register source")
}
opc := byte(0x22)
if c.ctl == 2 {
opc = 0x23
}
return e.emit(&instr{
opcode: []byte{0x0F, opc},
modrm: 0xC0 | (c.idx&7)<<3 | (g.idx & 7),
sib: -1, rexR: c.idx >= 8, rexB: g.idx >= 8,
})
}
// MMX register moves: MOVQ M0, mem and MOVQ mem, M0 are the MMX
// load/store pair 0F 6F/0F 7F (no prefix); a register pair takes the
// load opcode. The XMM MOVQ forms follow below.
if m, ok := src.(Reg); ok && m.mmx {
switch d := dst.(type) {
case Reg:
if !d.mmx {
return fmt.Errorf("MOV: MMX register moves stay inside the M bank")
}
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
if err := setRM(i, d, src, 8); err != nil {
return err
}
return e.emit(i)
case Mem:
i := &instr{opcode: []byte{0x0F, 0x7F}, modrm: -1, sib: -1}
if err := setRM(i, m, d, 8); err != nil {
return err
}
return e.emit(i)
}
return fmt.Errorf("MOV: invalid MMX destination")
}
if m, ok := dst.(Reg); ok && m.mmx {
srcM, ok := src.(Mem)
if !ok {
return fmt.Errorf("MOV: MMX load takes a memory source")
}
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
if err := setRM(i, m, srcM, 8); err != nil {
return err
}
return e.emit(i)
}
if srcVec || dstVec { if srcVec || dstVec {
if dstVec { if dstVec {
if g, ok := src.(Reg); ok && !g.isVec() { if g, ok := src.(Reg); ok && !g.isVec() {
@@ -192,6 +296,30 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
} }
return e.emit(i) return e.emit(i)
case TLSMem:
if !dstIsReg {
return fmt.Errorf("MOV: two memory operands")
}
// MOV r, off(TLS): the segment-prefixed absolute load, reg=dst,
// rm=src(tlsMem) through the SIB escape; the disp32 is the TLS slot
// offset with its R_TLSLE patch site.
i := newInstr(size, []byte{movRR(size)})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
case SegAbs:
if !dstIsReg {
return fmt.Errorf("MOV: two memory operands")
}
// MOV r, 0x30(GS): the segment-absolute load.
i := newInstr(size, []byte{movRR(size)})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
case Imm: case Imm:
if dstIsReg { if dstIsReg {
v := int64(src) v := int64(src)
@@ -232,11 +360,24 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
i.imm = imm i.imm = imm
return e.emit(i) return e.emit(i)
} }
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0. // MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0. An immediate in the
// destination slot is the absolute-address crash-store spelling,
// MOVL $0xf1, 0xf1: the parser reads the trailing bare constant
// as an immediate, and the store's disp32 carries the address.
op := byte(0xC7) op := byte(0xC7)
if size == 1 { if size == 1 {
op = 0xC6 op = 0xC6
} }
if d, ok := dst.(Imm); ok {
i := newInstr(size, []byte{op})
setSegAbs(i, 0, SegAbs{Disp: int64(d)})
immBytes, err := immediate(int64(src), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
i := newInstr(size, []byte{op}) i := newInstr(size, []byte{op})
if err := setRMDigit(i, 0, dst, size); err != nil { if err := setRMDigit(i, 0, dst, size); err != nil {
return err return err
@@ -481,6 +622,14 @@ func (e *enc) encodeLea(ops []Operand, size int) error {
default: default:
return fmt.Errorf("LEA: source must be a memory operand") return fmt.Errorf("LEA: source must be a memory operand")
} }
// LEA accepts the full unsigned 32-bit displacement span where the
// loads and stores reject it beyond the signed one; the wide values
// ride the same disp32 bytes as their two's-complement bit pattern.
if m, ok := src.(Mem); ok && m.Disp >= 1<<31 && m.Disp <= (1<<32)-1 {
c := m
c.Disp = int64(int32(uint32(m.Disp)))
src = c
}
i := newInstr(size, []byte{0x8D}) i := newInstr(size, []byte{0x8D})
if err := setRM(i, dstReg, src, size); err != nil { if err := setRM(i, dstReg, src, size); err != nil {
return err return err
@@ -626,8 +775,29 @@ func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error {
func (e *enc) encodeImul(ops []Operand, size int) error { func (e *enc) encodeImul(ops []Operand, size int) error {
switch len(ops) { switch len(ops) {
case 1:
// The one-operand form, IMUL r/m: F6/F7 /5 with AL/AX/EAX/RAX as the
// implied destination (the toolchain's one-register shape).
opc := byte(0xF7)
if size == 1 {
opc = 0xF6
}
i := newInstr(size, []byte{opc})
if err := setRMDigit(i, 5, ops[0], size); err != nil {
return err
}
return e.emit(i)
case 2: case 2:
// IMUL r, r/m: 0x0F 0xAF. // Two shapes. The leading-immediate spelling IMUL $imm, r multiplies
// r in place (dst = rm = r): the shape GOROOT's clock code writes.
// Otherwise IMUL r, r/m: 0x0F 0xAF.
if imm, ok := ops[0].(Imm); ok {
dstReg, isReg := ops[1].(Reg)
if !isReg {
return fmt.Errorf("IMUL: destination must be a register")
}
return e.encodeImulImm(imm, dstReg, dstReg, size)
}
dstReg, ok := ops[1].(Reg) dstReg, ok := ops[1].(Reg)
if !ok { if !ok {
return fmt.Errorf("IMUL: destination must be a register") return fmt.Errorf("IMUL: destination must be a register")
@@ -647,27 +817,34 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
if !ok { if !ok {
return fmt.Errorf("IMUL: immediate operand expected first") return fmt.Errorf("IMUL: immediate operand expected first")
} }
// Plan 9 order: IMUL $imm, src, dst. // Plan 9 order: IMUL $imm, src, dst; the source stays a general
if fits8(int64(imm)) { // r/m operand (setRM takes registers and memory alike).
i := newInstr(size, []byte{0x6B}) return e.encodeImulImm(imm, ops[1], dstReg, size)
if err := setRM(i, dstReg, ops[1], size); err != nil { }
return err return fmt.Errorf("IMUL expects 1, 2 or 3 operands, got %d", len(ops))
} }
i.imm = []byte{byte(int8(imm))}
return e.emit(i) // encodeImulImm emits the immediate multiply: 0x6B with a sign-extended imm8
} // when the value fits, 0x69 with a 32-bit immediate otherwise.
i := newInstr(size, []byte{0x69}) func (e *enc) encodeImulImm(imm Imm, rm Operand, dst Reg, size int) error {
if err := setRM(i, dstReg, ops[1], size); err != nil { if fits8(int64(imm)) {
i := newInstr(size, []byte{0x6B})
if err := setRM(i, dst, rm, size); err != nil {
return err return err
} }
immBytes, err := immediate(int64(imm), size, false) i.imm = []byte{byte(int8(imm))}
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i) return e.emit(i)
} }
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops)) i := newInstr(size, []byte{0x69})
if err := setRM(i, dst, rm, size); err != nil {
return err
}
immBytes, err := immediate(int64(imm), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
} }
// --- PUSH / POP ------------------------------------------------------------- // --- PUSH / POP -------------------------------------------------------------
@@ -689,6 +866,27 @@ func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
w16 := size == 2 w16 := size == 2
switch op := ops[0].(type) { switch op := ops[0].(type) {
case Reg: case Reg:
// Segment registers: FS and GS carry their own one-byte opcodes
// under 0F (A0/A8 push, A1/A9 pop); the other four spellings are
// not pushable in 64-bit mode.
if n, isSeg := op.segNumber(); isSeg {
switch n {
case 4: // FS
if push {
return e.emit(&instr{opcode: []byte{0x0F, 0xA0}, modrm: -1, sib: -1})
}
return e.emit(&instr{opcode: []byte{0x0F, 0xA1}, modrm: -1, sib: -1})
case 5: // GS
if push {
return e.emit(&instr{opcode: []byte{0x0F, 0xA8}, modrm: -1, sib: -1})
}
return e.emit(&instr{opcode: []byte{0x0F, 0xA9}, modrm: -1, sib: -1})
}
return fmt.Errorf("PUSH/POP: only FS and GS are encodable in 64-bit mode")
}
if op.mmx || op.isVec() || op.fp || op.ctl != 0 {
return fmt.Errorf("PUSH/POP: invalid register operand")
}
base := byte(0x50) // PUSH r; POP is 0x58 base := byte(0x50) // PUSH r; POP is 0x58
if !push { if !push {
base = 0x58 base = 0x58
@@ -1040,6 +1238,38 @@ var sseMoveTable = map[string]sseMove{
"MOVSS": {0xF3, 0x10, 0x11}, // scalar single "MOVSS": {0xF3, 0x10, 0x11}, // scalar single
} }
// sseStoreOnly holds the store-only SSE forms, OP xmm, mem: the XMM register
// rides the reg field and memory r/m (the non-temporal store).
var sseStoreOnly = map[string]struct {
prefix byte
op byte
}{
"MOVNTDQ": {0x66, 0xE7},
}
// encodeSSEStoreOnly encodes OP xmm, mem (reg = the XMM source, r/m = the
// destination memory).
func (e *enc) encodeSSEStoreOnly(mnem string, m struct {
prefix byte
op byte
}, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("%s source must be a vector register", mnem)
}
if !isX86Mem(ops[1]) {
return fmt.Errorf("%s destination must be a memory operand", mnem)
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, srcReg, ops[1], 8); err != nil {
return err
}
return e.emit(i)
}
// encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the // encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the
// load form (reg = destination), matching the Go assembler. // load form (reg = destination), matching the Go assembler.
func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error { func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
@@ -1255,14 +1485,23 @@ func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error {
} }
src, dst := ops[0], ops[1] src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg) dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() { if !ok || (!dstReg.isVec() && !dstReg.mmx) {
return fmt.Errorf("SSE binary destination must be a vector register") return fmt.Errorf("SSE binary destination must be a vector register")
} }
// The MMX twins of the packed-integer SSE2 ops drop the 0x66 prefix:
// PADDD M2, M1 is 0F FE where the XMM form is 66 0F FE.
prefix := m.prefix
if dstReg.mmx {
if prefix != 0x66 {
return fmt.Errorf("SSE binary: this form takes no MMX register operand")
}
prefix = 0
}
opcode := []byte{0x0F, m.op} opcode := []byte{0x0F, m.op}
if m.map38 { if m.map38 {
opcode = []byte{0x0F, 0x38, m.op} opcode = []byte{0x0F, 0x38, m.op}
} }
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1} i := &instr{prefix: prefix, opcode: opcode, modrm: -1, sib: -1}
if err := setRM(i, dstReg, src, 8); err != nil { if err := setRM(i, dstReg, src, 8); err != nil {
return err return err
} }
@@ -1738,12 +1977,19 @@ func (e *enc) encodeSSEShift(name string, ops []Operand) error {
// immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family: // immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family:
// F2 0F C2 with reg = dst, rm = src. // F2 0F C2 with reg = dst, rm = src.
func (e *enc) encodeCmpsd(ops []Operand) error { func (e *enc) encodeCmpsd(ops []Operand) error {
return e.encodeSSECmp("CMPSD", 0xF2, ops)
}
// encodeSSECmp encodes the SSE compare family (CMPSD/CMPSS/CMPPS/CMPPD):
// 0F C2 /r ib with the predicate immediate last in Plan 9 order
// (src, dst, $imm) and the packed forms' prefixes.
func (e *enc) encodeSSECmp(mnem string, prefix byte, ops []Operand) error {
if len(ops) != 3 { if len(ops) != 3 {
return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops)) return fmt.Errorf("%s expects 3 operands (src, dst, $imm), got %d", mnem, len(ops))
} }
imm, ok := ops[2].(Imm) imm, ok := ops[2].(Imm)
if !ok { if !ok {
return fmt.Errorf("CMPSD predicate must be an immediate") return fmt.Errorf("%s predicate must be an immediate", mnem)
} }
immByte, err := imm8(int64(imm)) immByte, err := imm8(int64(imm))
if err != nil { if err != nil {
@@ -1751,9 +1997,9 @@ func (e *enc) encodeCmpsd(ops []Operand) error {
} }
dstReg, ok2 := ops[1].(Reg) dstReg, ok2 := ops[1].(Reg)
if !ok2 || !dstReg.isVec() { if !ok2 || !dstReg.isVec() {
return fmt.Errorf("CMPSD destination must be a vector register") return fmt.Errorf("%s destination must be a vector register", mnem)
} }
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1} i := &instr{prefix: prefix, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil { if err := setRM(i, dstReg, ops[0], 8); err != nil {
return err return err
} }
+3 -3
View File
@@ -16,7 +16,7 @@ import (
"golang.org/x/arch/x86/x86asm" "golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel; // TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel;
@@ -25,7 +25,7 @@ import (
func TestAssembleGoFlacAVX2Kernel(t *testing.T) { func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx2_amd64.s" path := "../../go-libraries/go-flac/avx2_amd64.s"
if _, err := os.Stat(path); err != nil { if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit") t.Skip("go-libraries repository not present next to gasm-sdk")
} }
src, err := os.ReadFile(path) src, err := os.ReadFile(path)
if err != nil { if err != nil {
@@ -86,7 +86,7 @@ func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
func TestAssembleGoFlacAVX512Kernel(t *testing.T) { func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx512_amd64.s" path := "../../go-libraries/go-flac/avx512_amd64.s"
if _, err := os.Stat(path); err != nil { if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit") t.Skip("go-libraries repository not present next to gasm-sdk")
} }
src, err := os.ReadFile(path) src, err := os.ReadFile(path)
if err != nil { if err != nil {
+12 -2
View File
@@ -13,7 +13,7 @@ import (
"strings" "strings"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// The differential kernels for the DATA-path and front-end gaps are kept in // The differential kernels for the DATA-path and front-end gaps are kept in
@@ -23,9 +23,18 @@ import (
// bytes must agree with the relocation sites masked on both sides. // bytes must agree with the relocation sites masked on both sides.
// toolAsmObject assembles path with the installed toolchain's assembler for // toolAsmObject assembles path with the installed toolchain's assembler for
// goarch ("" = the host) and returns the object bytes. // goarch ("" = the host) and returns the object bytes. Every live-oracle
// comparison funnels through here, so this is also where the deliberate-run
// boundary sits: under -short (the push pipeline's mode) the comparisons
// skip, because each spawns a go tool asm subprocess and the small single-
// core runner pays seconds per spawn. The encodings stay pinned by the
// golden-byte tests in every mode; the live oracle runs in the local test
// gate and the dispatched workflows.
func toolAsmObject(t *testing.T, path, goarch string) []byte { func toolAsmObject(t *testing.T, path, goarch string) []byte {
t.Helper() t.Helper()
if testing.Short() {
t.Skip("live go tool asm oracle: skipped in -short mode")
}
goBin, err := exec.LookPath("go") goBin, err := exec.LookPath("go")
if err != nil { if err != nil {
t.Skip("no Go toolchain available") t.Skip("no Go toolchain available")
@@ -130,6 +139,7 @@ func TestDifferentialKernels(t *testing.T) {
{filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false}, {filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false}, {filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false}, {filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "forms_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true}, {filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true},
{filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true}, {filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true},
} { } {
+1 -1
View File
@@ -12,7 +12,7 @@ import (
"strings" "strings"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks: // TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
+37 -5
View File
@@ -9,7 +9,7 @@ import (
"sort" "sort"
"strconv" "strconv"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
) )
// Image is an assembled file: the function bodies laid out in source order, // Image is an assembled file: the function bodies laid out in source order,
@@ -134,7 +134,8 @@ type DataSymbol struct {
Offset int // byte offset within Data Offset int // byte offset within Data
Size int Size int
Static bool // the <> marker: file-local, not exported Static bool // the <> marker: file-local, not exported
Rodata bool // the RODATA flag: read-only data Rodata bool // the RODATA flag: read-only data (implies no pointers)
Noptr bool // the NOPTR flag: data with no pointers, kept out of GC scanning
Dupok bool // the DUPOK flag: duplicate-OK Dupok bool // the DUPOK flag: duplicate-OK
// Relocs carries the symbol-valued DATA initialisers ("DATA s+0(SB)/8, // Relocs carries the symbol-valued DATA initialisers ("DATA s+0(SB)/8,
// $other(SB)"): fields of this symbol's data that hold another symbol's // $other(SB)"): fields of this symbol's data that hold another symbol's
@@ -150,6 +151,18 @@ func (img *Image) Bytes() []byte {
return append(out, img.Data...) return append(out, img.Data...)
} }
// AssembleOption adjusts the file-level assembly context.
type AssembleOption func(*linkInfo)
// WithGOOS selects the target operating system for the forms that depend on
// it, the TLS access shape above all: linux and freebsd take the
// one-instruction form, windows and plan9 keep the two-instruction load.
func WithGOOS(goos string) AssembleOption {
return func(l *linkInfo) {
l.goos = goos
}
}
// AssembleFile assembles every TEXT function of a parsed file and lays out // AssembleFile assembles every TEXT function of a parsed file and lays out
// its static symbols (GLOBL/DATA) in a data section behind the code. Each // its static symbols (GLOBL/DATA) in a data section behind the code. Each
// reference to a file-local static symbol becomes a RIP-relative load whose // reference to a file-local static symbol becomes a RIP-relative load whose
@@ -157,7 +170,7 @@ func (img *Image) Bytes() []byte {
// GLOBL defines is recorded as an external relocation (Externals) with its // GLOBL defines is recorded as an external relocation (Externals) with its
// displacement left zero, the object-file emitters resolve it at link // displacement left zero, the object-file emitters resolve it at link
// time, while the raw image (Bytes) cannot represent it. // time, while the raw image (Bytes) cannot represent it.
func AssembleFile(f *ast.File) (*Image, error) { func AssembleFile(f *ast.File, opts ...AssembleOption) (*Image, error) {
dataSyms, err := collectData(f) dataSyms, err := collectData(f)
if err != nil { if err != nil {
return nil, err return nil, err
@@ -166,7 +179,17 @@ func AssembleFile(f *ast.File) (*Image, error) {
for _, d := range dataSyms { for _, d := range dataSyms {
known[d.name] = true known[d.name] = true
} }
// TEXT symbols are file-level definitions too: a symbol immediate
// ($fn(SB)) may name one, exactly as a data reference names a GLOBL.
for _, d := range f.Decls {
if t, ok := d.(*ast.Text); ok {
known[t.Name.Name] = true
}
}
link := &linkInfo{symbols: known, allowExternal: true} link := &linkInfo{symbols: known, allowExternal: true}
for _, o := range opts {
o(link)
}
poolSeen := map[string]bool{} poolSeen := map[string]bool{}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path} img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
@@ -247,6 +270,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
Size: len(d.buf), Size: len(d.buf),
Static: d.static, Static: d.static,
Rodata: d.rodata, Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok, Dupok: d.dupok,
}) })
img.Data = append(img.Data, d.buf...) img.Data = append(img.Data, d.buf...)
@@ -391,6 +415,7 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
Size: d.size, Size: d.size,
Static: d.static, Static: d.static,
Rodata: d.rodata, Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok, Dupok: d.dupok,
}) })
} }
@@ -463,6 +488,7 @@ func AssembleFileLOONG64(f *ast.File) (*Image, error) {
Size: d.size, Size: d.size,
Static: d.static, Static: d.static,
Rodata: d.rodata, Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok, Dupok: d.dupok,
}) })
} }
@@ -524,6 +550,7 @@ type dataSym struct {
size int size int
static bool static bool
rodata bool rodata bool
noptr bool
dupok bool dupok bool
// relocs are the symbol-valued DATA fields, in declaration order; Off // relocs are the symbol-valued DATA fields, in declaration order; Off
// is relative to the symbol's data start. // is relative to the symbol's data start.
@@ -565,12 +592,14 @@ func collectData(f *ast.File) ([]dataSym, error) {
switch f { switch f {
case "RODATA": case "RODATA":
ds.rodata = true ds.rodata = true
case "NOPTR":
ds.noptr = true
case "DUPOK": case "DUPOK":
ds.dupok = true ds.dupok = true
default: default:
// Legacy numeric flag constants (runtime/textflag.h): // Legacy numeric flag constants (runtime/textflag.h):
// DUPOK is 2, RODATA is 8; combinations arrive as one // DUPOK is 2, RODATA is 8, NOPTR is 16; combinations arrive
// number (e.g. 10 = RODATA|DUPOK). // as one number (e.g. 10 = RODATA|DUPOK).
if n, err := strconv.Atoi(f); err == nil { if n, err := strconv.Atoi(f); err == nil {
if n&2 != 0 { if n&2 != 0 {
ds.dupok = true ds.dupok = true
@@ -578,6 +607,9 @@ func collectData(f *ast.File) ([]dataSym, error) {
if n&8 != 0 { if n&8 != 0 {
ds.rodata = true ds.rodata = true
} }
if n&16 != 0 {
ds.noptr = true
}
} }
} }
} }
+12 -38
View File
@@ -12,7 +12,7 @@ import (
"strings" "strings"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// TestAssembleFileStaticData checks the whole-image layout; code, padding // TestAssembleFileStaticData checks the whole-image layout; code, padding
@@ -60,23 +60,15 @@ DATA small<>+0(SB)/4, $0x1234
} }
} }
// TestAssembleFileErrors checks the static-symbol error paths. // TestAssembleFileErrors checks the static-symbol error paths. A reference
// to a static symbol no GLOBL defines defers to the linker exactly as the
// toolchain does (an external relocation), so it is not an error here.
func TestAssembleFileErrors(t *testing.T) { func TestAssembleFileErrors(t *testing.T) {
cases := []struct { cases := []struct {
name string name string
src string src string
want string // substring of the error want string // substring of the error
}{ }{
{
"undefined symbol",
`
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
VMOVDQU nope<>(SB), X0
RET
`,
"undefined symbol",
},
{ {
"DATA without GLOBL", "DATA without GLOBL",
` `
@@ -369,23 +361,8 @@ func main() {
if err != nil { if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog) t.Fatalf("baseline build: %v\n%s", err, buildLog)
} }
var work, linkLine, asmObj string st := parseBuildLog(t, buildLog, "main_amd64.s")
for line := range strings.SplitSeq(string(buildLog), "\n") { defer os.RemoveAll(st.work)
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || linkLine == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
}
defer os.RemoveAll(work)
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
// Assemble the same source with gasm and substitute the object. // Assemble the same source with gasm and substitute the object.
src, err := os.ReadFile(filepath.Join(dir, "main_amd64.s")) src, err := os.ReadFile(filepath.Join(dir, "main_amd64.s"))
@@ -400,21 +377,18 @@ func main() {
if err != nil { if err != nil {
t.Fatalf("AssembleFile: %v", err) t.Fatalf("AssembleFile: %v", err)
} }
gasmObj, err := img.GOObject("dlink", "main_amd64.s") // The package path is "main": the linker resolves the Go code's
// references against main.<name>, so the object must define the symbols
// under that prefix whatever the module is called.
gasmObj, err := img.GOObject("main", "main_amd64.s")
if err != nil { if err != nil {
t.Fatalf("GOObject: %v", err) t.Fatalf("GOObject: %v", err)
} }
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil { substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"), gasmObj)
t.Fatalf("write gasm object: %v", err)
}
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
// The linked program must run and find the right function behind the // The linked program must run and find the right function behind the
// data word. // data word.
out, err := exec.Command(filepath.Join(dir, "prog")).CombinedOutput() out, err := exec.Command(filepath.Join(dir, "prog2")).CombinedOutput()
if err != nil { if err != nil {
t.Fatalf("linked program failed: %v\n%s", err, out) t.Fatalf("linked program failed: %v\n%s", err, out)
} }
+258 -7
View File
@@ -9,7 +9,7 @@ import (
"strconv" "strconv"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
) )
// assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into // assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into
@@ -371,6 +371,13 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
if mnem == "RET" { if mnem == "RET" {
return len(loong64Return(fi)) return len(loong64Return(fi))
} }
// BYTE lays down one raw byte per operand, a front-end pseudo-op the
// toolchain spells only on x86 but accepts here the same way the arm64
// and riscv64 encoders do (a superset spelling, shippable via the goobj
// path).
if mnem == "BYTE" {
return len(ops)
}
switch mnem { switch mnem {
case "END", "FUNCDATA", "PCDATA": case "END", "FUNCDATA", "PCDATA":
return 0 // bookkeeping statements contribute no bytes return 0 // bookkeeping statements contribute no bytes
@@ -484,6 +491,18 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops)) return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
} }
return l64wordLE(uint32(immFromOperand(ops[0]))), nil return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "BYTE":
// BYTE $b lays down one raw byte per operand, the same front-end
// pseudo-op the arm64 and riscv64 encoders accept.
var out []byte
for _, op := range ops {
b := l64Imm64(op)
if b < 0 || b > 0xFF {
return nil, fmt.Errorf("BYTE: immediate %d does not fit a byte", b)
}
out = append(out, byte(b))
}
return out, nil
case "END", "FUNCDATA", "PCDATA", "GETCALLERPC": case "END", "FUNCDATA", "PCDATA", "GETCALLERPC":
// The assembler's bookkeeping statements. END, FUNCDATA and PCDATA // The assembler's bookkeeping statements. END, FUNCDATA and PCDATA
// contribute no bytes, the same shapes GOARCH=loong64 go tool asm // contribute no bytes, the same shapes GOARCH=loong64 go tool asm
@@ -701,6 +720,35 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
} }
return l64wordLE(l64rr(enc.op, rj, rd)), nil return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Fllsc:
// LLACQ{W,V} (Rj), Rd loads and SCREL{W,V} Rd, (Rj) stores, both
// 2R encodings op | rj<<5 | rd against a zero-offset memory operand
// (the toolchain's C_ZOREG, which rejects any displacement).
rd, rj, off, _, err := l64MemOperands(ops, fi)
if err != nil {
return nil, fmt.Errorf("%s: %w", mnem, err)
}
if off != 0 {
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
}
return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Fscq:
// SCQ first, middle, (base): op | middle<<10 | base<<5 | first,
// against a zero-offset memory operand as with the LL/SC pair.
if len(ops) != 3 || !isMemOperand(ops[2]) || isMemOperand(ops[0]) || isMemOperand(ops[1]) {
return nil, fmt.Errorf("%s expects reg, reg, (reg)", mnem)
}
first, middle := l64Reg(ops[0]), l64Reg(ops[1])
rj, off := l64MemWithFrame(ops[2], fi)
if first < 0 || middle < 0 || rj < 0 {
return nil, fmt.Errorf("%s: invalid register operand", mnem)
}
if off != 0 {
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
}
return l64wordLE(l64rrr(enc.op, middle, rj, first)), nil
case l64Firr: case l64Firr:
// LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd. // LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd.
if len(ops) < 2 || !isImmOperand(ops[0]) { if len(ops) < 2 || !isImmOperand(ops[0]) {
@@ -1229,6 +1277,31 @@ func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off in
return rd, rj, off, load, nil return rd, rj, off, load, nil
} }
// l64ImmMem reads the `$off(rj)` immediate form off an operand's raw text:
// the shared immediate parse reduces it to the bare number and keeps only
// the text as a witness of the base register. ok reports the form was
// found, with the base's register number (or -1 when the name is not a
// general register).
func l64ImmMem(op *ast.Operand) (off int32, base int, ok bool) {
if op.Kind != ast.OpImmediate || !op.Imm.HasVal {
return 0, 0, false
}
raw := strings.ReplaceAll(op.Raw, " ", "")
if !strings.HasPrefix(raw, "$") || !strings.HasSuffix(raw, ")") {
return 0, 0, false
}
open := strings.LastIndexByte(raw, '(')
if open < 2 {
return 0, 0, false
}
base = loong64RegNum(raw[open+1 : len(raw)-1])
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return int32(v), base, base >= 0
}
// ---- the MOV pseudo-instruction ---- // ---- the MOV pseudo-instruction ----
// encodeLOONG64Mov encodes the MOV family, the load/store/immediate // encodeLOONG64Mov encodes the MOV family, the load/store/immediate
@@ -1262,6 +1335,29 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs
} }
return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil
} }
// MOVx $off(rj), rd computes an address: the toolchain's `mov
// $soreg, r` case, a plain addi.d whatever the move's width (both
// MOVW and MOVV $4(R4), R5 encode the same addi.d in its testdata).
// A wider offset materialises in R30 first (lu12i.w + ori + add.d,
// its case 10). The immediate's Raw carries the base register,
// which the shared immediate parse reduces to the bare number.
if off, base, ok := l64ImmMem(src); ok {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $imm(rj): invalid destination register", mnem)
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return nil, fmt.Errorf("%s $imm(rj): illegal combination with an F register destination", mnem)
}
if off >= -2048 && off <= 2047 {
return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(off), base, rd)), nil
}
return l64WordsLE(
l64ir(l64InstrTable["LU12IW"].op, int(off)>>12, 30),
l64irr(l64DualTable["OR"].imm, int(off)&0xFFF, 30, 30),
l64rrr(l64DualTable["ADDV"].rrr, 30, base, rd),
), nil
}
rd := l64Reg(dst) rd := l64Reg(dst)
if rd < 0 { if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem) return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
@@ -1356,6 +1452,14 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
return 8 // pcalau12i + addi.d return 8 // pcalau12i + addi.d
} }
// The $off(rj) address immediate: addi.d in the 12-bit window,
// lu12i.w + ori + add.d beyond it (the toolchain's case 10).
if off, _, ok := l64ImmMem(src); ok {
if off >= -2048 && off <= 2047 {
return 4
}
return 12
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP { if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid) return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid)
} }
@@ -1868,6 +1972,19 @@ func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
return l64Mem(op) return l64Mem(op)
} }
// l64VmovqMem resolves a VMOVQ/XVMOVQ memory operand. The toolchain's
// vector table falls back to the zero register as the FP-relative base
// (`VMOVQ V2, y+16(FP)` stores through R0 while MOVW reads the same operand
// through R3), so the vector moves keep the resolved offset but the zero
// base, exactly as `go tool asm` emits them.
func l64VmovqMem(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
rj, off = l64MemWithFrame(op, fi)
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" {
rj = 0
}
return rj, off
}
// l64MemOffset returns the resolved byte offset of a memory operand. // l64MemOffset returns the resolved byte offset of a memory operand.
func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 { func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 {
_, off := l64MemWithFrame(op, fi) _, off := l64MemWithFrame(op, fi)
@@ -1892,7 +2009,7 @@ func l64Label(op *ast.Operand) string {
type l64VecOperand struct { type l64VecOperand struct {
num int // 5-bit register number num int // 5-bit register number
lasx bool // X bank (LASX) rather than V (LSX) lasx bool // X bank (LASX) rather than V (LSX)
width byte // suffix width letter (B/H/W/V), 0 on a bare register width byte // suffix width letter (B/H/W/V/Q), 0 on a bare register
lanes int // lane count of a width suffix (B16 → 16) lanes int // lane count of a width suffix (B16 → 16)
elem int // element index of a .T[i] suffix elem int // element index of a .T[i] suffix
hasEl bool // the suffix names an element (.T[i]) hasEl bool // the suffix names an element (.T[i])
@@ -1932,7 +2049,7 @@ func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) {
} }
i++ i++
w := name[i] w := name[i]
if w != 'B' && w != 'H' && w != 'W' && w != 'V' { if w != 'B' && w != 'H' && w != 'W' && w != 'V' && w != 'Q' {
return v, false return v, false
} }
v.width, v.hasSuf = w, true v.width, v.hasSuf = w, true
@@ -2174,6 +2291,10 @@ func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]
// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register) // VMOVQ rj, vd.T vreplgr2vr (duplicate a general register)
// VMOVQ vj.T[i], rd vpickve2gr (extract one element) // VMOVQ vj.T[i], rd vpickve2gr (extract one element)
// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element) // VMOVQ rj, vd.T[i] vinsgr2vr (insert one element)
// VMOVQ vj.T[i], vd.T vreplvei (broadcast one element, LSX)
// XVMOVQ xj, xd.T xvreplve0 (broadcast element zero, LASX)
// XVMOVQ xj, xd.T[i] xvinsve0 (insert element zero, LASX)
// XVMOVQ xj.T[i], xd xvpickve (extract one element, LASX)
func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) { func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) {
enc := l64VmovqTable[lasx] enc := l64VmovqTable[lasx]
bank := "V" bank := "V"
@@ -2200,6 +2321,118 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
return loong64RegNum(name), nil return loong64RegNum(name), nil
} }
// Element broadcast: VMOVQ vj.T[i], vd.T (vreplvei.{b,h,w,d}), the
// source element width matching the destination arrangement. An LSX-only
// form: the toolchain's table gives vreplvei no LASX counterpart.
if srcVec && dstVec && src.hasEl && dst.hasSuf && !dst.hasEl {
if lasx || src.lasx || dst.lasx {
return nil, fmt.Errorf("VMOVQ: vreplvei has no %s-bank form", bank)
}
if src.unsig {
return nil, fmt.Errorf("VMOVQ: vreplvei takes no unsigned element suffix")
}
if src.width != dst.width {
return nil, fmt.Errorf("VMOVQ: element width does not match arrangement %q", ops[1].Raw)
}
if _, ok := l64VecSuffixWidth(false, dst); !ok {
return nil, fmt.Errorf("VMOVQ: invalid arrangement %q", ops[1].Raw)
}
var op uint32
limit := 0
switch src.width {
case 'B':
op, limit = enc.rveiB, 15
case 'H':
op, limit = enc.rveiH, 7
case 'W':
op, limit = enc.rveiW, 3
default:
op, limit = enc.rveiD, 1
}
if src.elem > limit {
return nil, fmt.Errorf("VMOVQ: element index %d out of range [0, %d]", src.elem, limit)
}
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Broadcast of element zero: XVMOVQ xj, xd.T (xvreplve0.{b,h,w,d,q}),
// a bare X source into an arranged X destination. LASX only.
if srcVec && dstVec && !src.hasSuf && dst.hasSuf && !dst.hasEl {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvreplve0 is the %s-bank form alone", bank)
}
var op uint32
switch dst.width {
case 'B':
if dst.lanes != 32 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0B
case 'H':
if dst.lanes != 16 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0H
case 'W':
if dst.lanes != 8 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0W
case 'V':
if dst.lanes != 4 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0D
case 'Q':
if dst.lanes != 2 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0Q
default:
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
return l64wordLE(op | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Insert of element zero: XVMOVQ xj, xd.T[i] (xvinsve0.{w,d}), a bare X
// source into one word or double-word lane. LASX only.
if srcVec && dstVec && !src.hasSuf && dst.hasEl {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvinsve0 is the %s-bank form alone", bank)
}
op, limit := enc.xinsW, 7
if dst.width != 'W' {
op, limit = enc.xinsD, 3
if dst.width != 'V' {
return nil, fmt.Errorf("XVMOVQ: xvinsve0 takes word or double-word lanes, got %q", ops[1].Raw)
}
}
if dst.elem > limit {
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", dst.elem, limit)
}
return l64wordLE(op | uint32(dst.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Element extract into a vector register: XVMOVQ xj.T[i], xd
// (xvpickve.{w,d}), one word or double-word lane out to a bare X
// register. LASX only.
if srcVec && src.hasEl && dstVec && !dst.hasSuf {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvpickve is the %s-bank form alone", bank)
}
op, limit := enc.xpickW, 7
if src.width != 'W' {
op, limit = enc.xpickD, 3
if src.width != 'V' {
return nil, fmt.Errorf("XVMOVQ: xvpickve takes word or double-word lanes, got %q", ops[0].Raw)
}
}
if src.elem > limit {
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", src.elem, limit)
}
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant), // Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant),
// both operands bare registers of the same bank. // both operands bare registers of the same bank.
if srcVec && dstVec { if srcVec && dstVec {
@@ -2224,7 +2457,7 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
} }
return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil
} }
rj, off := l64MemWithFrame(ops[1], fi) rj, off := l64VmovqMem(ops[1], fi)
if rj < 0 || off < -2048 || off > 2047 { if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]") return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]")
} }
@@ -2247,9 +2480,9 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
} }
return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil
} }
rj, off := l64MemWithFrame(ops[0], fi) rj, off := l64VmovqMem(ops[0], fi)
if rj < 0 || off < -2048 || off > 2047 { if rj < 0 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]") return nil, fmt.Errorf("VMOVQ: invalid load operand")
} }
op := enc.ld op := enc.ld
if dst.hasSuf { if dst.hasSuf {
@@ -2257,16 +2490,34 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
if !ok { if !ok {
return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw) return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw)
} }
// vldrepl keeps the byte offset raw for bytes and scales it by
// the element width for the wider forms, the immediate field
// shrinking a bit per scale exactly as the toolchain encodes it
// (the field mask keeps the two's complement inside its width).
scale, mask, lo, hi := 1, int32(0xFFF), -2048, 2047
switch w { switch w {
case 0: case 0:
op = enc.replB op = enc.replB
case 1: case 1:
op = enc.replH op = enc.replH
scale, mask, lo, hi = 2, 0x7FF, -1024, 1023
case 2: case 2:
op = enc.replW op = enc.replW
scale, mask, lo, hi = 4, 0x3FF, -512, 511
default: default:
op = enc.replD op = enc.replD
scale, mask, lo, hi = 8, 0x1FF, -256, 255
} }
if off%int32(scale) != 0 {
return nil, fmt.Errorf("VMOVQ: offset %d must be a multiple of %d", off, scale)
}
off /= int32(scale)
if off < int32(lo) || off > int32(hi) {
return nil, fmt.Errorf("VMOVQ: offset out of range [%d, %d]", lo*scale, hi*scale)
}
off &= mask
} else if off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
} }
return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil
} }
+32 -5
View File
@@ -279,6 +279,8 @@ const (
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd
l64Fllsc // acquire/release LL/SC (2R against a zero-offset memory operand)
l64Fscq // sc.q: op | middle<<10 | base<<5 | first against a zero-offset memory operand
) )
// l64Enc is one instruction's encoding: its bit layout (format) and the // l64Enc is one instruction's encoding: its bit layout (format) and the
@@ -341,8 +343,9 @@ var l64Vec2R = map[string]bool{}
// such as vshuf.b). // such as vshuf.b).
var l64Vec4R = map[string]bool{} var l64Vec4R = map[string]bool{}
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit // l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants, each pre-shifted to
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels. // its exact bit range, read off `go tool objdump` of GOARCH=loong64
// `go tool asm` kernels and the toolchain's specialLsxMovInst table.
type l64VmovqEnc struct { type l64VmovqEnc struct {
ld, st, ldx, stx uint32 // plain and indexed load/store ld, st, ldx, stx uint32 // plain and indexed load/store
replB, replH, replW, replD uint32 // vldrepl: load and replicate element replB, replH, replW, replD uint32 // vldrepl: load and replicate element
@@ -350,6 +353,11 @@ type l64VmovqEnc struct {
ins uint32 // vinsgr2vr element insert ins uint32 // vinsgr2vr element insert
dup uint32 // vreplgr2vr duplicate (width in [11:10]) dup uint32 // vreplgr2vr duplicate (width in [11:10])
move uint32 // vori.b/xvori.b $0 register move move uint32 // vori.b/xvori.b $0 register move
rveiB, rveiH, rveiW, rveiD uint32 // vreplvei: broadcast one element (LSX)
rve0B, rve0H, rve0W uint32 // xvreplve0 broadcast of element zero (LASX)
rve0D, rve0Q uint32 // xvreplve0.{d,q}, ditto
xinsW, xinsD uint32 // xvinsve0: insert element zero (LASX)
xpickW, xpickD uint32 // xvpickve: extract element (LASX)
} }
var l64VmovqTable = map[bool]l64VmovqEnc{ var l64VmovqTable = map[bool]l64VmovqEnc{
@@ -358,12 +366,17 @@ var l64VmovqTable = map[bool]l64VmovqEnc{
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15, replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15, pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15, ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
rveiB: 0x01CBDE << 14, rveiH: 0x0397BE << 13, rveiW: 0x072F7E << 12, rveiD: 0x0E5EFE << 11,
}, },
true: { // XVMOVQ, the LASX (X) bank true: { // XVMOVQ, the LASX (X) bank
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15, ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15, replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15, pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15, ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
rve0B: 0x1DC1C0 << 10, rve0H: 0x1DC1E0 << 10, rve0W: 0x1DC1F0 << 10,
rve0D: 0x1DC1F8 << 10, rve0Q: 0x1DC1FC << 10,
xinsW: 0x03B7FE << 13, xinsD: 0x076FFE << 12,
xpickW: 0x03B81E << 13, xpickD: 0x07703E << 12,
}, },
} }
@@ -373,7 +386,7 @@ func init() {
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15, "ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15, "SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
"SGT": 0x24 << 15, "SGTU": 0x25 << 15, "SGT": 0x24 << 15, "SGTU": 0x25 << 15,
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15, "MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15,
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15, "NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
"ORN": 0x2c << 15, "ANDN": 0x2d << 15, "ORN": 0x2c << 15, "ANDN": 0x2d << 15,
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15, "SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
@@ -467,6 +480,20 @@ func init() {
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10} l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10} l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
// Acquire/release LL/SC (2R against a zero-offset memory operand):
// LLACQV (Rj), Rd loads, SCRELV Rd, (Rj) stores, both encoding
// op | rj<<5 | rd. Opcodes from cmd/internal/obj/loong64/instOp.go
// (ll.acq.{w,d}, sc.rel.{w,d}).
l64InstrTable["LLACQW"] = l64Enc{format: l64Fllsc, op: 0x0E15E0 << 10}
l64InstrTable["SCRELW"] = l64Enc{format: l64Fllsc, op: 0x0E15E1 << 10}
l64InstrTable["LLACQV"] = l64Enc{format: l64Fllsc, op: 0x0E15E2 << 10}
l64InstrTable["SCRELV"] = l64Enc{format: l64Fllsc, op: 0x0E15E3 << 10}
// SCQ (sc.q first, middle, (base)) keeps its own operand order: the
// encoding is op | middle<<10 | base<<5 | first, the memory operand's
// base in the rj field, not the toolchain's generic 3R layout.
l64InstrTable["SCQ"] = l64Enc{format: l64Fscq, op: 0x070AE << 15}
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12), // The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
// selected by the operand kind; the shift mnemonics pair the 3R form // selected by the operand kind; the shift mnemonics pair the 3R form
// with a 5/6-bit shift immediate. // with a 5/6-bit shift immediate.
@@ -868,7 +895,7 @@ func init() {
"VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF}, "VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF},
"XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF}, "XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF},
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F}, "VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F}, "XVSEQB": {0xED00 << 15, true, -16, 15, 0, 0x1F},
// vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a // vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a
// 7-bit field, but the toolchain range-checks it down to si5 as well // 7-bit field, but the toolchain range-checks it down to si5 as well
// (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64). // (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64).
@@ -877,7 +904,7 @@ func init() {
"VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F}, "VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F},
"XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F}, "XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F},
"VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F}, "VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F},
"XVSEQV": {0xE903 << 15, true, -16, 15, 0, 0x7F}, "XVSEQV": {0xED03 << 15, true, -16, 15, 0, 0x7F},
// vslti compares against a signed (or, in the U spellings, unsigned) // vslti compares against a signed (or, in the U spellings, unsigned)
// si5/ui5 constant. // si5/ui5 constant.
"VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F}, "VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F},
+229 -2
View File
@@ -8,8 +8,8 @@ import (
"encoding/binary" "encoding/binary"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// firstTextLOONG64 parses assembly source and returns the first TEXT body. // firstTextLOONG64 parses assembly source and returns the first TEXT body.
@@ -831,3 +831,230 @@ TEXT ·atoms(SB), NOSPLIT, $0
0x4C000020, 0x4C000020,
) )
} }
// TestLOONG64_llacqScrel pins the acquire/release LL/SC pair. The oracle
// words come from GOARCH=loong64 go tool objdump and the toolchain's own
// loong64enc1.s golden bytes.
func TestLOONG64_llacqScrel(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·llsc(SB), NOSPLIT, $0
LLACQW (R5), R4
LLACQV (R5), R4
SCRELW R4, (R6)
SCRELV R4, (R6)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x385780A4, // ll.acq.w r4, r5
0x385788A4, // ll.acq.d r4, r5
0x385784C4, // sc.rel.w r4, r6
0x38578CC4, // sc.rel.d r4, r6
0x4C000020,
)
// The toolchain accepts the zero-offset memory form alone.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
LLACQW 4(R5), R4
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
SCRELV R4, 8(R6)
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_vmovqSuffixed pins the element-broadcast and element-move
// VMOVQ/XVMOVQ forms, with the oracle words lifted verbatim from the
// toolchain's loong64enc1.s.
func TestLOONG64_vmovqSuffixed(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·vmovq(SB), NOSPLIT, $0
VMOVQ V1.B[3], V9.B16
VMOVQ V2.H[2], V8.H8
VMOVQ V3.W[1], V7.W4
VMOVQ V4.V[0], V6.V2
XVMOVQ X0, X31.B32
XVMOVQ X1, X30.H16
XVMOVQ X2, X29.W8
XVMOVQ X3, X28.V4
XVMOVQ X3, X27.Q2
XVMOVQ X0, X31.W[7]
XVMOVQ X1, X29.W[0]
XVMOVQ X3, X28.V[3]
XVMOVQ X4, X27.V[0]
XVMOVQ X31.W[7], X0
XVMOVQ X29.W[0], X1
XVMOVQ X28.V[3], X8
XVMOVQ X27.V[0], X9
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x72F78C29, // vreplvei.b v9, v1, 3
0x72F7C848, // vreplvei.h v8, v2, 2
0x72F7E467, // vreplvei.w v7, v3, 1
0x72F7F086, // vreplvei.d v6, v4, 0
0x7707001F, // xvreplve0.b x31, x0
0x7707803E, // xvreplve0.h x30, x1
0x7707C05D, // xvreplve0.w x29, x2
0x7707E07C, // xvreplve0.d x28, x3
0x7707F07B, // xvreplve0.q x27, x3
0x76FFDC1F, // xvinsve0.w x31, x0, 7
0x76FFC03D, // xvinsve0.w x29, x1, 0
0x76FFEC7C, // xvinsve0.d x28, x3, 3
0x76FFE09B, // xvinsve0.d x27, x4, 0
0x7703DFE0, // xvpickve.w x0, x31, 7
0x7703C3A1, // xvpickve.w x1, x29, 0
0x7703EF88, // xvpickve.d x8, x28, 3
0x7703E369, // xvpickve.d x9, x27, 0
0x4C000020,
)
// The rejected shapes: a width mismatch between the element and the
// arrangement, an element index past the lane count, a wrong-bank
// vreplvei and an arrangement the LASX bank does not spell.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ V1.H[3], V9.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ V1.B[16], V9.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X1.B[3], X9.B32
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X0, X31.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X0, X31.W[8]
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_parityFixes pins the operand forms whose encodings were found
// diverging from the toolchain by the loong64enc1.s differential: the
// $off(reg) address immediate (addi.d), the SCQ operand order, the scaled
// vldrepl offsets (with their field masks), the XVSEQB/XVSEQV immediate
// opcodes and the zero-register base the toolchain gives FP-relative
// VMOVQ/XVMOVQ memory operands. Golden words from loong64enc1.s.
func TestLOONG64_parityFixes(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·parity(SB), NOSPLIT, $0-32
MOVW $4(R4), R5
MOVV $4(R4), R5
MOVW $65536(R4), R5
MOVW $-4096(R4), R5
SCQ R4, R5, (R6)
VMOVQ 2(R4), V1.H8
VMOVQ -6(R4), V1.H8
VMOVQ -12(R4), V2.W4
VMOVQ -16(R4), V3.V2
XVMOVQ -10(R4), X1.H16
XVSEQB $0, X2, X4
XVSEQH $3, X2, X4
XVSEQW $12, X2, X4
XVSEQV $15, X2, X4
XVSEQV $-15, X2, X4
VMOVQ V2, y+16(FP)
VMOVQ y+16(FP), V2
VMOVQ V2, x+2030(FP)
XVMOVQ X6, y+16(FP)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x02C01085, // addi.d $4, r4, r5
0x02C01085, // addi.d $4, r4, r5 (MOVW keeps the 64-bit addi.d)
0x1400021E, // lu12i.w $16, r30
0x038003DE, // ori $0, r30, r30
0x0010F885, // add.d r5, r4, r30
0x15FFFFFE, // lu12i.w $-1, r30
0x038003DE, // ori $0, r30, r30
0x0010F885, // add.d r5, r4, r30
0x385714C4, // sc.q r4, r5, (r6): middle<<10 | base<<5 | first
0x30400481, // vldrepl.h v1, 2(r4)
0x305FF481, // vldrepl.h v1, -6(r4)
0x302FF482, // vldrepl.w v2, -12(r4)
0x3017F883, // vldrepl.d v3, -16(r4)
0x325FEC81, // xvldrepl.h x1, -10(r4)
0x76800044, // xvseqi.b x4, x2, 0
0x76808C44, // xvseqi.h x4, x2, 3
0x76813044, // xvseqi.w x4, x2, 12
0x7681BC44, // xvseqi.d x4, x2, 15
0x7681C444, // xvseqi.d x4, x2, -15
0x2C406002, // vst v2, 24(r0): FP-relative keeps the zero base
0x2C006002, // vld v2, 24(r0)
0x2C5FD802, // vst v2, 2038(r0)
0x2CC06006, // xvst x6, 24(r0)
0x4C000020,
)
// Misaligned vldrepl offsets are rejected, as the toolchain does.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ 3(R4), V1.H8
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
MOVW $4(R4), F1
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_bytePseudo pins the BYTE literal-data pseudo-op, which the
// loong64 toolchain does not spell but the arm64 and riscv64 encoders of
// this package already accept for byte-exact data layout (a superset
// spelling, shippable via the goobj path).
func TestLOONG64_bytePseudo(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·bytes(SB), NOSPLIT, $0
BYTE $2
BYTE $1; BYTE $0
BYTE $255
RET
`)
code := assembleLOONG64Helper(t, fn)
// Four literal bytes, then RET (jirl r0, r1, 0); the trailing bytes pad
// the final word the way any sub-word tail does.
want := []byte{2, 1, 0, 0xFF, 0x20, 0x00, 0x00, 0x4C}
if !bytes.Equal(code[:len(want)], want) {
t.Errorf("bytes = % x, want % x", code, want)
}
for _, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
BYTE $256
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
BYTE $-1
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("%q: expected an error, got none", src)
}
}
}
+1 -1
View File
@@ -6,7 +6,7 @@ package asm
import ( import (
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
) )
// Loong64 frame mapping, matching the Go toolchain's loong64 backend. // Loong64 frame mapping, matching the Go toolchain's loong64 backend.
+1 -1
View File
@@ -7,7 +7,7 @@ import (
"bytes" "bytes"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// TestLOONG64_sys exercises the no-operand system instructions and the // TestLOONG64_sys exercises the no-operand system instructions and the
+1 -1
View File
@@ -6,7 +6,7 @@ package asm
import ( import (
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// TestLOONG64RelocOffsetsIncludePrologue pins the function-relative // TestLOONG64RelocOffsetsIncludePrologue pins the function-relative
+24
View File
@@ -36,6 +36,29 @@ type FloatImm struct {
func (FloatImm) isOperand() {} func (FloatImm) isOperand() {}
// TLSMem is a thread-local access, the source form off(base)(TLS*1) with the
// base dropped: the toolchain's one-instruction TLS rewrite assembles it as
// the segment-prefixed absolute whose disp32 carries an R_TLS_LE patch site
// (the linker fills the TLS slot offset).
type TLSMem struct {
Disp int64
Size int
Seg byte // the segment override: FS (0x64) or GS (0x65) on windows
}
func (TLSMem) isOperand() {}
// SegAbs is a segment-absolute access, 0x30(GS): the segment override
// prefixes a disp32 absolute reference with no relocation. The base
// register spellings GS and FS produce it.
type SegAbs struct {
Disp int64
Size int
Seg byte // 0x64 FS, 0x65 GS
}
func (SegAbs) isOperand() {}
// Mem is a memory operand of the form disp(base)(index*scale). // Mem is a memory operand of the form disp(base)(index*scale).
type Mem struct { type Mem struct {
Base Reg Base Reg
@@ -45,6 +68,7 @@ type Mem struct {
Size int // operand width in bytes Size int // operand width in bytes
HasBase bool HasBase bool
HasIndex bool HasIndex bool
Seg byte // segment override prefix (0x64 FS, 0x65 GS); 0 = none
} }
func (Mem) isOperand() {} func (Mem) isOperand() {}
+30 -1
View File
@@ -17,13 +17,28 @@ import "strings"
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which // size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share // occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one. The mask flag marks the AVX-512 opmask // those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0-K7, the fp flag the x87 stack registers F0-F7. // registers K0-K7, the fp flag the x87 stack registers F0-F7, the mmx flag the
// MMX registers M0-M7, the seg field a bare segment register (FS, GS) and the
// ctl field the control and debug registers, whose number rides an
// instruction's reg field rather than r/m.
type Reg struct { type Reg struct {
idx int idx int
size int // informational width implied by the name; the mnemonic decides size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH high bool // AH/CH/DH/BH
mask bool // K0-K7 opmask register mask bool // K0-K7 opmask register
fp bool // F0-F7 x87 stack register fp bool // F0-F7 x87 stack register
mmx bool // M0-M7 MMX register
seg int // segment register number plus one (ES=1..GS=6); 0 = not one
ctl byte // 0 none, 1 CRn control register, 2 DRn debug register
}
// segNumber returns the segment register number (ES=0..GS=5) when r names a
// bare segment register.
func (r Reg) segNumber() (int, bool) {
if r.seg == 0 {
return 0, false
}
return r.seg - 1, true
} }
// Index returns the register number (0-15 for GPRs, 0-31 for vectors). // Index returns the register number (0-15 for GPRs, 0-31 for vectors).
@@ -149,6 +164,20 @@ func buildRegByName() map[string]Reg {
for i := 0; i <= 7; i++ { for i := 0; i <= 7; i++ {
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true} m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
} }
// MMX: M0..M7.
for i := 0; i <= 7; i++ {
m["M"+itoa(i)] = Reg{idx: i, size: 8, mmx: true}
}
// Bare segment registers: ES, CS, SS, DS, FS, GS (the memory-base and
// index spellings of FS and GS are handled before register lookup).
for i, n := range []string{"ES", "CS", "SS", "DS", "FS", "GS"} {
m[n] = Reg{idx: i, size: 2, seg: i + 1}
}
// Control and debug registers: CR0..CR15, DR0..DR15.
for i := 0; i <= 15; i++ {
m["CR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 1}
m["DR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 2}
}
return m return m
} }
+1 -1
View File
@@ -10,7 +10,7 @@ import (
"slices" "slices"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
) )
// assembleRISCV assembles a RISC-V TEXT function body into machine code. // assembleRISCV assembles a RISC-V TEXT function body into machine code.
+2 -2
View File
@@ -10,8 +10,8 @@ import (
"strings" "strings"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// firstTextRISCV parses assembly source and returns the first TEXT function body. // firstTextRISCV parses assembly source and returns the first TEXT function body.
+1 -1
View File
@@ -7,7 +7,7 @@ import (
"fmt" "fmt"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
) )
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend. // RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
+1 -1
View File
@@ -6,7 +6,7 @@ package asm
import ( import (
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// TestRISCVFrameSpadjAndLines checks that a framed function records its // TestRISCVFrameSpadjAndLines checks that a framed function records its
+1 -1
View File
@@ -13,7 +13,7 @@ import (
"strings" "strings"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL // TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
+123 -4
View File
@@ -76,6 +76,10 @@ const (
// carries a vector length, so the register the L'L field follows is the // carries a vector length, so the register the L'L field follows is the
// XMM source. // XMM source.
vexExtractGPR vexExtractGPR
// vexBlend4 is the four-operand variable blend `OP mask, src2, src1,
// dst` (VPBLENDVB): ModRM.reg = dst (op3), VEX.vvvv = src1 (op2),
// ModRM.rm = src2 (op1) and the mask register in the /is4 byte (op0).
vexBlend4
) )
// vexSpec describes one VEX instruction's encoding parameters. // vexSpec describes one VEX instruction's encoding parameters.
@@ -202,8 +206,28 @@ var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8). // VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM}, "VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8). // VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8), and its
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM}, // double twin under op 01; the in-lane permutes under 04/05.
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM},
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM},
"VPERMILPD": {3, 0x05, 0, 1, -1, vexImmRM},
// VEX.66.0F3A.W0, the immediate-controlled AVX tail: the rounding
// pair, the AES key assistant and the string compares.
"VROUNDPD": {3, 0x09, 0, 1, -1, vexImmRM},
"VROUNDPS": {3, 0x08, 0, 1, -1, vexImmRM},
"VAESKEYGENASSIST": {3, 0xDF, 0, 1, -1, vexImmRM},
"VPCMPESTRI": {3, 0x61, 0, 1, -1, vexImmRM},
"VPCMPESTRM": {3, 0x60, 0, 1, -1, vexImmRM},
"VPCMPISTRI": {3, 0x63, 0, 1, -1, vexImmRM},
"VPCMPISTRM": {3, 0x62, 0, 1, -1, vexImmRM},
// VEX.128.66.0F3A.W0, the scalar lane extract to a GPR or memory
// (reg = the XMM source, r/m = the destination).
"VEXTRACTPS": {3, 0x17, 0, 1, -1, vexExtractGPR},
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR},
// VEX.128.66.0F3A.W0, the four-operand variable blend with its mask
// register in the /is4 byte.
"VPBLENDVB": {3, 0x4C, 0, 1, -1, vexBlend4},
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2, // VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// imm8). // imm8).
@@ -524,8 +548,16 @@ func isVex(mnemUpper string) bool {
if _, ok := vexTable[mnemUpper]; ok { if _, ok := vexTable[mnemUpper]; ok {
return true return true
} }
_, ok := vexMoveTable[mnemUpper] if _, ok := vexMoveTable[mnemUpper]; ok {
return ok return true
}
// The dual-shape moves (VMOVHPD/VMOVLPD) pick their VEX form by operand
// count in encodeVex.
switch mnemUpper {
case "VMOVHPD", "VMOVLPD":
return true
}
return false
} }
// encodeVex encodes a VEX instruction with operands in Plan 9 order. // encodeVex encodes a VEX instruction with operands in Plan 9 order.
@@ -559,6 +591,22 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops) return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
} }
} }
// The high/low double moves split by operand count: three operands
// load-and-insert (mem, src, dst, an NDS form), two store (xmm, m64,
// the reversed store layout).
if mnemUpper == "VMOVHPD" || mnemUpper == "VMOVLPD" {
loadOp, storeOp := byte(0x16), byte(0x17)
if mnemUpper == "VMOVLPD" {
loadOp, storeOp = 0x12, 0x13
}
switch len(ops) {
case 3:
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: loadOp, w: 0, pp: 1, opdigit: -1, form: vexNDS3}, ops)
case 2:
return e.encodeVexRMRev(vexSpec{mapSel: 1, opcode: storeOp, w: 0, pp: 1, opdigit: -1, form: vexRMRev}, ops)
}
return fmt.Errorf("%s expects 2 or 3 operands, got %d", mnemUpper, len(ops))
}
spec := vexTable[mnemUpper] spec := vexTable[mnemUpper]
switch spec.form { switch spec.form {
case vexNDS3: case vexNDS3:
@@ -573,6 +621,10 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3Imm(spec, ops) return e.encodeVexNDS3Imm(spec, ops)
case vexExtract: case vexExtract:
return e.encodeVexExtract(spec, ops) return e.encodeVexExtract(spec, ops)
case vexExtractGPR:
return e.encodeVexExtractGPR(spec, ops)
case vexBlend4:
return e.encodeVexBlend4(spec, ops)
case vexRMSrcLen: case vexRMSrcLen:
return e.encodeVexRMSrcLen(mnemUpper, spec, ops) return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero: case vexZero:
@@ -964,6 +1016,73 @@ func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1]) return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
} }
// encodeVexExtractGPR encodes the lane extract to a general-purpose register
// or memory (VEXTRACTPS): OP $imm, xsrc, gpr/mem with the XMM source in
// ModRM.reg and the destination in r/m, L = 0.
func (e *enc) encodeVexExtractGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() || srcReg.size != 16 {
return fmt.Errorf("extract source must be an XMM register")
}
if _, isReg := dst.(Reg); !isReg && !memOperand(dst) {
return fmt.Errorf("extract destination must be a register or memory")
}
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, 0, srcReg.idx&7, rBit, 15, dst); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexBlend4 encodes the four-operand variable blend (VPBLENDVB):
// OP mask, src2, src1, dst with ModRM.reg = dst, VEX.vvvv = src1, r/m =
// src2 and the mask XMM register in the trailing /is4 byte.
func (e *enc) encodeVexBlend4(spec vexSpec, ops []Operand) error {
if len(ops) != 4 {
return fmt.Errorf("blend expects 4 operands (mask, src2, src1, dst), got %d", len(ops))
}
mask, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
maskReg, ok := mask.(Reg)
if !ok || !maskReg.isVec() || maskReg.size != 16 {
return fmt.Errorf("blend mask must be an XMM register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("blend second source must be a vector register")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("blend destination must be a vector register")
}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, dstReg.vecLenBit(), dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2); err != nil {
return err
}
// The /is4 byte names the mask register: bits [3:0] its low nibble,
// bit 7 the fourth register bit (X8-X15).
e.out = append(e.out, byte(maskReg.idx&7)|byte((maskReg.idx&8)<<4))
return nil
}
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ, // encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector // VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
// move uses the store-form layout (reg = source, rm = destination), matching // move uses the store-form layout (reg = source, rm = destination), matching
+1 -1
View File
@@ -8,7 +8,7 @@
// to the arch package; the AST records syntax only. // to the arch package; the AST records syntax only.
package ast package ast
import "sourcedock.dev/petrbalvin/gasm-devkit/token" import "sourcedock.dev/petrbalvin/gasm-sdk/token"
// File is the parsed representation of one .s source file. // File is the parsed representation of one .s source file.
type File struct { type File struct {
+1 -1
View File
@@ -6,7 +6,7 @@ package ast
import ( import (
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/token" "sourcedock.dev/petrbalvin/gasm-sdk/token"
) )
func pos(line, col int) token.Position { return token.Position{Line: line, Column: col} } func pos(line, col int) token.Position { return token.Position{Line: line, Column: col} }
+1 -1
View File
@@ -17,7 +17,7 @@ import (
"regexp" "regexp"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch" "sourcedock.dev/petrbalvin/gasm-sdk/arch"
) )
// go_asm.h is the header the Go compiler writes for every package that // go_asm.h is the header the Go compiler writes for every package that
+57 -3
View File
@@ -5,6 +5,7 @@ package main
import ( import (
"os" "os"
"os/exec"
"path/filepath" "path/filepath"
"strings" "strings"
"testing" "testing"
@@ -392,14 +393,67 @@ func TestRunCorpusAuditGOOS(t *testing.T) {
} }
} }
// TestRunCorpusAuditBuildConstraint covers the //go:build classification end
// to end: a generic-named file whose constraint admits one target is
// attempted there alone (cpu_x86.s on amd64), and a file whose constraint
// admits none of the four targets is never attempted (the msan and
// goexperiment trees).
func TestRunCorpusAuditBuildConstraint(t *testing.T) {
dir := t.TempDir()
write := func(name, src string) {
t.Helper()
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
t.Fatal(err)
}
}
write("x86.s", "//go:build 386 || amd64\n\nTEXT \xc2\xb7f(SB), NOSPLIT, $0\n\tRET\n")
write("racey.s", "//go:build race\n\nTEXT \xc2\xb7r(SB), NOSPLIT, $0\n\tRET\n")
write("plain.s", "TEXT \xc2\xb7p(SB), NOSPLIT, $0\n\tRET\n")
stats, err := runCorpusAudit(dir, nil)
if err != nil {
t.Fatalf("runCorpusAudit: %v", err)
}
tally := func(name string) *corpusTally {
for i, tg := range stats.targets {
if tg.name == name {
return stats.tallies[i]
}
}
t.Fatalf("no tally for %s", name)
return nil
}
if stats.narrowed != 1 || stats.excluded != 1 || stats.generic != 1 {
t.Errorf("buckets = narrowed %d, excluded %d, generic %d; want 1, 1, 1", stats.narrowed, stats.excluded, stats.generic)
}
if a := tally("amd64"); a.attempted != 2 || a.assembled != 2 {
t.Errorf("amd64 = %d/%d, want 2/2 (x86.s and plain.s)", a.assembled, a.attempted)
}
for _, name := range []string{"arm64", "riscv64", "loong64"} {
if a := tally(name); a.attempted != 1 || a.assembled != 1 {
t.Errorf("%s = %d/%d, want 1/1 (plain.s only)", name, a.assembled, a.attempted)
}
}
if stats.full != 2 {
t.Errorf("full = %d, want 2 (x86.s over its one target, plain.s over all four)", stats.full)
}
}
// TestGenerateGoAsmHeaderRuntime pins the generator against the real thing: // TestGenerateGoAsmHeaderRuntime pins the generator against the real thing:
// the runtime package, whose header the toolchain's own -asmhdr output was // the runtime package of the ambient toolchain, whose header the toolchain's
// sampled from. Skipped in short mode: it type-checks the whole package. // own -asmhdr output was sampled from. Skipped in short mode: it type-checks
// the whole package. The GOROOT comes from the go command itself, so the
// test follows whatever toolchain the host provides.
func TestGenerateGoAsmHeaderRuntime(t *testing.T) { func TestGenerateGoAsmHeaderRuntime(t *testing.T) {
if testing.Short() { if testing.Short() {
t.Skip("type-checks the whole runtime package") t.Skip("type-checks the whole runtime package")
} }
dir, err := generateGoAsmHeader("/usr/local/go/src/runtime", "", "amd64", t.TempDir()) out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil {
t.Skipf("no Go toolchain: %v", err)
}
runtimeDir := filepath.Join(strings.TrimSpace(string(out)), "src", "runtime")
dir, err := generateGoAsmHeader(runtimeDir, "", "amd64", t.TempDir())
if err != nil { if err != nil {
t.Fatalf("generateGoAsmHeader(runtime): %v", err) t.Fatalf("generateGoAsmHeader(runtime): %v", err)
} }
+168 -32
View File
@@ -5,6 +5,7 @@ package main
import ( import (
"fmt" "fmt"
"go/build/constraint"
"maps" "maps"
"os" "os"
"os/exec" "os/exec"
@@ -15,9 +16,9 @@ import (
"strconv" "strconv"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch" "sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm" "sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// cmdAuditInstructions cross-checks a gasm encoder against the Go toolchain's // cmdAuditInstructions cross-checks a gasm encoder against the Go toolchain's
@@ -38,7 +39,7 @@ import (
// construction and are excluded from the diff; the other architectures list // construction and are excluded from the diff; the other architectures list
// their conditional branches outright. // their conditional branches outright.
func cmdAuditInstructions(args []string) error { func cmdAuditInstructions(args []string) error {
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [-I dir] [amd64|arm64|riscv64|loong64]", ` fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [--list] [-I dir] [amd64|arm64|riscv64|loong64]", `
Compare the gasm encoder for the given architecture (default amd64) against Compare the gasm encoder for the given architecture (default amd64) against
go tool asm and print the diff: superset encodings (gasm-only, shippable via go tool asm and print the diff: superset encodings (gasm-only, shippable via
gasm asm --format goobj) and known-but-unencodable names (the backlog). The gasm asm --format goobj) and known-but-unencodable names (the backlog). The
@@ -55,16 +56,19 @@ toolchain probing. A file whose name carries a recognisable _arch suffix is
attempted for that architecture; a file without one is attempted for all attempted for that architecture; a file without one is attempted for all
four, exactly as a GOARCH build would compile it. The report gives the four, exactly as a GOARCH build would compile it. The report gives the
per-architecture pass rates and the most common failure reasons, which drive per-architecture pass rates and the most common failure reasons, which drive
the encodability backlog by frequency rather than by table order. the encodability backlog by frequency rather than by table order. With
-list the report also prints every failing file with its reason, per
architecture.
`) `)
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons") corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
list := fs.Bool("list", false, "with --corpus, list every failing file with its reason, per architecture")
var dirs includeDirs var dirs includeDirs
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)") fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
if err := fs.Parse(args); err != nil { if err := fs.Parse(args); err != nil {
return err return err
} }
if *corpus { if *corpus {
return cmdAuditCorpus(fs.Args(), dirs) return cmdAuditCorpus(fs.Args(), dirs, *list)
} }
archName := "amd64" archName := "amd64"
switch n := len(fs.Args()); { switch n := len(fs.Args()); {
@@ -403,20 +407,30 @@ type corpusTally struct {
assembled int assembled int
reasons map[string]int // failure reason → count reasons map[string]int // failure reason → count
example map[string]string // failure reason → one representative file example map[string]string // failure reason → one representative file
fails []corpusFailure // every failure, in file order, for --list
} }
func (t *corpusTally) fail(path, reason string) { // corpusFailure is one failed attempt, recorded for the --list report.
type corpusFailure struct {
path string
reason string
detail string
}
func (t *corpusTally) fail(path string, err error) {
reason := corpusReason(err)
t.reasons[reason]++ t.reasons[reason]++
if t.example[reason] == "" { if t.example[reason] == "" {
t.example[reason] = path t.example[reason] = path
} }
t.fails = append(t.fails, corpusFailure{path: path, reason: reason, detail: firstLine(err.Error())})
} }
// cmdAuditCorpus implements audit-instructions --corpus. The include // cmdAuditCorpus implements audit-instructions --corpus. The include
// directories carry #include resolution over a corpus whose files refer to // directories carry #include resolution over a corpus whose files refer to
// headers such as GOROOT/pkg/include, the same -I a toolchain comparison // headers such as GOROOT/pkg/include, the same -I a toolchain comparison
// needs. // needs.
func cmdAuditCorpus(args []string, dirs includeDirs) error { func cmdAuditCorpus(args []string, dirs includeDirs, list bool) error {
if len(args) > 1 { if len(args) > 1 {
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")} return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
} }
@@ -454,7 +468,7 @@ func cmdAuditCorpus(args []string, dirs includeDirs) error {
if err != nil { if err != nil {
return err return err
} }
printCorpusStats(stats) printCorpusStats(stats, list)
return nil return nil
} }
@@ -463,8 +477,10 @@ type corpusStats struct {
root string root string
files int files int
generic int // files attempted for all four architectures generic int // files attempted for all four architectures
narrowed int // files whose //go:build admits a proper subset of the four
excluded int // files whose //go:build admits none of the four: never compiled
otherPort int // files named for another Go port: never attempted otherPort int // files named for another Go port: never attempted
full int // files that assembled for every target architecture full int // files that assembled for every applicable target architecture
targets []corpusTarget targets []corpusTarget
tallies []*corpusTally tallies []*corpusTally
} }
@@ -560,6 +576,57 @@ func otherGOOSFile(path string) bool {
return false return false
} }
// buildConstraint returns the file's leading //go:build expression, or nil
// when the file carries none. The constraint governs the same header block
// go/build reads: blank lines and comments may precede it, and the first
// line that is neither ends the block. A constraint that does not parse
// narrows nothing, so the file stays in the attempted set: the audit must
// never exclude a file the toolchain would compile.
func buildConstraint(src string) constraint.Expr {
for line := range strings.SplitSeq(src, "\n") {
t := strings.TrimSpace(line)
switch {
case t == "":
continue
case strings.HasPrefix(t, "//"):
if constraint.IsGoBuild(t) {
e, err := constraint.Parse(t)
if err != nil {
return nil
}
return e
}
continue
default:
return nil
}
}
return nil
}
// unixOS is go/build's unixOS set: the GOOSes the unix build tag admits.
var unixOS = map[string]bool{
"aix": true, "android": true, "darwin": true, "dragonfly": true,
"freebsd": true, "hurd": true, "illumos": true, "ios": true,
"linux": true, "netbsd": true, "openbsd": true, "solaris": true,
}
// constraintTags answers the build tags a plain `go build` sets for a
// target: the GOOS and GOARCH, gc, and unix on the unix-like GOOSes. No
// experiment, sanitiser or cgo tag is ever true: the audit models the
// default build, and no GOROOT assembly file's constraint hinges on cgo.
func constraintTags(goarch, goos string) func(string) bool {
return func(tag string) bool {
switch tag {
case goarch, goos, "gc":
return true
case "unix":
return unixOS[goos]
}
return false
}
}
func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) { func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
files, err := asmFiles(root) files, err := asmFiles(root)
if err != nil { if err != nil {
@@ -577,8 +644,8 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
tallies[i] = &corpusTally{reasons: map[string]int{}, example: map[string]string{}} tallies[i] = &corpusTally{reasons: map[string]int{}, example: map[string]string{}}
} }
// full is the north-star number: a file counts when every architecture // full is the north-star number: a file counts when every architecture
// its name allows assembles it. // its build admits assembles it.
full, generic, otherPort := 0, 0, 0 full, generic, otherPort, narrowedCount, excluded := 0, 0, 0, 0, 0
// Header generation is created on first use, so a corpus with no // Header generation is created on first use, so a corpus with no
// go_asm.h includes never pays for a temp directory. // go_asm.h includes never pays for a temp directory.
@@ -601,28 +668,78 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
// invisible to a file-name rule). // invisible to a file-name rule).
goos := goosFromFilename(path) goos := goosFromFilename(path)
// The GOOS the header generation type-checks under follows the
// file's name when the name carries one; the ambient GOOS is the
// honest guess otherwise.
namedArch := arch.FromFilename(path)
var wanted []int // indexes into targets var wanted []int // indexes into targets
if a := arch.FromFilename(path); a != arch.Unknown { other := false
switch {
case namedArch != arch.Unknown:
for i, tg := range targets { for i, tg := range targets {
if tg.a == a { if tg.a == namedArch {
wanted = append(wanted, i) wanted = append(wanted, i)
} }
} }
} else if otherPortFile(path) { case otherPortFile(path):
// A file named for a Go port gasm does not support (arm, // A file named for a Go port gasm does not support (arm,
// 386, s390x, ...) or for another GOOS is compiled by no // 386, s390x, ...) or for another GOOS is compiled by no
// supported-arch build, so it is neither generic nor a // supported-arch build, so it is neither generic nor a
// per-arch attempt: counting it as generic would make the // per-arch attempt: counting it as generic would make the
// headline unreachably low for reasons no supported target // headline unreachably low for reasons no supported target
// can fix. // can fix.
other = true
otherPort++ otherPort++
} else { default:
generic++
for i := range targets { for i := range targets {
wanted = append(wanted, i) wanted = append(wanted, i)
} }
} }
// A //go:build constraint narrows the set of targets the file is
// assembled for, the way the go command compiles the file only for
// the targets the expression admits: cpu_x86.s belongs to the x86
// build alone, and a file whose constraint admits none of the four
// targets (the goexperiment.runtimesecret and msan trees) is
// compiled by no supported build. The tags mirror what a plain
// `go build` sets: the GOOS and GOARCH, gc, and unix on the
// unix-like GOOSes; no experiment, sanitiser or cgo tag is ever
// true. The GOOS is the file's own when the name carries one,
// else the ambient one.
goosForEval := goos
if goosForEval == "" {
goosForEval = runtime.GOOS
}
narrowed := false
if len(wanted) > 0 {
if ce := buildConstraint(src); ce != nil {
kept := make([]int, 0, len(wanted))
for _, i := range wanted {
tg := targets[i]
if ce.Eval(constraintTags(goarchName(tg.a), goosForEval)) {
kept = append(kept, i)
}
}
if len(kept) < len(wanted) {
narrowed = true
}
wanted = kept
}
}
switch {
case other:
// already tallied above
case len(wanted) == 0:
excluded++
case namedArch != arch.Unknown:
// a per-arch attempt over the constraint's subset
case narrowed:
narrowedCount++
default:
generic++
}
// A file that includes go_asm.h parses against a per-target header: // A file that includes go_asm.h parses against a per-target header:
// the defines differ per architecture (internal/cpu's layout, for // the defines differ per architecture (internal/cpu's layout, for
// one) and per GOOS (sys_darwin_arm64.s's trampoline constants, // one) and per GOOS (sys_darwin_arm64.s's trampoline constants,
@@ -645,21 +762,22 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
hdrDir, err := hdr.dirFor(pkgDir, goos, goarchName(tg.a)) hdrDir, err := hdr.dirFor(pkgDir, goos, goarchName(tg.a))
if err != nil { if err != nil {
ok = false ok = false
t.fail(path, corpusReason(err)) t.fail(path, err)
continue continue
} }
f, errs := parser.ParseWithOptions(path, src, parser.Options{ f, errs := parser.ParseWithOptions(path, src, parser.Options{
Expand: true, Expand: true,
IncludeDirs: append(slices.Clone(dirs), hdrDir), IncludeDirs: append(slices.Clone(dirs), hdrDir),
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
}) })
if len(errs) > 0 { if len(errs) > 0 {
ok = false ok = false
t.fail(path, corpusReason(errs[0])) t.fail(path, errs[0])
continue continue
} }
if _, err := assembleFile(tg.a, f); err != nil { if _, err := assembleFile(tg.a, f, goos); err != nil {
ok = false ok = false
t.fail(path, corpusReason(err)) t.fail(path, err)
continue continue
} }
t.assembled++ t.assembled++
@@ -670,21 +788,28 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
continue continue
} }
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
ok := true ok := true
for _, i := range wanted { for _, i := range wanted {
tg, t := targets[i], tallies[i] tg, t := targets[i], tallies[i]
t.attempted++ t.attempted++
// The parse carries the target's platform predefines, so it
// cannot be shared across targets the way a header-free file's
// could: a #ifdef GOARCH_arm block must be live on arm64 and
// dead everywhere else.
f, errs := parser.ParseWithOptions(path, src, parser.Options{
Expand: true,
IncludeDirs: dirs,
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
})
var err error var err error
if len(errs) > 0 { if len(errs) > 0 {
err = errs[0] // a parse failure is a failure for every target err = errs[0] // a parse failure is a failure for every target
} else { } else {
_, err = assembleFile(tg.a, f) _, err = assembleFile(tg.a, f, goos)
} }
if err != nil { if err != nil {
ok = false ok = false
t.fail(path, corpusReason(err)) t.fail(path, err)
continue continue
} }
t.assembled++ t.assembled++
@@ -698,6 +823,8 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
root: root, root: root,
files: len(files), files: len(files),
generic: generic, generic: generic,
narrowed: narrowedCount,
excluded: excluded,
otherPort: otherPort, otherPort: otherPort,
full: full, full: full,
targets: targets, targets: targets,
@@ -706,14 +833,16 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
} }
// printCorpusStats renders the corpus audit report. // printCorpusStats renders the corpus audit report.
func printCorpusStats(s *corpusStats) { func printCorpusStats(s *corpusStats, list bool) {
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d named for other Go ports, never attempted)\n", s.root, s.files, s.generic, s.otherPort) fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d narrowed by //go:build; %d excluded by //go:build; %d named for other Go ports, never attempted)\n",
s.root, s.files, s.generic, s.narrowed, s.excluded, s.otherPort)
// The rate is over the files a supported build would attempt: the // The rate is over the files a supported build would attempt: the
// other ports' files sit in the count for completeness but can never // other ports' files and the ones no supported target compiles sit in
// assemble, so counting them in the denominator would report the gap // the count for completeness but can never assemble, so counting them
// of architectures gasm deliberately does not target. // in the denominator would report the gap of platforms gasm
attemptable := max(s.files-s.otherPort, 1) // deliberately does not target.
fmt.Printf(" assemble for every target architecture: %d of %d attemptable (%.1f%%)\n", s.full, attemptable, 100*float64(s.full)/float64(attemptable)) attemptable := max(s.files-s.otherPort-s.excluded, 1)
fmt.Printf(" assemble for every applicable target: %d of %d attemptable (%.1f%%)\n", s.full, attemptable, 100*float64(s.full)/float64(attemptable))
for i, tg := range s.targets { for i, tg := range s.targets {
t := s.tallies[i] t := s.tallies[i]
fmt.Printf(" %s: %d/%d attempted\n", tg.name, t.assembled, t.attempted) fmt.Printf(" %s: %d/%d attempted\n", tg.name, t.assembled, t.attempted)
@@ -721,6 +850,13 @@ func printCorpusStats(s *corpusStats) {
fmt.Printf(" %4d %s\n", t.reasons[r], r) fmt.Printf(" %4d %s\n", t.reasons[r], r)
fmt.Printf(" e.g. %s\n", t.example[r]) fmt.Printf(" e.g. %s\n", t.example[r])
} }
if !list {
continue
}
for _, f := range t.fails {
fmt.Printf(" FAIL %s\n", f.path)
fmt.Printf(" %s: %s\n", f.reason, f.detail)
}
} }
} }
+61 -1
View File
@@ -4,9 +4,10 @@
package main package main
import ( import (
"runtime"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch" "sourcedock.dev/petrbalvin/gasm-sdk/arch"
) )
func TestDerivedFamily(t *testing.T) { func TestDerivedFamily(t *testing.T) {
@@ -64,3 +65,62 @@ func TestGasmEncodable(t *testing.T) {
} }
} }
} }
// TestBuildConstraint pins the //go:build reader: the constraint governs the
// leading comment block, the first non-comment line ends it (a tag below a
// #include governs nothing, exactly as go/build drops it), and a file
// without one admits every target.
func TestBuildConstraint(t *testing.T) {
admits := func(src, goarch, goos string) bool {
t.Helper()
e := buildConstraint(src)
if e == nil {
return true
}
return e.Eval(constraintTags(goarch, goos))
}
const ret = "TEXT \xc2\xb7f(SB), NOSPLIT, $0\n\tRET\n"
cases := []struct {
name string
src string
amd64, arm64 bool
}{
{"no constraint", ret, true, true},
{"x86 only", "//go:build 386 || amd64\n\n" + ret, true, false},
{"arm64 and linux", "//go:build arm64 && linux\n\n" + ret, false, true},
{"msan never", "//go:build msan\n\n" + ret, false, false},
{"experiment never", "//go:build goexperiment.runtimesecret\n\n" + ret, false, false},
{"below an include governs nothing", "#include \"textflag.h\"\n//go:build amd64\n" + ret, true, true},
{"unparsable narrows nothing", "//go:build (amd64\n" + ret, true, true},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
if got := admits(c.src, "amd64", runtime.GOOS); got != c.amd64 {
t.Errorf("amd64 admission = %v, want %v", got, c.amd64)
}
if got := admits(c.src, "arm64", runtime.GOOS); got != c.arm64 {
t.Errorf("arm64 admission = %v, want %v", got, c.arm64)
}
})
}
}
// TestConstraintTags pins the tag set a plain `go build` sets: the GOOS and
// GOARCH, gc, unix on the unix-like GOOSes; nothing else is ever true.
func TestConstraintTags(t *testing.T) {
ok := constraintTags("amd64", "linux")
for _, tag := range []string{"amd64", "linux", "gc", "unix"} {
if !ok(tag) {
t.Errorf("tag %q = false, want true", tag)
}
}
for _, tag := range []string{"arm64", "freebsd", "darwin", "cgo", "race", "msan", "goexperiment.runtimesecret"} {
if ok(tag) {
t.Errorf("tag %q = true, want false", tag)
}
}
fb := constraintTags("arm64", "freebsd")
if !fb("unix") {
t.Error("unix on freebsd = false, want true")
}
}
+2 -2
View File
@@ -1,7 +1,7 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause // SPDX-License-Identifier: BSD-3-Clause
//go:build !linux //go:build !(linux || (freebsd && (amd64 || arm64 || riscv64)))
package main package main
@@ -11,6 +11,6 @@ import (
) )
func cmdDebug(args []string) int { func cmdDebug(args []string) int {
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux (ptrace)") fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux or FreeBSD (ptrace)")
return 1 return 1
} }
@@ -1,7 +1,7 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause // SPDX-License-Identifier: BSD-3-Clause
//go:build linux //go:build linux || (freebsd && (amd64 || arm64 || riscv64))
package main package main
@@ -13,8 +13,8 @@ import (
"strings" "strings"
"time" "time"
"sourcedock.dev/petrbalvin/gasm-devkit/debug" "sourcedock.dev/petrbalvin/gasm-sdk/debug"
"sourcedock.dev/petrbalvin/gasm-devkit/verify" "sourcedock.dev/petrbalvin/gasm-sdk/verify"
) )
func cmdDebug(args []string) int { func cmdDebug(args []string) int {
@@ -239,10 +239,12 @@ REPL commands:
fmt.Printf("gasm debug: cover: stopped on signal %v\n", sig) fmt.Printf("gasm debug: cover: stopped on signal %v\n", sig)
break break
} }
reason, _ := sess.StopInfo()
regs, rerr := sess.GetRegs() regs, rerr := sess.GetRegs()
if rerr != nil { if rerr != nil {
break break
} }
trapPC := regs.GetPC()
// HandleTrap restores the original byte, rewinds PC and counts // HandleTrap restores the original byte, rewinds PC and counts
// the hit on the breakpoint itself. Single-step over the // the hit on the breakpoint itself. Single-step over the
// restored instruction so the reinsertion at the top of the // restored instruction so the reinsertion at the top of the
@@ -251,6 +253,12 @@ REPL commands:
if err := sess.Step(); err != nil { if err := sess.Step(); err != nil {
break break
} }
} else if debug.TrapStray(sess, reason, trapPC) {
// A breakpoint-class trap that matches none of ours and left
// the PC in place: resuming would re-execute the trapping
// instruction forever, so the coverage run stops here.
fmt.Printf("gasm debug: cover: SIGTRAP at %#x matches no breakpoint; the PC did not advance\n", trapPC)
break
} }
} }
hits := map[uint64]int{} hits := map[uint64]int{}
+4 -4
View File
@@ -9,9 +9,9 @@ import (
"sort" "sort"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch" "sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm" "sourcedock.dev/petrbalvin/gasm-sdk/disasm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// cmdDis disassembles machine code: either a raw binary (standard input with // cmdDis disassembles machine code: either a raw binary (standard input with
@@ -84,7 +84,7 @@ func disSource(path string, target arch.Arch) int {
if len(errs) > 0 { if len(errs) > 0 {
return 1 return 1
} }
img, err := assembleFile(target, f) img, err := assembleFile(target, f, "")
if err != nil { if err != nil {
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err) fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
return 1 return 1
+43 -20
View File
@@ -27,15 +27,15 @@ import (
"sync" "sync"
"syscall" "syscall"
"sourcedock.dev/petrbalvin/gasm-devkit/arch" "sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm" "sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/format" "sourcedock.dev/petrbalvin/gasm-sdk/format"
"sourcedock.dev/petrbalvin/gasm-devkit/lexer" "sourcedock.dev/petrbalvin/gasm-sdk/lexer"
"sourcedock.dev/petrbalvin/gasm-devkit/lint" "sourcedock.dev/petrbalvin/gasm-sdk/lint"
"sourcedock.dev/petrbalvin/gasm-devkit/lsp" "sourcedock.dev/petrbalvin/gasm-sdk/lsp"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/verify" "sourcedock.dev/petrbalvin/gasm-sdk/verify"
) )
// version reports the release the toolchain recorded for this build: the // version reports the release the toolchain recorded for this build: the
@@ -574,7 +574,7 @@ naming the package.
defer cleanup() defer cleanup()
dirs = append(dirs, hdrDir) dirs = append(dirs, hdrDir)
} }
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs}) f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs, Predefines: platformPredefinesFor(string(targetArch), goos)})
for _, e := range errs { for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e) fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
} }
@@ -582,7 +582,7 @@ naming the package.
return 1 return 1
} }
img, err := assembleFile(targetArch, f) img, err := assembleFile(targetArch, f, goos)
if err != nil { if err != nil {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err) fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
return 1 return 1
@@ -789,11 +789,34 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
return 1 return 1
} }
// platformPredefines mirrors the go command's assembler invocation, which
// defines GOOS_<goos> and GOARCH_<arch> as -D macros: GOROOT headers
// (go_tls.h, asm_riscv64.h) select their platform blocks with #ifdef on
// exactly those names, so an assembler without them cannot see the platform
// definitions at all.
func platformPredefines(goarch, goos string) map[string]string {
return map[string]string{
"GOARCH_" + goarch: "1",
"GOOS_" + goos: "1",
}
}
// platformPredefinesFor resolves the ambient GOOS the way a build would: a
// file whose name carries one (sys_darwin_arm64.s) is compiled for that GOOS
// and nothing else.
func platformPredefinesFor(goarch string, fileGoos string) map[string]string {
goos := fileGoos
if goos == "" {
goos = runtime.GOOS
}
return platformPredefines(goarch, goos)
}
// assembleFile assembles a parsed file for the given architecture and returns the image. // assembleFile assembles a parsed file for the given architecture and returns the image.
func assembleFile(targetArch arch.Arch, f *ast.File) (*asm.Image, error) { func assembleFile(targetArch arch.Arch, f *ast.File, goos string) (*asm.Image, error) {
switch targetArch { switch targetArch {
case arch.AMD64: case arch.AMD64:
return asm.AssembleFile(f) return asm.AssembleFile(f, asm.WithGOOS(goos))
case arch.RISCV: case arch.RISCV:
return asm.AssembleFileRISCV(f) return asm.AssembleFileRISCV(f)
case arch.ARM64: case arch.ARM64:
@@ -813,18 +836,18 @@ func assemblePath(path string, forced arch.Arch, dirs includeDirs) (*asm.Image,
if err != nil { if err != nil {
return nil, err return nil, err
} }
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs}) target := forced
if target == arch.Unknown {
target = arch.FromFilename(path)
}
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs, Predefines: platformPredefinesFor(string(target), "")})
for _, e := range errs { for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e) fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
} }
if len(errs) > 0 { if len(errs) > 0 {
return nil, fmt.Errorf("parse errors") return nil, fmt.Errorf("parse errors")
} }
target := forced return assembleFile(target, f, "")
if target == arch.Unknown {
target = arch.FromFilename(path)
}
return assembleFile(target, f)
} }
// printByteDiff shows the first few byte differences between two code blocks. // printByteDiff shows the first few byte differences between two code blocks.
@@ -920,7 +943,7 @@ func cmdVerifyNonJIT(path string, targetArch arch.Arch, groundTruth, profile boo
if len(errs) > 0 { if len(errs) > 0 {
return 1 return 1
} }
img, err := assembleFile(targetArch, f) img, err := assembleFile(targetArch, f, "")
if err != nil { if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err) fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1 return 1
+2 -2
View File
@@ -14,8 +14,8 @@ import (
"syscall" "syscall"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch" "sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm" "sourcedock.dev/petrbalvin/gasm-sdk/asm"
) )
const clean = "#include \"textflag.h\"\n" + const clean = "#include \"textflag.h\"\n" +
+2 -2
View File
@@ -11,8 +11,8 @@ import (
"os" "os"
"strings" "strings"
gasmast "sourcedock.dev/petrbalvin/gasm-devkit/ast" gasmast "sourcedock.dev/petrbalvin/gasm-sdk/ast"
gasmparser "sourcedock.dev/petrbalvin/gasm-devkit/parser" gasmparser "sourcedock.dev/petrbalvin/gasm-sdk/parser"
) )
// cmdScaffold generates a differential test skeleton for every kernel in a // cmdScaffold generates a differential test skeleton for every kernel in a
+31 -7
View File
@@ -1,13 +1,17 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause // SPDX-License-Identifier: BSD-3-Clause
//go:build linux //go:build linux || (freebsd && (amd64 || arm64 || riscv64))
package debug package debug
import "strings" import "strings"
import "fmt" import (
"cmp"
"fmt"
"slices"
)
// Breakpoint is one software breakpoint in the debuggee. // Breakpoint is one software breakpoint in the debuggee.
type Breakpoint struct { type Breakpoint struct {
@@ -149,15 +153,16 @@ func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (
return bp, nil return bp, nil
} }
// Info returns a formatted list of all breakpoints. // Info returns a formatted list of all breakpoints, ordered by address so
// the numbering is stable across calls (map iteration order is not).
func (bm *Breakpoints) Info() string { func (bm *Breakpoints) Info() string {
if len(bm.bps) == 0 { if len(bm.bps) == 0 {
return "no breakpoints set\n" return "no breakpoints set\n"
} }
var result strings.Builder var result strings.Builder
i := 0 bps := bm.All()
for _, bp := range bm.bps { slices.SortFunc(bps, func(a, b *Breakpoint) int { return cmp.Compare(a.Addr, b.Addr) })
i++ for i, bp := range bps {
status := "enabled" status := "enabled"
if !bp.Enabled { if !bp.Enabled {
status = "disabled" status = "disabled"
@@ -170,7 +175,7 @@ func (bm *Breakpoints) Info() string {
if bp.Cond != nil { if bp.Cond != nil {
cond = " if " + bp.Cond.String() cond = " if " + bp.Cond.String()
} }
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond)) result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i+1, label, bp.Addr, status, bp.hits, cond))
} }
return result.String() return result.String()
} }
@@ -238,6 +243,25 @@ func (bm *Breakpoints) All() []*Breakpoint {
// Hits returns how many times the breakpoint has been hit. // Hits returns how many times the breakpoint has been hit.
func (bp *Breakpoint) Hits() int { return bp.hits } func (bp *Breakpoint) Hits() int { return bp.hits }
// TrapStray reports whether a stop is a breakpoint-class trap that matches
// no breakpoint of ours and cannot be resumed: the PC still stands on the
// trapping instruction (the kernel's own BRK, EBREAK or break, on an
// architecture that reports the trap in place), so the next resume would
// re-execute it and trap forever. trapPC is the PC the stop reported,
// before any HandleTrap rewinding; reason is the stop's StopInfo class.
// The debuggee's SIGSTOP barriers also stop without PC movement, and they
// never carry the breakpoint class, so they are unaffected.
func TrapStray(s *Session, reason StopReason, trapPC uint64) bool {
if reason != StopBreakpoint {
return false
}
after, err := s.GetRegs()
if err != nil {
return false
}
return after.GetPC() <= trapPC-uint64(breakpointPCAdjust)
}
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint { func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
// On amd64 the kernel reports the trap with RIP past the INT3; on the // On amd64 the kernel reports the trap with RIP past the INT3; on the
// other supported architectures the PC still stands on the trap // other supported architectures the PC still stands on the trap
+44
View File
@@ -10,6 +10,8 @@ package debug
// build on every supported linux architecture. // build on every supported linux architecture.
import ( import (
"fmt"
"slices"
"strings" "strings"
"testing" "testing"
) )
@@ -263,3 +265,45 @@ func TestConditionString(t *testing.T) {
} }
} }
} }
// TestBreakpointsInfoOrdered proves the listing is ordered by address: the
// numbers it prints are map keys rendered in iteration order otherwise, so
// the same set of breakpoints would renumber itself between calls.
func TestBreakpointsInfoOrdered(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
addrs := []uint64{0x9000, 0x1000, 0x7000, 0x3000, 0x8000, 0x2000,
0x6000, 0x4000, 0x5000, 0xa000}
for i, a := range addrs {
if _, err := bm.Set(a, fmt.Sprintf("bp%d", i)); err != nil {
t.Fatalf("Set(%#x): %v", a, err)
}
}
sorted := append([]uint64(nil), addrs...)
slices.Sort(sorted)
info := bm.Info()
for i, a := range sorted {
want := fmt.Sprintf(" %d: bp%d at %#x", i+1, indexOf(addrs, a), a)
if !strings.Contains(info, want) {
t.Errorf("Info() missing %q; listing:\n%s", want, info)
}
}
// The numbers themselves must ascend: "1:" before "2" ... "10".
pos := 0
for i := range len(addrs) {
next := strings.Index(info[pos:], fmt.Sprintf(" %d: ", i+1))
if next < 0 {
t.Fatalf("Info() has no entry %d; listing:\n%s", i+1, info)
}
pos += next
}
}
func indexOf(addrs []uint64, a uint64) int {
for i, v := range addrs {
if v == a {
return i
}
}
return -1
}
+325
View File
@@ -0,0 +1,325 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug
import (
"fmt"
"os"
"runtime"
"strings"
"testing"
"time"
)
// Regression tests for the debugger audit: memory access at mapping
// boundaries, watchpoint slot attribution, launch failure latency, stray
// trap instructions and the REPL's argument validation. All drive a real
// ptrace session, so they run on amd64 hosts only.
// memMap is one line of /proc/pid/maps.
type memMap struct {
lo, hi uint64
perms string
name string
}
// readMaps parses the debuggee's memory map.
func readMaps(t *testing.T, pid int) []memMap {
t.Helper()
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
if err != nil {
t.Fatalf("read maps: %v", err)
}
var out []memMap
for line := range strings.SplitSeq(string(data), "\n") {
fields := strings.Fields(line)
if len(fields) < 2 {
continue
}
var lo, hi uint64
if _, err := fmt.Sscanf(fields[0], "%x-%x", &lo, &hi); err != nil {
continue
}
m := memMap{lo: lo, hi: hi, perms: fields[1]}
if len(fields) >= 6 {
m.name = fields[5]
}
out = append(out, m)
}
return out
}
// boundaryByte returns the last byte of a writable, ordinary mapping that is
// followed by an unmapped gap: an access there is inside the mapping, while
// the 8-byte word starting at it crosses into unmapped memory.
func boundaryByte(t *testing.T, pid int) uint64 {
t.Helper()
maps := readMaps(t, pid)
for i, m := range maps {
if !strings.Contains(m.perms, "rw") ||
strings.Contains(m.name, "vvar") || strings.Contains(m.name, "vdso") ||
strings.Contains(m.name, "vsyscall") {
continue
}
gap := uint64(1) << 62
if i+1 < len(maps) {
gap = maps[i+1].lo - m.hi
}
if gap >= 4096 {
return m.hi - 1
}
}
t.Skip("no writable mapping followed by a hole; cannot construct the boundary")
return 0
}
// TestReadMemoryPageBoundary proves ReadMemory never reads past the requested
// range: one byte at the end of a mapping followed by a hole must be
// readable, which the old word-at-a-time tail read failed because its final
// 8-byte Peek crossed into the unmapped page.
func TestReadMemoryPageBoundary(t *testing.T) {
sess, _, _ := launchKernel(t, buildGasm(t), boundaryKernel(t), "boundary", nil)
addr := boundaryByte(t, sess.Pid())
mem, err := sess.ReadMemory(addr, 1)
if err != nil {
t.Fatalf("ReadMemory(%#x, 1): %v (the read must not cross into the unmapped page)", addr, err)
}
if len(mem) != 1 {
t.Fatalf("ReadMemory returned %d bytes, want 1", len(mem))
}
// A request whose own range crosses into the hole must still fail.
if _, err := sess.ReadMemory(addr, 8); err == nil {
t.Fatal("ReadMemory past the mapping end should fail")
}
}
// TestDisassemblePageBoundary proves the disassembler shrinks its read
// window at a mapping end instead of failing: the instruction stream cannot
// be decoded at all when the fixed 15-byte read crosses into the hole.
func TestDisassemblePageBoundary(t *testing.T) {
sess, _, _ := launchKernel(t, buildGasm(t), boundaryKernel(t), "boundary", nil)
addr := boundaryByte(t, sess.Pid())
if _, _, err := sess.Disassemble(addr); err != nil {
t.Fatalf("Disassemble(%#x): %v (the read window must shrink at the mapping end)", addr, err)
}
}
// TestWriteMemoryPageBoundary proves WriteMemory writes exactly the bytes it
// is given: one byte at the end of a mapping followed by a hole must be
// writable, which the old read-modify-write of the final partial word failed
// because its Peek crossed into the unmapped page.
func TestWriteMemoryPageBoundary(t *testing.T) {
sess, _, _ := launchKernel(t, buildGasm(t), boundaryKernel(t), "boundary", nil)
addr := boundaryByte(t, sess.Pid())
orig, err := sess.ReadMemory(addr, 1)
if err != nil {
t.Fatalf("ReadMemory(%#x, 1): %v", addr, err)
}
if err := sess.WriteMemory(addr, []byte{orig[0]}); err != nil {
t.Fatalf("WriteMemory(%#x, 1): %v (the write must not read past the range)", addr, err)
}
}
// TestWatchpointSlotAttribution proves a hit is attributed to the slot that
// fired, not to an earlier one whose DR6 status bit is still set: the B0-B3
// bits are sticky, so they must be acknowledged when read.
func TestWatchpointSlotAttribution(t *testing.T) {
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func wptwo(x, y int64) (a, b int64)
TEXT ·wptwo(SB), NOSPLIT, $0-32
MOVQ $0x1111, AX
MOVQ AX, a+16(FP)
MOVQ $0x2222, BX
MOVQ BX, b+24(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "wptwo", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
// FP sits one word above the entry stack pointer (the return address
// occupies [RSP]), so a+16(FP) = RSP+24 and b+24(FP) = RSP+32.
watchA := regs.RSP + 24
watchB := regs.RSP + 32
if err := sess.SetWatchpoint(0, watchA, WatchWrite, 8); err != nil {
t.Fatalf("SetWatchpoint(0): %v", err)
}
if err := sess.SetWatchpoint(1, watchB, WatchWrite, 8); err != nil {
t.Fatalf("SetWatchpoint(1): %v", err)
}
for i, want := range []uint64{watchA, watchB} {
if err := sess.Continue(); err != nil {
t.Fatalf("Continue (hit %d): %v", i+1, err)
}
reason, addr := sess.StopInfo()
if reason != StopWatchpoint {
t.Fatalf("hit %d: stop reason = %v, want StopWatchpoint", i+1, reason)
}
if addr != want {
t.Fatalf("hit %d reported %#x, want %#x (the sticky DR6 bit misattributes the slot)", i+1, addr, want)
}
}
// Clearing a watchpoint must zero its address register: a stale
// address in a disabled slot turns any sticky status bit into a
// misattributed report later.
if err := sess.ClearWatchpoint(0); err != nil {
t.Fatalf("ClearWatchpoint(0): %v", err)
}
dr0, err := ptracePeekUser(sess.Pid(), drOffset)
if err != nil {
t.Fatalf("read DR0: %v", err)
}
if dr0 != 0 {
t.Fatalf("DR0 = %#x after ClearWatchpoint, want 0 (the address register must be cleared)", dr0)
}
}
// TestLaunchFailsFastOnDeadDebuggee proves a debuggee that dies before
// signalling readiness surfaces promptly: the ready poll used to run its
// full 2.5 seconds before the wait discovered the exit.
func TestLaunchFailsFastOnDeadDebuggee(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
path := boundaryKernel(t)
start := time.Now()
sess, err := Launch(bin, path, "nosuchfunction", nil)
elapsed := time.Since(start)
if err == nil {
sess.Kill()
t.Fatal("Launch with an unknown function should fail")
}
if !strings.Contains(err.Error(), "before signalling readiness") &&
!strings.Contains(err.Error(), "debuggee exited") {
t.Errorf("error does not name the dead debuggee: %v", err)
}
if elapsed >= 1500*time.Millisecond {
t.Fatalf("Launch took %v to report the dead debuggee; the readiness poll must detect the exit, not time out", elapsed)
}
}
// TestStrayTrapRunsThrough proves the continue loop survives a trap
// instruction planted in the kernel itself (BYTE $0xCC, the same byte the
// debugger patches in): on architectures that report the trap in place the
// loop must surface the stop, and on amd64 it runs through to the exit. A
// regression here hangs, so a watchdog fails the run.
func TestStrayTrapRunsThrough(t *testing.T) {
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func stray() int64
TEXT ·stray(SB), NOSPLIT, $0-8
MOVQ $7, AX
BYTE $0xCC
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, _ := launchKernel(t, bin, path, "stray", nil)
timer := time.AfterFunc(time.Minute, func() {
panic("watchdog: the continue loop hung on the stray trap instruction")
})
defer timer.Stop()
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), 0, 0, 0, nil, nil,
strings.NewReader("continue\nquit\n"))
})
if !strings.Contains(out, "debuggee exited") {
t.Errorf("the stray trap wedged the continue loop; output:\n%s", out)
}
}
// TestStepIntoFaultReportsSignal proves the step command reports a genuine
// signal-delivery-stop instead of silently printing the faulting
// instruction as if the step had succeeded.
func TestStepIntoFaultReportsSignal(t *testing.T) {
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func crash() int64
TEXT ·crash(SB), NOSPLIT, $0-8
XORQ AX, AX
MOVQ (AX), AX
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "crash", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, nil, nil,
strings.NewReader("step 2\nquit\n"))
})
if !strings.Contains(out, "stopped on signal") {
t.Errorf("stepping into the fault did not report the signal; output:\n%s", out)
}
}
// TestREPLRejectsBadArguments proves the command loop reports malformed
// input instead of silently defaulting: an unknown label for x would read
// address 0, and a malformed count would silently step one instruction.
func TestREPLRejectsBadArguments(t *testing.T) {
bin := buildGasm(t)
path := boundaryKernel(t)
sess, bm, fl := launchKernel(t, bin, path, "boundary", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, nil, nil,
strings.NewReader("x nosuchlabel\nstep abc\ndisas abc\nwatch 0x1000 q 8\nquit\n"))
})
for _, want := range []string{
"unknown address: nosuchlabel",
"invalid count: abc",
"unknown watchpoint type: q",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
if got := strings.Count(out, "invalid count: abc"); got != 2 {
t.Errorf("invalid count reported %d times, want 2 (step and disas):\n%s", got, out)
}
}
// boundaryKernel is a minimal kernel for the boundary tests, which only need
// a live, stopped debuggee.
func boundaryKernel(t *testing.T) string {
t.Helper()
const kernel = `#include "textflag.h"
// func boundary() int64
TEXT ·boundary(SB), NOSPLIT, $0-8
MOVQ $1, AX
MOVQ AX, ret+0(FP)
RET
`
return writeKernel(t, kernel)
}
+65
View File
@@ -0,0 +1,65 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes. An amd64
// instruction is up to 15 bytes long, but the read must not reach past the
// end of the mapping: when the full 15-byte window crosses into unmapped
// memory the window shrinks, because an instruction at the mapping's end is
// by construction no longer than the readable bytes that hold it.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
var lastErr error
for _, n := range []int{15, 8, 4, 2, 1} {
mem, err := s.ReadMemory(addr, n)
if err != nil {
lastErr = err
continue
}
ins, derr := disasm.Decode(arch.AMD64, mem, addr)
if derr != nil {
return "", 0, derr
}
return ins.Text, ins.Len, nil
}
return "", 0, lastErr
}
// DisassembleN decodes up to n instructions starting at addr and returns
// them as a formatted string with addresses and byte offsets.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result strings.Builder
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
break
}
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
if length == 0 {
length = 1
}
pc += uint64(length)
}
return result.String()
}
// isCallInsn reports whether disassembled text (x86asm.IntelSyntax) is a
// call. The first token must match exactly: a prefix test would also catch
// unrelated mnemonics.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
return strings.ToLower(m) == "call"
}
+60
View File
@@ -0,0 +1,60 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && arm64
package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 4)
if err != nil {
return "", 0, err
}
ins, err := disasm.Decode(arch.ARM64, mem, addr)
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr and returns
// them as a formatted string with addresses and byte offsets.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result strings.Builder
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
break
}
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
if length == 0 {
length = 1
}
pc += uint64(length)
}
return result.String()
}
// isCallInsn reports whether disassembled text (arm64asm.GoSyntax) is a
// call. GoSyntax renders bl as CALL; the native mnemonic is accepted too.
// The first token must match exactly so branches never match.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "bl":
return true
}
return false
}
+70
View File
@@ -0,0 +1,70 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && riscv64
package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes. The read
// shrinks from 4 to 2 bytes when the full word crosses into unmapped memory:
// a compressed instruction at the mapping's end still fits the shorter
// window, and an instruction can never extend past the mapping that holds it.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
var lastErr error
for _, n := range []int{4, 2} {
mem, err := s.ReadMemory(addr, n)
if err != nil {
lastErr = err
continue
}
ins, derr := disasm.Decode(arch.RISCV, mem, addr)
if derr != nil {
return "", 0, derr
}
return ins.Text, ins.Len, nil
}
return "", 0, lastErr
}
// DisassembleN decodes up to n instructions starting at addr and returns
// them as a formatted string with addresses and byte offsets.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result strings.Builder
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
break
}
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
if length == 0 {
length = 1
}
pc += uint64(length)
}
return result.String()
}
// isCallInsn reports whether disassembled text (riscv64asm.GoSyntax) is a
// call. GoSyntax renders jal and jalr calls as CALL; the native mnemonics
// are accepted too. The first token must match exactly: a prefix test on
// "bl" would catch branches on other architectures, and jalr as ret prints
// RET, which must not be stepped over.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "jal", "jalr":
return true
}
return false
}
+20 -11
View File
@@ -9,22 +9,31 @@ import (
"fmt" "fmt"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch" "sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm" "sourcedock.dev/petrbalvin/gasm-sdk/disasm"
) )
// Disassemble decodes the instruction at the given address in the debuggee's // Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes. // memory and returns its text representation and length in bytes. An amd64
// instruction is up to 15 bytes long, but the read must not reach past the
// end of the mapping: when the full 15-byte window crosses into unmapped
// memory the window shrinks, because an instruction at the mapping's end is
// by construction no longer than the readable bytes that hold it.
func (s *Session) Disassemble(addr uint64) (string, int, error) { func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 15) var lastErr error
if err != nil { for _, n := range []int{15, 8, 4, 2, 1} {
return "", 0, err mem, err := s.ReadMemory(addr, n)
if err != nil {
lastErr = err
continue
}
ins, derr := disasm.Decode(arch.AMD64, mem, addr)
if derr != nil {
return "", 0, derr
}
return ins.Text, ins.Len, nil
} }
ins, err := disasm.Decode(arch.AMD64, mem, addr) return "", 0, lastErr
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
} }
// DisassembleN decodes up to n instructions starting at addr and returns // DisassembleN decodes up to n instructions starting at addr and returns
+2 -2
View File
@@ -9,8 +9,8 @@ import (
"fmt" "fmt"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch" "sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm" "sourcedock.dev/petrbalvin/gasm-sdk/disasm"
) )
// Disassemble decodes the instruction at the given address in the debuggee's // Disassemble decodes the instruction at the given address in the debuggee's
+2 -2
View File
@@ -9,8 +9,8 @@ import (
"fmt" "fmt"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch" "sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm" "sourcedock.dev/petrbalvin/gasm-sdk/disasm"
) )
// Disassemble decodes the instruction at the given address in the debuggee's // Disassemble decodes the instruction at the given address in the debuggee's
+19 -11
View File
@@ -9,22 +9,30 @@ import (
"fmt" "fmt"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch" "sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm" "sourcedock.dev/petrbalvin/gasm-sdk/disasm"
) )
// Disassemble decodes the instruction at the given address in the debuggee's // Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes. // memory and returns its text representation and length in bytes. The read
// shrinks from 4 to 2 bytes when the full word crosses into unmapped memory:
// a compressed instruction at the mapping's end still fits the shorter
// window, and an instruction can never extend past the mapping that holds it.
func (s *Session) Disassemble(addr uint64) (string, int, error) { func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 4) var lastErr error
if err != nil { for _, n := range []int{4, 2} {
return "", 0, err mem, err := s.ReadMemory(addr, n)
if err != nil {
lastErr = err
continue
}
ins, derr := disasm.Decode(arch.RISCV, mem, addr)
if derr != nil {
return "", 0, derr
}
return ins.Text, ins.Len, nil
} }
ins, err := disasm.Decode(arch.RISCV, mem, addr) return "", 0, lastErr
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
} }
// DisassembleN decodes up to n instructions starting at addr. // DisassembleN decodes up to n instructions starting at addr.
+136
View File
@@ -0,0 +1,136 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
import "fmt"
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" RIP = %#016x (func+%#x)\n", regs.RIP, regs.RIP-codeBase-funcOff)
fmt.Printf(" RSP = %#016x RBP = %#016x\n", regs.RSP, regs.RBP)
fmt.Printf(" RAX = %#016x RBX = %#016x\n", regs.RAX, regs.RBX)
fmt.Printf(" RCX = %#016x RDX = %#016x\n", regs.RCX, regs.RDX)
fmt.Printf(" RSI = %#016x RDI = %#016x\n", regs.RSI, regs.RDI)
fmt.Printf(" R8 = %#016x R9 = %#016x\n", regs.R8, regs.R9)
fmt.Printf(" R10 = %#016x R11 = %#016x\n", regs.R10, regs.R11)
fmt.Printf(" R12 = %#016x R13 = %#016x\n", regs.R12, regs.R13)
fmt.Printf(" R14 = %#016x R15 = %#016x\n", regs.R14, regs.R15)
fmt.Printf(" RFLAGS = %#x [%s]\n", regs.RFLAGS, decodeRflags(regs.RFLAGS))
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n Vector registers (YMM):")
for i := 0; i < 16; i += 2 {
fmt.Printf(" YMM%-2d = ", i)
printYMM(v.YMM[i][:])
fmt.Printf(" YMM%-2d = ", i+1)
printYMM(v.YMM[i+1][:])
fmt.Println()
}
}
func printYMM(b []byte) {
for j := 0; j < 32; j += 4 {
v := uint32(b[j]) | uint32(b[j+1])<<8 | uint32(b[j+2])<<16 | uint32(b[j+3])<<24
fmt.Printf("%08x ", v)
}
}
func decodeRflags(f uint64) string {
var flags string
if f&1 != 0 {
flags += "CF "
}
if f&(1<<2) != 0 {
flags += "PF "
}
if f&(1<<4) != 0 {
flags += "AF "
}
if f&(1<<6) != 0 {
flags += "ZF "
}
if f&(1<<7) != 0 {
flags += "SF "
}
if f&(1<<8) != 0 {
flags += "TF "
}
if f&(1<<9) != 0 {
flags += "IF "
}
if f&(1<<10) != 0 {
flags += "DF "
}
if f&(1<<11) != 0 {
flags += "OF "
}
if flags == "" {
return "none"
}
return flags[:len(flags)-1]
}
// SetReg modifies a register value in the debuggee.
func (s *Session) SetReg(name string, value uint64) error {
regs, err := s.GetRegs()
if err != nil {
return err
}
switch name {
case "rax", "eax", "ax", "al":
regs.RAX = value
case "rbx", "ebx", "bx", "bl":
regs.RBX = value
case "rcx", "ecx", "cx", "cl":
regs.RCX = value
case "rdx", "edx", "dx", "dl":
regs.RDX = value
case "rsi", "esi", "si":
regs.RSI = value
case "rdi", "edi", "di":
regs.RDI = value
case "rbp", "ebp", "bp":
regs.RBP = value
case "rsp", "esp", "sp":
regs.RSP = value
case "r8":
regs.R8 = value
case "r9":
regs.R9 = value
case "r10":
regs.R10 = value
case "r11":
regs.R11 = value
case "r12":
regs.R12 = value
case "r13":
regs.R13 = value
case "r14":
regs.R14 = value
case "r15":
regs.R15 = value
case "rip", "eip":
regs.RIP = value
default:
return fmt.Errorf("debug: unknown register %q", name)
}
return s.SetRegs(&regs)
}
// archReturnAddr reads the return address of the current frame (amd64
// ABI0 convention). A function that contains a CALL (or has a frame) is
// assembled with the prologue PUSHQ BP; MOVQ SP, BP, so mid-function the
// word at SP is the saved caller BP, a stack address, and the return
// address sits further up. Walk the stack from SP and take the first word
// that lies in an executable mapping: stack and data words never do, a
// return address always does. FreeBSD exposes no mapping list, so the
// walk degenerates to the raw entry convention, [SP] before any push.
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return s.Peek(regs.RSP)
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "RSP" }
+127
View File
@@ -0,0 +1,127 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && arm64
package debug
import (
"encoding/binary"
"fmt"
)
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.SP, regs.X29)
fmt.Printf(" LR = %#016x\n", regs.X30)
fmt.Printf(" X0 = %#016x X1 = %#016x\n", regs.X0, regs.X1)
fmt.Printf(" X2 = %#016x X3 = %#016x\n", regs.X2, regs.X3)
fmt.Printf(" X4 = %#016x X5 = %#016x\n", regs.X4, regs.X5)
fmt.Printf(" X6 = %#016x X7 = %#016x\n", regs.X6, regs.X7)
fmt.Printf(" X8 = %#016x X9 = %#016x\n", regs.X8, regs.X9)
fmt.Printf(" X10 = %#016x X11 = %#016x\n", regs.X10, regs.X11)
fmt.Printf(" X12 = %#016x X13 = %#016x\n", regs.X12, regs.X13)
fmt.Printf(" X14 = %#016x X15 = %#016x\n", regs.X14, regs.X15)
fmt.Printf(" X16 = %#016x X17 = %#016x\n", regs.X16, regs.X17)
fmt.Printf(" X18 = %#016x X19 = %#016x\n", regs.X18, regs.X19)
fmt.Printf(" X20 = %#016x X21 = %#016x\n", regs.X20, regs.X21)
fmt.Printf(" X22 = %#016x X23 = %#016x\n", regs.X22, regs.X23)
fmt.Printf(" X24 = %#016x X25 = %#016x\n", regs.X24, regs.X25)
fmt.Printf(" X26 = %#016x X27 = %#016x\n", regs.X26, regs.X27)
fmt.Printf(" X28 = %#016x PSTATE = %#x\n", regs.X28, regs.PSTATE)
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n Vector registers (V0-V31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" V%-2d = %016x%016x\n", i, binary.LittleEndian.Uint64(v.V[i][8:16]), binary.LittleEndian.Uint64(v.V[i][0:8]))
fmt.Printf(" V%-2d = %016x%016x\n", i+1, binary.LittleEndian.Uint64(v.V[i+1][8:16]), binary.LittleEndian.Uint64(v.V[i+1][0:8]))
}
}
// SetReg modifies a register value in the debuggee.
func (s *Session) SetReg(name string, value uint64) error {
regs, err := s.GetRegs()
if err != nil {
return err
}
switch name {
case "x0":
regs.X0 = value
case "x1":
regs.X1 = value
case "x2":
regs.X2 = value
case "x3":
regs.X3 = value
case "x4":
regs.X4 = value
case "x5":
regs.X5 = value
case "x6":
regs.X6 = value
case "x7":
regs.X7 = value
case "x8":
regs.X8 = value
case "x9":
regs.X9 = value
case "x10":
regs.X10 = value
case "x11":
regs.X11 = value
case "x12":
regs.X12 = value
case "x13":
regs.X13 = value
case "x14":
regs.X14 = value
case "x15":
regs.X15 = value
case "x16":
regs.X16 = value
case "x17":
regs.X17 = value
case "x18":
regs.X18 = value
case "x19":
regs.X19 = value
case "x20":
regs.X20 = value
case "x21":
regs.X21 = value
case "x22":
regs.X22 = value
case "x23":
regs.X23 = value
case "x24":
regs.X24 = value
case "x25":
regs.X25 = value
case "x26":
regs.X26 = value
case "x27":
regs.X27 = value
case "x28":
regs.X28 = value
case "x29", "fp":
regs.X29 = value
case "x30", "lr":
regs.X30 = value
case "sp":
regs.SP = value
case "pc":
regs.PC = value
default:
return fmt.Errorf("debug: unknown register %q", name)
}
return s.SetRegs(&regs)
}
// archReturnAddr reads the return address from LR (arm64 convention).
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return regs.X30, nil
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "SP" }
+121
View File
@@ -0,0 +1,121 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && riscv64
package debug
import "fmt"
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.Sp, regs.S0)
fmt.Printf(" RA = %#016x\n", regs.Ra)
fmt.Printf(" A0 = %#016x A1 = %#016x\n", regs.A0, regs.A1)
fmt.Printf(" A2 = %#016x A3 = %#016x\n", regs.A2, regs.A3)
fmt.Printf(" A4 = %#016x A5 = %#016x\n", regs.A4, regs.A5)
fmt.Printf(" A6 = %#016x A7 = %#016x\n", regs.A6, regs.A7)
fmt.Printf(" T0 = %#016x T1 = %#016x\n", regs.T0, regs.T1)
fmt.Printf(" T2 = %#016x T3 = %#016x\n", regs.T2, regs.T3)
fmt.Printf(" T4 = %#016x T5 = %#016x\n", regs.T4, regs.T5)
fmt.Printf(" T6 = %#016x\n", regs.T6)
fmt.Printf(" S1 = %#016x S2 = %#016x\n", regs.S1, regs.S2)
fmt.Printf(" S3 = %#016x S4 = %#016x\n", regs.S3, regs.S4)
fmt.Printf(" S5 = %#016x S6 = %#016x\n", regs.S5, regs.S6)
fmt.Printf(" S7 = %#016x S8 = %#016x\n", regs.S7, regs.S8)
fmt.Printf(" S9 = %#016x S10 = %#016x\n", regs.S9, regs.S10)
fmt.Printf(" S11 = %#016x\n", regs.S11)
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n FP registers (F0-F31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" F%-2d = %#018x F%-2d = %#018x\n", i, v.F[i], i+1, v.F[i+1])
}
fmt.Printf(" FCSR = %#x\n", v.FCSR)
}
// SetReg modifies a register value in the debuggee.
func (s *Session) SetReg(name string, value uint64) error {
regs, err := s.GetRegs()
if err != nil {
return err
}
switch name {
case "pc":
regs.PC = value
case "ra", "x1":
regs.Ra = value
case "sp", "x2":
regs.Sp = value
case "gp", "x3":
regs.Gp = value
case "tp", "x4":
regs.Tp = value
case "t0", "x5":
regs.T0 = value
case "t1", "x6":
regs.T1 = value
case "t2", "x7":
regs.T2 = value
case "s0", "fp", "x8":
regs.S0 = value
case "s1", "x9":
regs.S1 = value
case "a0", "x10":
regs.A0 = value
case "a1", "x11":
regs.A1 = value
case "a2", "x12":
regs.A2 = value
case "a3", "x13":
regs.A3 = value
case "a4", "x14":
regs.A4 = value
case "a5", "x15":
regs.A5 = value
case "a6", "x16":
regs.A6 = value
case "a7", "x17":
regs.A7 = value
case "s2", "x18":
regs.S2 = value
case "s3", "x19":
regs.S3 = value
case "s4", "x20":
regs.S4 = value
case "s5", "x21":
regs.S5 = value
case "s6", "x22":
regs.S6 = value
case "s7", "x23":
regs.S7 = value
case "s8", "x24":
regs.S8 = value
case "s9", "x25":
regs.S9 = value
case "s10", "x26":
regs.S10 = value
case "s11", "x27":
regs.S11 = value
case "t3", "x28":
regs.T3 = value
case "t4", "x29":
regs.T4 = value
case "t5", "x30":
regs.T5 = value
case "t6", "x31":
regs.T6 = value
default:
return fmt.Errorf("debug: unknown register %q", name)
}
return s.SetRegs(&regs)
}
// archReturnAddr reads the return address from RA (riscv64 convention).
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return regs.Ra, nil
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "SP" }
+2 -2
View File
@@ -17,8 +17,8 @@ import (
"time" "time"
"unsafe" "unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm" "sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/verify" "sourcedock.dev/petrbalvin/gasm-sdk/verify"
) )
// Integration tests beyond the basic entry breakpoint: hardware watchpoints, // Integration tests beyond the basic entry breakpoint: hardware watchpoints,
+310
View File
@@ -0,0 +1,310 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && (amd64 || arm64 || riscv64)
package debug
import (
"fmt"
"os"
"os/exec"
"path/filepath"
"runtime"
"strings"
"syscall"
"time"
"golang.org/x/sys/unix"
)
// Session is a ptrace debugging session controlling one debuggee process.
// The FreeBSD implementation sits behind the same surface as the Linux one:
// PT_TRACE_ME from the debuggee, PT_CONTINUE/PT_STEP from the tracer, and
// tracee memory through PT_IO (FreeBSD has no /proc/pid/mem to fall back
// on, so PT_IO is the only supported route).
type Session struct {
pid int
cmd *exec.Cmd
stopped bool
exited bool
codeBase uint64 // base address of the JIT code in the debuggee
tmpDir string // scratch directory of the session, removed on Kill
wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 dbw 0-15)
// lastSignal holds the signal of the most recent stop when that stop
// was a genuine signal-delivery-stop the caller must see (a fault such
// as SIGSEGV, SIGBUS, SIGFPE or SIGILL); 0 for breakpoint traps,
// single-steps, SIGSTOP and suppressed runtime signals.
lastSignal syscall.Signal
}
// Launch starts the debuggee subprocess (gasm debug --target ...) and
// attaches to it via ptrace.
func Launch(gasmBin, asmPath, funcName string, args []byte) (*Session, error) {
sess, _, err := LaunchWithBuffers(gasmBin, asmPath, funcName, args, "")
return sess, err
}
// LaunchWithBuffers is like Launch but also allocates buffers in the debuggee.
//
// It pins the calling goroutine to its OS thread and leaves it pinned: the
// debuggee's PT_TRACE_ME binds the tracer relation to the forking thread,
// and every ptrace request on the session must come from that same thread.
// All Session methods must therefore be called from the goroutine that
// launched the session (the REPL and coverage loops do exactly that).
func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec string) (*Session, []uint64, error) {
runtime.LockOSThread() // ptrace requests must stay on the forking thread
self, err := os.Executable()
if err != nil {
return nil, nil, fmt.Errorf("debug: cannot find gasm binary: %w", err)
}
if gasmBin != "" {
self = gasmBin
}
tmpDir, err := os.MkdirTemp("", "gasm-debug-*")
if err != nil {
return nil, nil, fmt.Errorf("debug: tempdir: %w", err)
}
argsFile := filepath.Join(tmpDir, "args.bin")
if err := os.WriteFile(argsFile, args, 0o644); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: write args: %w", err)
}
if bufSpec != "" {
if err := os.WriteFile(filepath.Join(tmpDir, "bufspec"), []byte(bufSpec), 0o644); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: write bufspec: %w", err)
}
}
cmd := exec.Command(self, "debug", "--func", funcName, "--args", argsFile, asmPath)
cmd.Env = append(os.Environ(), "GASM_DEBUG_TARGET=1", "GASM_DEBUG_TMP="+tmpDir)
cmd.Stdout = nil
cmd.Stderr = os.Stderr
cmd.SysProcAttr = &syscall.SysProcAttr{}
if err := cmd.Start(); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: start debuggee: %w", err)
}
s := &Session{pid: cmd.Process.Pid, cmd: cmd, tmpDir: tmpDir}
readyFile := filepath.Join(tmpDir, "ready")
for range 500 {
if _, err := os.Stat(readyFile); err == nil {
break
}
// A debuggee that died before signalling readiness (unknown
// function, unparseable source) writes its failure notice to the
// handshake directory; read it and fail fast. The poll never
// waits on the child: a wait here could consume the SIGSTOP park
// that waitStopped below must receive, hanging the launch.
if err := s.deadReason(); err != nil {
cmd.Wait()
os.RemoveAll(tmpDir)
return nil, nil, err
}
time.Sleep(5 * time.Millisecond)
}
// The debuggee parks itself with SIGSTOP once the JIT code is mapped.
// A Go tracee also reports SIGURG preemption as signal-delivery-stops,
// so the wait loops until a stop the debugger cares about instead of
// assuming the first event is the SIGSTOP.
if _, err := s.waitStopped(); err != nil {
cmd.Process.Kill()
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: wait for debuggee: %w", err)
}
s.stopped = true
// The debuggee reports its JIT mapping in the codebase file; that is
// the supported path on FreeBSD, where no /proc/pid/maps exists to
// scan for the RWX region as a fallback.
if data, err := os.ReadFile(filepath.Join(tmpDir, "codebase")); err == nil {
fmt.Sscanf(string(data), "%d", &s.codeBase)
}
var bufAddrs []uint64
if bufSpec != "" {
addrFile := filepath.Join(tmpDir, "bufaddrs")
if data, err := os.ReadFile(addrFile); err == nil {
for line := range strings.SplitSeq(strings.TrimSpace(string(data)), "\n") {
var addr uint64
if _, err := fmt.Sscanf(line, "%d", &addr); err == nil {
bufAddrs = append(bufAddrs, addr)
}
}
}
}
return s, bufAddrs, nil
}
// deadReason reports the debuggee's own failure notice, the file its
// failure paths write before exiting. A debuggee killed without a notice
// (a crash, SIGKILL) surfaces through waitStopped after the poll instead,
// which is why the poll's budget stays finite.
func (s *Session) deadReason() error {
data, err := os.ReadFile(filepath.Join(s.tmpDir, "dead"))
if err != nil {
return nil
}
return fmt.Errorf("debug: debuggee failed before signalling readiness: %s", strings.TrimSpace(string(data)))
}
// waitStopped consumes ptrace-stop events until one the debugger cares
// about arrives: SIGTRAP (a breakpoint or a completed single-step), the
// debuggee's own SIGSTOP, or a genuine signal-delivery-stop. A Go tracee's
// runtime raises SIGURG for asynchronous preemption, and every signal on a
// traced thread surfaces as a signal-delivery-stop, so SIGURG is suppressed
// and the tracee resumed without it. Every other signal (SIGSEGV, SIGBUS,
// SIGFPE, SIGILL, ...) is returned to the caller: resuming with signal 0
// would restart the faulting instruction and fault forever, so a faulting
// kernel must surface as a stop the caller reports.
func (s *Session) waitStopped() (syscall.Signal, error) {
for {
var ws syscall.WaitStatus
if _, err := syscall.Wait4(s.pid, &ws, syscall.WUNTRACED, nil); err != nil {
return 0, err
}
if ws.Exited() {
s.exited = true
return 0, fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
}
if ws.Signaled() {
s.exited = true
return 0, fmt.Errorf("debuggee killed by signal %v", ws.Signal())
}
switch sig := ws.StopSignal(); sig {
case syscall.SIGTRAP, syscall.SIGSTOP:
s.stopped = true
s.lastSignal = 0
return sig, nil
case syscall.SIGURG:
// Go runtime asynchronous preemption: resume the tracee
// without delivering the signal.
s.lastSignal = 0
if err := unix.PtraceCont(s.pid, 0); err != nil {
return 0, fmt.Errorf("debug: PT_CONTINUE: %w", err)
}
default:
// A genuine signal-delivery-stop. Report it; the caller
// decides how to proceed.
s.stopped = true
s.lastSignal = sig
return sig, nil
}
}
}
// LastSignal returns the signal of the most recent stop when that stop was
// a genuine signal-delivery-stop (a fault such as SIGSEGV, SIGFPE, SIGILL
// or SIGBUS), and 0 for breakpoint traps, single-steps, SIGSTOP and
// suppressed runtime signals.
func (s *Session) LastSignal() syscall.Signal { return s.lastSignal }
// Peek reads a word (8 bytes) from the debuggee's memory at addr, through
// PT_IO with PIOD_READ_D.
func (s *Session) Peek(addr uint64) (uint64, error) {
var buf [8]byte
if _, err := unix.PtraceIO(unix.PIOD_READ_D, s.pid, uintptr(addr), buf[:], len(buf)); err != nil {
return 0, fmt.Errorf("debug: read mem %#x: %w", addr, err)
}
return uint64(buf[0]) | uint64(buf[1])<<8 | uint64(buf[2])<<16 | uint64(buf[3])<<24 |
uint64(buf[4])<<32 | uint64(buf[5])<<40 | uint64(buf[6])<<48 | uint64(buf[7])<<56, nil
}
// Poke writes a word (8 bytes) to the debuggee's memory at addr, through
// PT_IO with PIOD_WRITE_D.
func (s *Session) Poke(addr, val uint64) error {
buf := []byte{byte(val), byte(val >> 8), byte(val >> 16), byte(val >> 24),
byte(val >> 32), byte(val >> 40), byte(val >> 48), byte(val >> 56)}
if _, err := unix.PtraceIO(unix.PIOD_WRITE_D, s.pid, uintptr(addr), buf, len(buf)); err != nil {
return fmt.Errorf("debug: write mem %#x: %w", addr, err)
}
return nil
}
// ReadMemory reads len bytes from the debuggee's memory at addr in one
// PT_IO request, the shape the request is built for.
func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) {
out := make([]byte, length)
n, err := unix.PtraceIO(unix.PIOD_READ_D, s.pid, uintptr(addr), out, length)
return out[:n], err
}
// WriteMemory writes bytes to the debuggee's memory at addr in one PT_IO
// request.
func (s *Session) WriteMemory(addr uint64, data []byte) error {
_, err := unix.PtraceIO(unix.PIOD_WRITE_D, s.pid, uintptr(addr), data, len(data))
return err
}
// Step executes a single instruction in the debuggee.
func (s *Session) Step() error {
if s.exited {
return fmt.Errorf("debug: debuggee has exited")
}
if err := unix.PtraceSingleStep(s.pid); err != nil {
return fmt.Errorf("debug: PT_STEP: %w", err)
}
_, err := s.waitStopped()
return err
}
// Continue resumes execution until the next breakpoint or exit.
func (s *Session) Continue() error {
if s.exited {
return fmt.Errorf("debug: debuggee has exited")
}
if err := unix.PtraceCont(s.pid, 0); err != nil {
return fmt.Errorf("debug: PT_CONTINUE: %w", err)
}
_, err := s.waitStopped()
return err
}
// Exited returns true if the debuggee has terminated.
func (s *Session) Exited() bool { return s.exited }
// Pid returns the debuggee's process ID.
func (s *Session) Pid() int { return s.pid }
// CodeBase returns the base address of the JIT code in the debuggee.
func (s *Session) CodeBase() uint64 { return s.codeBase }
// Kill terminates the debuggee and removes the session's scratch
// directory, so a successful session leaves no gasm-debug-* debris behind.
func (s *Session) Kill() {
if !s.exited {
syscall.Kill(s.pid, syscall.SIGKILL)
syscall.Wait4(s.pid, nil, 0, nil)
s.exited = true
}
if s.cmd != nil && s.cmd.Process != nil {
s.cmd.Wait()
}
if s.tmpDir != "" {
os.RemoveAll(s.tmpDir)
s.tmpDir = ""
}
}
// execRange is one executable mapping of the debuggee.
type execRange struct {
lo, hi uint64
}
// execRanges is a stub on FreeBSD: there is no /proc/pid/maps to parse,
// and procfs(5) is not guaranteed to be mounted. The callers degrade
// gracefully: archReturnAddr falls back to the raw stack convention and
// the mapping scan is skipped.
func execRanges(pid int) []execRange { return nil }
// findRWXMapping is a stub on FreeBSD for the same reason: the codebase
// handshake file is the supported way the JIT region is located.
func findRWXMapping(pid int) uint64 { return 0 }
+160
View File
@@ -0,0 +1,160 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
import (
"encoding/binary"
"fmt"
"unsafe"
"golang.org/x/sys/unix"
)
// GetRegs reads the general-purpose registers of the stopped debuggee and
// converts the FreeBSD struct reg into the portable layout.
func (s *Session) GetRegs() (Regs, error) {
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return Regs{}, fmt.Errorf("debug: PT_GETREGS: %w", err)
}
return Regs{
R15: uint64(ur.R15),
R14: uint64(ur.R14),
R13: uint64(ur.R13),
R12: uint64(ur.R12),
R11: uint64(ur.R11),
R10: uint64(ur.R10),
R9: uint64(ur.R9),
R8: uint64(ur.R8),
RDI: uint64(ur.Rdi),
RSI: uint64(ur.Rsi),
RBP: uint64(ur.Rbp),
RBX: uint64(ur.Rbx),
RDX: uint64(ur.Rdx),
RCX: uint64(ur.Rcx),
RAX: uint64(ur.Rax),
RIP: uint64(ur.Rip),
CS: uint64(ur.Cs),
RFLAGS: uint64(ur.Rflags),
RSP: uint64(ur.Rsp),
SS: uint64(ur.Ss),
FS: uint64(ur.Fs),
GS: uint64(ur.Gs),
DS: uint64(ur.Ds),
ES: uint64(ur.Es),
}, nil
}
// SetRegs writes the general-purpose registers of the stopped debuggee.
func (s *Session) SetRegs(regs *Regs) error {
// Read-modify-write keeps the fields FreeBSD owns (trapno, err) intact.
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return fmt.Errorf("debug: PT_GETREGS: %w", err)
}
ur.R15 = int64(regs.R15)
ur.R14 = int64(regs.R14)
ur.R13 = int64(regs.R13)
ur.R12 = int64(regs.R12)
ur.R11 = int64(regs.R11)
ur.R10 = int64(regs.R10)
ur.R9 = int64(regs.R9)
ur.R8 = int64(regs.R8)
ur.Rdi = int64(regs.RDI)
ur.Rsi = int64(regs.RSI)
ur.Rbp = int64(regs.RBP)
ur.Rbx = int64(regs.RBX)
ur.Rdx = int64(regs.RDX)
ur.Rcx = int64(regs.RCX)
ur.Rax = int64(regs.RAX)
ur.Rip = int64(regs.RIP)
ur.Cs = int64(regs.CS)
ur.Rflags = int64(regs.RFLAGS)
ur.Rsp = int64(regs.RSP)
ur.Ss = int64(regs.SS)
return unix.PtraceSetRegs(s.pid, &ur)
}
// FPRegs holds the x87 FPU and SSE (XMM) register state, the FXSAVE image
// the FreeBSD struct fpreg mirrors: XMM0-15 at the same offsets.
type FPRegs struct {
XMM [16][16]byte // XMM0-15
}
// GetFPRegs retrieves the FPU/SSE register state via PT_GETFPREGS. The
// FreeBSD struct fpreg mirrors the FXSAVE image: the x87 environment and
// stack in Env/Acc, XMM0-15 in Xacc.
func (s *Session) GetFPRegs() (FPRegs, error) {
var fp FPRegs
var fr unix.FpReg
if err := unix.PtraceGetFpRegs(s.pid, &fr); err != nil {
return fp, fmt.Errorf("debug: PT_GETFPREGS: %w", err)
}
for i := range 16 {
copy(fp.XMM[i][:], fr.Xacc[i][:])
}
return fp, nil
}
// VectorRegs holds the YMM register state.
type VectorRegs struct {
YMM [16][32]byte // YMM0-15 (full 256-bit values)
}
// The XSAVE area the PT_GETXSTATE request returns follows the architectural
// layout (Intel SDM vol 1, "XSAVE"): the 512-byte legacy FXSAVE image (x87
// state in 0-159, XMM0-15 in 160-511), then the 64-byte xsave header whose
// first 8 bytes are xstate_bv, then one component per set feature bit, each
// 64-byte aligned. The YMM high halves are the first extended component,
// at offset 576; XFEATURE_STATE_BIT_AVX is bit 2 of xstate_bv.
const (
xsaveXMMOffset = 160
xsaveHeaderOffset = 512
xsaveBVOffset = xsaveHeaderOffset
ymmOffset = xsaveHeaderOffset + 64 // 576
ymmSize = 256 // 16 registers, 16 bytes each
xfeatureMaskYMM = 1 << 2
xstateMaxBuffer = 4096 // PT_GETXSTATE_INFO bounds the size far below this
)
// GetVectorRegs retrieves the YMM registers via PT_GETXSTATE. The low
// (XMM) halves always come from the legacy image; the high halves are
// copied only when xstate_bv reports the AVX state, and read as zero
// otherwise. When the request fails the FP image still provides correct
// XMM halves, so that is the fallback.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
var v VectorRegs
buf := make([]byte, xstateMaxBuffer)
n, _, errno := unix.Syscall6(
unix.SYS_PTRACE,
uintptr(unix.PT_GETXSTATE),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(&buf[0])),
0, 0,
)
if errno != 0 {
fp, err := s.GetFPRegs()
if err != nil {
return v, err
}
for i := range 16 {
copy(v.YMM[i][:16], fp.XMM[i][:])
}
return v, nil
}
for i := range 16 {
copy(v.YMM[i][:16], buf[xsaveXMMOffset+16*i:xsaveXMMOffset+16*i+16])
}
if int(n) >= ymmOffset+ymmSize {
if binary.LittleEndian.Uint64(buf[xsaveBVOffset:xsaveBVOffset+8])&xfeatureMaskYMM != 0 {
for i := range 16 {
copy(v.YMM[i][16:], buf[ymmOffset+16*i:ymmOffset+16*i+16])
}
}
}
return v, nil
}
+114
View File
@@ -0,0 +1,114 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && arm64
package debug
import (
"fmt"
"golang.org/x/sys/unix"
)
// GetRegs reads the general-purpose registers of the stopped debuggee and
// converts the FreeBSD struct reg (x[30], lr, sp, elr, spsr) into the
// portable layout.
func (s *Session) GetRegs() (Regs, error) {
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return Regs{}, fmt.Errorf("debug: PT_GETREGS: %w", err)
}
return Regs{
X0: ur.X[0],
X1: ur.X[1],
X2: ur.X[2],
X3: ur.X[3],
X4: ur.X[4],
X5: ur.X[5],
X6: ur.X[6],
X7: ur.X[7],
X8: ur.X[8],
X9: ur.X[9],
X10: ur.X[10],
X11: ur.X[11],
X12: ur.X[12],
X13: ur.X[13],
X14: ur.X[14],
X15: ur.X[15],
X16: ur.X[16],
X17: ur.X[17],
X18: ur.X[18],
X19: ur.X[19],
X20: ur.X[20],
X21: ur.X[21],
X22: ur.X[22],
X23: ur.X[23],
X24: ur.X[24],
X25: ur.X[25],
X26: ur.X[26],
X27: ur.X[27],
X28: ur.X[28],
X29: ur.X[29],
X30: ur.Lr,
SP: ur.Sp,
PC: ur.Elr,
PSTATE: uint64(ur.Spsr),
}, nil
}
// SetRegs writes the general-purpose registers of the stopped debuggee.
func (s *Session) SetRegs(regs *Regs) error {
var ur unix.Reg
ur.X = [30]uint64{
regs.X0, regs.X1, regs.X2, regs.X3, regs.X4, regs.X5, regs.X6,
regs.X7, regs.X8, regs.X9, regs.X10, regs.X11, regs.X12, regs.X13,
regs.X14, regs.X15, regs.X16, regs.X17, regs.X18, regs.X19, regs.X20,
regs.X21, regs.X22, regs.X23, regs.X24, regs.X25, regs.X26, regs.X27,
regs.X28, regs.X29,
}
ur.Lr = regs.X30
ur.Sp = regs.SP
ur.Elr = regs.PC
ur.Spsr = uint32(regs.PSTATE)
return unix.PtraceSetRegs(s.pid, &ur)
}
// FPRegs holds the arm64 FP/NEON register state: the 32 128-bit V
// registers, then FPSR and FPCR (the user_fpsimd shape).
type FPRegs struct {
V [32][16]byte // V0-V31 (128-bit NEON/FP registers)
FPSR uint32
FPCR uint32
}
// GetFPRegs retrieves the FP/NEON register state via PT_GETFPREGS. The
// FreeBSD struct fpreg holds the 32 128-bit V registers followed by FPSR
// and FPCR, the user_fpsimd shape.
func (s *Session) GetFPRegs() (FPRegs, error) {
var fp FPRegs
var fr unix.FpReg
if err := unix.PtraceGetFpRegs(s.pid, &fr); err != nil {
return fp, fmt.Errorf("debug: PT_GETFPREGS: %w", err)
}
for i := range 32 {
copy(fp.V[i][:], fr.Q[i][:])
}
return fp, nil
}
// VectorRegs holds the full SIMD register state.
type VectorRegs struct {
V [32][16]byte // V0-V31 (128-bit)
}
// GetVectorRegs retrieves the SIMD registers.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
var v VectorRegs
fp, err := s.GetFPRegs()
if err != nil {
return v, err
}
copy(v.V[:][:], fp.V[:][:])
return v, nil
}
+115
View File
@@ -0,0 +1,115 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && riscv64
package debug
import (
"fmt"
"golang.org/x/sys/unix"
)
// GetRegs reads the general-purpose registers of the stopped debuggee and
// converts the FreeBSD struct reg into the portable layout. Sstatus rides
// the kernel's struct but the portable surface carries the GPRs and PC.
func (s *Session) GetRegs() (Regs, error) {
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return Regs{}, fmt.Errorf("debug: PT_GETREGS: %w", err)
}
return Regs{
PC: ur.Sepc,
Ra: ur.Ra,
Sp: ur.Sp,
Gp: ur.Gp,
Tp: ur.Tp,
T0: ur.T[0],
T1: ur.T[1],
T2: ur.T[2],
S0: ur.S[0],
S1: ur.S[1],
A0: ur.A[0],
A1: ur.A[1],
A2: ur.A[2],
A3: ur.A[3],
A4: ur.A[4],
A5: ur.A[5],
A6: ur.A[6],
A7: ur.A[7],
S2: ur.S[2],
S3: ur.S[3],
S4: ur.S[4],
S5: ur.S[5],
S6: ur.S[6],
S7: ur.S[7],
S8: ur.S[8],
S9: ur.S[9],
S10: ur.S[10],
S11: ur.S[11],
T3: ur.T[3],
T4: ur.T[4],
T5: ur.T[5],
T6: ur.T[6],
}, nil
}
// SetRegs writes the general-purpose registers of the stopped debuggee.
// Read-modify-write keeps sstatus, which the kernel owns, intact.
func (s *Session) SetRegs(regs *Regs) error {
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return fmt.Errorf("debug: PT_GETREGS: %w", err)
}
ur.Sepc = regs.PC
ur.Ra = regs.Ra
ur.Sp = regs.Sp
ur.Gp = regs.Gp
ur.Tp = regs.Tp
ur.T = [7]uint64{regs.T0, regs.T1, regs.T2, regs.T3, regs.T4, regs.T5, regs.T6}
ur.S = [12]uint64{regs.S0, regs.S1, regs.S2, regs.S3, regs.S4, regs.S5,
regs.S6, regs.S7, regs.S8, regs.S9, regs.S10, regs.S11}
ur.A = [8]uint64{regs.A0, regs.A1, regs.A2, regs.A3, regs.A4, regs.A5, regs.A6, regs.A7}
return unix.PtraceSetRegs(s.pid, &ur)
}
// FPRegs holds the RISC-V FP register state (32 64-bit FP registers plus
// fcsr).
type FPRegs struct {
F [32]uint64 // F0-F31 (64-bit FP registers)
FCSR uint32
}
// GetFPRegs retrieves the FP register state via PT_GETFPREGS. The FreeBSD
// struct fpreg carries each 64-bit FP register in a 128-bit slot (fp_x is
// the flat [64]-word area the x/sys type renders as [32][2]); the low word
// holds the register, and FCSR rides the tail.
func (s *Session) GetFPRegs() (FPRegs, error) {
var fp FPRegs
var fr unix.FpReg
if err := unix.PtraceGetFpRegs(s.pid, &fr); err != nil {
return fp, fmt.Errorf("debug: PT_GETFPREGS: %w", err)
}
for i := range 32 {
fp.F[i] = fr.X[i][0]
}
fp.FCSR = uint32(fr.Fcsr)
return fp, nil
}
// VectorRegs holds the FP register state shown by the regs command
// (riscv64 has 32 64-bit FP registers and fcsr).
type VectorRegs struct {
F [32]uint64
FCSR uint32
}
// GetVectorRegs retrieves the FP registers.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
fp, err := s.GetFPRegs()
if err != nil {
return VectorRegs{}, err
}
return VectorRegs{F: fp.F, FCSR: fp.FCSR}, nil
}
+91
View File
@@ -0,0 +1,91 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
import (
"os/exec"
"path/filepath"
"runtime"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
// TestLaunchAndBreakpoint is the FreeBSD twin of the Linux integration
// test: it drives the whole launch, breakpoint, trap and register-rewind
// flow end to end. It needs a real FreeBSD kernel (ptrace does not work
// under emulation), so it only runs where it can.
func TestLaunchAndBreakpoint(t *testing.T) {
if runtime.GOARCH != "amd64" {
t.Skip("runs only on amd64 hosts")
}
// The tracer is the OS thread that forked the debuggee (PT_TRACE_ME
// binds the relation to that thread); every ptrace request must come
// from the same thread, so pin the test goroutine to one thread.
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := filepath.Join(t.TempDir(), "gasm")
out, err := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm").CombinedOutput()
if err != nil {
t.Fatalf("build gasm: %v: %s", err, out)
}
const kernelPath = "../testdata/verify/basic_amd64.s"
k, err := verify.Load(kernelPath)
if err != nil {
t.Fatalf("Load: %v", err)
}
t.Cleanup(k.Close)
fl, err := k.Func("wideCopy")
if err != nil {
t.Fatalf("Func: %v", err)
}
sess, err := Launch(bin, kernelPath, "wideCopy", make([]byte, fl.Args))
if err != nil {
t.Fatalf("Launch: %v", err)
}
t.Cleanup(sess.Kill)
bm := NewBreakpoints(sess)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
// The INT3 must be visible in the debuggee's memory.
word, err := sess.Peek(entry)
if err != nil {
t.Fatalf("Peek: %v", err)
}
if b := word & 0xFF; b != 0xCC {
t.Fatalf("int3 not patched: first byte %#02x at %#x", b, entry)
}
// The debuggee raises a second SIGSTOP after the launch barrier (the
// child's RunTarget marks its entry), so like the REPL and the cover
// mode the test keeps resuming until the breakpoint trap arrives.
for range 10 {
if err := sess.Continue(); err != nil {
t.Fatalf("Continue: %v", err)
}
if sess.Exited() {
t.Fatal("debuggee exited instead of trapping on the breakpoint")
}
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if bp := bm.HandleTrap(&regs); bp != nil {
if bp.Addr != entry {
t.Fatalf("trap at %#x, want %#x", bp.Addr, entry)
}
return // trap on the entry breakpoint: the whole flow works
}
}
t.Fatal("no breakpoint trap after 10 resumes")
}
+11 -2
View File
@@ -14,17 +14,26 @@ import (
"strings" "strings"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/verify" "sourcedock.dev/petrbalvin/gasm-sdk/verify"
) )
// buildGasm produces the gasm binary the debugger spawns as its debuggee. // buildGasm produces the gasm binary the debugger spawns as its debuggee.
// Every live ptrace test funnels through here, so this is also where the
// deliberate-run boundary sits: under -short (the push pipeline's mode) the
// live sessions skip, because a real debuggee's launch handshake needs the
// machine to itself and a starved single-core runner turns each one into a
// timeout that burns the step's whole budget. The local test gate and the
// dispatched workflows run them in full.
func buildGasm(t *testing.T) string { func buildGasm(t *testing.T) string {
t.Helper() t.Helper()
if testing.Short() {
t.Skip("live ptrace session: skipped in -short mode")
}
if p := os.Getenv("GASM_TEST_BIN"); p != "" { if p := os.Getenv("GASM_TEST_BIN"); p != "" {
return p return p
} }
bin := filepath.Join(t.TempDir(), "gasm") bin := filepath.Join(t.TempDir(), "gasm")
cmd := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm") cmd := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm")
out, err := cmd.CombinedOutput() out, err := cmd.CombinedOutput()
if err != nil { if err != nil {
t.Fatalf("build gasm: %v: %s", err, out) t.Fatalf("build gasm: %v: %s", err, out)
+50 -27
View File
@@ -91,6 +91,16 @@ func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec s
if _, err := os.Stat(readyFile); err == nil { if _, err := os.Stat(readyFile); err == nil {
break break
} }
// A debuggee that died before signalling readiness (unknown
// function, unparseable source) writes its failure notice to the
// handshake directory; read it and fail fast. The poll never
// waits on the child: a wait here could consume the SIGSTOP park
// that waitStopped below must receive, hanging the launch.
if err := s.deadReason(); err != nil {
cmd.Wait()
os.RemoveAll(tmpDir)
return nil, nil, err
}
time.Sleep(5 * time.Millisecond) time.Sleep(5 * time.Millisecond)
} }
@@ -131,6 +141,18 @@ func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec s
return s, bufAddrs, nil return s, bufAddrs, nil
} }
// deadReason reports the debuggee's own failure notice, the file its
// failure paths write before exiting. A debuggee killed without a notice
// (a crash, SIGKILL) surfaces through waitStopped after the poll instead,
// which is why the poll's budget stays finite.
func (s *Session) deadReason() error {
data, err := os.ReadFile(filepath.Join(s.tmpDir, "dead"))
if err != nil {
return nil
}
return fmt.Errorf("debug: debuggee failed before signalling readiness: %s", strings.TrimSpace(string(data)))
}
// waitStopped consumes ptrace-stop events until one the debugger cares // waitStopped consumes ptrace-stop events until one the debugger cares
// about arrives: SIGTRAP (a breakpoint or a completed single-step), the // about arrives: SIGTRAP (a breakpoint or a completed single-step), the
// debuggee's own SIGSTOP, or a genuine signal-delivery-stop. A Go tracee's // debuggee's own SIGSTOP, or a genuine signal-delivery-stop. A Go tracee's
@@ -219,40 +241,41 @@ func (s *Session) Poke(addr, val uint64) error {
return nil return nil
} }
// ReadMemory reads len bytes from the debuggee's memory at addr. // ReadMemory reads len bytes from the debuggee's memory at addr. The read
// covers exactly the requested range: the old word-at-a-time loop read a
// whole 8-byte word for the final partial word, so a request that ended
// inside the last mapped page failed whenever the following page was
// unmapped, even though every requested byte was readable.
func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) { func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) {
out := make([]byte, length) out := make([]byte, length)
for i := 0; i < length; i += 8 { mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
word, err := s.Peek(addr + uint64(i)) if err != nil {
if err != nil { return out, fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
return out[:i], err }
} defer mem.Close()
for j := 0; j < 8 && i+j < length; j++ { n, err := mem.ReadAt(out, int64(addr))
out[i+j] = byte(word >> (8 * j)) if err != nil {
} return out[:n], fmt.Errorf("debug: read mem %#x: %w", addr, err)
} }
return out, nil return out, nil
} }
// WriteMemory writes bytes to the debuggee's memory at addr. // WriteMemory writes bytes to the debuggee's memory at addr. The write
// covers exactly the given bytes: /proc/pid/mem accepts writes of any
// length at any offset, so the word loop's read-modify-write of the final
// partial word (which read past the requested range and failed on an
// unmapped following page) is unnecessary.
func (s *Session) WriteMemory(addr uint64, data []byte) error { func (s *Session) WriteMemory(addr uint64, data []byte) error {
for i := 0; i < len(data); i += 8 { if len(data) == 0 {
end := min(i+8, len(data)) return nil
var word uint64 }
for j := 0; j < end-i; j++ { mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_WRONLY, 0)
word |= uint64(data[i+j]) << (8 * j) if err != nil {
} return fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
if end-i < 8 { }
existing, err := s.Peek(addr + uint64(i)) defer mem.Close()
if err != nil { if _, err := mem.WriteAt(data, int64(addr)); err != nil {
return err return fmt.Errorf("debug: write mem %#x: %w", addr, err)
}
mask := ^((uint64(1) << (8 * (end - i))) - 1)
word = (existing & mask) | word
}
if err := s.Poke(addr+uint64(i), word); err != nil {
return err
}
} }
return nil return nil
} }
+98
View File
@@ -0,0 +1,98 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
// Regs holds the full general-purpose register set of a traced process
// (the FreeBSD amd64 struct reg layout, sys/x86/include/reg.h). FreeBSD
// reports segment selectors (FS/GS/ES/DS), not the bases the Linux ptrace
// surface carries, and has no ORIG_RAX slot.
type Regs struct {
R15 uint64
R14 uint64
R13 uint64
R12 uint64
RBP uint64
RBX uint64
R11 uint64
R10 uint64
R9 uint64
R8 uint64
RAX uint64
RCX uint64
RDX uint64
RSI uint64
RDI uint64
RIP uint64
CS uint64
RFLAGS uint64
RSP uint64
SS uint64
FS uint64
GS uint64
DS uint64
ES uint64
}
// GetPC returns the program counter.
func (r *Regs) GetPC() uint64 { return r.RIP }
// SetPC sets the program counter.
func (r *Regs) SetPC(pc uint64) { r.RIP = pc }
// GetSP returns the stack pointer.
func (r *Regs) GetSP() uint64 { return r.RSP }
// RegValue returns the value of the named register, or false if unknown.
func (r *Regs) RegValue(name string) (uint64, bool) {
switch name {
case "rax", "eax", "ax", "al":
return r.RAX, true
case "rbx", "ebx", "bx", "bl":
return r.RBX, true
case "rcx", "ecx", "cx", "cl":
return r.RCX, true
case "rdx", "edx", "dx", "dl":
return r.RDX, true
case "rsi", "esi", "si":
return r.RSI, true
case "rdi", "edi", "di":
return r.RDI, true
case "rbp", "ebp", "bp":
return r.RBP, true
case "rsp", "esp", "sp":
return r.RSP, true
case "r8":
return r.R8, true
case "r9":
return r.R9, true
case "r10":
return r.R10, true
case "r11":
return r.R11, true
case "r12":
return r.R12, true
case "r13":
return r.R13, true
case "r14":
return r.R14, true
case "r15":
return r.R15, true
case "rip", "eip":
return r.RIP, true
default:
return 0, false
}
}
// breakpointInsn is the software breakpoint instruction.
var breakpointInsn = []byte{0xCC} // INT3
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap. INT3 leaves the hardware PC on the following instruction (Intel
// SDM vol 3, "Debug Exceptions") and the FreeBSD T_BPTFLT path delivers
// that frame unmodified (sys/amd64/amd64/trap.c), so the trap address is
// PC-1, the same correction the Linux side applies.
const breakpointPCAdjust = 1
+139
View File
@@ -0,0 +1,139 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && arm64
package debug
// Regs holds the full general-purpose register set of a traced process
// (the FreeBSD arm64 struct reg layout, sys/arm64/include/reg.h: x[30], lr,
// sp, elr, spsr).
type Regs struct {
X0 uint64
X1 uint64
X2 uint64
X3 uint64
X4 uint64
X5 uint64
X6 uint64
X7 uint64
X8 uint64
X9 uint64
X10 uint64
X11 uint64
X12 uint64
X13 uint64
X14 uint64
X15 uint64
X16 uint64
X17 uint64
X18 uint64
X19 uint64
X20 uint64
X21 uint64
X22 uint64
X23 uint64
X24 uint64
X25 uint64
X26 uint64
X27 uint64
X28 uint64
X29 uint64 // FP (frame pointer)
X30 uint64 // LR (link register)
SP uint64
PC uint64
PSTATE uint64
}
// GetPC returns the program counter.
func (r *Regs) GetPC() uint64 { return r.PC }
// SetPC sets the program counter.
func (r *Regs) SetPC(pc uint64) { r.PC = pc }
// GetSP returns the stack pointer.
func (r *Regs) GetSP() uint64 { return r.SP }
// RegValue returns the value of the named register, or false if unknown.
func (r *Regs) RegValue(name string) (uint64, bool) {
switch name {
case "x0":
return r.X0, true
case "x1":
return r.X1, true
case "x2":
return r.X2, true
case "x3":
return r.X3, true
case "x4":
return r.X4, true
case "x5":
return r.X5, true
case "x6":
return r.X6, true
case "x7":
return r.X7, true
case "x8":
return r.X8, true
case "x9":
return r.X9, true
case "x10":
return r.X10, true
case "x11":
return r.X11, true
case "x12":
return r.X12, true
case "x13":
return r.X13, true
case "x14":
return r.X14, true
case "x15":
return r.X15, true
case "x16":
return r.X16, true
case "x17":
return r.X17, true
case "x18":
return r.X18, true
case "x19":
return r.X19, true
case "x20":
return r.X20, true
case "x21":
return r.X21, true
case "x22":
return r.X22, true
case "x23":
return r.X23, true
case "x24":
return r.X24, true
case "x25":
return r.X25, true
case "x26":
return r.X26, true
case "x27":
return r.X27, true
case "x28":
return r.X28, true
case "x29", "fp":
return r.X29, true
case "x30", "lr":
return r.X30, true
case "sp":
return r.SP, true
case "pc":
return r.PC, true
default:
return 0, false
}
}
// breakpointInsn is the software breakpoint instruction (BRK #0).
var breakpointInsn = []byte{0x00, 0x00, 0x20, 0xD4} // BRK #0
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap: 0. The BRK synchronous exception leaves ELR_EL0 on the BRK
// itself (ARM DDI 0487), and the FreeBSD EXCP_BRKPT_EL0 handler delivers
// the frame's elr unmodified (sys/arm64/arm64/trap.c), so the trap address
// is the PC as reported.
const breakpointPCAdjust = 0
+134
View File
@@ -0,0 +1,134 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && riscv64
package debug
// Regs holds the full general-purpose register set of a traced process
// (the FreeBSD riscv64 struct reg layout: ra, sp, gp, tp, t0-t6, s0-s11,
// a0-a7, sepc, sstatus).
type Regs struct {
PC uint64 // sepc
Ra uint64 // x1 (return address)
Sp uint64 // x2
Gp uint64 // x3
Tp uint64 // x4
T0 uint64 // x5
T1 uint64 // x6
T2 uint64 // x7
S0 uint64 // x8 (frame pointer)
S1 uint64 // x9
A0 uint64 // x10
A1 uint64 // x11
A2 uint64 // x12
A3 uint64 // x13
A4 uint64 // x14
A5 uint64 // x15
A6 uint64 // x16
A7 uint64 // x17
S2 uint64 // x18
S3 uint64 // x19
S4 uint64 // x20
S5 uint64 // x21
S6 uint64 // x22
S7 uint64 // x23
S8 uint64 // x24
S9 uint64 // x25
S10 uint64 // x26
S11 uint64 // x27
T3 uint64 // x28
T4 uint64 // x29
T5 uint64 // x30
T6 uint64 // x31
}
// GetPC returns the program counter.
func (r *Regs) GetPC() uint64 { return r.PC }
// SetPC sets the program counter.
func (r *Regs) SetPC(pc uint64) { r.PC = pc }
// GetSP returns the stack pointer.
func (r *Regs) GetSP() uint64 { return r.Sp }
// RegValue returns the value of the named register, or false if unknown.
func (r *Regs) RegValue(name string) (uint64, bool) {
switch name {
case "pc":
return r.PC, true
case "ra", "x1":
return r.Ra, true
case "sp", "x2":
return r.Sp, true
case "gp", "x3":
return r.Gp, true
case "tp", "x4":
return r.Tp, true
case "t0", "x5":
return r.T0, true
case "t1", "x6":
return r.T1, true
case "t2", "x7":
return r.T2, true
case "s0", "fp", "x8":
return r.S0, true
case "s1", "x9":
return r.S1, true
case "a0", "x10":
return r.A0, true
case "a1", "x11":
return r.A1, true
case "a2", "x12":
return r.A2, true
case "a3", "x13":
return r.A3, true
case "a4", "x14":
return r.A4, true
case "a5", "x15":
return r.A5, true
case "a6", "x16":
return r.A6, true
case "a7", "x17":
return r.A7, true
case "s2", "x18":
return r.S2, true
case "s3", "x19":
return r.S3, true
case "s4", "x20":
return r.S4, true
case "s5", "x21":
return r.S5, true
case "s6", "x22":
return r.S6, true
case "s7", "x23":
return r.S7, true
case "s8", "x24":
return r.S8, true
case "s9", "x25":
return r.S9, true
case "s10", "x26":
return r.S10, true
case "s11", "x27":
return r.S11, true
case "t3", "x28":
return r.T3, true
case "t4", "x29":
return r.T4, true
case "t5", "x30":
return r.T5, true
case "t6", "x31":
return r.T6, true
default:
return 0, false
}
}
// breakpointInsn is the software breakpoint instruction (EBREAK).
var breakpointInsn = []byte{0x73, 0x00, 0x10, 0x00} // ebreak
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap: 0. The EBREAK synchronous exception leaves sepc on the ebreak
// itself (RISC-V privileged architecture), so the trap address is the PC as
// reported.
const breakpointPCAdjust = 0
+63 -15
View File
@@ -1,7 +1,7 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause // SPDX-License-Identifier: BSD-3-Clause
//go:build linux //go:build linux || (freebsd && (amd64 || arm64 || riscv64))
package debug package debug
@@ -71,7 +71,12 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
case "step", "s": case "step", "s":
n := 1 n := 1
if len(parts) > 1 { if len(parts) > 1 {
n, _ = strconv.Atoi(parts[1]) v, err := strconv.Atoi(parts[1])
if err != nil || v < 0 {
fmt.Printf("invalid count: %s\n", parts[1])
continue
}
n = v
} }
for range n { for range n {
if s.Exited() { if s.Exited() {
@@ -82,8 +87,13 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Println(err) fmt.Println(err)
break break
} }
if sig := s.LastSignal(); sig != 0 {
regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
break
}
} }
if !s.Exited() { if !s.Exited() && s.LastSignal() == 0 {
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
pc := regs.GetPC() pc := regs.GetPC()
text, _, _ := s.Disassemble(pc) text, _, _ := s.Disassemble(pc)
@@ -106,16 +116,16 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
} }
if err := s.Continue(); err != nil { if err := s.Continue(); err != nil {
fmt.Println(err) fmt.Println(err)
bm.Clear(afterAddr) clearNextBp(bm, afterAddr)
continue continue
} }
if s.Exited() { if s.Exited() {
bm.Clear(afterAddr) clearNextBp(bm, afterAddr)
fmt.Println("debuggee exited") fmt.Println("debuggee exited")
continue continue
} }
if sig := s.LastSignal(); sig != 0 { if sig := s.LastSignal(); sig != 0 {
bm.Clear(afterAddr) clearNextBp(bm, afterAddr)
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC()) fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
continue continue
@@ -125,12 +135,17 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
// snapshot, and a stale SetRegs would clobber live state. // snapshot, and a stale SetRegs would clobber live state.
regs, _ = s.GetRegs() regs, _ = s.GetRegs()
bm.HandleTrap(&regs) bm.HandleTrap(&regs)
bm.Clear(afterAddr) clearNextBp(bm, afterAddr)
} else { } else {
if err := s.Step(); err != nil { if err := s.Step(); err != nil {
fmt.Println(err) fmt.Println(err)
continue continue
} }
if sig := s.LastSignal(); sig != 0 {
regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
continue
}
} }
if !s.Exited() { if !s.Exited() {
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
@@ -156,16 +171,16 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
} }
if err := s.Continue(); err != nil { if err := s.Continue(); err != nil {
fmt.Println(err) fmt.Println(err)
bm.Clear(retAddr) clearNextBp(bm, retAddr)
continue continue
} }
if s.Exited() { if s.Exited() {
bm.Clear(retAddr) clearNextBp(bm, retAddr)
fmt.Println("debuggee exited") fmt.Println("debuggee exited")
continue continue
} }
if sig := s.LastSignal(); sig != 0 { if sig := s.LastSignal(); sig != 0 {
bm.Clear(retAddr) clearNextBp(bm, retAddr)
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC()) fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
continue continue
@@ -174,7 +189,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
// does: HandleTrap must see the PC the trap left behind. // does: HandleTrap must see the PC the trap left behind.
regs, _ = s.GetRegs() regs, _ = s.GetRegs()
bm.HandleTrap(&regs) bm.HandleTrap(&regs)
bm.Clear(retAddr) clearNextBp(bm, retAddr)
if s.Exited() { if s.Exited() {
fmt.Println("debuggee exited") fmt.Println("debuggee exited")
} else { } else {
@@ -214,6 +229,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
break break
} }
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
trapPC := regs.GetPC()
if bp := bm.HandleTrap(&regs); bp != nil { if bp := bm.HandleTrap(&regs); bp != nil {
// Execute the instruction under the restored breakpoint // Execute the instruction under the restored breakpoint
// so the next continue cannot re-trap on the same // so the next continue cannot re-trap on the same
@@ -229,6 +245,13 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Printf("breakpoint hit: %s (func+%#x)\n", name, bp.Addr-codeBase-uint64(funcOffset)) fmt.Printf("breakpoint hit: %s (func+%#x)\n", name, bp.Addr-codeBase-uint64(funcOffset))
break break
} }
// The trap matched no breakpoint of ours. When the PC still
// stands on the trapping instruction, resuming would re-execute
// it and trap forever, so surface the stop instead of spinning.
if TrapStray(s, reason, trapPC) {
fmt.Printf("SIGTRAP at %#x matches no breakpoint; the PC did not advance\n", trapPC-uint64(breakpointPCAdjust))
break
}
} }
case "break", "b": case "break", "b":
@@ -326,6 +349,10 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
length := 64 length := 64
if len(parts) > 1 { if len(parts) > 1 {
addr, _ = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels) addr, _ = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
if addr == 0 {
fmt.Printf("unknown address: %s\n", parts[1])
continue
}
} }
if len(parts) > 2 { if len(parts) > 2 {
// A malformed or non-positive length would panic // A malformed or non-positive length would panic
@@ -399,9 +426,13 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
case "disas", "u": case "disas", "u":
n := 5 n := 5
if len(parts) > 1 { if len(parts) > 1 {
n, _ = strconv.Atoi(parts[1]) v, err := strconv.Atoi(parts[1])
if n <= 0 { if err != nil {
n = 5 fmt.Printf("invalid count: %s\n", parts[1])
continue
}
if v > 0 {
n = v
} }
} }
regs, _ := s.GetRegs() regs, _ := s.GetRegs()
@@ -496,10 +527,18 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
typ = WatchRead typ = WatchRead
case "w": case "w":
typ = WatchWrite typ = WatchWrite
default:
fmt.Printf("unknown watchpoint type: %s (want r or w)\n", parts[2])
continue
} }
} }
if len(parts) > 3 { if len(parts) > 3 {
size, _ = strconv.Atoi(parts[3]) v, err := strconv.Atoi(parts[3])
if err != nil || v <= 0 {
fmt.Printf("invalid size: %s\n", parts[3])
continue
}
size = v
} }
slot := s.FindFreeWatchpointSlot() slot := s.FindFreeWatchpointSlot()
if slot < 0 { if slot < 0 {
@@ -543,6 +582,15 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
s.Kill() s.Kill()
} }
// clearNextBp removes one of the temporary breakpoints the next and finish
// commands plant, reporting a failure instead of silently leaving the trap
// instruction behind in the debuggee.
func clearNextBp(bm *Breakpoints, addr uint64) {
if err := bm.Clear(addr); err != nil {
fmt.Printf("cannot remove temporary breakpoint at %#x: %v\n", addr, err)
}
}
func hexDump(addr uint64, data []byte) { func hexDump(addr uint64, data []byte) {
for i := 0; i < len(data); i += 16 { for i := 0; i < len(data); i += 16 {
end := min(i+16, len(data)) end := min(i+16, len(data))
+73
View File
@@ -0,0 +1,73 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && (amd64 || arm64 || riscv64)
package debug
import (
"encoding/binary"
"syscall"
"unsafe"
"golang.org/x/sys/unix"
)
// FreeBSD TRAP_* si_code values (sys/signal.h). A breakpoint (INT3, BRK,
// EBREAK) arrives as TRAP_BRKPT on every supported architecture; TRAP_TRACE
// is shared by the completed single-step and the hardware watchpoint hit,
// so the watchpoint layer disambiguates from the debug registers.
const (
trapBRKPT = 1 // TRAP_BRKPT
trapTRACE = 2 // TRAP_TRACE
)
// StopReason describes why the debuggee stopped.
type StopReason int
const (
StopNone StopReason = iota
StopBreakpoint // software breakpoint hit
StopWatchpoint // hardware watchpoint triggered
StopSingleStep // single-step completed
StopSignal // stopped by a signal
StopExited // process exited
)
// StopInfo returns the reason the debuggee stopped and the faulting address
// (for watchpoints, the watched address that was accessed). FreeBSD has no
// PTRACE_GETSIGINFO; the stop's signal information comes from PT_LWPINFO,
// whose pl_siginfo carries the siginfo the kernel delivered. A ptrace stop
// with no signal behind it (a completed single-step, the initial attach)
// fills no siginfo at all.
func (s *Session) StopInfo() (StopReason, uint64) {
if s.exited {
return StopExited, 0
}
var info unix.PtraceLwpInfoStruct
if err := unix.PtraceLwpInfo(s.pid, &info); err != nil {
return StopNone, 0
}
// The siginfo layout is the FreeBSD siginfo_t: three leading ints
// (signo, errno, code), then the union, 8-byte aligned, whose _fault
// member puts the address at byte offset 16. The read is byte-wise
// because the blob's alignment is not guaranteed.
si := (*[64]byte)(unsafe.Pointer(&info.Siginfo))
signo := int32(binary.LittleEndian.Uint32(si[0:4]))
code := int32(binary.LittleEndian.Uint32(si[8:12]))
switch {
case signo == 0:
// A pure ptrace stop: single-step completion, attach, or the
// events the kernel resolves internally.
return StopSingleStep, 0
case signo != int32(syscall.SIGTRAP):
return StopSignal, uint64(code)
case code == trapBRKPT:
return StopBreakpoint, 0
case code == trapTRACE:
addr := binary.LittleEndian.Uint64(si[16:24])
return archStopTrace(s, addr)
default:
return StopSingleStep, 0
}
}
+206
View File
@@ -0,0 +1,206 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && (amd64 || arm64 || riscv64)
package debug
import (
"encoding/hex"
"fmt"
"os"
"runtime"
"strconv"
"strings"
"syscall"
"unsafe"
"golang.org/x/sys/unix"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
// mapRWX maps code into a read-write-execute region.
func mapRWX(code []byte) ([]byte, error) {
const pageSize = 4096
size := (len(code) + pageSize - 1) &^ (pageSize - 1)
mem, err := syscall.Mmap(-1, 0, size,
syscall.PROT_READ|syscall.PROT_WRITE|syscall.PROT_EXEC,
syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, err
}
copy(mem, code)
return mem, nil
}
// setupBuffers allocates buffers in the debuggee's memory.
func setupBuffers(spec string, args []byte, tmpDir string) ([]byte, error) {
type bufSpec struct {
name string
size int
pattern string
}
var specs []bufSpec
for part := range strings.SplitSeq(spec, ",") {
fields := strings.SplitN(part, ":", 3)
if len(fields) != 3 {
continue
}
size, err := strconv.Atoi(fields[1])
if err != nil || size <= 0 {
continue
}
specs = append(specs, bufSpec{name: fields[0], size: size, pattern: fields[2]})
}
if len(specs) == 0 {
return args, nil
}
var bufAddrs []uint64
for _, s := range specs {
buf, err := syscall.Mmap(-1, 0, s.size,
syscall.PROT_READ|syscall.PROT_WRITE,
syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, fmt.Errorf("mmap buffer %s: %w", s.name, err)
}
fillBuffer(buf, s.pattern)
bufAddrs = append(bufAddrs, uint64(uintptr(unsafe.Pointer(&buf[0]))))
}
addrFile, err := os.Create(tmpDir + "/bufaddrs")
if err != nil {
return nil, err
}
for _, addr := range bufAddrs {
fmt.Fprintf(addrFile, "%d\n", addr)
}
addrFile.Close()
return args, nil
}
// fillBuffer fills a buffer with the specified pattern.
func fillBuffer(buf []byte, pattern string) {
switch pattern {
case "zero":
case "ones":
for i := range buf {
buf[i] = 0xFF
}
case "seq":
for i := range buf {
buf[i] = byte(i)
}
default:
if data, err := hex.DecodeString(pattern); err == nil && len(data) > 0 {
for i := range buf {
buf[i] = data[i%len(data)]
}
}
}
}
// RunTarget is the debuggee entry point (gasm debug --target). A failure
// is marked in the handshake directory before the process exits, so the
// debugger's readiness poll fails fast on a dead debuggee instead of
// waiting out its whole budget.
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
err := runTarget(asmPath, funcName, argsFile, tmpDir)
if err != nil {
markDead(tmpDir, err.Error())
}
return err
}
func runTarget(asmPath, funcName, argsFile, tmpDir string) error {
src, err := os.ReadFile(asmPath)
if err != nil {
return fmt.Errorf("debug target: %w", err)
}
file, errs := parser.Parse(asmPath, string(src))
if len(errs) > 0 {
return fmt.Errorf("debug target: parse: %v", errs[0])
}
img, err := asm.AssembleFile(file)
if err != nil {
return fmt.Errorf("debug target: assemble: %w", err)
}
var fl *asm.FuncLayout
for i := range img.Funcs {
if img.Funcs[i].Name == funcName {
fl = &img.Funcs[i]
break
}
}
if fl == nil {
return fmt.Errorf("debug target: function %q not found", funcName)
}
code := img.Bytes()
exec, err := mapRWX(code)
if err != nil {
return fmt.Errorf("debug target: mmap: %w", err)
}
codeBase := uintptr(unsafe.Pointer(&exec[0]))
if err := os.WriteFile(tmpDir+"/codebase", []byte(fmt.Sprintf("%d", codeBase)), 0o644); err != nil {
return fmt.Errorf("debug target: write codebase: %w", err)
}
meta := fmt.Sprintf("%d %d %d", fl.Offset, fl.Size, fl.Args)
os.WriteFile(tmpDir+"/funcmeta", []byte(meta), 0o644)
labelsFile, _ := os.Create(tmpDir + "/labels")
if labelsFile != nil {
for label, off := range fl.Labels {
fmt.Fprintf(labelsFile, "%s %d\n", label, off)
}
labelsFile.Close()
}
args, err := os.ReadFile(argsFile)
if err != nil {
return fmt.Errorf("debug target: read args: %w", err)
}
if len(args) < fl.Args {
padded := make([]byte, fl.Args)
copy(padded, args)
args = padded
}
bufSpecFile := tmpDir + "/bufspec"
if bufSpec, err := os.ReadFile(bufSpecFile); err == nil && len(bufSpec) > 0 {
args, err = setupBuffers(string(bufSpec), args, tmpDir)
if err != nil {
return fmt.Errorf("debug target: setup buffers: %w", err)
}
}
runtime.LockOSThread()
if _, _, errno := unix.RawSyscall(unix.SYS_PTRACE, uintptr(unix.PT_TRACE_ME), 0, 0); errno != 0 {
return fmt.Errorf("debug target: PT_TRACE_ME: %v", errno)
}
os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644)
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
os.WriteFile(tmpDir+"/entry", []byte("ok"), 0o644)
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
fnAddr := codeBase + uintptr(fl.Offset)
stackArgs := make([]byte, fl.Args)
copy(stackArgs, args)
if _, callErr := verify.Call(fnAddr, stackArgs); callErr != nil {
os.Exit(1)
}
// Success returns to the caller, which exits with status 0; the JIT
// code has already run to its own trampoline by the time Call returns.
return nil
}
+15 -4
View File
@@ -12,13 +12,24 @@ import (
"syscall" "syscall"
"unsafe" "unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm" "sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/verify" "sourcedock.dev/petrbalvin/gasm-sdk/verify"
) )
// RunTarget is the debuggee entry point (gasm debug --target). // RunTarget is the debuggee entry point (gasm debug --target). A failure
// is marked in the handshake directory before the process exits, so the
// debugger's readiness poll fails fast on a dead debuggee instead of
// waiting out its whole budget.
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error { func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
err := runTarget(asmPath, funcName, argsFile, tmpDir)
if err != nil {
markDead(tmpDir, err.Error())
}
return err
}
func runTarget(asmPath, funcName, argsFile, tmpDir string) error {
src, err := os.ReadFile(asmPath) src, err := os.ReadFile(asmPath)
if err != nil { if err != nil {
return fmt.Errorf("debug target: %w", err) return fmt.Errorf("debug target: %w", err)
+15 -4
View File
@@ -12,13 +12,24 @@ import (
"syscall" "syscall"
"unsafe" "unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm" "sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/verify" "sourcedock.dev/petrbalvin/gasm-sdk/verify"
) )
// RunTarget is the debuggee entry point (gasm debug --target). // RunTarget is the debuggee entry point (gasm debug --target). A failure
// is marked in the handshake directory before the process exits, so the
// debugger's readiness poll fails fast on a dead debuggee instead of
// waiting out its whole budget.
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error { func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
err := runTarget(asmPath, funcName, argsFile, tmpDir)
if err != nil {
markDead(tmpDir, err.Error())
}
return err
}
func runTarget(asmPath, funcName, argsFile, tmpDir string) error {
src, err := os.ReadFile(asmPath) src, err := os.ReadFile(asmPath)
if err != nil { if err != nil {
return fmt.Errorf("debug target: %w", err) return fmt.Errorf("debug target: %w", err)
+15 -4
View File
@@ -12,13 +12,24 @@ import (
"syscall" "syscall"
"unsafe" "unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm" "sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/verify" "sourcedock.dev/petrbalvin/gasm-sdk/verify"
) )
// RunTarget is the debuggee entry point (gasm debug --target). // RunTarget is the debuggee entry point (gasm debug --target). A failure
// is marked in the handshake directory before the process exits, so the
// debugger's readiness poll fails fast on a dead debuggee instead of
// waiting out its whole budget.
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error { func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
err := runTarget(asmPath, funcName, argsFile, tmpDir)
if err != nil {
markDead(tmpDir, err.Error())
}
return err
}
func runTarget(asmPath, funcName, argsFile, tmpDir string) error {
src, err := os.ReadFile(asmPath) src, err := os.ReadFile(asmPath)
if err != nil { if err != nil {
return fmt.Errorf("debug target: %w", err) return fmt.Errorf("debug target: %w", err)
+15 -4
View File
@@ -12,13 +12,24 @@ import (
"syscall" "syscall"
"unsafe" "unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm" "sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-sdk/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/verify" "sourcedock.dev/petrbalvin/gasm-sdk/verify"
) )
// RunTarget is the debuggee entry point (gasm debug --target). // RunTarget is the debuggee entry point (gasm debug --target). A failure
// is marked in the handshake directory before the process exits, so the
// debugger's readiness poll fails fast on a dead debuggee instead of
// waiting out its whole budget.
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error { func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
err := runTarget(asmPath, funcName, argsFile, tmpDir)
if err != nil {
markDead(tmpDir, err.Error())
}
return err
}
func runTarget(asmPath, funcName, argsFile, tmpDir string) error {
src, err := os.ReadFile(asmPath) src, err := os.ReadFile(asmPath)
if err != nil { if err != nil {
return fmt.Errorf("debug target: %w", err) return fmt.Errorf("debug target: %w", err)
+17 -1
View File
@@ -1,10 +1,26 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause // SPDX-License-Identifier: BSD-3-Clause
//go:build linux //go:build linux || (freebsd && (amd64 || arm64 || riscv64))
package debug package debug
import (
"os"
"path/filepath"
)
// markDead records the target's failure reason in the handshake directory,
// the death notice the debugger's readiness poll reads. The poll must not
// wait on the child while it polls: a wait there could consume the SIGSTOP
// park the debugger's own waitStopped must receive, hanging the launch, so
// the file is the only fast death notice that is safe to read. A debuggee
// killed without a notice (a crash, SIGKILL) surfaces through the ordinary
// wait after the poll instead.
func markDead(tmpDir, reason string) {
os.WriteFile(filepath.Join(tmpDir, "dead"), []byte(reason), 0o644)
}
// tracer abstracts the minimal ptrace operations needed by the breakpoint // tracer abstracts the minimal ptrace operations needed by the breakpoint
// manager and the stop-information helpers. The live implementation is // manager and the stop-information helpers. The live implementation is
// *Session (ptrace_linux_amd64.go); tests supply a mock. // *Session (ptrace_linux_amd64.go); tests supply a mock.
+200
View File
@@ -0,0 +1,200 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
import (
"fmt"
"unsafe"
"golang.org/x/sys/unix"
)
// Hardware watchpoint support via x86-64 debug registers (DR0-DR3, DR7),
// read and written as one blob through PT_GETDBREGS/PT_SETDBREGS. The
// FreeBSD struct dbreg is the raw DR file: dr[16], where DR0-DR3 are the
// address registers, DR6 the status and DR7 the control (sys/x86/include/
// reg.h; the DBREG_DRX accessor indexes the same array).
// dbreg mirrors FreeBSD's struct dbreg for PT_GETDBREGS/PT_SETDBREGS.
type dbreg struct {
Dr [16]uint64
}
// dbreg indices of the registers the watchpoint layer drives.
const (
drStatus = 6 // DR6: the trap status register
drControl = 7 // DR7: the debug control register
)
// WatchpointType selects what triggers the watchpoint.
type WatchpointType int
const (
WatchWrite WatchpointType = 1 // trigger on write
WatchRead WatchpointType = 3 // trigger on read or write
)
// maxWatchpoints reports the number of hardware watchpoint slots the
// architecture provides: four address registers, DR0-DR3.
func maxWatchpoints() int { return 4 }
// getDbRegs reads the debug register file of the stopped debuggee.
func (s *Session) getDbRegs() (*dbreg, error) {
var dr dbreg
if _, _, errno := unix.Syscall6(
unix.SYS_PTRACE,
uintptr(unix.PT_GETDBREGS),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(&dr)),
0, 0,
); errno != 0 {
return nil, errno
}
return &dr, nil
}
// setDbRegs writes the debug register file of the stopped debuggee.
func (s *Session) setDbRegs(dr *dbreg) error {
if _, _, errno := unix.Syscall6(
unix.SYS_PTRACE,
uintptr(unix.PT_SETDBREGS),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(dr)),
0, 0,
); errno != 0 {
return errno
}
return nil
}
// archStopTrace classifies a TRAP_TRACE stop. On amd64 the kernel
// delivers both the completed single-step and the debug-register hit
// through T_TRCTRAP with TRAP_TRACE (sys/amd64/amd64/trap.c), and DR6's
// B0-B3 bits name the watchpoint that fired. The bits are sticky ("the
// processor never clears DR6", Intel SDM vol 3, "Debug Registers"), so
// they are acknowledged here: cleared once read, or the next hit on a
// different slot would still see this slot's bit set and report this
// slot's address again.
func archStopTrace(s *Session, siAddr uint64) (StopReason, uint64) {
dr, err := s.getDbRegs()
if err != nil {
return StopSingleStep, 0
}
status := dr.Dr[drStatus] & 0xF
if status == 0 {
return StopSingleStep, 0
}
// Best effort: the write-back only fails for a dead debuggee, for
// which no further watchpoint can fire anyway.
dr.Dr[drStatus] &^= 0xF
_ = s.setDbRegs(dr)
for slot := range 4 {
if status&(1<<slot) != 0 && dr.Dr[slot] != 0 {
return StopWatchpoint, dr.Dr[slot]
}
}
return StopSingleStep, 0
}
// FindFreeWatchpointSlot returns the index of the first free watchpoint slot
// (0-3), or -1 if all four hardware watchpoints are in use.
func (s *Session) FindFreeWatchpointSlot() int {
for i := range 4 {
if !s.wpSlots[i] {
return i
}
}
return -1
}
// IsWatchpointSlotUsed reports whether slot (0-3) currently holds a watchpoint.
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot > 3 {
return false
}
return s.wpSlots[slot]
}
// SetWatchpoint installs a hardware watchpoint on the given address.
// DR7's encoding is architectural: a 2-bit local/global enable pair per
// slot at bit 2*slot, the R/W field at 16+4*slot and the length field at
// 18+4*slot (Intel SDM vol 3, "Debug Registers").
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
if slot < 0 || slot > 3 {
return fmt.Errorf("debug: watchpoint slot must be 0-3")
}
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
var lenBits uint64
switch size {
case 1:
lenBits = 0
case 2:
lenBits = 1
case 4:
lenBits = 3
case 8:
lenBits = 2
default:
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
}
dr, err := s.getDbRegs()
if err != nil {
return fmt.Errorf("debug: read debug registers: %w", err)
}
dr.Dr[slot] = addr
dr7 := dr.Dr[drControl]
enableBit := uint64(1) << (2 * slot)
rwBits := uint64(typ) << (16 + 4*slot)
lenField := lenBits << (18 + 4*slot)
mask := ^((uint64(1) << (2 * slot)) | (uint64(3) << (16 + 4*slot)) | (uint64(3) << (18 + 4*slot)))
dr.Dr[drControl] = (dr7 & mask) | enableBit | rwBits | lenField
if err := s.setDbRegs(dr); err != nil {
return fmt.Errorf("debug: set debug registers: %w", err)
}
s.wpSlots[slot] = true
return nil
}
// ClearWatchpoint removes a hardware watchpoint.
func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot > 3 {
return fmt.Errorf("debug: watchpoint slot must be 0-3")
}
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
dr, err := s.getDbRegs()
if err != nil {
return err
}
dr.Dr[slot] = 0
dr.Dr[drControl] &^= uint64(1) << (2 * slot)
if err := s.setDbRegs(dr); err != nil {
return err
}
s.wpSlots[slot] = false
return nil
}
// ClearAllWatchpoints removes all hardware watchpoints.
func (s *Session) ClearAllWatchpoints() error {
for slot := range maxWatchpoints() {
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
}
}
}
return nil
}
+216
View File
@@ -0,0 +1,216 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && arm64
package debug
import (
"fmt"
"math/bits"
"unsafe"
"golang.org/x/sys/unix"
)
// Hardware watchpoint support via arm64 debug registers, read and written
// as one blob through PT_GETDBREGS/PT_SETDBREGS. The FreeBSD struct dbreg
// (sys/arm64/include/reg.h) opens with the debug-facility header and then
// carries 16 breakpoint and 16 watchpoint pairs of {address, control}.
// dbreg mirrors FreeBSD's struct dbreg for PT_GETDBREGS/PT_SETDBREGS.
type dbreg struct {
DbDebugVer uint8
DbNbkpts uint8
DbNwtpts uint8
_ [5]byte
DbBreakregs [16]struct {
Addr uint64
Ctrl uint32
_ uint32
}
DbWatchregs [16]struct {
Addr uint64
Ctrl uint32
_ uint32
}
}
// WatchpointType selects what triggers the watchpoint.
type WatchpointType int
const (
WatchWrite WatchpointType = 1 // trigger on write
WatchRead WatchpointType = 3 // trigger on read or write
)
// maxWatchpoints reports the number of hardware watchpoint slots the
// architecture provides: DBGWVR0-DBGWCR15.
func maxWatchpoints() int { return 16 }
// getDbRegs reads the debug register file of the stopped debuggee.
func (s *Session) getDbRegs() (*dbreg, error) {
var dr dbreg
if _, _, errno := unix.Syscall6(
unix.SYS_PTRACE,
uintptr(unix.PT_GETDBREGS),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(&dr)),
0, 0,
); errno != 0 {
return nil, errno
}
return &dr, nil
}
// setDbRegs writes the debug register file of the stopped debuggee.
func (s *Session) setDbRegs(dr *dbreg) error {
if _, _, errno := unix.Syscall6(
unix.SYS_PTRACE,
uintptr(unix.PT_SETDBREGS),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(dr)),
0, 0,
); errno != 0 {
return errno
}
return nil
}
// archStopTrace classifies a TRAP_TRACE stop. On arm64 the kernel
// delivers both the software single step and the watchpoint hit through
// EXCP_SOFTSTP_EL0/EXCP_WATCHPT_EL0 with TRAP_TRACE (sys/arm64/arm64/
// trap.c); the watchpoint address rides the FAR register, so a stop whose
// reported address falls inside an armed watchpoint's byte range is a
// watchpoint and everything else is a single step. The byte range comes
// from DBGWCR's byte-address-select bits (ARM DDI 0487, DBGWCR<n>_EL1):
// bit i watches the address plus i, so the range spans the lowest set bit
// to the highest set bit inclusive.
func archStopTrace(s *Session, siAddr uint64) (StopReason, uint64) {
dr, err := s.getDbRegs()
if err != nil {
return StopSingleStep, 0
}
for slot := range 16 {
ctrl := uint64(dr.DbWatchregs[slot].Ctrl)
if ctrl&1 == 0 || dr.DbWatchregs[slot].Addr == 0 {
continue
}
bas := uint8((ctrl >> 5) & 0xFF)
if bas == 0 {
continue
}
lo := bits.TrailingZeros8(bas)
hi := 7 - bits.LeadingZeros8(bas)
addr := dr.DbWatchregs[slot].Addr
if siAddr >= addr+uint64(lo) && siAddr < addr+uint64(hi)+1 {
return StopWatchpoint, siAddr
}
}
return StopSingleStep, 0
}
// FindFreeWatchpointSlot returns the index of the first free watchpoint
// slot, or -1 if all of them are in use.
func (s *Session) FindFreeWatchpointSlot() int {
for i := range maxWatchpoints() {
if !s.wpSlots[i] {
return i
}
}
return -1
}
// IsWatchpointSlotUsed reports whether slot currently holds a watchpoint.
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot >= maxWatchpoints() {
return false
}
return s.wpSlots[slot]
}
// SetWatchpoint installs a hardware watchpoint on the given address. The
// control word is the architectural DBGWCR (ARM DDI 0487): bit 0 enables,
// bits 3-4 select the access type (10 store, 11 load+store) and bits 5-12
// are the byte-address select, so the watch stays 8-byte aligned and names
// its watched bytes through BAS.
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
if slot < 0 || slot >= maxWatchpoints() {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints()-1)
}
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
var bas uint64
switch size {
case 1:
bas = 0x01
case 2:
bas = 0x03
case 4:
bas = 0x0F
case 8:
bas = 0xFF
default:
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
}
dr, err := s.getDbRegs()
if err != nil {
return fmt.Errorf("debug: read debug registers: %w", err)
}
if uint8(slot) >= dr.DbNwtpts && dr.DbNwtpts != 0 {
return fmt.Errorf("debug: slot %d exceeds available watchpoints (%d)", slot, dr.DbNwtpts)
}
ctrl := uint64(1) // enable
switch typ {
case WatchWrite:
ctrl |= 2 << 3 // store only
case WatchRead:
ctrl |= 3 << 3 // load+store
}
ctrl |= bas << 5
dr.DbWatchregs[slot].Addr = addr
dr.DbWatchregs[slot].Ctrl = uint32(ctrl)
if err := s.setDbRegs(dr); err != nil {
return fmt.Errorf("debug: set debug registers: %w", err)
}
s.wpSlots[slot] = true
return nil
}
// ClearWatchpoint removes a hardware watchpoint.
func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot >= maxWatchpoints() {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints()-1)
}
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
dr, err := s.getDbRegs()
if err != nil {
return err
}
dr.DbWatchregs[slot].Addr = 0
dr.DbWatchregs[slot].Ctrl = 0
if err := s.setDbRegs(dr); err != nil {
return err
}
s.wpSlots[slot] = false
return nil
}
// ClearAllWatchpoints removes all hardware watchpoints.
func (s *Session) ClearAllWatchpoints() error {
for slot := range maxWatchpoints() {
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
}
}
}
return nil
}

Some files were not shown because too many files have changed in this diff Show More