Compare commits

..
51 Commits
Author SHA1 Message Date
petrbalvin 5a8e9acbf3 feat(arch): add the extended-instruction layer with SVE arithmetic
Test / test (push) Successful in 3m38s
2026-10-02 20:39:33 +02:00
petrbalvin 2747fce7d3 feat(asm): encode the arm64 system registers and structure loads 2026-10-02 20:39:26 +02:00
petrbalvin e02918c17b fix(ci): keep the push suite inside the runner's memory and time budget
Test / test (push) Successful in 2m55s
Assisted-by: GLM 5.3 Flash
2026-10-02 17:20:31 +02:00
petrbalvin 02a6359c1f ci: shrink the push pipeline to the affordable gate set
Test / test (push) Failing after 5m26s
Assisted-by: GLM 5.3 Flash
2026-10-02 16:47:52 +02:00
petrbalvin b4c1e133c0 ci: keep the GOOBJ link parity gate off the push pipeline
Test / test (push) Failing after 12m18s
Assisted-by: GLM 5.3 Flash
2026-10-02 16:15:02 +02:00
petrbalvin 1107928870 build(justfile): run the test recipes under the memory fence
Assisted-by: GLM 5.3 Flash
2026-10-02 16:15:02 +02:00
petrbalvin bc4ac93fd9 style(asm): reindent the evex comment gofmt asks for
Test / test (push) Failing after 21m29s
Assisted-by: GLM 5.3
2026-10-02 00:41:46 +02:00
petrbalvin c1bca7ce7e docs: record the development deltas in the changelog
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin fefb76beb9 docs(asm): correct the reference against the assemblers' behaviour
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin 42bc1669d7 feat(lsp): document directives on hover and widen completion
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin 69dcbec8ef feat(lint): eleven new rules over directives, data and addressing
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin f405cea5bc fix(cmd): stop the coverage run on a stray in-place trap
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin f57377abb9 fix(debug): handle mapping edges, stray traps and dying debuggees
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin dd1782c538 test(verify): gate GOOBJ link parity with cmd/link
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin 17cc49fee4 fix(asm): emit NOPTR data as its own symbol kind
Assisted-by: GLM 5.3
2026-10-02 00:40:43 +02:00
petrbalvin bafb2fd130 feat(asm): encode the amd64 and loong64 tails of the corpus testdata
Assisted-by: GLM 5.3
2026-10-02 00:40:43 +02:00
petrbalvin 2f679326c2 style(testdata): canonicalise forms_amd64.s
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 96f2dd65b4 test(lexer): fuzz the token stream invariants
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin ca887d3927 fix(parser): bound folding depth and macro expansion work
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 6570709226 fix(parser): peel stacked labels the way the formatter renders them
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 9a5217d9c1 fix(format): keep every token of a line in the canonical output
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 4d01bb3ecf build: rename the module to sourcedock.dev/petrbalvin/gasm-sdk
Test / test (push) Successful in 4m18s
2026-09-26 11:08:43 +02:00
petrbalvin 332c63e440 ci: compile-gate FreeBSD in the test pipeline
Test / test (push) Successful in 4m10s
Assisted-by: GLM 5.3 Flash
2026-09-25 21:46:49 +02:00
petrbalvin b306c210c6 feat(debug): port the debugger to FreeBSD
Assisted-by: GLM 5.3 Flash
2026-09-25 21:46:40 +02:00
petrbalvin b9015e1c2e fix(verify): make the executable mapping build on FreeBSD
Assisted-by: GLM 5.3 Flash
2026-09-25 21:46:31 +02:00
petrbalvin 26c5008136 fix(cmd): honour //go:build in the corpus audit
Test / test (push) Successful in 2m32s
Assisted-by: GLM 5.3 Flash
2026-09-23 21:03:16 +02:00
petrbalvin 74d6b90d69 fix(asm): read the arm64 move-wide immediate as an unsigned pattern
Assisted-by: GLM 5.3 Flash
2026-09-23 21:03:03 +02:00
petrbalvin 7b11c62f53 fix(asm): resolve negative numeric PC-relative jumps
Assisted-by: GLM 5.3 Flash
2026-09-23 21:02:50 +02:00
petrbalvin 8eed54b3da feat(lsp): quick fixes for the textflag include and the argument area
Test / test (push) Successful in 2m56s
Assisted-by: GLM 5.3 Flash
2026-09-23 20:23:35 +02:00
petrbalvin 4be16dcdf5 feat(lsp): resolve symbols across workspace files
Assisted-by: GLM 5.3 Flash
2026-09-23 20:21:58 +02:00
petrbalvin ded9cabdf4 fix(lsp): apply each rename edit to its own document
Assisted-by: GLM 5.3 Flash
2026-09-23 20:19:21 +02:00
petrbalvin cf6bc6987e fix(ci): pass the upload file to curl, not its interpolation
Test / test (push) Successful in 2m26s
Release / gates (push) Successful in 2m25s
Release / build (amd64, linux) (push) Successful in 1m15s
Release / build (arm64, linux) (push) Successful in 1m16s
Release / build (loong64, linux) (push) Successful in 1m16s
Release / build (riscv64, linux) (push) Successful in 1m16s
Release / release (push) Successful in 34s
2026-09-22 01:31:49 +02:00
petrbalvin ff7b1452b1 docs: name 0.35.0 as the supported release
Test / test (push) Successful in 2m33s
Release / gates (push) Successful in 2m29s
Release / build (amd64, linux) (push) Successful in 1m18s
Release / build (arm64, linux) (push) Successful in 1m20s
Release / build (loong64, linux) (push) Successful in 1m17s
Release / build (riscv64, linux) (push) Successful in 1m26s
Release / release (push) Failing after 35s
2026-09-22 00:52:56 +02:00
petrbalvin 517c1cea25 chore: prepare release v0.35.0
Test / test (push) Successful in 2m33s
Release / gates (push) Failing after 46s
Release / build (amd64, linux) (push) Skipped
Release / build (arm64, linux) (push) Skipped
Release / build (loong64, linux) (push) Skipped
Release / build (riscv64, linux) (push) Skipped
Release / release (push) Skipped
2026-09-22 00:44:10 +02:00
petrbalvin a3e3010e0f fix(cmd): resolve the runtime header test GOROOT from the go command
Test / test (push) Successful in 2m39s
2026-09-21 22:46:07 +02:00
petrbalvin 057c4eb545 docs: complete the release delta in the changelog and readme 2026-09-21 22:45:56 +02:00
petrbalvin f720381d43 feat(asm): the segment-absolute and crash-store forms GOROOT writes
Test / test (push) Failing after 2m28s
Assisted-by: GLM 5.3 Flash
2026-09-21 22:19:53 +02:00
petrbalvin 2c9042d62c feat(asm): PCALIGN alignment on amd64
Assisted-by: GLM 5.3 Flash
2026-09-21 22:00:30 +02:00
petrbalvin 82ef289d3a feat(asm): the immediate multiply and arm64 indirect branches GOROOT writes
Assisted-by: GLM 5.3 Flash
2026-09-21 21:50:11 +02:00
petrbalvin 7246b0e002 feat(asm): the TLS access pair in the toolchain's one-instruction form
Assisted-by: GLM 5.3 Flash
2026-09-21 21:35:15 +02:00
petrbalvin 8cfd40aac8 feat(asm): the operand forms and defines GOROOT writes
Assisted-by: GLM 5.3 Flash
2026-09-21 21:17:34 +02:00
petrbalvin 5382c9a8e4 feat(audit): list every corpus failure per architecture 2026-09-21 21:17:34 +02:00
petrbalvin 53de91b2df docs(asm): describe the four target architectures
Test / test (push) Failing after 2m23s
Assisted-by: GLM 5.3 Flash
2026-09-21 20:15:55 +02:00
petrbalvin 8a36af7c7d docs(asm): generate the instruction appendices
Assisted-by: GLM 5.3 Flash
2026-09-21 20:15:55 +02:00
petrbalvin e9789ce3f4 chore(arch): regenerate the instruction tables 2026-09-21 20:15:55 +02:00
petrbalvin 837231c068 docs(asm): open the assembly language reference
Assisted-by: GLM 5.3 Flash
2026-09-21 19:49:04 +02:00
petrbalvin 95025be1bc docs(changelog): describe the encoder entries by content
Test / test (push) Failing after 2m33s
2026-09-21 19:20:01 +02:00
petrbalvin 03a964bb2d docs(goobj): document the GOOBJ object file format 2026-09-21 19:19:53 +02:00
petrbalvin 123a16e346 docs(readme): state the documentation goal 2026-09-21 18:35:27 +02:00
petrbalvin 9701812bee docs: changelog for the completeness waves
Test / test (push) Failing after 3m6s
Assisted-by: GLM 5.3 Flash
2026-09-21 02:04:44 +02:00
petrbalvin 29ac03468e feat(amd64): floating-point immediates through a synthesised pool
Assisted-by: GLM 5.3 Flash
2026-09-21 02:04:44 +02:00
190 changed files with 18996 additions and 843 deletions
+42
View File
@@ -0,0 +1,42 @@
# FreeBSD compile gates. Dispatched by hand, never on a push.
#
# The debugger's ptrace surface and the JIT substrate are the two
# FreeBSD-portable layers the tree carries; the forge has no FreeBSD runner,
# so they can only be compile-gated, and three foreign-GOOS builds of the
# whole module are minutes of one-core work the push pipeline's budget cannot
# carry. The push pipeline stays fast and light; this workflow is the
# deliberate run, before a release or after touching the ported layers.
# Running the ptrace suite itself needs real FreeBSD hardware.
#
# A dispatched workflow takes no concurrency block: it is one deliberate run.
name: FreeBSD build
on:
workflow_dispatch:
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
build:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
# The module is the source of truth for the version, so it cannot drift.
go-version-file: go.mod
cache: true
- name: FreeBSD build (amd64)
run: GOOS=freebsd GOARCH=amd64 go build ./...
- name: FreeBSD build (arm64)
run: GOOS=freebsd GOARCH=arm64 go build ./...
- name: FreeBSD build (riscv64)
run: GOOS=freebsd GOARCH=riscv64 go build ./...
+4 -1
View File
@@ -342,7 +342,10 @@ jobs:
my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}},
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
q{-H}, q{Content-Type: application/octet-stream},
q{-X}, q{POST}, q{--data-binary}, qq{@$path},
# The @ must not sit inside a qq{} string: there it starts an
# array interpolation and the upload body collapses to empty,
# which Gitea stores as a 201-created zero-byte attachment.
q{-X}, q{POST}, q{--data-binary}, q{@} . $path,
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name});
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
my $code = <$curl>;
+19 -11
View File
@@ -6,6 +6,11 @@
# everything runs in one job. Extra jobs would duplicate the checkout, the Go setup and
# the dependency download three times without buying any parallelism.
#
# The budget is part of the contract: a push run is fast and light, about two minutes,
# and nothing that cannot run natively on the runner belongs here. The FreeBSD compile
# gates live in freebsd.yml behind workflow_dispatch for that reason; the GOOBJ link
# parity campaign is an opt-in local verification (just link-parity).
#
# Every step is one command, so the step that fails is the gate that failed, and no shell
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
# not Python: Perl behaves the same on both runner images, there is no bashism to trip over
@@ -46,10 +51,9 @@ jobs:
go-version-file: go.mod
cache: true
- name: Install Perl
# The runner images are minimal and Perl is not guaranteed. The install is a
# no-op where it is already present; drop this step once verified on the box.
run: dnf install -y perl
# No Install Perl step: the fedora image carries perl (verified by run
# 76: the install degraded into a package upgrade costing ~50 s), and a
# dnf on the push path is network work the budget does not need.
# The steps follow the `gates` order of the justfile contract: build, format,
# vet, test. The vet gate is go vet and go fix -diff, two steps here.
@@ -82,7 +86,11 @@ jobs:
# command, so the floor is the same number everywhere. ./verify/... carries the
# live oracle-parity comparison against `go tool asm` (the TestGroundTruth
# suites); the runner's Go setup provides both the tool and GOROOT.
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
# -short skips the deliberate-run categories inside the suites (the live
# ptrace sessions above all): they need the machine to themselves and a
# starved single-core runner turns each into a timeout the budget cannot
# carry. The local `just test` gate runs everything, in full.
run: go test -short -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Tests outside the coverage set
# The CLI and the debugger sit outside `packages` because a thin main and a
@@ -90,12 +98,12 @@ jobs:
# shipped surfaces: the command exit codes, the manual pages against the
# binary's own help, and the debugger's architecture-neutral units. They run
# here so the floor stays a product measure and nothing is left untested.
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
- name: Oracle parity
# Re-run the live go-tool-asm comparison as its own step so that a parity
# regression names the gate that failed instead of hiding inside the suite.
run: go test -count=1 -timeout 10m -run 'TestGroundTruth' ./verify/...
# -short skips the debugger's live ptrace sessions, the deliberate-run
# category the runner cannot starve-proof. The live go-tool-asm oracle
# comparison (TestGroundTruth in ./verify/...) runs inside the coverage
# sweep above; it is not re-run as its own step, because every second on
# this box is budget.
run: go test -short -count=1 -timeout 10m ./cmd/... ./debug/...
- name: Coverage floor
run: |
+248 -18
View File
@@ -1,6 +1,6 @@
# Changelog
All notable changes to gasm-devkit are documented here.
All notable changes to gasm-sdk are documented here.
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
@@ -9,6 +9,178 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added
- **The FreeBSD port of the debugger.** `gasm debug` runs on FreeBSD on
amd64, arm64 and riscv64 with the same interactive surface as on Linux:
breakpoints, hardware watchpoints (x86 debug registers, the arm64 debug
register file), single-stepping, register and memory access, all behind
the kernel's own ptrace requests, with tracee memory through `PT_IO` and
stop reports through `PT_LWPINFO`. The JIT substrate maps executable
memory through `golang.org/x/sys/unix`, so `verify` builds on FreeBSD
too. The pipeline compile-gates all three architectures; live
validation awaits a FreeBSD machine.
- **Workspace-wide navigation in the language server.** `gasm lsp` indexes
the `.s` files under the workspace root beyond the documents the editor
has open, so go-to-definition, find references and workspace symbol search
reach files that were never opened. An open buffer always shadows its
disk copy, and watched-file events together with a per-query freshness
check keep the index current.
- **Quick fixes for the textflag include and the argument area.** The
`missing-textflag-include` warning offers to add the include after the
last one in the file, and the `abi-argsize` warning offers to set the
TEXT argument area to the size the `// func` signature implies, computed
by the new `lint.ExpectedArgSize`.
- **Eleven new lint rules over the directives, the data section and the
sharpest addressing edges.** `missing-argsize` flags a TEXT that
declares no argument area its `// func` signature implies;
`noframe-frame-size` a NOFRAME with a positive frame;
`unnamed-fp-reference` a nameless `0(FP)`, which both assemblers
reject; `hardware-sp-addressing` a negative offset off the hardware SP
rather than the virtual frame; `vex-sse-mixing` a kernel that mixes VEX
and legacy SSE spellings and pays the transition penalty;
`unnamed-result` a `ret+N(FP)` the signature names; `data-width`,
`data-value-overflow`, `data-string-width`, `data-without-globl` and
`data-exceeds-globl` police the DATA width against its value type and
the GLOBL size behind it. `missing-ret` now also flags a function
whose tail can fall off its end even though a RET sits somewhere in the
body, and `invalid-textflag` also reports a flag misplaced between TEXT
and GLOBL.
- **Hover documentation and wider completions in the language server.**
Hovering a directive or pseudo-operation (TEXT, DATA, GLOBL, PCALIGN,
FUNCDATA, PCDATA, the BYTE family) shows its grammar and rules in
preference to the empty instruction-table entry, and completion offers
those names beside the instruction set. The `missing-argsize` warning
carries a quick fix that declares the argument area the signature
implies (`$0` becomes `$0-16`).
- **The GOOBJ link parity gate.** A regression test assembles a kernel
per architecture through `gasm asm --format goobj`, substitutes the
object into a real `go build`'s package archive, proves the archive
carries it byte for byte and re-links with cmd/link on amd64, arm64,
riscv64 and loong64. The binaries run, natively on amd64 and under
qemu-user on arm64 and riscv64, and their output must match the
toolchain-built baseline; loong64 is link-only. The pipeline installs
qemu-user and runs the gate on every push.
- **The amd64 and loong64 encoders close four more corpus files.** The
whole-tree measure moves to 272 of 322 (84.5 %): amd64 gains the
one-operand IMUL, the SSE compare family (CMPPD, CMPPS, CMPSS), RETFL,
the LOOP family, the MMX register bank with its bank-crossing moves,
MOVNTDQ, the CR and DR register moves, PUSH and POP of FS and GS, the
`(TLS)` pseudo-base, the wait and cache controls (CLWB, CLDEMOTE,
TPAUSE, UMONITOR, UMWAIT, RDPID, ENDBR64), indirect branches with the
star spelling (`JMP *(R12)(R13*4)`), `RET sym(SB)` as the tail jump,
the colon shift spelling (`SHLL CX, R11:AX`) and the EVEX and VEX forms
of the rounds, AES key assist, string compares, extracts, blends and
permutes; loong64 gains the acquire and release pair (LLACQ, SCREL,
with the vector widths), the VMOVQ and XVMOVQ lane forms and the BYTE
literal-data escape hatch the other architectures already take.
### Changed
- **The corpus audit assembles like the build.** A file's `//go:build`
constraint decides which target architectures attempt it: cpu_x86.s is
an x86 build alone, and the msan and goexperiment.runtimesecret trees
are compiled by no supported build, so they leave the measured set
instead of failing it. The headline now reads "assemble for every
applicable target": every real-code GOROOT assembly file, the tree
without testdata, assembles for all four architectures (250 of 250,
100 %); over the whole tree including testdata the measure is 272 of
322 (84.5 %).
- **The module moves to `sourcedock.dev/petrbalvin/gasm-sdk`.** The
repository and the module rename together with the product, now the
GAsm Software Development Kit. Fresh installs become
`go install sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm@latest`, and
installs pinned to the old `gasm-sdk` path stop resolving once the
repository takes the new name: reinstall from the new path. The
binary stays `gasm`.
### Fixed
- **Rename edits land in their own documents.** A rename collected the
ranges of every reference across the open documents but applied them all
to the document that started it, so renaming a symbol used in a second
file moved that file's text into the first. Each edit now applies to the
document it was collected in.
- **Negative numeric PC-relative jumps.** `JMP -3(PC)`, the shape the
runtime's exit loops write (sys_linux_amd64.s, sys_netbsd_amd64.s),
resolved to nothing: only the forward forms counted. A negative count
now walks the same instruction statements backwards, labels excluded,
byte-identical with the toolchain.
- **The arm64 move-wide family reads its immediate as an unsigned
pattern.** `MOVK $(40000<<48)` folds to a negative int64 and was
rejected; the toolchain picks the 16-bit lane from the 64-bit bit
pattern, so the encoder now does the same, and a zero immediate is
rejected where the toolchain rejects it.
- **NOPTR data emits its own symbol kind.** A GLOBL with NOPTR was
emitted as plain SDATA, the kind the linker holds to its Go type
information requirement, so every gasm object carrying runtime-shaped
data (`GLOBL ·x(SB), NOPTR, ...`) died in cmd/link with "missing Go
type information". NOPTR data is now SNOPTRDATA, the toolchain's kind
for pointer-free globals, and RODATA still wins where both flags
appear, exactly as the toolchain chooses. The three end-to-end link
tests that should have caught this substituted the gasm object after
the package archive was already packed, so they passed vacuously; they
now substitute inside the archive, prove the substitution byte for
byte and re-link.
- **Latent encoder divergences against the toolchain, found by the
whole-file differential harness.** On loong64, `MOVx $off(reg)` lost
its base register, SCQ swapped its operand fields, the vector lane
inserts did not scale offsets by the element width and two immediate
opcodes were mistyped; on amd64, NOP with operands encoded 0x90 where
the toolchain emits nothing at all, the VEX gather length bit ignored
the VSIB index width, four permute and extract families were pushed
into EVEX where the toolchain stays VEX, and a reference to a static
symbol no GLOBL defines failed where the toolchain defers it to the
linker as an external relocation.
- **Seven front-end defects found by fuzzing.** The formatter swallowed
the statement after a leading block comment (`/* head */ MOVQ AX, BX`
formatted to the comment alone), dropped trailing tokens after an
`#include` header, and trimmed the trailing whitespace inside a string
literal; the parser peeled only one label of a stacked pair (`a: b:`
parsed differently than it formatted); deeply nested parentheses in an
immediate overflowed the stack through the constant folder, which now
stops at a bounded depth and falls back to the ordinary operand paths;
macro expansion is linear in the invocation count instead of
quadratic; and amplifying macros (the billion-laughs shape) stop at a
work budget sized by the line and the macro table, reported as a
diagnostic instead of running for hours.
- **The debugger handles the edges its tests now reach.** Memory reads
and writes cover exactly the requested bytes, so a request ending in
the last page of a mapping no longer fails on the unmapped page behind
it; disassembly shrinks its instruction window at a mapping's end
instead of failing; a debuggee that dies before signalling readiness
writes a failure notice the launcher reads, so launch fails fast with
the reason instead of after the whole poll (and the poll no longer
waits on the child, which could consume the SIGSTOP park and hang the
launch); amd64 watchpoints acknowledge the sticky DR6 hit bits and
clear the address register on release; the breakpoint listing is
ordered by address so its numbering is stable; the REPL rejects bad
counts, sizes and watchpoint types instead of silently guessing,
reports stops on signals during stepping, and a breakpoint-class trap
that matches no breakpoint and leaves the PC in place surfaces instead
of spinning the continue loop and the coverage run forever.
## [0.35.0] - 2026-09-22
### Added
- **The go_asm.h generator.** `gasm asm` generates the package's go_asm.h
itself when an assembly file includes it: the Go files beside the source
are type-checked for the target architecture and the constants and field
offsets become assembler defines, so package-context files assemble with
no compiler and no `go build` in the loop. `-GOOS` selects the
type-checking GOOS for GOOS-specific files, and the corpus audit derives
the GOOS from the file name.
- **ELF data relocations on arm64, riscv64 and loong64.** `gasm asm
--format elf` emits `.rela.data` for symbol-valued DATA initialisers on
every architecture (amd64 carried them already), so standalone ELF
objects link on all four targets.
- **Corpus failure listing.** `gasm audit-instructions --corpus --list`
prints every failing file with its failure reason, per architecture,
instead of one representative file per reason.
- **DATA with symbol values and relaxed symbol spellings.** DATA
initialisers accept `$symbol(SB)` values, laid down as an absolute
relocation at the data field (GOOBJ on all four architectures and ELF
on all four as of this release), and U+2215 is accepted inside symbol
package paths.
- **Macro expansion and include splicing.** `gasm asm`, `gasm diff` and
`gasm audit-instructions` now preprocess assembly the way the
toolchain does: object and parameterised `#define` macros expand at
@@ -19,7 +191,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
(`$(32-7)`, `$~63`, `(index*4)(base)`) fold at parse. Expansion
happens only on the assembly path: `gasm lint`, `gasm fmt` and the
language server keep reading the raw file.
- **The GOROOT instruction wave, part 1.** The encoder now covers the
- **Encoder coverage: the instruction families GOROOT's real code
uses.** The encoder now covers the
instruction families GOROOT's real code uses that gasm lacked,
byte-verified against `go tool asm`: on amd64 the carry ALU, the
atomics (CMPXCHG, XADD, XCHG), AES-NI, SHA-1/256, PCLMULQDQ, CRC32,
@@ -35,7 +208,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
Also fixed on the way: arm64 `CASD`/`CASW` lacked an opcode bit, and
riscv64 `VSETVLI` with an immediate length now canonicalises to
`vsetivli` as the toolchain does.
- **The GOROOT instruction wave, part 2.** The encoder gains the
- **Encoder coverage: quad-register AVX-512 and floating-point
immediates.** The encoder gains the
quad-register AVX-512 families (4FMAPS, 4FNMADD, 4VNNIW, VP4DPWSSD,
VP4DPWSSDS) with the register list riding the inverted V'VVVV field,
floating-point immediates on the SSE scalar moves and arithmetic
@@ -47,21 +221,77 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
ranges, index-only VSIB memory operands and bare trailing immediates;
macro substitution reaches parameters used with element suffixes
(`A.S4`), and `;` separates statements in plain files.
- **`gasm asm -GOOS`.** The go_asm.h generator type-checks per target
GOOS, so the darwin-only and windows-only runtime files assemble with
their own defines; the corpus audit derives the GOOS from the file
name. DATA initialisers accept `$symbol(SB)` values (an absolute
relocation at the data field, GOOBJ on all four architectures and ELF
on amd64, arm64, riscv64 and loong64), and U+2215 is accepted inside
symbol package paths.
- **The corpus audit measures honestly.** Files named for Go ports gasm
does not target (arm, 386, s390x, ...) are no longer attempted for the
four supported architectures (no supported build compiles them), and
the headline rate is reported over attemptable files: 136 of 433 on
the full corpus (31.4 %), 135 of 383 on real code (35.2 %), from the
127 that the previous release measured. The probe battery that
decides encodability gained the operand shapes the new families use.
-
- **Per-architecture reference pages.** [docs/asm/](docs/asm/README.md)
gains AMD64, ARM64, RISCV64 and LOONG64: the register files and the
roles the ABI fixes, addressing, operand order with every special form,
constants and materialisation, alignment, fences and the relocations
each target emits. An instruction inventory appendix per architecture
is generated from the toolchain's own tables by `just gen`, and the
regenerated tables recognise 147 more mnemonics than the previous
release carried (arm64 107, riscv64 31, loong64 9).
- **The Plan 9 assembly language reference.** [docs/asm/](docs/asm/README.md)
opens the complete language reference with its common core: the lexicon,
statement structure and constant expressions, the operand grammar with
the pseudo-registers and symbol naming, the directives and the function
flag vocabulary, preprocessing with `#define` and `#include`, and the
Go-embedded layer (ABI0, prototypes, `go_asm.h`, `funcdata.h` and the
runtime contract). Every claim is verified against `go tool asm` of
Go 1.27.1 and gasm's differential tests; the per-architecture pages and
generated instruction appendices follow.
- **GOOBJ format specification.** [docs/GOOBJ.md](docs/GOOBJ.md)
documents the Go object file format in full: both containers, the 96
byte header and all 19 blocks, every structure with its byte
offsets, symbol kinds and flag bits, all 106 relocation types with
the weak variants, aux symbols, the FuncInfo payload, the pc-value
table encoding, the content hashes and the builtin table, all
verified byte for byte against objects produced by Go 1.27.1's own
tools.
### Changed
- **The corpus audit measures like a build.** Files named for a Go port
gasm does not target (arm, 386, s390x, ...) are never attempted, because
no supported build compiles them; the GOOS comes from the file name; and
each target's go_asm.h is generated on the fly. The headline is reported
over attemptable files: 291 of 353 on the full corpus (82.4 %) assemble
for every target architecture and 295 of 303 on real code (97.4 %),
against 108 of 627 over all files (17.2 %) that the previous release
measured.
### Fixed
- **The operand forms GOROOT writes.** Numeric PC-relative jumps
(`JEQ 2(PC)`, the park loop `JMP 0(PC)`) resolve with the toolchain's
own instruction counting and fold jump-to-jump chains exactly as its
branch optimiser does; symbol immediates (`MOVQ $sym(SB), AX`)
assemble to the toolchain's RIP-relative LEA with an R_PCREL
relocation; negated constant expressions in operands (`ADJSP
$-(REGS - 8)`, the shape the cgo ABI macros write) fold; the immediate
multiply (`IMULQ $1000000000, AX`) encodes with the toolchain's
0x69/0x6B selection; the TLS access pair assembles as the toolchain's
one-instruction form (the bare `MOVQ TLS, r` load nops out and
`off(r)(TLS*1)` folds to the segment-prefixed absolute whose disp32
carries the R_TLSLE relocation, per-GOOS); arm64 accepts the
bare-register indirect branch (`BL R9` beside `BL (R9)`, both BLR) and
the zero-immediate store (`MOVD $0, mem` through the zero register,
rejecting non-zero immediates as the toolchain does); `PCALIGN` now
aligns on amd64, padding with the toolchain's greedy
single-instruction NOPs; the segment-absolute forms (`MOVQ 0x30(GS),
AX` and the store direction) and the absolute crash-store
(`MOVL $0xf1, 0xf1`) encode; and `gasm asm` predefines the
`GOARCH_<arch>` and `GOOS_<goos>` macros the go command passes to
`go tool asm`, so GOROOT headers' `#ifdef GOARCH_amd64` platform
blocks (`go_tls.h`'s `get_tls` and friends) select as intended. The
GOROOT corpus measure moves to 291 of 353 files assembling for every
target architecture (82.4 %), 97.4 % of the real-code corpus, from
70.8 % and 82.2 %.
- **Tool corrections across the pipeline.** The formatter keeps square
brackets in SIMD operands, statement separators and canonical macro
bodies; the linter drops false positives on shift counts, SETcc
spellings and ABIInternal references; the lexer treats a trailing
carriage return as a line end so comment text stays idempotent; and
arm64 rejects bare BTI with a diagnostic while accepting the full
family.
## [0.34.0] - 2026-09-20
+4 -4
View File
@@ -1,6 +1,6 @@
# Contributing
Contributions to **gasm-devkit** are governed by the Contributor terms
Contributions to **gasm-sdk** are governed by the Contributor terms
below; submitting one means you accept them.
## Contributor terms
@@ -29,8 +29,8 @@ compiler (gcc), because `just gates` includes `just race` and the race
detector needs cgo.
```sh
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
cd gasm-devkit
git clone https://sourcedock.dev/petrbalvin/gasm-sdk.git
cd gasm-sdk
just build
just gates
```
@@ -126,7 +126,7 @@ tag, where it would double the time and the memory a shared runner cannot spare.
## Reporting bugs
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-devkit/issues` with the
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-sdk/issues` with the
version, the operating system and architecture, the exact command, the full output,
and the expected against the actual behaviour.
+66 -20
View File
@@ -1,6 +1,6 @@
# Plan 9 assembly tooling, inside and outside Go
# GAsm: Software Development Kit for Plan 9 Assembly
> **Warning: this is an experiment.** gasm-devkit is under active
> **Warning: this is an experiment.** gasm-sdk is under active
> development and is not stable. The version is 0.x.x: commands, flags,
> output formats and behaviour can change without warning at any time.
> A 1.0.0 release is light years away. Nothing in this document is a
@@ -13,7 +13,7 @@
there is no formatter, no linter and no debugger for `.s` files, and no
assembler that works without a Go installation. Developers write
assembly blind, validate it by benchmark, and debug it by print
statement. gasm-devkit is the missing toolkit: a single, self-contained
statement. gasm-sdk is the missing toolkit: a single, self-contained
binary, `gasm`, that serves both purposes.
- **Help develop Plan 9 assembly.** Formatting, linting, disassembly,
@@ -58,7 +58,7 @@ Plan 9 (Go): MOVQ AX, total-16(SP)
The same lines, but only one of them tells you what the number is for.
The syntax is uppercase, regular and boring, which is the highest
compliment a language for machine code can earn. gasm-devkit exists
compliment a language for machine code can earn. gasm-sdk exists
to give that syntax the tooling it deserves.
## Features
@@ -81,7 +81,10 @@ to give that syntax the tooling it deserves.
GOOBJ format, which needs the installed toolchain and which `go build`
consumes in place of the toolchain's output. Framed functions get the
stack-split guard and the morestack block, byte-identical to the
toolchain's, so split functions link too.
toolchain's, so split functions link too. The assembler preprocesses
like the toolchain (`#define`, `#include` with `-I`, `#ifdef`), generates
`go_asm.h` from the package's Go files, and carries `PCALIGN`, the
`LOCK`/`REP` prefixes and the literal-data pseudo-ops.
- **Disassembler.** `gasm dis` lists a `.s` file's functions at their real
offsets after assembling, or disassembles raw bytes from a file or stdin.
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
@@ -91,13 +94,16 @@ to give that syntax the tooling it deserves.
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
breakpoints (optionally conditional), hardware watchpoints, register and
memory inspection, and headless script runs that report instruction and
label coverage.
label coverage; it runs on Linux (all four architectures) and FreeBSD
(amd64, arm64, riscv64).
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
push and pull diagnostics, semantic-token highlighting, go-to-definition,
find references, rename, formatting, inlay hints, code actions, signature
help, document highlights, workspace symbol search, #include document
links and folding ranges over stdio; definition, references and rename
work across every open document.
work across every open document and the indexed workspace files beyond
them, and the quick fixes add a missing textflag.h include and set the
argument area from the // func signature.
- **Comparators and audits.** `gasm diff` compares the machine code of two
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
`gasm audit-instructions` diffs the encoder against the installed toolchain,
@@ -110,9 +116,9 @@ Four architectures, the four that matter in practice:
| Architecture | GOARCH | File suffix | Instructions recognised |
|--------------|-------------|--------------|---------------------------------------------|
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
| ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes |
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes |
| LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes |
| ARM64 | `arm64` | `_arm64.s` | 645 + common opcodes |
| RISC-V | `riscv64` | `_riscv64.s` | 992 + common opcodes |
| LoongArch | `loong64` | `_loong64.s` | 808 + common opcodes |
"Common opcodes" are the instructions shared by every architecture (`RET`,
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
@@ -124,10 +130,13 @@ can emit today is narrower, and a recognised but unencodable instruction is
reported as an explicit error, never as a wrong byte.
The same measurement runs over GOROOT's whole assembly corpus:
`gasm audit-instructions --corpus` reports 136 of 433 attemptable files
(31.4 %) assembling for every target architecture today (files named for
other Go ports are counted but never attempted), with the top failure
reasons per architecture; the number moves with every release.
`gasm audit-instructions --corpus` reports every real-code GOROOT assembly
file (the tree without testdata) assembling for every target its build
admits: 250 of 250, 100 %. Over the whole tree including testdata the
measure is 271 of 322 attemptable (84.2 %); files named for other Go ports
are counted but never attempted, and `//go:build` constraints decide which
targets attempt a file at all, exactly as the build does. The number moves
with every release.
### Validation status
@@ -142,7 +151,7 @@ actually been executed.
|---|---|---|
| Encoding: byte-for-byte against `go tool asm` | native hardware | native hardware (the toolchain cross-assembles any GOARCH on any host) |
| Execution: JIT calls, ABI checks, differential fuzzing | native hardware | qemu-user emulation |
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more |
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more. FreeBSD (amd64, arm64, riscv64) is in the same position: the port compiles behind the cross-build gate and its integration test is ready, but no FreeBSD machine has executed it |
Consequences, stated plainly. An emulator is a model of a CPU, not the
CPU: instruction semantics are implemented in software and can differ
@@ -157,6 +166,39 @@ been compiled and read, never executed. Its architecture-neutral units
run under `go test ./...`, which the race workflow and a manual run
perform; the default `just test` gate does not sweep `./debug/...`.
## The documentation goal
The toolkit is the primary goal. The secondary one is documentation: a
specification of the Plan 9 assembly language and of the GOOBJ object
format that is 100 % complete, detailed enough to implement against,
and written to a professional standard. These are the two subjects this
project works with every day, and they are the two for which no usable
documentation exists.
Go documents the language on a single page, "A Quick Guide to Go's
Assembler", which carries no section for loong64, one of the four
architectures gasm supports, and covers a fraction of what each
assembler accepts. What exists beyond it lives as comments inside the
toolchain's internal source: per-architecture reference manuals for
arm64, ppc64, riscv64 and loong64, written for the toolchain's own
maintainers rather than for an outside reader, and none at all for
amd64. GOOBJ fares worst of all. The format that `go build` consumes
has no specification anywhere: it is described by a comment in an
internal package, it is not a stable interface, and it can change with
any toolchain release.
The gap is therefore filled the only way it can be filled: by reverse
engineering the toolchain itself, the same work the encoders already
perform. Most of the documentation can come from nowhere else, and it
is written as that knowledge is produced during development. It is
verified the way the code is verified: an encoding documented here is
one that differential tests against `go tool asm` confirm
byte-for-byte, and a format field documented here is one the linker
demonstrably reads. The work has begun: [docs/GOOBJ.md](docs/GOOBJ.md)
specifies the object file format completely, and
[docs/asm/README.md](docs/asm/README.md) opens the language reference
with its common core. The per-architecture pages follow.
## Direction
The plan, in the order it is being worked:
@@ -179,9 +221,11 @@ The plan, in the order it is being worked:
toolchain itself does not support; through ELF, Plan 9 assembly becomes
usable outside Go entirely.
- **Platforms: Linux and FreeBSD.** Linux is supported today on all four
architectures and is where the binary builds. FreeBSD follows: the
JIT's executable-memory mapping and the ptrace debugger layer are the
two pieces of porting work. Other unix systems may follow those two.
architectures and is where the binary builds. FreeBSD follows on amd64,
arm64 and riscv64: the JIT's executable-memory mapping and the ptrace
debugger layer are ported (the debugger's live validation awaits a
FreeBSD machine, as the validation status states). Other unix systems
may follow those two.
- **Four architectures, no more.** amd64, arm64, riscv64 and loong64.
No others are planned.
@@ -189,11 +233,11 @@ The plan, in the order it is being worked:
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
linux/loong64 are on the
[releases page](https://sourcedock.dev/petrbalvin/gasm-devkit/releases).
[releases page](https://sourcedock.dev/petrbalvin/gasm-sdk/releases).
From source (Go 1.27.1):
```sh
go install sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm@latest
go install sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm@latest
```
Or from a repository checkout:
@@ -282,6 +326,8 @@ recipe.
~/.local/share/man (MANDIR overrides); `just uninstall-man` removes
them
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/GOOBJ.md](docs/GOOBJ.md): the GOOBJ object file format specification
- [docs/asm/](docs/asm/README.md): the Plan 9 assembly language reference
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
- [CHANGELOG.md](CHANGELOG.md): release history
+1 -1
View File
@@ -7,7 +7,7 @@ releases do not receive them.
| Version | Supported |
|---|---|
| 0.34.0 | yes |
| 0.35.0 | yes |
| older releases | no |
## Reporting a vulnerability
+105 -7
View File
@@ -5,9 +5,13 @@
// toolchain's own assembler source. Go's Plan 9 assembler defines the exact,
// complete set of mnemonics it accepts for each architecture in
// $GOROOT/src/cmd/internal/obj/<arch>/anames.go; this tool extracts those
// names so gasm-devkit supports every instruction the real assembler does,
// names so gasm-sdk supports every instruction the real assembler does,
// with no hand-maintained (and therefore inevitably incomplete) lists.
//
// The same data feeds the generated instruction appendices of the assembly
// language reference, docs/asm/INSTRUCTIONS-<ARCH>.md, so that the reference
// cannot drift from the tables it documents.
//
// Usage (via the justfile):
//
// just gen
@@ -26,9 +30,12 @@ import (
"path/filepath"
"sort"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
)
// archDirs maps a gasm-devkit architecture name to its obj sub-directory.
// archDirs maps a gasm-sdk architecture name to its obj sub-directory.
var archDirs = []struct {
arch string
sub string
@@ -39,11 +46,30 @@ var archDirs = []struct {
{"loong64", "loong64"},
}
// docPages maps an architecture to its generated appendix in the language
// reference. The amd64 page carries a per-mnemonic encodability column,
// decided by asm.Encodable, which mirrors the encoder's own dispatch; the
// other targets have no single cheap predicate, so their pages carry the
// inventory and point at the live measurement instead.
var docPages = []struct {
arch arch.Arch
title string
file string
anames string
encodable bool
}{
{arch.AMD64, "AMD64", "INSTRUCTIONS-AMD64.md", "cmd/internal/obj/x86/anames.go", true},
{arch.ARM64, "ARM64", "INSTRUCTIONS-ARM64.md", "cmd/internal/obj/arm64/anames.go", false},
{arch.RISCV, "RISC-V 64", "INSTRUCTIONS-RISCV64.md", "cmd/internal/obj/riscv/anames.go", false},
{arch.LOONG64, "LoongArch 64", "INSTRUCTIONS-LOONG64.md", "cmd/internal/obj/loong64/anames.go", false},
}
func main() {
goroot := strings.TrimSpace(runGoEnvGOROOT())
if goroot == "" {
fatal("could not determine GOROOT")
}
version := strings.TrimSpace(runGoEnv("GOVERSION"))
// The common opcodes shared by every architecture (RET, JMP, NOP, CALL,
// TEXT, FUNCDATA, …) live in cmd/internal/obj/util.go.
commonPath := filepath.Join(goroot, "src", "cmd", "internal", "obj", "util.go")
@@ -57,16 +83,24 @@ func main() {
}
fmt.Printf("%-8s %4d instructions -> arch/common_gen.go\n", "common", len(common))
names := map[string][]string{}
for _, a := range archDirs {
path := filepath.Join(goroot, "src", "cmd", "internal", "obj", a.sub, "anames.go")
names, err := extractInstrs(path)
names[a.arch], err = extractInstrs(path)
if err != nil {
fatal("extract %s: %v", a.arch, err)
}
if err := writeGen(a.arch, a.sub, names); err != nil {
if err := writeGen(a.arch, a.sub, names[a.arch]); err != nil {
fatal("write %s: %v", a.arch, err)
}
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names), a.arch)
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names[a.arch]), a.arch)
}
for _, p := range docPages {
if err := writeDocPage(p.arch, p.title, p.file, p.anames, version, p.encodable); err != nil {
fatal("write %s: %v", p.file, err)
}
fmt.Printf("%-8s -> docs/asm/%s\n", p.arch, p.file)
}
}
@@ -85,7 +119,7 @@ func filterCommon(names []string) []string {
// writeCommon emits arch/common_gen.go.
func writeCommon(names []string) error {
var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
@@ -156,7 +190,7 @@ func stringLit(elt ast.Expr) string {
// writeGen emits arch/<arch>_gen.go.
func writeGen(arch, sub string, names []string) error {
var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
@@ -172,6 +206,61 @@ func writeGen(arch, sub string, names []string) error {
return os.WriteFile(filepath.Join("arch", arch+"_gen.go"), []byte(b.String()), 0o644)
}
// writeDocPage emits docs/asm/<file>, the generated instruction appendix of
// the language reference for one architecture: every mnemonic the toolchain
// accepts, with the curated summary where the architecture table carries one
// and, on amd64, a per-mnemonic encodability column.
func writeDocPage(a arch.Arch, title, file, anames, version string, encodable bool) error {
table := arch.ForArch(a)
instrs := table.Instructions()
var b strings.Builder
b.WriteString("# " + title + ": instruction inventory\n\n")
b.WriteString("Generated by gasm-sdk's `_gen` from the Go toolchain's instruction table\n")
b.WriteString("(`" + anames + "`, " + version + "); DO NOT EDIT. This page lists every mnemonic\n")
b.WriteString("`go tool asm` accepts on this target, which is the upper bound of the\n")
b.WriteString("language on it: a name absent here is not an instruction of the target,\n")
b.WriteString("and a name present here may still be one gasm's encoder cannot emit yet.\n\n")
encodableCount := 0
if encodable {
b.WriteString("The `gasm encodes` column reports whether gasm's encoder can emit the\n")
b.WriteString("mnemonic today; the gap is the encoder backlog, measured live by\n")
b.WriteString("`gasm audit-instructions`.\n\n")
b.WriteString("| Mnemonic | gasm encodes | Notes |\n")
b.WriteString("|---|---|---|\n")
for _, in := range instrs {
ok := asm.Encodable(in.Name)
if ok {
encodableCount++
}
b.WriteString("| `" + in.Name + "` | " + yesNo(ok) + " | " + in.Summary + " |\n")
}
b.WriteString("\n")
fmt.Fprintf(&b, "Recognised: %d mnemonics. gasm encodes: %d.\n", len(instrs), encodableCount)
} else {
b.WriteString("The inventory carries no per-mnemonic encoder column: on this target\n")
b.WriteString("encodability is decided per operand shape, and the live measured\n")
b.WriteString("coverage is reported by `gasm audit-instructions`.\n\n")
b.WriteString("| Mnemonic | Notes |\n")
b.WriteString("|---|---|\n")
for _, in := range instrs {
b.WriteString("| `" + in.Name + "` | " + in.Summary + " |\n")
}
b.WriteString("\n")
fmt.Fprintf(&b, "Recognised: %d mnemonics.\n", len(instrs))
}
return os.WriteFile(filepath.Join("docs", "asm", file), []byte(b.String()), 0o644)
}
// yesNo renders a boolean as the word the appendix tables use.
func yesNo(v bool) string {
if v {
return "yes"
}
return "no"
}
func runGoEnvGOROOT() string {
out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil {
@@ -180,6 +269,15 @@ func runGoEnvGOROOT() string {
return string(out)
}
// runGoEnv runs `go env` for a single variable.
func runGoEnv(name string) string {
out, err := exec.Command("go", "env", name).Output()
if err != nil {
return ""
}
return string(out)
}
func fatal(format string, args ...any) {
fmt.Fprintf(os.Stderr, "gen: "+format+"\n", args...)
os.Exit(1)
+1 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
+624
View File
@@ -0,0 +1,624 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// This file carries the extended-instruction layer: instructions the Go
// toolchain does not know at all, described as data and validated against
// golden vectors from the Arm Architecture Reference Manual rather than
// against the toolchain. It sits beside the generated tables, never inside
// them: arch/arm64_gen.go stays untouched, and Extensions returns the layer
// per architecture so a later amd64 table attaches through the same door.
//
// The first entry is the arm64 SVE and SVE2 integer add/subtract/multiply
// family (twenty-three forms over four word shapes). The encodings are
// transcribed from the manual and cross-checked against the GNU assembler's
// and LLVM's published encodings; the golden vectors in arm64_ext_test.go pin
// the bytes.
package arch
import "fmt"
// ExtOperandKind classifies one operand of an extended instruction.
type ExtOperandKind uint8
// Operand kinds.
const (
ExtZReg ExtOperandKind = iota // scalable vector register Z0-Z31
ExtPReg // predicate register P0-P15
ExtImm // immediate
)
// String returns a short label for the kind.
func (k ExtOperandKind) String() string {
switch k {
case ExtZReg:
return "scalable vector register"
case ExtPReg:
return "predicate register"
case ExtImm:
return "immediate"
default:
return "operand"
}
}
// ExtArrangement is the element-size suffix a scalable vector operand
// carries: .B, .H, .S, .D or .Q. ExtArrNone means the operand is written
// bare, which the SVE forms in this layer reject.
type ExtArrangement uint8
// Arrangements, widest last.
const (
ExtArrNone ExtArrangement = iota
ExtArrB // 8-bit elements
ExtArrH // 16-bit elements
ExtArrS // 32-bit elements
ExtArrD // 64-bit elements
ExtArrQ // 128-bit elements
)
// String returns the assembler suffix, with the leading dot.
func (a ExtArrangement) String() string {
switch a {
case ExtArrB:
return ".B"
case ExtArrH:
return ".H"
case ExtArrS:
return ".S"
case ExtArrD:
return ".D"
case ExtArrQ:
return ".Q"
default:
return ""
}
}
// Width returns the byte width of one element under the arrangement.
func (a ExtArrangement) Width() int {
switch a {
case ExtArrB:
return 1
case ExtArrH:
return 2
case ExtArrS:
return 4
case ExtArrD:
return 8
case ExtArrQ:
return 16
default:
return 0
}
}
// sizeBits maps the arrangement onto the two-bit size field the integer SVE
// classes carry at bits 23..22: 00=B, 01=H, 10=S, 11=D. ok is false for the
// arrangements no such class accepts (.Q and the bare spelling).
func (a ExtArrangement) sizeBits() (uint32, bool) {
switch a {
case ExtArrB, ExtArrH, ExtArrS, ExtArrD:
return uint32(a) - 1, true
default:
return 0, false
}
}
// ExtQualifier is the predicate qualifier spelled after the slash.
type ExtQualifier uint8
// Predicate qualifiers.
const (
ExtQualNone ExtQualifier = iota // bare Pn (non-predicating position)
ExtQualMerging // /M, inactive lanes keep the destination
ExtQualZeroing // /Z, inactive lanes become zero
)
// String returns the assembler spelling, with the leading slash.
func (q ExtQualifier) String() string {
switch q {
case ExtQualMerging:
return "/M"
case ExtQualZeroing:
return "/Z"
default:
return ""
}
}
// ExtOperand is one operand of an extended instruction, already resolved to
// its pieces: a register with its arrangement and qualifier, or an immediate
// with its optional left shift. The assembler's future hook constructs these
// from the parsed statement; Encode consumes them.
type ExtOperand struct {
Kind ExtOperandKind
Reg int // register number (Z: 0..31, P: 0..15)
Arr ExtArrangement // element-size suffix; ExtArrNone when bare
Qual ExtQualifier // predicate qualifier; ExtQualNone elsewhere
Imm int64 // immediate value (ExtImm only)
// Shift carries the LSL amount an immediate form shifts the constant by
// before use (0 or 8 in the SVE add/subtract immediate class). HasShift
// separates a spelled shift (validated as written) from an unshifted
// operand (the encoder may derive the sh bit from the value).
Shift int
HasShift bool
}
// ExtVector builds a scalable vector operand, ADD Z1.S style.
func ExtVector(reg int, arr ExtArrangement) ExtOperand {
return ExtOperand{Kind: ExtZReg, Reg: reg, Arr: arr}
}
// ExtPredicate builds a predicate operand with its qualifier, P0/M style.
func ExtPredicate(reg int, qual ExtQualifier) ExtOperand {
return ExtOperand{Kind: ExtPReg, Reg: reg, Qual: qual}
}
// ExtImmediate builds an unshifted immediate operand.
func ExtImmediate(v int64) ExtOperand {
return ExtOperand{Kind: ExtImm, Imm: v}
}
// ExtShiftedImmediate builds an immediate operand with a spelled LSL amount.
func ExtShiftedImmediate(v int64, shift int) ExtOperand {
return ExtOperand{Kind: ExtImm, Imm: v, Shift: shift, HasShift: true}
}
// ExtField is one named field of the 32-bit encoding word: a bit offset from
// the least significant end and the field's width.
type ExtField struct {
Off uint8
Width uint8
}
// extMask returns the field's bits as a mask.
func extMask(f ExtField) uint32 {
return ^uint32(0) >> (32 - f.Width)
}
// extSet ORs v into the field of word.
func extSet(word uint32, f ExtField, v uint32) uint32 {
return word | (v&extMask(f))<<f.Off
}
// The fields the SVE integer classes use. The 5-bit register fields are
// named after their role in the three-vector class; the predicated class
// reuses extFieldRn for its Zm operand and extFieldPg for the governing
// predicate, which that class narrows to three bits (P0-P7).
var (
extFieldRd = ExtField{0, 5} // destination (Zd or Zdn)
extFieldRn = ExtField{5, 5} // first source (Zn, or Zm in the predicated class)
extFieldRm = ExtField{16, 5} // second source (Zm in the three-vector class)
extFieldPg = ExtField{10, 3} // governing predicate P0-P7 (predicated class)
extFieldImm8 = ExtField{5, 8} // the immediate, bits 12..5
extFieldSh = ExtField{13, 1} // the shift flag: 1 means LSL #8
extSizeBHSD = ExtField{22, 2} // element-size field of every class here, bits 23..22
)
// ExtForm enumerates the operand shapes the extension layer defines, in Plan
// 9 order (sources first, destination last). A destructive SVE operand is
// written once, in destination position: the encoding carries no second copy.
type ExtForm uint8
// Operand shapes.
const (
// ExtFormVectors is the unpredicated three-vector form, the SVE integer
// add/subtract (unpredicated) class: ADD Z0.S, Z1.S, Z2.S computes
// Z0 = Z1 + Z2. Operands: Zn, Zm, Zd.
ExtFormVectors ExtForm = iota
// ExtFormPredicated is the governed destructive form, the SVE integer
// add/subtract vectors (predicated) class: ADD Z1.S, P0/M, Z0.S computes
// Z0 = Z0 + Z1 for the active lanes. Operands: Zm, Pg/M, Zdn. The
// governing predicate is a 3-bit field, so only P0-P7 encode here, and
// the class takes the merging qualifier alone: a zeroing form would need
// a MOVPRFX expansion, which one data word cannot carry.
ExtFormPredicated
// ExtFormImmediate is the add/subtract immediate form, the SVE integer
// add/subtract (immediate) class: ADD $255, Z0.S computes
// Z0 = Z0 + 255. Operands: imm{, LSL #8}, Zdn. The constant is an
// unsigned imm8, optionally shifted left by 8 bits; a bare multiple of
// 256 (up to 65280) derives the shift, the spelling the GNU assembler
// canonicalises too. .B takes no shift.
ExtFormImmediate
// ExtFormSignedImmediate is the signed immediate form of the SVE integer
// multiply (immediate) class: MUL $-128, Z0.B computes Z0 = Z0 * -128.
// Operands: simm8, Zdn. No shift exists in this class.
ExtFormSignedImmediate
)
// Arity returns the operand count the form takes.
func (f ExtForm) Arity() int {
switch f {
case ExtFormVectors, ExtFormPredicated:
return 3
case ExtFormImmediate, ExtFormSignedImmediate:
return 2
default:
return 0
}
}
// Kinds returns the operand kind each position of the form wants, in the
// order the operands arrive. The registry uses the list to pick the most
// specific rejection when every matching form refuses an operand list.
func (f ExtForm) Kinds() []ExtOperandKind {
switch f {
case ExtFormVectors:
return []ExtOperandKind{ExtZReg, ExtZReg, ExtZReg}
case ExtFormPredicated:
return []ExtOperandKind{ExtZReg, ExtPReg, ExtZReg}
case ExtFormImmediate, ExtFormSignedImmediate:
return []ExtOperandKind{ExtImm, ExtZReg}
default:
return nil
}
}
// String returns a short label for the form, for diagnostics.
func (f ExtForm) String() string {
switch f {
case ExtFormVectors:
return "unpredicated vectors"
case ExtFormPredicated:
return "predicated (merging)"
case ExtFormImmediate:
return "unsigned immediate"
case ExtFormSignedImmediate:
return "signed immediate"
default:
return "unknown form"
}
}
// ExtFeature names the architecture feature an extended instruction belongs
// to. The field is metadata: the assembler offers every instruction it
// registers, and a feature check is the caller's decision, not the encoder's.
type ExtFeature string
// The features the arm64 layer covers.
const (
ExtFeatureSVE ExtFeature = "sve"
ExtFeatureSVE2 ExtFeature = "sve2"
)
// ExtInstr is one extended instruction: the metadata a lookup needs and the
// encoding as data. Word holds the fixed bits of the 32-bit encoding with
// every operand field and the size field zero; the form says which fields the
// operands fill; the size field receives the arrangement's bits at encode
// time. Ref names the manual entry the encoding is transcribed from, the
// golden source in place of a toolchain oracle.
type ExtInstr struct {
Name string // upper-case mnemonic
Summary string // one line of hover documentation
Word uint32 // fixed encoding bits, operand fields zero
Form ExtForm // operand shape
Size ExtField // element-size field the arrangement fills
Feature ExtFeature // sve or sve2
Ref string // the ARM ARM entry the encoding comes from
}
// Encode assembles the operands into the 4 little-endian bytes of the
// instruction word. The operand kinds, register ranges, arrangements and
// immediate ranges are validated against the form; an operand the class
// cannot carry is an error, never a silent mis-encoding.
func (in ExtInstr) Encode(ops []ExtOperand) ([]byte, error) {
if len(ops) != in.Form.Arity() {
return nil, fmt.Errorf("%s: the %s form takes %d operands, got %d",
in.Name, in.Form, in.Form.Arity(), len(ops))
}
switch in.Form {
case ExtFormVectors:
return in.encodeVectors(ops)
case ExtFormPredicated:
return in.encodePredicated(ops)
case ExtFormImmediate:
return in.encodeImmediate(ops)
case ExtFormSignedImmediate:
return in.encodeSignedImmediate(ops)
default:
return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form)
}
}
// encodeVectors fills the unpredicated three-vector form: Zn, Zm, Zd, all
// under one required arrangement.
func (in ExtInstr) encodeVectors(ops []ExtOperand) ([]byte, error) {
for i, op := range ops {
if op.Kind != ExtZReg {
return nil, fmt.Errorf("%s: operand %d wants a scalable vector register, got %s",
in.Name, i+1, op.Kind)
}
if op.Reg < 0 || op.Reg > 31 {
return nil, fmt.Errorf("%s: operand %d is Z%d, outside Z0-Z31", in.Name, i+1, op.Reg)
}
}
arr, err := in.sharedArrangement(ops)
if err != nil {
return nil, err
}
size, ok := arr.sizeBits()
if !ok {
return nil, fmt.Errorf("%s: arrangement %s has no size encoding in this class", in.Name, arr)
}
word := in.Word
word = extSet(word, extFieldRn, uint32(ops[0].Reg))
word = extSet(word, extFieldRm, uint32(ops[1].Reg))
word = extSet(word, extFieldRd, uint32(ops[2].Reg))
word = extSet(word, in.Size, size)
return extWordLE(word), nil
}
// encodePredicated fills the governed destructive form: Zm, Pg/M, Zdn. The
// predicate is a 3-bit field, the merging qualifier alone, and carries no
// arrangement suffix in this class.
func (in ExtInstr) encodePredicated(ops []ExtOperand) ([]byte, error) {
zm, pg, zdn := ops[0], ops[1], ops[2]
if zm.Kind != ExtZReg {
return nil, fmt.Errorf("%s: operand 1 wants a scalable vector register, got %s",
in.Name, zm.Kind)
}
if zm.Reg < 0 || zm.Reg > 31 {
return nil, fmt.Errorf("%s: operand 1 is Z%d, outside Z0-Z31", in.Name, zm.Reg)
}
if pg.Kind != ExtPReg {
return nil, fmt.Errorf("%s: operand 2 wants a predicate register, got %s",
in.Name, pg.Kind)
}
if pg.Reg < 0 || pg.Reg > 7 {
return nil, fmt.Errorf("%s: operand 2 is P%d, outside P0-P7 in this class", in.Name, pg.Reg)
}
if pg.Qual != ExtQualMerging {
return nil, fmt.Errorf("%s: operand 2 wants the merging qualifier /M, got %q",
in.Name, pg.Qual)
}
if pg.Arr != ExtArrNone {
return nil, fmt.Errorf("%s: the governing predicate carries no arrangement suffix, got %s",
in.Name, pg.Arr)
}
if zdn.Kind != ExtZReg {
return nil, fmt.Errorf("%s: operand 3 wants a scalable vector register, got %s",
in.Name, zdn.Kind)
}
if zdn.Reg < 0 || zdn.Reg > 31 {
return nil, fmt.Errorf("%s: operand 3 is Z%d, outside Z0-Z31", in.Name, zdn.Reg)
}
if zm.Arr != zdn.Arr {
return nil, fmt.Errorf("%s: operands 1 and 3 carry arrangements %s and %s, they must match",
in.Name, zm.Arr, zdn.Arr)
}
size, ok := zdn.Arr.sizeBits()
if !ok {
return nil, fmt.Errorf("%s: arrangement %s has no size encoding in this class", in.Name, zdn.Arr)
}
word := in.Word
word = extSet(word, extFieldRn, uint32(zm.Reg))
word = extSet(word, extFieldPg, uint32(pg.Reg))
word = extSet(word, extFieldRd, uint32(zdn.Reg))
word = extSet(word, in.Size, size)
return extWordLE(word), nil
}
// encodeImmediate fills the add/subtract immediate form: imm{, LSL #8}, Zdn.
// The class encodes an unsigned imm8 with one shift bit, so a bare multiple
// of 256 derives the shift the way the GNU assembler canonicalises it.
func (in ExtInstr) encodeImmediate(ops []ExtOperand) ([]byte, error) {
imm, zdn := ops[0], ops[1]
imm8, sh, err := in.addSubImmediate(imm, zdn.Arr)
if err != nil {
return nil, err
}
word := in.Word
word = extSet(word, extFieldImm8, uint32(imm8))
if sh != 0 {
word = extSet(word, extFieldSh, 1)
}
word, err = in.setDestAndSize(word, zdn)
if err != nil {
return nil, err
}
return extWordLE(word), nil
}
// encodeSignedImmediate fills the multiply immediate form: simm8, Zdn, with
// no shift bit in the class.
func (in ExtInstr) encodeSignedImmediate(ops []ExtOperand) ([]byte, error) {
imm, zdn := ops[0], ops[1]
if imm.Kind != ExtImm {
return nil, fmt.Errorf("%s: operand 1 wants an immediate, got %s", in.Name, imm.Kind)
}
if imm.HasShift {
return nil, fmt.Errorf("%s: the signed immediate class takes no shift", in.Name)
}
if imm.Imm < -128 || imm.Imm > 127 {
return nil, fmt.Errorf("%s: immediate %d is outside the signed 8-bit range -128..127",
in.Name, imm.Imm)
}
word := in.Word
word = extSet(word, extFieldImm8, uint32(imm.Imm))
word, err := in.setDestAndSize(word, zdn)
if err != nil {
return nil, err
}
return extWordLE(word), nil
}
// addSubImmediate resolves the immediate operand of the add/subtract
// immediate class into its imm8 and shift bit: a spelled shift is validated
// as written, a bare multiple of 256 (on .H, .S or .D) derives one.
func (in ExtInstr) addSubImmediate(op ExtOperand, arr ExtArrangement) (imm8, sh int, err error) {
if op.Kind != ExtImm {
return 0, 0, fmt.Errorf("%s: operand 1 wants an immediate, got %s", in.Name, op.Kind)
}
switch {
case op.HasShift:
if op.Shift != 0 && op.Shift != 8 {
return 0, 0, fmt.Errorf("%s: the shift amount must be 0 or 8, got %d", in.Name, op.Shift)
}
if arr == ExtArrB && op.Shift != 0 {
return 0, 0, fmt.Errorf("%s: arrangement .B takes no shift", in.Name)
}
if op.Imm < 0 || op.Imm > 255 {
return 0, 0, fmt.Errorf("%s: immediate %d is outside the unsigned 8-bit range 0..255",
in.Name, op.Imm)
}
return int(op.Imm), op.Shift, nil
case op.Imm >= 0 && op.Imm <= 255:
return int(op.Imm), 0, nil
case arr != ExtArrB && op.Imm >= 256 && op.Imm <= 255<<8 && op.Imm%256 == 0:
// A bare multiple of 256 rides the shift bit, 65280 = 255<<8 included.
return int(op.Imm / 256), 8, nil
default:
return 0, 0, fmt.Errorf("%s: immediate %d is not an unsigned imm8%s, nor a multiple of 256 the shift bit can carry",
in.Name, op.Imm, arr.shiftNote())
}
}
// shiftNote describes where a shifted constant is expressible, for the
// immediate range error.
func (arr ExtArrangement) shiftNote() string {
if arr == ExtArrB {
return " (and .B takes no shifted constant)"
}
return " (a multiple of 256 up to 65280 shifts)"
}
// setDestAndSize fills the destructive destination register and the size
// field from the arrangement the vector carries.
func (in ExtInstr) setDestAndSize(word uint32, zdn ExtOperand) (uint32, error) {
if zdn.Kind != ExtZReg {
return 0, fmt.Errorf("%s: operand 2 wants a scalable vector register, got %s",
in.Name, zdn.Kind)
}
if zdn.Reg < 0 || zdn.Reg > 31 {
return 0, fmt.Errorf("%s: operand 2 is Z%d, outside Z0-Z31", in.Name, zdn.Reg)
}
size, ok := zdn.Arr.sizeBits()
if !ok {
return 0, fmt.Errorf("%s: arrangement %s has no size encoding in this class", in.Name, zdn.Arr)
}
word = extSet(word, extFieldRd, uint32(zdn.Reg))
word = extSet(word, in.Size, size)
return word, nil
}
// sharedArrangement returns the one arrangement all vector operands carry, or
// an error when any operand is bare or they disagree.
func (in ExtInstr) sharedArrangement(ops []ExtOperand) (ExtArrangement, error) {
arr := ops[0].Arr
for i, op := range ops {
if op.Arr == ExtArrNone {
return 0, fmt.Errorf("%s: operand %d carries no arrangement suffix", in.Name, i+1)
}
if op.Arr != arr {
return 0, fmt.Errorf("%s: operand %d carries arrangement %s, want %s",
in.Name, i+1, op.Arr, arr)
}
}
return arr, nil
}
// extWordLE returns a 32-bit encoding word as 4 little-endian bytes.
func extWordLE(w uint32) []byte {
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
}
// --- the arm64 SVE/SVE2 table ------------------------------------------------
// arm64Extensions is the extended instruction layer of arm64: the SVE and
// SVE2 integer add/subtract/multiply family. The Go toolchain knows none of
// these; the encodings are transcribed from the ARM Architecture Reference
// Manual (DDI 0487J, Part C, Chapter C8, the alphabetical list of SVE
// instructions) and cross-checked against the GNU assembler's and LLVM's
// published encodings.
var arm64Extensions = []ExtInstr{
// Unpredicated three-vector forms: ADD Z0.S, Z1.S, Z2.S.
{Name: "ADD", Summary: "Add scalable vector elements, unpredicated",
Word: 0x04200000, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: ADD (vectors, unpredicated)"},
{Name: "SUB", Summary: "Subtract scalable vector elements, unpredicated",
Word: 0x04200400, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUB (vectors, unpredicated)"},
{Name: "SQADD", Summary: "Add signed saturating scalable vector elements, unpredicated",
Word: 0x04201000, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SQADD (vectors, unpredicated)"},
{Name: "UQADD", Summary: "Add unsigned saturating scalable vector elements, unpredicated",
Word: 0x04201400, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UQADD (vectors, unpredicated)"},
{Name: "SQSUB", Summary: "Subtract signed saturating scalable vector elements, unpredicated",
Word: 0x04201800, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SQSUB (vectors, unpredicated)"},
{Name: "UQSUB", Summary: "Subtract unsigned saturating scalable vector elements, unpredicated",
Word: 0x04201c00, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UQSUB (vectors, unpredicated)"},
{Name: "MUL", Summary: "Multiply scalable vector elements, unpredicated",
Word: 0x04206000, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE2,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: MUL (vectors, unpredicated)"},
{Name: "SMULH", Summary: "Multiply signed scalable vector elements, keeping the high half, unpredicated",
Word: 0x04206800, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE2,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SMULH (vectors, unpredicated)"},
{Name: "UMULH", Summary: "Multiply unsigned scalable vector elements, keeping the high half, unpredicated",
Word: 0x04206c00, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE2,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UMULH (vectors, unpredicated)"},
// Governed destructive forms, merging: ADD Z1.S, P0/M, Z0.S.
{Name: "ADD", Summary: "Add scalable vector elements under a governing predicate, merging",
Word: 0x04000000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: ADD (vectors, predicated)"},
{Name: "SUB", Summary: "Subtract scalable vector elements under a governing predicate, merging",
Word: 0x04010000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUB (vectors, predicated)"},
{Name: "SUBR", Summary: "Reverse-subtract scalable vector elements under a governing predicate, merging",
Word: 0x04030000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUBR (vectors, predicated)"},
{Name: "MUL", Summary: "Multiply scalable vector elements under a governing predicate, merging",
Word: 0x04100000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: MUL (vectors, predicated)"},
{Name: "SMULH", Summary: "Multiply signed scalable vector elements, keeping the high half, under a governing predicate, merging",
Word: 0x04120000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SMULH (vectors, predicated)"},
{Name: "UMULH", Summary: "Multiply unsigned scalable vector elements, keeping the high half, under a governing predicate, merging",
Word: 0x04130000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UMULH (vectors, predicated)"},
// Immediate forms: ADD $255, Z0.S.
{Name: "ADD", Summary: "Add an unsigned immediate to scalable vector elements",
Word: 0x2520c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: ADD (vectors, immediate)"},
{Name: "SUB", Summary: "Subtract an unsigned immediate from scalable vector elements",
Word: 0x2521c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUB (vectors, immediate)"},
{Name: "SUBR", Summary: "Subtract scalable vector elements from an unsigned immediate",
Word: 0x2523c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUBR (vectors, immediate)"},
{Name: "SQADD", Summary: "Add a signed saturating unsigned immediate to scalable vector elements",
Word: 0x2524c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SQADD (vectors, immediate)"},
{Name: "UQADD", Summary: "Add an unsigned saturating immediate to scalable vector elements",
Word: 0x2525c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UQADD (vectors, immediate)"},
{Name: "SQSUB", Summary: "Subtract an unsigned immediate from scalable vector elements with signed saturation",
Word: 0x2526c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SQSUB (vectors, immediate)"},
{Name: "UQSUB", Summary: "Subtract an unsigned immediate from scalable vector elements with unsigned saturation",
Word: 0x2527c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UQSUB (vectors, immediate)"},
// The signed immediate of the multiply class: MUL $-128, Z0.B.
{Name: "MUL", Summary: "Multiply scalable vector elements by a signed immediate",
Word: 0x2530c000, Form: ExtFormSignedImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: MUL (vectors, immediate)"},
}
// Extensions returns the extended-instruction layer registered for a, outside
// the generated tables. An architecture whose extended layer is not built
// yet returns nothing: the mechanism is ordinary code, not a build tag, and
// it simply offers no instruction where none is registered.
func Extensions(a Arch) []ExtInstr {
switch a {
case ARM64:
return arm64Extensions
default:
return nil
}
}
+400
View File
@@ -0,0 +1,400 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
import (
"encoding/hex"
"strings"
"testing"
)
// The SVE encodings have no toolchain oracle: go tool asm knows no SVE at
// all. The golden words below are therefore transcribed from the ARM
// Architecture Reference Manual (DDI 0487J, Part C, Chapter C8, the
// alphabetical list of SVE instructions) and cross-checked against two
// independent implementations of the manual, the GNU assembler and LLVM:
// the rows marked "GNU" match a vector in binutils-gdb's own
// gas/testsuite/gas/aarch64/sve.d (assembled under -march=armv8-a+sve), the
// rows marked "LLVM" match the Inst field assignments in
// SVEInstrFormats.td's sve_int_bin_cons_arit_0, sve_int_bin_pred_arit_log,
// sve_int_arith_imm0 and sve2_int_mul classes. Every class is covered by at
// least one vector of each source.
func extInstruction(t *testing.T, mnem string, form ExtForm) ExtInstr {
t.Helper()
for _, in := range Extensions(ARM64) {
if in.Name == mnem && in.Form == form {
return in
}
}
t.Fatalf("no extended %s with the %s form", mnem, form)
return ExtInstr{}
}
func TestArm64ExtGoldenBytes(t *testing.T) {
for _, tt := range []struct {
name string
mnem string
form ExtForm
ops []ExtOperand
want uint32
GNUas string // the matching binutils-gdb sve.d line, empty when the class evidence comes from LLVM alone
}{
// Unpredicated three-vector forms: Zn, Zm, Zd.
{"add z0.b, z0.b, z0.b", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04200000, "04200000 add z0.b, z0.b, z0.b"},
{"add z0.b, z0.b, z31.b", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(31, ExtArrB), ExtVector(0, ExtArrB)},
0x043f0000, "043f0000 add z0.b, z0.b, z31.b"},
{"add z31.b, z0.b, z0.b", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(31, ExtArrB)},
0x0420001f, "0420001f add z31.b, z0.b, z0.b"},
{"add z0.b, z2.b, z0.b", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(2, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04200040, "04200040 add z0.b, z2.b, z0.b"},
{"add z0.h, z0.h, z0.h", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrH), ExtVector(0, ExtArrH), ExtVector(0, ExtArrH)},
0x04600000, "04600000 add z0.h, z0.h, z0.h"},
{"add z0.s, z0.s, z0.s", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrS), ExtVector(0, ExtArrS), ExtVector(0, ExtArrS)},
0x04a00000, "04a00000 add z0.s, z0.s, z0.s"},
{"add z0.d, z0.d, z0.d", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrD), ExtVector(0, ExtArrD), ExtVector(0, ExtArrD)},
0x04e00000, "04e00000 add z0.d, z0.d, z0.d"},
{"sub z0.b, z0.b, z0.b", "SUB", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04200400, "04200400 sub z0.b, z0.b, z0.b"},
{"sub z0.b, z0.b, z3.b", "SUB", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(3, ExtArrB), ExtVector(0, ExtArrB)},
0x04230400, "04230400 sub z0.b, z0.b, z3.b"},
{"sqadd z0.b, z0.b, z0.b", "SQADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04201000, "04201000 sqadd z0.b, z0.b, z0.b"},
{"sqadd z0.b, z0.b, z3.b", "SQADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(3, ExtArrB), ExtVector(0, ExtArrB)},
0x04231000, "04231000 sqadd z0.b, z0.b, z3.b"},
{"sqadd z0.d, z0.d, z0.d", "SQADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrD), ExtVector(0, ExtArrD), ExtVector(0, ExtArrD)},
0x04e01000, "04e01000 sqadd z0.d, z0.d, z0.d"},
{"uqadd z0.b, z0.b, z0.b", "UQADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04201400, "04201400 uqadd z0.b, z0.b, z0.b"},
{"sqsub z0.b, z0.b, z0.b", "SQSUB", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04201800, "04201800 sqsub z0.b, z0.b, z0.b"},
{"sqsub z0.h, z0.h, z0.h", "SQSUB", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrH), ExtVector(0, ExtArrH), ExtVector(0, ExtArrH)},
0x04601800, "04601800 sqsub z0.h, z0.h, z0.h"},
{"uqsub z0.b, z0.b, z0.b", "UQSUB", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04201c00, "04201c00 uqsub z0.b, z0.b, z0.b"},
{"mul z0.b, z0.b, z0.b (sve2)", "MUL", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04206000, "04206000 mul z0.b, z0.b, z0.b"},
{"mul z17.b, z21.b, z27.b (sve2)", "MUL", ExtFormVectors,
[]ExtOperand{ExtVector(21, ExtArrB), ExtVector(27, ExtArrB), ExtVector(17, ExtArrB)},
0x043b62b1, "043b62b1 mul z17.b, z21.b, z27.b"},
{"mul z0.d, z0.d, z0.d (sve2)", "MUL", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrD), ExtVector(0, ExtArrD), ExtVector(0, ExtArrD)},
0x04e06000, "04e06000 mul z0.d, z0.d, z0.d"},
{"smulh z0.b, z0.b, z0.b (sve2)", "SMULH", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04206800, "04206800 smulh z0.b, z0.b, z0.b"},
{"smulh z17.b, z21.b, z27.b (sve2)", "SMULH", ExtFormVectors,
[]ExtOperand{ExtVector(21, ExtArrB), ExtVector(27, ExtArrB), ExtVector(17, ExtArrB)},
0x043b6ab1, "043b6ab1 smulh z17.b, z21.b, z27.b"},
{"umulh z0.b, z0.b, z0.b (sve2)", "UMULH", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04206c00, "04206c00 umulh z0.b, z0.b, z0.b"},
{"umulh z17.b, z21.b, z27.b (sve2)", "UMULH", ExtFormVectors,
[]ExtOperand{ExtVector(21, ExtArrB), ExtVector(27, ExtArrB), ExtVector(17, ExtArrB)},
0x043b6eb1, "043b6eb1 umulh z17.b, z21.b, z27.b"},
// Governed destructive forms, merging: Zm, Pg/M, Zdn.
{"add z0.b, p0/m, z0.b", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04000000, "04000000 add z0.b, p0/m, z0.b, z0.b"},
{"add z0.b, p2/m, z0.b", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(2, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04000800, "04000800 add z0.b, p2/m, z0.b, z0.b"},
{"add z0.b, p7/m, z0.b", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(7, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04001c00, "04001c00 add z0.b, p7/m, z0.b, z0.b"},
{"add z31.b, p0/m, z31.b", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(31, ExtArrB)},
0x0400001f, "0400001f add z31.b, p0/m, z31.b, z0.b"},
{"add z0.b, p0/m, z0.b (zm 31)", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(31, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x040003e0, "040003e0 add z0.b, p0/m, z0.b, z31.b"},
{"add z0.s, p0/m, z0.s", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrS), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrS)},
0x04800000, "04800000 add z0.s, p0/m, z0.s, z0.s"},
{"sub z0.b, p0/m, z0.b", "SUB", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04010000, "04010000 sub z0.b, p0/m, z0.b, z0.b"},
{"sub z0.b, p7/m, z0.b", "SUB", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(7, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04011c00, "04011c00 sub z0.b, p7/m, z0.b, z0.b"},
{"sub z3.b, p0/m, z3.b", "SUB", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(3, ExtArrB)},
0x04010003, "04010003 sub z3.b, p0/m, z3.b, z0.b"},
{"subr z0.b, p0/m, z0.b", "SUBR", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04030000, "04030000 subr z0.b, p0/m, z0.b, z0.b"},
{"subr z0.h, p0/m, z0.h", "SUBR", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrH), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrH)},
0x04430000, "04430000 subr z0.h, p0/m, z0.h, z0.h"},
{"mul z0.b, p0/m, z0.b", "MUL", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04100000, "04100000 mul z0.b, p0/m, z0.b, z0.b"},
{"mul z0.b, p2/m, z0.b", "MUL", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(2, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04100800, "04100800 mul z0.b, p2/m, z0.b, z0.b"},
{"mul z0.b, p0/m, z0.b (zm 31)", "MUL", ExtFormPredicated,
[]ExtOperand{ExtVector(31, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x041003e0, "041003e0 mul z0.b, p0/m, z0.b, z31.b"},
{"smulh z0.b, p0/m, z0.b", "SMULH", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04120000, "04120000 smulh z0.b, p0/m, z0.b, z0.b"},
{"smulh z0.b, p2/m, z0.b", "SMULH", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(2, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04120800, "04120800 smulh z0.b, p2/m, z0.b, z0.b"},
{"smulh z0.s, p0/m, z0.s", "SMULH", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrS), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrS)},
0x04920000, "04920000 smulh z0.s, p0/m, z0.s, z0.s"},
{"umulh z0.b, p0/m, z0.b", "UMULH", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04130000, "04130000 umulh z0.b, p0/m, z0.b, z0.b"},
// Immediate forms: imm{, LSL #8}, Zdn.
{"add z0.b, z0.b, #0", "ADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2520c000, "2520c000 add z0.b, z0.b, #0"},
{"add z0.b, z0.b, #127", "ADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(127), ExtVector(0, ExtArrB)},
0x2520cfe0, "2520cfe0 add z0.b, z0.b, #127"},
{"add z0.h, z0.h, #0, lsl #8", "ADD", ExtFormImmediate,
[]ExtOperand{ExtShiftedImmediate(0, 8), ExtVector(0, ExtArrH)},
0x2560e000, "2560e000 add z0.h, z0.h, #0, lsl #8"},
{"add z0.h, z0.h, #32512 (derived shift)", "ADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(32512), ExtVector(0, ExtArrH)},
0x2560efe0, "2560efe0 is sqsub's GNU word for #32512; the classes share the encoding"},
{"sub z0.b, z0.b, #0", "SUB", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2521c000, "2521c000 sub z0.b, z0.b, #0"},
{"subr z0.b, z0.b, #0", "SUBR", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2523c000, "2523c000 subr z0.b, z0.b, #0"},
{"subr z0.b, z0.b, #255", "SUBR", ExtFormImmediate,
[]ExtOperand{ExtImmediate(255), ExtVector(0, ExtArrB)},
0x2523dfe0, "2523dfe0 subr z0.b, z0.b, #255"},
{"subr z0.h, z0.h, #0, lsl #8", "SUBR", ExtFormImmediate,
[]ExtOperand{ExtShiftedImmediate(0, 8), ExtVector(0, ExtArrH)},
0x2563e000, "2563e000 subr z0.h, z0.h, #0, lsl #8"},
{"sqadd z0.b, z0.b, #0", "SQADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2524c000, "2524c000 sqadd z0.b, z0.b, #0"},
{"uqadd z0.b, z0.b, #0", "UQADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2525c000, "2525c000 uqadd z0.b, z0.b, #0"},
{"sqsub z0.b, z0.b, #0", "SQSUB", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2526c000, "2526c000 sqsub z0.b, z0.b, #0"},
{"sqsub z0.b, z0.b, #255", "SQSUB", ExtFormImmediate,
[]ExtOperand{ExtImmediate(255), ExtVector(0, ExtArrB)},
0x2526dfe0, "2526dfe0 sqsub z0.b, z0.b, #255"},
{"sqsub z0.h, z0.h, #0, lsl #8", "SQSUB", ExtFormImmediate,
[]ExtOperand{ExtShiftedImmediate(0, 8), ExtVector(0, ExtArrH)},
0x2566e000, "2566e000 sqsub z0.h, z0.h, #0, lsl #8"},
{"uqsub z0.b, z0.b, #0", "UQSUB", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2527c000, "2527c000 uqsub z0.b, z0.b, #0 (class vector from the GNU table and LLVM: sve_int_arith_imm0 opc 0b111)"},
{"mul z0.b, z0.b, #0", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2530c000, "2530c000 mul z0.b, z0.b, #0"},
{"mul z0.b, z0.b, #127", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(127), ExtVector(0, ExtArrB)},
0x2530cfe0, "2530cfe0 mul z0.b, z0.b, #127"},
{"mul z0.b, z0.b, #-128", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(-128), ExtVector(0, ExtArrB)},
0x2530d000, "2530d000 mul z0.b, z0.b, #-128"},
{"mul z0.b, z0.b, #-1", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(-1), ExtVector(0, ExtArrB)},
0x2530dfe0, "2530dfe0 mul z0.b, z0.b, #-1"},
{"mul z0.h, z0.h, #0", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrH)},
0x2570c000, "2570c000 mul z0.h, z0.h, #0"},
} {
in := extInstruction(t, tt.mnem, tt.form)
got, err := in.Encode(tt.ops)
if err != nil {
t.Errorf("%s: encode: %v", tt.name, err)
continue
}
if want := hex.EncodeToString(extWordLE(tt.want)); hex.EncodeToString(got) != want {
t.Errorf("%s:\n got %x\n want %s", tt.name, got, want)
}
}
}
// TestArm64ExtGoldenSources pins the cross-check contract: every encoding
// class in the table carries at least one GNU-assembler vector, so no class
// rests on transcription alone.
func TestArm64ExtGoldenSources(t *testing.T) {
classes := map[ExtForm]bool{}
for _, tt := range []struct {
mnem string
form ExtForm
}{
{"ADD", ExtFormVectors}, {"SUB", ExtFormVectors}, {"SQADD", ExtFormVectors},
{"UQADD", ExtFormVectors}, {"SQSUB", ExtFormVectors}, {"UQSUB", ExtFormVectors},
{"MUL", ExtFormVectors}, {"SMULH", ExtFormVectors}, {"UMULH", ExtFormVectors},
{"ADD", ExtFormPredicated}, {"SUB", ExtFormPredicated}, {"SUBR", ExtFormPredicated},
{"MUL", ExtFormPredicated}, {"SMULH", ExtFormPredicated}, {"UMULH", ExtFormPredicated},
{"ADD", ExtFormImmediate}, {"SUB", ExtFormImmediate}, {"SUBR", ExtFormImmediate},
{"SQADD", ExtFormImmediate}, {"UQADD", ExtFormImmediate},
{"SQSUB", ExtFormImmediate}, {"UQSUB", ExtFormImmediate},
{"MUL", ExtFormSignedImmediate},
} {
if _, ok := extInstructionQuiet(tt.mnem, tt.form); !ok {
t.Errorf("the table lacks %s with the %s form", tt.mnem, tt.form)
}
classes[tt.form] = true
}
for _, form := range []ExtForm{ExtFormVectors, ExtFormPredicated, ExtFormImmediate, ExtFormSignedImmediate} {
if !classes[form] {
t.Errorf("no golden vectors cover the %s form", form)
}
}
}
func extInstructionQuiet(mnem string, form ExtForm) (ExtInstr, bool) {
for _, in := range Extensions(ARM64) {
if in.Name == mnem && in.Form == form {
return in, true
}
}
return ExtInstr{}, false
}
// TestArm64ExtTableIntegrity checks the metadata contract: every entry names
// its manual reference, summary and feature, and the element-size field sits
// at bits 23..22 where the manual puts it for every class in the family.
func TestArm64ExtTableIntegrity(t *testing.T) {
for _, in := range Extensions(ARM64) {
if in.Name == "" || in.Summary == "" || in.Ref == "" {
t.Errorf("%+v: name, summary and reference are mandatory", in)
}
if in.Feature != ExtFeatureSVE && in.Feature != ExtFeatureSVE2 {
t.Errorf("%s: feature %q is neither sve nor sve2", in.Name, in.Feature)
}
if in.Form.Arity() < 2 || in.Form.Arity() > 3 {
t.Errorf("%s: form %d carries an unusable arity %d", in.Name, in.Form, in.Form.Arity())
}
if in.Size.Off != 22 || in.Size.Width != 2 {
t.Errorf("%s: the size field sits at bits %d..%d, the classes here put it at 23..22",
in.Name, in.Size.Off, in.Size.Off+in.Size.Width-1)
}
// The destination register field and the element-size field are
// operands everywhere in this family, so Word carries both zero; the
// class opcodes live around them and stay where they are.
if in.Word&0x1f != 0 || in.Word&(0x3<<22) != 0 {
t.Errorf("%s: word %08x carries destination or size bits, want them zero", in.Name, in.Word)
}
}
}
func TestArm64ExtRejects(t *testing.T) {
rgb := func(rs ...int) []ExtOperand {
ops := make([]ExtOperand, len(rs))
for i, r := range rs {
ops[i] = ExtVector(r, ExtArrB)
}
return ops
}
for _, tt := range []struct {
name string
mnem string
form ExtForm
ops []ExtOperand
quote string // a fragment the error carries
}{
{"no arrangement", "ADD", ExtFormVectors,
[]ExtOperand{{Kind: ExtZReg, Reg: 0}, ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
"no arrangement"},
{"mismatched arrangements", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrS), ExtVector(0, ExtArrB)},
"want .B"},
{"wrong arity", "ADD", ExtFormVectors, rgb(0, 0), "takes 3 operands"},
{"predicate in a vector position", "ADD", ExtFormVectors,
[]ExtOperand{ExtPredicate(0, ExtQualNone), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
"scalable vector register"},
{"quadword arrangement has no size encoding", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrQ), ExtVector(0, ExtArrQ), ExtVector(0, ExtArrQ)},
"no size encoding"},
{"zeroing qualifier", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualZeroing), ExtVector(0, ExtArrB)},
"/M"},
{"predicate beyond the 3-bit field", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(8, ExtQualMerging), ExtVector(0, ExtArrB)},
"P0-P7"},
{"predicate arrangement suffix", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtOperand{Kind: ExtPReg, Reg: 0, Qual: ExtQualMerging, Arr: ExtArrB}, ExtVector(0, ExtArrB)},
"arrangement"},
{"predicate operands disagree on arrangement", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrS)},
"must match"},
{"bare 256 on .B", "ADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(256), ExtVector(0, ExtArrB)},
"immediate 256"},
{"negative unsigned immediate", "ADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(-1), ExtVector(0, ExtArrB)},
"immediate -1"},
{"multiple of 256 beyond the imm8 span", "ADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(65536), ExtVector(0, ExtArrH)},
"immediate 65536"},
{"shift amount other than 0 or 8", "ADD", ExtFormImmediate,
[]ExtOperand{ExtShiftedImmediate(1, 4), ExtVector(0, ExtArrS)},
"0 or 8"},
{"shifted constant on .B", "ADD", ExtFormImmediate,
[]ExtOperand{ExtShiftedImmediate(1, 8), ExtVector(0, ExtArrB)},
".B takes no shift"},
{"register where the immediate belongs", "ADD", ExtFormImmediate,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
"wants an immediate"},
{"signed immediate over the top", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(128), ExtVector(0, ExtArrB)},
"128"},
{"signed immediate under the floor", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(-129), ExtVector(0, ExtArrB)},
"-129"},
{"shift in the signed class", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtShiftedImmediate(1, 8), ExtVector(0, ExtArrB)},
"no shift"},
} {
in := extInstruction(t, tt.mnem, tt.form)
_, err := in.Encode(tt.ops)
if err == nil {
t.Errorf("%s: encode succeeded, want an error", tt.name)
continue
}
if !strings.Contains(err.Error(), tt.quote) {
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
}
}
}
// TestExtensionsArchBinding pins the registry's architecture binding: the
// extended layer exists for arm64 alone until an amd64 table attaches, and no
// other architecture sees a single SVE instruction.
func TestExtensionsArchBinding(t *testing.T) {
for _, a := range []Arch{AMD64, RISCV, LOONG64, Unknown} {
if got := Extensions(a); len(got) != 0 {
t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got))
}
}
if got := Extensions(ARM64); len(got) == 0 {
t.Error("Extensions(ARM64) is empty")
}
}
+108 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
@@ -364,6 +364,8 @@ var arm64GeneratedInstrs = []string{
"REVW",
"ROR",
"RORW",
"RPRFM",
"SB",
"SBC",
"SBCS",
"SBCSW",
@@ -477,23 +479,68 @@ var arm64GeneratedInstrs = []string{
"UXTH",
"UXTHW",
"UXTW",
"VABS",
"VADD",
"VADDP",
"VADDV",
"VAND",
"VBCAX",
"VBIC",
"VBIF",
"VBIT",
"VBSL",
"VCLS",
"VCLZ",
"VCMEQ",
"VCMGE",
"VCMGT",
"VCMHI",
"VCMHS",
"VCMLE",
"VCMLT",
"VCMTST",
"VCNT",
"VDUP",
"VEOR",
"VEOR3",
"VEXT",
"VFABS",
"VFADD",
"VFADDP",
"VFCMEQ",
"VFCMGE",
"VFCMGT",
"VFCMLE",
"VFCMLT",
"VFCVTL",
"VFCVTL2",
"VFCVTN",
"VFCVTN2",
"VFCVTZS",
"VFCVTZU",
"VFDIV",
"VFMAX",
"VFMAXNM",
"VFMAXNMP",
"VFMAXNMV",
"VFMAXP",
"VFMAXV",
"VFMIN",
"VFMINNM",
"VFMINNMP",
"VFMINNMV",
"VFMINP",
"VFMINV",
"VFMLA",
"VFMLS",
"VFMUL",
"VFNEG",
"VFRINTM",
"VFRINTN",
"VFRINTP",
"VFRINTZ",
"VFSQRT",
"VFSUB",
"VLD1",
"VLD1R",
"VLD2",
@@ -502,11 +549,17 @@ var arm64GeneratedInstrs = []string{
"VLD3R",
"VLD4",
"VLD4R",
"VMLA",
"VMLS",
"VMOV",
"VMOVD",
"VMOVI",
"VMOVQ",
"VMOVS",
"VMUL",
"VNEG",
"VNOT",
"VORN",
"VORR",
"VPMULL",
"VPMULL2",
@@ -515,14 +568,47 @@ var arm64GeneratedInstrs = []string{
"VREV16",
"VREV32",
"VREV64",
"VSCVTF",
"VSHADD",
"VSHL",
"VSHRN",
"VSHRN2",
"VSLI",
"VSMAX",
"VSMAXP",
"VSMAXV",
"VSMIN",
"VSMINP",
"VSMINV",
"VSMLAL",
"VSMLAL2",
"VSMLSL",
"VSMLSL2",
"VSMULL",
"VSMULL2",
"VSQABS",
"VSQADD",
"VSQNEG",
"VSQSHL",
"VSQSUB",
"VSQXTN",
"VSQXTN2",
"VSQXTUN",
"VSQXTUN2",
"VSRHADD",
"VSRI",
"VSRSHR",
"VSSHL",
"VSSHLL",
"VSSHLL2",
"VSSHR",
"VST1",
"VST2",
"VST3",
"VST4",
"VSUB",
"VSXTL",
"VSXTL2",
"VTBL",
"VTBX",
"VTRN1",
@@ -530,8 +616,27 @@ var arm64GeneratedInstrs = []string{
"VUADDLV",
"VUADDW",
"VUADDW2",
"VUCVTF",
"VUHADD",
"VUMAX",
"VUMAXP",
"VUMAXV",
"VUMIN",
"VUMINP",
"VUMINV",
"VUMLAL",
"VUMLAL2",
"VUMLSL",
"VUMLSL2",
"VUMULL",
"VUMULL2",
"VUQADD",
"VUQSHL",
"VUQSUB",
"VUQXTN",
"VUQXTN2",
"VURHADD",
"VUSHL",
"VUSHLL",
"VUSHLL2",
"VUSHR",
@@ -541,6 +646,8 @@ var arm64GeneratedInstrs = []string{
"VUZP1",
"VUZP2",
"VXAR",
"VXTN",
"VXTN2",
"VZIP1",
"VZIP2",
"WFE",
+1 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/util.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
+10 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
@@ -152,6 +152,8 @@ var loong64GeneratedInstrs = []string{
"FNMADDF",
"FNMSUBD",
"FNMSUBF",
"FRINTD",
"FRINTF",
"FSCALEBD",
"FSCALEBF",
"FSEL",
@@ -177,7 +179,10 @@ var loong64GeneratedInstrs = []string{
"FTINTWF",
"JIRL",
"LL",
"LLACQV",
"LLACQW",
"LLV",
"LLW",
"LU12IW",
"LU32ID",
"LU52ID",
@@ -248,7 +253,11 @@ var loong64GeneratedInstrs = []string{
"ROTR",
"ROTRV",
"SC",
"SCQ",
"SCRELV",
"SCRELW",
"SCV",
"SCW",
"SGT",
"SGTU",
"SLL",
+32 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
@@ -81,6 +81,9 @@ var riscvGeneratedInstrs = []string{
"CLD",
"CLDSP",
"CLI",
"CLMUL",
"CLMULH",
"CLMULR",
"CLUI",
"CLW",
"CLWSP",
@@ -95,13 +98,20 @@ var riscvGeneratedInstrs = []string{
"CSDSP",
"CSLLI",
"CSRAI",
"CSRC",
"CSRCI",
"CSRLI",
"CSRR",
"CSRRC",
"CSRRCI",
"CSRRS",
"CSRRSI",
"CSRRW",
"CSRRWI",
"CSRS",
"CSRSI",
"CSRW",
"CSRWI",
"CSUB",
"CSUBW",
"CSW",
@@ -259,6 +269,7 @@ var riscvGeneratedInstrs = []string{
"ORCB",
"ORI",
"ORN",
"PAUSE",
"RDCYCLE",
"RDINSTRET",
"RDTIME",
@@ -322,6 +333,8 @@ var riscvGeneratedInstrs = []string{
"VADDVI",
"VADDVV",
"VADDVX",
"VANDNVV",
"VANDNVX",
"VANDVI",
"VANDVV",
"VANDVX",
@@ -329,8 +342,17 @@ var riscvGeneratedInstrs = []string{
"VASUBUVX",
"VASUBVV",
"VASUBVX",
"VBREV8V",
"VBREVV",
"VCLMULHVV",
"VCLMULHVX",
"VCLMULVV",
"VCLMULVX",
"VCLZV",
"VCOMPRESSVM",
"VCPOPM",
"VCPOPV",
"VCTZV",
"VDIVUVV",
"VDIVUVX",
"VDIVVV",
@@ -743,10 +765,16 @@ var riscvGeneratedInstrs = []string{
"VREMUVX",
"VREMVV",
"VREMVX",
"VREV8V",
"VRGATHEREI16VV",
"VRGATHERVI",
"VRGATHERVV",
"VRGATHERVX",
"VROLVV",
"VROLVX",
"VRORVI",
"VRORVV",
"VRORVX",
"VRSUBVI",
"VRSUBVX",
"VS1RV",
@@ -950,6 +978,9 @@ var riscvGeneratedInstrs = []string{
"VWMULVX",
"VWREDSUMUVS",
"VWREDSUMVS",
"VWSLLVI",
"VWSLLVV",
"VWSLLVX",
"VWSUBUVV",
"VWSUBUVX",
"VWSUBUWV",
+15 -66
View File
@@ -11,7 +11,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestGOObjectAARCH64Structure checks the basic structure of the emitted
@@ -178,26 +178,10 @@ func main() {
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
}
defer os.RemoveAll(work)
st := parseBuildLog(t, buildLog, "main_arm64.s")
defer os.RemoveAll(st.work)
// Expand $WORK in the object path.
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
// Read the toolchain-produced object and assemble the same source with gasm.
// Assemble the same source with gasm and substitute the object.
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
if err != nil {
t.Fatal(err)
@@ -210,30 +194,16 @@ func main() {
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
gasmObj, err := img.GOObjectAARCH64("a64link", "main_arm64.s")
// The package path is "main", the prefix the Go code's references carry.
gasmObj, err := img.GOObjectAARCH64("main", "main_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
// Replace the toolchain-produced object with gasm's.
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
// Re-link.
if linkLine == "" {
t.Skip("could not find link command in build log")
}
// Expand $WORK in the link command.
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"),
gasmObj, "GOARCH=arm64")
// Verify the binary exists and contains the symbol.
binPath := filepath.Join(dir, "prog")
binPath := filepath.Join(dir, "prog2")
if _, err := os.Stat(binPath); err != nil {
t.Fatalf("binary not found: %v", err)
}
@@ -296,23 +266,8 @@ func main() {
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || linkLine == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
}
defer os.RemoveAll(work)
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
st := parseBuildLog(t, buildLog, "main_arm64.s")
defer os.RemoveAll(st.work)
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
if err != nil {
@@ -326,19 +281,13 @@ func main() {
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
gasmObj, err := img.GOObjectAARCH64("a64dlink", "main_arm64.s")
gasmObj, err := img.GOObjectAARCH64("main", "main_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
binData, err := os.ReadFile(filepath.Join(dir, "prog"))
substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"),
gasmObj, "GOARCH=arm64")
binData, err := os.ReadFile(filepath.Join(dir, "prog2"))
if err != nil {
t.Fatal(err)
}
+510 -74
View File
@@ -9,7 +9,7 @@ import (
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// assembleARM64 assembles an AArch64 (arm64) TEXT function body into machine
@@ -458,10 +458,13 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64Excl(mnem, enc.op, ops)
}
// LSE atomics (LDADD, CAS, SWP).
// LSE atomics (LDADD, CAS, SWP) and the compare-and-swap pair.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FLSE {
return encodeARM64LSEAtom(mnem, enc.op, ops)
}
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCASP {
return encodeARM64CASP(mnem, enc.op, ops)
}
// Bitfield/shift (ASR, LSL, LSR, ROR, BFI, BFXIL, SBFM, UBFM).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfield {
@@ -555,16 +558,26 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
// this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist
// only in the zero-immediate form (a64SimdVZero), so they route here with
// an empty register-form spec.
// an empty register-form spec. VSQSHL/VUQSHL keep their shift-by-
// immediate route when the first operand is an immediate.
if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 {
if !arm64SimdShiftImmRoute(mnem, ops) {
return encodeARM64SimdV(mnem, spec, ops)
}
}
// Arrangement-aware SIMD two-register (VREV32, VREV64, VUADDLV, VMOV).
if spec, ok := a64SimdV2Table[mnem]; ok {
return encodeARM64SimdV2(mnem, spec, ops)
}
// Narrow/long/wide SIMD families whose size and Q bits read off one
// designated operand (VXTN, VSXTL, VUADDW, VUMULL, VSHRN, VSSHLL, VFCVTN
// and friends).
if spec, ok := a64SimdNLTable[mnem]; ok {
return encodeARM64SimdNL(mnem, spec, ops)
}
// SIMD four-register and immediate three-register (VEOR3, VBCAX, VXAR,
// VEXT).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSIMDV4 {
@@ -587,6 +600,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64ShiftImm(mnem, enc.op, ops)
}
// SIMD move immediate: VMOVI $imm8, Vd.B8/B16.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FVMoviImm {
return encodeARM64MoviImm(ops)
}
// VMOVS/VMOVD/VMOVQ with a large constant: a literal pool load.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FMoviLit {
return encodeARM64MoviLit(mnem, enc.op, ops, relocs, lits)
@@ -637,6 +655,20 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
return a64wordLE(a64UncondBranch(opc, uint32(rn), 0)), nil
}
// The bare spelling BL R9 is the same indirect branch: the parser reads
// a bare identifier as a symbol, and one named for a register is an
// indirect branch through it, which the toolchain accepts alongside the
// parenthesised form (BL (R3) and BL R3 both encode BLR R3).
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" && op.Addr.Index == "" {
if rn := arm64RegNum(op.Addr.Sym.Name); rn >= 0 {
opc := uint32(0) // BR
if link {
opc = 1 // BLR
}
return a64wordLE(a64UncondBranch(opc, uint32(rn), 0)), nil
}
}
// Symbol reference: BL sym(SB), or B sym(SB) for a tail call, against a
// relocation (R_CALLARM64 either way).
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
@@ -1454,6 +1486,15 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo,
}
return encodeARM64SBAddr(src.Imm.Sym, rd, relocs), nil
}
// Immediate → memory: only storing zero is encodable (the ZR
// register); the toolchain rejects any other immediate-to-memory
// combination ("illegal combination").
if isMemOperand(dst) {
if arm64Imm64(src) != 0 {
return nil, fmt.Errorf("%s: illegal combination: an immediate store must be zero", mnem)
}
return encodeARM64MemOp(mnem, dst, 31, false, fi, "")
}
rd := arm64RegNum(operandRegName(dst))
if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
@@ -2545,6 +2586,263 @@ func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
}
// encodeARM64CASP encodes the compare-and-swap pair: CASP (Rs, Rs+1), (Rn),
// (Rt, Rt+1). Both pairs must start on an even register and be contiguous;
// the second register of each pair rides no encoding field.
func encodeARM64CASP(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects (Rs, Rs+1), (Rn), (Rt, Rt+1), got %d operands", mnem, len(ops))
}
rs, rs1, ok := arm64PairOf(ops[0])
if !ok {
return nil, fmt.Errorf("%s expects a source register pair (Rs, Rs+1)", mnem)
}
rn, err := arm64ExclMem(mnem, ops[1])
if err != nil {
return nil, err
}
rt, rt1, ok := arm64PairOf(ops[2])
if !ok {
return nil, fmt.Errorf("%s expects a destination register pair (Rt, Rt+1)", mnem)
}
if rs&1 != 0 {
return nil, fmt.Errorf("%s: source register pair must start from an even register", mnem)
}
if rt&1 != 0 {
return nil, fmt.Errorf("%s: destination register pair must start from an even register", mnem)
}
if rs != rs1-1 {
return nil, fmt.Errorf("%s: source register pair must be contiguous", mnem)
}
if rt != rt1-1 {
return nil, fmt.Errorf("%s: destination register pair must be contiguous", mnem)
}
if rt == 31 {
return nil, fmt.Errorf("%s: illegal destination register", mnem)
}
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
}
// encodeARM64MoviImm encodes VMOVI $imm8, Vd.B8/B16: the modified-immediate
// form of the SIMD move (asm7.go case 86). Only the byte arrangements exist
// and the immediate is one unsigned byte.
func encodeARM64MoviImm(ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("VMOVI expects $immediate, Vd.<T>")
}
vd, ok := arm64VecOf(ops[1])
if !ok || vd.hasIdx || (vd.arr != "B8" && vd.arr != "B16") {
return nil, fmt.Errorf("VMOVI: destination arrangement must be B8 or B16")
}
imm := arm64Imm64(ops[0])
if imm < 0 || imm > 255 {
return nil, fmt.Errorf("VMOVI: immediate constant %d out of range (0..255)", imm)
}
q := uint32(0)
if vd.arr == "B16" {
q = 1 << 30
}
w := 0x0f00e400 | q | uint32(imm>>5&7)<<16 | uint32(imm&0x1f)<<5 | uint32(vd.reg)
return a64wordLE(w), nil
}
// arm64SimdShiftImmRoute reports whether a mnemonic carries both a shift-by-
// immediate and a register form and the operands spell the immediate one: the
// dedicated shift route keeps them.
func arm64SimdShiftImmRoute(mnem string, ops []*ast.Operand) bool {
if mnem != "VSQSHL" && mnem != "VUQSHL" {
return false
}
return len(ops) > 0 && isImmOperand(ops[0])
}
// arm64SimdNLArr describes one arrangement for the narrow/long/wide families:
// the element width in bytes and whether the spelling names the 128-bit form.
func arm64SimdNLArr(arr string) (esize int, wide bool, ok bool) {
switch arr {
case "B8", "B16":
return 1, arr == "B16", true
case "H4", "H8":
return 2, arr == "H8", true
case "S2", "S4":
return 4, arr == "S4", true
case "D1", "D2":
return 8, arr == "D2", true
}
return 0, false, false
}
// arm64SimdLongPair validates a long pairing (source narrow, destination
// wide): the destination element is twice the source's, the destination is
// always spelled the wide way (H8/S4/D2) and the source carries the 128-bit
// flag exactly for the .2 spellings.
func arm64SimdLongPair(mnem, src, dst string, two bool) error {
se, sw, ok1 := arm64SimdNLArr(src)
de, dw, ok2 := arm64SimdNLArr(dst)
if !ok1 || !ok2 || de != 2*se {
return fmt.Errorf("%s: incompatible arrangements %s and %s", mnem, src, dst)
}
if !dw || sw != two {
return fmt.Errorf("%s: operand mismatch for the %s spelling", mnem, mnem)
}
return nil
}
// arm64SimdNarrowPair validates a narrow pairing (source wide, destination
// narrow): the mirror image of arm64SimdLongPair.
func arm64SimdNarrowPair(mnem, src, dst string, two bool) error {
se, sw, ok1 := arm64SimdNLArr(src)
de, dw, ok2 := arm64SimdNLArr(dst)
if !ok1 || !ok2 || se != 2*de {
return fmt.Errorf("%s: incompatible arrangements %s and %s", mnem, src, dst)
}
if !sw || dw != two {
return fmt.Errorf("%s: operand mismatch for the %s spelling", mnem, mnem)
}
return nil
}
// arm64SimdNLArrBits returns the arrangement bits a narrow/long/wide
// instruction contributes: the driving arrangement's size and Q bits, or for
// the FCVT family only the Q bit, whose size field is fixed in the base.
func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
if spec.qonly {
if two {
return 1 << 30
}
return 0
}
return a64ArrBits[a64ArrIndex(drive)]
}
// encodeARM64SimdNL encodes the narrow/long/wide SIMD families
// (a64SimdNLTable): XTN and FCVTN narrow a wide source, SXTL and FCVTL
// lengthen, the MULL/MLAL/MLSL group multiplies long, UADDW widens, and the
// SSHLL/USHLL and SHRN shifts carry their immediate in the immh:immb field.
// The size and Q bits read off the designated driving operand, and the .2
// spellings force the 128-bit side through their own arrangement.
func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]byte, error) {
two := strings.HasSuffix(mnem, "2")
// vecAt parses operand i as a vector register with an arrangement.
vecAt := func(i int) (a64Vec, bool) {
if i >= len(ops) {
return a64Vec{}, false
}
v, ok := arm64VecOf(ops[i])
if !ok || v.hasIdx {
return a64Vec{}, false
}
return v, true
}
switch spec.form {
case a64NLTwoNarrow, a64NLTwoLong:
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vn, ok1 := vecAt(0)
vd, ok2 := vecAt(1)
if !ok1 || !ok2 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
var drive string
var pairErr error
if spec.form == a64NLTwoNarrow {
// XTN/FCVTN: wide source into a narrow destination; the
// arrangement bits follow the destination.
drive, pairErr = vd.arr, arm64SimdNarrowPair(mnem, vn.arr, vd.arr, two)
} else {
// SXTL/UXTL/FCVTL: narrow source into a wide destination; the
// arrangement bits follow the source.
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vn.arr, vd.arr, two)
}
if pairErr != nil {
return nil, pairErr
}
arrBits := arm64SimdNLArrBits(spec, drive, two)
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
case a64NLThreeLongMul, a64NLThreeWide:
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
vm, ok1 := vecAt(0)
vn, ok2 := vecAt(1)
vd, ok3 := vecAt(2)
if !ok1 || !ok2 || !ok3 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
var drive string
var pairErr error
if spec.form == a64NLThreeWide {
// UADDW: Vn and Vd spell the wide arrangement, Vm the narrow one;
// the arrangement bits follow the wide side.
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
} else {
// MULL/MLAL/MLSL: Vm and Vn spell the narrow arrangement, Vd the
// wide one; the arrangement bits follow the narrow source.
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vn.arr, vd.arr, two)
}
if pairErr != nil {
return nil, pairErr
}
if spec.form == a64NLThreeWide && vd.arr != vn.arr {
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vn.arr, vd.arr)
}
if spec.form == a64NLThreeLongMul && vm.arr != vn.arr {
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vm.arr, vn.arr)
}
arrBits := arm64SimdNLArrBits(spec, drive, two)
return a64wordLE(spec.base | arrBits | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
case a64NLThreeLongShift, a64NLThreeNarrowShift:
if len(ops) != 3 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects ($shift, Vn.<T>, Vd.<T>)", mnem)
}
sh := arm64Imm64(ops[0])
vn, ok1 := vecAt(1)
vd, ok2 := vecAt(2)
if !ok1 || !ok2 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
if spec.form == a64NLThreeLongShift {
// SSHLL/USHLL: the narrow source drives the immediate's size
// (immh:immb = esize + shift), so the arrangement bits carry
// the Q bit alone: the size field belongs to immh, and ORing
// the source's size bits into it would collide with immb.
if err := arm64SimdLongPair(mnem, vn.arr, vd.arr, two); err != nil {
return nil, err
}
se, _, _ := arm64SimdNLArr(vn.arr)
esize := se * 8
if sh < 0 || sh >= int64(esize) {
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
}
var qBit uint32
if two {
qBit = 1 << 30
}
return a64wordLE(spec.base | uint32(esize+int(sh))<<16 | qBit | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
// SHRN: the narrow destination drives the immediate's size
// (immh:immb = esize - shift over the wide source element), so the
// arrangement bits carry the Q bit alone, exactly as above.
if err := arm64SimdNarrowPair(mnem, vn.arr, vd.arr, two); err != nil {
return nil, err
}
se, _, _ := arm64SimdNLArr(vn.arr)
esize := se * 8
if sh < 1 || sh >= int64(esize) {
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize-1)
}
var qBit uint32
if two {
qBit = 1 << 30
}
return a64wordLE(spec.base | uint32(esize-int(sh))<<16 | qBit | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
}
// encodeARM64DP1 encodes a data-processing (1 source) instruction:
// RBIT, REV, CLZ and friends take (Rn, Rd), word = base | Rn<<5 | Rd.
func encodeARM64DP1(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
@@ -2561,7 +2859,8 @@ func encodeARM64DP1(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, err
// encodeARM64ADR encodes ADR/ADRP: (label, Rd), the byte distance to the
// target split into immlo (bits 30:29) and immhi (bits 23:5), with bit 31
// selecting the page form.
// selecting the page form. An n(PC) operand resolves to the instruction's
// own address: the toolchain rewrites it away and encodes displacement 0.
func encodeARM64ADR(mnem string, page uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
@@ -2570,15 +2869,18 @@ func encodeARM64ADR(mnem string, page uint32, ops []*ast.Operand, pc int, offset
if rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
var rel int64
if _, pcRel := arm64PCRelOffset(ops[0]); !pcRel {
target := resolve(arm64Label(ops[0]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q", target)
}
rel := int64(targetOff - pc)
rel = int64(targetOff - pc)
if rel < -(1<<20) || rel >= 1<<20 {
return nil, fmt.Errorf("%s to %q too far (21-bit range)", mnem, target)
}
}
return a64wordLE(a64ADR(page, int32(rel>>2), uint32(rel)&3, uint32(rd))), nil
}
@@ -2856,11 +3158,16 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
// BRK [$imm16] SVC $imm16
// DMB|DSB|ISB $imm4 DC <op>, Rn
// MRS <sysreg>, Rd MSR $imm4, <sysreg>
// PRFM (Rn), $imm|<op>
// PRFM (Rn), $imm|<op> RPRFM (Rn), Rm, <op|$imm6>
// SYS $imm[, Rn] SYSL $imm, Rd
// TLBI <op>[, Rn] SB, PACIASP, PACIBSP
//
// The system registers, TLBI and DC aliases and the range-prefetch operations
// come from the toolchain's own data tables in arm64_sysregs.go.
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
// Operand-less returns and pointer-authentication hints.
if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0,
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff,
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff, "PACIASP": 0xd503233f, "PACIBSP": 0xd503237f,
"AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f,
"YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f,
"SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok {
@@ -2896,6 +3203,12 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
}
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem]
return a64wordLE(base | uint32(v)<<8), nil
case "SB":
// Speculation barrier: DSB with a fixed barrier domain.
if len(ops) != 0 {
return nil, fmt.Errorf("%s expects no operand", mnem)
}
return a64wordLE(0xd50330ff), nil
case "HINT":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
@@ -2934,7 +3247,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("DC expects <op>, Rn")
}
base, ok := a64DCOps[operandRegName(ops[0])]
inst, ok := a64DCOps2[operandRegName(ops[0])]
if !ok {
return nil, fmt.Errorf("DC: unknown cache operation %q", operandRegName(ops[0]))
}
@@ -2942,51 +3255,115 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
if rn < 0 {
return nil, fmt.Errorf("DC: invalid register operand")
}
return a64wordLE(base | uint32(rn)&31), nil
w := 0xd5080000 | inst.op1<<16 | 7<<12 | inst.cm<<8 | inst.op2<<5
return a64wordLE(w | uint32(rn)&31), nil
case "TLBI":
// The register operand is optional: TLBI VMALLE1IS alone means ZR.
if len(ops) != 1 && len(ops) != 2 {
return nil, fmt.Errorf("TLBI expects <op>[, Rn]")
}
inst, ok := a64TLBIOps[operandRegName(ops[0])]
if !ok {
return nil, fmt.Errorf("TLBI: unknown operation %q", operandRegName(ops[0]))
}
rt := 31
if len(ops) == 2 {
if rt = arm64RegNum(operandRegName(ops[1])); rt < 0 {
return nil, fmt.Errorf("TLBI: invalid register operand")
}
}
w := 0xd5080000 | inst.op1<<16 | 8<<12 | inst.cm<<8 | inst.op2<<5
return a64wordLE(w | uint32(rt)&31), nil
case "SYS", "SYSL":
// SYS $imm[, Rn] / SYSL $imm, Rd: the immediate packs
// op1<<16 | CRn<<12 | CRm<<8 | op2<<5, the register defaults to ZR.
if len(ops) != 1 && len(ops) != 2 {
return nil, fmt.Errorf("%s expects $immediate[, Rn]", mnem)
}
if len(ops) == 1 && mnem == "SYSL" {
return nil, fmt.Errorf("SYSL expects $immediate, Rd")
}
if !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate[, Rn]", mnem)
}
imm := arm64Imm64(ops[0])
if imm < 0 || imm&^0x7FFE0 != 0 {
return nil, fmt.Errorf("%s: illegal SYS argument %d", mnem, imm)
}
rt := 31
if len(ops) == 2 {
if rt = arm64RegNum(operandRegName(ops[1])); rt < 0 {
return nil, fmt.Errorf("%s: invalid register operand", mnem)
}
}
base := uint32(0xd5080000)
if mnem == "SYSL" {
base = 0xd5280000
}
return a64wordLE(base | uint32(imm) | uint32(rt)&31), nil
case "MRS":
if len(ops) != 2 {
return nil, fmt.Errorf("MRS expects <sysreg>, Rd")
}
base, ok := a64MRSOps[operandRegName(ops[0])]
reg, ok := a64SysRegs[operandRegName(ops[0])]
if !ok {
return nil, fmt.Errorf("MRS: unknown system register %q", operandRegName(ops[0]))
}
if !reg.read {
return nil, fmt.Errorf("MRS: system register is not readable: %q", operandRegName(ops[0]))
}
rd := arm64RegNum(operandRegName(ops[1]))
if rd < 0 {
return nil, fmt.Errorf("MRS: invalid register operand")
}
return a64wordLE(base | uint32(rd)&31), nil
return a64wordLE(0xd5300000 | reg.v | uint32(rd)&31), nil
case "MSR":
if len(ops) != 2 {
return nil, fmt.Errorf("MSR expects $immediate, <sysreg> or Rn, <sysreg>")
}
if !isImmOperand(ops[0]) {
// Register form: MSR Rn, <sysreg> (the a64MSRRegOps words).
base, ok := a64MSRRegOps[operandRegName(ops[1])]
if isImmOperand(ops[0]) {
v := arm64Imm64(ops[0])
// The PSTATE fields keep their dedicated immediate form.
if base, ok := a64MSROps[operandRegName(ops[1])]; ok {
if v < 0 || v > 15 {
return nil, fmt.Errorf("MSR: immediate %d out of range (0..15)", v)
}
return a64wordLE(base | uint32(v)<<8 | 31), nil
}
// A $0 against a full system register writes it from ZR, exactly
// the way the toolchain preprocesses the constant away; any other
// immediate is the PSTATE-form error.
if v != 0 {
return nil, fmt.Errorf("MSR: illegal PSTATE field for immediate move: %q", operandRegName(ops[1]))
}
reg, ok := a64SysRegs[operandRegName(ops[1])]
if !ok {
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
}
if !reg.write {
return nil, fmt.Errorf("MSR: system register is not writable: %q", operandRegName(ops[1]))
}
return a64wordLE(0xd5100000 | reg.v | 31), nil
}
// Register form: MSR Rn, <sysreg>.
reg, ok := a64SysRegs[operandRegName(ops[1])]
if !ok {
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
}
if !reg.write {
return nil, fmt.Errorf("MSR: system register is not writable: %q", operandRegName(ops[1]))
}
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, fmt.Errorf("MSR: invalid source register")
}
return a64wordLE(base | uint32(rs)&31), nil
}
base, ok := a64MSROps[operandRegName(ops[1])]
if !ok {
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
}
v := arm64Imm64(ops[0])
if v < 0 || v > 15 {
return nil, fmt.Errorf("MSR: immediate %d out of range (0..15)", v)
}
return a64wordLE(base | uint32(v)<<8 | 31), nil
return a64wordLE(0xd5100000 | reg.v | uint32(rs)&31), nil
case "PRFM":
if len(ops) != 2 {
return nil, fmt.Errorf("PRFM expects (Rn), $immediate|<op>")
}
rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{})
if rn < 0 || off != 0 {
if rn < 0 || off < 0 || off%8 != 0 || off/8 >= 4096 {
return nil, fmt.Errorf("PRFM: invalid memory operand")
}
var prfop int64
@@ -3002,7 +3379,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
}
prfop = int64(p)
}
return a64wordLE(0xf9800000 | uint32(rn)<<5 | uint32(prfop)), nil
return a64wordLE(0xf9800000 | uint32(off/8)<<10 | uint32(rn)<<5 | uint32(prfop)), nil
case "RPRFM":
// RPRFM (Rn), Rm, <op|$imm6>: the 6-bit operation scatters across
// bits 15, 13, 12 and 2:0 (asm7.go case 110).
if len(ops) != 3 {
return nil, fmt.Errorf("RPRFM expects (Rn), Rm, <op|$immediate>")
}
rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{})
if rn < 0 || off != 0 {
return nil, fmt.Errorf("RPRFM: invalid memory operand")
}
rm := arm64RegNum(operandRegName(ops[1]))
if rm < 0 {
return nil, fmt.Errorf("RPRFM: invalid register operand")
}
var op uint64
if isImmOperand(ops[2]) {
op = uint64(arm64Imm64(ops[2]))
if op > 63 {
return nil, fmt.Errorf("RPRFM: range prefetch immediate %d out of range (0..63)", op)
}
} else {
v, ok := a64RPRFOps[operandRegName(ops[2])]
if !ok {
return nil, fmt.Errorf("RPRFM: unknown range prefetch operation %q", operandRegName(ops[2]))
}
op = uint64(v)
}
scatter := (op&(1<<5))<<10 | (op&(1<<4))<<9 | (op&(1<<3))<<9 | op&7
return a64wordLE(0xf8a04818 | uint32(rm)<<16 | uint32(rn)<<5 | uint32(scatter)), nil
}
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
}
@@ -3046,29 +3452,29 @@ func encodeARM64MoveWide(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte
// base, so MOVZ and MOVN come along for free.
opc := baseOp >> 29 & 3
sf := baseOp >> 31 & 1
v := arm64Imm64(ops[0])
if v < 0 {
return nil, fmt.Errorf("%s: negative immediate %d", mnem, v)
// The toolchain's optab case 33, shared by the whole family in both
// widths: the immediate is one unsigned 64-bit pattern (a high-lane
// constant such as $(40000<<48) arrives negative through int64
// folding), it must occupy exactly one 16-bit lane, zero is rejected,
// and the W forms cannot reach the top half.
u := uint64(arm64Imm64(ops[0]))
if u == 0 {
return nil, fmt.Errorf("%s: zero immediate cannot be handled", mnem)
}
hw := -1
for i := range 4 {
if v>>(uint(i)*16)&0xFFFF != 0 {
hw = i
for lane := range 4 {
if u&^(uint64(0xFFFF)<<(lane*16)) == 0 {
hw = lane
break
}
}
if hw < 0 {
hw = 0 // zero: every chunk is zero, hw = 0 carries it
}
for i := hw + 1; i < 4; i++ {
if v>>(uint(i)*16)&0xFFFF != 0 {
return nil, fmt.Errorf("%s: immediate %d does not fit one 16-bit chunk", mnem, v)
}
return nil, fmt.Errorf("%s: immediate %#x does not fit one 16-bit chunk", mnem, u)
}
if sf == 0 && hw > 1 {
return nil, fmt.Errorf("%s: immediate %d out of range for the 32-bit form", mnem, v)
return nil, fmt.Errorf("%s: immediate %#x out of range for the 32-bit form", mnem, u)
}
return a64wordLE(a64MoveWide(sf, opc, uint32(hw), uint32(v>>uint(hw*16)&0xFFFF), uint32(rd))), nil
return a64wordLE(a64MoveWide(sf, opc, uint32(hw), uint32(u>>uint(hw*16)&0xFFFF), uint32(rd))), nil
}
// ---- Bitfield/EXTR encoding ----
@@ -3425,7 +3831,7 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("invalid destination register in VTBL")
}
for i, t := range ts {
if t.hasIdx || t.reg != ts[0].reg+i {
if t.hasIdx || (ts[0].reg+i)&31 != t.reg {
return nil, fmt.Errorf("VTBL table registers must be consecutive")
}
}
@@ -3582,15 +3988,17 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
// encodeARM64VLDST encodes the SIMD structure loads and stores:
//
// VLD1 (Rn), [Vt.arr, ...] VST1 [Vt.arr, ...], (Rn)
// VLD1.P off(Rn), [Vt.arr, ...] VST1.P [Vt.arr, ...], off(Rn)
// VLD1.P off(Rn), Vt.T[i] VST1.P Vt.T[i], off(Rn) (one lane)
// VLD1R (Rn), [Vt.arr] VLD4R (Rn), [Vt.arr, Vt+1, Vt+2, Vt+3]
// VLD1|2|3|4 (Rn), [Vt.arr, ...] VST1|2|3|4 [Vt.arr, ...], (Rn)
// VLD1|2|3|4.P off(Rn), [Vt.arr, ...] VST1|2|3|4.P [Vt.arr, ...], off(Rn)
// VLD1|2|3|4.P (Rn)(Rm), [Vt.arr, ...] VST1|2|3|4.P [Vt.arr, ...], (Rn)(Rm)
// VLD1|2|3|4R (Rn), [Vt.arr, ...] (replicating loads)
// VLD1 off(Rn), Vt.T[i] VST1 Vt.T[i], off(Rn) (one lane)
//
// The post-index forms set the post bit and Rm = 11111. A spelled offset
// rides along (the encoding ignores it; the toolchain only checks that it
// matches the access size), and a multi-register post-index list takes no
// offset at all, the increment following from the list.
// The post-index forms set the post bit and carry Rm: 11111 for an immediate
// increment, the spelled register for (Rn)(Rm). A register list may wrap
// around V31: the toolchain checks only (first+i) mod 32. A spelled offset
// rides along on the one-lane forms (the toolchain only checks that it
// matches the access size).
func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, error) {
load := strings.HasPrefix(mnem, "VLD")
@@ -3628,24 +4036,31 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err
if off != 0 && post == 0 {
return nil, fmt.Errorf("%s: offset %d not supported, plain and list accesses take a plain (Rn) operand", mnem, off)
}
// VLD1R loads one register and replicates; VLD4R loads four.
if strings.HasPrefix(mnem, "VLD1R") || strings.HasPrefix(mnem, "VLD4R") {
want := 1
base := uint32(0x0d40c000)
if strings.HasPrefix(mnem, "VLD4R") {
want, base = 4, 0x0d60e000
// The post-index increment: 11111 for an immediate offset, else the
// spelled (Rn)(Rm) register.
rm := 31
if post != 0 {
if idx := ops[memIdx].Addr.Index; idx != "" {
if rm = arm64RegNum(idx); rm < 0 {
return nil, fmt.Errorf("%s: invalid post-index register %q", mnem, idx)
}
if len(vs) != want {
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, want)
}
}
// The replicating loads: VLD1R through VLD4R load one register and
// replicate it across the whole list.
if base := strings.TrimSuffix(mnem, ".P"); load && strings.HasSuffix(base, "R") && len(base) == 5 {
n := int(base[3] - '0')
if len(vs) != n {
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, n)
}
size, q, ok := a64ArrSizeQ(vs[0].arr)
if !ok {
return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr)
}
w := base | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg)
w := a64VLDNReplicate[n] | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg)
if post != 0 {
w |= 1<<23 | 0x1f<<16
w |= 1<<23 | uint32(rm)<<16
}
return a64wordLE(w), nil
}
@@ -3654,7 +4069,7 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err
return nil, fmt.Errorf("%s expects a list of one to four registers", mnem)
}
for i, v := range vs {
if v.hasIdx || v.reg != vs[0].reg+i {
if v.hasIdx || (vs[0].reg+i)&31 != v.reg {
return nil, fmt.Errorf("%s: register list must be consecutive", mnem)
}
_, _, okArr := a64ArrSizeQ(v.arr)
@@ -3666,21 +4081,36 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err
if !ok {
return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr)
}
base := a64VLD1Base[len(vs)]
n := len(vs)
base := a64VLD1Base[n]
if !load {
base = a64VST1Base[len(vs)]
base = a64VST1Base[n]
}
// VLD2/VLD3/VLD4 and VST2/VST3/VST4 name the register count in the
// mnemonic and carry their own opcode fields. The count digit sits at
// index 3 of the mnemonic (VLD2, VST3.P, ...), before any .P suffix.
if stem := strings.TrimSuffix(mnem, ".P"); len(stem) >= 4 && stem[3] >= '2' && stem[3] <= '4' {
n := int(stem[3] - '0')
if n != len(vs) {
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, n)
}
if load {
base = a64VLDNBase[n]
} else {
base = a64VSTNBase[n]
}
}
postBits := uint32(0)
if post != 0 {
postBits = 0x9f0000
postBits = 1<<23 | uint32(rm)<<16
}
return a64wordLE(base | q<<30 | size<<10 | postBits | uint32(rn)<<5 | uint32(vs[0].reg)), nil
}
// encodeARM64VLDSTLane encodes the one-lane structure forms:
// VLD1 off(Rn), Vt.T[i] (post-index adds the post bit and Rm=11111) and
// VST1.P Vt.T[i], off(Rn); the plain VST1 lane form does not exist in the
// toolchain's table and is rejected.
// VLD1 off(Rn), Vt.T[i] and VST1 Vt.T[i], off(Rn); the post-index spellings
// add the post bit and Rm: 11111 for an immediate increment, the spelled
// register for (Rn)(Rm).
func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load bool, laneIdx int, v a64Vec) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects a memory operand and one Vt.T[i] lane operand", mnem)
@@ -3690,14 +4120,19 @@ func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load boo
if rn < 0 {
return nil, fmt.Errorf("%s: invalid memory operand", mnem)
}
if !load && post == 0 {
return nil, fmt.Errorf("%s: the toolchain only spells a post-index single-lane store", mnem)
rm := 31
if post != 0 {
if idx := ops[memIdx].Addr.Index; idx != "" {
if rm = arm64RegNum(idx); rm < 0 {
return nil, fmt.Errorf("%s: invalid post-index register %q", mnem, idx)
}
}
}
w := uint32(0x0d400000)
switch strings.ToUpper(v.arr) {
case "B":
// Index at bits 12:10 (the size field doubles as the low index bits).
w |= uint32(v.idx) << 10
// Index<3> rides bit 30, index<2:0> the size field at bits 12:10.
w |= uint32(v.idx&7)<<10 | uint32(v.idx>>3&1)<<30
case "H":
// Index<2> at bit 30, index<1> at bit 12, index<0> at bit 11.
w |= 1<<14 | uint32(v.idx&1)<<11 | uint32(v.idx>>1&1)<<12 | uint32(v.idx>>2&1)<<30
@@ -3711,12 +4146,12 @@ func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load boo
return nil, fmt.Errorf("%s: invalid lane arrangement %q", mnem, v.arr)
}
// The base carries bit 22 (L) set; a store clears it. The post-index
// forms add bit 23 and Rm = 11111.
// forms add bit 23 and Rm.
if !load {
w &^= 1 << 22
}
if post != 0 {
w |= 1<<23 | 0x1f<<16
w |= 1<<23 | uint32(rm)<<16
}
return a64wordLE(w | uint32(rn)<<5 | uint32(v.reg)), nil
}
@@ -3980,6 +4415,7 @@ func AssembleFileARM64(f *ast.File) (*Image, error) {
Size: d.size,
Static: d.static,
Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok,
})
}
+144 -12
View File
@@ -33,7 +33,7 @@ import (
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
@@ -374,6 +374,7 @@ const (
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
a64FAcqRel // acquire/release: LDAR family, STLR family
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
a64FCASP // compare and swap pair: CASP
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
@@ -384,6 +385,7 @@ const (
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
a64FVMoviImm // SIMD move immediate: VMOVI $imm8, Vd.B8/B16
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
)
@@ -770,7 +772,7 @@ func init() {
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
// ---- system operations ----
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM", "RPRFM", "SYS", "SYSL", "TLBI", "SB", "PACIASP", "PACIBSP"} {
a64InstrTable[m] = a64Enc{format: a64FSys}
}
@@ -783,12 +785,20 @@ func init() {
a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000}
// ---- load/store pair (signed offset) ----
// The scale column of a64LoadTable does not reach the pair forms, so each
// entry states its own access width through the imm7 divisor the pair
// encoder derives from the opc field (8 for D, 4 for W and SW, 16 for Q).
a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000}
a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000}
a64InstrTable["LDPSW"] = a64Enc{format: a64FPair, op: 0x69400000}
a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000}
a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000}
a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000}
a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000}
a64InstrTable["FLDPS"] = a64Enc{format: a64FPair, op: 0x2d400000}
a64InstrTable["FSTPS"] = a64Enc{format: a64FPair, op: 0x2d000000}
a64InstrTable["FLDPQ"] = a64Enc{format: a64FPair, op: 0xad400000}
a64InstrTable["FSTPQ"] = a64Enc{format: a64FPair, op: 0xad000000}
// ---- acquire/release loads and stores ----
a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00}
@@ -806,11 +816,23 @@ func init() {
lse := map[string]uint32{
"CASALD": 0xc8e0fc00,
"CASALW": 0x88e0fc00,
"CASB": 0x08a07c00,
"CASAB": 0x08e07c00,
"CASH": 0x48a07c00,
"CASLD": 0xc8a0fc00,
"CASLH": 0x48a0fc00,
"CASAW": 0x88e07c00,
"CASAD": 0xc8e07c00,
"CASALH": 0x48e07c00,
"LDADDALD": 0xf8e00000,
"LDADDALW": 0xb8e00000,
"LDADDAD": 0xf8a00000,
"LDADDAW": 0xb8a00000,
"LDCLRALB": 0x38e01000,
"LDCLRALW": 0xb8e01000,
"LDCLRALD": 0xf8e01000,
"LDCLRAD": 0xf8a01000,
"LDCLRAW": 0xb8a01000,
"LDORALB": 0x38e03000,
"LDORALW": 0xb8e03000,
"LDORALD": 0xf8e03000,
@@ -889,6 +911,12 @@ func init() {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// Compare and swap pair: the second register of each pair is implicit
// (Rs+1 and Rt+1), so the encoding carries Rs and Rt alone over a preset
// fixed field (asm7.go atomicCASP).
a64InstrTable["CASPD"] = a64Enc{format: a64FCASP, op: 1<<30 | 0x41<<21 | 0x1f<<10}
a64InstrTable["CASPW"] = a64Enc{format: a64FCASP, op: 0x41<<21 | 0x1f<<10}
// ---- carry-setting/carry-using arithmetic and widening multiply ----
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
// register preset to ZR (bits 14:10 = 11111).
@@ -940,11 +968,13 @@ func init() {
a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000}
a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000}
a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000}
a64InstrTable["VMOVI"] = a64Enc{format: a64FVMoviImm}
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 7<<10}
a64InstrTable["VUSRA"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 5<<10}
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
@@ -957,6 +987,24 @@ func init() {
a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1}
// Multi-register structure accesses beyond VLD1/VST1: VLD2/VLD3/VLD4 and
// the replicate loads VLD2R/VLD3R, each with the post-index spelling.
a64InstrTable["VLD2"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD2.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD3"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD3.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD4"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD2R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD2R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD3R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD3R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST2"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST2.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST3"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST3.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST4"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST4.P"] = a64Enc{format: a64FVLDST, op: 1}
}
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
@@ -1120,6 +1168,78 @@ var a64SimdVTable = map[string]a64SimdVSpec{
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
// Saturating shifts, register forms (the immediate spellings route to
// a64FShiftImm).
"VSQSHL": {0x0e204c00, 0x7f, false},
"VUQSHL": {0x2e204c00, 0x7f, false},
}
// a64SimdNLForm classifies the narrow/long/wide SIMD families whose
// arrangement does not travel on every operand: the encoding's size and Q
// bits read off one designated operand and the element widths pair up across
// the operands.
type a64SimdNLForm uint8
const (
a64NLTwoNarrow a64SimdNLForm = iota // (Vn.wide, Vd.narrow): size/Q from Vd
a64NLTwoLong // (Vn.narrow, Vd.long): size/Q from Vn
a64NLThreeLongMul // (Vm.narrow, Vn.narrow, Vd.long): size/Q from Vn
a64NLThreeWide // (Vm.narrow, Vn.wide, Vd.wide): size/Q from Vn
a64NLThreeLongShift // ($sh, Vn.narrow, Vd.long): size/Q from Vn, immh = esize+sh
a64NLThreeNarrowShift // ($sh, Vn.wide, Vd.narrow): size/Q from Vd, immh = esize-sh
)
// a64SimdNLSpec is one narrow/long/wide instruction: the base word (U, opcode
// and fixed bits positioned) and the arrangement form. qonly marks the FCVT
// family, whose size field is fixed in the base and only the Q bit follows
// the driving arrangement.
type a64SimdNLSpec struct {
base uint32
form a64SimdNLForm
qonly bool
}
// a64SimdNLTable holds the families the arrangement-driven three-register and
// two-register encoders cannot express. The .2 spellings force the 128-bit
// side of the pair through their operand arrangements, so the base carries no
// arrangement bits of its own.
var a64SimdNLTable = map[string]a64SimdNLSpec{
"VSHRN": {0x0f008400, a64NLThreeNarrowShift, false},
"VSHRN2": {0x0f008400, a64NLThreeNarrowShift, false},
"VSXTL": {0x0f00a400, a64NLTwoLong, false},
"VSXTL2": {0x0f00a400, a64NLTwoLong, false},
"VUXTL": {0x2f00a400, a64NLTwoLong, false},
"VUXTL2": {0x2f00a400, a64NLTwoLong, false},
"VXTN": {0x0e202800, a64NLTwoNarrow, false},
"VXTN2": {0x0e202800, a64NLTwoNarrow, false},
"VSQXTN": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTN2": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTUN": {0x2e202800, a64NLTwoNarrow, false},
"VSQXTUN2": {0x2e202800, a64NLTwoNarrow, false},
"VUQXTN": {0x2e204800, a64NLTwoNarrow, false},
"VUQXTN2": {0x2e204800, a64NLTwoNarrow, false},
"VFCVTN": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTN2": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTL": {0x0e217800, a64NLTwoLong, true},
"VFCVTL2": {0x0e217800, a64NLTwoLong, true},
"VSSHLL": {0x0f00a400, a64NLThreeLongShift, false},
"VSSHLL2": {0x0f00a400, a64NLThreeLongShift, false},
"VUSHLL": {0x2f00a400, a64NLThreeLongShift, false},
"VUSHLL2": {0x2f00a400, a64NLThreeLongShift, false},
"VUADDW": {0x2e201000, a64NLThreeWide, false},
"VUADDW2": {0x2e201000, a64NLThreeWide, false},
"VUMULL": {0x2e20c000, a64NLThreeLongMul, false},
"VUMULL2": {0x2e20c000, a64NLThreeLongMul, false},
"VSMULL": {0x0e20c000, a64NLThreeLongMul, false},
"VSMULL2": {0x0e20c000, a64NLThreeLongMul, false},
"VUMLAL": {0x2e208000, a64NLThreeLongMul, false},
"VUMLAL2": {0x2e208000, a64NLThreeLongMul, false},
"VSMLAL": {0x0e208000, a64NLThreeLongMul, false},
"VSMLAL2": {0x0e208000, a64NLThreeLongMul, false},
"VUMLSL": {0x2e20a000, a64NLThreeLongMul, false},
"VUMLSL2": {0x2e20a000, a64NLThreeLongMul, false},
"VSMLSL": {0x0e20a000, a64NLThreeLongMul, false},
"VSMLSL2": {0x0e20a000, a64NLThreeLongMul, false},
}
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
@@ -1243,6 +1363,16 @@ var a64PRFOps = map[string]int{
var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000}
var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000}
// a64VLDNBase and a64VSTNBase hold the VLD2/VLD3/VLD4 and VST2/VST3/VST4
// fixed words (indexed by register count 2..4): the opcode field at bits
// 15:12 carries the access kind.
var a64VLDNBase = [5]uint32{0, 0, 0x0c408000, 0x0c404000, 0x0c400000}
var a64VSTNBase = [5]uint32{0, 0, 0x0c008000, 0x0c004000, 0x0c000000}
// a64VLDNReplicate holds the VLD2R/VLD3R fixed words beside the existing
// VLD1R (0x0d40c000) and VLD4R (0x0d60e000) bases.
var a64VLDNReplicate = [5]uint32{0, 0x0d40c000, 0x0d60c000, 0x0d40e000, 0x0d60e000}
// a64Vec is a parsed vector operand: the register number, the arrangement
// ("" when the operand spells none) and, for element forms, the lane index.
type a64Vec struct {
@@ -1366,20 +1496,22 @@ type a64LSType struct {
size int // 0=byte, 1=half, 2=word, 3=dword
V int // 0=integer, 1=FP
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
scale int // access width in bytes; the unsigned offset divides by it
}
// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters.
// For loads, opc selects signed vs unsigned; for stores, we flip the opc.
var a64LoadTable = map[string]a64LSType{
"MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
"MOVD": {3, 0, 1, 8}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1, 4}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2, 4}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1, 2}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2, 1}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1, 4}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1, 8}, // LDR D (64-bit FP)
"FMOVQ": {0, 1, 3, 16}, // LDR/STR Q (128-bit FP): opc=11 selects it
}
// a64StoreOpc returns the store opc for a given load type: integer and FP
+37 -2
View File
@@ -7,8 +7,8 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
func TestArm64LDRSTREncoding(t *testing.T) {
@@ -1051,6 +1051,41 @@ func TestArm64MOVK(t *testing.T) {
}
}
// TestArm64MOVKHighLane pins the shifted high-lane immediate the arm64 test
// kernels write: $(40000<<48) folds to a negative int64, and the toolchain
// reads the value as an unsigned 64-bit pattern when it picks the lane.
func TestArm64MOVKHighLane(t *testing.T) {
got := arm64Words(t, "\tMOVK $(40000<<48), R0\n\tMOVK $0x9c40000000000000, R1\n")
want := []uint32{
0xf2f38800, // MOVK $(40000<<48), R0 (go tool asm: f2f38800)
0xf2f38801, // MOVK hw=3
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64MoveWideZeroImmediate pins the toolchain's rejection of a zero
// immediate in the move-wide family (optab case 33: "zero shifts cannot be
// handled"): every lane is zero, so no hw field can carry it.
func TestArm64MoveWideZeroImmediate(t *testing.T) {
for _, mnem := range []string{"MOVK", "MOVZ", "MOVN"} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\t"+mnem+" $0, R0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", mnem, errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s $0: expected error, got nil", mnem)
}
}
}
// TestArm64LoadImm64 tests 64-bit immediate loading.
func TestArm64LoadImm64(t *testing.T) {
src := `#include "textflag.h"
+1 -1
View File
@@ -53,7 +53,7 @@ package asm
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
+1 -1
View File
@@ -7,7 +7,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// parseArm64File is a helper assembling one arm64 source file.
+588
View File
@@ -0,0 +1,588 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// arm64 system registers and system-instruction aliases.
//
// The tables are transcribed from the data the Go toolchain itself carries
// (cmd/internal/obj/arm64/sysRegEnc.go and the sysInstFields map of asm7.go),
// which the ARM ARM defines: every system register is the packed field set
// op0<<19 | op1<<16 | CRn<<12 | CRm<<8 | op2<<5, and the read/write flags are
// the toolchain's own access classification. The encoding tables live here so
// the encoder stays testable against the GOROOT testdata word for word.
// a64SysReg is one system register: the packed encoding fields and the
// directions the register supports.
type a64SysReg struct {
v uint32
read bool
write bool
}
// a64SysRegs maps the system register names the toolchain knows to their
// encodings. MRS reads 0xd5300000 | v | Rd and MSR writes
// 0xd5100000 | v | Rt.
var a64SysRegs = map[string]a64SysReg{
"ACTLR_EL1": a64SysReg{0x181020, true, true},
"AFSR0_EL1": a64SysReg{0x185100, true, true},
"AFSR1_EL1": a64SysReg{0x185120, true, true},
"AIDR_EL1": a64SysReg{0x1900e0, true, false},
"AMAIR_EL1": a64SysReg{0x18a300, true, true},
"AMCFGR_EL0": a64SysReg{0x1bd220, true, false},
"AMCGCR_EL0": a64SysReg{0x1bd240, true, false},
"AMCNTENCLR0_EL0": a64SysReg{0x1bd280, true, true},
"AMCNTENCLR1_EL0": a64SysReg{0x1bd300, true, true},
"AMCNTENSET0_EL0": a64SysReg{0x1bd2a0, true, true},
"AMCNTENSET1_EL0": a64SysReg{0x1bd320, true, true},
"AMCR_EL0": a64SysReg{0x1bd200, true, true},
"AMEVCNTR00_EL0": a64SysReg{0x1bd400, true, true},
"AMEVCNTR01_EL0": a64SysReg{0x1bd420, true, true},
"AMEVCNTR02_EL0": a64SysReg{0x1bd440, true, true},
"AMEVCNTR03_EL0": a64SysReg{0x1bd460, true, true},
"AMEVCNTR04_EL0": a64SysReg{0x1bd480, true, true},
"AMEVCNTR05_EL0": a64SysReg{0x1bd4a0, true, true},
"AMEVCNTR06_EL0": a64SysReg{0x1bd4c0, true, true},
"AMEVCNTR07_EL0": a64SysReg{0x1bd4e0, true, true},
"AMEVCNTR08_EL0": a64SysReg{0x1bd500, true, true},
"AMEVCNTR09_EL0": a64SysReg{0x1bd520, true, true},
"AMEVCNTR010_EL0": a64SysReg{0x1bd540, true, true},
"AMEVCNTR011_EL0": a64SysReg{0x1bd560, true, true},
"AMEVCNTR012_EL0": a64SysReg{0x1bd580, true, true},
"AMEVCNTR013_EL0": a64SysReg{0x1bd5a0, true, true},
"AMEVCNTR014_EL0": a64SysReg{0x1bd5c0, true, true},
"AMEVCNTR015_EL0": a64SysReg{0x1bd5e0, true, true},
"AMEVCNTR10_EL0": a64SysReg{0x1bdc00, true, true},
"AMEVCNTR11_EL0": a64SysReg{0x1bdc20, true, true},
"AMEVCNTR12_EL0": a64SysReg{0x1bdc40, true, true},
"AMEVCNTR13_EL0": a64SysReg{0x1bdc60, true, true},
"AMEVCNTR14_EL0": a64SysReg{0x1bdc80, true, true},
"AMEVCNTR15_EL0": a64SysReg{0x1bdca0, true, true},
"AMEVCNTR16_EL0": a64SysReg{0x1bdcc0, true, true},
"AMEVCNTR17_EL0": a64SysReg{0x1bdce0, true, true},
"AMEVCNTR18_EL0": a64SysReg{0x1bdd00, true, true},
"AMEVCNTR19_EL0": a64SysReg{0x1bdd20, true, true},
"AMEVCNTR110_EL0": a64SysReg{0x1bdd40, true, true},
"AMEVCNTR111_EL0": a64SysReg{0x1bdd60, true, true},
"AMEVCNTR112_EL0": a64SysReg{0x1bdd80, true, true},
"AMEVCNTR113_EL0": a64SysReg{0x1bdda0, true, true},
"AMEVCNTR114_EL0": a64SysReg{0x1bddc0, true, true},
"AMEVCNTR115_EL0": a64SysReg{0x1bdde0, true, true},
"AMEVTYPER00_EL0": a64SysReg{0x1bd600, true, false},
"AMEVTYPER01_EL0": a64SysReg{0x1bd620, true, false},
"AMEVTYPER02_EL0": a64SysReg{0x1bd640, true, false},
"AMEVTYPER03_EL0": a64SysReg{0x1bd660, true, false},
"AMEVTYPER04_EL0": a64SysReg{0x1bd680, true, false},
"AMEVTYPER05_EL0": a64SysReg{0x1bd6a0, true, false},
"AMEVTYPER06_EL0": a64SysReg{0x1bd6c0, true, false},
"AMEVTYPER07_EL0": a64SysReg{0x1bd6e0, true, false},
"AMEVTYPER08_EL0": a64SysReg{0x1bd700, true, false},
"AMEVTYPER09_EL0": a64SysReg{0x1bd720, true, false},
"AMEVTYPER010_EL0": a64SysReg{0x1bd740, true, false},
"AMEVTYPER011_EL0": a64SysReg{0x1bd760, true, false},
"AMEVTYPER012_EL0": a64SysReg{0x1bd780, true, false},
"AMEVTYPER013_EL0": a64SysReg{0x1bd7a0, true, false},
"AMEVTYPER014_EL0": a64SysReg{0x1bd7c0, true, false},
"AMEVTYPER015_EL0": a64SysReg{0x1bd7e0, true, false},
"AMEVTYPER10_EL0": a64SysReg{0x1bde00, true, true},
"AMEVTYPER11_EL0": a64SysReg{0x1bde20, true, true},
"AMEVTYPER12_EL0": a64SysReg{0x1bde40, true, true},
"AMEVTYPER13_EL0": a64SysReg{0x1bde60, true, true},
"AMEVTYPER14_EL0": a64SysReg{0x1bde80, true, true},
"AMEVTYPER15_EL0": a64SysReg{0x1bdea0, true, true},
"AMEVTYPER16_EL0": a64SysReg{0x1bdec0, true, true},
"AMEVTYPER17_EL0": a64SysReg{0x1bdee0, true, true},
"AMEVTYPER18_EL0": a64SysReg{0x1bdf00, true, true},
"AMEVTYPER19_EL0": a64SysReg{0x1bdf20, true, true},
"AMEVTYPER110_EL0": a64SysReg{0x1bdf40, true, true},
"AMEVTYPER111_EL0": a64SysReg{0x1bdf60, true, true},
"AMEVTYPER112_EL0": a64SysReg{0x1bdf80, true, true},
"AMEVTYPER113_EL0": a64SysReg{0x1bdfa0, true, true},
"AMEVTYPER114_EL0": a64SysReg{0x1bdfc0, true, true},
"AMEVTYPER115_EL0": a64SysReg{0x1bdfe0, true, true},
"AMUSERENR_EL0": a64SysReg{0x1bd260, true, true},
"APDAKeyHi_EL1": a64SysReg{0x182220, true, true},
"APDAKeyLo_EL1": a64SysReg{0x182200, true, true},
"APDBKeyHi_EL1": a64SysReg{0x182260, true, true},
"APDBKeyLo_EL1": a64SysReg{0x182240, true, true},
"APGAKeyHi_EL1": a64SysReg{0x182320, true, true},
"APGAKeyLo_EL1": a64SysReg{0x182300, true, true},
"APIAKeyHi_EL1": a64SysReg{0x182120, true, true},
"APIAKeyLo_EL1": a64SysReg{0x182100, true, true},
"APIBKeyHi_EL1": a64SysReg{0x182160, true, true},
"APIBKeyLo_EL1": a64SysReg{0x182140, true, true},
"CCSIDR2_EL1": a64SysReg{0x190040, true, false},
"CCSIDR_EL1": a64SysReg{0x190000, true, false},
"CLIDR_EL1": a64SysReg{0x190020, true, false},
"CNTFRQ_EL0": a64SysReg{0x1be000, true, true},
"CNTKCTL_EL1": a64SysReg{0x18e100, true, true},
"CNTP_CTL_EL0": a64SysReg{0x1be220, true, true},
"CNTP_CVAL_EL0": a64SysReg{0x1be240, true, true},
"CNTP_TVAL_EL0": a64SysReg{0x1be200, true, true},
"CNTPCT_EL0": a64SysReg{0x1be020, true, false},
"CNTPS_CTL_EL1": a64SysReg{0x1fe220, true, true},
"CNTPS_CVAL_EL1": a64SysReg{0x1fe240, true, true},
"CNTPS_TVAL_EL1": a64SysReg{0x1fe200, true, true},
"CNTV_CTL_EL0": a64SysReg{0x1be320, true, true},
"CNTV_CVAL_EL0": a64SysReg{0x1be340, true, true},
"CNTV_TVAL_EL0": a64SysReg{0x1be300, true, true},
"CNTVCT_EL0": a64SysReg{0x1be040, true, false},
"CONTEXTIDR_EL1": a64SysReg{0x18d020, true, true},
"CPACR_EL1": a64SysReg{0x181040, true, true},
"CSSELR_EL1": a64SysReg{0x1a0000, true, true},
"CTR_EL0": a64SysReg{0x1b0020, true, false},
"CurrentEL": a64SysReg{0x184240, true, false},
"DAIF": a64SysReg{0x1b4220, true, true},
"DBGAUTHSTATUS_EL1": a64SysReg{0x107ec0, true, false},
"DBGBCR0_EL1": a64SysReg{0x1000a0, true, true},
"DBGBCR1_EL1": a64SysReg{0x1001a0, true, true},
"DBGBCR2_EL1": a64SysReg{0x1002a0, true, true},
"DBGBCR3_EL1": a64SysReg{0x1003a0, true, true},
"DBGBCR4_EL1": a64SysReg{0x1004a0, true, true},
"DBGBCR5_EL1": a64SysReg{0x1005a0, true, true},
"DBGBCR6_EL1": a64SysReg{0x1006a0, true, true},
"DBGBCR7_EL1": a64SysReg{0x1007a0, true, true},
"DBGBCR8_EL1": a64SysReg{0x1008a0, true, true},
"DBGBCR9_EL1": a64SysReg{0x1009a0, true, true},
"DBGBCR10_EL1": a64SysReg{0x100aa0, true, true},
"DBGBCR11_EL1": a64SysReg{0x100ba0, true, true},
"DBGBCR12_EL1": a64SysReg{0x100ca0, true, true},
"DBGBCR13_EL1": a64SysReg{0x100da0, true, true},
"DBGBCR14_EL1": a64SysReg{0x100ea0, true, true},
"DBGBCR15_EL1": a64SysReg{0x100fa0, true, true},
"DBGBVR0_EL1": a64SysReg{0x100080, true, true},
"DBGBVR1_EL1": a64SysReg{0x100180, true, true},
"DBGBVR2_EL1": a64SysReg{0x100280, true, true},
"DBGBVR3_EL1": a64SysReg{0x100380, true, true},
"DBGBVR4_EL1": a64SysReg{0x100480, true, true},
"DBGBVR5_EL1": a64SysReg{0x100580, true, true},
"DBGBVR6_EL1": a64SysReg{0x100680, true, true},
"DBGBVR7_EL1": a64SysReg{0x100780, true, true},
"DBGBVR8_EL1": a64SysReg{0x100880, true, true},
"DBGBVR9_EL1": a64SysReg{0x100980, true, true},
"DBGBVR10_EL1": a64SysReg{0x100a80, true, true},
"DBGBVR11_EL1": a64SysReg{0x100b80, true, true},
"DBGBVR12_EL1": a64SysReg{0x100c80, true, true},
"DBGBVR13_EL1": a64SysReg{0x100d80, true, true},
"DBGBVR14_EL1": a64SysReg{0x100e80, true, true},
"DBGBVR15_EL1": a64SysReg{0x100f80, true, true},
"DBGCLAIMCLR_EL1": a64SysReg{0x1079c0, true, true},
"DBGCLAIMSET_EL1": a64SysReg{0x1078c0, true, true},
"DBGDTR_EL0": a64SysReg{0x130400, true, true},
"DBGDTRRX_EL0": a64SysReg{0x130500, true, false},
"DBGDTRTX_EL0": a64SysReg{0x130500, false, true},
"DBGPRCR_EL1": a64SysReg{0x101480, true, true},
"DBGWCR0_EL1": a64SysReg{0x1000e0, true, true},
"DBGWCR1_EL1": a64SysReg{0x1001e0, true, true},
"DBGWCR2_EL1": a64SysReg{0x1002e0, true, true},
"DBGWCR3_EL1": a64SysReg{0x1003e0, true, true},
"DBGWCR4_EL1": a64SysReg{0x1004e0, true, true},
"DBGWCR5_EL1": a64SysReg{0x1005e0, true, true},
"DBGWCR6_EL1": a64SysReg{0x1006e0, true, true},
"DBGWCR7_EL1": a64SysReg{0x1007e0, true, true},
"DBGWCR8_EL1": a64SysReg{0x1008e0, true, true},
"DBGWCR9_EL1": a64SysReg{0x1009e0, true, true},
"DBGWCR10_EL1": a64SysReg{0x100ae0, true, true},
"DBGWCR11_EL1": a64SysReg{0x100be0, true, true},
"DBGWCR12_EL1": a64SysReg{0x100ce0, true, true},
"DBGWCR13_EL1": a64SysReg{0x100de0, true, true},
"DBGWCR14_EL1": a64SysReg{0x100ee0, true, true},
"DBGWCR15_EL1": a64SysReg{0x100fe0, true, true},
"DBGWVR0_EL1": a64SysReg{0x1000c0, true, true},
"DBGWVR1_EL1": a64SysReg{0x1001c0, true, true},
"DBGWVR2_EL1": a64SysReg{0x1002c0, true, true},
"DBGWVR3_EL1": a64SysReg{0x1003c0, true, true},
"DBGWVR4_EL1": a64SysReg{0x1004c0, true, true},
"DBGWVR5_EL1": a64SysReg{0x1005c0, true, true},
"DBGWVR6_EL1": a64SysReg{0x1006c0, true, true},
"DBGWVR7_EL1": a64SysReg{0x1007c0, true, true},
"DBGWVR8_EL1": a64SysReg{0x1008c0, true, true},
"DBGWVR9_EL1": a64SysReg{0x1009c0, true, true},
"DBGWVR10_EL1": a64SysReg{0x100ac0, true, true},
"DBGWVR11_EL1": a64SysReg{0x100bc0, true, true},
"DBGWVR12_EL1": a64SysReg{0x100cc0, true, true},
"DBGWVR13_EL1": a64SysReg{0x100dc0, true, true},
"DBGWVR14_EL1": a64SysReg{0x100ec0, true, true},
"DBGWVR15_EL1": a64SysReg{0x100fc0, true, true},
"DCZID_EL0": a64SysReg{0x1b00e0, true, false},
"DISR_EL1": a64SysReg{0x18c120, true, true},
"DIT": a64SysReg{0x1b42a0, true, true},
"DLR_EL0": a64SysReg{0x1b4520, true, true},
"DSPSR_EL0": a64SysReg{0x1b4500, true, true},
"ELR_EL1": a64SysReg{0x184020, true, true},
"ERRIDR_EL1": a64SysReg{0x185300, true, false},
"ERRSELR_EL1": a64SysReg{0x185320, true, true},
"ERXADDR_EL1": a64SysReg{0x185460, true, true},
"ERXCTLR_EL1": a64SysReg{0x185420, true, true},
"ERXFR_EL1": a64SysReg{0x185400, true, false},
"ERXMISC0_EL1": a64SysReg{0x185500, true, true},
"ERXMISC1_EL1": a64SysReg{0x185520, true, true},
"ERXMISC2_EL1": a64SysReg{0x185540, true, true},
"ERXMISC3_EL1": a64SysReg{0x185560, true, true},
"ERXPFGCDN_EL1": a64SysReg{0x1854c0, true, true},
"ERXPFGCTL_EL1": a64SysReg{0x1854a0, true, true},
"ERXPFGF_EL1": a64SysReg{0x185480, true, false},
"ERXSTATUS_EL1": a64SysReg{0x185440, true, true},
"ESR_EL1": a64SysReg{0x185200, true, true},
"FAR_EL1": a64SysReg{0x186000, true, true},
"FPCR": a64SysReg{0x1b4400, true, true},
"FPSR": a64SysReg{0x1b4420, true, true},
"GCR_EL1": a64SysReg{0x1810c0, true, true},
"GMID_EL1": a64SysReg{0x31400, true, false},
"ICC_AP0R0_EL1": a64SysReg{0x18c880, true, true},
"ICC_AP0R1_EL1": a64SysReg{0x18c8a0, true, true},
"ICC_AP0R2_EL1": a64SysReg{0x18c8c0, true, true},
"ICC_AP0R3_EL1": a64SysReg{0x18c8e0, true, true},
"ICC_AP1R0_EL1": a64SysReg{0x18c900, true, true},
"ICC_AP1R1_EL1": a64SysReg{0x18c920, true, true},
"ICC_AP1R2_EL1": a64SysReg{0x18c940, true, true},
"ICC_AP1R3_EL1": a64SysReg{0x18c960, true, true},
"ICC_ASGI1R_EL1": a64SysReg{0x18cbc0, false, true},
"ICC_BPR0_EL1": a64SysReg{0x18c860, true, true},
"ICC_BPR1_EL1": a64SysReg{0x18cc60, true, true},
"ICC_CTLR_EL1": a64SysReg{0x18cc80, true, true},
"ICC_DIR_EL1": a64SysReg{0x18cb20, false, true},
"ICC_EOIR0_EL1": a64SysReg{0x18c820, false, true},
"ICC_EOIR1_EL1": a64SysReg{0x18cc20, false, true},
"ICC_HPPIR0_EL1": a64SysReg{0x18c840, true, false},
"ICC_HPPIR1_EL1": a64SysReg{0x18cc40, true, false},
"ICC_IAR0_EL1": a64SysReg{0x18c800, true, false},
"ICC_IAR1_EL1": a64SysReg{0x18cc00, true, false},
"ICC_IGRPEN0_EL1": a64SysReg{0x18ccc0, true, true},
"ICC_IGRPEN1_EL1": a64SysReg{0x18cce0, true, true},
"ICC_PMR_EL1": a64SysReg{0x184600, true, true},
"ICC_RPR_EL1": a64SysReg{0x18cb60, true, false},
"ICC_SGI0R_EL1": a64SysReg{0x18cbe0, false, true},
"ICC_SGI1R_EL1": a64SysReg{0x18cba0, false, true},
"ICC_SRE_EL1": a64SysReg{0x18cca0, true, true},
"ICV_AP0R0_EL1": a64SysReg{0x18c880, true, true},
"ICV_AP0R1_EL1": a64SysReg{0x18c8a0, true, true},
"ICV_AP0R2_EL1": a64SysReg{0x18c8c0, true, true},
"ICV_AP0R3_EL1": a64SysReg{0x18c8e0, true, true},
"ICV_AP1R0_EL1": a64SysReg{0x18c900, true, true},
"ICV_AP1R1_EL1": a64SysReg{0x18c920, true, true},
"ICV_AP1R2_EL1": a64SysReg{0x18c940, true, true},
"ICV_AP1R3_EL1": a64SysReg{0x18c960, true, true},
"ICV_BPR0_EL1": a64SysReg{0x18c860, true, true},
"ICV_BPR1_EL1": a64SysReg{0x18cc60, true, true},
"ICV_CTLR_EL1": a64SysReg{0x18cc80, true, true},
"ICV_DIR_EL1": a64SysReg{0x18cb20, false, true},
"ICV_EOIR0_EL1": a64SysReg{0x18c820, false, true},
"ICV_EOIR1_EL1": a64SysReg{0x18cc20, false, true},
"ICV_HPPIR0_EL1": a64SysReg{0x18c840, true, false},
"ICV_HPPIR1_EL1": a64SysReg{0x18cc40, true, false},
"ICV_IAR0_EL1": a64SysReg{0x18c800, true, false},
"ICV_IAR1_EL1": a64SysReg{0x18cc00, true, false},
"ICV_IGRPEN0_EL1": a64SysReg{0x18ccc0, true, true},
"ICV_IGRPEN1_EL1": a64SysReg{0x18cce0, true, true},
"ICV_PMR_EL1": a64SysReg{0x184600, true, true},
"ICV_RPR_EL1": a64SysReg{0x18cb60, true, false},
"ID_AA64AFR0_EL1": a64SysReg{0x180580, true, false},
"ID_AA64AFR1_EL1": a64SysReg{0x1805a0, true, false},
"ID_AA64DFR0_EL1": a64SysReg{0x180500, true, false},
"ID_AA64DFR1_EL1": a64SysReg{0x180520, true, false},
"ID_AA64ISAR0_EL1": a64SysReg{0x180600, true, false},
"ID_AA64ISAR1_EL1": a64SysReg{0x180620, true, false},
"ID_AA64MMFR0_EL1": a64SysReg{0x180700, true, false},
"ID_AA64MMFR1_EL1": a64SysReg{0x180720, true, false},
"ID_AA64MMFR2_EL1": a64SysReg{0x180740, true, false},
"ID_AA64PFR0_EL1": a64SysReg{0x180400, true, false},
"ID_AA64PFR1_EL1": a64SysReg{0x180420, true, false},
"ID_AA64ZFR0_EL1": a64SysReg{0x180480, true, false},
"ID_AFR0_EL1": a64SysReg{0x180160, true, false},
"ID_DFR0_EL1": a64SysReg{0x180140, true, false},
"ID_ISAR0_EL1": a64SysReg{0x180200, true, false},
"ID_ISAR1_EL1": a64SysReg{0x180220, true, false},
"ID_ISAR2_EL1": a64SysReg{0x180240, true, false},
"ID_ISAR3_EL1": a64SysReg{0x180260, true, false},
"ID_ISAR4_EL1": a64SysReg{0x180280, true, false},
"ID_ISAR5_EL1": a64SysReg{0x1802a0, true, false},
"ID_ISAR6_EL1": a64SysReg{0x1802e0, true, false},
"ID_MMFR0_EL1": a64SysReg{0x180180, true, false},
"ID_MMFR1_EL1": a64SysReg{0x1801a0, true, false},
"ID_MMFR2_EL1": a64SysReg{0x1801c0, true, false},
"ID_MMFR3_EL1": a64SysReg{0x1801e0, true, false},
"ID_MMFR4_EL1": a64SysReg{0x1802c0, true, false},
"ID_PFR0_EL1": a64SysReg{0x180100, true, false},
"ID_PFR1_EL1": a64SysReg{0x180120, true, false},
"ID_PFR2_EL1": a64SysReg{0x180380, true, false},
"ISR_EL1": a64SysReg{0x18c100, true, false},
"LORC_EL1": a64SysReg{0x18a460, true, true},
"LOREA_EL1": a64SysReg{0x18a420, true, true},
"LORID_EL1": a64SysReg{0x18a4e0, true, false},
"LORN_EL1": a64SysReg{0x18a440, true, true},
"LORSA_EL1": a64SysReg{0x18a400, true, true},
"MAIR_EL1": a64SysReg{0x18a200, true, true},
"MDCCINT_EL1": a64SysReg{0x100200, true, true},
"MDCCSR_EL0": a64SysReg{0x130100, true, false},
"MDRAR_EL1": a64SysReg{0x101000, true, false},
"MDSCR_EL1": a64SysReg{0x100240, true, true},
"MIDR_EL1": a64SysReg{0x180000, true, false},
"MPAM0_EL1": a64SysReg{0x18a520, true, true},
"MPAM1_EL1": a64SysReg{0x18a500, true, true},
"MPAMIDR_EL1": a64SysReg{0x18a480, true, false},
"MPIDR_EL1": a64SysReg{0x1800a0, true, false},
"MVFR0_EL1": a64SysReg{0x180300, true, false},
"MVFR1_EL1": a64SysReg{0x180320, true, false},
"MVFR2_EL1": a64SysReg{0x180340, true, false},
"NZCV": a64SysReg{0x1b4200, true, true},
"OSDLR_EL1": a64SysReg{0x101380, true, true},
"OSDTRRX_EL1": a64SysReg{0x100040, true, true},
"OSDTRTX_EL1": a64SysReg{0x100340, true, true},
"OSECCR_EL1": a64SysReg{0x100640, true, true},
"OSLAR_EL1": a64SysReg{0x101080, false, true},
"OSLSR_EL1": a64SysReg{0x101180, true, false},
"PAN": a64SysReg{0x184260, true, true},
"PAR_EL1": a64SysReg{0x187400, true, true},
"PMBIDR_EL1": a64SysReg{0x189ae0, true, false},
"PMBLIMITR_EL1": a64SysReg{0x189a00, true, true},
"PMBPTR_EL1": a64SysReg{0x189a20, true, true},
"PMBSR_EL1": a64SysReg{0x189a60, true, true},
"PMCCFILTR_EL0": a64SysReg{0x1befe0, true, true},
"PMCCNTR_EL0": a64SysReg{0x1b9d00, true, true},
"PMCEID0_EL0": a64SysReg{0x1b9cc0, true, false},
"PMCEID1_EL0": a64SysReg{0x1b9ce0, true, false},
"PMCNTENCLR_EL0": a64SysReg{0x1b9c40, true, true},
"PMCNTENSET_EL0": a64SysReg{0x1b9c20, true, true},
"PMCR_EL0": a64SysReg{0x1b9c00, true, true},
"PMEVCNTR0_EL0": a64SysReg{0x1be800, true, true},
"PMEVCNTR1_EL0": a64SysReg{0x1be820, true, true},
"PMEVCNTR2_EL0": a64SysReg{0x1be840, true, true},
"PMEVCNTR3_EL0": a64SysReg{0x1be860, true, true},
"PMEVCNTR4_EL0": a64SysReg{0x1be880, true, true},
"PMEVCNTR5_EL0": a64SysReg{0x1be8a0, true, true},
"PMEVCNTR6_EL0": a64SysReg{0x1be8c0, true, true},
"PMEVCNTR7_EL0": a64SysReg{0x1be8e0, true, true},
"PMEVCNTR8_EL0": a64SysReg{0x1be900, true, true},
"PMEVCNTR9_EL0": a64SysReg{0x1be920, true, true},
"PMEVCNTR10_EL0": a64SysReg{0x1be940, true, true},
"PMEVCNTR11_EL0": a64SysReg{0x1be960, true, true},
"PMEVCNTR12_EL0": a64SysReg{0x1be980, true, true},
"PMEVCNTR13_EL0": a64SysReg{0x1be9a0, true, true},
"PMEVCNTR14_EL0": a64SysReg{0x1be9c0, true, true},
"PMEVCNTR15_EL0": a64SysReg{0x1be9e0, true, true},
"PMEVCNTR16_EL0": a64SysReg{0x1bea00, true, true},
"PMEVCNTR17_EL0": a64SysReg{0x1bea20, true, true},
"PMEVCNTR18_EL0": a64SysReg{0x1bea40, true, true},
"PMEVCNTR19_EL0": a64SysReg{0x1bea60, true, true},
"PMEVCNTR20_EL0": a64SysReg{0x1bea80, true, true},
"PMEVCNTR21_EL0": a64SysReg{0x1beaa0, true, true},
"PMEVCNTR22_EL0": a64SysReg{0x1beac0, true, true},
"PMEVCNTR23_EL0": a64SysReg{0x1beae0, true, true},
"PMEVCNTR24_EL0": a64SysReg{0x1beb00, true, true},
"PMEVCNTR25_EL0": a64SysReg{0x1beb20, true, true},
"PMEVCNTR26_EL0": a64SysReg{0x1beb40, true, true},
"PMEVCNTR27_EL0": a64SysReg{0x1beb60, true, true},
"PMEVCNTR28_EL0": a64SysReg{0x1beb80, true, true},
"PMEVCNTR29_EL0": a64SysReg{0x1beba0, true, true},
"PMEVCNTR30_EL0": a64SysReg{0x1bebc0, true, true},
"PMEVTYPER0_EL0": a64SysReg{0x1bec00, true, true},
"PMEVTYPER1_EL0": a64SysReg{0x1bec20, true, true},
"PMEVTYPER2_EL0": a64SysReg{0x1bec40, true, true},
"PMEVTYPER3_EL0": a64SysReg{0x1bec60, true, true},
"PMEVTYPER4_EL0": a64SysReg{0x1bec80, true, true},
"PMEVTYPER5_EL0": a64SysReg{0x1beca0, true, true},
"PMEVTYPER6_EL0": a64SysReg{0x1becc0, true, true},
"PMEVTYPER7_EL0": a64SysReg{0x1bece0, true, true},
"PMEVTYPER8_EL0": a64SysReg{0x1bed00, true, true},
"PMEVTYPER9_EL0": a64SysReg{0x1bed20, true, true},
"PMEVTYPER10_EL0": a64SysReg{0x1bed40, true, true},
"PMEVTYPER11_EL0": a64SysReg{0x1bed60, true, true},
"PMEVTYPER12_EL0": a64SysReg{0x1bed80, true, true},
"PMEVTYPER13_EL0": a64SysReg{0x1beda0, true, true},
"PMEVTYPER14_EL0": a64SysReg{0x1bedc0, true, true},
"PMEVTYPER15_EL0": a64SysReg{0x1bede0, true, true},
"PMEVTYPER16_EL0": a64SysReg{0x1bee00, true, true},
"PMEVTYPER17_EL0": a64SysReg{0x1bee20, true, true},
"PMEVTYPER18_EL0": a64SysReg{0x1bee40, true, true},
"PMEVTYPER19_EL0": a64SysReg{0x1bee60, true, true},
"PMEVTYPER20_EL0": a64SysReg{0x1bee80, true, true},
"PMEVTYPER21_EL0": a64SysReg{0x1beea0, true, true},
"PMEVTYPER22_EL0": a64SysReg{0x1beec0, true, true},
"PMEVTYPER23_EL0": a64SysReg{0x1beee0, true, true},
"PMEVTYPER24_EL0": a64SysReg{0x1bef00, true, true},
"PMEVTYPER25_EL0": a64SysReg{0x1bef20, true, true},
"PMEVTYPER26_EL0": a64SysReg{0x1bef40, true, true},
"PMEVTYPER27_EL0": a64SysReg{0x1bef60, true, true},
"PMEVTYPER28_EL0": a64SysReg{0x1bef80, true, true},
"PMEVTYPER29_EL0": a64SysReg{0x1befa0, true, true},
"PMEVTYPER30_EL0": a64SysReg{0x1befc0, true, true},
"PMINTENCLR_EL1": a64SysReg{0x189e40, true, true},
"PMINTENSET_EL1": a64SysReg{0x189e20, true, true},
"PMMIR_EL1": a64SysReg{0x189ec0, true, false},
"PMOVSCLR_EL0": a64SysReg{0x1b9c60, true, true},
"PMOVSSET_EL0": a64SysReg{0x1b9e60, true, true},
"PMSCR_EL1": a64SysReg{0x189900, true, true},
"PMSELR_EL0": a64SysReg{0x1b9ca0, true, true},
"PMSEVFR_EL1": a64SysReg{0x1899a0, true, true},
"PMSFCR_EL1": a64SysReg{0x189980, true, true},
"PMSICR_EL1": a64SysReg{0x189940, true, true},
"PMSIDR_EL1": a64SysReg{0x1899e0, true, false},
"PMSIRR_EL1": a64SysReg{0x189960, true, true},
"PMSLATFR_EL1": a64SysReg{0x1899c0, true, true},
"PMSWINC_EL0": a64SysReg{0x1b9c80, false, true},
"PMUSERENR_EL0": a64SysReg{0x1b9e00, true, true},
"PMXEVCNTR_EL0": a64SysReg{0x1b9d40, true, true},
"PMXEVTYPER_EL0": a64SysReg{0x1b9d20, true, true},
"REVIDR_EL1": a64SysReg{0x1800c0, true, false},
"RGSR_EL1": a64SysReg{0x1810a0, true, true},
"RMR_EL1": a64SysReg{0x18c040, true, true},
"RNDR": a64SysReg{0x1b2400, true, false},
"RNDRRS": a64SysReg{0x1b2420, true, false},
"RVBAR_EL1": a64SysReg{0x18c020, true, false},
"SCTLR_EL1": a64SysReg{0x181000, true, true},
"SCXTNUM_EL0": a64SysReg{0x1bd0e0, true, true},
"SCXTNUM_EL1": a64SysReg{0x18d0e0, true, true},
"SP_EL0": a64SysReg{0x184100, true, true},
"SP_EL1": a64SysReg{0x1c4100, true, true},
"SPSel": a64SysReg{0x184200, true, true},
"SPSR_abt": a64SysReg{0x1c4320, true, true},
"SPSR_EL1": a64SysReg{0x184000, true, true},
"SPSR_fiq": a64SysReg{0x1c4360, true, true},
"SPSR_irq": a64SysReg{0x1c4300, true, true},
"SPSR_und": a64SysReg{0x1c4340, true, true},
"SSBS": a64SysReg{0x1b42c0, true, true},
"TCO": a64SysReg{0x1b42e0, true, true},
"TCR_EL1": a64SysReg{0x182040, true, true},
"TFSR_EL1": a64SysReg{0x185600, true, true},
"TFSRE0_EL1": a64SysReg{0x185620, true, true},
"TPIDR_EL0": a64SysReg{0x1bd040, true, true},
"TPIDR_EL1": a64SysReg{0x18d080, true, true},
"TPIDRRO_EL0": a64SysReg{0x1bd060, true, true},
"TRFCR_EL1": a64SysReg{0x181220, true, true},
"TTBR0_EL1": a64SysReg{0x182000, true, true},
"TTBR1_EL1": a64SysReg{0x182020, true, true},
"UAO": a64SysReg{0x184280, true, true},
"VBAR_EL1": a64SysReg{0x18c000, true, true},
"ZCR_EL1": a64SysReg{0x181200, true, true},
}
// a64SysInst is one TLBI alias: the fields the SYS encoding carries beside
// the fixed op0 = 01 and CRn = 8.
type a64SysInst struct {
op1, cm, op2 uint32
}
// a64TLBIOps maps the TLBI operation names to their fields; the register
// operand is optional and defaults to ZR.
var a64TLBIOps = map[string]a64SysInst{
"ALLE1": {0x4, 0x7, 0x4},
"ALLE1IS": {0x4, 0x3, 0x4},
"ALLE1OS": {0x4, 0x1, 0x4},
"ALLE2": {0x4, 0x7, 0x0},
"ALLE2IS": {0x4, 0x3, 0x0},
"ALLE2OS": {0x4, 0x1, 0x0},
"ALLE3": {0x6, 0x7, 0x0},
"ALLE3IS": {0x6, 0x3, 0x0},
"ALLE3OS": {0x6, 0x1, 0x0},
"ASIDE1": {0x0, 0x7, 0x2},
"ASIDE1IS": {0x0, 0x3, 0x2},
"ASIDE1OS": {0x0, 0x1, 0x2},
"IPAS2E1": {0x4, 0x4, 0x1},
"IPAS2E1IS": {0x4, 0x0, 0x1},
"IPAS2E1OS": {0x4, 0x4, 0x0},
"IPAS2LE1": {0x4, 0x4, 0x5},
"IPAS2LE1IS": {0x4, 0x0, 0x5},
"IPAS2LE1OS": {0x4, 0x4, 0x4},
"RIPAS2E1": {0x4, 0x4, 0x2},
"RIPAS2E1IS": {0x4, 0x0, 0x2},
"RIPAS2E1OS": {0x4, 0x4, 0x3},
"RIPAS2LE1": {0x4, 0x4, 0x6},
"RIPAS2LE1IS": {0x4, 0x0, 0x6},
"RIPAS2LE1OS": {0x4, 0x4, 0x7},
"RVAAE1": {0x0, 0x6, 0x3},
"RVAAE1IS": {0x0, 0x2, 0x3},
"RVAAE1OS": {0x0, 0x5, 0x3},
"RVAALE1": {0x0, 0x6, 0x7},
"RVAALE1IS": {0x0, 0x2, 0x7},
"RVAALE1OS": {0x0, 0x5, 0x7},
"RVAE1": {0x0, 0x6, 0x1},
"RVAE1IS": {0x0, 0x2, 0x1},
"RVAE1OS": {0x0, 0x5, 0x1},
"RVAE2": {0x4, 0x6, 0x1},
"RVAE2IS": {0x4, 0x2, 0x1},
"RVAE2OS": {0x4, 0x5, 0x1},
"RVAE3": {0x6, 0x6, 0x1},
"RVAE3IS": {0x6, 0x2, 0x1},
"RVAE3OS": {0x6, 0x5, 0x1},
"RVALE1": {0x0, 0x6, 0x5},
"RVALE1IS": {0x0, 0x2, 0x5},
"RVALE1OS": {0x0, 0x5, 0x5},
"RVALE2": {0x4, 0x6, 0x5},
"RVALE2IS": {0x4, 0x2, 0x5},
"RVALE2OS": {0x4, 0x5, 0x5},
"RVALE3": {0x6, 0x6, 0x5},
"RVALE3IS": {0x6, 0x2, 0x5},
"RVALE3OS": {0x6, 0x5, 0x5},
"VAAE1": {0x0, 0x7, 0x3},
"VAAE1IS": {0x0, 0x3, 0x3},
"VAAE1OS": {0x0, 0x1, 0x3},
"VAALE1": {0x0, 0x7, 0x7},
"VAALE1IS": {0x0, 0x3, 0x7},
"VAALE1OS": {0x0, 0x1, 0x7},
"VAE1": {0x0, 0x7, 0x1},
"VAE1IS": {0x0, 0x3, 0x1},
"VAE1OS": {0x0, 0x1, 0x1},
"VAE2": {0x4, 0x7, 0x1},
"VAE2IS": {0x4, 0x3, 0x1},
"VAE2OS": {0x4, 0x1, 0x1},
"VAE3": {0x6, 0x7, 0x1},
"VAE3IS": {0x6, 0x3, 0x1},
"VAE3OS": {0x6, 0x1, 0x1},
"VALE1": {0x0, 0x7, 0x5},
"VALE1IS": {0x0, 0x3, 0x5},
"VALE1OS": {0x0, 0x1, 0x5},
"VALE2": {0x4, 0x7, 0x5},
"VALE2IS": {0x4, 0x3, 0x5},
"VALE2OS": {0x4, 0x1, 0x5},
"VALE3": {0x6, 0x7, 0x5},
"VALE3IS": {0x6, 0x3, 0x5},
"VALE3OS": {0x6, 0x1, 0x5},
"VMALLE1": {0x0, 0x7, 0x0},
"VMALLE1IS": {0x0, 0x3, 0x0},
"VMALLE1OS": {0x0, 0x1, 0x0},
"VMALLS12E1": {0x4, 0x7, 0x6},
"VMALLS12E1IS": {0x4, 0x3, 0x6},
"VMALLS12E1OS": {0x4, 0x1, 0x6},
}
// a64DCOps2 maps the DC operation names to their fields; the register
// operand is mandatory.
var a64DCOps2 = map[string]a64SysInst{
"CGDSW": {0x0, 0xa, 0x6},
"CGDVAC": {0x3, 0xa, 0x5},
"CGDVADP": {0x3, 0xd, 0x5},
"CGDVAP": {0x3, 0xc, 0x5},
"CGSW": {0x0, 0xa, 0x4},
"CGVAC": {0x3, 0xa, 0x3},
"CGVADP": {0x3, 0xd, 0x3},
"CGVAP": {0x3, 0xc, 0x3},
"CIGDSW": {0x0, 0xe, 0x6},
"CIGDVAC": {0x3, 0xe, 0x5},
"CIGSW": {0x0, 0xe, 0x4},
"CIGVAC": {0x3, 0xe, 0x3},
"CISW": {0x0, 0xe, 0x2},
"CIVAC": {0x3, 0xe, 0x1},
"CSW": {0x0, 0xa, 0x2},
"CVAC": {0x3, 0xa, 0x1},
"CVADP": {0x3, 0xd, 0x1},
"CVAP": {0x3, 0xc, 0x1},
"CVAU": {0x3, 0xb, 0x1},
"GVA": {0x3, 0x4, 0x3},
"GZVA": {0x3, 0x4, 0x4},
"IGDSW": {0x0, 0x6, 0x6},
"IGDVAC": {0x0, 0x6, 0x5},
"IGSW": {0x0, 0x6, 0x4},
"IGVAC": {0x0, 0x6, 0x3},
"ISW": {0x0, 0x6, 0x2},
"IVAC": {0x0, 0x6, 0x1},
"ZVA": {0x3, 0x4, 0x1},
}
// a64RPRFOps maps the range-prefetch operation names to their 6-bit values.
var a64RPRFOps = map[string]uint32{
"PLDKEEP": 0,
"PLDSTRM": 4,
"PSTKEEP": 1,
"PSTSTRM": 5,
}
+164
View File
@@ -0,0 +1,164 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
"os"
"path/filepath"
"slices"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestARM64SysRegsDifferential proves the whole system-register table against
// the toolchain at once: one TEXT whose body reads every register the table
// carries (and writes every writable one), assembled by gasm and by
// go tool asm, must agree byte for byte. A single wrong op0/op1/CRn/CRm/op2
// packing names its register through the first differing word.
func TestARM64SysRegsDifferential(t *testing.T) {
names := make([]string, 0, len(a64SysRegs))
for name := range a64SysRegs {
names = append(names, name)
}
slices.Sort(names)
var body strings.Builder
for i, name := range names {
// R18 is the arm64 platform register and R29-R31 carry dedicated
// meanings; a plain read/write destination keeps to R0-R17.
reg := fmt.Sprintf("R%d", i%18)
if a64SysRegs[name].read {
body.WriteString(fmt.Sprintf("\tMRS %s, %s\n", name, reg))
}
if a64SysRegs[name].write {
body.WriteString(fmt.Sprintf("\tMSR %s, %s\n", reg, name))
}
}
src := "#include \"textflag.h\"\n\nTEXT ·sysregs(SB), NOSPLIT, $0\n" + body.String() + "\tRET\n"
dir := t.TempDir()
path := filepath.Join(dir, "sysregs_arm64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
assertARM64Differential(t, path, src, "sysregs")
}
// TestARM64FamiliesDifferential pins the non-sysreg families the arm64
// campaign added: the LSE compare-and-swap pairs, the VMOVI immediate, the
// SIMD narrow/long shift pairs, the VLD2/VLD3/VLD4 and VST2/VST3/VST4
// structure accesses with their post-index and replicate forms, LDPSW, the
// pointer-authentication hint and the DC maintenance operation. Every
// spelling is the toolchain's own, taken from its arm64 testdata, and the
// bytes must agree word for word.
func TestARM64FamiliesDifferential(t *testing.T) {
src := `#include "textflag.h"
TEXT ·families(SB), NOSPLIT, $0
CASPD (R2, R3), (R2), (R8, R9)
CASPW (R6, R7), (R8), (R4, R5)
VMOVI $82, V0.B16
VMOVI $146, V22.B16
VSSHLL $0, V1.B8, V2.H8
VSSHLL $7, V1.B8, V2.H8
VSSHLL2 $0, V1.B16, V2.H8
VSHRN $7, V1.H8, V0.B8
VSHRN2 $31, V1.D2, V0.S4
VLD2 (R29), [V23.H8, V24.H8]
VLD2.P 16(R0), [V18.B8, V19.B8]
VLD2.P (R1)(R2), [V15.S2, V16.S2]
VLD3 (R27), [V11.S4, V12.S4, V13.S4]
VLD3.P 48(RSP), [V11.S4, V12.S4, V13.S4]
VLD4 (R15), [V10.H4, V11.H4, V12.H4, V13.H4]
VLD4.P 32(R24), [V31.B8, V0.B8, V1.B8, V2.B8]
VLD1R (R1), [V9.B8]
VLD1R.P (R0), [V0.B16]
VLD1R.P 2(R1), [V2.H4]
VLD2R (R15), [V15.H4, V16.H4]
VLD2R.P 16(R0), [V0.D2, V1.D2]
VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]
VLD4R.P 16(RSP), [V31.S4, V0.S4, V1.S4, V2.S4]
VST2 [V22.H8, V23.H8], (R23)
VST2.P [V14.H4, V15.H4], 16(R17)
VST2.P [V14.H4, V15.H4], (R3)(R17)
VST3 [V1.D2, V2.D2, V3.D2], (R11)
VST3.P [V18.S4, V19.S4, V20.S4], 48(R25)
VST4 [V22.D2, V23.D2, V24.D2, V25.D2], (R3)
VST4.P [V14.D2, V15.D2, V16.D2, V17.D2], 64(R15)
LDPSW (R0), (R1, R2)
LDPSW 4(R0), (R1, R2)
LDPSW -4(R0), (R1, R2)
PACIASP
DC IVAC, R1
RET
`
dir := t.TempDir()
path := filepath.Join(dir, "families_arm64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
assertARM64Differential(t, path, src, "families")
}
// assertARM64Differential assembles the same source with gasm and with the
// toolchain for arm64 and requires the named function's code bytes to agree.
// The live oracle is a deliberate-run comparison, so -short skips it (the
// push pipeline's mode); the golden bytes of the individual encoders are
// pinned separately in every mode.
func assertARM64Differential(t *testing.T, path, src, fn string) {
t.Helper()
oracle := oracleFuncCode(t, toolAsmObject(t, path, "arm64"))
// The oracle keys its functions by the qualified object name
// (pkg.name); match on the local part.
want := map[string][]byte{}
for name, code := range oracle {
if _, after, ok := strings.Cut(name, "."); ok {
want[after] = code
} else {
want[name] = code
}
}
if want[fn] == nil {
t.Fatalf("the oracle object carries no function %q (has %v)", fn, keysOf(want))
}
f, perrs := parser.Parse(path, src)
if len(perrs) > 0 {
t.Fatalf("parse: %v", perrs[0])
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := trimTrailingZeroWords(img.Code)
wantB := trimTrailingZeroWords(want[fn])
if len(got) != len(wantB) {
t.Fatalf("gasm %d bytes, oracle %d bytes", len(got), len(wantB))
}
for i := range wantB {
if got[i] != wantB[i] {
t.Fatalf("word %d differs: gasm %08x, oracle %08x", i/4,
binary.LittleEndian.Uint32(got[i:i+4]), binary.LittleEndian.Uint32(wantB[i:i+4]))
}
}
}
// trimTrailingZeroWords drops whole zero words off the end of a code span:
// an object pads a function to its alignment, and the raw image does not.
// A difference in the middle survives the trim untouched.
func trimTrailingZeroWords(b []byte) []byte {
for len(b) >= 4 {
last := b[len(b)-4:]
if last[0]|last[1]|last[2]|last[3] != 0 {
break
}
b = b[:len(b)-4]
}
return b
}
+574 -28
View File
@@ -8,7 +8,7 @@ import (
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// Assemble encodes the body of a TEXT function into x86-64 machine code,
@@ -38,10 +38,25 @@ func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
// rejects SB operands outright (single-function assembly cannot resolve
// them). When allowExternal is set, a reference to a symbol no GLOBL in the
// file defines is recorded as an external relocation instead of failing
// the object-file emitters resolve it at link time.
// the object-file emitters resolve it at link time. goos selects the TLS
// access form: the empty default behaves as linux.
type linkInfo struct {
symbols map[string]bool
allowExternal bool
goos string
}
// tlsOneInsn reports the one-instruction TLS form, obj6.go's
// CanUse1InsnTLS for the GOOS gasm supports: the bare TLS load nops out and
// the (TLS*1) index folds to a segment-absolute access. Windows and plan9
// keep the two-instruction form; shared linux does too, which gasm's raw
// path does not model and therefore does not select.
func (l *linkInfo) tlsOneInsn() bool {
switch l.goos {
case "", "linux", "freebsd":
return true
}
return false
}
// sbPatch is a function-relative static-symbol relocation: the disp32 field
@@ -86,6 +101,10 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
// outgrows the short form.
long := make([]bool, len(t.Body))
sizes := make([]int, len(t.Body))
numTargets := make([]int, len(t.Body))
for i := range numTargets {
numTargets[i] = -1
}
offsets := map[string]int{}
pcs := make([]int, len(t.Body))
var guardJBlong, guardJBElong, moreJMPlong bool
@@ -94,23 +113,106 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
for {
guard := fi.guardLen(guardJBlong, guardJBElong)
pos := guard + len(fi.prologue)
for i := range numTargets {
numTargets[i] = -1
}
idxAtPc := map[int]int{}
for i, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
// The alignment pseudo-statement: its size is the
// padding to the next boundary at this very position,
// filled with NOPs at emission.
pad, err := pcAlignPad(pcAlignValue(s), pos)
if err != nil {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("PCALIGN: %w", err)
}
sizes[i] = pad
pcs[i] = pos
pos += pad
continue
}
sz, err := instrSize(s, fi, long[i], link)
if err != nil {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
}
sizes[i] = sz
pcs[i] = pos
idxAtPc[pos] = i
pos += sz
}
}
bodyLen := pos - (guard + len(fi.prologue))
// Expand any short jump whose displacement no longer fits rel8.
changed := false
// Numeric ±N(PC) jumps resolve against this iteration's layout; the
// emission pass reads the same table after the loop converges. A
// target that is itself an unconditional local JMP is chased to the
// ultimate target: the toolchain's brloop pass collapses branch-to-
// branch chains before it encodes, so matching its bytes requires
// the same redirection.
for i := range numTargets {
numTargets[i] = -1
}
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
if !ok {
continue
}
if len(s.Operands) == 1 {
if n, isNum := pcJumpOffset(s.Operands[0]); isNum {
if target, okT := pcJumpTarget(t, i, n, pcs); okT {
numTargets[i] = target
}
}
}
}
for i := range numTargets {
if numTargets[i] < 0 {
continue
}
tgt := numTargets[i]
for hop := 0; hop < len(t.Body); hop++ {
idx, ok := idxAtPc[tgt]
if !ok {
break
}
in, ok := t.Body[idx].(*ast.Instr)
if !ok || strings.ToUpper(in.Mnemonic.Text) != "JMP" || len(in.Operands) != 1 {
break
}
if name, isLabel := labelName(in.Operands[0]); isLabel {
tgt = offsets[resolve(name)]
continue
}
if n, isNum := pcJumpOffset(in.Operands[0]); isNum {
next, okT := pcJumpTarget(t, idx, n, pcs)
if !okT {
break
}
tgt = next
continue
}
break // JMP through a register or memory: the chain ends
}
numTargets[i] = tgt
}
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
if !ok {
continue
}
if numTargets[i] >= 0 && !long[i] {
rel := int64(numTargets[i] - (pcs[i] + jumpSize(strings.ToUpper(s.Mnemonic.Text), false)))
if !fits8(rel) {
long[i] = true
changed = true
}
}
}
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
if !ok {
@@ -232,7 +334,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
spadjStep{pos + epi, 0},
)
}
code, ps, pool, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
code, ps, pool, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link, numTargets[i])
if err != nil {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
}
@@ -428,6 +530,99 @@ func computeFrame(t *ast.Text) frameInfo {
return fi
}
// pcJumpOffset recognises the numeric relative jump operand ±N(PC) and
// returns N: the toolchain counts instructions, not bytes, so +2(PC) targets
// the second instruction boundary after the branch.
func pcJumpOffset(op *ast.Operand) (int, bool) {
if op.Kind != ast.OpAddr || op.Addr.Base != "PC" {
return 0, false
}
return int(op.Addr.Offset), true
}
// pcJumpTarget resolves a numeric jump at statement index j: N counts the
// instruction statements after the jump itself (N = 0 is the jump's own
// address, the classic park loop), and the target is the start of the Nth
// one. A negative N counts the same way backwards, before the jump: the
// exit loops write JMP -3(PC) to land three instructions earlier. Labels
// count not, in either direction. It reports false when the count runs
// past the end of the function, or before its first instruction.
func pcJumpTarget(t *ast.Text, j, n int, pcs []int) (int, bool) {
if n == 0 {
return pcs[j], true
}
if n < 0 {
seen := 0
for k := j - 1; k >= 0; k-- {
if _, ok := t.Body[k].(*ast.Instr); !ok {
continue
}
seen--
if seen == n {
return pcs[k], true
}
}
return 0, false
}
seen := 0
for k := j + 1; k < len(t.Body); k++ {
if _, ok := t.Body[k].(*ast.Instr); !ok {
continue
}
seen++
if seen == n {
return pcs[k], true
}
}
return 0, false
}
// x86 NOP encodings, single-instruction no-ops of lengths 1 to 9 (the
// toolchain's asm6.go nop table); longer padding repeats the largest that
// fits, greedy from the end.
var x86Nops = [][]byte{
{0x90},
{0x66, 0x90},
{0x0F, 0x1F, 0x00},
{0x0F, 0x1F, 0x40, 0x00},
{0x0F, 0x1F, 0x44, 0x00, 0x00},
{0x66, 0x0F, 0x1F, 0x44, 0x00, 0x00},
{0x0F, 0x1F, 0x80, 0x00, 0x00, 0x00, 0x00},
{0x0F, 0x1F, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00},
{0x66, 0x0F, 0x1F, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00},
}
// fillNOPs fills p with the greedy largest single-instruction NOPs, exactly
// the toolchain's fillnop.
func fillNOPs(p []byte) {
for len(p) > 0 {
m := min(len(p), len(x86Nops))
copy(p[:m], x86Nops[m-1])
p = p[m:]
}
}
// pcAlignPad computes the padding PCALIGN $align inserts at pos: the
// alignment must be a power of two in [8, 2048] and the padding runs to the
// next boundary (zero when the position is already aligned).
func pcAlignPad(align, pos int) (int, error) {
if align <= 0 || align&(align-1) != 0 || align < 8 || align > 2048 {
return 0, fmt.Errorf("alignment value of an instruction must be a power of two and in the range [8, 2048], got %d", align)
}
if lob := pos & (align - 1); lob != 0 {
return align - lob, nil
}
return 0, nil
}
// pcAlignValue reads a PCALIGN statement's alignment operand.
func pcAlignValue(s *ast.Instr) int {
if len(s.Operands) == 1 && s.Operands[0].Kind == ast.OpImmediate && s.Operands[0].Imm.HasVal {
return int(s.Operands[0].Imm.Val)
}
return 0 // rejected by pcAlignPad's range check
}
// hasCall reports whether the function body contains a CALL instruction.
func hasCall(t *ast.Text) bool {
for _, stmt := range t.Body {
@@ -602,7 +797,7 @@ func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, erro
}
return jumpSize(mnem, long), nil
}
code, _, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
code, _, _, err := encodeInstr(s, 0, nil, fi, false, nil, link, -1)
if err != nil {
return 0, err
}
@@ -613,17 +808,43 @@ func isJumpMnemonic(mnem string) bool {
if mnem == "JMP" || mnem == "CALL" {
return true
}
if isLoopMnemonic(mnem) {
return true
}
_, ok := condCode(mnem)
return ok
}
// isLoopMnemonic reports the LOOP family, rel8 alone (E0-E2).
func isLoopMnemonic(mnem string) bool {
switch mnem {
case "LOOP", "LOOPE", "LOOPNE":
return true
}
return false
}
// loopOpcode maps the LOOP family to its E0-E2 opcode.
func loopOpcode(mnem string) byte {
switch mnem {
case "LOOPE":
return 0xE1
case "LOOPNE":
return 0xE0
}
return 0xE2
}
// jumpSize returns the length of a jump instruction in the requested form:
// short (rel8) where available, otherwise the rel32 form. CALL is always
// rel32.
// rel32; the LOOP family is rel8 alone.
func jumpSize(mnem string, long bool) int {
if mnem == "CALL" {
return 5 // opcode + rel32
}
if isLoopMnemonic(mnem) {
return 2 // opcode + rel8, the only form
}
if !long {
return 2 // opcode + rel8
}
@@ -637,9 +858,21 @@ func jumpSize(mnem string, long bool) int {
// (relative to pc, the instruction's own offset). A RET in a frame-pointer
// function is prefixed with the epilogue. resolve, when non-nil, redirects a
// jump label through the jump-to-jump chain before the offset lookup.
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo, numTarget int) ([]byte, []sbPatch, []floatPoolEntry, error) {
mnem := strings.ToUpper(s.Mnemonic.Text)
if mnem == "PCALIGN" {
// The layout pass already accounted the padding; emit the same
// amount of NOP bytes for the statement's own position.
pad, err := pcAlignPad(pcAlignValue(s), pc)
if err != nil {
return nil, nil, nil, err
}
out := make([]byte, pad)
fillNOPs(out)
return out, nil, nil, nil
}
var prefix []byte
if mnem == "RET" && fi.useFP {
prefix = fi.epilogue
@@ -677,7 +910,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
}
return append(prefix, code...), nil, nil, nil
}
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve, numTarget)
} else {
code, ps, pool, err = encodeNormal(s, fi, link)
}
@@ -703,6 +936,51 @@ func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch
}
return code, nil, nil, nil
}
// MOVQ $sym±off(SB), r64: the toolchain assembles a symbol immediate as
// LEAQ disp32(RIP), r64 with an R_PCREL relocation at the disp32 field,
// never as a 64-bit absolute immediate (verified against go tool asm).
// MOVD is the MOVQ alias; the narrower widths reject the form outright.
if (mnemUpper == "MOVQ" || mnemUpper == "MOVD") && len(s.Operands) == 2 &&
s.Operands[0].Kind == ast.OpImmediate && s.Operands[0].Imm.Sym != nil &&
s.Operands[0].Imm.Sym.Pseudo == "SB" {
mem := &ast.Operand{Kind: ast.OpAddr, Addr: ast.Address{Sym: s.Operands[0].Imm.Sym}}
src, err := operandFromAST(mnemUpper, mem, 8, fi, link)
if err != nil {
return nil, nil, nil, err
}
dst, err := operandFromAST(mnemUpper, s.Operands[1], 8, fi, link)
if err != nil {
return nil, nil, nil, err
}
e := &enc{}
if err := e.encodeLea([]Operand{src, dst}, 8); err != nil {
return nil, nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
}
return e.out, ps, nil, nil
}
// MOVQ/MOVL TLS, r: the bare TLS load. The toolchain's progedit nops
// it out on the one-instruction TLS systems (linux and freebsd, not
// shared) and encodes the segment-prefixed load elsewhere; get_tls(r),
// the macro GOROOT's go_tls.h defines, expands to exactly this
// statement, and the toolchain's pairing pass removes it whenever the
// following instruction's (TLS*1) index folds.
if (mnemUpper == "MOVQ" || mnemUpper == "MOVL") && len(s.Operands) == 2 && isBareTLS(s.Operands[0]) {
return encodeTLSBaseLoad(s, fi, link)
}
// The old paired-register shift spelling, SHLL CX, R11:AX (a colon
// between the two registers), is the toolchain's SHLD family: SHLDL CL,
// AX, R11 with the count register first, the paired source in the reg
// field and the pair's head in r/m.
if code, ps, err := encodeColonShift(s, mnemUpper, fi, link); code != nil || err != nil {
if err != nil {
return nil, nil, nil, err
}
return code, ps, nil, nil
}
_, size := splitSize(mnemUpper)
if size == 0 {
size = 8
@@ -722,10 +1000,67 @@ func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch
ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
if p.tls {
ps[i].kind = RelTLSLE
}
}
return e.out, ps, e.floatPoolList(), nil
}
// isBareTLS reports whether the operand is the bare TLS pseudo-register
// load source, the expansion of go_tls.h's get_tls(r) macro.
func isBareTLS(op *ast.Operand) bool {
return op.Kind == ast.OpAddr && op.Addr.Sym != nil &&
op.Addr.Sym.Pseudo == "" && op.Addr.Sym.Name == "TLS" &&
op.Addr.Base == "" && op.Addr.Index == ""
}
// encodeTLSBaseLoad assembles MOVQ/MOVL TLS, r. On the one-instruction TLS
// systems (linux and freebsd outside -shared, obj6.go's CanUse1InsnTLS) the
// statement nops out: the following (TLS*1) access folds to a direct
// segment-absolute load. The two-instruction systems keep the segment load,
// nine bytes with the R_TLSLE patch site at the disp32.
func encodeTLSBaseLoad(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
if size == 0 {
size = 8
}
dst, err := operandFromAST("MOVQ", s.Operands[1], 8, fi, link)
if err != nil {
return nil, nil, nil, err
}
reg, ok := dst.(Reg)
if !ok || reg.isVec() {
return nil, nil, nil, fmt.Errorf("TLS: destination must be a general register")
}
if link == nil || link.tlsOneInsn() {
return nil, nil, nil, nil // noped out
}
seg := byte(0x64) // FS
if link.goos == "windows" {
seg = 0x65 // GS
}
e := &enc{}
i := &instr{
prefix: seg,
rexW: size == 8,
rexR: reg.idx >= 8,
opcode: []byte{0x8B},
modrm: 0x04 | (reg.idx&7)<<3,
sib: 0x25,
disp: le32(0),
tls: true,
}
if err := e.emit(i); err != nil {
return nil, nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend, kind: RelTLSLE}
}
return e.out, ps, nil, nil
}
// encodeBookkeeping accepts-and-ignores FUNCDATA and PCDATA at the statement
// level, before operand conversion: the toolchain's shapes are FUNCDATA
// $n, sym(SB) and PCDATA $n, $m, and neither contributes a byte to the
@@ -752,23 +1087,91 @@ func encodeBookkeeping(upper string, s *ast.Instr) ([]byte, error) {
return nil, nil
}
// encodeColonShift encodes the paired-register shift spellings, SHLx CX,
// dst:src: the toolchain reads them as the SHLD family (double-precision
// shift by CL), reg = the paired source, r/m = the pair's head. The second
// operand's raw text carries the colon; ok reports the spelling was found.
func encodeColonShift(s *ast.Instr, mnemUpper string, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) {
base, _ := strings.CutPrefix(mnemUpper, "SHL")
if base == mnemUpper || len(s.Operands) != 2 {
return nil, nil, nil
}
_, size := splitSize(mnemUpper)
raw := strings.ReplaceAll(s.Operands[1].Raw, " ", "")
head, tail, ok := strings.Cut(raw, ":")
if !ok || head == "" || tail == "" {
return nil, nil, nil
}
headReg, ok1 := ParseReg(head)
srcReg, ok2 := ParseReg(tail)
if !ok1 || !ok2 {
return nil, nil, fmt.Errorf("%s: invalid paired register %q", mnemUpper, s.Operands[1].Raw)
}
cnt, err := operandFromAST(mnemUpper, s.Operands[0], size, fi, link)
if err != nil {
return nil, nil, err
}
cntReg, ok := cnt.(Reg)
if !ok || cntReg.idx != 1 {
return nil, nil, fmt.Errorf("%s: the paired-register form counts in CL", mnemUpper)
}
// SHLD r/m, reg, CL: 0F A5 (REX.W for the 64-bit width).
e := &enc{}
i := &instr{rexW: size == 8, opcode: []byte{0x0F, 0xA5}, modrm: -1, sib: -1}
if err := setRM(i, srcReg, headReg, size); err != nil {
return nil, nil, err
}
if err := e.emit(i); err != nil {
return nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for j, p := range e.patches {
ps[j] = sbPatch{off: p.off, name: p.name, addend: p.addend}
}
return e.out, ps, nil
}
// trailingIndexGroup recovers a trailing "(index*scale)" or "(index)" group
// from an operand's raw text: the symbol-pseudo parse returns before the
// index group, so foo(SP)(AX*1) keeps its index only in the spelling.
func trailingIndexGroup(raw string) (string, int, bool) {
compact := strings.ReplaceAll(raw, " ", "")
if !strings.HasSuffix(compact, ")") {
return "", 0, false
}
open := strings.LastIndex(compact, "(")
if open < 2 || !strings.Contains(compact[:open], ")") {
return "", 0, false // one group alone: no trailing index
}
name, scale, _, ok := cutParenGroup(compact[open:])
return name, scale, ok
}
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
// target label, in the short (rel8) or long (rel32) form.
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long bool, resolve func(string) string) ([]byte, error) {
// target label or from a numeric ±N(PC) instruction count, in the short
// (rel8) or long (rel32) form. numTarget is the resolved byte offset of a
// numeric operand, negative when the operand is not one.
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long bool, resolve func(string) string, numTarget int) ([]byte, error) {
if len(s.Operands) != 1 {
return nil, fmt.Errorf("jump expects 1 operand, got %d", len(s.Operands))
}
name, ok := labelName(s.Operands[0])
if !ok {
name, isLabel := labelName(s.Operands[0])
if !isLabel && numTarget < 0 {
return nil, fmt.Errorf("jump target must be a local label")
}
var target int
if isLabel {
if resolve != nil && mnem != "CALL" {
name = resolve(name)
}
target, ok := offsets[name]
t, ok := offsets[name]
if !ok {
return nil, fmt.Errorf("undefined label %q", name)
}
target = t
} else {
target = numTarget
}
rel := int64(target - (pc + jumpSize(mnem, long)))
if !long {
@@ -778,9 +1181,15 @@ func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long
if mnem == "JMP" {
return []byte{0xEB, byte(int8(rel))}, nil
}
if isLoopMnemonic(mnem) {
return []byte{loopOpcode(mnem), byte(int8(rel))}, nil
}
cc, _ := condCode(mnem)
return []byte{0x70 + byte(cc), byte(int8(rel))}, nil
}
if isLoopMnemonic(mnem) {
return nil, fmt.Errorf("%s has no long form", mnem)
}
switch mnem {
case "JMP":
return append([]byte{0xE9}, le32(rel)...), nil
@@ -832,6 +1241,72 @@ func labelName(op *ast.Operand) (string, bool) {
return "", false
}
// jumpOperand returns the branch-target operand of a JMP/CALL, rewriting the
// `*`-prefixed indirect spellings (JMP *(R12), JMP *4(SP)) into their plain
// memory form. The star marks an indirect target and changes no bytes; the
// address parser leaves the operand's address empty because of the leading
// star, so the fields are rebuilt from the raw text onto a copy of the
// operand, never on the shared syntax tree.
func jumpOperand(s *ast.Instr) *ast.Operand {
if len(s.Operands) != 1 {
return nil
}
op := s.Operands[0]
compact := strings.ReplaceAll(op.Raw, " ", "")
inner, ok := strings.CutPrefix(compact, "*")
if !ok {
return op
}
var addr ast.Address
if i := strings.IndexByte(inner, '('); i > 0 {
v, err := strconv.ParseInt(inner[:i], 0, 64)
if err != nil {
return op
}
addr.Offset, addr.HasOff = v, true
inner = inner[i:]
}
base, _, rest, ok := cutParenGroup(inner)
if !ok {
return op
}
if base != "" {
addr.Base = base
}
if rest != "" {
idx, scale, _, ok := cutParenGroup(rest)
if ok && idx != "" {
addr.Index = idx
addr.Scale = scale
}
}
c := *op
c.Addr = addr
return &c
}
// cutParenGroup splits a leading "(name)" or "(name*n)" off s, returning the
// inner text, the scale it names (1 when the group spells no multiplier) and
// the remainder.
func cutParenGroup(s string) (name string, scale int, rest string, ok bool) {
if !strings.HasPrefix(s, "(") {
return "", 0, "", false
}
i := strings.IndexByte(s, ')')
if i < 0 {
return "", 0, "", false
}
inner, rest := s[1:i], s[i+1:]
if before, after, ok := strings.Cut(inner, "*"); ok {
n, err := strconv.Atoi(after)
if err != nil {
return "", 0, "", false
}
return before, n, rest, true
}
return inner, 1, rest, true
}
// indirectJumpTarget reports whether the JMP/CALL operand addresses a
// register or a memory location rather than a label or a static symbol.
// A bare identifier is a register when the register table knows the name and
@@ -840,7 +1315,19 @@ func indirectJumpTarget(s *ast.Instr) bool {
if len(s.Operands) != 1 || s.Operands[0].Kind != ast.OpAddr {
return false
}
a := s.Operands[0].Addr
op := jumpOperand(s)
if op == nil {
return false
}
if op != s.Operands[0] {
return true // the star marker spells an indirect target
}
a := op.Addr
// ±N(PC) is the numeric relative form, the PC counts instructions from
// the branch: relative, not indirect.
if a.Base == "PC" || a.Index == "PC" {
return false
}
if a.Base != "" || a.Index != "" {
return true
}
@@ -853,18 +1340,20 @@ func indirectJumpTarget(s *ast.Instr) bool {
}
// encodeIndirectJump assembles a JMP/CALL through a register or memory
// operand, which carries no relocation and no label to resolve.
// operand, which carries no relocation and no label to resolve. The
// `*`-prefixed spellings go through jumpOperand first, their star rebuilt
// into a plain memory operand.
func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
ops := make([]Operand, len(s.Operands))
for i, op := range s.Operands {
op := s.Operands[0]
if cleaned := jumpOperand(s); cleaned != nil {
op = cleaned
}
o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil)
if err != nil {
return nil, err
}
ops[i] = o
}
e := &enc{}
if err := e.encodeIndirectBranch(mnem, ops); err != nil {
if err := e.encodeIndirectBranch(mnem, []Operand{o}); err != nil {
return nil, err
}
return e.out, nil
@@ -933,37 +1422,89 @@ func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, l
off := a.Sym.Offset + fi.fpAdjust
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
}
// SP-relative local: x-N(SP) → (spAdjust + offset)(SP).
// SP-relative local: x-N(SP) → (spAdjust + offset)(SP), keeping a scaled
// index beside the virtual stack pointer (foo(SP)(AX*1)). The
// symbol-pseudo parse returns before the index group, so the index
// is recovered from the raw text when the address lacks it.
if a.Sym != nil && a.Sym.Pseudo == "SP" && a.Base == "" {
off := fi.spAdjust + a.Sym.Offset
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
m := Mem{Base: spReg, Disp: off, HasBase: true, Size: size}
if name, scale, ok := trailingIndexGroup(op.Raw); ok {
idx, ok := ParseReg(name)
if !ok {
return nil, fmt.Errorf("unknown index register %q", name)
}
m.Index = idx
m.Scale = scale
m.HasIndex = true
}
return m, nil
}
// SB (global symbol): a symbol defined in the same file (GLOBL) is
// encoded RIP-relative and resolved by the file-level layout;
// anything not defined here needs object-file emission.
// anything not defined here needs object-file emission. A static
// (file-local) spelling of an undefined symbol defers the same way
// the toolchain does: the relocation names it and the linker decides.
if a.Sym != nil && a.Sym.Pseudo == "SB" {
if link == nil || link.symbols == nil {
return nil, fmt.Errorf("symbol %q needs file-level assembly (AssembleFile)", a.Sym.Name)
}
if !link.symbols[a.Sym.Name] {
if a.Sym.Static {
return nil, fmt.Errorf("undefined symbol %q", a.Sym.Name)
}
if !link.allowExternal {
if !link.symbols[a.Sym.Name] && !link.allowExternal {
return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name)
}
}
return sbMem{size: size, name: a.Sym.Name, addend: a.Sym.Offset}, nil
}
// Memory with a real base register: (base), off(base), (base)(index*scale).
if a.Base != "" {
// The TLS pseudo-base, off(TLS): the segment-prefixed absolute
// the thread-local access lowers to, 64 8B 04 25 with its
// R_TLS_LE patch site on the disp32.
if a.Base == "TLS" {
seg := byte(0x64) // FS on linux, freebsd, plan9
if link != nil && link.goos == "windows" {
seg = 0x65 // GS
}
return TLSMem{Disp: a.Offset, Size: size, Seg: seg}, nil
}
// Segment-absolute: 0x30(GS) and 0x28(FS), the windows TLS
// spellings. The segment override prefixes a disp32 absolute
// reference with no relocation.
if a.Base == "GS" || a.Base == "FS" {
seg := byte(0x64)
if a.Base == "GS" {
seg = 0x65
}
return SegAbs{Disp: a.Offset, Size: size, Seg: seg}, nil
}
base, ok := ParseReg(a.Base)
if !ok {
return nil, fmt.Errorf("unknown base register %q", a.Base)
}
m := Mem{Base: base, Disp: a.Offset, HasBase: true, Size: size}
if a.Index != "" {
if a.Index == "TLS" {
// off(base)(TLS*1): the thread-local annotation. The
// one-instruction TLS form folds it to off(TLS), the
// segment-prefixed absolute whose disp32 carries an
// R_TLS_LE patch site; the base register disappears
// from the encoding, exactly as the toolchain's
// progedit rewrites the address.
seg := byte(0x64) // FS on linux, freebsd, plan9
if link != nil && link.goos == "windows" {
seg = 0x65 // GS
}
return TLSMem{Disp: a.Offset, Size: size, Seg: seg}, nil
}
if a.Index == "GS" || a.Index == "FS" {
// 0(CX)(GS): the segment annotation rides the base
// access as the override prefix.
m.Seg = 0x64
if a.Index == "GS" {
m.Seg = 0x65
}
return m, nil
}
idx, ok := ParseReg(a.Index)
if !ok {
return nil, fmt.Errorf("unknown index register %q", a.Index)
@@ -984,6 +1525,11 @@ func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, l
}
return Mem{Index: idx, Scale: a.Scale, Disp: a.Offset, HasIndex: true, Size: size}, nil
}
// A bare displacement with no base: the absolute address form,
// MOVL $0xf1, 0xf1. No segment and no relocation.
if a.Sym == nil && a.Base == "" && a.Index == "" && a.HasOff {
return SegAbs{Disp: a.Offset, Size: size}, nil
}
// Bare register.
if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" {
if r, ok := ParseReg(a.Sym.Name); ok {
+54 -2
View File
@@ -10,8 +10,8 @@ import (
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// firstText parses src and returns its first TEXT function.
@@ -370,6 +370,58 @@ end:
}
}
// TestAssembleNumericPCJumps pins the numeric ±N(PC) branch operands: N
// counts instruction statements, skipping labels, in both directions (the
// runtime's exit loops write JMP -3(PC)), N = 0 parks on the jump itself.
func TestAssembleNumericPCJumps(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·exit(SB), NOSPLIT, $0
MOVB $1, AL
lab:
MOVB $2, AL
MOVB $3, AL
JMP -3(PC)
MOVB $4, AL
park:
JMP 0(PC)
MOVB $5, AL
JMP 2(PC)
MOVB $6, AL
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
// From the Go-assembled function:
// MOVB $1, AL b001
// MOVB $2, AL b002
// MOVB $3, AL b003
// JMP -3(PC) ebf8 (three instructions back, past lab:)
// MOVB $4, AL b004
// JMP 0(PC) ebfe (the park loop)
// MOVB $5, AL b005
// JMP 2(PC) eb02 (over MOVB $6 to the RET)
// MOVB $6, AL b006
// RET c3
want := []byte{
0xb0, 0x01,
0xb0, 0x02,
0xb0, 0x03,
0xeb, 0xf8,
0xb0, 0x04,
0xeb, 0xfe,
0xb0, 0x05,
0xeb, 0x02,
0xb0, 0x06,
0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("numeric-PC mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
func TestAssemblePrefetch(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
+1 -1
View File
@@ -8,7 +8,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// ulebIter reads ULEB128 values, the .debug_abbrev and line-header
+2 -2
View File
@@ -12,8 +12,8 @@ import (
"path/filepath"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// The object-file tests share one source: two exported functions, one
+1 -1
View File
@@ -9,7 +9,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
+1 -1
View File
@@ -9,7 +9,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
+1 -1
View File
@@ -9,7 +9,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestELFRISCVObjectDataRelocation checks that a symbol-valued DATA field
+13
View File
@@ -20,11 +20,24 @@ func Encodable(mnemonic string) bool {
switch upper {
case "RET", "NOP", "CALL", "JMP",
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
// The SSE compare family sharing CMPSD's predicate-last shape, the
// far return with its stack pop, the loop family, the bank-crossing
// MMX moves and the one-operand system controls.
"CMPSS", "CMPPS", "CMPPD", "RETFL",
"LOOP", "LOOPE", "LOOPNE",
"MOVDQ2Q", "MOVQ2DQ",
"ENDBR64", "CLWB", "TPAUSE", "UMONITOR", "UMWAIT", "RDPID", "CLDEMOTE",
// The literal-data pseudo-ops, the accepted-and-ignored END and
// bookkeeping statements, and the SP adjust.
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA":
return true
}
if _, ok := sysUnaryTable[upper]; ok {
return true
}
if _, ok := sseStoreOnly[upper]; ok {
return true
}
if _, ok := noOperandTable[upper]; ok {
return true
}
+108 -3
View File
@@ -64,6 +64,7 @@ type encPatch struct {
off int
name string
addend int64
tls bool // a TLS slot offset: the patch is R_TLSLE with no symbol
}
func (e *enc) encode(mnem string, ops []Operand) error {
@@ -72,9 +73,23 @@ func (e *enc) encode(mnem string, ops []Operand) error {
// Fixed-name instructions (no size suffix).
switch {
case upper == "RET":
// RET sym(SB), the absolute return: the toolchain encodes it as a
// tail jump, E9 rel32 with a call relocation against the symbol.
if len(ops) == 1 {
if m, ok := ops[0].(sbMem); ok {
return e.emit(&instr{opcode: []byte{0xE9}, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}})
}
return fmt.Errorf("RET: unsupported operand")
}
if len(ops) != 0 {
return fmt.Errorf("RET expects no operands, got %d", len(ops))
}
return e.encodeRet()
case upper == "NOP":
return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1})
// The toolchain consumes every NOP statement as a pseudo and emits
// nothing for it, operands included (a bare NOP, NOP AX and
// NOP sym(SB) all vanish from the object).
return nil
case upper == "CALL" || upper == "JMP":
// Through a register or memory: FF /2 (CALL) or FF /4 (JMP).
// Anything else is a rel32 against a label resolved by the assembler.
@@ -101,6 +116,15 @@ func (e *enc) encode(mnem string, ops []Operand) error {
}
return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
}
// One-operand system instructions whose reg field is a fixed digit:
// the cache and wait controls under 0F AE/0F 1C and the RDPID read.
if m, ok := sysUnaryTable[upper]; ok {
return e.encodeSysUnary(upper, m, ops)
}
// The store-only SSE moves (the non-temporal store).
if m, ok := sseStoreOnly[upper]; ok {
return e.encodeSSEStoreOnly(upper, m, ops)
}
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
// are rejected by go tool asm in 64-bit mode, so they stay unsupported.
switch upper {
@@ -116,14 +140,63 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
case "INT":
return e.encodeInt(ops)
// The LOOP family outside the assembler's label settlement: the operand
// is the already-computed rel8 (E0-E2).
case "LOOP", "LOOPE", "LOOPNE":
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 operand, got %d", upper, len(ops))
}
imm, ok := ops[0].(Imm)
if !ok || !fits8(int64(imm)) {
return fmt.Errorf("%s: relative offset must be a signed byte", upper)
}
return e.emit(&instr{opcode: []byte{loopOpcode(upper)}, modrm: -1, sib: -1, imm: []byte{byte(int8(imm))}})
case "LDMXCSR":
return e.encodeMxcsr(2, ops)
case "STMXCSR":
return e.encodeMxcsr(3, ops)
// CMPSD is the scalar double compare, whose predicate immediate comes
// LAST in Plan 9 order (src, dst, $imm).
// LAST in Plan 9 order (src, dst, $imm); the family shares the shape.
case "CMPSD":
return e.encodeCmpsd(ops)
return e.encodeSSECmp("CMPSD", 0xF2, ops)
case "CMPSS":
return e.encodeSSECmp("CMPSS", 0xF3, ops)
case "CMPPS":
return e.encodeSSECmp("CMPPS", 0x00, ops)
case "CMPPD":
return e.encodeSSECmp("CMPPD", 0x66, ops)
// RETFL pops the immediate's worth of bytes after the far return
// (LRET iw: CA imm16), the toolchain's RETF spelling with a stack
// adjustment.
case "RETFL":
if len(ops) != 1 {
return fmt.Errorf("RETFL expects 1 operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("RETFL expects an immediate")
}
return e.emit(&instr{opcode: []byte{0xCA}, modrm: -1, sib: -1, imm: le16(int64(imm))})
// MOVDQ2Q/MOVQ2DQ cross the MMX and XMM banks (F2 0F D6), the register
// in the reg field, the other bank's in r/m.
case "MOVDQ2Q", "MOVQ2DQ":
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
srcReg, ok1 := ops[0].(Reg)
dstReg, ok2 := ops[1].(Reg)
if !ok1 || !ok2 {
return fmt.Errorf("%s takes register operands alone", upper)
}
if upper == "MOVDQ2Q" && (!srcReg.isVec() || !dstReg.mmx) ||
upper == "MOVQ2DQ" && (!srcReg.mmx || !dstReg.isVec()) {
return fmt.Errorf("%s crosses the XMM and MMX banks in that order", upper)
}
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, srcReg, 8); err != nil {
return err
}
return e.emit(i)
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
case "SHA256RNDS2":
return e.encodeSha256rnds2(ops)
@@ -578,6 +651,7 @@ type instr struct {
disp []byte
imm []byte
sb *sbRef // static-symbol displacement in disp, awaiting resolution
tls bool // the displacement is a TLS slot offset, patched R_TLSLE
}
// sbRef records that an instruction's displacement refers to a static symbol
@@ -620,6 +694,9 @@ func (e *enc) emit(i *instr) error {
if i.sb != nil {
e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend})
}
if i.tls {
e.patches = append(e.patches, encPatch{off: len(e.out), tls: true})
}
e.out = append(e.out, i.disp...)
e.out = append(e.out, i.imm...)
return nil
@@ -674,12 +751,30 @@ func setRMReg(i *instr, regField int, rexR, regForced bool, rm Operand, opSize i
i.disp = le32(0)
i.sb = &sbRef{name: r.name, addend: r.addend}
return nil
case TLSMem:
// off(TLS): the segment-prefixed absolute access, mod=00 with the
// SIB escape's disp32 absolute form. The displacement is the TLS
// slot offset, patched by the linker's TLS relocation.
i.prefix = r.Seg
i.modrm = 0x04 | regField<<3
i.sib = 0x25
i.disp = le32(r.Disp)
i.tls = true
return nil
case SegAbs:
// 0x30(GS): the segment override with the SIB escape's disp32
// absolute form, no relocation.
setSegAbs(i, regField, r)
return nil
default:
return fmt.Errorf("invalid r/m operand %T", rm)
}
}
func setMem(i *instr, regField int, m Mem) error {
if m.Seg != 0 {
i.prefix = m.Seg
}
modrm, sib, disp, xBit, bBit, err := memComponents(regField, m)
if err != nil {
return err
@@ -692,6 +787,16 @@ func setMem(i *instr, regField int, m Mem) error {
return nil
}
// setSegAbs assembles a segment-absolute operand, 0x30(GS): the segment
// override with the mod=00 SIB escape's disp32 absolute form and no
// relocation.
func setSegAbs(i *instr, regField int, m SegAbs) {
i.prefix = m.Seg
i.modrm = 0x04 | regField<<3
i.sib = 0x25
i.disp = le32(m.Disp)
}
// memComponents computes the ModR/M byte (with the given reg field), the SIB
// byte (-1 if none), the displacement bytes, and the high index/base bits, for
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
+158 -4
View File
@@ -10,8 +10,8 @@ import (
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// decode encodes an instruction and decodes it back, returning the decoded
@@ -265,7 +265,11 @@ func TestImul(t *testing.T) {
func TestControl(t *testing.T) {
checkSyntax(t, "ret", "RET")
checkSyntax(t, "nop", "NOP")
// NOP contributes nothing on amd64, consumed whole by the toolchain as
// a pseudo; only the bytes pin it (no decodable instruction remains).
if code, err := Encode("NOP"); err != nil || len(code) != 0 {
t.Errorf("NOP: bytes %x (err %v), want empty", code, err)
}
checkOp(t, x86asm.JMP, "JMP", Imm(0))
checkOp(t, x86asm.CALL, "CALL", Imm(0))
checkOp(t, x86asm.JGE, "JGE", Imm(0))
@@ -1193,7 +1197,7 @@ func TestBookkeepingGroundTruth(t *testing.T) {
if err != nil {
t.Fatalf("assemble: %v", err)
}
want := "90c3"
want := "c3"
if got := hexCompact(img.Code); got != want {
t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want)
}
@@ -1221,3 +1225,153 @@ func mustParse(t *testing.T, src string) *ast.File {
}
return f
}
// TestCorpusTailSystem pins the system and control forms the toolchain's own
// amd64 testdata carries, byte for byte: the one-operand IMUL, the compare
// family, the far return, the loop, the MMX moves, the CR/DR and segment
// register moves, the TLS pseudo-base and the 0F AE/1C/C7 controls.
func TestCorpusTailSystem(t *testing.T) {
regBPT := Reg{idx: 5, size: 8}
X0, X1, X2 := vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")
Y1, Y2, Y7 := vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y7")
X5, X20 := vreg(t, "X5"), vreg(t, "X20")
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"IMUL one-op byte", "IMULB", []Operand{DX}, "f6ea"},
{"IMUL one-op long", "IMULL", []Operand{AX}, "f7e8"},
{"CMPPD", "CMPPD", []Operand{X1, X2, Imm(4)}, "660fc2d104"},
{"CMPSS", "CMPSS", []Operand{X1, X2, Imm(4)}, "f30fc2d104"},
{"CMPPS", "CMPPS", []Operand{X1, X2, Imm(4)}, "0fc2d104"},
{"RETFL", "RETFL", []Operand{Imm(4)}, "ca0400"},
{"LOOP", "LOOP", []Operand{Imm(-2)}, "e2fe"},
{"LOOPE", "LOOPE", []Operand{Imm(-2)}, "e1fe"},
{"LOOPNE", "LOOPNE", []Operand{Imm(-2)}, "e0fe"},
{"PADDD MMX", "PADDD", []Operand{Reg{idx: 2, size: 8, mmx: true}, Reg{idx: 1, size: 8, mmx: true}}, "0ffeca"},
{"MOVDQ2Q", "MOVDQ2Q", []Operand{X1, Reg{idx: 1, size: 8, mmx: true}}, "f20fd6c9"},
{"MOVNTDQ", "MOVNTDQ", []Operand{X1, Ptr(AX, 0, 16)}, "660fe708"},
{"MOVQ mmx load", "MOVQ", []Operand{Ptr(AX, 0, 8), Reg{idx: 0, size: 8, mmx: true}}, "0f6f00"},
{"MOVQ mmx store", "MOVQ", []Operand{Reg{idx: 0, size: 8, mmx: true}, Ptr(SI, 0, 8)}, "0f7f06"},
{"MOVQ CR0 load", "MOVQ", []Operand{Reg{idx: 0, size: 8, ctl: 1}, AX}, "0f20c0"},
{"MOVQ CR4 load", "MOVQ", []Operand{Reg{idx: 4, size: 8, ctl: 1}, DI}, "0f20e7"},
{"MOVQ CR0 store", "MOVQ", []Operand{AX, Reg{idx: 0, size: 8, ctl: 1}}, "0f22c0"},
{"MOVQ DR0 load", "MOVQ", []Operand{Reg{idx: 0, size: 8, ctl: 2}, AX}, "0f21c0"},
{"MOVQ DR7 load", "MOVQ", []Operand{Reg{idx: 7, size: 8, ctl: 2}, SI}, "0f21fe"},
{"PUSHQ FS", "PUSHQ", []Operand{Reg{idx: 4, size: 2, seg: 5}}, "0fa0"},
{"PUSHQ GS", "PUSHQ", []Operand{Reg{idx: 5, size: 2, seg: 6}}, "0fa8"},
{"POPQ FS", "POPQ", []Operand{Reg{idx: 4, size: 2, seg: 5}}, "0fa1"},
{"POPQ GS", "POPQ", []Operand{Reg{idx: 5, size: 2, seg: 6}}, "0fa9"},
{"ENDBR64", "ENDBR64", nil, "f30f1efa"},
{"CLWB", "CLWB", []Operand{Ptr(BX, 0, 8)}, "660fae33"},
{"CLDEMOTE", "CLDEMOTE", []Operand{Ptr(BX, 0, 8)}, "0f1c03"},
{"TPAUSE", "TPAUSE", []Operand{BX}, "660faef3"},
{"UMONITOR", "UMONITOR", []Operand{BX}, "f30faef3"},
{"UMWAIT", "UMWAIT", []Operand{BX}, "f20faef3"},
{"RDPID", "RDPID", []Operand{DX}, "f30fc7fa"},
{"RDPID r11", "RDPID", []Operand{Reg{idx: 11, size: 8}}, "f3410fc7fb"},
{"LEAL wide disp", "LEAL", []Operand{Idx(regBPT, Reg{idx: 10, size: 8}, 1, 0x8f1bbcdc, 8), regBPT}, "428dac15dcbc1b8f"},
{"VPERMPD", "VPERMPD", []Operand{Imm(0xd8), Y7, Y7}, "c4e3fd01ffd8"},
{"VPERMILPD", "VPERMILPD", []Operand{Imm(0xff), X1, X2}, "c4e37905d1ff"},
{"VPERMILPS", "VPERMILPS", []Operand{Imm(0xff), X1, X2}, "c4e37904d1ff"},
{"VROUNDPD", "VROUNDPD", []Operand{Imm(-1), X1, X2}, "c4e37909d1ff"},
{"VROUNDPS", "VROUNDPS", []Operand{Imm(-1), Y1, Y2}, "c4e37d08d1ff"},
{"VAESKEYGENASSIST", "VAESKEYGENASSIST", []Operand{Imm(-1), X1, X2}, "c4e379dfd1ff"},
{"VPCMPESTRI", "VPCMPESTRI", []Operand{Imm(-1), X1, X2}, "c4e37961d1ff"},
{"VPCMPESTRM", "VPCMPESTRM", []Operand{Imm(-1), X1, X2}, "c4e37960d1ff"},
{"VPCMPISTRI", "VPCMPISTRI", []Operand{Imm(-1), X1, X2}, "c4e37963d1ff"},
{"VPCMPISTRM", "VPCMPISTRM", []Operand{Imm(-1), X1, X2}, "c4e37962d1ff"},
{"VEXTRACTPS", "VEXTRACTPS", []Operand{Imm(-1), X1, AX}, "c4e37917c8ff"},
{"VPEXTRW", "VPEXTRW", []Operand{Imm(0xff), X1, AX}, "c4e37915c8ff"},
{"VPBLENDVB", "VPBLENDVB", []Operand{X0, Ptr(BX, 0, 16), X1, X2}, "c4e3714c1300"},
{"VMOVHPD load", "VMOVHPD", []Operand{Ptr(AX, 0, 8), X5, X5}, "c5d11628"},
{"VMOVHPD load disp", "VMOVHPD", []Operand{Ptr(DX, 7, 8), X5, X5}, "c5d1166a07"},
{"VMOVHPD store", "VMOVHPD", []Operand{X5, Ptr(AX, 0, 8)}, "c5f91728"},
{"VMOVLPD load", "VMOVLPD", []Operand{Ptr(AX, 0, 8), X5, X5}, "c5d11228"},
{"VMOVLPD store", "VMOVLPD", []Operand{X5, Ptr(AX, 0, 8)}, "c5f91328"},
{"VMOVQ EVEX gpr load", "VMOVQ", []Operand{Reg{idx: 4, size: 8}, X20}, "62e1fd086ee4"},
{"VMOVQ EVEX mem store", "VMOVQ", []Operand{X20, Ptr(AX, 0, 8)}, "62e1fd087e20"},
{"VMOVQ EVEX mem load", "VMOVQ", []Operand{Ptr(AX, 0, 8), X20}, "62e1fd086e20"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
}
// TestCorpusTailFileForms pins the file-level forms the toolchain's amd64
// testdata carries: the star-marked indirect jumps, the TLS pseudo-base, the
// paired-register shift spelling, the absolute RET and the jump to an
// undefined static symbol (its displacement and the TLS slot offsets are
// relocation sites, zeroed here as the kernel parity suites do).
func TestCorpusTailFileForms(t *testing.T) {
mask32 := func(b []byte, at int) { b[at], b[at+1], b[at+2], b[at+3] = 0, 0, 0, 0 }
cases := []struct {
name string
src string
want string // hex, with X marking a masked 32-bit relocation site
}{
{"star reg jump", "\tJMP *(R12)\n\tRET\n", "41ff2424c3"},
{"star sp jump", "\tJMP *4(SP)\n\tRET\n", "ff642404c3"},
{"star indexed jump", "\tJMP *(R12)(R13*4)\n\tRET\n", "43ff24acc3"},
{"TLS load", "\tMOVQ (TLS), AX\n\tRET\n", "64488b0425XXXXXXXXc3"},
{"TLS load offset", "\tMOVQ 8(TLS), DX\n\tRET\n", "64488b1425XXXXXXXXc3"},
{"colon shift", "\tSHLL CX, R11:AX\n\tRET\n", "410fa5c3c3"},
{"SP indexed local", "\tMOVQ foo(SP)(AX*1), BX\n\tRET\n", "488b1c04c3"},
}
for _, c := range cases {
f, errs := parser.Parse("t_amd64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n"+c.src)
if len(errs) > 0 {
t.Errorf("%s: parse: %v", c.name, errs)
continue
}
img, err := AssembleFile(f)
if err != nil {
t.Errorf("%s: assemble: %v", c.name, err)
continue
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
if at := strings.Index(c.want, "XXXXXXXX"); at >= 0 {
mask32(code, at/2) // the masked relocation site
}
if got := hexCompact(code); got != strings.ReplaceAll(c.want, "X", "0") {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
// JMP to an undefined static symbol and the absolute RET: their rel32
// carries a call relocation against the symbol, masked to zero here.
for _, c := range []struct{ name, src string }{
{"static external jump", "\tJMP bar<>+4(SB)\n\tRET\n"},
{"static external indexed jump", "\tJMP bar<>+4(SB)(R11*4)\n\tRET\n"},
{"absolute ret", "\tRET\n\tRET foo(SB)\n"},
} {
f, errs := parser.Parse("t_amd64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n"+c.src)
if len(errs) > 0 {
t.Errorf("%s: parse: %v", c.name, errs)
continue
}
img, err := AssembleFile(f)
if err != nil {
t.Errorf("%s: assemble: %v", c.name, err)
continue
}
fn := img.Funcs[0]
code := maskCode(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
want := "e900000000c3"
if c.name == "absolute ret" {
want = "c3e900000000"
}
if got := hexCompact(code); got != want {
t.Errorf("%s: bytes %s, want %s", c.name, got, want)
}
}
}
+66 -17
View File
@@ -856,37 +856,41 @@ type evexMoveSpec struct {
vecOK bool // the non-memory operand may be a vector register
xmmOnly bool // wider than XMM registers are rejected
nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS)
gprOK bool // the r/m side may be a general-purpose register (VMOVQ)
}
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false},
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false},
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true},
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true, false},
// EVEX.128.F2.0F.W1, scalar double move: memory operands and the
// three-operand register form (VMOVSD dst, src1, src2).
"VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true},
"VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true, false},
// EVEX.128/256/512.0F.W0, unaligned packed single move.
"VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128.66.0F.W1, the 64-bit GPR/memory ↔ XMM move (VMOVQ RSP, X20
// and friends, the EVEX spelling the high registers demand).
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, [3]int{8, 8, 8}, true, true, false, true},
}
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
@@ -900,6 +904,9 @@ func isEvex(mnemUpper string) bool {
if _, ok := evexMoveTable[mnemUpper]; ok {
return true
}
if _, ok := evexHptrTable[mnemUpper]; ok {
return true
}
return isEvexQuad(mnemUpper)
}
@@ -911,8 +918,14 @@ func evexRequired(upper string, ops []Operand) bool {
_, inVex := vexTable[upper]
_, inVexMove := vexMoveTable[upper]
if !inVex && !inVexMove {
// The dual-shape moves pick their VEX form by operand count, so
// they are not EVEX-only either.
switch upper {
case "VMOVHPD", "VMOVLPD":
default:
return true // EVEX-only mnemonic
}
}
// The byte-quad shifts have VEX register forms but EVEX-only memory
// forms: a memory count source forces the EVEX encoding.
if upper == "VPSLLDQ" || upper == "VPSRLDQ" {
@@ -1018,6 +1031,8 @@ var evexRound = map[string]bool{
"VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true,
"VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true,
"VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true,
// The scalar compares suppress exceptions on their LIG encoding.
"VCMPSD": true, "VCMPSS": true,
}
// evexBcstN maps an instruction accepting .BCST to the broadcast element
@@ -1082,6 +1097,14 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
return e.encodeEvexRM(spec, ops, 0, sfx)
}
spec, inTable := evexTable[mnemUpper]
// A high/low half move that lives in the hptr table alone (the packed
// double twins) reaches the same inTable block below, which completes
// its spec from the hptr entry.
if !inTable {
if _, ok := evexHptrTable[mnemUpper]; ok {
inTable = true
}
}
if q, ok := evexQuadTable[mnemUpper]; ok {
// The quad-register family carries no rounding, SAE or broadcast;
// only masking and zeroing apply.
@@ -1446,6 +1469,19 @@ func (e *enc) encodeEvexExtractGPR(spec evexSpec, ops []Operand, mask int, sfx e
// assembler. The scalar moves also carry a three-operand register form
// (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3
// opens.
// validEvexMoveOther reports whether the non-vector side of an EVEX move may
// take the operand: memory always, a general-purpose register when gprOK.
func validEvexMoveOther(ms evexMoveSpec, op Operand) bool {
if memOperand(op) {
return true
}
if !ms.gprOK {
return false
}
r, ok := op.(Reg)
return ok && !r.isVec() && !r.mask && r.ctl == 0 && !r.mmx && !r.fp
}
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) == 3 {
if !ms.nds3 {
@@ -1453,7 +1489,7 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
}
// The masked scalar register form keeps the Go assembler's own
// layout: the store opcode with reg = op0, vvvv = op1 and the
// destination in r/m (op2) — the bytes go tool asm emits, not
// destination in r/m (op2), the bytes go tool asm emits, not
// the manual's NDS reading.
src, src1, dst := ops[0], ops[1], ops[2]
reg, ok := src.(Reg)
@@ -1494,12 +1530,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
}
reg, rm = srcReg, dst
case srcIsVec:
if !memOperand(dst) {
if !validEvexMoveOther(ms, dst) {
return fmt.Errorf("%s: invalid destination operand", mnem)
}
reg, rm = srcReg, dst
case dstIsVec:
if !memOperand(src) {
if !validEvexMoveOther(ms, src) {
return fmt.Errorf("%s: invalid source operand", mnem)
}
op = ms.load
@@ -1890,6 +1926,15 @@ var evexHptrTable = map[string]evexHptrSpec{
"VMOVLHPS": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
},
// The packed-double twins, 66-prefixed.
"VMOVHPD": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 1, pp: 1, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x17, w: 1, pp: 1, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
"VMOVLPD": {
insert: evexSpec{mapSel: 1, opcode: 0x12, w: 1, pp: 1, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x13, w: 1, pp: 1, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
}
// encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib.
@@ -1954,7 +1999,7 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
if !ok || !maskReg.isVec() {
return fmt.Errorf("%s: mask must be a vector register", upper)
}
vsib, _, err := vsibLen(rest[1], upper)
vsib, idxLen, err := vsibLen(rest[1], upper)
if err != nil {
return err
}
@@ -1962,12 +2007,16 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", upper)
}
// The L bit is the wider of the data register and the VSIB index
// lengths (a YMM index under an XMM destination selects 256-bit, the
// bytes go tool asm emits).
ll := max(idxLen, dst.vecLenBit())
spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1}
rBit := 0
if dst.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib)
return e.emitVexFields(spec, ll, dst.idx&7, rBit, 15-maskReg.idx, vsib)
}
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src,
+159
View File
@@ -0,0 +1,159 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// The extended-instruction registry: the lookup over and above the generated
// architecture tables. The generated tables (arch/*_gen.go) list the
// mnemonics the Go toolchain knows; the extension layer carries the
// instructions it does not, and this file indexes them per architecture so
// the assembler and the linter can consult the layer without touching the
// generated lists or the main encoders. A later hook wires
// ExtensionEncodable into the Encodable mirror and EncodeExtension into the
// per-architecture assembly paths; nothing existing changes until then.
package asm
import (
"fmt"
"slices"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
)
// extensionIndex is the per-architecture index of the extension layer, keyed
// by upper-case mnemonic. One mnemonic registers several forms (the SVE ADD
// carries unpredicated, predicated and immediate shapes), so the value is the
// full candidate list in table order.
type extensionIndex struct {
byName map[string][]arch.ExtInstr
}
// extensionIndexes builds one index per known architecture. Architectures
// whose extension layer is not built yet get an empty index, which keeps the
// queries answering false rather than failing on a missing entry.
var extensionIndexes = buildExtensionIndexes()
func buildExtensionIndexes() map[arch.Arch]*extensionIndex {
m := make(map[arch.Arch]*extensionIndex)
for _, a := range []arch.Arch{arch.AMD64, arch.ARM64, arch.RISCV, arch.LOONG64} {
idx := &extensionIndex{byName: make(map[string][]arch.ExtInstr)}
for _, in := range arch.Extensions(a) {
key := strings.ToUpper(in.Name)
idx.byName[key] = append(idx.byName[key], in)
}
m[a] = idx
}
return m
}
// LookupExtension returns the extended instructions registered for the
// mnemonic on a, outside the generated architecture table. It reports false
// when a carries no extended layer or the mnemonic is not in it; a mnemonic
// the base table knows is not thereby covered, the layers stay independent.
func LookupExtension(a arch.Arch, mnemonic string) ([]arch.ExtInstr, bool) {
idx, ok := extensionIndexes[a]
if !ok || idx == nil {
return nil, false
}
cands, ok := idx.byName[strings.ToUpper(mnemonic)]
return cands, ok && len(cands) > 0
}
// ExtensionNames returns the mnemonics the extension layer of a registers,
// in table order, without duplicates.
func ExtensionNames(a arch.Arch) []string {
var names []string
seen := make(map[string]bool)
for _, in := range arch.Extensions(a) {
key := strings.ToUpper(in.Name)
if !seen[key] {
seen[key] = true
names = append(names, in.Name)
}
}
return names
}
// EncodeExtension encodes one extended instruction on a: it resolves the
// mnemonic through the extension registry, picks the registered form whose
// arity matches the operands and encodes against it. The first form that
// encodes wins. When every matching form rejects the operands, the error
// comes from the form whose operand kinds the list points at (the one with
// the most matching positions), so a mis-spelled predicate qualifier is
// diagnosed as one, not as the unpredicated form's register complaint.
func EncodeExtension(a arch.Arch, mnemonic string, ops ...arch.ExtOperand) ([]byte, error) {
cands, ok := LookupExtension(a, mnemonic)
if !ok {
return nil, fmt.Errorf("%s registers no extended instruction %q", a, mnemonic)
}
var bestErr error
var bestScore int
var tried int
for _, in := range cands {
if in.Form.Arity() != len(ops) {
continue
}
tried++
b, err := in.Encode(ops)
if err == nil {
return b, nil
}
if score := kindScore(in.Form, ops); bestErr == nil || score > bestScore {
bestErr, bestScore = err, score
}
}
if tried == 0 {
return nil, fmt.Errorf("%s: extended %q takes %s, got %d operands",
a, mnemonic, extensionAritySummary(cands), len(ops))
}
return nil, bestErr
}
// kindScore counts the positions whose operand kind matches what the form
// wants, the tie-break that picks the most specific rejection.
func kindScore(form arch.ExtForm, ops []arch.ExtOperand) int {
kinds := form.Kinds()
score := 0
for i, op := range ops {
if i < len(kinds) && op.Kind == kinds[i] {
score++
}
}
return score
}
// ExtensionEncodable reports whether the extension layer of a encodes the
// mnemonic with these operands. It mirrors asm.Encodable for the extension
// layer: the predicate the linter consults once the hook wires it in.
func ExtensionEncodable(a arch.Arch, mnemonic string, ops ...arch.ExtOperand) bool {
_, err := EncodeExtension(a, mnemonic, ops...)
return err == nil
}
// extensionAritySummary describes the operand counts the candidate forms
// take, "2 or 3" style, for the arity error.
func extensionAritySummary(cands []arch.ExtInstr) string {
counts := make([]int, 0, len(cands))
seen := make(map[int]bool)
for _, in := range cands {
n := in.Form.Arity()
if !seen[n] {
seen[n] = true
counts = append(counts, n)
}
}
slices.Sort(counts)
var b strings.Builder
for i, n := range counts {
if i > 0 {
if i == len(counts)-1 {
b.WriteString(" or ")
} else {
b.WriteString(", ")
}
}
fmt.Fprintf(&b, "%d", n)
}
b.WriteString(" operands")
return b.String()
}
+197
View File
@@ -0,0 +1,197 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/hex"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
)
// TestExtensionRegistryARM64 checks the mnemonic lookup over and above the
// generated arm64 table: one mnemonic, several forms, case-insensitive, and
// nothing offered for a spelling the layer does not carry.
func TestExtensionRegistryARM64(t *testing.T) {
add, ok := LookupExtension(arch.ARM64, "ADD")
if !ok {
t.Fatal("LookupExtension(ARM64, ADD) found nothing")
}
var forms []arch.ExtForm
for _, in := range add {
if in.Name != "ADD" {
t.Errorf("candidate %q leaked into the ADD lookup", in.Name)
}
forms = append(forms, in.Form)
}
if len(forms) != 3 ||
forms[0] != arch.ExtFormVectors ||
forms[1] != arch.ExtFormPredicated ||
forms[2] != arch.ExtFormImmediate {
t.Errorf("ADD registers forms %v, want unpredicated, predicated and immediate", forms)
}
if _, ok := LookupExtension(arch.ARM64, "add"); !ok {
t.Error("the lookup is case-sensitive")
}
if _, ok := LookupExtension(arch.ARM64, "NOSUCHINSTR"); ok {
t.Error("a non-extended mnemonic resolved")
}
sqadd, ok := LookupExtension(arch.ARM64, "SQADD")
if !ok || len(sqadd) != 2 {
t.Errorf("SQADD registers %d forms, want the unpredicated and immediate pair", len(sqadd))
}
}
// TestExtensionAboveGeneratedTable pins the layering: SQADD is nowhere in the
// generated arm64 table (the toolchain knows only the NEON spelling VSQADD)
// yet the extension layer carries it, while ADD sits in both layers
// independently.
func TestExtensionAboveGeneratedTable(t *testing.T) {
if _, found := arch.ForArch(arch.ARM64).Lookup("SQADD"); found {
t.Error("SQADD is in the generated table, the layering assumption broke")
}
if _, ok := LookupExtension(arch.ARM64, "SQADD"); !ok {
t.Error("SQADD is missing from the extension layer")
}
if _, found := arch.ForArch(arch.ARM64).Lookup("ADD"); !found {
t.Error("ADD vanished from the generated table")
}
if add, ok := LookupExtension(arch.ARM64, "ADD"); !ok || len(add) != 3 {
t.Errorf("ADD carries %d extension forms, want 3", len(add))
}
}
// TestEncodeExtensionGolden encodes through the registry and pins the same
// golden words the arch table tests pin, proving the registry resolves to the
// right encoding.
func TestEncodeExtensionGolden(t *testing.T) {
for _, tt := range []struct {
name string
mnem string
ops []arch.ExtOperand
want uint32
}{
{"unpredicated add", "ADD",
[]arch.ExtOperand{
arch.ExtVector(2, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB),
},
0x04200040},
{"predicated mul", "MUL",
[]arch.ExtOperand{
arch.ExtVector(0, arch.ExtArrB), arch.ExtPredicate(2, arch.ExtQualMerging), arch.ExtVector(0, arch.ExtArrB),
},
0x04100800},
{"immediate add with derived shift", "ADD",
[]arch.ExtOperand{arch.ExtImmediate(32512), arch.ExtVector(0, arch.ExtArrH)},
0x2560efe0},
{"signed immediate mul", "MUL",
[]arch.ExtOperand{arch.ExtImmediate(-1), arch.ExtVector(0, arch.ExtArrB)},
0x2530dfe0},
} {
got, err := EncodeExtension(arch.ARM64, tt.mnem, tt.ops...)
if err != nil {
t.Errorf("%s: encode: %v", tt.name, err)
continue
}
if want := hex.EncodeToString([]byte{
byte(tt.want), byte(tt.want >> 8), byte(tt.want >> 16), byte(tt.want >> 24),
}); hex.EncodeToString(got) != want {
t.Errorf("%s:\n got %x\n want %s", tt.name, got, want)
}
}
}
// TestEncodeExtensionErrors checks the registry's diagnostics: a wrong arity
// names every form's count, an operand the first candidate rejects surfaces
// its own message once a later form takes over.
func TestEncodeExtensionErrors(t *testing.T) {
if _, err := EncodeExtension(arch.ARM64, "ADD", arch.ExtVector(0, arch.ExtArrB)); err == nil {
t.Error("one operand encoded, want an arity error")
} else if !strings.Contains(err.Error(), "2 or 3 operands") {
t.Errorf("arity error %q does not name the counts", err)
}
// The predicated candidate must answer for its own operands: the /Z
// qualifier is rejected with the merging message, not the unpredicated
// form's register-kind complaint.
_, err := EncodeExtension(arch.ARM64, "ADD",
arch.ExtVector(0, arch.ExtArrB), arch.ExtPredicate(0, arch.ExtQualZeroing), arch.ExtVector(0, arch.ExtArrB))
if err == nil {
t.Fatal("/Z encoded, want an error")
}
if !strings.Contains(err.Error(), "/M") {
t.Errorf("error %q does not name the merging qualifier", err)
}
if _, err := EncodeExtension(arch.ARM64, "NOSUCHINSTR", arch.ExtVector(0, arch.ExtArrB)); err == nil ||
!strings.Contains(err.Error(), "registers no extended instruction") {
t.Errorf("unknown mnemonic error = %v", err)
}
}
// TestExtensionEncodable checks the predicate the later Encodable hook will
// call: true exactly when the registry encodes the operand list.
func TestExtensionEncodable(t *testing.T) {
if !ExtensionEncodable(arch.ARM64, "ADD",
arch.ExtVector(0, arch.ExtArrS), arch.ExtVector(1, arch.ExtArrS), arch.ExtVector(2, arch.ExtArrS)) {
t.Error("an encodable unpredicated add reported false")
}
if !ExtensionEncodable(arch.ARM64, "ADD",
arch.ExtVector(1, arch.ExtArrS), arch.ExtPredicate(0, arch.ExtQualMerging), arch.ExtVector(0, arch.ExtArrS)) {
t.Error("an encodable predicated add reported false")
}
if ExtensionEncodable(arch.ARM64, "ADD",
arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrS), arch.ExtVector(0, arch.ExtArrB)) {
t.Error("mismatched arrangements reported encodable")
}
if ExtensionEncodable(arch.ARM64, "ADD", arch.ExtVector(0, arch.ExtArrB)) {
t.Error("a one-operand add reported encodable")
}
if ExtensionEncodable(arch.ARM64, "NOSUCHINSTR") {
t.Error("an unregistered mnemonic reported encodable")
}
}
// TestExtensionArchIsolation is the architecture-binding negative case: the
// extension layer is registered for arm64 alone, and no other architecture
// answers its queries, not even for a mnemonic the amd64 base table carries.
func TestExtensionArchIsolation(t *testing.T) {
ops := []arch.ExtOperand{
arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB),
}
for _, a := range []arch.Arch{arch.AMD64, arch.RISCV, arch.LOONG64, arch.Unknown} {
if cands, ok := LookupExtension(a, "ADD"); ok || cands != nil {
t.Errorf("LookupExtension(%s, ADD) offered %d candidates", a, len(cands))
}
if cands, ok := LookupExtension(a, "MUL"); ok || cands != nil {
t.Errorf("LookupExtension(%s, MUL) offered %d candidates", a, len(cands))
}
if got, err := EncodeExtension(a, "ADD", ops...); err == nil {
t.Errorf("EncodeExtension(%s, ADD) encoded %x, want a refusal", a, got)
} else if !strings.Contains(err.Error(), string(a)) {
t.Errorf("EncodeExtension(%s) error %q does not name the architecture", a, err)
}
if ExtensionEncodable(a, "ADD", ops...) {
t.Errorf("ExtensionEncodable(%s, ADD) reported true", a)
}
if names := ExtensionNames(a); len(names) != 0 {
t.Errorf("ExtensionNames(%s) = %v, want none", a, names)
}
if got := arch.Extensions(a); len(got) != 0 {
t.Errorf("arch.Extensions(%s) carries %d instructions", a, len(got))
}
}
}
// TestExtensionNamesARM64 checks the completion-facing name list: every
// distinct mnemonic of the family, first-occurrence order, no duplicates.
func TestExtensionNamesARM64(t *testing.T) {
want := []string{"ADD", "SUB", "SQADD", "UQADD", "SQSUB", "UQSUB", "MUL", "SMULH", "UMULH", "SUBR"}
got := ExtensionNames(arch.ARM64)
if strings.Join(got, ",") != strings.Join(want, ",") {
t.Errorf("ExtensionNames(ARM64) = %v, want %v", got, want)
}
if n := len(arch.Extensions(arch.ARM64)); n != 23 {
t.Errorf("the family registers %d instructions, want 23", n)
}
}
+8
View File
@@ -63,6 +63,7 @@ const (
const (
kindSTEXT = 1
kindSRODATA = 3
kindSNOPTRDATA = 5
kindSDATA = 7
kindSDWARFFCN = 14
kindSDWARFLINES = 20
@@ -305,9 +306,16 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
if !d.Static {
name = pkgPath + "." + name
}
// RODATA implies no pointers, so it wins over NOPTR: the kind is
// SRODATA either way, exactly as the toolchain chooses it. Plain
// NOPTR data is SNOPTRDATA, which the linker keeps out of the GC's
// type scan; a plain SDATA symbol would demand Go type information
// no assembly file can supply, and the link would fail.
typ := uint8(kindSDATA)
if d.Rodata {
typ = kindSRODATA
} else if d.Noptr {
typ = kindSNOPTRDATA
}
flag := uint8(0)
if d.Dupok {
+3
View File
@@ -45,6 +45,9 @@ func TestReadRuntimeSymbols(t *testing.T) {
// TestResolveExternalSymbols verifies end-to-end resolution of external
// symbol references.
func TestResolveExternalSymbols(t *testing.T) {
if testing.Short() {
t.Skip("resolves through a live go list -export: skipped in -short mode")
}
if _, err := exec.LookPath("go"); err != nil {
t.Skip("go toolchain not available")
}
+143 -1
View File
@@ -12,7 +12,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// goobjView is a minimal parsed view of a GOOBJ payload, enough to check
@@ -514,6 +514,145 @@ func fieldAfter(line, flag string) string {
return ""
}
// buildLogSteps is what a substitution test needs from a `go build -x -work`
// log: the work directory, the assembler's object, the package archive and
// the link command line.
type buildLogSteps struct {
work string
asmObj string // $WORK expanded
pkgArch string // $WORK expanded
linkLine string // still carries $WORK placeholders
}
// parseBuildLog extracts the build steps from a `go build -x -work` log.
// asmFile names the assembly file whose object the test substitutes. A
// missing step is a failure, not a skip: the toolchain changed shape and the
// substitution would silently test nothing.
func parseBuildLog(t *testing.T, log []byte, asmFile string) buildLogSteps {
t.Helper()
var st buildLogSteps
for line := range strings.SplitSeq(string(log), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
st.work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, asmFile) && !strings.Contains(line, "-gensymabis"):
st.asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
rest := strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
st.pkgArch = strings.Fields(strings.SplitN(rest, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
st.linkLine = line
}
}
if st.work == "" || st.asmObj == "" || st.pkgArch == "" || st.linkLine == "" {
t.Fatalf("could not locate the build steps (work=%q asmObj=%q pkgArch=%q link=%q):\n%s",
st.work, st.asmObj, st.pkgArch, st.linkLine, log)
}
st.asmObj = strings.ReplaceAll(st.asmObj, "$WORK", st.work)
st.pkgArch = strings.ReplaceAll(st.pkgArch, "$WORK", st.work)
return st
}
// substituteAndRelink swaps the gasm object into the package archive the
// baseline build produced and re-runs the captured link line against the
// rebuilt archive, writing the binary to outBin (the -x log's link step
// always targets the action graph's internal a.out, which the helper
// redirects; the copy to the -o target is a separate build action the helper
// does not need). The archive handed to the linker is proven to carry the
// gasm object byte for byte, so a build-layout change that skipped the
// substitution fails here instead of passing vacuously.
func substituteAndRelink(t *testing.T, goBin, dir string, st buildLogSteps, outBin string, gasmObj []byte, extraEnv ...string) {
t.Helper()
// The deliberate-run boundary: this path drives a real `go build` and
// cmd/link per invocation, minutes-scale work on the small single-core
// CI runner. Under -short (the push pipeline's mode) it skips; the
// local test gate and the dispatched workflows run it in full.
if testing.Short() {
t.Skip("end-to-end go build and link: skipped in -short mode")
}
// Extract the archive, overwrite the assembler's member with the gasm
// object and repack (go tool pack has no replace-in-place).
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract := exec.Command(goBin, "tool", "pack", "x", st.pkgArch)
extract.Dir = membersDir
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
member := filepath.Join(membersDir, filepath.Base(st.asmObj))
if _, err := os.Stat(member); err != nil {
t.Fatalf("the assembler's archive member was not extracted: %v", err)
}
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, gasmObj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", st.pkgArch)
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for m := range strings.FieldsSeq(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, m)
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
// Prove the substitution: the archive the linker is about to consume
// holds the gasm object, byte for byte.
checkDir := filepath.Join(dir, "check")
if err := os.MkdirAll(checkDir, 0o755); err != nil {
t.Fatal(err)
}
check := exec.Command(goBin, "tool", "pack", "x", newArch)
check.Dir = checkDir
if out, err := check.CombinedOutput(); err != nil {
t.Fatalf("pack x (verification): %v\n%s", err, out)
}
got, err := os.ReadFile(filepath.Join(checkDir, filepath.Base(st.asmObj)))
if err != nil {
t.Fatalf("read the substituted member back: %v", err)
}
if !bytes.Equal(got, gasmObj) {
t.Fatal("the repacked archive does not carry the gasm object")
}
// Re-link. The line carries a GOROOT assignment and $WORK placeholders;
// GOEXPERIMENT must match the toolchain's own, because the linker
// compares the object header against its configuration.
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
linkLine := strings.ReplaceAll(st.linkLine, "$WORK", st.work)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(st.work, "b001", "_pkg_.a"), newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(st.work, "b001", "exe", "a.out"), outBin)
env := append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)))
env = append(env, extraEnv...)
link := exec.Command("sh", "-c", linkLine)
link.Dir = dir
link.Env = env
if out, err := link.CombinedOutput(); err != nil {
t.Fatalf("link with the gasm object: %v\n%s", err, out)
}
}
// TestGOObjectExternalPackageLink is the cross-package end-to-end check: a
// GOOBJ whose code references a real external package symbol (runtime's
// morestack, a plain reference rather than the builtin noctxt form) must
@@ -524,6 +663,9 @@ func fieldAfter(line, flag string) string {
// failed. The binary is not run: morestack returns to the call site's
// stack check, which a hand-written caller has none of.
func TestGOObjectExternalPackageLink(t *testing.T) {
if testing.Short() {
t.Skip("end-to-end go build and link: skipped in -short mode")
}
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
+6 -6
View File
@@ -9,8 +9,8 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64,
@@ -307,12 +307,12 @@ func TestStackGuardBytesLOONG64(t *testing.T) {
func TestStackGuardGOObjInternalCall(t *testing.T) {
for _, tt := range []struct {
src string
assemble func(*ast.File) (*Image, error)
assemble func(*ast.File, ...AssembleOption) (*Image, error)
}{
{"g_amd64.s", AssembleFile},
{"g_arm64.s", AssembleFileARM64},
{"g_riscv64.s", AssembleFileRISCV},
{"g_loong64.s", AssembleFileLOONG64},
{"g_arm64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileARM64(f) }},
{"g_riscv64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileRISCV(f) }},
{"g_loong64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileLOONG64(f) }},
} {
f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
+259 -13
View File
@@ -90,6 +90,40 @@ var noOperandTable = map[string][]byte{
"LOCK": {0xF0},
"REP": {0xF3},
"REPN": {0xF2},
"ENDBR64": {0xF3, 0x0F, 0x1E, 0xFA},
}
// sysUnaryTable maps the one-operand system instructions to their bytes:
// the prefix, the opcode and the /digit the reg field carries. The operand
// is a register or memory in r/m.
var sysUnaryTable = map[string]struct {
prefix byte
opcode []byte
digit int
}{
"CLWB": {0x66, []byte{0x0F, 0xAE}, 6},
"TPAUSE": {0x66, []byte{0x0F, 0xAE}, 6},
"UMONITOR": {0xF3, []byte{0x0F, 0xAE}, 6},
"UMWAIT": {0xF2, []byte{0x0F, 0xAE}, 6},
"RDPID": {0xF3, []byte{0x0F, 0xC7}, 7},
"CLDEMOTE": {0x00, []byte{0x0F, 0x1C}, 0},
}
// encodeSysUnary emits a one-operand system instruction: the operand in r/m
// under the fixed /digit, no REX.W.
func (e *enc) encodeSysUnary(mnem string, m struct {
prefix byte
opcode []byte
digit int
}, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
}
i := &instr{prefix: m.prefix, opcode: m.opcode, modrm: -1, sib: -1}
if err := setRMDigit(i, m.digit, ops[0], 8); err != nil {
return err
}
return e.emit(i)
}
// --- MOV --------------------------------------------------------------------
@@ -111,6 +145,76 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
// silently emit REX.W 8B with the wrong operand meaning.
_, srcVec := vecReg(src)
dstReg, dstVec := vecReg(dst)
// Control and debug register moves: 0F 20 (CRn→r64), 0F 22 (r64→CRn),
// 0F 21 (DRn→r64) and 0F 23 (r64→DRn). The CR/DR number rides the reg
// field, the general register r/m; CR8+/DR8+ take REX.R.
if c, ok := src.(Reg); ok && c.ctl != 0 {
g, ok := dst.(Reg)
if !ok || g.isVec() || g.ctl != 0 {
return fmt.Errorf("MOV: control/debug register load needs a general register destination")
}
opc := byte(0x20)
if c.ctl == 2 {
opc = 0x21
}
return e.emit(&instr{
opcode: []byte{0x0F, opc},
modrm: 0xC0 | (c.idx&7)<<3 | (g.idx & 7),
sib: -1, rexR: c.idx >= 8, rexB: g.idx >= 8,
})
}
if c, ok := dst.(Reg); ok && c.ctl != 0 {
g, ok := src.(Reg)
if !ok || g.isVec() || g.ctl != 0 {
return fmt.Errorf("MOV: control/debug register store needs a general register source")
}
opc := byte(0x22)
if c.ctl == 2 {
opc = 0x23
}
return e.emit(&instr{
opcode: []byte{0x0F, opc},
modrm: 0xC0 | (c.idx&7)<<3 | (g.idx & 7),
sib: -1, rexR: c.idx >= 8, rexB: g.idx >= 8,
})
}
// MMX register moves: MOVQ M0, mem and MOVQ mem, M0 are the MMX
// load/store pair 0F 6F/0F 7F (no prefix); a register pair takes the
// load opcode. The XMM MOVQ forms follow below.
if m, ok := src.(Reg); ok && m.mmx {
switch d := dst.(type) {
case Reg:
if !d.mmx {
return fmt.Errorf("MOV: MMX register moves stay inside the M bank")
}
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
if err := setRM(i, d, src, 8); err != nil {
return err
}
return e.emit(i)
case Mem:
i := &instr{opcode: []byte{0x0F, 0x7F}, modrm: -1, sib: -1}
if err := setRM(i, m, d, 8); err != nil {
return err
}
return e.emit(i)
}
return fmt.Errorf("MOV: invalid MMX destination")
}
if m, ok := dst.(Reg); ok && m.mmx {
srcM, ok := src.(Mem)
if !ok {
return fmt.Errorf("MOV: MMX load takes a memory source")
}
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
if err := setRM(i, m, srcM, 8); err != nil {
return err
}
return e.emit(i)
}
if srcVec || dstVec {
if dstVec {
if g, ok := src.(Reg); ok && !g.isVec() {
@@ -192,6 +296,30 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
}
return e.emit(i)
case TLSMem:
if !dstIsReg {
return fmt.Errorf("MOV: two memory operands")
}
// MOV r, off(TLS): the segment-prefixed absolute load, reg=dst,
// rm=src(tlsMem) through the SIB escape; the disp32 is the TLS slot
// offset with its R_TLSLE patch site.
i := newInstr(size, []byte{movRR(size)})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
case SegAbs:
if !dstIsReg {
return fmt.Errorf("MOV: two memory operands")
}
// MOV r, 0x30(GS): the segment-absolute load.
i := newInstr(size, []byte{movRR(size)})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
case Imm:
if dstIsReg {
v := int64(src)
@@ -232,11 +360,24 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
i.imm = imm
return e.emit(i)
}
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0. An immediate in the
// destination slot is the absolute-address crash-store spelling,
// MOVL $0xf1, 0xf1: the parser reads the trailing bare constant
// as an immediate, and the store's disp32 carries the address.
op := byte(0xC7)
if size == 1 {
op = 0xC6
}
if d, ok := dst.(Imm); ok {
i := newInstr(size, []byte{op})
setSegAbs(i, 0, SegAbs{Disp: int64(d)})
immBytes, err := immediate(int64(src), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
i := newInstr(size, []byte{op})
if err := setRMDigit(i, 0, dst, size); err != nil {
return err
@@ -481,6 +622,14 @@ func (e *enc) encodeLea(ops []Operand, size int) error {
default:
return fmt.Errorf("LEA: source must be a memory operand")
}
// LEA accepts the full unsigned 32-bit displacement span where the
// loads and stores reject it beyond the signed one; the wide values
// ride the same disp32 bytes as their two's-complement bit pattern.
if m, ok := src.(Mem); ok && m.Disp >= 1<<31 && m.Disp <= (1<<32)-1 {
c := m
c.Disp = int64(int32(uint32(m.Disp)))
src = c
}
i := newInstr(size, []byte{0x8D})
if err := setRM(i, dstReg, src, size); err != nil {
return err
@@ -626,8 +775,29 @@ func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error {
func (e *enc) encodeImul(ops []Operand, size int) error {
switch len(ops) {
case 1:
// The one-operand form, IMUL r/m: F6/F7 /5 with AL/AX/EAX/RAX as the
// implied destination (the toolchain's one-register shape).
opc := byte(0xF7)
if size == 1 {
opc = 0xF6
}
i := newInstr(size, []byte{opc})
if err := setRMDigit(i, 5, ops[0], size); err != nil {
return err
}
return e.emit(i)
case 2:
// IMUL r, r/m: 0x0F 0xAF.
// Two shapes. The leading-immediate spelling IMUL $imm, r multiplies
// r in place (dst = rm = r): the shape GOROOT's clock code writes.
// Otherwise IMUL r, r/m: 0x0F 0xAF.
if imm, ok := ops[0].(Imm); ok {
dstReg, isReg := ops[1].(Reg)
if !isReg {
return fmt.Errorf("IMUL: destination must be a register")
}
return e.encodeImulImm(imm, dstReg, dstReg, size)
}
dstReg, ok := ops[1].(Reg)
if !ok {
return fmt.Errorf("IMUL: destination must be a register")
@@ -647,17 +817,26 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
if !ok {
return fmt.Errorf("IMUL: immediate operand expected first")
}
// Plan 9 order: IMUL $imm, src, dst.
// Plan 9 order: IMUL $imm, src, dst; the source stays a general
// r/m operand (setRM takes registers and memory alike).
return e.encodeImulImm(imm, ops[1], dstReg, size)
}
return fmt.Errorf("IMUL expects 1, 2 or 3 operands, got %d", len(ops))
}
// encodeImulImm emits the immediate multiply: 0x6B with a sign-extended imm8
// when the value fits, 0x69 with a 32-bit immediate otherwise.
func (e *enc) encodeImulImm(imm Imm, rm Operand, dst Reg, size int) error {
if fits8(int64(imm)) {
i := newInstr(size, []byte{0x6B})
if err := setRM(i, dstReg, ops[1], size); err != nil {
if err := setRM(i, dst, rm, size); err != nil {
return err
}
i.imm = []byte{byte(int8(imm))}
return e.emit(i)
}
i := newInstr(size, []byte{0x69})
if err := setRM(i, dstReg, ops[1], size); err != nil {
if err := setRM(i, dst, rm, size); err != nil {
return err
}
immBytes, err := immediate(int64(imm), size, false)
@@ -666,8 +845,6 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
}
i.imm = immBytes
return e.emit(i)
}
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
}
// --- PUSH / POP -------------------------------------------------------------
@@ -689,6 +866,27 @@ func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
w16 := size == 2
switch op := ops[0].(type) {
case Reg:
// Segment registers: FS and GS carry their own one-byte opcodes
// under 0F (A0/A8 push, A1/A9 pop); the other four spellings are
// not pushable in 64-bit mode.
if n, isSeg := op.segNumber(); isSeg {
switch n {
case 4: // FS
if push {
return e.emit(&instr{opcode: []byte{0x0F, 0xA0}, modrm: -1, sib: -1})
}
return e.emit(&instr{opcode: []byte{0x0F, 0xA1}, modrm: -1, sib: -1})
case 5: // GS
if push {
return e.emit(&instr{opcode: []byte{0x0F, 0xA8}, modrm: -1, sib: -1})
}
return e.emit(&instr{opcode: []byte{0x0F, 0xA9}, modrm: -1, sib: -1})
}
return fmt.Errorf("PUSH/POP: only FS and GS are encodable in 64-bit mode")
}
if op.mmx || op.isVec() || op.fp || op.ctl != 0 {
return fmt.Errorf("PUSH/POP: invalid register operand")
}
base := byte(0x50) // PUSH r; POP is 0x58
if !push {
base = 0x58
@@ -1040,6 +1238,38 @@ var sseMoveTable = map[string]sseMove{
"MOVSS": {0xF3, 0x10, 0x11}, // scalar single
}
// sseStoreOnly holds the store-only SSE forms, OP xmm, mem: the XMM register
// rides the reg field and memory r/m (the non-temporal store).
var sseStoreOnly = map[string]struct {
prefix byte
op byte
}{
"MOVNTDQ": {0x66, 0xE7},
}
// encodeSSEStoreOnly encodes OP xmm, mem (reg = the XMM source, r/m = the
// destination memory).
func (e *enc) encodeSSEStoreOnly(mnem string, m struct {
prefix byte
op byte
}, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("%s source must be a vector register", mnem)
}
if !isX86Mem(ops[1]) {
return fmt.Errorf("%s destination must be a memory operand", mnem)
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, srcReg, ops[1], 8); err != nil {
return err
}
return e.emit(i)
}
// encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the
// load form (reg = destination), matching the Go assembler.
func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
@@ -1255,14 +1485,23 @@ func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error {
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
if !ok || (!dstReg.isVec() && !dstReg.mmx) {
return fmt.Errorf("SSE binary destination must be a vector register")
}
// The MMX twins of the packed-integer SSE2 ops drop the 0x66 prefix:
// PADDD M2, M1 is 0F FE where the XMM form is 66 0F FE.
prefix := m.prefix
if dstReg.mmx {
if prefix != 0x66 {
return fmt.Errorf("SSE binary: this form takes no MMX register operand")
}
prefix = 0
}
opcode := []byte{0x0F, m.op}
if m.map38 {
opcode = []byte{0x0F, 0x38, m.op}
}
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
i := &instr{prefix: prefix, opcode: opcode, modrm: -1, sib: -1}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
@@ -1738,12 +1977,19 @@ func (e *enc) encodeSSEShift(name string, ops []Operand) error {
// immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family:
// F2 0F C2 with reg = dst, rm = src.
func (e *enc) encodeCmpsd(ops []Operand) error {
return e.encodeSSECmp("CMPSD", 0xF2, ops)
}
// encodeSSECmp encodes the SSE compare family (CMPSD/CMPSS/CMPPS/CMPPD):
// 0F C2 /r ib with the predicate immediate last in Plan 9 order
// (src, dst, $imm) and the packed forms' prefixes.
func (e *enc) encodeSSECmp(mnem string, prefix byte, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops))
return fmt.Errorf("%s expects 3 operands (src, dst, $imm), got %d", mnem, len(ops))
}
imm, ok := ops[2].(Imm)
if !ok {
return fmt.Errorf("CMPSD predicate must be an immediate")
return fmt.Errorf("%s predicate must be an immediate", mnem)
}
immByte, err := imm8(int64(imm))
if err != nil {
@@ -1751,9 +1997,9 @@ func (e *enc) encodeCmpsd(ops []Operand) error {
}
dstReg, ok2 := ops[1].(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("CMPSD destination must be a vector register")
return fmt.Errorf("%s destination must be a vector register", mnem)
}
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
i := &instr{prefix: prefix, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return err
}
+3 -3
View File
@@ -16,7 +16,7 @@ import (
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel;
@@ -25,7 +25,7 @@ import (
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx2_amd64.s"
if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit")
t.Skip("go-libraries repository not present next to gasm-sdk")
}
src, err := os.ReadFile(path)
if err != nil {
@@ -86,7 +86,7 @@ func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx512_amd64.s"
if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit")
t.Skip("go-libraries repository not present next to gasm-sdk")
}
src, err := os.ReadFile(path)
if err != nil {
+12 -2
View File
@@ -13,7 +13,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// The differential kernels for the DATA-path and front-end gaps are kept in
@@ -23,9 +23,18 @@ import (
// bytes must agree with the relocation sites masked on both sides.
// toolAsmObject assembles path with the installed toolchain's assembler for
// goarch ("" = the host) and returns the object bytes.
// goarch ("" = the host) and returns the object bytes. Every live-oracle
// comparison funnels through here, so this is also where the deliberate-run
// boundary sits: under -short (the push pipeline's mode) the comparisons
// skip, because each spawns a go tool asm subprocess and the small single-
// core runner pays seconds per spawn. The encodings stay pinned by the
// golden-byte tests in every mode; the live oracle runs in the local test
// gate and the dispatched workflows.
func toolAsmObject(t *testing.T, path, goarch string) []byte {
t.Helper()
if testing.Short() {
t.Skip("live go tool asm oracle: skipped in -short mode")
}
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
@@ -130,6 +139,7 @@ func TestDifferentialKernels(t *testing.T) {
{filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "forms_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true},
{filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true},
} {
+1 -1
View File
@@ -12,7 +12,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
+37 -5
View File
@@ -9,7 +9,7 @@ import (
"sort"
"strconv"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// Image is an assembled file: the function bodies laid out in source order,
@@ -134,7 +134,8 @@ type DataSymbol struct {
Offset int // byte offset within Data
Size int
Static bool // the <> marker: file-local, not exported
Rodata bool // the RODATA flag: read-only data
Rodata bool // the RODATA flag: read-only data (implies no pointers)
Noptr bool // the NOPTR flag: data with no pointers, kept out of GC scanning
Dupok bool // the DUPOK flag: duplicate-OK
// Relocs carries the symbol-valued DATA initialisers ("DATA s+0(SB)/8,
// $other(SB)"): fields of this symbol's data that hold another symbol's
@@ -150,6 +151,18 @@ func (img *Image) Bytes() []byte {
return append(out, img.Data...)
}
// AssembleOption adjusts the file-level assembly context.
type AssembleOption func(*linkInfo)
// WithGOOS selects the target operating system for the forms that depend on
// it, the TLS access shape above all: linux and freebsd take the
// one-instruction form, windows and plan9 keep the two-instruction load.
func WithGOOS(goos string) AssembleOption {
return func(l *linkInfo) {
l.goos = goos
}
}
// AssembleFile assembles every TEXT function of a parsed file and lays out
// its static symbols (GLOBL/DATA) in a data section behind the code. Each
// reference to a file-local static symbol becomes a RIP-relative load whose
@@ -157,7 +170,7 @@ func (img *Image) Bytes() []byte {
// GLOBL defines is recorded as an external relocation (Externals) with its
// displacement left zero, the object-file emitters resolve it at link
// time, while the raw image (Bytes) cannot represent it.
func AssembleFile(f *ast.File) (*Image, error) {
func AssembleFile(f *ast.File, opts ...AssembleOption) (*Image, error) {
dataSyms, err := collectData(f)
if err != nil {
return nil, err
@@ -166,7 +179,17 @@ func AssembleFile(f *ast.File) (*Image, error) {
for _, d := range dataSyms {
known[d.name] = true
}
// TEXT symbols are file-level definitions too: a symbol immediate
// ($fn(SB)) may name one, exactly as a data reference names a GLOBL.
for _, d := range f.Decls {
if t, ok := d.(*ast.Text); ok {
known[t.Name.Name] = true
}
}
link := &linkInfo{symbols: known, allowExternal: true}
for _, o := range opts {
o(link)
}
poolSeen := map[string]bool{}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
@@ -247,6 +270,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
Size: len(d.buf),
Static: d.static,
Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok,
})
img.Data = append(img.Data, d.buf...)
@@ -391,6 +415,7 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
Size: d.size,
Static: d.static,
Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok,
})
}
@@ -463,6 +488,7 @@ func AssembleFileLOONG64(f *ast.File) (*Image, error) {
Size: d.size,
Static: d.static,
Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok,
})
}
@@ -524,6 +550,7 @@ type dataSym struct {
size int
static bool
rodata bool
noptr bool
dupok bool
// relocs are the symbol-valued DATA fields, in declaration order; Off
// is relative to the symbol's data start.
@@ -565,12 +592,14 @@ func collectData(f *ast.File) ([]dataSym, error) {
switch f {
case "RODATA":
ds.rodata = true
case "NOPTR":
ds.noptr = true
case "DUPOK":
ds.dupok = true
default:
// Legacy numeric flag constants (runtime/textflag.h):
// DUPOK is 2, RODATA is 8; combinations arrive as one
// number (e.g. 10 = RODATA|DUPOK).
// DUPOK is 2, RODATA is 8, NOPTR is 16; combinations arrive
// as one number (e.g. 10 = RODATA|DUPOK).
if n, err := strconv.Atoi(f); err == nil {
if n&2 != 0 {
ds.dupok = true
@@ -578,6 +607,9 @@ func collectData(f *ast.File) ([]dataSym, error) {
if n&8 != 0 {
ds.rodata = true
}
if n&16 != 0 {
ds.noptr = true
}
}
}
}
+12 -38
View File
@@ -12,7 +12,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestAssembleFileStaticData checks the whole-image layout; code, padding
@@ -60,23 +60,15 @@ DATA small<>+0(SB)/4, $0x1234
}
}
// TestAssembleFileErrors checks the static-symbol error paths.
// TestAssembleFileErrors checks the static-symbol error paths. A reference
// to a static symbol no GLOBL defines defers to the linker exactly as the
// toolchain does (an external relocation), so it is not an error here.
func TestAssembleFileErrors(t *testing.T) {
cases := []struct {
name string
src string
want string // substring of the error
}{
{
"undefined symbol",
`
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
VMOVDQU nope<>(SB), X0
RET
`,
"undefined symbol",
},
{
"DATA without GLOBL",
`
@@ -369,23 +361,8 @@ func main() {
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || linkLine == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
}
defer os.RemoveAll(work)
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
st := parseBuildLog(t, buildLog, "main_amd64.s")
defer os.RemoveAll(st.work)
// Assemble the same source with gasm and substitute the object.
src, err := os.ReadFile(filepath.Join(dir, "main_amd64.s"))
@@ -400,21 +377,18 @@ func main() {
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
gasmObj, err := img.GOObject("dlink", "main_amd64.s")
// The package path is "main": the linker resolves the Go code's
// references against main.<name>, so the object must define the symbols
// under that prefix whatever the module is called.
gasmObj, err := img.GOObject("main", "main_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"), gasmObj)
// The linked program must run and find the right function behind the
// data word.
out, err := exec.Command(filepath.Join(dir, "prog")).CombinedOutput()
out, err := exec.Command(filepath.Join(dir, "prog2")).CombinedOutput()
if err != nil {
t.Fatalf("linked program failed: %v\n%s", err, out)
}
+258 -7
View File
@@ -9,7 +9,7 @@ import (
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into
@@ -371,6 +371,13 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
if mnem == "RET" {
return len(loong64Return(fi))
}
// BYTE lays down one raw byte per operand, a front-end pseudo-op the
// toolchain spells only on x86 but accepts here the same way the arm64
// and riscv64 encoders do (a superset spelling, shippable via the goobj
// path).
if mnem == "BYTE" {
return len(ops)
}
switch mnem {
case "END", "FUNCDATA", "PCDATA":
return 0 // bookkeeping statements contribute no bytes
@@ -484,6 +491,18 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "BYTE":
// BYTE $b lays down one raw byte per operand, the same front-end
// pseudo-op the arm64 and riscv64 encoders accept.
var out []byte
for _, op := range ops {
b := l64Imm64(op)
if b < 0 || b > 0xFF {
return nil, fmt.Errorf("BYTE: immediate %d does not fit a byte", b)
}
out = append(out, byte(b))
}
return out, nil
case "END", "FUNCDATA", "PCDATA", "GETCALLERPC":
// The assembler's bookkeeping statements. END, FUNCDATA and PCDATA
// contribute no bytes, the same shapes GOARCH=loong64 go tool asm
@@ -701,6 +720,35 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
}
return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Fllsc:
// LLACQ{W,V} (Rj), Rd loads and SCREL{W,V} Rd, (Rj) stores, both
// 2R encodings op | rj<<5 | rd against a zero-offset memory operand
// (the toolchain's C_ZOREG, which rejects any displacement).
rd, rj, off, _, err := l64MemOperands(ops, fi)
if err != nil {
return nil, fmt.Errorf("%s: %w", mnem, err)
}
if off != 0 {
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
}
return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Fscq:
// SCQ first, middle, (base): op | middle<<10 | base<<5 | first,
// against a zero-offset memory operand as with the LL/SC pair.
if len(ops) != 3 || !isMemOperand(ops[2]) || isMemOperand(ops[0]) || isMemOperand(ops[1]) {
return nil, fmt.Errorf("%s expects reg, reg, (reg)", mnem)
}
first, middle := l64Reg(ops[0]), l64Reg(ops[1])
rj, off := l64MemWithFrame(ops[2], fi)
if first < 0 || middle < 0 || rj < 0 {
return nil, fmt.Errorf("%s: invalid register operand", mnem)
}
if off != 0 {
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
}
return l64wordLE(l64rrr(enc.op, middle, rj, first)), nil
case l64Firr:
// LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd.
if len(ops) < 2 || !isImmOperand(ops[0]) {
@@ -1229,6 +1277,31 @@ func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off in
return rd, rj, off, load, nil
}
// l64ImmMem reads the `$off(rj)` immediate form off an operand's raw text:
// the shared immediate parse reduces it to the bare number and keeps only
// the text as a witness of the base register. ok reports the form was
// found, with the base's register number (or -1 when the name is not a
// general register).
func l64ImmMem(op *ast.Operand) (off int32, base int, ok bool) {
if op.Kind != ast.OpImmediate || !op.Imm.HasVal {
return 0, 0, false
}
raw := strings.ReplaceAll(op.Raw, " ", "")
if !strings.HasPrefix(raw, "$") || !strings.HasSuffix(raw, ")") {
return 0, 0, false
}
open := strings.LastIndexByte(raw, '(')
if open < 2 {
return 0, 0, false
}
base = loong64RegNum(raw[open+1 : len(raw)-1])
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return int32(v), base, base >= 0
}
// ---- the MOV pseudo-instruction ----
// encodeLOONG64Mov encodes the MOV family, the load/store/immediate
@@ -1262,6 +1335,29 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs
}
return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil
}
// MOVx $off(rj), rd computes an address: the toolchain's `mov
// $soreg, r` case, a plain addi.d whatever the move's width (both
// MOVW and MOVV $4(R4), R5 encode the same addi.d in its testdata).
// A wider offset materialises in R30 first (lu12i.w + ori + add.d,
// its case 10). The immediate's Raw carries the base register,
// which the shared immediate parse reduces to the bare number.
if off, base, ok := l64ImmMem(src); ok {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $imm(rj): invalid destination register", mnem)
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return nil, fmt.Errorf("%s $imm(rj): illegal combination with an F register destination", mnem)
}
if off >= -2048 && off <= 2047 {
return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(off), base, rd)), nil
}
return l64WordsLE(
l64ir(l64InstrTable["LU12IW"].op, int(off)>>12, 30),
l64irr(l64DualTable["OR"].imm, int(off)&0xFFF, 30, 30),
l64rrr(l64DualTable["ADDV"].rrr, 30, base, rd),
), nil
}
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
@@ -1356,6 +1452,14 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
return 8 // pcalau12i + addi.d
}
// The $off(rj) address immediate: addi.d in the 12-bit window,
// lu12i.w + ori + add.d beyond it (the toolchain's case 10).
if off, _, ok := l64ImmMem(src); ok {
if off >= -2048 && off <= 2047 {
return 4
}
return 12
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid)
}
@@ -1868,6 +1972,19 @@ func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
return l64Mem(op)
}
// l64VmovqMem resolves a VMOVQ/XVMOVQ memory operand. The toolchain's
// vector table falls back to the zero register as the FP-relative base
// (`VMOVQ V2, y+16(FP)` stores through R0 while MOVW reads the same operand
// through R3), so the vector moves keep the resolved offset but the zero
// base, exactly as `go tool asm` emits them.
func l64VmovqMem(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
rj, off = l64MemWithFrame(op, fi)
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" {
rj = 0
}
return rj, off
}
// l64MemOffset returns the resolved byte offset of a memory operand.
func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 {
_, off := l64MemWithFrame(op, fi)
@@ -1892,7 +2009,7 @@ func l64Label(op *ast.Operand) string {
type l64VecOperand struct {
num int // 5-bit register number
lasx bool // X bank (LASX) rather than V (LSX)
width byte // suffix width letter (B/H/W/V), 0 on a bare register
width byte // suffix width letter (B/H/W/V/Q), 0 on a bare register
lanes int // lane count of a width suffix (B16 → 16)
elem int // element index of a .T[i] suffix
hasEl bool // the suffix names an element (.T[i])
@@ -1932,7 +2049,7 @@ func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) {
}
i++
w := name[i]
if w != 'B' && w != 'H' && w != 'W' && w != 'V' {
if w != 'B' && w != 'H' && w != 'W' && w != 'V' && w != 'Q' {
return v, false
}
v.width, v.hasSuf = w, true
@@ -2174,6 +2291,10 @@ func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]
// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register)
// VMOVQ vj.T[i], rd vpickve2gr (extract one element)
// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element)
// VMOVQ vj.T[i], vd.T vreplvei (broadcast one element, LSX)
// XVMOVQ xj, xd.T xvreplve0 (broadcast element zero, LASX)
// XVMOVQ xj, xd.T[i] xvinsve0 (insert element zero, LASX)
// XVMOVQ xj.T[i], xd xvpickve (extract one element, LASX)
func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) {
enc := l64VmovqTable[lasx]
bank := "V"
@@ -2200,6 +2321,118 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
return loong64RegNum(name), nil
}
// Element broadcast: VMOVQ vj.T[i], vd.T (vreplvei.{b,h,w,d}), the
// source element width matching the destination arrangement. An LSX-only
// form: the toolchain's table gives vreplvei no LASX counterpart.
if srcVec && dstVec && src.hasEl && dst.hasSuf && !dst.hasEl {
if lasx || src.lasx || dst.lasx {
return nil, fmt.Errorf("VMOVQ: vreplvei has no %s-bank form", bank)
}
if src.unsig {
return nil, fmt.Errorf("VMOVQ: vreplvei takes no unsigned element suffix")
}
if src.width != dst.width {
return nil, fmt.Errorf("VMOVQ: element width does not match arrangement %q", ops[1].Raw)
}
if _, ok := l64VecSuffixWidth(false, dst); !ok {
return nil, fmt.Errorf("VMOVQ: invalid arrangement %q", ops[1].Raw)
}
var op uint32
limit := 0
switch src.width {
case 'B':
op, limit = enc.rveiB, 15
case 'H':
op, limit = enc.rveiH, 7
case 'W':
op, limit = enc.rveiW, 3
default:
op, limit = enc.rveiD, 1
}
if src.elem > limit {
return nil, fmt.Errorf("VMOVQ: element index %d out of range [0, %d]", src.elem, limit)
}
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Broadcast of element zero: XVMOVQ xj, xd.T (xvreplve0.{b,h,w,d,q}),
// a bare X source into an arranged X destination. LASX only.
if srcVec && dstVec && !src.hasSuf && dst.hasSuf && !dst.hasEl {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvreplve0 is the %s-bank form alone", bank)
}
var op uint32
switch dst.width {
case 'B':
if dst.lanes != 32 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0B
case 'H':
if dst.lanes != 16 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0H
case 'W':
if dst.lanes != 8 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0W
case 'V':
if dst.lanes != 4 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0D
case 'Q':
if dst.lanes != 2 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0Q
default:
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
return l64wordLE(op | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Insert of element zero: XVMOVQ xj, xd.T[i] (xvinsve0.{w,d}), a bare X
// source into one word or double-word lane. LASX only.
if srcVec && dstVec && !src.hasSuf && dst.hasEl {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvinsve0 is the %s-bank form alone", bank)
}
op, limit := enc.xinsW, 7
if dst.width != 'W' {
op, limit = enc.xinsD, 3
if dst.width != 'V' {
return nil, fmt.Errorf("XVMOVQ: xvinsve0 takes word or double-word lanes, got %q", ops[1].Raw)
}
}
if dst.elem > limit {
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", dst.elem, limit)
}
return l64wordLE(op | uint32(dst.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Element extract into a vector register: XVMOVQ xj.T[i], xd
// (xvpickve.{w,d}), one word or double-word lane out to a bare X
// register. LASX only.
if srcVec && src.hasEl && dstVec && !dst.hasSuf {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvpickve is the %s-bank form alone", bank)
}
op, limit := enc.xpickW, 7
if src.width != 'W' {
op, limit = enc.xpickD, 3
if src.width != 'V' {
return nil, fmt.Errorf("XVMOVQ: xvpickve takes word or double-word lanes, got %q", ops[0].Raw)
}
}
if src.elem > limit {
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", src.elem, limit)
}
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant),
// both operands bare registers of the same bank.
if srcVec && dstVec {
@@ -2224,7 +2457,7 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
}
return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil
}
rj, off := l64MemWithFrame(ops[1], fi)
rj, off := l64VmovqMem(ops[1], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]")
}
@@ -2247,9 +2480,9 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
}
return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil
}
rj, off := l64MemWithFrame(ops[0], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
rj, off := l64VmovqMem(ops[0], fi)
if rj < 0 {
return nil, fmt.Errorf("VMOVQ: invalid load operand")
}
op := enc.ld
if dst.hasSuf {
@@ -2257,16 +2490,34 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw)
}
// vldrepl keeps the byte offset raw for bytes and scales it by
// the element width for the wider forms, the immediate field
// shrinking a bit per scale exactly as the toolchain encodes it
// (the field mask keeps the two's complement inside its width).
scale, mask, lo, hi := 1, int32(0xFFF), -2048, 2047
switch w {
case 0:
op = enc.replB
case 1:
op = enc.replH
scale, mask, lo, hi = 2, 0x7FF, -1024, 1023
case 2:
op = enc.replW
scale, mask, lo, hi = 4, 0x3FF, -512, 511
default:
op = enc.replD
scale, mask, lo, hi = 8, 0x1FF, -256, 255
}
if off%int32(scale) != 0 {
return nil, fmt.Errorf("VMOVQ: offset %d must be a multiple of %d", off, scale)
}
off /= int32(scale)
if off < int32(lo) || off > int32(hi) {
return nil, fmt.Errorf("VMOVQ: offset out of range [%d, %d]", lo*scale, hi*scale)
}
off &= mask
} else if off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
}
return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil
}
+32 -5
View File
@@ -279,6 +279,8 @@ const (
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd
l64Fllsc // acquire/release LL/SC (2R against a zero-offset memory operand)
l64Fscq // sc.q: op | middle<<10 | base<<5 | first against a zero-offset memory operand
)
// l64Enc is one instruction's encoding: its bit layout (format) and the
@@ -341,8 +343,9 @@ var l64Vec2R = map[string]bool{}
// such as vshuf.b).
var l64Vec4R = map[string]bool{}
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels.
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants, each pre-shifted to
// its exact bit range, read off `go tool objdump` of GOARCH=loong64
// `go tool asm` kernels and the toolchain's specialLsxMovInst table.
type l64VmovqEnc struct {
ld, st, ldx, stx uint32 // plain and indexed load/store
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
@@ -350,6 +353,11 @@ type l64VmovqEnc struct {
ins uint32 // vinsgr2vr element insert
dup uint32 // vreplgr2vr duplicate (width in [11:10])
move uint32 // vori.b/xvori.b $0 register move
rveiB, rveiH, rveiW, rveiD uint32 // vreplvei: broadcast one element (LSX)
rve0B, rve0H, rve0W uint32 // xvreplve0 broadcast of element zero (LASX)
rve0D, rve0Q uint32 // xvreplve0.{d,q}, ditto
xinsW, xinsD uint32 // xvinsve0: insert element zero (LASX)
xpickW, xpickD uint32 // xvpickve: extract element (LASX)
}
var l64VmovqTable = map[bool]l64VmovqEnc{
@@ -358,12 +366,17 @@ var l64VmovqTable = map[bool]l64VmovqEnc{
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
rveiB: 0x01CBDE << 14, rveiH: 0x0397BE << 13, rveiW: 0x072F7E << 12, rveiD: 0x0E5EFE << 11,
},
true: { // XVMOVQ, the LASX (X) bank
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
rve0B: 0x1DC1C0 << 10, rve0H: 0x1DC1E0 << 10, rve0W: 0x1DC1F0 << 10,
rve0D: 0x1DC1F8 << 10, rve0Q: 0x1DC1FC << 10,
xinsW: 0x03B7FE << 13, xinsD: 0x076FFE << 12,
xpickW: 0x03B81E << 13, xpickD: 0x07703E << 12,
},
}
@@ -373,7 +386,7 @@ func init() {
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
"SGT": 0x24 << 15, "SGTU": 0x25 << 15,
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15,
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15,
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
"ORN": 0x2c << 15, "ANDN": 0x2d << 15,
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
@@ -467,6 +480,20 @@ func init() {
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
// Acquire/release LL/SC (2R against a zero-offset memory operand):
// LLACQV (Rj), Rd loads, SCRELV Rd, (Rj) stores, both encoding
// op | rj<<5 | rd. Opcodes from cmd/internal/obj/loong64/instOp.go
// (ll.acq.{w,d}, sc.rel.{w,d}).
l64InstrTable["LLACQW"] = l64Enc{format: l64Fllsc, op: 0x0E15E0 << 10}
l64InstrTable["SCRELW"] = l64Enc{format: l64Fllsc, op: 0x0E15E1 << 10}
l64InstrTable["LLACQV"] = l64Enc{format: l64Fllsc, op: 0x0E15E2 << 10}
l64InstrTable["SCRELV"] = l64Enc{format: l64Fllsc, op: 0x0E15E3 << 10}
// SCQ (sc.q first, middle, (base)) keeps its own operand order: the
// encoding is op | middle<<10 | base<<5 | first, the memory operand's
// base in the rj field, not the toolchain's generic 3R layout.
l64InstrTable["SCQ"] = l64Enc{format: l64Fscq, op: 0x070AE << 15}
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
// selected by the operand kind; the shift mnemonics pair the 3R form
// with a 5/6-bit shift immediate.
@@ -868,7 +895,7 @@ func init() {
"VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF},
"XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF},
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F},
"XVSEQB": {0xED00 << 15, true, -16, 15, 0, 0x1F},
// vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a
// 7-bit field, but the toolchain range-checks it down to si5 as well
// (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64).
@@ -877,7 +904,7 @@ func init() {
"VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F},
"XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F},
"VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F},
"XVSEQV": {0xE903 << 15, true, -16, 15, 0, 0x7F},
"XVSEQV": {0xED03 << 15, true, -16, 15, 0, 0x7F},
// vslti compares against a signed (or, in the U spellings, unsigned)
// si5/ui5 constant.
"VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F},
+229 -2
View File
@@ -8,8 +8,8 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// firstTextLOONG64 parses assembly source and returns the first TEXT body.
@@ -831,3 +831,230 @@ TEXT ·atoms(SB), NOSPLIT, $0
0x4C000020,
)
}
// TestLOONG64_llacqScrel pins the acquire/release LL/SC pair. The oracle
// words come from GOARCH=loong64 go tool objdump and the toolchain's own
// loong64enc1.s golden bytes.
func TestLOONG64_llacqScrel(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·llsc(SB), NOSPLIT, $0
LLACQW (R5), R4
LLACQV (R5), R4
SCRELW R4, (R6)
SCRELV R4, (R6)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x385780A4, // ll.acq.w r4, r5
0x385788A4, // ll.acq.d r4, r5
0x385784C4, // sc.rel.w r4, r6
0x38578CC4, // sc.rel.d r4, r6
0x4C000020,
)
// The toolchain accepts the zero-offset memory form alone.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
LLACQW 4(R5), R4
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
SCRELV R4, 8(R6)
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_vmovqSuffixed pins the element-broadcast and element-move
// VMOVQ/XVMOVQ forms, with the oracle words lifted verbatim from the
// toolchain's loong64enc1.s.
func TestLOONG64_vmovqSuffixed(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·vmovq(SB), NOSPLIT, $0
VMOVQ V1.B[3], V9.B16
VMOVQ V2.H[2], V8.H8
VMOVQ V3.W[1], V7.W4
VMOVQ V4.V[0], V6.V2
XVMOVQ X0, X31.B32
XVMOVQ X1, X30.H16
XVMOVQ X2, X29.W8
XVMOVQ X3, X28.V4
XVMOVQ X3, X27.Q2
XVMOVQ X0, X31.W[7]
XVMOVQ X1, X29.W[0]
XVMOVQ X3, X28.V[3]
XVMOVQ X4, X27.V[0]
XVMOVQ X31.W[7], X0
XVMOVQ X29.W[0], X1
XVMOVQ X28.V[3], X8
XVMOVQ X27.V[0], X9
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x72F78C29, // vreplvei.b v9, v1, 3
0x72F7C848, // vreplvei.h v8, v2, 2
0x72F7E467, // vreplvei.w v7, v3, 1
0x72F7F086, // vreplvei.d v6, v4, 0
0x7707001F, // xvreplve0.b x31, x0
0x7707803E, // xvreplve0.h x30, x1
0x7707C05D, // xvreplve0.w x29, x2
0x7707E07C, // xvreplve0.d x28, x3
0x7707F07B, // xvreplve0.q x27, x3
0x76FFDC1F, // xvinsve0.w x31, x0, 7
0x76FFC03D, // xvinsve0.w x29, x1, 0
0x76FFEC7C, // xvinsve0.d x28, x3, 3
0x76FFE09B, // xvinsve0.d x27, x4, 0
0x7703DFE0, // xvpickve.w x0, x31, 7
0x7703C3A1, // xvpickve.w x1, x29, 0
0x7703EF88, // xvpickve.d x8, x28, 3
0x7703E369, // xvpickve.d x9, x27, 0
0x4C000020,
)
// The rejected shapes: a width mismatch between the element and the
// arrangement, an element index past the lane count, a wrong-bank
// vreplvei and an arrangement the LASX bank does not spell.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ V1.H[3], V9.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ V1.B[16], V9.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X1.B[3], X9.B32
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X0, X31.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X0, X31.W[8]
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_parityFixes pins the operand forms whose encodings were found
// diverging from the toolchain by the loong64enc1.s differential: the
// $off(reg) address immediate (addi.d), the SCQ operand order, the scaled
// vldrepl offsets (with their field masks), the XVSEQB/XVSEQV immediate
// opcodes and the zero-register base the toolchain gives FP-relative
// VMOVQ/XVMOVQ memory operands. Golden words from loong64enc1.s.
func TestLOONG64_parityFixes(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·parity(SB), NOSPLIT, $0-32
MOVW $4(R4), R5
MOVV $4(R4), R5
MOVW $65536(R4), R5
MOVW $-4096(R4), R5
SCQ R4, R5, (R6)
VMOVQ 2(R4), V1.H8
VMOVQ -6(R4), V1.H8
VMOVQ -12(R4), V2.W4
VMOVQ -16(R4), V3.V2
XVMOVQ -10(R4), X1.H16
XVSEQB $0, X2, X4
XVSEQH $3, X2, X4
XVSEQW $12, X2, X4
XVSEQV $15, X2, X4
XVSEQV $-15, X2, X4
VMOVQ V2, y+16(FP)
VMOVQ y+16(FP), V2
VMOVQ V2, x+2030(FP)
XVMOVQ X6, y+16(FP)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x02C01085, // addi.d $4, r4, r5
0x02C01085, // addi.d $4, r4, r5 (MOVW keeps the 64-bit addi.d)
0x1400021E, // lu12i.w $16, r30
0x038003DE, // ori $0, r30, r30
0x0010F885, // add.d r5, r4, r30
0x15FFFFFE, // lu12i.w $-1, r30
0x038003DE, // ori $0, r30, r30
0x0010F885, // add.d r5, r4, r30
0x385714C4, // sc.q r4, r5, (r6): middle<<10 | base<<5 | first
0x30400481, // vldrepl.h v1, 2(r4)
0x305FF481, // vldrepl.h v1, -6(r4)
0x302FF482, // vldrepl.w v2, -12(r4)
0x3017F883, // vldrepl.d v3, -16(r4)
0x325FEC81, // xvldrepl.h x1, -10(r4)
0x76800044, // xvseqi.b x4, x2, 0
0x76808C44, // xvseqi.h x4, x2, 3
0x76813044, // xvseqi.w x4, x2, 12
0x7681BC44, // xvseqi.d x4, x2, 15
0x7681C444, // xvseqi.d x4, x2, -15
0x2C406002, // vst v2, 24(r0): FP-relative keeps the zero base
0x2C006002, // vld v2, 24(r0)
0x2C5FD802, // vst v2, 2038(r0)
0x2CC06006, // xvst x6, 24(r0)
0x4C000020,
)
// Misaligned vldrepl offsets are rejected, as the toolchain does.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ 3(R4), V1.H8
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
MOVW $4(R4), F1
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_bytePseudo pins the BYTE literal-data pseudo-op, which the
// loong64 toolchain does not spell but the arm64 and riscv64 encoders of
// this package already accept for byte-exact data layout (a superset
// spelling, shippable via the goobj path).
func TestLOONG64_bytePseudo(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·bytes(SB), NOSPLIT, $0
BYTE $2
BYTE $1; BYTE $0
BYTE $255
RET
`)
code := assembleLOONG64Helper(t, fn)
// Four literal bytes, then RET (jirl r0, r1, 0); the trailing bytes pad
// the final word the way any sub-word tail does.
want := []byte{2, 1, 0, 0xFF, 0x20, 0x00, 0x00, 0x4C}
if !bytes.Equal(code[:len(want)], want) {
t.Errorf("bytes = % x, want % x", code, want)
}
for _, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
BYTE $256
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
BYTE $-1
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("%q: expected an error, got none", src)
}
}
}
+1 -1
View File
@@ -6,7 +6,7 @@ package asm
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// Loong64 frame mapping, matching the Go toolchain's loong64 backend.
+1 -1
View File
@@ -7,7 +7,7 @@ import (
"bytes"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestLOONG64_sys exercises the no-operand system instructions and the
+1 -1
View File
@@ -6,7 +6,7 @@ package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestLOONG64RelocOffsetsIncludePrologue pins the function-relative
+24
View File
@@ -36,6 +36,29 @@ type FloatImm struct {
func (FloatImm) isOperand() {}
// TLSMem is a thread-local access, the source form off(base)(TLS*1) with the
// base dropped: the toolchain's one-instruction TLS rewrite assembles it as
// the segment-prefixed absolute whose disp32 carries an R_TLS_LE patch site
// (the linker fills the TLS slot offset).
type TLSMem struct {
Disp int64
Size int
Seg byte // the segment override: FS (0x64) or GS (0x65) on windows
}
func (TLSMem) isOperand() {}
// SegAbs is a segment-absolute access, 0x30(GS): the segment override
// prefixes a disp32 absolute reference with no relocation. The base
// register spellings GS and FS produce it.
type SegAbs struct {
Disp int64
Size int
Seg byte // 0x64 FS, 0x65 GS
}
func (SegAbs) isOperand() {}
// Mem is a memory operand of the form disp(base)(index*scale).
type Mem struct {
Base Reg
@@ -45,6 +68,7 @@ type Mem struct {
Size int // operand width in bytes
HasBase bool
HasIndex bool
Seg byte // segment override prefix (0x64 FS, 0x65 GS); 0 = none
}
func (Mem) isOperand() {}
+30 -1
View File
@@ -17,13 +17,28 @@ import "strings"
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0-K7, the fp flag the x87 stack registers F0-F7.
// registers K0-K7, the fp flag the x87 stack registers F0-F7, the mmx flag the
// MMX registers M0-M7, the seg field a bare segment register (FS, GS) and the
// ctl field the control and debug registers, whose number rides an
// instruction's reg field rather than r/m.
type Reg struct {
idx int
size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH
mask bool // K0-K7 opmask register
fp bool // F0-F7 x87 stack register
mmx bool // M0-M7 MMX register
seg int // segment register number plus one (ES=1..GS=6); 0 = not one
ctl byte // 0 none, 1 CRn control register, 2 DRn debug register
}
// segNumber returns the segment register number (ES=0..GS=5) when r names a
// bare segment register.
func (r Reg) segNumber() (int, bool) {
if r.seg == 0 {
return 0, false
}
return r.seg - 1, true
}
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
@@ -149,6 +164,20 @@ func buildRegByName() map[string]Reg {
for i := 0; i <= 7; i++ {
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
}
// MMX: M0..M7.
for i := 0; i <= 7; i++ {
m["M"+itoa(i)] = Reg{idx: i, size: 8, mmx: true}
}
// Bare segment registers: ES, CS, SS, DS, FS, GS (the memory-base and
// index spellings of FS and GS are handled before register lookup).
for i, n := range []string{"ES", "CS", "SS", "DS", "FS", "GS"} {
m[n] = Reg{idx: i, size: 2, seg: i + 1}
}
// Control and debug registers: CR0..CR15, DR0..DR15.
for i := 0; i <= 15; i++ {
m["CR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 1}
m["DR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 2}
}
return m
}
+1 -1
View File
@@ -10,7 +10,7 @@ import (
"slices"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// assembleRISCV assembles a RISC-V TEXT function body into machine code.
+2 -2
View File
@@ -10,8 +10,8 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// firstTextRISCV parses assembly source and returns the first TEXT function body.
+1 -1
View File
@@ -7,7 +7,7 @@ import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
+1 -1
View File
@@ -6,7 +6,7 @@ package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestRISCVFrameSpadjAndLines checks that a framed function records its
+1 -1
View File
@@ -13,7 +13,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
+122 -3
View File
@@ -76,6 +76,10 @@ const (
// carries a vector length, so the register the L'L field follows is the
// XMM source.
vexExtractGPR
// vexBlend4 is the four-operand variable blend `OP mask, src2, src1,
// dst` (VPBLENDVB): ModRM.reg = dst (op3), VEX.vvvv = src1 (op2),
// ModRM.rm = src2 (op1) and the mask register in the /is4 byte (op0).
vexBlend4
)
// vexSpec describes one VEX instruction's encoding parameters.
@@ -202,8 +206,28 @@ var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8).
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8), and its
// double twin under op 01; the in-lane permutes under 04/05.
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM},
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM},
"VPERMILPD": {3, 0x05, 0, 1, -1, vexImmRM},
// VEX.66.0F3A.W0, the immediate-controlled AVX tail: the rounding
// pair, the AES key assistant and the string compares.
"VROUNDPD": {3, 0x09, 0, 1, -1, vexImmRM},
"VROUNDPS": {3, 0x08, 0, 1, -1, vexImmRM},
"VAESKEYGENASSIST": {3, 0xDF, 0, 1, -1, vexImmRM},
"VPCMPESTRI": {3, 0x61, 0, 1, -1, vexImmRM},
"VPCMPESTRM": {3, 0x60, 0, 1, -1, vexImmRM},
"VPCMPISTRI": {3, 0x63, 0, 1, -1, vexImmRM},
"VPCMPISTRM": {3, 0x62, 0, 1, -1, vexImmRM},
// VEX.128.66.0F3A.W0, the scalar lane extract to a GPR or memory
// (reg = the XMM source, r/m = the destination).
"VEXTRACTPS": {3, 0x17, 0, 1, -1, vexExtractGPR},
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR},
// VEX.128.66.0F3A.W0, the four-operand variable blend with its mask
// register in the /is4 byte.
"VPBLENDVB": {3, 0x4C, 0, 1, -1, vexBlend4},
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// imm8).
@@ -524,8 +548,16 @@ func isVex(mnemUpper string) bool {
if _, ok := vexTable[mnemUpper]; ok {
return true
}
_, ok := vexMoveTable[mnemUpper]
return ok
if _, ok := vexMoveTable[mnemUpper]; ok {
return true
}
// The dual-shape moves (VMOVHPD/VMOVLPD) pick their VEX form by operand
// count in encodeVex.
switch mnemUpper {
case "VMOVHPD", "VMOVLPD":
return true
}
return false
}
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
@@ -559,6 +591,22 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
}
}
// The high/low double moves split by operand count: three operands
// load-and-insert (mem, src, dst, an NDS form), two store (xmm, m64,
// the reversed store layout).
if mnemUpper == "VMOVHPD" || mnemUpper == "VMOVLPD" {
loadOp, storeOp := byte(0x16), byte(0x17)
if mnemUpper == "VMOVLPD" {
loadOp, storeOp = 0x12, 0x13
}
switch len(ops) {
case 3:
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: loadOp, w: 0, pp: 1, opdigit: -1, form: vexNDS3}, ops)
case 2:
return e.encodeVexRMRev(vexSpec{mapSel: 1, opcode: storeOp, w: 0, pp: 1, opdigit: -1, form: vexRMRev}, ops)
}
return fmt.Errorf("%s expects 2 or 3 operands, got %d", mnemUpper, len(ops))
}
spec := vexTable[mnemUpper]
switch spec.form {
case vexNDS3:
@@ -573,6 +621,10 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3Imm(spec, ops)
case vexExtract:
return e.encodeVexExtract(spec, ops)
case vexExtractGPR:
return e.encodeVexExtractGPR(spec, ops)
case vexBlend4:
return e.encodeVexBlend4(spec, ops)
case vexRMSrcLen:
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero:
@@ -964,6 +1016,73 @@ func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
}
// encodeVexExtractGPR encodes the lane extract to a general-purpose register
// or memory (VEXTRACTPS): OP $imm, xsrc, gpr/mem with the XMM source in
// ModRM.reg and the destination in r/m, L = 0.
func (e *enc) encodeVexExtractGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() || srcReg.size != 16 {
return fmt.Errorf("extract source must be an XMM register")
}
if _, isReg := dst.(Reg); !isReg && !memOperand(dst) {
return fmt.Errorf("extract destination must be a register or memory")
}
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, 0, srcReg.idx&7, rBit, 15, dst); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexBlend4 encodes the four-operand variable blend (VPBLENDVB):
// OP mask, src2, src1, dst with ModRM.reg = dst, VEX.vvvv = src1, r/m =
// src2 and the mask XMM register in the trailing /is4 byte.
func (e *enc) encodeVexBlend4(spec vexSpec, ops []Operand) error {
if len(ops) != 4 {
return fmt.Errorf("blend expects 4 operands (mask, src2, src1, dst), got %d", len(ops))
}
mask, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
maskReg, ok := mask.(Reg)
if !ok || !maskReg.isVec() || maskReg.size != 16 {
return fmt.Errorf("blend mask must be an XMM register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("blend second source must be a vector register")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("blend destination must be a vector register")
}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, dstReg.vecLenBit(), dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2); err != nil {
return err
}
// The /is4 byte names the mask register: bits [3:0] its low nibble,
// bit 7 the fourth register bit (X8-X15).
e.out = append(e.out, byte(maskReg.idx&7)|byte((maskReg.idx&8)<<4))
return nil
}
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
// move uses the store-form layout (reg = source, rm = destination), matching
+1 -1
View File
@@ -8,7 +8,7 @@
// to the arch package; the AST records syntax only.
package ast
import "sourcedock.dev/petrbalvin/gasm-devkit/token"
import "sourcedock.dev/petrbalvin/gasm-sdk/token"
// File is the parsed representation of one .s source file.
type File struct {
+1 -1
View File
@@ -6,7 +6,7 @@ package ast
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
"sourcedock.dev/petrbalvin/gasm-sdk/token"
)
func pos(line, col int) token.Position { return token.Position{Line: line, Column: col} }
+1 -1
View File
@@ -17,7 +17,7 @@ import (
"regexp"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
)
// go_asm.h is the header the Go compiler writes for every package that
+57 -3
View File
@@ -5,6 +5,7 @@ package main
import (
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
@@ -392,14 +393,67 @@ func TestRunCorpusAuditGOOS(t *testing.T) {
}
}
// TestRunCorpusAuditBuildConstraint covers the //go:build classification end
// to end: a generic-named file whose constraint admits one target is
// attempted there alone (cpu_x86.s on amd64), and a file whose constraint
// admits none of the four targets is never attempted (the msan and
// goexperiment trees).
func TestRunCorpusAuditBuildConstraint(t *testing.T) {
dir := t.TempDir()
write := func(name, src string) {
t.Helper()
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
t.Fatal(err)
}
}
write("x86.s", "//go:build 386 || amd64\n\nTEXT \xc2\xb7f(SB), NOSPLIT, $0\n\tRET\n")
write("racey.s", "//go:build race\n\nTEXT \xc2\xb7r(SB), NOSPLIT, $0\n\tRET\n")
write("plain.s", "TEXT \xc2\xb7p(SB), NOSPLIT, $0\n\tRET\n")
stats, err := runCorpusAudit(dir, nil)
if err != nil {
t.Fatalf("runCorpusAudit: %v", err)
}
tally := func(name string) *corpusTally {
for i, tg := range stats.targets {
if tg.name == name {
return stats.tallies[i]
}
}
t.Fatalf("no tally for %s", name)
return nil
}
if stats.narrowed != 1 || stats.excluded != 1 || stats.generic != 1 {
t.Errorf("buckets = narrowed %d, excluded %d, generic %d; want 1, 1, 1", stats.narrowed, stats.excluded, stats.generic)
}
if a := tally("amd64"); a.attempted != 2 || a.assembled != 2 {
t.Errorf("amd64 = %d/%d, want 2/2 (x86.s and plain.s)", a.assembled, a.attempted)
}
for _, name := range []string{"arm64", "riscv64", "loong64"} {
if a := tally(name); a.attempted != 1 || a.assembled != 1 {
t.Errorf("%s = %d/%d, want 1/1 (plain.s only)", name, a.assembled, a.attempted)
}
}
if stats.full != 2 {
t.Errorf("full = %d, want 2 (x86.s over its one target, plain.s over all four)", stats.full)
}
}
// TestGenerateGoAsmHeaderRuntime pins the generator against the real thing:
// the runtime package, whose header the toolchain's own -asmhdr output was
// sampled from. Skipped in short mode: it type-checks the whole package.
// the runtime package of the ambient toolchain, whose header the toolchain's
// own -asmhdr output was sampled from. Skipped in short mode: it type-checks
// the whole package. The GOROOT comes from the go command itself, so the
// test follows whatever toolchain the host provides.
func TestGenerateGoAsmHeaderRuntime(t *testing.T) {
if testing.Short() {
t.Skip("type-checks the whole runtime package")
}
dir, err := generateGoAsmHeader("/usr/local/go/src/runtime", "", "amd64", t.TempDir())
out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil {
t.Skipf("no Go toolchain: %v", err)
}
runtimeDir := filepath.Join(strings.TrimSpace(string(out)), "src", "runtime")
dir, err := generateGoAsmHeader(runtimeDir, "", "amd64", t.TempDir())
if err != nil {
t.Fatalf("generateGoAsmHeader(runtime): %v", err)
}
+168 -32
View File
@@ -5,6 +5,7 @@ package main
import (
"fmt"
"go/build/constraint"
"maps"
"os"
"os/exec"
@@ -15,9 +16,9 @@ import (
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// cmdAuditInstructions cross-checks a gasm encoder against the Go toolchain's
@@ -38,7 +39,7 @@ import (
// construction and are excluded from the diff; the other architectures list
// their conditional branches outright.
func cmdAuditInstructions(args []string) error {
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [-I dir] [amd64|arm64|riscv64|loong64]", `
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [--list] [-I dir] [amd64|arm64|riscv64|loong64]", `
Compare the gasm encoder for the given architecture (default amd64) against
go tool asm and print the diff: superset encodings (gasm-only, shippable via
gasm asm --format goobj) and known-but-unencodable names (the backlog). The
@@ -55,16 +56,19 @@ toolchain probing. A file whose name carries a recognisable _arch suffix is
attempted for that architecture; a file without one is attempted for all
four, exactly as a GOARCH build would compile it. The report gives the
per-architecture pass rates and the most common failure reasons, which drive
the encodability backlog by frequency rather than by table order.
the encodability backlog by frequency rather than by table order. With
-list the report also prints every failing file with its reason, per
architecture.
`)
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
list := fs.Bool("list", false, "with --corpus, list every failing file with its reason, per architecture")
var dirs includeDirs
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
if err := fs.Parse(args); err != nil {
return err
}
if *corpus {
return cmdAuditCorpus(fs.Args(), dirs)
return cmdAuditCorpus(fs.Args(), dirs, *list)
}
archName := "amd64"
switch n := len(fs.Args()); {
@@ -403,20 +407,30 @@ type corpusTally struct {
assembled int
reasons map[string]int // failure reason → count
example map[string]string // failure reason → one representative file
fails []corpusFailure // every failure, in file order, for --list
}
func (t *corpusTally) fail(path, reason string) {
// corpusFailure is one failed attempt, recorded for the --list report.
type corpusFailure struct {
path string
reason string
detail string
}
func (t *corpusTally) fail(path string, err error) {
reason := corpusReason(err)
t.reasons[reason]++
if t.example[reason] == "" {
t.example[reason] = path
}
t.fails = append(t.fails, corpusFailure{path: path, reason: reason, detail: firstLine(err.Error())})
}
// cmdAuditCorpus implements audit-instructions --corpus. The include
// directories carry #include resolution over a corpus whose files refer to
// headers such as GOROOT/pkg/include, the same -I a toolchain comparison
// needs.
func cmdAuditCorpus(args []string, dirs includeDirs) error {
func cmdAuditCorpus(args []string, dirs includeDirs, list bool) error {
if len(args) > 1 {
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
}
@@ -454,7 +468,7 @@ func cmdAuditCorpus(args []string, dirs includeDirs) error {
if err != nil {
return err
}
printCorpusStats(stats)
printCorpusStats(stats, list)
return nil
}
@@ -463,8 +477,10 @@ type corpusStats struct {
root string
files int
generic int // files attempted for all four architectures
narrowed int // files whose //go:build admits a proper subset of the four
excluded int // files whose //go:build admits none of the four: never compiled
otherPort int // files named for another Go port: never attempted
full int // files that assembled for every target architecture
full int // files that assembled for every applicable target architecture
targets []corpusTarget
tallies []*corpusTally
}
@@ -560,6 +576,57 @@ func otherGOOSFile(path string) bool {
return false
}
// buildConstraint returns the file's leading //go:build expression, or nil
// when the file carries none. The constraint governs the same header block
// go/build reads: blank lines and comments may precede it, and the first
// line that is neither ends the block. A constraint that does not parse
// narrows nothing, so the file stays in the attempted set: the audit must
// never exclude a file the toolchain would compile.
func buildConstraint(src string) constraint.Expr {
for line := range strings.SplitSeq(src, "\n") {
t := strings.TrimSpace(line)
switch {
case t == "":
continue
case strings.HasPrefix(t, "//"):
if constraint.IsGoBuild(t) {
e, err := constraint.Parse(t)
if err != nil {
return nil
}
return e
}
continue
default:
return nil
}
}
return nil
}
// unixOS is go/build's unixOS set: the GOOSes the unix build tag admits.
var unixOS = map[string]bool{
"aix": true, "android": true, "darwin": true, "dragonfly": true,
"freebsd": true, "hurd": true, "illumos": true, "ios": true,
"linux": true, "netbsd": true, "openbsd": true, "solaris": true,
}
// constraintTags answers the build tags a plain `go build` sets for a
// target: the GOOS and GOARCH, gc, and unix on the unix-like GOOSes. No
// experiment, sanitiser or cgo tag is ever true: the audit models the
// default build, and no GOROOT assembly file's constraint hinges on cgo.
func constraintTags(goarch, goos string) func(string) bool {
return func(tag string) bool {
switch tag {
case goarch, goos, "gc":
return true
case "unix":
return unixOS[goos]
}
return false
}
}
func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
files, err := asmFiles(root)
if err != nil {
@@ -577,8 +644,8 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
tallies[i] = &corpusTally{reasons: map[string]int{}, example: map[string]string{}}
}
// full is the north-star number: a file counts when every architecture
// its name allows assembles it.
full, generic, otherPort := 0, 0, 0
// its build admits assembles it.
full, generic, otherPort, narrowedCount, excluded := 0, 0, 0, 0, 0
// Header generation is created on first use, so a corpus with no
// go_asm.h includes never pays for a temp directory.
@@ -601,28 +668,78 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
// invisible to a file-name rule).
goos := goosFromFilename(path)
// The GOOS the header generation type-checks under follows the
// file's name when the name carries one; the ambient GOOS is the
// honest guess otherwise.
namedArch := arch.FromFilename(path)
var wanted []int // indexes into targets
if a := arch.FromFilename(path); a != arch.Unknown {
other := false
switch {
case namedArch != arch.Unknown:
for i, tg := range targets {
if tg.a == a {
if tg.a == namedArch {
wanted = append(wanted, i)
}
}
} else if otherPortFile(path) {
case otherPortFile(path):
// A file named for a Go port gasm does not support (arm,
// 386, s390x, ...) or for another GOOS is compiled by no
// supported-arch build, so it is neither generic nor a
// per-arch attempt: counting it as generic would make the
// headline unreachably low for reasons no supported target
// can fix.
other = true
otherPort++
} else {
generic++
default:
for i := range targets {
wanted = append(wanted, i)
}
}
// A //go:build constraint narrows the set of targets the file is
// assembled for, the way the go command compiles the file only for
// the targets the expression admits: cpu_x86.s belongs to the x86
// build alone, and a file whose constraint admits none of the four
// targets (the goexperiment.runtimesecret and msan trees) is
// compiled by no supported build. The tags mirror what a plain
// `go build` sets: the GOOS and GOARCH, gc, and unix on the
// unix-like GOOSes; no experiment, sanitiser or cgo tag is ever
// true. The GOOS is the file's own when the name carries one,
// else the ambient one.
goosForEval := goos
if goosForEval == "" {
goosForEval = runtime.GOOS
}
narrowed := false
if len(wanted) > 0 {
if ce := buildConstraint(src); ce != nil {
kept := make([]int, 0, len(wanted))
for _, i := range wanted {
tg := targets[i]
if ce.Eval(constraintTags(goarchName(tg.a), goosForEval)) {
kept = append(kept, i)
}
}
if len(kept) < len(wanted) {
narrowed = true
}
wanted = kept
}
}
switch {
case other:
// already tallied above
case len(wanted) == 0:
excluded++
case namedArch != arch.Unknown:
// a per-arch attempt over the constraint's subset
case narrowed:
narrowedCount++
default:
generic++
}
// A file that includes go_asm.h parses against a per-target header:
// the defines differ per architecture (internal/cpu's layout, for
// one) and per GOOS (sys_darwin_arm64.s's trampoline constants,
@@ -645,21 +762,22 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
hdrDir, err := hdr.dirFor(pkgDir, goos, goarchName(tg.a))
if err != nil {
ok = false
t.fail(path, corpusReason(err))
t.fail(path, err)
continue
}
f, errs := parser.ParseWithOptions(path, src, parser.Options{
Expand: true,
IncludeDirs: append(slices.Clone(dirs), hdrDir),
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
})
if len(errs) > 0 {
ok = false
t.fail(path, corpusReason(errs[0]))
t.fail(path, errs[0])
continue
}
if _, err := assembleFile(tg.a, f); err != nil {
if _, err := assembleFile(tg.a, f, goos); err != nil {
ok = false
t.fail(path, corpusReason(err))
t.fail(path, err)
continue
}
t.assembled++
@@ -670,21 +788,28 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
continue
}
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
ok := true
for _, i := range wanted {
tg, t := targets[i], tallies[i]
t.attempted++
// The parse carries the target's platform predefines, so it
// cannot be shared across targets the way a header-free file's
// could: a #ifdef GOARCH_arm block must be live on arm64 and
// dead everywhere else.
f, errs := parser.ParseWithOptions(path, src, parser.Options{
Expand: true,
IncludeDirs: dirs,
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
})
var err error
if len(errs) > 0 {
err = errs[0] // a parse failure is a failure for every target
} else {
_, err = assembleFile(tg.a, f)
_, err = assembleFile(tg.a, f, goos)
}
if err != nil {
ok = false
t.fail(path, corpusReason(err))
t.fail(path, err)
continue
}
t.assembled++
@@ -698,6 +823,8 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
root: root,
files: len(files),
generic: generic,
narrowed: narrowedCount,
excluded: excluded,
otherPort: otherPort,
full: full,
targets: targets,
@@ -706,14 +833,16 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
}
// printCorpusStats renders the corpus audit report.
func printCorpusStats(s *corpusStats) {
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d named for other Go ports, never attempted)\n", s.root, s.files, s.generic, s.otherPort)
func printCorpusStats(s *corpusStats, list bool) {
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d narrowed by //go:build; %d excluded by //go:build; %d named for other Go ports, never attempted)\n",
s.root, s.files, s.generic, s.narrowed, s.excluded, s.otherPort)
// The rate is over the files a supported build would attempt: the
// other ports' files sit in the count for completeness but can never
// assemble, so counting them in the denominator would report the gap
// of architectures gasm deliberately does not target.
attemptable := max(s.files-s.otherPort, 1)
fmt.Printf(" assemble for every target architecture: %d of %d attemptable (%.1f%%)\n", s.full, attemptable, 100*float64(s.full)/float64(attemptable))
// other ports' files and the ones no supported target compiles sit in
// the count for completeness but can never assemble, so counting them
// in the denominator would report the gap of platforms gasm
// deliberately does not target.
attemptable := max(s.files-s.otherPort-s.excluded, 1)
fmt.Printf(" assemble for every applicable target: %d of %d attemptable (%.1f%%)\n", s.full, attemptable, 100*float64(s.full)/float64(attemptable))
for i, tg := range s.targets {
t := s.tallies[i]
fmt.Printf(" %s: %d/%d attempted\n", tg.name, t.assembled, t.attempted)
@@ -721,6 +850,13 @@ func printCorpusStats(s *corpusStats) {
fmt.Printf(" %4d %s\n", t.reasons[r], r)
fmt.Printf(" e.g. %s\n", t.example[r])
}
if !list {
continue
}
for _, f := range t.fails {
fmt.Printf(" FAIL %s\n", f.path)
fmt.Printf(" %s: %s\n", f.reason, f.detail)
}
}
}
+61 -1
View File
@@ -4,9 +4,10 @@
package main
import (
"runtime"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
)
func TestDerivedFamily(t *testing.T) {
@@ -64,3 +65,62 @@ func TestGasmEncodable(t *testing.T) {
}
}
}
// TestBuildConstraint pins the //go:build reader: the constraint governs the
// leading comment block, the first non-comment line ends it (a tag below a
// #include governs nothing, exactly as go/build drops it), and a file
// without one admits every target.
func TestBuildConstraint(t *testing.T) {
admits := func(src, goarch, goos string) bool {
t.Helper()
e := buildConstraint(src)
if e == nil {
return true
}
return e.Eval(constraintTags(goarch, goos))
}
const ret = "TEXT \xc2\xb7f(SB), NOSPLIT, $0\n\tRET\n"
cases := []struct {
name string
src string
amd64, arm64 bool
}{
{"no constraint", ret, true, true},
{"x86 only", "//go:build 386 || amd64\n\n" + ret, true, false},
{"arm64 and linux", "//go:build arm64 && linux\n\n" + ret, false, true},
{"msan never", "//go:build msan\n\n" + ret, false, false},
{"experiment never", "//go:build goexperiment.runtimesecret\n\n" + ret, false, false},
{"below an include governs nothing", "#include \"textflag.h\"\n//go:build amd64\n" + ret, true, true},
{"unparsable narrows nothing", "//go:build (amd64\n" + ret, true, true},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
if got := admits(c.src, "amd64", runtime.GOOS); got != c.amd64 {
t.Errorf("amd64 admission = %v, want %v", got, c.amd64)
}
if got := admits(c.src, "arm64", runtime.GOOS); got != c.arm64 {
t.Errorf("arm64 admission = %v, want %v", got, c.arm64)
}
})
}
}
// TestConstraintTags pins the tag set a plain `go build` sets: the GOOS and
// GOARCH, gc, unix on the unix-like GOOSes; nothing else is ever true.
func TestConstraintTags(t *testing.T) {
ok := constraintTags("amd64", "linux")
for _, tag := range []string{"amd64", "linux", "gc", "unix"} {
if !ok(tag) {
t.Errorf("tag %q = false, want true", tag)
}
}
for _, tag := range []string{"arm64", "freebsd", "darwin", "cgo", "race", "msan", "goexperiment.runtimesecret"} {
if ok(tag) {
t.Errorf("tag %q = true, want false", tag)
}
}
fb := constraintTags("arm64", "freebsd")
if !fb("unix") {
t.Error("unix on freebsd = false, want true")
}
}
+2 -2
View File
@@ -1,7 +1,7 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build !linux
//go:build !(linux || (freebsd && (amd64 || arm64 || riscv64)))
package main
@@ -11,6 +11,6 @@ import (
)
func cmdDebug(args []string) int {
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux (ptrace)")
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux or FreeBSD (ptrace)")
return 1
}
@@ -1,7 +1,7 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
//go:build linux || (freebsd && (amd64 || arm64 || riscv64))
package main
@@ -13,8 +13,8 @@ import (
"strings"
"time"
"sourcedock.dev/petrbalvin/gasm-devkit/debug"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
"sourcedock.dev/petrbalvin/gasm-sdk/debug"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
func cmdDebug(args []string) int {
@@ -239,10 +239,12 @@ REPL commands:
fmt.Printf("gasm debug: cover: stopped on signal %v\n", sig)
break
}
reason, _ := sess.StopInfo()
regs, rerr := sess.GetRegs()
if rerr != nil {
break
}
trapPC := regs.GetPC()
// HandleTrap restores the original byte, rewinds PC and counts
// the hit on the breakpoint itself. Single-step over the
// restored instruction so the reinsertion at the top of the
@@ -251,6 +253,12 @@ REPL commands:
if err := sess.Step(); err != nil {
break
}
} else if debug.TrapStray(sess, reason, trapPC) {
// A breakpoint-class trap that matches none of ours and left
// the PC in place: resuming would re-execute the trapping
// instruction forever, so the coverage run stops here.
fmt.Printf("gasm debug: cover: SIGTRAP at %#x matches no breakpoint; the PC did not advance\n", trapPC)
break
}
}
hits := map[uint64]int{}
+4 -4
View File
@@ -9,9 +9,9 @@ import (
"sort"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// cmdDis disassembles machine code: either a raw binary (standard input with
@@ -84,7 +84,7 @@ func disSource(path string, target arch.Arch) int {
if len(errs) > 0 {
return 1
}
img, err := assembleFile(target, f)
img, err := assembleFile(target, f, "")
if err != nil {
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
return 1
+43 -20
View File
@@ -27,15 +27,15 @@ import (
"sync"
"syscall"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/format"
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
"sourcedock.dev/petrbalvin/gasm-devkit/lsp"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/format"
"sourcedock.dev/petrbalvin/gasm-sdk/lexer"
"sourcedock.dev/petrbalvin/gasm-sdk/lint"
"sourcedock.dev/petrbalvin/gasm-sdk/lsp"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
// version reports the release the toolchain recorded for this build: the
@@ -574,7 +574,7 @@ naming the package.
defer cleanup()
dirs = append(dirs, hdrDir)
}
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs, Predefines: platformPredefinesFor(string(targetArch), goos)})
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
@@ -582,7 +582,7 @@ naming the package.
return 1
}
img, err := assembleFile(targetArch, f)
img, err := assembleFile(targetArch, f, goos)
if err != nil {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
return 1
@@ -789,11 +789,34 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
return 1
}
// platformPredefines mirrors the go command's assembler invocation, which
// defines GOOS_<goos> and GOARCH_<arch> as -D macros: GOROOT headers
// (go_tls.h, asm_riscv64.h) select their platform blocks with #ifdef on
// exactly those names, so an assembler without them cannot see the platform
// definitions at all.
func platformPredefines(goarch, goos string) map[string]string {
return map[string]string{
"GOARCH_" + goarch: "1",
"GOOS_" + goos: "1",
}
}
// platformPredefinesFor resolves the ambient GOOS the way a build would: a
// file whose name carries one (sys_darwin_arm64.s) is compiled for that GOOS
// and nothing else.
func platformPredefinesFor(goarch string, fileGoos string) map[string]string {
goos := fileGoos
if goos == "" {
goos = runtime.GOOS
}
return platformPredefines(goarch, goos)
}
// assembleFile assembles a parsed file for the given architecture and returns the image.
func assembleFile(targetArch arch.Arch, f *ast.File) (*asm.Image, error) {
func assembleFile(targetArch arch.Arch, f *ast.File, goos string) (*asm.Image, error) {
switch targetArch {
case arch.AMD64:
return asm.AssembleFile(f)
return asm.AssembleFile(f, asm.WithGOOS(goos))
case arch.RISCV:
return asm.AssembleFileRISCV(f)
case arch.ARM64:
@@ -813,18 +836,18 @@ func assemblePath(path string, forced arch.Arch, dirs includeDirs) (*asm.Image,
if err != nil {
return nil, err
}
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
target := forced
if target == arch.Unknown {
target = arch.FromFilename(path)
}
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs, Predefines: platformPredefinesFor(string(target), "")})
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return nil, fmt.Errorf("parse errors")
}
target := forced
if target == arch.Unknown {
target = arch.FromFilename(path)
}
return assembleFile(target, f)
return assembleFile(target, f, "")
}
// printByteDiff shows the first few byte differences between two code blocks.
@@ -920,7 +943,7 @@ func cmdVerifyNonJIT(path string, targetArch arch.Arch, groundTruth, profile boo
if len(errs) > 0 {
return 1
}
img, err := assembleFile(targetArch, f)
img, err := assembleFile(targetArch, f, "")
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
+2 -2
View File
@@ -14,8 +14,8 @@ import (
"syscall"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
)
const clean = "#include \"textflag.h\"\n" +
+2 -2
View File
@@ -11,8 +11,8 @@ import (
"os"
"strings"
gasmast "sourcedock.dev/petrbalvin/gasm-devkit/ast"
gasmparser "sourcedock.dev/petrbalvin/gasm-devkit/parser"
gasmast "sourcedock.dev/petrbalvin/gasm-sdk/ast"
gasmparser "sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// cmdScaffold generates a differential test skeleton for every kernel in a
+31 -7
View File
@@ -1,13 +1,17 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
//go:build linux || (freebsd && (amd64 || arm64 || riscv64))
package debug
import "strings"
import "fmt"
import (
"cmp"
"fmt"
"slices"
)
// Breakpoint is one software breakpoint in the debuggee.
type Breakpoint struct {
@@ -149,15 +153,16 @@ func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (
return bp, nil
}
// Info returns a formatted list of all breakpoints.
// Info returns a formatted list of all breakpoints, ordered by address so
// the numbering is stable across calls (map iteration order is not).
func (bm *Breakpoints) Info() string {
if len(bm.bps) == 0 {
return "no breakpoints set\n"
}
var result strings.Builder
i := 0
for _, bp := range bm.bps {
i++
bps := bm.All()
slices.SortFunc(bps, func(a, b *Breakpoint) int { return cmp.Compare(a.Addr, b.Addr) })
for i, bp := range bps {
status := "enabled"
if !bp.Enabled {
status = "disabled"
@@ -170,7 +175,7 @@ func (bm *Breakpoints) Info() string {
if bp.Cond != nil {
cond = " if " + bp.Cond.String()
}
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond))
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i+1, label, bp.Addr, status, bp.hits, cond))
}
return result.String()
}
@@ -238,6 +243,25 @@ func (bm *Breakpoints) All() []*Breakpoint {
// Hits returns how many times the breakpoint has been hit.
func (bp *Breakpoint) Hits() int { return bp.hits }
// TrapStray reports whether a stop is a breakpoint-class trap that matches
// no breakpoint of ours and cannot be resumed: the PC still stands on the
// trapping instruction (the kernel's own BRK, EBREAK or break, on an
// architecture that reports the trap in place), so the next resume would
// re-execute it and trap forever. trapPC is the PC the stop reported,
// before any HandleTrap rewinding; reason is the stop's StopInfo class.
// The debuggee's SIGSTOP barriers also stop without PC movement, and they
// never carry the breakpoint class, so they are unaffected.
func TrapStray(s *Session, reason StopReason, trapPC uint64) bool {
if reason != StopBreakpoint {
return false
}
after, err := s.GetRegs()
if err != nil {
return false
}
return after.GetPC() <= trapPC-uint64(breakpointPCAdjust)
}
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
// On amd64 the kernel reports the trap with RIP past the INT3; on the
// other supported architectures the PC still stands on the trap
+44
View File
@@ -10,6 +10,8 @@ package debug
// build on every supported linux architecture.
import (
"fmt"
"slices"
"strings"
"testing"
)
@@ -263,3 +265,45 @@ func TestConditionString(t *testing.T) {
}
}
}
// TestBreakpointsInfoOrdered proves the listing is ordered by address: the
// numbers it prints are map keys rendered in iteration order otherwise, so
// the same set of breakpoints would renumber itself between calls.
func TestBreakpointsInfoOrdered(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
addrs := []uint64{0x9000, 0x1000, 0x7000, 0x3000, 0x8000, 0x2000,
0x6000, 0x4000, 0x5000, 0xa000}
for i, a := range addrs {
if _, err := bm.Set(a, fmt.Sprintf("bp%d", i)); err != nil {
t.Fatalf("Set(%#x): %v", a, err)
}
}
sorted := append([]uint64(nil), addrs...)
slices.Sort(sorted)
info := bm.Info()
for i, a := range sorted {
want := fmt.Sprintf(" %d: bp%d at %#x", i+1, indexOf(addrs, a), a)
if !strings.Contains(info, want) {
t.Errorf("Info() missing %q; listing:\n%s", want, info)
}
}
// The numbers themselves must ascend: "1:" before "2" ... "10".
pos := 0
for i := range len(addrs) {
next := strings.Index(info[pos:], fmt.Sprintf(" %d: ", i+1))
if next < 0 {
t.Fatalf("Info() has no entry %d; listing:\n%s", i+1, info)
}
pos += next
}
}
func indexOf(addrs []uint64, a uint64) int {
for i, v := range addrs {
if v == a {
return i
}
}
return -1
}
+325
View File
@@ -0,0 +1,325 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug
import (
"fmt"
"os"
"runtime"
"strings"
"testing"
"time"
)
// Regression tests for the debugger audit: memory access at mapping
// boundaries, watchpoint slot attribution, launch failure latency, stray
// trap instructions and the REPL's argument validation. All drive a real
// ptrace session, so they run on amd64 hosts only.
// memMap is one line of /proc/pid/maps.
type memMap struct {
lo, hi uint64
perms string
name string
}
// readMaps parses the debuggee's memory map.
func readMaps(t *testing.T, pid int) []memMap {
t.Helper()
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
if err != nil {
t.Fatalf("read maps: %v", err)
}
var out []memMap
for line := range strings.SplitSeq(string(data), "\n") {
fields := strings.Fields(line)
if len(fields) < 2 {
continue
}
var lo, hi uint64
if _, err := fmt.Sscanf(fields[0], "%x-%x", &lo, &hi); err != nil {
continue
}
m := memMap{lo: lo, hi: hi, perms: fields[1]}
if len(fields) >= 6 {
m.name = fields[5]
}
out = append(out, m)
}
return out
}
// boundaryByte returns the last byte of a writable, ordinary mapping that is
// followed by an unmapped gap: an access there is inside the mapping, while
// the 8-byte word starting at it crosses into unmapped memory.
func boundaryByte(t *testing.T, pid int) uint64 {
t.Helper()
maps := readMaps(t, pid)
for i, m := range maps {
if !strings.Contains(m.perms, "rw") ||
strings.Contains(m.name, "vvar") || strings.Contains(m.name, "vdso") ||
strings.Contains(m.name, "vsyscall") {
continue
}
gap := uint64(1) << 62
if i+1 < len(maps) {
gap = maps[i+1].lo - m.hi
}
if gap >= 4096 {
return m.hi - 1
}
}
t.Skip("no writable mapping followed by a hole; cannot construct the boundary")
return 0
}
// TestReadMemoryPageBoundary proves ReadMemory never reads past the requested
// range: one byte at the end of a mapping followed by a hole must be
// readable, which the old word-at-a-time tail read failed because its final
// 8-byte Peek crossed into the unmapped page.
func TestReadMemoryPageBoundary(t *testing.T) {
sess, _, _ := launchKernel(t, buildGasm(t), boundaryKernel(t), "boundary", nil)
addr := boundaryByte(t, sess.Pid())
mem, err := sess.ReadMemory(addr, 1)
if err != nil {
t.Fatalf("ReadMemory(%#x, 1): %v (the read must not cross into the unmapped page)", addr, err)
}
if len(mem) != 1 {
t.Fatalf("ReadMemory returned %d bytes, want 1", len(mem))
}
// A request whose own range crosses into the hole must still fail.
if _, err := sess.ReadMemory(addr, 8); err == nil {
t.Fatal("ReadMemory past the mapping end should fail")
}
}
// TestDisassemblePageBoundary proves the disassembler shrinks its read
// window at a mapping end instead of failing: the instruction stream cannot
// be decoded at all when the fixed 15-byte read crosses into the hole.
func TestDisassemblePageBoundary(t *testing.T) {
sess, _, _ := launchKernel(t, buildGasm(t), boundaryKernel(t), "boundary", nil)
addr := boundaryByte(t, sess.Pid())
if _, _, err := sess.Disassemble(addr); err != nil {
t.Fatalf("Disassemble(%#x): %v (the read window must shrink at the mapping end)", addr, err)
}
}
// TestWriteMemoryPageBoundary proves WriteMemory writes exactly the bytes it
// is given: one byte at the end of a mapping followed by a hole must be
// writable, which the old read-modify-write of the final partial word failed
// because its Peek crossed into the unmapped page.
func TestWriteMemoryPageBoundary(t *testing.T) {
sess, _, _ := launchKernel(t, buildGasm(t), boundaryKernel(t), "boundary", nil)
addr := boundaryByte(t, sess.Pid())
orig, err := sess.ReadMemory(addr, 1)
if err != nil {
t.Fatalf("ReadMemory(%#x, 1): %v", addr, err)
}
if err := sess.WriteMemory(addr, []byte{orig[0]}); err != nil {
t.Fatalf("WriteMemory(%#x, 1): %v (the write must not read past the range)", addr, err)
}
}
// TestWatchpointSlotAttribution proves a hit is attributed to the slot that
// fired, not to an earlier one whose DR6 status bit is still set: the B0-B3
// bits are sticky, so they must be acknowledged when read.
func TestWatchpointSlotAttribution(t *testing.T) {
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func wptwo(x, y int64) (a, b int64)
TEXT ·wptwo(SB), NOSPLIT, $0-32
MOVQ $0x1111, AX
MOVQ AX, a+16(FP)
MOVQ $0x2222, BX
MOVQ BX, b+24(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "wptwo", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
// FP sits one word above the entry stack pointer (the return address
// occupies [RSP]), so a+16(FP) = RSP+24 and b+24(FP) = RSP+32.
watchA := regs.RSP + 24
watchB := regs.RSP + 32
if err := sess.SetWatchpoint(0, watchA, WatchWrite, 8); err != nil {
t.Fatalf("SetWatchpoint(0): %v", err)
}
if err := sess.SetWatchpoint(1, watchB, WatchWrite, 8); err != nil {
t.Fatalf("SetWatchpoint(1): %v", err)
}
for i, want := range []uint64{watchA, watchB} {
if err := sess.Continue(); err != nil {
t.Fatalf("Continue (hit %d): %v", i+1, err)
}
reason, addr := sess.StopInfo()
if reason != StopWatchpoint {
t.Fatalf("hit %d: stop reason = %v, want StopWatchpoint", i+1, reason)
}
if addr != want {
t.Fatalf("hit %d reported %#x, want %#x (the sticky DR6 bit misattributes the slot)", i+1, addr, want)
}
}
// Clearing a watchpoint must zero its address register: a stale
// address in a disabled slot turns any sticky status bit into a
// misattributed report later.
if err := sess.ClearWatchpoint(0); err != nil {
t.Fatalf("ClearWatchpoint(0): %v", err)
}
dr0, err := ptracePeekUser(sess.Pid(), drOffset)
if err != nil {
t.Fatalf("read DR0: %v", err)
}
if dr0 != 0 {
t.Fatalf("DR0 = %#x after ClearWatchpoint, want 0 (the address register must be cleared)", dr0)
}
}
// TestLaunchFailsFastOnDeadDebuggee proves a debuggee that dies before
// signalling readiness surfaces promptly: the ready poll used to run its
// full 2.5 seconds before the wait discovered the exit.
func TestLaunchFailsFastOnDeadDebuggee(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
path := boundaryKernel(t)
start := time.Now()
sess, err := Launch(bin, path, "nosuchfunction", nil)
elapsed := time.Since(start)
if err == nil {
sess.Kill()
t.Fatal("Launch with an unknown function should fail")
}
if !strings.Contains(err.Error(), "before signalling readiness") &&
!strings.Contains(err.Error(), "debuggee exited") {
t.Errorf("error does not name the dead debuggee: %v", err)
}
if elapsed >= 1500*time.Millisecond {
t.Fatalf("Launch took %v to report the dead debuggee; the readiness poll must detect the exit, not time out", elapsed)
}
}
// TestStrayTrapRunsThrough proves the continue loop survives a trap
// instruction planted in the kernel itself (BYTE $0xCC, the same byte the
// debugger patches in): on architectures that report the trap in place the
// loop must surface the stop, and on amd64 it runs through to the exit. A
// regression here hangs, so a watchdog fails the run.
func TestStrayTrapRunsThrough(t *testing.T) {
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func stray() int64
TEXT ·stray(SB), NOSPLIT, $0-8
MOVQ $7, AX
BYTE $0xCC
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, _ := launchKernel(t, bin, path, "stray", nil)
timer := time.AfterFunc(time.Minute, func() {
panic("watchdog: the continue loop hung on the stray trap instruction")
})
defer timer.Stop()
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), 0, 0, 0, nil, nil,
strings.NewReader("continue\nquit\n"))
})
if !strings.Contains(out, "debuggee exited") {
t.Errorf("the stray trap wedged the continue loop; output:\n%s", out)
}
}
// TestStepIntoFaultReportsSignal proves the step command reports a genuine
// signal-delivery-stop instead of silently printing the faulting
// instruction as if the step had succeeded.
func TestStepIntoFaultReportsSignal(t *testing.T) {
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func crash() int64
TEXT ·crash(SB), NOSPLIT, $0-8
XORQ AX, AX
MOVQ (AX), AX
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "crash", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, nil, nil,
strings.NewReader("step 2\nquit\n"))
})
if !strings.Contains(out, "stopped on signal") {
t.Errorf("stepping into the fault did not report the signal; output:\n%s", out)
}
}
// TestREPLRejectsBadArguments proves the command loop reports malformed
// input instead of silently defaulting: an unknown label for x would read
// address 0, and a malformed count would silently step one instruction.
func TestREPLRejectsBadArguments(t *testing.T) {
bin := buildGasm(t)
path := boundaryKernel(t)
sess, bm, fl := launchKernel(t, bin, path, "boundary", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, nil, nil,
strings.NewReader("x nosuchlabel\nstep abc\ndisas abc\nwatch 0x1000 q 8\nquit\n"))
})
for _, want := range []string{
"unknown address: nosuchlabel",
"invalid count: abc",
"unknown watchpoint type: q",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
if got := strings.Count(out, "invalid count: abc"); got != 2 {
t.Errorf("invalid count reported %d times, want 2 (step and disas):\n%s", got, out)
}
}
// boundaryKernel is a minimal kernel for the boundary tests, which only need
// a live, stopped debuggee.
func boundaryKernel(t *testing.T) string {
t.Helper()
const kernel = `#include "textflag.h"
// func boundary() int64
TEXT ·boundary(SB), NOSPLIT, $0-8
MOVQ $1, AX
MOVQ AX, ret+0(FP)
RET
`
return writeKernel(t, kernel)
}
+65
View File
@@ -0,0 +1,65 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes. An amd64
// instruction is up to 15 bytes long, but the read must not reach past the
// end of the mapping: when the full 15-byte window crosses into unmapped
// memory the window shrinks, because an instruction at the mapping's end is
// by construction no longer than the readable bytes that hold it.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
var lastErr error
for _, n := range []int{15, 8, 4, 2, 1} {
mem, err := s.ReadMemory(addr, n)
if err != nil {
lastErr = err
continue
}
ins, derr := disasm.Decode(arch.AMD64, mem, addr)
if derr != nil {
return "", 0, derr
}
return ins.Text, ins.Len, nil
}
return "", 0, lastErr
}
// DisassembleN decodes up to n instructions starting at addr and returns
// them as a formatted string with addresses and byte offsets.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result strings.Builder
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
break
}
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
if length == 0 {
length = 1
}
pc += uint64(length)
}
return result.String()
}
// isCallInsn reports whether disassembled text (x86asm.IntelSyntax) is a
// call. The first token must match exactly: a prefix test would also catch
// unrelated mnemonics.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
return strings.ToLower(m) == "call"
}
+60
View File
@@ -0,0 +1,60 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && arm64
package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 4)
if err != nil {
return "", 0, err
}
ins, err := disasm.Decode(arch.ARM64, mem, addr)
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr and returns
// them as a formatted string with addresses and byte offsets.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result strings.Builder
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
break
}
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
if length == 0 {
length = 1
}
pc += uint64(length)
}
return result.String()
}
// isCallInsn reports whether disassembled text (arm64asm.GoSyntax) is a
// call. GoSyntax renders bl as CALL; the native mnemonic is accepted too.
// The first token must match exactly so branches never match.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "bl":
return true
}
return false
}
+70
View File
@@ -0,0 +1,70 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && riscv64
package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes. The read
// shrinks from 4 to 2 bytes when the full word crosses into unmapped memory:
// a compressed instruction at the mapping's end still fits the shorter
// window, and an instruction can never extend past the mapping that holds it.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
var lastErr error
for _, n := range []int{4, 2} {
mem, err := s.ReadMemory(addr, n)
if err != nil {
lastErr = err
continue
}
ins, derr := disasm.Decode(arch.RISCV, mem, addr)
if derr != nil {
return "", 0, derr
}
return ins.Text, ins.Len, nil
}
return "", 0, lastErr
}
// DisassembleN decodes up to n instructions starting at addr and returns
// them as a formatted string with addresses and byte offsets.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result strings.Builder
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
break
}
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
if length == 0 {
length = 1
}
pc += uint64(length)
}
return result.String()
}
// isCallInsn reports whether disassembled text (riscv64asm.GoSyntax) is a
// call. GoSyntax renders jal and jalr calls as CALL; the native mnemonics
// are accepted too. The first token must match exactly: a prefix test on
// "bl" would catch branches on other architectures, and jalr as ret prints
// RET, which must not be stepped over.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "jal", "jalr":
return true
}
return false
}
+17 -8
View File
@@ -9,22 +9,31 @@ import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
// memory and returns its text representation and length in bytes. An amd64
// instruction is up to 15 bytes long, but the read must not reach past the
// end of the mapping: when the full 15-byte window crosses into unmapped
// memory the window shrinks, because an instruction at the mapping's end is
// by construction no longer than the readable bytes that hold it.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 15)
var lastErr error
for _, n := range []int{15, 8, 4, 2, 1} {
mem, err := s.ReadMemory(addr, n)
if err != nil {
return "", 0, err
lastErr = err
continue
}
ins, err := disasm.Decode(arch.AMD64, mem, addr)
if err != nil {
return "", 0, err
ins, derr := disasm.Decode(arch.AMD64, mem, addr)
if derr != nil {
return "", 0, derr
}
return ins.Text, ins.Len, nil
}
return "", 0, lastErr
}
// DisassembleN decodes up to n instructions starting at addr and returns
+2 -2
View File
@@ -9,8 +9,8 @@ import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
+2 -2
View File
@@ -9,8 +9,8 @@ import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
+16 -8
View File
@@ -9,22 +9,30 @@ import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
// memory and returns its text representation and length in bytes. The read
// shrinks from 4 to 2 bytes when the full word crosses into unmapped memory:
// a compressed instruction at the mapping's end still fits the shorter
// window, and an instruction can never extend past the mapping that holds it.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 4)
var lastErr error
for _, n := range []int{4, 2} {
mem, err := s.ReadMemory(addr, n)
if err != nil {
return "", 0, err
lastErr = err
continue
}
ins, err := disasm.Decode(arch.RISCV, mem, addr)
if err != nil {
return "", 0, err
ins, derr := disasm.Decode(arch.RISCV, mem, addr)
if derr != nil {
return "", 0, derr
}
return ins.Text, ins.Len, nil
}
return "", 0, lastErr
}
// DisassembleN decodes up to n instructions starting at addr.
+136
View File
@@ -0,0 +1,136 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
import "fmt"
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" RIP = %#016x (func+%#x)\n", regs.RIP, regs.RIP-codeBase-funcOff)
fmt.Printf(" RSP = %#016x RBP = %#016x\n", regs.RSP, regs.RBP)
fmt.Printf(" RAX = %#016x RBX = %#016x\n", regs.RAX, regs.RBX)
fmt.Printf(" RCX = %#016x RDX = %#016x\n", regs.RCX, regs.RDX)
fmt.Printf(" RSI = %#016x RDI = %#016x\n", regs.RSI, regs.RDI)
fmt.Printf(" R8 = %#016x R9 = %#016x\n", regs.R8, regs.R9)
fmt.Printf(" R10 = %#016x R11 = %#016x\n", regs.R10, regs.R11)
fmt.Printf(" R12 = %#016x R13 = %#016x\n", regs.R12, regs.R13)
fmt.Printf(" R14 = %#016x R15 = %#016x\n", regs.R14, regs.R15)
fmt.Printf(" RFLAGS = %#x [%s]\n", regs.RFLAGS, decodeRflags(regs.RFLAGS))
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n Vector registers (YMM):")
for i := 0; i < 16; i += 2 {
fmt.Printf(" YMM%-2d = ", i)
printYMM(v.YMM[i][:])
fmt.Printf(" YMM%-2d = ", i+1)
printYMM(v.YMM[i+1][:])
fmt.Println()
}
}
func printYMM(b []byte) {
for j := 0; j < 32; j += 4 {
v := uint32(b[j]) | uint32(b[j+1])<<8 | uint32(b[j+2])<<16 | uint32(b[j+3])<<24
fmt.Printf("%08x ", v)
}
}
func decodeRflags(f uint64) string {
var flags string
if f&1 != 0 {
flags += "CF "
}
if f&(1<<2) != 0 {
flags += "PF "
}
if f&(1<<4) != 0 {
flags += "AF "
}
if f&(1<<6) != 0 {
flags += "ZF "
}
if f&(1<<7) != 0 {
flags += "SF "
}
if f&(1<<8) != 0 {
flags += "TF "
}
if f&(1<<9) != 0 {
flags += "IF "
}
if f&(1<<10) != 0 {
flags += "DF "
}
if f&(1<<11) != 0 {
flags += "OF "
}
if flags == "" {
return "none"
}
return flags[:len(flags)-1]
}
// SetReg modifies a register value in the debuggee.
func (s *Session) SetReg(name string, value uint64) error {
regs, err := s.GetRegs()
if err != nil {
return err
}
switch name {
case "rax", "eax", "ax", "al":
regs.RAX = value
case "rbx", "ebx", "bx", "bl":
regs.RBX = value
case "rcx", "ecx", "cx", "cl":
regs.RCX = value
case "rdx", "edx", "dx", "dl":
regs.RDX = value
case "rsi", "esi", "si":
regs.RSI = value
case "rdi", "edi", "di":
regs.RDI = value
case "rbp", "ebp", "bp":
regs.RBP = value
case "rsp", "esp", "sp":
regs.RSP = value
case "r8":
regs.R8 = value
case "r9":
regs.R9 = value
case "r10":
regs.R10 = value
case "r11":
regs.R11 = value
case "r12":
regs.R12 = value
case "r13":
regs.R13 = value
case "r14":
regs.R14 = value
case "r15":
regs.R15 = value
case "rip", "eip":
regs.RIP = value
default:
return fmt.Errorf("debug: unknown register %q", name)
}
return s.SetRegs(&regs)
}
// archReturnAddr reads the return address of the current frame (amd64
// ABI0 convention). A function that contains a CALL (or has a frame) is
// assembled with the prologue PUSHQ BP; MOVQ SP, BP, so mid-function the
// word at SP is the saved caller BP, a stack address, and the return
// address sits further up. Walk the stack from SP and take the first word
// that lies in an executable mapping: stack and data words never do, a
// return address always does. FreeBSD exposes no mapping list, so the
// walk degenerates to the raw entry convention, [SP] before any push.
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return s.Peek(regs.RSP)
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "RSP" }
+127
View File
@@ -0,0 +1,127 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && arm64
package debug
import (
"encoding/binary"
"fmt"
)
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.SP, regs.X29)
fmt.Printf(" LR = %#016x\n", regs.X30)
fmt.Printf(" X0 = %#016x X1 = %#016x\n", regs.X0, regs.X1)
fmt.Printf(" X2 = %#016x X3 = %#016x\n", regs.X2, regs.X3)
fmt.Printf(" X4 = %#016x X5 = %#016x\n", regs.X4, regs.X5)
fmt.Printf(" X6 = %#016x X7 = %#016x\n", regs.X6, regs.X7)
fmt.Printf(" X8 = %#016x X9 = %#016x\n", regs.X8, regs.X9)
fmt.Printf(" X10 = %#016x X11 = %#016x\n", regs.X10, regs.X11)
fmt.Printf(" X12 = %#016x X13 = %#016x\n", regs.X12, regs.X13)
fmt.Printf(" X14 = %#016x X15 = %#016x\n", regs.X14, regs.X15)
fmt.Printf(" X16 = %#016x X17 = %#016x\n", regs.X16, regs.X17)
fmt.Printf(" X18 = %#016x X19 = %#016x\n", regs.X18, regs.X19)
fmt.Printf(" X20 = %#016x X21 = %#016x\n", regs.X20, regs.X21)
fmt.Printf(" X22 = %#016x X23 = %#016x\n", regs.X22, regs.X23)
fmt.Printf(" X24 = %#016x X25 = %#016x\n", regs.X24, regs.X25)
fmt.Printf(" X26 = %#016x X27 = %#016x\n", regs.X26, regs.X27)
fmt.Printf(" X28 = %#016x PSTATE = %#x\n", regs.X28, regs.PSTATE)
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n Vector registers (V0-V31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" V%-2d = %016x%016x\n", i, binary.LittleEndian.Uint64(v.V[i][8:16]), binary.LittleEndian.Uint64(v.V[i][0:8]))
fmt.Printf(" V%-2d = %016x%016x\n", i+1, binary.LittleEndian.Uint64(v.V[i+1][8:16]), binary.LittleEndian.Uint64(v.V[i+1][0:8]))
}
}
// SetReg modifies a register value in the debuggee.
func (s *Session) SetReg(name string, value uint64) error {
regs, err := s.GetRegs()
if err != nil {
return err
}
switch name {
case "x0":
regs.X0 = value
case "x1":
regs.X1 = value
case "x2":
regs.X2 = value
case "x3":
regs.X3 = value
case "x4":
regs.X4 = value
case "x5":
regs.X5 = value
case "x6":
regs.X6 = value
case "x7":
regs.X7 = value
case "x8":
regs.X8 = value
case "x9":
regs.X9 = value
case "x10":
regs.X10 = value
case "x11":
regs.X11 = value
case "x12":
regs.X12 = value
case "x13":
regs.X13 = value
case "x14":
regs.X14 = value
case "x15":
regs.X15 = value
case "x16":
regs.X16 = value
case "x17":
regs.X17 = value
case "x18":
regs.X18 = value
case "x19":
regs.X19 = value
case "x20":
regs.X20 = value
case "x21":
regs.X21 = value
case "x22":
regs.X22 = value
case "x23":
regs.X23 = value
case "x24":
regs.X24 = value
case "x25":
regs.X25 = value
case "x26":
regs.X26 = value
case "x27":
regs.X27 = value
case "x28":
regs.X28 = value
case "x29", "fp":
regs.X29 = value
case "x30", "lr":
regs.X30 = value
case "sp":
regs.SP = value
case "pc":
regs.PC = value
default:
return fmt.Errorf("debug: unknown register %q", name)
}
return s.SetRegs(&regs)
}
// archReturnAddr reads the return address from LR (arm64 convention).
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return regs.X30, nil
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "SP" }
+121
View File
@@ -0,0 +1,121 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && riscv64
package debug
import "fmt"
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.Sp, regs.S0)
fmt.Printf(" RA = %#016x\n", regs.Ra)
fmt.Printf(" A0 = %#016x A1 = %#016x\n", regs.A0, regs.A1)
fmt.Printf(" A2 = %#016x A3 = %#016x\n", regs.A2, regs.A3)
fmt.Printf(" A4 = %#016x A5 = %#016x\n", regs.A4, regs.A5)
fmt.Printf(" A6 = %#016x A7 = %#016x\n", regs.A6, regs.A7)
fmt.Printf(" T0 = %#016x T1 = %#016x\n", regs.T0, regs.T1)
fmt.Printf(" T2 = %#016x T3 = %#016x\n", regs.T2, regs.T3)
fmt.Printf(" T4 = %#016x T5 = %#016x\n", regs.T4, regs.T5)
fmt.Printf(" T6 = %#016x\n", regs.T6)
fmt.Printf(" S1 = %#016x S2 = %#016x\n", regs.S1, regs.S2)
fmt.Printf(" S3 = %#016x S4 = %#016x\n", regs.S3, regs.S4)
fmt.Printf(" S5 = %#016x S6 = %#016x\n", regs.S5, regs.S6)
fmt.Printf(" S7 = %#016x S8 = %#016x\n", regs.S7, regs.S8)
fmt.Printf(" S9 = %#016x S10 = %#016x\n", regs.S9, regs.S10)
fmt.Printf(" S11 = %#016x\n", regs.S11)
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n FP registers (F0-F31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" F%-2d = %#018x F%-2d = %#018x\n", i, v.F[i], i+1, v.F[i+1])
}
fmt.Printf(" FCSR = %#x\n", v.FCSR)
}
// SetReg modifies a register value in the debuggee.
func (s *Session) SetReg(name string, value uint64) error {
regs, err := s.GetRegs()
if err != nil {
return err
}
switch name {
case "pc":
regs.PC = value
case "ra", "x1":
regs.Ra = value
case "sp", "x2":
regs.Sp = value
case "gp", "x3":
regs.Gp = value
case "tp", "x4":
regs.Tp = value
case "t0", "x5":
regs.T0 = value
case "t1", "x6":
regs.T1 = value
case "t2", "x7":
regs.T2 = value
case "s0", "fp", "x8":
regs.S0 = value
case "s1", "x9":
regs.S1 = value
case "a0", "x10":
regs.A0 = value
case "a1", "x11":
regs.A1 = value
case "a2", "x12":
regs.A2 = value
case "a3", "x13":
regs.A3 = value
case "a4", "x14":
regs.A4 = value
case "a5", "x15":
regs.A5 = value
case "a6", "x16":
regs.A6 = value
case "a7", "x17":
regs.A7 = value
case "s2", "x18":
regs.S2 = value
case "s3", "x19":
regs.S3 = value
case "s4", "x20":
regs.S4 = value
case "s5", "x21":
regs.S5 = value
case "s6", "x22":
regs.S6 = value
case "s7", "x23":
regs.S7 = value
case "s8", "x24":
regs.S8 = value
case "s9", "x25":
regs.S9 = value
case "s10", "x26":
regs.S10 = value
case "s11", "x27":
regs.S11 = value
case "t3", "x28":
regs.T3 = value
case "t4", "x29":
regs.T4 = value
case "t5", "x30":
regs.T5 = value
case "t6", "x31":
regs.T6 = value
default:
return fmt.Errorf("debug: unknown register %q", name)
}
return s.SetRegs(&regs)
}
// archReturnAddr reads the return address from RA (riscv64 convention).
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return regs.Ra, nil
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "SP" }
+2 -2
View File
@@ -17,8 +17,8 @@ import (
"time"
"unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
// Integration tests beyond the basic entry breakpoint: hardware watchpoints,
+310
View File
@@ -0,0 +1,310 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && (amd64 || arm64 || riscv64)
package debug
import (
"fmt"
"os"
"os/exec"
"path/filepath"
"runtime"
"strings"
"syscall"
"time"
"golang.org/x/sys/unix"
)
// Session is a ptrace debugging session controlling one debuggee process.
// The FreeBSD implementation sits behind the same surface as the Linux one:
// PT_TRACE_ME from the debuggee, PT_CONTINUE/PT_STEP from the tracer, and
// tracee memory through PT_IO (FreeBSD has no /proc/pid/mem to fall back
// on, so PT_IO is the only supported route).
type Session struct {
pid int
cmd *exec.Cmd
stopped bool
exited bool
codeBase uint64 // base address of the JIT code in the debuggee
tmpDir string // scratch directory of the session, removed on Kill
wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 dbw 0-15)
// lastSignal holds the signal of the most recent stop when that stop
// was a genuine signal-delivery-stop the caller must see (a fault such
// as SIGSEGV, SIGBUS, SIGFPE or SIGILL); 0 for breakpoint traps,
// single-steps, SIGSTOP and suppressed runtime signals.
lastSignal syscall.Signal
}
// Launch starts the debuggee subprocess (gasm debug --target ...) and
// attaches to it via ptrace.
func Launch(gasmBin, asmPath, funcName string, args []byte) (*Session, error) {
sess, _, err := LaunchWithBuffers(gasmBin, asmPath, funcName, args, "")
return sess, err
}
// LaunchWithBuffers is like Launch but also allocates buffers in the debuggee.
//
// It pins the calling goroutine to its OS thread and leaves it pinned: the
// debuggee's PT_TRACE_ME binds the tracer relation to the forking thread,
// and every ptrace request on the session must come from that same thread.
// All Session methods must therefore be called from the goroutine that
// launched the session (the REPL and coverage loops do exactly that).
func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec string) (*Session, []uint64, error) {
runtime.LockOSThread() // ptrace requests must stay on the forking thread
self, err := os.Executable()
if err != nil {
return nil, nil, fmt.Errorf("debug: cannot find gasm binary: %w", err)
}
if gasmBin != "" {
self = gasmBin
}
tmpDir, err := os.MkdirTemp("", "gasm-debug-*")
if err != nil {
return nil, nil, fmt.Errorf("debug: tempdir: %w", err)
}
argsFile := filepath.Join(tmpDir, "args.bin")
if err := os.WriteFile(argsFile, args, 0o644); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: write args: %w", err)
}
if bufSpec != "" {
if err := os.WriteFile(filepath.Join(tmpDir, "bufspec"), []byte(bufSpec), 0o644); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: write bufspec: %w", err)
}
}
cmd := exec.Command(self, "debug", "--func", funcName, "--args", argsFile, asmPath)
cmd.Env = append(os.Environ(), "GASM_DEBUG_TARGET=1", "GASM_DEBUG_TMP="+tmpDir)
cmd.Stdout = nil
cmd.Stderr = os.Stderr
cmd.SysProcAttr = &syscall.SysProcAttr{}
if err := cmd.Start(); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: start debuggee: %w", err)
}
s := &Session{pid: cmd.Process.Pid, cmd: cmd, tmpDir: tmpDir}
readyFile := filepath.Join(tmpDir, "ready")
for range 500 {
if _, err := os.Stat(readyFile); err == nil {
break
}
// A debuggee that died before signalling readiness (unknown
// function, unparseable source) writes its failure notice to the
// handshake directory; read it and fail fast. The poll never
// waits on the child: a wait here could consume the SIGSTOP park
// that waitStopped below must receive, hanging the launch.
if err := s.deadReason(); err != nil {
cmd.Wait()
os.RemoveAll(tmpDir)
return nil, nil, err
}
time.Sleep(5 * time.Millisecond)
}
// The debuggee parks itself with SIGSTOP once the JIT code is mapped.
// A Go tracee also reports SIGURG preemption as signal-delivery-stops,
// so the wait loops until a stop the debugger cares about instead of
// assuming the first event is the SIGSTOP.
if _, err := s.waitStopped(); err != nil {
cmd.Process.Kill()
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: wait for debuggee: %w", err)
}
s.stopped = true
// The debuggee reports its JIT mapping in the codebase file; that is
// the supported path on FreeBSD, where no /proc/pid/maps exists to
// scan for the RWX region as a fallback.
if data, err := os.ReadFile(filepath.Join(tmpDir, "codebase")); err == nil {
fmt.Sscanf(string(data), "%d", &s.codeBase)
}
var bufAddrs []uint64
if bufSpec != "" {
addrFile := filepath.Join(tmpDir, "bufaddrs")
if data, err := os.ReadFile(addrFile); err == nil {
for line := range strings.SplitSeq(strings.TrimSpace(string(data)), "\n") {
var addr uint64
if _, err := fmt.Sscanf(line, "%d", &addr); err == nil {
bufAddrs = append(bufAddrs, addr)
}
}
}
}
return s, bufAddrs, nil
}
// deadReason reports the debuggee's own failure notice, the file its
// failure paths write before exiting. A debuggee killed without a notice
// (a crash, SIGKILL) surfaces through waitStopped after the poll instead,
// which is why the poll's budget stays finite.
func (s *Session) deadReason() error {
data, err := os.ReadFile(filepath.Join(s.tmpDir, "dead"))
if err != nil {
return nil
}
return fmt.Errorf("debug: debuggee failed before signalling readiness: %s", strings.TrimSpace(string(data)))
}
// waitStopped consumes ptrace-stop events until one the debugger cares
// about arrives: SIGTRAP (a breakpoint or a completed single-step), the
// debuggee's own SIGSTOP, or a genuine signal-delivery-stop. A Go tracee's
// runtime raises SIGURG for asynchronous preemption, and every signal on a
// traced thread surfaces as a signal-delivery-stop, so SIGURG is suppressed
// and the tracee resumed without it. Every other signal (SIGSEGV, SIGBUS,
// SIGFPE, SIGILL, ...) is returned to the caller: resuming with signal 0
// would restart the faulting instruction and fault forever, so a faulting
// kernel must surface as a stop the caller reports.
func (s *Session) waitStopped() (syscall.Signal, error) {
for {
var ws syscall.WaitStatus
if _, err := syscall.Wait4(s.pid, &ws, syscall.WUNTRACED, nil); err != nil {
return 0, err
}
if ws.Exited() {
s.exited = true
return 0, fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
}
if ws.Signaled() {
s.exited = true
return 0, fmt.Errorf("debuggee killed by signal %v", ws.Signal())
}
switch sig := ws.StopSignal(); sig {
case syscall.SIGTRAP, syscall.SIGSTOP:
s.stopped = true
s.lastSignal = 0
return sig, nil
case syscall.SIGURG:
// Go runtime asynchronous preemption: resume the tracee
// without delivering the signal.
s.lastSignal = 0
if err := unix.PtraceCont(s.pid, 0); err != nil {
return 0, fmt.Errorf("debug: PT_CONTINUE: %w", err)
}
default:
// A genuine signal-delivery-stop. Report it; the caller
// decides how to proceed.
s.stopped = true
s.lastSignal = sig
return sig, nil
}
}
}
// LastSignal returns the signal of the most recent stop when that stop was
// a genuine signal-delivery-stop (a fault such as SIGSEGV, SIGFPE, SIGILL
// or SIGBUS), and 0 for breakpoint traps, single-steps, SIGSTOP and
// suppressed runtime signals.
func (s *Session) LastSignal() syscall.Signal { return s.lastSignal }
// Peek reads a word (8 bytes) from the debuggee's memory at addr, through
// PT_IO with PIOD_READ_D.
func (s *Session) Peek(addr uint64) (uint64, error) {
var buf [8]byte
if _, err := unix.PtraceIO(unix.PIOD_READ_D, s.pid, uintptr(addr), buf[:], len(buf)); err != nil {
return 0, fmt.Errorf("debug: read mem %#x: %w", addr, err)
}
return uint64(buf[0]) | uint64(buf[1])<<8 | uint64(buf[2])<<16 | uint64(buf[3])<<24 |
uint64(buf[4])<<32 | uint64(buf[5])<<40 | uint64(buf[6])<<48 | uint64(buf[7])<<56, nil
}
// Poke writes a word (8 bytes) to the debuggee's memory at addr, through
// PT_IO with PIOD_WRITE_D.
func (s *Session) Poke(addr, val uint64) error {
buf := []byte{byte(val), byte(val >> 8), byte(val >> 16), byte(val >> 24),
byte(val >> 32), byte(val >> 40), byte(val >> 48), byte(val >> 56)}
if _, err := unix.PtraceIO(unix.PIOD_WRITE_D, s.pid, uintptr(addr), buf, len(buf)); err != nil {
return fmt.Errorf("debug: write mem %#x: %w", addr, err)
}
return nil
}
// ReadMemory reads len bytes from the debuggee's memory at addr in one
// PT_IO request, the shape the request is built for.
func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) {
out := make([]byte, length)
n, err := unix.PtraceIO(unix.PIOD_READ_D, s.pid, uintptr(addr), out, length)
return out[:n], err
}
// WriteMemory writes bytes to the debuggee's memory at addr in one PT_IO
// request.
func (s *Session) WriteMemory(addr uint64, data []byte) error {
_, err := unix.PtraceIO(unix.PIOD_WRITE_D, s.pid, uintptr(addr), data, len(data))
return err
}
// Step executes a single instruction in the debuggee.
func (s *Session) Step() error {
if s.exited {
return fmt.Errorf("debug: debuggee has exited")
}
if err := unix.PtraceSingleStep(s.pid); err != nil {
return fmt.Errorf("debug: PT_STEP: %w", err)
}
_, err := s.waitStopped()
return err
}
// Continue resumes execution until the next breakpoint or exit.
func (s *Session) Continue() error {
if s.exited {
return fmt.Errorf("debug: debuggee has exited")
}
if err := unix.PtraceCont(s.pid, 0); err != nil {
return fmt.Errorf("debug: PT_CONTINUE: %w", err)
}
_, err := s.waitStopped()
return err
}
// Exited returns true if the debuggee has terminated.
func (s *Session) Exited() bool { return s.exited }
// Pid returns the debuggee's process ID.
func (s *Session) Pid() int { return s.pid }
// CodeBase returns the base address of the JIT code in the debuggee.
func (s *Session) CodeBase() uint64 { return s.codeBase }
// Kill terminates the debuggee and removes the session's scratch
// directory, so a successful session leaves no gasm-debug-* debris behind.
func (s *Session) Kill() {
if !s.exited {
syscall.Kill(s.pid, syscall.SIGKILL)
syscall.Wait4(s.pid, nil, 0, nil)
s.exited = true
}
if s.cmd != nil && s.cmd.Process != nil {
s.cmd.Wait()
}
if s.tmpDir != "" {
os.RemoveAll(s.tmpDir)
s.tmpDir = ""
}
}
// execRange is one executable mapping of the debuggee.
type execRange struct {
lo, hi uint64
}
// execRanges is a stub on FreeBSD: there is no /proc/pid/maps to parse,
// and procfs(5) is not guaranteed to be mounted. The callers degrade
// gracefully: archReturnAddr falls back to the raw stack convention and
// the mapping scan is skipped.
func execRanges(pid int) []execRange { return nil }
// findRWXMapping is a stub on FreeBSD for the same reason: the codebase
// handshake file is the supported way the JIT region is located.
func findRWXMapping(pid int) uint64 { return 0 }
+160
View File
@@ -0,0 +1,160 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
import (
"encoding/binary"
"fmt"
"unsafe"
"golang.org/x/sys/unix"
)
// GetRegs reads the general-purpose registers of the stopped debuggee and
// converts the FreeBSD struct reg into the portable layout.
func (s *Session) GetRegs() (Regs, error) {
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return Regs{}, fmt.Errorf("debug: PT_GETREGS: %w", err)
}
return Regs{
R15: uint64(ur.R15),
R14: uint64(ur.R14),
R13: uint64(ur.R13),
R12: uint64(ur.R12),
R11: uint64(ur.R11),
R10: uint64(ur.R10),
R9: uint64(ur.R9),
R8: uint64(ur.R8),
RDI: uint64(ur.Rdi),
RSI: uint64(ur.Rsi),
RBP: uint64(ur.Rbp),
RBX: uint64(ur.Rbx),
RDX: uint64(ur.Rdx),
RCX: uint64(ur.Rcx),
RAX: uint64(ur.Rax),
RIP: uint64(ur.Rip),
CS: uint64(ur.Cs),
RFLAGS: uint64(ur.Rflags),
RSP: uint64(ur.Rsp),
SS: uint64(ur.Ss),
FS: uint64(ur.Fs),
GS: uint64(ur.Gs),
DS: uint64(ur.Ds),
ES: uint64(ur.Es),
}, nil
}
// SetRegs writes the general-purpose registers of the stopped debuggee.
func (s *Session) SetRegs(regs *Regs) error {
// Read-modify-write keeps the fields FreeBSD owns (trapno, err) intact.
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return fmt.Errorf("debug: PT_GETREGS: %w", err)
}
ur.R15 = int64(regs.R15)
ur.R14 = int64(regs.R14)
ur.R13 = int64(regs.R13)
ur.R12 = int64(regs.R12)
ur.R11 = int64(regs.R11)
ur.R10 = int64(regs.R10)
ur.R9 = int64(regs.R9)
ur.R8 = int64(regs.R8)
ur.Rdi = int64(regs.RDI)
ur.Rsi = int64(regs.RSI)
ur.Rbp = int64(regs.RBP)
ur.Rbx = int64(regs.RBX)
ur.Rdx = int64(regs.RDX)
ur.Rcx = int64(regs.RCX)
ur.Rax = int64(regs.RAX)
ur.Rip = int64(regs.RIP)
ur.Cs = int64(regs.CS)
ur.Rflags = int64(regs.RFLAGS)
ur.Rsp = int64(regs.RSP)
ur.Ss = int64(regs.SS)
return unix.PtraceSetRegs(s.pid, &ur)
}
// FPRegs holds the x87 FPU and SSE (XMM) register state, the FXSAVE image
// the FreeBSD struct fpreg mirrors: XMM0-15 at the same offsets.
type FPRegs struct {
XMM [16][16]byte // XMM0-15
}
// GetFPRegs retrieves the FPU/SSE register state via PT_GETFPREGS. The
// FreeBSD struct fpreg mirrors the FXSAVE image: the x87 environment and
// stack in Env/Acc, XMM0-15 in Xacc.
func (s *Session) GetFPRegs() (FPRegs, error) {
var fp FPRegs
var fr unix.FpReg
if err := unix.PtraceGetFpRegs(s.pid, &fr); err != nil {
return fp, fmt.Errorf("debug: PT_GETFPREGS: %w", err)
}
for i := range 16 {
copy(fp.XMM[i][:], fr.Xacc[i][:])
}
return fp, nil
}
// VectorRegs holds the YMM register state.
type VectorRegs struct {
YMM [16][32]byte // YMM0-15 (full 256-bit values)
}
// The XSAVE area the PT_GETXSTATE request returns follows the architectural
// layout (Intel SDM vol 1, "XSAVE"): the 512-byte legacy FXSAVE image (x87
// state in 0-159, XMM0-15 in 160-511), then the 64-byte xsave header whose
// first 8 bytes are xstate_bv, then one component per set feature bit, each
// 64-byte aligned. The YMM high halves are the first extended component,
// at offset 576; XFEATURE_STATE_BIT_AVX is bit 2 of xstate_bv.
const (
xsaveXMMOffset = 160
xsaveHeaderOffset = 512
xsaveBVOffset = xsaveHeaderOffset
ymmOffset = xsaveHeaderOffset + 64 // 576
ymmSize = 256 // 16 registers, 16 bytes each
xfeatureMaskYMM = 1 << 2
xstateMaxBuffer = 4096 // PT_GETXSTATE_INFO bounds the size far below this
)
// GetVectorRegs retrieves the YMM registers via PT_GETXSTATE. The low
// (XMM) halves always come from the legacy image; the high halves are
// copied only when xstate_bv reports the AVX state, and read as zero
// otherwise. When the request fails the FP image still provides correct
// XMM halves, so that is the fallback.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
var v VectorRegs
buf := make([]byte, xstateMaxBuffer)
n, _, errno := unix.Syscall6(
unix.SYS_PTRACE,
uintptr(unix.PT_GETXSTATE),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(&buf[0])),
0, 0,
)
if errno != 0 {
fp, err := s.GetFPRegs()
if err != nil {
return v, err
}
for i := range 16 {
copy(v.YMM[i][:16], fp.XMM[i][:])
}
return v, nil
}
for i := range 16 {
copy(v.YMM[i][:16], buf[xsaveXMMOffset+16*i:xsaveXMMOffset+16*i+16])
}
if int(n) >= ymmOffset+ymmSize {
if binary.LittleEndian.Uint64(buf[xsaveBVOffset:xsaveBVOffset+8])&xfeatureMaskYMM != 0 {
for i := range 16 {
copy(v.YMM[i][16:], buf[ymmOffset+16*i:ymmOffset+16*i+16])
}
}
}
return v, nil
}
+114
View File
@@ -0,0 +1,114 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && arm64
package debug
import (
"fmt"
"golang.org/x/sys/unix"
)
// GetRegs reads the general-purpose registers of the stopped debuggee and
// converts the FreeBSD struct reg (x[30], lr, sp, elr, spsr) into the
// portable layout.
func (s *Session) GetRegs() (Regs, error) {
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return Regs{}, fmt.Errorf("debug: PT_GETREGS: %w", err)
}
return Regs{
X0: ur.X[0],
X1: ur.X[1],
X2: ur.X[2],
X3: ur.X[3],
X4: ur.X[4],
X5: ur.X[5],
X6: ur.X[6],
X7: ur.X[7],
X8: ur.X[8],
X9: ur.X[9],
X10: ur.X[10],
X11: ur.X[11],
X12: ur.X[12],
X13: ur.X[13],
X14: ur.X[14],
X15: ur.X[15],
X16: ur.X[16],
X17: ur.X[17],
X18: ur.X[18],
X19: ur.X[19],
X20: ur.X[20],
X21: ur.X[21],
X22: ur.X[22],
X23: ur.X[23],
X24: ur.X[24],
X25: ur.X[25],
X26: ur.X[26],
X27: ur.X[27],
X28: ur.X[28],
X29: ur.X[29],
X30: ur.Lr,
SP: ur.Sp,
PC: ur.Elr,
PSTATE: uint64(ur.Spsr),
}, nil
}
// SetRegs writes the general-purpose registers of the stopped debuggee.
func (s *Session) SetRegs(regs *Regs) error {
var ur unix.Reg
ur.X = [30]uint64{
regs.X0, regs.X1, regs.X2, regs.X3, regs.X4, regs.X5, regs.X6,
regs.X7, regs.X8, regs.X9, regs.X10, regs.X11, regs.X12, regs.X13,
regs.X14, regs.X15, regs.X16, regs.X17, regs.X18, regs.X19, regs.X20,
regs.X21, regs.X22, regs.X23, regs.X24, regs.X25, regs.X26, regs.X27,
regs.X28, regs.X29,
}
ur.Lr = regs.X30
ur.Sp = regs.SP
ur.Elr = regs.PC
ur.Spsr = uint32(regs.PSTATE)
return unix.PtraceSetRegs(s.pid, &ur)
}
// FPRegs holds the arm64 FP/NEON register state: the 32 128-bit V
// registers, then FPSR and FPCR (the user_fpsimd shape).
type FPRegs struct {
V [32][16]byte // V0-V31 (128-bit NEON/FP registers)
FPSR uint32
FPCR uint32
}
// GetFPRegs retrieves the FP/NEON register state via PT_GETFPREGS. The
// FreeBSD struct fpreg holds the 32 128-bit V registers followed by FPSR
// and FPCR, the user_fpsimd shape.
func (s *Session) GetFPRegs() (FPRegs, error) {
var fp FPRegs
var fr unix.FpReg
if err := unix.PtraceGetFpRegs(s.pid, &fr); err != nil {
return fp, fmt.Errorf("debug: PT_GETFPREGS: %w", err)
}
for i := range 32 {
copy(fp.V[i][:], fr.Q[i][:])
}
return fp, nil
}
// VectorRegs holds the full SIMD register state.
type VectorRegs struct {
V [32][16]byte // V0-V31 (128-bit)
}
// GetVectorRegs retrieves the SIMD registers.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
var v VectorRegs
fp, err := s.GetFPRegs()
if err != nil {
return v, err
}
copy(v.V[:][:], fp.V[:][:])
return v, nil
}
+115
View File
@@ -0,0 +1,115 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && riscv64
package debug
import (
"fmt"
"golang.org/x/sys/unix"
)
// GetRegs reads the general-purpose registers of the stopped debuggee and
// converts the FreeBSD struct reg into the portable layout. Sstatus rides
// the kernel's struct but the portable surface carries the GPRs and PC.
func (s *Session) GetRegs() (Regs, error) {
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return Regs{}, fmt.Errorf("debug: PT_GETREGS: %w", err)
}
return Regs{
PC: ur.Sepc,
Ra: ur.Ra,
Sp: ur.Sp,
Gp: ur.Gp,
Tp: ur.Tp,
T0: ur.T[0],
T1: ur.T[1],
T2: ur.T[2],
S0: ur.S[0],
S1: ur.S[1],
A0: ur.A[0],
A1: ur.A[1],
A2: ur.A[2],
A3: ur.A[3],
A4: ur.A[4],
A5: ur.A[5],
A6: ur.A[6],
A7: ur.A[7],
S2: ur.S[2],
S3: ur.S[3],
S4: ur.S[4],
S5: ur.S[5],
S6: ur.S[6],
S7: ur.S[7],
S8: ur.S[8],
S9: ur.S[9],
S10: ur.S[10],
S11: ur.S[11],
T3: ur.T[3],
T4: ur.T[4],
T5: ur.T[5],
T6: ur.T[6],
}, nil
}
// SetRegs writes the general-purpose registers of the stopped debuggee.
// Read-modify-write keeps sstatus, which the kernel owns, intact.
func (s *Session) SetRegs(regs *Regs) error {
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return fmt.Errorf("debug: PT_GETREGS: %w", err)
}
ur.Sepc = regs.PC
ur.Ra = regs.Ra
ur.Sp = regs.Sp
ur.Gp = regs.Gp
ur.Tp = regs.Tp
ur.T = [7]uint64{regs.T0, regs.T1, regs.T2, regs.T3, regs.T4, regs.T5, regs.T6}
ur.S = [12]uint64{regs.S0, regs.S1, regs.S2, regs.S3, regs.S4, regs.S5,
regs.S6, regs.S7, regs.S8, regs.S9, regs.S10, regs.S11}
ur.A = [8]uint64{regs.A0, regs.A1, regs.A2, regs.A3, regs.A4, regs.A5, regs.A6, regs.A7}
return unix.PtraceSetRegs(s.pid, &ur)
}
// FPRegs holds the RISC-V FP register state (32 64-bit FP registers plus
// fcsr).
type FPRegs struct {
F [32]uint64 // F0-F31 (64-bit FP registers)
FCSR uint32
}
// GetFPRegs retrieves the FP register state via PT_GETFPREGS. The FreeBSD
// struct fpreg carries each 64-bit FP register in a 128-bit slot (fp_x is
// the flat [64]-word area the x/sys type renders as [32][2]); the low word
// holds the register, and FCSR rides the tail.
func (s *Session) GetFPRegs() (FPRegs, error) {
var fp FPRegs
var fr unix.FpReg
if err := unix.PtraceGetFpRegs(s.pid, &fr); err != nil {
return fp, fmt.Errorf("debug: PT_GETFPREGS: %w", err)
}
for i := range 32 {
fp.F[i] = fr.X[i][0]
}
fp.FCSR = uint32(fr.Fcsr)
return fp, nil
}
// VectorRegs holds the FP register state shown by the regs command
// (riscv64 has 32 64-bit FP registers and fcsr).
type VectorRegs struct {
F [32]uint64
FCSR uint32
}
// GetVectorRegs retrieves the FP registers.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
fp, err := s.GetFPRegs()
if err != nil {
return VectorRegs{}, err
}
return VectorRegs{F: fp.F, FCSR: fp.FCSR}, nil
}
+91
View File
@@ -0,0 +1,91 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
import (
"os/exec"
"path/filepath"
"runtime"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
// TestLaunchAndBreakpoint is the FreeBSD twin of the Linux integration
// test: it drives the whole launch, breakpoint, trap and register-rewind
// flow end to end. It needs a real FreeBSD kernel (ptrace does not work
// under emulation), so it only runs where it can.
func TestLaunchAndBreakpoint(t *testing.T) {
if runtime.GOARCH != "amd64" {
t.Skip("runs only on amd64 hosts")
}
// The tracer is the OS thread that forked the debuggee (PT_TRACE_ME
// binds the relation to that thread); every ptrace request must come
// from the same thread, so pin the test goroutine to one thread.
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := filepath.Join(t.TempDir(), "gasm")
out, err := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm").CombinedOutput()
if err != nil {
t.Fatalf("build gasm: %v: %s", err, out)
}
const kernelPath = "../testdata/verify/basic_amd64.s"
k, err := verify.Load(kernelPath)
if err != nil {
t.Fatalf("Load: %v", err)
}
t.Cleanup(k.Close)
fl, err := k.Func("wideCopy")
if err != nil {
t.Fatalf("Func: %v", err)
}
sess, err := Launch(bin, kernelPath, "wideCopy", make([]byte, fl.Args))
if err != nil {
t.Fatalf("Launch: %v", err)
}
t.Cleanup(sess.Kill)
bm := NewBreakpoints(sess)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
// The INT3 must be visible in the debuggee's memory.
word, err := sess.Peek(entry)
if err != nil {
t.Fatalf("Peek: %v", err)
}
if b := word & 0xFF; b != 0xCC {
t.Fatalf("int3 not patched: first byte %#02x at %#x", b, entry)
}
// The debuggee raises a second SIGSTOP after the launch barrier (the
// child's RunTarget marks its entry), so like the REPL and the cover
// mode the test keeps resuming until the breakpoint trap arrives.
for range 10 {
if err := sess.Continue(); err != nil {
t.Fatalf("Continue: %v", err)
}
if sess.Exited() {
t.Fatal("debuggee exited instead of trapping on the breakpoint")
}
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if bp := bm.HandleTrap(&regs); bp != nil {
if bp.Addr != entry {
t.Fatalf("trap at %#x, want %#x", bp.Addr, entry)
}
return // trap on the entry breakpoint: the whole flow works
}
}
t.Fatal("no breakpoint trap after 10 resumes")
}
+11 -2
View File
@@ -14,17 +14,26 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
// buildGasm produces the gasm binary the debugger spawns as its debuggee.
// Every live ptrace test funnels through here, so this is also where the
// deliberate-run boundary sits: under -short (the push pipeline's mode) the
// live sessions skip, because a real debuggee's launch handshake needs the
// machine to itself and a starved single-core runner turns each one into a
// timeout that burns the step's whole budget. The local test gate and the
// dispatched workflows run them in full.
func buildGasm(t *testing.T) string {
t.Helper()
if testing.Short() {
t.Skip("live ptrace session: skipped in -short mode")
}
if p := os.Getenv("GASM_TEST_BIN"); p != "" {
return p
}
bin := filepath.Join(t.TempDir(), "gasm")
cmd := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm")
cmd := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm")
out, err := cmd.CombinedOutput()
if err != nil {
t.Fatalf("build gasm: %v: %s", err, out)
+45 -22
View File
@@ -91,6 +91,16 @@ func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec s
if _, err := os.Stat(readyFile); err == nil {
break
}
// A debuggee that died before signalling readiness (unknown
// function, unparseable source) writes its failure notice to the
// handshake directory; read it and fail fast. The poll never
// waits on the child: a wait here could consume the SIGSTOP park
// that waitStopped below must receive, hanging the launch.
if err := s.deadReason(); err != nil {
cmd.Wait()
os.RemoveAll(tmpDir)
return nil, nil, err
}
time.Sleep(5 * time.Millisecond)
}
@@ -131,6 +141,18 @@ func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec s
return s, bufAddrs, nil
}
// deadReason reports the debuggee's own failure notice, the file its
// failure paths write before exiting. A debuggee killed without a notice
// (a crash, SIGKILL) surfaces through waitStopped after the poll instead,
// which is why the poll's budget stays finite.
func (s *Session) deadReason() error {
data, err := os.ReadFile(filepath.Join(s.tmpDir, "dead"))
if err != nil {
return nil
}
return fmt.Errorf("debug: debuggee failed before signalling readiness: %s", strings.TrimSpace(string(data)))
}
// waitStopped consumes ptrace-stop events until one the debugger cares
// about arrives: SIGTRAP (a breakpoint or a completed single-step), the
// debuggee's own SIGSTOP, or a genuine signal-delivery-stop. A Go tracee's
@@ -219,40 +241,41 @@ func (s *Session) Poke(addr, val uint64) error {
return nil
}
// ReadMemory reads len bytes from the debuggee's memory at addr.
// ReadMemory reads len bytes from the debuggee's memory at addr. The read
// covers exactly the requested range: the old word-at-a-time loop read a
// whole 8-byte word for the final partial word, so a request that ended
// inside the last mapped page failed whenever the following page was
// unmapped, even though every requested byte was readable.
func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) {
out := make([]byte, length)
for i := 0; i < length; i += 8 {
word, err := s.Peek(addr + uint64(i))
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
if err != nil {
return out[:i], err
}
for j := 0; j < 8 && i+j < length; j++ {
out[i+j] = byte(word >> (8 * j))
return out, fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
}
defer mem.Close()
n, err := mem.ReadAt(out, int64(addr))
if err != nil {
return out[:n], fmt.Errorf("debug: read mem %#x: %w", addr, err)
}
return out, nil
}
// WriteMemory writes bytes to the debuggee's memory at addr.
// WriteMemory writes bytes to the debuggee's memory at addr. The write
// covers exactly the given bytes: /proc/pid/mem accepts writes of any
// length at any offset, so the word loop's read-modify-write of the final
// partial word (which read past the requested range and failed on an
// unmapped following page) is unnecessary.
func (s *Session) WriteMemory(addr uint64, data []byte) error {
for i := 0; i < len(data); i += 8 {
end := min(i+8, len(data))
var word uint64
for j := 0; j < end-i; j++ {
word |= uint64(data[i+j]) << (8 * j)
if len(data) == 0 {
return nil
}
if end-i < 8 {
existing, err := s.Peek(addr + uint64(i))
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_WRONLY, 0)
if err != nil {
return err
}
mask := ^((uint64(1) << (8 * (end - i))) - 1)
word = (existing & mask) | word
}
if err := s.Poke(addr+uint64(i), word); err != nil {
return err
return fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
}
defer mem.Close()
if _, err := mem.WriteAt(data, int64(addr)); err != nil {
return fmt.Errorf("debug: write mem %#x: %w", addr, err)
}
return nil
}
+98
View File
@@ -0,0 +1,98 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
// Regs holds the full general-purpose register set of a traced process
// (the FreeBSD amd64 struct reg layout, sys/x86/include/reg.h). FreeBSD
// reports segment selectors (FS/GS/ES/DS), not the bases the Linux ptrace
// surface carries, and has no ORIG_RAX slot.
type Regs struct {
R15 uint64
R14 uint64
R13 uint64
R12 uint64
RBP uint64
RBX uint64
R11 uint64
R10 uint64
R9 uint64
R8 uint64
RAX uint64
RCX uint64
RDX uint64
RSI uint64
RDI uint64
RIP uint64
CS uint64
RFLAGS uint64
RSP uint64
SS uint64
FS uint64
GS uint64
DS uint64
ES uint64
}
// GetPC returns the program counter.
func (r *Regs) GetPC() uint64 { return r.RIP }
// SetPC sets the program counter.
func (r *Regs) SetPC(pc uint64) { r.RIP = pc }
// GetSP returns the stack pointer.
func (r *Regs) GetSP() uint64 { return r.RSP }
// RegValue returns the value of the named register, or false if unknown.
func (r *Regs) RegValue(name string) (uint64, bool) {
switch name {
case "rax", "eax", "ax", "al":
return r.RAX, true
case "rbx", "ebx", "bx", "bl":
return r.RBX, true
case "rcx", "ecx", "cx", "cl":
return r.RCX, true
case "rdx", "edx", "dx", "dl":
return r.RDX, true
case "rsi", "esi", "si":
return r.RSI, true
case "rdi", "edi", "di":
return r.RDI, true
case "rbp", "ebp", "bp":
return r.RBP, true
case "rsp", "esp", "sp":
return r.RSP, true
case "r8":
return r.R8, true
case "r9":
return r.R9, true
case "r10":
return r.R10, true
case "r11":
return r.R11, true
case "r12":
return r.R12, true
case "r13":
return r.R13, true
case "r14":
return r.R14, true
case "r15":
return r.R15, true
case "rip", "eip":
return r.RIP, true
default:
return 0, false
}
}
// breakpointInsn is the software breakpoint instruction.
var breakpointInsn = []byte{0xCC} // INT3
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap. INT3 leaves the hardware PC on the following instruction (Intel
// SDM vol 3, "Debug Exceptions") and the FreeBSD T_BPTFLT path delivers
// that frame unmodified (sys/amd64/amd64/trap.c), so the trap address is
// PC-1, the same correction the Linux side applies.
const breakpointPCAdjust = 1
+139
View File
@@ -0,0 +1,139 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && arm64
package debug
// Regs holds the full general-purpose register set of a traced process
// (the FreeBSD arm64 struct reg layout, sys/arm64/include/reg.h: x[30], lr,
// sp, elr, spsr).
type Regs struct {
X0 uint64
X1 uint64
X2 uint64
X3 uint64
X4 uint64
X5 uint64
X6 uint64
X7 uint64
X8 uint64
X9 uint64
X10 uint64
X11 uint64
X12 uint64
X13 uint64
X14 uint64
X15 uint64
X16 uint64
X17 uint64
X18 uint64
X19 uint64
X20 uint64
X21 uint64
X22 uint64
X23 uint64
X24 uint64
X25 uint64
X26 uint64
X27 uint64
X28 uint64
X29 uint64 // FP (frame pointer)
X30 uint64 // LR (link register)
SP uint64
PC uint64
PSTATE uint64
}
// GetPC returns the program counter.
func (r *Regs) GetPC() uint64 { return r.PC }
// SetPC sets the program counter.
func (r *Regs) SetPC(pc uint64) { r.PC = pc }
// GetSP returns the stack pointer.
func (r *Regs) GetSP() uint64 { return r.SP }
// RegValue returns the value of the named register, or false if unknown.
func (r *Regs) RegValue(name string) (uint64, bool) {
switch name {
case "x0":
return r.X0, true
case "x1":
return r.X1, true
case "x2":
return r.X2, true
case "x3":
return r.X3, true
case "x4":
return r.X4, true
case "x5":
return r.X5, true
case "x6":
return r.X6, true
case "x7":
return r.X7, true
case "x8":
return r.X8, true
case "x9":
return r.X9, true
case "x10":
return r.X10, true
case "x11":
return r.X11, true
case "x12":
return r.X12, true
case "x13":
return r.X13, true
case "x14":
return r.X14, true
case "x15":
return r.X15, true
case "x16":
return r.X16, true
case "x17":
return r.X17, true
case "x18":
return r.X18, true
case "x19":
return r.X19, true
case "x20":
return r.X20, true
case "x21":
return r.X21, true
case "x22":
return r.X22, true
case "x23":
return r.X23, true
case "x24":
return r.X24, true
case "x25":
return r.X25, true
case "x26":
return r.X26, true
case "x27":
return r.X27, true
case "x28":
return r.X28, true
case "x29", "fp":
return r.X29, true
case "x30", "lr":
return r.X30, true
case "sp":
return r.SP, true
case "pc":
return r.PC, true
default:
return 0, false
}
}
// breakpointInsn is the software breakpoint instruction (BRK #0).
var breakpointInsn = []byte{0x00, 0x00, 0x20, 0xD4} // BRK #0
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap: 0. The BRK synchronous exception leaves ELR_EL0 on the BRK
// itself (ARM DDI 0487), and the FreeBSD EXCP_BRKPT_EL0 handler delivers
// the frame's elr unmodified (sys/arm64/arm64/trap.c), so the trap address
// is the PC as reported.
const breakpointPCAdjust = 0
+134
View File
@@ -0,0 +1,134 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && riscv64
package debug
// Regs holds the full general-purpose register set of a traced process
// (the FreeBSD riscv64 struct reg layout: ra, sp, gp, tp, t0-t6, s0-s11,
// a0-a7, sepc, sstatus).
type Regs struct {
PC uint64 // sepc
Ra uint64 // x1 (return address)
Sp uint64 // x2
Gp uint64 // x3
Tp uint64 // x4
T0 uint64 // x5
T1 uint64 // x6
T2 uint64 // x7
S0 uint64 // x8 (frame pointer)
S1 uint64 // x9
A0 uint64 // x10
A1 uint64 // x11
A2 uint64 // x12
A3 uint64 // x13
A4 uint64 // x14
A5 uint64 // x15
A6 uint64 // x16
A7 uint64 // x17
S2 uint64 // x18
S3 uint64 // x19
S4 uint64 // x20
S5 uint64 // x21
S6 uint64 // x22
S7 uint64 // x23
S8 uint64 // x24
S9 uint64 // x25
S10 uint64 // x26
S11 uint64 // x27
T3 uint64 // x28
T4 uint64 // x29
T5 uint64 // x30
T6 uint64 // x31
}
// GetPC returns the program counter.
func (r *Regs) GetPC() uint64 { return r.PC }
// SetPC sets the program counter.
func (r *Regs) SetPC(pc uint64) { r.PC = pc }
// GetSP returns the stack pointer.
func (r *Regs) GetSP() uint64 { return r.Sp }
// RegValue returns the value of the named register, or false if unknown.
func (r *Regs) RegValue(name string) (uint64, bool) {
switch name {
case "pc":
return r.PC, true
case "ra", "x1":
return r.Ra, true
case "sp", "x2":
return r.Sp, true
case "gp", "x3":
return r.Gp, true
case "tp", "x4":
return r.Tp, true
case "t0", "x5":
return r.T0, true
case "t1", "x6":
return r.T1, true
case "t2", "x7":
return r.T2, true
case "s0", "fp", "x8":
return r.S0, true
case "s1", "x9":
return r.S1, true
case "a0", "x10":
return r.A0, true
case "a1", "x11":
return r.A1, true
case "a2", "x12":
return r.A2, true
case "a3", "x13":
return r.A3, true
case "a4", "x14":
return r.A4, true
case "a5", "x15":
return r.A5, true
case "a6", "x16":
return r.A6, true
case "a7", "x17":
return r.A7, true
case "s2", "x18":
return r.S2, true
case "s3", "x19":
return r.S3, true
case "s4", "x20":
return r.S4, true
case "s5", "x21":
return r.S5, true
case "s6", "x22":
return r.S6, true
case "s7", "x23":
return r.S7, true
case "s8", "x24":
return r.S8, true
case "s9", "x25":
return r.S9, true
case "s10", "x26":
return r.S10, true
case "s11", "x27":
return r.S11, true
case "t3", "x28":
return r.T3, true
case "t4", "x29":
return r.T4, true
case "t5", "x30":
return r.T5, true
case "t6", "x31":
return r.T6, true
default:
return 0, false
}
}
// breakpointInsn is the software breakpoint instruction (EBREAK).
var breakpointInsn = []byte{0x73, 0x00, 0x10, 0x00} // ebreak
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap: 0. The EBREAK synchronous exception leaves sepc on the ebreak
// itself (RISC-V privileged architecture), so the trap address is the PC as
// reported.
const breakpointPCAdjust = 0
+63 -15
View File
@@ -1,7 +1,7 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
//go:build linux || (freebsd && (amd64 || arm64 || riscv64))
package debug
@@ -71,7 +71,12 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
case "step", "s":
n := 1
if len(parts) > 1 {
n, _ = strconv.Atoi(parts[1])
v, err := strconv.Atoi(parts[1])
if err != nil || v < 0 {
fmt.Printf("invalid count: %s\n", parts[1])
continue
}
n = v
}
for range n {
if s.Exited() {
@@ -82,8 +87,13 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Println(err)
break
}
if sig := s.LastSignal(); sig != 0 {
regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
break
}
if !s.Exited() {
}
if !s.Exited() && s.LastSignal() == 0 {
regs, _ := s.GetRegs()
pc := regs.GetPC()
text, _, _ := s.Disassemble(pc)
@@ -106,16 +116,16 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
}
if err := s.Continue(); err != nil {
fmt.Println(err)
bm.Clear(afterAddr)
clearNextBp(bm, afterAddr)
continue
}
if s.Exited() {
bm.Clear(afterAddr)
clearNextBp(bm, afterAddr)
fmt.Println("debuggee exited")
continue
}
if sig := s.LastSignal(); sig != 0 {
bm.Clear(afterAddr)
clearNextBp(bm, afterAddr)
regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
continue
@@ -125,12 +135,17 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
// snapshot, and a stale SetRegs would clobber live state.
regs, _ = s.GetRegs()
bm.HandleTrap(&regs)
bm.Clear(afterAddr)
clearNextBp(bm, afterAddr)
} else {
if err := s.Step(); err != nil {
fmt.Println(err)
continue
}
if sig := s.LastSignal(); sig != 0 {
regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
continue
}
}
if !s.Exited() {
regs, _ := s.GetRegs()
@@ -156,16 +171,16 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
}
if err := s.Continue(); err != nil {
fmt.Println(err)
bm.Clear(retAddr)
clearNextBp(bm, retAddr)
continue
}
if s.Exited() {
bm.Clear(retAddr)
clearNextBp(bm, retAddr)
fmt.Println("debuggee exited")
continue
}
if sig := s.LastSignal(); sig != 0 {
bm.Clear(retAddr)
clearNextBp(bm, retAddr)
regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
continue
@@ -174,7 +189,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
// does: HandleTrap must see the PC the trap left behind.
regs, _ = s.GetRegs()
bm.HandleTrap(&regs)
bm.Clear(retAddr)
clearNextBp(bm, retAddr)
if s.Exited() {
fmt.Println("debuggee exited")
} else {
@@ -214,6 +229,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
break
}
regs, _ := s.GetRegs()
trapPC := regs.GetPC()
if bp := bm.HandleTrap(&regs); bp != nil {
// Execute the instruction under the restored breakpoint
// so the next continue cannot re-trap on the same
@@ -229,6 +245,13 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Printf("breakpoint hit: %s (func+%#x)\n", name, bp.Addr-codeBase-uint64(funcOffset))
break
}
// The trap matched no breakpoint of ours. When the PC still
// stands on the trapping instruction, resuming would re-execute
// it and trap forever, so surface the stop instead of spinning.
if TrapStray(s, reason, trapPC) {
fmt.Printf("SIGTRAP at %#x matches no breakpoint; the PC did not advance\n", trapPC-uint64(breakpointPCAdjust))
break
}
}
case "break", "b":
@@ -326,6 +349,10 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
length := 64
if len(parts) > 1 {
addr, _ = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
if addr == 0 {
fmt.Printf("unknown address: %s\n", parts[1])
continue
}
}
if len(parts) > 2 {
// A malformed or non-positive length would panic
@@ -399,9 +426,13 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
case "disas", "u":
n := 5
if len(parts) > 1 {
n, _ = strconv.Atoi(parts[1])
if n <= 0 {
n = 5
v, err := strconv.Atoi(parts[1])
if err != nil {
fmt.Printf("invalid count: %s\n", parts[1])
continue
}
if v > 0 {
n = v
}
}
regs, _ := s.GetRegs()
@@ -496,10 +527,18 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
typ = WatchRead
case "w":
typ = WatchWrite
default:
fmt.Printf("unknown watchpoint type: %s (want r or w)\n", parts[2])
continue
}
}
if len(parts) > 3 {
size, _ = strconv.Atoi(parts[3])
v, err := strconv.Atoi(parts[3])
if err != nil || v <= 0 {
fmt.Printf("invalid size: %s\n", parts[3])
continue
}
size = v
}
slot := s.FindFreeWatchpointSlot()
if slot < 0 {
@@ -543,6 +582,15 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
s.Kill()
}
// clearNextBp removes one of the temporary breakpoints the next and finish
// commands plant, reporting a failure instead of silently leaving the trap
// instruction behind in the debuggee.
func clearNextBp(bm *Breakpoints, addr uint64) {
if err := bm.Clear(addr); err != nil {
fmt.Printf("cannot remove temporary breakpoint at %#x: %v\n", addr, err)
}
}
func hexDump(addr uint64, data []byte) {
for i := 0; i < len(data); i += 16 {
end := min(i+16, len(data))
+73
View File
@@ -0,0 +1,73 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && (amd64 || arm64 || riscv64)
package debug
import (
"encoding/binary"
"syscall"
"unsafe"
"golang.org/x/sys/unix"
)
// FreeBSD TRAP_* si_code values (sys/signal.h). A breakpoint (INT3, BRK,
// EBREAK) arrives as TRAP_BRKPT on every supported architecture; TRAP_TRACE
// is shared by the completed single-step and the hardware watchpoint hit,
// so the watchpoint layer disambiguates from the debug registers.
const (
trapBRKPT = 1 // TRAP_BRKPT
trapTRACE = 2 // TRAP_TRACE
)
// StopReason describes why the debuggee stopped.
type StopReason int
const (
StopNone StopReason = iota
StopBreakpoint // software breakpoint hit
StopWatchpoint // hardware watchpoint triggered
StopSingleStep // single-step completed
StopSignal // stopped by a signal
StopExited // process exited
)
// StopInfo returns the reason the debuggee stopped and the faulting address
// (for watchpoints, the watched address that was accessed). FreeBSD has no
// PTRACE_GETSIGINFO; the stop's signal information comes from PT_LWPINFO,
// whose pl_siginfo carries the siginfo the kernel delivered. A ptrace stop
// with no signal behind it (a completed single-step, the initial attach)
// fills no siginfo at all.
func (s *Session) StopInfo() (StopReason, uint64) {
if s.exited {
return StopExited, 0
}
var info unix.PtraceLwpInfoStruct
if err := unix.PtraceLwpInfo(s.pid, &info); err != nil {
return StopNone, 0
}
// The siginfo layout is the FreeBSD siginfo_t: three leading ints
// (signo, errno, code), then the union, 8-byte aligned, whose _fault
// member puts the address at byte offset 16. The read is byte-wise
// because the blob's alignment is not guaranteed.
si := (*[64]byte)(unsafe.Pointer(&info.Siginfo))
signo := int32(binary.LittleEndian.Uint32(si[0:4]))
code := int32(binary.LittleEndian.Uint32(si[8:12]))
switch {
case signo == 0:
// A pure ptrace stop: single-step completion, attach, or the
// events the kernel resolves internally.
return StopSingleStep, 0
case signo != int32(syscall.SIGTRAP):
return StopSignal, uint64(code)
case code == trapBRKPT:
return StopBreakpoint, 0
case code == trapTRACE:
addr := binary.LittleEndian.Uint64(si[16:24])
return archStopTrace(s, addr)
default:
return StopSingleStep, 0
}
}
+206
View File
@@ -0,0 +1,206 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && (amd64 || arm64 || riscv64)
package debug
import (
"encoding/hex"
"fmt"
"os"
"runtime"
"strconv"
"strings"
"syscall"
"unsafe"
"golang.org/x/sys/unix"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
// mapRWX maps code into a read-write-execute region.
func mapRWX(code []byte) ([]byte, error) {
const pageSize = 4096
size := (len(code) + pageSize - 1) &^ (pageSize - 1)
mem, err := syscall.Mmap(-1, 0, size,
syscall.PROT_READ|syscall.PROT_WRITE|syscall.PROT_EXEC,
syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, err
}
copy(mem, code)
return mem, nil
}
// setupBuffers allocates buffers in the debuggee's memory.
func setupBuffers(spec string, args []byte, tmpDir string) ([]byte, error) {
type bufSpec struct {
name string
size int
pattern string
}
var specs []bufSpec
for part := range strings.SplitSeq(spec, ",") {
fields := strings.SplitN(part, ":", 3)
if len(fields) != 3 {
continue
}
size, err := strconv.Atoi(fields[1])
if err != nil || size <= 0 {
continue
}
specs = append(specs, bufSpec{name: fields[0], size: size, pattern: fields[2]})
}
if len(specs) == 0 {
return args, nil
}
var bufAddrs []uint64
for _, s := range specs {
buf, err := syscall.Mmap(-1, 0, s.size,
syscall.PROT_READ|syscall.PROT_WRITE,
syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, fmt.Errorf("mmap buffer %s: %w", s.name, err)
}
fillBuffer(buf, s.pattern)
bufAddrs = append(bufAddrs, uint64(uintptr(unsafe.Pointer(&buf[0]))))
}
addrFile, err := os.Create(tmpDir + "/bufaddrs")
if err != nil {
return nil, err
}
for _, addr := range bufAddrs {
fmt.Fprintf(addrFile, "%d\n", addr)
}
addrFile.Close()
return args, nil
}
// fillBuffer fills a buffer with the specified pattern.
func fillBuffer(buf []byte, pattern string) {
switch pattern {
case "zero":
case "ones":
for i := range buf {
buf[i] = 0xFF
}
case "seq":
for i := range buf {
buf[i] = byte(i)
}
default:
if data, err := hex.DecodeString(pattern); err == nil && len(data) > 0 {
for i := range buf {
buf[i] = data[i%len(data)]
}
}
}
}
// RunTarget is the debuggee entry point (gasm debug --target). A failure
// is marked in the handshake directory before the process exits, so the
// debugger's readiness poll fails fast on a dead debuggee instead of
// waiting out its whole budget.
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
err := runTarget(asmPath, funcName, argsFile, tmpDir)
if err != nil {
markDead(tmpDir, err.Error())
}
return err
}
func runTarget(asmPath, funcName, argsFile, tmpDir string) error {
src, err := os.ReadFile(asmPath)
if err != nil {
return fmt.Errorf("debug target: %w", err)
}
file, errs := parser.Parse(asmPath, string(src))
if len(errs) > 0 {
return fmt.Errorf("debug target: parse: %v", errs[0])
}
img, err := asm.AssembleFile(file)
if err != nil {
return fmt.Errorf("debug target: assemble: %w", err)
}
var fl *asm.FuncLayout
for i := range img.Funcs {
if img.Funcs[i].Name == funcName {
fl = &img.Funcs[i]
break
}
}
if fl == nil {
return fmt.Errorf("debug target: function %q not found", funcName)
}
code := img.Bytes()
exec, err := mapRWX(code)
if err != nil {
return fmt.Errorf("debug target: mmap: %w", err)
}
codeBase := uintptr(unsafe.Pointer(&exec[0]))
if err := os.WriteFile(tmpDir+"/codebase", []byte(fmt.Sprintf("%d", codeBase)), 0o644); err != nil {
return fmt.Errorf("debug target: write codebase: %w", err)
}
meta := fmt.Sprintf("%d %d %d", fl.Offset, fl.Size, fl.Args)
os.WriteFile(tmpDir+"/funcmeta", []byte(meta), 0o644)
labelsFile, _ := os.Create(tmpDir + "/labels")
if labelsFile != nil {
for label, off := range fl.Labels {
fmt.Fprintf(labelsFile, "%s %d\n", label, off)
}
labelsFile.Close()
}
args, err := os.ReadFile(argsFile)
if err != nil {
return fmt.Errorf("debug target: read args: %w", err)
}
if len(args) < fl.Args {
padded := make([]byte, fl.Args)
copy(padded, args)
args = padded
}
bufSpecFile := tmpDir + "/bufspec"
if bufSpec, err := os.ReadFile(bufSpecFile); err == nil && len(bufSpec) > 0 {
args, err = setupBuffers(string(bufSpec), args, tmpDir)
if err != nil {
return fmt.Errorf("debug target: setup buffers: %w", err)
}
}
runtime.LockOSThread()
if _, _, errno := unix.RawSyscall(unix.SYS_PTRACE, uintptr(unix.PT_TRACE_ME), 0, 0); errno != 0 {
return fmt.Errorf("debug target: PT_TRACE_ME: %v", errno)
}
os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644)
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
os.WriteFile(tmpDir+"/entry", []byte("ok"), 0o644)
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
fnAddr := codeBase + uintptr(fl.Offset)
stackArgs := make([]byte, fl.Args)
copy(stackArgs, args)
if _, callErr := verify.Call(fnAddr, stackArgs); callErr != nil {
os.Exit(1)
}
// Success returns to the caller, which exits with status 0; the JIT
// code has already run to its own trampoline by the time Call returns.
return nil
}
+15 -4
View File
@@ -12,13 +12,24 @@ import (
"syscall"
"unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
// RunTarget is the debuggee entry point (gasm debug --target).
// RunTarget is the debuggee entry point (gasm debug --target). A failure
// is marked in the handshake directory before the process exits, so the
// debugger's readiness poll fails fast on a dead debuggee instead of
// waiting out its whole budget.
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
err := runTarget(asmPath, funcName, argsFile, tmpDir)
if err != nil {
markDead(tmpDir, err.Error())
}
return err
}
func runTarget(asmPath, funcName, argsFile, tmpDir string) error {
src, err := os.ReadFile(asmPath)
if err != nil {
return fmt.Errorf("debug target: %w", err)

Some files were not shown because too many files have changed in this diff Show More