Compare commits

..
61 Commits
Author SHA1 Message Date
petrbalvin 5a8e9acbf3 feat(arch): add the extended-instruction layer with SVE arithmetic
Test / test (push) Successful in 3m38s
2026-10-02 20:39:33 +02:00
petrbalvin 2747fce7d3 feat(asm): encode the arm64 system registers and structure loads 2026-10-02 20:39:26 +02:00
petrbalvin e02918c17b fix(ci): keep the push suite inside the runner's memory and time budget
Test / test (push) Successful in 2m55s
Assisted-by: GLM 5.3 Flash
2026-10-02 17:20:31 +02:00
petrbalvin 02a6359c1f ci: shrink the push pipeline to the affordable gate set
Test / test (push) Failing after 5m26s
Assisted-by: GLM 5.3 Flash
2026-10-02 16:47:52 +02:00
petrbalvin b4c1e133c0 ci: keep the GOOBJ link parity gate off the push pipeline
Test / test (push) Failing after 12m18s
Assisted-by: GLM 5.3 Flash
2026-10-02 16:15:02 +02:00
petrbalvin 1107928870 build(justfile): run the test recipes under the memory fence
Assisted-by: GLM 5.3 Flash
2026-10-02 16:15:02 +02:00
petrbalvin bc4ac93fd9 style(asm): reindent the evex comment gofmt asks for
Test / test (push) Failing after 21m29s
Assisted-by: GLM 5.3
2026-10-02 00:41:46 +02:00
petrbalvin c1bca7ce7e docs: record the development deltas in the changelog
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin fefb76beb9 docs(asm): correct the reference against the assemblers' behaviour
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin 42bc1669d7 feat(lsp): document directives on hover and widen completion
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin 69dcbec8ef feat(lint): eleven new rules over directives, data and addressing
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin f405cea5bc fix(cmd): stop the coverage run on a stray in-place trap
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin f57377abb9 fix(debug): handle mapping edges, stray traps and dying debuggees
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin dd1782c538 test(verify): gate GOOBJ link parity with cmd/link
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin 17cc49fee4 fix(asm): emit NOPTR data as its own symbol kind
Assisted-by: GLM 5.3
2026-10-02 00:40:43 +02:00
petrbalvin bafb2fd130 feat(asm): encode the amd64 and loong64 tails of the corpus testdata
Assisted-by: GLM 5.3
2026-10-02 00:40:43 +02:00
petrbalvin 2f679326c2 style(testdata): canonicalise forms_amd64.s
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 96f2dd65b4 test(lexer): fuzz the token stream invariants
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin ca887d3927 fix(parser): bound folding depth and macro expansion work
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 6570709226 fix(parser): peel stacked labels the way the formatter renders them
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 9a5217d9c1 fix(format): keep every token of a line in the canonical output
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 4d01bb3ecf build: rename the module to sourcedock.dev/petrbalvin/gasm-sdk
Test / test (push) Successful in 4m18s
2026-09-26 11:08:43 +02:00
petrbalvin 332c63e440 ci: compile-gate FreeBSD in the test pipeline
Test / test (push) Successful in 4m10s
Assisted-by: GLM 5.3 Flash
2026-09-25 21:46:49 +02:00
petrbalvin b306c210c6 feat(debug): port the debugger to FreeBSD
Assisted-by: GLM 5.3 Flash
2026-09-25 21:46:40 +02:00
petrbalvin b9015e1c2e fix(verify): make the executable mapping build on FreeBSD
Assisted-by: GLM 5.3 Flash
2026-09-25 21:46:31 +02:00
petrbalvin 26c5008136 fix(cmd): honour //go:build in the corpus audit
Test / test (push) Successful in 2m32s
Assisted-by: GLM 5.3 Flash
2026-09-23 21:03:16 +02:00
petrbalvin 74d6b90d69 fix(asm): read the arm64 move-wide immediate as an unsigned pattern
Assisted-by: GLM 5.3 Flash
2026-09-23 21:03:03 +02:00
petrbalvin 7b11c62f53 fix(asm): resolve negative numeric PC-relative jumps
Assisted-by: GLM 5.3 Flash
2026-09-23 21:02:50 +02:00
petrbalvin 8eed54b3da feat(lsp): quick fixes for the textflag include and the argument area
Test / test (push) Successful in 2m56s
Assisted-by: GLM 5.3 Flash
2026-09-23 20:23:35 +02:00
petrbalvin 4be16dcdf5 feat(lsp): resolve symbols across workspace files
Assisted-by: GLM 5.3 Flash
2026-09-23 20:21:58 +02:00
petrbalvin ded9cabdf4 fix(lsp): apply each rename edit to its own document
Assisted-by: GLM 5.3 Flash
2026-09-23 20:19:21 +02:00
petrbalvin cf6bc6987e fix(ci): pass the upload file to curl, not its interpolation
Test / test (push) Successful in 2m26s
Release / gates (push) Successful in 2m25s
Release / build (amd64, linux) (push) Successful in 1m15s
Release / build (arm64, linux) (push) Successful in 1m16s
Release / build (loong64, linux) (push) Successful in 1m16s
Release / build (riscv64, linux) (push) Successful in 1m16s
Release / release (push) Successful in 34s
2026-09-22 01:31:49 +02:00
petrbalvin ff7b1452b1 docs: name 0.35.0 as the supported release
Test / test (push) Successful in 2m33s
Release / gates (push) Successful in 2m29s
Release / build (amd64, linux) (push) Successful in 1m18s
Release / build (arm64, linux) (push) Successful in 1m20s
Release / build (loong64, linux) (push) Successful in 1m17s
Release / build (riscv64, linux) (push) Successful in 1m26s
Release / release (push) Failing after 35s
2026-09-22 00:52:56 +02:00
petrbalvin 517c1cea25 chore: prepare release v0.35.0
Test / test (push) Successful in 2m33s
Release / gates (push) Failing after 46s
Release / build (amd64, linux) (push) Skipped
Release / build (arm64, linux) (push) Skipped
Release / build (loong64, linux) (push) Skipped
Release / build (riscv64, linux) (push) Skipped
Release / release (push) Skipped
2026-09-22 00:44:10 +02:00
petrbalvin a3e3010e0f fix(cmd): resolve the runtime header test GOROOT from the go command
Test / test (push) Successful in 2m39s
2026-09-21 22:46:07 +02:00
petrbalvin 057c4eb545 docs: complete the release delta in the changelog and readme 2026-09-21 22:45:56 +02:00
petrbalvin f720381d43 feat(asm): the segment-absolute and crash-store forms GOROOT writes
Test / test (push) Failing after 2m28s
Assisted-by: GLM 5.3 Flash
2026-09-21 22:19:53 +02:00
petrbalvin 2c9042d62c feat(asm): PCALIGN alignment on amd64
Assisted-by: GLM 5.3 Flash
2026-09-21 22:00:30 +02:00
petrbalvin 82ef289d3a feat(asm): the immediate multiply and arm64 indirect branches GOROOT writes
Assisted-by: GLM 5.3 Flash
2026-09-21 21:50:11 +02:00
petrbalvin 7246b0e002 feat(asm): the TLS access pair in the toolchain's one-instruction form
Assisted-by: GLM 5.3 Flash
2026-09-21 21:35:15 +02:00
petrbalvin 8cfd40aac8 feat(asm): the operand forms and defines GOROOT writes
Assisted-by: GLM 5.3 Flash
2026-09-21 21:17:34 +02:00
petrbalvin 5382c9a8e4 feat(audit): list every corpus failure per architecture 2026-09-21 21:17:34 +02:00
petrbalvin 53de91b2df docs(asm): describe the four target architectures
Test / test (push) Failing after 2m23s
Assisted-by: GLM 5.3 Flash
2026-09-21 20:15:55 +02:00
petrbalvin 8a36af7c7d docs(asm): generate the instruction appendices
Assisted-by: GLM 5.3 Flash
2026-09-21 20:15:55 +02:00
petrbalvin e9789ce3f4 chore(arch): regenerate the instruction tables 2026-09-21 20:15:55 +02:00
petrbalvin 837231c068 docs(asm): open the assembly language reference
Assisted-by: GLM 5.3 Flash
2026-09-21 19:49:04 +02:00
petrbalvin 95025be1bc docs(changelog): describe the encoder entries by content
Test / test (push) Failing after 2m33s
2026-09-21 19:20:01 +02:00
petrbalvin 03a964bb2d docs(goobj): document the GOOBJ object file format 2026-09-21 19:19:53 +02:00
petrbalvin 123a16e346 docs(readme): state the documentation goal 2026-09-21 18:35:27 +02:00
petrbalvin 9701812bee docs: changelog for the completeness waves
Test / test (push) Failing after 3m6s
Assisted-by: GLM 5.3 Flash
2026-09-21 02:04:44 +02:00
petrbalvin 29ac03468e feat(amd64): floating-point immediates through a synthesised pool
Assisted-by: GLM 5.3 Flash
2026-09-21 02:04:44 +02:00
petrbalvin bfb7701db1 feat(amd64): emit the quad-register EVEX families
Assisted-by: GLM 5.3 Flash
2026-09-21 02:02:19 +02:00
petrbalvin e8b6ff5d7c fix(parser): fold a signed parenthesised displacement expression
Test / test (push) Failing after 2m21s
Assisted-by: GLM 5.3 Flash
2026-09-21 00:45:33 +02:00
petrbalvin 1456907000 feat(riscv64,loong64): operand tail, float DATA and honest port classification
Assisted-by: GLM 5.3 Flash
2026-09-21 00:44:47 +02:00
petrbalvin ec1c521187 feat(cmd): GOOS-aware headers, audit battery shapes and semicolon spacing
Test / test (push) Failing after 2m21s
Assisted-by: GLM 5.3 Flash
2026-09-20 22:02:46 +02:00
petrbalvin a7744c24bd fix(parser): substitute macro parameters behind element selectors
Assisted-by: GLM 5.3 Flash
2026-09-20 22:02:19 +02:00
petrbalvin 522e6f2ae8 feat(parser): bracket register ranges, index-only VSIB and bare trailing immediates
Assisted-by: GLM 5.3 Flash
2026-09-20 22:02:19 +02:00
petrbalvin 81d4bd81e4 test(verify): register the wave kernels
Test / test (push) Failing after 2m20s
Assisted-by: GLM 5.3 Flash
2026-09-20 21:17:31 +02:00
petrbalvin 687678a2ea feat(elf): emit data relocations on arm64, riscv64 and loong64
Assisted-by: GLM 5.3 Flash
2026-09-20 21:17:20 +02:00
petrbalvin b0f9071bf5 feat(arm64): whole-vector moves, bookkeeping ops and truncating-move lowering
Assisted-by: GLM 5.3 Flash
2026-09-20 21:17:20 +02:00
petrbalvin 81e2673923 feat(amd64): encode the AVX-512 and BMI corpus families
Assisted-by: GLM 5.3 Flash
2026-09-20 21:17:20 +02:00
209 changed files with 23347 additions and 1074 deletions
+42
View File
@@ -0,0 +1,42 @@
# FreeBSD compile gates. Dispatched by hand, never on a push.
#
# The debugger's ptrace surface and the JIT substrate are the two
# FreeBSD-portable layers the tree carries; the forge has no FreeBSD runner,
# so they can only be compile-gated, and three foreign-GOOS builds of the
# whole module are minutes of one-core work the push pipeline's budget cannot
# carry. The push pipeline stays fast and light; this workflow is the
# deliberate run, before a release or after touching the ported layers.
# Running the ptrace suite itself needs real FreeBSD hardware.
#
# A dispatched workflow takes no concurrency block: it is one deliberate run.
name: FreeBSD build
on:
workflow_dispatch:
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
build:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
# The module is the source of truth for the version, so it cannot drift.
go-version-file: go.mod
cache: true
- name: FreeBSD build (amd64)
run: GOOS=freebsd GOARCH=amd64 go build ./...
- name: FreeBSD build (arm64)
run: GOOS=freebsd GOARCH=arm64 go build ./...
- name: FreeBSD build (riscv64)
run: GOOS=freebsd GOARCH=riscv64 go build ./...
+4 -1
View File
@@ -342,7 +342,10 @@ jobs:
my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}},
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
q{-H}, q{Content-Type: application/octet-stream},
q{-X}, q{POST}, q{--data-binary}, qq{@$path},
# The @ must not sit inside a qq{} string: there it starts an
# array interpolation and the upload body collapses to empty,
# which Gitea stores as a 201-created zero-byte attachment.
q{-X}, q{POST}, q{--data-binary}, q{@} . $path,
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name});
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
my $code = <$curl>;
+19 -11
View File
@@ -6,6 +6,11 @@
# everything runs in one job. Extra jobs would duplicate the checkout, the Go setup and
# the dependency download three times without buying any parallelism.
#
# The budget is part of the contract: a push run is fast and light, about two minutes,
# and nothing that cannot run natively on the runner belongs here. The FreeBSD compile
# gates live in freebsd.yml behind workflow_dispatch for that reason; the GOOBJ link
# parity campaign is an opt-in local verification (just link-parity).
#
# Every step is one command, so the step that fails is the gate that failed, and no shell
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
# not Python: Perl behaves the same on both runner images, there is no bashism to trip over
@@ -46,10 +51,9 @@ jobs:
go-version-file: go.mod
cache: true
- name: Install Perl
# The runner images are minimal and Perl is not guaranteed. The install is a
# no-op where it is already present; drop this step once verified on the box.
run: dnf install -y perl
# No Install Perl step: the fedora image carries perl (verified by run
# 76: the install degraded into a package upgrade costing ~50 s), and a
# dnf on the push path is network work the budget does not need.
# The steps follow the `gates` order of the justfile contract: build, format,
# vet, test. The vet gate is go vet and go fix -diff, two steps here.
@@ -82,7 +86,11 @@ jobs:
# command, so the floor is the same number everywhere. ./verify/... carries the
# live oracle-parity comparison against `go tool asm` (the TestGroundTruth
# suites); the runner's Go setup provides both the tool and GOROOT.
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
# -short skips the deliberate-run categories inside the suites (the live
# ptrace sessions above all): they need the machine to themselves and a
# starved single-core runner turns each into a timeout the budget cannot
# carry. The local `just test` gate runs everything, in full.
run: go test -short -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Tests outside the coverage set
# The CLI and the debugger sit outside `packages` because a thin main and a
@@ -90,12 +98,12 @@ jobs:
# shipped surfaces: the command exit codes, the manual pages against the
# binary's own help, and the debugger's architecture-neutral units. They run
# here so the floor stays a product measure and nothing is left untested.
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
- name: Oracle parity
# Re-run the live go-tool-asm comparison as its own step so that a parity
# regression names the gate that failed instead of hiding inside the suite.
run: go test -count=1 -timeout 10m -run 'TestGroundTruth' ./verify/...
# -short skips the debugger's live ptrace sessions, the deliberate-run
# category the runner cannot starve-proof. The live go-tool-asm oracle
# comparison (TestGroundTruth in ./verify/...) runs inside the coverage
# sweep above; it is not re-run as its own step, because every second on
# this box is budget.
run: go test -short -count=1 -timeout 10m ./cmd/... ./debug/...
- name: Coverage floor
run: |
+259 -10
View File
@@ -1,6 +1,6 @@
# Changelog
All notable changes to gasm-devkit are documented here.
All notable changes to gasm-sdk are documented here.
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
@@ -9,6 +9,178 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added
- **The FreeBSD port of the debugger.** `gasm debug` runs on FreeBSD on
amd64, arm64 and riscv64 with the same interactive surface as on Linux:
breakpoints, hardware watchpoints (x86 debug registers, the arm64 debug
register file), single-stepping, register and memory access, all behind
the kernel's own ptrace requests, with tracee memory through `PT_IO` and
stop reports through `PT_LWPINFO`. The JIT substrate maps executable
memory through `golang.org/x/sys/unix`, so `verify` builds on FreeBSD
too. The pipeline compile-gates all three architectures; live
validation awaits a FreeBSD machine.
- **Workspace-wide navigation in the language server.** `gasm lsp` indexes
the `.s` files under the workspace root beyond the documents the editor
has open, so go-to-definition, find references and workspace symbol search
reach files that were never opened. An open buffer always shadows its
disk copy, and watched-file events together with a per-query freshness
check keep the index current.
- **Quick fixes for the textflag include and the argument area.** The
`missing-textflag-include` warning offers to add the include after the
last one in the file, and the `abi-argsize` warning offers to set the
TEXT argument area to the size the `// func` signature implies, computed
by the new `lint.ExpectedArgSize`.
- **Eleven new lint rules over the directives, the data section and the
sharpest addressing edges.** `missing-argsize` flags a TEXT that
declares no argument area its `// func` signature implies;
`noframe-frame-size` a NOFRAME with a positive frame;
`unnamed-fp-reference` a nameless `0(FP)`, which both assemblers
reject; `hardware-sp-addressing` a negative offset off the hardware SP
rather than the virtual frame; `vex-sse-mixing` a kernel that mixes VEX
and legacy SSE spellings and pays the transition penalty;
`unnamed-result` a `ret+N(FP)` the signature names; `data-width`,
`data-value-overflow`, `data-string-width`, `data-without-globl` and
`data-exceeds-globl` police the DATA width against its value type and
the GLOBL size behind it. `missing-ret` now also flags a function
whose tail can fall off its end even though a RET sits somewhere in the
body, and `invalid-textflag` also reports a flag misplaced between TEXT
and GLOBL.
- **Hover documentation and wider completions in the language server.**
Hovering a directive or pseudo-operation (TEXT, DATA, GLOBL, PCALIGN,
FUNCDATA, PCDATA, the BYTE family) shows its grammar and rules in
preference to the empty instruction-table entry, and completion offers
those names beside the instruction set. The `missing-argsize` warning
carries a quick fix that declares the argument area the signature
implies (`$0` becomes `$0-16`).
- **The GOOBJ link parity gate.** A regression test assembles a kernel
per architecture through `gasm asm --format goobj`, substitutes the
object into a real `go build`'s package archive, proves the archive
carries it byte for byte and re-links with cmd/link on amd64, arm64,
riscv64 and loong64. The binaries run, natively on amd64 and under
qemu-user on arm64 and riscv64, and their output must match the
toolchain-built baseline; loong64 is link-only. The pipeline installs
qemu-user and runs the gate on every push.
- **The amd64 and loong64 encoders close four more corpus files.** The
whole-tree measure moves to 272 of 322 (84.5 %): amd64 gains the
one-operand IMUL, the SSE compare family (CMPPD, CMPPS, CMPSS), RETFL,
the LOOP family, the MMX register bank with its bank-crossing moves,
MOVNTDQ, the CR and DR register moves, PUSH and POP of FS and GS, the
`(TLS)` pseudo-base, the wait and cache controls (CLWB, CLDEMOTE,
TPAUSE, UMONITOR, UMWAIT, RDPID, ENDBR64), indirect branches with the
star spelling (`JMP *(R12)(R13*4)`), `RET sym(SB)` as the tail jump,
the colon shift spelling (`SHLL CX, R11:AX`) and the EVEX and VEX forms
of the rounds, AES key assist, string compares, extracts, blends and
permutes; loong64 gains the acquire and release pair (LLACQ, SCREL,
with the vector widths), the VMOVQ and XVMOVQ lane forms and the BYTE
literal-data escape hatch the other architectures already take.
### Changed
- **The corpus audit assembles like the build.** A file's `//go:build`
constraint decides which target architectures attempt it: cpu_x86.s is
an x86 build alone, and the msan and goexperiment.runtimesecret trees
are compiled by no supported build, so they leave the measured set
instead of failing it. The headline now reads "assemble for every
applicable target": every real-code GOROOT assembly file, the tree
without testdata, assembles for all four architectures (250 of 250,
100 %); over the whole tree including testdata the measure is 272 of
322 (84.5 %).
- **The module moves to `sourcedock.dev/petrbalvin/gasm-sdk`.** The
repository and the module rename together with the product, now the
GAsm Software Development Kit. Fresh installs become
`go install sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm@latest`, and
installs pinned to the old `gasm-sdk` path stop resolving once the
repository takes the new name: reinstall from the new path. The
binary stays `gasm`.
### Fixed
- **Rename edits land in their own documents.** A rename collected the
ranges of every reference across the open documents but applied them all
to the document that started it, so renaming a symbol used in a second
file moved that file's text into the first. Each edit now applies to the
document it was collected in.
- **Negative numeric PC-relative jumps.** `JMP -3(PC)`, the shape the
runtime's exit loops write (sys_linux_amd64.s, sys_netbsd_amd64.s),
resolved to nothing: only the forward forms counted. A negative count
now walks the same instruction statements backwards, labels excluded,
byte-identical with the toolchain.
- **The arm64 move-wide family reads its immediate as an unsigned
pattern.** `MOVK $(40000<<48)` folds to a negative int64 and was
rejected; the toolchain picks the 16-bit lane from the 64-bit bit
pattern, so the encoder now does the same, and a zero immediate is
rejected where the toolchain rejects it.
- **NOPTR data emits its own symbol kind.** A GLOBL with NOPTR was
emitted as plain SDATA, the kind the linker holds to its Go type
information requirement, so every gasm object carrying runtime-shaped
data (`GLOBL ·x(SB), NOPTR, ...`) died in cmd/link with "missing Go
type information". NOPTR data is now SNOPTRDATA, the toolchain's kind
for pointer-free globals, and RODATA still wins where both flags
appear, exactly as the toolchain chooses. The three end-to-end link
tests that should have caught this substituted the gasm object after
the package archive was already packed, so they passed vacuously; they
now substitute inside the archive, prove the substitution byte for
byte and re-link.
- **Latent encoder divergences against the toolchain, found by the
whole-file differential harness.** On loong64, `MOVx $off(reg)` lost
its base register, SCQ swapped its operand fields, the vector lane
inserts did not scale offsets by the element width and two immediate
opcodes were mistyped; on amd64, NOP with operands encoded 0x90 where
the toolchain emits nothing at all, the VEX gather length bit ignored
the VSIB index width, four permute and extract families were pushed
into EVEX where the toolchain stays VEX, and a reference to a static
symbol no GLOBL defines failed where the toolchain defers it to the
linker as an external relocation.
- **Seven front-end defects found by fuzzing.** The formatter swallowed
the statement after a leading block comment (`/* head */ MOVQ AX, BX`
formatted to the comment alone), dropped trailing tokens after an
`#include` header, and trimmed the trailing whitespace inside a string
literal; the parser peeled only one label of a stacked pair (`a: b:`
parsed differently than it formatted); deeply nested parentheses in an
immediate overflowed the stack through the constant folder, which now
stops at a bounded depth and falls back to the ordinary operand paths;
macro expansion is linear in the invocation count instead of
quadratic; and amplifying macros (the billion-laughs shape) stop at a
work budget sized by the line and the macro table, reported as a
diagnostic instead of running for hours.
- **The debugger handles the edges its tests now reach.** Memory reads
and writes cover exactly the requested bytes, so a request ending in
the last page of a mapping no longer fails on the unmapped page behind
it; disassembly shrinks its instruction window at a mapping's end
instead of failing; a debuggee that dies before signalling readiness
writes a failure notice the launcher reads, so launch fails fast with
the reason instead of after the whole poll (and the poll no longer
waits on the child, which could consume the SIGSTOP park and hang the
launch); amd64 watchpoints acknowledge the sticky DR6 hit bits and
clear the address register on release; the breakpoint listing is
ordered by address so its numbering is stable; the REPL rejects bad
counts, sizes and watchpoint types instead of silently guessing,
reports stops on signals during stepping, and a breakpoint-class trap
that matches no breakpoint and leaves the PC in place surfaces instead
of spinning the continue loop and the coverage run forever.
## [0.35.0] - 2026-09-22
### Added
- **The go_asm.h generator.** `gasm asm` generates the package's go_asm.h
itself when an assembly file includes it: the Go files beside the source
are type-checked for the target architecture and the constants and field
offsets become assembler defines, so package-context files assemble with
no compiler and no `go build` in the loop. `-GOOS` selects the
type-checking GOOS for GOOS-specific files, and the corpus audit derives
the GOOS from the file name.
- **ELF data relocations on arm64, riscv64 and loong64.** `gasm asm
--format elf` emits `.rela.data` for symbol-valued DATA initialisers on
every architecture (amd64 carried them already), so standalone ELF
objects link on all four targets.
- **Corpus failure listing.** `gasm audit-instructions --corpus --list`
prints every failing file with its failure reason, per architecture,
instead of one representative file per reason.
- **DATA with symbol values and relaxed symbol spellings.** DATA
initialisers accept `$symbol(SB)` values, laid down as an absolute
relocation at the data field (GOOBJ on all four architectures and ELF
on all four as of this release), and U+2215 is accepted inside symbol
package paths.
- **Macro expansion and include splicing.** `gasm asm`, `gasm diff` and
`gasm audit-instructions` now preprocess assembly the way the
toolchain does: object and parameterised `#define` macros expand at
@@ -19,7 +191,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
(`$(32-7)`, `$~63`, `(index*4)(base)`) fold at parse. Expansion
happens only on the assembly path: `gasm lint`, `gasm fmt` and the
language server keep reading the raw file.
- **The GOROOT instruction wave, part 1.** The encoder now covers the
- **Encoder coverage: the instruction families GOROOT's real code
uses.** The encoder now covers the
instruction families GOROOT's real code uses that gasm lacked,
byte-verified against `go tool asm`: on amd64 the carry ALU, the
atomics (CMPXCHG, XADD, XCHG), AES-NI, SHA-1/256, PCLMULQDQ, CRC32,
@@ -35,14 +208,90 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
Also fixed on the way: arm64 `CASD`/`CASW` lacked an opcode bit, and
riscv64 `VSETVLI` with an immediate length now canonicalises to
`vsetivli` as the toolchain does.
- **The corpus audit measures honestly.** Files named for Go ports gasm
does not target (arm, 386, s390x, ...) are no longer attempted for the
four supported architectures (no supported build compiles them), and
the headline rate is reported over attemptable files: 136 of 433 on
the full corpus (31.4 %), 135 of 383 on real code (35.2 %), from the
127 that the previous release measured. The probe battery that
decides encodability gained the operand shapes the new families use.
-
- **Encoder coverage: quad-register AVX-512 and floating-point
immediates.** The encoder gains the
quad-register AVX-512 families (4FMAPS, 4FNMADD, 4VNNIW, VP4DPWSSD,
VP4DPWSSDS) with the register list riding the inverted V'VVVV field,
floating-point immediates on the SSE scalar moves and arithmetic
(the constant lands in a synthesised read-only pool, a positive zero
collapses to XORPS exactly as the toolchain does), accept-and-ignore
FUNCDATA and PCDATA, three-operand double shifts, static-symbol
operands for the legacy SSE moves, and the pooled 64-bit immediate
materialisation on riscv64. The parser carries bracketed register
ranges, index-only VSIB memory operands and bare trailing immediates;
macro substitution reaches parameters used with element suffixes
(`A.S4`), and `;` separates statements in plain files.
- **Per-architecture reference pages.** [docs/asm/](docs/asm/README.md)
gains AMD64, ARM64, RISCV64 and LOONG64: the register files and the
roles the ABI fixes, addressing, operand order with every special form,
constants and materialisation, alignment, fences and the relocations
each target emits. An instruction inventory appendix per architecture
is generated from the toolchain's own tables by `just gen`, and the
regenerated tables recognise 147 more mnemonics than the previous
release carried (arm64 107, riscv64 31, loong64 9).
- **The Plan 9 assembly language reference.** [docs/asm/](docs/asm/README.md)
opens the complete language reference with its common core: the lexicon,
statement structure and constant expressions, the operand grammar with
the pseudo-registers and symbol naming, the directives and the function
flag vocabulary, preprocessing with `#define` and `#include`, and the
Go-embedded layer (ABI0, prototypes, `go_asm.h`, `funcdata.h` and the
runtime contract). Every claim is verified against `go tool asm` of
Go 1.27.1 and gasm's differential tests; the per-architecture pages and
generated instruction appendices follow.
- **GOOBJ format specification.** [docs/GOOBJ.md](docs/GOOBJ.md)
documents the Go object file format in full: both containers, the 96
byte header and all 19 blocks, every structure with its byte
offsets, symbol kinds and flag bits, all 106 relocation types with
the weak variants, aux symbols, the FuncInfo payload, the pc-value
table encoding, the content hashes and the builtin table, all
verified byte for byte against objects produced by Go 1.27.1's own
tools.
### Changed
- **The corpus audit measures like a build.** Files named for a Go port
gasm does not target (arm, 386, s390x, ...) are never attempted, because
no supported build compiles them; the GOOS comes from the file name; and
each target's go_asm.h is generated on the fly. The headline is reported
over attemptable files: 291 of 353 on the full corpus (82.4 %) assemble
for every target architecture and 295 of 303 on real code (97.4 %),
against 108 of 627 over all files (17.2 %) that the previous release
measured.
### Fixed
- **The operand forms GOROOT writes.** Numeric PC-relative jumps
(`JEQ 2(PC)`, the park loop `JMP 0(PC)`) resolve with the toolchain's
own instruction counting and fold jump-to-jump chains exactly as its
branch optimiser does; symbol immediates (`MOVQ $sym(SB), AX`)
assemble to the toolchain's RIP-relative LEA with an R_PCREL
relocation; negated constant expressions in operands (`ADJSP
$-(REGS - 8)`, the shape the cgo ABI macros write) fold; the immediate
multiply (`IMULQ $1000000000, AX`) encodes with the toolchain's
0x69/0x6B selection; the TLS access pair assembles as the toolchain's
one-instruction form (the bare `MOVQ TLS, r` load nops out and
`off(r)(TLS*1)` folds to the segment-prefixed absolute whose disp32
carries the R_TLSLE relocation, per-GOOS); arm64 accepts the
bare-register indirect branch (`BL R9` beside `BL (R9)`, both BLR) and
the zero-immediate store (`MOVD $0, mem` through the zero register,
rejecting non-zero immediates as the toolchain does); `PCALIGN` now
aligns on amd64, padding with the toolchain's greedy
single-instruction NOPs; the segment-absolute forms (`MOVQ 0x30(GS),
AX` and the store direction) and the absolute crash-store
(`MOVL $0xf1, 0xf1`) encode; and `gasm asm` predefines the
`GOARCH_<arch>` and `GOOS_<goos>` macros the go command passes to
`go tool asm`, so GOROOT headers' `#ifdef GOARCH_amd64` platform
blocks (`go_tls.h`'s `get_tls` and friends) select as intended. The
GOROOT corpus measure moves to 291 of 353 files assembling for every
target architecture (82.4 %), 97.4 % of the real-code corpus, from
70.8 % and 82.2 %.
- **Tool corrections across the pipeline.** The formatter keeps square
brackets in SIMD operands, statement separators and canonical macro
bodies; the linter drops false positives on shift counts, SETcc
spellings and ABIInternal references; the lexer treats a trailing
carriage return as a line end so comment text stays idempotent; and
arm64 rejects bare BTI with a diagnostic while accepting the full
family.
## [0.34.0] - 2026-09-20
+4 -4
View File
@@ -1,6 +1,6 @@
# Contributing
Contributions to **gasm-devkit** are governed by the Contributor terms
Contributions to **gasm-sdk** are governed by the Contributor terms
below; submitting one means you accept them.
## Contributor terms
@@ -29,8 +29,8 @@ compiler (gcc), because `just gates` includes `just race` and the race
detector needs cgo.
```sh
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
cd gasm-devkit
git clone https://sourcedock.dev/petrbalvin/gasm-sdk.git
cd gasm-sdk
just build
just gates
```
@@ -126,7 +126,7 @@ tag, where it would double the time and the memory a shared runner cannot spare.
## Reporting bugs
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-devkit/issues` with the
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-sdk/issues` with the
version, the operating system and architecture, the exact command, the full output,
and the expected against the actual behaviour.
+66 -20
View File
@@ -1,6 +1,6 @@
# Plan 9 assembly tooling, inside and outside Go
# GAsm: Software Development Kit for Plan 9 Assembly
> **Warning: this is an experiment.** gasm-devkit is under active
> **Warning: this is an experiment.** gasm-sdk is under active
> development and is not stable. The version is 0.x.x: commands, flags,
> output formats and behaviour can change without warning at any time.
> A 1.0.0 release is light years away. Nothing in this document is a
@@ -13,7 +13,7 @@
there is no formatter, no linter and no debugger for `.s` files, and no
assembler that works without a Go installation. Developers write
assembly blind, validate it by benchmark, and debug it by print
statement. gasm-devkit is the missing toolkit: a single, self-contained
statement. gasm-sdk is the missing toolkit: a single, self-contained
binary, `gasm`, that serves both purposes.
- **Help develop Plan 9 assembly.** Formatting, linting, disassembly,
@@ -58,7 +58,7 @@ Plan 9 (Go): MOVQ AX, total-16(SP)
The same lines, but only one of them tells you what the number is for.
The syntax is uppercase, regular and boring, which is the highest
compliment a language for machine code can earn. gasm-devkit exists
compliment a language for machine code can earn. gasm-sdk exists
to give that syntax the tooling it deserves.
## Features
@@ -81,7 +81,10 @@ to give that syntax the tooling it deserves.
GOOBJ format, which needs the installed toolchain and which `go build`
consumes in place of the toolchain's output. Framed functions get the
stack-split guard and the morestack block, byte-identical to the
toolchain's, so split functions link too.
toolchain's, so split functions link too. The assembler preprocesses
like the toolchain (`#define`, `#include` with `-I`, `#ifdef`), generates
`go_asm.h` from the package's Go files, and carries `PCALIGN`, the
`LOCK`/`REP` prefixes and the literal-data pseudo-ops.
- **Disassembler.** `gasm dis` lists a `.s` file's functions at their real
offsets after assembling, or disassembles raw bytes from a file or stdin.
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
@@ -91,13 +94,16 @@ to give that syntax the tooling it deserves.
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
breakpoints (optionally conditional), hardware watchpoints, register and
memory inspection, and headless script runs that report instruction and
label coverage.
label coverage; it runs on Linux (all four architectures) and FreeBSD
(amd64, arm64, riscv64).
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
push and pull diagnostics, semantic-token highlighting, go-to-definition,
find references, rename, formatting, inlay hints, code actions, signature
help, document highlights, workspace symbol search, #include document
links and folding ranges over stdio; definition, references and rename
work across every open document.
work across every open document and the indexed workspace files beyond
them, and the quick fixes add a missing textflag.h include and set the
argument area from the // func signature.
- **Comparators and audits.** `gasm diff` compares the machine code of two
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
`gasm audit-instructions` diffs the encoder against the installed toolchain,
@@ -110,9 +116,9 @@ Four architectures, the four that matter in practice:
| Architecture | GOARCH | File suffix | Instructions recognised |
|--------------|-------------|--------------|---------------------------------------------|
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
| ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes |
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes |
| LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes |
| ARM64 | `arm64` | `_arm64.s` | 645 + common opcodes |
| RISC-V | `riscv64` | `_riscv64.s` | 992 + common opcodes |
| LoongArch | `loong64` | `_loong64.s` | 808 + common opcodes |
"Common opcodes" are the instructions shared by every architecture (`RET`,
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
@@ -124,10 +130,13 @@ can emit today is narrower, and a recognised but unencodable instruction is
reported as an explicit error, never as a wrong byte.
The same measurement runs over GOROOT's whole assembly corpus:
`gasm audit-instructions --corpus` reports 136 of 433 attemptable files
(31.4 %) assembling for every target architecture today (files named for
other Go ports are counted but never attempted), with the top failure
reasons per architecture; the number moves with every release.
`gasm audit-instructions --corpus` reports every real-code GOROOT assembly
file (the tree without testdata) assembling for every target its build
admits: 250 of 250, 100 %. Over the whole tree including testdata the
measure is 271 of 322 attemptable (84.2 %); files named for other Go ports
are counted but never attempted, and `//go:build` constraints decide which
targets attempt a file at all, exactly as the build does. The number moves
with every release.
### Validation status
@@ -142,7 +151,7 @@ actually been executed.
|---|---|---|
| Encoding: byte-for-byte against `go tool asm` | native hardware | native hardware (the toolchain cross-assembles any GOARCH on any host) |
| Execution: JIT calls, ABI checks, differential fuzzing | native hardware | qemu-user emulation |
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more |
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more. FreeBSD (amd64, arm64, riscv64) is in the same position: the port compiles behind the cross-build gate and its integration test is ready, but no FreeBSD machine has executed it |
Consequences, stated plainly. An emulator is a model of a CPU, not the
CPU: instruction semantics are implemented in software and can differ
@@ -157,6 +166,39 @@ been compiled and read, never executed. Its architecture-neutral units
run under `go test ./...`, which the race workflow and a manual run
perform; the default `just test` gate does not sweep `./debug/...`.
## The documentation goal
The toolkit is the primary goal. The secondary one is documentation: a
specification of the Plan 9 assembly language and of the GOOBJ object
format that is 100 % complete, detailed enough to implement against,
and written to a professional standard. These are the two subjects this
project works with every day, and they are the two for which no usable
documentation exists.
Go documents the language on a single page, "A Quick Guide to Go's
Assembler", which carries no section for loong64, one of the four
architectures gasm supports, and covers a fraction of what each
assembler accepts. What exists beyond it lives as comments inside the
toolchain's internal source: per-architecture reference manuals for
arm64, ppc64, riscv64 and loong64, written for the toolchain's own
maintainers rather than for an outside reader, and none at all for
amd64. GOOBJ fares worst of all. The format that `go build` consumes
has no specification anywhere: it is described by a comment in an
internal package, it is not a stable interface, and it can change with
any toolchain release.
The gap is therefore filled the only way it can be filled: by reverse
engineering the toolchain itself, the same work the encoders already
perform. Most of the documentation can come from nowhere else, and it
is written as that knowledge is produced during development. It is
verified the way the code is verified: an encoding documented here is
one that differential tests against `go tool asm` confirm
byte-for-byte, and a format field documented here is one the linker
demonstrably reads. The work has begun: [docs/GOOBJ.md](docs/GOOBJ.md)
specifies the object file format completely, and
[docs/asm/README.md](docs/asm/README.md) opens the language reference
with its common core. The per-architecture pages follow.
## Direction
The plan, in the order it is being worked:
@@ -179,9 +221,11 @@ The plan, in the order it is being worked:
toolchain itself does not support; through ELF, Plan 9 assembly becomes
usable outside Go entirely.
- **Platforms: Linux and FreeBSD.** Linux is supported today on all four
architectures and is where the binary builds. FreeBSD follows: the
JIT's executable-memory mapping and the ptrace debugger layer are the
two pieces of porting work. Other unix systems may follow those two.
architectures and is where the binary builds. FreeBSD follows on amd64,
arm64 and riscv64: the JIT's executable-memory mapping and the ptrace
debugger layer are ported (the debugger's live validation awaits a
FreeBSD machine, as the validation status states). Other unix systems
may follow those two.
- **Four architectures, no more.** amd64, arm64, riscv64 and loong64.
No others are planned.
@@ -189,11 +233,11 @@ The plan, in the order it is being worked:
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
linux/loong64 are on the
[releases page](https://sourcedock.dev/petrbalvin/gasm-devkit/releases).
[releases page](https://sourcedock.dev/petrbalvin/gasm-sdk/releases).
From source (Go 1.27.1):
```sh
go install sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm@latest
go install sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm@latest
```
Or from a repository checkout:
@@ -282,6 +326,8 @@ recipe.
~/.local/share/man (MANDIR overrides); `just uninstall-man` removes
them
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/GOOBJ.md](docs/GOOBJ.md): the GOOBJ object file format specification
- [docs/asm/](docs/asm/README.md): the Plan 9 assembly language reference
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
- [CHANGELOG.md](CHANGELOG.md): release history
+1 -1
View File
@@ -7,7 +7,7 @@ releases do not receive them.
| Version | Supported |
|---|---|
| 0.34.0 | yes |
| 0.35.0 | yes |
| older releases | no |
## Reporting a vulnerability
+105 -7
View File
@@ -5,9 +5,13 @@
// toolchain's own assembler source. Go's Plan 9 assembler defines the exact,
// complete set of mnemonics it accepts for each architecture in
// $GOROOT/src/cmd/internal/obj/<arch>/anames.go; this tool extracts those
// names so gasm-devkit supports every instruction the real assembler does,
// names so gasm-sdk supports every instruction the real assembler does,
// with no hand-maintained (and therefore inevitably incomplete) lists.
//
// The same data feeds the generated instruction appendices of the assembly
// language reference, docs/asm/INSTRUCTIONS-<ARCH>.md, so that the reference
// cannot drift from the tables it documents.
//
// Usage (via the justfile):
//
// just gen
@@ -26,9 +30,12 @@ import (
"path/filepath"
"sort"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
)
// archDirs maps a gasm-devkit architecture name to its obj sub-directory.
// archDirs maps a gasm-sdk architecture name to its obj sub-directory.
var archDirs = []struct {
arch string
sub string
@@ -39,11 +46,30 @@ var archDirs = []struct {
{"loong64", "loong64"},
}
// docPages maps an architecture to its generated appendix in the language
// reference. The amd64 page carries a per-mnemonic encodability column,
// decided by asm.Encodable, which mirrors the encoder's own dispatch; the
// other targets have no single cheap predicate, so their pages carry the
// inventory and point at the live measurement instead.
var docPages = []struct {
arch arch.Arch
title string
file string
anames string
encodable bool
}{
{arch.AMD64, "AMD64", "INSTRUCTIONS-AMD64.md", "cmd/internal/obj/x86/anames.go", true},
{arch.ARM64, "ARM64", "INSTRUCTIONS-ARM64.md", "cmd/internal/obj/arm64/anames.go", false},
{arch.RISCV, "RISC-V 64", "INSTRUCTIONS-RISCV64.md", "cmd/internal/obj/riscv/anames.go", false},
{arch.LOONG64, "LoongArch 64", "INSTRUCTIONS-LOONG64.md", "cmd/internal/obj/loong64/anames.go", false},
}
func main() {
goroot := strings.TrimSpace(runGoEnvGOROOT())
if goroot == "" {
fatal("could not determine GOROOT")
}
version := strings.TrimSpace(runGoEnv("GOVERSION"))
// The common opcodes shared by every architecture (RET, JMP, NOP, CALL,
// TEXT, FUNCDATA, …) live in cmd/internal/obj/util.go.
commonPath := filepath.Join(goroot, "src", "cmd", "internal", "obj", "util.go")
@@ -57,16 +83,24 @@ func main() {
}
fmt.Printf("%-8s %4d instructions -> arch/common_gen.go\n", "common", len(common))
names := map[string][]string{}
for _, a := range archDirs {
path := filepath.Join(goroot, "src", "cmd", "internal", "obj", a.sub, "anames.go")
names, err := extractInstrs(path)
names[a.arch], err = extractInstrs(path)
if err != nil {
fatal("extract %s: %v", a.arch, err)
}
if err := writeGen(a.arch, a.sub, names); err != nil {
if err := writeGen(a.arch, a.sub, names[a.arch]); err != nil {
fatal("write %s: %v", a.arch, err)
}
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names), a.arch)
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names[a.arch]), a.arch)
}
for _, p := range docPages {
if err := writeDocPage(p.arch, p.title, p.file, p.anames, version, p.encodable); err != nil {
fatal("write %s: %v", p.file, err)
}
fmt.Printf("%-8s -> docs/asm/%s\n", p.arch, p.file)
}
}
@@ -85,7 +119,7 @@ func filterCommon(names []string) []string {
// writeCommon emits arch/common_gen.go.
func writeCommon(names []string) error {
var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
@@ -156,7 +190,7 @@ func stringLit(elt ast.Expr) string {
// writeGen emits arch/<arch>_gen.go.
func writeGen(arch, sub string, names []string) error {
var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
@@ -172,6 +206,61 @@ func writeGen(arch, sub string, names []string) error {
return os.WriteFile(filepath.Join("arch", arch+"_gen.go"), []byte(b.String()), 0o644)
}
// writeDocPage emits docs/asm/<file>, the generated instruction appendix of
// the language reference for one architecture: every mnemonic the toolchain
// accepts, with the curated summary where the architecture table carries one
// and, on amd64, a per-mnemonic encodability column.
func writeDocPage(a arch.Arch, title, file, anames, version string, encodable bool) error {
table := arch.ForArch(a)
instrs := table.Instructions()
var b strings.Builder
b.WriteString("# " + title + ": instruction inventory\n\n")
b.WriteString("Generated by gasm-sdk's `_gen` from the Go toolchain's instruction table\n")
b.WriteString("(`" + anames + "`, " + version + "); DO NOT EDIT. This page lists every mnemonic\n")
b.WriteString("`go tool asm` accepts on this target, which is the upper bound of the\n")
b.WriteString("language on it: a name absent here is not an instruction of the target,\n")
b.WriteString("and a name present here may still be one gasm's encoder cannot emit yet.\n\n")
encodableCount := 0
if encodable {
b.WriteString("The `gasm encodes` column reports whether gasm's encoder can emit the\n")
b.WriteString("mnemonic today; the gap is the encoder backlog, measured live by\n")
b.WriteString("`gasm audit-instructions`.\n\n")
b.WriteString("| Mnemonic | gasm encodes | Notes |\n")
b.WriteString("|---|---|---|\n")
for _, in := range instrs {
ok := asm.Encodable(in.Name)
if ok {
encodableCount++
}
b.WriteString("| `" + in.Name + "` | " + yesNo(ok) + " | " + in.Summary + " |\n")
}
b.WriteString("\n")
fmt.Fprintf(&b, "Recognised: %d mnemonics. gasm encodes: %d.\n", len(instrs), encodableCount)
} else {
b.WriteString("The inventory carries no per-mnemonic encoder column: on this target\n")
b.WriteString("encodability is decided per operand shape, and the live measured\n")
b.WriteString("coverage is reported by `gasm audit-instructions`.\n\n")
b.WriteString("| Mnemonic | Notes |\n")
b.WriteString("|---|---|\n")
for _, in := range instrs {
b.WriteString("| `" + in.Name + "` | " + in.Summary + " |\n")
}
b.WriteString("\n")
fmt.Fprintf(&b, "Recognised: %d mnemonics.\n", len(instrs))
}
return os.WriteFile(filepath.Join("docs", "asm", file), []byte(b.String()), 0o644)
}
// yesNo renders a boolean as the word the appendix tables use.
func yesNo(v bool) string {
if v {
return "yes"
}
return "no"
}
func runGoEnvGOROOT() string {
out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil {
@@ -180,6 +269,15 @@ func runGoEnvGOROOT() string {
return string(out)
}
// runGoEnv runs `go env` for a single variable.
func runGoEnv(name string) string {
out, err := exec.Command("go", "env", name).Output()
if err != nil {
return ""
}
return string(out)
}
func fatal(format string, args ...any) {
fmt.Fprintf(os.Stderr, "gen: "+format+"\n", args...)
os.Exit(1)
+1 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
+1
View File
@@ -33,6 +33,7 @@ func arm64Registers() []Register {
for i := 0; i <= 30; i++ {
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
}
add("R18_PLATFORM", GPR, "R18 under its toolchain-reserved Windows name (an alias of R18)")
add("ZR", Special, "zero register (reads as 0)")
add("SP", Special, "stack pointer")
add("LR", Special, "link register (alias of R30)")
+624
View File
@@ -0,0 +1,624 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// This file carries the extended-instruction layer: instructions the Go
// toolchain does not know at all, described as data and validated against
// golden vectors from the Arm Architecture Reference Manual rather than
// against the toolchain. It sits beside the generated tables, never inside
// them: arch/arm64_gen.go stays untouched, and Extensions returns the layer
// per architecture so a later amd64 table attaches through the same door.
//
// The first entry is the arm64 SVE and SVE2 integer add/subtract/multiply
// family (twenty-three forms over four word shapes). The encodings are
// transcribed from the manual and cross-checked against the GNU assembler's
// and LLVM's published encodings; the golden vectors in arm64_ext_test.go pin
// the bytes.
package arch
import "fmt"
// ExtOperandKind classifies one operand of an extended instruction.
type ExtOperandKind uint8
// Operand kinds.
const (
ExtZReg ExtOperandKind = iota // scalable vector register Z0-Z31
ExtPReg // predicate register P0-P15
ExtImm // immediate
)
// String returns a short label for the kind.
func (k ExtOperandKind) String() string {
switch k {
case ExtZReg:
return "scalable vector register"
case ExtPReg:
return "predicate register"
case ExtImm:
return "immediate"
default:
return "operand"
}
}
// ExtArrangement is the element-size suffix a scalable vector operand
// carries: .B, .H, .S, .D or .Q. ExtArrNone means the operand is written
// bare, which the SVE forms in this layer reject.
type ExtArrangement uint8
// Arrangements, widest last.
const (
ExtArrNone ExtArrangement = iota
ExtArrB // 8-bit elements
ExtArrH // 16-bit elements
ExtArrS // 32-bit elements
ExtArrD // 64-bit elements
ExtArrQ // 128-bit elements
)
// String returns the assembler suffix, with the leading dot.
func (a ExtArrangement) String() string {
switch a {
case ExtArrB:
return ".B"
case ExtArrH:
return ".H"
case ExtArrS:
return ".S"
case ExtArrD:
return ".D"
case ExtArrQ:
return ".Q"
default:
return ""
}
}
// Width returns the byte width of one element under the arrangement.
func (a ExtArrangement) Width() int {
switch a {
case ExtArrB:
return 1
case ExtArrH:
return 2
case ExtArrS:
return 4
case ExtArrD:
return 8
case ExtArrQ:
return 16
default:
return 0
}
}
// sizeBits maps the arrangement onto the two-bit size field the integer SVE
// classes carry at bits 23..22: 00=B, 01=H, 10=S, 11=D. ok is false for the
// arrangements no such class accepts (.Q and the bare spelling).
func (a ExtArrangement) sizeBits() (uint32, bool) {
switch a {
case ExtArrB, ExtArrH, ExtArrS, ExtArrD:
return uint32(a) - 1, true
default:
return 0, false
}
}
// ExtQualifier is the predicate qualifier spelled after the slash.
type ExtQualifier uint8
// Predicate qualifiers.
const (
ExtQualNone ExtQualifier = iota // bare Pn (non-predicating position)
ExtQualMerging // /M, inactive lanes keep the destination
ExtQualZeroing // /Z, inactive lanes become zero
)
// String returns the assembler spelling, with the leading slash.
func (q ExtQualifier) String() string {
switch q {
case ExtQualMerging:
return "/M"
case ExtQualZeroing:
return "/Z"
default:
return ""
}
}
// ExtOperand is one operand of an extended instruction, already resolved to
// its pieces: a register with its arrangement and qualifier, or an immediate
// with its optional left shift. The assembler's future hook constructs these
// from the parsed statement; Encode consumes them.
type ExtOperand struct {
Kind ExtOperandKind
Reg int // register number (Z: 0..31, P: 0..15)
Arr ExtArrangement // element-size suffix; ExtArrNone when bare
Qual ExtQualifier // predicate qualifier; ExtQualNone elsewhere
Imm int64 // immediate value (ExtImm only)
// Shift carries the LSL amount an immediate form shifts the constant by
// before use (0 or 8 in the SVE add/subtract immediate class). HasShift
// separates a spelled shift (validated as written) from an unshifted
// operand (the encoder may derive the sh bit from the value).
Shift int
HasShift bool
}
// ExtVector builds a scalable vector operand, ADD Z1.S style.
func ExtVector(reg int, arr ExtArrangement) ExtOperand {
return ExtOperand{Kind: ExtZReg, Reg: reg, Arr: arr}
}
// ExtPredicate builds a predicate operand with its qualifier, P0/M style.
func ExtPredicate(reg int, qual ExtQualifier) ExtOperand {
return ExtOperand{Kind: ExtPReg, Reg: reg, Qual: qual}
}
// ExtImmediate builds an unshifted immediate operand.
func ExtImmediate(v int64) ExtOperand {
return ExtOperand{Kind: ExtImm, Imm: v}
}
// ExtShiftedImmediate builds an immediate operand with a spelled LSL amount.
func ExtShiftedImmediate(v int64, shift int) ExtOperand {
return ExtOperand{Kind: ExtImm, Imm: v, Shift: shift, HasShift: true}
}
// ExtField is one named field of the 32-bit encoding word: a bit offset from
// the least significant end and the field's width.
type ExtField struct {
Off uint8
Width uint8
}
// extMask returns the field's bits as a mask.
func extMask(f ExtField) uint32 {
return ^uint32(0) >> (32 - f.Width)
}
// extSet ORs v into the field of word.
func extSet(word uint32, f ExtField, v uint32) uint32 {
return word | (v&extMask(f))<<f.Off
}
// The fields the SVE integer classes use. The 5-bit register fields are
// named after their role in the three-vector class; the predicated class
// reuses extFieldRn for its Zm operand and extFieldPg for the governing
// predicate, which that class narrows to three bits (P0-P7).
var (
extFieldRd = ExtField{0, 5} // destination (Zd or Zdn)
extFieldRn = ExtField{5, 5} // first source (Zn, or Zm in the predicated class)
extFieldRm = ExtField{16, 5} // second source (Zm in the three-vector class)
extFieldPg = ExtField{10, 3} // governing predicate P0-P7 (predicated class)
extFieldImm8 = ExtField{5, 8} // the immediate, bits 12..5
extFieldSh = ExtField{13, 1} // the shift flag: 1 means LSL #8
extSizeBHSD = ExtField{22, 2} // element-size field of every class here, bits 23..22
)
// ExtForm enumerates the operand shapes the extension layer defines, in Plan
// 9 order (sources first, destination last). A destructive SVE operand is
// written once, in destination position: the encoding carries no second copy.
type ExtForm uint8
// Operand shapes.
const (
// ExtFormVectors is the unpredicated three-vector form, the SVE integer
// add/subtract (unpredicated) class: ADD Z0.S, Z1.S, Z2.S computes
// Z0 = Z1 + Z2. Operands: Zn, Zm, Zd.
ExtFormVectors ExtForm = iota
// ExtFormPredicated is the governed destructive form, the SVE integer
// add/subtract vectors (predicated) class: ADD Z1.S, P0/M, Z0.S computes
// Z0 = Z0 + Z1 for the active lanes. Operands: Zm, Pg/M, Zdn. The
// governing predicate is a 3-bit field, so only P0-P7 encode here, and
// the class takes the merging qualifier alone: a zeroing form would need
// a MOVPRFX expansion, which one data word cannot carry.
ExtFormPredicated
// ExtFormImmediate is the add/subtract immediate form, the SVE integer
// add/subtract (immediate) class: ADD $255, Z0.S computes
// Z0 = Z0 + 255. Operands: imm{, LSL #8}, Zdn. The constant is an
// unsigned imm8, optionally shifted left by 8 bits; a bare multiple of
// 256 (up to 65280) derives the shift, the spelling the GNU assembler
// canonicalises too. .B takes no shift.
ExtFormImmediate
// ExtFormSignedImmediate is the signed immediate form of the SVE integer
// multiply (immediate) class: MUL $-128, Z0.B computes Z0 = Z0 * -128.
// Operands: simm8, Zdn. No shift exists in this class.
ExtFormSignedImmediate
)
// Arity returns the operand count the form takes.
func (f ExtForm) Arity() int {
switch f {
case ExtFormVectors, ExtFormPredicated:
return 3
case ExtFormImmediate, ExtFormSignedImmediate:
return 2
default:
return 0
}
}
// Kinds returns the operand kind each position of the form wants, in the
// order the operands arrive. The registry uses the list to pick the most
// specific rejection when every matching form refuses an operand list.
func (f ExtForm) Kinds() []ExtOperandKind {
switch f {
case ExtFormVectors:
return []ExtOperandKind{ExtZReg, ExtZReg, ExtZReg}
case ExtFormPredicated:
return []ExtOperandKind{ExtZReg, ExtPReg, ExtZReg}
case ExtFormImmediate, ExtFormSignedImmediate:
return []ExtOperandKind{ExtImm, ExtZReg}
default:
return nil
}
}
// String returns a short label for the form, for diagnostics.
func (f ExtForm) String() string {
switch f {
case ExtFormVectors:
return "unpredicated vectors"
case ExtFormPredicated:
return "predicated (merging)"
case ExtFormImmediate:
return "unsigned immediate"
case ExtFormSignedImmediate:
return "signed immediate"
default:
return "unknown form"
}
}
// ExtFeature names the architecture feature an extended instruction belongs
// to. The field is metadata: the assembler offers every instruction it
// registers, and a feature check is the caller's decision, not the encoder's.
type ExtFeature string
// The features the arm64 layer covers.
const (
ExtFeatureSVE ExtFeature = "sve"
ExtFeatureSVE2 ExtFeature = "sve2"
)
// ExtInstr is one extended instruction: the metadata a lookup needs and the
// encoding as data. Word holds the fixed bits of the 32-bit encoding with
// every operand field and the size field zero; the form says which fields the
// operands fill; the size field receives the arrangement's bits at encode
// time. Ref names the manual entry the encoding is transcribed from, the
// golden source in place of a toolchain oracle.
type ExtInstr struct {
Name string // upper-case mnemonic
Summary string // one line of hover documentation
Word uint32 // fixed encoding bits, operand fields zero
Form ExtForm // operand shape
Size ExtField // element-size field the arrangement fills
Feature ExtFeature // sve or sve2
Ref string // the ARM ARM entry the encoding comes from
}
// Encode assembles the operands into the 4 little-endian bytes of the
// instruction word. The operand kinds, register ranges, arrangements and
// immediate ranges are validated against the form; an operand the class
// cannot carry is an error, never a silent mis-encoding.
func (in ExtInstr) Encode(ops []ExtOperand) ([]byte, error) {
if len(ops) != in.Form.Arity() {
return nil, fmt.Errorf("%s: the %s form takes %d operands, got %d",
in.Name, in.Form, in.Form.Arity(), len(ops))
}
switch in.Form {
case ExtFormVectors:
return in.encodeVectors(ops)
case ExtFormPredicated:
return in.encodePredicated(ops)
case ExtFormImmediate:
return in.encodeImmediate(ops)
case ExtFormSignedImmediate:
return in.encodeSignedImmediate(ops)
default:
return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form)
}
}
// encodeVectors fills the unpredicated three-vector form: Zn, Zm, Zd, all
// under one required arrangement.
func (in ExtInstr) encodeVectors(ops []ExtOperand) ([]byte, error) {
for i, op := range ops {
if op.Kind != ExtZReg {
return nil, fmt.Errorf("%s: operand %d wants a scalable vector register, got %s",
in.Name, i+1, op.Kind)
}
if op.Reg < 0 || op.Reg > 31 {
return nil, fmt.Errorf("%s: operand %d is Z%d, outside Z0-Z31", in.Name, i+1, op.Reg)
}
}
arr, err := in.sharedArrangement(ops)
if err != nil {
return nil, err
}
size, ok := arr.sizeBits()
if !ok {
return nil, fmt.Errorf("%s: arrangement %s has no size encoding in this class", in.Name, arr)
}
word := in.Word
word = extSet(word, extFieldRn, uint32(ops[0].Reg))
word = extSet(word, extFieldRm, uint32(ops[1].Reg))
word = extSet(word, extFieldRd, uint32(ops[2].Reg))
word = extSet(word, in.Size, size)
return extWordLE(word), nil
}
// encodePredicated fills the governed destructive form: Zm, Pg/M, Zdn. The
// predicate is a 3-bit field, the merging qualifier alone, and carries no
// arrangement suffix in this class.
func (in ExtInstr) encodePredicated(ops []ExtOperand) ([]byte, error) {
zm, pg, zdn := ops[0], ops[1], ops[2]
if zm.Kind != ExtZReg {
return nil, fmt.Errorf("%s: operand 1 wants a scalable vector register, got %s",
in.Name, zm.Kind)
}
if zm.Reg < 0 || zm.Reg > 31 {
return nil, fmt.Errorf("%s: operand 1 is Z%d, outside Z0-Z31", in.Name, zm.Reg)
}
if pg.Kind != ExtPReg {
return nil, fmt.Errorf("%s: operand 2 wants a predicate register, got %s",
in.Name, pg.Kind)
}
if pg.Reg < 0 || pg.Reg > 7 {
return nil, fmt.Errorf("%s: operand 2 is P%d, outside P0-P7 in this class", in.Name, pg.Reg)
}
if pg.Qual != ExtQualMerging {
return nil, fmt.Errorf("%s: operand 2 wants the merging qualifier /M, got %q",
in.Name, pg.Qual)
}
if pg.Arr != ExtArrNone {
return nil, fmt.Errorf("%s: the governing predicate carries no arrangement suffix, got %s",
in.Name, pg.Arr)
}
if zdn.Kind != ExtZReg {
return nil, fmt.Errorf("%s: operand 3 wants a scalable vector register, got %s",
in.Name, zdn.Kind)
}
if zdn.Reg < 0 || zdn.Reg > 31 {
return nil, fmt.Errorf("%s: operand 3 is Z%d, outside Z0-Z31", in.Name, zdn.Reg)
}
if zm.Arr != zdn.Arr {
return nil, fmt.Errorf("%s: operands 1 and 3 carry arrangements %s and %s, they must match",
in.Name, zm.Arr, zdn.Arr)
}
size, ok := zdn.Arr.sizeBits()
if !ok {
return nil, fmt.Errorf("%s: arrangement %s has no size encoding in this class", in.Name, zdn.Arr)
}
word := in.Word
word = extSet(word, extFieldRn, uint32(zm.Reg))
word = extSet(word, extFieldPg, uint32(pg.Reg))
word = extSet(word, extFieldRd, uint32(zdn.Reg))
word = extSet(word, in.Size, size)
return extWordLE(word), nil
}
// encodeImmediate fills the add/subtract immediate form: imm{, LSL #8}, Zdn.
// The class encodes an unsigned imm8 with one shift bit, so a bare multiple
// of 256 derives the shift the way the GNU assembler canonicalises it.
func (in ExtInstr) encodeImmediate(ops []ExtOperand) ([]byte, error) {
imm, zdn := ops[0], ops[1]
imm8, sh, err := in.addSubImmediate(imm, zdn.Arr)
if err != nil {
return nil, err
}
word := in.Word
word = extSet(word, extFieldImm8, uint32(imm8))
if sh != 0 {
word = extSet(word, extFieldSh, 1)
}
word, err = in.setDestAndSize(word, zdn)
if err != nil {
return nil, err
}
return extWordLE(word), nil
}
// encodeSignedImmediate fills the multiply immediate form: simm8, Zdn, with
// no shift bit in the class.
func (in ExtInstr) encodeSignedImmediate(ops []ExtOperand) ([]byte, error) {
imm, zdn := ops[0], ops[1]
if imm.Kind != ExtImm {
return nil, fmt.Errorf("%s: operand 1 wants an immediate, got %s", in.Name, imm.Kind)
}
if imm.HasShift {
return nil, fmt.Errorf("%s: the signed immediate class takes no shift", in.Name)
}
if imm.Imm < -128 || imm.Imm > 127 {
return nil, fmt.Errorf("%s: immediate %d is outside the signed 8-bit range -128..127",
in.Name, imm.Imm)
}
word := in.Word
word = extSet(word, extFieldImm8, uint32(imm.Imm))
word, err := in.setDestAndSize(word, zdn)
if err != nil {
return nil, err
}
return extWordLE(word), nil
}
// addSubImmediate resolves the immediate operand of the add/subtract
// immediate class into its imm8 and shift bit: a spelled shift is validated
// as written, a bare multiple of 256 (on .H, .S or .D) derives one.
func (in ExtInstr) addSubImmediate(op ExtOperand, arr ExtArrangement) (imm8, sh int, err error) {
if op.Kind != ExtImm {
return 0, 0, fmt.Errorf("%s: operand 1 wants an immediate, got %s", in.Name, op.Kind)
}
switch {
case op.HasShift:
if op.Shift != 0 && op.Shift != 8 {
return 0, 0, fmt.Errorf("%s: the shift amount must be 0 or 8, got %d", in.Name, op.Shift)
}
if arr == ExtArrB && op.Shift != 0 {
return 0, 0, fmt.Errorf("%s: arrangement .B takes no shift", in.Name)
}
if op.Imm < 0 || op.Imm > 255 {
return 0, 0, fmt.Errorf("%s: immediate %d is outside the unsigned 8-bit range 0..255",
in.Name, op.Imm)
}
return int(op.Imm), op.Shift, nil
case op.Imm >= 0 && op.Imm <= 255:
return int(op.Imm), 0, nil
case arr != ExtArrB && op.Imm >= 256 && op.Imm <= 255<<8 && op.Imm%256 == 0:
// A bare multiple of 256 rides the shift bit, 65280 = 255<<8 included.
return int(op.Imm / 256), 8, nil
default:
return 0, 0, fmt.Errorf("%s: immediate %d is not an unsigned imm8%s, nor a multiple of 256 the shift bit can carry",
in.Name, op.Imm, arr.shiftNote())
}
}
// shiftNote describes where a shifted constant is expressible, for the
// immediate range error.
func (arr ExtArrangement) shiftNote() string {
if arr == ExtArrB {
return " (and .B takes no shifted constant)"
}
return " (a multiple of 256 up to 65280 shifts)"
}
// setDestAndSize fills the destructive destination register and the size
// field from the arrangement the vector carries.
func (in ExtInstr) setDestAndSize(word uint32, zdn ExtOperand) (uint32, error) {
if zdn.Kind != ExtZReg {
return 0, fmt.Errorf("%s: operand 2 wants a scalable vector register, got %s",
in.Name, zdn.Kind)
}
if zdn.Reg < 0 || zdn.Reg > 31 {
return 0, fmt.Errorf("%s: operand 2 is Z%d, outside Z0-Z31", in.Name, zdn.Reg)
}
size, ok := zdn.Arr.sizeBits()
if !ok {
return 0, fmt.Errorf("%s: arrangement %s has no size encoding in this class", in.Name, zdn.Arr)
}
word = extSet(word, extFieldRd, uint32(zdn.Reg))
word = extSet(word, in.Size, size)
return word, nil
}
// sharedArrangement returns the one arrangement all vector operands carry, or
// an error when any operand is bare or they disagree.
func (in ExtInstr) sharedArrangement(ops []ExtOperand) (ExtArrangement, error) {
arr := ops[0].Arr
for i, op := range ops {
if op.Arr == ExtArrNone {
return 0, fmt.Errorf("%s: operand %d carries no arrangement suffix", in.Name, i+1)
}
if op.Arr != arr {
return 0, fmt.Errorf("%s: operand %d carries arrangement %s, want %s",
in.Name, i+1, op.Arr, arr)
}
}
return arr, nil
}
// extWordLE returns a 32-bit encoding word as 4 little-endian bytes.
func extWordLE(w uint32) []byte {
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
}
// --- the arm64 SVE/SVE2 table ------------------------------------------------
// arm64Extensions is the extended instruction layer of arm64: the SVE and
// SVE2 integer add/subtract/multiply family. The Go toolchain knows none of
// these; the encodings are transcribed from the ARM Architecture Reference
// Manual (DDI 0487J, Part C, Chapter C8, the alphabetical list of SVE
// instructions) and cross-checked against the GNU assembler's and LLVM's
// published encodings.
var arm64Extensions = []ExtInstr{
// Unpredicated three-vector forms: ADD Z0.S, Z1.S, Z2.S.
{Name: "ADD", Summary: "Add scalable vector elements, unpredicated",
Word: 0x04200000, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: ADD (vectors, unpredicated)"},
{Name: "SUB", Summary: "Subtract scalable vector elements, unpredicated",
Word: 0x04200400, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUB (vectors, unpredicated)"},
{Name: "SQADD", Summary: "Add signed saturating scalable vector elements, unpredicated",
Word: 0x04201000, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SQADD (vectors, unpredicated)"},
{Name: "UQADD", Summary: "Add unsigned saturating scalable vector elements, unpredicated",
Word: 0x04201400, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UQADD (vectors, unpredicated)"},
{Name: "SQSUB", Summary: "Subtract signed saturating scalable vector elements, unpredicated",
Word: 0x04201800, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SQSUB (vectors, unpredicated)"},
{Name: "UQSUB", Summary: "Subtract unsigned saturating scalable vector elements, unpredicated",
Word: 0x04201c00, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UQSUB (vectors, unpredicated)"},
{Name: "MUL", Summary: "Multiply scalable vector elements, unpredicated",
Word: 0x04206000, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE2,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: MUL (vectors, unpredicated)"},
{Name: "SMULH", Summary: "Multiply signed scalable vector elements, keeping the high half, unpredicated",
Word: 0x04206800, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE2,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SMULH (vectors, unpredicated)"},
{Name: "UMULH", Summary: "Multiply unsigned scalable vector elements, keeping the high half, unpredicated",
Word: 0x04206c00, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE2,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UMULH (vectors, unpredicated)"},
// Governed destructive forms, merging: ADD Z1.S, P0/M, Z0.S.
{Name: "ADD", Summary: "Add scalable vector elements under a governing predicate, merging",
Word: 0x04000000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: ADD (vectors, predicated)"},
{Name: "SUB", Summary: "Subtract scalable vector elements under a governing predicate, merging",
Word: 0x04010000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUB (vectors, predicated)"},
{Name: "SUBR", Summary: "Reverse-subtract scalable vector elements under a governing predicate, merging",
Word: 0x04030000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUBR (vectors, predicated)"},
{Name: "MUL", Summary: "Multiply scalable vector elements under a governing predicate, merging",
Word: 0x04100000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: MUL (vectors, predicated)"},
{Name: "SMULH", Summary: "Multiply signed scalable vector elements, keeping the high half, under a governing predicate, merging",
Word: 0x04120000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SMULH (vectors, predicated)"},
{Name: "UMULH", Summary: "Multiply unsigned scalable vector elements, keeping the high half, under a governing predicate, merging",
Word: 0x04130000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UMULH (vectors, predicated)"},
// Immediate forms: ADD $255, Z0.S.
{Name: "ADD", Summary: "Add an unsigned immediate to scalable vector elements",
Word: 0x2520c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: ADD (vectors, immediate)"},
{Name: "SUB", Summary: "Subtract an unsigned immediate from scalable vector elements",
Word: 0x2521c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUB (vectors, immediate)"},
{Name: "SUBR", Summary: "Subtract scalable vector elements from an unsigned immediate",
Word: 0x2523c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUBR (vectors, immediate)"},
{Name: "SQADD", Summary: "Add a signed saturating unsigned immediate to scalable vector elements",
Word: 0x2524c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SQADD (vectors, immediate)"},
{Name: "UQADD", Summary: "Add an unsigned saturating immediate to scalable vector elements",
Word: 0x2525c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UQADD (vectors, immediate)"},
{Name: "SQSUB", Summary: "Subtract an unsigned immediate from scalable vector elements with signed saturation",
Word: 0x2526c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SQSUB (vectors, immediate)"},
{Name: "UQSUB", Summary: "Subtract an unsigned immediate from scalable vector elements with unsigned saturation",
Word: 0x2527c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UQSUB (vectors, immediate)"},
// The signed immediate of the multiply class: MUL $-128, Z0.B.
{Name: "MUL", Summary: "Multiply scalable vector elements by a signed immediate",
Word: 0x2530c000, Form: ExtFormSignedImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: MUL (vectors, immediate)"},
}
// Extensions returns the extended-instruction layer registered for a, outside
// the generated tables. An architecture whose extended layer is not built
// yet returns nothing: the mechanism is ordinary code, not a build tag, and
// it simply offers no instruction where none is registered.
func Extensions(a Arch) []ExtInstr {
switch a {
case ARM64:
return arm64Extensions
default:
return nil
}
}
+400
View File
@@ -0,0 +1,400 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
import (
"encoding/hex"
"strings"
"testing"
)
// The SVE encodings have no toolchain oracle: go tool asm knows no SVE at
// all. The golden words below are therefore transcribed from the ARM
// Architecture Reference Manual (DDI 0487J, Part C, Chapter C8, the
// alphabetical list of SVE instructions) and cross-checked against two
// independent implementations of the manual, the GNU assembler and LLVM:
// the rows marked "GNU" match a vector in binutils-gdb's own
// gas/testsuite/gas/aarch64/sve.d (assembled under -march=armv8-a+sve), the
// rows marked "LLVM" match the Inst field assignments in
// SVEInstrFormats.td's sve_int_bin_cons_arit_0, sve_int_bin_pred_arit_log,
// sve_int_arith_imm0 and sve2_int_mul classes. Every class is covered by at
// least one vector of each source.
func extInstruction(t *testing.T, mnem string, form ExtForm) ExtInstr {
t.Helper()
for _, in := range Extensions(ARM64) {
if in.Name == mnem && in.Form == form {
return in
}
}
t.Fatalf("no extended %s with the %s form", mnem, form)
return ExtInstr{}
}
func TestArm64ExtGoldenBytes(t *testing.T) {
for _, tt := range []struct {
name string
mnem string
form ExtForm
ops []ExtOperand
want uint32
GNUas string // the matching binutils-gdb sve.d line, empty when the class evidence comes from LLVM alone
}{
// Unpredicated three-vector forms: Zn, Zm, Zd.
{"add z0.b, z0.b, z0.b", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04200000, "04200000 add z0.b, z0.b, z0.b"},
{"add z0.b, z0.b, z31.b", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(31, ExtArrB), ExtVector(0, ExtArrB)},
0x043f0000, "043f0000 add z0.b, z0.b, z31.b"},
{"add z31.b, z0.b, z0.b", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(31, ExtArrB)},
0x0420001f, "0420001f add z31.b, z0.b, z0.b"},
{"add z0.b, z2.b, z0.b", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(2, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04200040, "04200040 add z0.b, z2.b, z0.b"},
{"add z0.h, z0.h, z0.h", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrH), ExtVector(0, ExtArrH), ExtVector(0, ExtArrH)},
0x04600000, "04600000 add z0.h, z0.h, z0.h"},
{"add z0.s, z0.s, z0.s", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrS), ExtVector(0, ExtArrS), ExtVector(0, ExtArrS)},
0x04a00000, "04a00000 add z0.s, z0.s, z0.s"},
{"add z0.d, z0.d, z0.d", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrD), ExtVector(0, ExtArrD), ExtVector(0, ExtArrD)},
0x04e00000, "04e00000 add z0.d, z0.d, z0.d"},
{"sub z0.b, z0.b, z0.b", "SUB", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04200400, "04200400 sub z0.b, z0.b, z0.b"},
{"sub z0.b, z0.b, z3.b", "SUB", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(3, ExtArrB), ExtVector(0, ExtArrB)},
0x04230400, "04230400 sub z0.b, z0.b, z3.b"},
{"sqadd z0.b, z0.b, z0.b", "SQADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04201000, "04201000 sqadd z0.b, z0.b, z0.b"},
{"sqadd z0.b, z0.b, z3.b", "SQADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(3, ExtArrB), ExtVector(0, ExtArrB)},
0x04231000, "04231000 sqadd z0.b, z0.b, z3.b"},
{"sqadd z0.d, z0.d, z0.d", "SQADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrD), ExtVector(0, ExtArrD), ExtVector(0, ExtArrD)},
0x04e01000, "04e01000 sqadd z0.d, z0.d, z0.d"},
{"uqadd z0.b, z0.b, z0.b", "UQADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04201400, "04201400 uqadd z0.b, z0.b, z0.b"},
{"sqsub z0.b, z0.b, z0.b", "SQSUB", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04201800, "04201800 sqsub z0.b, z0.b, z0.b"},
{"sqsub z0.h, z0.h, z0.h", "SQSUB", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrH), ExtVector(0, ExtArrH), ExtVector(0, ExtArrH)},
0x04601800, "04601800 sqsub z0.h, z0.h, z0.h"},
{"uqsub z0.b, z0.b, z0.b", "UQSUB", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04201c00, "04201c00 uqsub z0.b, z0.b, z0.b"},
{"mul z0.b, z0.b, z0.b (sve2)", "MUL", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04206000, "04206000 mul z0.b, z0.b, z0.b"},
{"mul z17.b, z21.b, z27.b (sve2)", "MUL", ExtFormVectors,
[]ExtOperand{ExtVector(21, ExtArrB), ExtVector(27, ExtArrB), ExtVector(17, ExtArrB)},
0x043b62b1, "043b62b1 mul z17.b, z21.b, z27.b"},
{"mul z0.d, z0.d, z0.d (sve2)", "MUL", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrD), ExtVector(0, ExtArrD), ExtVector(0, ExtArrD)},
0x04e06000, "04e06000 mul z0.d, z0.d, z0.d"},
{"smulh z0.b, z0.b, z0.b (sve2)", "SMULH", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04206800, "04206800 smulh z0.b, z0.b, z0.b"},
{"smulh z17.b, z21.b, z27.b (sve2)", "SMULH", ExtFormVectors,
[]ExtOperand{ExtVector(21, ExtArrB), ExtVector(27, ExtArrB), ExtVector(17, ExtArrB)},
0x043b6ab1, "043b6ab1 smulh z17.b, z21.b, z27.b"},
{"umulh z0.b, z0.b, z0.b (sve2)", "UMULH", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
0x04206c00, "04206c00 umulh z0.b, z0.b, z0.b"},
{"umulh z17.b, z21.b, z27.b (sve2)", "UMULH", ExtFormVectors,
[]ExtOperand{ExtVector(21, ExtArrB), ExtVector(27, ExtArrB), ExtVector(17, ExtArrB)},
0x043b6eb1, "043b6eb1 umulh z17.b, z21.b, z27.b"},
// Governed destructive forms, merging: Zm, Pg/M, Zdn.
{"add z0.b, p0/m, z0.b", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04000000, "04000000 add z0.b, p0/m, z0.b, z0.b"},
{"add z0.b, p2/m, z0.b", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(2, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04000800, "04000800 add z0.b, p2/m, z0.b, z0.b"},
{"add z0.b, p7/m, z0.b", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(7, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04001c00, "04001c00 add z0.b, p7/m, z0.b, z0.b"},
{"add z31.b, p0/m, z31.b", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(31, ExtArrB)},
0x0400001f, "0400001f add z31.b, p0/m, z31.b, z0.b"},
{"add z0.b, p0/m, z0.b (zm 31)", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(31, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x040003e0, "040003e0 add z0.b, p0/m, z0.b, z31.b"},
{"add z0.s, p0/m, z0.s", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrS), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrS)},
0x04800000, "04800000 add z0.s, p0/m, z0.s, z0.s"},
{"sub z0.b, p0/m, z0.b", "SUB", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04010000, "04010000 sub z0.b, p0/m, z0.b, z0.b"},
{"sub z0.b, p7/m, z0.b", "SUB", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(7, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04011c00, "04011c00 sub z0.b, p7/m, z0.b, z0.b"},
{"sub z3.b, p0/m, z3.b", "SUB", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(3, ExtArrB)},
0x04010003, "04010003 sub z3.b, p0/m, z3.b, z0.b"},
{"subr z0.b, p0/m, z0.b", "SUBR", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04030000, "04030000 subr z0.b, p0/m, z0.b, z0.b"},
{"subr z0.h, p0/m, z0.h", "SUBR", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrH), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrH)},
0x04430000, "04430000 subr z0.h, p0/m, z0.h, z0.h"},
{"mul z0.b, p0/m, z0.b", "MUL", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04100000, "04100000 mul z0.b, p0/m, z0.b, z0.b"},
{"mul z0.b, p2/m, z0.b", "MUL", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(2, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04100800, "04100800 mul z0.b, p2/m, z0.b, z0.b"},
{"mul z0.b, p0/m, z0.b (zm 31)", "MUL", ExtFormPredicated,
[]ExtOperand{ExtVector(31, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x041003e0, "041003e0 mul z0.b, p0/m, z0.b, z31.b"},
{"smulh z0.b, p0/m, z0.b", "SMULH", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04120000, "04120000 smulh z0.b, p0/m, z0.b, z0.b"},
{"smulh z0.b, p2/m, z0.b", "SMULH", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(2, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04120800, "04120800 smulh z0.b, p2/m, z0.b, z0.b"},
{"smulh z0.s, p0/m, z0.s", "SMULH", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrS), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrS)},
0x04920000, "04920000 smulh z0.s, p0/m, z0.s, z0.s"},
{"umulh z0.b, p0/m, z0.b", "UMULH", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
0x04130000, "04130000 umulh z0.b, p0/m, z0.b, z0.b"},
// Immediate forms: imm{, LSL #8}, Zdn.
{"add z0.b, z0.b, #0", "ADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2520c000, "2520c000 add z0.b, z0.b, #0"},
{"add z0.b, z0.b, #127", "ADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(127), ExtVector(0, ExtArrB)},
0x2520cfe0, "2520cfe0 add z0.b, z0.b, #127"},
{"add z0.h, z0.h, #0, lsl #8", "ADD", ExtFormImmediate,
[]ExtOperand{ExtShiftedImmediate(0, 8), ExtVector(0, ExtArrH)},
0x2560e000, "2560e000 add z0.h, z0.h, #0, lsl #8"},
{"add z0.h, z0.h, #32512 (derived shift)", "ADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(32512), ExtVector(0, ExtArrH)},
0x2560efe0, "2560efe0 is sqsub's GNU word for #32512; the classes share the encoding"},
{"sub z0.b, z0.b, #0", "SUB", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2521c000, "2521c000 sub z0.b, z0.b, #0"},
{"subr z0.b, z0.b, #0", "SUBR", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2523c000, "2523c000 subr z0.b, z0.b, #0"},
{"subr z0.b, z0.b, #255", "SUBR", ExtFormImmediate,
[]ExtOperand{ExtImmediate(255), ExtVector(0, ExtArrB)},
0x2523dfe0, "2523dfe0 subr z0.b, z0.b, #255"},
{"subr z0.h, z0.h, #0, lsl #8", "SUBR", ExtFormImmediate,
[]ExtOperand{ExtShiftedImmediate(0, 8), ExtVector(0, ExtArrH)},
0x2563e000, "2563e000 subr z0.h, z0.h, #0, lsl #8"},
{"sqadd z0.b, z0.b, #0", "SQADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2524c000, "2524c000 sqadd z0.b, z0.b, #0"},
{"uqadd z0.b, z0.b, #0", "UQADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2525c000, "2525c000 uqadd z0.b, z0.b, #0"},
{"sqsub z0.b, z0.b, #0", "SQSUB", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2526c000, "2526c000 sqsub z0.b, z0.b, #0"},
{"sqsub z0.b, z0.b, #255", "SQSUB", ExtFormImmediate,
[]ExtOperand{ExtImmediate(255), ExtVector(0, ExtArrB)},
0x2526dfe0, "2526dfe0 sqsub z0.b, z0.b, #255"},
{"sqsub z0.h, z0.h, #0, lsl #8", "SQSUB", ExtFormImmediate,
[]ExtOperand{ExtShiftedImmediate(0, 8), ExtVector(0, ExtArrH)},
0x2566e000, "2566e000 sqsub z0.h, z0.h, #0, lsl #8"},
{"uqsub z0.b, z0.b, #0", "UQSUB", ExtFormImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2527c000, "2527c000 uqsub z0.b, z0.b, #0 (class vector from the GNU table and LLVM: sve_int_arith_imm0 opc 0b111)"},
{"mul z0.b, z0.b, #0", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
0x2530c000, "2530c000 mul z0.b, z0.b, #0"},
{"mul z0.b, z0.b, #127", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(127), ExtVector(0, ExtArrB)},
0x2530cfe0, "2530cfe0 mul z0.b, z0.b, #127"},
{"mul z0.b, z0.b, #-128", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(-128), ExtVector(0, ExtArrB)},
0x2530d000, "2530d000 mul z0.b, z0.b, #-128"},
{"mul z0.b, z0.b, #-1", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(-1), ExtVector(0, ExtArrB)},
0x2530dfe0, "2530dfe0 mul z0.b, z0.b, #-1"},
{"mul z0.h, z0.h, #0", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrH)},
0x2570c000, "2570c000 mul z0.h, z0.h, #0"},
} {
in := extInstruction(t, tt.mnem, tt.form)
got, err := in.Encode(tt.ops)
if err != nil {
t.Errorf("%s: encode: %v", tt.name, err)
continue
}
if want := hex.EncodeToString(extWordLE(tt.want)); hex.EncodeToString(got) != want {
t.Errorf("%s:\n got %x\n want %s", tt.name, got, want)
}
}
}
// TestArm64ExtGoldenSources pins the cross-check contract: every encoding
// class in the table carries at least one GNU-assembler vector, so no class
// rests on transcription alone.
func TestArm64ExtGoldenSources(t *testing.T) {
classes := map[ExtForm]bool{}
for _, tt := range []struct {
mnem string
form ExtForm
}{
{"ADD", ExtFormVectors}, {"SUB", ExtFormVectors}, {"SQADD", ExtFormVectors},
{"UQADD", ExtFormVectors}, {"SQSUB", ExtFormVectors}, {"UQSUB", ExtFormVectors},
{"MUL", ExtFormVectors}, {"SMULH", ExtFormVectors}, {"UMULH", ExtFormVectors},
{"ADD", ExtFormPredicated}, {"SUB", ExtFormPredicated}, {"SUBR", ExtFormPredicated},
{"MUL", ExtFormPredicated}, {"SMULH", ExtFormPredicated}, {"UMULH", ExtFormPredicated},
{"ADD", ExtFormImmediate}, {"SUB", ExtFormImmediate}, {"SUBR", ExtFormImmediate},
{"SQADD", ExtFormImmediate}, {"UQADD", ExtFormImmediate},
{"SQSUB", ExtFormImmediate}, {"UQSUB", ExtFormImmediate},
{"MUL", ExtFormSignedImmediate},
} {
if _, ok := extInstructionQuiet(tt.mnem, tt.form); !ok {
t.Errorf("the table lacks %s with the %s form", tt.mnem, tt.form)
}
classes[tt.form] = true
}
for _, form := range []ExtForm{ExtFormVectors, ExtFormPredicated, ExtFormImmediate, ExtFormSignedImmediate} {
if !classes[form] {
t.Errorf("no golden vectors cover the %s form", form)
}
}
}
func extInstructionQuiet(mnem string, form ExtForm) (ExtInstr, bool) {
for _, in := range Extensions(ARM64) {
if in.Name == mnem && in.Form == form {
return in, true
}
}
return ExtInstr{}, false
}
// TestArm64ExtTableIntegrity checks the metadata contract: every entry names
// its manual reference, summary and feature, and the element-size field sits
// at bits 23..22 where the manual puts it for every class in the family.
func TestArm64ExtTableIntegrity(t *testing.T) {
for _, in := range Extensions(ARM64) {
if in.Name == "" || in.Summary == "" || in.Ref == "" {
t.Errorf("%+v: name, summary and reference are mandatory", in)
}
if in.Feature != ExtFeatureSVE && in.Feature != ExtFeatureSVE2 {
t.Errorf("%s: feature %q is neither sve nor sve2", in.Name, in.Feature)
}
if in.Form.Arity() < 2 || in.Form.Arity() > 3 {
t.Errorf("%s: form %d carries an unusable arity %d", in.Name, in.Form, in.Form.Arity())
}
if in.Size.Off != 22 || in.Size.Width != 2 {
t.Errorf("%s: the size field sits at bits %d..%d, the classes here put it at 23..22",
in.Name, in.Size.Off, in.Size.Off+in.Size.Width-1)
}
// The destination register field and the element-size field are
// operands everywhere in this family, so Word carries both zero; the
// class opcodes live around them and stay where they are.
if in.Word&0x1f != 0 || in.Word&(0x3<<22) != 0 {
t.Errorf("%s: word %08x carries destination or size bits, want them zero", in.Name, in.Word)
}
}
}
func TestArm64ExtRejects(t *testing.T) {
rgb := func(rs ...int) []ExtOperand {
ops := make([]ExtOperand, len(rs))
for i, r := range rs {
ops[i] = ExtVector(r, ExtArrB)
}
return ops
}
for _, tt := range []struct {
name string
mnem string
form ExtForm
ops []ExtOperand
quote string // a fragment the error carries
}{
{"no arrangement", "ADD", ExtFormVectors,
[]ExtOperand{{Kind: ExtZReg, Reg: 0}, ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
"no arrangement"},
{"mismatched arrangements", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrS), ExtVector(0, ExtArrB)},
"want .B"},
{"wrong arity", "ADD", ExtFormVectors, rgb(0, 0), "takes 3 operands"},
{"predicate in a vector position", "ADD", ExtFormVectors,
[]ExtOperand{ExtPredicate(0, ExtQualNone), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
"scalable vector register"},
{"quadword arrangement has no size encoding", "ADD", ExtFormVectors,
[]ExtOperand{ExtVector(0, ExtArrQ), ExtVector(0, ExtArrQ), ExtVector(0, ExtArrQ)},
"no size encoding"},
{"zeroing qualifier", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualZeroing), ExtVector(0, ExtArrB)},
"/M"},
{"predicate beyond the 3-bit field", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(8, ExtQualMerging), ExtVector(0, ExtArrB)},
"P0-P7"},
{"predicate arrangement suffix", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtOperand{Kind: ExtPReg, Reg: 0, Qual: ExtQualMerging, Arr: ExtArrB}, ExtVector(0, ExtArrB)},
"arrangement"},
{"predicate operands disagree on arrangement", "ADD", ExtFormPredicated,
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrS)},
"must match"},
{"bare 256 on .B", "ADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(256), ExtVector(0, ExtArrB)},
"immediate 256"},
{"negative unsigned immediate", "ADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(-1), ExtVector(0, ExtArrB)},
"immediate -1"},
{"multiple of 256 beyond the imm8 span", "ADD", ExtFormImmediate,
[]ExtOperand{ExtImmediate(65536), ExtVector(0, ExtArrH)},
"immediate 65536"},
{"shift amount other than 0 or 8", "ADD", ExtFormImmediate,
[]ExtOperand{ExtShiftedImmediate(1, 4), ExtVector(0, ExtArrS)},
"0 or 8"},
{"shifted constant on .B", "ADD", ExtFormImmediate,
[]ExtOperand{ExtShiftedImmediate(1, 8), ExtVector(0, ExtArrB)},
".B takes no shift"},
{"register where the immediate belongs", "ADD", ExtFormImmediate,
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
"wants an immediate"},
{"signed immediate over the top", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(128), ExtVector(0, ExtArrB)},
"128"},
{"signed immediate under the floor", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtImmediate(-129), ExtVector(0, ExtArrB)},
"-129"},
{"shift in the signed class", "MUL", ExtFormSignedImmediate,
[]ExtOperand{ExtShiftedImmediate(1, 8), ExtVector(0, ExtArrB)},
"no shift"},
} {
in := extInstruction(t, tt.mnem, tt.form)
_, err := in.Encode(tt.ops)
if err == nil {
t.Errorf("%s: encode succeeded, want an error", tt.name)
continue
}
if !strings.Contains(err.Error(), tt.quote) {
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
}
}
}
// TestExtensionsArchBinding pins the registry's architecture binding: the
// extended layer exists for arm64 alone until an amd64 table attaches, and no
// other architecture sees a single SVE instruction.
func TestExtensionsArchBinding(t *testing.T) {
for _, a := range []Arch{AMD64, RISCV, LOONG64, Unknown} {
if got := Extensions(a); len(got) != 0 {
t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got))
}
}
if got := Extensions(ARM64); len(got) == 0 {
t.Error("Extensions(ARM64) is empty")
}
}
+108 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
@@ -364,6 +364,8 @@ var arm64GeneratedInstrs = []string{
"REVW",
"ROR",
"RORW",
"RPRFM",
"SB",
"SBC",
"SBCS",
"SBCSW",
@@ -477,23 +479,68 @@ var arm64GeneratedInstrs = []string{
"UXTH",
"UXTHW",
"UXTW",
"VABS",
"VADD",
"VADDP",
"VADDV",
"VAND",
"VBCAX",
"VBIC",
"VBIF",
"VBIT",
"VBSL",
"VCLS",
"VCLZ",
"VCMEQ",
"VCMGE",
"VCMGT",
"VCMHI",
"VCMHS",
"VCMLE",
"VCMLT",
"VCMTST",
"VCNT",
"VDUP",
"VEOR",
"VEOR3",
"VEXT",
"VFABS",
"VFADD",
"VFADDP",
"VFCMEQ",
"VFCMGE",
"VFCMGT",
"VFCMLE",
"VFCMLT",
"VFCVTL",
"VFCVTL2",
"VFCVTN",
"VFCVTN2",
"VFCVTZS",
"VFCVTZU",
"VFDIV",
"VFMAX",
"VFMAXNM",
"VFMAXNMP",
"VFMAXNMV",
"VFMAXP",
"VFMAXV",
"VFMIN",
"VFMINNM",
"VFMINNMP",
"VFMINNMV",
"VFMINP",
"VFMINV",
"VFMLA",
"VFMLS",
"VFMUL",
"VFNEG",
"VFRINTM",
"VFRINTN",
"VFRINTP",
"VFRINTZ",
"VFSQRT",
"VFSUB",
"VLD1",
"VLD1R",
"VLD2",
@@ -502,11 +549,17 @@ var arm64GeneratedInstrs = []string{
"VLD3R",
"VLD4",
"VLD4R",
"VMLA",
"VMLS",
"VMOV",
"VMOVD",
"VMOVI",
"VMOVQ",
"VMOVS",
"VMUL",
"VNEG",
"VNOT",
"VORN",
"VORR",
"VPMULL",
"VPMULL2",
@@ -515,14 +568,47 @@ var arm64GeneratedInstrs = []string{
"VREV16",
"VREV32",
"VREV64",
"VSCVTF",
"VSHADD",
"VSHL",
"VSHRN",
"VSHRN2",
"VSLI",
"VSMAX",
"VSMAXP",
"VSMAXV",
"VSMIN",
"VSMINP",
"VSMINV",
"VSMLAL",
"VSMLAL2",
"VSMLSL",
"VSMLSL2",
"VSMULL",
"VSMULL2",
"VSQABS",
"VSQADD",
"VSQNEG",
"VSQSHL",
"VSQSUB",
"VSQXTN",
"VSQXTN2",
"VSQXTUN",
"VSQXTUN2",
"VSRHADD",
"VSRI",
"VSRSHR",
"VSSHL",
"VSSHLL",
"VSSHLL2",
"VSSHR",
"VST1",
"VST2",
"VST3",
"VST4",
"VSUB",
"VSXTL",
"VSXTL2",
"VTBL",
"VTBX",
"VTRN1",
@@ -530,8 +616,27 @@ var arm64GeneratedInstrs = []string{
"VUADDLV",
"VUADDW",
"VUADDW2",
"VUCVTF",
"VUHADD",
"VUMAX",
"VUMAXP",
"VUMAXV",
"VUMIN",
"VUMINP",
"VUMINV",
"VUMLAL",
"VUMLAL2",
"VUMLSL",
"VUMLSL2",
"VUMULL",
"VUMULL2",
"VUQADD",
"VUQSHL",
"VUQSUB",
"VUQXTN",
"VUQXTN2",
"VURHADD",
"VUSHL",
"VUSHLL",
"VUSHLL2",
"VUSHR",
@@ -541,6 +646,8 @@ var arm64GeneratedInstrs = []string{
"VUZP1",
"VUZP2",
"VXAR",
"VXTN",
"VXTN2",
"VZIP1",
"VZIP2",
"WFE",
+1 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/util.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
+10 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
@@ -152,6 +152,8 @@ var loong64GeneratedInstrs = []string{
"FNMADDF",
"FNMSUBD",
"FNMSUBF",
"FRINTD",
"FRINTF",
"FSCALEBD",
"FSCALEBF",
"FSEL",
@@ -177,7 +179,10 @@ var loong64GeneratedInstrs = []string{
"FTINTWF",
"JIRL",
"LL",
"LLACQV",
"LLACQW",
"LLV",
"LLW",
"LU12IW",
"LU32ID",
"LU52ID",
@@ -248,7 +253,11 @@ var loong64GeneratedInstrs = []string{
"ROTR",
"ROTRV",
"SC",
"SCQ",
"SCRELV",
"SCRELW",
"SCV",
"SCW",
"SGT",
"SGTU",
"SLL",
+32 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
@@ -81,6 +81,9 @@ var riscvGeneratedInstrs = []string{
"CLD",
"CLDSP",
"CLI",
"CLMUL",
"CLMULH",
"CLMULR",
"CLUI",
"CLW",
"CLWSP",
@@ -95,13 +98,20 @@ var riscvGeneratedInstrs = []string{
"CSDSP",
"CSLLI",
"CSRAI",
"CSRC",
"CSRCI",
"CSRLI",
"CSRR",
"CSRRC",
"CSRRCI",
"CSRRS",
"CSRRSI",
"CSRRW",
"CSRRWI",
"CSRS",
"CSRSI",
"CSRW",
"CSRWI",
"CSUB",
"CSUBW",
"CSW",
@@ -259,6 +269,7 @@ var riscvGeneratedInstrs = []string{
"ORCB",
"ORI",
"ORN",
"PAUSE",
"RDCYCLE",
"RDINSTRET",
"RDTIME",
@@ -322,6 +333,8 @@ var riscvGeneratedInstrs = []string{
"VADDVI",
"VADDVV",
"VADDVX",
"VANDNVV",
"VANDNVX",
"VANDVI",
"VANDVV",
"VANDVX",
@@ -329,8 +342,17 @@ var riscvGeneratedInstrs = []string{
"VASUBUVX",
"VASUBVV",
"VASUBVX",
"VBREV8V",
"VBREVV",
"VCLMULHVV",
"VCLMULHVX",
"VCLMULVV",
"VCLMULVX",
"VCLZV",
"VCOMPRESSVM",
"VCPOPM",
"VCPOPV",
"VCTZV",
"VDIVUVV",
"VDIVUVX",
"VDIVVV",
@@ -743,10 +765,16 @@ var riscvGeneratedInstrs = []string{
"VREMUVX",
"VREMVV",
"VREMVX",
"VREV8V",
"VRGATHEREI16VV",
"VRGATHERVI",
"VRGATHERVV",
"VRGATHERVX",
"VROLVV",
"VROLVX",
"VRORVI",
"VRORVV",
"VRORVX",
"VRSUBVI",
"VRSUBVX",
"VS1RV",
@@ -950,6 +978,9 @@ var riscvGeneratedInstrs = []string{
"VWMULVX",
"VWREDSUMUVS",
"VWREDSUMVS",
"VWSLLVI",
"VWSLLVV",
"VWSLLVX",
"VWSUBUVV",
"VWSUBUVX",
"VWSUBUWV",
+15 -66
View File
@@ -11,7 +11,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestGOObjectAARCH64Structure checks the basic structure of the emitted
@@ -178,26 +178,10 @@ func main() {
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
}
defer os.RemoveAll(work)
st := parseBuildLog(t, buildLog, "main_arm64.s")
defer os.RemoveAll(st.work)
// Expand $WORK in the object path.
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
// Read the toolchain-produced object and assemble the same source with gasm.
// Assemble the same source with gasm and substitute the object.
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
if err != nil {
t.Fatal(err)
@@ -210,30 +194,16 @@ func main() {
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
gasmObj, err := img.GOObjectAARCH64("a64link", "main_arm64.s")
// The package path is "main", the prefix the Go code's references carry.
gasmObj, err := img.GOObjectAARCH64("main", "main_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
// Replace the toolchain-produced object with gasm's.
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
// Re-link.
if linkLine == "" {
t.Skip("could not find link command in build log")
}
// Expand $WORK in the link command.
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"),
gasmObj, "GOARCH=arm64")
// Verify the binary exists and contains the symbol.
binPath := filepath.Join(dir, "prog")
binPath := filepath.Join(dir, "prog2")
if _, err := os.Stat(binPath); err != nil {
t.Fatalf("binary not found: %v", err)
}
@@ -296,23 +266,8 @@ func main() {
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || linkLine == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
}
defer os.RemoveAll(work)
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
st := parseBuildLog(t, buildLog, "main_arm64.s")
defer os.RemoveAll(st.work)
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
if err != nil {
@@ -326,19 +281,13 @@ func main() {
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
gasmObj, err := img.GOObjectAARCH64("a64dlink", "main_arm64.s")
gasmObj, err := img.GOObjectAARCH64("main", "main_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
binData, err := os.ReadFile(filepath.Join(dir, "prog"))
substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"),
gasmObj, "GOARCH=arm64")
binData, err := os.ReadFile(filepath.Join(dir, "prog2"))
if err != nil {
t.Fatal(err)
}
+646 -110
View File
File diff suppressed because it is too large Load Diff
+152 -15
View File
@@ -33,7 +33,7 @@ import (
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
@@ -79,6 +79,11 @@ func arm64RegNum(name string) int {
return 17
case "R18":
return 18
case "R18_PLATFORM":
// The toolchain's Windows spelling: R18 is renamed R18_PLATFORM in
// cmd/asm/internal/arch so assembly cannot use it by accident, and
// sys_windows_arm64.s references it only through this name.
return 18
case "R19":
return 19
case "R20":
@@ -369,6 +374,7 @@ const (
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
a64FAcqRel // acquire/release: LDAR family, STLR family
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
a64FCASP // compare and swap pair: CASP
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
@@ -379,6 +385,7 @@ const (
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
a64FVMoviImm // SIMD move immediate: VMOVI $imm8, Vd.B8/B16
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
)
@@ -765,7 +772,7 @@ func init() {
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
// ---- system operations ----
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM", "RPRFM", "SYS", "SYSL", "TLBI", "SB", "PACIASP", "PACIBSP"} {
a64InstrTable[m] = a64Enc{format: a64FSys}
}
@@ -778,12 +785,20 @@ func init() {
a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000}
// ---- load/store pair (signed offset) ----
// The scale column of a64LoadTable does not reach the pair forms, so each
// entry states its own access width through the imm7 divisor the pair
// encoder derives from the opc field (8 for D, 4 for W and SW, 16 for Q).
a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000}
a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000}
a64InstrTable["LDPSW"] = a64Enc{format: a64FPair, op: 0x69400000}
a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000}
a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000}
a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000}
a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000}
a64InstrTable["FLDPS"] = a64Enc{format: a64FPair, op: 0x2d400000}
a64InstrTable["FSTPS"] = a64Enc{format: a64FPair, op: 0x2d000000}
a64InstrTable["FLDPQ"] = a64Enc{format: a64FPair, op: 0xad400000}
a64InstrTable["FSTPQ"] = a64Enc{format: a64FPair, op: 0xad000000}
// ---- acquire/release loads and stores ----
a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00}
@@ -801,11 +816,23 @@ func init() {
lse := map[string]uint32{
"CASALD": 0xc8e0fc00,
"CASALW": 0x88e0fc00,
"CASB": 0x08a07c00,
"CASAB": 0x08e07c00,
"CASH": 0x48a07c00,
"CASLD": 0xc8a0fc00,
"CASLH": 0x48a0fc00,
"CASAW": 0x88e07c00,
"CASAD": 0xc8e07c00,
"CASALH": 0x48e07c00,
"LDADDALD": 0xf8e00000,
"LDADDALW": 0xb8e00000,
"LDADDAD": 0xf8a00000,
"LDADDAW": 0xb8a00000,
"LDCLRALB": 0x38e01000,
"LDCLRALW": 0xb8e01000,
"LDCLRALD": 0xf8e01000,
"LDCLRAD": 0xf8a01000,
"LDCLRAW": 0xb8a01000,
"LDORALB": 0x38e03000,
"LDORALW": 0xb8e03000,
"LDORALD": 0xf8e03000,
@@ -884,6 +911,12 @@ func init() {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// Compare and swap pair: the second register of each pair is implicit
// (Rs+1 and Rt+1), so the encoding carries Rs and Rt alone over a preset
// fixed field (asm7.go atomicCASP).
a64InstrTable["CASPD"] = a64Enc{format: a64FCASP, op: 1<<30 | 0x41<<21 | 0x1f<<10}
a64InstrTable["CASPW"] = a64Enc{format: a64FCASP, op: 0x41<<21 | 0x1f<<10}
// ---- carry-setting/carry-using arithmetic and widening multiply ----
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
// register preset to ZR (bits 14:10 = 11111).
@@ -935,11 +968,13 @@ func init() {
a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000}
a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000}
a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000}
a64InstrTable["VMOVI"] = a64Enc{format: a64FVMoviImm}
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 7<<10}
a64InstrTable["VUSRA"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 5<<10}
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
@@ -952,6 +987,24 @@ func init() {
a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1}
// Multi-register structure accesses beyond VLD1/VST1: VLD2/VLD3/VLD4 and
// the replicate loads VLD2R/VLD3R, each with the post-index spelling.
a64InstrTable["VLD2"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD2.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD3"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD3.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD4"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD2R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD2R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD3R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD3R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST2"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST2.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST3"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST3.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST4"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST4.P"] = a64Enc{format: a64FVLDST, op: 1}
}
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
@@ -1115,6 +1168,78 @@ var a64SimdVTable = map[string]a64SimdVSpec{
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
// Saturating shifts, register forms (the immediate spellings route to
// a64FShiftImm).
"VSQSHL": {0x0e204c00, 0x7f, false},
"VUQSHL": {0x2e204c00, 0x7f, false},
}
// a64SimdNLForm classifies the narrow/long/wide SIMD families whose
// arrangement does not travel on every operand: the encoding's size and Q
// bits read off one designated operand and the element widths pair up across
// the operands.
type a64SimdNLForm uint8
const (
a64NLTwoNarrow a64SimdNLForm = iota // (Vn.wide, Vd.narrow): size/Q from Vd
a64NLTwoLong // (Vn.narrow, Vd.long): size/Q from Vn
a64NLThreeLongMul // (Vm.narrow, Vn.narrow, Vd.long): size/Q from Vn
a64NLThreeWide // (Vm.narrow, Vn.wide, Vd.wide): size/Q from Vn
a64NLThreeLongShift // ($sh, Vn.narrow, Vd.long): size/Q from Vn, immh = esize+sh
a64NLThreeNarrowShift // ($sh, Vn.wide, Vd.narrow): size/Q from Vd, immh = esize-sh
)
// a64SimdNLSpec is one narrow/long/wide instruction: the base word (U, opcode
// and fixed bits positioned) and the arrangement form. qonly marks the FCVT
// family, whose size field is fixed in the base and only the Q bit follows
// the driving arrangement.
type a64SimdNLSpec struct {
base uint32
form a64SimdNLForm
qonly bool
}
// a64SimdNLTable holds the families the arrangement-driven three-register and
// two-register encoders cannot express. The .2 spellings force the 128-bit
// side of the pair through their operand arrangements, so the base carries no
// arrangement bits of its own.
var a64SimdNLTable = map[string]a64SimdNLSpec{
"VSHRN": {0x0f008400, a64NLThreeNarrowShift, false},
"VSHRN2": {0x0f008400, a64NLThreeNarrowShift, false},
"VSXTL": {0x0f00a400, a64NLTwoLong, false},
"VSXTL2": {0x0f00a400, a64NLTwoLong, false},
"VUXTL": {0x2f00a400, a64NLTwoLong, false},
"VUXTL2": {0x2f00a400, a64NLTwoLong, false},
"VXTN": {0x0e202800, a64NLTwoNarrow, false},
"VXTN2": {0x0e202800, a64NLTwoNarrow, false},
"VSQXTN": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTN2": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTUN": {0x2e202800, a64NLTwoNarrow, false},
"VSQXTUN2": {0x2e202800, a64NLTwoNarrow, false},
"VUQXTN": {0x2e204800, a64NLTwoNarrow, false},
"VUQXTN2": {0x2e204800, a64NLTwoNarrow, false},
"VFCVTN": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTN2": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTL": {0x0e217800, a64NLTwoLong, true},
"VFCVTL2": {0x0e217800, a64NLTwoLong, true},
"VSSHLL": {0x0f00a400, a64NLThreeLongShift, false},
"VSSHLL2": {0x0f00a400, a64NLThreeLongShift, false},
"VUSHLL": {0x2f00a400, a64NLThreeLongShift, false},
"VUSHLL2": {0x2f00a400, a64NLThreeLongShift, false},
"VUADDW": {0x2e201000, a64NLThreeWide, false},
"VUADDW2": {0x2e201000, a64NLThreeWide, false},
"VUMULL": {0x2e20c000, a64NLThreeLongMul, false},
"VUMULL2": {0x2e20c000, a64NLThreeLongMul, false},
"VSMULL": {0x0e20c000, a64NLThreeLongMul, false},
"VSMULL2": {0x0e20c000, a64NLThreeLongMul, false},
"VUMLAL": {0x2e208000, a64NLThreeLongMul, false},
"VUMLAL2": {0x2e208000, a64NLThreeLongMul, false},
"VSMLAL": {0x0e208000, a64NLThreeLongMul, false},
"VSMLAL2": {0x0e208000, a64NLThreeLongMul, false},
"VUMLSL": {0x2e20a000, a64NLThreeLongMul, false},
"VUMLSL2": {0x2e20a000, a64NLThreeLongMul, false},
"VSMLSL": {0x0e20a000, a64NLThreeLongMul, false},
"VSMLSL2": {0x0e20a000, a64NLThreeLongMul, false},
}
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
@@ -1238,6 +1363,16 @@ var a64PRFOps = map[string]int{
var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000}
var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000}
// a64VLDNBase and a64VSTNBase hold the VLD2/VLD3/VLD4 and VST2/VST3/VST4
// fixed words (indexed by register count 2..4): the opcode field at bits
// 15:12 carries the access kind.
var a64VLDNBase = [5]uint32{0, 0, 0x0c408000, 0x0c404000, 0x0c400000}
var a64VSTNBase = [5]uint32{0, 0, 0x0c008000, 0x0c004000, 0x0c000000}
// a64VLDNReplicate holds the VLD2R/VLD3R fixed words beside the existing
// VLD1R (0x0d40c000) and VLD4R (0x0d60e000) bases.
var a64VLDNReplicate = [5]uint32{0, 0x0d40c000, 0x0d60c000, 0x0d40e000, 0x0d60e000}
// a64Vec is a parsed vector operand: the register number, the arrangement
// ("" when the operand spells none) and, for element forms, the lane index.
type a64Vec struct {
@@ -1358,23 +1493,25 @@ func a64VecListOf(ops []*ast.Operand, start int) (vs []a64Vec, end int, ok bool)
// a64LSType describes the load/store parameters for a MOV width mnemonic.
type a64LSType struct {
size int // 0=byte, 1=half, 2=word, 3=dword
V int // 0=integer, 1=FP
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
size int // 0=byte, 1=half, 2=word, 3=dword
V int // 0=integer, 1=FP
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
scale int // access width in bytes; the unsigned offset divides by it
}
// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters.
// For loads, opc selects signed vs unsigned; for stores, we flip the opc.
var a64LoadTable = map[string]a64LSType{
"MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
"MOVD": {3, 0, 1, 8}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1, 4}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2, 4}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1, 2}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2, 1}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1, 4}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1, 8}, // LDR D (64-bit FP)
"FMOVQ": {0, 1, 3, 16}, // LDR/STR Q (128-bit FP): opc=11 selects it
}
// a64StoreOpc returns the store opc for a given load type: integer and FP
+131 -2
View File
@@ -7,8 +7,8 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
func TestArm64LDRSTREncoding(t *testing.T) {
@@ -105,6 +105,7 @@ func TestArm64RegNum(t *testing.T) {
}{
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
{"R18_PLATFORM", 18},
{"F0", 0}, {"F4", 4}, {"F31", 31},
{"INVALID", -1}, {"X0", -1}, {"", -1},
}
@@ -861,6 +862,99 @@ func TestArm64SIMDElement(t *testing.T) {
}
}
// TestArm64GPIntoVector pins the whole-vector moves VMOV/VDUP Rs, Vd.<T>
// against `go tool asm -S` output (Go 1.27, arm64): word = Q | 7<<25 |
// imm5<<16 | 3<<10 | rs<<5 | rd, shared by both mnemonics, the form
// sys_windows_arm64.s and the bytealg loops use. The D1 destination is
// rejected, as the toolchain rejects it.
func TestArm64GPIntoVector(t *testing.T) {
got := arm64Words(t, "\tVMOV R5, V5.B16\n\tVMOV R1, V2.B8\n\tVMOV R3, V4.H4\n"+
"\tVMOV R9, V10.S4\n\tVMOV R7, V31.H8\n\tVMOV R11, V12.D2\n"+
"\tVDUP R5, V5.B16\n\tVDUP R9, V10.H8\n\tVMOV V4.B16, V20.B16\n")
want := []uint32{
0x4e010ca5, // VMOV R5, V5.B16
0x0e010c22, // VMOV R1, V2.B8
0x0e020c64, // VMOV R3, V4.H4
0x4e040d2a, // VMOV R9, V10.S4
0x4e020cff, // VMOV R7, V31.H8
0x4e080d6c, // VMOV R11, V12.D2
0x4e010ca5, // VDUP R5, V5.B16 (same word as VMOV)
0x4e020d2a, // VDUP R9, V10.H8
0x4ea41c94, // VMOV V4.B16, V20.B16 (vector to vector stays ORR)
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tVMOV R7, V8.D1\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("VMOV R7, V8.D1 assembled, want an arrangement error")
}
}
// TestArm64SimdTwoOperand pins the two-operand accumulate spellings
// VADD/VSUB Vm, Vn against `go tool asm -S` output (Go 1.27, arm64):
// word = 5<<28|7<<25|7<<21|1<<15|1<<10 for VADD (7<<28 for VSUB) with
// rf<<16 | rn<<5 | rn, bare V registers only (asm7.go case 89).
func TestArm64SimdTwoOperand(t *testing.T) {
got := arm64Words(t, "\tVADD V7, V8\n\tVSUB V7, V8\n\tVADD V1, V2\n\tVADD V0.B16, V1.B16, V2.B16\n")
want := []uint32{
0x5ee78508, // VADD V7, V8
0x7ee78508, // VSUB V7, V8
0x5ee18442, // VADD V1, V2
0x4e208422, // VADD arranged: the ordinary three-register path
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64TruncMove pins the truncating register moves against
// `go tool asm -S` output (Go 1.27, arm64): the signed forms lower to SXTB,
// SXTH and SXTW (SBFM), the unsigned byte and halfword forms to UXTB and
// UXTH (UBFM), MOVWU to a W ORR, and a narrow move out of the zero register
// drops to the W ORR too (asm7.go case 45).
func TestArm64TruncMove(t *testing.T) {
got := arm64Words(t, "\tMOVB R3, R4\n\tMOVH R5, R6\n\tMOVW R9, R10\n"+
"\tMOVBU R3, R4\n\tMOVHU R3, R4\n\tMOVWU R3, R4\n\tMOVD R3, R4\n"+
"\tMOVD ZR, R4\n\tMOVB ZR, R4\n\tMOVWU ZR, R5\n")
want := []uint32{
0x93401c64, // MOVB = SXTB
0x93403ca6, // MOVH = SXTH
0x93407d2a, // MOVW = SXTW
0xd3401c64, // MOVBU = UXTB
0xd3403c64, // MOVHU = UXTH
0x2a0303e4, // MOVWU = ORR W
0xaa0303e4, // MOVD = ORR X
0xaa1f03e4, // MOVD ZR, R4 keeps the X form
0x2a1f03e4, // MOVB ZR, R4 drops to the W form
0x2a1f03e5, // MOVWU ZR, R5
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDLoadStore pins the structure loads and stores.
func TestArm64SIMDLoadStore(t *testing.T) {
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+
@@ -957,6 +1051,41 @@ func TestArm64MOVK(t *testing.T) {
}
}
// TestArm64MOVKHighLane pins the shifted high-lane immediate the arm64 test
// kernels write: $(40000<<48) folds to a negative int64, and the toolchain
// reads the value as an unsigned 64-bit pattern when it picks the lane.
func TestArm64MOVKHighLane(t *testing.T) {
got := arm64Words(t, "\tMOVK $(40000<<48), R0\n\tMOVK $0x9c40000000000000, R1\n")
want := []uint32{
0xf2f38800, // MOVK $(40000<<48), R0 (go tool asm: f2f38800)
0xf2f38801, // MOVK hw=3
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64MoveWideZeroImmediate pins the toolchain's rejection of a zero
// immediate in the move-wide family (optab case 33: "zero shifts cannot be
// handled"): every lane is zero, so no hw field can carry it.
func TestArm64MoveWideZeroImmediate(t *testing.T) {
for _, mnem := range []string{"MOVK", "MOVZ", "MOVN"} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\t"+mnem+" $0, R0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", mnem, errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s $0: expected error, got nil", mnem)
}
}
}
// TestArm64LoadImm64 tests 64-bit immediate loading.
func TestArm64LoadImm64(t *testing.T) {
src := `#include "textflag.h"
+1 -1
View File
@@ -53,7 +53,7 @@ package asm
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
+1 -1
View File
@@ -7,7 +7,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// parseArm64File is a helper assembling one arm64 source file.
+588
View File
@@ -0,0 +1,588 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// arm64 system registers and system-instruction aliases.
//
// The tables are transcribed from the data the Go toolchain itself carries
// (cmd/internal/obj/arm64/sysRegEnc.go and the sysInstFields map of asm7.go),
// which the ARM ARM defines: every system register is the packed field set
// op0<<19 | op1<<16 | CRn<<12 | CRm<<8 | op2<<5, and the read/write flags are
// the toolchain's own access classification. The encoding tables live here so
// the encoder stays testable against the GOROOT testdata word for word.
// a64SysReg is one system register: the packed encoding fields and the
// directions the register supports.
type a64SysReg struct {
v uint32
read bool
write bool
}
// a64SysRegs maps the system register names the toolchain knows to their
// encodings. MRS reads 0xd5300000 | v | Rd and MSR writes
// 0xd5100000 | v | Rt.
var a64SysRegs = map[string]a64SysReg{
"ACTLR_EL1": a64SysReg{0x181020, true, true},
"AFSR0_EL1": a64SysReg{0x185100, true, true},
"AFSR1_EL1": a64SysReg{0x185120, true, true},
"AIDR_EL1": a64SysReg{0x1900e0, true, false},
"AMAIR_EL1": a64SysReg{0x18a300, true, true},
"AMCFGR_EL0": a64SysReg{0x1bd220, true, false},
"AMCGCR_EL0": a64SysReg{0x1bd240, true, false},
"AMCNTENCLR0_EL0": a64SysReg{0x1bd280, true, true},
"AMCNTENCLR1_EL0": a64SysReg{0x1bd300, true, true},
"AMCNTENSET0_EL0": a64SysReg{0x1bd2a0, true, true},
"AMCNTENSET1_EL0": a64SysReg{0x1bd320, true, true},
"AMCR_EL0": a64SysReg{0x1bd200, true, true},
"AMEVCNTR00_EL0": a64SysReg{0x1bd400, true, true},
"AMEVCNTR01_EL0": a64SysReg{0x1bd420, true, true},
"AMEVCNTR02_EL0": a64SysReg{0x1bd440, true, true},
"AMEVCNTR03_EL0": a64SysReg{0x1bd460, true, true},
"AMEVCNTR04_EL0": a64SysReg{0x1bd480, true, true},
"AMEVCNTR05_EL0": a64SysReg{0x1bd4a0, true, true},
"AMEVCNTR06_EL0": a64SysReg{0x1bd4c0, true, true},
"AMEVCNTR07_EL0": a64SysReg{0x1bd4e0, true, true},
"AMEVCNTR08_EL0": a64SysReg{0x1bd500, true, true},
"AMEVCNTR09_EL0": a64SysReg{0x1bd520, true, true},
"AMEVCNTR010_EL0": a64SysReg{0x1bd540, true, true},
"AMEVCNTR011_EL0": a64SysReg{0x1bd560, true, true},
"AMEVCNTR012_EL0": a64SysReg{0x1bd580, true, true},
"AMEVCNTR013_EL0": a64SysReg{0x1bd5a0, true, true},
"AMEVCNTR014_EL0": a64SysReg{0x1bd5c0, true, true},
"AMEVCNTR015_EL0": a64SysReg{0x1bd5e0, true, true},
"AMEVCNTR10_EL0": a64SysReg{0x1bdc00, true, true},
"AMEVCNTR11_EL0": a64SysReg{0x1bdc20, true, true},
"AMEVCNTR12_EL0": a64SysReg{0x1bdc40, true, true},
"AMEVCNTR13_EL0": a64SysReg{0x1bdc60, true, true},
"AMEVCNTR14_EL0": a64SysReg{0x1bdc80, true, true},
"AMEVCNTR15_EL0": a64SysReg{0x1bdca0, true, true},
"AMEVCNTR16_EL0": a64SysReg{0x1bdcc0, true, true},
"AMEVCNTR17_EL0": a64SysReg{0x1bdce0, true, true},
"AMEVCNTR18_EL0": a64SysReg{0x1bdd00, true, true},
"AMEVCNTR19_EL0": a64SysReg{0x1bdd20, true, true},
"AMEVCNTR110_EL0": a64SysReg{0x1bdd40, true, true},
"AMEVCNTR111_EL0": a64SysReg{0x1bdd60, true, true},
"AMEVCNTR112_EL0": a64SysReg{0x1bdd80, true, true},
"AMEVCNTR113_EL0": a64SysReg{0x1bdda0, true, true},
"AMEVCNTR114_EL0": a64SysReg{0x1bddc0, true, true},
"AMEVCNTR115_EL0": a64SysReg{0x1bdde0, true, true},
"AMEVTYPER00_EL0": a64SysReg{0x1bd600, true, false},
"AMEVTYPER01_EL0": a64SysReg{0x1bd620, true, false},
"AMEVTYPER02_EL0": a64SysReg{0x1bd640, true, false},
"AMEVTYPER03_EL0": a64SysReg{0x1bd660, true, false},
"AMEVTYPER04_EL0": a64SysReg{0x1bd680, true, false},
"AMEVTYPER05_EL0": a64SysReg{0x1bd6a0, true, false},
"AMEVTYPER06_EL0": a64SysReg{0x1bd6c0, true, false},
"AMEVTYPER07_EL0": a64SysReg{0x1bd6e0, true, false},
"AMEVTYPER08_EL0": a64SysReg{0x1bd700, true, false},
"AMEVTYPER09_EL0": a64SysReg{0x1bd720, true, false},
"AMEVTYPER010_EL0": a64SysReg{0x1bd740, true, false},
"AMEVTYPER011_EL0": a64SysReg{0x1bd760, true, false},
"AMEVTYPER012_EL0": a64SysReg{0x1bd780, true, false},
"AMEVTYPER013_EL0": a64SysReg{0x1bd7a0, true, false},
"AMEVTYPER014_EL0": a64SysReg{0x1bd7c0, true, false},
"AMEVTYPER015_EL0": a64SysReg{0x1bd7e0, true, false},
"AMEVTYPER10_EL0": a64SysReg{0x1bde00, true, true},
"AMEVTYPER11_EL0": a64SysReg{0x1bde20, true, true},
"AMEVTYPER12_EL0": a64SysReg{0x1bde40, true, true},
"AMEVTYPER13_EL0": a64SysReg{0x1bde60, true, true},
"AMEVTYPER14_EL0": a64SysReg{0x1bde80, true, true},
"AMEVTYPER15_EL0": a64SysReg{0x1bdea0, true, true},
"AMEVTYPER16_EL0": a64SysReg{0x1bdec0, true, true},
"AMEVTYPER17_EL0": a64SysReg{0x1bdee0, true, true},
"AMEVTYPER18_EL0": a64SysReg{0x1bdf00, true, true},
"AMEVTYPER19_EL0": a64SysReg{0x1bdf20, true, true},
"AMEVTYPER110_EL0": a64SysReg{0x1bdf40, true, true},
"AMEVTYPER111_EL0": a64SysReg{0x1bdf60, true, true},
"AMEVTYPER112_EL0": a64SysReg{0x1bdf80, true, true},
"AMEVTYPER113_EL0": a64SysReg{0x1bdfa0, true, true},
"AMEVTYPER114_EL0": a64SysReg{0x1bdfc0, true, true},
"AMEVTYPER115_EL0": a64SysReg{0x1bdfe0, true, true},
"AMUSERENR_EL0": a64SysReg{0x1bd260, true, true},
"APDAKeyHi_EL1": a64SysReg{0x182220, true, true},
"APDAKeyLo_EL1": a64SysReg{0x182200, true, true},
"APDBKeyHi_EL1": a64SysReg{0x182260, true, true},
"APDBKeyLo_EL1": a64SysReg{0x182240, true, true},
"APGAKeyHi_EL1": a64SysReg{0x182320, true, true},
"APGAKeyLo_EL1": a64SysReg{0x182300, true, true},
"APIAKeyHi_EL1": a64SysReg{0x182120, true, true},
"APIAKeyLo_EL1": a64SysReg{0x182100, true, true},
"APIBKeyHi_EL1": a64SysReg{0x182160, true, true},
"APIBKeyLo_EL1": a64SysReg{0x182140, true, true},
"CCSIDR2_EL1": a64SysReg{0x190040, true, false},
"CCSIDR_EL1": a64SysReg{0x190000, true, false},
"CLIDR_EL1": a64SysReg{0x190020, true, false},
"CNTFRQ_EL0": a64SysReg{0x1be000, true, true},
"CNTKCTL_EL1": a64SysReg{0x18e100, true, true},
"CNTP_CTL_EL0": a64SysReg{0x1be220, true, true},
"CNTP_CVAL_EL0": a64SysReg{0x1be240, true, true},
"CNTP_TVAL_EL0": a64SysReg{0x1be200, true, true},
"CNTPCT_EL0": a64SysReg{0x1be020, true, false},
"CNTPS_CTL_EL1": a64SysReg{0x1fe220, true, true},
"CNTPS_CVAL_EL1": a64SysReg{0x1fe240, true, true},
"CNTPS_TVAL_EL1": a64SysReg{0x1fe200, true, true},
"CNTV_CTL_EL0": a64SysReg{0x1be320, true, true},
"CNTV_CVAL_EL0": a64SysReg{0x1be340, true, true},
"CNTV_TVAL_EL0": a64SysReg{0x1be300, true, true},
"CNTVCT_EL0": a64SysReg{0x1be040, true, false},
"CONTEXTIDR_EL1": a64SysReg{0x18d020, true, true},
"CPACR_EL1": a64SysReg{0x181040, true, true},
"CSSELR_EL1": a64SysReg{0x1a0000, true, true},
"CTR_EL0": a64SysReg{0x1b0020, true, false},
"CurrentEL": a64SysReg{0x184240, true, false},
"DAIF": a64SysReg{0x1b4220, true, true},
"DBGAUTHSTATUS_EL1": a64SysReg{0x107ec0, true, false},
"DBGBCR0_EL1": a64SysReg{0x1000a0, true, true},
"DBGBCR1_EL1": a64SysReg{0x1001a0, true, true},
"DBGBCR2_EL1": a64SysReg{0x1002a0, true, true},
"DBGBCR3_EL1": a64SysReg{0x1003a0, true, true},
"DBGBCR4_EL1": a64SysReg{0x1004a0, true, true},
"DBGBCR5_EL1": a64SysReg{0x1005a0, true, true},
"DBGBCR6_EL1": a64SysReg{0x1006a0, true, true},
"DBGBCR7_EL1": a64SysReg{0x1007a0, true, true},
"DBGBCR8_EL1": a64SysReg{0x1008a0, true, true},
"DBGBCR9_EL1": a64SysReg{0x1009a0, true, true},
"DBGBCR10_EL1": a64SysReg{0x100aa0, true, true},
"DBGBCR11_EL1": a64SysReg{0x100ba0, true, true},
"DBGBCR12_EL1": a64SysReg{0x100ca0, true, true},
"DBGBCR13_EL1": a64SysReg{0x100da0, true, true},
"DBGBCR14_EL1": a64SysReg{0x100ea0, true, true},
"DBGBCR15_EL1": a64SysReg{0x100fa0, true, true},
"DBGBVR0_EL1": a64SysReg{0x100080, true, true},
"DBGBVR1_EL1": a64SysReg{0x100180, true, true},
"DBGBVR2_EL1": a64SysReg{0x100280, true, true},
"DBGBVR3_EL1": a64SysReg{0x100380, true, true},
"DBGBVR4_EL1": a64SysReg{0x100480, true, true},
"DBGBVR5_EL1": a64SysReg{0x100580, true, true},
"DBGBVR6_EL1": a64SysReg{0x100680, true, true},
"DBGBVR7_EL1": a64SysReg{0x100780, true, true},
"DBGBVR8_EL1": a64SysReg{0x100880, true, true},
"DBGBVR9_EL1": a64SysReg{0x100980, true, true},
"DBGBVR10_EL1": a64SysReg{0x100a80, true, true},
"DBGBVR11_EL1": a64SysReg{0x100b80, true, true},
"DBGBVR12_EL1": a64SysReg{0x100c80, true, true},
"DBGBVR13_EL1": a64SysReg{0x100d80, true, true},
"DBGBVR14_EL1": a64SysReg{0x100e80, true, true},
"DBGBVR15_EL1": a64SysReg{0x100f80, true, true},
"DBGCLAIMCLR_EL1": a64SysReg{0x1079c0, true, true},
"DBGCLAIMSET_EL1": a64SysReg{0x1078c0, true, true},
"DBGDTR_EL0": a64SysReg{0x130400, true, true},
"DBGDTRRX_EL0": a64SysReg{0x130500, true, false},
"DBGDTRTX_EL0": a64SysReg{0x130500, false, true},
"DBGPRCR_EL1": a64SysReg{0x101480, true, true},
"DBGWCR0_EL1": a64SysReg{0x1000e0, true, true},
"DBGWCR1_EL1": a64SysReg{0x1001e0, true, true},
"DBGWCR2_EL1": a64SysReg{0x1002e0, true, true},
"DBGWCR3_EL1": a64SysReg{0x1003e0, true, true},
"DBGWCR4_EL1": a64SysReg{0x1004e0, true, true},
"DBGWCR5_EL1": a64SysReg{0x1005e0, true, true},
"DBGWCR6_EL1": a64SysReg{0x1006e0, true, true},
"DBGWCR7_EL1": a64SysReg{0x1007e0, true, true},
"DBGWCR8_EL1": a64SysReg{0x1008e0, true, true},
"DBGWCR9_EL1": a64SysReg{0x1009e0, true, true},
"DBGWCR10_EL1": a64SysReg{0x100ae0, true, true},
"DBGWCR11_EL1": a64SysReg{0x100be0, true, true},
"DBGWCR12_EL1": a64SysReg{0x100ce0, true, true},
"DBGWCR13_EL1": a64SysReg{0x100de0, true, true},
"DBGWCR14_EL1": a64SysReg{0x100ee0, true, true},
"DBGWCR15_EL1": a64SysReg{0x100fe0, true, true},
"DBGWVR0_EL1": a64SysReg{0x1000c0, true, true},
"DBGWVR1_EL1": a64SysReg{0x1001c0, true, true},
"DBGWVR2_EL1": a64SysReg{0x1002c0, true, true},
"DBGWVR3_EL1": a64SysReg{0x1003c0, true, true},
"DBGWVR4_EL1": a64SysReg{0x1004c0, true, true},
"DBGWVR5_EL1": a64SysReg{0x1005c0, true, true},
"DBGWVR6_EL1": a64SysReg{0x1006c0, true, true},
"DBGWVR7_EL1": a64SysReg{0x1007c0, true, true},
"DBGWVR8_EL1": a64SysReg{0x1008c0, true, true},
"DBGWVR9_EL1": a64SysReg{0x1009c0, true, true},
"DBGWVR10_EL1": a64SysReg{0x100ac0, true, true},
"DBGWVR11_EL1": a64SysReg{0x100bc0, true, true},
"DBGWVR12_EL1": a64SysReg{0x100cc0, true, true},
"DBGWVR13_EL1": a64SysReg{0x100dc0, true, true},
"DBGWVR14_EL1": a64SysReg{0x100ec0, true, true},
"DBGWVR15_EL1": a64SysReg{0x100fc0, true, true},
"DCZID_EL0": a64SysReg{0x1b00e0, true, false},
"DISR_EL1": a64SysReg{0x18c120, true, true},
"DIT": a64SysReg{0x1b42a0, true, true},
"DLR_EL0": a64SysReg{0x1b4520, true, true},
"DSPSR_EL0": a64SysReg{0x1b4500, true, true},
"ELR_EL1": a64SysReg{0x184020, true, true},
"ERRIDR_EL1": a64SysReg{0x185300, true, false},
"ERRSELR_EL1": a64SysReg{0x185320, true, true},
"ERXADDR_EL1": a64SysReg{0x185460, true, true},
"ERXCTLR_EL1": a64SysReg{0x185420, true, true},
"ERXFR_EL1": a64SysReg{0x185400, true, false},
"ERXMISC0_EL1": a64SysReg{0x185500, true, true},
"ERXMISC1_EL1": a64SysReg{0x185520, true, true},
"ERXMISC2_EL1": a64SysReg{0x185540, true, true},
"ERXMISC3_EL1": a64SysReg{0x185560, true, true},
"ERXPFGCDN_EL1": a64SysReg{0x1854c0, true, true},
"ERXPFGCTL_EL1": a64SysReg{0x1854a0, true, true},
"ERXPFGF_EL1": a64SysReg{0x185480, true, false},
"ERXSTATUS_EL1": a64SysReg{0x185440, true, true},
"ESR_EL1": a64SysReg{0x185200, true, true},
"FAR_EL1": a64SysReg{0x186000, true, true},
"FPCR": a64SysReg{0x1b4400, true, true},
"FPSR": a64SysReg{0x1b4420, true, true},
"GCR_EL1": a64SysReg{0x1810c0, true, true},
"GMID_EL1": a64SysReg{0x31400, true, false},
"ICC_AP0R0_EL1": a64SysReg{0x18c880, true, true},
"ICC_AP0R1_EL1": a64SysReg{0x18c8a0, true, true},
"ICC_AP0R2_EL1": a64SysReg{0x18c8c0, true, true},
"ICC_AP0R3_EL1": a64SysReg{0x18c8e0, true, true},
"ICC_AP1R0_EL1": a64SysReg{0x18c900, true, true},
"ICC_AP1R1_EL1": a64SysReg{0x18c920, true, true},
"ICC_AP1R2_EL1": a64SysReg{0x18c940, true, true},
"ICC_AP1R3_EL1": a64SysReg{0x18c960, true, true},
"ICC_ASGI1R_EL1": a64SysReg{0x18cbc0, false, true},
"ICC_BPR0_EL1": a64SysReg{0x18c860, true, true},
"ICC_BPR1_EL1": a64SysReg{0x18cc60, true, true},
"ICC_CTLR_EL1": a64SysReg{0x18cc80, true, true},
"ICC_DIR_EL1": a64SysReg{0x18cb20, false, true},
"ICC_EOIR0_EL1": a64SysReg{0x18c820, false, true},
"ICC_EOIR1_EL1": a64SysReg{0x18cc20, false, true},
"ICC_HPPIR0_EL1": a64SysReg{0x18c840, true, false},
"ICC_HPPIR1_EL1": a64SysReg{0x18cc40, true, false},
"ICC_IAR0_EL1": a64SysReg{0x18c800, true, false},
"ICC_IAR1_EL1": a64SysReg{0x18cc00, true, false},
"ICC_IGRPEN0_EL1": a64SysReg{0x18ccc0, true, true},
"ICC_IGRPEN1_EL1": a64SysReg{0x18cce0, true, true},
"ICC_PMR_EL1": a64SysReg{0x184600, true, true},
"ICC_RPR_EL1": a64SysReg{0x18cb60, true, false},
"ICC_SGI0R_EL1": a64SysReg{0x18cbe0, false, true},
"ICC_SGI1R_EL1": a64SysReg{0x18cba0, false, true},
"ICC_SRE_EL1": a64SysReg{0x18cca0, true, true},
"ICV_AP0R0_EL1": a64SysReg{0x18c880, true, true},
"ICV_AP0R1_EL1": a64SysReg{0x18c8a0, true, true},
"ICV_AP0R2_EL1": a64SysReg{0x18c8c0, true, true},
"ICV_AP0R3_EL1": a64SysReg{0x18c8e0, true, true},
"ICV_AP1R0_EL1": a64SysReg{0x18c900, true, true},
"ICV_AP1R1_EL1": a64SysReg{0x18c920, true, true},
"ICV_AP1R2_EL1": a64SysReg{0x18c940, true, true},
"ICV_AP1R3_EL1": a64SysReg{0x18c960, true, true},
"ICV_BPR0_EL1": a64SysReg{0x18c860, true, true},
"ICV_BPR1_EL1": a64SysReg{0x18cc60, true, true},
"ICV_CTLR_EL1": a64SysReg{0x18cc80, true, true},
"ICV_DIR_EL1": a64SysReg{0x18cb20, false, true},
"ICV_EOIR0_EL1": a64SysReg{0x18c820, false, true},
"ICV_EOIR1_EL1": a64SysReg{0x18cc20, false, true},
"ICV_HPPIR0_EL1": a64SysReg{0x18c840, true, false},
"ICV_HPPIR1_EL1": a64SysReg{0x18cc40, true, false},
"ICV_IAR0_EL1": a64SysReg{0x18c800, true, false},
"ICV_IAR1_EL1": a64SysReg{0x18cc00, true, false},
"ICV_IGRPEN0_EL1": a64SysReg{0x18ccc0, true, true},
"ICV_IGRPEN1_EL1": a64SysReg{0x18cce0, true, true},
"ICV_PMR_EL1": a64SysReg{0x184600, true, true},
"ICV_RPR_EL1": a64SysReg{0x18cb60, true, false},
"ID_AA64AFR0_EL1": a64SysReg{0x180580, true, false},
"ID_AA64AFR1_EL1": a64SysReg{0x1805a0, true, false},
"ID_AA64DFR0_EL1": a64SysReg{0x180500, true, false},
"ID_AA64DFR1_EL1": a64SysReg{0x180520, true, false},
"ID_AA64ISAR0_EL1": a64SysReg{0x180600, true, false},
"ID_AA64ISAR1_EL1": a64SysReg{0x180620, true, false},
"ID_AA64MMFR0_EL1": a64SysReg{0x180700, true, false},
"ID_AA64MMFR1_EL1": a64SysReg{0x180720, true, false},
"ID_AA64MMFR2_EL1": a64SysReg{0x180740, true, false},
"ID_AA64PFR0_EL1": a64SysReg{0x180400, true, false},
"ID_AA64PFR1_EL1": a64SysReg{0x180420, true, false},
"ID_AA64ZFR0_EL1": a64SysReg{0x180480, true, false},
"ID_AFR0_EL1": a64SysReg{0x180160, true, false},
"ID_DFR0_EL1": a64SysReg{0x180140, true, false},
"ID_ISAR0_EL1": a64SysReg{0x180200, true, false},
"ID_ISAR1_EL1": a64SysReg{0x180220, true, false},
"ID_ISAR2_EL1": a64SysReg{0x180240, true, false},
"ID_ISAR3_EL1": a64SysReg{0x180260, true, false},
"ID_ISAR4_EL1": a64SysReg{0x180280, true, false},
"ID_ISAR5_EL1": a64SysReg{0x1802a0, true, false},
"ID_ISAR6_EL1": a64SysReg{0x1802e0, true, false},
"ID_MMFR0_EL1": a64SysReg{0x180180, true, false},
"ID_MMFR1_EL1": a64SysReg{0x1801a0, true, false},
"ID_MMFR2_EL1": a64SysReg{0x1801c0, true, false},
"ID_MMFR3_EL1": a64SysReg{0x1801e0, true, false},
"ID_MMFR4_EL1": a64SysReg{0x1802c0, true, false},
"ID_PFR0_EL1": a64SysReg{0x180100, true, false},
"ID_PFR1_EL1": a64SysReg{0x180120, true, false},
"ID_PFR2_EL1": a64SysReg{0x180380, true, false},
"ISR_EL1": a64SysReg{0x18c100, true, false},
"LORC_EL1": a64SysReg{0x18a460, true, true},
"LOREA_EL1": a64SysReg{0x18a420, true, true},
"LORID_EL1": a64SysReg{0x18a4e0, true, false},
"LORN_EL1": a64SysReg{0x18a440, true, true},
"LORSA_EL1": a64SysReg{0x18a400, true, true},
"MAIR_EL1": a64SysReg{0x18a200, true, true},
"MDCCINT_EL1": a64SysReg{0x100200, true, true},
"MDCCSR_EL0": a64SysReg{0x130100, true, false},
"MDRAR_EL1": a64SysReg{0x101000, true, false},
"MDSCR_EL1": a64SysReg{0x100240, true, true},
"MIDR_EL1": a64SysReg{0x180000, true, false},
"MPAM0_EL1": a64SysReg{0x18a520, true, true},
"MPAM1_EL1": a64SysReg{0x18a500, true, true},
"MPAMIDR_EL1": a64SysReg{0x18a480, true, false},
"MPIDR_EL1": a64SysReg{0x1800a0, true, false},
"MVFR0_EL1": a64SysReg{0x180300, true, false},
"MVFR1_EL1": a64SysReg{0x180320, true, false},
"MVFR2_EL1": a64SysReg{0x180340, true, false},
"NZCV": a64SysReg{0x1b4200, true, true},
"OSDLR_EL1": a64SysReg{0x101380, true, true},
"OSDTRRX_EL1": a64SysReg{0x100040, true, true},
"OSDTRTX_EL1": a64SysReg{0x100340, true, true},
"OSECCR_EL1": a64SysReg{0x100640, true, true},
"OSLAR_EL1": a64SysReg{0x101080, false, true},
"OSLSR_EL1": a64SysReg{0x101180, true, false},
"PAN": a64SysReg{0x184260, true, true},
"PAR_EL1": a64SysReg{0x187400, true, true},
"PMBIDR_EL1": a64SysReg{0x189ae0, true, false},
"PMBLIMITR_EL1": a64SysReg{0x189a00, true, true},
"PMBPTR_EL1": a64SysReg{0x189a20, true, true},
"PMBSR_EL1": a64SysReg{0x189a60, true, true},
"PMCCFILTR_EL0": a64SysReg{0x1befe0, true, true},
"PMCCNTR_EL0": a64SysReg{0x1b9d00, true, true},
"PMCEID0_EL0": a64SysReg{0x1b9cc0, true, false},
"PMCEID1_EL0": a64SysReg{0x1b9ce0, true, false},
"PMCNTENCLR_EL0": a64SysReg{0x1b9c40, true, true},
"PMCNTENSET_EL0": a64SysReg{0x1b9c20, true, true},
"PMCR_EL0": a64SysReg{0x1b9c00, true, true},
"PMEVCNTR0_EL0": a64SysReg{0x1be800, true, true},
"PMEVCNTR1_EL0": a64SysReg{0x1be820, true, true},
"PMEVCNTR2_EL0": a64SysReg{0x1be840, true, true},
"PMEVCNTR3_EL0": a64SysReg{0x1be860, true, true},
"PMEVCNTR4_EL0": a64SysReg{0x1be880, true, true},
"PMEVCNTR5_EL0": a64SysReg{0x1be8a0, true, true},
"PMEVCNTR6_EL0": a64SysReg{0x1be8c0, true, true},
"PMEVCNTR7_EL0": a64SysReg{0x1be8e0, true, true},
"PMEVCNTR8_EL0": a64SysReg{0x1be900, true, true},
"PMEVCNTR9_EL0": a64SysReg{0x1be920, true, true},
"PMEVCNTR10_EL0": a64SysReg{0x1be940, true, true},
"PMEVCNTR11_EL0": a64SysReg{0x1be960, true, true},
"PMEVCNTR12_EL0": a64SysReg{0x1be980, true, true},
"PMEVCNTR13_EL0": a64SysReg{0x1be9a0, true, true},
"PMEVCNTR14_EL0": a64SysReg{0x1be9c0, true, true},
"PMEVCNTR15_EL0": a64SysReg{0x1be9e0, true, true},
"PMEVCNTR16_EL0": a64SysReg{0x1bea00, true, true},
"PMEVCNTR17_EL0": a64SysReg{0x1bea20, true, true},
"PMEVCNTR18_EL0": a64SysReg{0x1bea40, true, true},
"PMEVCNTR19_EL0": a64SysReg{0x1bea60, true, true},
"PMEVCNTR20_EL0": a64SysReg{0x1bea80, true, true},
"PMEVCNTR21_EL0": a64SysReg{0x1beaa0, true, true},
"PMEVCNTR22_EL0": a64SysReg{0x1beac0, true, true},
"PMEVCNTR23_EL0": a64SysReg{0x1beae0, true, true},
"PMEVCNTR24_EL0": a64SysReg{0x1beb00, true, true},
"PMEVCNTR25_EL0": a64SysReg{0x1beb20, true, true},
"PMEVCNTR26_EL0": a64SysReg{0x1beb40, true, true},
"PMEVCNTR27_EL0": a64SysReg{0x1beb60, true, true},
"PMEVCNTR28_EL0": a64SysReg{0x1beb80, true, true},
"PMEVCNTR29_EL0": a64SysReg{0x1beba0, true, true},
"PMEVCNTR30_EL0": a64SysReg{0x1bebc0, true, true},
"PMEVTYPER0_EL0": a64SysReg{0x1bec00, true, true},
"PMEVTYPER1_EL0": a64SysReg{0x1bec20, true, true},
"PMEVTYPER2_EL0": a64SysReg{0x1bec40, true, true},
"PMEVTYPER3_EL0": a64SysReg{0x1bec60, true, true},
"PMEVTYPER4_EL0": a64SysReg{0x1bec80, true, true},
"PMEVTYPER5_EL0": a64SysReg{0x1beca0, true, true},
"PMEVTYPER6_EL0": a64SysReg{0x1becc0, true, true},
"PMEVTYPER7_EL0": a64SysReg{0x1bece0, true, true},
"PMEVTYPER8_EL0": a64SysReg{0x1bed00, true, true},
"PMEVTYPER9_EL0": a64SysReg{0x1bed20, true, true},
"PMEVTYPER10_EL0": a64SysReg{0x1bed40, true, true},
"PMEVTYPER11_EL0": a64SysReg{0x1bed60, true, true},
"PMEVTYPER12_EL0": a64SysReg{0x1bed80, true, true},
"PMEVTYPER13_EL0": a64SysReg{0x1beda0, true, true},
"PMEVTYPER14_EL0": a64SysReg{0x1bedc0, true, true},
"PMEVTYPER15_EL0": a64SysReg{0x1bede0, true, true},
"PMEVTYPER16_EL0": a64SysReg{0x1bee00, true, true},
"PMEVTYPER17_EL0": a64SysReg{0x1bee20, true, true},
"PMEVTYPER18_EL0": a64SysReg{0x1bee40, true, true},
"PMEVTYPER19_EL0": a64SysReg{0x1bee60, true, true},
"PMEVTYPER20_EL0": a64SysReg{0x1bee80, true, true},
"PMEVTYPER21_EL0": a64SysReg{0x1beea0, true, true},
"PMEVTYPER22_EL0": a64SysReg{0x1beec0, true, true},
"PMEVTYPER23_EL0": a64SysReg{0x1beee0, true, true},
"PMEVTYPER24_EL0": a64SysReg{0x1bef00, true, true},
"PMEVTYPER25_EL0": a64SysReg{0x1bef20, true, true},
"PMEVTYPER26_EL0": a64SysReg{0x1bef40, true, true},
"PMEVTYPER27_EL0": a64SysReg{0x1bef60, true, true},
"PMEVTYPER28_EL0": a64SysReg{0x1bef80, true, true},
"PMEVTYPER29_EL0": a64SysReg{0x1befa0, true, true},
"PMEVTYPER30_EL0": a64SysReg{0x1befc0, true, true},
"PMINTENCLR_EL1": a64SysReg{0x189e40, true, true},
"PMINTENSET_EL1": a64SysReg{0x189e20, true, true},
"PMMIR_EL1": a64SysReg{0x189ec0, true, false},
"PMOVSCLR_EL0": a64SysReg{0x1b9c60, true, true},
"PMOVSSET_EL0": a64SysReg{0x1b9e60, true, true},
"PMSCR_EL1": a64SysReg{0x189900, true, true},
"PMSELR_EL0": a64SysReg{0x1b9ca0, true, true},
"PMSEVFR_EL1": a64SysReg{0x1899a0, true, true},
"PMSFCR_EL1": a64SysReg{0x189980, true, true},
"PMSICR_EL1": a64SysReg{0x189940, true, true},
"PMSIDR_EL1": a64SysReg{0x1899e0, true, false},
"PMSIRR_EL1": a64SysReg{0x189960, true, true},
"PMSLATFR_EL1": a64SysReg{0x1899c0, true, true},
"PMSWINC_EL0": a64SysReg{0x1b9c80, false, true},
"PMUSERENR_EL0": a64SysReg{0x1b9e00, true, true},
"PMXEVCNTR_EL0": a64SysReg{0x1b9d40, true, true},
"PMXEVTYPER_EL0": a64SysReg{0x1b9d20, true, true},
"REVIDR_EL1": a64SysReg{0x1800c0, true, false},
"RGSR_EL1": a64SysReg{0x1810a0, true, true},
"RMR_EL1": a64SysReg{0x18c040, true, true},
"RNDR": a64SysReg{0x1b2400, true, false},
"RNDRRS": a64SysReg{0x1b2420, true, false},
"RVBAR_EL1": a64SysReg{0x18c020, true, false},
"SCTLR_EL1": a64SysReg{0x181000, true, true},
"SCXTNUM_EL0": a64SysReg{0x1bd0e0, true, true},
"SCXTNUM_EL1": a64SysReg{0x18d0e0, true, true},
"SP_EL0": a64SysReg{0x184100, true, true},
"SP_EL1": a64SysReg{0x1c4100, true, true},
"SPSel": a64SysReg{0x184200, true, true},
"SPSR_abt": a64SysReg{0x1c4320, true, true},
"SPSR_EL1": a64SysReg{0x184000, true, true},
"SPSR_fiq": a64SysReg{0x1c4360, true, true},
"SPSR_irq": a64SysReg{0x1c4300, true, true},
"SPSR_und": a64SysReg{0x1c4340, true, true},
"SSBS": a64SysReg{0x1b42c0, true, true},
"TCO": a64SysReg{0x1b42e0, true, true},
"TCR_EL1": a64SysReg{0x182040, true, true},
"TFSR_EL1": a64SysReg{0x185600, true, true},
"TFSRE0_EL1": a64SysReg{0x185620, true, true},
"TPIDR_EL0": a64SysReg{0x1bd040, true, true},
"TPIDR_EL1": a64SysReg{0x18d080, true, true},
"TPIDRRO_EL0": a64SysReg{0x1bd060, true, true},
"TRFCR_EL1": a64SysReg{0x181220, true, true},
"TTBR0_EL1": a64SysReg{0x182000, true, true},
"TTBR1_EL1": a64SysReg{0x182020, true, true},
"UAO": a64SysReg{0x184280, true, true},
"VBAR_EL1": a64SysReg{0x18c000, true, true},
"ZCR_EL1": a64SysReg{0x181200, true, true},
}
// a64SysInst is one TLBI alias: the fields the SYS encoding carries beside
// the fixed op0 = 01 and CRn = 8.
type a64SysInst struct {
op1, cm, op2 uint32
}
// a64TLBIOps maps the TLBI operation names to their fields; the register
// operand is optional and defaults to ZR.
var a64TLBIOps = map[string]a64SysInst{
"ALLE1": {0x4, 0x7, 0x4},
"ALLE1IS": {0x4, 0x3, 0x4},
"ALLE1OS": {0x4, 0x1, 0x4},
"ALLE2": {0x4, 0x7, 0x0},
"ALLE2IS": {0x4, 0x3, 0x0},
"ALLE2OS": {0x4, 0x1, 0x0},
"ALLE3": {0x6, 0x7, 0x0},
"ALLE3IS": {0x6, 0x3, 0x0},
"ALLE3OS": {0x6, 0x1, 0x0},
"ASIDE1": {0x0, 0x7, 0x2},
"ASIDE1IS": {0x0, 0x3, 0x2},
"ASIDE1OS": {0x0, 0x1, 0x2},
"IPAS2E1": {0x4, 0x4, 0x1},
"IPAS2E1IS": {0x4, 0x0, 0x1},
"IPAS2E1OS": {0x4, 0x4, 0x0},
"IPAS2LE1": {0x4, 0x4, 0x5},
"IPAS2LE1IS": {0x4, 0x0, 0x5},
"IPAS2LE1OS": {0x4, 0x4, 0x4},
"RIPAS2E1": {0x4, 0x4, 0x2},
"RIPAS2E1IS": {0x4, 0x0, 0x2},
"RIPAS2E1OS": {0x4, 0x4, 0x3},
"RIPAS2LE1": {0x4, 0x4, 0x6},
"RIPAS2LE1IS": {0x4, 0x0, 0x6},
"RIPAS2LE1OS": {0x4, 0x4, 0x7},
"RVAAE1": {0x0, 0x6, 0x3},
"RVAAE1IS": {0x0, 0x2, 0x3},
"RVAAE1OS": {0x0, 0x5, 0x3},
"RVAALE1": {0x0, 0x6, 0x7},
"RVAALE1IS": {0x0, 0x2, 0x7},
"RVAALE1OS": {0x0, 0x5, 0x7},
"RVAE1": {0x0, 0x6, 0x1},
"RVAE1IS": {0x0, 0x2, 0x1},
"RVAE1OS": {0x0, 0x5, 0x1},
"RVAE2": {0x4, 0x6, 0x1},
"RVAE2IS": {0x4, 0x2, 0x1},
"RVAE2OS": {0x4, 0x5, 0x1},
"RVAE3": {0x6, 0x6, 0x1},
"RVAE3IS": {0x6, 0x2, 0x1},
"RVAE3OS": {0x6, 0x5, 0x1},
"RVALE1": {0x0, 0x6, 0x5},
"RVALE1IS": {0x0, 0x2, 0x5},
"RVALE1OS": {0x0, 0x5, 0x5},
"RVALE2": {0x4, 0x6, 0x5},
"RVALE2IS": {0x4, 0x2, 0x5},
"RVALE2OS": {0x4, 0x5, 0x5},
"RVALE3": {0x6, 0x6, 0x5},
"RVALE3IS": {0x6, 0x2, 0x5},
"RVALE3OS": {0x6, 0x5, 0x5},
"VAAE1": {0x0, 0x7, 0x3},
"VAAE1IS": {0x0, 0x3, 0x3},
"VAAE1OS": {0x0, 0x1, 0x3},
"VAALE1": {0x0, 0x7, 0x7},
"VAALE1IS": {0x0, 0x3, 0x7},
"VAALE1OS": {0x0, 0x1, 0x7},
"VAE1": {0x0, 0x7, 0x1},
"VAE1IS": {0x0, 0x3, 0x1},
"VAE1OS": {0x0, 0x1, 0x1},
"VAE2": {0x4, 0x7, 0x1},
"VAE2IS": {0x4, 0x3, 0x1},
"VAE2OS": {0x4, 0x1, 0x1},
"VAE3": {0x6, 0x7, 0x1},
"VAE3IS": {0x6, 0x3, 0x1},
"VAE3OS": {0x6, 0x1, 0x1},
"VALE1": {0x0, 0x7, 0x5},
"VALE1IS": {0x0, 0x3, 0x5},
"VALE1OS": {0x0, 0x1, 0x5},
"VALE2": {0x4, 0x7, 0x5},
"VALE2IS": {0x4, 0x3, 0x5},
"VALE2OS": {0x4, 0x1, 0x5},
"VALE3": {0x6, 0x7, 0x5},
"VALE3IS": {0x6, 0x3, 0x5},
"VALE3OS": {0x6, 0x1, 0x5},
"VMALLE1": {0x0, 0x7, 0x0},
"VMALLE1IS": {0x0, 0x3, 0x0},
"VMALLE1OS": {0x0, 0x1, 0x0},
"VMALLS12E1": {0x4, 0x7, 0x6},
"VMALLS12E1IS": {0x4, 0x3, 0x6},
"VMALLS12E1OS": {0x4, 0x1, 0x6},
}
// a64DCOps2 maps the DC operation names to their fields; the register
// operand is mandatory.
var a64DCOps2 = map[string]a64SysInst{
"CGDSW": {0x0, 0xa, 0x6},
"CGDVAC": {0x3, 0xa, 0x5},
"CGDVADP": {0x3, 0xd, 0x5},
"CGDVAP": {0x3, 0xc, 0x5},
"CGSW": {0x0, 0xa, 0x4},
"CGVAC": {0x3, 0xa, 0x3},
"CGVADP": {0x3, 0xd, 0x3},
"CGVAP": {0x3, 0xc, 0x3},
"CIGDSW": {0x0, 0xe, 0x6},
"CIGDVAC": {0x3, 0xe, 0x5},
"CIGSW": {0x0, 0xe, 0x4},
"CIGVAC": {0x3, 0xe, 0x3},
"CISW": {0x0, 0xe, 0x2},
"CIVAC": {0x3, 0xe, 0x1},
"CSW": {0x0, 0xa, 0x2},
"CVAC": {0x3, 0xa, 0x1},
"CVADP": {0x3, 0xd, 0x1},
"CVAP": {0x3, 0xc, 0x1},
"CVAU": {0x3, 0xb, 0x1},
"GVA": {0x3, 0x4, 0x3},
"GZVA": {0x3, 0x4, 0x4},
"IGDSW": {0x0, 0x6, 0x6},
"IGDVAC": {0x0, 0x6, 0x5},
"IGSW": {0x0, 0x6, 0x4},
"IGVAC": {0x0, 0x6, 0x3},
"ISW": {0x0, 0x6, 0x2},
"IVAC": {0x0, 0x6, 0x1},
"ZVA": {0x3, 0x4, 0x1},
}
// a64RPRFOps maps the range-prefetch operation names to their 6-bit values.
var a64RPRFOps = map[string]uint32{
"PLDKEEP": 0,
"PLDSTRM": 4,
"PSTKEEP": 1,
"PSTSTRM": 5,
}
+164
View File
@@ -0,0 +1,164 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
"os"
"path/filepath"
"slices"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestARM64SysRegsDifferential proves the whole system-register table against
// the toolchain at once: one TEXT whose body reads every register the table
// carries (and writes every writable one), assembled by gasm and by
// go tool asm, must agree byte for byte. A single wrong op0/op1/CRn/CRm/op2
// packing names its register through the first differing word.
func TestARM64SysRegsDifferential(t *testing.T) {
names := make([]string, 0, len(a64SysRegs))
for name := range a64SysRegs {
names = append(names, name)
}
slices.Sort(names)
var body strings.Builder
for i, name := range names {
// R18 is the arm64 platform register and R29-R31 carry dedicated
// meanings; a plain read/write destination keeps to R0-R17.
reg := fmt.Sprintf("R%d", i%18)
if a64SysRegs[name].read {
body.WriteString(fmt.Sprintf("\tMRS %s, %s\n", name, reg))
}
if a64SysRegs[name].write {
body.WriteString(fmt.Sprintf("\tMSR %s, %s\n", reg, name))
}
}
src := "#include \"textflag.h\"\n\nTEXT ·sysregs(SB), NOSPLIT, $0\n" + body.String() + "\tRET\n"
dir := t.TempDir()
path := filepath.Join(dir, "sysregs_arm64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
assertARM64Differential(t, path, src, "sysregs")
}
// TestARM64FamiliesDifferential pins the non-sysreg families the arm64
// campaign added: the LSE compare-and-swap pairs, the VMOVI immediate, the
// SIMD narrow/long shift pairs, the VLD2/VLD3/VLD4 and VST2/VST3/VST4
// structure accesses with their post-index and replicate forms, LDPSW, the
// pointer-authentication hint and the DC maintenance operation. Every
// spelling is the toolchain's own, taken from its arm64 testdata, and the
// bytes must agree word for word.
func TestARM64FamiliesDifferential(t *testing.T) {
src := `#include "textflag.h"
TEXT ·families(SB), NOSPLIT, $0
CASPD (R2, R3), (R2), (R8, R9)
CASPW (R6, R7), (R8), (R4, R5)
VMOVI $82, V0.B16
VMOVI $146, V22.B16
VSSHLL $0, V1.B8, V2.H8
VSSHLL $7, V1.B8, V2.H8
VSSHLL2 $0, V1.B16, V2.H8
VSHRN $7, V1.H8, V0.B8
VSHRN2 $31, V1.D2, V0.S4
VLD2 (R29), [V23.H8, V24.H8]
VLD2.P 16(R0), [V18.B8, V19.B8]
VLD2.P (R1)(R2), [V15.S2, V16.S2]
VLD3 (R27), [V11.S4, V12.S4, V13.S4]
VLD3.P 48(RSP), [V11.S4, V12.S4, V13.S4]
VLD4 (R15), [V10.H4, V11.H4, V12.H4, V13.H4]
VLD4.P 32(R24), [V31.B8, V0.B8, V1.B8, V2.B8]
VLD1R (R1), [V9.B8]
VLD1R.P (R0), [V0.B16]
VLD1R.P 2(R1), [V2.H4]
VLD2R (R15), [V15.H4, V16.H4]
VLD2R.P 16(R0), [V0.D2, V1.D2]
VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]
VLD4R.P 16(RSP), [V31.S4, V0.S4, V1.S4, V2.S4]
VST2 [V22.H8, V23.H8], (R23)
VST2.P [V14.H4, V15.H4], 16(R17)
VST2.P [V14.H4, V15.H4], (R3)(R17)
VST3 [V1.D2, V2.D2, V3.D2], (R11)
VST3.P [V18.S4, V19.S4, V20.S4], 48(R25)
VST4 [V22.D2, V23.D2, V24.D2, V25.D2], (R3)
VST4.P [V14.D2, V15.D2, V16.D2, V17.D2], 64(R15)
LDPSW (R0), (R1, R2)
LDPSW 4(R0), (R1, R2)
LDPSW -4(R0), (R1, R2)
PACIASP
DC IVAC, R1
RET
`
dir := t.TempDir()
path := filepath.Join(dir, "families_arm64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
assertARM64Differential(t, path, src, "families")
}
// assertARM64Differential assembles the same source with gasm and with the
// toolchain for arm64 and requires the named function's code bytes to agree.
// The live oracle is a deliberate-run comparison, so -short skips it (the
// push pipeline's mode); the golden bytes of the individual encoders are
// pinned separately in every mode.
func assertARM64Differential(t *testing.T, path, src, fn string) {
t.Helper()
oracle := oracleFuncCode(t, toolAsmObject(t, path, "arm64"))
// The oracle keys its functions by the qualified object name
// (pkg.name); match on the local part.
want := map[string][]byte{}
for name, code := range oracle {
if _, after, ok := strings.Cut(name, "."); ok {
want[after] = code
} else {
want[name] = code
}
}
if want[fn] == nil {
t.Fatalf("the oracle object carries no function %q (has %v)", fn, keysOf(want))
}
f, perrs := parser.Parse(path, src)
if len(perrs) > 0 {
t.Fatalf("parse: %v", perrs[0])
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := trimTrailingZeroWords(img.Code)
wantB := trimTrailingZeroWords(want[fn])
if len(got) != len(wantB) {
t.Fatalf("gasm %d bytes, oracle %d bytes", len(got), len(wantB))
}
for i := range wantB {
if got[i] != wantB[i] {
t.Fatalf("word %d differs: gasm %08x, oracle %08x", i/4,
binary.LittleEndian.Uint32(got[i:i+4]), binary.LittleEndian.Uint32(wantB[i:i+4]))
}
}
}
// trimTrailingZeroWords drops whole zero words off the end of a code span:
// an object pads a function to its alignment, and the raw image does not.
// A difference in the middle survives the trim untouched.
func trimTrailingZeroWords(b []byte) []byte {
for len(b) >= 4 {
last := b[len(b)-4:]
if last[0]|last[1]|last[2]|last[3] != 0 {
break
}
b = b[:len(b)-4]
}
return b
}
+730 -59
View File
File diff suppressed because it is too large Load Diff
+79 -2
View File
@@ -10,8 +10,8 @@ import (
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// firstText parses src and returns its first TEXT function.
@@ -370,6 +370,58 @@ end:
}
}
// TestAssembleNumericPCJumps pins the numeric ±N(PC) branch operands: N
// counts instruction statements, skipping labels, in both directions (the
// runtime's exit loops write JMP -3(PC)), N = 0 parks on the jump itself.
func TestAssembleNumericPCJumps(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·exit(SB), NOSPLIT, $0
MOVB $1, AL
lab:
MOVB $2, AL
MOVB $3, AL
JMP -3(PC)
MOVB $4, AL
park:
JMP 0(PC)
MOVB $5, AL
JMP 2(PC)
MOVB $6, AL
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
// From the Go-assembled function:
// MOVB $1, AL b001
// MOVB $2, AL b002
// MOVB $3, AL b003
// JMP -3(PC) ebf8 (three instructions back, past lab:)
// MOVB $4, AL b004
// JMP 0(PC) ebfe (the park loop)
// MOVB $5, AL b005
// JMP 2(PC) eb02 (over MOVB $6 to the RET)
// MOVB $6, AL b006
// RET c3
want := []byte{
0xb0, 0x01,
0xb0, 0x02,
0xb0, 0x03,
0xeb, 0xf8,
0xb0, 0x04,
0xeb, 0xfe,
0xb0, 0x05,
0xeb, 0x02,
0xb0, 0x06,
0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("numeric-PC mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
func TestAssemblePrefetch(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
@@ -568,3 +620,28 @@ TEXT ·framed(SB), $16-8
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
// TestAssembleRegRange pins the bracketed register range at the statement
// level: exactly four consecutive same-width vector registers assemble, the
// toolchain's rejected shapes all report an error.
func TestAssembleRegRange(t *testing.T) {
asm := func(t *testing.T, op string) ([]byte, error) {
t.Helper()
f, errs := parser.Parse("f_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tV4FMADDPS 17(SP), "+op+", K2, Z0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse %s: %v", op, errs)
}
code, _, err := Assemble(f.Decls[0].(*ast.Text))
return code, err
}
for _, op := range []string{"[Z0-Z3]", "[Z4-Z7]", "[Z28-Z31]"} {
if _, err := asm(t, op); err != nil {
t.Errorf("%s: %v", op, err)
}
}
for _, op := range []string{"[Z0-Z4]", "[Z0-Z2]", "[Z0-Z0]", "[Z4-Z0]", "[Z1-Z0]", "[AX-Z3]", "[Z0-AX]"} {
if _, err := asm(t, op); err == nil {
t.Errorf("%s: assembled, want an error", op)
}
}
}
+1 -1
View File
@@ -8,7 +8,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// ulebIter reads ULEB128 values, the .debug_abbrev and line-header
+2 -2
View File
@@ -12,8 +12,8 @@ import (
"path/filepath"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// The object-file tests share one source: two exported functions, one
+68 -11
View File
@@ -18,12 +18,16 @@ const (
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset)
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset)
// R_AARCH64_ABS32 (debug/elf 258): the absolute 32-bit address of a
// symbol, the R_ADDR shape a 4-byte DATA field carries. ABS64 (257)
// lives with the DWARF fixup constants as rAARCH64Abs64.
rArm64Abs32 = 258
)
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
// optional .rela.text.
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
// optional .rela.text and an optional .rela.data.
func (img *Image) ELFAARCH64Object() ([]byte, error) {
le := binary.LittleEndian
@@ -133,6 +137,50 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
}
}
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
// $other(SB)") become .rela.data entries: an absolute relocation of the
// DATA line's width at the field's data-section offset, S + A with no
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
// cannot hold an address, so they are refused rather than truncated.
var dataRelas []elfRela
for _, d := range img.DataSyms {
for _, r := range d.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
}
var typ uint32
switch r.Siz {
case 8:
typ = rAARCH64Abs64
case 4:
typ = rArm64Abs32
default:
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
}
dataRelas = append(dataRelas, elfRela{
off: uint64(d.Offset + r.Off),
sym: idx,
typ: typ,
addend: r.Addend,
})
}
}
// Section presence: .rela.text only when there are code relocations,
// .rela.data only when a DATA line holds a symbol value.
hasRela := len(relas) > 0
hasDataRela := len(dataRelas) > 0
nSections := 6
if hasRela {
nSections++
}
if hasDataRela {
nSections++
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// String tables.
stNames := newElfStrtab()
for _, s := range syms {
@@ -142,18 +190,13 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
if hasDataRela {
stSections.add(".rela.data")
}
for _, n := range dwarfSectionNames {
stSections.add(n)
}
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout.
var out []byte
out = append(out, make([]byte, 64)...)
@@ -188,7 +231,7 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
var relaOff, relaDataOff int
if hasRela {
align(8)
relaOff = len(out)
@@ -200,6 +243,17 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
out = append(out, b[:]...)
}
}
if hasDataRela {
align(8)
relaDataOff = len(out)
for _, r := range dataRelas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
@@ -257,6 +311,9 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
if hasDataRela {
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order.
if dw != nil {
+116 -1
View File
@@ -9,7 +9,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
@@ -197,3 +197,118 @@ TEXT ·add(SB), NOSPLIT, $0-24
t.Error("unexpected .rela.text section when there are no relocations")
}
}
// TestELFAARCH64ObjectDataRelocation checks that a symbol-valued DATA field
// ("DATA s+0(SB)/8, $other(SB)") reaches the AArch64 ELF object as a
// .rela.data entry: an R_AARCH64_ABS64 (ABS32 for a width-4 field) at the
// field's offset within .data, against the named symbol, external targets
// included.
func TestELFAARCH64ObjectDataRelocation(t *testing.T) {
f, errs := parser.Parse("t_arm64.s", `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-0
RET
GLOBL holder(SB), NOPTR, $32
DATA holder+0(SB)/8, $·Keep+5(SB)
DATA holder+8(SB)/8, $holder(SB)
DATA holder+16(SB)/8, $extvar(SB)
DATA holder+24(SB)/4, $Keep(SB)
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.ELFAARCH64Object()
if err != nil {
t.Fatalf("ELFAARCH64Object: %v", err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaData := ef.Section(".rela.data")
if relaData == nil {
t.Fatal("missing .rela.data section")
}
if relaData.Type != elf.SHT_RELA {
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
}
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
}
if ef.Sections[relaData.Info].Name != ".data" {
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
}
relas, err := relaData.Data()
if err != nil {
t.Fatal(err)
}
var got []struct {
off uint64
sym uint32
typ uint32
addend int64
}
for i := 0; i+24 <= len(relas); i += 24 {
got = append(got, struct {
off uint64
sym uint32
typ uint32
addend int64
}{
off: binary.LittleEndian.Uint64(relas[i:]),
// r_info packs the type in the low dword and the symbol index
// in the high dword.
typ: binary.LittleEndian.Uint32(relas[i+8:]),
sym: binary.LittleEndian.Uint32(relas[i+12:]),
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
})
}
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
name := func(idx uint32) string {
if idx >= 1 && int(idx) <= len(syms) {
return syms[idx-1].Name
}
return ""
}
// The offsets are data-section-relative: the field's DATA offset plus
// the symbol's position in .data (the layout aligns each symbol to 16).
base := uint64(0)
for _, d := range img.DataSyms {
if d.Name == "holder" {
base = uint64(d.Offset)
}
}
want := []struct {
off uint64
typ uint32
addend int64
target string
}{
{off: base + 0, typ: uint32(elf.R_AARCH64_ABS64), addend: 5, target: "Keep"},
{off: base + 8, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "holder"},
{off: base + 16, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "extvar"},
{off: base + 24, typ: uint32(elf.R_AARCH64_ABS32), addend: 0, target: "Keep"},
}
if len(got) != len(want) {
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
}
for i, w := range want {
g := got[i]
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
}
if n := name(g.sym); n != w.target {
t.Errorf("entry %d names %q, want %q", i, n, w.target)
}
}
}
+68 -11
View File
@@ -23,12 +23,16 @@ const (
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
rLarchB26 = 66 // R_LARCH_B26 (b/bl, matches the Go linker's mapping)
// R_LARCH_32 (debug/elf 1): the absolute 32-bit address of a symbol,
// the R_ADDR shape a 4-byte DATA field carries. R_LARCH_64 (2) lives
// with the DWARF fixup constants as rLarchAbs64.
rLarchAbs32 = 1
)
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
// optional .rela.text.
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
// optional .rela.text and an optional .rela.data.
func (img *Image) ELFLOONG64Object() ([]byte, error) {
le := binary.LittleEndian
@@ -117,6 +121,50 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
}
}
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
// $other(SB)") become .rela.data entries: an absolute relocation of the
// DATA line's width at the field's data-section offset, S + A with no
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
// cannot hold an address, so they are refused rather than truncated.
var dataRelas []elfRela
for _, d := range img.DataSyms {
for _, r := range d.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
}
var typ uint32
switch r.Siz {
case 8:
typ = rLarchAbs64
case 4:
typ = rLarchAbs32
default:
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
}
dataRelas = append(dataRelas, elfRela{
off: uint64(d.Offset + r.Off),
sym: idx,
typ: typ,
addend: r.Addend,
})
}
}
// Section presence: .rela.text only when there are code relocations,
// .rela.data only when a DATA line holds a symbol value.
hasRela := len(relas) > 0
hasDataRela := len(dataRelas) > 0
nSections := 6
if hasRela {
nSections++
}
if hasDataRela {
nSections++
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// String tables.
stNames := newElfStrtab()
for _, s := range syms {
@@ -126,18 +174,13 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
if hasDataRela {
stSections.add(".rela.data")
}
for _, n := range dwarfSectionNames {
stSections.add(n)
}
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout.
var out []byte
out = append(out, make([]byte, 64)...)
@@ -172,7 +215,7 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
var relaOff, relaDataOff int
if hasRela {
align(8)
relaOff = len(out)
@@ -184,6 +227,17 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
out = append(out, b[:]...)
}
}
if hasDataRela {
align(8)
relaDataOff = len(out)
for _, r := range dataRelas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
@@ -239,6 +293,9 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
if hasDataRela {
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order.
if dw != nil {
+116 -1
View File
@@ -9,7 +9,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
@@ -245,3 +245,118 @@ func TestELFLOONG64BranchRelocation(t *testing.T) {
}
}
}
// TestELFLOONG64ObjectDataRelocation checks that a symbol-valued DATA field
// ("DATA s+0(SB)/8, $other(SB)") reaches the LoongArch ELF object as a
// .rela.data entry: an R_LARCH_64 (R_LARCH_32 for a width-4 field) at the
// field's offset within .data, against the named symbol, external targets
// included.
func TestELFLOONG64ObjectDataRelocation(t *testing.T) {
f, errs := parser.Parse("t_loong64.s", `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-0
RET
GLOBL holder(SB), NOPTR, $32
DATA holder+0(SB)/8, $·Keep+5(SB)
DATA holder+8(SB)/8, $holder(SB)
DATA holder+16(SB)/8, $extvar(SB)
DATA holder+24(SB)/4, $Keep(SB)
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.ELFLOONG64Object()
if err != nil {
t.Fatalf("ELFLOONG64Object: %v", err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaData := ef.Section(".rela.data")
if relaData == nil {
t.Fatal("missing .rela.data section")
}
if relaData.Type != elf.SHT_RELA {
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
}
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
}
if ef.Sections[relaData.Info].Name != ".data" {
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
}
relas, err := relaData.Data()
if err != nil {
t.Fatal(err)
}
var got []struct {
off uint64
sym uint32
typ uint32
addend int64
}
for i := 0; i+24 <= len(relas); i += 24 {
got = append(got, struct {
off uint64
sym uint32
typ uint32
addend int64
}{
off: binary.LittleEndian.Uint64(relas[i:]),
// r_info packs the type in the low dword and the symbol index
// in the high dword.
typ: binary.LittleEndian.Uint32(relas[i+8:]),
sym: binary.LittleEndian.Uint32(relas[i+12:]),
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
})
}
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
name := func(idx uint32) string {
if idx >= 1 && int(idx) <= len(syms) {
return syms[idx-1].Name
}
return ""
}
// The offsets are data-section-relative: the field's DATA offset plus
// the symbol's position in .data (the layout aligns each symbol to 16).
base := uint64(0)
for _, d := range img.DataSyms {
if d.Name == "holder" {
base = uint64(d.Offset)
}
}
want := []struct {
off uint64
typ uint32
addend int64
target string
}{
{off: base + 0, typ: uint32(elf.R_LARCH_64), addend: 5, target: "Keep"},
{off: base + 8, typ: uint32(elf.R_LARCH_64), addend: 0, target: "holder"},
{off: base + 16, typ: uint32(elf.R_LARCH_64), addend: 0, target: "extvar"},
{off: base + 24, typ: uint32(elf.R_LARCH_32), addend: 0, target: "Keep"},
}
if len(got) != len(want) {
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
}
for i, w := range want {
g := got[i]
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
}
if n := name(g.sym); n != w.target {
t.Errorf("entry %d names %q, want %q", i, n, w.target)
}
}
}
+68 -10
View File
@@ -24,11 +24,16 @@ const (
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
// R_RISCV_32 (debug/elf 1): the absolute 32-bit address of a symbol,
// the R_ADDR shape a 4-byte DATA field carries. R_RISCV_64 (2) lives
// with the DWARF fixup constants as rRISCVAbs64.
rRISVCAbs32 = 1
)
// ELFRISCVObject returns the image as an ELF64 relocatable object file for
// RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64
// ELF emission: .text, .data, .symtab, .strtab and optional .rela.text.
// ELF emission: .text, .data, .symtab, .strtab, an optional .rela.text and
// an optional .rela.data.
func (img *Image) ELFRISCVObject() ([]byte, error) {
le := binary.LittleEndian
@@ -129,6 +134,50 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
}
}
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
// $other(SB)") become .rela.data entries: an absolute relocation of the
// DATA line's width at the field's data-section offset, S + A with no
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
// cannot hold an address, so they are refused rather than truncated.
var dataRelas []elfRela
for _, d := range img.DataSyms {
for _, r := range d.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
}
var typ uint32
switch r.Siz {
case 8:
typ = rRISCVAbs64
case 4:
typ = rRISVCAbs32
default:
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
}
dataRelas = append(dataRelas, elfRela{
off: uint64(d.Offset + r.Off),
sym: idx,
typ: typ,
addend: r.Addend,
})
}
}
// Section presence: .rela.text only when there are code relocations,
// .rela.data only when a DATA line holds a symbol value.
hasRela := len(relas) > 0
hasDataRela := len(dataRelas) > 0
nSections := 6
if hasRela {
nSections++
}
if hasDataRela {
nSections++
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// String tables.
stNames := newElfStrtab()
for _, s := range syms {
@@ -138,18 +187,13 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
if hasDataRela {
stSections.add(".rela.data")
}
for _, n := range dwarfSectionNames {
stSections.add(n)
}
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout.
var out []byte
out = append(out, make([]byte, 64)...)
@@ -184,7 +228,7 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
var relaOff, relaDataOff int
if hasRela {
align(8)
relaOff = len(out)
@@ -196,6 +240,17 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
out = append(out, b[:]...)
}
}
if hasDataRela {
align(8)
relaDataOff = len(out)
for _, r := range dataRelas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
@@ -251,6 +306,9 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
if hasDataRela {
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order.
if dw != nil {
+128
View File
@@ -0,0 +1,128 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestELFRISCVObjectDataRelocation checks that a symbol-valued DATA field
// ("DATA s+0(SB)/8, $other(SB)") reaches the RISC-V ELF object as a
// .rela.data entry: an R_RISCV_64 (R_RISCV_32 for a width-4 field) at the
// field's offset within .data, against the named symbol, external targets
// included.
func TestELFRISCVObjectDataRelocation(t *testing.T) {
f, errs := parser.Parse("t_riscv64.s", `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-0
RET
GLOBL holder(SB), NOPTR, $32
DATA holder+0(SB)/8, $·Keep+5(SB)
DATA holder+8(SB)/8, $holder(SB)
DATA holder+16(SB)/8, $extvar(SB)
DATA holder+24(SB)/4, $Keep(SB)
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
obj, err := img.ELFRISCVObject()
if err != nil {
t.Fatalf("ELFRISCVObject: %v", err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaData := ef.Section(".rela.data")
if relaData == nil {
t.Fatal("missing .rela.data section")
}
if relaData.Type != elf.SHT_RELA {
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
}
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
}
if ef.Sections[relaData.Info].Name != ".data" {
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
}
relas, err := relaData.Data()
if err != nil {
t.Fatal(err)
}
var got []struct {
off uint64
sym uint32
typ uint32
addend int64
}
for i := 0; i+24 <= len(relas); i += 24 {
got = append(got, struct {
off uint64
sym uint32
typ uint32
addend int64
}{
off: binary.LittleEndian.Uint64(relas[i:]),
// r_info packs the type in the low dword and the symbol index
// in the high dword.
typ: binary.LittleEndian.Uint32(relas[i+8:]),
sym: binary.LittleEndian.Uint32(relas[i+12:]),
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
})
}
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
name := func(idx uint32) string {
if idx >= 1 && int(idx) <= len(syms) {
return syms[idx-1].Name
}
return ""
}
// The offsets are data-section-relative: the field's DATA offset plus
// the symbol's position in .data (the layout aligns each symbol to 16).
base := uint64(0)
for _, d := range img.DataSyms {
if d.Name == "holder" {
base = uint64(d.Offset)
}
}
want := []struct {
off uint64
typ uint32
addend int64
target string
}{
{off: base + 0, typ: uint32(elf.R_RISCV_64), addend: 5, target: "Keep"},
{off: base + 8, typ: uint32(elf.R_RISCV_64), addend: 0, target: "holder"},
{off: base + 16, typ: uint32(elf.R_RISCV_64), addend: 0, target: "extvar"},
{off: base + 24, typ: uint32(elf.R_RISCV_32), addend: 0, target: "Keep"},
}
if len(got) != len(want) {
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
}
for i, w := range want {
g := got[i]
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
}
if n := name(g.sym); n != w.target {
t.Errorf("entry %d names %q, want %q", i, n, w.target)
}
}
}
+16 -3
View File
@@ -20,9 +20,22 @@ func Encodable(mnemonic string) bool {
switch upper {
case "RET", "NOP", "CALL", "JMP",
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
// The literal-data pseudo-ops, the accepted-and-ignored END and the
// SP adjust.
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP":
// The SSE compare family sharing CMPSD's predicate-last shape, the
// far return with its stack pop, the loop family, the bank-crossing
// MMX moves and the one-operand system controls.
"CMPSS", "CMPPS", "CMPPD", "RETFL",
"LOOP", "LOOPE", "LOOPNE",
"MOVDQ2Q", "MOVQ2DQ",
"ENDBR64", "CLWB", "TPAUSE", "UMONITOR", "UMWAIT", "RDPID", "CLDEMOTE",
// The literal-data pseudo-ops, the accepted-and-ignored END and
// bookkeeping statements, and the SP adjust.
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA":
return true
}
if _, ok := sysUnaryTable[upper]; ok {
return true
}
if _, ok := sseStoreOnly[upper]; ok {
return true
}
if _, ok := noOperandTable[upper]; ok {
+304 -5
View File
@@ -5,6 +5,8 @@ package asm
import (
"fmt"
"math"
"strconv"
"strings"
)
@@ -21,6 +23,39 @@ func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
type enc struct {
out []byte
patches []encPatch // disp32 fields awaiting static-symbol resolution
// FloatPool collects the pooled constants the floating-point
// immediates reference, in first-use order.
floatPool []floatPoolEntry
floatPoolSeen map[string]bool
}
// floatPoolEntry is one pooled floating-point constant: the symbol name
// the emitted RIP-relative load refers to and its IEEE-754 bytes.
type floatPoolEntry struct {
name string
data []byte
}
// addFloatPool records a pooled constant, deduplicated by symbol name.
func (e *enc) addFloatPool(name string, bits uint64, width int) {
if e.floatPoolSeen == nil {
e.floatPoolSeen = map[string]bool{}
}
if e.floatPoolSeen[name] {
return
}
e.floatPoolSeen[name] = true
data := make([]byte, width)
for i := range width {
data[i] = byte(bits >> (8 * i))
}
e.floatPool = append(e.floatPool, floatPoolEntry{name: name, data: data})
}
// floatPoolList returns the pooled constants in first-use order.
func (e *enc) floatPoolList() []floatPoolEntry {
return e.floatPool
}
// encPatch marks a 4-byte displacement field in enc.out that must receive the
@@ -29,6 +64,7 @@ type encPatch struct {
off int
name string
addend int64
tls bool // a TLS slot offset: the patch is R_TLSLE with no symbol
}
func (e *enc) encode(mnem string, ops []Operand) error {
@@ -37,9 +73,23 @@ func (e *enc) encode(mnem string, ops []Operand) error {
// Fixed-name instructions (no size suffix).
switch {
case upper == "RET":
// RET sym(SB), the absolute return: the toolchain encodes it as a
// tail jump, E9 rel32 with a call relocation against the symbol.
if len(ops) == 1 {
if m, ok := ops[0].(sbMem); ok {
return e.emit(&instr{opcode: []byte{0xE9}, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}})
}
return fmt.Errorf("RET: unsupported operand")
}
if len(ops) != 0 {
return fmt.Errorf("RET expects no operands, got %d", len(ops))
}
return e.encodeRet()
case upper == "NOP":
return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1})
// The toolchain consumes every NOP statement as a pseudo and emits
// nothing for it, operands included (a bare NOP, NOP AX and
// NOP sym(SB) all vanish from the object).
return nil
case upper == "CALL" || upper == "JMP":
// Through a register or memory: FF /2 (CALL) or FF /4 (JMP).
// Anything else is a rel32 against a label resolved by the assembler.
@@ -66,6 +116,15 @@ func (e *enc) encode(mnem string, ops []Operand) error {
}
return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
}
// One-operand system instructions whose reg field is a fixed digit:
// the cache and wait controls under 0F AE/0F 1C and the RDPID read.
if m, ok := sysUnaryTable[upper]; ok {
return e.encodeSysUnary(upper, m, ops)
}
// The store-only SSE moves (the non-temporal store).
if m, ok := sseStoreOnly[upper]; ok {
return e.encodeSSEStoreOnly(upper, m, ops)
}
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
// are rejected by go tool asm in 64-bit mode, so they stay unsupported.
switch upper {
@@ -81,14 +140,63 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
case "INT":
return e.encodeInt(ops)
// The LOOP family outside the assembler's label settlement: the operand
// is the already-computed rel8 (E0-E2).
case "LOOP", "LOOPE", "LOOPNE":
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 operand, got %d", upper, len(ops))
}
imm, ok := ops[0].(Imm)
if !ok || !fits8(int64(imm)) {
return fmt.Errorf("%s: relative offset must be a signed byte", upper)
}
return e.emit(&instr{opcode: []byte{loopOpcode(upper)}, modrm: -1, sib: -1, imm: []byte{byte(int8(imm))}})
case "LDMXCSR":
return e.encodeMxcsr(2, ops)
case "STMXCSR":
return e.encodeMxcsr(3, ops)
// CMPSD is the scalar double compare, whose predicate immediate comes
// LAST in Plan 9 order (src, dst, $imm).
// LAST in Plan 9 order (src, dst, $imm); the family shares the shape.
case "CMPSD":
return e.encodeCmpsd(ops)
return e.encodeSSECmp("CMPSD", 0xF2, ops)
case "CMPSS":
return e.encodeSSECmp("CMPSS", 0xF3, ops)
case "CMPPS":
return e.encodeSSECmp("CMPPS", 0x00, ops)
case "CMPPD":
return e.encodeSSECmp("CMPPD", 0x66, ops)
// RETFL pops the immediate's worth of bytes after the far return
// (LRET iw: CA imm16), the toolchain's RETF spelling with a stack
// adjustment.
case "RETFL":
if len(ops) != 1 {
return fmt.Errorf("RETFL expects 1 operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("RETFL expects an immediate")
}
return e.emit(&instr{opcode: []byte{0xCA}, modrm: -1, sib: -1, imm: le16(int64(imm))})
// MOVDQ2Q/MOVQ2DQ cross the MMX and XMM banks (F2 0F D6), the register
// in the reg field, the other bank's in r/m.
case "MOVDQ2Q", "MOVQ2DQ":
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
srcReg, ok1 := ops[0].(Reg)
dstReg, ok2 := ops[1].(Reg)
if !ok1 || !ok2 {
return fmt.Errorf("%s takes register operands alone", upper)
}
if upper == "MOVDQ2Q" && (!srcReg.isVec() || !dstReg.mmx) ||
upper == "MOVQ2DQ" && (!srcReg.mmx || !dstReg.isVec()) {
return fmt.Errorf("%s crosses the XMM and MMX banks in that order", upper)
}
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, srcReg, 8); err != nil {
return err
}
return e.emit(i)
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
case "SHA256RNDS2":
return e.encodeSha256rnds2(ops)
@@ -102,6 +210,11 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeEnd(ops)
case "ADJSP":
return e.encodeAdjsp(ops)
// The runtime's bookkeeping statements carry no text bytes: go tool asm
// records FUNCDATA and PCDATA in the program list only, so the encoded
// body shows nothing, on every architecture.
case "FUNCDATA", "PCDATA":
return e.encodeFuncdata(upper, ops)
}
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
@@ -113,6 +226,7 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return err
}
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
isEvexPrefGather(base) ||
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
return e.encodeVec(base, ops, sfx)
}
@@ -140,11 +254,18 @@ func (e *enc) encode(mnem string, ops []Operand) error {
}
// Legacy SSE packed binaries dispatch on the full name: the packed
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
// which the size split must not eat.
// which the size split must not eat. A floating-point immediate
// rewrites into a pooled-constant read on the scalar members.
if m, ok := sseBinTable[upper]; ok {
if f, isFloat := floatImmOperand(ops); isFloat {
return e.encodeSSEFloatBin(upper, m, f, ops)
}
return e.encodeSSEBin(m, ops)
}
if m, ok := sseBinTable[base]; ok {
if f, isFloat := floatImmOperand(ops); isFloat {
return e.encodeSSEFloatBin(upper, m, f, ops)
}
return e.encodeSSEBin(m, ops)
}
// The imm8-controlled legacy instructions, the lane extracts and inserts
@@ -221,7 +342,12 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeCvtInt(base, ops, size)
case "FMOVD":
return e.encodeFmov(ops)
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
case "MOVSD", "MOVSS":
if f, isFloat := floatImmOperand(ops); isFloat {
return e.encodeSSEFloatMove(upper, f, ops)
}
return e.encodeSSEMove(sseMoveTable[base], ops)
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD":
return e.encodeSSEMove(sseMoveTable[base], ops)
}
return fmt.Errorf("unsupported instruction %q", mnem)
@@ -284,6 +410,33 @@ func (e *enc) encodeData(mnem string, ops []Operand) error {
return nil
}
// encodeFuncdata accepts-and-ignores the runtime bookkeeping statements:
// FUNCDATA $n, sym(SB) and PCDATA $n, $m. go tool asm emits no text bytes
// for either (the entries live in the object's ancillary tables, not the
// function body), and the operand shapes it takes are exactly these: an
// integer count first, then a symbol reference for FUNCDATA and an integer
// value for PCDATA. The other architectures accept-and-ignore the same
// statements; amd64 now matches.
func (e *enc) encodeFuncdata(upper string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
if _, ok := ops[0].(Imm); !ok {
return fmt.Errorf("%s: first operand must be an integer immediate", upper)
}
switch upper {
case "FUNCDATA":
if _, ok := ops[1].(sbMem); !ok {
return fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
}
case "PCDATA":
if _, ok := ops[1].(Imm); !ok {
return fmt.Errorf("PCDATA: second operand must be an integer immediate")
}
}
return nil
}
// encodeEnd accepts-and-ignores END. go tool asm drops the statement
// entirely: the AEND Prog is skipped when the program list is flushed, so
// the statements after an END still belong to the same function and the
@@ -318,6 +471,120 @@ func (e *enc) encodeAdjsp(ops []Operand) error {
return nil
}
// --- floating-point immediates ----------------------------------------------
// sseFloatImm lists the mnemonics whose first operand may be a floating-point
// immediate, the set go tool asm rewrites into a pooled-constant read: the
// scalar moves, the four scalar arithmetic pairs and the scalar compares.
// The packed members and the uniform forms (MAXSD, MINSD, SQRTSD, CMPSD)
// reject the immediate in the toolchain and are absent here on purpose.
var sseFloatImm = map[string]bool{
"MOVSD": true, "MOVSS": true,
"ADDSD": true, "ADDSS": true,
"SUBSD": true, "SUBSS": true,
"MULSD": true, "MULSS": true,
"DIVSD": true, "DIVSS": true,
"COMISD": true, "COMISS": true,
"UCOMISD": true, "UCOMISS": true,
}
// floatImmOperand reports whether the operand list opens with a
// floating-point immediate in the two-operand spelling (imm, dst).
func floatImmOperand(ops []Operand) (FloatImm, bool) {
if len(ops) != 2 {
return FloatImm{}, false
}
f, ok := ops[0].(FloatImm)
return f, ok
}
// floatPoolValue evaluates a floating-point immediate at the width its
// mnemonic encodes and names the pool constant the toolchain synthesises:
// $f64.<16 hex> for the doubles, $f32.<8 hex> for the singles (the float32
// rounding of the parsed value). The name carries the IEEE-754 bits; the
// section holds them little-endian.
func floatPoolValue(mnem string, f FloatImm) (bits uint64, name string, err error) {
v, err := strconv.ParseFloat(f.Text, 64)
if err != nil {
return 0, "", fmt.Errorf("invalid floating-point immediate %q", f.Text)
}
if f.Neg {
v = -v
}
if strings.HasSuffix(mnem, "D") {
bits = math.Float64bits(v)
return bits, fmt.Sprintf("$f64.%016x", bits), nil
}
bits = uint64(math.Float32bits(float32(v)))
return bits, fmt.Sprintf("$f32.%08x", bits), nil
}
// encodeSSEFloatMove encodes MOVSD/MOVSS with a floating-point immediate
// source. A positive zero needs no memory read: the toolchain emits
// XORPS dst, dst. Anything else loads the pooled constant RIP-relative
// ($f64.<hex>(SB) / $f32.<hex>(SB)), the displacement a patch site the
// file-level layout or the linker resolves.
func (e *enc) encodeSSEFloatMove(mnem string, f FloatImm, ops []Operand) error {
if !sseFloatImm[mnem] {
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
}
dst, ok := ops[1].(Reg)
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
bits, name, err := floatPoolValue(mnem, f)
if err != nil {
return err
}
e.addFloatPool(name, bits, mwidth(mnem))
if bits == 0 {
i := &instr{opcode: []byte{0x0F, 0x57}, modrm: -1, sib: -1} // XORPS
if err := setRM(i, dst, dst, 8); err != nil {
return err
}
return e.emit(i)
}
m := sseMoveTable[mnem]
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.load}, modrm: -1, sib: -1}
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
return err
}
return e.emit(i)
}
// encodeSSEFloatBin encodes the scalar arithmetic and compare mnemonics with
// a floating-point immediate source: the constant is read from the pool into
// the instruction's r/m side (reg = destination), the rewrite go tool asm
// performs at the source level.
func (e *enc) encodeSSEFloatBin(mnem string, m sseBin, f FloatImm, ops []Operand) error {
if !sseFloatImm[mnem] {
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
}
dst, ok := ops[1].(Reg)
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
bits, name, err := floatPoolValue(mnem, f)
if err != nil {
return err
}
e.addFloatPool(name, bits, mwidth(mnem))
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
return err
}
return e.emit(i)
}
// mwidth returns the operand width a scalar SSE mnemonic encodes: the double
// spellings end in D, the single spellings in S.
func mwidth(mnem string) int {
if strings.HasSuffix(mnem, "D") {
return 8
}
return 4
}
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
func splitSize(upper string) (base string, size int) {
if upper == "" {
@@ -384,6 +651,7 @@ type instr struct {
disp []byte
imm []byte
sb *sbRef // static-symbol displacement in disp, awaiting resolution
tls bool // the displacement is a TLS slot offset, patched R_TLSLE
}
// sbRef records that an instruction's displacement refers to a static symbol
@@ -426,6 +694,9 @@ func (e *enc) emit(i *instr) error {
if i.sb != nil {
e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend})
}
if i.tls {
e.patches = append(e.patches, encPatch{off: len(e.out), tls: true})
}
e.out = append(e.out, i.disp...)
e.out = append(e.out, i.imm...)
return nil
@@ -480,12 +751,30 @@ func setRMReg(i *instr, regField int, rexR, regForced bool, rm Operand, opSize i
i.disp = le32(0)
i.sb = &sbRef{name: r.name, addend: r.addend}
return nil
case TLSMem:
// off(TLS): the segment-prefixed absolute access, mod=00 with the
// SIB escape's disp32 absolute form. The displacement is the TLS
// slot offset, patched by the linker's TLS relocation.
i.prefix = r.Seg
i.modrm = 0x04 | regField<<3
i.sib = 0x25
i.disp = le32(r.Disp)
i.tls = true
return nil
case SegAbs:
// 0x30(GS): the segment override with the SIB escape's disp32
// absolute form, no relocation.
setSegAbs(i, regField, r)
return nil
default:
return fmt.Errorf("invalid r/m operand %T", rm)
}
}
func setMem(i *instr, regField int, m Mem) error {
if m.Seg != 0 {
i.prefix = m.Seg
}
modrm, sib, disp, xBit, bBit, err := memComponents(regField, m)
if err != nil {
return err
@@ -498,6 +787,16 @@ func setMem(i *instr, regField int, m Mem) error {
return nil
}
// setSegAbs assembles a segment-absolute operand, 0x30(GS): the segment
// override with the mod=00 SIB escape's disp32 absolute form and no
// relocation.
func setSegAbs(i *instr, regField int, m SegAbs) {
i.prefix = m.Seg
i.modrm = 0x04 | regField<<3
i.sib = 0x25
i.disp = le32(m.Disp)
}
// memComponents computes the ModR/M byte (with the given reg field), the SIB
// byte (-1 if none), the displacement bytes, and the high index/base bits, for
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
+312 -1
View File
@@ -9,6 +9,9 @@ import (
"testing"
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// decode encodes an instruction and decodes it back, returning the decoded
@@ -262,7 +265,11 @@ func TestImul(t *testing.T) {
func TestControl(t *testing.T) {
checkSyntax(t, "ret", "RET")
checkSyntax(t, "nop", "NOP")
// NOP contributes nothing on amd64, consumed whole by the toolchain as
// a pseudo; only the bytes pin it (no decodable instruction remains).
if code, err := Encode("NOP"); err != nil || len(code) != 0 {
t.Errorf("NOP: bytes %x (err %v), want empty", code, err)
}
checkOp(t, x86asm.JMP, "JMP", Imm(0))
checkOp(t, x86asm.CALL, "CALL", Imm(0))
checkOp(t, x86asm.JGE, "JGE", Imm(0))
@@ -1064,3 +1071,307 @@ func TestAdjsp(t *testing.T) {
t.Error("ADJSP AX assembled, want an error")
}
}
// TestFloatImmediateGroundTruth pins the floating-point immediate rewrite
// byte for byte against go tool asm: the scalar moves and the scalar
// arithmetic read the constant from a synthesised read-only pool symbol
// ($f64.<hex>, $f32.<hex>) RIP-relative with the displacement left to the
// relocation, and a positive zero on the moves collapses to XORPS dst, dst.
func TestFloatImmediateGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"MOVSD -1.0", "MOVSD", []Operand{FloatImm{Text: "1.0", Neg: true}, vreg(t, "X2")}, "f20f101500000000"},
{"MOVSD 1.5", "MOVSD", []Operand{FloatImm{Text: "1.5"}, vreg(t, "X3")}, "f20f101d00000000"},
{"MOVSS 2.5", "MOVSS", []Operand{FloatImm{Text: "2.5"}, vreg(t, "X4")}, "f30f102500000000"},
{"MOVSS -0.5", "MOVSS", []Operand{FloatImm{Text: "0.5", Neg: true}, vreg(t, "X5")}, "f30f102d00000000"},
{"MOVSS +0.0 is XORPS", "MOVSS", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X10")}, "450f57d2"},
{"MOVSD +0.0 is XORPS", "MOVSD", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X6")}, "0f57f6"},
{"ADDSD 1.0", "ADDSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f580500000000"},
{"ADDSS 0.5", "ADDSS", []Operand{FloatImm{Text: "0.5"}, vreg(t, "X1")}, "f30f580d00000000"},
{"SUBSD 2.0", "SUBSD", []Operand{FloatImm{Text: "2.0"}, vreg(t, "X3")}, "f20f5c1d00000000"},
{"MULSD -2.5", "MULSD", []Operand{FloatImm{Text: "2.5", Neg: true}, vreg(t, "X3")}, "f20f591d00000000"},
{"DIVSD 1.0", "DIVSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f5e0500000000"},
{"COMISD 1.0", "COMISD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "660f2f0500000000"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
// The pool names carry the IEEE-754 bits, the float32 narrowing for the
// single spellings; negative zero keeps its sign bit and never takes the
// XORPS shortcut.
for _, c := range []struct {
mnem string
imm FloatImm
want string
}{
{"MOVSD", FloatImm{Text: "1.0", Neg: true}, "$f64.bff0000000000000"},
{"MOVSD", FloatImm{Text: "0.5"}, "$f64.3fe0000000000000"},
{"MOVSS", FloatImm{Text: "2.5"}, "$f32.40200000"},
{"MOVSS", FloatImm{Text: "0.5", Neg: true}, "$f32.bf000000"},
{"MOVSD", FloatImm{Text: "0.0", Neg: true}, "$f64.8000000000000000"},
} {
_, name, err := floatPoolValue(c.mnem, c.imm)
if err != nil {
t.Errorf("%s %s: %v", c.mnem, c.imm.Text, err)
continue
}
if name != c.want {
t.Errorf("%s $%s: pool name %s, want %s", c.mnem, c.imm.Text, name, c.want)
}
}
// The shapes the toolchain's parser rejects: the packed and uniform
// forms, a non-vector destination, and the integer spellings.
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"MAXSD rejects the immediate", "MAXSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
{"MINSD rejects the immediate", "MINSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
{"SQRTSD rejects the immediate", "SQRTSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
{"integer destination", "MOVSD", []Operand{FloatImm{Text: "1.0"}, AX}},
} {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
// TestBookkeepingGroundTruth pins FUNCDATA and PCDATA as accept-and-ignore:
// go tool asm emits no text bytes for either, on every architecture.
func TestBookkeepingGroundTruth(t *testing.T) {
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"FUNCDATA", "FUNCDATA", []Operand{Imm(3), sbMem{name: "\u00b7f.arginfo0"}}},
{"PCDATA", "PCDATA", []Operand{Imm(1), Imm(-1)}},
} {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if len(code) != 0 {
t.Errorf("%s: emitted %x, want no bytes", c.name, code)
}
}
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"FUNCDATA arity", "FUNCDATA", []Operand{Imm(3)}},
{"FUNCDATA missing the count", "FUNCDATA", []Operand{sbMem{name: "x"}}},
{"FUNCDATA integer value", "FUNCDATA", []Operand{Imm(3), Imm(4)}},
{"PCDATA arity", "PCDATA", []Operand{Imm(1)}},
{"PCDATA register value", "PCDATA", []Operand{Imm(1), AX}},
} {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
// At the statement level the bookkeeping lines sit between real
// instructions and contribute nothing to the body, symbol reference
// included: the FUNCDATA operand never needs file-level resolution.
f, errs := parser.Parse("t_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tNOP\n\tFUNCDATA $3, \u00b7f.arginfo0(SB)\n\tPCDATA $1, $-1\n\tFUNCDATA $0, x<>(SB)\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
want := "c3"
if got := hexCompact(img.Code); got != want {
t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want)
}
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tFUNCDATA $1, X0\n\tRET\n")); err == nil {
t.Error("FUNCDATA $1, X0 assembled, want an error")
}
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tPCDATA $1, X0\n\tRET\n")); err == nil {
t.Error("PCDATA $1, X0 assembled, want an error")
}
// Encodable mirrors Encode for the names this work touched.
for _, mnem := range []string{"FUNCDATA", "PCDATA", "V4FMADDPS", "V4FMADDSS", "V4FNMADDPS", "V4FNMADDSS", "VP4DPWSSD", "VP4DPWSSDS"} {
if !Encodable(mnem) {
t.Errorf("Encodable(%s) = false, want true", mnem)
}
}
}
// mustParse parses src or fails the test.
func mustParse(t *testing.T, src string) *ast.File {
t.Helper()
f, errs := parser.Parse("t_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
return f
}
// TestCorpusTailSystem pins the system and control forms the toolchain's own
// amd64 testdata carries, byte for byte: the one-operand IMUL, the compare
// family, the far return, the loop, the MMX moves, the CR/DR and segment
// register moves, the TLS pseudo-base and the 0F AE/1C/C7 controls.
func TestCorpusTailSystem(t *testing.T) {
regBPT := Reg{idx: 5, size: 8}
X0, X1, X2 := vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")
Y1, Y2, Y7 := vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y7")
X5, X20 := vreg(t, "X5"), vreg(t, "X20")
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"IMUL one-op byte", "IMULB", []Operand{DX}, "f6ea"},
{"IMUL one-op long", "IMULL", []Operand{AX}, "f7e8"},
{"CMPPD", "CMPPD", []Operand{X1, X2, Imm(4)}, "660fc2d104"},
{"CMPSS", "CMPSS", []Operand{X1, X2, Imm(4)}, "f30fc2d104"},
{"CMPPS", "CMPPS", []Operand{X1, X2, Imm(4)}, "0fc2d104"},
{"RETFL", "RETFL", []Operand{Imm(4)}, "ca0400"},
{"LOOP", "LOOP", []Operand{Imm(-2)}, "e2fe"},
{"LOOPE", "LOOPE", []Operand{Imm(-2)}, "e1fe"},
{"LOOPNE", "LOOPNE", []Operand{Imm(-2)}, "e0fe"},
{"PADDD MMX", "PADDD", []Operand{Reg{idx: 2, size: 8, mmx: true}, Reg{idx: 1, size: 8, mmx: true}}, "0ffeca"},
{"MOVDQ2Q", "MOVDQ2Q", []Operand{X1, Reg{idx: 1, size: 8, mmx: true}}, "f20fd6c9"},
{"MOVNTDQ", "MOVNTDQ", []Operand{X1, Ptr(AX, 0, 16)}, "660fe708"},
{"MOVQ mmx load", "MOVQ", []Operand{Ptr(AX, 0, 8), Reg{idx: 0, size: 8, mmx: true}}, "0f6f00"},
{"MOVQ mmx store", "MOVQ", []Operand{Reg{idx: 0, size: 8, mmx: true}, Ptr(SI, 0, 8)}, "0f7f06"},
{"MOVQ CR0 load", "MOVQ", []Operand{Reg{idx: 0, size: 8, ctl: 1}, AX}, "0f20c0"},
{"MOVQ CR4 load", "MOVQ", []Operand{Reg{idx: 4, size: 8, ctl: 1}, DI}, "0f20e7"},
{"MOVQ CR0 store", "MOVQ", []Operand{AX, Reg{idx: 0, size: 8, ctl: 1}}, "0f22c0"},
{"MOVQ DR0 load", "MOVQ", []Operand{Reg{idx: 0, size: 8, ctl: 2}, AX}, "0f21c0"},
{"MOVQ DR7 load", "MOVQ", []Operand{Reg{idx: 7, size: 8, ctl: 2}, SI}, "0f21fe"},
{"PUSHQ FS", "PUSHQ", []Operand{Reg{idx: 4, size: 2, seg: 5}}, "0fa0"},
{"PUSHQ GS", "PUSHQ", []Operand{Reg{idx: 5, size: 2, seg: 6}}, "0fa8"},
{"POPQ FS", "POPQ", []Operand{Reg{idx: 4, size: 2, seg: 5}}, "0fa1"},
{"POPQ GS", "POPQ", []Operand{Reg{idx: 5, size: 2, seg: 6}}, "0fa9"},
{"ENDBR64", "ENDBR64", nil, "f30f1efa"},
{"CLWB", "CLWB", []Operand{Ptr(BX, 0, 8)}, "660fae33"},
{"CLDEMOTE", "CLDEMOTE", []Operand{Ptr(BX, 0, 8)}, "0f1c03"},
{"TPAUSE", "TPAUSE", []Operand{BX}, "660faef3"},
{"UMONITOR", "UMONITOR", []Operand{BX}, "f30faef3"},
{"UMWAIT", "UMWAIT", []Operand{BX}, "f20faef3"},
{"RDPID", "RDPID", []Operand{DX}, "f30fc7fa"},
{"RDPID r11", "RDPID", []Operand{Reg{idx: 11, size: 8}}, "f3410fc7fb"},
{"LEAL wide disp", "LEAL", []Operand{Idx(regBPT, Reg{idx: 10, size: 8}, 1, 0x8f1bbcdc, 8), regBPT}, "428dac15dcbc1b8f"},
{"VPERMPD", "VPERMPD", []Operand{Imm(0xd8), Y7, Y7}, "c4e3fd01ffd8"},
{"VPERMILPD", "VPERMILPD", []Operand{Imm(0xff), X1, X2}, "c4e37905d1ff"},
{"VPERMILPS", "VPERMILPS", []Operand{Imm(0xff), X1, X2}, "c4e37904d1ff"},
{"VROUNDPD", "VROUNDPD", []Operand{Imm(-1), X1, X2}, "c4e37909d1ff"},
{"VROUNDPS", "VROUNDPS", []Operand{Imm(-1), Y1, Y2}, "c4e37d08d1ff"},
{"VAESKEYGENASSIST", "VAESKEYGENASSIST", []Operand{Imm(-1), X1, X2}, "c4e379dfd1ff"},
{"VPCMPESTRI", "VPCMPESTRI", []Operand{Imm(-1), X1, X2}, "c4e37961d1ff"},
{"VPCMPESTRM", "VPCMPESTRM", []Operand{Imm(-1), X1, X2}, "c4e37960d1ff"},
{"VPCMPISTRI", "VPCMPISTRI", []Operand{Imm(-1), X1, X2}, "c4e37963d1ff"},
{"VPCMPISTRM", "VPCMPISTRM", []Operand{Imm(-1), X1, X2}, "c4e37962d1ff"},
{"VEXTRACTPS", "VEXTRACTPS", []Operand{Imm(-1), X1, AX}, "c4e37917c8ff"},
{"VPEXTRW", "VPEXTRW", []Operand{Imm(0xff), X1, AX}, "c4e37915c8ff"},
{"VPBLENDVB", "VPBLENDVB", []Operand{X0, Ptr(BX, 0, 16), X1, X2}, "c4e3714c1300"},
{"VMOVHPD load", "VMOVHPD", []Operand{Ptr(AX, 0, 8), X5, X5}, "c5d11628"},
{"VMOVHPD load disp", "VMOVHPD", []Operand{Ptr(DX, 7, 8), X5, X5}, "c5d1166a07"},
{"VMOVHPD store", "VMOVHPD", []Operand{X5, Ptr(AX, 0, 8)}, "c5f91728"},
{"VMOVLPD load", "VMOVLPD", []Operand{Ptr(AX, 0, 8), X5, X5}, "c5d11228"},
{"VMOVLPD store", "VMOVLPD", []Operand{X5, Ptr(AX, 0, 8)}, "c5f91328"},
{"VMOVQ EVEX gpr load", "VMOVQ", []Operand{Reg{idx: 4, size: 8}, X20}, "62e1fd086ee4"},
{"VMOVQ EVEX mem store", "VMOVQ", []Operand{X20, Ptr(AX, 0, 8)}, "62e1fd087e20"},
{"VMOVQ EVEX mem load", "VMOVQ", []Operand{Ptr(AX, 0, 8), X20}, "62e1fd086e20"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
}
// TestCorpusTailFileForms pins the file-level forms the toolchain's amd64
// testdata carries: the star-marked indirect jumps, the TLS pseudo-base, the
// paired-register shift spelling, the absolute RET and the jump to an
// undefined static symbol (its displacement and the TLS slot offsets are
// relocation sites, zeroed here as the kernel parity suites do).
func TestCorpusTailFileForms(t *testing.T) {
mask32 := func(b []byte, at int) { b[at], b[at+1], b[at+2], b[at+3] = 0, 0, 0, 0 }
cases := []struct {
name string
src string
want string // hex, with X marking a masked 32-bit relocation site
}{
{"star reg jump", "\tJMP *(R12)\n\tRET\n", "41ff2424c3"},
{"star sp jump", "\tJMP *4(SP)\n\tRET\n", "ff642404c3"},
{"star indexed jump", "\tJMP *(R12)(R13*4)\n\tRET\n", "43ff24acc3"},
{"TLS load", "\tMOVQ (TLS), AX\n\tRET\n", "64488b0425XXXXXXXXc3"},
{"TLS load offset", "\tMOVQ 8(TLS), DX\n\tRET\n", "64488b1425XXXXXXXXc3"},
{"colon shift", "\tSHLL CX, R11:AX\n\tRET\n", "410fa5c3c3"},
{"SP indexed local", "\tMOVQ foo(SP)(AX*1), BX\n\tRET\n", "488b1c04c3"},
}
for _, c := range cases {
f, errs := parser.Parse("t_amd64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n"+c.src)
if len(errs) > 0 {
t.Errorf("%s: parse: %v", c.name, errs)
continue
}
img, err := AssembleFile(f)
if err != nil {
t.Errorf("%s: assemble: %v", c.name, err)
continue
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
if at := strings.Index(c.want, "XXXXXXXX"); at >= 0 {
mask32(code, at/2) // the masked relocation site
}
if got := hexCompact(code); got != strings.ReplaceAll(c.want, "X", "0") {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
// JMP to an undefined static symbol and the absolute RET: their rel32
// carries a call relocation against the symbol, masked to zero here.
for _, c := range []struct{ name, src string }{
{"static external jump", "\tJMP bar<>+4(SB)\n\tRET\n"},
{"static external indexed jump", "\tJMP bar<>+4(SB)(R11*4)\n\tRET\n"},
{"absolute ret", "\tRET\n\tRET foo(SB)\n"},
} {
f, errs := parser.Parse("t_amd64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n"+c.src)
if len(errs) > 0 {
t.Errorf("%s: parse: %v", c.name, errs)
continue
}
img, err := AssembleFile(f)
if err != nil {
t.Errorf("%s: assemble: %v", c.name, err)
continue
}
fn := img.Funcs[0]
code := maskCode(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
want := "e900000000c3"
if c.name == "absolute ret" {
want = "c3e900000000"
}
if got := hexCompact(code); got != want {
t.Errorf("%s: bytes %s, want %s", c.name, got, want)
}
}
}
+647 -29
View File
@@ -5,6 +5,7 @@ package asm
import (
"fmt"
"slices"
"strings"
)
@@ -91,7 +92,7 @@ var evexTable = map[string]evexSpec{
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
// the W bit distinguishes it from VPSRAD's E2 form).
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRAQ": {1, 0x72, 1, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
// rm=src, no vvvv).
@@ -138,7 +139,7 @@ var evexTable = map[string]evexSpec{
// EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFPS": {1, 0xC6, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
@@ -202,7 +203,7 @@ var evexTable = map[string]evexSpec{
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLVW": {2, 0x10, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -324,7 +325,7 @@ var evexTable = map[string]evexSpec{
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PS": {1, 0x7A, 0, 3, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F38, half-precision convert (half-width source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
@@ -494,6 +495,331 @@ var evexTable = map[string]evexSpec{
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
// --- the AVX-512 families the avx512enc corpus exercises, read off
// the toolchain opcodetables ---
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VALIGNQ": {3, 0x03, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VANDNPD": {1, 0x55, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VANDPD": {1, 0x54, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VBLENDMPD": {2, 0x65, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VBLENDMPS": {2, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VBROADCASTF32X2": {2, 0x19, 0, 1, -1, vexRM, [3]int{0, 8, 8}},
"VBROADCASTF32X4": {2, 0x1A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTF32X8": {2, 0x1B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
"VBROADCASTF64X2": {2, 0x1A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTF64X4": {2, 0x1B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
"VBROADCASTI32X2": {2, 0x59, 0, 1, -1, vexRM, [3]int{8, 8, 8}},
"VBROADCASTI32X4": {2, 0x5A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTI32X8": {2, 0x5B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
"VBROADCASTI64X2": {2, 0x5A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTI64X4": {2, 0x5B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
"VCOMISD": {1, 0x2F, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
"VCVTSD2SS": {1, 0x5A, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
"VDBPSADBW": {3, 0x42, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VEXP2PD": {2, 0xC8, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
"VEXP2PS": {2, 0xC8, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev, [3]int{16, 32, 64}},
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VMOVNTPD": {1, 0x2B, 1, 1, -1, vexRMRev, [3]int{16, 32, 64}},
"VORPD": {1, 0x56, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMB": {2, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMD": {2, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMQ": {2, 0x64, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMW": {2, 0x66, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBROADCASTMB2Q": {2, 0x2A, 1, 2, -1, vexRM, [3]int{0, 0, 0}},
"VPBROADCASTMW2D": {2, 0x3A, 0, 2, -1, vexRM, [3]int{0, 0, 0}},
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPEQQ": {2, 0x29, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTQ": {2, 0x37, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCOMPRESSB": {2, 0x63, 0, 1, -1, vexRMRev, [3]int{1, 1, 1}},
"VPCOMPRESSW": {2, 0x63, 1, 1, -1, vexRMRev, [3]int{2, 2, 2}},
"VPCONFLICTD": {2, 0xC4, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPCONFLICTQ": {2, 0xC4, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPDPBUSD": {2, 0x50, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPDPBUSDS": {2, 0x51, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPDPWSSD": {2, 0x52, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPDPWSSDS": {2, 0x53, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2PD": {2, 0x77, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2PS": {2, 0x77, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2W": {2, 0x75, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
"VPERMT2B": {2, 0x7D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2PS": {2, 0x7F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2W": {2, 0x7D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPEXPANDB": {2, 0x62, 0, 1, -1, vexRM, [3]int{1, 1, 1}},
"VPEXPANDW": {2, 0x62, 1, 1, -1, vexRM, [3]int{2, 2, 2}},
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm, [3]int{4, 0, 0}},
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm, [3]int{8, 0, 0}},
"VPLZCNTD": {2, 0x44, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPLZCNTQ": {2, 0x44, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPMADD52HUQ": {2, 0xB5, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMADD52LUQ": {2, 0xB4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULDQ": {2, 0x28, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULTISHIFTQB": {2, 0x83, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULUDQ": {1, 0xF4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPOPCNTW": {2, 0x54, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPORD": {1, 0xEB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPROLVD": {2, 0x15, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPROLVQ": {2, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPRORVD": {2, 0x14, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPRORVQ": {2, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDD": {3, 0x71, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHLDQ": {3, 0x71, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHLDVD": {2, 0x71, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDVQ": {2, 0x71, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDVW": {2, 0x70, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDW": {3, 0x70, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHRDD": {3, 0x73, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHRDQ": {3, 0x73, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHRDVD": {2, 0x73, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHRDVQ": {2, 0x73, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHRDVW": {2, 0x72, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHRDW": {3, 0x72, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHUFBITQMB": {2, 0x8F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRAVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.66.0F73 /7, the byte-quad shift left (the count is always an
// immediate; there is no register-count twin).
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0, the plain-prefix (no 66) packed spellings
// whose EVEX form drops the legacy prefix entirely.
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
"VCOMISS": {1, 0x2F, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev, [3]int{16, 32, 64}},
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMB": {2, 0x26, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMD": {2, 0x27, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMQ": {2, 0x27, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMW": {2, 0x26, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMB": {2, 0x26, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMD": {2, 0x27, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMQ": {2, 0x27, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMW": {2, 0x26, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHQDQ": {1, 0x6D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLQDQ": {1, 0x6C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VRCP28PD": {2, 0xCA, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRCP28PS": {2, 0xCA, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRCP28SD": {2, 0xCB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VRCP28SS": {2, 0xCB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VRSQRT28PD": {2, 0xCC, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRSQRT28PS": {2, 0xCC, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRSQRT28SD": {2, 0xCD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VRSQRT28SS": {2, 0xCD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VSQRTPD": {1, 0x51, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VSQRTSD": {1, 0x51, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
"VUCOMISD": {1, 0x2E, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
"VXORPD": {1, 0x57, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.F3/F2.W0, word shuffles with an immediate
// ($imm, src, dst: reg = dst, rm = src, imm8). The F3/F2 prefixes
// split the high/low lane spellings.
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM, [3]int{16, 32, 64}},
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.128.66.0F3A, lane extract to a general-purpose register or
// memory ($imm, xsrc, GPR/mem dst: reg = source, rm = destination).
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtractGPR, [3]int{1, 1, 1}},
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR, [3]int{2, 2, 2}},
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtractGPR, [3]int{4, 4, 4}},
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtractGPR, [3]int{8, 8, 8}},
// EVEX.66.0F3A.W1, the qword permutes with an immediate control
// ($imm, src, dst: reg = dst, rm = src, imm8); the register-count
// forms live in evexRegFormTable.
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.66.0F3A, the packed permute shuffles with an immediate control.
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPERMILPD": {3, 0x05, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.128.0F.W0, high/low half moves. VMOVHPS carries the
// three-operand insert form (rm = m64 source, vvvv = preserved,
// reg = dst) and the two-operand store (reg = source, rm = m64);
// the encoder splits on the operand count. VMOVLHPS is the
// three-operand form alone.
"VMOVHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
"VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
}
// evexQuad describes one quad-register instruction: the opcode under
// EVEX.0F38.W0 with the F2 mandatory prefix, and the width of the vector
// registers the bracketed list and the destination take (512-bit ZMM for
// the packed forms, 128-bit XMM for the scalar ones).
type evexQuad struct {
opcode byte
width int // register width in bytes: 64 (ZMM) or 16 (XMM)
}
// evexQuadTable maps the quad-register instructions (the 4FMAPS and 4VNNIW
// families) to their encoding. The operand shape is fixed: a single memory
// source in r/m, the bracketed register list whose LOW register travels the
// inverted 5-bit V'VVVV field, an optional opmask in aaa and the vector
// destination in reg. The vector length follows the destination (512-bit
// for the ZMM list forms, 128-bit for the scalar ones) while the disp8×N
// multiplier stays 16 for every member, the toolchain's own tuple choice.
var evexQuadTable = map[string]evexQuad{
"V4FMADDPS": {0x9A, 64},
"V4FMADDSS": {0x9B, 16},
"V4FNMADDPS": {0xAA, 64},
"V4FNMADDSS": {0xAB, 16},
"VP4DPWSSD": {0x52, 64},
"VP4DPWSSDS": {0x53, 64},
}
// isEvexQuad reports whether the mnemonic is a quad-register instruction.
func isEvexQuad(upper string) bool {
_, ok := evexQuadTable[upper]
return ok
}
// encodeEvexQuad encodes the quad-register form: OP mem, [Zn-Zn+3], (K), dst.
// The register list is the VVVV-side source: its low register fills the
// inverted V'VVVV bits, which is why an indexed memory source above Z15 (no
// spare EVEX.X bit once V' is taken) is refused. Masking rides the standard
// aaa field, zeroing keeps the usual requires-a-mask rule, and no other
// suffix applies.
func (e *enc) encodeEvexQuad(mnem string, q evexQuad, ops []Operand, sfx evexSuffix) error {
if len(ops) != 3 && len(ops) != 4 {
return fmt.Errorf("%s expects 3 or 4 operands (mem, [Zn-Zn+3], (K), dst), got %d", mnem, len(ops))
}
mem, lst := ops[0], ops[1]
dst := ops[len(ops)-1]
mask := 0
if len(ops) == 4 {
k, ok := ops[2].(Reg)
if !ok || !k.mask {
return fmt.Errorf("%s: third operand must be an opmask register", mnem)
}
if k.idx == 0 {
return fmt.Errorf("k0 is not a usable mask register")
}
mask = k.idx
}
list, ok := lst.(RegList)
if !ok {
return fmt.Errorf("%s: second operand must be a four-register list", mnem)
}
if list.Lo.size != q.width {
return fmt.Errorf("%s: the register list must hold %d-bit vector registers", mnem, q.width*8)
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
if dstReg.size != q.width {
return fmt.Errorf("%s: the destination must be a %d-bit vector register", mnem, q.width*8)
}
if !memOperand(mem) {
return fmt.Errorf("%s: the source must be a memory operand", mnem)
}
// The list owns V'VVVV; a scaled index in the EVEX-only half would fold
// its fifth bit into the same field the list's low register occupies.
if m, ok := mem.(Mem); ok && m.HasIndex && m.Index.idx >= 16 {
return fmt.Errorf("%s: an index register above Z15 has no EVEX bit free", mnem)
}
if sfx.zeroing && mask == 0 {
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnem)
}
spec := evexSpec{mapSel: 2, opcode: q.opcode, w: 0, pp: 3, opdigit: -1, n: [3]int{16, 16, 16}}
// The vector length follows the destination (512-bit for the ZMM forms,
// 128-bit for the scalar ones), exactly as the oracle encodes it.
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, list.Lo.idx, mem, mask, sfx)
}
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
@@ -529,32 +855,42 @@ type evexMoveSpec struct {
n [3]int
vecOK bool // the non-memory operand may be a vector register
xmmOnly bool // wider than XMM registers are rejected
nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS)
gprOK bool // the r/m side may be a general-purpose register (VMOVQ)
}
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false},
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false},
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true},
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true, false},
// EVEX.128.F2.0F.W1, scalar double move: memory operands and the
// three-operand register form (VMOVSD dst, src1, src2).
"VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true, false},
// EVEX.128/256/512.0F.W0, unaligned packed single move.
"VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128.66.0F.W1, the 64-bit GPR/memory ↔ XMM move (VMOVQ RSP, X20
// and friends, the EVEX spelling the high registers demand).
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, [3]int{8, 8, 8}, true, true, false, true},
}
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
@@ -565,8 +901,13 @@ func isEvex(mnemUpper string) bool {
if _, ok := evexBcastTable[mnemUpper]; ok {
return true
}
_, ok := evexMoveTable[mnemUpper]
return ok
if _, ok := evexMoveTable[mnemUpper]; ok {
return true
}
if _, ok := evexHptrTable[mnemUpper]; ok {
return true
}
return isEvexQuad(mnemUpper)
}
// evexRequired reports whether the operands force the EVEX encoding of a
@@ -577,7 +918,20 @@ func evexRequired(upper string, ops []Operand) bool {
_, inVex := vexTable[upper]
_, inVexMove := vexMoveTable[upper]
if !inVex && !inVexMove {
return true // EVEX-only mnemonic
// The dual-shape moves pick their VEX form by operand count, so
// they are not EVEX-only either.
switch upper {
case "VMOVHPD", "VMOVLPD":
default:
return true // EVEX-only mnemonic
}
}
// The byte-quad shifts have VEX register forms but EVEX-only memory
// forms: a memory count source forces the EVEX encoding.
if upper == "VPSLLDQ" || upper == "VPSRLDQ" {
if slices.ContainsFunc(ops, memOperand) {
return true
}
}
for _, op := range ops {
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
@@ -677,6 +1031,8 @@ var evexRound = map[string]bool{
"VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true,
"VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true,
"VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true,
// The scalar compares suppress exceptions on their LIG encoding.
"VCMPSD": true, "VCMPSS": true,
}
// evexBcstN maps an instruction accepting .BCST to the broadcast element
@@ -741,6 +1097,22 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
return e.encodeEvexRM(spec, ops, 0, sfx)
}
spec, inTable := evexTable[mnemUpper]
// A high/low half move that lives in the hptr table alone (the packed
// double twins) reaches the same inTable block below, which completes
// its spec from the hptr entry.
if !inTable {
if _, ok := evexHptrTable[mnemUpper]; ok {
inTable = true
}
}
if q, ok := evexQuadTable[mnemUpper]; ok {
// The quad-register family carries no rounding, SAE or broadcast;
// only masking and zeroing apply.
if sfx.sae || sfx.bcst || sfx.rounding >= 0 {
return fmt.Errorf("%s takes no rounding/SAE/broadcast suffix", mnemUpper)
}
return e.encodeEvexQuad(mnemUpper, q, ops, sfx)
}
if inTable {
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper)
@@ -752,6 +1124,27 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
}
spec.n = [3]int{n, n, n}
}
// A mnemonic with an immediate and a register spelling (the
// variable-count shifts, the permutes) encodes the register one
// when the first operand is not an immediate.
if len(ops) > 0 {
if _, isImm := ops[0].(Imm); !isImm {
if alt, ok := evexRegFormTable[mnemUpper]; ok {
spec, inTable = alt, true
}
}
}
// The high/low half moves split by operand count: three operands
// insert, two store (VMOVHPS m64, X1).
if hs, ok := evexHptrTable[mnemUpper]; ok {
if len(ops) == 2 {
if hs.store.opcode == 0 {
return fmt.Errorf("%s has no two-operand form", mnemUpper)
}
return e.encodeEvexRMRev(hs.store, ops, 0, sfx)
}
spec = hs.insert
}
} else if sfx.evexOnly() {
return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper)
}
@@ -811,6 +1204,12 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
}
return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx)
}
if ps, ok := evexPrefGatherTable[mnemUpper]; ok {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
}
return e.encodeEvexPrefGather(mnemUpper, ps, ops, mask, sfx)
}
if !inTable {
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
}
@@ -829,6 +1228,8 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
return e.encodeEvexNDS3Imm(spec, ops, mask, sfx)
case vexExtract:
return e.encodeEvexExtract(spec, ops, mask, sfx)
case vexExtractGPR:
return e.encodeEvexExtractGPR(spec, ops, mask, sfx)
case vexRMSrcLen:
return e.encodeEvexRMSrcLen(spec, ops, mask, sfx)
}
@@ -908,6 +1309,11 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSu
if dstReg.mask {
if r, ok := src.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
} else if l, err := soleLen(spec.n); err == nil {
// A memory source with a length-fixed mnemonic
// (VFPCLASSPDX/Y/Z): the length comes from the table's
// single valid slot, not from the operand.
ll = l
}
} else if r, ok := src.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
@@ -934,9 +1340,11 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve
if !ok {
return fmt.Errorf("shift count must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("shift source must be a vector register")
// The count source is a vector register or memory; the length the L'L
// field and the disp8×N multiplier follow is the destination's either
// way.
if !vecOrMem(src) {
return fmt.Errorf("shift source must be a vector register or memory")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
@@ -946,7 +1354,7 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve
if err != nil {
return err
}
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, sfx); err != nil {
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, src, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
@@ -1018,10 +1426,90 @@ func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, sfx evex
return nil
}
// encodeEvexExtractGPR encodes the lane extract to a general-purpose
// register or memory: OP $imm, xsrc, dst (reg = the XMM source, rm = the
// destination, imm8). The encoding is 128-bit regardless of register
// numbers, so L'L is fixed at 0 and the disp8×N multiplier is the extracted
// element size the table carries.
func (e *enc) encodeEvexExtractGPR(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("extract source must be a vector register")
}
switch dst.(type) {
case Reg:
if dst.(Reg).isVec() {
return fmt.Errorf("extract destination must be a general-purpose register or memory")
}
case Mem, sbMem:
default:
return fmt.Errorf("extract destination must be a general-purpose register or memory")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, 0, srcReg.idx, -1, dst, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
// the store-form opcode (reg = source, rm = destination), matching the Go
// assembler.
// assembler. The scalar moves also carry a three-operand register form
// (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3
// opens.
// validEvexMoveOther reports whether the non-vector side of an EVEX move may
// take the operand: memory always, a general-purpose register when gprOK.
func validEvexMoveOther(ms evexMoveSpec, op Operand) bool {
if memOperand(op) {
return true
}
if !ms.gprOK {
return false
}
r, ok := op.(Reg)
return ok && !r.isVec() && !r.mask && r.ctl == 0 && !r.mmx && !r.fp
}
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) == 3 {
if !ms.nds3 {
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
}
// The masked scalar register form keeps the Go assembler's own
// layout: the store opcode with reg = op0, vvvv = op1 and the
// destination in r/m (op2), the bytes go tool asm emits, not
// the manual's NDS reading.
src, src1, dst := ops[0], ops[1], ops[2]
reg, ok := src.(Reg)
if !ok || !reg.isVec() {
return fmt.Errorf("%s: first operand must be a vector register", mnem)
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("%s: second operand must be a vector register", mnem)
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
if ms.xmmOnly && (reg.size != 16 || vvvvReg.size != 16 || dstReg.size != 16) {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
spec := evexSpec{mapSel: ms.mapSel, opcode: ms.store, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
return e.emitEvexFields(spec, dstReg.vecLenBit(), reg.idx, vvvvReg.idx, dst, mask, sfx)
}
if len(ops) != 2 {
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
}
@@ -1042,12 +1530,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
}
reg, rm = srcReg, dst
case srcIsVec:
if !memOperand(dst) {
if !validEvexMoveOther(ms, dst) {
return fmt.Errorf("%s: invalid destination operand", mnem)
}
reg, rm = srcReg, dst
case dstIsVec:
if !memOperand(src) {
if !validEvexMoveOther(ms, src) {
return fmt.Errorf("%s: invalid source operand", mnem)
}
op = ms.load
@@ -1141,12 +1629,20 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx eve
return fmt.Errorf("broadcast destination must be a vector register")
}
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
switch src.(type) {
switch r := src.(type) {
case Mem, sbMem:
spec.opcode = bs.opMem
spec.n = [3]int{bs.n, bs.n, bs.n}
case Reg:
spec.opcode = bs.opReg
// A GPR source uses the register broadcast opcode; a vector
// source shares the xmm/mem one (the low byte is copied from
// the lane or from the memory operand).
if r.isVec() {
spec.opcode = bs.opMem
spec.n = [3]int{bs.n, bs.n, bs.n}
} else {
spec.opcode = bs.opReg
}
default:
return fmt.Errorf("broadcast source must be a register or memory")
}
@@ -1349,6 +1845,111 @@ func isScatter(upper string) bool {
return ok
}
// isEvexPrefGather reports whether the mnemonic is a gather/scatter
// prefetch hint.
func isEvexPrefGather(upper string) bool {
_, ok := evexPrefGatherTable[upper]
return ok
}
// evexRegFormTable holds the register-count twin of the immediate-form
// entries in evexTable. Several mnemonics name two encodings: an immediate
// count or control ($imm, src, dst …) and a register-count one whose second
// operand is a vector register or memory (count, src2, src1, dst). The
// immediate spelling lives in evexTable, this table carries the register
// spelling, and encodeEvex picks by whether the first operand is an
// immediate, the way vexVarShift does on the VEX side.
var evexRegFormTable = map[string]evexSpec{
"VPSLLD": {1, 0xF2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSLLQ": {1, 0xF3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSLLW": {1, 0xF1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRAD": {1, 0xE2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRAW": {1, 0xE1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRLD": {1, 0xD2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRLQ": {1, 0xD3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRLW": {1, 0xD1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
// EVEX.NDS.0F38.W1, the register-count permutes (the immediate
// controls live in evexTable under 0F3A).
"VPERMQ": {2, 0x36, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMPD": {2, 0x16, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.NDS.0F38, the register-count permil shuffles.
"VPERMILPS": {2, 0x0C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMILPD": {2, 0x0D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
}
// evexPrefGatherSpec describes a gather/scatter prefetch hint: one memory
// operand with a VSIB index and an opmask register, no destination. The
// ModRM.reg field carries a fixed /digit, the L'L field is fixed at 512, and
// the mask register is the instruction's only register operand.
type evexPrefGatherSpec struct {
mapSel int
opcode byte
w int
pp int
opdigit int
n int
}
var evexPrefGatherTable = map[string]evexPrefGatherSpec{
"VGATHERPF0DPD": {2, 0xC6, 1, 1, 1, 8},
"VGATHERPF0DPS": {2, 0xC6, 0, 1, 1, 4},
"VGATHERPF0QPD": {2, 0xC7, 1, 1, 1, 8},
"VGATHERPF0QPS": {2, 0xC7, 0, 1, 1, 4},
"VGATHERPF1DPD": {2, 0xC6, 1, 1, 2, 8},
"VGATHERPF1DPS": {2, 0xC6, 0, 1, 2, 4},
"VGATHERPF1QPD": {2, 0xC7, 1, 1, 2, 8},
"VGATHERPF1QPS": {2, 0xC7, 0, 1, 2, 4},
"VSCATTERPF0DPD": {2, 0xC6, 1, 1, 5, 8},
"VSCATTERPF0DPS": {2, 0xC6, 0, 1, 5, 4},
"VSCATTERPF0QPD": {2, 0xC7, 1, 1, 5, 8},
"VSCATTERPF0QPS": {2, 0xC7, 0, 1, 5, 4},
"VSCATTERPF1DPD": {2, 0xC6, 1, 1, 6, 8},
"VSCATTERPF1DPS": {2, 0xC6, 0, 1, 6, 4},
"VSCATTERPF1QPD": {2, 0xC7, 1, 1, 6, 8},
"VSCATTERPF1QPS": {2, 0xC7, 0, 1, 6, 4},
}
// evexHptrSpec describes the high/low half moves (VMOVHPS family): the
// three-operand insert shares an opcode with a two-operand store whose
// source is the vector register and whose destination is m64.
type evexHptrSpec struct {
insert evexSpec
store evexSpec // store.opcode == 0 when the mnemonic has no store form
}
var evexHptrTable = map[string]evexHptrSpec{
"VMOVHPS": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x17, w: 0, pp: 0, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
"VMOVLHPS": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
},
// The packed-double twins, 66-prefixed.
"VMOVHPD": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 1, pp: 1, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x17, w: 1, pp: 1, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
"VMOVLPD": {
insert: evexSpec{mapSel: 1, opcode: 0x12, w: 1, pp: 1, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x13, w: 1, pp: 1, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
}
// encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib.
func (e *enc) encodeEvexPrefGather(upper string, ps evexPrefGatherSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 2 operands (K, vsib memory), got %d", upper, len(ops)+1)
}
m, ok := ops[0].(Mem)
if !ok || !m.HasIndex || !m.Index.isVec() {
return fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", upper)
}
spec := evexSpec{mapSel: ps.mapSel, opcode: ps.opcode, w: ps.w, pp: ps.pp, opdigit: ps.opdigit, n: [3]int{ps.n, ps.n, ps.n}}
return e.emitEvexFields(spec, 2, ps.opdigit, -1, m, mask, sfx)
}
// vsibLen validates a VSIB memory operand (the index must be a vector
// register) and returns it with the vector length the index selects, the
// EVEX L'L field follows the index register, not the data register.
@@ -1370,7 +1971,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
return err
}
if mask != 0 || sfx.any() {
// EVEX form: OP vsib, K, dst.
// EVEX form: OP vsib, K, dst. The L'L field is the wider of the
// index and the data register lengths (the Go assembler's
// layout); the disp8×N multiplier stays the index element size.
if len(rest) != 2 {
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
}
@@ -1382,6 +1985,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", upper)
}
if d := dst.vecLenBit(); d > ll {
ll = d
}
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
}
@@ -1393,7 +1999,7 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
if !ok || !maskReg.isVec() {
return fmt.Errorf("%s: mask must be a vector register", upper)
}
vsib, _, err := vsibLen(rest[1], upper)
vsib, idxLen, err := vsibLen(rest[1], upper)
if err != nil {
return err
}
@@ -1401,12 +2007,16 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", upper)
}
// The L bit is the wider of the data register and the VSIB index
// lengths (a YMM index under an XMM destination selects 256-bit, the
// bytes go tool asm emits).
ll := max(idxLen, dst.vecLenBit())
spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1}
rBit := 0
if dst.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib)
return e.emitVexFields(spec, ll, dst.idx&7, rBit, 15-maskReg.idx, vsib)
}
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src,
@@ -1431,6 +2041,11 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
if err != nil {
return err
}
// The L'L field is the wider of the data register and the VSIB index
// lengths, the bytes go tool asm emits.
if d := src.vecLenBit(); d > ll {
ll = d
}
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
}
@@ -1441,6 +2056,8 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
var evexKOperand = map[string]bool{
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
// The K-to-vector broadcast reads its opmask source from r/m.
"VPBROADCASTMB2Q": true, "VPBROADCASTMW2D": true,
}
// kmovSpec describes a KMOV width: the opcode depends on the operand
@@ -1541,6 +2158,7 @@ var kOpsTable = map[string]kOpSpec{
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
"KUNPCKWD": {1, 0x4B, 0, 0, 1, vexNDS3},
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
+213
View File
@@ -721,3 +721,216 @@ func hexCompact(b []byte) string {
}
return string(out)
}
// TestAvx512CorpusFamilies pins representative encodings of the AVX-512
// families the toolchain's avx512enc corpus exercises: the bytes are the
// go tool asm output for exactly these operands, and the same families are
// covered end to end by the avx512_amd64.s differential kernel.
func TestAvx512CorpusFamilies(t *testing.T) {
vsib := func(base, idx string, scale int) Operand {
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
}
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// AES rounds (EVEX NDS, VEX twin routed by operand width).
{"VAESDEC Z", "VAESDEC", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d48ded9"},
// Integer VNNI and the bit algorithm group.
{"VPDPBUSD", "VPDPBUSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f26d4a50d9"},
{"VPOPCNTW", "VPOPCNTW", []Operand{vreg(t, "Z1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f2fd4b54d1"},
{"VPCONFLICTD", "VPCONFLICTD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d49c4d1"},
{"VPLZCNTQ masked", "VPLZCNTQ", []Operand{vreg(t, "Z7"), vreg(t, "K1"), vreg(t, "Z8")}, "6272fd4944c7"},
{"VPERMT2B", "VPERMT2B", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f26d497dd9"},
{"VPMULTISHIFTQB", "VPMULTISHIFTQB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f2ed4b83e1"},
{"VDBPSADBW", "VDBPSADBW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z3")}, "62f36d4b42d903"},
{"VPSHUFBITQMB", "VPSHUFBITQMB", []Operand{vreg(t, "Z9"), vreg(t, "Z10"), vreg(t, "K3")}, "62d22d488fd9"},
{"VPTESTNMQ", "VPTESTNMQ", []Operand{vreg(t, "Z13"), vreg(t, "Z14"), vreg(t, "K5")}, "62d28e4827ed"},
// Permutations: immediate and register counts.
{"VALIGNQ", "VALIGNQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f3ed4903d903"},
{"VPERMQ imm", "VPERMQ", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f3fd4a00d101"},
{"VPERMQ reg", "VPERMQ", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K2"), vreg(t, "Z5")}, "62f2dd4a36eb"},
{"VPERMPD reg", "VPERMPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4816d9"},
{"VPERMILPS imm", "VPERMILPS", []Operand{Imm(5), vreg(t, "Z9"), vreg(t, "K2"), vreg(t, "Z10")}, "62537d4a04d105"},
{"VPERMILPS reg", "VPERMILPS", []Operand{vreg(t, "Z11"), vreg(t, "Z12"), vreg(t, "K2"), vreg(t, "Z13")}, "62521d4a0ceb"},
// Shifts: immediate, register-count and memory-count forms; the
// count source carries its own XMM tuple width.
{"VPSLLW imm mask", "VPSLLW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f16d4a71f103"},
{"VPSLLD reg count", "VPSLLD", []Operand{vreg(t, "X1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49f2d9"},
{"VPSLLDQ", "VPSLLDQ", []Operand{Imm(9), vreg(t, "Z7"), vreg(t, "Z8")}, "62f13d4873ff09"},
{"VPSRLDQ mem", "VPSRLDQ", []Operand{Imm(11), Ptr(SI, 16, 16), vreg(t, "Z4")}, "62f15d48739e100000000b"},
{"VPSRLVW", "VPSRLVW", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K1"), vreg(t, "Z5")}, "62f2dd4910eb"},
// Conversions and shuffles with the F2 prefix and no prefix.
{"VCVTUDQ2PS", "VCVTUDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f17f497ad1"},
{"VSHUFPS", "VSHUFPS", []Operand{Imm(2), vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f15449c6f402"},
// Gather and scatter prefetch hints (memory-only, /digit in reg).
{"VGATHERPF0DPD", "VGATHERPF0DPD", []Operand{vreg(t, "K5"), vsib("R10", "Y29", 8)}, "6292fd45c60cea"},
{"VSCATTERPF1DPS", "VSCATTERPF1DPS", []Operand{vreg(t, "K2"), vsib("R10", "Z28", 4)}, "62927d42c634a2"},
// Opmask broadcasts and the K logic.
{"VPBROADCASTMB2Q", "VPBROADCASTMB2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe482ad1"},
{"VPBROADCASTMW2D", "VPBROADCASTMW2D", []Operand{vreg(t, "K3"), vreg(t, "Z4")}, "62f27e483ae3"},
{"KUNPCKWD", "KUNPCKWD", []Operand{vreg(t, "K6"), vreg(t, "K4"), vreg(t, "K1")}, "c5dc4bce"},
{"KADDB", "KADDB", []Operand{vreg(t, "K2"), vreg(t, "K3"), vreg(t, "K5")}, "c5e54aea"},
// Lane extracts to general registers (EVEX and VEX routes).
{"VPEXTRB", "VPEXTRB", []Operand{Imm(3), vreg(t, "X26"), AX}, "62637d0814d003"},
{"VPEXTRD", "VPEXTRD", []Operand{Imm(1), vreg(t, "X26"), vreg(t, "R9")}, "62437d0816d101"},
{"VPEXTRD vex", "VPEXTRD", []Operand{Imm(1), vreg(t, "X2"), DI}, "c4e37916d701"},
{"VPINSRQ", "VPINSRQ", []Operand{Imm(1), DI, vreg(t, "X3"), vreg(t, "X4")}, "c4e3e122e701"},
// Moves: masked unaligned, masked scalar register form, half moves
// and non-temporal stores.
{"VMOVUPS mask", "VMOVUPS", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f17c4a11cb"},
{"VMOVSD 3op", "VMOVSD", []Operand{vreg(t, "X14"), vreg(t, "X5"), vreg(t, "K3"), vreg(t, "X22")}, "6231d70b11f6"},
{"VMOVSS 3op", "VMOVSS", []Operand{vreg(t, "X18"), vreg(t, "X3"), vreg(t, "K2"), vreg(t, "X25")}, "6281660a11d1"},
{"VMOVHPS insert", "VMOVHPS", []Operand{Ptr(SI, 0, 8), vreg(t, "X18"), vreg(t, "X19")}, "62e16c00161e"},
{"VMOVHPS store", "VMOVHPS", []Operand{vreg(t, "X20"), Ptr(SI, 8, 8)}, "62e17c08176601"},
{"VMOVLHPS", "VMOVLHPS", []Operand{vreg(t, "X16"), vreg(t, "X5"), vreg(t, "X17")}, "62a1540816c8"},
{"VMOVNTDQ", "VMOVNTDQ", []Operand{vreg(t, "Z7"), Ptr(SI, 0, 64)}, "62f17d48e73e"},
{"VMOVNTDQA", "VMOVNTDQA", []Operand{Ptr(SI, 64, 64), vreg(t, "Z8")}, "62727d482a4601"},
{"VMOVNTPS", "VMOVNTPS", []Operand{vreg(t, "Z9"), Ptr(SI, 0, 64)}, "62717c482b0e"},
// Scalar compares with and without the 66 prefix.
{"VCOMISD", "VCOMISD", []Operand{vreg(t, "X5"), vreg(t, "X6")}, "c5f92ff5"},
{"VUCOMISS", "VUCOMISS", []Operand{vreg(t, "X7"), vreg(t, "X8")}, "c5782ec7"},
// Floating point helpers.
{"VSQRTSD", "VSQRTSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f1ef0951d9"},
{"VEXP2PD", "VEXP2PD", []Operand{vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f2fd49c8f5"},
{"VRCP28SD", "VRCP28SD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "K1"), vreg(t, "X10")}, "6252bd09cbd1"},
{"VBROADCASTF32X2", "VBROADCASTF32X2", []Operand{vreg(t, "X1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d4919d1"},
{"VPCOMPRESSB", "VPCOMPRESSB", []Operand{vreg(t, "Z1"), vreg(t, "K1"), Ptr(SI, 0, 64)}, "62f27d49630e"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
}
// TestEvexQuadRegisterGroundTruth pins the quad-register instructions (the
// 4FMAPS and 4VNNIW families) byte for byte against go tool asm: the memory
// source keeps r/m, the bracketed list's LOW register travels the inverted
// 5-bit V'VVVV field, the destination sits in reg, the opmask rides aaa and
// the vector length follows the destination (L'L=512 for the ZMM forms,
// 128 for the scalar ones) while the disp8×N multiplier stays 16 for every
// member. The x86 decoder has no view of these forms, so no decode check
// runs.
func TestEvexQuadRegisterGroundTruth(t *testing.T) {
sp := vreg(t, "RSP")
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"V4FMADDPS 17(SP) [Z0-Z3] K2 Z0", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f27f4a9a842411000000"},
{"V4FMADDPS [Z10-Z13]", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z10"), vreg(t, "Z13")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f22f4a9a842411000000"},
{"V4FMADDPS [Z20-Z23]", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z20"), vreg(t, "Z23")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f25f429a842411000000"},
{"V4FMADDPS Z8 dst", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z8")},
"62727f4a9a842411000000"},
{"V4FMADDPS disp8x16", "V4FMADDPS",
[]Operand{Ptr(sp, 64, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f27f4a9a442404"},
{"V4FMADDPS unmasked", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
"62f27f489a842411000000"},
{"V4FMADDSS 7(AX) [X0-X3] K5 X22", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
"62e27f0d9bb007000000"},
{"V4FMADDSS (DI)", "V4FMADDSS",
[]Operand{Ptr(DI, 0, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
"62e27f0d9b37"},
{"V4FMADDSS [X10-X13]", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X10"), vreg(t, "X13")}, vreg(t, "K5"), vreg(t, "X22")},
"62e22f0d9bb007000000"},
{"V4FMADDSS [X20-X23]", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X22")},
"62e25f059bb007000000"},
{"V4FMADDSS X30 dst", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X30")},
"62627f0d9bb007000000"},
{"V4FMADDSS X3 dst", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X3")},
"62f27f0d9b9807000000"},
{"V4FMADDSS disp8x16", "V4FMADDSS",
[]Operand{Ptr(AX, 16, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X30")},
"62625f059b7001"},
{"V4FNMADDPS", "V4FNMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f27f4aaa842411000000"},
{"V4FNMADDSS", "V4FNMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
"62e27f0dabb007000000"},
{"VP4DPWSSD", "VP4DPWSSD",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f27f4a52842411000000"},
{"VP4DPWSSDS unmasked", "VP4DPWSSDS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
"62f27f4853842411000000"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
}
// TestEvexQuadRegisterErrors pins the operand shapes the toolchain rejects:
// the register class the list and the destination take is fixed per
// instruction, the source is memory only, the opmask slot is positional and
// the list's low register owns V'VVVV.
func TestEvexQuadRegisterErrors(t *testing.T) {
sp := vreg(t, "RSP")
list := func(lo, hi string) RegList {
return RegList{vreg(t, lo), vreg(t, hi)}
}
cases := []struct {
name string
mnem string
ops []Operand
}{
{"X list on the PS form", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("X0", "X3"), vreg(t, "K2"), vreg(t, "Z0")}},
{"Z list on the SS form", "V4FMADDSS",
[]Operand{Ptr(AX, 0, 8), list("Z0", "Z3"), vreg(t, "K5"), vreg(t, "X22")}},
{"Y destination", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Y0")}},
{"register source", "V4FMADDPS",
[]Operand{vreg(t, "Z1"), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
{"non-mask third operand", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z4"), vreg(t, "Z0")}},
{"k0 mask", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K0"), vreg(t, "Z0")}},
{"K after the destination", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0"), vreg(t, "K2")}},
{"zeroing without a mask", "V4FMADDPS.Z",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0")}},
{"SAE suffix", "V4FMADDPS.SAE",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
{"high index source", "VP4DPWSSD",
[]Operand{Idx(DI, vreg(t, "X16"), 1, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
{"short operand list", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3")}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
+159
View File
@@ -0,0 +1,159 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// The extended-instruction registry: the lookup over and above the generated
// architecture tables. The generated tables (arch/*_gen.go) list the
// mnemonics the Go toolchain knows; the extension layer carries the
// instructions it does not, and this file indexes them per architecture so
// the assembler and the linter can consult the layer without touching the
// generated lists or the main encoders. A later hook wires
// ExtensionEncodable into the Encodable mirror and EncodeExtension into the
// per-architecture assembly paths; nothing existing changes until then.
package asm
import (
"fmt"
"slices"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
)
// extensionIndex is the per-architecture index of the extension layer, keyed
// by upper-case mnemonic. One mnemonic registers several forms (the SVE ADD
// carries unpredicated, predicated and immediate shapes), so the value is the
// full candidate list in table order.
type extensionIndex struct {
byName map[string][]arch.ExtInstr
}
// extensionIndexes builds one index per known architecture. Architectures
// whose extension layer is not built yet get an empty index, which keeps the
// queries answering false rather than failing on a missing entry.
var extensionIndexes = buildExtensionIndexes()
func buildExtensionIndexes() map[arch.Arch]*extensionIndex {
m := make(map[arch.Arch]*extensionIndex)
for _, a := range []arch.Arch{arch.AMD64, arch.ARM64, arch.RISCV, arch.LOONG64} {
idx := &extensionIndex{byName: make(map[string][]arch.ExtInstr)}
for _, in := range arch.Extensions(a) {
key := strings.ToUpper(in.Name)
idx.byName[key] = append(idx.byName[key], in)
}
m[a] = idx
}
return m
}
// LookupExtension returns the extended instructions registered for the
// mnemonic on a, outside the generated architecture table. It reports false
// when a carries no extended layer or the mnemonic is not in it; a mnemonic
// the base table knows is not thereby covered, the layers stay independent.
func LookupExtension(a arch.Arch, mnemonic string) ([]arch.ExtInstr, bool) {
idx, ok := extensionIndexes[a]
if !ok || idx == nil {
return nil, false
}
cands, ok := idx.byName[strings.ToUpper(mnemonic)]
return cands, ok && len(cands) > 0
}
// ExtensionNames returns the mnemonics the extension layer of a registers,
// in table order, without duplicates.
func ExtensionNames(a arch.Arch) []string {
var names []string
seen := make(map[string]bool)
for _, in := range arch.Extensions(a) {
key := strings.ToUpper(in.Name)
if !seen[key] {
seen[key] = true
names = append(names, in.Name)
}
}
return names
}
// EncodeExtension encodes one extended instruction on a: it resolves the
// mnemonic through the extension registry, picks the registered form whose
// arity matches the operands and encodes against it. The first form that
// encodes wins. When every matching form rejects the operands, the error
// comes from the form whose operand kinds the list points at (the one with
// the most matching positions), so a mis-spelled predicate qualifier is
// diagnosed as one, not as the unpredicated form's register complaint.
func EncodeExtension(a arch.Arch, mnemonic string, ops ...arch.ExtOperand) ([]byte, error) {
cands, ok := LookupExtension(a, mnemonic)
if !ok {
return nil, fmt.Errorf("%s registers no extended instruction %q", a, mnemonic)
}
var bestErr error
var bestScore int
var tried int
for _, in := range cands {
if in.Form.Arity() != len(ops) {
continue
}
tried++
b, err := in.Encode(ops)
if err == nil {
return b, nil
}
if score := kindScore(in.Form, ops); bestErr == nil || score > bestScore {
bestErr, bestScore = err, score
}
}
if tried == 0 {
return nil, fmt.Errorf("%s: extended %q takes %s, got %d operands",
a, mnemonic, extensionAritySummary(cands), len(ops))
}
return nil, bestErr
}
// kindScore counts the positions whose operand kind matches what the form
// wants, the tie-break that picks the most specific rejection.
func kindScore(form arch.ExtForm, ops []arch.ExtOperand) int {
kinds := form.Kinds()
score := 0
for i, op := range ops {
if i < len(kinds) && op.Kind == kinds[i] {
score++
}
}
return score
}
// ExtensionEncodable reports whether the extension layer of a encodes the
// mnemonic with these operands. It mirrors asm.Encodable for the extension
// layer: the predicate the linter consults once the hook wires it in.
func ExtensionEncodable(a arch.Arch, mnemonic string, ops ...arch.ExtOperand) bool {
_, err := EncodeExtension(a, mnemonic, ops...)
return err == nil
}
// extensionAritySummary describes the operand counts the candidate forms
// take, "2 or 3" style, for the arity error.
func extensionAritySummary(cands []arch.ExtInstr) string {
counts := make([]int, 0, len(cands))
seen := make(map[int]bool)
for _, in := range cands {
n := in.Form.Arity()
if !seen[n] {
seen[n] = true
counts = append(counts, n)
}
}
slices.Sort(counts)
var b strings.Builder
for i, n := range counts {
if i > 0 {
if i == len(counts)-1 {
b.WriteString(" or ")
} else {
b.WriteString(", ")
}
}
fmt.Fprintf(&b, "%d", n)
}
b.WriteString(" operands")
return b.String()
}
+197
View File
@@ -0,0 +1,197 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/hex"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
)
// TestExtensionRegistryARM64 checks the mnemonic lookup over and above the
// generated arm64 table: one mnemonic, several forms, case-insensitive, and
// nothing offered for a spelling the layer does not carry.
func TestExtensionRegistryARM64(t *testing.T) {
add, ok := LookupExtension(arch.ARM64, "ADD")
if !ok {
t.Fatal("LookupExtension(ARM64, ADD) found nothing")
}
var forms []arch.ExtForm
for _, in := range add {
if in.Name != "ADD" {
t.Errorf("candidate %q leaked into the ADD lookup", in.Name)
}
forms = append(forms, in.Form)
}
if len(forms) != 3 ||
forms[0] != arch.ExtFormVectors ||
forms[1] != arch.ExtFormPredicated ||
forms[2] != arch.ExtFormImmediate {
t.Errorf("ADD registers forms %v, want unpredicated, predicated and immediate", forms)
}
if _, ok := LookupExtension(arch.ARM64, "add"); !ok {
t.Error("the lookup is case-sensitive")
}
if _, ok := LookupExtension(arch.ARM64, "NOSUCHINSTR"); ok {
t.Error("a non-extended mnemonic resolved")
}
sqadd, ok := LookupExtension(arch.ARM64, "SQADD")
if !ok || len(sqadd) != 2 {
t.Errorf("SQADD registers %d forms, want the unpredicated and immediate pair", len(sqadd))
}
}
// TestExtensionAboveGeneratedTable pins the layering: SQADD is nowhere in the
// generated arm64 table (the toolchain knows only the NEON spelling VSQADD)
// yet the extension layer carries it, while ADD sits in both layers
// independently.
func TestExtensionAboveGeneratedTable(t *testing.T) {
if _, found := arch.ForArch(arch.ARM64).Lookup("SQADD"); found {
t.Error("SQADD is in the generated table, the layering assumption broke")
}
if _, ok := LookupExtension(arch.ARM64, "SQADD"); !ok {
t.Error("SQADD is missing from the extension layer")
}
if _, found := arch.ForArch(arch.ARM64).Lookup("ADD"); !found {
t.Error("ADD vanished from the generated table")
}
if add, ok := LookupExtension(arch.ARM64, "ADD"); !ok || len(add) != 3 {
t.Errorf("ADD carries %d extension forms, want 3", len(add))
}
}
// TestEncodeExtensionGolden encodes through the registry and pins the same
// golden words the arch table tests pin, proving the registry resolves to the
// right encoding.
func TestEncodeExtensionGolden(t *testing.T) {
for _, tt := range []struct {
name string
mnem string
ops []arch.ExtOperand
want uint32
}{
{"unpredicated add", "ADD",
[]arch.ExtOperand{
arch.ExtVector(2, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB),
},
0x04200040},
{"predicated mul", "MUL",
[]arch.ExtOperand{
arch.ExtVector(0, arch.ExtArrB), arch.ExtPredicate(2, arch.ExtQualMerging), arch.ExtVector(0, arch.ExtArrB),
},
0x04100800},
{"immediate add with derived shift", "ADD",
[]arch.ExtOperand{arch.ExtImmediate(32512), arch.ExtVector(0, arch.ExtArrH)},
0x2560efe0},
{"signed immediate mul", "MUL",
[]arch.ExtOperand{arch.ExtImmediate(-1), arch.ExtVector(0, arch.ExtArrB)},
0x2530dfe0},
} {
got, err := EncodeExtension(arch.ARM64, tt.mnem, tt.ops...)
if err != nil {
t.Errorf("%s: encode: %v", tt.name, err)
continue
}
if want := hex.EncodeToString([]byte{
byte(tt.want), byte(tt.want >> 8), byte(tt.want >> 16), byte(tt.want >> 24),
}); hex.EncodeToString(got) != want {
t.Errorf("%s:\n got %x\n want %s", tt.name, got, want)
}
}
}
// TestEncodeExtensionErrors checks the registry's diagnostics: a wrong arity
// names every form's count, an operand the first candidate rejects surfaces
// its own message once a later form takes over.
func TestEncodeExtensionErrors(t *testing.T) {
if _, err := EncodeExtension(arch.ARM64, "ADD", arch.ExtVector(0, arch.ExtArrB)); err == nil {
t.Error("one operand encoded, want an arity error")
} else if !strings.Contains(err.Error(), "2 or 3 operands") {
t.Errorf("arity error %q does not name the counts", err)
}
// The predicated candidate must answer for its own operands: the /Z
// qualifier is rejected with the merging message, not the unpredicated
// form's register-kind complaint.
_, err := EncodeExtension(arch.ARM64, "ADD",
arch.ExtVector(0, arch.ExtArrB), arch.ExtPredicate(0, arch.ExtQualZeroing), arch.ExtVector(0, arch.ExtArrB))
if err == nil {
t.Fatal("/Z encoded, want an error")
}
if !strings.Contains(err.Error(), "/M") {
t.Errorf("error %q does not name the merging qualifier", err)
}
if _, err := EncodeExtension(arch.ARM64, "NOSUCHINSTR", arch.ExtVector(0, arch.ExtArrB)); err == nil ||
!strings.Contains(err.Error(), "registers no extended instruction") {
t.Errorf("unknown mnemonic error = %v", err)
}
}
// TestExtensionEncodable checks the predicate the later Encodable hook will
// call: true exactly when the registry encodes the operand list.
func TestExtensionEncodable(t *testing.T) {
if !ExtensionEncodable(arch.ARM64, "ADD",
arch.ExtVector(0, arch.ExtArrS), arch.ExtVector(1, arch.ExtArrS), arch.ExtVector(2, arch.ExtArrS)) {
t.Error("an encodable unpredicated add reported false")
}
if !ExtensionEncodable(arch.ARM64, "ADD",
arch.ExtVector(1, arch.ExtArrS), arch.ExtPredicate(0, arch.ExtQualMerging), arch.ExtVector(0, arch.ExtArrS)) {
t.Error("an encodable predicated add reported false")
}
if ExtensionEncodable(arch.ARM64, "ADD",
arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrS), arch.ExtVector(0, arch.ExtArrB)) {
t.Error("mismatched arrangements reported encodable")
}
if ExtensionEncodable(arch.ARM64, "ADD", arch.ExtVector(0, arch.ExtArrB)) {
t.Error("a one-operand add reported encodable")
}
if ExtensionEncodable(arch.ARM64, "NOSUCHINSTR") {
t.Error("an unregistered mnemonic reported encodable")
}
}
// TestExtensionArchIsolation is the architecture-binding negative case: the
// extension layer is registered for arm64 alone, and no other architecture
// answers its queries, not even for a mnemonic the amd64 base table carries.
func TestExtensionArchIsolation(t *testing.T) {
ops := []arch.ExtOperand{
arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB),
}
for _, a := range []arch.Arch{arch.AMD64, arch.RISCV, arch.LOONG64, arch.Unknown} {
if cands, ok := LookupExtension(a, "ADD"); ok || cands != nil {
t.Errorf("LookupExtension(%s, ADD) offered %d candidates", a, len(cands))
}
if cands, ok := LookupExtension(a, "MUL"); ok || cands != nil {
t.Errorf("LookupExtension(%s, MUL) offered %d candidates", a, len(cands))
}
if got, err := EncodeExtension(a, "ADD", ops...); err == nil {
t.Errorf("EncodeExtension(%s, ADD) encoded %x, want a refusal", a, got)
} else if !strings.Contains(err.Error(), string(a)) {
t.Errorf("EncodeExtension(%s) error %q does not name the architecture", a, err)
}
if ExtensionEncodable(a, "ADD", ops...) {
t.Errorf("ExtensionEncodable(%s, ADD) reported true", a)
}
if names := ExtensionNames(a); len(names) != 0 {
t.Errorf("ExtensionNames(%s) = %v, want none", a, names)
}
if got := arch.Extensions(a); len(got) != 0 {
t.Errorf("arch.Extensions(%s) carries %d instructions", a, len(got))
}
}
}
// TestExtensionNamesARM64 checks the completion-facing name list: every
// distinct mnemonic of the family, first-occurrence order, no duplicates.
func TestExtensionNamesARM64(t *testing.T) {
want := []string{"ADD", "SUB", "SQADD", "UQADD", "SQSUB", "UQSUB", "MUL", "SMULH", "UMULH", "SUBR"}
got := ExtensionNames(arch.ARM64)
if strings.Join(got, ",") != strings.Join(want, ",") {
t.Errorf("ExtensionNames(ARM64) = %v, want %v", got, want)
}
if n := len(arch.Extensions(arch.ARM64)); n != 23 {
t.Errorf("the family registers %d instructions, want 23", n)
}
}
+8
View File
@@ -63,6 +63,7 @@ const (
const (
kindSTEXT = 1
kindSRODATA = 3
kindSNOPTRDATA = 5
kindSDATA = 7
kindSDWARFFCN = 14
kindSDWARFLINES = 20
@@ -305,9 +306,16 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
if !d.Static {
name = pkgPath + "." + name
}
// RODATA implies no pointers, so it wins over NOPTR: the kind is
// SRODATA either way, exactly as the toolchain chooses it. Plain
// NOPTR data is SNOPTRDATA, which the linker keeps out of the GC's
// type scan; a plain SDATA symbol would demand Go type information
// no assembly file can supply, and the link would fail.
typ := uint8(kindSDATA)
if d.Rodata {
typ = kindSRODATA
} else if d.Noptr {
typ = kindSNOPTRDATA
}
flag := uint8(0)
if d.Dupok {
+3
View File
@@ -45,6 +45,9 @@ func TestReadRuntimeSymbols(t *testing.T) {
// TestResolveExternalSymbols verifies end-to-end resolution of external
// symbol references.
func TestResolveExternalSymbols(t *testing.T) {
if testing.Short() {
t.Skip("resolves through a live go list -export: skipped in -short mode")
}
if _, err := exec.LookPath("go"); err != nil {
t.Skip("go toolchain not available")
}
+143 -1
View File
@@ -12,7 +12,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// goobjView is a minimal parsed view of a GOOBJ payload, enough to check
@@ -514,6 +514,145 @@ func fieldAfter(line, flag string) string {
return ""
}
// buildLogSteps is what a substitution test needs from a `go build -x -work`
// log: the work directory, the assembler's object, the package archive and
// the link command line.
type buildLogSteps struct {
work string
asmObj string // $WORK expanded
pkgArch string // $WORK expanded
linkLine string // still carries $WORK placeholders
}
// parseBuildLog extracts the build steps from a `go build -x -work` log.
// asmFile names the assembly file whose object the test substitutes. A
// missing step is a failure, not a skip: the toolchain changed shape and the
// substitution would silently test nothing.
func parseBuildLog(t *testing.T, log []byte, asmFile string) buildLogSteps {
t.Helper()
var st buildLogSteps
for line := range strings.SplitSeq(string(log), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
st.work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, asmFile) && !strings.Contains(line, "-gensymabis"):
st.asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
rest := strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
st.pkgArch = strings.Fields(strings.SplitN(rest, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
st.linkLine = line
}
}
if st.work == "" || st.asmObj == "" || st.pkgArch == "" || st.linkLine == "" {
t.Fatalf("could not locate the build steps (work=%q asmObj=%q pkgArch=%q link=%q):\n%s",
st.work, st.asmObj, st.pkgArch, st.linkLine, log)
}
st.asmObj = strings.ReplaceAll(st.asmObj, "$WORK", st.work)
st.pkgArch = strings.ReplaceAll(st.pkgArch, "$WORK", st.work)
return st
}
// substituteAndRelink swaps the gasm object into the package archive the
// baseline build produced and re-runs the captured link line against the
// rebuilt archive, writing the binary to outBin (the -x log's link step
// always targets the action graph's internal a.out, which the helper
// redirects; the copy to the -o target is a separate build action the helper
// does not need). The archive handed to the linker is proven to carry the
// gasm object byte for byte, so a build-layout change that skipped the
// substitution fails here instead of passing vacuously.
func substituteAndRelink(t *testing.T, goBin, dir string, st buildLogSteps, outBin string, gasmObj []byte, extraEnv ...string) {
t.Helper()
// The deliberate-run boundary: this path drives a real `go build` and
// cmd/link per invocation, minutes-scale work on the small single-core
// CI runner. Under -short (the push pipeline's mode) it skips; the
// local test gate and the dispatched workflows run it in full.
if testing.Short() {
t.Skip("end-to-end go build and link: skipped in -short mode")
}
// Extract the archive, overwrite the assembler's member with the gasm
// object and repack (go tool pack has no replace-in-place).
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract := exec.Command(goBin, "tool", "pack", "x", st.pkgArch)
extract.Dir = membersDir
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
member := filepath.Join(membersDir, filepath.Base(st.asmObj))
if _, err := os.Stat(member); err != nil {
t.Fatalf("the assembler's archive member was not extracted: %v", err)
}
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, gasmObj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", st.pkgArch)
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for m := range strings.FieldsSeq(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, m)
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
// Prove the substitution: the archive the linker is about to consume
// holds the gasm object, byte for byte.
checkDir := filepath.Join(dir, "check")
if err := os.MkdirAll(checkDir, 0o755); err != nil {
t.Fatal(err)
}
check := exec.Command(goBin, "tool", "pack", "x", newArch)
check.Dir = checkDir
if out, err := check.CombinedOutput(); err != nil {
t.Fatalf("pack x (verification): %v\n%s", err, out)
}
got, err := os.ReadFile(filepath.Join(checkDir, filepath.Base(st.asmObj)))
if err != nil {
t.Fatalf("read the substituted member back: %v", err)
}
if !bytes.Equal(got, gasmObj) {
t.Fatal("the repacked archive does not carry the gasm object")
}
// Re-link. The line carries a GOROOT assignment and $WORK placeholders;
// GOEXPERIMENT must match the toolchain's own, because the linker
// compares the object header against its configuration.
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
linkLine := strings.ReplaceAll(st.linkLine, "$WORK", st.work)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(st.work, "b001", "_pkg_.a"), newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(st.work, "b001", "exe", "a.out"), outBin)
env := append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)))
env = append(env, extraEnv...)
link := exec.Command("sh", "-c", linkLine)
link.Dir = dir
link.Env = env
if out, err := link.CombinedOutput(); err != nil {
t.Fatalf("link with the gasm object: %v\n%s", err, out)
}
}
// TestGOObjectExternalPackageLink is the cross-package end-to-end check: a
// GOOBJ whose code references a real external package symbol (runtime's
// morestack, a plain reference rather than the builtin noctxt form) must
@@ -524,6 +663,9 @@ func fieldAfter(line, flag string) string {
// failed. The binary is not run: morestack returns to the call site's
// stack check, which a hand-written caller has none of.
func TestGOObjectExternalPackageLink(t *testing.T) {
if testing.Short() {
t.Skip("end-to-end go build and link: skipped in -short mode")
}
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
+6 -6
View File
@@ -9,8 +9,8 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64,
@@ -307,12 +307,12 @@ func TestStackGuardBytesLOONG64(t *testing.T) {
func TestStackGuardGOObjInternalCall(t *testing.T) {
for _, tt := range []struct {
src string
assemble func(*ast.File) (*Image, error)
assemble func(*ast.File, ...AssembleOption) (*Image, error)
}{
{"g_amd64.s", AssembleFile},
{"g_arm64.s", AssembleFileARM64},
{"g_riscv64.s", AssembleFileRISCV},
{"g_loong64.s", AssembleFileLOONG64},
{"g_arm64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileARM64(f) }},
{"g_riscv64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileRISCV(f) }},
{"g_loong64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileLOONG64(f) }},
} {
f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
+271 -25
View File
@@ -90,6 +90,40 @@ var noOperandTable = map[string][]byte{
"LOCK": {0xF0},
"REP": {0xF3},
"REPN": {0xF2},
"ENDBR64": {0xF3, 0x0F, 0x1E, 0xFA},
}
// sysUnaryTable maps the one-operand system instructions to their bytes:
// the prefix, the opcode and the /digit the reg field carries. The operand
// is a register or memory in r/m.
var sysUnaryTable = map[string]struct {
prefix byte
opcode []byte
digit int
}{
"CLWB": {0x66, []byte{0x0F, 0xAE}, 6},
"TPAUSE": {0x66, []byte{0x0F, 0xAE}, 6},
"UMONITOR": {0xF3, []byte{0x0F, 0xAE}, 6},
"UMWAIT": {0xF2, []byte{0x0F, 0xAE}, 6},
"RDPID": {0xF3, []byte{0x0F, 0xC7}, 7},
"CLDEMOTE": {0x00, []byte{0x0F, 0x1C}, 0},
}
// encodeSysUnary emits a one-operand system instruction: the operand in r/m
// under the fixed /digit, no REX.W.
func (e *enc) encodeSysUnary(mnem string, m struct {
prefix byte
opcode []byte
digit int
}, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
}
i := &instr{prefix: m.prefix, opcode: m.opcode, modrm: -1, sib: -1}
if err := setRMDigit(i, m.digit, ops[0], 8); err != nil {
return err
}
return e.emit(i)
}
// --- MOV --------------------------------------------------------------------
@@ -111,6 +145,76 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
// silently emit REX.W 8B with the wrong operand meaning.
_, srcVec := vecReg(src)
dstReg, dstVec := vecReg(dst)
// Control and debug register moves: 0F 20 (CRn→r64), 0F 22 (r64→CRn),
// 0F 21 (DRn→r64) and 0F 23 (r64→DRn). The CR/DR number rides the reg
// field, the general register r/m; CR8+/DR8+ take REX.R.
if c, ok := src.(Reg); ok && c.ctl != 0 {
g, ok := dst.(Reg)
if !ok || g.isVec() || g.ctl != 0 {
return fmt.Errorf("MOV: control/debug register load needs a general register destination")
}
opc := byte(0x20)
if c.ctl == 2 {
opc = 0x21
}
return e.emit(&instr{
opcode: []byte{0x0F, opc},
modrm: 0xC0 | (c.idx&7)<<3 | (g.idx & 7),
sib: -1, rexR: c.idx >= 8, rexB: g.idx >= 8,
})
}
if c, ok := dst.(Reg); ok && c.ctl != 0 {
g, ok := src.(Reg)
if !ok || g.isVec() || g.ctl != 0 {
return fmt.Errorf("MOV: control/debug register store needs a general register source")
}
opc := byte(0x22)
if c.ctl == 2 {
opc = 0x23
}
return e.emit(&instr{
opcode: []byte{0x0F, opc},
modrm: 0xC0 | (c.idx&7)<<3 | (g.idx & 7),
sib: -1, rexR: c.idx >= 8, rexB: g.idx >= 8,
})
}
// MMX register moves: MOVQ M0, mem and MOVQ mem, M0 are the MMX
// load/store pair 0F 6F/0F 7F (no prefix); a register pair takes the
// load opcode. The XMM MOVQ forms follow below.
if m, ok := src.(Reg); ok && m.mmx {
switch d := dst.(type) {
case Reg:
if !d.mmx {
return fmt.Errorf("MOV: MMX register moves stay inside the M bank")
}
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
if err := setRM(i, d, src, 8); err != nil {
return err
}
return e.emit(i)
case Mem:
i := &instr{opcode: []byte{0x0F, 0x7F}, modrm: -1, sib: -1}
if err := setRM(i, m, d, 8); err != nil {
return err
}
return e.emit(i)
}
return fmt.Errorf("MOV: invalid MMX destination")
}
if m, ok := dst.(Reg); ok && m.mmx {
srcM, ok := src.(Mem)
if !ok {
return fmt.Errorf("MOV: MMX load takes a memory source")
}
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
if err := setRM(i, m, srcM, 8); err != nil {
return err
}
return e.emit(i)
}
if srcVec || dstVec {
if dstVec {
if g, ok := src.(Reg); ok && !g.isVec() {
@@ -192,6 +296,30 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
}
return e.emit(i)
case TLSMem:
if !dstIsReg {
return fmt.Errorf("MOV: two memory operands")
}
// MOV r, off(TLS): the segment-prefixed absolute load, reg=dst,
// rm=src(tlsMem) through the SIB escape; the disp32 is the TLS slot
// offset with its R_TLSLE patch site.
i := newInstr(size, []byte{movRR(size)})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
case SegAbs:
if !dstIsReg {
return fmt.Errorf("MOV: two memory operands")
}
// MOV r, 0x30(GS): the segment-absolute load.
i := newInstr(size, []byte{movRR(size)})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
case Imm:
if dstIsReg {
v := int64(src)
@@ -232,11 +360,24 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
i.imm = imm
return e.emit(i)
}
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0. An immediate in the
// destination slot is the absolute-address crash-store spelling,
// MOVL $0xf1, 0xf1: the parser reads the trailing bare constant
// as an immediate, and the store's disp32 carries the address.
op := byte(0xC7)
if size == 1 {
op = 0xC6
}
if d, ok := dst.(Imm); ok {
i := newInstr(size, []byte{op})
setSegAbs(i, 0, SegAbs{Disp: int64(d)})
immBytes, err := immediate(int64(src), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
i := newInstr(size, []byte{op})
if err := setRMDigit(i, 0, dst, size); err != nil {
return err
@@ -481,6 +622,14 @@ func (e *enc) encodeLea(ops []Operand, size int) error {
default:
return fmt.Errorf("LEA: source must be a memory operand")
}
// LEA accepts the full unsigned 32-bit displacement span where the
// loads and stores reject it beyond the signed one; the wide values
// ride the same disp32 bytes as their two's-complement bit pattern.
if m, ok := src.(Mem); ok && m.Disp >= 1<<31 && m.Disp <= (1<<32)-1 {
c := m
c.Disp = int64(int32(uint32(m.Disp)))
src = c
}
i := newInstr(size, []byte{0x8D})
if err := setRM(i, dstReg, src, size); err != nil {
return err
@@ -626,8 +775,29 @@ func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error {
func (e *enc) encodeImul(ops []Operand, size int) error {
switch len(ops) {
case 1:
// The one-operand form, IMUL r/m: F6/F7 /5 with AL/AX/EAX/RAX as the
// implied destination (the toolchain's one-register shape).
opc := byte(0xF7)
if size == 1 {
opc = 0xF6
}
i := newInstr(size, []byte{opc})
if err := setRMDigit(i, 5, ops[0], size); err != nil {
return err
}
return e.emit(i)
case 2:
// IMUL r, r/m: 0x0F 0xAF.
// Two shapes. The leading-immediate spelling IMUL $imm, r multiplies
// r in place (dst = rm = r): the shape GOROOT's clock code writes.
// Otherwise IMUL r, r/m: 0x0F 0xAF.
if imm, ok := ops[0].(Imm); ok {
dstReg, isReg := ops[1].(Reg)
if !isReg {
return fmt.Errorf("IMUL: destination must be a register")
}
return e.encodeImulImm(imm, dstReg, dstReg, size)
}
dstReg, ok := ops[1].(Reg)
if !ok {
return fmt.Errorf("IMUL: destination must be a register")
@@ -647,27 +817,34 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
if !ok {
return fmt.Errorf("IMUL: immediate operand expected first")
}
// Plan 9 order: IMUL $imm, src, dst.
if fits8(int64(imm)) {
i := newInstr(size, []byte{0x6B})
if err := setRM(i, dstReg, ops[1], size); err != nil {
return err
}
i.imm = []byte{byte(int8(imm))}
return e.emit(i)
}
i := newInstr(size, []byte{0x69})
if err := setRM(i, dstReg, ops[1], size); err != nil {
// Plan 9 order: IMUL $imm, src, dst; the source stays a general
// r/m operand (setRM takes registers and memory alike).
return e.encodeImulImm(imm, ops[1], dstReg, size)
}
return fmt.Errorf("IMUL expects 1, 2 or 3 operands, got %d", len(ops))
}
// encodeImulImm emits the immediate multiply: 0x6B with a sign-extended imm8
// when the value fits, 0x69 with a 32-bit immediate otherwise.
func (e *enc) encodeImulImm(imm Imm, rm Operand, dst Reg, size int) error {
if fits8(int64(imm)) {
i := newInstr(size, []byte{0x6B})
if err := setRM(i, dst, rm, size); err != nil {
return err
}
immBytes, err := immediate(int64(imm), size, false)
if err != nil {
return err
}
i.imm = immBytes
i.imm = []byte{byte(int8(imm))}
return e.emit(i)
}
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
i := newInstr(size, []byte{0x69})
if err := setRM(i, dst, rm, size); err != nil {
return err
}
immBytes, err := immediate(int64(imm), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
// --- PUSH / POP -------------------------------------------------------------
@@ -689,6 +866,27 @@ func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
w16 := size == 2
switch op := ops[0].(type) {
case Reg:
// Segment registers: FS and GS carry their own one-byte opcodes
// under 0F (A0/A8 push, A1/A9 pop); the other four spellings are
// not pushable in 64-bit mode.
if n, isSeg := op.segNumber(); isSeg {
switch n {
case 4: // FS
if push {
return e.emit(&instr{opcode: []byte{0x0F, 0xA0}, modrm: -1, sib: -1})
}
return e.emit(&instr{opcode: []byte{0x0F, 0xA1}, modrm: -1, sib: -1})
case 5: // GS
if push {
return e.emit(&instr{opcode: []byte{0x0F, 0xA8}, modrm: -1, sib: -1})
}
return e.emit(&instr{opcode: []byte{0x0F, 0xA9}, modrm: -1, sib: -1})
}
return fmt.Errorf("PUSH/POP: only FS and GS are encodable in 64-bit mode")
}
if op.mmx || op.isVec() || op.fp || op.ctl != 0 {
return fmt.Errorf("PUSH/POP: invalid register operand")
}
base := byte(0x50) // PUSH r; POP is 0x58
if !push {
base = 0x58
@@ -1040,6 +1238,38 @@ var sseMoveTable = map[string]sseMove{
"MOVSS": {0xF3, 0x10, 0x11}, // scalar single
}
// sseStoreOnly holds the store-only SSE forms, OP xmm, mem: the XMM register
// rides the reg field and memory r/m (the non-temporal store).
var sseStoreOnly = map[string]struct {
prefix byte
op byte
}{
"MOVNTDQ": {0x66, 0xE7},
}
// encodeSSEStoreOnly encodes OP xmm, mem (reg = the XMM source, r/m = the
// destination memory).
func (e *enc) encodeSSEStoreOnly(mnem string, m struct {
prefix byte
op byte
}, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("%s source must be a vector register", mnem)
}
if !isX86Mem(ops[1]) {
return fmt.Errorf("%s destination must be a memory operand", mnem)
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, srcReg, ops[1], 8); err != nil {
return err
}
return e.emit(i)
}
// encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the
// load form (reg = destination), matching the Go assembler.
func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
@@ -1255,14 +1485,23 @@ func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error {
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
if !ok || (!dstReg.isVec() && !dstReg.mmx) {
return fmt.Errorf("SSE binary destination must be a vector register")
}
// The MMX twins of the packed-integer SSE2 ops drop the 0x66 prefix:
// PADDD M2, M1 is 0F FE where the XMM form is 66 0F FE.
prefix := m.prefix
if dstReg.mmx {
if prefix != 0x66 {
return fmt.Errorf("SSE binary: this form takes no MMX register operand")
}
prefix = 0
}
opcode := []byte{0x0F, m.op}
if m.map38 {
opcode = []byte{0x0F, 0x38, m.op}
}
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
i := &instr{prefix: prefix, opcode: opcode, modrm: -1, sib: -1}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
@@ -1738,12 +1977,19 @@ func (e *enc) encodeSSEShift(name string, ops []Operand) error {
// immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family:
// F2 0F C2 with reg = dst, rm = src.
func (e *enc) encodeCmpsd(ops []Operand) error {
return e.encodeSSECmp("CMPSD", 0xF2, ops)
}
// encodeSSECmp encodes the SSE compare family (CMPSD/CMPSS/CMPPS/CMPPD):
// 0F C2 /r ib with the predicate immediate last in Plan 9 order
// (src, dst, $imm) and the packed forms' prefixes.
func (e *enc) encodeSSECmp(mnem string, prefix byte, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops))
return fmt.Errorf("%s expects 3 operands (src, dst, $imm), got %d", mnem, len(ops))
}
imm, ok := ops[2].(Imm)
if !ok {
return fmt.Errorf("CMPSD predicate must be an immediate")
return fmt.Errorf("%s predicate must be an immediate", mnem)
}
immByte, err := imm8(int64(imm))
if err != nil {
@@ -1751,9 +1997,9 @@ func (e *enc) encodeCmpsd(ops []Operand) error {
}
dstReg, ok2 := ops[1].(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("CMPSD destination must be a vector register")
return fmt.Errorf("%s destination must be a vector register", mnem)
}
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
i := &instr{prefix: prefix, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return err
}
+3 -3
View File
@@ -16,7 +16,7 @@ import (
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel;
@@ -25,7 +25,7 @@ import (
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx2_amd64.s"
if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit")
t.Skip("go-libraries repository not present next to gasm-sdk")
}
src, err := os.ReadFile(path)
if err != nil {
@@ -86,7 +86,7 @@ func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx512_amd64.s"
if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit")
t.Skip("go-libraries repository not present next to gasm-sdk")
}
src, err := os.ReadFile(path)
if err != nil {
+20 -9
View File
@@ -13,7 +13,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// The differential kernels for the DATA-path and front-end gaps are kept in
@@ -23,9 +23,18 @@ import (
// bytes must agree with the relocation sites masked on both sides.
// toolAsmObject assembles path with the installed toolchain's assembler for
// goarch ("" = the host) and returns the object bytes.
// goarch ("" = the host) and returns the object bytes. Every live-oracle
// comparison funnels through here, so this is also where the deliberate-run
// boundary sits: under -short (the push pipeline's mode) the comparisons
// skip, because each spawns a go tool asm subprocess and the small single-
// core runner pays seconds per spawn. The encodings stay pinned by the
// golden-byte tests in every mode; the live oracle runs in the local test
// gate and the dispatched workflows.
func toolAsmObject(t *testing.T, path, goarch string) []byte {
t.Helper()
if testing.Short() {
t.Skip("live go tool asm oracle: skipped in -short mode")
}
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
@@ -63,7 +72,10 @@ func toolAsmObject(t *testing.T, path, goarch string) []byte {
}
// oracleFuncCode extracts the non-package TEXT functions' code bytes from a
// toolchain object, keyed by the name the object records (pkg.name).
// toolchain object, keyed by the name the object records (pkg.name). Each
// function's span is its own symbol size: a toolchain object that follows
// the text with data symbols (the synthesised float-constant pool) would
// otherwise fold them into the last function's bytes.
func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
t.Helper()
v := openGoobj(t, obj)
@@ -76,18 +88,13 @@ func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef} {
preceding += len(v.blk(bi)) / symSize
}
total := preceding + len(nps)
out := make(map[string][]byte, len(nps))
for i, s := range nps {
if s.typ != kindSTEXT {
continue
}
start := le.Uint32(didx[4*(preceding+i):])
end := uint32(len(data))
if preceding+i+1 < total {
end = le.Uint32(didx[4*(preceding+i+1):])
}
out[s.name] = data[start:end]
out[s.name] = data[start : start+s.size]
}
return out
}
@@ -129,6 +136,10 @@ func TestDifferentialKernels(t *testing.T) {
{filepath.Join("..", "testdata", "verify", "datarel_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "divslash_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "semicolons_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "forms_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true},
{filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true},
} {
+1 -1
View File
@@ -12,7 +12,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
+122 -12
View File
@@ -5,10 +5,11 @@ package asm
import (
"fmt"
"math"
"sort"
"strconv"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// Image is an assembled file: the function bodies laid out in source order,
@@ -133,7 +134,8 @@ type DataSymbol struct {
Offset int // byte offset within Data
Size int
Static bool // the <> marker: file-local, not exported
Rodata bool // the RODATA flag: read-only data
Rodata bool // the RODATA flag: read-only data (implies no pointers)
Noptr bool // the NOPTR flag: data with no pointers, kept out of GC scanning
Dupok bool // the DUPOK flag: duplicate-OK
// Relocs carries the symbol-valued DATA initialisers ("DATA s+0(SB)/8,
// $other(SB)"): fields of this symbol's data that hold another symbol's
@@ -149,6 +151,18 @@ func (img *Image) Bytes() []byte {
return append(out, img.Data...)
}
// AssembleOption adjusts the file-level assembly context.
type AssembleOption func(*linkInfo)
// WithGOOS selects the target operating system for the forms that depend on
// it, the TLS access shape above all: linux and freebsd take the
// one-instruction form, windows and plan9 keep the two-instruction load.
func WithGOOS(goos string) AssembleOption {
return func(l *linkInfo) {
l.goos = goos
}
}
// AssembleFile assembles every TEXT function of a parsed file and lays out
// its static symbols (GLOBL/DATA) in a data section behind the code. Each
// reference to a file-local static symbol becomes a RIP-relative load whose
@@ -156,7 +170,7 @@ func (img *Image) Bytes() []byte {
// GLOBL defines is recorded as an external relocation (Externals) with its
// displacement left zero, the object-file emitters resolve it at link
// time, while the raw image (Bytes) cannot represent it.
func AssembleFile(f *ast.File) (*Image, error) {
func AssembleFile(f *ast.File, opts ...AssembleOption) (*Image, error) {
dataSyms, err := collectData(f)
if err != nil {
return nil, err
@@ -165,7 +179,18 @@ func AssembleFile(f *ast.File) (*Image, error) {
for _, d := range dataSyms {
known[d.name] = true
}
// TEXT symbols are file-level definitions too: a symbol immediate
// ($fn(SB)) may name one, exactly as a data reference names a GLOBL.
for _, d := range f.Decls {
if t, ok := d.(*ast.Text); ok {
known[t.Name.Name] = true
}
}
link := &linkInfo{symbols: known, allowExternal: true}
for _, o := range opts {
o(link)
}
poolSeen := map[string]bool{}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
textOff := map[string]int{}
@@ -179,7 +204,26 @@ func AssembleFile(f *ast.File) (*Image, error) {
if !ok {
continue
}
code, patches, labels, steps, lines, err := assemble(t, link)
code, patches, labels, steps, lines, pool, err := assemble(t, link)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
// The pooled floating-point constants join the declared data as
// read-only symbols, deduplicated across the file (the toolchain
// synthesises the same symbols into its rodata).
for _, entry := range pool {
if poolSeen[entry.name] {
continue
}
poolSeen[entry.name] = true
dataSyms = append(dataSyms, dataSym{
name: entry.name,
buf: entry.data,
size: len(entry.data),
rodata: true,
dupok: true,
})
}
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
@@ -226,6 +270,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
Size: len(d.buf),
Static: d.static,
Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok,
})
img.Data = append(img.Data, d.buf...)
@@ -299,6 +344,10 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
if err != nil {
return nil, err
}
// The pooled $i64 constants the wide MOV immediate loads refer to join
// the declared data as read-only symbols, deduplicated across the file
// (the toolchain synthesises the same symbols into its rodata).
litSeen := map[string]bool{}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
for _, d := range f.Decls {
@@ -306,10 +355,23 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
if !ok {
continue
}
code, labels, relocs, lines, spadj, err := assembleRISCV(t)
code, labels, relocs, lines, spadj, lits, err := assembleRISCV(t)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
for _, lit := range lits {
if litSeen[lit.Name] {
continue
}
litSeen[lit.Name] = true
dataSyms = append(dataSyms, dataSym{
name: lit.Name,
buf: lit.Data,
size: len(lit.Data),
rodata: true,
dupok: true,
})
}
fl := FuncLayout{
Name: t.Name.Name,
Pkg: t.Name.Pkg,
@@ -353,6 +415,7 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
Size: d.size,
Static: d.static,
Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok,
})
}
@@ -425,6 +488,7 @@ func AssembleFileLOONG64(f *ast.File) (*Image, error) {
Size: d.size,
Static: d.static,
Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok,
})
}
@@ -486,6 +550,7 @@ type dataSym struct {
size int
static bool
rodata bool
noptr bool
dupok bool
// relocs are the symbol-valued DATA fields, in declaration order; Off
// is relative to the symbol's data start.
@@ -527,12 +592,14 @@ func collectData(f *ast.File) ([]dataSym, error) {
switch f {
case "RODATA":
ds.rodata = true
case "NOPTR":
ds.noptr = true
case "DUPOK":
ds.dupok = true
default:
// Legacy numeric flag constants (runtime/textflag.h):
// DUPOK is 2, RODATA is 8; combinations arrive as one
// number (e.g. 10 = RODATA|DUPOK).
// DUPOK is 2, RODATA is 8, NOPTR is 16; combinations arrive
// as one number (e.g. 10 = RODATA|DUPOK).
if n, err := strconv.Atoi(f); err == nil {
if n&2 != 0 {
ds.dupok = true
@@ -540,6 +607,9 @@ func collectData(f *ast.File) ([]dataSym, error) {
if n&8 != 0 {
ds.rodata = true
}
if n&16 != 0 {
ds.noptr = true
}
}
}
}
@@ -561,11 +631,6 @@ func collectData(f *ast.File) ([]dataSym, error) {
return nil, fmt.Errorf("DATA %q: missing value", dd.Name.Name)
}
w := dd.Width
switch w {
case 1, 2, 4, 8:
default:
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
}
off := dd.Name.Offset
buf := syms[i].buf
if off < 0 || off+int64(w) > int64(len(buf)) {
@@ -587,9 +652,54 @@ func collectData(f *ast.File) ([]dataSym, error) {
})
continue
}
// A string or rune value ("DATA s+0(SB)/20, $"text"") writes its
// bytes into the field and leaves the rest zero, the toolchain's
// WriteString: the declared width must hold every byte, and any
// width is legal.
if s := dd.Value.Imm.Str; s != "" && !dd.Value.Imm.HasVal {
text, err := strconv.Unquote(s)
if err != nil {
return nil, fmt.Errorf("DATA %q: invalid string value %s", dd.Name.Name, s)
}
if len(text) > w {
return nil, fmt.Errorf("DATA %q: string of %d bytes does not fit width %d", dd.Name.Name, len(text), w)
}
copy(buf[off:], text)
continue
}
// A floating-point value stores its IEEE-754 bits: /4 the float32
// rounding of the parsed double, /8 the full 64 bits, the
// toolchain's WriteFloat32 and WriteFloat64.
if f := dd.Value.Imm.Float; f != "" && !dd.Value.Imm.HasVal {
num, err := strconv.ParseFloat(f, 64)
if err != nil {
return nil, fmt.Errorf("DATA %q: invalid floating-point value %q", dd.Name.Name, f)
}
if dd.Value.Imm.Neg {
num = -num
}
var v uint64
switch w {
case 4:
v = uint64(math.Float32bits(float32(num)))
case 8:
v = math.Float64bits(num)
default:
return nil, fmt.Errorf("DATA %q: invalid width %d for a float (want 4 or 8)", dd.Name.Name, w)
}
for j := range w {
buf[off+int64(j)] = byte(v >> (8 * j))
}
continue
}
if !dd.Value.Imm.HasVal {
return nil, fmt.Errorf("DATA %q: value must be an integer immediate or a symbol address", dd.Name.Name)
}
switch w {
case 1, 2, 4, 8:
default:
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
}
v := dd.Value.Imm.Val
if dd.Value.Imm.Neg {
v = -v
+85 -38
View File
@@ -5,13 +5,14 @@ package asm
import (
"encoding/binary"
"fmt"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestAssembleFileStaticData checks the whole-image layout; code, padding
@@ -59,23 +60,15 @@ DATA small<>+0(SB)/4, $0x1234
}
}
// TestAssembleFileErrors checks the static-symbol error paths.
// TestAssembleFileErrors checks the static-symbol error paths. A reference
// to a static symbol no GLOBL defines defers to the linker exactly as the
// toolchain does (an external relocation), so it is not an error here.
func TestAssembleFileErrors(t *testing.T) {
cases := []struct {
name string
src string
want string // substring of the error
}{
{
"undefined symbol",
`
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
VMOVDQU nope<>(SB), X0
RET
`,
"undefined symbol",
},
{
"DATA without GLOBL",
`
@@ -368,23 +361,8 @@ func main() {
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || linkLine == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
}
defer os.RemoveAll(work)
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
st := parseBuildLog(t, buildLog, "main_amd64.s")
defer os.RemoveAll(st.work)
// Assemble the same source with gasm and substitute the object.
src, err := os.ReadFile(filepath.Join(dir, "main_amd64.s"))
@@ -399,22 +377,91 @@ func main() {
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
gasmObj, err := img.GOObject("dlink", "main_amd64.s")
// The package path is "main": the linker resolves the Go code's
// references against main.<name>, so the object must define the symbols
// under that prefix whatever the module is called.
gasmObj, err := img.GOObject("main", "main_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"), gasmObj)
// The linked program must run and find the right function behind the
// data word.
out, err := exec.Command(filepath.Join(dir, "prog")).CombinedOutput()
out, err := exec.Command(filepath.Join(dir, "prog2")).CombinedOutput()
if err != nil {
t.Fatalf("linked program failed: %v\n%s", err, out)
}
}
// TestCollectDataFloatAndStringValues covers the non-integer DATA values the
// runtime's math and asm files use: floating-point initialisers store their
// IEEE-754 bits (/4 the float32 rounding, /8 the full double) and string
// initialisers write their bytes zero-padded within the declared width.
func TestCollectDataFloatAndStringValues(t *testing.T) {
src := `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-8
RET
GLOBL vals<>(SB), RODATA, $44
DATA vals<>+0(SB)/8, $0.5
DATA vals<>+8(SB)/8, $-1.0
DATA vals<>+16(SB)/4, $1.5
DATA vals<>+20(SB)/16, $"call frame too "
DATA vals<>+36(SB)/4, $"hi"
`
f, errs := parser.Parse("fvals_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
byName := map[string]DataSymbol{}
for _, d := range img.DataSyms {
byName[d.Name] = d
}
d := byName["vals"]
if d.Size != 44 {
t.Fatalf("vals size = %d, want 44", d.Size)
}
buf := img.Data[d.Offset : d.Offset+44]
// 0.5 = 0x3FE0000000000000, -1.0 = 0xBFF0000000000000 (float64);
// 1.5 = 0x3FC00000 (float32).
for _, c := range []struct {
off int
want []byte
}{
{0, []byte{0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xE0, 0x3F}},
{8, []byte{0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xF0, 0xBF}},
{16, []byte{0x00, 0x00, 0xC0, 0x3F}},
{20, []byte("call frame too ")},
{36, []byte{'h', 'i', 0x00, 0x00}},
} {
if string(buf[c.off:c.off+len(c.want)]) != string(c.want) {
t.Errorf("vals+%d: got % x, want % x", c.off, buf[c.off:c.off+len(c.want)], c.want)
}
}
}
// TestCollectDataValueErrors pins the value-kind width rules: a float needs
// width 4 or 8, a string must fit its declared width, and a bad float
// literal is diagnosed rather than stored.
func TestCollectDataValueErrors(t *testing.T) {
cases := []string{
`GLOBL v<>(SB), RODATA, $4
DATA v<>+0(SB)/1, $0.5`,
`GLOBL v<>(SB), RODATA, $2
DATA v<>+0(SB)/2, $"toolarge"`,
}
for i, src := range cases {
full := "#include \"textflag.h\"\nTEXT ·Keep(SB), NOSPLIT, $0-8\n\tRET\n" + src
f, errs := parser.Parse(fmt.Sprintf("verr%d_amd64.s", i), full)
if len(errs) > 0 {
t.Fatalf("case %d parse: %v", i, errs)
}
if _, err := AssembleFile(f); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
+258 -7
View File
@@ -9,7 +9,7 @@ import (
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into
@@ -371,6 +371,13 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
if mnem == "RET" {
return len(loong64Return(fi))
}
// BYTE lays down one raw byte per operand, a front-end pseudo-op the
// toolchain spells only on x86 but accepts here the same way the arm64
// and riscv64 encoders do (a superset spelling, shippable via the goobj
// path).
if mnem == "BYTE" {
return len(ops)
}
switch mnem {
case "END", "FUNCDATA", "PCDATA":
return 0 // bookkeeping statements contribute no bytes
@@ -484,6 +491,18 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "BYTE":
// BYTE $b lays down one raw byte per operand, the same front-end
// pseudo-op the arm64 and riscv64 encoders accept.
var out []byte
for _, op := range ops {
b := l64Imm64(op)
if b < 0 || b > 0xFF {
return nil, fmt.Errorf("BYTE: immediate %d does not fit a byte", b)
}
out = append(out, byte(b))
}
return out, nil
case "END", "FUNCDATA", "PCDATA", "GETCALLERPC":
// The assembler's bookkeeping statements. END, FUNCDATA and PCDATA
// contribute no bytes, the same shapes GOARCH=loong64 go tool asm
@@ -701,6 +720,35 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
}
return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Fllsc:
// LLACQ{W,V} (Rj), Rd loads and SCREL{W,V} Rd, (Rj) stores, both
// 2R encodings op | rj<<5 | rd against a zero-offset memory operand
// (the toolchain's C_ZOREG, which rejects any displacement).
rd, rj, off, _, err := l64MemOperands(ops, fi)
if err != nil {
return nil, fmt.Errorf("%s: %w", mnem, err)
}
if off != 0 {
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
}
return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Fscq:
// SCQ first, middle, (base): op | middle<<10 | base<<5 | first,
// against a zero-offset memory operand as with the LL/SC pair.
if len(ops) != 3 || !isMemOperand(ops[2]) || isMemOperand(ops[0]) || isMemOperand(ops[1]) {
return nil, fmt.Errorf("%s expects reg, reg, (reg)", mnem)
}
first, middle := l64Reg(ops[0]), l64Reg(ops[1])
rj, off := l64MemWithFrame(ops[2], fi)
if first < 0 || middle < 0 || rj < 0 {
return nil, fmt.Errorf("%s: invalid register operand", mnem)
}
if off != 0 {
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
}
return l64wordLE(l64rrr(enc.op, middle, rj, first)), nil
case l64Firr:
// LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd.
if len(ops) < 2 || !isImmOperand(ops[0]) {
@@ -1229,6 +1277,31 @@ func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off in
return rd, rj, off, load, nil
}
// l64ImmMem reads the `$off(rj)` immediate form off an operand's raw text:
// the shared immediate parse reduces it to the bare number and keeps only
// the text as a witness of the base register. ok reports the form was
// found, with the base's register number (or -1 when the name is not a
// general register).
func l64ImmMem(op *ast.Operand) (off int32, base int, ok bool) {
if op.Kind != ast.OpImmediate || !op.Imm.HasVal {
return 0, 0, false
}
raw := strings.ReplaceAll(op.Raw, " ", "")
if !strings.HasPrefix(raw, "$") || !strings.HasSuffix(raw, ")") {
return 0, 0, false
}
open := strings.LastIndexByte(raw, '(')
if open < 2 {
return 0, 0, false
}
base = loong64RegNum(raw[open+1 : len(raw)-1])
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return int32(v), base, base >= 0
}
// ---- the MOV pseudo-instruction ----
// encodeLOONG64Mov encodes the MOV family, the load/store/immediate
@@ -1262,6 +1335,29 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs
}
return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil
}
// MOVx $off(rj), rd computes an address: the toolchain's `mov
// $soreg, r` case, a plain addi.d whatever the move's width (both
// MOVW and MOVV $4(R4), R5 encode the same addi.d in its testdata).
// A wider offset materialises in R30 first (lu12i.w + ori + add.d,
// its case 10). The immediate's Raw carries the base register,
// which the shared immediate parse reduces to the bare number.
if off, base, ok := l64ImmMem(src); ok {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $imm(rj): invalid destination register", mnem)
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return nil, fmt.Errorf("%s $imm(rj): illegal combination with an F register destination", mnem)
}
if off >= -2048 && off <= 2047 {
return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(off), base, rd)), nil
}
return l64WordsLE(
l64ir(l64InstrTable["LU12IW"].op, int(off)>>12, 30),
l64irr(l64DualTable["OR"].imm, int(off)&0xFFF, 30, 30),
l64rrr(l64DualTable["ADDV"].rrr, 30, base, rd),
), nil
}
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
@@ -1356,6 +1452,14 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
return 8 // pcalau12i + addi.d
}
// The $off(rj) address immediate: addi.d in the 12-bit window,
// lu12i.w + ori + add.d beyond it (the toolchain's case 10).
if off, _, ok := l64ImmMem(src); ok {
if off >= -2048 && off <= 2047 {
return 4
}
return 12
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid)
}
@@ -1868,6 +1972,19 @@ func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
return l64Mem(op)
}
// l64VmovqMem resolves a VMOVQ/XVMOVQ memory operand. The toolchain's
// vector table falls back to the zero register as the FP-relative base
// (`VMOVQ V2, y+16(FP)` stores through R0 while MOVW reads the same operand
// through R3), so the vector moves keep the resolved offset but the zero
// base, exactly as `go tool asm` emits them.
func l64VmovqMem(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
rj, off = l64MemWithFrame(op, fi)
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" {
rj = 0
}
return rj, off
}
// l64MemOffset returns the resolved byte offset of a memory operand.
func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 {
_, off := l64MemWithFrame(op, fi)
@@ -1892,7 +2009,7 @@ func l64Label(op *ast.Operand) string {
type l64VecOperand struct {
num int // 5-bit register number
lasx bool // X bank (LASX) rather than V (LSX)
width byte // suffix width letter (B/H/W/V), 0 on a bare register
width byte // suffix width letter (B/H/W/V/Q), 0 on a bare register
lanes int // lane count of a width suffix (B16 → 16)
elem int // element index of a .T[i] suffix
hasEl bool // the suffix names an element (.T[i])
@@ -1932,7 +2049,7 @@ func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) {
}
i++
w := name[i]
if w != 'B' && w != 'H' && w != 'W' && w != 'V' {
if w != 'B' && w != 'H' && w != 'W' && w != 'V' && w != 'Q' {
return v, false
}
v.width, v.hasSuf = w, true
@@ -2174,6 +2291,10 @@ func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]
// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register)
// VMOVQ vj.T[i], rd vpickve2gr (extract one element)
// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element)
// VMOVQ vj.T[i], vd.T vreplvei (broadcast one element, LSX)
// XVMOVQ xj, xd.T xvreplve0 (broadcast element zero, LASX)
// XVMOVQ xj, xd.T[i] xvinsve0 (insert element zero, LASX)
// XVMOVQ xj.T[i], xd xvpickve (extract one element, LASX)
func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) {
enc := l64VmovqTable[lasx]
bank := "V"
@@ -2200,6 +2321,118 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
return loong64RegNum(name), nil
}
// Element broadcast: VMOVQ vj.T[i], vd.T (vreplvei.{b,h,w,d}), the
// source element width matching the destination arrangement. An LSX-only
// form: the toolchain's table gives vreplvei no LASX counterpart.
if srcVec && dstVec && src.hasEl && dst.hasSuf && !dst.hasEl {
if lasx || src.lasx || dst.lasx {
return nil, fmt.Errorf("VMOVQ: vreplvei has no %s-bank form", bank)
}
if src.unsig {
return nil, fmt.Errorf("VMOVQ: vreplvei takes no unsigned element suffix")
}
if src.width != dst.width {
return nil, fmt.Errorf("VMOVQ: element width does not match arrangement %q", ops[1].Raw)
}
if _, ok := l64VecSuffixWidth(false, dst); !ok {
return nil, fmt.Errorf("VMOVQ: invalid arrangement %q", ops[1].Raw)
}
var op uint32
limit := 0
switch src.width {
case 'B':
op, limit = enc.rveiB, 15
case 'H':
op, limit = enc.rveiH, 7
case 'W':
op, limit = enc.rveiW, 3
default:
op, limit = enc.rveiD, 1
}
if src.elem > limit {
return nil, fmt.Errorf("VMOVQ: element index %d out of range [0, %d]", src.elem, limit)
}
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Broadcast of element zero: XVMOVQ xj, xd.T (xvreplve0.{b,h,w,d,q}),
// a bare X source into an arranged X destination. LASX only.
if srcVec && dstVec && !src.hasSuf && dst.hasSuf && !dst.hasEl {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvreplve0 is the %s-bank form alone", bank)
}
var op uint32
switch dst.width {
case 'B':
if dst.lanes != 32 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0B
case 'H':
if dst.lanes != 16 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0H
case 'W':
if dst.lanes != 8 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0W
case 'V':
if dst.lanes != 4 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0D
case 'Q':
if dst.lanes != 2 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0Q
default:
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
return l64wordLE(op | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Insert of element zero: XVMOVQ xj, xd.T[i] (xvinsve0.{w,d}), a bare X
// source into one word or double-word lane. LASX only.
if srcVec && dstVec && !src.hasSuf && dst.hasEl {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvinsve0 is the %s-bank form alone", bank)
}
op, limit := enc.xinsW, 7
if dst.width != 'W' {
op, limit = enc.xinsD, 3
if dst.width != 'V' {
return nil, fmt.Errorf("XVMOVQ: xvinsve0 takes word or double-word lanes, got %q", ops[1].Raw)
}
}
if dst.elem > limit {
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", dst.elem, limit)
}
return l64wordLE(op | uint32(dst.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Element extract into a vector register: XVMOVQ xj.T[i], xd
// (xvpickve.{w,d}), one word or double-word lane out to a bare X
// register. LASX only.
if srcVec && src.hasEl && dstVec && !dst.hasSuf {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvpickve is the %s-bank form alone", bank)
}
op, limit := enc.xpickW, 7
if src.width != 'W' {
op, limit = enc.xpickD, 3
if src.width != 'V' {
return nil, fmt.Errorf("XVMOVQ: xvpickve takes word or double-word lanes, got %q", ops[0].Raw)
}
}
if src.elem > limit {
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", src.elem, limit)
}
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant),
// both operands bare registers of the same bank.
if srcVec && dstVec {
@@ -2224,7 +2457,7 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
}
return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil
}
rj, off := l64MemWithFrame(ops[1], fi)
rj, off := l64VmovqMem(ops[1], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]")
}
@@ -2247,9 +2480,9 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
}
return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil
}
rj, off := l64MemWithFrame(ops[0], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
rj, off := l64VmovqMem(ops[0], fi)
if rj < 0 {
return nil, fmt.Errorf("VMOVQ: invalid load operand")
}
op := enc.ld
if dst.hasSuf {
@@ -2257,16 +2490,34 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw)
}
// vldrepl keeps the byte offset raw for bytes and scales it by
// the element width for the wider forms, the immediate field
// shrinking a bit per scale exactly as the toolchain encodes it
// (the field mask keeps the two's complement inside its width).
scale, mask, lo, hi := 1, int32(0xFFF), -2048, 2047
switch w {
case 0:
op = enc.replB
case 1:
op = enc.replH
scale, mask, lo, hi = 2, 0x7FF, -1024, 1023
case 2:
op = enc.replW
scale, mask, lo, hi = 4, 0x3FF, -512, 511
default:
op = enc.replD
scale, mask, lo, hi = 8, 0x1FF, -256, 255
}
if off%int32(scale) != 0 {
return nil, fmt.Errorf("VMOVQ: offset %d must be a multiple of %d", off, scale)
}
off /= int32(scale)
if off < int32(lo) || off > int32(hi) {
return nil, fmt.Errorf("VMOVQ: offset out of range [%d, %d]", lo*scale, hi*scale)
}
off &= mask
} else if off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
}
return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil
}
+32 -5
View File
@@ -279,6 +279,8 @@ const (
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd
l64Fllsc // acquire/release LL/SC (2R against a zero-offset memory operand)
l64Fscq // sc.q: op | middle<<10 | base<<5 | first against a zero-offset memory operand
)
// l64Enc is one instruction's encoding: its bit layout (format) and the
@@ -341,8 +343,9 @@ var l64Vec2R = map[string]bool{}
// such as vshuf.b).
var l64Vec4R = map[string]bool{}
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels.
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants, each pre-shifted to
// its exact bit range, read off `go tool objdump` of GOARCH=loong64
// `go tool asm` kernels and the toolchain's specialLsxMovInst table.
type l64VmovqEnc struct {
ld, st, ldx, stx uint32 // plain and indexed load/store
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
@@ -350,6 +353,11 @@ type l64VmovqEnc struct {
ins uint32 // vinsgr2vr element insert
dup uint32 // vreplgr2vr duplicate (width in [11:10])
move uint32 // vori.b/xvori.b $0 register move
rveiB, rveiH, rveiW, rveiD uint32 // vreplvei: broadcast one element (LSX)
rve0B, rve0H, rve0W uint32 // xvreplve0 broadcast of element zero (LASX)
rve0D, rve0Q uint32 // xvreplve0.{d,q}, ditto
xinsW, xinsD uint32 // xvinsve0: insert element zero (LASX)
xpickW, xpickD uint32 // xvpickve: extract element (LASX)
}
var l64VmovqTable = map[bool]l64VmovqEnc{
@@ -358,12 +366,17 @@ var l64VmovqTable = map[bool]l64VmovqEnc{
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
rveiB: 0x01CBDE << 14, rveiH: 0x0397BE << 13, rveiW: 0x072F7E << 12, rveiD: 0x0E5EFE << 11,
},
true: { // XVMOVQ, the LASX (X) bank
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
rve0B: 0x1DC1C0 << 10, rve0H: 0x1DC1E0 << 10, rve0W: 0x1DC1F0 << 10,
rve0D: 0x1DC1F8 << 10, rve0Q: 0x1DC1FC << 10,
xinsW: 0x03B7FE << 13, xinsD: 0x076FFE << 12,
xpickW: 0x03B81E << 13, xpickD: 0x07703E << 12,
},
}
@@ -373,7 +386,7 @@ func init() {
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
"SGT": 0x24 << 15, "SGTU": 0x25 << 15,
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15,
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15,
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
"ORN": 0x2c << 15, "ANDN": 0x2d << 15,
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
@@ -467,6 +480,20 @@ func init() {
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
// Acquire/release LL/SC (2R against a zero-offset memory operand):
// LLACQV (Rj), Rd loads, SCRELV Rd, (Rj) stores, both encoding
// op | rj<<5 | rd. Opcodes from cmd/internal/obj/loong64/instOp.go
// (ll.acq.{w,d}, sc.rel.{w,d}).
l64InstrTable["LLACQW"] = l64Enc{format: l64Fllsc, op: 0x0E15E0 << 10}
l64InstrTable["SCRELW"] = l64Enc{format: l64Fllsc, op: 0x0E15E1 << 10}
l64InstrTable["LLACQV"] = l64Enc{format: l64Fllsc, op: 0x0E15E2 << 10}
l64InstrTable["SCRELV"] = l64Enc{format: l64Fllsc, op: 0x0E15E3 << 10}
// SCQ (sc.q first, middle, (base)) keeps its own operand order: the
// encoding is op | middle<<10 | base<<5 | first, the memory operand's
// base in the rj field, not the toolchain's generic 3R layout.
l64InstrTable["SCQ"] = l64Enc{format: l64Fscq, op: 0x070AE << 15}
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
// selected by the operand kind; the shift mnemonics pair the 3R form
// with a 5/6-bit shift immediate.
@@ -868,7 +895,7 @@ func init() {
"VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF},
"XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF},
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F},
"XVSEQB": {0xED00 << 15, true, -16, 15, 0, 0x1F},
// vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a
// 7-bit field, but the toolchain range-checks it down to si5 as well
// (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64).
@@ -877,7 +904,7 @@ func init() {
"VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F},
"XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F},
"VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F},
"XVSEQV": {0xE903 << 15, true, -16, 15, 0, 0x7F},
"XVSEQV": {0xED03 << 15, true, -16, 15, 0, 0x7F},
// vslti compares against a signed (or, in the U spellings, unsigned)
// si5/ui5 constant.
"VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F},
+229 -2
View File
@@ -8,8 +8,8 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// firstTextLOONG64 parses assembly source and returns the first TEXT body.
@@ -831,3 +831,230 @@ TEXT ·atoms(SB), NOSPLIT, $0
0x4C000020,
)
}
// TestLOONG64_llacqScrel pins the acquire/release LL/SC pair. The oracle
// words come from GOARCH=loong64 go tool objdump and the toolchain's own
// loong64enc1.s golden bytes.
func TestLOONG64_llacqScrel(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·llsc(SB), NOSPLIT, $0
LLACQW (R5), R4
LLACQV (R5), R4
SCRELW R4, (R6)
SCRELV R4, (R6)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x385780A4, // ll.acq.w r4, r5
0x385788A4, // ll.acq.d r4, r5
0x385784C4, // sc.rel.w r4, r6
0x38578CC4, // sc.rel.d r4, r6
0x4C000020,
)
// The toolchain accepts the zero-offset memory form alone.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
LLACQW 4(R5), R4
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
SCRELV R4, 8(R6)
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_vmovqSuffixed pins the element-broadcast and element-move
// VMOVQ/XVMOVQ forms, with the oracle words lifted verbatim from the
// toolchain's loong64enc1.s.
func TestLOONG64_vmovqSuffixed(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·vmovq(SB), NOSPLIT, $0
VMOVQ V1.B[3], V9.B16
VMOVQ V2.H[2], V8.H8
VMOVQ V3.W[1], V7.W4
VMOVQ V4.V[0], V6.V2
XVMOVQ X0, X31.B32
XVMOVQ X1, X30.H16
XVMOVQ X2, X29.W8
XVMOVQ X3, X28.V4
XVMOVQ X3, X27.Q2
XVMOVQ X0, X31.W[7]
XVMOVQ X1, X29.W[0]
XVMOVQ X3, X28.V[3]
XVMOVQ X4, X27.V[0]
XVMOVQ X31.W[7], X0
XVMOVQ X29.W[0], X1
XVMOVQ X28.V[3], X8
XVMOVQ X27.V[0], X9
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x72F78C29, // vreplvei.b v9, v1, 3
0x72F7C848, // vreplvei.h v8, v2, 2
0x72F7E467, // vreplvei.w v7, v3, 1
0x72F7F086, // vreplvei.d v6, v4, 0
0x7707001F, // xvreplve0.b x31, x0
0x7707803E, // xvreplve0.h x30, x1
0x7707C05D, // xvreplve0.w x29, x2
0x7707E07C, // xvreplve0.d x28, x3
0x7707F07B, // xvreplve0.q x27, x3
0x76FFDC1F, // xvinsve0.w x31, x0, 7
0x76FFC03D, // xvinsve0.w x29, x1, 0
0x76FFEC7C, // xvinsve0.d x28, x3, 3
0x76FFE09B, // xvinsve0.d x27, x4, 0
0x7703DFE0, // xvpickve.w x0, x31, 7
0x7703C3A1, // xvpickve.w x1, x29, 0
0x7703EF88, // xvpickve.d x8, x28, 3
0x7703E369, // xvpickve.d x9, x27, 0
0x4C000020,
)
// The rejected shapes: a width mismatch between the element and the
// arrangement, an element index past the lane count, a wrong-bank
// vreplvei and an arrangement the LASX bank does not spell.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ V1.H[3], V9.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ V1.B[16], V9.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X1.B[3], X9.B32
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X0, X31.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X0, X31.W[8]
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_parityFixes pins the operand forms whose encodings were found
// diverging from the toolchain by the loong64enc1.s differential: the
// $off(reg) address immediate (addi.d), the SCQ operand order, the scaled
// vldrepl offsets (with their field masks), the XVSEQB/XVSEQV immediate
// opcodes and the zero-register base the toolchain gives FP-relative
// VMOVQ/XVMOVQ memory operands. Golden words from loong64enc1.s.
func TestLOONG64_parityFixes(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·parity(SB), NOSPLIT, $0-32
MOVW $4(R4), R5
MOVV $4(R4), R5
MOVW $65536(R4), R5
MOVW $-4096(R4), R5
SCQ R4, R5, (R6)
VMOVQ 2(R4), V1.H8
VMOVQ -6(R4), V1.H8
VMOVQ -12(R4), V2.W4
VMOVQ -16(R4), V3.V2
XVMOVQ -10(R4), X1.H16
XVSEQB $0, X2, X4
XVSEQH $3, X2, X4
XVSEQW $12, X2, X4
XVSEQV $15, X2, X4
XVSEQV $-15, X2, X4
VMOVQ V2, y+16(FP)
VMOVQ y+16(FP), V2
VMOVQ V2, x+2030(FP)
XVMOVQ X6, y+16(FP)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x02C01085, // addi.d $4, r4, r5
0x02C01085, // addi.d $4, r4, r5 (MOVW keeps the 64-bit addi.d)
0x1400021E, // lu12i.w $16, r30
0x038003DE, // ori $0, r30, r30
0x0010F885, // add.d r5, r4, r30
0x15FFFFFE, // lu12i.w $-1, r30
0x038003DE, // ori $0, r30, r30
0x0010F885, // add.d r5, r4, r30
0x385714C4, // sc.q r4, r5, (r6): middle<<10 | base<<5 | first
0x30400481, // vldrepl.h v1, 2(r4)
0x305FF481, // vldrepl.h v1, -6(r4)
0x302FF482, // vldrepl.w v2, -12(r4)
0x3017F883, // vldrepl.d v3, -16(r4)
0x325FEC81, // xvldrepl.h x1, -10(r4)
0x76800044, // xvseqi.b x4, x2, 0
0x76808C44, // xvseqi.h x4, x2, 3
0x76813044, // xvseqi.w x4, x2, 12
0x7681BC44, // xvseqi.d x4, x2, 15
0x7681C444, // xvseqi.d x4, x2, -15
0x2C406002, // vst v2, 24(r0): FP-relative keeps the zero base
0x2C006002, // vld v2, 24(r0)
0x2C5FD802, // vst v2, 2038(r0)
0x2CC06006, // xvst x6, 24(r0)
0x4C000020,
)
// Misaligned vldrepl offsets are rejected, as the toolchain does.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ 3(R4), V1.H8
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
MOVW $4(R4), F1
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_bytePseudo pins the BYTE literal-data pseudo-op, which the
// loong64 toolchain does not spell but the arm64 and riscv64 encoders of
// this package already accept for byte-exact data layout (a superset
// spelling, shippable via the goobj path).
func TestLOONG64_bytePseudo(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·bytes(SB), NOSPLIT, $0
BYTE $2
BYTE $1; BYTE $0
BYTE $255
RET
`)
code := assembleLOONG64Helper(t, fn)
// Four literal bytes, then RET (jirl r0, r1, 0); the trailing bytes pad
// the final word the way any sub-word tail does.
want := []byte{2, 1, 0, 0xFF, 0x20, 0x00, 0x00, 0x4C}
if !bytes.Equal(code[:len(want)], want) {
t.Errorf("bytes = % x, want % x", code, want)
}
for _, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
BYTE $256
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
BYTE $-1
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("%q: expected an error, got none", src)
}
}
}
+1 -1
View File
@@ -6,7 +6,7 @@ package asm
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// Loong64 frame mapping, matching the Go toolchain's loong64 backend.
+1 -1
View File
@@ -7,7 +7,7 @@ import (
"bytes"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestLOONG64_sys exercises the no-operand system instructions and the
+1 -1
View File
@@ -6,7 +6,7 @@ package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestLOONG64RelocOffsetsIncludePrologue pins the function-relative
+46
View File
@@ -14,6 +14,51 @@ type Imm int64
func (Imm) isOperand() {}
// RegList is a bracketed register range, [Z0-Z3]: the four-register source
// of the 4FMAPS and 4VNNIW families. The EVEX emit path carries the list's
// low register through the inverted 5-bit V'VVVV field; the three higher
// registers are implied by the instruction, so only the pair travels here.
type RegList struct {
Lo Reg
Hi Reg // implied by the encoding; Lo.idx+3 by construction
}
func (RegList) isOperand() {}
// FloatImm is a floating-point immediate ($-1.0). The SSE mnemonics whose
// encoding takes an XMM/memory source at that position rewrite it as a read
// from a read-only pool constant ($f64.<hex> or $f32.<hex>), the toolchain's
// own behaviour; every other instruction rejects it.
type FloatImm struct {
Text string // the numeric text as written, sign excluded
Neg bool // a leading minus
}
func (FloatImm) isOperand() {}
// TLSMem is a thread-local access, the source form off(base)(TLS*1) with the
// base dropped: the toolchain's one-instruction TLS rewrite assembles it as
// the segment-prefixed absolute whose disp32 carries an R_TLS_LE patch site
// (the linker fills the TLS slot offset).
type TLSMem struct {
Disp int64
Size int
Seg byte // the segment override: FS (0x64) or GS (0x65) on windows
}
func (TLSMem) isOperand() {}
// SegAbs is a segment-absolute access, 0x30(GS): the segment override
// prefixes a disp32 absolute reference with no relocation. The base
// register spellings GS and FS produce it.
type SegAbs struct {
Disp int64
Size int
Seg byte // 0x64 FS, 0x65 GS
}
func (SegAbs) isOperand() {}
// Mem is a memory operand of the form disp(base)(index*scale).
type Mem struct {
Base Reg
@@ -23,6 +68,7 @@ type Mem struct {
Size int // operand width in bytes
HasBase bool
HasIndex bool
Seg byte // segment override prefix (0x64 FS, 0x65 GS); 0 = none
}
func (Mem) isOperand() {}
+30 -1
View File
@@ -17,13 +17,28 @@ import "strings"
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0-K7, the fp flag the x87 stack registers F0-F7.
// registers K0-K7, the fp flag the x87 stack registers F0-F7, the mmx flag the
// MMX registers M0-M7, the seg field a bare segment register (FS, GS) and the
// ctl field the control and debug registers, whose number rides an
// instruction's reg field rather than r/m.
type Reg struct {
idx int
size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH
mask bool // K0-K7 opmask register
fp bool // F0-F7 x87 stack register
mmx bool // M0-M7 MMX register
seg int // segment register number plus one (ES=1..GS=6); 0 = not one
ctl byte // 0 none, 1 CRn control register, 2 DRn debug register
}
// segNumber returns the segment register number (ES=0..GS=5) when r names a
// bare segment register.
func (r Reg) segNumber() (int, bool) {
if r.seg == 0 {
return 0, false
}
return r.seg - 1, true
}
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
@@ -149,6 +164,20 @@ func buildRegByName() map[string]Reg {
for i := 0; i <= 7; i++ {
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
}
// MMX: M0..M7.
for i := 0; i <= 7; i++ {
m["M"+itoa(i)] = Reg{idx: i, size: 8, mmx: true}
}
// Bare segment registers: ES, CS, SS, DS, FS, GS (the memory-base and
// index spellings of FS and GS are handled before register lookup).
for i, n := range []string{"ES", "CS", "SS", "DS", "FS", "GS"} {
m[n] = Reg{idx: i, size: 2, seg: i + 1}
}
// Control and debug registers: CR0..CR15, DR0..DR15.
for i := 0; i <= 15; i++ {
m["CR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 1}
m["DR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 2}
}
return m
}
+287 -36
View File
@@ -6,21 +6,23 @@ package asm
import (
"errors"
"fmt"
"math/bits"
"slices"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// assembleRISCV assembles a RISC-V TEXT function body into machine code.
// It handles the full RV64IMAFDC instruction set including RVC compression.
func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, []RiscvLiteral, error) {
fi := riscvComputeFrame(t)
prologue := riscvPrologue(fi)
guardLen, err := riscvGuardLen(fi)
if err != nil {
return nil, nil, nil, nil, nil, err
return nil, nil, nil, nil, nil, nil, err
}
lits := &riscvLiterals{}
var relocs []Reloc
var spadj []SpadjStep
@@ -74,9 +76,9 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
pc := len(prologue)
for i := range recs {
branchLike := isBranchLike(recs[i].instr.Mnemonic.Text) || riscvIsCondBranch(recs[i].instr.Mnemonic.Text)
code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil, nil) // no relocs in Pass 2
code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil, nil, lits) // no relocs in Pass 2
if err != nil && !(branchLike && riscvIsRangeError(err)) {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err)
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err)
}
if err != nil {
code = make([]byte, 4)
@@ -178,13 +180,20 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
}
}
if !changed {
// Capture the final pcs for the N(PC) branch forms: their target
// is the instruction N source slots away, resolved by index.
// Capture the final pcs for the N(PC) branch and jump forms: the
// target is the instruction N source slots away (N=0 the branch
// itself, N negative backwards), resolved by index against the
// final layout.
pcRelPcs = map[*ast.Instr]int{}
for i := range recs {
if _, ok := riscvPCRelOffset(recs[i].instr); ok {
pcRelPcs[recs[i].instr] = pcs[i]
n, ok := riscvPCRelOffset(recs[i].instr)
if !ok {
continue
}
if i+n < 0 || i+n >= len(recs) {
continue
}
pcRelPcs[recs[i].instr] = pcs[i+n]
}
break
}
@@ -198,7 +207,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
var out []byte
guardBytes, guardReloc, err := riscvGuard(fi)
if err != nil {
return nil, nil, nil, nil, nil, err
return nil, nil, nil, nil, nil, nil, err
}
if fi.needSplit {
out = append(out, guardBytes...)
@@ -220,11 +229,11 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
// The JMP a relaxation inserted: JAL X0 to the original target.
targetOff, ok := offsets[r.jmpTo]
if !ok {
return nil, nil, nil, nil, nil, fmt.Errorf("undefined label %q", r.jmpTo)
return nil, nil, nil, nil, nil, nil, fmt.Errorf("undefined label %q", r.jmpTo)
}
offset := int32(targetOff - pc)
if err := riscvCheckJumpOffset(r.jmpTo, offset); err != nil {
return nil, nil, nil, nil, nil, err
return nil, nil, nil, nil, nil, nil, err
}
word := riscvJType(0, offset)
code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
@@ -233,7 +242,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
// JMP, always the very next instruction (offset 4).
enc, rs1, rs2, ok := riscvInvertedBranchEnc(strings.ToUpper(r.instr.Mnemonic.Text), r.instr.Operands)
if !ok {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: cannot relax branch", r.instr.Mnemonic.Text)
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: cannot relax branch", r.instr.Mnemonic.Text)
}
word := riscvBType(enc, rs1, rs2, 4)
code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
@@ -241,9 +250,9 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
code = r.code
default:
var err error
code, err = encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs, pcRelPcs)
code, err = encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs, pcRelPcs, lits)
if err != nil {
return nil, nil, nil, nil, nil, err
return nil, nil, nil, nil, nil, nil, err
}
if c16, ok := tryCompressRVC(r.instr, fi); ok {
code = []byte{byte(c16), byte(c16 >> 8)}
@@ -270,7 +279,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
if fi.needSplit {
relocs = append(relocs, guardReloc)
}
return out, offsets, relocs, lines, spadj, nil
return out, offsets, relocs, lines, spadj, lits.list(), nil
}
// riscvImmAlias maps the R-type ALU mnemonics onto their I-type immediate
@@ -348,6 +357,11 @@ func riscvPadBytes(pad int) []byte {
func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
mnem := instr.Mnemonic.Text
ops := instr.Operands
mnem = riscvNormalisePseudo(mnem)
if mnem == "FUNCDATA" || mnem == "PCDATA" {
// The bookkeeping statements contribute no bytes.
return 0
}
var immNeg bool
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
if mnem == "RET" {
@@ -368,7 +382,25 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
}
// MOV $imm, rd → size depends on the immediate and RVC compression.
if isImmOperand(ops[0]) && ops[0].Imm.Sym == nil {
return riscvMovImmSize(regFromOperand(ops[1]), immFromOperand(ops[0]))
imm := riscvOperandImm64(ops[0])
if int64(int32(imm)) != imm {
return riscvMovImm64Size(regFromOperand(ops[1]), imm)
}
return riscvMovImmSize(regFromOperand(ops[1]), int32(imm))
}
// MOV $sym+off(FP|SP), rd → the frame-adjusted offset as an ADDI,
// compressed like riscvSPAddiBytes encodes it.
if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil &&
(ops[0].Imm.Sym.Pseudo == "FP" || ops[0].Imm.Sym.Pseudo == "SP") {
rd := regFromOperand(ops[1])
_, off := riscvResolvePseudo(ops[0].Imm.Sym, fi)
if rd > 0 && off == 0 {
return 2 // C.MV rd, SP
}
if isRVCIntReg(rd) && off > 0 && off < 1024 && off%4 == 0 {
return 2 // C.ADDI4SPN
}
return riscvItypeImmediateSize("ADDI", off)
}
// Frame-relative loads and stores: a frame offset beyond the signed
// 12-bit range materialises the address in X31 first.
@@ -664,9 +696,10 @@ func riscvCheckJumpOffset(target string, off int32) error {
}
// encodeRISCVInstr encodes a single RISC-V instruction.
func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int, lits *riscvLiterals) ([]byte, error) {
mnem := instr.Mnemonic.Text
ops := instr.Operands
mnem = riscvNormalisePseudo(mnem)
var immNeg bool
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
var word uint32
@@ -677,6 +710,22 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
// RET = epilogue (restore LR and close the frame when present) +
// uncompressed JALR X0, 0(X1) (the toolchain never compresses RET).
return riscvReturn(fi), nil
case "FUNCDATA":
// The assembler's bookkeeping statement, the expanded form of the
// GO_ARGS and NO_LOCAL_POINTERS macros: FUNCDATA $n, sym(SB)
// contributes no bytes, exactly as the toolchain's listing shows
// (the FUNCDATA entries and the instruction after them share a PC).
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)")
}
return nil, nil
case "PCDATA":
// The other bookkeeping statement, the expanded form of
// GO_RESULTS_INITIALIZED: PCDATA $n, $m contributes no bytes too.
if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
return nil, fmt.Errorf("PCDATA expects $n, $m")
}
return nil, nil
case "WORD":
// WORD $w lays down a raw 32-bit little-endian word.
if len(ops) != 1 {
@@ -739,6 +788,22 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
target = labelFromOperand(ops[0])
// JMP N(PC): the PC-relative slot form, resolved like the
// branches (the toolchain counts source instructions at a
// uniform 4 bytes, so JMP 0(PC) is a self-loop and JMP -3(PC)
// reaches twelve bytes back). It must be recognised before the
// indirect-register form, whose operand it resembles.
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
offset := int32(off - pc)
if err := riscvCheckJumpOffset("", offset); err != nil {
return nil, err
}
word = riscvJType(0, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
// JMP (X5): an indirect branch, the toolchain's JALR X0, 0(X5).
if ops[0].Addr.Sym == nil && ops[0].Addr.Base != "" {
if ops[0].Addr.Offset != 0 || ops[0].Addr.Index != "" {
@@ -751,17 +816,6 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
word = riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, rs1, 0)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
offset := int32(off - pc)
if err := riscvCheckJumpOffset("", offset); err != nil {
return nil, err
}
word = riscvJType(0, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
}
targetOff, ok := offsets[target]
if !ok {
@@ -810,7 +864,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
// (MOVB/MOVH/MOVW and unsigned forms) select the access width, and
// MOVD/MOVF address the FP registers.
case "MOV", "MOVB", "MOVBU", "MOVH", "MOVHU", "MOVW", "MOVWU", "MOVF", "MOVD":
return encodeRISCVMov(instr, fi, relocs)
return encodeRISCVMov(instr, fi, relocs, lits)
// JALR: indirect jump/call. Plan 9: JALR rs1, rd or JALR offset(rs1).
case "JALR":
@@ -1316,7 +1370,7 @@ func isImmOperand(op *ast.Operand) bool {
// - MOV Rs, (Rd) register-relative store
// - MOV Rs, Rd register-to-register move (ADDI $0)
// - MOV $imm, Rd load immediate (ADDI or LUI+ADDIW)
func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) {
func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc, lits *riscvLiterals) ([]byte, error) {
ops := instr.Operands
if len(ops) != 2 {
return nil, fmt.Errorf("MOV expects 2 operands, got %d", len(ops))
@@ -1335,8 +1389,21 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
}
return encodeRISCVSBAddr(src.Imm.Sym, rd, relocs), nil
}
// MOV $sym+off(FP|SP), rd: the address of a frame slot as an
// immediate is the frame-adjusted offset against the hardware SP,
// the toolchain's ADDI $adj, SP, rd (argframe+0(FP) in the runtime's
// reflect trampolines is the spelling).
if src.Imm.Sym != nil && (src.Imm.Sym.Pseudo == "FP" || src.Imm.Sym.Pseudo == "SP") {
rd := regFromOperand(dst)
if rd < 0 {
return nil, fmt.Errorf("MOV $%s(%s): invalid destination register", src.Imm.Sym.Name, src.Imm.Sym.Pseudo)
}
_, off := riscvResolvePseudo(src.Imm.Sym, fi)
return riscvSPAddiBytes(rd, off), nil
}
// MOV $sym(FP/SP), rd, not supported: immediate symbol references
// other than SB cannot be encoded as a simple immediate.
// other than the frame pseudos cannot be encoded as a simple
// immediate.
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo != "" {
return nil, fmt.Errorf("MOV $%s(%s): unsupported immediate symbol reference (only SB is supported)", src.Imm.Sym.Name, src.Imm.Sym.Pseudo)
}
@@ -1344,11 +1411,14 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
if rd < 0 {
return nil, fmt.Errorf("MOV $imm: invalid destination register")
}
imm, err := riscvImm32FromOperand(src, false)
if err != nil {
return nil, err
imm := riscvOperandImm64(src)
if int64(int32(imm)) != imm {
// Beyond the signed 32-bit span the toolchain either builds the
// value from a shifted 32-bit part or loads it from the pooled
// $i64 constant it synthesises for the purpose.
return riscvLoadImm64(rd, imm, lits, relocs), nil
}
return encodeRISCVLoadImm(rd, imm), nil
return encodeRISCVLoadImm(rd, int32(imm)), nil
}
// Memory → register (load).
@@ -1557,6 +1627,186 @@ func splitRISCV32Imm(imm int32) (low, high int32) {
return low, high
}
// riscvNormalisePseudo rewrites the toolchain's UNDEF spelling onto EBREAK:
// the assembler accepts UNDEF where the hardware wants the trap instruction
// and emits ebreak (compressed to C.EBREAK under RVC), so every pass sees the
// canonical name.
func riscvNormalisePseudo(mnem string) string {
if strings.EqualFold(mnem, "UNDEF") {
return "EBREAK"
}
return mnem
}
// riscvOperandImm64 reads an immediate operand as a full signed 64-bit value,
// where immFromOperand would truncate to int32; the MOV immediate path uses
// it to classify the wide constants.
func riscvOperandImm64(op *ast.Operand) int64 {
if !op.Imm.HasVal {
return 0
}
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return v
}
// riscvSplitShiftConst mirrors cmd/internal/obj/riscv's splitShiftConst: it
// looks for the signed 32-bit integer a constant can be rebuilt from with a
// left shift, a left-and-right shift pair (a run of ones), or a zero-extended
// 32-bit pattern. A constant that fits none of the shapes is materialised
// from the pooled $i64 data symbol instead.
func riscvSplitShiftConst(v int64) (imm int64, lsh int, rsh int, ok bool) {
// Rebuild from a signed 32-bit integer shifted left.
lsh = bits.TrailingZeros64(uint64(v))
c := v >> lsh
if int64(int32(c)) == c {
return c, lsh, 0, true
}
// Rebuild from a small negative constant: shift left into place, then
// shift the sign-extended ones run right.
rsh = bits.LeadingZeros64(uint64(v))
ones := bits.OnesCount64((uint64(v) >> lsh) >> 11)
if rsh+ones+lsh+11 == 64 {
c = (1<<11 | ((v >> lsh) & 0x7ff)) << 52 >> 52 // sign extend 12 bits
if lsh > 0 || c != -1 {
lsh += rsh
}
return c, lsh, rsh, true
}
// Rebuild from a zero-extended signed 32-bit integer.
if int64(uint32(c)) == c {
c = int64(int32(c))
lsh, rsh = 32, 32-lsh
return c, lsh, rsh, true
}
return 0, 0, 0, false
}
// riscvSPAddiBytes encodes ADDI rd, SP, imm for the frame-address immediates
// (the MOV $sym+off(FP|SP) form), using the compressed forms the toolchain
// picks under RVC: C.ADDI4SPN for a positive 4-byte multiple that fits, C.MV
// for the zero offset, the plain ADDI otherwise.
func riscvSPAddiBytes(rd int, imm int32) []byte {
if rd != 0 && imm == 0 {
return word16(rvcCR(0x8, uint32(rd), 2)) // C.MV rd, SP
}
if isRVCIntReg(rd) && imm > 0 && imm < 1024 && imm%4 == 0 {
return word16(rvcCIW(0x0, rvcReg3(rd), uint32(imm)))
}
return wordLE(riscvIType(riscvInstrTable["ADDI"], rd, 2, imm))
}
// riscvMovImm64Size returns the encoded byte length of MOV $imm, rd when the
// immediate sits outside the signed 32-bit span: the shifted-part sequences
// of riscvLoadImm64, or the 8-byte AUIPC+LD pool load.
func riscvMovImm64Size(rd int, imm int64) int {
c, lsh, rsh, ok := riscvSplitShiftConst(imm)
if !ok {
return 8 // AUIPC + LD against the $i64 pool symbol
}
size := riscvMovImmSize(rd, int32(c))
if lsh > 0 {
size += riscvShiftImmSize(rd, true)
}
if rsh > 0 {
size += riscvShiftImmSize(rd, false)
}
return size
}
// riscvShiftImmSize returns the encoded size of one SLLI/SRLI expansion
// part: two bytes under RVC when the destination can carry a compressed
// shift (C.SLLI admits every register but X0, C.SRLI only X8 to X15), four
// otherwise.
func riscvShiftImmSize(rd int, left bool) int {
if rd != 0 && (left || isRVCIntReg(rd)) {
return 2
}
return 4
}
// riscvLoadImm64 encodes MOV $imm, rd for an immediate beyond the signed
// 32-bit span, mirroring the toolchain's instructionsForMOVConst: when a
// shifted 32-bit part rebuilds the value it emits that part (compressed like
// any written MOV) followed by the SLLI and SRLI shifts; otherwise it loads
// the constant from the pooled read-only $i64.<hex> symbol via AUIPC + LD
// and registers the literal so the data section carries its bytes.
func riscvLoadImm64(rd int, imm int64, lits *riscvLiterals, relocs *[]Reloc) []byte {
c, lsh, rsh, ok := riscvSplitShiftConst(imm)
if !ok {
name := fmt.Sprintf("$i64.%016x", uint64(imm))
if lits != nil {
lits.add(name, riscvLiteralBytes(imm))
}
return encodeRISCVSBLoad(&ast.Symbol{Name: name}, rd, relocs)
}
out := encodeRISCVLoadImm(rd, int32(c))
if lsh > 0 {
out = append(out, riscvShiftImmBytes(rd, lsh, true)...)
}
if rsh > 0 {
out = append(out, riscvShiftImmBytes(rd, rsh, false)...)
}
return out
}
// riscvShiftImmBytes encodes one SLLI (left) or SRLI expansion part, using
// the compressed form the toolchain picks under RVC: C.SLLI admits every
// register but X0, C.SRLI only X8 to X15.
func riscvShiftImmBytes(rd, shamt int, left bool) []byte {
if rd != 0 && shamt >= 1 && shamt <= 63 && (left || isRVCIntReg(rd)) {
if left {
return word16(rvcSLLI(uint32(rd), uint32(shamt)&0x3F))
}
return word16(rvcCBShift(0x0, rvcReg3(rd), uint32(shamt)&0x3F))
}
enc := riscvEnc{0x13, 0x1, 0x00} // SLLI
imm := int32(shamt)
if !left {
enc = riscvEnc{0x13, 0x5, 0x00} // SRLI: funct6 000000, funct3 101
}
return wordLE(riscvIType(enc, rd, rd, imm))
}
// riscvLiteralBytes renders a 64-bit constant as the little-endian bytes the
// $i64 pool symbol holds.
func riscvLiteralBytes(v int64) []byte {
return []byte{byte(v), byte(v >> 8), byte(v >> 16), byte(v >> 24),
byte(v >> 32), byte(v >> 40), byte(v >> 48), byte(v >> 56)}
}
// RiscvLiteral is one pooled 64-bit constant: a MOV whose immediate sits
// beyond both the 32-bit span and the shift sequences loads its bits from a
// read-only data symbol named like the toolchain's $i64 pool.
type RiscvLiteral struct {
Name string
Data []byte
}
// riscvLiterals collects the pooled constants the MOV expansions refer to,
// deduplicated by name, in first-use order.
type riscvLiterals struct {
order []RiscvLiteral
seen map[string]bool
}
func (l *riscvLiterals) add(name string, data []byte) {
if l.seen == nil {
l.seen = map[string]bool{}
}
if !l.seen[name] {
l.seen[name] = true
l.order = append(l.order, RiscvLiteral{Name: name, Data: data})
}
}
func (l *riscvLiterals) list() []RiscvLiteral { return l.order }
// encodeRISCVItypeImmediate encodes an I-type arithmetic instruction, expanding
// large immediates for ADDI/ANDI/ORI/XORI into LUI+ADDIW+op (or two ADDIs for
// ADDI), matching the Go assembler.
@@ -1734,6 +1984,7 @@ func encodeRISCVJALR(instr *ast.Instr, fi riscvFrameInfo) ([]byte, error) {
// RVC form. It returns the compressed instruction word and true on success.
func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
mnem := riscvCompressMnem(instr)
mnem = riscvNormalisePseudo(mnem)
ops := instr.Operands
// The immediate aliases fold onto their I-type mnemonics before
// compression: the toolchain compresses ADD $imm, rd as c.addi, exactly
+1 -1
View File
@@ -63,7 +63,7 @@ func riscvRegNum(name string) int {
return 24
case "X25", "S9":
return 25
case "X26", "S10":
case "X26", "S10", "CTXT":
return 26
case "X27", "S11", "g":
return 27
+154 -21
View File
@@ -10,8 +10,8 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// firstTextRISCV parses assembly source and returns the first TEXT function body.
@@ -33,7 +33,7 @@ func firstTextRISCV(t *testing.T, src string) *ast.Text {
// assembleRISCVHelper assembles one TEXT function and returns its code bytes.
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
t.Helper()
code, _, _, _, _, err := assembleRISCV(fn)
code, _, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
@@ -785,16 +785,140 @@ TEXT ·sys(SB), NOSPLIT, $0
}
}
func TestRISCV_MOV_sym_FP_error(t *testing.T) {
// MOV $sym(FP), rd should return an error (unsupported).
func TestRISCV_MOV_sym_FP(t *testing.T) {
// MOV $sym(FP), rd lowers to the frame-adjusted ADDI against SP: the
// toolchain's argframe spelling. A zero frame leaves the offset at the
// 8-byte link slot, compressed to C.ADDI4SPN.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·badfp(SB), NOSPLIT, $0
TEXT ·argfp(SB), NOSPLIT, $0
MOV $arg(FP), X10
RET
`)
_, _, _, _, _, err := assembleRISCV(fn)
if err == nil {
t.Error("expected error for MOV $arg(FP), got nil")
code, _, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
// prologue (0: leaf, zero frame) + C.ADDI4SPN (2) + RET (4) = 6
want := []byte{0x28, 0x00, 0x67, 0x80, 0x00, 0x00}
if string(code) != string(want) {
t.Errorf("got % x, want % x", code, want)
}
}
func TestRISCV_Bookkeeping(t *testing.T) {
// FUNCDATA and PCDATA contribute no bytes; UNDEF is the toolchain's
// ebreak, compressed to C.EBREAK under RVC.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·book(SB), NOSPLIT, $0-8
FUNCDATA $0, marks<>(SB)
PCDATA $1, $1
UNDEF
MOV $1, X10
MOV X10, ret+0(FP)
RET
`)
code, _, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
// C.EBREAK (2) + C.LI X10, 1 (2) + C.SWSP (2) + RET (4) = 10: the
// FUNCDATA and PCDATA statements contribute nothing.
want := []byte{0x02, 0x90, 0x05, 0x45, 0x2a, 0xe4, 0x67, 0x80, 0x00, 0x00}
if string(code) != string(want) {
t.Errorf("got % x, want % x", code, want)
}
}
func TestRISCV_JMPPCRel(t *testing.T) {
// JMP N(PC): the displacement tracks the instruction N source slots
// away in the final layout (0 the jump itself, negative backwards).
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·slots(SB), NOSPLIT, $0-0
JMP 2(PC)
MOV $1, X11
MOV $2, X12
MOV X12, X11
JMP -3(PC)
RET
`)
code, _, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
// JMP 2(PC) lands on the C.MV six bytes ahead; JMP -3(PC) lands back on
// the first C.LI, six bytes behind.
want := []byte{
0x6f, 0x00, 0x60, 0x00, // JAL X0, 6
0x85, 0x45, // C.LI X11, 1
0x09, 0x46, // C.LI X12, 2
0xb2, 0x85, // C.MV X11, X12
0x6f, 0xf0, 0xbf, 0xff, // JAL X0, -6
0x67, 0x80, 0x00, 0x00, // RET
}
if string(code) != string(want) {
t.Errorf("got % x, want % x", code, want)
}
}
func TestRISCV_MOVWideImm(t *testing.T) {
// Shift-sequence constants compress like the toolchain's expansion.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·wide(SB), NOSPLIT, $0-0
MOV $0x8000000000000000, X5
MOV $0x100000000, X5
MOV $0x000fffffffffffda, X5
RET
`)
code, _, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
// C.LI -1, C.SLLI 63; C.LI 1, C.SLLI 32; C.LI -19, C.SLLI 13, SRLI 12.
want := []byte{
0xfd, 0x52, 0xfe, 0x12,
0x85, 0x42, 0x82, 0x12,
0xb5, 0x52, 0xb6, 0x02, 0x93, 0xd2, 0xc2, 0x00,
0x67, 0x80, 0x00, 0x00,
}
if string(code) != string(want) {
t.Errorf("got % x, want % x", code, want)
}
}
func TestRISCV_MOVImmPool(t *testing.T) {
// A constant outside the shift shapes loads from the pooled $i64 data
// symbol via AUIPC+LD, named like the toolchain's pool.
src := `#include "textflag.h"
TEXT ·pool(SB), NOSPLIT, $0-8
MOV $0x0101010101010101, X16
MOV X16, ret+0(FP)
RET
`
f, errs := parser.Parse("pool_riscv64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
// AUIPC X16, 0 + LD X16, 0(X16): the relocation pair carries the symbol.
wantCode := []byte{0x17, 0x08, 0x00, 0x00, 0x03, 0x38, 0x08, 0x00}
if string(img.Code[0:8]) != string(wantCode) {
t.Errorf("pool load: got % x", img.Code[0:8])
}
var lit *DataSymbol
for i := range img.DataSyms {
if img.DataSyms[i].Name == "$i64.0101010101010101" {
lit = &img.DataSyms[i]
}
}
if lit == nil {
t.Fatalf("pool symbol missing: %v", img.DataSyms)
}
wantData := []byte{0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01}
if string(img.Data[lit.Offset:lit.Offset+8]) != string(wantData) {
t.Errorf("pool bytes: got % x", img.Data[lit.Offset:lit.Offset+8])
}
}
@@ -805,7 +929,7 @@ TEXT ·calltest(SB), NOSPLIT, $0
CALL ext(SB)
RET
`)
code, _, relocs, _, _, err := assembleRISCV(fn)
code, _, relocs, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
@@ -834,7 +958,7 @@ TEXT ·calllocal(SB), NOSPLIT, $0
sub:
RET
`)
_, _, _, _, _, err := assembleRISCV(fn)
_, _, _, _, _, _, err := assembleRISCV(fn)
if err == nil {
t.Error("expected error for CALL to local label, got nil")
}
@@ -868,7 +992,7 @@ func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]in
t.Helper()
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
instr := fn.Body[0].(*ast.Instr)
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil, nil)
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil, nil, nil)
}
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span
@@ -917,7 +1041,7 @@ func TestRISCVBranchFarBody(t *testing.T) {
}
sb.WriteString("done:\n\tRET\n")
fn := firstTextRISCV(t, sb.String())
out, _, _, _, _, err := assembleRISCV(fn)
out, _, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
@@ -943,7 +1067,7 @@ TEXT ·csrhi(SB), NOSPLIT, $0
CSRRW $4096, X10, X11
RET
`)
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
if _, _, _, _, _, _, err := assembleRISCV(fn); err == nil {
t.Error("expected an out-of-range error for CSR $4096, got none")
}
fn = firstTextRISCV(t, `#include "textflag.h"
@@ -951,25 +1075,24 @@ TEXT ·csrmax(SB), NOSPLIT, $0
CSRRW $4095, X10, X11
RET
`)
if _, _, _, _, _, err := assembleRISCV(fn); err != nil {
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
t.Errorf("CSR $4095 must assemble: %v", err)
}
}
// TestRISCV_Imm64Rejected checks that immediates outside the signed 32-bit
// span are diagnosed instead of silently truncated to their low 32 bits (the
// toolchain materialises such constants via SLLI expansion, which this
// assembler does not implement).
// span are diagnosed instead of silently truncated to their low 32 bits for
// the I-type arithmetic; the MOV forms materialise the wide constant instead
// (shift sequence or pooled load), like the toolchain.
func TestRISCV_Imm64Rejected(t *testing.T) {
cases := []string{
"MOV $0x123456789, X10",
"ADDI $0x100000000, X10, X11",
"ANDI $-0x800000001, X10, X11",
"SUB $0x100000000, X10, X11",
}
for _, src := range cases {
fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·wide(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
if _, _, _, _, _, _, err := assembleRISCV(fn); err == nil {
t.Errorf("%s: expected an out-of-range error, got none", src)
}
}
@@ -982,9 +1105,19 @@ TEXT ·edge(SB), NOSPLIT, $0
SUB $0x80000000, X12, X13
RET
`)
if _, _, _, _, _, err := assembleRISCV(fn); err != nil {
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
t.Errorf("int32-span immediates must assemble: %v", err)
}
// Beyond the span the MOV forms materialise the constant like the
// toolchain instead of diagnosing it.
fn = firstTextRISCV(t, `#include "textflag.h"
TEXT ·pool(SB), NOSPLIT, $0
MOV $0x123456789, X10
RET
`)
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
t.Errorf("MOV with a 64-bit immediate must assemble: %v", err)
}
}
// riscvWants decodes code as little-endian words and pins each one; the
+1 -1
View File
@@ -7,7 +7,7 @@ import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
+1 -1
View File
@@ -6,7 +6,7 @@ package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestRISCVFrameSpadjAndLines checks that a framed function records its
+1 -1
View File
@@ -13,7 +13,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
+344 -11
View File
@@ -61,6 +61,25 @@ const (
// vexImmRMGPR is the immediate form over general-purpose registers
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
vexImmRMGPR
// vexRMOpGPR is the two-operand /digit form over general-purpose
// registers (BLSI, BLSMSK, BLSR): ModRM.reg = /digit, ModRM.rm = src
// (op0), VEX.vvvv = dst (op1), L = 0.
vexRMOpGPR
// vexCountGPR is the three-operand count form over general-purpose
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): the first operand rides
// VEX.vvvv and the second is r/m, the opposite pairing of the ANDN
// family, with reg = dst (op2), L = 0.
vexCountGPR
// vexExtractGPR is the lane-extract-to-GPR form `OP $imm, xsrc, GPR/mem
// dst`: ModRM.reg = xsrc (op1), ModRM.rm = destination (op2), imm8 =
// op0, the VPEXTRB/W/D/Q layout. EVEX only; the destination never
// carries a vector length, so the register the L'L field follows is the
// XMM source.
vexExtractGPR
// vexBlend4 is the four-operand variable blend `OP mask, src2, src1,
// dst` (VPBLENDVB): ModRM.reg = dst (op3), VEX.vvvv = src1 (op2),
// ModRM.rm = src2 (op1) and the mask register in the /is4 byte (op0).
vexBlend4
)
// vexSpec describes one VEX instruction's encoding parameters.
@@ -187,8 +206,28 @@ var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8).
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8), and its
// double twin under op 01; the in-lane permutes under 04/05.
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM},
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM},
"VPERMILPD": {3, 0x05, 0, 1, -1, vexImmRM},
// VEX.66.0F3A.W0, the immediate-controlled AVX tail: the rounding
// pair, the AES key assistant and the string compares.
"VROUNDPD": {3, 0x09, 0, 1, -1, vexImmRM},
"VROUNDPS": {3, 0x08, 0, 1, -1, vexImmRM},
"VAESKEYGENASSIST": {3, 0xDF, 0, 1, -1, vexImmRM},
"VPCMPESTRI": {3, 0x61, 0, 1, -1, vexImmRM},
"VPCMPESTRM": {3, 0x60, 0, 1, -1, vexImmRM},
"VPCMPISTRI": {3, 0x63, 0, 1, -1, vexImmRM},
"VPCMPISTRM": {3, 0x62, 0, 1, -1, vexImmRM},
// VEX.128.66.0F3A.W0, the scalar lane extract to a GPR or memory
// (reg = the XMM source, r/m = the destination).
"VEXTRACTPS": {3, 0x17, 0, 1, -1, vexExtractGPR},
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR},
// VEX.128.66.0F3A.W0, the four-operand variable blend with its mask
// register in the /is4 byte.
"VPBLENDVB": {3, 0x4C, 0, 1, -1, vexBlend4},
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// imm8).
@@ -230,8 +269,34 @@ var vexTable = map[string]vexSpec{
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
// VEX.NDS.LZ.0F38, the BMI2 three-operand bit ops: BEXTR and BZHI
// share the F7/F5 opcodes across W, the variable shifts carry their
// direction in the prefix (SHLX 66, SHRX F2, SARX F3) and PDEP/PEXT
// in F2/F3.
"BEXTRL": {2, 0xF7, 0, 0, -1, vexCountGPR},
"BEXTRQ": {2, 0xF7, 1, 0, -1, vexCountGPR},
"BZHIL": {2, 0xF5, 0, 0, -1, vexCountGPR},
"BZHIQ": {2, 0xF5, 1, 0, -1, vexCountGPR},
"SARXL": {2, 0xF7, 0, 2, -1, vexCountGPR},
"SARXQ": {2, 0xF7, 1, 2, -1, vexCountGPR},
"SHLXL": {2, 0xF7, 0, 1, -1, vexCountGPR},
"SHLXQ": {2, 0xF7, 1, 1, -1, vexCountGPR},
"SHRXL": {2, 0xF7, 0, 3, -1, vexCountGPR},
"SHRXQ": {2, 0xF7, 1, 3, -1, vexCountGPR},
"PDEPL": {2, 0xF5, 0, 3, -1, vexNDS3GPR},
"PDEPQ": {2, 0xF5, 1, 3, -1, vexNDS3GPR},
"PEXTL": {2, 0xF5, 0, 2, -1, vexNDS3GPR},
"PEXTQ": {2, 0xF5, 1, 2, -1, vexNDS3GPR},
// VEX.LZ.0F38.W, the BMI1 unary bit ops (src, dst: ModRM.reg = /digit,
// rm = src, vvvv = dst).
"BLSIL": {2, 0xF3, 0, 0, 3, vexRMOpGPR},
"BLSIQ": {2, 0xF3, 1, 0, 3, vexRMOpGPR},
"BLSMSKL": {2, 0xF3, 0, 0, 2, vexRMOpGPR},
"BLSMSKQ": {2, 0xF3, 1, 0, 2, vexRMOpGPR},
"BLSRL": {2, 0xF3, 0, 0, 1, vexRMOpGPR},
"BLSRQ": {2, 0xF3, 1, 0, 1, vexRMOpGPR},
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
@@ -290,6 +355,128 @@ var vexTable = map[string]vexSpec{
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
// --- the VEX forms the avx512enc corpus exercises alongside the EVEX
// spellings, read off the toolchain opcode tables ---
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3},
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3},
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3},
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3},
"VANDNPD": {1, 0x55, 0, 1, -1, vexNDS3},
"VANDPD": {1, 0x54, 0, 1, -1, vexNDS3},
"VCOMISD": {1, 0x2F, 0, 1, -1, vexRM},
"VCVTSD2SS": {1, 0x5A, 0, 3, -1, vexNDS3},
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3},
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3},
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3},
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3},
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3},
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3},
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3},
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3},
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3},
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3},
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3},
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3},
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3},
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3},
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3},
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3},
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3},
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3},
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3},
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3},
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3},
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3},
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3},
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3},
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3},
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3},
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3},
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3},
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3},
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3},
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3},
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3},
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3},
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3},
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3},
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3},
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3},
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3},
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3},
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3},
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3},
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3},
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3},
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3},
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3},
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3},
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3},
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3},
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3},
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3},
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3},
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3},
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3},
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3},
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3},
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3},
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3},
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3},
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm},
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3},
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM},
"VMOVNTPD": {1, 0x2B, 0, 1, -1, vexRMRev},
"VORPD": {1, 0x56, 0, 1, -1, vexNDS3},
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3},
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3},
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3},
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3},
"VPCMPEQQ": {2, 0x29, 0, 1, -1, vexNDS3},
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3},
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3},
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3},
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3},
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3},
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtract},
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtract},
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtract},
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm},
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm},
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3},
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3},
"VPMULUDQ": {1, 0xF4, 0, 1, -1, vexNDS3},
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3},
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3},
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3},
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3},
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3},
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3},
"VPUNPCKHQDQ": {1, 0x6D, 0, 1, -1, vexNDS3},
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3},
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3},
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3},
"VSQRTPD": {1, 0x51, 0, 1, -1, vexRM},
"VSQRTSD": {1, 0x51, 0, 3, -1, vexNDS3},
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3},
"VUCOMISD": {1, 0x2E, 0, 1, -1, vexRM},
// VEX.0F.WIG, the plain-prefix single/double arithmetic and unpack
// spellings (no 66 prefix; WIG, so W = 0).
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3},
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3},
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3},
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3},
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3},
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM},
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev},
// VEX.128.66.0F, the scalar and packed compare forms.
"VCOMISS": {1, 0x2F, 0, 1, -1, vexRM},
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM},
// VEX.128.0F.F3/F2.W0, the high/low word shuffles ($imm, src, dst).
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM},
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM},
}
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
@@ -361,8 +548,16 @@ func isVex(mnemUpper string) bool {
if _, ok := vexTable[mnemUpper]; ok {
return true
}
_, ok := vexMoveTable[mnemUpper]
return ok
if _, ok := vexMoveTable[mnemUpper]; ok {
return true
}
// The dual-shape moves (VMOVHPD/VMOVLPD) pick their VEX form by operand
// count in encodeVex.
switch mnemUpper {
case "VMOVHPD", "VMOVLPD":
return true
}
return false
}
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
@@ -396,6 +591,22 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
}
}
// The high/low double moves split by operand count: three operands
// load-and-insert (mem, src, dst, an NDS form), two store (xmm, m64,
// the reversed store layout).
if mnemUpper == "VMOVHPD" || mnemUpper == "VMOVLPD" {
loadOp, storeOp := byte(0x16), byte(0x17)
if mnemUpper == "VMOVLPD" {
loadOp, storeOp = 0x12, 0x13
}
switch len(ops) {
case 3:
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: loadOp, w: 0, pp: 1, opdigit: -1, form: vexNDS3}, ops)
case 2:
return e.encodeVexRMRev(vexSpec{mapSel: 1, opcode: storeOp, w: 0, pp: 1, opdigit: -1, form: vexRMRev}, ops)
}
return fmt.Errorf("%s expects 2 or 3 operands, got %d", mnemUpper, len(ops))
}
spec := vexTable[mnemUpper]
switch spec.form {
case vexNDS3:
@@ -410,6 +621,10 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3Imm(spec, ops)
case vexExtract:
return e.encodeVexExtract(spec, ops)
case vexExtractGPR:
return e.encodeVexExtractGPR(spec, ops)
case vexBlend4:
return e.encodeVexBlend4(spec, ops)
case vexRMSrcLen:
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero:
@@ -420,6 +635,10 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3GPR(spec, ops)
case vexImmRMGPR:
return e.encodeVexImmRMGPR(spec, ops)
case vexRMOpGPR:
return e.encodeVexRMOpGPR(spec, ops)
case vexCountGPR:
return e.encodeVexCountGPR(spec, ops)
case vexRMRev:
return e.encodeVexRMRev(spec, ops)
}
@@ -523,9 +742,10 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
if !ok {
return fmt.Errorf("shift count must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("shift source must be a vector register")
// The count source is a vector register or memory; the VEX length
// follows the destination register either way.
if !vecOrMem(src) {
return fmt.Errorf("shift source must be a vector register or memory")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
@@ -533,7 +753,7 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
}
vvvvBar := 15 - (dstReg.idx & 15)
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, srcReg); err != nil {
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, src); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
@@ -700,7 +920,11 @@ func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
if !ok || vvvvReg.isVec() {
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
}
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15-(vvvvReg.idx&15), src2)
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2)
}
// encodeVexImmRMGPR encodes the immediate form over general-purpose
@@ -729,6 +953,48 @@ func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
return nil
}
// encodeVexRMOpGPR encodes the two-operand /digit form over general-purpose
// registers (BLSI, BLSMSK, BLSR): OP src, dst with ModRM.reg = /digit,
// ModRM.rm = src and VEX.vvvv = dst.
func (e *enc) encodeVexRMOpGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("instruction expects 2 operands (src, dst), got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
return e.emitVexFields(spec, 0, spec.opdigit, 0, 15-(dstReg.idx&15), src)
}
// encodeVexCountGPR encodes the three-operand count form over general-purpose
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): OP src, count, dst with
// VEX.vvvv = src (op0), ModRM.rm = count (op1), ModRM.reg = dst (op2).
func (e *enc) encodeVexCountGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX count instruction expects 3 operands, got %d", len(ops))
}
src, count, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
countReg, ok := count.(Reg)
if !ok || countReg.isVec() {
return fmt.Errorf("VEX count operand must be a general-purpose register")
}
srcReg, ok := src.(Reg)
if !ok || srcReg.isVec() {
return fmt.Errorf("VEX count source must be a general-purpose register")
}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(srcReg.idx&15), count)
}
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
// a store with no register-destination form).
@@ -750,6 +1016,73 @@ func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
}
// encodeVexExtractGPR encodes the lane extract to a general-purpose register
// or memory (VEXTRACTPS): OP $imm, xsrc, gpr/mem with the XMM source in
// ModRM.reg and the destination in r/m, L = 0.
func (e *enc) encodeVexExtractGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() || srcReg.size != 16 {
return fmt.Errorf("extract source must be an XMM register")
}
if _, isReg := dst.(Reg); !isReg && !memOperand(dst) {
return fmt.Errorf("extract destination must be a register or memory")
}
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, 0, srcReg.idx&7, rBit, 15, dst); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexBlend4 encodes the four-operand variable blend (VPBLENDVB):
// OP mask, src2, src1, dst with ModRM.reg = dst, VEX.vvvv = src1, r/m =
// src2 and the mask XMM register in the trailing /is4 byte.
func (e *enc) encodeVexBlend4(spec vexSpec, ops []Operand) error {
if len(ops) != 4 {
return fmt.Errorf("blend expects 4 operands (mask, src2, src1, dst), got %d", len(ops))
}
mask, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
maskReg, ok := mask.(Reg)
if !ok || !maskReg.isVec() || maskReg.size != 16 {
return fmt.Errorf("blend mask must be an XMM register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("blend second source must be a vector register")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("blend destination must be a vector register")
}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, dstReg.vecLenBit(), dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2); err != nil {
return err
}
// The /is4 byte names the mask register: bits [3:0] its low nibble,
// bit 7 the fourth register bit (X8-X15).
e.out = append(e.out, byte(maskReg.idx&7)|byte((maskReg.idx&8)<<4))
return nil
}
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
// move uses the store-form layout (reg = source, rm = destination), matching
+57
View File
@@ -31,6 +31,51 @@ var x86asmUnrecognised = map[string]bool{
"RORXQ": true,
"VFMADD213SD": true,
"VFNMADD231SD": true,
// The scalar FMA spellings the decoder's tables lack entirely.
"VFMADD132SD": true,
"VFMADD132SS": true,
"VFMADD213SS": true,
"VFMADD231SD": true,
"VFMADD231SS": true,
"VFMSUB132SD": true,
"VFMSUB132SS": true,
"VFMSUB213SD": true,
"VFMSUB213SS": true,
"VFMSUB231SD": true,
"VFMSUB231SS": true,
"VFNMADD132SD": true,
"VFNMADD132SS": true,
"VFNMADD213SD": true,
"VFNMADD213SS": true,
"VFNMADD231SS": true,
"VFNMSUB132SD": true,
"VFNMSUB132SS": true,
"VFNMSUB213SD": true,
"VFNMSUB213SS": true,
"VFNMSUB231SD": true,
"VFNMSUB231SS": true,
// The BMI1 unary bit ops the decoder's AVX tables lack.
"BLSIL": true,
"BLSIQ": true,
"BLSMSKL": true,
"BLSMSKQ": true,
"BLSRL": true,
"BLSRQ": true,
// The BMI2 bit ops whose W1/LZ rows the decoder misses.
"BEXTRL": true,
"BEXTRQ": true,
"BZHIL": true,
"BZHIQ": true,
"PDEPL": true,
"PDEPQ": true,
"PEXTL": true,
"PEXTQ": true,
"SARXL": true,
"SARXQ": true,
"SHLXL": true,
"SHLXQ": true,
"SHRXL": true,
"SHRXQ": true,
}
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
@@ -237,6 +282,18 @@ func TestVexGroundTruth(t *testing.T) {
{"MULXQ AX,BX,CX", "MULXQ", []Operand{AX, BX, CX}, "c4e2e3f6c8", ""},
{"RORXL $3,AX,CX", "RORXL", []Operand{Imm(3), AX, CX}, "c4e37bf0c803", ""},
{"RORXQ $3,AX,CX", "RORXQ", []Operand{Imm(3), AX, CX}, "c4e3fbf0c803", ""},
// BMI2 variable shifts and bit ops (three general registers).
{"SHLXL AX,CX,R15", "SHLXL", []Operand{AX, CX, vreg(t, "R15")}, "c46279f7f9", ""},
{"SHRXQ R8,DX,AX", "SHRXQ", []Operand{vreg(t, "R8"), DX, AX}, "c4e2bbf7c2", ""},
{"SARXQ AX,DX,R9", "SARXQ", []Operand{AX, DX, vreg(t, "R9")}, "c462faf7ca", ""},
{"BEXTRL AX,CX,R15", "BEXTRL", []Operand{AX, CX, vreg(t, "R15")}, "c46278f7f9", ""},
{"BZHIQ AX,CX,R15", "BZHIQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f8f5f9", ""},
{"PDEPQ AX,CX,R15", "PDEPQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f3f5f8", ""},
{"PEXTQ AX,CX,R15", "PEXTQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f2f5f8", ""},
// BMI1 unary bit ops (src, dst: /digit in ModRM.reg, dst in vvvv).
{"BLSIL AX,CX", "BLSIL", []Operand{AX, CX}, "c4e270f3d8", ""},
{"BLSRQ AX,CX", "BLSRQ", []Operand{AX, CX}, "c4e2f0f3c8", ""},
{"BLSMSKQ AX,CX", "BLSMSKQ", []Operand{AX, CX}, "c4e2f0f3d0", ""},
// Two-operand reg/rm form (v̄vvv must be 1111).
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
+18 -8
View File
@@ -8,7 +8,7 @@
// to the arch package; the AST records syntax only.
package ast
import "sourcedock.dev/petrbalvin/gasm-devkit/token"
import "sourcedock.dev/petrbalvin/gasm-sdk/token"
// File is the parsed representation of one .s source file.
type File struct {
@@ -151,11 +151,21 @@ type Immediate struct {
// Address is a non-immediate operand: a register, a memory reference, a symbol
// reference or a label. Fields are populated best-effort from the syntax.
type Address struct {
Sym *Symbol // name reference (bare ident, or name+off(pseudo))
Base string // base register, from (base)
Index string // index register, from (index*scale)
Scale int // index scale; 0 when absent
Offset int64 // leading displacement, from off(base)
HasOff bool // a leading displacement is present
Shift string // verbatim arm64 shift suffix, e.g. "<< 2"
Sym *Symbol // name reference (bare ident, or name+off(pseudo))
Base string // base register, from (base)
Index string // index register, from (index*scale)
Scale int // index scale; 0 when absent
Offset int64 // leading displacement, from off(base)
HasOff bool // a leading displacement is present
Shift string // verbatim arm64 shift suffix, e.g. "<< 2"
Range *RegRange // bracketed register range; nil for every other form
}
// RegRange is a bracketed register range, [Z0-Z3]: the amd64 spelling of
// the four-register source of the 4FMAPS/4VNNIW families. Lo and Hi carry
// the verbatim register spellings; the range is inclusive at both ends.
type RegRange struct {
Lo string
Hi string
Pos token.Position
}
+1 -1
View File
@@ -6,7 +6,7 @@ package ast
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
"sourcedock.dev/petrbalvin/gasm-sdk/token"
)
func pos(line, col int) token.Position { return token.Position{Line: line, Column: col} }
+35 -22
View File
@@ -17,7 +17,7 @@ import (
"regexp"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
)
// go_asm.h is the header the Go compiler writes for every package that
@@ -27,7 +27,10 @@ import (
// GOROOT assembly includes it, and a standalone assembler has no compiler
// to have produced it, so gasm generates the equivalent itself: the package
// the .s file lives in is parsed and type-checked here, with the target
// architecture's own sizes, and the same defines are written out.
// architecture's own sizes, and the same defines are written out. The
// type-checking GOOS is selected by the caller: a GOOS-specific file
// (sys_darwin_arm64.s) needs its platform's defines, which a header from
// the ambient GOOS silently omits.
//
// The emitter mirrors cmd/compile's dumpasmhdr exactly: constants come out
// as "const_NAME", struct entries as "NAME__size" followed by the fields in
@@ -67,11 +70,15 @@ func goAsmHeaderResolved(asmDir string, dirs []string) bool {
return false
}
// generateGoAsmHeader type-checks the Go package in pkgDir for goarch,
// writes its go_asm.h equivalent into dir, and returns dir. The caller
// owns the directory and its removal.
func generateGoAsmHeader(pkgDir, goarch, dir string) (string, error) {
imp := newSourceImporter(goarch)
// generateGoAsmHeader type-checks the Go package in pkgDir for goos and
// goarch, writes its go_asm.h equivalent into dir, and returns dir. An
// empty goos means the ambient one. The caller owns the directory and its
// removal.
func generateGoAsmHeader(pkgDir, goos, goarch, dir string) (string, error) {
if goos == "" {
goos = build.Default.GOOS
}
imp := newSourceImporter(goos, goarch)
if imp.sizes == nil {
return "", fmt.Errorf("go_asm.h: unknown GOARCH %q", goarch)
}
@@ -89,7 +96,7 @@ func generateGoAsmHeader(pkgDir, goarch, dir string) (string, error) {
}
var b strings.Builder
fmt.Fprintf(&b, "// generated by gasm from package %s (GOARCH %s)\n\n", bp.Name, goarch)
fmt.Fprintf(&b, "// generated by gasm from package %s (GOOS %s, GOARCH %s)\n\n", bp.Name, goos, goarch)
// Files in the build's own order and declarations in source order: the
// same walk the compiler's reader makes, so the header reads the same
// way the toolchain's does. Order carries no meaning to the assembler
@@ -203,13 +210,14 @@ type sourceImporter struct {
pkgs map[string]*types.Package
}
// newSourceImporter returns the importer for one target architecture.
// newSourceImporter returns the importer for one target GOOS and GOARCH.
// Cgo is disabled so the file set is deterministic and independent of the
// host's C toolchain: cgo-tagged files drop out of the build exactly as
// they do from a CGO_ENABLED=0 build, whose assembly is what gasm targets.
func newSourceImporter(goarch string) *sourceImporter {
func newSourceImporter(goos, goarch string) *sourceImporter {
ctxt := new(build.Context)
*ctxt = build.Default
ctxt.GOOS = goos
ctxt.GOARCH = goarch
ctxt.CgoEnabled = false
return &sourceImporter{
@@ -291,7 +299,7 @@ func (im *sourceImporter) checkPackage(bp *build.Package, files []*ast.File) (*t
// type-check must not be re-checked once per file.
type asmhdrCache struct {
root string
dirs map[string]string // "pkgDir\x00goarch" -> directory holding go_asm.h
dirs map[string]string // "pkgDir\x00goos\x00goarch" -> directory holding go_asm.h
errs map[string]error
}
@@ -304,17 +312,22 @@ func newAsmhdrCache() (*asmhdrCache, error) {
}
// dirFor returns the directory holding the generated go_asm.h for pkgDir
// and goarch, generating it on first use.
func (c *asmhdrCache) dirFor(pkgDir, goarch string) (string, error) {
key := pkgDir + "\x00" + goarch
// under goos and goarch, generating it on first use. An empty goos means
// the ambient one, resolved here so that one package cannot generate twice
// under an explicit and an implicit spelling of the same GOOS.
func (c *asmhdrCache) dirFor(pkgDir, goos, goarch string) (string, error) {
if goos == "" {
goos = build.Default.GOOS
}
key := pkgDir + "\x00" + goos + "\x00" + goarch
if dir, ok := c.dirs[key]; ok {
return dir, nil
}
if err, ok := c.errs[key]; ok {
return "", err
}
dir := filepath.Join(c.root, fmt.Sprintf("h%d_%s", len(c.dirs), goarch))
if _, err := generateGoAsmHeader(pkgDir, goarch, dir); err != nil {
dir := filepath.Join(c.root, fmt.Sprintf("h%d_%s_%s", len(c.dirs), goos, goarch))
if _, err := generateGoAsmHeader(pkgDir, goos, goarch, dir); err != nil {
c.errs[key] = err
return "", err
}
@@ -327,10 +340,10 @@ func (c *asmhdrCache) close() { os.RemoveAll(c.root) }
// ensureGoAsmHeader prepares the include directory a file that includes
// go_asm.h needs: the generated header for the package in path's directory,
// for the file's target architecture. It reports a usage error when the
// architecture cannot be determined, and passes through the generator's
// diagnostics, which name the package.
func ensureGoAsmHeader(path string, target arch.Arch, cache *asmhdrCache) (string, func(), error) {
// for the file's target GOOS and architecture. It reports a usage error
// when the architecture cannot be determined, and passes through the
// generator's diagnostics, which name the package.
func ensureGoAsmHeader(path string, target arch.Arch, goos string, cache *asmhdrCache) (string, func(), error) {
if path == "-" {
return "", nil, errors.New("cannot generate go_asm.h for standard input (no package directory)")
}
@@ -338,14 +351,14 @@ func ensureGoAsmHeader(path string, target arch.Arch, cache *asmhdrCache) (strin
return "", nil, errors.New("a file that includes go_asm.h needs a target architecture: name the file _<arch>.s or pass -GOARCH")
}
if cache != nil {
dir, err := cache.dirFor(filepath.Dir(path), goarchName(target))
dir, err := cache.dirFor(filepath.Dir(path), goos, goarchName(target))
return dir, func() {}, err
}
root, err := os.MkdirTemp("", "gasm-asmhdr")
if err != nil {
return "", nil, err
}
dir, err := generateGoAsmHeader(filepath.Dir(path), goarchName(target), root)
dir, err := generateGoAsmHeader(filepath.Dir(path), goos, goarchName(target), root)
if err != nil {
os.RemoveAll(root)
return "", nil, err
+178 -13
View File
@@ -5,6 +5,7 @@ package main
import (
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
@@ -22,12 +23,13 @@ func writePkg(t *testing.T, files map[string]string) string {
return dir
}
// generateFor generates the header for dir and returns its text.
func generateFor(t *testing.T, dir, goarch string) string {
// generateFor generates the header for dir and returns its text. An empty
// goos means the ambient one.
func generateFor(t *testing.T, dir, goos, goarch string) string {
t.Helper()
hdrDir, err := generateGoAsmHeader(dir, goarch, t.TempDir())
hdrDir, err := generateGoAsmHeader(dir, goos, goarch, t.TempDir())
if err != nil {
t.Fatalf("generateGoAsmHeader(%q, %s): %v", dir, goarch, err)
t.Fatalf("generateGoAsmHeader(%q, %s, %s): %v", dir, goos, goarch, err)
}
b, err := os.ReadFile(filepath.Join(hdrDir, "go_asm.h"))
if err != nil {
@@ -72,7 +74,7 @@ type aliased struct {
type alias = aliased
`})
hdr := generateFor(t, dir, "amd64")
hdr := generateFor(t, dir, "", "amd64")
want := []string{
"#define const_bufSize 1024",
// iota resolves through go/types, one define per name.
@@ -133,8 +135,8 @@ package perarch
const flavour = 2
`,
})
amd64 := generateFor(t, dir, "amd64")
arm64 := generateFor(t, dir, "arm64")
amd64 := generateFor(t, dir, "", "amd64")
arm64 := generateFor(t, dir, "", "arm64")
if !strings.Contains(amd64, "#define const_flavour 1\n") {
t.Errorf("amd64 header misses const_flavour 1:\n%s", amd64)
}
@@ -149,19 +151,90 @@ const flavour = 2
if !strings.Contains(amd64, "#define layout__size 16\n") || !strings.Contains(amd64, "#define layout_p 8\n") {
t.Errorf("amd64 layout wrong:\n%s", amd64)
}
w386 := generateFor(t, dir, "386")
w386 := generateFor(t, dir, "", "386")
if !strings.Contains(w386, "#define layout__size 8\n") || !strings.Contains(w386, "#define layout_p 4\n") {
t.Errorf("386 layout wrong:\n%s", w386)
}
}
// TestGenerateGoAsmHeaderGOOS pins the GOOS half of the target: only the
// platform's own files type-check into the header, which is why
// sys_darwin_arm64.s cannot assemble against a linux-generated one.
func TestGenerateGoAsmHeaderGOOS(t *testing.T) {
dir := writePkg(t, map[string]string{
"common.go": `package goosaware
type shared struct {
a int32
}
`,
"plat_darwin.go": `//go:build darwin
package goosaware
type platform struct {
trampoline_numer int64
}
`,
"plat_windows.go": `//go:build windows
package goosaware
type platform struct {
callbackArgs__size int32
}
`,
})
darwin := generateFor(t, dir, "darwin", "arm64")
if !strings.Contains(darwin, "#define platform__size 8\n") || !strings.Contains(darwin, "#define platform_trampoline_numer 0\n") {
t.Errorf("darwin header misses the darwin layout:\n%s", darwin)
}
if strings.Contains(darwin, "callbackArgs") {
t.Errorf("darwin header must not carry the windows layout:\n%s", darwin)
}
windows := generateFor(t, dir, "windows", "arm64")
if !strings.Contains(windows, "#define platform_callbackArgs__size 0\n") {
t.Errorf("windows header misses the windows layout:\n%s", windows)
}
if strings.Contains(windows, "trampoline_numer") {
t.Errorf("windows header must not carry the darwin layout:\n%s", windows)
}
// The ambient GOOS is neither of the two, so only shared's defines are
// emitted; the shared type keeps its layout there.
ambient := generateFor(t, dir, "", "arm64")
if !strings.Contains(ambient, "#define shared__size 4\n") {
t.Errorf("ambient header misses the shared layout:\n%s", ambient)
}
if strings.Contains(ambient, "#define platform_") {
t.Errorf("ambient header must not carry either platform layout:\n%s", ambient)
}
}
func TestGoosFromFilename(t *testing.T) {
for path, want := range map[string]string{
"/x/sys_darwin_arm64.s": "darwin",
"/x/sys_windows_arm64.s": "windows",
"/x/asm_linux_amd64.s": "linux",
"/x/rt0_darwin_arm64.s": "darwin",
"/x/vgetrandom_zos_s390x.s": "zos",
"/x/rt0_js_wasm.s": "js",
"/x/memmove_amd64.s": "",
"/x/vlop_arm.s": "",
"/x/stubs.s": "",
} {
if got := goosFromFilename(path); got != want {
t.Errorf("goosFromFilename(%q) = %q, want %q", path, got, want)
}
}
}
func TestGenerateGoAsmHeaderErrors(t *testing.T) {
t.Run("type error", func(t *testing.T) {
dir := writePkg(t, map[string]string{"bad.go": `package bad
const x = undefinedIdent
`})
_, err := generateGoAsmHeader(dir, "amd64", t.TempDir())
_, err := generateGoAsmHeader(dir, "", "amd64", t.TempDir())
if err == nil {
t.Fatal("generation must fail for a package that does not type-check")
}
@@ -174,7 +247,7 @@ const x = undefinedIdent
})
t.Run("no go files", func(t *testing.T) {
dir := t.TempDir()
_, err := generateGoAsmHeader(dir, "amd64", t.TempDir())
_, err := generateGoAsmHeader(dir, "", "amd64", t.TempDir())
if err == nil {
t.Fatal("generation must fail without Go files")
}
@@ -281,14 +354,106 @@ type header struct {
}
}
// TestRunCorpusAuditGOOS covers the filename-derived GOOS end to end: a
// kernel whose name names darwin must have its header type-checked with
// GOOS=darwin, so the darwin-only constant it offsets with is defined. The
// operand mirrors sys_darwin_arm64.s's trampoline, where a missing define
// leaves an unexpanded symbol in the offset and fails.
func TestRunCorpusAuditGOOS(t *testing.T) {
dir := t.TempDir()
write := func(name, src string) {
t.Helper()
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
t.Fatal(err)
}
}
write("pkg.go", "package corpus\n")
write("plat_darwin.go", "//go:build darwin\n\npackage corpus\n\nconst trampolineNumer = 8\n")
write("kern_darwin_arm64.s", "#include \"go_asm.h\"\n"+
"GLOBL timebase<>(SB), NOPTR, $16\n"+
"TEXT \xc2\xb7g(SB), NOSPLIT, $0-0\n"+
"\tMOVD\ttimebase<>+const_trampolineNumer(SB), R0\n"+
"\tRET\n")
stats, err := runCorpusAudit(dir, nil)
if err != nil {
t.Fatalf("runCorpusAudit: %v", err)
}
var arm *corpusTally
for i, tg := range stats.targets {
if tg.name == "arm64" {
arm = stats.tallies[i]
}
}
if arm == nil {
t.Fatal("no arm64 tally")
}
if arm.attempted != 1 || arm.assembled != 1 {
t.Errorf("arm64 = %d/%d, want 1/1; reasons: %v", arm.assembled, arm.attempted, arm.reasons)
}
}
// TestRunCorpusAuditBuildConstraint covers the //go:build classification end
// to end: a generic-named file whose constraint admits one target is
// attempted there alone (cpu_x86.s on amd64), and a file whose constraint
// admits none of the four targets is never attempted (the msan and
// goexperiment trees).
func TestRunCorpusAuditBuildConstraint(t *testing.T) {
dir := t.TempDir()
write := func(name, src string) {
t.Helper()
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
t.Fatal(err)
}
}
write("x86.s", "//go:build 386 || amd64\n\nTEXT \xc2\xb7f(SB), NOSPLIT, $0\n\tRET\n")
write("racey.s", "//go:build race\n\nTEXT \xc2\xb7r(SB), NOSPLIT, $0\n\tRET\n")
write("plain.s", "TEXT \xc2\xb7p(SB), NOSPLIT, $0\n\tRET\n")
stats, err := runCorpusAudit(dir, nil)
if err != nil {
t.Fatalf("runCorpusAudit: %v", err)
}
tally := func(name string) *corpusTally {
for i, tg := range stats.targets {
if tg.name == name {
return stats.tallies[i]
}
}
t.Fatalf("no tally for %s", name)
return nil
}
if stats.narrowed != 1 || stats.excluded != 1 || stats.generic != 1 {
t.Errorf("buckets = narrowed %d, excluded %d, generic %d; want 1, 1, 1", stats.narrowed, stats.excluded, stats.generic)
}
if a := tally("amd64"); a.attempted != 2 || a.assembled != 2 {
t.Errorf("amd64 = %d/%d, want 2/2 (x86.s and plain.s)", a.assembled, a.attempted)
}
for _, name := range []string{"arm64", "riscv64", "loong64"} {
if a := tally(name); a.attempted != 1 || a.assembled != 1 {
t.Errorf("%s = %d/%d, want 1/1 (plain.s only)", name, a.assembled, a.attempted)
}
}
if stats.full != 2 {
t.Errorf("full = %d, want 2 (x86.s over its one target, plain.s over all four)", stats.full)
}
}
// TestGenerateGoAsmHeaderRuntime pins the generator against the real thing:
// the runtime package, whose header the toolchain's own -asmhdr output was
// sampled from. Skipped in short mode: it type-checks the whole package.
// the runtime package of the ambient toolchain, whose header the toolchain's
// own -asmhdr output was sampled from. Skipped in short mode: it type-checks
// the whole package. The GOROOT comes from the go command itself, so the
// test follows whatever toolchain the host provides.
func TestGenerateGoAsmHeaderRuntime(t *testing.T) {
if testing.Short() {
t.Skip("type-checks the whole runtime package")
}
dir, err := generateGoAsmHeader("/usr/local/go/src/runtime", "amd64", t.TempDir())
out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil {
t.Skipf("no Go toolchain: %v", err)
}
runtimeDir := filepath.Join(strings.TrimSpace(string(out)), "src", "runtime")
dir, err := generateGoAsmHeader(runtimeDir, "", "amd64", t.TempDir())
if err != nil {
t.Fatalf("generateGoAsmHeader(runtime): %v", err)
}
+238 -43
View File
@@ -5,6 +5,8 @@ package main
import (
"fmt"
"go/build/constraint"
"maps"
"os"
"os/exec"
"path/filepath"
@@ -14,9 +16,9 @@ import (
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// cmdAuditInstructions cross-checks a gasm encoder against the Go toolchain's
@@ -37,7 +39,7 @@ import (
// construction and are excluded from the diff; the other architectures list
// their conditional branches outright.
func cmdAuditInstructions(args []string) error {
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [-I dir] [amd64|arm64|riscv64|loong64]", `
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [--list] [-I dir] [amd64|arm64|riscv64|loong64]", `
Compare the gasm encoder for the given architecture (default amd64) against
go tool asm and print the diff: superset encodings (gasm-only, shippable via
gasm asm --format goobj) and known-but-unencodable names (the backlog). The
@@ -54,16 +56,19 @@ toolchain probing. A file whose name carries a recognisable _arch suffix is
attempted for that architecture; a file without one is attempted for all
four, exactly as a GOARCH build would compile it. The report gives the
per-architecture pass rates and the most common failure reasons, which drive
the encodability backlog by frequency rather than by table order.
the encodability backlog by frequency rather than by table order. With
-list the report also prints every failing file with its reason, per
architecture.
`)
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
list := fs.Bool("list", false, "with --corpus, list every failing file with its reason, per architecture")
var dirs includeDirs
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
if err := fs.Parse(args); err != nil {
return err
}
if *corpus {
return cmdAuditCorpus(fs.Args(), dirs)
return cmdAuditCorpus(fs.Args(), dirs, *list)
}
archName := "amd64"
switch n := len(fs.Args()); {
@@ -288,6 +293,11 @@ func probeShapes(a arch.Arch) []string {
"V1.B16, [V2.B16], V3.B16", "V1.B8, [V2.B16, V3.B16], V4.B8",
"$4, V1.B16, V2.B16, V3.B16", "$15, V1", "V1, V2, p2",
"R0, R1, $1, $4, p2",
// The landing-pad kind, the compiler's PCDATA
// bookkeeping and the four-operand bitfield
// insert/extract family, as the toolchain's own
// testdata spells them.
"C", "$1, $0", "$0, R1, $1, R2",
}
case arch.RISCV:
return []string{
@@ -307,6 +317,9 @@ func probeShapes(a arch.Arch) []string {
"X5, X6, p2", "R5, R6, p2",
"X5, E8, M8, TA, MA, X6", "$4, E32, M1, TA, MA, X1",
"(X5), X6, V1, V2",
// The CSR immediate forms the toolchain's testdata spells:
// immediate, CSR name, destination.
"$2, TIME, X5",
"",
}
case arch.LOONG64:
@@ -323,6 +336,12 @@ func probeShapes(a arch.Arch) []string {
"V1, V2, V3", "X1, X2, X3", "V1, V2", "X1, X2", "V1", "X1",
// The vector compare-to-flag forms land in an FCC register.
"V1, FCC0", "X1, FCC0",
// The compiler's bookkeeping pair and the raw spellings the
// toolchain's own testdata carries: JIRL rd, rj, offset (the
// form RET lowers to), the prefetch with a 32-bit address and
// hint, and the byte-shuffle quads.
"$1, $0", "R1, R5, 0", "0(R7), $5, $0", "(R7), $5, $0",
"V1, V2, V3, V4", "X1, X2, X3, X4",
"",
}
}
@@ -388,20 +407,30 @@ type corpusTally struct {
assembled int
reasons map[string]int // failure reason → count
example map[string]string // failure reason → one representative file
fails []corpusFailure // every failure, in file order, for --list
}
func (t *corpusTally) fail(path, reason string) {
// corpusFailure is one failed attempt, recorded for the --list report.
type corpusFailure struct {
path string
reason string
detail string
}
func (t *corpusTally) fail(path string, err error) {
reason := corpusReason(err)
t.reasons[reason]++
if t.example[reason] == "" {
t.example[reason] = path
}
t.fails = append(t.fails, corpusFailure{path: path, reason: reason, detail: firstLine(err.Error())})
}
// cmdAuditCorpus implements audit-instructions --corpus. The include
// directories carry #include resolution over a corpus whose files refer to
// headers such as GOROOT/pkg/include, the same -I a toolchain comparison
// needs.
func cmdAuditCorpus(args []string, dirs includeDirs) error {
func cmdAuditCorpus(args []string, dirs includeDirs, list bool) error {
if len(args) > 1 {
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
}
@@ -439,7 +468,7 @@ func cmdAuditCorpus(args []string, dirs includeDirs) error {
if err != nil {
return err
}
printCorpusStats(stats)
printCorpusStats(stats, list)
return nil
}
@@ -448,8 +477,10 @@ type corpusStats struct {
root string
files int
generic int // files attempted for all four architectures
narrowed int // files whose //go:build admits a proper subset of the four
excluded int // files whose //go:build admits none of the four: never compiled
otherPort int // files named for another Go port: never attempted
full int // files that assembled for every target architecture
full int // files that assembled for every applicable target architecture
targets []corpusTarget
tallies []*corpusTally
}
@@ -460,8 +491,8 @@ type corpusStats struct {
// set, even when gasm does not support the architecture.
var goPortSuffixes = []string{
"386", "amd64", "arm", "arm64", "loong64", "mips", "mips64",
"mips64le", "mipsle", "ppc64", "ppc64le", "riscv", "riscv64",
"s390x", "wasm",
"mips64le", "mipsle", "mips64x", "mipsx", "ppc64", "ppc64le",
"ppc64x", "riscv", "riscv64", "s390x", "wasm",
}
// otherPortFile reports whether the file belongs to a build no supported
@@ -487,10 +518,47 @@ func otherPortFile(path string) bool {
// goOSNames are the GOOS values go/build recognises in file names.
var goOSNames = map[string]bool{
"aix": true, "darwin": true, "dragonfly": true, "freebsd": true,
"ios": true, "js": true, "linux": true, "netbsd": true,
"aix": true, "android": true, "darwin": true, "dragonfly": true,
"freebsd": true, "hurd": true, "illumos": true, "ios": true,
"js": true, "linux": true, "nacl": true, "netbsd": true,
"openbsd": true, "plan9": true, "solaris": true, "wasip1": true,
"windows": true,
"windows": true, "zos": true,
}
// resolveGOOS validates a -GOOS flag value, mirroring the architecture
// check's surface: a usage error naming what the tool accepts.
func resolveGOOS(name string) (string, error) {
lower := strings.ToLower(name)
if goOSNames[lower] {
return lower, nil
}
return "", &usageError{fmt.Errorf("unknown GOOS %q: want one of %s", name, strings.Join(slices.Sorted(maps.Keys(goOSNames)), ", "))}
}
// goosFromFilename returns the GOOS the file's name carries, by go/build's
// goodOSArchFile rule: the GOOS segment sits last, or last before the
// architecture segment (sys_darwin_arm64.s, vlop_arm.s carries none). An
// empty result means the name names no GOOS and the ambient one applies.
func goosFromFilename(path string) string {
base := path
if i := strings.LastIndexByte(base, '/'); i >= 0 {
base = base[i+1:]
}
base = strings.TrimSuffix(base, ".s")
// go/build ignores everything before the first underscore, so a GOOS
// segment is only ever looked for from there on.
i := strings.IndexByte(base, '_')
if i < 0 {
return ""
}
segs := strings.Split(base[i:], "_")
if n := len(segs); n >= 2 && goOSNames[segs[n-2]] && slices.Contains(goPortSuffixes, segs[n-1]) {
return segs[n-2]
}
if goOSNames[segs[len(segs)-1]] {
return segs[len(segs)-1]
}
return ""
}
// otherGOOSFile reports whether the file's name names a GOOS other than the
@@ -508,6 +576,57 @@ func otherGOOSFile(path string) bool {
return false
}
// buildConstraint returns the file's leading //go:build expression, or nil
// when the file carries none. The constraint governs the same header block
// go/build reads: blank lines and comments may precede it, and the first
// line that is neither ends the block. A constraint that does not parse
// narrows nothing, so the file stays in the attempted set: the audit must
// never exclude a file the toolchain would compile.
func buildConstraint(src string) constraint.Expr {
for line := range strings.SplitSeq(src, "\n") {
t := strings.TrimSpace(line)
switch {
case t == "":
continue
case strings.HasPrefix(t, "//"):
if constraint.IsGoBuild(t) {
e, err := constraint.Parse(t)
if err != nil {
return nil
}
return e
}
continue
default:
return nil
}
}
return nil
}
// unixOS is go/build's unixOS set: the GOOSes the unix build tag admits.
var unixOS = map[string]bool{
"aix": true, "android": true, "darwin": true, "dragonfly": true,
"freebsd": true, "hurd": true, "illumos": true, "ios": true,
"linux": true, "netbsd": true, "openbsd": true, "solaris": true,
}
// constraintTags answers the build tags a plain `go build` sets for a
// target: the GOOS and GOARCH, gc, and unix on the unix-like GOOSes. No
// experiment, sanitiser or cgo tag is ever true: the audit models the
// default build, and no GOROOT assembly file's constraint hinges on cgo.
func constraintTags(goarch, goos string) func(string) bool {
return func(tag string) bool {
switch tag {
case goarch, goos, "gc":
return true
case "unix":
return unixOS[goos]
}
return false
}
}
func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
files, err := asmFiles(root)
if err != nil {
@@ -525,8 +644,8 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
tallies[i] = &corpusTally{reasons: map[string]int{}, example: map[string]string{}}
}
// full is the north-star number: a file counts when every architecture
// its name allows assembles it.
full, generic, otherPort := 0, 0, 0
// its build admits assembles it.
full, generic, otherPort, narrowedCount, excluded := 0, 0, 0, 0, 0
// Header generation is created on first use, so a corpus with no
// go_asm.h includes never pays for a temp directory.
@@ -543,35 +662,92 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
return nil, err
}
// The GOOS the header generation type-checks under follows the
// file's name when the name carries one; the ambient GOOS is the
// honest guess otherwise (a build tag naming another GOOS is
// invisible to a file-name rule).
goos := goosFromFilename(path)
// The GOOS the header generation type-checks under follows the
// file's name when the name carries one; the ambient GOOS is the
// honest guess otherwise.
namedArch := arch.FromFilename(path)
var wanted []int // indexes into targets
if a := arch.FromFilename(path); a != arch.Unknown {
other := false
switch {
case namedArch != arch.Unknown:
for i, tg := range targets {
if tg.a == a {
if tg.a == namedArch {
wanted = append(wanted, i)
}
}
} else if otherPortFile(path) {
case otherPortFile(path):
// A file named for a Go port gasm does not support (arm,
// 386, s390x, ...) or for another GOOS is compiled by no
// supported-arch build, so it is neither generic nor a
// per-arch attempt: counting it as generic would make the
// headline unreachably low for reasons no supported target
// can fix.
other = true
otherPort++
} else {
generic++
default:
for i := range targets {
wanted = append(wanted, i)
}
}
// A //go:build constraint narrows the set of targets the file is
// assembled for, the way the go command compiles the file only for
// the targets the expression admits: cpu_x86.s belongs to the x86
// build alone, and a file whose constraint admits none of the four
// targets (the goexperiment.runtimesecret and msan trees) is
// compiled by no supported build. The tags mirror what a plain
// `go build` sets: the GOOS and GOARCH, gc, and unix on the
// unix-like GOOSes; no experiment, sanitiser or cgo tag is ever
// true. The GOOS is the file's own when the name carries one,
// else the ambient one.
goosForEval := goos
if goosForEval == "" {
goosForEval = runtime.GOOS
}
narrowed := false
if len(wanted) > 0 {
if ce := buildConstraint(src); ce != nil {
kept := make([]int, 0, len(wanted))
for _, i := range wanted {
tg := targets[i]
if ce.Eval(constraintTags(goarchName(tg.a), goosForEval)) {
kept = append(kept, i)
}
}
if len(kept) < len(wanted) {
narrowed = true
}
wanted = kept
}
}
switch {
case other:
// already tallied above
case len(wanted) == 0:
excluded++
case namedArch != arch.Unknown:
// a per-arch attempt over the constraint's subset
case narrowed:
narrowedCount++
default:
generic++
}
// A file that includes go_asm.h parses against a per-target header:
// the defines differ per architecture (internal/cpu's layout, for
// one), so the parse cannot be shared the way a header-free file's
// can. A generation failure is a failure for every target, named
// for the package rather than a bare "include not found". A header
// already resolvable in the package directory or the -I list is
// left alone.
// one) and per GOOS (sys_darwin_arm64.s's trampoline constants,
// for another), so the parse cannot be shared the way a
// header-free file's can. A generation failure is a failure for
// every target, named for the package rather than a bare "include
// not found". A header already resolvable in the package
// directory or the -I list is left alone.
if len(wanted) > 0 && needsGoAsmHeader(src) && !goAsmHeaderResolved(filepath.Dir(path), dirs) {
if hdr == nil {
if hdr, err = newAsmhdrCache(); err != nil {
@@ -583,24 +759,25 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
for _, i := range wanted {
tg, t := targets[i], tallies[i]
t.attempted++
hdrDir, err := hdr.dirFor(pkgDir, goarchName(tg.a))
hdrDir, err := hdr.dirFor(pkgDir, goos, goarchName(tg.a))
if err != nil {
ok = false
t.fail(path, corpusReason(err))
t.fail(path, err)
continue
}
f, errs := parser.ParseWithOptions(path, src, parser.Options{
Expand: true,
IncludeDirs: append(slices.Clone(dirs), hdrDir),
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
})
if len(errs) > 0 {
ok = false
t.fail(path, corpusReason(errs[0]))
t.fail(path, errs[0])
continue
}
if _, err := assembleFile(tg.a, f); err != nil {
if _, err := assembleFile(tg.a, f, goos); err != nil {
ok = false
t.fail(path, corpusReason(err))
t.fail(path, err)
continue
}
t.assembled++
@@ -611,21 +788,28 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
continue
}
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
ok := true
for _, i := range wanted {
tg, t := targets[i], tallies[i]
t.attempted++
// The parse carries the target's platform predefines, so it
// cannot be shared across targets the way a header-free file's
// could: a #ifdef GOARCH_arm block must be live on arm64 and
// dead everywhere else.
f, errs := parser.ParseWithOptions(path, src, parser.Options{
Expand: true,
IncludeDirs: dirs,
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
})
var err error
if len(errs) > 0 {
err = errs[0] // a parse failure is a failure for every target
} else {
_, err = assembleFile(tg.a, f)
_, err = assembleFile(tg.a, f, goos)
}
if err != nil {
ok = false
t.fail(path, corpusReason(err))
t.fail(path, err)
continue
}
t.assembled++
@@ -639,6 +823,8 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
root: root,
files: len(files),
generic: generic,
narrowed: narrowedCount,
excluded: excluded,
otherPort: otherPort,
full: full,
targets: targets,
@@ -647,14 +833,16 @@ func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
}
// printCorpusStats renders the corpus audit report.
func printCorpusStats(s *corpusStats) {
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d named for other Go ports, never attempted)\n", s.root, s.files, s.generic, s.otherPort)
func printCorpusStats(s *corpusStats, list bool) {
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d narrowed by //go:build; %d excluded by //go:build; %d named for other Go ports, never attempted)\n",
s.root, s.files, s.generic, s.narrowed, s.excluded, s.otherPort)
// The rate is over the files a supported build would attempt: the
// other ports' files sit in the count for completeness but can never
// assemble, so counting them in the denominator would report the gap
// of architectures gasm deliberately does not target.
attemptable := max(s.files-s.otherPort, 1)
fmt.Printf(" assemble for every target architecture: %d of %d attemptable (%.1f%%)\n", s.full, attemptable, 100*float64(s.full)/float64(attemptable))
// other ports' files and the ones no supported target compiles sit in
// the count for completeness but can never assemble, so counting them
// in the denominator would report the gap of platforms gasm
// deliberately does not target.
attemptable := max(s.files-s.otherPort-s.excluded, 1)
fmt.Printf(" assemble for every applicable target: %d of %d attemptable (%.1f%%)\n", s.full, attemptable, 100*float64(s.full)/float64(attemptable))
for i, tg := range s.targets {
t := s.tallies[i]
fmt.Printf(" %s: %d/%d attempted\n", tg.name, t.assembled, t.attempted)
@@ -662,6 +850,13 @@ func printCorpusStats(s *corpusStats) {
fmt.Printf(" %4d %s\n", t.reasons[r], r)
fmt.Printf(" e.g. %s\n", t.example[r])
}
if !list {
continue
}
for _, f := range t.fails {
fmt.Printf(" FAIL %s\n", f.path)
fmt.Printf(" %s: %s\n", f.reason, f.detail)
}
}
}
+61 -1
View File
@@ -4,9 +4,10 @@
package main
import (
"runtime"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
)
func TestDerivedFamily(t *testing.T) {
@@ -64,3 +65,62 @@ func TestGasmEncodable(t *testing.T) {
}
}
}
// TestBuildConstraint pins the //go:build reader: the constraint governs the
// leading comment block, the first non-comment line ends it (a tag below a
// #include governs nothing, exactly as go/build drops it), and a file
// without one admits every target.
func TestBuildConstraint(t *testing.T) {
admits := func(src, goarch, goos string) bool {
t.Helper()
e := buildConstraint(src)
if e == nil {
return true
}
return e.Eval(constraintTags(goarch, goos))
}
const ret = "TEXT \xc2\xb7f(SB), NOSPLIT, $0\n\tRET\n"
cases := []struct {
name string
src string
amd64, arm64 bool
}{
{"no constraint", ret, true, true},
{"x86 only", "//go:build 386 || amd64\n\n" + ret, true, false},
{"arm64 and linux", "//go:build arm64 && linux\n\n" + ret, false, true},
{"msan never", "//go:build msan\n\n" + ret, false, false},
{"experiment never", "//go:build goexperiment.runtimesecret\n\n" + ret, false, false},
{"below an include governs nothing", "#include \"textflag.h\"\n//go:build amd64\n" + ret, true, true},
{"unparsable narrows nothing", "//go:build (amd64\n" + ret, true, true},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
if got := admits(c.src, "amd64", runtime.GOOS); got != c.amd64 {
t.Errorf("amd64 admission = %v, want %v", got, c.amd64)
}
if got := admits(c.src, "arm64", runtime.GOOS); got != c.arm64 {
t.Errorf("arm64 admission = %v, want %v", got, c.arm64)
}
})
}
}
// TestConstraintTags pins the tag set a plain `go build` sets: the GOOS and
// GOARCH, gc, unix on the unix-like GOOSes; nothing else is ever true.
func TestConstraintTags(t *testing.T) {
ok := constraintTags("amd64", "linux")
for _, tag := range []string{"amd64", "linux", "gc", "unix"} {
if !ok(tag) {
t.Errorf("tag %q = false, want true", tag)
}
}
for _, tag := range []string{"arm64", "freebsd", "darwin", "cgo", "race", "msan", "goexperiment.runtimesecret"} {
if ok(tag) {
t.Errorf("tag %q = true, want false", tag)
}
}
fb := constraintTags("arm64", "freebsd")
if !fb("unix") {
t.Error("unix on freebsd = false, want true")
}
}
+2 -2
View File
@@ -1,7 +1,7 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build !linux
//go:build !(linux || (freebsd && (amd64 || arm64 || riscv64)))
package main
@@ -11,6 +11,6 @@ import (
)
func cmdDebug(args []string) int {
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux (ptrace)")
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux or FreeBSD (ptrace)")
return 1
}
@@ -1,7 +1,7 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
//go:build linux || (freebsd && (amd64 || arm64 || riscv64))
package main
@@ -13,8 +13,8 @@ import (
"strings"
"time"
"sourcedock.dev/petrbalvin/gasm-devkit/debug"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
"sourcedock.dev/petrbalvin/gasm-sdk/debug"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
func cmdDebug(args []string) int {
@@ -239,10 +239,12 @@ REPL commands:
fmt.Printf("gasm debug: cover: stopped on signal %v\n", sig)
break
}
reason, _ := sess.StopInfo()
regs, rerr := sess.GetRegs()
if rerr != nil {
break
}
trapPC := regs.GetPC()
// HandleTrap restores the original byte, rewinds PC and counts
// the hit on the breakpoint itself. Single-step over the
// restored instruction so the reinsertion at the top of the
@@ -251,6 +253,12 @@ REPL commands:
if err := sess.Step(); err != nil {
break
}
} else if debug.TrapStray(sess, reason, trapPC) {
// A breakpoint-class trap that matches none of ours and left
// the PC in place: resuming would re-execute the trapping
// instruction forever, so the coverage run stops here.
fmt.Printf("gasm debug: cover: SIGTRAP at %#x matches no breakpoint; the PC did not advance\n", trapPC)
break
}
}
hits := map[uint64]int{}
+4 -4
View File
@@ -9,9 +9,9 @@ import (
"sort"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// cmdDis disassembles machine code: either a raw binary (standard input with
@@ -84,7 +84,7 @@ func disSource(path string, target arch.Arch) int {
if len(errs) > 0 {
return 1
}
img, err := assembleFile(target, f)
img, err := assembleFile(target, f, "")
if err != nil {
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
return 1
+61 -25
View File
@@ -27,15 +27,15 @@ import (
"sync"
"syscall"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/format"
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
"sourcedock.dev/petrbalvin/gasm-devkit/lsp"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/format"
"sourcedock.dev/petrbalvin/gasm-sdk/lexer"
"sourcedock.dev/petrbalvin/gasm-sdk/lint"
"sourcedock.dev/petrbalvin/gasm-sdk/lsp"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
// version reports the release the toolchain recorded for this build: the
@@ -486,7 +486,7 @@ hover, document symbols, diagnostics and semantic-token highlighting.
}
func cmdAsm(args []string) int {
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>", `
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-GOOS os] [-o out] <file>", `
Assemble FILE without the Go toolchain: every TEXT function is encoded to
machine code and printed as a hex dump. Supported architectures: amd64
(including VEX/AVX2 and EVEX/AVX-512), arm64 (AArch64 integer, FP,
@@ -507,18 +507,22 @@ and the format version from go version).
A file that includes go_asm.h gets that header generated automatically from
the package it lives in (the .go files beside it, type-checked for the
target architecture, the toolchain's own defines), so GOROOT assembly
assembles without a compiler. A package that has no Go files for the
target or does not type-check is a hard error naming the package.
assembles without a compiler. -GOOS selects the type-checking GOOS for
that header: a GOOS-specific file (sys_darwin_arm64.s) needs its platform's
defines, which a header from the ambient GOOS silently omits. A package
that has no Go files for the target or does not type-check is a hard error
naming the package.
`)
out := fs.String("o", "", "write the output to this file")
format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
archName := fs.String("GOARCH", "", "target architecture: amd64, arm64, riscv64 or loong64 (overrides the file-name suffix)")
goosName := fs.String("GOOS", "", "operating system for go_asm.h generation: a GOOS go/build recognises (default: the host's)")
var dirs includeDirs
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>")
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-GOOS os] [-o out] <file>")
return 2
}
// The format is validated before anything else, so a bogus value exits 2
@@ -539,6 +543,15 @@ target or does not type-check is a hard error naming the package.
}
targetArch = a
}
goos := ""
if *goosName != "" {
g, err := resolveGOOS(*goosName)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm asm: %v\n", err)
return 2
}
goos = g
}
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
@@ -553,7 +566,7 @@ target or does not type-check is a hard error naming the package.
// A go_asm.h that already resolves (placed by hand, or passed with -I)
// is left alone.
if needsGoAsmHeader(src) && !goAsmHeaderResolved(filepath.Dir(path), dirs) {
hdrDir, cleanup, err := ensureGoAsmHeader(path, targetArch, nil)
hdrDir, cleanup, err := ensureGoAsmHeader(path, targetArch, goos, nil)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm asm:", err)
return 1
@@ -561,7 +574,7 @@ target or does not type-check is a hard error naming the package.
defer cleanup()
dirs = append(dirs, hdrDir)
}
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs, Predefines: platformPredefinesFor(string(targetArch), goos)})
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
@@ -569,7 +582,7 @@ target or does not type-check is a hard error naming the package.
return 1
}
img, err := assembleFile(targetArch, f)
img, err := assembleFile(targetArch, f, goos)
if err != nil {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
return 1
@@ -776,11 +789,34 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
return 1
}
// platformPredefines mirrors the go command's assembler invocation, which
// defines GOOS_<goos> and GOARCH_<arch> as -D macros: GOROOT headers
// (go_tls.h, asm_riscv64.h) select their platform blocks with #ifdef on
// exactly those names, so an assembler without them cannot see the platform
// definitions at all.
func platformPredefines(goarch, goos string) map[string]string {
return map[string]string{
"GOARCH_" + goarch: "1",
"GOOS_" + goos: "1",
}
}
// platformPredefinesFor resolves the ambient GOOS the way a build would: a
// file whose name carries one (sys_darwin_arm64.s) is compiled for that GOOS
// and nothing else.
func platformPredefinesFor(goarch string, fileGoos string) map[string]string {
goos := fileGoos
if goos == "" {
goos = runtime.GOOS
}
return platformPredefines(goarch, goos)
}
// assembleFile assembles a parsed file for the given architecture and returns the image.
func assembleFile(targetArch arch.Arch, f *ast.File) (*asm.Image, error) {
func assembleFile(targetArch arch.Arch, f *ast.File, goos string) (*asm.Image, error) {
switch targetArch {
case arch.AMD64:
return asm.AssembleFile(f)
return asm.AssembleFile(f, asm.WithGOOS(goos))
case arch.RISCV:
return asm.AssembleFileRISCV(f)
case arch.ARM64:
@@ -800,18 +836,18 @@ func assemblePath(path string, forced arch.Arch, dirs includeDirs) (*asm.Image,
if err != nil {
return nil, err
}
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
target := forced
if target == arch.Unknown {
target = arch.FromFilename(path)
}
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs, Predefines: platformPredefinesFor(string(target), "")})
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return nil, fmt.Errorf("parse errors")
}
target := forced
if target == arch.Unknown {
target = arch.FromFilename(path)
}
return assembleFile(target, f)
return assembleFile(target, f, "")
}
// printByteDiff shows the first few byte differences between two code blocks.
@@ -907,7 +943,7 @@ func cmdVerifyNonJIT(path string, targetArch arch.Arch, groundTruth, profile boo
if len(errs) > 0 {
return 1
}
img, err := assembleFile(targetArch, f)
img, err := assembleFile(targetArch, f, "")
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
+2 -2
View File
@@ -14,8 +14,8 @@ import (
"syscall"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
)
const clean = "#include \"textflag.h\"\n" +
+2 -2
View File
@@ -11,8 +11,8 @@ import (
"os"
"strings"
gasmast "sourcedock.dev/petrbalvin/gasm-devkit/ast"
gasmparser "sourcedock.dev/petrbalvin/gasm-devkit/parser"
gasmast "sourcedock.dev/petrbalvin/gasm-sdk/ast"
gasmparser "sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// cmdScaffold generates a differential test skeleton for every kernel in a
+31 -7
View File
@@ -1,13 +1,17 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
//go:build linux || (freebsd && (amd64 || arm64 || riscv64))
package debug
import "strings"
import "fmt"
import (
"cmp"
"fmt"
"slices"
)
// Breakpoint is one software breakpoint in the debuggee.
type Breakpoint struct {
@@ -149,15 +153,16 @@ func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (
return bp, nil
}
// Info returns a formatted list of all breakpoints.
// Info returns a formatted list of all breakpoints, ordered by address so
// the numbering is stable across calls (map iteration order is not).
func (bm *Breakpoints) Info() string {
if len(bm.bps) == 0 {
return "no breakpoints set\n"
}
var result strings.Builder
i := 0
for _, bp := range bm.bps {
i++
bps := bm.All()
slices.SortFunc(bps, func(a, b *Breakpoint) int { return cmp.Compare(a.Addr, b.Addr) })
for i, bp := range bps {
status := "enabled"
if !bp.Enabled {
status = "disabled"
@@ -170,7 +175,7 @@ func (bm *Breakpoints) Info() string {
if bp.Cond != nil {
cond = " if " + bp.Cond.String()
}
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond))
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i+1, label, bp.Addr, status, bp.hits, cond))
}
return result.String()
}
@@ -238,6 +243,25 @@ func (bm *Breakpoints) All() []*Breakpoint {
// Hits returns how many times the breakpoint has been hit.
func (bp *Breakpoint) Hits() int { return bp.hits }
// TrapStray reports whether a stop is a breakpoint-class trap that matches
// no breakpoint of ours and cannot be resumed: the PC still stands on the
// trapping instruction (the kernel's own BRK, EBREAK or break, on an
// architecture that reports the trap in place), so the next resume would
// re-execute it and trap forever. trapPC is the PC the stop reported,
// before any HandleTrap rewinding; reason is the stop's StopInfo class.
// The debuggee's SIGSTOP barriers also stop without PC movement, and they
// never carry the breakpoint class, so they are unaffected.
func TrapStray(s *Session, reason StopReason, trapPC uint64) bool {
if reason != StopBreakpoint {
return false
}
after, err := s.GetRegs()
if err != nil {
return false
}
return after.GetPC() <= trapPC-uint64(breakpointPCAdjust)
}
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
// On amd64 the kernel reports the trap with RIP past the INT3; on the
// other supported architectures the PC still stands on the trap
+44
View File
@@ -10,6 +10,8 @@ package debug
// build on every supported linux architecture.
import (
"fmt"
"slices"
"strings"
"testing"
)
@@ -263,3 +265,45 @@ func TestConditionString(t *testing.T) {
}
}
}
// TestBreakpointsInfoOrdered proves the listing is ordered by address: the
// numbers it prints are map keys rendered in iteration order otherwise, so
// the same set of breakpoints would renumber itself between calls.
func TestBreakpointsInfoOrdered(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
addrs := []uint64{0x9000, 0x1000, 0x7000, 0x3000, 0x8000, 0x2000,
0x6000, 0x4000, 0x5000, 0xa000}
for i, a := range addrs {
if _, err := bm.Set(a, fmt.Sprintf("bp%d", i)); err != nil {
t.Fatalf("Set(%#x): %v", a, err)
}
}
sorted := append([]uint64(nil), addrs...)
slices.Sort(sorted)
info := bm.Info()
for i, a := range sorted {
want := fmt.Sprintf(" %d: bp%d at %#x", i+1, indexOf(addrs, a), a)
if !strings.Contains(info, want) {
t.Errorf("Info() missing %q; listing:\n%s", want, info)
}
}
// The numbers themselves must ascend: "1:" before "2" ... "10".
pos := 0
for i := range len(addrs) {
next := strings.Index(info[pos:], fmt.Sprintf(" %d: ", i+1))
if next < 0 {
t.Fatalf("Info() has no entry %d; listing:\n%s", i+1, info)
}
pos += next
}
}
func indexOf(addrs []uint64, a uint64) int {
for i, v := range addrs {
if v == a {
return i
}
}
return -1
}
+325
View File
@@ -0,0 +1,325 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug
import (
"fmt"
"os"
"runtime"
"strings"
"testing"
"time"
)
// Regression tests for the debugger audit: memory access at mapping
// boundaries, watchpoint slot attribution, launch failure latency, stray
// trap instructions and the REPL's argument validation. All drive a real
// ptrace session, so they run on amd64 hosts only.
// memMap is one line of /proc/pid/maps.
type memMap struct {
lo, hi uint64
perms string
name string
}
// readMaps parses the debuggee's memory map.
func readMaps(t *testing.T, pid int) []memMap {
t.Helper()
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
if err != nil {
t.Fatalf("read maps: %v", err)
}
var out []memMap
for line := range strings.SplitSeq(string(data), "\n") {
fields := strings.Fields(line)
if len(fields) < 2 {
continue
}
var lo, hi uint64
if _, err := fmt.Sscanf(fields[0], "%x-%x", &lo, &hi); err != nil {
continue
}
m := memMap{lo: lo, hi: hi, perms: fields[1]}
if len(fields) >= 6 {
m.name = fields[5]
}
out = append(out, m)
}
return out
}
// boundaryByte returns the last byte of a writable, ordinary mapping that is
// followed by an unmapped gap: an access there is inside the mapping, while
// the 8-byte word starting at it crosses into unmapped memory.
func boundaryByte(t *testing.T, pid int) uint64 {
t.Helper()
maps := readMaps(t, pid)
for i, m := range maps {
if !strings.Contains(m.perms, "rw") ||
strings.Contains(m.name, "vvar") || strings.Contains(m.name, "vdso") ||
strings.Contains(m.name, "vsyscall") {
continue
}
gap := uint64(1) << 62
if i+1 < len(maps) {
gap = maps[i+1].lo - m.hi
}
if gap >= 4096 {
return m.hi - 1
}
}
t.Skip("no writable mapping followed by a hole; cannot construct the boundary")
return 0
}
// TestReadMemoryPageBoundary proves ReadMemory never reads past the requested
// range: one byte at the end of a mapping followed by a hole must be
// readable, which the old word-at-a-time tail read failed because its final
// 8-byte Peek crossed into the unmapped page.
func TestReadMemoryPageBoundary(t *testing.T) {
sess, _, _ := launchKernel(t, buildGasm(t), boundaryKernel(t), "boundary", nil)
addr := boundaryByte(t, sess.Pid())
mem, err := sess.ReadMemory(addr, 1)
if err != nil {
t.Fatalf("ReadMemory(%#x, 1): %v (the read must not cross into the unmapped page)", addr, err)
}
if len(mem) != 1 {
t.Fatalf("ReadMemory returned %d bytes, want 1", len(mem))
}
// A request whose own range crosses into the hole must still fail.
if _, err := sess.ReadMemory(addr, 8); err == nil {
t.Fatal("ReadMemory past the mapping end should fail")
}
}
// TestDisassemblePageBoundary proves the disassembler shrinks its read
// window at a mapping end instead of failing: the instruction stream cannot
// be decoded at all when the fixed 15-byte read crosses into the hole.
func TestDisassemblePageBoundary(t *testing.T) {
sess, _, _ := launchKernel(t, buildGasm(t), boundaryKernel(t), "boundary", nil)
addr := boundaryByte(t, sess.Pid())
if _, _, err := sess.Disassemble(addr); err != nil {
t.Fatalf("Disassemble(%#x): %v (the read window must shrink at the mapping end)", addr, err)
}
}
// TestWriteMemoryPageBoundary proves WriteMemory writes exactly the bytes it
// is given: one byte at the end of a mapping followed by a hole must be
// writable, which the old read-modify-write of the final partial word failed
// because its Peek crossed into the unmapped page.
func TestWriteMemoryPageBoundary(t *testing.T) {
sess, _, _ := launchKernel(t, buildGasm(t), boundaryKernel(t), "boundary", nil)
addr := boundaryByte(t, sess.Pid())
orig, err := sess.ReadMemory(addr, 1)
if err != nil {
t.Fatalf("ReadMemory(%#x, 1): %v", addr, err)
}
if err := sess.WriteMemory(addr, []byte{orig[0]}); err != nil {
t.Fatalf("WriteMemory(%#x, 1): %v (the write must not read past the range)", addr, err)
}
}
// TestWatchpointSlotAttribution proves a hit is attributed to the slot that
// fired, not to an earlier one whose DR6 status bit is still set: the B0-B3
// bits are sticky, so they must be acknowledged when read.
func TestWatchpointSlotAttribution(t *testing.T) {
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func wptwo(x, y int64) (a, b int64)
TEXT ·wptwo(SB), NOSPLIT, $0-32
MOVQ $0x1111, AX
MOVQ AX, a+16(FP)
MOVQ $0x2222, BX
MOVQ BX, b+24(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "wptwo", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
// FP sits one word above the entry stack pointer (the return address
// occupies [RSP]), so a+16(FP) = RSP+24 and b+24(FP) = RSP+32.
watchA := regs.RSP + 24
watchB := regs.RSP + 32
if err := sess.SetWatchpoint(0, watchA, WatchWrite, 8); err != nil {
t.Fatalf("SetWatchpoint(0): %v", err)
}
if err := sess.SetWatchpoint(1, watchB, WatchWrite, 8); err != nil {
t.Fatalf("SetWatchpoint(1): %v", err)
}
for i, want := range []uint64{watchA, watchB} {
if err := sess.Continue(); err != nil {
t.Fatalf("Continue (hit %d): %v", i+1, err)
}
reason, addr := sess.StopInfo()
if reason != StopWatchpoint {
t.Fatalf("hit %d: stop reason = %v, want StopWatchpoint", i+1, reason)
}
if addr != want {
t.Fatalf("hit %d reported %#x, want %#x (the sticky DR6 bit misattributes the slot)", i+1, addr, want)
}
}
// Clearing a watchpoint must zero its address register: a stale
// address in a disabled slot turns any sticky status bit into a
// misattributed report later.
if err := sess.ClearWatchpoint(0); err != nil {
t.Fatalf("ClearWatchpoint(0): %v", err)
}
dr0, err := ptracePeekUser(sess.Pid(), drOffset)
if err != nil {
t.Fatalf("read DR0: %v", err)
}
if dr0 != 0 {
t.Fatalf("DR0 = %#x after ClearWatchpoint, want 0 (the address register must be cleared)", dr0)
}
}
// TestLaunchFailsFastOnDeadDebuggee proves a debuggee that dies before
// signalling readiness surfaces promptly: the ready poll used to run its
// full 2.5 seconds before the wait discovered the exit.
func TestLaunchFailsFastOnDeadDebuggee(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
path := boundaryKernel(t)
start := time.Now()
sess, err := Launch(bin, path, "nosuchfunction", nil)
elapsed := time.Since(start)
if err == nil {
sess.Kill()
t.Fatal("Launch with an unknown function should fail")
}
if !strings.Contains(err.Error(), "before signalling readiness") &&
!strings.Contains(err.Error(), "debuggee exited") {
t.Errorf("error does not name the dead debuggee: %v", err)
}
if elapsed >= 1500*time.Millisecond {
t.Fatalf("Launch took %v to report the dead debuggee; the readiness poll must detect the exit, not time out", elapsed)
}
}
// TestStrayTrapRunsThrough proves the continue loop survives a trap
// instruction planted in the kernel itself (BYTE $0xCC, the same byte the
// debugger patches in): on architectures that report the trap in place the
// loop must surface the stop, and on amd64 it runs through to the exit. A
// regression here hangs, so a watchdog fails the run.
func TestStrayTrapRunsThrough(t *testing.T) {
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func stray() int64
TEXT ·stray(SB), NOSPLIT, $0-8
MOVQ $7, AX
BYTE $0xCC
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, _ := launchKernel(t, bin, path, "stray", nil)
timer := time.AfterFunc(time.Minute, func() {
panic("watchdog: the continue loop hung on the stray trap instruction")
})
defer timer.Stop()
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), 0, 0, 0, nil, nil,
strings.NewReader("continue\nquit\n"))
})
if !strings.Contains(out, "debuggee exited") {
t.Errorf("the stray trap wedged the continue loop; output:\n%s", out)
}
}
// TestStepIntoFaultReportsSignal proves the step command reports a genuine
// signal-delivery-stop instead of silently printing the faulting
// instruction as if the step had succeeded.
func TestStepIntoFaultReportsSignal(t *testing.T) {
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func crash() int64
TEXT ·crash(SB), NOSPLIT, $0-8
XORQ AX, AX
MOVQ (AX), AX
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "crash", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, nil, nil,
strings.NewReader("step 2\nquit\n"))
})
if !strings.Contains(out, "stopped on signal") {
t.Errorf("stepping into the fault did not report the signal; output:\n%s", out)
}
}
// TestREPLRejectsBadArguments proves the command loop reports malformed
// input instead of silently defaulting: an unknown label for x would read
// address 0, and a malformed count would silently step one instruction.
func TestREPLRejectsBadArguments(t *testing.T) {
bin := buildGasm(t)
path := boundaryKernel(t)
sess, bm, fl := launchKernel(t, bin, path, "boundary", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, nil, nil,
strings.NewReader("x nosuchlabel\nstep abc\ndisas abc\nwatch 0x1000 q 8\nquit\n"))
})
for _, want := range []string{
"unknown address: nosuchlabel",
"invalid count: abc",
"unknown watchpoint type: q",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
if got := strings.Count(out, "invalid count: abc"); got != 2 {
t.Errorf("invalid count reported %d times, want 2 (step and disas):\n%s", got, out)
}
}
// boundaryKernel is a minimal kernel for the boundary tests, which only need
// a live, stopped debuggee.
func boundaryKernel(t *testing.T) string {
t.Helper()
const kernel = `#include "textflag.h"
// func boundary() int64
TEXT ·boundary(SB), NOSPLIT, $0-8
MOVQ $1, AX
MOVQ AX, ret+0(FP)
RET
`
return writeKernel(t, kernel)
}
+65
View File
@@ -0,0 +1,65 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes. An amd64
// instruction is up to 15 bytes long, but the read must not reach past the
// end of the mapping: when the full 15-byte window crosses into unmapped
// memory the window shrinks, because an instruction at the mapping's end is
// by construction no longer than the readable bytes that hold it.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
var lastErr error
for _, n := range []int{15, 8, 4, 2, 1} {
mem, err := s.ReadMemory(addr, n)
if err != nil {
lastErr = err
continue
}
ins, derr := disasm.Decode(arch.AMD64, mem, addr)
if derr != nil {
return "", 0, derr
}
return ins.Text, ins.Len, nil
}
return "", 0, lastErr
}
// DisassembleN decodes up to n instructions starting at addr and returns
// them as a formatted string with addresses and byte offsets.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result strings.Builder
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
break
}
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
if length == 0 {
length = 1
}
pc += uint64(length)
}
return result.String()
}
// isCallInsn reports whether disassembled text (x86asm.IntelSyntax) is a
// call. The first token must match exactly: a prefix test would also catch
// unrelated mnemonics.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
return strings.ToLower(m) == "call"
}
+60
View File
@@ -0,0 +1,60 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && arm64
package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 4)
if err != nil {
return "", 0, err
}
ins, err := disasm.Decode(arch.ARM64, mem, addr)
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr and returns
// them as a formatted string with addresses and byte offsets.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result strings.Builder
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
break
}
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
if length == 0 {
length = 1
}
pc += uint64(length)
}
return result.String()
}
// isCallInsn reports whether disassembled text (arm64asm.GoSyntax) is a
// call. GoSyntax renders bl as CALL; the native mnemonic is accepted too.
// The first token must match exactly so branches never match.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "bl":
return true
}
return false
}
+70
View File
@@ -0,0 +1,70 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && riscv64
package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes. The read
// shrinks from 4 to 2 bytes when the full word crosses into unmapped memory:
// a compressed instruction at the mapping's end still fits the shorter
// window, and an instruction can never extend past the mapping that holds it.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
var lastErr error
for _, n := range []int{4, 2} {
mem, err := s.ReadMemory(addr, n)
if err != nil {
lastErr = err
continue
}
ins, derr := disasm.Decode(arch.RISCV, mem, addr)
if derr != nil {
return "", 0, derr
}
return ins.Text, ins.Len, nil
}
return "", 0, lastErr
}
// DisassembleN decodes up to n instructions starting at addr and returns
// them as a formatted string with addresses and byte offsets.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result strings.Builder
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
break
}
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
if length == 0 {
length = 1
}
pc += uint64(length)
}
return result.String()
}
// isCallInsn reports whether disassembled text (riscv64asm.GoSyntax) is a
// call. GoSyntax renders jal and jalr calls as CALL; the native mnemonics
// are accepted too. The first token must match exactly: a prefix test on
// "bl" would catch branches on other architectures, and jalr as ret prints
// RET, which must not be stepped over.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "jal", "jalr":
return true
}
return false
}
+20 -11
View File
@@ -9,22 +9,31 @@ import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
// memory and returns its text representation and length in bytes. An amd64
// instruction is up to 15 bytes long, but the read must not reach past the
// end of the mapping: when the full 15-byte window crosses into unmapped
// memory the window shrinks, because an instruction at the mapping's end is
// by construction no longer than the readable bytes that hold it.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 15)
if err != nil {
return "", 0, err
var lastErr error
for _, n := range []int{15, 8, 4, 2, 1} {
mem, err := s.ReadMemory(addr, n)
if err != nil {
lastErr = err
continue
}
ins, derr := disasm.Decode(arch.AMD64, mem, addr)
if derr != nil {
return "", 0, derr
}
return ins.Text, ins.Len, nil
}
ins, err := disasm.Decode(arch.AMD64, mem, addr)
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
return "", 0, lastErr
}
// DisassembleN decodes up to n instructions starting at addr and returns
+2 -2
View File
@@ -9,8 +9,8 @@ import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
+2 -2
View File
@@ -9,8 +9,8 @@ import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
+19 -11
View File
@@ -9,22 +9,30 @@ import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
// memory and returns its text representation and length in bytes. The read
// shrinks from 4 to 2 bytes when the full word crosses into unmapped memory:
// a compressed instruction at the mapping's end still fits the shorter
// window, and an instruction can never extend past the mapping that holds it.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 4)
if err != nil {
return "", 0, err
var lastErr error
for _, n := range []int{4, 2} {
mem, err := s.ReadMemory(addr, n)
if err != nil {
lastErr = err
continue
}
ins, derr := disasm.Decode(arch.RISCV, mem, addr)
if derr != nil {
return "", 0, derr
}
return ins.Text, ins.Len, nil
}
ins, err := disasm.Decode(arch.RISCV, mem, addr)
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
return "", 0, lastErr
}
// DisassembleN decodes up to n instructions starting at addr.
+136
View File
@@ -0,0 +1,136 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
import "fmt"
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" RIP = %#016x (func+%#x)\n", regs.RIP, regs.RIP-codeBase-funcOff)
fmt.Printf(" RSP = %#016x RBP = %#016x\n", regs.RSP, regs.RBP)
fmt.Printf(" RAX = %#016x RBX = %#016x\n", regs.RAX, regs.RBX)
fmt.Printf(" RCX = %#016x RDX = %#016x\n", regs.RCX, regs.RDX)
fmt.Printf(" RSI = %#016x RDI = %#016x\n", regs.RSI, regs.RDI)
fmt.Printf(" R8 = %#016x R9 = %#016x\n", regs.R8, regs.R9)
fmt.Printf(" R10 = %#016x R11 = %#016x\n", regs.R10, regs.R11)
fmt.Printf(" R12 = %#016x R13 = %#016x\n", regs.R12, regs.R13)
fmt.Printf(" R14 = %#016x R15 = %#016x\n", regs.R14, regs.R15)
fmt.Printf(" RFLAGS = %#x [%s]\n", regs.RFLAGS, decodeRflags(regs.RFLAGS))
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n Vector registers (YMM):")
for i := 0; i < 16; i += 2 {
fmt.Printf(" YMM%-2d = ", i)
printYMM(v.YMM[i][:])
fmt.Printf(" YMM%-2d = ", i+1)
printYMM(v.YMM[i+1][:])
fmt.Println()
}
}
func printYMM(b []byte) {
for j := 0; j < 32; j += 4 {
v := uint32(b[j]) | uint32(b[j+1])<<8 | uint32(b[j+2])<<16 | uint32(b[j+3])<<24
fmt.Printf("%08x ", v)
}
}
func decodeRflags(f uint64) string {
var flags string
if f&1 != 0 {
flags += "CF "
}
if f&(1<<2) != 0 {
flags += "PF "
}
if f&(1<<4) != 0 {
flags += "AF "
}
if f&(1<<6) != 0 {
flags += "ZF "
}
if f&(1<<7) != 0 {
flags += "SF "
}
if f&(1<<8) != 0 {
flags += "TF "
}
if f&(1<<9) != 0 {
flags += "IF "
}
if f&(1<<10) != 0 {
flags += "DF "
}
if f&(1<<11) != 0 {
flags += "OF "
}
if flags == "" {
return "none"
}
return flags[:len(flags)-1]
}
// SetReg modifies a register value in the debuggee.
func (s *Session) SetReg(name string, value uint64) error {
regs, err := s.GetRegs()
if err != nil {
return err
}
switch name {
case "rax", "eax", "ax", "al":
regs.RAX = value
case "rbx", "ebx", "bx", "bl":
regs.RBX = value
case "rcx", "ecx", "cx", "cl":
regs.RCX = value
case "rdx", "edx", "dx", "dl":
regs.RDX = value
case "rsi", "esi", "si":
regs.RSI = value
case "rdi", "edi", "di":
regs.RDI = value
case "rbp", "ebp", "bp":
regs.RBP = value
case "rsp", "esp", "sp":
regs.RSP = value
case "r8":
regs.R8 = value
case "r9":
regs.R9 = value
case "r10":
regs.R10 = value
case "r11":
regs.R11 = value
case "r12":
regs.R12 = value
case "r13":
regs.R13 = value
case "r14":
regs.R14 = value
case "r15":
regs.R15 = value
case "rip", "eip":
regs.RIP = value
default:
return fmt.Errorf("debug: unknown register %q", name)
}
return s.SetRegs(&regs)
}
// archReturnAddr reads the return address of the current frame (amd64
// ABI0 convention). A function that contains a CALL (or has a frame) is
// assembled with the prologue PUSHQ BP; MOVQ SP, BP, so mid-function the
// word at SP is the saved caller BP, a stack address, and the return
// address sits further up. Walk the stack from SP and take the first word
// that lies in an executable mapping: stack and data words never do, a
// return address always does. FreeBSD exposes no mapping list, so the
// walk degenerates to the raw entry convention, [SP] before any push.
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return s.Peek(regs.RSP)
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "RSP" }
+127
View File
@@ -0,0 +1,127 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && arm64
package debug
import (
"encoding/binary"
"fmt"
)
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.SP, regs.X29)
fmt.Printf(" LR = %#016x\n", regs.X30)
fmt.Printf(" X0 = %#016x X1 = %#016x\n", regs.X0, regs.X1)
fmt.Printf(" X2 = %#016x X3 = %#016x\n", regs.X2, regs.X3)
fmt.Printf(" X4 = %#016x X5 = %#016x\n", regs.X4, regs.X5)
fmt.Printf(" X6 = %#016x X7 = %#016x\n", regs.X6, regs.X7)
fmt.Printf(" X8 = %#016x X9 = %#016x\n", regs.X8, regs.X9)
fmt.Printf(" X10 = %#016x X11 = %#016x\n", regs.X10, regs.X11)
fmt.Printf(" X12 = %#016x X13 = %#016x\n", regs.X12, regs.X13)
fmt.Printf(" X14 = %#016x X15 = %#016x\n", regs.X14, regs.X15)
fmt.Printf(" X16 = %#016x X17 = %#016x\n", regs.X16, regs.X17)
fmt.Printf(" X18 = %#016x X19 = %#016x\n", regs.X18, regs.X19)
fmt.Printf(" X20 = %#016x X21 = %#016x\n", regs.X20, regs.X21)
fmt.Printf(" X22 = %#016x X23 = %#016x\n", regs.X22, regs.X23)
fmt.Printf(" X24 = %#016x X25 = %#016x\n", regs.X24, regs.X25)
fmt.Printf(" X26 = %#016x X27 = %#016x\n", regs.X26, regs.X27)
fmt.Printf(" X28 = %#016x PSTATE = %#x\n", regs.X28, regs.PSTATE)
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n Vector registers (V0-V31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" V%-2d = %016x%016x\n", i, binary.LittleEndian.Uint64(v.V[i][8:16]), binary.LittleEndian.Uint64(v.V[i][0:8]))
fmt.Printf(" V%-2d = %016x%016x\n", i+1, binary.LittleEndian.Uint64(v.V[i+1][8:16]), binary.LittleEndian.Uint64(v.V[i+1][0:8]))
}
}
// SetReg modifies a register value in the debuggee.
func (s *Session) SetReg(name string, value uint64) error {
regs, err := s.GetRegs()
if err != nil {
return err
}
switch name {
case "x0":
regs.X0 = value
case "x1":
regs.X1 = value
case "x2":
regs.X2 = value
case "x3":
regs.X3 = value
case "x4":
regs.X4 = value
case "x5":
regs.X5 = value
case "x6":
regs.X6 = value
case "x7":
regs.X7 = value
case "x8":
regs.X8 = value
case "x9":
regs.X9 = value
case "x10":
regs.X10 = value
case "x11":
regs.X11 = value
case "x12":
regs.X12 = value
case "x13":
regs.X13 = value
case "x14":
regs.X14 = value
case "x15":
regs.X15 = value
case "x16":
regs.X16 = value
case "x17":
regs.X17 = value
case "x18":
regs.X18 = value
case "x19":
regs.X19 = value
case "x20":
regs.X20 = value
case "x21":
regs.X21 = value
case "x22":
regs.X22 = value
case "x23":
regs.X23 = value
case "x24":
regs.X24 = value
case "x25":
regs.X25 = value
case "x26":
regs.X26 = value
case "x27":
regs.X27 = value
case "x28":
regs.X28 = value
case "x29", "fp":
regs.X29 = value
case "x30", "lr":
regs.X30 = value
case "sp":
regs.SP = value
case "pc":
regs.PC = value
default:
return fmt.Errorf("debug: unknown register %q", name)
}
return s.SetRegs(&regs)
}
// archReturnAddr reads the return address from LR (arm64 convention).
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return regs.X30, nil
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "SP" }
+121
View File
@@ -0,0 +1,121 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && riscv64
package debug
import "fmt"
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.Sp, regs.S0)
fmt.Printf(" RA = %#016x\n", regs.Ra)
fmt.Printf(" A0 = %#016x A1 = %#016x\n", regs.A0, regs.A1)
fmt.Printf(" A2 = %#016x A3 = %#016x\n", regs.A2, regs.A3)
fmt.Printf(" A4 = %#016x A5 = %#016x\n", regs.A4, regs.A5)
fmt.Printf(" A6 = %#016x A7 = %#016x\n", regs.A6, regs.A7)
fmt.Printf(" T0 = %#016x T1 = %#016x\n", regs.T0, regs.T1)
fmt.Printf(" T2 = %#016x T3 = %#016x\n", regs.T2, regs.T3)
fmt.Printf(" T4 = %#016x T5 = %#016x\n", regs.T4, regs.T5)
fmt.Printf(" T6 = %#016x\n", regs.T6)
fmt.Printf(" S1 = %#016x S2 = %#016x\n", regs.S1, regs.S2)
fmt.Printf(" S3 = %#016x S4 = %#016x\n", regs.S3, regs.S4)
fmt.Printf(" S5 = %#016x S6 = %#016x\n", regs.S5, regs.S6)
fmt.Printf(" S7 = %#016x S8 = %#016x\n", regs.S7, regs.S8)
fmt.Printf(" S9 = %#016x S10 = %#016x\n", regs.S9, regs.S10)
fmt.Printf(" S11 = %#016x\n", regs.S11)
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n FP registers (F0-F31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" F%-2d = %#018x F%-2d = %#018x\n", i, v.F[i], i+1, v.F[i+1])
}
fmt.Printf(" FCSR = %#x\n", v.FCSR)
}
// SetReg modifies a register value in the debuggee.
func (s *Session) SetReg(name string, value uint64) error {
regs, err := s.GetRegs()
if err != nil {
return err
}
switch name {
case "pc":
regs.PC = value
case "ra", "x1":
regs.Ra = value
case "sp", "x2":
regs.Sp = value
case "gp", "x3":
regs.Gp = value
case "tp", "x4":
regs.Tp = value
case "t0", "x5":
regs.T0 = value
case "t1", "x6":
regs.T1 = value
case "t2", "x7":
regs.T2 = value
case "s0", "fp", "x8":
regs.S0 = value
case "s1", "x9":
regs.S1 = value
case "a0", "x10":
regs.A0 = value
case "a1", "x11":
regs.A1 = value
case "a2", "x12":
regs.A2 = value
case "a3", "x13":
regs.A3 = value
case "a4", "x14":
regs.A4 = value
case "a5", "x15":
regs.A5 = value
case "a6", "x16":
regs.A6 = value
case "a7", "x17":
regs.A7 = value
case "s2", "x18":
regs.S2 = value
case "s3", "x19":
regs.S3 = value
case "s4", "x20":
regs.S4 = value
case "s5", "x21":
regs.S5 = value
case "s6", "x22":
regs.S6 = value
case "s7", "x23":
regs.S7 = value
case "s8", "x24":
regs.S8 = value
case "s9", "x25":
regs.S9 = value
case "s10", "x26":
regs.S10 = value
case "s11", "x27":
regs.S11 = value
case "t3", "x28":
regs.T3 = value
case "t4", "x29":
regs.T4 = value
case "t5", "x30":
regs.T5 = value
case "t6", "x31":
regs.T6 = value
default:
return fmt.Errorf("debug: unknown register %q", name)
}
return s.SetRegs(&regs)
}
// archReturnAddr reads the return address from RA (riscv64 convention).
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return regs.Ra, nil
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "SP" }
+2 -2
View File
@@ -17,8 +17,8 @@ import (
"time"
"unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
// Integration tests beyond the basic entry breakpoint: hardware watchpoints,
+310
View File
@@ -0,0 +1,310 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && (amd64 || arm64 || riscv64)
package debug
import (
"fmt"
"os"
"os/exec"
"path/filepath"
"runtime"
"strings"
"syscall"
"time"
"golang.org/x/sys/unix"
)
// Session is a ptrace debugging session controlling one debuggee process.
// The FreeBSD implementation sits behind the same surface as the Linux one:
// PT_TRACE_ME from the debuggee, PT_CONTINUE/PT_STEP from the tracer, and
// tracee memory through PT_IO (FreeBSD has no /proc/pid/mem to fall back
// on, so PT_IO is the only supported route).
type Session struct {
pid int
cmd *exec.Cmd
stopped bool
exited bool
codeBase uint64 // base address of the JIT code in the debuggee
tmpDir string // scratch directory of the session, removed on Kill
wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 dbw 0-15)
// lastSignal holds the signal of the most recent stop when that stop
// was a genuine signal-delivery-stop the caller must see (a fault such
// as SIGSEGV, SIGBUS, SIGFPE or SIGILL); 0 for breakpoint traps,
// single-steps, SIGSTOP and suppressed runtime signals.
lastSignal syscall.Signal
}
// Launch starts the debuggee subprocess (gasm debug --target ...) and
// attaches to it via ptrace.
func Launch(gasmBin, asmPath, funcName string, args []byte) (*Session, error) {
sess, _, err := LaunchWithBuffers(gasmBin, asmPath, funcName, args, "")
return sess, err
}
// LaunchWithBuffers is like Launch but also allocates buffers in the debuggee.
//
// It pins the calling goroutine to its OS thread and leaves it pinned: the
// debuggee's PT_TRACE_ME binds the tracer relation to the forking thread,
// and every ptrace request on the session must come from that same thread.
// All Session methods must therefore be called from the goroutine that
// launched the session (the REPL and coverage loops do exactly that).
func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec string) (*Session, []uint64, error) {
runtime.LockOSThread() // ptrace requests must stay on the forking thread
self, err := os.Executable()
if err != nil {
return nil, nil, fmt.Errorf("debug: cannot find gasm binary: %w", err)
}
if gasmBin != "" {
self = gasmBin
}
tmpDir, err := os.MkdirTemp("", "gasm-debug-*")
if err != nil {
return nil, nil, fmt.Errorf("debug: tempdir: %w", err)
}
argsFile := filepath.Join(tmpDir, "args.bin")
if err := os.WriteFile(argsFile, args, 0o644); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: write args: %w", err)
}
if bufSpec != "" {
if err := os.WriteFile(filepath.Join(tmpDir, "bufspec"), []byte(bufSpec), 0o644); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: write bufspec: %w", err)
}
}
cmd := exec.Command(self, "debug", "--func", funcName, "--args", argsFile, asmPath)
cmd.Env = append(os.Environ(), "GASM_DEBUG_TARGET=1", "GASM_DEBUG_TMP="+tmpDir)
cmd.Stdout = nil
cmd.Stderr = os.Stderr
cmd.SysProcAttr = &syscall.SysProcAttr{}
if err := cmd.Start(); err != nil {
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: start debuggee: %w", err)
}
s := &Session{pid: cmd.Process.Pid, cmd: cmd, tmpDir: tmpDir}
readyFile := filepath.Join(tmpDir, "ready")
for range 500 {
if _, err := os.Stat(readyFile); err == nil {
break
}
// A debuggee that died before signalling readiness (unknown
// function, unparseable source) writes its failure notice to the
// handshake directory; read it and fail fast. The poll never
// waits on the child: a wait here could consume the SIGSTOP park
// that waitStopped below must receive, hanging the launch.
if err := s.deadReason(); err != nil {
cmd.Wait()
os.RemoveAll(tmpDir)
return nil, nil, err
}
time.Sleep(5 * time.Millisecond)
}
// The debuggee parks itself with SIGSTOP once the JIT code is mapped.
// A Go tracee also reports SIGURG preemption as signal-delivery-stops,
// so the wait loops until a stop the debugger cares about instead of
// assuming the first event is the SIGSTOP.
if _, err := s.waitStopped(); err != nil {
cmd.Process.Kill()
os.RemoveAll(tmpDir)
return nil, nil, fmt.Errorf("debug: wait for debuggee: %w", err)
}
s.stopped = true
// The debuggee reports its JIT mapping in the codebase file; that is
// the supported path on FreeBSD, where no /proc/pid/maps exists to
// scan for the RWX region as a fallback.
if data, err := os.ReadFile(filepath.Join(tmpDir, "codebase")); err == nil {
fmt.Sscanf(string(data), "%d", &s.codeBase)
}
var bufAddrs []uint64
if bufSpec != "" {
addrFile := filepath.Join(tmpDir, "bufaddrs")
if data, err := os.ReadFile(addrFile); err == nil {
for line := range strings.SplitSeq(strings.TrimSpace(string(data)), "\n") {
var addr uint64
if _, err := fmt.Sscanf(line, "%d", &addr); err == nil {
bufAddrs = append(bufAddrs, addr)
}
}
}
}
return s, bufAddrs, nil
}
// deadReason reports the debuggee's own failure notice, the file its
// failure paths write before exiting. A debuggee killed without a notice
// (a crash, SIGKILL) surfaces through waitStopped after the poll instead,
// which is why the poll's budget stays finite.
func (s *Session) deadReason() error {
data, err := os.ReadFile(filepath.Join(s.tmpDir, "dead"))
if err != nil {
return nil
}
return fmt.Errorf("debug: debuggee failed before signalling readiness: %s", strings.TrimSpace(string(data)))
}
// waitStopped consumes ptrace-stop events until one the debugger cares
// about arrives: SIGTRAP (a breakpoint or a completed single-step), the
// debuggee's own SIGSTOP, or a genuine signal-delivery-stop. A Go tracee's
// runtime raises SIGURG for asynchronous preemption, and every signal on a
// traced thread surfaces as a signal-delivery-stop, so SIGURG is suppressed
// and the tracee resumed without it. Every other signal (SIGSEGV, SIGBUS,
// SIGFPE, SIGILL, ...) is returned to the caller: resuming with signal 0
// would restart the faulting instruction and fault forever, so a faulting
// kernel must surface as a stop the caller reports.
func (s *Session) waitStopped() (syscall.Signal, error) {
for {
var ws syscall.WaitStatus
if _, err := syscall.Wait4(s.pid, &ws, syscall.WUNTRACED, nil); err != nil {
return 0, err
}
if ws.Exited() {
s.exited = true
return 0, fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
}
if ws.Signaled() {
s.exited = true
return 0, fmt.Errorf("debuggee killed by signal %v", ws.Signal())
}
switch sig := ws.StopSignal(); sig {
case syscall.SIGTRAP, syscall.SIGSTOP:
s.stopped = true
s.lastSignal = 0
return sig, nil
case syscall.SIGURG:
// Go runtime asynchronous preemption: resume the tracee
// without delivering the signal.
s.lastSignal = 0
if err := unix.PtraceCont(s.pid, 0); err != nil {
return 0, fmt.Errorf("debug: PT_CONTINUE: %w", err)
}
default:
// A genuine signal-delivery-stop. Report it; the caller
// decides how to proceed.
s.stopped = true
s.lastSignal = sig
return sig, nil
}
}
}
// LastSignal returns the signal of the most recent stop when that stop was
// a genuine signal-delivery-stop (a fault such as SIGSEGV, SIGFPE, SIGILL
// or SIGBUS), and 0 for breakpoint traps, single-steps, SIGSTOP and
// suppressed runtime signals.
func (s *Session) LastSignal() syscall.Signal { return s.lastSignal }
// Peek reads a word (8 bytes) from the debuggee's memory at addr, through
// PT_IO with PIOD_READ_D.
func (s *Session) Peek(addr uint64) (uint64, error) {
var buf [8]byte
if _, err := unix.PtraceIO(unix.PIOD_READ_D, s.pid, uintptr(addr), buf[:], len(buf)); err != nil {
return 0, fmt.Errorf("debug: read mem %#x: %w", addr, err)
}
return uint64(buf[0]) | uint64(buf[1])<<8 | uint64(buf[2])<<16 | uint64(buf[3])<<24 |
uint64(buf[4])<<32 | uint64(buf[5])<<40 | uint64(buf[6])<<48 | uint64(buf[7])<<56, nil
}
// Poke writes a word (8 bytes) to the debuggee's memory at addr, through
// PT_IO with PIOD_WRITE_D.
func (s *Session) Poke(addr, val uint64) error {
buf := []byte{byte(val), byte(val >> 8), byte(val >> 16), byte(val >> 24),
byte(val >> 32), byte(val >> 40), byte(val >> 48), byte(val >> 56)}
if _, err := unix.PtraceIO(unix.PIOD_WRITE_D, s.pid, uintptr(addr), buf, len(buf)); err != nil {
return fmt.Errorf("debug: write mem %#x: %w", addr, err)
}
return nil
}
// ReadMemory reads len bytes from the debuggee's memory at addr in one
// PT_IO request, the shape the request is built for.
func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) {
out := make([]byte, length)
n, err := unix.PtraceIO(unix.PIOD_READ_D, s.pid, uintptr(addr), out, length)
return out[:n], err
}
// WriteMemory writes bytes to the debuggee's memory at addr in one PT_IO
// request.
func (s *Session) WriteMemory(addr uint64, data []byte) error {
_, err := unix.PtraceIO(unix.PIOD_WRITE_D, s.pid, uintptr(addr), data, len(data))
return err
}
// Step executes a single instruction in the debuggee.
func (s *Session) Step() error {
if s.exited {
return fmt.Errorf("debug: debuggee has exited")
}
if err := unix.PtraceSingleStep(s.pid); err != nil {
return fmt.Errorf("debug: PT_STEP: %w", err)
}
_, err := s.waitStopped()
return err
}
// Continue resumes execution until the next breakpoint or exit.
func (s *Session) Continue() error {
if s.exited {
return fmt.Errorf("debug: debuggee has exited")
}
if err := unix.PtraceCont(s.pid, 0); err != nil {
return fmt.Errorf("debug: PT_CONTINUE: %w", err)
}
_, err := s.waitStopped()
return err
}
// Exited returns true if the debuggee has terminated.
func (s *Session) Exited() bool { return s.exited }
// Pid returns the debuggee's process ID.
func (s *Session) Pid() int { return s.pid }
// CodeBase returns the base address of the JIT code in the debuggee.
func (s *Session) CodeBase() uint64 { return s.codeBase }
// Kill terminates the debuggee and removes the session's scratch
// directory, so a successful session leaves no gasm-debug-* debris behind.
func (s *Session) Kill() {
if !s.exited {
syscall.Kill(s.pid, syscall.SIGKILL)
syscall.Wait4(s.pid, nil, 0, nil)
s.exited = true
}
if s.cmd != nil && s.cmd.Process != nil {
s.cmd.Wait()
}
if s.tmpDir != "" {
os.RemoveAll(s.tmpDir)
s.tmpDir = ""
}
}
// execRange is one executable mapping of the debuggee.
type execRange struct {
lo, hi uint64
}
// execRanges is a stub on FreeBSD: there is no /proc/pid/maps to parse,
// and procfs(5) is not guaranteed to be mounted. The callers degrade
// gracefully: archReturnAddr falls back to the raw stack convention and
// the mapping scan is skipped.
func execRanges(pid int) []execRange { return nil }
// findRWXMapping is a stub on FreeBSD for the same reason: the codebase
// handshake file is the supported way the JIT region is located.
func findRWXMapping(pid int) uint64 { return 0 }
+160
View File
@@ -0,0 +1,160 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
import (
"encoding/binary"
"fmt"
"unsafe"
"golang.org/x/sys/unix"
)
// GetRegs reads the general-purpose registers of the stopped debuggee and
// converts the FreeBSD struct reg into the portable layout.
func (s *Session) GetRegs() (Regs, error) {
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return Regs{}, fmt.Errorf("debug: PT_GETREGS: %w", err)
}
return Regs{
R15: uint64(ur.R15),
R14: uint64(ur.R14),
R13: uint64(ur.R13),
R12: uint64(ur.R12),
R11: uint64(ur.R11),
R10: uint64(ur.R10),
R9: uint64(ur.R9),
R8: uint64(ur.R8),
RDI: uint64(ur.Rdi),
RSI: uint64(ur.Rsi),
RBP: uint64(ur.Rbp),
RBX: uint64(ur.Rbx),
RDX: uint64(ur.Rdx),
RCX: uint64(ur.Rcx),
RAX: uint64(ur.Rax),
RIP: uint64(ur.Rip),
CS: uint64(ur.Cs),
RFLAGS: uint64(ur.Rflags),
RSP: uint64(ur.Rsp),
SS: uint64(ur.Ss),
FS: uint64(ur.Fs),
GS: uint64(ur.Gs),
DS: uint64(ur.Ds),
ES: uint64(ur.Es),
}, nil
}
// SetRegs writes the general-purpose registers of the stopped debuggee.
func (s *Session) SetRegs(regs *Regs) error {
// Read-modify-write keeps the fields FreeBSD owns (trapno, err) intact.
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return fmt.Errorf("debug: PT_GETREGS: %w", err)
}
ur.R15 = int64(regs.R15)
ur.R14 = int64(regs.R14)
ur.R13 = int64(regs.R13)
ur.R12 = int64(regs.R12)
ur.R11 = int64(regs.R11)
ur.R10 = int64(regs.R10)
ur.R9 = int64(regs.R9)
ur.R8 = int64(regs.R8)
ur.Rdi = int64(regs.RDI)
ur.Rsi = int64(regs.RSI)
ur.Rbp = int64(regs.RBP)
ur.Rbx = int64(regs.RBX)
ur.Rdx = int64(regs.RDX)
ur.Rcx = int64(regs.RCX)
ur.Rax = int64(regs.RAX)
ur.Rip = int64(regs.RIP)
ur.Cs = int64(regs.CS)
ur.Rflags = int64(regs.RFLAGS)
ur.Rsp = int64(regs.RSP)
ur.Ss = int64(regs.SS)
return unix.PtraceSetRegs(s.pid, &ur)
}
// FPRegs holds the x87 FPU and SSE (XMM) register state, the FXSAVE image
// the FreeBSD struct fpreg mirrors: XMM0-15 at the same offsets.
type FPRegs struct {
XMM [16][16]byte // XMM0-15
}
// GetFPRegs retrieves the FPU/SSE register state via PT_GETFPREGS. The
// FreeBSD struct fpreg mirrors the FXSAVE image: the x87 environment and
// stack in Env/Acc, XMM0-15 in Xacc.
func (s *Session) GetFPRegs() (FPRegs, error) {
var fp FPRegs
var fr unix.FpReg
if err := unix.PtraceGetFpRegs(s.pid, &fr); err != nil {
return fp, fmt.Errorf("debug: PT_GETFPREGS: %w", err)
}
for i := range 16 {
copy(fp.XMM[i][:], fr.Xacc[i][:])
}
return fp, nil
}
// VectorRegs holds the YMM register state.
type VectorRegs struct {
YMM [16][32]byte // YMM0-15 (full 256-bit values)
}
// The XSAVE area the PT_GETXSTATE request returns follows the architectural
// layout (Intel SDM vol 1, "XSAVE"): the 512-byte legacy FXSAVE image (x87
// state in 0-159, XMM0-15 in 160-511), then the 64-byte xsave header whose
// first 8 bytes are xstate_bv, then one component per set feature bit, each
// 64-byte aligned. The YMM high halves are the first extended component,
// at offset 576; XFEATURE_STATE_BIT_AVX is bit 2 of xstate_bv.
const (
xsaveXMMOffset = 160
xsaveHeaderOffset = 512
xsaveBVOffset = xsaveHeaderOffset
ymmOffset = xsaveHeaderOffset + 64 // 576
ymmSize = 256 // 16 registers, 16 bytes each
xfeatureMaskYMM = 1 << 2
xstateMaxBuffer = 4096 // PT_GETXSTATE_INFO bounds the size far below this
)
// GetVectorRegs retrieves the YMM registers via PT_GETXSTATE. The low
// (XMM) halves always come from the legacy image; the high halves are
// copied only when xstate_bv reports the AVX state, and read as zero
// otherwise. When the request fails the FP image still provides correct
// XMM halves, so that is the fallback.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
var v VectorRegs
buf := make([]byte, xstateMaxBuffer)
n, _, errno := unix.Syscall6(
unix.SYS_PTRACE,
uintptr(unix.PT_GETXSTATE),
uintptr(s.pid),
0,
uintptr(unsafe.Pointer(&buf[0])),
0, 0,
)
if errno != 0 {
fp, err := s.GetFPRegs()
if err != nil {
return v, err
}
for i := range 16 {
copy(v.YMM[i][:16], fp.XMM[i][:])
}
return v, nil
}
for i := range 16 {
copy(v.YMM[i][:16], buf[xsaveXMMOffset+16*i:xsaveXMMOffset+16*i+16])
}
if int(n) >= ymmOffset+ymmSize {
if binary.LittleEndian.Uint64(buf[xsaveBVOffset:xsaveBVOffset+8])&xfeatureMaskYMM != 0 {
for i := range 16 {
copy(v.YMM[i][16:], buf[ymmOffset+16*i:ymmOffset+16*i+16])
}
}
}
return v, nil
}
+114
View File
@@ -0,0 +1,114 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && arm64
package debug
import (
"fmt"
"golang.org/x/sys/unix"
)
// GetRegs reads the general-purpose registers of the stopped debuggee and
// converts the FreeBSD struct reg (x[30], lr, sp, elr, spsr) into the
// portable layout.
func (s *Session) GetRegs() (Regs, error) {
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return Regs{}, fmt.Errorf("debug: PT_GETREGS: %w", err)
}
return Regs{
X0: ur.X[0],
X1: ur.X[1],
X2: ur.X[2],
X3: ur.X[3],
X4: ur.X[4],
X5: ur.X[5],
X6: ur.X[6],
X7: ur.X[7],
X8: ur.X[8],
X9: ur.X[9],
X10: ur.X[10],
X11: ur.X[11],
X12: ur.X[12],
X13: ur.X[13],
X14: ur.X[14],
X15: ur.X[15],
X16: ur.X[16],
X17: ur.X[17],
X18: ur.X[18],
X19: ur.X[19],
X20: ur.X[20],
X21: ur.X[21],
X22: ur.X[22],
X23: ur.X[23],
X24: ur.X[24],
X25: ur.X[25],
X26: ur.X[26],
X27: ur.X[27],
X28: ur.X[28],
X29: ur.X[29],
X30: ur.Lr,
SP: ur.Sp,
PC: ur.Elr,
PSTATE: uint64(ur.Spsr),
}, nil
}
// SetRegs writes the general-purpose registers of the stopped debuggee.
func (s *Session) SetRegs(regs *Regs) error {
var ur unix.Reg
ur.X = [30]uint64{
regs.X0, regs.X1, regs.X2, regs.X3, regs.X4, regs.X5, regs.X6,
regs.X7, regs.X8, regs.X9, regs.X10, regs.X11, regs.X12, regs.X13,
regs.X14, regs.X15, regs.X16, regs.X17, regs.X18, regs.X19, regs.X20,
regs.X21, regs.X22, regs.X23, regs.X24, regs.X25, regs.X26, regs.X27,
regs.X28, regs.X29,
}
ur.Lr = regs.X30
ur.Sp = regs.SP
ur.Elr = regs.PC
ur.Spsr = uint32(regs.PSTATE)
return unix.PtraceSetRegs(s.pid, &ur)
}
// FPRegs holds the arm64 FP/NEON register state: the 32 128-bit V
// registers, then FPSR and FPCR (the user_fpsimd shape).
type FPRegs struct {
V [32][16]byte // V0-V31 (128-bit NEON/FP registers)
FPSR uint32
FPCR uint32
}
// GetFPRegs retrieves the FP/NEON register state via PT_GETFPREGS. The
// FreeBSD struct fpreg holds the 32 128-bit V registers followed by FPSR
// and FPCR, the user_fpsimd shape.
func (s *Session) GetFPRegs() (FPRegs, error) {
var fp FPRegs
var fr unix.FpReg
if err := unix.PtraceGetFpRegs(s.pid, &fr); err != nil {
return fp, fmt.Errorf("debug: PT_GETFPREGS: %w", err)
}
for i := range 32 {
copy(fp.V[i][:], fr.Q[i][:])
}
return fp, nil
}
// VectorRegs holds the full SIMD register state.
type VectorRegs struct {
V [32][16]byte // V0-V31 (128-bit)
}
// GetVectorRegs retrieves the SIMD registers.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
var v VectorRegs
fp, err := s.GetFPRegs()
if err != nil {
return v, err
}
copy(v.V[:][:], fp.V[:][:])
return v, nil
}
+115
View File
@@ -0,0 +1,115 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && riscv64
package debug
import (
"fmt"
"golang.org/x/sys/unix"
)
// GetRegs reads the general-purpose registers of the stopped debuggee and
// converts the FreeBSD struct reg into the portable layout. Sstatus rides
// the kernel's struct but the portable surface carries the GPRs and PC.
func (s *Session) GetRegs() (Regs, error) {
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return Regs{}, fmt.Errorf("debug: PT_GETREGS: %w", err)
}
return Regs{
PC: ur.Sepc,
Ra: ur.Ra,
Sp: ur.Sp,
Gp: ur.Gp,
Tp: ur.Tp,
T0: ur.T[0],
T1: ur.T[1],
T2: ur.T[2],
S0: ur.S[0],
S1: ur.S[1],
A0: ur.A[0],
A1: ur.A[1],
A2: ur.A[2],
A3: ur.A[3],
A4: ur.A[4],
A5: ur.A[5],
A6: ur.A[6],
A7: ur.A[7],
S2: ur.S[2],
S3: ur.S[3],
S4: ur.S[4],
S5: ur.S[5],
S6: ur.S[6],
S7: ur.S[7],
S8: ur.S[8],
S9: ur.S[9],
S10: ur.S[10],
S11: ur.S[11],
T3: ur.T[3],
T4: ur.T[4],
T5: ur.T[5],
T6: ur.T[6],
}, nil
}
// SetRegs writes the general-purpose registers of the stopped debuggee.
// Read-modify-write keeps sstatus, which the kernel owns, intact.
func (s *Session) SetRegs(regs *Regs) error {
var ur unix.Reg
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
return fmt.Errorf("debug: PT_GETREGS: %w", err)
}
ur.Sepc = regs.PC
ur.Ra = regs.Ra
ur.Sp = regs.Sp
ur.Gp = regs.Gp
ur.Tp = regs.Tp
ur.T = [7]uint64{regs.T0, regs.T1, regs.T2, regs.T3, regs.T4, regs.T5, regs.T6}
ur.S = [12]uint64{regs.S0, regs.S1, regs.S2, regs.S3, regs.S4, regs.S5,
regs.S6, regs.S7, regs.S8, regs.S9, regs.S10, regs.S11}
ur.A = [8]uint64{regs.A0, regs.A1, regs.A2, regs.A3, regs.A4, regs.A5, regs.A6, regs.A7}
return unix.PtraceSetRegs(s.pid, &ur)
}
// FPRegs holds the RISC-V FP register state (32 64-bit FP registers plus
// fcsr).
type FPRegs struct {
F [32]uint64 // F0-F31 (64-bit FP registers)
FCSR uint32
}
// GetFPRegs retrieves the FP register state via PT_GETFPREGS. The FreeBSD
// struct fpreg carries each 64-bit FP register in a 128-bit slot (fp_x is
// the flat [64]-word area the x/sys type renders as [32][2]); the low word
// holds the register, and FCSR rides the tail.
func (s *Session) GetFPRegs() (FPRegs, error) {
var fp FPRegs
var fr unix.FpReg
if err := unix.PtraceGetFpRegs(s.pid, &fr); err != nil {
return fp, fmt.Errorf("debug: PT_GETFPREGS: %w", err)
}
for i := range 32 {
fp.F[i] = fr.X[i][0]
}
fp.FCSR = uint32(fr.Fcsr)
return fp, nil
}
// VectorRegs holds the FP register state shown by the regs command
// (riscv64 has 32 64-bit FP registers and fcsr).
type VectorRegs struct {
F [32]uint64
FCSR uint32
}
// GetVectorRegs retrieves the FP registers.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
fp, err := s.GetFPRegs()
if err != nil {
return VectorRegs{}, err
}
return VectorRegs{F: fp.F, FCSR: fp.FCSR}, nil
}
+91
View File
@@ -0,0 +1,91 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build freebsd && amd64
package debug
import (
"os/exec"
"path/filepath"
"runtime"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
// TestLaunchAndBreakpoint is the FreeBSD twin of the Linux integration
// test: it drives the whole launch, breakpoint, trap and register-rewind
// flow end to end. It needs a real FreeBSD kernel (ptrace does not work
// under emulation), so it only runs where it can.
func TestLaunchAndBreakpoint(t *testing.T) {
if runtime.GOARCH != "amd64" {
t.Skip("runs only on amd64 hosts")
}
// The tracer is the OS thread that forked the debuggee (PT_TRACE_ME
// binds the relation to that thread); every ptrace request must come
// from the same thread, so pin the test goroutine to one thread.
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := filepath.Join(t.TempDir(), "gasm")
out, err := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm").CombinedOutput()
if err != nil {
t.Fatalf("build gasm: %v: %s", err, out)
}
const kernelPath = "../testdata/verify/basic_amd64.s"
k, err := verify.Load(kernelPath)
if err != nil {
t.Fatalf("Load: %v", err)
}
t.Cleanup(k.Close)
fl, err := k.Func("wideCopy")
if err != nil {
t.Fatalf("Func: %v", err)
}
sess, err := Launch(bin, kernelPath, "wideCopy", make([]byte, fl.Args))
if err != nil {
t.Fatalf("Launch: %v", err)
}
t.Cleanup(sess.Kill)
bm := NewBreakpoints(sess)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
// The INT3 must be visible in the debuggee's memory.
word, err := sess.Peek(entry)
if err != nil {
t.Fatalf("Peek: %v", err)
}
if b := word & 0xFF; b != 0xCC {
t.Fatalf("int3 not patched: first byte %#02x at %#x", b, entry)
}
// The debuggee raises a second SIGSTOP after the launch barrier (the
// child's RunTarget marks its entry), so like the REPL and the cover
// mode the test keeps resuming until the breakpoint trap arrives.
for range 10 {
if err := sess.Continue(); err != nil {
t.Fatalf("Continue: %v", err)
}
if sess.Exited() {
t.Fatal("debuggee exited instead of trapping on the breakpoint")
}
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if bp := bm.HandleTrap(&regs); bp != nil {
if bp.Addr != entry {
t.Fatalf("trap at %#x, want %#x", bp.Addr, entry)
}
return // trap on the entry breakpoint: the whole flow works
}
}
t.Fatal("no breakpoint trap after 10 resumes")
}
+11 -2
View File
@@ -14,17 +14,26 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
)
// buildGasm produces the gasm binary the debugger spawns as its debuggee.
// Every live ptrace test funnels through here, so this is also where the
// deliberate-run boundary sits: under -short (the push pipeline's mode) the
// live sessions skip, because a real debuggee's launch handshake needs the
// machine to itself and a starved single-core runner turns each one into a
// timeout that burns the step's whole budget. The local test gate and the
// dispatched workflows run them in full.
func buildGasm(t *testing.T) string {
t.Helper()
if testing.Short() {
t.Skip("live ptrace session: skipped in -short mode")
}
if p := os.Getenv("GASM_TEST_BIN"); p != "" {
return p
}
bin := filepath.Join(t.TempDir(), "gasm")
cmd := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm")
cmd := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm")
out, err := cmd.CombinedOutput()
if err != nil {
t.Fatalf("build gasm: %v: %s", err, out)
+50 -27
View File
@@ -91,6 +91,16 @@ func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec s
if _, err := os.Stat(readyFile); err == nil {
break
}
// A debuggee that died before signalling readiness (unknown
// function, unparseable source) writes its failure notice to the
// handshake directory; read it and fail fast. The poll never
// waits on the child: a wait here could consume the SIGSTOP park
// that waitStopped below must receive, hanging the launch.
if err := s.deadReason(); err != nil {
cmd.Wait()
os.RemoveAll(tmpDir)
return nil, nil, err
}
time.Sleep(5 * time.Millisecond)
}
@@ -131,6 +141,18 @@ func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec s
return s, bufAddrs, nil
}
// deadReason reports the debuggee's own failure notice, the file its
// failure paths write before exiting. A debuggee killed without a notice
// (a crash, SIGKILL) surfaces through waitStopped after the poll instead,
// which is why the poll's budget stays finite.
func (s *Session) deadReason() error {
data, err := os.ReadFile(filepath.Join(s.tmpDir, "dead"))
if err != nil {
return nil
}
return fmt.Errorf("debug: debuggee failed before signalling readiness: %s", strings.TrimSpace(string(data)))
}
// waitStopped consumes ptrace-stop events until one the debugger cares
// about arrives: SIGTRAP (a breakpoint or a completed single-step), the
// debuggee's own SIGSTOP, or a genuine signal-delivery-stop. A Go tracee's
@@ -219,40 +241,41 @@ func (s *Session) Poke(addr, val uint64) error {
return nil
}
// ReadMemory reads len bytes from the debuggee's memory at addr.
// ReadMemory reads len bytes from the debuggee's memory at addr. The read
// covers exactly the requested range: the old word-at-a-time loop read a
// whole 8-byte word for the final partial word, so a request that ended
// inside the last mapped page failed whenever the following page was
// unmapped, even though every requested byte was readable.
func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) {
out := make([]byte, length)
for i := 0; i < length; i += 8 {
word, err := s.Peek(addr + uint64(i))
if err != nil {
return out[:i], err
}
for j := 0; j < 8 && i+j < length; j++ {
out[i+j] = byte(word >> (8 * j))
}
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
if err != nil {
return out, fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
}
defer mem.Close()
n, err := mem.ReadAt(out, int64(addr))
if err != nil {
return out[:n], fmt.Errorf("debug: read mem %#x: %w", addr, err)
}
return out, nil
}
// WriteMemory writes bytes to the debuggee's memory at addr.
// WriteMemory writes bytes to the debuggee's memory at addr. The write
// covers exactly the given bytes: /proc/pid/mem accepts writes of any
// length at any offset, so the word loop's read-modify-write of the final
// partial word (which read past the requested range and failed on an
// unmapped following page) is unnecessary.
func (s *Session) WriteMemory(addr uint64, data []byte) error {
for i := 0; i < len(data); i += 8 {
end := min(i+8, len(data))
var word uint64
for j := 0; j < end-i; j++ {
word |= uint64(data[i+j]) << (8 * j)
}
if end-i < 8 {
existing, err := s.Peek(addr + uint64(i))
if err != nil {
return err
}
mask := ^((uint64(1) << (8 * (end - i))) - 1)
word = (existing & mask) | word
}
if err := s.Poke(addr+uint64(i), word); err != nil {
return err
}
if len(data) == 0 {
return nil
}
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_WRONLY, 0)
if err != nil {
return fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
}
defer mem.Close()
if _, err := mem.WriteAt(data, int64(addr)); err != nil {
return fmt.Errorf("debug: write mem %#x: %w", addr, err)
}
return nil
}

Some files were not shown because too many files have changed in this diff Show More