Compare commits

..
36 Commits
Author SHA1 Message Date
petrbalvin 96e81cc98d docs: add the Plan 9 assembly case and real-use note to the README
Test / test (push) Successful in 2m6s
2026-09-19 18:06:18 +02:00
petrbalvin c834d98210 docs: bring the document set into the standard shape
Test / test (push) Successful in 2m28s
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin 03d6d4da54 style: put the repository assembly in gasm fmt canonical form
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin 0b42ce7952 style: use one spelling for colour across the CLI
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin 288a64ccd2 ci: align the pipelines with the hand-written templates
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin 5fddfa704b build: declare the exact toolchain and the canonical recipes
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin a2bb5eeb4e chore: drop the stale comment from the ignore list
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:14 +02:00
petrbalvin 48449b7a7f build: declare the go1.27.1 toolchain
Assisted-by: GLM 5.3 Flash
2026-09-16 23:12:31 +02:00
petrbalvin 3de043c494 docs: add SECURITY.md and record the round in the CHANGELOG
Assisted-by: GLM 5.3 Flash
2026-09-16 23:12:31 +02:00
petrbalvin 0078f7be5c style: purge em dashes from the produced text
Assisted-by: GLM 5.3 Flash
2026-09-16 23:12:31 +02:00
petrbalvin 6a7317d141 chore: trim the ignore list to the convention
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin d08523caa5 docs: move the recipe and version descriptions with the behaviour
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin 20e4b8d9c4 ci: align the pipelines with the hand-written templates
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin 61f4247cef refactor(gasm): report the toolchain-recorded version
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin 049872ddff build: restore the canonical justfile recipe set
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin 3669f64ff6 build: install the gasm binary into the user-local bin directory
Test / vet (push) Successful in 46s
Test / test (push) Successful in 2m44s
Test / build (push) Successful in 42s
2026-09-14 23:41:21 +02:00
petrbalvin fff9f75595 chore: prepare release v0.33.0
Release / build (amd64, linux) (push) Successful in 49s
Release / build (arm64, linux) (push) Successful in 43s
Release / build (loong64, linux) (push) Successful in 46s
Release / build (riscv64, linux) (push) Successful in 45s
Test / vet (push) Successful in 47s
Release / release (push) Successful in 18s
Test / test (push) Successful in 2m39s
Test / build (push) Successful in 43s
2026-09-14 23:36:19 +02:00
petrbalvin 40476546df fix(asm): close the oracle parity gaps in frame addressing and calls 2026-09-14 23:25:14 +02:00
petrbalvin 70218e84ba feat(asm): emit the loong64 stack-split guard for big frames 2026-09-14 22:38:58 +02:00
petrbalvin db50b98179 feat(asm): emit the loong64 stack-split guard for small and medium frames 2026-09-14 21:21:58 +02:00
petrbalvin 2e2c0b82a0 feat(asm): emit the riscv64 stack-split guard and fix large-frame addressing 2026-09-14 21:09:13 +02:00
petrbalvin 8dc1e98ca1 feat(asm): emit the arm64 stack-split guard and morestack block 2026-09-14 20:49:03 +02:00
petrbalvin 1d8e68c574 feat(asm): emit the amd64 stack-split guard and morestack block 2026-09-14 20:35:55 +02:00
petrbalvin 89fa6ea15e feat(lsp): resolve definition and references across open documents 2026-09-14 18:50:55 +02:00
petrbalvin edc20ffa97 feat(cmd): add gasm dis and share the decoder with the debugger 2026-09-14 18:47:08 +02:00
petrbalvin 50db6615b2 feat(cmd): add gofmt-style -l and -d modes to gasm fmt 2026-09-14 18:47:08 +02:00
petrbalvin 95f1d6f083 style: replace em dashes in the scaffold comments 2026-09-14 18:22:25 +02:00
petrbalvin f43e791e5a chore: add .qwen to the gitignore metadata block 2026-09-14 18:22:18 +02:00
petrbalvin 1691c81095 style: replace em and en dashes across sources 2026-09-14 18:22:18 +02:00
petrbalvin 2db563be07 refactor(cmd): consolidate cross-arch verify and drop dead code 2026-09-14 18:22:00 +02:00
petrbalvin 4f190ee1a2 refactor(debug): move watchpoint slot state into the session 2026-09-14 18:22:00 +02:00
petrbalvin 909f874797 fix(lsp): recover from handler panics and decode client uris 2026-09-14 18:22:00 +02:00
petrbalvin e307bf830f fix(lint): guard unnamed TEXT and refresh the textflag table 2026-09-14 18:22:00 +02:00
petrbalvin 953c258d6a fix(asm): make arm64 and loong64 relocations match the toolchain 2026-09-14 18:22:00 +02:00
petrbalvin c6f0286732 fix(asm): encode amd64 frame adjustments above 127 bytes with imm32 2026-09-14 18:22:00 +02:00
petrbalvin f5fc22d390 fix(parser): reject malformed TEXT frames and parse signed frame sizes 2026-09-14 18:22:00 +02:00
137 changed files with 5862 additions and 2307 deletions
+37
View File
@@ -0,0 +1,37 @@
# Race, Go. Dispatched by hand, and never a gate on a push or a tag: the release tag is
# cut only after `just gates` has already raced the tree, so this workflow is the
# explicit second opinion, not a step of the release.
#
# The race detector roughly doubles both time and memory, which the shared runner box
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once per
# task; here it is a decision rather than a routine.
#
# Every step is one command, so the step that fails is the gate that failed.
name: Race
on:
workflow_dispatch:
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
race:
runs-on: fedora
timeout-minutes: 20
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
cache: true
- name: Install gcc
# The race detector needs cgo and the runner image carries no C compiler.
run: dnf install -y gcc
- name: Race
run: go test -race -count=1 -timeout 10m ./...
+255 -86
View File
@@ -1,73 +1,198 @@
# Release — gasm binaries. Runs on version tags (v0.28.0) pushed to main.
# Release, Go binaries. Runs on version tags (v1.2.3) pushed to main.
#
# The module sits at the repository root: the toolchain records a version only for a root
# module, measured on go1.27.1, so a build of a module in a subdirectory reports (devel)
# even at its own <module>/vX.Y.Z tag and this workflow's smoke test can never pass for
# it. A Go repository is one module at the root.
#
# The version contract these steps implement is in the `release` skill, and its point is
# that nothing is injected: the toolchain records the tag into the binary's build
# information, so the build simply has to happen at the tag, which the trigger guarantees.
#
# The gates run in their own job, once, before the matrix, minus the race detector: race
# never runs on a push path or a tag, and the local gate raced this tree before the tag
# was cut. Putting the gates inside the matrix would run the whole suite once per target
# on the box that also hosts the forge. Each job validates the tag for itself rather than
# passing a value between jobs, so no workflow feature has to be trusted for the version
# to reach the file name.
name: Release
on:
push:
tags: ["v*"]
env:
# The box is shared with the forge, so parallelism is bounded on purpose. The gates job
# needs it most; the build jobs inherit it for their parallel compilation.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
gates:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
cache: true
- name: Install Perl
# Perl for the steps below. The install is a no-op where the package
# is already present.
run: dnf install -y perl
- name: Validate the tag
env:
VERSION: ${{ gitea.ref_name }}
run: |
perl -e '
my $v = $ENV{VERSION} // q{};
$v =~ m{^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$}
or die qq{ERROR: expected a semver tag like v1.2.3, got: $v\n};
print qq{tag $v\n};
'
- name: Build
run: go build ./...
- name: Format
run: |
perl -e '
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
my @bad = <$g>;
close($g);
print @bad;
exit(@bad ? 1 : 0);
'
- name: Vet
run: go vet ./...
- name: Modernise
run: go fix -diff ./...
- name: Tests
# The same command as in test.yml, so the floor is the same number everywhere.
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Coverage floor
run: |
perl -e '
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
my $total;
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
close($c);
die qq{no total line in coverage.out\n} unless defined $total;
printf qq{Total coverage: %s%%\n}, $total;
exit($total < 80 ? 1 : 0);
'
build:
runs-on: fedora
timeout-minutes: 25
needs: gates
strategy:
fail-fast: false
matrix:
# Portable targets: amd64, arm64, loong64 and riscv64 on Linux, at the toolchain
# default level. No 32-bit, no wasm, no macOS, no Windows. FreeBSD stays out until
# verify/jit.go ports off syscall.Mprotect: the Go syscall package defines no
# Mprotect for freebsd, and verify/jit.go:50 calls it to drop the write bit from
# the JIT mapping, so every freebsd target fails to build with "undefined:
# syscall.Mprotect" (verified for amd64, arm64 and riscv64 on go1.27.1).
include:
- goos: linux
goarch: amd64
- goos: linux
goarch: arm64
- goos: linux
goarch: riscv64
- goos: linux
goarch: loong64
- goos: linux
goarch: riscv64
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version: "1.27"
go-version-file: go.mod
cache: true
- name: Download dependencies
run: go mod download
- name: Install Perl
run: dnf install -y perl
- name: Validate tag and build
id: build
- name: Validate the tag
id: version
env:
VERSION: ${{ gitea.ref_name }}
run: |
set -euo pipefail
perl -e '
my $v = $ENV{VERSION} // q{};
$v =~ m{^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$}
or die qq{ERROR: expected a semver tag like v1.2.3, got: $v\n};
(my $nv = $v) =~ s{^v}{};
open(my $o, q{>>}, $ENV{GITEA_OUTPUT}) or die qq{GITEA_OUTPUT: $!};
print $o qq{version_no_v=$nv\n};
close($o);
print qq{version $nv\n};
'
if ! echo "$VERSION" | grep -qE '^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$'; then
echo "ERROR: expected a semver tag like v1.2.3, got: '$VERSION'"
exit 1
fi
VERSION_NO_V="${VERSION#v}"
echo "version_no_v=${VERSION_NO_V}" >> "$GITEA_OUTPUT"
mkdir -p bin
GOOS=${{ matrix.goos }} GOARCH=${{ matrix.goarch }} CGO_ENABLED=0 \
go build -ldflags "-s -w -X main.version=${VERSION_NO_V}" \
-o "bin/gasm-${VERSION_NO_V}-${{ matrix.goos }}-${{ matrix.goarch }}" \
./cmd/gasm
- name: Build
env:
VERSION_NO_V: ${{ steps.version.outputs.version_no_v }}
GOOS: ${{ matrix.goos }}
GOARCH: ${{ matrix.goarch }}
CGO_ENABLED: "0"
run: |
# Nothing is injected. The toolchain records the tag into the binary's build
# information, so the version is right because this build happens at the tag, and
# there is no path for anyone to get wrong. -s -w only strips symbols.
go build -ldflags "-s -w" -o "bin/gasm-${VERSION_NO_V}-${GOOS}-${GOARCH}" ./cmd/gasm
# Artifacts stay on v3: v4 and later detect Gitea as GHES and abort.
- name: Upload artifact
uses: actions/upload-artifact@v3
with:
name: gasm-${{ matrix.goos }}-${{ matrix.goarch }}
path: bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
path: bin/gasm-${{ steps.version.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
if-no-files-found: error
- name: Smoke test
# Only a binary matching the runner can be run here. The check is not that --version
# exits cleanly but that it reports the tag and nothing more: a build outside version
# control reports (devel), and a build whose tree was dirty reports +dirty, and both
# would otherwise be published.
if: matrix.goos == 'linux' && matrix.goarch == 'amd64'
env:
TAG: ${{ gitea.ref_name }}
BIN: bin/gasm-${{ steps.version.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
run: |
chmod +x bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
./bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }} --version
perl -e '
my $want = $ENV{TAG} // die qq{ERROR: no tag\n};
open(my $bin, q{-|}, $ENV{BIN}, q{--version}) or die qq{$ENV{BIN}: $!};
my $got = <$bin>;
close($bin);
$got = defined $got ? $got : q{};
chomp $got;
index($got, $want) >= 0
or die qq{ERROR: the binary printed "$got", which does not contain $want. Version control was disabled, so there is no recorded version.\n};
index($got, q{+dirty}) < 0
or die qq{ERROR: the binary printed "$got". The tree was dirty at build time, which means the checkout was not the tag, or the build artefacts are not ignored.\n};
print qq{$ENV{BIN} reports $got\n};
'
release:
runs-on: fedora
timeout-minutes: 15
needs: build
permissions:
# contents: read is required for the checkout: a job that declares any
# permissions gets a token scoped to exactly those, and releases: write
# alone leaves the fetch with no read access, which Gitea answers with
# a 404 "Repository not found". Verified on the instance 2026-09-16.
contents: read
releases: write
steps:
- uses: actions/checkout@v7
@@ -77,81 +202,125 @@ jobs:
with:
path: dist
- name: Extract CHANGELOG section
- name: Install Perl
run: dnf install -y perl
- name: Extract the CHANGELOG section
env:
VERSION: ${{ gitea.ref_name }}
run: |
set -euo pipefail
VERSION_NO_V="${VERSION#v}"
# Each step derives what it needs from the tag, so no value has to travel between
# jobs.
perl -e '
my $v = $ENV{VERSION} // q{};
$v =~ s{^v}{};
open(my $vout, q{>}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
print $vout $v;
close($vout);
open(my $in, q{<}, q{CHANGELOG.md}) or die qq{CHANGELOG.md: $!};
my @lines = <$in>;
close($in);
my ($start, $end) = (-1, scalar @lines);
for my $i (0 .. $#lines) {
if ($start < 0) { $start = $i if $lines[$i] =~ m{^##\s+\[\Q$v\E\]} }
elsif ($lines[$i] =~ m{^##\s+\[}) { $end = $i; last }
}
$start >= 0 or die qq{ERROR: no CHANGELOG section for $v, expected a heading like: ## [$v] - YYYY-MM-DD\n};
my @body = grep { m{\S} } @lines[$start + 1 .. $end - 1];
@body or die qq{ERROR: the CHANGELOG section for $v is empty\n};
open(my $out, q{>}, q{release-body.md}) or die qq{release-body.md: $!};
print $out @body;
close($out);
printf qq{notes for %s: %d lines\n}, $v, scalar @body;
'
sed -n "/^## \[${VERSION_NO_V}\] /,/^## \[/p" CHANGELOG.md \
| sed '$d' \
| tail -n +2 \
> release-body.md
- name: Build the release request
run: |
perl -e '
open(my $vin, q{<}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
my $v = <$vin>;
close($vin);
chomp $v;
open(my $in, q{<:raw}, q{release-body.md}) or die qq{release-body.md: $!};
my $body = do { local $/; <$in> };
close($in);
# Byte-oriented escaping: JSON is UTF-8, so non-ASCII passes through and only the
# characters JSON forbids are rewritten.
$body =~ s/([\\"])/\\$1/g;
$body =~ s/\t/\\t/g;
$body =~ s/\r//g;
$body =~ s/\n/\\n/g;
$body =~ s/([\x00-\x08\x0b\x0c\x0e-\x1f])/sprintf(q{\u%04x}, ord($1))/ge;
my $json = sprintf(qq{{"tag_name":"v%s","name":"v%s","body":"%s","draft":false,"prerelease":false}}, $v, $v, $body);
open(my $out, q{>}, q{release.json}) or die qq{release.json: $!};
print $out $json;
close($out);
print qq{release.json written for v$v\n};
'
if [ ! -s release-body.md ]; then
echo "ERROR: no CHANGELOG section found for ${VERSION_NO_V}"
echo "Expected a heading like: ## [${VERSION_NO_V}] — YYYY-MM-DD"
exit 1
fi
- name: Create release
- name: Create the release
env:
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
GITEA_SERVER_URL: ${{ gitea.server_url }}
GITEA_REPOSITORY: ${{ gitea.repository }}
GITEA_REF_NAME: ${{ gitea.ref_name }}
run: |
set -euo pipefail
BODY=$(sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' -e 's/\t/\\t/g' -e 's/\r//g' release-body.md | sed ':a;N;$!ba;s/\n/\\n/g')
BODY="\"${BODY}\""
response=$(curl -sS -w '\n%{http_code}' \
-H "Authorization: token ${GITEA_TOKEN}" \
-H "Content-Type: application/json" \
-X POST \
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases" \
-d "{\"tag_name\":\"${GITEA_REF_NAME}\",\"name\":\"${GITEA_REF_NAME}\",\"body\":${BODY},\"draft\":false,\"prerelease\":false}")
http_code=$(echo "$response" | tail -1)
payload=$(echo "$response" | sed '$d')
echo "HTTP ${http_code}"
if [ "$http_code" != "201" ]; then
echo "Failed to create release: ${payload}"
exit 1
fi
RELEASE_ID=$(echo "$payload" | grep -oE '"id"[[:space:]]*:[[:space:]]*[0-9]+' | head -1 | grep -oE '[0-9]+')
echo "Created release ID=${RELEASE_ID}"
printf '%s' "${RELEASE_ID}" > release-id.txt
perl -e '
my @cmd = (q{curl}, q{-sS}, q{-o}, q{response.json}, q{-w}, q{%{http_code}},
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
q{-H}, q{Content-Type: application/json},
q{-X}, q{POST},
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases},
q{--data-binary}, q{@release.json});
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
my $code = <$curl>;
my $ok = close($curl);
my $exit = $? >> 8;
$code = defined $code ? $code : q{};
$ok or die qq{ERROR: curl failed (exit $exit) calling $ENV{GITEA_SERVER_URL}\n};
open(my $r, q{<:raw}, q{response.json}) or die qq{response.json: $!};
my $body = do { local $/; <$r> };
close($r);
$code eq q{201} or die qq{ERROR: the release was not created, HTTP $code: $body\n};
$body =~ m{"id"\s*:\s*([0-9]+)} or die qq{ERROR: no release id in the response: $body\n};
open(my $o, q{>}, q{release-id.txt}) or die qq{release-id.txt: $!};
print $o $1;
close($o);
print qq{release id $1\n};
'
- name: Upload assets
env:
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
GITEA_SERVER_URL: ${{ gitea.server_url }}
GITEA_REPOSITORY: ${{ gitea.repository }}
GITEA_REF_NAME: ${{ gitea.ref_name }}
run: |
set -euo pipefail
RELEASE_ID=$(cat release-id.txt)
for binary in dist/gasm-*/gasm-*; do
[ -f "$binary" ] || continue
fname=$(basename "$binary")
echo "Uploading ${fname}..."
http_code=$(curl -sS -o /dev/null -w '%{http_code}' \
-H "Authorization: token ${GITEA_TOKEN}" \
-H "Content-Type: application/octet-stream" \
-X POST \
--data-binary "@${binary}" \
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases/${RELEASE_ID}/assets?name=${fname}")
echo " HTTP ${http_code}"
if [ "$http_code" != "201" ]; then
echo "Failed to upload ${fname}"
exit 1
fi
done
echo "Release ${GITEA_REF_NAME} is live."
perl -e '
open(my $f, q{<}, q{release-id.txt}) or die qq{release-id.txt: $!};
my $id = <$f>;
close($f);
chomp $id;
my @files = grep { -f $_ } glob(q{dist/*/*});
@files or die qq{ERROR: no assets under dist/\n};
my $bad = 0;
for my $path (@files) {
(my $name = $path) =~ s{.*/}{};
my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}},
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
q{-H}, q{Content-Type: application/octet-stream},
q{-X}, q{POST}, q{--data-binary}, qq{@$path},
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name});
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
my $code = <$curl>;
my $ok = close($curl);
my $exit = $? >> 8;
$code = defined $code ? $code : q{};
unless ($ok) {
printf qq{%s: curl failed (exit %d)\n}, $name, $exit;
$bad = 1;
next;
}
printf qq{%s: HTTP %s\n}, $name, $code;
$bad = 1 if $code ne q{201};
}
exit($bad ? 1 : 0);
'
+75 -76
View File
@@ -1,4 +1,17 @@
# Test — gasm-devkit. Runs on push and pull request to development.
# Test, Go. Push and pull request to development. Never on main.
#
# The gates are the ones the justfile's `gates` recipe runs, minus race: the shared
# runner box cannot afford the race detector on every push, so it lives in race.yml.
# The box is one core and 2 GB beside Gitea, so parallelism is bounded on purpose and
# everything runs in one job. Extra jobs would duplicate the checkout, the Go setup and
# the dependency download three times without buying any parallelism.
#
# Every step is one command, so the step that fails is the gate that failed, and no shell
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
# not Python: Perl behaves the same on both runner images, there is no bashism to trip over
# on ash, and it is one language instead of two. The Perl uses builtins only, because
# Fedora packages the Perl modules separately and nothing beyond `perl` itself may be
# assumed present.
name: Test
on:
@@ -7,90 +20,76 @@ on:
pull_request:
branches: [development]
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1
GOMAXPROCS: "2"
# A superseded run of the same ref is cancelled instead of queueing behind one that
# no longer matters. Verified on Gitea 1.27.1 on 2026-09-17: a queued run whose ref
# moved on is cancelled before it ever reaches the runner, while a run already
# dispatched there runs to completion.
concurrency:
group: ${{ gitea.workflow }}-${{ gitea.ref }}
cancel-in-progress: true
jobs:
vet:
runs-on: fedora
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version: "1.27"
- name: Download dependencies
run: go mod download
- name: gofmt
run: |
set -euo pipefail
unformatted=$(gofmt -l .)
if [ -n "$unformatted" ]; then
echo "These files need gofmt:"
echo "$unformatted"
exit 1
fi
- name: go vet
run: go vet ./...
test:
runs-on: fedora
needs: vet
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version: "1.27"
# The module is the source of truth for the version, so it cannot drift.
go-version-file: go.mod
cache: true
- name: Download dependencies
run: go mod download
- name: Install gcc
run: dnf install -y gcc
- name: go test -race
run: go test -race -count=1 ./...
- name: Coverage gate — 80 % minimum
run: |
set -euo pipefail
# Exclude packages inherently untestable without hardware:
# debug — interactive ptrace, requires a live process
# cmd/gasm — CLI glue, covered by integration tests
go test -coverprofile=coverage.out \
sourcedock.dev/petrbalvin/gasm-devkit/arch \
sourcedock.dev/petrbalvin/gasm-devkit/asm \
sourcedock.dev/petrbalvin/gasm-devkit/ast \
sourcedock.dev/petrbalvin/gasm-devkit/format \
sourcedock.dev/petrbalvin/gasm-devkit/lexer \
sourcedock.dev/petrbalvin/gasm-devkit/lint \
sourcedock.dev/petrbalvin/gasm-devkit/lsp \
sourcedock.dev/petrbalvin/gasm-devkit/parser \
sourcedock.dev/petrbalvin/gasm-devkit/token \
sourcedock.dev/petrbalvin/gasm-devkit/verify
coverage=$(go tool cover -func=coverage.out | awk '/^total:/ { gsub("%", "", $3); print $3 }')
echo "Total coverage: ${coverage}%"
if awk -v c="$coverage" 'BEGIN { exit !(c+0 < 80) }'; then
echo "ERROR: coverage ${coverage}% is below the 80% threshold"
exit 1
fi
build:
runs-on: fedora
needs: test
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version: "1.27"
- name: Download dependencies
run: go mod download
- name: Install Perl
# The runner images are minimal and Perl is not guaranteed. The install is a
# no-op where it is already present; drop this step once verified on the box.
run: dnf install -y perl
# The steps follow the `gates` order of the justfile contract: build, format,
# vet, test. The vet gate is go vet and go fix -diff, two steps here.
- name: Build
run: go build -ldflags="-s -w" -o bin/gasm ./cmd/gasm
run: go build ./...
- name: Smoke test
run: ./bin/gasm --version
- name: Format
run: |
perl -e '
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
my @bad = <$g>;
close($g);
print @bad;
exit(@bad ? 1 : 0);
'
- name: Vet
run: go vet ./...
- name: Modernise
# Exits non-zero when it has something to rewrite, so it needs no output capture.
run: go fix -diff ./...
- name: Tests
# The suite must be fast: a push pipeline that cannot finish in a few minutes moves
# its heavy part behind a dispatch. The inner timeout matches the job's, so a
# hanging test reports its own goroutine dump rather than a silent job kill.
# The pattern is `packages` in the project's justfile: the logic packages, since a
# thin cmd/ would drag the total under the floor. release.yml runs the same
# command, so the floor is the same number everywhere.
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Coverage floor
run: |
perl -e '
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
my $total;
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
close($c);
die qq{no total line in coverage.out\n} unless defined $total;
printf qq{Total coverage: %s%%\n}, $total;
exit($total < 80 ? 1 : 0);
'
+3 -14
View File
@@ -1,24 +1,13 @@
# Metadata (always first, per repo convention)
.idea/
.zcode/
.mimocode/
# Binaries
/gasm
# Build output
/bin/
*.exe
# Test and coverage artefacts
/gasm
coverage.out
*.test
# Crash dumps
# Crash dumps from the emulator runs
core
core.*
*.core
# Scratch / temporary work
_scratch/
# ZCode workspace
.zcode
+243 -139
View File
@@ -3,15 +3,119 @@
All notable changes to gasm-devkit are documented here.
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
and this project adheres to [Conventional Commits](https://www.conventionalcommits.org/).
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
## [development]
### Changed
- **Canonical just recipes.** `just gates` is the definition of done
(build, fmt-check, vet, test, race). `install` now builds and copies
the binary into `~/.local/bin` (`BINDIR` overrides) instead of
downloading module dependencies, and `install-bin` is gone. The test
gate sweeps the logic packages (arch through verify; the hardware-bound
`debug` and the thin `cmd/gasm` sit outside it), so the coverage floor
is computed over the product code and the number is identical locally
and in CI. `fuzz` requires its target package.
- **The reported version comes from the build.** `gasm --version`
prints the version the toolchain recorded: the tag on a tagged
checkout, a pseudo-version naming the commit below one, `+dirty` on a
dirty tree and `(devel)` outside version control. Nothing is
injected with `-ldflags -X` any more.
- **CI realigned with the gate set.** The push pipeline runs the gates
minus race in one job, in the `gates` order, with a cached Go setup and
the module as the version source; a superseded run of the same branch
is cancelled instead of queueing; every `go test` runs under a
ten-minute bound that matches its job's; the race detector moved to a
hand-dispatched workflow and runs in the local gate before a tag is
cut, never on a push or a tag; the release builds without injection and
its smoke test requires the recorded tag and rejects `+dirty`.
- **The documents follow the standard set.** `docs/ARCHITECTURE.md` is
organised as Overview, Packages, Data flow, State and lifetime and
Dependencies, and carries a sequence diagram of the assembly path;
`docs/DEVELOPMENT.md` lists every recipe in one table and documents the
coverage floor, the CI and the release flow; `docs/CLI.md` gives the
synopsis, the commands, every flag with its default, the exit codes and
worked examples; `CONTRIBUTING.md` carries the Contributor terms and
states the commit trailer form, the one-logical-change rule and the
licence header rule. The repository's own assembly (the `verify`
trampolines and the test kernels) is in `gasm fmt` canonical form.
- **The README states the project's purpose and status.** It opens with
a warning that the tool is an experiment under active development,
version 0.x.x, free to change without warning, with 1.0.0 far off,
and already in active use on real assembly work. It describes both
goals (tooling for Plan 9 assembly, and Plan 9 assembly outside the
Go toolchain), argues the case for the syntax in a new Why Plan 9
assembly section, and carries a Direction section: extended
instruction support, full GOOBJ and ELF compilation, Linux and
FreeBSD, and the four architectures.
### Fixed
- **The dependency statement was wrong.** `golang.org/x/arch` is not
test-only: `gasm dis` and the debugger's listings decode through it, so
it is linked into the binary. `CONTRIBUTING.md` and
`docs/ARCHITECTURE.md` said otherwise.
- **The CLI reference listed 17 of the 18 lint rules.** The missing
`reserved-register-write` is documented with the rest.
## [0.33.0] - 2026-09-14
### Added
-
- **Stack-split guards in `gasm asm`.** Every framed function now gets
the morestack prologue check and the trailing morestack block
(`CALL runtime.morestack_noctxt`), byte-identical to the toolchain's
`stacksplit` output on all four architectures: the small, medium and
big frame classes, auto-NOSPLIT leaves, the materialised constants of
large frames (arm64 `R27`, riscv64 `X31`, loong64 `R30`) and the
arm64 extrasize rule. Assembled objects are therefore linkable for
split functions, not only `NOSPLIT` leaves.
- **`gasm dis`.** Standalone disassembly through `golang.org/x/arch`:
a `.s` file is assembled and listed per `TEXT` function with local
labels at their real offsets, or raw bytes from a file or stdin are
disassembled linearly (`-a` selects the architecture). The debugger
shares the same decoder instead of carrying its own.
- **`gasm fmt -l` and `-d`.** Check mode lists files whose formatting
differs; diff mode prints a unified diff from the project's own
LCS-based differ, with GNU header semantics.
- **LSP cross-file navigation.** Go-to-definition and find references
fall back from local labels to the `TEXT` functions of every open
document, and rename follows the same cross-file matching.
- **Large frame offsets on riscv64 and arm64.** Frame-relative loads
and stores beyond the signed 12-bit immediate range materialise the
address through the toolchain temp register (riscv64 `X31`, arm64
`R27`) instead of silently truncating the offset (riscv64) or
rejecting the instruction (arm64); arm64 frame sizes now add the
toolchain's extrasize exactly (+8 when the frame leaves an alignment
gap, +16 when it is already aligned).
- **Tail calls `JMP sym(SB)`** on all four architectures (amd64 `E9`,
arm64 `B`, riscv64 `JAL X0`, loong64 `B`) with the call relocation.
- **Live oracle-parity tests.** Kernel files covering every guard
class, large-offset pattern and tail call are assembled by gasm and
by the installed `go tool asm` and compared byte-for-byte on all four
architectures, alongside the existing pinned-byte tests.
## [0.32.0] — 2026-08-31
### Fixed
- The v0.32.0 review findings: the parser rejects malformed `TEXT`
frames and parses signed frame sizes; amd64 frame adjustments above
127 bytes encode with imm32; arm64 and loong64 relocation encodings
match the toolchain; the linter guards unnamed `TEXT` directives and
refreshes its textflag table; the LSP recovers from handler panics
and decodes client URIs; watchpoint slot state moved into the debug
session; dead verify code removed; em and en dashes replaced across
sources.
- `gasm asm --format goobj`: internal calls to `TEXT` symbols of the
same file resolve on every architecture (the reference check accepted
only the amd64 call kind).
- `gasm asm --format elf` on loong64: branch relocations now map to
`R_LARCH_B26` instead of falling into `R_LARCH_PCALA_HI20`.
- arm64 large-prologue `ADD`/`SUB` use the extended-register encoding
the toolchain picks, and the morestack block saves the link register
with the toolchain's `OR` form on loong64.
## [0.32.0] - 2026-08-31
### Added
@@ -162,7 +266,7 @@ and this project adheres to [Conventional Commits](https://www.conventionalcommi
outputs and operand strictness with `go tool asm`.
- **asm help text.** Updated to list arm64 as a supported architecture.
## [0.31.1] — 2026-08-20
## [0.31.1] - 2026-08-20
### Fixed
@@ -170,9 +274,9 @@ and this project adheres to [Conventional Commits](https://www.conventionalcommi
because the version variables in `justfile` and `cmd/gasm/main.go` were not
bumped during the release commit.
## [0.31.0] — 2026-08-20
## [0.31.0] - 2026-08-20
The arm64 encoder (Phase 5 — complete) ships with ELF64 and GOOBJ emission,
The arm64 encoder (Phase 5; complete) ships with ELF64 and GOOBJ emission,
verified byte-for-byte against `GOARCH=arm64 go tool asm` and linked into a
real `go build`. The encoder covers the full integer instruction set, FP
arithmetic, conditional select, CRC32, and the MOV pseudo-instruction with
@@ -180,7 +284,7 @@ bitmask immediate encoding. The project now requires Go 1.27.
### Added
- **arm64 encoder (Phase 5 — complete).** `gasm asm` can now assemble `_arm64.s`
- **arm64 encoder (Phase 5; complete).** `gasm asm` can now assemble `_arm64.s`
files: the AArch64 integer instruction set with the MOV pseudo-instruction and
its immediate-constant expansions (MOVZ/MOVN/MOVK for wide immediates, ORR with
logical bitmask encoding for values like `$1`), data-processing (shifted
@@ -189,7 +293,7 @@ bitmask immediate encoding. The project now requires Go 1.27.
SB/global symbol references (ADRP+ADD pairs with `R_ADDRARM64` relocations),
jump chain folding, and ELF64 emission (`gasm asm --format elf`). Ground-truth
verification against `GOARCH=arm64 go tool asm` matches byte-for-byte. Phase 5
(the other architectures — RISC-V, LoongArch, arm64) is now complete.
(the other architectures; RISC-V, LoongArch, arm64) is now complete.
### Changed
@@ -197,7 +301,7 @@ bitmask immediate encoding. The project now requires Go 1.27.
The `R_DWTXTADDR_U4` relocation type is detected at runtime for backward
compatibility.
## [0.30.0] — 2026-08-13
## [0.30.0] - 2026-08-13
The LoongArch encoder (Phase 5) ships with ELF64 and GOOBJ emission, verified
byte-for-byte against `GOARCH=loong64 go tool asm` and linked into a real
@@ -222,7 +326,7 @@ tracks four hardware watchpoint slots, and the toolkit is Linux-only.
- **GOOBJ DWARF symbols.** The GOOBJ emitters now write the per-function
DWARF symbols the linker requires (the subprogram DIE and the `.debug_line`
program, byte-identical to `cmd/asm`'s), and the pc-value table deltas are
in the architecture's MinLC units as the runtime expects — the amd64 link
in the architecture's MinLC units as the runtime expects; the amd64 link
test now genuinely substitutes the gasm object, and the amd64/loong64
end-to-end GOOBJ link tests pass.
- **RISC-V GOOBJ emission via the shared emitter.** RISC-V GOOBJ output is
@@ -291,7 +395,7 @@ tracks four hardware watchpoint slots, and the toolkit is Linux-only.
`GOARCH=riscv64 go tool asm`.
- **Debugger watchpoint slots.** `gasm debug`'s `watch` command always used
hardware watchpoint slot 0, so a second `watch` call silently overwrote
the first. Watchpoint slots are now tracked in the `Session` (DR0–DR3);
the first. Watchpoint slots are now tracked in the `Session` (DR0-DR3);
`watch` picks the first free slot and reports an error if all four are in
use, and `unwatch <slot>` clears one (no argument clears all).
@@ -303,7 +407,7 @@ tracks four hardware watchpoint slots, and the toolkit is Linux-only.
disassembly at PC, memory-write, watchpoints, and source-line mapping are
all shipped.
## [0.29.0] — 2026-08-07
## [0.29.0] - 2026-08-07
RISC-V GOOBJ emission, YMM vector register display, named buffer allocation
in the debugger, two new CLI commands (`diff`, `profile`), go-to-definition in
@@ -313,30 +417,30 @@ new CLI commands. A signature-parser fix corrects grouped Go parameters.
### Added
- **RISC-V GOOBJ emission** — `gasm asm --format goobj` for RISC-V produces
- **RISC-V GOOBJ emission**; `gasm asm --format goobj` for RISC-V produces
linkable Go objects with funcdata, pc-value tables, and RISC-V relocation
types (same format as amd64 GOOBJ, with the RISC-V architecture marker).
- **`gasm diff`** — compare the machine code of two assembly files byte-for-byte;
- **`gasm diff`**; compare the machine code of two assembly files byte-for-byte;
shows which functions differ and the first few differing bytes.
- **`gasm profile`** — show the basic-block structure of each function: labels,
- **`gasm profile`**; show the basic-block structure of each function: labels,
offsets, frame size, and NOSPLIT flag.
- **LSP go-to-definition** — `textDocument/definition` navigates from a label
- **LSP go-to-definition**; `textDocument/definition` navigates from a label
reference to its definition.
- **did-you-mean** — when the RISC-V assembler encounters an undefined label, it
- **did-you-mean**; when the RISC-V assembler encounters an undefined label, it
suggests the closest existing label using Levenshtein distance.
- **YMM vector register display** — `regs` in the debugger now shows YMM
- **YMM vector register display**; `regs` in the debugger now shows YMM
registers via `PTRACE_GETFPREGS` (falls back to XMM when XSAVE is unavailable).
- **Named buffer allocation** — `gasm debug --buf name:size:pattern` allocates
- **Named buffer allocation**; `gasm debug --buf name:size:pattern` allocates
buffers in the debuggee filled with `zero`, `ones`, `seq`, or a hex pattern;
buffer pointers are placed into the argument block at the matching positions.
- **Crash input storage** — `FuzzResult.CrashInput` stores the input that caused
- **Crash input storage**; `FuzzResult.CrashInput` stores the input that caused
a crash or mismatch for reproducibility.
- **ABI + fuzz combined** — `gasm verify --fuzz` now runs ABI checks (sentinel
- **ABI + fuzz combined**; `gasm verify --fuzz` now runs ABI checks (sentinel
registers, canary, stack bounds) alongside differential fuzz testing.
- **`gasm diff --map`** — compare functions whose names differ between files
- **`gasm diff --map`**; compare functions whose names differ between files
(e.g. `--map wideCopyAVX2=wideCopyAVX512` pairs two variants regardless
of suffix). Unmapped functions fall back to the original name match.
- **`gasm verify --call`** — invoke a single function with user-supplied buffers
- **`gasm verify --call`**; invoke a single function with user-supplied buffers
(`--buf name:size:pattern`) instead of the smoke/abi/fuzz sweeps. Patterns:
`zero`, `ones`, `seq`, or a hex blob. Useful for partial functions (e.g.
decoders) that crash on random input but should succeed on valid data.
@@ -346,23 +450,23 @@ new CLI commands. A signature-parser fix corrects grouped Go parameters.
### Fixed
- **Signature parser** — grouped Go parameters like `dst, src []byte` are now
- **Signature parser**; grouped Go parameters like `dst, src []byte` are now
parsed correctly (both get type `[]byte`). Previously the first name was
treated as its own type (`dst` with size 8), causing wrong ABI0 arg-block
layout in both `verify --call` and the fuzzer.
- **Flaky JIT tests** — `runtime.KeepAlive` guards and package-level buffers
- **Flaky JIT tests**; `runtime.KeepAlive` guards and package-level buffers
prevent GC from collecting heap objects whose addresses were passed to JIT
code via `unsafe.Pointer`; all verify tests pass 100/100 under `-race`.
### Changed
- **Removed external kernel test dependencies** — the verify test suite no
- **Removed external kernel test dependencies**; the verify test suite no
longer references production kernels from the separate go-libraries project.
The remaining test suite uses only `testdata/verify/*.s` kernels, which are
part of this repository. Coverage is identical locally and in CI (80.3 %).
## [0.28.0] — 2026-08-03
## [0.28.0] - 2026-08-03
RISC-V encoder: full RV64IMAFDC instruction set with RVC compression, MOV
pseudo-instruction, SB/global symbol references, ELF64 object emission, and
@@ -370,20 +474,20 @@ ground-truth verification against `GOARCH=riscv64 go tool asm`.
### Added
- **RISC-V encoder** — RV64I, RV64M, RV64A, RV64F/D, FMA, CSR, JALR.
- **MOV pseudo-instruction** — load, store, reg-to-reg, immediate, frame mapping.
- **RVC compression** — 22 compressed instruction types (C.LDSP, C.SDSP, C.FLDSP,
- **RISC-V encoder**; RV64I, RV64M, RV64A, RV64F/D, FMA, CSR, JALR.
- **MOV pseudo-instruction**; load, store, reg-to-reg, immediate, frame mapping.
- **RVC compression**; 22 compressed instruction types (C.LDSP, C.SDSP, C.FLDSP,
C.FSDSP, C.ADDI, C.LI, C.LUI, C.ADDIW, C.MV, C.ADD, C.SUB, C.XOR, C.OR, C.AND,
C.SLLI, C.SRLI, C.SRAI, C.ANDI, C.BEQZ, C.BNEZ, C.J, C.JR).
- **SB/global symbols** — `MOV $sym(SB)`, `MOV sym(SB)`, `MOV rd, sym(SB)`
- **SB/global symbols**; `MOV $sym(SB)`, `MOV sym(SB)`, `MOV rd, sym(SB)`
encoded as AUIPC pairs with R_RISCV_PCREL_HI20/LO12 relocations.
- **GLOBL/DATA** — data section layout in `AssembleFileRISCV`.
- **ELF64 emission** — `gasm asm --format elf` produces EM_RISCV objects
- **GLOBL/DATA**; data section layout in `AssembleFileRISCV`.
- **ELF64 emission**; `gasm asm --format elf` produces EM_RISCV objects
(.text, .data, .symtab, .rela.text).
- **`gasm verify --ground-truth`** — byte-exact comparison against
- **`gasm verify --ground-truth`**; byte-exact comparison against
`GOARCH=riscv64 go tool asm`.
- **`gasm verify --profile`** — function layout listing for RISC-V.
- **CALL** — AUIPC + JALR pair encoding.
- **`gasm verify --profile`**; function layout listing for RISC-V.
- **CALL**; AUIPC + JALR pair encoding.
### Fixed
@@ -393,7 +497,7 @@ ground-truth verification against `GOARCH=riscv64 go tool asm`.
(bit-interleaved format).
## [0.27.0] — 2026-08-01
## [0.27.0] - 2026-08-01
Subprocess isolation for `--fuzz`: each function is fuzzed in its own child
process, so a partial function (decoder) that faults on random garbage is
@@ -404,7 +508,7 @@ the parent. CRASH is informational (exit 0); only MISMATCH is an error.
- `gasm verify --fuzz` no longer crashes the process on partial functions.
## [0.26.0] — 2026-07-31
## [0.26.0] - 2026-07-31
Universal differential fuzzing: `gasm verify --fuzz` needs no hand-written
reference. It parses the `// func` signature from the assembly source,
@@ -415,7 +519,7 @@ area bit-for-bit.
### Added
- `verify`: `FuzzFunc` / `ExtractSignatures` / `parseFuncSig` — universal
- `verify`: `FuzzFunc` / `ExtractSignatures` / `parseFuncSig`; universal
differential fuzz driven by the conventional `// func` comment. Each
version gets its own buffer set (deep copy) so functions that write to
their arguments (histogram increments) don't corrupt the other's input.
@@ -430,16 +534,16 @@ area bit-for-bit.
over-copy paths read past the buffer on random garbage input. Subprocess
isolation (fork per function) is planned. Use `--ground-truth` for decoders.
## [0.25.0] — 2026-07-30
## [0.25.0] - 2026-07-30
Universal ground-truth verification: `gasm verify --ground-truth` assembles
any `.s` file with both gasm and `go tool asm`, then compares the machine
code byte-for-byte per function (relocation sites masked). No hand-written
reference needed — the Go toolchain IS the oracle.
reference needed; the Go toolchain IS the oracle.
### Added
- `verify`: `GroundTruth` — shells out to `go tool asm`, parses the GOOBJ
- `verify`: `GroundTruth`; shells out to `go tool asm`, parses the GOOBJ
output (minimal reader: block offsets, nonpkg symbol table, data index)
and returns per-function code bytes.
- `gasm verify --ground-truth`: compares gasm's output against the Go
@@ -448,11 +552,11 @@ reference needed — the Go toolchain IS the oracle.
linker fills) are masked before comparison.
- Verified: go-lz4 AVX2 2/2, go-flac AVX2 17/17 functions byte-identical.
## [0.24.0] — 2026-07-29
## [0.24.0] - 2026-07-29
The full analyze family and stereo PCM decode are now differentially tested.
15 of 17 go-flac AVX2 kernels have bit-for-bit differential coverage; the
two remaining (autocorrAVX2 — FMA reassociation, lpcResidualAVX2 — complex
two remaining (autocorrAVX2: FMA reassociation, lpcResidualAVX2: complex
multi-arg) are deferred.
### Added
@@ -462,7 +566,7 @@ multi-arg) are deferred.
- `verify`: `decodeStereo16AVX2` differential test (500 random interleaved
stereo PCM buffers, both channels compared sample-by-sample).
## [0.23.0] — 2026-07-28
## [0.23.0] - 2026-07-28
The analyze family and 24-bit PCM decode join the differential suite.
@@ -474,7 +578,7 @@ The analyze family and 24-bit PCM decode join the differential suite.
- `verify`: `decodeMono24AVX2` differential test (500 random 24-bit PCM
buffers, sign-extension compared sample-by-sample).
## [0.22.0] — 2026-07-27
## [0.22.0] - 2026-07-27
The remaining go-flac encoder kernels join the differential suite.
@@ -488,39 +592,39 @@ The remaining go-flac encoder kernels join the differential suite.
loop).
## [0.21.0] — 2026-07-26
## [0.21.0] - 2026-07-26
Differential testing extended to all four production kernels and the CLI
exposes the full dynamic-analysis toolkit.
### Added
- `verify`: go-flac AVX2 differential tests — `decodeMono16AVX2` (500
- `verify`: go-flac AVX2 differential tests; `decodeMono16AVX2` (500
random PCM buffers), `pack16AVX2` (500 random int32→int16 packings) and
all four decorrelation kernels (200 iterations each: left-side, side-right,
mid-side, interleave) compared bit-for-bit against the portable Go
references.
- `verify`: go-lz4 AVX-512 differential tests — `decodeBlockAVX512` (3 000
fuzzed LZ4 blocks + known answers) and `wideCopyAVX512` (0–1024 bytes)
- `verify`: go-lz4 AVX-512 differential tests; `decodeBlockAVX512` (3 000
fuzzed LZ4 blocks + known answers) and `wideCopyAVX512` (0-1024 bytes)
against the same portable oracle as the AVX2 suite.
- `gasm verify --abi`: runs each NOSPLIT function with sentinel registers
and a red-zone canary, reporting violations.
- `gasm verify --profile`: lists the static basic-block count per function.
## [0.20.0] — 2026-07-25
## [0.20.0] - 2026-07-25
Coverage profiling: the third pillar of Phase 3. Static basic-block
enumeration from the assembler's label map, combined with multi-input path
diversity measurement — how many observationally distinct execution paths a
diversity measurement; how many observationally distinct execution paths a
test corpus exercises.
### Added
- `verify`: `Kernel.Blocks` / `Kernel.BlockCount` — enumerate basic blocks
- `verify`: `Kernel.Blocks` / `Kernel.BlockCount`; enumerate basic blocks
from the assembler's local-label map (every jump target is a block
boundary; the function entry is always a block). `decodeBlockAVX2` has
27 blocks.
- `verify`: `Kernel.ProfilePaths` — run the function with a corpus of
- `verify`: `Kernel.ProfilePaths`; run the function with a corpus of
argument blocks and collect distinct output fingerprints (the result
words); reports path diversity as a lower bound on code coverage.
@@ -532,7 +636,7 @@ rt_sigaction handlers fragile in a Go process. The static + path-diversity
approach delivers the project's goal (proving the SIMD path and tail handling
execute) without fighting the runtime.
## [0.19.0] — 2026-07-24
## [0.19.0] - 2026-07-24
Runtime ABI checks: the second pillar of Phase 3. The JIT trampoline now
has an ABI-checking variant that sets sentinels in the callee-saved registers
@@ -542,41 +646,41 @@ detects any illegal write below the stack pointer.
### Added
- `verify`: `CallChecked` / `Kernel.CallFuncChecked` — ABI-checking JIT call
- `verify`: `CallChecked` / `Kernel.CallFuncChecked`; ABI-checking JIT call
with sentinel registers and red-zone canary; returns an `ABIReport`
(BPClobbered, R14Clobbered, RedZoneHit).
- `verify`: the raw `leaveJITCheckedRaw` trampoline — a TEXT symbol with no
- `verify`: the raw `leaveJITCheckedRaw` trampoline; a TEXT symbol with no
ABIInternal wrapper (address obtained via GLOBL/DATA), so the JIT
function's RET lands directly in the check code and sees the registers
exactly as the function left them.
- Tests: deliberate BP/R14 clobberers detected; both go-lz4 kernels
confirmed ABI-clean (BP preserved, R14 preserved, red zone intact).
## [0.18.0] — 2026-07-23
## [0.18.0] - 2026-07-23
Differential testing: the JIT-assembled go-lz4 `decodeBlockAVX2` kernel is
fuzzed against a portable Go reference — 5 000 valid LZ4 blocks compared
fuzzed against a portable Go reference; 5 000 valid LZ4 blocks compared
bit-for-bit, plus 2 000 hostile (random garbage) inputs with matching error
codes. This is the automated form of the project's bit-identical contract.
### Added
- `verify`: differential fuzz tests — a random LZ4 block generator produces
- `verify`: differential fuzz tests; a random LZ4 block generator produces
valid blocks (literals, overlapping matches, extension bytes) and the
JIT-assembled kernel's output is compared byte-for-byte against a portable
Go decoder; a hostile-input suite confirms error-code agreement on random
garbage (no crashes, same classification).
## [0.17.0] — 2026-07-22
## [0.17.0] - 2026-07-22
Phase 3 begins: dynamic analysis. A JIT execution substrate that assembles
Plan 9 amd64 kernels into executable memory and calls them directly — pure Go
Plan 9 amd64 kernels into executable memory and calls them directly; pure Go
(stdlib only, `syscall.Mmap` + an assembly trampoline), no cgo, no external
toolchain.
### Added
- `verify` package: JIT infrastructure — `Map` copies machine code into a
- `verify` package: JIT infrastructure; `Map` copies machine code into a
W^X memory mapping, `Call` invokes it through an ABI0 trampoline that
switches to a prepared stack and back. `Load`/`LoadSource`/`LoadAST`
parse, assemble and map a `.s` file in one step; `Kernel.CallFunc`
@@ -585,34 +689,34 @@ toolchain.
available functions; with `-smoke`, calls each NOSPLIT function with
zeroed arguments to confirm the trampoline works end-to-end.
- Integration tests: the go-lz4 `decodeBlockAVX2` and `wideCopyAVX2`
kernels (699 and 146 bytes) assemble, map and execute correctly —
known-answer LZ4 blocks decode bit-for-bit, wide copies of 0–1024 bytes
kernels (699 and 146 bytes) assemble, map and execute correctly;
known-answer LZ4 blocks decode bit-for-bit, wide copies of 0-1024 bytes
match, malformed input returns the correct error codes.
## [0.16.0] — 2026-07-21
## [0.16.0] - 2026-07-21
The scalar conversions between vector and general-purpose registers — the
The scalar conversions between vector and general-purpose registers; the
last of the amd64 EVEX instruction set.
### Added
- `asm`: the GPR-interchanging conversions, byte for byte against the Go
assembler (28 ground-truth cases including memory sources and extended
GPRs): vector to GPR — the signed and truncated VCVT{,T}S{D,S}2SI{,Q}
GPRs): vector to GPR; the signed and truncated VCVT{,T}S{D,S}2SI{,Q}
in both VEX and EVEX, and the unsigned VCVT{,T}S{D,S}2USI{L,Q}
(EVEX only); GPR to vector — VCVTSI2SD{L,Q}/VCVTSI2SS{L,Q} (VEX and
(EVEX only); GPR to vector; VCVTSI2SD{L,Q}/VCVTSI2SS{L,Q} (VEX and
EVEX) and VCVTUSI2SD{L,Q}/VCVTUSI2SS{L,Q} (EVEX only), whose preserved
vector source sits in vvvv (three Plan 9 operands).
## [0.15.0] — 2026-07-20
## [0.15.0] - 2026-07-20
The last of the EVEX conversions and narrowing/extending moves — the EVEX
The last of the EVEX conversions and narrowing/extending moves; the EVEX
instruction set is now complete save for the GPR-interchanging forms.
### Added
- `asm`: the unsigned and truncating conversions — VCVTPD2PS (and the X/Y
- `asm`: the unsigned and truncating conversions; VCVTPD2PS (and the X/Y
spellings, whose length the spelling fixes), VCVTPD2UDQ (X/Y),
VCVTTPD2UDQ (X/Y), VCVTTPD2UQQ, VCVTPS2UDQ, VCVTTPS2UDQ, VCVTPS2UQQ,
VCVTTPS2UQQ, VCVTTPD2QQ, VCVTTPS2QQ, VCVTUQQ2PD, VCVTUQQ2PS (X/Y) and
@@ -625,20 +729,20 @@ instruction set is now complete save for the GPR-interchanging forms.
D2M/Q2M), whose K register is a genuine operand rather than a mask and
which therefore take no masking suffixes.
## [0.14.0] — 2026-07-19
## [0.14.0] - 2026-07-19
The floating-point helper and conversion tail of the AVX-512 set, plus
gather and scatter with VSIB addressing — every encoding verified byte for
gather and scatter with VSIB addressing; every encoding verified byte for
byte against the Go assembler.
### Added
- `asm`: the floating-point helpers — reciprocals and reciprocal square
- `asm`: the floating-point helpers; reciprocals and reciprocal square
roots (VRCP14/VRSQRT14 PD/PS/SD/SS), exponents and mantissas (VGETEXP*,
VGETMANT*), scaling by powers of two (VSCALEF*), rounding (VRNDSCALE*),
reduction (VREDUCE*), immediate fixup (VFIXUPIMM*) and range selection
(VRANGE*), and floating-point class tests (VFPCLASSPD/PS X/Y/Z and
VFPCLASSSD/SS — a new immediate form whose reg field carries the opmask
VFPCLASSSD/SS; a new immediate form whose reg field carries the opmask
destination).
- `asm`: **gather and scatter with VSIB addressing.** The gathers take
both Go spellings: the VEX form with a vector mask register (OP mask,
@@ -647,14 +751,14 @@ byte against the Go assembler.
data register (a ZMM index with an YMM destination encodes L'L = 10, as
the Go assembler emits). The scatters (VSCATTER*/VPSCATTER*) are EVEX
only (OP src, K, vsib). All eight gather and eight scatter widths.
- `asm`: the remaining conversions — VCVTQQ2PS (the 512-bit source sets
- `asm`: the remaining conversions; VCVTQQ2PS (the 512-bit source sets
the length), VCVTPD2QQ/UQQ, VCVTPS2QQ, VCVTUDQ2PD/PS, the half-precision
VCVTPH2PS and VCVTPS2PH (the extract layout with an immediate).
## [0.13.0] — 2026-07-18
## [0.13.0] - 2026-07-18
The wider AVX-512 set: ternary logic, permutes, compares, expand/compress,
the opmask instructions and the EVEX rounding/SAE/broadcast suffixes — every
the opmask instructions and the EVEX rounding/SAE/broadcast suffixes; every
encoding verified byte for byte against the Go assembler.
### Added
@@ -662,7 +766,7 @@ encoding verified byte for byte against the Go assembler.
- `asm`: the wider EVEX/AVX-512 set, across roughly sixty new ground-truth
cases: ternary logic (VPTERNLOGD/Q), the lane shuffles/inserts/extracts
(VSHUF{F,I}{32,64}X{2,4}, the VINSERT*/VEXTRACT* {F,I}{32,64}X{2,4,8}
family, VPALIGNR), compares with an opmask destination (VCMPPD/PS/SD/SS —
family, VPALIGNR), compares with an opmask destination (VCMPPD/PS/SD/SS;
a new NDS-plus-immediate form with the K register in the reg field), the
permutes (VPERMB/W, VPERMI2/T2 D/Q/PD), the wider integer families
(VPMADDWD/UBSW, VPMULHUW, VPACKSSWB/USWB/SSDW/USDW, VPABS B/W/D/Q, the
@@ -676,15 +780,15 @@ encoding verified byte for byte against the Go assembler.
(VMOVSLDUP/VMOVSHDUP), the conversions (VCVTPS2DQ, VCVTTPS2DQ) and the
remaining extending and narrowing moves (VPMOVSXBW, VPMOVZXBW, VPMOVWB,
VPMOVQB).
- `asm`: the EVEX mnemonic suffixes the Go assembler accepts — the rounding
- `asm`: the EVEX mnemonic suffixes the Go assembler accepts; the rounding
modes `.RN_SAE`, `.RD_SAE`, `.RU_SAE`, `.RZ_SAE` (the EVEX b bit with the
rounding control in L'L), suppress-all-exceptions `.SAE`, and memory
broadcast `.BCST` (the b bit, the vector length preserved, disp8×N scaled
by the element size) — each combinable with the `.Z` zeroing suffix,
by the element size); each combinable with the `.Z` zeroing suffix,
validated against the Go assembler's bytes, and rejected on instructions
that do not support them.
## [0.12.0] — 2026-07-17
## [0.12.0] - 2026-07-17
GOOBJ emission: gasm-assembled functions drop into a `go build` without the
Go assembler.
@@ -692,7 +796,7 @@ Go assembler.
### Added
- `asm`: **GOOBJ object output.** `gasm asm --format goobj -p <pkgpath>`
writes the Go toolchain's own object format — the one `cmd/link` consumes
writes the Go toolchain's own object format; the one `cmd/link` consumes
directly: the functions as non-package symbols qualified with the package
path (exactly as `cmd/asm` records assembly symbols), the `GLOBL` data,
one serialized `FuncInfo` per function (argument/frame sizes, the asm
@@ -701,8 +805,8 @@ Go assembler.
real stack deltas: the assembler now tracks every stack-adjustment
boundary through the prologue (`PUSHQ BP`, `SUBQ $frame, SP`) and each
`RET`'s epilogue, so frame-pointer functions unwind correctly. The
object preamble — the version-and-experiment header the linker compares
verbatim — is captured from the installed `go tool asm`, so the output is
object preamble; the version-and-experiment header the linker compares
verbatim; is captured from the installed `go tool asm`, so the output is
always consistent with the toolchain that links it.
- `asm`: relocations against file-local `GLOBL` symbols become `R_PCREL`
entries in the GOOBJ output, with the instruction's displacement field
@@ -715,7 +819,7 @@ Go assembler.
pattern, instead of being rejected as non-integer.
## [0.11.0] — 2026-07-16
## [0.11.0] - 2026-07-16
Linkable object output: external symbols and relocatable ELF / Mach-O
objects.
@@ -734,7 +838,7 @@ objects.
external symbol; the Mach-O output is verified structurally with
`debug/macho`.
- `asm`: **external symbol references.** A reference to a symbol no
`GLOBL` in the file defines no longer aborts assembly — it is recorded
`GLOBL` in the file defines no longer aborts assembly; it is recorded
as an external relocation (`Image.Externals`, `FuncLayout.Relocs`) and
becomes an undefined global symbol in the object output. The raw image
s (`--format raw`, the default) still reports them: only an object
@@ -746,7 +850,7 @@ objects.
writes; without `--format` the behaviour is unchanged (the concatenated
image).
## [0.10.0] — 2026-07-15
## [0.10.0] - 2026-07-15
The EVEX floating-point and conversion set: the packed-double arithmetic,
the scalar SD/SS forms, VMOVDDUP and the width-changing conversions, each
@@ -754,23 +858,23 @@ verified byte for byte against the Go assembler.
### Added
- `asm`: the rest of the common EVEX/VEX floating-point set — packed double
- `asm`: the rest of the common EVEX/VEX floating-point set; packed double
arithmetic (VSUBPD, VDIVPD, VMINPD, VMAXPD, VUNPCKLPD and the EVEX form of
VUNPCKHPD), the scalar double and single operations (VSUBSD, VDIVSD,
VMINSD, VMAXSD and the full VADDSS/VSUBSS/VMULSS/VDIVSS/VMINSS/VMAXSS
family in both VEX and EVEX — the EVEX scalar forms exist for masked and
family in both VEX and EVEX; the EVEX scalar forms exist for masked and
zeroing use), and VMOVDDUP (lane duplication, VEX and EVEX).
- `asm`: the width-changing conversions — VCVTDQ2PS and VCVTPS2PD (VEX and
- `asm`: the width-changing conversions; VCVTDQ2PS and VCVTPS2PD (VEX and
EVEX; the destination sets the length for PS→PD), the EVEX form of
VCVTDQ2PD, and the packed-double → dword family: VCVTPD2DQ/VCVTTPD2DQ
(EVEX-512 only, a ZMM source and an XMM destination) and their X/Y
spellings (VCVTPD2DQX/Y, VCVTTPD2DQX/Y), whose length follows the wider
source — a new operand form, since the destination is always XMM while
source; a new operand form, since the destination is always XMM while
VEX.L / EVEX.L'L ride with the source (fixed by the spelling even for a
memory source).
- `asm`: masking and zeroing on every new form — the scalar SD/SS
- `asm`: masking and zeroing on every new form; the scalar SD/SS
arithmetic, the unpacks, VMOVDDUP and the conversions all accept the
explicit K1–K7 operand and the `.Z` suffix the way Go writes them.
explicit K1-K7 operand and the `.Z` suffix the way Go writes them.
### Changed
@@ -781,35 +885,35 @@ verified byte for byte against the Go assembler.
shares the convention).
## [0.9.0] — 2026-07-14
## [0.9.0] - 2026-07-14
AVX-512 masking and a wider EVEX integer set.
### Added
- `asm`: **EVEX masking** the way Go writes it — an explicit `K1`–`K7`
- `asm`: **EVEX masking** the way Go writes it; an explicit `K1`-`K7`
operand placed among the operands (merging mask), and a `.Z` mnemonic
suffix for zeroing (`VPADDD.Z Z1, Z2, K2, Z3`). Supported across the NDS,
reg/rm, immediate-shift, align, extract, convert and move forms, including
masked comparisons with a K destination (`VPCMPEQD Z0, Z3, K2, K1`). K0 is
rejected as an explicit mask, and `.Z` without a mask is an error, matching
the Go assembler.
- `asm`: the common AVX-512 F/BW integer set — VPADDB/W, VPSUBB/W, VPANDD/Q,
- `asm`: the common AVX-512 F/BW integer set; VPADDB/W, VPSUBB/W, VPANDD/Q,
VPANDND/Q, VPMULLW, VPAVGB/W, the signed/unsigned min/max family for
B/W/D/Q elements, the variable shifts VPSLLVD/Q, VPSRLVD/Q, VPSRAVD/Q, the
EVEX forms of VPSHUFD/VPSHUFB, and the VMOVDQU8/VMOVDQU16 move aliases.
Register indices 16–31 encode correctly (the mod=11 quirk carries rm[4]
Register indices 16-31 encode correctly (the mod=11 quirk carries rm[4]
in X̄). All verified byte for byte against the Go assembler.
- `lint`: masked EVEX forms (`.Z` suffix, K operands) are recognised by
`unknown-instruction` and exempted from `operand-count`.
### Fixed
- `asm`: EVEX register–register operands with indices 16–31 encoded rm[4]
- `asm`: EVEX register-register operands with indices 16-31 encoded rm[4]
into B̄ instead of X̄ (the EVEX mod=11 extension quirk), producing wrong
prefix bytes for X16+/Y16+ r/m operands.
## [0.8.0] — 2026-07-13
## [0.8.0] - 2026-07-13
Standard CLI ergonomics.
@@ -825,20 +929,20 @@ Standard CLI ergonomics.
- The version is primarily available as the standard `gasm --version` / `-V`
flag; the `gasm version` spelling remains as an alias.
## [0.7.0] — 2026-07-12
## [0.7.0] - 2026-07-12
The formatter behaves like `go fmt` and canonicalises block separation.
### Added
- `gasm fmt` now works like `go fmt`: with no arguments — or with a directory
argument — it reformats every `.s` file below it in place and lists the
- `gasm fmt` now works like `go fmt`: with no arguments; or with a directory
argument; it reformats every `.s` file below it in place and lists the
changed files, skipping `.` and `_` directories (`.git`, `_refs`, …).
Explicit file arguments keep the `-w` / standard-output behaviour.
### Changed
- `s`: canonical blank-line layout — a new block (a label, `TEXT` or
- `s`: canonical blank-line layout; a new block (a label, `TEXT` or
`GLOBL`) is preceded by exactly one blank line, neither more nor less.
Comments leading a block stay with it (the blank line goes before them),
stacked labels share their block, the function's first label keeps hugging
@@ -847,7 +951,7 @@ The formatter behaves like `go fmt` and canonicalises block separation.
kernels were reformatted with this release and remain byte-identical when
assembled.
## [0.6.0] — 2026-07-11
## [0.6.0] - 2026-07-11
Calibrated to the Go ABI: `register-clobber` stops reporting legal code, and
the encoder learns the legacy SSE moves.
@@ -856,13 +960,13 @@ the encoder learns the legacy SSE moves.
- `lint`: **`register-clobber` is now calibrated to the Go ABI**
(`cmd/compile/abi-internal.md`), not the platform ABI. Go's stack-based
ABI0 has no System V style callee-saved registers — amd64 `BX`, `R12`–`R15`
ABI0 has no System V style callee-saved registers; amd64 `BX`, `R12`-`R15`
and the arm64/riscv64/loong64 scratch sets are caller-saved or permanent
scratch, and hand-written kernels may clobber them freely. The rule now
audits only the registers Go fixes across calls: the frame pointer and the
goroutine pointer (amd64 `BP`/`R14`, arm64 `R18`/`R28`/`R29`, riscv64
`X27`, loong64 `R22`), and the goroutine pointer is reported only when the
function can reach the runtime (is not `NOSPLIT` or makes a call) — the
function can reach the runtime (is not `NOSPLIT` or makes a call); the
ABI0 transition restores it on those paths, and NOSPLIT call-free leaves
may use it, exactly as the runtime's own assembly does. Both go-flac
kernels now lint with zero diagnostics.
@@ -879,21 +983,21 @@ the encoder learns the legacy SSE moves.
### Added
- `asm`: the legacy (non-VEX) SSE moves — `MOVOU`/`MOVO` (the Plan 9 names
- `asm`: the legacy (non-VEX) SSE moves; `MOVOU`/`MOVO` (the Plan 9 names
for MOVDQU/MOVDQA), `MOVUPS`/`MOVAPS`/`MOVUPD`/`MOVAPD` and the scalar
`MOVSD`/`MOVSS` — and `VMOVDQU64` in the EVEX set. All verified byte for
`MOVSD`/`MOVSS`; and `VMOVDQU64` in the EVEX set. All verified byte for
byte against the Go assembler.
## [0.5.0] — 2026-07-10
## [0.5.0] - 2026-07-10
EVEX / AVX-512: the go-flac AVX-512 kernel now assembles, byte-identically to
the Go toolchain, completing the production-kernel coverage.
### Added
- `asm`: **EVEX (AVX-512) encoding** — the four-byte EVEX prefix with the
5-bit register fields (Z0–Z31, X/Y 16–31, with the reg-r/m X̄ quirk and
V'̄ shared between vvvv and the SIB index), opmask registers (K0–K7) as
- `asm`: **EVEX (AVX-512) encoding**; the four-byte EVEX prefix with the
5-bit register fields (Z0-Z31, X/Y 16-31, with the reg-r/m X̄ quirk and
V'̄ shared between vvvv and the SIB index), opmask registers (K0-K7) as
operands and as mask destinations, and the compressed disp8×N displacement
(the multiplier follows the memory operand's size, as the Go assembler's
opcode tables prescribe). Covers every AVX-512 instruction the go-flac
@@ -903,20 +1007,20 @@ the Go toolchain, completing the production-kernel coverage.
extracts VEXTRACTI64X4/VEXTRACTF64X4, VFMADD231PD, VADDPD, VMULPD, the
broadcasts VPBROADCASTD/Q (GPR and memory sources take different opcodes)
and the mask moves KMOVW/KTESTW. Masking/zeroing suffixes are out of scope
— the kernels use neither.
; the kernels use neither.
- `asm`: `AssembleFile` now accepts file-defined global (`non-<>`) symbols
too; a reference is external only when no `GLOBL` in the file defines it.
### Fixed
- `asm`: registers X16–Y31 force the EVEX encoding of dual-form mnemonics;
- `asm`: registers X16-Y31 force the EVEX encoding of dual-form mnemonics;
previously a `VPBROADCASTD AX, Y30` fell into the VEX encoder, which cannot
represent indices above 15 and silently truncated them.
- `asm`: the VEX encoder now rejects vector register indices 16–31 instead of
- `asm`: the VEX encoder now rejects vector register indices 16-31 instead of
encoding a truncated (wrong) register.
## [0.4.0] — 2026-07-09
## [0.4.0] - 2026-07-09
The standalone assembler reaches the whole go-flac AVX2 kernel: static
symbols assemble, and all 17 kernel functions now match the Go toolchain's
@@ -924,10 +1028,10 @@ machine code byte for byte.
### Added
- `asm`: **file-level assembly** — `AssembleFile` turns a parsed file into an
- `asm`: **file-level assembly**; `AssembleFile` turns a parsed file into an
`Image`: the function bodies in source order followed by a data section
built from the file's `GLOBL`/`DATA` directives (each symbol 16-aligned).
- `asm`: **static-symbol (`SB`) operands** — `mask<>(SB)` references encode as
- `asm`: **static-symbol (`SB`) operands**; `mask<>(SB)` references encode as
RIP-relative loads with a patched disp32, resolved against the image layout
so the output is self-consistent and position-independent. External
(non-file-local) symbols are rejected with a clear error: they need
@@ -936,7 +1040,7 @@ machine code byte for byte.
and writes the whole image (code + data) with `-o`.
## [0.3.0] — 2026-07-08
## [0.3.0] - 2026-07-08
The assembler reaches byte-identical parity with the Go toolchain on the
production go-flac AVX2 kernels: every one of the 15 kernel functions that
@@ -946,18 +1050,18 @@ support).
### Added
- `asm`: the scalar instruction families the kernels use — `CMOVcc` and
- `asm`: the scalar instruction families the kernels use; `CMOVcc` and
`SETcc` (conditions spelled exactly like the jumps), `LZCNT`/`TZCNT`
(legacy `F3 0F BD/BC`), the sign/zero-extending moves (`MOVBLZX`, `MOVBQZX`,
`MOVWLZX`, `MOVWQZX`, `MOVWLSX`, `MOVLQSX`), `CVTSL2SD`/`CVTSQ2SD` (the
legacy SSE encoding, as the Go assembler emits it), the traditional
three-operand `IMUL3{W,L,Q}`, and the variable-count vector shifts
(`VPSRLQ X0, Y8, Y8` — the count in an XMM register or memory takes the
(`VPSRLQ X0, Y8, Y8`; the count in an XMM register or memory takes the
ordinary NDS form).
- `asm`: **jump relaxation** — jumps start in the short (rel8) form and
- `asm`: **jump relaxation**; jumps start in the short (rel8) form and
expand to rel32 when the settled displacement does not fit, iterating the
layout to a fixed point (CALL is always rel32).
- `asm`: **jump-to-jump folding** — a conditional jump to a label whose only
- `asm`: **jump-to-jump folding**; a conditional jump to a label whose only
instruction is an unconditional jump is redirected to the ultimate
target, replicating the Go toolchain's linker, which chases such chains
before it encodes branches.
@@ -969,12 +1073,12 @@ support).
- `asm`: `CMP` with a register or memory operand computed **second − first**
instead of first − second, silently inverting every condition that followed
(`CMPQ SI, R10; JGE` tested R10 ≥ SI). The encoding now always records
first − second — `CMP r/m, r` with the first operand in r/m, `CMP r, r/m`
with the first operand in reg — and is byte-identical to the Go assembler.
first − second; `CMP r/m, r` with the first operand in r/m, `CMP r, r/m`
with the first operand in reg; and is byte-identical to the Go assembler.
- `asm`: register-to-register `MOV` now uses the `r/m ← r` opcode (reg =
source), the Go assembler's choice; the output is byte-identical.
## [0.2.0] — 2026-07-07
## [0.2.0] - 2026-07-07
The Phase 2 assembler grows the SIMD set: shuffles, extract/insert, permute
and the moves, on top of the Phase 1 VEX forms.
@@ -987,10 +1091,10 @@ and the moves, on top of the Phase 1 VEX forms.
- the immediate shuffle (`VPSHUFD`, `VPERMQ`),
- the three-operand-plus-immediate form (`VSHUFPD`, `VPERM2I128`,
`VINSERTI128`),
- the lane extract (`VEXTRACTI128`, `VEXTRACTF128` — the YMM source occupies
- the lane extract (`VEXTRACTI128`, `VEXTRACTF128`; the YMM source occupies
the ModRM.reg field, the XMM/memory destination the r/m field),
- the direction-sensitive moves (`VMOVDQU`, `VMOVUPD`, `VMOVD`, `VMOVQ`,
`VMOVSD` — each direction picks its own opcode and VEX.W; a vector→vector
`VMOVSD`; each direction picks its own opcode and VEX.W; a vector→vector
move uses the store-form layout, matching the Go assembler),
- the no-operand `VZEROUPPER`, and `VPERMD` in the NDS form,
- the floating-point and FMA set (`VADDPD`, `VMULPD`, `VXORPD`,
@@ -999,21 +1103,21 @@ and the moves, on top of the Phase 1 VEX forms.
the encoder now covers every integer, shuffle and FP instruction the
go-flac AVX2 kernels use.
- `asm`: `CMP` accepts the immediate in the second operand position
(`CMPL CX, $31`) — the spelling the Go assembler accepts — encoding it
(`CMPL CX, $31`), the spelling the Go assembler accepts, encoding it
identically to the immediate-first form.
### Fixed
- `asm`: an unused VEX.vvvv field is now stored as `1111` (v̄vvv = 1111), as
the hardware requires — the previous value (`0000`) made the two-operand
the hardware requires; the previous value (`0000`) made the two-operand
reg/rm forms (VPMOVSXWD, VPBROADCASTD, VMOVMSKPS, …) raise #UD on real CPUs
and differ from the Go assembler's bytes. The round-trip decoder ignores
the field on these instructions, which is why the byte-for-byte Go
comparison (added this release) is now part of the test suite.
## [0.1.0] — 2026-07-06
## [0.1.0] - 2026-07-06
Initial release — the Phase 1 foundation.
Initial release; the Phase 1 foundation.
### Added
@@ -1026,11 +1130,11 @@ Initial release — the Phase 1 foundation.
- `arch`: register files and **complete** instruction tables for amd64,
arm64, riscv64 and loong64, with the middle-dot symbol separator and static
(`<>`) symbols. Instruction names are generated from the Go toolchain's own
assembler source (`just gen`) — the `anames` opcode lists plus the common
assembler source (`just gen`); the `anames` opcode lists plus the common
opcodes and the per-architecture front-end aliases (arm64 `B`/`BL`, the
`.P`/`.W` addressing suffixes, loong64 `JAL`, the x86 conditional-jump
spellings) — so every mnemonic the real assembler accepts is recognised.
- `lint`: conservative rules — `unknown-instruction`, `operand-count`,
spellings); so every mnemonic the real assembler accepts is recognised.
- `lint`: conservative rules; `unknown-instruction`, `operand-count`,
`undefined-label`, `duplicate-label`, `missing-ret`,
`missing-textflag-include`, `abi-argsize` and `unreachable-code`. Macro
invocations are recognised (in-file `#define` names and underscore
@@ -1052,9 +1156,9 @@ Initial release — the Phase 1 foundation.
- `lsp`: a Language Server Protocol server over stdio providing completion,
hover documentation, document symbols, publish-diagnostics and semantic-token
highlighting.
- `asm`: a standalone amd64 (x86-64) assembler — an instruction encoder (REX/
- `asm`: a standalone amd64 (x86-64) assembler; an instruction encoder (REX/
ModR-M/SIB/displacement/immediate plus the scalar instruction set, and VEX/
AVX2 SIMD across three operand forms — NDS, reg/rm and immediate-shift —
AVX2 SIMD across the three operand forms NDS, reg/rm and immediate-shift,
covering the bulk of the integer SIMD set) validated by round-trip decoding
against `golang.org/x/arch`, and an assembler that drives the parser's AST
into the encoder with local-label resolution and `FP`/`SP` frame mapping
+98 -78
View File
@@ -1,107 +1,127 @@
# Contributing to gasm-devkit
# Contributing
Thanks for contributing to gasm-devkit.
Contributions to **gasm-devkit** are governed by the Contributor terms
below; submitting one means you accept them.
## Contributor terms
1. This project belongs to its owner alone. The owner decides what is
accepted, in what form and when; the decision is final and needs no
justification.
2. By submitting a contribution you assign to Petr Balvín
<opensource@petrbalvin.org> all present and future copyright and
related rights in it, worldwide, for the full term of the rights,
with the right to relicense and sublicense without restriction,
including under proprietary terms.
3. Where that assignment is not effective, it counts as a perpetual,
irrevocable, royalty-free licence with the same scope.
4. To the fullest extent permitted by law, you waive any right of
attribution and integrity in the contribution. The project names no
contributors and keeps no credits list.
5. By submitting you represent that the work is yours and that you
hold the rights to assign it as above.
## Development setup
Requirements: Go 1.27 or later, the [just](https://github.com/casey/just)
command runner, and a Linux host on amd64, arm64, riscv64 or loong64.
Requirements: Go 1.27.1, the exact version the `go` directive in `go.mod`
declares, and [just](https://github.com/casey/just) for the recipes.
```sh
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
cd gasm-devkit
just install # download module dependencies
just build # go vet + gofmt check
just test # full suite, race detector, 80 % coverage gate
just build
just gates
```
## Workflow
1. Branch from `development`; never commit directly to `main` (`main` is
release-only: merge from `development`, then tag).
2. Commit with [Conventional Commits](https://www.conventionalcommits.org/):
`type(scope): description`: subject line only, imperative mood,
lowercase after the colon, no trailing dot. Allowed types: `feat`,
`fix`, `docs`, `style`, `refactor`, `perf`, `test`, `chore`, `ci`,
`build`, `revert`. The only line after the subject is the trailer:
`Assisted-by: <model-name>`. No `Co-Authored-By`, no `Signed-off-by`,
no other trailers.
3. Record every user-visible change in `CHANGELOG.md` under
`## [development]` (categories: Added, Changed, Fixed, Removed,
Security).
4. Add or update tests; coverage must stay **at or above 80 %** (hard
gate, enforced by CI).
5. Update the documentation when behaviour, flags or the public surface
change.
6. Open a pull request against `development`.
1. Branch from `development`. Never commit directly to `main`, which is release-only.
2. Commit in [Conventional Commits](https://www.conventionalcommits.org/) form:
`type(scope): description`, subject line only, imperative mood, lowercase after the
colon, no trailing full stop. Allowed types: `feat`, `fix`, `docs`, `style`,
`refactor`, `perf`, `test`, `chore`, `ci`, `build`, `revert`.
3. One logical change per commit. A refactor, a behaviour change and a formatting pass
are three commits, never one.
4. Record every user-visible change in `CHANGELOG.md` under `## [development]`.
5. Add or update tests. Coverage stays at 80 percent or more; it is a hard gate.
6. Update the documentation when the public API, the configuration or the behaviour
changes.
7. Open a pull request against `development`.
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`;
CI builds and publishes the binaries for all four architectures.
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The release
workflow builds the assets and publishes the release and its notes.
## Code style
`gofmt` and `go vet` via `just fmt` / `just build`; both must pass with
zero output; `go fix -diff ./...` must report nothing on touched packages.
`gofmt` and `go vet` run through `just fmt` and `just vet`, with zero diff and zero
warnings tolerated. `just gates` is the definition of done in one command, and the recipe
file names what it contains. Errors are checked explicitly, wrapped as
`fmt.Errorf("context: %w", err)`, and nothing panics outside `main`. The `golang`
skill holds the rules the project follows; the recipe file holds the commands.
- Standard library only in production code; `golang.org/x/arch` is used
in tests only (round-trip decoding) and is never linked into the `gasm`
binary.
- No cgo, no C, no external toolchains at runtime.
- Explicit `if err != nil`; errors wrapped with
`fmt.Errorf("context: %w", err)`; no panics outside `main`.
- The parser, lexer and formatter are hand-written; the `arch` instruction
tables are generated only via `_gen/gen.go` (`just gen`), never edited.
- `golang.org/x/arch` is the one module dependency, and it is linked into the binary:
`gasm dis` and the debugger's listings decode through it. Everything else is the
standard library.
- No cgo, no C, no external toolchain at runtime.
- The parser, lexer and formatter are hand-written; the `arch` instruction tables are
generated only by `_gen/gen.go` (`just gen`) and never edited by hand.
- Assembly committed to the repository goes through `gasm fmt` and `gasm lint`, so a
`.s` file that `gasm fmt -l .` lists is unfinished.
## Running a single test
New source files open with the project's two-line licence header, whose SPDX
identifier matches `LICENSE`. Configuration files, workflows and dotfiles do not carry
it.
```sh
go test -run TestVexGroundTruth ./asm/
go test -run TestGroundTruthBasic ./verify/
go test -run TestGOObjectLinkAndRun ./asm/
go test -run TestFuzzWideCopy ./verify/
```
## AI contribution policy
The interactive debugger (`gasm debug`) requires a compiled binary on
`$PATH`; `go run` does not work for the traced child process. Install
first with `just install-bin`.
AI tools are welcome as productivity aids and are a normal part of modern software
development. What matters is that the contribution stays understandable, reviewable and
genuinely useful.
## CI (Gitea Actions)
- **Disclose the assistance.** If AI helped draft any part of a commit, issue, pull
request or review, say so.
- **Commit messages carry exactly one trailer**, as a git trailer on the line after a
blank line that closes the subject:
Workflows live in `.gitea/workflows/` and run on self-hosted runners:
```
Assisted-by: MODEL
```
Name the model that did the work, spelled the way its maker spells it, for example
`GLM 5.3`, `DeepSeek V4.1 Flash` or `Qwen 3.8 Flash`. No `Co-Authored-By`, no `Signed-off-by`,
no other trailers, and no prose: the trailer is the disclosure.
- **Issues and pull requests** attribute the assistance in a comment, for example
`_Assisted-by: GLM 5.3_`. It does not belong in the pull request description.
- **Take responsibility.** You are accountable for the accuracy, completeness and
intent of everything you submit, whether or not AI produced it.
- **Review before marking ready.** Read the diff carefully, run it locally, and add the
tests it needs. Do not mark a pull request ready until you can defend every change in
it.
- **Quality over quantity.** Contributions that look like un-reviewed output, or whose
author cannot engage substantively during review, may be closed.
- **Preferred models.** Prefer open-weight models with transparent training data and
minimal output filtering.
AI assists. It does not replace judgement.
## Continuous integration
Workflows live in `.gitea/workflows/` and run on the project's own runners:
| Workflow | Trigger | What it does |
|----------|---------|--------------|
| Test | push / PR to `development` | gofmt check, `go vet`, `go test -race`, 80 % coverage gate |
| Release | tag `v*` | cross-compiles binaries for linux/{amd64,arm64,riscv64,loong64} and publishes the Gitea release |
|---|---|---|
| Test | push or pull request to `development` | build, format check, vet, modernisation, the test suite with the coverage floor |
| Release | a `v*` tag | the same gates as Test, then the matrix build, the proven version and the release itself; the race detector runs locally in `just gates` before the tag is cut |
The Definition of Done (`just build` + `just test` + `just fmt`) must
still pass locally before pushing.
## AI Contribution Policy
AI tools are welcome as productivity aids. What matters is that
contributions remain understandable, reviewable, and genuinely useful.
- **Disclose AI use.** If you used AI to draft or generate any part of a
commit, issue, pull request, or code review, say so clearly.
- **Commit messages:** end every commit with exactly one trailer:
`Assisted-by: <model-name>` (e.g. `Assisted-by: GLM 5.3`).
- **Pull requests and issues:** attribute AI assistance in one trailing
line, e.g. `_Assisted-by: GLM 5.3_`. Do not paste it into the PR
description as a section.
- **Take responsibility.** You remain accountable for the accuracy,
completeness, and intent of everything you submit.
- **Review before marking ready.** Read AI-generated diffs carefully, run
them locally, and add or update tests where appropriate.
- **Preferred models.** Prefer open-weight models with transparent
training data: **GLM**, **DeepSeek**, and **MiMo**.
The local equivalent is `just gates`, which is the same set plus the race detector. The
race detector also has its own workflow, dispatched by hand; it never runs on a push or a
tag, where it would double the time and the memory a shared runner cannot spare.
## Reporting bugs
Open an issue at
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit/issues)
with the version (`gasm --version`), OS and architecture, the exact
command, the full output, and the expected versus actual behaviour.
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-devkit/issues` with the
version, the operating system and architecture, the exact command, the full output,
and the expected against the actual behaviour.
**Security issues:** email **opensource@petrbalvin.org** instead of opening
a public issue.
**Security issues do not go in the issue tracker.** Report them as
[SECURITY.md](SECURITY.md) describes, to **opensource@petrbalvin.org**.
+115 -25
View File
@@ -1,13 +1,63 @@
# gasm-devkit
# Plan 9 assembly tooling, inside and outside Go
Developer tooling for **GAsm**, Go's built-in Plan 9 assembler.
> **Warning: this is an experiment.** gasm-devkit is under active
> development and is not stable. The version is 0.x.x: commands, flags,
> output formats and behaviour can change without warning at any time.
> A 1.0.0 release is light years away. Nothing in this document is a
> stability promise. For all of that, this is not a paper project: gasm
> is already in active use and is tested on real assembly work.
Go ships an assembler but no tooling for it: there is no syntax highlighting,
no autocomplete, no linter, no static analyser, no formatter, no standalone
assembler and no debugger for `.s` files. Developers write assembly blind,
validate it by benchmark, and debug it by print statement. gasm-devkit is the
missing toolkit: a single, self-contained binary, `gasm`, that brings proper
developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
**GAsm** is Go's Plan 9 assembler, and Go ships it without tooling:
there is no formatter, no linter, no static analyser, no standalone
assembler and no debugger for `.s` files. Developers write assembly
blind, validate it by benchmark, and debug it by print statement.
gasm-devkit is the missing toolkit: a single, self-contained binary,
`gasm`, that serves both purposes.
- **Help develop Plan 9 assembly.** Formatting, linting, disassembly,
dynamic verification, a source-level debugger and a language server,
for `.s` files in Go programs.
- **Use Plan 9 assembly outside the Go toolchain.** `gasm asm` encodes
on its own, with no Go installation in the loop, and writes raw
images, linkable ELF objects with DWARF5 debug sections, or the Go
toolchain's own GOOBJ format, which `go build` consumes in place of
the toolchain's output.
## Why Plan 9 assembly
Plan 9 assembly is the quiet triumph of the field. One syntax across
every architecture Go builds for: the same source-first operand order,
the same four pseudo-registers, the same frame convention, whether the
target is x86, ARM, RISC-V or LoongArch. Learn it once and you can
read a kernel on any of them.
Compare the alternatives. Intel syntax and AT&T syntax disagree on the
one question every instruction answers, which operand is the source
and which is the destination, so half the world writes it one way,
half the other, and every assembly programmer carries both in their
head forever. GNU as settles the argument with directives that switch
dialects mid-file (`.intel_syntax noprefix`), a percent sign on every
register and a dollar on every immediate: punctuation that carries
nothing the operand order did not already say. And the x86 family
fragments again underneath: NASM is not MASM is not GAS, each with its
own directive zoo and macro language, so every project picks a dialect
and every reader learns a different one by accident.
Plan 9 assembly has none of it. Registers are bare names. Memory is
one notation, `offset(base)`, extended by an index and a scale when
the instruction needs it. Arguments arrive named and offset-checked:
`x+0(FP)` is the argument x, on every architecture, and `go vet`
polices the offsets against the Go prototype.
```text
AT&T (GNU as): movq %rax, -16(%rbp)
Plan 9 (Go): MOVQ AX, total-16(SP)
```
The same lines, but only one of them tells you what the number is for.
The syntax is uppercase, regular and boring, which is the highest
compliment a language for machine code can earn. gasm-devkit exists
to give that syntax the tooling it deserves.
## Features
@@ -16,7 +66,8 @@ developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
directly.
- **Formatter.** `gasm fmt` canonicalises indentation, operand spacing,
per-function mnemonic alignment and blank-line layout: `gofmt` for assembly,
operating recursively on directories the way `go fmt` does.
operating recursively on directories the way `go fmt` does. `-l` lists
files whose formatting differs and `-d` prints a unified diff.
- **Linter.** `gasm lint` runs 18 conservative static checks, among them
`undefined-label`, `abi-argsize` (declared frame vs the `// func` signature),
`register-clobber` (Go ABI register liveness over the control-flow graph),
@@ -24,7 +75,11 @@ developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
- **Standalone assembler.** `gasm asm` encodes all four architectures without
the Go toolchain and writes raw images, linkable ELF objects (with DWARF5
debug sections) or the Go toolchain's own GOOBJ format, which `go build`
consumes in place of the toolchain's output.
consumes in place of the toolchain's output. Framed functions get the
stack-split guard and the morestack block, byte-identical to the
toolchain's, so split functions link too.
- **Disassembler.** `gasm dis` lists a `.s` file's functions at their real
offsets after assembling, or disassembles raw bytes from a file or stdin.
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
executable memory: smoke calls, ABI checks (sentinel registers, red-zone
canary), differential fuzzing against the `go tool asm` build, and
@@ -36,17 +91,17 @@ developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
push and pull diagnostics, semantic-token highlighting, go-to-definition,
find references, rename, formatting, inlay hints, code actions, signature
help, document highlights, workspace symbol search, #include document
links and folding ranges over stdio.
links and folding ranges over stdio; definition, references and rename
work across every open document.
- **Comparators and audits.** `gasm diff` compares the machine code of two
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
`gasm audit-instructions` diffs the encoder against the installed toolchain,
and `gasm scaffold` generates a differential test skeleton for a kernel.
- **Complete instruction coverage.** The instruction tables are generated
from the Go toolchain's own assembler source, so the toolkit recognises
every mnemonic the real assembler accepts; `just gen` refreshes them.
### Architecture support
Four architectures, the four that matter in practice:
| Architecture | GOARCH | File suffix | Instructions recognised |
|--------------|-------------|--------------|---------------------------------------------|
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
@@ -57,27 +112,59 @@ developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
"Common opcodes" are the instructions shared by every architecture (`RET`,
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
...) that the assembler accepts as aliases. Regenerating the tables is one
command (`just gen`) and requires only a Go installation; the committed output
has no runtime dependency on the toolchain.
...) that the assembler accepts as aliases. The tables are generated from
the Go toolchain's own assembler source (`just gen` refreshes them), so
every mnemonic the real assembler accepts is recognised; what the encoder
can emit today is narrower, and a recognised but unencodable instruction is
reported as an explicit error, never as a wrong byte.
## Direction
The plan, in the order it is being worked:
- **Extended instruction support.** Two layers. First, encoding
coverage for every mnemonic the Go toolchain itself accepts, closed in
order of how often real code needs each instruction;
`gasm audit-instructions` measures the gap. Second, the larger work:
an extended instruction set the toolchain does not know at all. The
toolchain-derived tables stay generated and untouched; only the
extended instructions are hand-maintained, with their own spellings
and encoders, verified by execution on real hardware because the
toolchain offers no ground truth to compare against. The gaps exist
on every architecture, amd64 included.
- **Full GOOBJ and ELF compilation.** The destination is a complete,
standalone compilation path: linkable ELF objects for consumers outside
Go, and GOOBJ objects that `go build` links directly. Through GOOBJ, a
Go program will be able to use machine instructions that the Go
toolchain itself does not support; through ELF, Plan 9 assembly becomes
usable outside Go entirely.
- **Platforms: Linux and FreeBSD.** Linux is supported today on all four
architectures and is where the binary builds. FreeBSD follows: the
JIT's executable-memory mapping and the ptrace debugger layer are the
two pieces of porting work. Other unix systems may follow those two.
- **Four architectures, no more.** amd64, arm64, riscv64 and loong64.
No others are planned.
## Install
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
linux/loong64 are on the
[releases page](https://sourcedock.dev/petrbalvin/gasm-devkit/releases).
From source (Go 1.27 or later):
From source (Go 1.27.1):
```sh
go install sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm@latest
```
Or from a repository checkout, with the development version stamped:
Or from a repository checkout:
```sh
just install-bin
just install
```
The installed binary reports the version the toolchain recorded: the tag
on a tagged checkout, a pseudo-version naming the commit below one.
## Quick start
```sh
@@ -102,9 +189,13 @@ gasm verify --call add --args a=2,b=3 hello_amd64.s # JIT-call it with argumen
```sh
gasm fmt # reformat every .s below here, like go fmt
gasm fmt -w kernel_amd64.s # canonicalise one file in place
gasm fmt -l *.s # list files whose formatting differs
gasm fmt -d kernel_amd64.s # print a unified diff instead
gasm lint *.s # static checks
gasm asm --format elf -o k.o k.s # assemble to a linkable ELF object
gasm asm --format goobj -p pkg/path -o k.o k.s # Go object, consumed by go build
gasm dis k.s # assemble, then list each function
gasm dis -a amd64 - < dump.bin # disassemble raw bytes from stdin
gasm verify --ground-truth k.s # byte-for-byte vs go tool asm
gasm verify --fuzz k.s # differential fuzz vs the go tool asm build
gasm debug --func name k.s # interactive debugger
@@ -132,9 +223,9 @@ infers the target architecture from the file-name suffix
## Development
```sh
just install # download module dependencies
just build # go vet + gofmt check, zero errors and zero warnings
just test # full suite, race detector, 80 % coverage gate
just build # compile, zero errors and zero warnings
just test # the suite, no cache, the 80 % coverage floor
just gates # build, fmt-check, vet, test, race: the definition of done
just fmt # gofmt the tree
just gen # regenerate the instruction tables from the Go toolchain
```
@@ -148,11 +239,10 @@ recipe.
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/CLI.md](docs/CLI.md): full command reference
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
- [docs/DECISIONS.md](docs/DECISIONS.md): deferred design decisions
- [CHANGELOG.md](CHANGELOG.md): release history
## Licence
BSD-3-Clause — see [LICENSE](LICENSE).
BSD-3-Clause; see [LICENSE](LICENSE).
Copyright © 2026 [Petr Balvín](https://petrbalvin.org)
+40
View File
@@ -0,0 +1,40 @@
# Security policy
## Supported versions
Security fixes go to the newest release and to the `development` branch. Older
releases do not receive them.
| Version | Supported |
|---|---|
| 0.33.0 | yes |
| older releases | no |
## Reporting a vulnerability
**Do not open a public issue for a security problem.** A public report tells everyone
about the flaw before there is a fix. Report it privately to
**opensource@petrbalvin.org**.
Include:
- the version or commit you tested, and the platform
- what the problem is, and what an attacker gains from it
- the smallest reproducer you have, ideally a test or a single command
- a suggested fix, if you have one
## What to expect
- A human reads the report, and you get an acknowledgement.
- You are kept informed while the fix is being made, and told when it ships.
- The fix is released before the details are published, and the timing is agreed with
you.
- The reporter is credited in the release notes unless they ask otherwise.
## Out of scope
- Findings that require the attacker to already run code as the user, or to have local
access.
- Missing hardening with no demonstrated impact.
- Flaws in a third-party dependency: report them to that project, and to this one only
when this project's use of it makes them reachable.
+1 -1
View File
@@ -29,7 +29,7 @@ func arm64Registers() []Register {
regs = append(regs, Register{Name: name, Class: class, Desc: desc})
}
// General-purpose integer registers R0–R30.
// General-purpose integer registers R0-R30.
for i := 0; i <= 30; i++ {
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
}
+113 -53
View File
@@ -12,17 +12,21 @@ import (
// assembleARM64 assembles an AArch64 (arm64) TEXT function body into machine
// code. Every instruction is 4 bytes; the MOV pseudo-instruction and the
// immediate-arithmetic forms expand to 2–4 instructions when the immediate
// immediate-arithmetic forms expand to 2-4 instructions when the immediate
// does not fit, so the layout is computed in two passes (sizes, then encoding
// with resolved branch targets).
//
// The emitted bytes match the Go toolchain's arm64 assembler, which is the
// ground-truth oracle: prologue/epilogue, FP/SP frame mapping, branch
// encodings and the MOV immediate expansions all follow cmd/internal/obj/
// arm64's asmout cases.
// arm64's asmout cases. One deliberate difference: the stack-growth guard
// (the morestack check in the prologue and the call back into the runtime in
// the epilogue) is not emitted, so the bytes match only for NOSPLIT functions
// or zero-frame leaves, where the toolchain emits no guard either.
func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
fi := arm64ComputeFrame(t)
prologue := arm64Prologue(fi)
guardLen := arm64GuardLen(fi)
chain := arm64JumpChain(t)
resolve := func(name string) string {
if r, ok := chain[name]; ok {
@@ -35,14 +39,14 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
var spadj []SpadjStep
// The prologue (3 instructions when a small frame, 4 for large)
// raises the SP delta by autosize.
// raises the SP delta by autosize. The guard prefix shifts its PC.
if fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: arm64PrologueSpadjPC(fi), Value: fi.autosize})
spadj = append(spadj, SpadjStep{PC: guardLen + arm64PrologueSpadjPC(fi), Value: fi.autosize})
}
// Pass 1: label offsets from the instruction sizes.
offsets := map[string]int{}
pos := len(prologue)
pos := guardLen + len(prologue)
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
@@ -52,9 +56,25 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
}
}
// Pass 2: encode. Relocation offsets are recorded function-relative.
out := append([]byte(nil), prologue...)
pc := len(prologue)
// Pass 2: encode. The guard prefix precedes the prologue; its branches
// target the morestack block at the end of the function, whose position
// the first pass has settled.
bodyLen := 0
{
p := guardLen + len(prologue)
for _, stmt := range t.Body {
if in, ok := stmt.(*ast.Instr); ok {
p += arm64InstrSize(in, fi)
}
}
bodyLen = p - (guardLen + len(prologue))
}
var out []byte
if fi.needSplit {
out = append(out, arm64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...)
}
out = append(out, prologue...)
pc := guardLen + len(prologue)
preCount := len(relocs)
var lines []LineEntry
for _, stmt := range t.Body {
@@ -67,7 +87,12 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
}
for j := preCount; j < len(relocs); j++ {
relocs[j].Off += pc - len(prologue)
// Make the relocation offsets function-relative: each instruction
// records its reloc offset relative to its own start, and pc is
// that instruction's offset from the function start (prologue
// included). After shifts by the same amount.
relocs[j].Off += pc
relocs[j].After += pc
}
preCount = len(relocs)
lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line})
@@ -79,6 +104,12 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
out = append(out, code...)
pc += len(code)
}
if fi.needSplit {
block, blReloc := arm64MoreStackBlock(pc)
out = append(out, block...)
relocs = append(relocs, blReloc)
pc += len(block)
}
return out, offsets, relocs, lines, spadj, nil
}
@@ -191,7 +222,7 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
return a64wordLE(uint32(immFromOperand(ops[0]))), nil
case "B":
case "B", "JMP":
return encodeARM64Branch(mnem, ops, pc, offsets, false, relocs, resolve)
case "BL", "CALL":
return encodeARM64Branch(mnem, ops, pc, offsets, true, relocs, resolve)
@@ -306,8 +337,9 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
}
op := ops[0]
// External symbol reference: BL sym(SB).
if link && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
// Symbol reference: BL sym(SB), or B sym(SB) for a tail call, against a
// relocation (R_CALLARM64 either way).
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
if relocs != nil {
*relocs = append(*relocs, Reloc{
Off: 0,
@@ -317,8 +349,12 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
Kind: RelArm64Branch,
})
}
// Emit BL with zero offset; the linker fills in the target.
return a64wordLE(a64Branch(1, 0)), nil
// Emit B/BL with zero offset; the linker fills in the target.
bop := uint32(0) // B
if link {
bop = 1 // BL
}
return a64wordLE(a64Branch(bop, 0)), nil
}
target := resolve(arm64Label(op))
@@ -462,7 +498,7 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
// ---- MOV pseudo-instruction ----
// encodeARM64Mov encodes the MOV family — the load/store/immediate workhorse
// encodeARM64Mov encodes the MOV family, the load/store/immediate workhorse
// of Go's arm64 assembly. MOV is an alias of MOVD (the width mnemonics
// select the access width). The forms, mirroring the toolchain:
//
@@ -562,9 +598,9 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
}
_, off := arm64MemWithFrame(mem, fi)
// Scaled unsigned offset fits if aligned and in range.
lt := a64LoadTable[mnem]
if lt.size == 0 {
lt.size = 3 // default to64-bit for MOV
lt, ok := a64LoadTable[mnem]
if !ok {
lt = a64LoadTable["MOVD"] // the MOV pseudo is a 64-bit access
}
scale := int32(1) << uint(lt.size)
if off >= 0 && off%scale == 0 && off/scale < 4096 {
@@ -573,7 +609,10 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
if off >= -256 && off <= 255 {
return 4 // unscaled
}
return 12 // materialise offset + LDR/STR
if _, _, _, ok := arm64SplitOffset(off, scale); ok {
return 8 // ADD base, REGTMP + access
}
return 12 // literal pool range: encoding reports it as unsupported
default:
return 4 // register move
}
@@ -793,33 +832,53 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6
}
scale := int32(1) << uint(lt.size)
if load {
// Try scaled unsigned offset first.
if off >= 0 && off%scale == 0 {
imm12 := uint32(off / scale)
if imm12 < 4096 {
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), imm12, uint32(rn), uint32(reg))), nil
}
}
// Try unscaled (9-bit signed).
if off >= -256 && off <= 255 {
return a64wordLE(a64LSUnscaled(lt.size, lt.V, lt.opc, off, rn, reg)), nil
}
// Large offset: materialise in R20 (TMP) and use register-offset.
return nil, fmt.Errorf("%s: offset %d out of range", mnem, off)
}
// Store: same encoding but opc bits indicate store.
storeOpc := a64StoreOpc(lt)
if off >= 0 && off%scale == 0 {
imm12 := uint32(off / scale)
if imm12 < 4096 {
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), imm12, uint32(rn), uint32(reg))), nil
}
var opc int
if load {
opc = lt.opc
} else {
opc = storeOpc
}
// Scaled unsigned offset first, then the unscaled ±255 form.
if off >= 0 && off%scale == 0 && off/scale < 4096 {
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(off/scale), uint32(rn), uint32(reg))), nil
}
if off >= -256 && off <= 255 {
return a64wordLE(a64LSUnscaled(lt.size, lt.V, storeOpc, off, rn, reg)), nil
return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, off, rn, reg)), nil
}
return nil, fmt.Errorf("%s: offset %d out of range", mnem, off)
// Large offset: materialise the base in REGTMP (R27) the way the
// toolchain does and access what remains.
addImm, addShift, access, ok := arm64SplitOffset(off, scale)
if !ok {
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
}
return a64WordsLE(
a64AddSub(1, 0, 0, addShift, uint32(addImm), 31, 27), // ADD $addImm<<shift, SP, R27
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(access/scale), 27, uint32(reg)),
), nil
}
// arm64SplitOffset decomposes an out-of-range frame offset for a REGTMP
// base: an ADD (plain, or shifted left by 12) brings SP near the target and
// the access covers what remains. ok is false when no decomposition exists
// (offsets at or beyond 16 MiB, where the toolchain falls back to a literal
// pool).
func arm64SplitOffset(off int32, scale int32) (addImm, addShift uint32, access int32, ok bool) {
if off < 0 {
return 0, 0, 0, false
}
// Plain ADD: bring SP to within the largest scaled access.
l := min(off, 4095*scale)
l -= l % scale
if a := off - l; a <= 4095 {
return uint32(a), 0, l, true
}
// Shifted ADD: cover everything but the bits the access imm12 carries.
rest := off &^ (0xFFF * scale)
if rest >= 0 && rest>>12 <= 4095 {
return uint32(rest >> 12), 1, off - rest, true
}
return 0, 0, 0, false
}
// ---- static symbol references (ADRP + offset) ----
@@ -839,7 +898,9 @@ func encodeARM64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
)
}
// encodeARM64SBLoad emits ADRP R20, 0; LDR Rd, [R20, 0] with relocations.
// encodeARM64SBLoad emits ADRP R27, 0; LDR Rd, [R27, 0] with relocations,
// matching the toolchain: the scratch register is REGTMP (R27) and the pair
// carries R_ARM64_PCREL_LDST64.
func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([]byte, error) {
lt, ok := a64LoadTable[mnem]
if !ok {
@@ -847,17 +908,17 @@ func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([
}
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
Reloc{Off: 0, After: 8, Name: sym.Name, Kind: RelArm64LDST64, Addend: sym.Offset},
)
}
return a64WordsLE(
a64ADR(1, 0, 0, 20), // ADRP R20, 0
a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), 0, 20, uint32(rd)), // LDR Rd, [R20, #0]
a64ADR(1, 0, 0, 27), // ADRP R27, 0
a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), 0, 27, uint32(rd)), // LDR Rd, [R27, #0]
), nil
}
// encodeARM64SBStore emits ADRP R20, 0; STR Rs, [R20, 0] with relocations.
// encodeARM64SBStore emits ADRP R27, 0; STR Rs, [R27, 0] with relocations,
// matching the toolchain's R27 scratch and R_ARM64_PCREL_LDST64 pair.
func encodeARM64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) ([]byte, error) {
lt, ok := a64LoadTable[mnem]
if !ok {
@@ -866,13 +927,12 @@ func encodeARM64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) (
storeOpc := a64StoreOpc(lt)
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
Reloc{Off: 0, After: 8, Name: sym.Name, Kind: RelArm64LDST64, Addend: sym.Offset},
)
}
return a64WordsLE(
a64ADR(1, 0, 0, 20), // ADRP R20, 0
a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), 0, 20, uint32(rs)), // STR Rs, [R20, #0]
a64ADR(1, 0, 0, 27), // ADRP R27, 0
a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), 0, 27, uint32(rs)), // STR Rs, [R27, #0]
), nil
}
@@ -1092,7 +1152,7 @@ func encodeARM64CSEL(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
return a64wordLE(baseOp | uint32(rn)<<16 | invCond<<12 | uint32(rn)<<5 | uint32(rd)), nil
}
// CSEL cond, Rn, Rm, Rd (4 operands) — condition first.
// CSEL cond, Rn, Rm, Rd (4 operands), condition first.
// Go assembler syntax: CSEL cond, Rn, Rm, Rd
// ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5], Rd in bits[4:0].
if len(ops) != 4 {
+3 -3
View File
@@ -9,7 +9,7 @@ package asm
// an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own arm64
// backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm`
// exactly — the ground-truth oracle for the verify suite.
// exactly, the ground-truth oracle for the verify suite.
//
// All AArch64 instructions are 32 bits, little-endian. The formats used here
// (per the ARM Architecture Reference Manual):
@@ -28,7 +28,7 @@ package asm
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
// R0–R30 (integer), F0–F31 (floating point), and the ABI aliases the
// R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
// runtime's assembly uses. Returns -1 for an unrecognised name.
func arm64RegNum(name string) int {
switch name {
@@ -99,7 +99,7 @@ func arm64RegNum(name string) int {
case "SP":
return 31 // SP and ZR share encoding 31; context determines meaning
}
// F0–F31.
// F0-F31.
if len(name) >= 1 && name[0] == 'F' {
n := 0
for i := 1; i < len(name); i++ {
+186 -14
View File
@@ -63,6 +63,12 @@ type arm64FrameInfo struct {
args int // the declared -argsize
noSplit bool // the NOSPLIT flag
leaf bool // no call instructions in the body
// Stack-split guard state: needSplit mirrors the toolchain, which skips
// the check for NOSPLIT functions and auto-marks leaf functions with an
// autosize below StackSmall as NOSPLIT.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
}
// arm64ComputeFrame derives the frame layout for a TEXT function.
@@ -80,15 +86,68 @@ func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
if fi.frame != 0 || !fi.leaf {
fi.autosize = fi.frame + 8 // space for the saved LR
if fi.autosize%16 != 0 {
// The toolchain aligns to 16: if autosize%16 == 8, add 8;
// otherwise add whatever is needed.
// The toolchain always adds an extrasize: 8 when the total leaves a
// 16-byte alignment gap, another 16 when already aligned.
switch fi.autosize % 16 {
case 8:
fi.autosize += 8
case 0:
fi.autosize += 16
default:
// The toolchain rejects unaligned frames; round up so such
// sources still assemble.
fi.autosize += 16 - (fi.autosize % 16)
}
}
switch {
case fi.noSplit:
case fi.autosize < stackSmall && fi.leaf:
// Auto-NOSPLIT, as the toolchain's leaf mark concludes.
default:
fi.needSplit = true
switch {
case fi.autosize <= stackSmall:
fi.splitClass = 0
case fi.autosize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi
}
// arm64GuardLen returns the byte length of the stack-split guard prefix
// (zero when the function needs no guard). The big class materialises
// framesize-StackSmall into REGTMP, whose MOVZ/MOVK sequence length varies.
func arm64GuardLen(fi arm64FrameInfo) int {
if !fi.needSplit {
return 0
}
switch fi.splitClass {
case 0:
return 12
case 1:
return 16
default:
n, err := arm64LoadImmLen(int64(fi.autosize - stackSmall))
if err != nil {
return 0
}
return 4 + n + 4 + 4 + 4 + 4
}
}
// arm64LoadImmLen returns the byte length of the MOVZ/MOVK sequence that
// loads v into a register.
func arm64LoadImmLen(v int64) (int, error) {
b, err := encodeARM64LoadImm(27, v, "MOVD")
if err != nil {
return 0, err
}
return len(b), nil
}
// arm64IsLeaf reports whether a function contains no call instructions
// (BL/CALL), matching the toolchain's LEAF mark.
func arm64IsLeaf(t *ast.Text) bool {
@@ -119,12 +178,47 @@ func arm64Prologue(fi arm64FrameInfo) []byte {
)
}
// Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP
return a64WordsLE(
a64AddSub(1, 1, 0, 0, uint32(fi.autosize), 31, 20), // SUB $autosize, SP, R20
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
ws := arm64SubImmWords(uint32(fi.autosize), 20)
ws = append(ws,
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
)
return a64WordsLE(ws...)
}
// arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value
// fits the imm12 field (plain, or shifted left by 12 when it is a multiple
// of 4096); otherwise the toolchain materialises it into REGTMP (R27) and
// subtracts the register in the extended-register form.
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 0xFFF {
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
}
if imm <= 4095<<12 && imm&0xFFF == 0 {
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
}
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
}
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12
// and REGTMP fallback ladder.
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 0xFFF {
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
}
if imm <= 4095<<12 && imm&0xFFF == 0 {
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
}
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
}
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
@@ -134,10 +228,8 @@ func arm64Return(fi arm64FrameInfo) []byte {
if fi.autosize != 0 {
if fi.leaf {
// Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP
ws = append(ws,
a64AddSub(1, 0, 0, 0, uint32(fi.autosize-8), 31, 29), // ADD $autosize-8, SP, FP
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
)
ws = append(ws, arm64AddImmWords(uint32(fi.autosize-8), 29)...)
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
} else if fi.autosize <= 0xf0 {
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize
ws = append(ws,
@@ -147,9 +239,9 @@ func arm64Return(fi arm64FrameInfo) []byte {
} else {
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
ws = append(ws,
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
)
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
}
}
// RET: BR LR (0xd65f03c0)
@@ -235,3 +327,83 @@ func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// Data-processing (shifted register) base opcodes for the guard blocks.
const (
arm64OpAdd = 1<<31 | 0<<30 | 0<<29 | 0x0b<<24
arm64OpSub = 1<<31 | 1<<30 | 0<<29 | 0x0b<<24
arm64OpSubs = 1<<31 | 1<<30 | 1<<29 | 0x0b<<24
)
// arm64DPSRWords builds one data-processing (shifted register) word:
// OP Rm, Rn, Rd in the Go assembler's operand order.
func arm64DPSRWords(base uint32, rm, rn, rd uint32) uint32 {
return base | rm<<16 | rn<<5 | rd
}
// arm64DPExtWords builds one data-processing (extended register) word, the
// form the toolchain picks when a large immediate was materialised into
// REGTMP before the operation: base | 1<<21 | Rm<<16 | UXTX<<13 | Rn<<5 | Rd.
func arm64DPExtWords(base, rm, rn, rd uint32) uint32 {
return base | 1<<21 | rm<<16 | 3<<13 | rn<<5 | rd
}
// wordsOf converts little-endian instruction bytes back to words.
func wordsOf(b []byte) []uint32 {
ws := make([]uint32, 0, len(b)/4)
for i := 0; i+4 <= len(b); i += 4 {
ws = append(ws, uint32(b[i])|uint32(b[i+1])<<8|uint32(b[i+2])<<16|uint32(b[i+3])<<24)
}
return ws
}
// arm64GuardBytes emits the stack-split guard prefix; blockStart is the
// function-relative byte address of the morestack block the branches target.
func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte {
// MOVD 16(R28), R16 (g.stackguard0)
ws := []uint32{a64LSU(3, 0, 1, 2, 28, 16)}
br := func(from int, cond uint32) uint32 {
return a64BranchCond(int32((blockStart-from)>>2), cond)
}
switch fi.splitClass {
case 0:
// CMP R16, RSP in the exact encoding go tool asm emits for it.
ws = append(ws, 0xeb3063ff)
ws = append(ws, br(8, a64CondLS))
case 1:
ws = append(ws, a64AddSub(1, 1, 0, 0, uint32(fi.autosize-stackSmall), 31, 17))
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
ws = append(ws, br(12, a64CondLS))
default:
mov, err := encodeARM64LoadImm(27, int64(fi.autosize-stackSmall), "MOVD")
if err != nil {
mov = nil
}
ws = append(ws, wordsOf(mov)...)
ml := len(mov) / 4
ws = append(ws, arm64DPExtWords(arm64OpSubs, 27, 31, 17)) // SUBS R17, RSP, R27
ws = append(ws, br(8+ml, a64CondLO))
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
ws = append(ws, br(8+ml+8, a64CondLS))
}
return a64WordsLE(ws...)
}
// arm64MoreStackBlock emits the trailing block: MOVD R30, R3 (save LR),
// BL runtime.morestack_noctxt, B back to the function start. The BL carries
// the R_CALLARM64 relocation.
func arm64MoreStackBlock(blockStart int) ([]byte, Reloc) {
ws := []uint32{
1<<31 | 1<<29 | 0x0a<<24 | 30<<16 | 31<<5 | 3, // MOVD R30, R3
a64Branch(1, 0), // BL, patched by the linker
}
bPC := blockStart + 8
ws = append(ws, a64Branch(0, int32(-bPC>>2))) // B back to the entry
reloc := Reloc{
Off: blockStart + 4,
After: blockStart + 8,
Name: "runtime\u00b7morestack_noctxt",
Kind: RelArm64Branch,
}
return a64WordsLE(ws...), reloc
}
+126
View File
@@ -0,0 +1,126 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// parseArm64File is a helper assembling one arm64 source file.
func parseArm64File(t *testing.T, src string) *Image {
t.Helper()
f, errs := parser.Parse("k_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
return img
}
// TestArm64RelocOffsetsIncludePrologue pins the function-relative relocation
// offsets of a framed function: the offsets used to exclude the prologue, so
// every relocation landed on a prologue instruction in the GOOBJ/ELF output.
// The function calls an external, so it is a non-leaf and carries the
// stack-split guard (12 bytes, small class) before the prologue.
func TestArm64RelocOffsetsIncludePrologue(t *testing.T) {
img := parseArm64File(t, "TEXT \u00b7f(SB), $16-0\n"+
"\tBL ext\u00b7foo(SB)\n"+
"\tMOVD $gdata(SB), R5\n"+
"\tMOVD $extsym(SB), R6\n"+
"\tRET\n"+
"GLOBL gdata(SB), $8\n")
fn := img.Funcs[0]
// Layout: 12-byte guard, 12-byte prologue, BL (24), ADRP+ADD (28, 32),
// ADRP+ADD (36, 40), 12-byte epilogue with RET, 12-byte morestack block.
want := []struct {
off int
after int
name string
kind RelocKind
external bool
}{
{24, 28, "foo", RelArm64Branch, true},
{28, 28, "gdata", RelArm64Addr, false},
{32, 32, "gdata", RelArm64Addr, false},
{36, 36, "extsym", RelArm64Addr, true},
{40, 40, "extsym", RelArm64Addr, true},
{60, 64, "runtime\u00b7morestack_noctxt", RelArm64Branch, true},
}
if len(fn.Relocs) != len(want) {
t.Fatalf("relocs = %d, want %d", len(fn.Relocs), len(want))
}
for i, w := range want {
r := fn.Relocs[i]
if r.Off != w.off || r.After != w.after || r.Name != w.name || r.Kind != w.kind || r.External != w.external {
t.Errorf("reloc %d = {off %d after %d name %q kind %d ext %v}, want {off %d after %d name %q kind %d ext %v}",
i, r.Off, r.After, r.Name, r.Kind, r.External, w.off, w.after, w.name, w.kind, w.external)
}
}
// The BL with a zero offset sits exactly at the first reloc site.
code := img.Code[fn.Offset : fn.Offset+fn.Size]
if w := binary.LittleEndian.Uint32(code[24:28]); w != 0x94000000 {
t.Errorf("BL word = %08x, want 94000000", w)
}
}
// TestArm64SBLoadStoreMatchesToolchain pins the ADRP scratch register
// (REGTMP, R27) and the LDST64 relocation kind for sym loads and stores,
// against the bytes go tool asm emits for MOVD sym(SB), R5.
func TestArm64SBLoadStoreMatchesToolchain(t *testing.T) {
img := parseArm64File(t, "TEXT \u00b7ld(SB), NOSPLIT, $0\n"+
"\tMOVD sym(SB), R5\n"+
"\tMOVD R5, sym(SB)\n"+
"\tRET\n"+
"GLOBL sym(SB), $8\n")
fn := img.Funcs[0]
code := img.Code[fn.Offset : fn.Offset+fn.Size]
// go tool asm: ADRP 0(PC), R27 (9000001b); MOVD (R27), R5 (f9400365);
// ADRP 0(PC), R27; MOVD R5, (R27) (f9000365).
for off, want := range map[int]uint32{0: 0x9000001b, 4: 0xf9400365, 8: 0x9000001b, 12: 0xf9000365} {
if got := binary.LittleEndian.Uint32(code[off : off+4]); got != want {
t.Errorf("word at %d = %08x, want %08x", off, got, want)
}
}
if len(fn.Relocs) != 2 {
t.Fatalf("relocs = %d, want 2", len(fn.Relocs))
}
for i, w := range []struct{ off, after int }{{0, 8}, {8, 16}} {
r := fn.Relocs[i]
if r.Kind != RelArm64LDST64 {
t.Errorf("reloc %d kind = %d, want RelArm64LDST64 (%d)", i, r.Kind, RelArm64LDST64)
}
if r.Off != w.off || r.After != w.after {
t.Errorf("reloc %d = {off %d after %d}, want {off %d after %d}", i, r.Off, r.After, w.off, w.after)
}
}
}
// TestArm64GOObjRelocTypes checks that GOOBJ emission succeeds with the new
// relocation kinds in play; the detailed layout is covered by the goobj tests.
func TestArm64GOObjRelocTypes(t *testing.T) {
img := parseArm64File(t, "TEXT \u00b7ld(SB), NOSPLIT, $0\n"+
"\tMOVD sym(SB), R5\n"+
"\tMOVD R5, sym(SB)\n"+
"\tRET\n"+
"GLOBL sym(SB), $8\n")
obj, err := img.GOObjectAARCH64("testpkg", "k_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
if len(obj) == 0 {
t.Fatal("empty object")
}
// The detailed layout is covered by the goobj tests; here we only pin
// that emission succeeds with the new relocation kinds in play.
}
+299 -14
View File
@@ -22,6 +22,11 @@ import (
// FP/SP frame-relative operands, and local-label jumps. SB (global symbol)
// operands require relocations and are not yet supported; the SIMD (VEX/AVX2)
// integer and shuffle/extract/permute/move set is in.
//
// Like the other architectures, the stack-growth guard (the morestack check
// in the prologue and the call back into the runtime in the epilogue) is not
// emitted: the bytes match go tool asm only for NOSPLIT functions or
// zero-frame leaves, where the toolchain emits no guard either.
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
code, _, labels, _, _, err := assemble(t, nil)
return code, labels, err
@@ -31,7 +36,7 @@ func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
// the set of static symbols a GLOBL in the same file defines. A nil link
// rejects SB operands outright (single-function assembly cannot resolve
// them). When allowExternal is set, a reference to a symbol no GLOBL in the
// file defines is recorded as an external relocation instead of failing —
// file defines is recorded as an external relocation instead of failing
// the object-file emitters resolve it at link time.
type linkInfo struct {
symbols map[string]bool
@@ -46,6 +51,7 @@ type sbPatch struct {
after int
name string
addend int64
kind RelocKind
}
// spadjStep is one stack-adjustment boundary within a function: Value is the
@@ -70,13 +76,18 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
return name
}
// Layout: iterate jump sizes to a fixed point.
// Layout: iterate jump sizes to a fixed point. The stack-split guard
// prefix and the trailing morestack block participate in the iteration:
// their conditional branches relax from rel8 to rel32 when the body
// outgrows the short form.
long := make([]bool, len(t.Body))
sizes := make([]int, len(t.Body))
offsets := map[string]int{}
pcs := make([]int, len(t.Body))
var guardJBlong, guardJBElong, moreJMPlong bool
for {
pos := len(fi.prologue)
guard := fi.guardLen(guardJBlong, guardJBElong)
pos := guard + len(fi.prologue)
for i, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
@@ -91,6 +102,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
pos += sz
}
}
bodyLen := pos - (guard + len(fi.prologue))
// Expand any short jump whose displacement no longer fits rel8.
changed := false
for i, stmt := range t.Body {
@@ -116,25 +128,75 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
changed = true
}
}
// The guard's conditional branches target the morestack block, which
// starts right after the body: the JBE measures from the end of the
// guard, so its displacement is the prologue plus the body.
if !guardJBElong && !fits8(int64(len(fi.prologue)+bodyLen)) {
guardJBElong = true
changed = true
}
if fi.splitClass == 2 && !guardJBlong {
// The underflow JB sits before the CMPQ; its displacement spans
// the rest of the guard plus the prologue and the body.
jbLen := 2
if guardJBlong {
jbLen = 6
}
rest := fi.guardLen(guardJBlong, guardJBElong) - (9 + 3 + 7 + jbLen)
if !fits8(int64(rest + len(fi.prologue) + bodyLen)) {
guardJBlong = true
changed = true
}
}
// The morestack JMP returns to the function start, so its
// displacement is the negated distance from its own end.
if !moreJMPlong {
jmpLen := 2
if moreJMPlong {
jmpLen = 5
}
if !fits8(-int64(guard + len(fi.prologue) + bodyLen + 5 + jmpLen)) {
moreJMPlong = true
changed = true
}
}
if !changed {
break
}
}
// Pass 2: emit.
out := append([]byte(nil), fi.prologue...)
// Pass 2: emit. The guard comes first, then the prologue, the body and
// the morestack block.
guardLen := fi.guardLen(guardJBlong, guardJBElong)
bodyLen := 0
{
pos := guardLen + len(fi.prologue)
for i, stmt := range t.Body {
if _, ok := stmt.(*ast.Instr); ok {
pos += sizes[i]
}
}
bodyLen = pos - (guardLen + len(fi.prologue))
}
var out []byte
var patches []sbPatch
if fi.needSplit {
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+2)+len(fi.prologue)+bodyLen))
out = append(out, guard...)
patches = append(patches, tlsPatch)
}
out = append(out, fi.prologue...)
var steps []spadjStep
var lines []LineEntry
if fi.useFP {
// PUSHQ BP saves the return-address-relative base (+8); the MOVQ
// changes nothing; SUBQ $size, SP completes the frame.
steps = append(steps,
spadjStep{1, 8},
spadjStep{len(fi.prologue), 8 + fi.size},
spadjStep{guardLen + 1, 8},
spadjStep{guardLen + len(fi.prologue), 8 + fi.size},
)
}
pos := len(fi.prologue)
pos := guardLen + len(fi.prologue)
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
if !ok {
@@ -156,11 +218,32 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
if len(code) != sizes[i] {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
}
if strings.ToUpper(s.Mnemonic.Text) == "CALL" {
for k := range ps {
ps[k].kind = RelCall
}
}
patches = append(patches, ps...)
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
out = append(out, code...)
pos += len(code)
}
if fi.needSplit {
// The morestack block: CALL runtime.morestack_noctxt, then a JMP
// back to the function entry.
jmpLen := 2
if moreJMPlong {
jmpLen = 5
}
jmpDisp := -int64(pos + 5 + jmpLen)
suffix, callPatch := buildMoreStack(int32(jmpDisp))
callPatch.off += pos
callPatch.after = pos + 5
patches = append(patches, callPatch)
out = append(out, suffix...)
pos += len(suffix)
}
_ = pos
return out, patches, offsets, steps, lines, nil
}
@@ -224,10 +307,31 @@ type frameInfo struct {
spAdjust int64 // x-N(SP) becomes (spAdjust - N)(SP)
prologue []byte
epilogue []byte
// Stack-split guard state (matching the toolchain's stacksplit): needSplit
// is false for NOSPLIT functions and for leaf functions whose frame is
// below StackSmall, which the toolchain auto-marks NOSPLIT.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
framesize int // the size the guard checks: frame+8 for framed functions
}
// Stack-frame size classes from runtime/stack.go.
const (
stackSmall = 128
stackBig = 4096
)
// sbPatch gains a kind so the emitters can tell CALL and TLS patches from
// plain PC-relative displacements.
// computeFrame derives the frame layout, matching the Go assembler's default
// (a frame pointer is used whenever the function has a non-zero frame).
// (a frame pointer is used whenever the function has a non-zero frame). It
// also decides whether the function needs the stack-split guard, mirroring
// obj6: a NOSPLIT function never splits, and a leaf function whose frame is
// below StackSmall is auto-marked NOSPLIT. One deliberate deviation: the
// toolchain treats zero-argument runtime calls (duffcopy and friends) as
// leaf-compatible; here any CALL makes the function a non-leaf.
func computeFrame(t *ast.Text) frameInfo {
fi := frameInfo{}
if t.Frame != nil && t.Frame.Imm.HasVal {
@@ -242,9 +346,141 @@ func computeFrame(t *ast.Text) frameInfo {
} else {
fi.fpAdjust = 8 // return address only
}
noSplit := false
for _, f := range t.Flags {
if strings.EqualFold(f, "NOSPLIT") {
noSplit = true
}
}
// The toolchain's autoffset: the frame plus the saved base pointer.
framesize := fi.size
if framesize > 0 {
framesize += 8
}
switch {
case noSplit:
case framesize < stackSmall && !hasCall(t):
// Auto-NOSPLIT, as the toolchain's leaf search concludes.
default:
fi.needSplit = true
fi.framesize = framesize
switch {
case framesize <= stackSmall:
fi.splitClass = 0
case framesize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi
}
// hasCall reports whether the function body contains a CALL instruction.
func hasCall(t *ast.Text) bool {
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
if strings.ToUpper(in.Mnemonic.Text) == "CALL" {
return true
}
}
return false
}
// guardLen returns the byte length of the stack-split guard prefix. The
// final conditional branch (JBE, and JB in the big class) is 2 bytes in the
// short form and 6 in the long form.
func (fi frameInfo) guardLen(jbLong, jbeLong bool) int {
if !fi.needSplit {
return 0
}
jb, jbe := 2, 2
if jbLong {
jb = 6
}
if jbeLong {
jbe = 6
}
switch fi.splitClass {
case 0:
return 9 + 4 + jbe
case 1:
return 9 + 8 + 4 + jbe
default:
return 9 + 3 + 7 + jb + 4 + jbe
}
}
// moreLen returns the byte length of the trailing morestack block: the CALL
// (always rel32) plus the JMP back to the function start.
func moreLen(jmpLong bool) int {
jmp := 2
if jmpLong {
jmp = 5
}
return 5 + jmp
}
// buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the
// already-computed displacements of the conditional branches that jump to the
// morestack block (unused in classes without them). The TLS load carries a
// R_TLS_LE patch site at offset 5.
func buildGuard(fi frameInfo, jbeDisp, jbDisp int32) ([]byte, sbPatch) {
out := []byte{
0x64, 0x4c, 0x8b, 0x34, 0x25, // MOVQ FS:0, R14
0, 0, 0, 0, // TLS slot offset, filled by the linker
}
tls := sbPatch{off: 5, after: 9, kind: RelTLSLE}
jmp := func(op8, op32 byte, disp int32) []byte {
if disp >= -128 && disp <= 127 {
return []byte{op8, byte(disp)}
}
return append([]byte{0x0F, op32}, le32(int64(disp))...)
}
switch fi.splitClass {
case 0:
// CMPQ SP, 16(R14)
out = append(out, 0x49, 0x3b, 0x66, 0x10)
out = append(out, jmp(0x76, 0x86, jbeDisp)...)
case 1:
// LEAQ -(framesize-StackSmall)(SP), R12; CMPQ R12, 16(R14)
out = append(out, 0x4c, 0x8d, 0xa4, 0x24)
out = append(out, le32(-int64(fi.framesize-stackSmall))...)
out = append(out, 0x4d, 0x3b, 0x66, 0x10)
out = append(out, jmp(0x76, 0x86, jbeDisp)...)
default:
// MOVQ SP, R12; SUBQ $(framesize-StackSmall), R12; JB; CMPQ R12, 16(R14)
out = append(out, 0x49, 0x89, 0xe4)
out = append(out, 0x49, 0x81, 0xec)
out = append(out, le32(int64(fi.framesize-stackSmall))...)
out = append(out, jmp(0x72, 0x82, jbDisp)...)
out = append(out, 0x4d, 0x3b, 0x66, 0x10)
out = append(out, jmp(0x76, 0x86, jbeDisp)...)
}
return out, tls
}
// buildMoreStack emits the trailing block: CALL runtime.morestack_noctxt
// (patched by the linker) and a JMP back to the function start.
func buildMoreStack(jmpDisp int32) ([]byte, sbPatch) {
out := []byte{0xE8, 0, 0, 0, 0}
call := sbPatch{off: 1, after: 5, name: "runtime\u00b7morestack_noctxt", kind: RelCall}
out = append(out, jmpBytes(jmpDisp)...)
return out, call
}
// jmpBytes encodes a near JMP in the short or long form.
func jmpBytes(disp int32) []byte {
if disp >= -128 && disp <= 127 {
return []byte{0xEB, byte(disp)}
}
return append([]byte{0xE9}, le32(int64(disp))...)
}
// prologueBytes emits: PUSHQ BP; MOVQ SP, BP; SUBQ $size, SP.
func prologueBytes(size int) []byte {
out := []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
@@ -258,14 +494,11 @@ func epilogueBytes(size int) []byte {
}
func subSP(size int) []byte { // SUBQ $size, SP
// imm8 holds -128..127; anything larger takes the imm32 form, exactly as
// the Go assembler encodes it (verified for 8, 128, 200 and 255).
if size >= -128 && size <= 127 {
return []byte{0x48, 0x83, 0xEC, byte(int8(size))}
}
// 128..255 do not fit SUB's unsigned imm8, but the Go assembler
// switches to ADDQ $-size, SP whose sign-extended imm8 does.
if size >= -255 && size <= 255 {
return []byte{0x48, 0x83, 0xC4, byte(int8(-size))}
}
return append([]byte{0x48, 0x81, 0xEC}, le32(int64(size))...)
}
@@ -282,6 +515,9 @@ func addSP(size int) []byte { // ADDQ $size, SP
func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, error) {
mnem := strings.ToUpper(s.Mnemonic.Text)
if isJumpMnemonic(mnem) {
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
return 5, nil // opcode + rel32, always the long form
}
return jumpSize(mnem, long), nil
}
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
@@ -331,6 +567,24 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
var ps []sbPatch
var err error
if isJumpMnemonic(mnem) {
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
// CALL/JMP sym(SB): a rel32 call (or tail call) against a
// static or external symbol, resolved by the file-level layout
// or the linker.
code, ps, err = encodeSBCall(s, link)
if err != nil {
return nil, nil, err
}
for i := range ps {
ps[i].kind = RelCall
}
body := pc + len(prefix)
for i := range ps {
ps[i].off += body
ps[i].after = body + len(code)
}
return append(prefix, code...), ps, nil
}
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
} else {
code, ps, err = encodeNormal(s, fi, link)
@@ -412,6 +666,37 @@ func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long
}
}
// isSBCall reports whether the CALL operand is a symbol reference.
func isSBCall(s *ast.Instr) bool {
return len(s.Operands) == 1 && s.Operands[0].Kind == ast.OpAddr &&
s.Operands[0].Addr.Sym != nil && s.Operands[0].Addr.Sym.Pseudo == "SB"
}
// encodeSBCall encodes CALL sym(SB) as E8 rel32 with a patch site.
func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) {
o, err := operandFromAST(s.Operands[0], 8, frameInfo{}, link)
if err != nil {
return nil, nil, err
}
m, ok := o.(sbMem)
if !ok {
return nil, nil, fmt.Errorf("CALL: unsupported operand")
}
opcode := []byte{0xE8}
if strings.ToUpper(s.Mnemonic.Text) == "JMP" {
opcode = []byte{0xE9} // a tail call, no return address pushed
}
e := &enc{}
if err := e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}}); err != nil {
return nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend, kind: RelCall}
}
return e.out, ps, nil
}
// labelName extracts a local-label name from a jump operand.
func labelName(op *ast.Operand) (string, bool) {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
+26 -2
View File
@@ -4,6 +4,7 @@
package asm
import (
"bytes"
"strings"
"testing"
@@ -201,8 +202,8 @@ TEXT ·withframe(SB), NOSPLIT, $16-16
}
// TestAssembleVexKernel assembles the horizontal-sum reduction the go-flac
// kernels end with — exercising the VEX moves, shuffle and extract forms
// through the full parser → encoder path — and checks the output is
// kernels end with; exercising the VEX moves, shuffle and extract forms
// through the full parser → encoder path; and checks the output is
// byte-identical to the Go assembler's.
func TestAssembleVexKernel(t *testing.T) {
fn := firstText(t, `
@@ -351,3 +352,26 @@ TEXT ·pf(SB), NOSPLIT, $0
t.Errorf("PREFETCHT0 bytes: got %s, want 0f 18 0b", hex)
}
}
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
// larger. The intermediate 129..255 range used to encode an ADD with a
// truncated immediate, moving SP the wrong way.
func TestSubSPEncodings(t *testing.T) {
for _, tt := range []struct {
size int
want []byte
}{
{8, []byte{0x48, 0x83, 0xEC, 0x08}},
{127, []byte{0x48, 0x83, 0xEC, 0x7F}},
{128, []byte{0x48, 0x81, 0xEC, 0x80, 0x00, 0x00, 0x00}},
{200, []byte{0x48, 0x81, 0xEC, 0xC8, 0x00, 0x00, 0x00}},
{255, []byte{0x48, 0x81, 0xEC, 0xFF, 0x00, 0x00, 0x00}},
{4096, []byte{0x48, 0x81, 0xEC, 0x00, 0x10, 0x00, 0x00}},
} {
got := subSP(tt.size)
if !bytes.Equal(got, tt.want) {
t.Errorf("subSP(%d) = %x, want %x", tt.size, got, tt.want)
}
}
}
+14 -4
View File
@@ -12,7 +12,7 @@ import (
// Image: a .text section holding the function bodies, a .data section
// holding the GLOBL initialisers, a symbol table with one symbol per TEXT
// and GLOBL (file-local <> symbols are STB_LOCAL, the rest STB_GLOBAL), and
// a .rela.text relocation table — one R_X86_64_PC32 entry per static-symbol
// a .rela.text relocation table, one R_X86_64_PC32 entry per static-symbol
// reference, internal references resolving against the local data symbols
// and external ones against undefined globals. The output links with the
// system toolchain (cc/ld) the way a hand-assembled .o would.
@@ -42,7 +42,8 @@ const (
sttSection = 3
stInfoShift = 4
rX8664PC32 = 2
rX8664PC32 = 2
rX8664TPOFF32 = 20
)
// elfSym is one symbol-table entry in construction.
@@ -71,7 +72,7 @@ func (img *Image) ELFObject() ([]byte, error) {
// Build the symbol table: the null entry and the two section symbols
// come first, then the local symbols (static TEXT and GLOBL), then the
// globals (exported TEXT and GLOBL, and the undefined externals) — ELF
// globals (exported TEXT and GLOBL, and the undefined externals), ELF
// requires every local to precede every global, and sh_info records the
// boundary. symIdx maps a symbol name to its index for the relocations.
var locals, globals []elfSym
@@ -125,11 +126,19 @@ func (img *Image) ELFObject() ([]byte, error) {
type elfRela struct {
off uint64
sym int
typ uint32
addend int64
}
var relas []elfRela
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
var typ uint32 = rX8664PC32
if r.Kind == RelTLSLE {
// R_X86_64_TPOFF32 resolves to the local-exec TLS offset and
// carries no symbol.
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), sym: 0, typ: rX8664TPOFF32})
continue
}
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
@@ -137,6 +146,7 @@ func (img *Image) ELFObject() ([]byte, error) {
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
sym: idx,
typ: typ,
// R_X86_64_PC32 computes S + A − P with P the patch site; the
// assembler measures the symbol from the instruction end,
// After − Off bytes past the field, so the addend carries
@@ -218,7 +228,7 @@ func (img *Image) ELFObject() ([]byte, error) {
shstrOff := len(out)
out = append(out, stSections.bytes()...)
// DWARF debug sections (no relocations — the linker resolves DWARF fixups).
// DWARF debug sections (no relocations, the linker resolves DWARF fixups).
dwAlign := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
+1 -1
View File
@@ -282,7 +282,7 @@ func dwarfBuildFrameSection(img *Image) []byte {
// Patch CIE length.
le.PutUint32(b[cieStart:], uint32(len(b)-cieStart-4))
// FDEs (Frame Description Entries) — one per function.
// FDEs (Frame Description Entries), one per function.
for _, fn := range img.Funcs {
fdeStart := len(b)
b = append(b, 0, 0, 0, 0) // length (placeholder)
+1 -1
View File
@@ -52,7 +52,7 @@ func elfTestImage(t *testing.T) *Image {
}
// TestAssembleFileExternals checks that a reference to a symbol no GLOBL
// defines is recorded as an external relocation instead of failing — the
// defines is recorded as an external relocation instead of failing; the
// raw image leaves the displacement zero, the object emitters carry it.
func TestAssembleFileExternals(t *testing.T) {
img := elfTestImage(t)
+12 -3
View File
@@ -15,8 +15,9 @@ const (
// AArch64 relocation types (the ELF psABI).
rArm64PrelPgHi21 = 275 // R_AARCH64_ADR_PREL_PG_HI21 (ADRP page)
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD/STR/LDR page offset)
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset)
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset)
)
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
@@ -81,7 +82,13 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
// Build relocations. Each SB reference is an ADRP pair:
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21
// ADD/LDR/STR → R_AARCH64_ADD_ABS_LO12_NC
// ADD → R_AARCH64_ADD_ABS_LO12_NC
// LDR/STR X → R_AARCH64_LDST64_ABS_LO12_NC
// BL → R_AARCH64_CALL26
// Addends stay raw: ADR_PREL_PG_HI21 and the ABS_LO12_NC forms resolve
// against S+A, and CALL26 branches take the branch instruction's own
// place as the PC-relative base, so subtracting the field width (the
// amd64 R_PCREL convention) would misplace every branch by 4 bytes.
type elfRela struct {
off uint64
typ uint32
@@ -99,6 +106,8 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
switch {
case r.Kind == RelArm64Branch:
typ = rArm64Call26
case r.Kind == RelArm64LDST64 && r.Off%4 == 4:
typ = rArm64Ldst64Lo12NC
case r.Kind == RelArm64Addr && r.Off%4 == 4:
typ = rArm64AddAbsLo12NC
default:
@@ -108,7 +117,7 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
off: uint64(fn.Offset + r.Off),
typ: typ,
sym: idx,
addend: r.Addend - int64(r.After-r.Off),
addend: r.Addend,
})
}
}
+6 -2
View File
@@ -16,6 +16,7 @@ const (
// LoongArch relocation types (the ELF psABI).
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
rLarchB26 = 66 // R_LARCH_B26 (b/bl, matches the Go linker's mapping)
)
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
@@ -95,14 +96,17 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
typ := uint32(rLarchPCALAHI20)
if r.Kind == RelLoong64AddrLo {
switch r.Kind {
case RelLoong64AddrLo:
typ = rLarchPCALALO12
case RelLoong64Branch:
typ = rLarchB26
}
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
typ: typ,
sym: idx,
addend: r.Addend - int64(r.After-r.Off),
addend: r.Addend,
})
}
}
+42
View File
@@ -198,3 +198,45 @@ TEXT ·nop(SB), NOSPLIT, $0
t.Error("function symbol nop not found")
}
}
// TestELFLOONG64BranchRelocation checks that the morestack call and an
// internal CALL both carry R_LARCH_B26 in the emitted object, matching the
// Go linker's mapping of its call relocation.
func TestELFLOONG64BranchRelocation(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", "TEXT \u00b7callbig(SB), $8192-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.ELFLOONG64Object()
if err != nil {
t.Fatalf("ELFLOONG64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaSec := ef.Section(".rela.text")
if relaSec == nil {
t.Fatal("missing .rela.text")
}
raw, err := relaSec.Data()
if err != nil {
t.Fatal(err)
}
// The guard's morestack call plus the body's CALL to other.
if len(raw)%24 != 0 || len(raw)/24 != 2 {
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
}
le := binary.LittleEndian
for i := range 2 {
info := le.Uint64(raw[i*24+8:])
if elf.R_LARCH(info&0xffffffff) != elf.R_LARCH_B26 {
t.Errorf("relocation %d type = %v, want R_LARCH_B26", i, elf.R_LARCH(info&0xffffffff))
}
}
}
+1 -1
View File
@@ -284,7 +284,7 @@ func setRM(i *instr, reg Reg, rm Operand, opSize int) error {
}
// setRMDigit fills in the ModR/M for an instruction whose reg field is an
// opcode /digit extension (0–7), which carries none of the register REX rules.
// opcode /digit extension (0-7), which carries none of the register REX rules.
func setRMDigit(i *instr, digit int, rm Operand, opSize int) error {
return setRMReg(i, digit, false, false, rm, opSize)
}
+2 -2
View File
@@ -230,7 +230,7 @@ func TestSSEMoveGroundTruth(t *testing.T) {
// TestGoFlacScalarTail encodes the scalar tail of an analyze kernel to confirm
// the encoder handles a realistic instruction sequence.
func TestGoFlacScalarTail(t *testing.T) {
// MOVQ swin_base+0(FP), SI — modelled as MOVQ disp(reg), reg.
// MOVQ swin_base+0(FP), SI; modelled as MOVQ disp(reg), reg.
checkSyntax(t, "mov rsi, qword ptr [rax+0x10]", "MOVQ", Ptr(AX, 0x10, 8), SI)
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
checkSyntax(t, "and r10, -0x8", "ANDQ", Imm(-8), Reg{idx: 10, size: 8})
@@ -421,7 +421,7 @@ func TestSSEShuffleGroundTruth(t *testing.T) {
}
// TestMOVQXMMGroundTruth pins the SSE2 packed-quadword move encodings:
// loads and register moves on F3 0F 7E, stores on 66 0F D6 — the forms
// loads and register moves on F3 0F 7E, stores on 66 0F D6; the forms
// the GPR-move fallback silently corrupted.
func TestMOVQXMMGroundTruth(t *testing.T) {
cases := []struct {
+82 -82
View File
@@ -9,10 +9,10 @@ import (
)
// This file implements EVEX (AVX-512) instruction encoding: the four-byte
// EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the
// EVEX prefix with 5-bit vector register fields (Z0-Z31, X/Y 16-31), the
// compressed disp8×N displacement, and the operand shapes the go-flac
// AVX-512 kernels use plus the common floating-point and conversion set.
// Masking follows the Go assembler's spelling: an explicit K1–K7 operand
// Masking follows the Go assembler's spelling: an explicit K1-K7 operand
// anywhere among the operands (merging) plus a ".Z" mnemonic suffix for
// zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are
// supported too.
@@ -36,7 +36,7 @@ type evexSpec struct {
// are taken from the Go assembler's opcode tables, which are authoritative
// for byte-for-byte agreement.
var evexTable = map[string]evexSpec{
// EVEX.128/256/512.66.0F — integer arithmetic / logic, NDS form.
// EVEX.128/256/512.66.0F, integer arithmetic / logic, NDS form.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -48,25 +48,25 @@ var evexTable = map[string]evexSpec{
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — packed double arithmetic.
// EVEX.128/256/512.66.0F.W1, packed double arithmetic.
"VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0 — packed single arithmetic.
// EVEX.128/256/512.0F.W0, packed single arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — packed double unpack.
// EVEX.128/256/512.66.0F.W1, packed double unpack.
"VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128.F2.0F.W1 — scalar double arithmetic (the packed opcodes with
// EVEX.128.F2.0F.W1, scalar double arithmetic (the packed opcodes with
// an F2 pp; the EVEX forms exist for masked and zeroing use). The
// memory operand is a single double, so disp8×N = 8.
"VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
@@ -76,7 +76,7 @@ var evexTable = map[string]evexSpec{
"VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128.F3.0F.W0 — scalar single arithmetic (disp8×N = 4).
// EVEX.128.F3.0F.W0, scalar single arithmetic (disp8×N = 4).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
@@ -84,38 +84,38 @@ var evexTable = map[string]evexSpec{
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.512.66.0F3A — align (NDS + imm8).
// EVEX.512.66.0F3A, align (NDS + imm8).
"VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F — immediate shift (VPSRAD /4).
// EVEX.128/256/512.66.0F, immediate shift (VPSRAD /4).
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — variable shift with an XMM count (VPSRAQ;
// EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
// the W bit distinguishes it from VPSRAD's E2 form).
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst,
// EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
// rm=src, no vvvv).
"VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1 — duplicate the low double (reg=dst,
// EVEX.128/256/512.F2.0F.W1, duplicate the low double (reg=dst,
// rm=src, no vvvv): a 128-bit destination reads a single double from
// memory (disp8×8), the wider ones read the full operand.
"VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}},
// EVEX.128/256/512.0F.W0 — signed dword to packed single (reg=dst,
// rm=src, no vvvv, no mandatory prefix — as in the VEX form).
// EVEX.128/256/512.0F.W0, signed dword to packed single (reg=dst,
// rm=src, no vvvv, no mandatory prefix, as in the VEX form).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0 — packed single to packed double: the
// EVEX.128/256/512.0F.W0, packed single to packed double: the
// destination is twice the source width and sets the length; disp8×N
// follows the narrow memory source. No F3 prefix: the Go assembler
// emits this instruction with pp = 00 (Intel's maps would call that
// undefined) and gasm reproduces the Go assembler's bytes — its machine
// undefined) and gasm reproduces the Go assembler's bytes, its machine
// code is the oracle, not the manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.128/256/512.F3.0F.W0 — signed dword to packed double (the EVEX
// EVEX.128/256/512.F3.0F.W0, signed dword to packed double (the EVEX
// form of the VEX instruction; the destination sets the length, disp8×N
// follows the narrow memory source).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
// EVEX packed double → dword conversions: the source is the wide
// operand and the mnemonic fixes the length — the bare names are
// operand and the mnemonic fixes the length, the bare names are
// 512-bit only (ZMM source, XMM destination), the X/Y spellings are
// EVEX-128/256. Exactly one slot of n is valid; it names the vector
// length (and the disp8×N multiplier) a register or memory source
@@ -127,7 +127,7 @@ var evexTable = map[string]evexSpec{
"VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.66.0F3A — ternary logic and lane shuffles (NDS + imm8).
// EVEX.66.0F3A, ternary logic and lane shuffles (NDS + imm8).
"VPTERNLOGD": {3, 0x25, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPTERNLOGQ": {3, 0x25, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFI32X4": {3, 0x43, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
@@ -136,11 +136,11 @@ var evexTable = map[string]evexSpec{
"VSHUFF64X2": {3, 0x23, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F — the EVEX forms of the VEX two-source shuffle.
// EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F3A — lane insert ($imm, xsrc, zsrc1, zdst).
// EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTF32X8": {3, 0x1A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
"VINSERTF64X2": {3, 0x18, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
@@ -150,7 +150,7 @@ var evexTable = map[string]evexSpec{
"VINSERTI64X2": {3, 0x38, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTI64X4": {3, 0x3A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
// EVEX.66.0F3A — lane extract (reg=source, rm=XMM/YMM destination,
// EVEX.66.0F3A, lane extract (reg=source, rm=XMM/YMM destination,
// imm8).
"VEXTRACTF32X4": {3, 0x19, 0, 1, -1, vexExtract, [3]int{0, 16, 16}},
"VEXTRACTF32X8": {3, 0x1B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
@@ -159,14 +159,14 @@ var evexTable = map[string]evexSpec{
"VEXTRACTI32X8": {3, 0x3B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTI64X2": {3, 0x39, 1, 1, -1, vexExtract, [3]int{0, 16, 16}},
// EVEX.66.0F — compare with an opmask destination ($imm, src2, src1,
// EVEX.66.0F, compare with an opmask destination ($imm, src2, src1,
// kdst): NDS3Imm with the K register in the reg field.
"VCMPPD": {1, 0xC2, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VCMPPS": {1, 0xC2, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}},
// EVEX.66.0F3A — integer compares with an opmask destination, the same
// EVEX.66.0F3A, integer compares with an opmask destination, the same
// NDS3Imm-with-k-reg shape as the floating-point compares; W selects the
// operand width (byte/word vs dword/qword), the opcode the signedness.
// The memory form takes a full vector, so disp8×N is 16/32/64.
@@ -179,7 +179,7 @@ var evexTable = map[string]evexSpec{
"VPCMPQ": {3, 0x1F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F38 — permutes (NDS form).
// EVEX.66.0F38, permutes (NDS form).
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -188,7 +188,7 @@ var evexTable = map[string]evexSpec{
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F — the wider integer set (NDS form).
// EVEX.66.0F, the wider integer set (NDS form).
"VPMADDWD": {1, 0xF5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -199,34 +199,34 @@ var evexTable = map[string]evexSpec{
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKUSDW": {2, 0x2B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F38 — absolute values and replicating moves (reg=dst,
// EVEX.66.0F38, absolute values and replicating moves (reg=dst,
// rm=src).
"VPABSB": {2, 0x1C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSW": {2, 0x1D, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSD": {2, 0x1E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSQ": {2, 0x1F, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.F3.0F — replicate even/odd singles.
// EVEX.F3.0F, replicate even/odd singles.
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — sign/zero-extending moves; the memory source is the
// EVEX.66.0F38, sign/zero-extending moves; the memory source is the
// narrow half (here byte to word).
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F — packed single conversions (reg=dst, rm=src).
// EVEX.66.0F, packed single conversions (reg=dst, rm=src).
"VCVTPS2DQ": {1, 0x5B, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTTPS2DQ": {1, 0x5B, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — broadcast a single/double to all lanes (reg=dst,
// EVEX.66.0F38, broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; disp8×N is the element size).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VBROADCASTSD": {2, 0x19, 1, 1, -1, vexRM, [3]int{0, 8, 8}},
// EVEX.66.0F38 — expand loads (rm → vector register destination).
// EVEX.66.0F38, expand loads (rm → vector register destination).
"VEXPANDPD": {2, 0x88, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
"VEXPANDPS": {2, 0x88, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VPEXPANDD": {2, 0x89, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VPEXPANDQ": {2, 0x89, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
// EVEX.66.0F38 — compress stores (vector register source → rm), and the
// EVEX.66.0F38, compress stores (vector register source → rm), and the
// remaining narrowing stores.
"VCOMPRESSPD": {2, 0x8A, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}},
"VCOMPRESSPS": {2, 0x8A, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}},
@@ -235,7 +235,7 @@ var evexTable = map[string]evexSpec{
"VPMOVWB": {2, 0x30, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQB": {2, 0x32, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
// EVEX.66.0F — rotates (immediate form: /0 right, /1 left).
// EVEX.66.0F, rotates (immediate form: /0 right, /1 left).
"VPRORD": {1, 0x72, 0, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
"VPRORQ": {1, 0x72, 1, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
"VPROLD": {1, 0x72, 0, 1, 1, vexShiftImm, [3]int{16, 32, 64}},
@@ -248,14 +248,14 @@ var evexTable = map[string]evexSpec{
"VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
"VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.66.0F38 — floating-point helpers, packed (reg=dst, rm=src).
// EVEX.66.0F38, floating-point helpers, packed (reg=dst, rm=src).
"VRCP14PD": {2, 0x4C, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRCP14PS": {2, 0x4C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRSQRT14PD": {2, 0x4E, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRSQRT14PS": {2, 0x4E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VGETEXPPD": {2, 0x42, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VGETEXPPS": {2, 0x42, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — floating-point helpers, scalar (NDS form: src2 is
// EVEX.66.0F38, floating-point helpers, scalar (NDS form: src2 is
// rm, src1 is vvvv, the XMM destination is reg). Like the scalar 0F3A
// forms, these take the 66 prefix; W selects double/single.
"VRCP14SD": {2, 0x4D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
@@ -264,13 +264,13 @@ var evexTable = map[string]evexSpec{
"VRSQRT14SS": {2, 0x4F, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
"VGETEXPSD": {2, 0x43, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VGETEXPSS": {2, 0x43, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.66.0F38 — scale by a power of two (NDS form).
// EVEX.66.0F38, scale by a power of two (NDS form).
"VSCALEFPD": {2, 0x2C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSCALEFPS": {2, 0x2C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSCALEFSD": {2, 0x2D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VSCALEFSS": {2, 0x2D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.66.0F3A — packed round/getmant/reduce ($imm, src, dst: reg=dst,
// EVEX.66.0F3A, packed round/getmant/reduce ($imm, src, dst: reg=dst,
// rm=src, imm8).
"VRNDSCALEPD": {3, 0x09, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VRNDSCALEPS": {3, 0x08, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
@@ -278,7 +278,7 @@ var evexTable = map[string]evexSpec{
"VGETMANTPS": {3, 0x26, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VREDUCEPD": {3, 0x56, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VREDUCEPS": {3, 0x56, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.66.0F3A — scalar round/getmant/reduce and fixup/range (NDS +
// EVEX.66.0F3A, scalar round/getmant/reduce and fixup/range (NDS +
// imm8: $imm, src2, src1, dst). The scalar 0F3A forms all take the 66
// prefix; W selects double/single.
"VRNDSCALESD": {3, 0x0B, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
@@ -296,7 +296,7 @@ var evexTable = map[string]evexSpec{
"VRANGESD": {3, 0x51, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VRANGESS": {3, 0x51, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
// EVEX.66.0F3A — floating-point class test ($imm, src, kdst): the
// EVEX.66.0F3A, floating-point class test ($imm, src, kdst): the
// reg field carries the opmask destination. The packed forms carry an
// explicit length in the mnemonic (X/Y/Z).
"VFPCLASSPDX": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{16, 0, 0}},
@@ -308,7 +308,7 @@ var evexTable = map[string]evexSpec{
"VFPCLASSSD": {3, 0x67, 1, 1, -1, vexImmRM, [3]int{8, 0, 0}},
"VFPCLASSSS": {3, 0x67, 0, 1, -1, vexImmRM, [3]int{4, 0, 0}},
// EVEX — the remaining conversions. VCVTQQ2PS narrows (the 512-bit
// EVEX, the remaining conversions. VCVTQQ2PS narrows (the 512-bit
// source sets the length); the rest follow the destination.
"VCVTQQ2PS": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2QQ": {1, 0x7B, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
@@ -316,13 +316,13 @@ var evexTable = map[string]evexSpec{
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F38 — half-precision convert (half-width source).
// EVEX.66.0F38, half-precision convert (half-width source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src,
// rm=dst, imm8 — the extract layout).
// EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
// rm=dst, imm8, the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
// EVEX — unsigned and truncating conversions. The PD sources are the
// EVEX, unsigned and truncating conversions. The PD sources are the
// wide operand (the bare names are 512-bit only, the X/Y spellings fix
// the length); the PS/UQQ destinations are wide and follow the
// destination.
@@ -349,7 +349,7 @@ var evexTable = map[string]evexSpec{
"VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.66.0F38 — the remaining sign/zero-extending moves (narrow
// EVEX.66.0F38, the remaining sign/zero-extending moves (narrow
// source; disp8×N follows its size).
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
@@ -361,7 +361,7 @@ var evexTable = map[string]evexSpec{
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.F3.0F38 — the remaining narrowing stores (vector source in reg,
// EVEX.F3.0F38, the remaining narrowing stores (vector source in reg,
// narrow destination in r/m): signed, unsigned and the D/Q truncations.
"VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
@@ -378,7 +378,7 @@ var evexTable = map[string]evexSpec{
"VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
// EVEX.F3.0F38 — mask/vector conversions: M2* moves an opmask register
// EVEX.F3.0F38, mask/vector conversions: M2* moves an opmask register
// into a vector (rm = K source, reg = vector destination), *2M does the
// reverse (reg = K destination, rm = vector source, the length follows
// the vector).
@@ -391,7 +391,7 @@ var evexTable = map[string]evexSpec{
"VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX — scalar conversions between vector and general-purpose
// EVEX, scalar conversions between vector and general-purpose
// registers. Vector to GPR (two operands: vec/mem source, GPR
// destination, vvvv unused): the signed and truncated pair, and the
// unsigned forms (EVEX only).
@@ -421,22 +421,22 @@ var evexTable = map[string]evexSpec{
"VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
// EVEX.128/256/512.66.0F38.W0, sign-extend dwords to qwords; the memory
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
// the xmm/ymm/zmm destination lengths).
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.512.66.0F3A.W1 — lane extract (reg=ZMM source, rm=YMM/memory
// EVEX.512.66.0F3A.W1, lane extract (reg=ZMM source, rm=YMM/memory
// destination, imm8).
"VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
// EVEX.66.0F38 — more integer NDS forms (W distinguishes D/Q).
// EVEX.66.0F38, more integer NDS forms (W distinguishes D/Q).
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
// EVEX.128/256/512 — the wider integer set (AVX-512 F/BW): byte/word
// EVEX.128/256/512, the wider integer set (AVX-512 F/BW): byte/word
// arithmetic, the bitwise ops with D/Q suffixes, min/max, averages and
// variable shifts. All NDS form; W distinguishes element size.
"VPADDB": {1, 0xFC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -474,21 +474,21 @@ var evexTable = map[string]evexSpec{
"VPSRAVQ": {2, 0x46, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX forms of instructions that also exist in VEX (selected when a ZMM
// or K register, or indices 16–31, demand EVEX).
// or K register, or indices 16-31, demand EVEX).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F — immediate shift (VPSLLD /6).
// EVEX.66.0F, immediate shift (VPSLLD /6).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.F3.0F38.W0 — narrowing stores: reg = wide source, rm = narrow
// EVEX.F3.0F38.W0, narrowing stores: reg = wide source, rm = narrow
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
}
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
// depends on the source kind — a GPR source uses opReg, a memory source uses
// depends on the source kind, a GPR source uses opReg, a memory source uses
// opMem with a disp8×N of n.
type evexBcastSpec struct {
mapSel int
@@ -499,10 +499,10 @@ type evexBcastSpec struct {
}
var evexBcastTable = map[string]evexBcastSpec{
// EVEX.128/256/512.66.0F38 — broadcast a dword/qword to all lanes.
// EVEX.128/256/512.66.0F38, broadcast a dword/qword to all lanes.
"VPBROADCASTD": {2, 0x7C, 0x58, 0, 4},
"VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8},
// EVEX.128/256/512.66.0F38 — broadcast a byte/word (GPR or memory
// EVEX.128/256/512.66.0F38, broadcast a byte/word (GPR or memory
// source) to all lanes.
"VPBROADCASTB": {2, 0x7A, 0x78, 0, 1},
"VPBROADCASTW": {2, 0x7B, 0x79, 0, 2},
@@ -521,26 +521,26 @@ type evexMoveSpec struct {
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0 — unaligned integer move.
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1 — unaligned qword move.
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W0 — unaligned byte move (byte/word moves use the
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1 — unaligned word move (shares the qword
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — unaligned packed double move.
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512 — aligned packed moves.
// EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F — aligned integer moves.
// EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128.F3.0F.W0 — scalar single move, memory operands (the
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}},
}
@@ -559,7 +559,7 @@ func isEvex(mnemUpper string) bool {
// evexRequired reports whether the operands force the EVEX encoding of a
// mnemonic that also has a VEX form: ZMM and K registers do, and so do
// register indices 16–31, which only EVEX can represent (X16–Y31 exist
// register indices 16-31, which only EVEX can represent (X16-Y31 exist
// solely under AVX-512).
func evexRequired(upper string, ops []Operand) bool {
_, inVex := vexTable[upper]
@@ -578,7 +578,7 @@ func evexRequired(upper string, ops []Operand) bool {
// evexSuffix carries the EVEX mnemonic suffixes the Go assembler accepts:
// zeroing (.Z), a rounding mode (.RN_SAE, .RD_SAE, .RU_SAE, .RZ_SAE),
// suppress-all-exceptions (.SAE) and memory broadcast (.BCST). Masking is
// not a suffix — Go writes it as an explicit K operand.
// not a suffix, Go writes it as an explicit K operand.
type evexSuffix struct {
zeroing bool
sae bool
@@ -668,7 +668,7 @@ var evexRound = map[string]bool{
}
// evexBcstN maps an instruction accepting .BCST to the broadcast element
// size — the disp8×N multiplier for its memory operand.
// size, the disp8×N multiplier for its memory operand.
var evexBcstN = map[string]int{
"VADDPD": 8, "VSUBPD": 8, "VMULPD": 8, "VDIVPD": 8,
"VMINPD": 8, "VMAXPD": 8,
@@ -689,7 +689,7 @@ var evexBcstN = map[string]int{
"VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8,
}
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
// splitMask extracts an explicit mask register (K1-K7) from the operand list,
// returning the remaining operands and the mask index. K0 is not a usable
// mask (aaa = 0 means "no mask"), matching the assembler.
func splitMask(ops []Operand) ([]Operand, int, error) {
@@ -712,7 +712,7 @@ func splitMask(ops []Operand) ([]Operand, int, error) {
}
// encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The
// mask, when present, is an explicit K1–K7 operand anywhere among the
// mask, when present, is an explicit K1-K7 operand anywhere among the
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
// broadcast.
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
@@ -1042,7 +1042,7 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
}
// encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the length fixed by the mnemonic — the
// the destination always XMM and the length fixed by the mnemonic, the
// single valid slot of spec.n names the vector length (and the disp8×N
// multiplier) a register or memory source encodes.
func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
@@ -1061,7 +1061,7 @@ func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx eve
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
}
// soleLen returns the vector-length index of the single valid slot of n —
// soleLen returns the vector-length index of the single valid slot of n
// the length a length-fixed mnemonic (the EVEX conversion spellings) encodes
// regardless of its operands.
func soleLen(n [3]int) (int, error) {
@@ -1131,8 +1131,8 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx eve
// emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement
// (disp8×N compressed) for the given precomputed fields. regIdx is the
// unextended reg-field register index, or a /digit (0–7); vvvvIdx is the
// vvvv register index, or -1 when unused. mask (K1–K7, 0 = unmasked) and
// unextended reg-field register index, or a /digit (0-7); vvvvIdx is the
// vvvv register index, or -1 when unused. mask (K1-K7, 0 = unmasked) and
// zeroing fill the aaa and z bits of the P2 byte.
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, sfx evexSuffix) error {
if ll > 2 {
@@ -1324,7 +1324,7 @@ func isScatter(upper string) bool {
}
// vsibLen validates a VSIB memory operand (the index must be a vector
// register) and returns it with the vector length the index selects — the
// register) and returns it with the vector length the index selects, the
// EVEX L'L field follows the index register, not the data register.
func vsibLen(op Operand, what string) (Mem, int, error) {
m, ok := op.(Mem)
@@ -1383,7 +1383,7 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib)
}
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib — reg = src,
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src,
// rm = the VSIB memory operand, the K mask in aaa and L following the VSIB
// index.
func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evexSuffix) error {
@@ -1411,15 +1411,15 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
// evexKOperand lists the instructions whose K register is a genuine operand
// (the source or destination of a mask/vector conversion) rather than a
// mask modifier — the M2 and 2M conversions. They take no masking.
// mask modifier, the M2 and 2M conversions. They take no masking.
var evexKOperand = map[string]bool{
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
}
// kmovSpec describes a KMOV width: the opcode depends on the operand
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory
// direction, kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
// gprk (GPR/mem → K), kgpr (K → GPR), and the GPR forms carry a mandatory
// prefix and W for the wider widths.
type kmovSpec struct {
kk, kmem, gprk, kgpr byte
+13 -13
View File
@@ -16,7 +16,7 @@ import (
// kernels use: NDS arithmetic, immediate and variable shifts, shuffles with
// an immediate, lane extracts, narrowing stores, broadcasts from a GPR or
// memory, mask destinations, mask moves, disp8×N compression and the 5-bit
// register fields (X/Y 16–31, Z 0–31).
// register fields (X/Y 16-31, Z 0-31).
func TestEvexGroundTruth(t *testing.T) {
cases := []struct {
name string
@@ -58,7 +58,7 @@ func TestEvexGroundTruth(t *testing.T) {
{"VMOVDQU32 16(SI)(R15*4),Z4", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 16, 64), vreg(t, "Z4")}, "62b17e486fa4be10000000"},
{"VMOVDQU32 Z0,4(SI)(AX*1)", "VMOVDQU32", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f17e487f840604000000"},
{"VMOVDQU32 Z3,(DI)(R15*4)", "VMOVDQU32", []Operand{vreg(t, "Z3"), Idx(DI, vreg(t, "R15"), 4, 0, 64)}, "62b17e487f1cbf"},
// VMOVDQU64 — the W1 qword variant.
// VMOVDQU64; the W1 qword variant.
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
@@ -77,7 +77,7 @@ func TestEvexGroundTruth(t *testing.T) {
{"VPSHUFB Z1,Z2,Z3", "VPSHUFB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4800d9"},
{"VMOVDQU8 Z1,Z2", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17f487fca"},
{"VMOVDQU16 Z1,Z2", "VMOVDQU16", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff487fca"},
// Indices 16–31: rm[4] rides in X̄ for register operands.
// Indices 16-31: rm[4] rides in X̄ for register operands.
{"VPSHUFD $1,X16,X17", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}, "62a17d0870c801"},
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
@@ -96,7 +96,7 @@ func TestEvexGroundTruth(t *testing.T) {
{"VPBROADCASTD 4(SI),Z10", "VPBROADCASTD", []Operand{Ptr(SI, 4, 4), vreg(t, "Z10")}, "62727d48585601"},
{"VPBROADCASTQ R8,X31", "VPBROADCASTQ", []Operand{vreg(t, "R8"), vreg(t, "X31")}, "6242fd087cf8"},
{"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"},
// Register indices 16–31 exist only in EVEX encodings.
// Register indices 16-31 exist only in EVEX encodings.
{"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"},
// Packed double arithmetic / unpack (EVEX forms carry W=1).
{"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"},
@@ -107,7 +107,7 @@ func TestEvexGroundTruth(t *testing.T) {
{"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"},
{"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"},
{"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"},
// VMOVDDUP — duplicate the low double; disp8×N = 64 at 512 bits, and
// VMOVDDUP; duplicate the low double; disp8×N = 64 at 512 bits, and
// X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄).
{"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"},
{"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"},
@@ -150,7 +150,7 @@ func TestEvexGroundTruth(t *testing.T) {
}
}
// TestEvexMasking checks the AVX-512 mask operand (K1–K7, placed freely among
// TestEvexMasking checks the AVX-512 mask operand (K1-K7, placed freely among
// the operands) and the .Z zeroing suffix, byte for byte against the Go
// assembler.
func TestEvexMasking(t *testing.T) {
@@ -241,11 +241,11 @@ func TestEvexMasking(t *testing.T) {
}
}
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set — ternary
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set; ternary
// logic, lane shuffles/inserts/extracts, compares with a K destination,
// permutes, the wider integer families, expand/compress, broadcasts,
// rotates and word shifts, the opmask instructions, the EVEX suffixes
// (rounding/SAE/broadcast) and the aligned/scalar moves — byte for byte
// (rounding/SAE/broadcast) and the aligned/scalar moves; byte for byte
// against the Go assembler.
func TestEvexExtendedGroundTruth(t *testing.T) {
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
@@ -275,7 +275,7 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
// Packed single arithmetic (same opcodes, no mandatory prefix) —
// Packed single arithmetic (same opcodes, no mandatory prefix);
// ZMM, YMM and XMM widths, rounding and broadcast.
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
@@ -399,8 +399,8 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
}
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef,
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions —
// tail of the EVEX set; reciprocals, rsqrt, getexp/getmant, scalef,
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions;
// plus gather/scatter with VSIB addressing, byte for byte against the Go
// assembler.
func TestEvexHelperGroundTruth(t *testing.T) {
@@ -502,9 +502,9 @@ func TestEvexHelperGroundTruth(t *testing.T) {
}
// TestEvexGprGroundTruth covers the scalar conversions between vector and
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
// general-purpose registers; the signed and truncated VCVT{,T}S{D,S}2SI
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv; byte
// for byte against the Go assembler, including memory sources and extended
// GPRs.
func TestEvexGprGroundTruth(t *testing.T) {
+99 -17
View File
@@ -14,8 +14,8 @@ import (
"sync"
)
// This file emits GOOBJ — the Go toolchain's object format, which cmd/link
// consumes directly — so gasm-assembled functions drop into a go build
// This file emits GOOBJ, the Go toolchain's object format, which cmd/link
// consumes directly, so gasm-assembled functions drop into a go build
// without the Go assembler. The layout follows cmd/internal/goobj: a
// toolchain preamble ("go object ...\n!\n"), the go120ld header with its
// block offsets, a string table, symbol definitions, the relocation /
@@ -94,10 +94,12 @@ const (
)
// Relocation types (cmd/internal/objabi).
// R_PCREL and R_ADDR are stable across Go versions.
// R_ADDR, R_CALL, R_PCREL and R_TLS_LE are stable across Go versions.
const (
relocPCRel = 14 // R_PCREL
relocAddr = 1 // R_ADDR
relocCall = 7 // R_CALL
relocPCRel = 14 // R_PCREL
relocTLSLE = 15 // R_TLS_LE
)
// relocDWTXTADDRU4 returns the R_DWTXTADDR_U4 relocation type for the
@@ -140,10 +142,30 @@ func isGo127OrLater() bool {
// Special package indices for symbol references.
const (
pkgIdxNone = 0x7fffffff
pkgIdxSelf = 0x7ffffffb
pkgIdxNone = 0x7fffffff
pkgIdxSelf = 0x7ffffffb
pkgIdxBuiltin = 0x7ffffffc
)
// goobjBuiltinMorestackNoctxt is the index of runtime.morestack_noctxt in
// cmd/internal/goobj/builtinlist.go of the toolchain the object targets
// (246 since Go 1.25; the list is append-only).
const goobjBuiltinMorestackNoctxt = 246
// goobjBuiltinMorestack is the builtin reference the toolchain emits for the
// stack-guard call.
var goobjBuiltinMorestack = "runtime\u00b7morestack_noctxt"
// isCallReloc reports whether k is one of the per-arch call relocations a
// direct branch to a TEXT symbol carries.
func isCallReloc(k RelocKind) bool {
switch k {
case RelCall, RelRISCVJal, RelArm64Branch, RelLoong64Branch:
return true
}
return false
}
const goobjMagic = "\x00go120ld"
// goSym is one symbol definition under construction.
@@ -178,14 +200,24 @@ type dwarfRelocSet struct {
// does with its -p flag). srcPath names the source file recorded in the
// object's file table and line tables. The toolchain's object preamble is
// captured from the installed go tool asm, so the output links with the
// toolchain it was produced on — exactly like a real assembly object.
// toolchain it was produced on, exactly like a real assembly object.
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreamble()
if err != nil {
return nil, err
}
// amd64: MinLC 1, R_PCREL for the code relocations.
return img.emitGOObject(pkgPath, srcPath, pre, 1, func(Reloc) (uint16, uint8) { return relocPCRel, 4 })
// amd64: MinLC 1, R_PCREL for displacements, R_CALL for calls and
// R_TLS_LE for the stack-guard TLS load.
return img.emitGOObject(pkgPath, srcPath, pre, 1, func(r Reloc) (uint16, uint8) {
switch r.Kind {
case RelCall:
return relocCall, 4
case RelTLSLE:
return relocTLSLE, 4
default:
return relocPCRel, 4
}
})
}
// emitGOObject assembles the GOOBJ payload for any architecture. pre is
@@ -198,7 +230,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
}
// The non-package definitions first — the DWARF symbols reference the
// The non-package definitions first, the DWARF symbols reference the
// functions by these indices: per function the four pc-value tables
// and the function itself, as cmd/asm lays them out.
type npSym struct {
@@ -247,7 +279,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
// their relocations cover whole AUIPC/pcalau12i pairs, so
// zeroing r.Off would erase the opcode/register bits the linker
// preserves when it patches only the immediate.
if r.Kind != RelPCRel32 {
if r.Kind != RelPCRel32 && r.Kind != RelCall {
continue
}
if r.Off >= 0 && r.Off+4 <= len(code) {
@@ -320,6 +352,13 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
)
}
// Index the non-package TEXT definitions by short name for the internal
// call references.
textNpIdx := map[string]int{}
for i, fn := range img.Funcs {
textNpIdx[fn.Name] = fnNpIdx[i]
}
// Resolve external symbol references (cross-package). Build the
// package index table and determine each external symbol's SymIdx
// by reading the target package's export data.
@@ -327,10 +366,20 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
var extPkgIdx map[string]int
var extSymIdx map[string]int
if len(img.Externals) > 0 {
var err error
extPkgTable, extPkgIdx, extSymIdx, err = resolveExternalSymbols(img.Externals)
if err != nil {
return nil, fmt.Errorf("GOOBJ emission: resolving external symbols: %w", err)
// The morestack call is a builtin reference, not a resolved external.
var need []string
for _, n := range img.Externals {
if n == goobjBuiltinMorestack {
continue
}
need = append(need, n)
}
if len(need) > 0 {
var err error
extPkgTable, extPkgIdx, extSymIdx, err = resolveExternalSymbols(need)
if err != nil {
return nil, fmt.Errorf("GOOBJ emission: resolving external symbols: %w", err)
}
}
}
@@ -342,6 +391,31 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
si := len(defs) + fnNpIdx[i]
for _, r := range fn.Relocs {
typ, size := relocField(r)
if r.Kind == RelTLSLE {
// The TLS load has no symbol: {0, 0} is the nil ref.
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], 0)
binary.LittleEndian.PutUint32(rec[19:], 0)
symRelocs[si] = append(symRelocs[si], rec[:]...)
continue
}
if r.External && r.Name == goobjBuiltinMorestack {
// The stack-guard morestack call uses the toolchain's
// builtin reference.
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxBuiltin)
binary.LittleEndian.PutUint32(rec[19:], goobjBuiltinMorestackNoctxt)
symRelocs[si] = append(symRelocs[si], rec[:]...)
continue
}
if r.External {
// Split package-qualified name: "runtime·morestack" → runtime, morestack.
pkg, name := splitQualified(r.Name)
@@ -366,16 +440,24 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
symRelocs[si] = append(symRelocs[si], rec[:]...)
continue
}
pkg := uint32(pkgIdxSelf)
di, ok := defIdx[r.Name]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
// A call to a TEXT function of the same file references the
// non-package definition table.
ni, isText := textNpIdx[r.Name]
if !isText || !isCallReloc(r.Kind) {
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
}
pkg = pkgIdxNone
di = ni
}
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size // field width
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
binary.LittleEndian.PutUint32(rec[15:], pkg)
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
symRelocs[si] = append(symRelocs[si], rec[:]...)
}
+1 -1
View File
@@ -84,7 +84,7 @@ func sortedPkgRefs(refs map[string][]string) []pkgRef {
for pkg, syms := range refs {
pkgs = append(pkgs, pkgRef{pkg, syms})
}
// Simple insertion sort — the list is tiny (usually 1–3 packages).
// Simple insertion sort, the list is tiny (usually 1-3 packages).
for i := 1; i < len(pkgs); i++ {
for j := i; j > 0 && pkgs[j-1].path > pkgs[j].path; j-- {
pkgs[j-1], pkgs[j] = pkgs[j], pkgs[j-1]
+4 -4
View File
@@ -158,8 +158,8 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
t.Errorf("funcinfo bytes %x", fi)
}
// The pc-value tables of addq (non-package indices 0–3, so global
// indices 7–10): pcsp a flat zero over the whole function, pcinline a
// The pc-value tables of addq (non-package indices 0-3, so global
// indices 7-10): pcsp a flat zero over the whole function, pcinline a
// flat -1, both with the pc delta in MinLC (1) units.
pcsp := data[le.Uint32(didx[4*7:]):]
if got := pcsp[:3]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
@@ -294,7 +294,7 @@ TEXT ·framed(SB), NOSPLIT, $8-0
}
for i := range wantPCs {
if pcs[i] != wantPCs[i] || vals[i] != wantVals[i] {
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d) — all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d); all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
}
}
// The last two steps unwind the epilogue to zero.
@@ -333,7 +333,7 @@ TEXT ·useext(SB), NOSPLIT, $0-8
// TestGOObjectLinkAndRun is the end-to-end check: assemble the test
// functions to a GOOBJ, swap it into a go build in place of the toolchain's
// assembly object, link, and run — the output must match the baseline
// assembly object, link, and run; the output must match the baseline
// binary the Go assembler produced. Skipped when no Go toolchain is
// available.
func TestGOObjectLinkAndRun(t *testing.T) {
+13 -8
View File
@@ -13,28 +13,33 @@ import (
)
// GOObjectAARCH64 emits a GOOBJ object file for AArch64. The layout is
// the shared one in goobj.go — the toolchain preamble, the go120ld header
// the shared one in goobj.go, the toolchain preamble, the go120ld header
// with its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the arm64 preamble, the MinLC of 4
// for the pc-value deltas, and R_ADDRARM64 relocation types for the
// ADRP+ADD/LDR/STR address pairs.
// reloc/aux/data index arrays, with the arm64 preamble, the MinLC of 4
// for the pc-value deltas, and the arm64 relocation types for the ADRP
// pairs and BL calls.
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreambleAARCH64()
if err != nil {
return nil, err
}
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
if r.Kind == RelArm64Branch {
switch r.Kind {
case RelArm64Branch:
return relocArm64Branch, 4
case RelArm64LDST64:
return relocArm64LDST64, 4
default:
return relocArm64Addr, 4
}
return relocArm64Addr, 4
})
}
// arm64 relocation types (cmd/internal/objabi).
const (
relocArm64Addr = 3 // R_ADDRARM64 — ADRP+ADD/LDR/STR pair
relocArm64Branch = 9 // R_CALLARM64 — BL instruction
relocArm64Addr = 3 // R_ADDRARM64, ADRP+ADD pair
relocArm64Branch = 9 // R_CALLARM64, BL instruction
relocArm64LDST64 = 40 // R_ARM64_PCREL_LDST64, ADRP+LDR/STR pair
)
// toolchainObjectPreambleAARCH64 returns the "go object ...\n!\n" header
+11 -5
View File
@@ -13,9 +13,9 @@ import (
)
// GOObjectLOONG64 emits a GOOBJ object file for LoongArch. The layout is
// the shared one in goobj.go — the toolchain preamble, the go120ld header
// the shared one in goobj.go, the toolchain preamble, the go120ld header
// with its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the loong64 preamble, the MinLC of 4
// reloc/aux/data index arrays, with the loong64 preamble, the MinLC of 4
// for the pc-value deltas, and R_LOONG64_ADDR_HI/LO relocation types for
// the pcalau12i+addi.d address pairs.
func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
@@ -25,11 +25,16 @@ func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
}
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
// A pcalau12i+addi.d pair: the high part carries
// R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO.
if r.Kind == RelLoong64AddrLo {
// R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO; the guard's
// morestack call carries R_CALLLOONG64.
switch {
case r.Kind == RelLoong64AddrLo:
return relocLoong64AddrLo, 4
case r.Kind == RelLoong64Branch:
return relocCallLoong64, 4
default:
return relocLoong64AddrHi, 4
}
return relocLoong64AddrHi, 4
})
}
@@ -39,6 +44,7 @@ func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
const (
relocLoong64AddrHi = 77 // R_LOONG64_ADDR_HI
relocLoong64AddrLo = 78 // R_LOONG64_ADDR_LO
relocCallLoong64 = 84 // R_CALLLOONG64
)
// toolchainObjectPreambleLOONG64 returns the "go object ...\n!\n" header
+2 -2
View File
@@ -13,9 +13,9 @@ import (
)
// GOObjectRISCV emits a GOOBJ object file for RISC-V. The layout is the
// shared one in goobj.go — the toolchain preamble, the go120ld header with
// shared one in goobj.go, the toolchain preamble, the go120ld header with
// its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the RISC-V preamble, the MinLC of 2 for
// reloc/aux/data index arrays, with the RISC-V preamble, the MinLC of 2 for
// the pc-value deltas, and the single R_RISCV_PCREL_ITYPE/STYPE relocation
// per AUIPC pair, matching `go tool asm`'s model (each pair is one 8-byte
// relocation, not the ELF HI20/LO12 pair).
+283
View File
@@ -0,0 +1,283 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/hex"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64,
// verified with go tool objdump): the stack-split guard classes, the morestack
// block and the auto-NOSPLIT leaf behaviour.
func TestStackGuardBytes(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"554889e54883ec104883c4105dc3"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"644c8b3425000000004c8da42478ffffff4d3b66107614554889e54881ec000100004881c4000100005dc3e800000000ebce"},
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"644c8b3425000000004989e44981ec881f0000721a4d3b66107614554889e54881ec002000004881c4002000005dc3e800000000ebca"},
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"644c8b342500000000493b66107613554889e54883ec10e8000000004883c4105dc3e800000000ebd7"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"554889e54883ec104883c4105dc3"},
} {
f, errs := parser.Parse("g_amd64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
// The toolchain's object leaves every relocation field zero for the
// linker, while the gasm image resolves file-internal references, so
// the comparison masks the patch sites the way verify's ground truth
// does.
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// TestStackGuardRelocs checks the guard's patch sites: the TLS slot and the
// morestack call.
func TestStackGuardRelocs(t *testing.T) {
f, errs := parser.Parse("g_amd64.s", "TEXT \u00b7f(SB), $256-0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
relocs := img.Funcs[0].Relocs
if len(relocs) != 2 {
t.Fatalf("relocs = %d, want 2", len(relocs))
}
tls, call := relocs[0], relocs[1]
if tls.Kind != RelTLSLE || tls.Off != 5 || tls.Name != "" || tls.External {
t.Errorf("tls reloc = %+v, want RelTLSLE at 5 with no symbol", tls)
}
if call.Kind != RelCall || call.Name != "runtime\u00b7morestack_noctxt" || !call.External {
t.Errorf("call reloc = %+v, want RelCall to runtime.morestack_noctxt", call)
}
}
// TestStackGuardGOObj emissions succeed with the guard's TLS and builtin
// references in play.
func TestStackGuardGOObj(t *testing.T) {
f, errs := parser.Parse("g_amd64.s", "TEXT \u00b7f(SB), $256-0\n\tCALL \u00b7helper(SB)\n\tRET\nTEXT \u00b7helper(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
obj, err := img.GOObject("testpkg", "g_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
if !bytes.Contains(obj, []byte("go120ld")) {
t.Fatal("object lacks the GOOBJ magic")
}
}
// The arm64 stack-split guard, pinned from `go tool asm` (Go 1.27, arm64):
// the guard classes, the auto-NOSPLIT leaf behaviour and the morestack
// block. Relocation fields are masked: the toolchain's object leaves them
// zero for the linker, the gasm image resolves file-internal references.
func TestStackGuardBytesARM64(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"900b40f9f14302d13f0210eb09010054f44304d19dfa3fa99f020091fd2300d1fd230491ff430491c0035fd6e3031eaa00000000f3ffff17"},
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"900b40f91bf283d2f1633beba30100543f0210eb690100541b0284d2f4633bcb9dfa3fa99f020091fd2300d11b0184d2fd633b8b1b0284d2ff633b8bc0035fd6e3031eaa00000000eeffff17"},
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"900b40f9ff6330eb09010054fe0f1ef8fd831ff8fd2300d100000000fd835ff8fe0742f8c0035fd6e3031eaa00000000f4ffff17"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"},
} {
f, errs := parser.Parse("g_arm64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// The riscv64 stack-split guard, pinned from `go tool asm` (Go 1.27,
// riscv64): the morestack call sits between the guard and the body, and the
// guard branches forward over it. Relocation fields are masked.
func TestStackGuardBytesRISCV64(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"03b30d0163662300000000006ff05fff233411fe211106e08260610167800000"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"03b30d01930381f763667300000000006ff01fff233c11ee130181ef06e082601301811067800000"},
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"03b30d0189639b8383f863697100f97f9b8f8f07b303f10163667300000000006ff01ffef97f8a9f23bc1ffef97fe13f7e9106e08260896fa12f7e9167800000"},
{"frameless", "TEXT \u00b7frameless(SB), $0-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"03b30d0163662300000000006ff05fff233c11fe611106e0000000008260210167800000"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"233411fe211106e08260610167800000"},
} {
f, errs := parser.Parse("g_riscv64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// The loong64 stack-split guard, pinned from `go tool asm` (Go 1.27,
// loong64): every guard class (including the medium class with the
// materialised constant and the big class with the ORI-less constants), the
// auto-NOSPLIT leaf behaviour, the large-frame R30 prologue/epilogue forms
// and the morestack block. Relocation fields are masked.
func TestStackGuardBytesLOONG64(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"61a0ff2963a0ff026100c0296360c0022000004c"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"d442c02878e0fd0294e21200801a004061e0fb2963e0fb026100c0296320c4022000004c3f00150000000000ffd7ff53"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"61a0ff2963a0ff026100c0296360c0022000004c"},
// The LR store leaves the 12-bit store-offset range while the SP
// adjust immediate still fits, and the epilogue adjusts through a
// single ORI.
{"fit2048", "TEXT \u00b7fit2048(SB), $2040-0\n\tRET\n",
"d442c0287800e20294e21200802600401e000014de8f1000c103e0296300e0026100c0291e00a00363f810002000004c3f00150000000000ffcbff53"},
// Medium class at the materialisation boundary (off = 2048 still
// immediate, 2049+ goes through R30).
{"med2048off", "TEXT \u00b7med2048off(SB), $2168-0\n\tRET\n",
"d442c0287800e00294e21200802e0040feffff15de8f1000c103de29feffff15de039e0363f810006100c0291e00a20363f810002000004c3f00150000000000ffc3ff53"},
{"medmat", "TEXT \u00b7medmat(SB), $2176-0\n\tRET\n",
"d442c028feffff15dee39f0378f8100094e21200802e0040feffff15de8f1000c1e3dd29feffff15dee39d0363f810006100c0291e20a20363f810002000004c3f00150000000000ffbbff53"},
// Big class with the rounding-split store and the floor-split adjust.
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"d442c0283e000014de23be0378f8120000470044deffff15dee3810378f8100094e2120080320040deffff15de8f1000c1e3ff29beffff15dee3bf0363f810006100c0295e000014de23800363f810002000004c3f00150000000000ffa7ff53"},
// Zero low 12 bits drop the ORI from the store, the adjust and the
// epilogue materialisation.
{"bigzero", "TEXT \u00b7bigzero(SB), $4088-0\n\tRET\n",
"d442c028feffff15de03820378f8100094e21200802a0040feffff15de8f1000c103c029feffff1563f810006100c0293e00001463f810002000004c3f00150000000000ffbfff53"},
// Big class whose first constant has a zero high part: a single ORI.
{"big3976", "TEXT \u00b7big3976(SB), $4096-0\n\tRET\n",
"d442c0281e20be0378f8120000470044feffff15dee3810378f8100094e2120080320040feffff15de8f1000c1e3ff29deffff15dee3bf0363f810006100c0293e000014de23800363f810002000004c3f00150000000000ffabff53"},
// Big class at a multiple of 4096: both guard constants lose their
// ORI word.
{"giantlo0", "TEXT \u00b7giantlo0(SB), $4216-0\n\tRET\n",
"d442c0283e00001478f8120000430044feffff1578f8100094e2120080320040feffff15de8f1000c103fe29deffff15de03be0363f810006100c0293e000014de03820363f810002000004c3f00150000000000ffafff53"},
// Non-leaf big frame: the body call plus the LR restore epilogue.
{"callbig", "TEXT \u00b7callbig(SB), $8192-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"d442c0283e000014de23be0378f81200004f0044deffff15dee3810378f8100094e21200803a0040deffff15de8f1000c1e3ff29beffff15dee3bf0363f810006100c029000000006100c0285e000014de23800363f810002000004c3f00150000000000ff9fff53"},
} {
f, errs := parser.Parse("g_loong64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// TestStackGuardGOObjInternalCall checks that GOOBJ emission succeeds when a
// guarded function calls a TEXT symbol of the same file, for every arch's
// call relocation kind.
func TestStackGuardGOObjInternalCall(t *testing.T) {
for _, tt := range []struct {
src string
assemble func(*ast.File) (*Image, error)
}{
{"g_amd64.s", AssembleFile},
{"g_arm64.s", AssembleFileARM64},
{"g_riscv64.s", AssembleFileRISCV},
{"g_loong64.s", AssembleFileLOONG64},
} {
f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.src, errs)
}
img, err := tt.assemble(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.src, err)
}
if _, err := img.GOObject("testpkg", tt.src); err != nil {
t.Errorf("%s: GOObject: %v", tt.src, err)
}
}
}
+14 -14
View File
@@ -21,7 +21,7 @@ var aluOp = map[string]struct {
}
// unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use
// the 0xFE/0xFF group (the short 0x40–0x4F forms are REX prefixes in 64-bit
// the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes in 64-bit
// mode); NEG/NOT use the 0xF6/0xF7 group.
var unaryOp = map[string]struct {
digit int
@@ -33,7 +33,7 @@ var unaryOp = map[string]struct {
"NEG": {3, 0xF7},
}
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0–0xD3 group.
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0-0xD3 group.
var shiftOp = map[string]int{
"SHL": 4,
"SHR": 5,
@@ -50,7 +50,7 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
// Integer scalar XMM moves: MOVQ with an XMM operand is the SSE2
// packed-quadword move, NOT a GPR move: mem→xmm encodes as F3 0F 7E
// (reg = dst, no REX.W — the Go assembler's form), xmm→mem as
// (reg = dst, no REX.W, the Go assembler's form), xmm→mem as
// 66 0F D6 (rm = xmm). Register forms against a GPR use the MOVD
// opcodes with REX.W instead: 66 REX.W 0F 6E (gpr→xmm) and
// 66 REX.W 0F 7E (xmm→gpr); the memory opcodes with a register r/m
@@ -103,7 +103,7 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
switch src := src.(type) {
case Reg:
if dstIsReg {
// MOV r/m, r: 0x88/0x89, reg=src, rm=dst — the form the Go
// MOV r/m, r: 0x88/0x89, reg=src, rm=dst, the form the Go
// assembler emits for register-to-register moves.
i := newInstr(size, []byte{movRM(size)})
if err := setRM(i, src, dst, size); err != nil {
@@ -147,7 +147,7 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
// a signed int32, choosing per sign:
// v >= 0: B8+rd imm32 without REX.W (zero-extended by the
// hardware, REX.B still emitted for R8-R15);
// v < 0: REX.W C7 /0 imm32 (sign-extended — the plain B8+rd
// v < 0: REX.W C7 /0 imm32 (sign-extended, the plain B8+rd
// form would zero-extend and corrupt the value).
// Out-of-range immediates keep the B8+rd imm64 form.
if size == 8 && v >= 0 && v <= (1<<31)-1 {
@@ -226,8 +226,8 @@ func (e *enc) encodeALU(op struct {
return e.encodeALUImm(op.digit, dst, int64(imm), size)
}
// CMP accepts the immediate in the second position too — CMPL CX, $31 is
// the form the Go assembler itself accepts — and encodes it identically
// CMP accepts the immediate in the second position too, CMPL CX, $31 is
// the form the Go assembler itself accepts, and encodes it identically
// (CMP r/m, imm sets the flags as first − second). No other ALU op takes
// an immediate destination.
if imm, ok := dst.(Imm); ok {
@@ -313,7 +313,7 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
i.imm = []byte{byte(int8(imm))}
return e.emit(i)
}
// 0x81 /digit, imm16/imm32 — or the Go assembler's accumulator short
// 0x81 /digit, imm16/imm32, or the Go assembler's accumulator short
// form (opcode+5, no ModR/M) when the destination is AX/AL, which it
// prefers over the generic form exactly here.
if r, ok := dst.(Reg); ok && r.idx == 0 {
@@ -339,7 +339,7 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
}
src, dst := ops[0], ops[1]
if imm, ok := src.(Imm); ok {
// TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0 — but the Go assembler
// TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0, but the Go assembler
// always uses the accumulator forms (A8/A9, no ModR/M) when the
// register operand is AL/AX, whatever the immediate's width.
if r, ok := dst.(Reg); ok && r.idx == 0 {
@@ -678,7 +678,7 @@ func (e *enc) encodeCmov(upper string, ops []Operand) error {
}
// encodeSet encodes a conditional byte set: SET + condition (SETNE, SETEQ, …),
// always a byte write — 0F 90+cc /0 into a register or memory operand.
// always a byte write, 0F 90+cc /0 into a register or memory operand.
func (e *enc) encodeSet(upper string, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("SETcc expects 1 operand, got %d", len(ops))
@@ -711,8 +711,8 @@ var countOp = map[string]struct {
"POPCNT": {0xB8, 0xF3},
}
// encodeCount encodes the bit-scan and bit-count family — BSF (0F BC),
// BSR (0F BD), TZCNT (F3 0F BC), LZCNT (F3 0F BD) and POPCNT (F3 0F B8) —
// encodeCount encodes the bit-scan and bit-count family, BSF (0F BC),
// BSR (0F BD), TZCNT (F3 0F BC), LZCNT (F3 0F BD) and POPCNT (F3 0F B8)
// with reg = dst and rm = src. The size suffix selects the operand width
// (BSFQ, TZCNTL, …). Note BSF/BSR leave the destination undefined when the
// source is zero (unlike their F3-prefixed counterparts); callers must
@@ -801,8 +801,8 @@ type sseMove struct {
}
var sseMoveTable = map[string]sseMove{
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU — unaligned octa
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA — aligned octa
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
+3 -3
View File
@@ -19,8 +19,8 @@ import (
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel —
// all functions plus the file-local mask24 constant — and checks that every
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel;
// all functions plus the file-local mask24 constant; and checks that every
// static-symbol load resolves to the right bytes in the image.
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx2_amd64.s"
@@ -81,7 +81,7 @@ func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
}
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
// kernel — all functions plus the file-global idx16 constant — and checks
// kernel, all functions plus the file-global idx16 constant, and checks
// that the static-symbol load resolves to the right bytes in the image.
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx512_amd64.s"
+2 -2
View File
@@ -94,9 +94,9 @@ DATA ·table<>+0(SB)/8, $0x1122334455667788
}
// The debug_line program: LNE_set_address (the R_ADDR relocation
// carries the function address), then one row per line change — the
// carries the function address), then one row per line change; the
// TEXT is on line 4 (a leading blank line precedes the include), the
// instructions on lines 5–9 — an advance to the 20-byte end and an
// instructions on lines 5-9; an advance to the 20-byte end and an
// end-of-sequence.
linesOff := le.Uint32(dataIdx[4*2:])
lines := dataBlk[linesOff : linesOff+21]
+37 -11
View File
@@ -6,6 +6,7 @@ package asm
import (
"fmt"
"sort"
"strconv"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
@@ -15,7 +16,7 @@ import (
// file-local static symbols are encoded RIP-relative and resolved within the
// image, so the raw bytes are self-consistent and executable at any base
// address; references to external symbols are recorded as relocations
// (Funcs[i].Relocs, Externals) and left unresolved — the object-file
// (Funcs[i].Relocs, Externals) and left unresolved, the object-file
// emitters turn them into linker relocations.
type Image struct {
Code []byte // concatenated function bodies
@@ -90,14 +91,18 @@ type RelocKind int
const (
RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64)
RelCall // R_CALL: CALL to a function symbol (amd64)
RelTLSLE // R_TLS_LE: local-exec TLS load, no symbol (amd64 guard)
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
RelRISCVJal // R_RISCV_JAL (J-type call)
RelPCRelAbs // 32-bit absolute (R_RISCV_32)
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
RelArm64Addr // R_ADDRARM64 (ADRP + ADD/LDR/STR pair)
RelArm64Addr // R_ADDRARM64 (ADRP + ADD pair)
RelArm64Branch // R_CALLARM64 (BL instruction)
RelArm64LDST64 // R_ARM64_PCREL_LDST64 (ADRP + 64-bit LDR/STR pair)
RelLoong64Branch // R_CALLLOONG64 (BL instruction)
)
type Reloc struct {
@@ -132,7 +137,7 @@ func (img *Image) Bytes() []byte {
// reference to a file-local static symbol becomes a RIP-relative load whose
// displacement is resolved against that layout; a reference to a symbol no
// GLOBL defines is recorded as an external relocation (Externals) with its
// displacement left zero — the object-file emitters resolve it at link
// displacement left zero, the object-file emitters resolve it at link
// time, while the raw image (Bytes) cannot represent it.
func AssembleFile(f *ast.File) (*Image, error) {
dataSyms, err := collectData(f)
@@ -146,6 +151,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
link := &linkInfo{symbols: known, allowExternal: true}
img := &Image{Symbols: map[string]int{}}
textOff := map[string]int{}
type asmFunc struct {
name string
patches []sbPatch
@@ -183,6 +189,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
for _, s := range steps {
fl.Spadj = append(fl.Spadj, SpadjStep{PC: s.pc, Value: s.value})
}
textOff[t.Name.Name] = len(img.Code)
img.Funcs = append(img.Funcs, fl)
img.Code = append(img.Code, code...)
funcs = append(funcs, asmFunc{name: t.Name.Name, patches: patches})
@@ -215,13 +222,27 @@ func AssembleFile(f *ast.File) (*Image, error) {
base := img.Funcs[i].Offset
code := img.Code[base : base+img.Funcs[i].Size]
for _, p := range fn.patches {
reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend}
reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend, Kind: p.kind}
if p.kind == RelTLSLE {
// The TLS slot has no symbol: the linker fills the offset
// from the runtime's TLS layout.
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
continue
}
if imgOff, ok := img.Symbols[p.name]; ok {
rel := int64(imgOff) + p.addend - int64(base+p.after)
if rel < -1<<31 || rel >= 1<<31 {
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
}
copy(code[p.off:p.off+4], le32(rel))
} else if imgOff, ok := textOff[p.name]; ok {
// A CALL to a TEXT function of the same file: resolve the
// displacement against the function's layout position.
rel := int64(imgOff) + p.addend - int64(base+p.after)
if rel < -1<<31 || rel >= 1<<31 {
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
}
copy(code[p.off:p.off+4], le32(rel))
} else {
reloc.External = true
externals[p.name] = true
@@ -451,13 +472,18 @@ func collectData(f *ast.File) ([]dataSym, error) {
ds.rodata = true
case "DUPOK":
ds.dupok = true
case "1":
ds.dupok = true
case "8":
ds.rodata = true
case "9":
ds.dupok = true
ds.rodata = true
default:
// Legacy numeric flag constants (runtime/textflag.h):
// DUPOK is 2, RODATA is 8; combinations arrive as one
// number (e.g. 10 = RODATA|DUPOK).
if n, err := strconv.Atoi(f); err == nil {
if n&2 != 0 {
ds.dupok = true
}
if n&8 != 0 {
ds.rodata = true
}
}
}
}
syms = append(syms, ds)
+40 -2
View File
@@ -10,8 +10,8 @@ import (
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestAssembleFileStaticData checks the whole-image layout — code, padding
// and the data section — and that the RIP-relative displacements of static
// TestAssembleFileStaticData checks the whole-image layout; code, padding
// and the data section; and that the RIP-relative displacements of static
// symbol loads resolve to the right bytes.
func TestAssembleFileStaticData(t *testing.T) {
f, errs := parser.Parse("d_amd64.s", `
@@ -128,3 +128,41 @@ DATA x<>+0(SB)/4, $1
t.Errorf("single-function SB: error %v, want a file-level-assembly error", err)
}
}
// TestCollectDataNumericFlags pins the numeric GLOBL flag constants from
// runtime/textflag.h: DUPOK is 2, RODATA is 8, and combinations arrive as
// one number (9 = NOPROF|RODATA, 10 = RODATA|DUPOK).
func TestCollectDataNumericFlags(t *testing.T) {
tests := []struct {
flags string
rodata bool
dupok bool
}{
{"2", false, true},
{"8", true, false},
{"9", true, false}, // NOPROF|RODATA, not DUPOK
{"10", true, true}, // RODATA|DUPOK
{"RODATA", true, false},
{"DUPOK", false, true},
{"RODATA|DUPOK", true, true},
}
for _, tt := range tests {
src := "TEXT \u00b7f(SB), NOSPLIT, $0\n\tRET\nGLOBL sym(SB), " + tt.flags + ", $8\n"
f, errs := parser.Parse("f_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse %q: %v", tt.flags, errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble %q: %v", tt.flags, err)
}
if len(img.DataSyms) != 1 {
t.Fatalf("%q: data syms = %d, want 1", tt.flags, len(img.DataSyms))
}
d := img.DataSyms[0]
if d.Rodata != tt.rodata || d.Dupok != tt.dupok {
t.Errorf("flags %q: rodata=%v dupok=%v, want rodata=%v dupok=%v",
tt.flags, d.Rodata, d.Dupok, tt.rodata, tt.dupok)
}
}
}
+66 -28
View File
@@ -13,17 +13,19 @@ import (
// assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into
// machine code. Every instruction is 4 bytes; the MOV pseudo-instruction and
// the immediate-arithmetic forms expand to 2–5 instructions when the
// the immediate-arithmetic forms expand to 2-5 instructions when the
// immediate does not fit, so the layout is computed in two passes (sizes,
// then encoding with resolved branch targets).
//
// The emitted bytes match the Go toolchain's loong64 assembler, which is the
// ground-truth oracle: prologue/epilogue, FP/SP frame mapping, branch
// encodings and the MOV immediate expansions all follow cmd/internal/obj/
// loong64's asmout cases.
// ground-truth oracle: prologue/epilogue (including the large-frame R30
// materialisations), FP/SP frame mapping, the stack-split guard classes, and
// branch encodings all follow cmd/internal/obj/loong64. The morestack block
// at the end of split functions carries the runtime.morestack_noctxt call.
func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
fi := loong64ComputeFrame(t)
prologue := loong64Prologue(fi)
guardLen := loong64GuardLen(fi)
chain := loong64JumpChain(t)
resolve := func(name string) string {
if r, ok := chain[name]; ok {
@@ -35,16 +37,17 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
var relocs []Reloc
var spadj []SpadjStep
// The prologue (3 instructions when a frame is present) raises the SP
// delta by autosize; the boundary is reported at the third instruction's
// pc, exactly as the toolchain's pctospadj does.
// The prologue raises the SP delta by autosize; the boundary is reported
// after the SP adjust instruction, exactly as the toolchain's pctospadj
// does. The prologue (3 instructions when a frame is present) may
// materialise its store or adjust through R30, which widens it.
if fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: 8, Value: fi.autosize})
spadj = append(spadj, SpadjStep{PC: guardLen + (loong64StoreWords(fi.autosize)+loong64AdjustWords(-int64(fi.autosize)))*4, Value: fi.autosize})
}
// Pass 1: label offsets from the instruction sizes.
offsets := map[string]int{}
pos := len(prologue)
pos := guardLen + len(prologue)
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
@@ -54,9 +57,25 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
}
}
// Pass 2: encode. Relocation offsets are recorded function-relative.
out := append([]byte(nil), prologue...)
pc := len(prologue)
// Pass 2: encode. The guard prefix precedes the prologue; its branches
// target the morestack block at the end of the function, which the first
// pass has sized.
bodyLen := 0
{
p := guardLen + len(prologue)
for _, stmt := range t.Body {
if in, ok := stmt.(*ast.Instr); ok {
p += loong64InstrSize(in, fi)
}
}
bodyLen = p - (guardLen + len(prologue))
}
var out []byte
if fi.needSplit {
out = append(out, loong64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...)
}
out = append(out, prologue...)
pc := guardLen + len(prologue)
preCount := len(relocs)
var lines []LineEntry
for _, stmt := range t.Body {
@@ -69,23 +88,29 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
}
for j := preCount; j < len(relocs); j++ {
relocs[j].Off += pc - len(prologue)
// Make the relocation offsets function-relative: each instruction
// records its reloc offset relative to its own start, and pc is
// that instruction's offset from the function start (prologue
// included). After shifts by the same amount.
relocs[j].Off += pc
relocs[j].After += pc
}
preCount = len(relocs)
lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line})
// The RET's epilogue closes the frame: the SP delta returns to zero
// after the addi.d (one instruction for a leaf, two for a non-leaf
// with the LR restore).
// after the frame-deallocating ADDV.
if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 {
epi := 4
if !fi.leaf {
epi = 8
}
spadj = append(spadj, SpadjStep{PC: pc + epi, Value: 0})
spadj = append(spadj, SpadjStep{PC: pc + loong64EpilogueWords(fi)*4, Value: 0})
}
out = append(out, code...)
pc += len(code)
}
if fi.needSplit {
block, blReloc := loong64MoreStackBlock(pc)
out = append(out, block...)
relocs = append(relocs, blReloc)
pc += len(block)
}
return out, offsets, relocs, lines, spadj, nil
}
@@ -224,9 +249,9 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
}
return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "JMP", "B":
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve)
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs)
case "JAL", "CALL", "BL":
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve)
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs)
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return encodeLOONG64Mov(instr, mnem, fi, relocs)
}
@@ -417,7 +442,7 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
case l64Firrr:
// ALSL: INSTR $sa, rj, rk, rd (the toolchain's optab places rj in
// the second register position); the source amount is 1–4, encoded
// the second register position); the source amount is 1-4, encoded
// as sa-1.
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
@@ -485,7 +510,7 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
//
// JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0
// JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string) ([]byte, error) {
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc) ([]byte, error) {
if len(instr.Operands) != 1 {
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands))
}
@@ -502,6 +527,19 @@ func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[stri
}
return l64wordLE(l64irr16(l64branchTable["JIRL"], 0, rj, rd)), nil
}
// Direct symbol: sym+off(SB) → b/bl with an R_CALLLOONG64 relocation
// (the linker fills the offset), as the toolchain does for CALL/BL/JAL
// and for tail-calling JMP.
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
opc := l64jumpTable["B"]
if link {
opc = l64jumpTable["BL"]
}
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelLoong64Branch, Addend: op.Addr.Sym.Offset})
}
return l64wordLE(l64bbl(opc, 0)), nil
}
// Direct: label → b/bl.
target := resolve(l64Label(op))
targetOff, ok := offsets[target]
@@ -578,8 +616,8 @@ func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, o
// encodeLOONG64Branch21 encodes a single-register branch: BLTZ/BGEZ and
// BFPT/BFPF use the 21-bit offset form (register in the rj field), while
// BGTZ/BLEZ — which the toolchain encodes with the register in the rd field
// and a 16-bit offset — are handled separately.
// BGTZ/BLEZ, which the toolchain encodes with the register in the rd field
// and a 16-bit offset, are handled separately.
func encodeLOONG64Branch21(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
@@ -692,7 +730,7 @@ func encodeLOONG64ImmArith(mnem string, de l64DualEnc, ops []*ast.Operand) ([]by
}
// isLoong64ShiftD reports whether a shift-immediate opcode constant is one of
// the 6-bit (.d) variants — the toolchain distinguishes them by the bit
// the 6-bit (.d) variants, the toolchain distinguishes them by the bit
// position of the opcode field (bits [25:16]).
func isLoong64ShiftD(op uint32) bool {
return op&0x03ff0000 != 0 && op>>25 == 0
@@ -740,7 +778,7 @@ func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off in
// ---- the MOV pseudo-instruction ----
// encodeLOONG64Mov encodes the MOV family — the load/store/immediate
// encodeLOONG64Mov encodes the MOV family, the load/store/immediate
// workhorse of Go's loong64 assembly. MOV is an alias of MOVV (the width
// mnemonics MOVB/MOVH/MOVW/MOVV/MOVBU/MOVHU/MOVWU/MOVF/MOVD select the
// access width). The forms, mirroring the toolchain:
+14 -14
View File
@@ -9,7 +9,7 @@ package asm
// an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own loong64
// backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm`
// exactly — the ground-truth oracle for the verify suite.
// exactly, the ground-truth oracle for the verify suite.
//
// All LoongArch instructions are 32 bits, little-endian. The formats used
// here (per the LoongArch Volume I specification):
@@ -33,8 +33,8 @@ package asm
import "maps"
// loong64RegNum returns the 5-bit register number for a LoongArch register
// name: R0–R31 (integer), F0–F31 (floating point), FCC0–FCC7 (condition
// flags), FCSR0–FCSR31 (control/status) and the ABI aliases the runtime's
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
// flags), FCSR0-FCSR31 (control/status) and the ABI aliases the runtime's
// assembly uses. Returns -1 for an unrecognised name.
func loong64RegNum(name string) int {
switch name {
@@ -103,7 +103,7 @@ func loong64RegNum(name string) int {
case "R31", "S8":
return 31
}
// F0–F31, FCC0–FCC7, FCSR0–FCSR31.
// F0-F31, FCC0-FCC7, FCSR0-FCSR31.
if len(name) >= 4 && name[:4] == "FCSR" {
return loong64RegSpecial(name[4:], 31)
}
@@ -199,7 +199,7 @@ func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
}
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
// The msb/lsb fields are 6 bits wide (0–63) and are validated by the caller.
// The msb/lsb fields are 6 bits wide (0-63) and are validated by the caller.
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
}
@@ -280,7 +280,7 @@ var l64DualTable = map[string]l64DualEnc{}
var l64InstrTable = map[string]l64Enc{}
func init() {
// 3R — integer.
// 3R, integer.
rrr := map[string]uint32{
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
@@ -300,7 +300,7 @@ func init() {
"CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15,
"CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15,
}
// 3R — floating point.
// 3R, floating point.
rrr["MULF"] = 0x209 << 15
rrr["MULD"] = 0x20a << 15
rrr["DIVF"] = 0x20d << 15
@@ -390,12 +390,12 @@ func init() {
"ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true},
})
// 2RI12 — pure immediate arithmetic (LU52ID has no register form).
// 2RI12, pure immediate arithmetic (LU52ID has no register form).
l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22}
// ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16.
l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26}
// 2RI14 — LL/SC are aliased by the Go assembler to the pointer loads and
// 2RI14, LL/SC are aliased by the Go assembler to the pointer loads and
// stores (ldptr/stptr), with the offset scaled by 4.
l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w
l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d
@@ -414,7 +414,7 @@ func init() {
// LUI is the Plan 9 spelling of lu12i.w.
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
// 4R — fused multiply-add.
// 4R, fused multiply-add.
rrrr := map[string]uint32{
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
@@ -425,7 +425,7 @@ func init() {
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
}
// IRIR — bit-field insert/extract.
// IRIR, bit-field insert/extract.
irir := map[string]uint32{
"BSTRINSW": 0x3<<21 | 0x0<<15,
"BSTRINSV": 0x2 << 22,
@@ -436,7 +436,7 @@ func init() {
l64InstrTable[m] = l64Enc{format: l64Firir, op: op}
}
// 3RI2 — ALSL.
// 3RI2, ALSL.
irrr := map[string]uint32{
"ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17,
}
@@ -452,7 +452,7 @@ func init() {
// PRELD.
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
// Atomics — 3R with the AM field order (rk=value, rj=address, rd=result).
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
am := map[string]uint32{
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
@@ -477,7 +477,7 @@ func init() {
}
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
// register move between the integer and floating-point register banks — the
// register move between the integer and floating-point register banks, the
// MOVW/MOVV specials the Go assembler accepts.
var l64FpMovTable = map[string]uint32{
"MOVV.R.F": 0x452a << 10, // movgr2fr.d
+223 -12
View File
@@ -20,14 +20,20 @@ import (
// (the toolchain aligns frames with `if autosize&4 != 0 { autosize += 4 }`).
// A leaf function (no calls) with a zero frame gets no prologue at all.
//
// Prologue (autosize > 0), byte-identical to the toolchain:
// Prologue (autosize > 0, small), byte-identical to the toolchain:
//
// MOVV R1, -autosize(R3) // save LR below the new SP (traceback-safe)
// ADDV $-autosize, R3 // open the frame
// MOVV R1, 0(R3) // save LR again at SP (signal-safety)
//
// Large frames (autosize past the 12-bit offset or immediate ranges) expand
// the store and the adjust through REGTMP (R30) exactly as the toolchain's
// assembler does: the store via the rounding LU12IW split, the adjust via
// the floor LU12IW/ORI split.
//
// Epilogue: MOVV 0(R3), R1; ADDV $autosize, R3 (non-leaf only for the LR
// restore); the RET's jirl r0, r1, 0 follows.
// restore; the adjust materialised when the immediate does not fit); the
// RET's jirl r0, r1, 0 follows.
// loong64FrameInfo holds the frame layout derived from a TEXT directive.
type loong64FrameInfo struct {
@@ -36,6 +42,11 @@ type loong64FrameInfo struct {
args int // the declared -argsize
noSplit bool // the NOSPLIT flag
leaf bool // no call instructions in the body
// Stack-split guard state: like amd64 and arm64, a leaf function with a
// small autosize is auto-marked NOSPLIT by the toolchain.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
}
// loong64ComputeFrame derives the frame layout for a TEXT function.
@@ -59,9 +70,157 @@ func loong64ComputeFrame(t *ast.Text) loong64FrameInfo {
// A zero-frame non-leaf function still opens an 8-byte frame for LR.
fi.autosize = 8
}
switch {
case fi.noSplit:
case fi.autosize < stackSmall && fi.leaf:
// Auto-NOSPLIT, as the toolchain's leaf mark concludes.
default:
fi.needSplit = true
switch {
case fi.autosize <= stackSmall:
fi.splitClass = 0
case fi.autosize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi
}
// loong64GuardLen returns the byte length of the stack-split guard prefix
// (zero when the function needs no guard). The big class materialises two
// constants through R30; each materialisation shrinks by one word when the
// constant's low 12 bits are zero.
func loong64GuardLen(fi loong64FrameInfo) int {
if !fi.needSplit {
return 0
}
off := int64(fi.autosize - stackSmall)
switch fi.splitClass {
case 0:
return 12
case 1:
if off <= 2048 {
return 16 // ADDV $-off fits the signed 12-bit immediate
}
return 24 // MOVV + LU12IW + ORI + ADDV + SGTU + BEQ
default:
// MOVV + [mat] + SGTU + BNE + [mat] + ADDV + SGTU + BEQ
return (6 + loong64MatLen(off) + loong64MatLen(-off)) * 4
}
}
// loong64MatLen reports the word count of materialising v in R30: a value
// with a zero high part needs only the ORI (the toolchain's MOVW $v, R30),
// one with a zero low part only the LU12IW.
func loong64MatLen(v int64) int {
if v>>12 == 0 || v&0xFFF == 0 {
return 1
}
return 2
}
// loong64MatWords appends the words that materialise v in R30, splitting it
// as v>>12 plus the zero-extended low 12 bits.
func loong64MatWords(ws []uint32, v int64) []uint32 {
hi := v >> 12
lo := v & 0xFFF
if hi == 0 {
return append(ws, l64irr(l64OriOp, int(v), 0, 30))
}
ws = append(ws, l64ir(l64Lu12iwOp, int(hi), 30))
if lo != 0 {
ws = append(ws, l64irr(l64OriOp, int(lo), 30, 30))
}
return ws
}
// The LU12IW and ORI opcode bases (2RI20 and 2RI12 formats); the ORI reads
// and writes rd itself.
const (
l64Lu12iwOp = 0x0a << 25
l64OriOp = 0x0e << 22
)
// loong64Imm12 reports whether v fits a signed 12-bit immediate.
func loong64Imm12(v int64) bool { return v >= -2048 && v <= 2047 }
// loong64GuardBytes emits the stack-split guard prefix. blockStart is the
// function-relative address of the morestack call at the end of the function;
// branch displacements are in instructions and are computed from each
// branch's own position.
func loong64GuardBytes(fi loong64FrameInfo, blockStart int) []byte {
// MOVV 16(g), R20 (g.stackguard0), g = R22.
ws := []uint32{l64irr(l64loadStoreTable["MOVV"].ld, 16, 22, 20)}
off := int64(fi.autosize - stackSmall)
// beq appends BEQ R20, blockStart from the branch's own position.
beq := func() {
ws = append(ws, loong64Beqz(20, int32((blockStart-len(ws)*4)>>2)))
}
switch fi.splitClass {
case 0:
// SGTU SP, R20, R20; BEQ R20, more
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 3, 20, 20))
beq()
case 1:
ws = append(ws, loong64MediumWords(off)...)
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 24, 20, 20))
beq()
default:
// SGTU $off, SP, R24 catches the SP underflow a huge frame would
// cause; BNE jumps to morestack in that case.
ws = append(ws, loong64MatWords(nil, off)...)
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 30, 3, 24))
ws = append(ws, loong64Bnez(24, int32((blockStart-len(ws)*4)>>2)))
ws = append(ws, loong64MatWords(nil, -off)...)
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 24))
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 24, 20, 20))
beq()
}
return l64WordsLE(ws...)
}
// loong64MediumWords emits the medium-class stack check for offset off: the
// ADDV immediate when it fits, otherwise the same sequence with the constant
// materialised in R30.
func loong64MediumWords(off int64) []uint32 {
if off <= 2048 {
return []uint32{l64irr(l64DualTable["ADDV"].imm, int(-off), 3, 24)}
}
ws := loong64MatWords(nil, -off)
return append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 24))
}
// loong64Beqz/loong64Bnez build the 21-bit conditional branches against R0
// that the toolchain emits for its guard compares.
func loong64Beqz(rj int, dispInstr int32) uint32 {
return l64ir21(l64branch21Table["BEQZ"], int(dispInstr), rj)
}
func loong64Bnez(rj int, dispInstr int32) uint32 {
return l64ir21(l64branch21Table["BNEZ"], int(dispInstr), rj)
}
// loong64MoreStackBlock emits the trailing block: MOVV R1, R31 (save LR, the
// toolchain's OR R1, R0, R31 expansion), BL runtime.morestack_noctxt, B back
// to the function entry.
func loong64MoreStackBlock(blockStart int) ([]byte, Reloc) {
ws := []uint32{
l64rrr(l64DualTable["OR"].rrr, 0, 1, 31), // MOVV R1, R31 (OR R1, R0, R31)
l64bbl(l64jumpTable["BL"], 0), // BL, patched by the linker
}
disp := (-(blockStart + 8)) >> 2
ws = append(ws, l64bbl(l64jumpTable["B"], int(disp)))
reloc := Reloc{
Off: blockStart + 4,
After: blockStart + 8,
Name: "runtime\u00b7morestack_noctxt",
Kind: RelLoong64Branch,
}
return l64WordsLE(ws...), reloc
}
// loong64IsLeaf reports whether a function contains no call instructions
// (JAL/BL/CALL), matching the toolchain's LEAF mark, which drives the frame
// and the epilogue shape.
@@ -79,17 +238,37 @@ func loong64IsLeaf(t *ast.Text) bool {
return true
}
// loong64Prologue returns the prologue bytes for a loong64 function.
// loong64Prologue returns the prologue bytes for a loong64 function. When
// the LR store offset leaves the toolchain's 12-bit store range ([-2046,
// 2045], BIG_12 = 2046) or the SP adjust immediate its 12-bit immediate
// range, each switches to the R30 materialisation the assembler expands it
// to: the store uses the rounding %hi/%lo split (LU12IW of (v+2048)>>12,
// REGTMP += SP, store at the raw offset), the adjust the floor split
// (LU12IW, ORI when the low part is non-zero, REGTMP += SP).
func loong64Prologue(fi loong64FrameInfo) []byte {
if fi.autosize == 0 {
return nil
}
addiD := l64DualTable["ADDV"].imm
return l64WordsLE(
l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, 3, 1), // MOVV R1, -autosize(R3)
l64irr(addiD, -fi.autosize, 3, 3), // ADDV $-autosize, R3
l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1), // MOVV R1, 0(R3)
)
var ws []uint32
storeBase := 3
if fi.autosize > 2046 {
// The store goes through REGTMP: LU12IW of the rounding split,
// REGTMP += SP, then the store at REGTMP with the truncated offset.
v := -int64(fi.autosize)
ws = append(ws, l64ir(l64Lu12iwOp, int((v+2048)>>12), 30))
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 3, 30, 30))
storeBase = 30
}
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, storeBase, 1)) // MOVV R1, -autosize(base)
if loong64Imm12(-int64(fi.autosize)) {
ws = append(ws, l64irr(addiD, -fi.autosize, 3, 3)) // ADDV $-autosize, R3
} else {
ws = append(ws, loong64MatWords(nil, -int64(fi.autosize))...)
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 3))
}
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1)) // MOVV R1, 0(R3)
return l64WordsLE(ws...)
}
// loong64Return returns the bytes for a RET: the epilogue (restore LR and
@@ -98,17 +277,49 @@ func loong64Return(fi loong64FrameInfo) []byte {
var ws []uint32
if fi.autosize != 0 {
if !fi.leaf {
// MOVV 0(R3), R1 — restore the link register.
// MOVV 0(R3), R1, restore the link register.
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1))
}
// ADDV $autosize, R3 — close the frame.
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
// ADDV $autosize, R3, close the frame (materialised when the
// immediate does not fit).
if loong64Imm12(int64(fi.autosize)) {
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
} else {
ws = append(ws, loong64MatWords(nil, int64(fi.autosize))...)
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 3))
}
}
// jirl r0, r1, 0 — return.
// jirl r0, r1, 0, return.
ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0))
return l64WordsLE(ws...)
}
// loong64StoreWords reports the prologue word count of the LR store, and
// loong64AdjustWords the word count of an SP adjust of v: the immediate
// forms when they fit, otherwise the R30 materialisation sequences.
func loong64StoreWords(autosize int) int {
if autosize > 2046 {
return 3
}
return 1
}
func loong64AdjustWords(v int64) int {
if loong64Imm12(v) {
return 1
}
return loong64MatLen(v) + 1
}
// loong64EpilogueWords reports the epilogue word count the RET expands to.
func loong64EpilogueWords(fi loong64FrameInfo) int {
n := loong64AdjustWords(int64(fi.autosize))
if !fi.leaf {
n++
}
return n
}
// loong64ResolvePseudo translates a pseudo-register memory reference into a
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
// x-N(SP) → (autosize - N)(SP). Returns base = -1 for an unresolvable
+1 -1
View File
@@ -280,7 +280,7 @@ done:
// TestLOONG64_pcsp checks the stack-adjustment table of a framed function:
// the prologue raises the SP delta by autosize (in effect from the third
// instruction) and the RET's epilogue restores it to zero, with the pc deltas
// in MinLC (4) units — byte-identical to `go tool asm`.
// in MinLC (4) units; byte-identical to `go tool asm`.
func TestLOONG64_pcsp(t *testing.T) {
cases := []struct {
name string
+42
View File
@@ -0,0 +1,42 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestLOONG64RelocOffsetsIncludePrologue pins the function-relative
// relocation offsets of a framed loong64 function: the offsets used to
// exclude the prologue, so every relocation landed on a prologue
// instruction in the GOOBJ/ELF output.
func TestLOONG64RelocOffsetsIncludePrologue(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", "TEXT \u00b7f(SB), $16-0\n"+
"\tMOVV $gdata(SB), R4\n"+
"\tRET\n"+
"GLOBL gdata(SB), $8\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
fn := img.Funcs[0]
// Layout: 12-byte prologue (autosize 32), pcalau12i+addi.d (12, 16),
// epilogue with RET.
if len(fn.Relocs) != 2 {
t.Fatalf("relocs = %d, want 2", len(fn.Relocs))
}
hi, lo := fn.Relocs[0], fn.Relocs[1]
if hi.Kind != RelLoong64AddrHi || hi.Off != 12 || hi.After != 12 {
t.Errorf("hi reloc = {off %d after %d kind %d}, want {off 12 after 12 kind RelLoong64AddrHi}", hi.Off, hi.After, hi.Kind)
}
if lo.Kind != RelLoong64AddrLo || lo.Off != 16 || lo.After != 16 {
t.Errorf("lo reloc = {off %d after %d kind %d}, want {off 16 after 16 kind RelLoong64AddrLo}", lo.Off, lo.After, lo.Kind)
}
}
+9 -9
View File
@@ -12,33 +12,33 @@ import "maps"
import "strings"
// Reg is an x86-64 register. In Plan 9 assembly the classic names (AX, BX, …)
// are size-agnostic — the instruction suffix (MOVQ vs MOVL) fixes the width —
// are size-agnostic, the instruction suffix (MOVQ vs MOVL) fixes the width
// so the encoder keys off the register's index and lets the mnemonic supply the
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4–7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0–K7.
// registers K0-K7.
type Reg struct {
idx int
size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH
mask bool // K0–K7 opmask register
mask bool // K0-K7 opmask register
}
// Index returns the register number (0–15 for GPRs, 0–31 for vectors).
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
func (r Reg) Index() int { return r.idx }
// Size returns the width in bytes implied by the register's name.
func (r Reg) Size() int { return r.size }
// IsMask reports whether r is an AVX-512 opmask register (K0–K7).
// IsMask reports whether r is an AVX-512 opmask register (K0-K7).
func (r Reg) IsMask() bool { return r.mask }
func (r Reg) isOperand() {}
// needsREX reports whether this register forces a REX prefix at the given
// operand size: the extended registers R8–R15 always do, and at byte size the
// low registers SPL/BPL/SIL/DIL (indices 4–7, not high) do as well.
// operand size: the extended registers R8-R15 always do, and at byte size the
// low registers SPL/BPL/SIL/DIL (indices 4-7, not high) do as well.
func (r Reg) needsREX(opSize int) bool {
if r.idx >= 8 {
return true
@@ -133,7 +133,7 @@ func buildRegByName() map[string]Reg {
}
// Vector: X0..X31 (128-bit, size 16), Y0..Y31 (256-bit, size 32),
// Z0..Z31 (512-bit, size 64). Indices 16–31 are only encodable in EVEX
// Z0..Z31 (512-bit, size 64). Indices 16-31 are only encodable in EVEX
// (AVX-512) instructions; the encoder validates that through its tables.
for i := 0; i <= 31; i++ {
m["X"+itoa(i)] = Reg{idx: i, size: 16}
+85 -20
View File
@@ -15,14 +15,16 @@ import (
func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
fi := riscvComputeFrame(t)
prologue := riscvPrologue(fi)
guardLen := riscvGuardLen(fi)
var relocs []Reloc
var spadj []SpadjStep
// The prologue raises the SP delta by autosize; the boundary is reported
// at the pc just past its ADDI, exactly as the toolchain's pctospadj does.
// The guard prefix shifts its PC.
if fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: riscvPrologueSpadjPC(fi), Value: fi.autosize})
spadj = append(spadj, SpadjStep{PC: guardLen + riscvPrologueSpadjPC(fi), Value: fi.autosize})
}
// Pass 1: collect instructions and compute label offsets assuming 4 bytes
@@ -34,7 +36,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
}
var recs []instrRec
offsets := map[string]int{}
pos := len(prologue)
pos := guardLen + len(prologue)
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
@@ -66,7 +68,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
// Pass 4: recompute offsets with actual sizes.
offsets = map[string]int{}
pos = len(prologue)
pos = guardLen + len(prologue)
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
@@ -82,9 +84,17 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
}
// Pass 5: re-encode branches with corrected offsets. Record relocations
// during this final pass (relocation offsets are relative to instruction start).
out := append([]byte(nil), prologue...)
pc = len(prologue)
// during this final pass (relocation offsets are relative to instruction
// start). The guard prefix precedes the prologue; its branches target
// the morestack block at the end of the function, which the previous
// passes have sized.
var out []byte
guardBytes, guardReloc := riscvGuard(fi)
if fi.needSplit {
out = append(out, guardBytes...)
}
out = append(out, prologue...)
pc = guardLen + len(prologue)
preCount := len(relocs)
var lines []LineEntry
for _, r := range recs {
@@ -119,6 +129,9 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
pc += len(code)
}
}
if fi.needSplit {
relocs = append(relocs, guardReloc)
}
return out, offsets, relocs, lines, spadj, nil
}
@@ -149,6 +162,14 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
if isImmOperand(ops[0]) && ops[0].Imm.Sym == nil {
return riscvMovImmSize(regFromOperand(ops[1]), immFromOperand(ops[0]))
}
// Frame-relative loads and stores: a frame offset beyond the signed
// 12-bit range materialises the address in X31 first.
if isMemOperand(ops[0]) && !isMemOperand(ops[1]) {
return riscvFrameMemSize(ops[0], fi)
}
if isMemOperand(ops[1]) && !isMemOperand(ops[0]) {
return riscvFrameMemSize(ops[1], fi)
}
}
// I-type arithmetic with a large immediate expands to several instructions.
if (mnem == "ADDI" || mnem == "ANDI" || mnem == "ORI" || mnem == "XORI") && len(ops) >= 1 && isImmOperand(ops[0]) {
@@ -192,13 +213,21 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelRISCVJal, Addend: op.Addr.Sym.Offset})
}
word = riscvJType(1, 0) // JAL X1, 0 — the linker fills the offset
word = riscvJType(1, 0) // JAL X1, 0, the linker fills the offset
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
case "JMP":
// JMP = JAL X0, target. The Go assembler never compresses this to
// C.J, so always emit the 32-bit JAL.
var target string
if len(ops) >= 1 {
// JMP sym(SB): a tail call, JAL X0 against a symbol relocation.
if ops[0].Addr.Sym != nil && ops[0].Addr.Sym.Pseudo == "SB" {
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: ops[0].Addr.Sym.Name, Kind: RelRISCVJal, Addend: ops[0].Addr.Sym.Offset})
}
word = riscvJType(0, 0)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
target = labelFromOperand(ops[0])
}
targetOff, ok := offsets[target]
@@ -395,7 +424,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
}
word = riscvSType(enc, rs1, rs2, imm)
// LR (load-reserved): INSTR (addr), dst — 2 operands.
// LR (load-reserved): INSTR (addr), dst, 2 operands.
case len(ops) == 2 && isLRInstr(mnem):
rs1, _ := memFromOperandWithFrame(ops[0], fi)
rd := regFromOperand(ops[1])
@@ -404,7 +433,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
}
word = riscvAMOType(enc, rd, rs1, 0) // rs2=0 for LR
// SC (store-conditional): INSTR src, (addr), dst — 3 operands.
// SC (store-conditional): INSTR src, (addr), dst, 3 operands.
case len(ops) == 3 && isSCInstr(mnem):
rs2 := regFromOperand(ops[0])
rs1, _ := memFromOperandWithFrame(ops[1], fi)
@@ -443,7 +472,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
}
return encodeRISCVItypeImmediate(mnem, enc, rd, rd, imm)
// Loads: rd, offset(rs1) — Plan 9 order is LD src, dst.
// Loads: rd, offset(rs1), Plan 9 order is LD src, dst.
case len(ops) == 2 && isLoadInstr(mnem):
rd := regFromOperand(ops[1]) // destination (last operand)
rs1, imm := memFromOperandWithFrame(ops[0], fi) // memory source (first operand)
@@ -538,7 +567,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
// Immediate → register.
if isImmOperand(src) {
// MOV $sym(SB), rd — load address of a static symbol or external.
// MOV $sym(SB), rd, load address of a static symbol or external.
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
rd := regFromOperand(dst)
if rd < 0 {
@@ -546,7 +575,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
}
return encodeRISCVSBAddr(src.Imm.Sym, rd, relocs), nil
}
// MOV $sym(FP/SP), rd — not supported: immediate symbol references
// MOV $sym(FP/SP), rd, not supported: immediate symbol references
// other than SB cannot be encoded as a simple immediate.
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo != "" {
return nil, fmt.Errorf("MOV $%s(%s): unsupported immediate symbol reference (only SB is supported)", src.Imm.Sym.Name, src.Imm.Sym.Pseudo)
@@ -562,7 +591,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
// Memory → register (load).
if isMemOperand(src) && !isMemOperand(dst) {
rd := regFromOperand(dst)
// MOV sym(SB), rd — load from static data.
// MOV sym(SB), rd, load from static data.
if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" {
if rd < 0 {
return nil, fmt.Errorf("MOV sym(SB): invalid destination register")
@@ -573,14 +602,13 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("MOV load: invalid operand")
}
word := riscvIType(riscvEnc{0x03, 0x3, 0x00}, rd, rs1, off)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
return riscvFrameMemOp(riscvEnc{0x03, 0x3, 0x00}, false, rd, rs1, off), nil
}
// Register → memory (store).
if !isMemOperand(src) && isMemOperand(dst) {
rs2 := regFromOperand(src)
// MOV rd, sym(SB) — store to static data.
// MOV rd, sym(SB), store to static data.
if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" {
if rs2 < 0 {
return nil, fmt.Errorf("MOV rd, sym(SB): invalid source register")
@@ -591,8 +619,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
if rs2 < 0 || rs1 < 0 {
return nil, fmt.Errorf("MOV store: invalid operand")
}
word := riscvSType(riscvEnc{0x23, 0x3, 0x00}, rs1, rs2, off)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
return riscvFrameMemOp(riscvEnc{0x23, 0x3, 0x00}, true, rs2, rs1, off), nil
}
// Register → register (ADDI $0, src, dst).
@@ -607,6 +634,44 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
}
}
// riscvFrameMemOp encodes a register-relative load (store=false, I-type
// width 0x03) or store (store=true, S-type width 0x23) of the 64-bit width
// at off(rs1). Offsets beyond the signed 12-bit range materialise the
// address in X31 first: LUI hi (the rounding split), then ADD X31, rs1,
// matching the toolchain's large-frame addressing; the access uses the
// sign-extended low part, which always fits.
func riscvFrameMemOp(enc riscvEnc, store bool, reg, rs1 int, off int32) []byte {
if fits12(off) {
var word uint32
if store {
word = riscvSType(enc, rs1, reg, off)
} else {
word = riscvIType(enc, reg, rs1, off)
}
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
}
lo := off - (splitHi(off) << 12)
out := riscvAddressInX31WithBase(off, rs1)
var word uint32
if store {
word = riscvSType(enc, 31, reg, lo)
} else {
word = riscvIType(enc, reg, 31, lo)
}
return append(out, wordLE(word)...)
}
// riscvFrameMemSize returns the encoded size of a frame-relative MOV for the
// layout pass: 4 bytes when the offset fits, otherwise the X31
// materialisation plus the access.
func riscvFrameMemSize(op *ast.Operand, fi riscvFrameInfo) int {
rs1, off := memFromOperandWithFrame(op, fi)
if fits12(off) {
return 4
}
return len(riscvAddressInX31WithBase(off, rs1)) + 4
}
// encodeRISCVLoadImm encodes loading an immediate into a register (MOV $imm,
// rd), matching the toolchain's instructionsForMOVConst. For 12-bit
// immediates it emits ADDI $imm, ZERO, rd (compressed to C.LI when it fits
@@ -1006,7 +1071,7 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
}
case "ADDW", "SUBW":
// C.ADDW (0x27,1) / C.SUBW (0x27,0) — CA-type, prime regs.
// C.ADDW (0x27,1) / C.SUBW (0x27,0), CA-type, prime regs.
if len(ops) == 3 {
funct2 := uint32(0x0)
if mnem == "ADDW" {
@@ -1314,7 +1379,7 @@ func suggestLabel(target string, offsets map[string]int) string {
}
// Only suggest if the distance is small enough.
if bestDist <= 3 && bestDist < len(target)/2+1 {
return fmt.Sprintf(" — did you mean %q?", best)
return fmt.Sprintf("; did you mean %q?", best)
}
return ""
}
+14 -14
View File
@@ -154,7 +154,7 @@ type riscvEnc struct {
// riscvInstrTable maps RISC-V mnemonics to their encoding.
var riscvInstrTable = map[string]riscvEnc{
// RV64I — R-type arithmetic/logic.
// RV64I, R-type arithmetic/logic.
"ADD": {0x33, 0x0, 0x00},
"SUB": {0x33, 0x0, 0x20},
"SLL": {0x33, 0x1, 0x00},
@@ -165,20 +165,20 @@ var riscvInstrTable = map[string]riscvEnc{
"SRA": {0x33, 0x5, 0x20},
"OR": {0x33, 0x6, 0x00},
"AND": {0x33, 0x7, 0x00},
// RV64I — 32-bit variants (W suffix).
// RV64I, 32-bit variants (W suffix).
"ADDW": {0x3B, 0x0, 0x00},
"SUBW": {0x3B, 0x0, 0x20},
"SLLW": {0x3B, 0x1, 0x00},
"SRLW": {0x3B, 0x5, 0x00},
"SRAW": {0x3B, 0x5, 0x20},
// RV64I — I-type shift-immediate (shamt in rs2 field).
// RV64I, I-type shift-immediate (shamt in rs2 field).
"SLLI": {0x13, 0x1, 0x00},
"SRLI": {0x13, 0x5, 0x00},
"SRAI": {0x13, 0x5, 0x20},
"SLLIW": {0x1B, 0x1, 0x00},
"SRLIW": {0x1B, 0x5, 0x00},
"SRAIW": {0x1B, 0x5, 0x20},
// RV64M — multiply/divide.
// RV64M, multiply/divide.
"MUL": {0x33, 0x0, 0x01},
"MULH": {0x33, 0x1, 0x01},
"MULHSU": {0x33, 0x2, 0x01},
@@ -187,13 +187,13 @@ var riscvInstrTable = map[string]riscvEnc{
"DIVU": {0x33, 0x5, 0x01},
"REM": {0x33, 0x6, 0x01},
"REMU": {0x33, 0x7, 0x01},
// RV64M — 32-bit variants.
// RV64M, 32-bit variants.
"MULW": {0x3B, 0x0, 0x01},
"DIVW": {0x3B, 0x4, 0x01},
"DIVUW": {0x3B, 0x5, 0x01},
"REMW": {0x3B, 0x6, 0x01},
"REMUW": {0x3B, 0x7, 0x01},
// RV64I — I-type arithmetic.
// RV64I, I-type arithmetic.
"ADDI": {0x13, 0x0, 0x00},
"ADDIW": {0x1B, 0x0, 0x00},
"SLTI": {0x13, 0x2, 0x00},
@@ -228,10 +228,10 @@ var riscvInstrTable = map[string]riscvEnc{
"ECALL": {0x73, 0x0, 0x00},
"EBREAK": {0x73, 0x0, 0x00},
"FENCE": {0x0F, 0x0, 0x00},
// JALR — indirect jump/call (I-type).
// JALR, indirect jump/call (I-type).
"JALR": {0x67, 0x0, 0x00},
// RV64A — atomics (AMO opcode 0x2F).
// RV64A, atomics (AMO opcode 0x2F).
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
@@ -252,7 +252,7 @@ var riscvInstrTable = map[string]riscvEnc{
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
// RV64F/D — floating-point arithmetic.
// RV64F/D, floating-point arithmetic.
"FADDS": {0x53, 0x0, 0x00},
"FSUBS": {0x53, 0x0, 0x04},
"FMULS": {0x53, 0x0, 0x08},
@@ -274,13 +274,13 @@ var riscvInstrTable = map[string]riscvEnc{
"FMIND": {0x53, 0x0, 0x15},
"FMAXD": {0x53, 0x1, 0x15},
// RV64A — load-reserved / store-conditional (funct5 0x02 / 0x03).
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
"LRW": {0x2F, 0x2, 0x02 << 2},
"LRD": {0x2F, 0x3, 0x02 << 2},
"SCW": {0x2F, 0x2, 0x03 << 2},
"SCD": {0x2F, 0x3, 0x03 << 2},
// FP compare — result in integer register (funct7 0x50/0x51).
// FP compare, result in integer register (funct7 0x50/0x51).
"FEQS": {0x53, 0x2, 0x50},
"FLTS": {0x53, 0x1, 0x50},
"FLES": {0x53, 0x0, 0x50},
@@ -442,10 +442,10 @@ func riscvJType(rd int, offset int32) uint32 {
// ---- RVC (compressed) encoding helpers ----
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
// prime register field used by compressed instructions (x8–x15).
// prime register field used by compressed instructions (x8-x15).
func isRVCIntReg(r int) bool { return r >= 8 && r <= 15 }
// rvcReg3 returns the 3-bit encoding for registers x8–x15 (0–7).
// rvcReg3 returns the 3-bit encoding for registers x8-x15 (0-7).
func rvcReg3(r int) uint32 { return uint32(r - 8) }
// rvcCR encodes a CR-type (register) compressed instruction.
@@ -455,7 +455,7 @@ func rvcCR(funct4, rd, rs2 uint32) uint16 {
}
// rvcCI encodes a CI-type (immediate) compressed instruction.
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW — linear 6-bit immediate.
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW, linear 6-bit immediate.
func rvcCI(funct3, rd uint32, imm uint32) uint16 {
return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x1)
}
+1 -1
View File
@@ -289,7 +289,7 @@ TEXT ·cmp(SB), NOSPLIT, $0
}
func TestRISCV_forwardBranch(t *testing.T) {
// Forward label reference — must not fail.
// Forward label reference; must not fail.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·fwd(SB), NOSPLIT, $0
ADDI $1, X10, X10
+185 -13
View File
@@ -32,6 +32,13 @@ import (
// riscvFrameInfo holds the frame layout derived from a TEXT directive.
type riscvFrameInfo struct {
autosize int // the real SP adjustment (locals + saved LR)
// Stack-split guard state: the toolchain emits the check for every
// non-NOSPLIT function whose autosize is nonzero (a zero autosize is
// "effectively NOSPLIT"); unlike amd64 and arm64 there is no leaf
// auto-NOSPLIT.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
}
// riscvComputeFrame derives the frame layout for a TEXT function.
@@ -40,11 +47,34 @@ func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
if frame != 0 || !riscvIsLeaf(t) {
// FixedFrameSize = 8: space for the saved link register. A
// zero-frame non-leaf function still opens an 8-byte frame for LR.
return riscvFrameInfo{autosize: frame + 8}
autosize := frame + 8
fi := riscvFrameInfo{autosize: autosize}
if !hasNoSplitFlag(t) {
fi.needSplit = true
switch {
case autosize <= stackSmall:
fi.splitClass = 0
case autosize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi
}
return riscvFrameInfo{}
}
// hasNoSplitFlag reports whether the TEXT directive carries NOSPLIT.
func hasNoSplitFlag(t *ast.Text) bool {
for _, f := range t.Flags {
if strings.EqualFold(f, "NOSPLIT") {
return true
}
}
return false
}
// riscvIsLeaf reports whether a function contains no call instructions.
// CALL always links; JAL/JALR link only when their destination register is
// the link register (X1), matching cmd/internal/obj/riscv's containsCall.
@@ -58,12 +88,12 @@ func riscvIsLeaf(t *ast.Text) bool {
case "CALL":
return false
case "JAL":
// JAL rd, target — a call only when rd is the link register.
// JAL rd, target, a call only when rd is the link register.
if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 {
return false
}
case "JALR":
// JALR rs1, rd — a call when rd is X1; JALR offset(rs1) always
// JALR rs1, rd, a call when rd is X1; JALR offset(rs1) always
// links to X1.
if len(in.Operands) == 1 {
return false
@@ -84,28 +114,99 @@ func riscvPrologue(fi riscvFrameInfo) []byte {
return nil
}
var out []byte
// MOV LR, -autosize(SP) — SD X1, -autosize(X2). The negative offset is
// not compressible to C.SDSP (unsigned), so it stays 4 bytes.
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
// ADDI $-autosize, SP, SP — open the frame (C.ADDI when it fits).
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
// MOV LR, 0(SP) — SD X1, 0(X2) → C.SDSP X1, 0.
// MOV LR, -autosize(SP), SD X1, -autosize(X2). The negative offset is
// not compressible to C.SDSP (unsigned), so it stays 4 bytes. Beyond
// the imm12 range the toolchain materialises the address in X31.
if fits12(int32(-fi.autosize)) {
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
} else {
out = append(out, riscvAddressInX31(int32(-fi.autosize))...)
lo := int32(-fi.autosize) - (splitHi(int32(-fi.autosize)) << 12)
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 1, lo))...)
}
// ADDI $-autosize, SP, SP, open the frame (C.ADDI when it fits; X31
// materialisation beyond imm12).
if fits12(int32(-fi.autosize)) {
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
} else {
out = append(out, riscvAddToSP(int32(-fi.autosize))...)
}
// MOV LR, 0(SP), SD X1, 0(X2) → C.SDSP X1, 0.
c := rvcSSP(0x7, 1, 0)
out = append(out, byte(c), byte(c>>8))
return out
}
func fits12(v int32) bool { return v >= -2048 && v <= 2047 }
// splitHi returns the LUI half of the hi/lo split of v (what remains is the
// sign-extended 12-bit low part).
func splitHi(v int32) int32 {
_, high := splitRISCV32Imm(v)
return high
}
// riscvAddressInX31 materialises hi(v) into X31 against the stack pointer,
// matching the toolchain's large-frame addressing: C.LUI (or LUI) X31, hi;
// C.ADD (or ADD) X31, SP.
func riscvAddressInX31(v int32) []byte {
return riscvAddressInX31WithBase(v, 2)
}
// riscvAddressInX31WithBase materialises hi(v) into X31 against an arbitrary
// base register: LUI (or C.LUI) X31, hi; C.ADD X31, rs1. The CR rs2 field
// carries the full 5-bit register, so the compressed form is always
// available.
func riscvAddressInX31WithBase(v int32, rs1 int) []byte {
hi := splitHi(v)
var out []byte
if hi >= -32 && hi <= 31 {
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
}
c := rvcCR(0x9, 31, uint32(rs1))
return append(out, byte(c), byte(c>>8))
}
// riscvAddToSP adds v to SP through X31 for the values imm12 cannot carry:
// C.LUI X31, hi; C.ADDIW X31, lo; C.ADD SP, X31 (the toolchain's form).
func riscvAddToSP(v int32) []byte {
hi := splitHi(v)
lo := v - (hi << 12)
var out []byte
if hi >= -32 && hi <= 31 {
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
}
if lo >= -32 && lo <= 31 {
c := rvcCI(0x1, 31, uint32(lo)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvIType(riscvEnc{0x1b, 0x0, 0x00}, 31, 31, lo))...)
}
c := rvcCR(0x9, 2, 31)
return append(out, byte(c), byte(c>>8))
}
// riscvReturn returns the bytes for a RET: the epilogue (restore LR and
// deallocate the frame when present) followed by the uncompressed JALR X0,
// 0(X1) the toolchain emits for RET (it never compresses RET to C.JR).
func riscvReturn(fi riscvFrameInfo) []byte {
var out []byte
if fi.autosize != 0 {
// MOV 0(SP), LR — LD X1, 0(X2) → C.LDSP X1, 0.
// MOV 0(SP), LR, LD X1, 0(X2) → C.LDSP X1, 0.
c := rvcLSP(0x3, 1, 0)
out = append(out, byte(c), byte(c>>8))
// ADDI $autosize, SP, SP — close the frame (C.ADDI when it fits).
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
// ADDI $autosize, SP, SP, close the frame (C.ADDI when it fits).
if fits12(int32(fi.autosize)) {
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
} else {
out = append(out, riscvAddToSP(int32(fi.autosize))...)
}
}
// JALR X0, 0(X1).
return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...)
@@ -143,7 +244,7 @@ func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
}
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final JALR — the point where SP is restored.
// (but not including) the final JALR, the point where SP is restored.
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
if fi.autosize == 0 {
return 0
@@ -180,3 +281,74 @@ func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32
}
return -1, 0
}
// riscvGuardLen returns the byte length of the stack-split guard prefix
// including the inline morestack call (zero when the function needs no
// guard). Unlike amd64 and arm64, the toolchain places the morestack call
// between the guard and the body: the guard branches forward over it.
func riscvGuardLen(fi riscvFrameInfo) int {
_, reloc := riscvGuard(fi)
_ = reloc
return len(riscvGuardBytes(fi))
}
// riscvGuard emits the stack-split guard prefix with the inline morestack
// call: the branch skips forward over JAL X5 and JAL X0 straight into the
// body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are
// relative to the guard itself, which sits at function offset 0.
func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
if !fi.needSplit {
return nil, Reloc{}
}
// MOV 16(g), X6 (g.stackguard0), g = X27.
out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16))
jalBack := func() []byte {
// JAL X0 back to the function start: it sits right after the JAL X5,
// so its displacement is minus the current offset.
return wordLE(riscvJType(0, int32(-len(out))))
}
var reloc Reloc
switch fi.splitClass {
case 0:
// BLTU X6, SP, done (+8: over the CALL and the JMP back)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
case 1:
// ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+8)
off := int32(fi.autosize - stackSmall)
out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
default:
// MOV $(framesize-StackSmall), X7; BLTU SP, X7, call;
// ADD $-(framesize-StackSmall), SP, X7; BLTU X6, X7, call
off := int32(fi.autosize - stackSmall)
mov := encodeRISCVLoadImm(7, off)
out = append(out, mov...)
addiLen := riscvItypeImmediateSize("ADDI", -off)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...)
addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off)
if err != nil {
addi = nil
}
out = append(out, addi...)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
}
return out, reloc
}
// riscvGuardBytes emits the guard prefix bytes alone (sizing helper).
func riscvGuardBytes(fi riscvFrameInfo) []byte {
g, _ := riscvGuard(fi)
return g
}
+43 -43
View File
@@ -36,11 +36,11 @@ const (
vexNDS3Imm
// vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg =
// ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM
// source lives in the reg field, the destination in r/m — the PEXTR-style
// source lives in the reg field, the destination in r/m, the PEXTR-style
// layout. VEXTRACTI128 and VEXTRACTF128 use this shape.
vexExtract
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
// in ModRM.reg and the destination in r/m — the layout of the EVEX
// in ModRM.reg and the destination in r/m, the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD).
vexRMRev
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
@@ -68,7 +68,7 @@ type vexSpec struct {
// incrementally; every entry is covered by a byte-for-byte ground-truth test
// against the Go assembler.
var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG — integer arithmetic / logic / compare.
// VEX.128/256.66.0F.WIG, integer arithmetic / logic / compare.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3},
"VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3},
@@ -82,7 +82,7 @@ var vexTable = map[string]vexSpec{
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3},
"VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3},
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3},
// VEX.256.66.0F38.W0 — dword permute (three-operand NDS form).
// VEX.256.66.0F38.W0, dword permute (three-operand NDS form).
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG.
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3},
@@ -90,14 +90,14 @@ var vexTable = map[string]vexSpec{
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3},
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic.
// VEX.128/256.66.0F.WIG, packed double-precision arithmetic / logic.
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG — packed single-precision arithmetic.
// VEX.128/256.0F.WIG, packed single-precision arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
@@ -107,7 +107,7 @@ var vexTable = map[string]vexSpec{
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
// VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed
// VEX.128.F2.0F.WIG, scalar double-precision arithmetic (the packed
// opcodes with an F2 pp).
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
@@ -115,7 +115,7 @@ var vexTable = map[string]vexSpec{
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
// VEX.128.F3.0F.WIG — scalar single-precision arithmetic (the packed
// VEX.128.F3.0F.WIG, scalar single-precision arithmetic (the packed
// opcodes with an F3 pp).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
@@ -123,10 +123,10 @@ var vexTable = map[string]vexSpec{
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
// VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form).
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG — sign/zero extend and broadcast (reg=dst, rm=src,
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
// no vvvv).
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
@@ -143,70 +143,70 @@ var vexTable = map[string]vexSpec{
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
"VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM},
"VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM},
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
// VEX.128/256.F3.0F.WIG, signed dword to packed double conversion
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
// VEX.128/256.0F.WIG — signed dword to packed single conversion
// VEX.128/256.0F.WIG, signed dword to packed single conversion
// (reg=dst, rm=src, no vvvv, no mandatory prefix).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
// VEX.128/256.0F.WIG — packed single to packed double conversion
// VEX.128/256.0F.WIG, packed single to packed double conversion
// (reg=dst, rm=src; the destination is the wide operand and sets the
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
// the Go assembler emits the instruction with pp = 00, and gasm follows
// the Go assembler's bytes — its machine code is the oracle, not the
// the Go assembler's bytes, its machine code is the oracle, not the
// manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
// VEX.128.F2.0F.WIG — duplicate the low double of each 128-bit lane
// VEX.128.F2.0F.WIG, duplicate the low double of each 128-bit lane
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
// VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src).
// VEX.128/256.66.0F.WIG, move mask to a GPR (reg=gpr dst, rm=vec src).
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
// VEX.128/256.66.0F.WIG — immediate shifts (opdigit selects the shift).
// VEX.128/256.66.0F.WIG, immediate shifts (opdigit selects the shift).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm},
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm},
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm},
"VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm},
"VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm},
// VEX.128/256.66.0F.WIG — immediate shuffle (reg=dst, rm=src, imm8).
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
// VEX.256.66.0F3A.W1 — qword permute (reg=dst, rm=src, imm8).
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8).
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
// VEX.128/256.66.0F.WIG — two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// imm8).
"VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm},
// VEX.256.66.0F3A.W0 — permute / insert (same shape; VINSERTI128's rm is
// VEX.256.66.0F3A.W0, permute / insert (same shape; VINSERTI128's rm is
// the XMM or memory source).
"VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm},
"VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm},
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
// VEX.256.66.0F3A.W0, lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
// VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst:
// reg=src, rm=XMM/memory dst, imm8 — the extract layout).
// VEX.128/256.66.0F3A.W0, half-precision convert back ($imm, src, dst:
// reg=src, rm=XMM/memory dst, imm8, the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
// VEX.128.0F.W0 — no operands.
// VEX.128.0F.W0, no operands.
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
// VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src).
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
// VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst,
// VEX.66.0F38.W0, broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; SD is 256-bit only).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
// VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
// source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
// VEX.F3.0F.WIG, replicate even/odd singles (reg=dst, rm=src).
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y
// VEX.66.0F.WIG, packed double to packed single conversion, the X/Y
// spellings: the destination is always XMM and the spelling fixes the
// source length (X = 128, Y = 256).
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
@@ -230,14 +230,14 @@ var vexTable = map[string]vexSpec{
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
// VEX.128/256.66.0F.WIG, word shifts (opdigit selects the shift).
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
// VEX.F2.0F — packed double to packed dword conversions, truncating and
// VEX.F2.0F, packed double to packed dword conversions, truncating and
// non-truncating. The destination is always XMM; the X/Y spellings fix
// the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen.
// the source length (XMM/YMM), and VEX.L follows it, see vexSrcLen.
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
@@ -257,7 +257,7 @@ var vexSrcLen = map[string]int{
"VCVTPD2PSY": 1,
}
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
// vexVarShift maps the shift mnemonics to their variable-count opcode, the
// form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8),
// an ordinary NDS encoding rather than the /digit immediate form above.
var vexVarShift = map[string]byte{
@@ -288,20 +288,20 @@ type vexMoveSpec struct {
// vexMoveTable maps an upper-case move mnemonic to its encoding.
var vexMoveTable = map[string]vexMoveSpec{
// VEX.128/256.F3.0F.WIG — unaligned integer move.
// VEX.128/256.F3.0F.WIG, unaligned integer move.
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG — unaligned packed double move.
// VEX.128/256.66.0F.WIG, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
// VEX.128.66.0F.W0 — 32-bit GPR/memory ↔ XMM.
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
"VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true},
// VMOVQ — 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
// VMOVQ, 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true},
// VEX.128.F2.0F.WIG — scalar double move, memory operands only (the
// VEX.128.F2.0F.WIG, scalar double move, memory operands only (the
// register form takes three operands and is not supported yet).
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128.F3.0F.WIG — scalar single move, memory operands only.
// VEX.128.F3.0F.WIG, scalar single move, memory operands only.
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128/256 — aligned packed moves.
// VEX.128/256, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
}
@@ -317,7 +317,7 @@ func isVex(mnemUpper string) bool {
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
// Vector register indices 16–31 exist only in EVEX encodings; fail
// Vector register indices 16-31 exist only in EVEX encodings; fail
// loudly rather than silently truncating the index.
for _, op := range ops {
if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 {
@@ -420,7 +420,7 @@ func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
}
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the VEX.L bit following the source — fixed
// the destination always XMM and the VEX.L bit following the source, fixed
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
// the source is memory.
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
+5 -5
View File
@@ -173,7 +173,7 @@ func TestVexGroundTruth(t *testing.T) {
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""},
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""},
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""},
// Floating point (packed and scalar) and FMA — same NDS form, the pp
// Floating point (packed and scalar) and FMA; same NDS form, the pp
// bits and map select the operation.
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""},
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""},
@@ -217,7 +217,7 @@ func TestVexGroundTruth(t *testing.T) {
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""},
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""},
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""},
// Moves — each direction picks its own opcode and VEX.W.
// Moves; each direction picks its own opcode and VEX.W.
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""},
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""},
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""},
@@ -234,7 +234,7 @@ func TestVexGroundTruth(t *testing.T) {
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""},
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""},
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""},
// Packed double arithmetic and unpack — the NDS form, the opcode
// Packed double arithmetic and unpack; the NDS form, the opcode
// selects the operation.
{"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""},
{"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""},
@@ -255,12 +255,12 @@ func TestVexGroundTruth(t *testing.T) {
{"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""},
{"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""},
{"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""},
// VMOVDDUP — duplicate the low double (reg=dst, rm=src, F2 pp).
// VMOVDDUP; duplicate the low double (reg=dst, rm=src, F2 pp).
{"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""},
{"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""},
{"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""},
// Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3
// prefix — see the table comment), DQ→PD.
// prefix; see the table comment), DQ→PD.
{"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""},
{"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""},
{"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""},
+1 -1
View File
@@ -17,7 +17,7 @@ type File struct {
Orphans []Stmt // labels/instructions seen before any TEXT directive
// Macros holds the names introduced by #define directives in this file.
// The linter uses it to avoid flagging macro invocations as unknown
// instructions (macro expansion itself is out of scope — see the docs).
// instructions (macro expansion itself is out of scope, see the docs).
Macros map[string]bool
}
+1 -4
View File
@@ -84,7 +84,7 @@ REPL commands:
if *timeout > 0 {
go func() {
time.Sleep(*timeout)
fmt.Fprintf(os.Stderr, "gasm debug: timeout (%s) — killing the debuggee\n", *timeout)
fmt.Fprintf(os.Stderr, "gasm debug: timeout (%s), killing the debuggee\n", *timeout)
os.Exit(3)
}()
}
@@ -140,7 +140,6 @@ REPL commands:
}
// Construct the argument block with buffer pointers at the correct positions.
bufIdx := 0
for _, arg := range layout {
if !arg.IsPtr {
continue
@@ -170,12 +169,10 @@ REPL commands:
argBlock[off+16+j] = byte(size >> (j * 8))
}
}
bufIdx++
break
}
}
}
_ = bufIdx
} else {
argBlock = make([]byte, fl.Args)
sess, err = debug.Launch("", path, *funcName, argBlock)
+148
View File
@@ -0,0 +1,148 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"fmt"
"os"
"sort"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// cmdDis disassembles machine code: either a raw binary (standard input with
// "-") whose architecture is given with -a, or a .s file, which is assembled
// first so the listing shows the real function and label layout.
func cmdDis(args []string) int {
fs := newCommand("dis", "gasm dis [-a arch] <file>", `
Disassemble machine code to instruction text (via golang.org/x/arch).
With a .s file, the file is assembled first and the listing follows the
real layout: one block per TEXT function, local labels printed at their
offsets. The architecture comes from the file name suffix, or from -a.
With any other file, or "-" for standard input, the bytes are disassembled
linearly and -a selects the architecture (amd64, arm64, riscv64 or
loong64).
`)
archName := fs.String("a", "", "architecture for raw input: amd64, arm64, riscv64 or loong64")
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm dis [-a arch] <file>")
return 2
}
path := fs.Arg(0)
var target arch.Arch
if *archName != "" {
var err error
target, err = auditArch(*archName)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
return 2
}
}
if strings.HasSuffix(path, ".s") {
if target == arch.Unknown {
target = arch.FromFilename(path)
}
if target == arch.Unknown {
fmt.Fprintln(os.Stderr, "gasm dis: cannot infer the architecture from the file name; use -a")
return 2
}
return disSource(path, target)
}
if target == arch.Unknown {
fmt.Fprintln(os.Stderr, "gasm dis: raw input needs -a (amd64, arm64, riscv64 or loong64)")
return 2
}
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm dis:", err)
return 1
}
printListing(target, []byte(src), 0, nil)
return 0
}
// disSource assembles a .s file and prints one listing block per function.
func disSource(path string, target arch.Arch) int {
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm dis:", err)
return 1
}
f, errs := parser.Parse(path, src)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return 1
}
img, err := assembleFile(target, f)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
return 1
}
if len(img.Funcs) == 0 {
fmt.Fprintln(os.Stderr, "gasm dis: no assemblable TEXT functions found")
return 1
}
for _, fn := range img.Funcs {
code := img.Code[fn.Offset : fn.Offset+fn.Size]
fmt.Printf("%s: %d bytes\n", fn.Name, fn.Size)
labels := make(map[int][]string, len(fn.Labels))
for name, off := range fn.Labels {
labels[off] = append(labels[off], name)
}
for off := range labels {
sort.Strings(labels[off])
}
printListing(target, code, uint64(fn.Offset), labels)
}
if len(img.Data) > 0 {
fmt.Printf("data: %d bytes at 0x%x\n", len(img.Data), len(img.Code))
}
return 0
}
// printListing decodes code linearly from offset base, printing label lines
// (label name to offset within the block) as they are reached.
func printListing(a arch.Arch, code []byte, base uint64, labels map[int][]string) {
pc := 0
for pc < len(code) {
for _, name := range labels[pc] {
fmt.Printf("%s:\n", name)
}
ins, err := disasm.Decode(a, code[pc:], base+uint64(pc))
if err != nil {
break
}
end := min(pc+ins.Len, len(code))
fmt.Printf(" %04x: %-16s %s\n", base+uint64(pc), hexBytes(code[pc:end]), ins.Text)
if ins.Len <= 0 {
break
}
pc += ins.Len
}
}
// hexBytes renders up to 8 bytes as contiguous hex.
func hexBytes(b []byte) string {
var sb strings.Builder
for i, c := range b {
if i == 8 {
break
}
if i > 0 {
sb.WriteByte(' ')
}
fmt.Fprintf(&sb, "%02x", c)
}
return sb.String()
}
+144 -304
View File
@@ -1,7 +1,7 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Command gasm is the developer frontend for GAsm — Go's Plan 9 assembler.
// Command gasm is the developer frontend for GAsm, Go's Plan 9 assembler.
// It bundles a token dumper, a parser, a formatter, a linter and a language
// server into one binary. Every subcommand works headlessly so it can be
// driven from scripts and CI as well as from an editor.
@@ -18,6 +18,7 @@ import (
"os/exec"
"path/filepath"
"runtime"
"runtime/debug"
"slices"
"sort"
"strconv"
@@ -36,9 +37,16 @@ import (
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
)
// version is the release version, stamped at build time via
// -ldflags "-X main.version=…" (defaulting to the current release).
var version = "0.32.0"
// version reports the release the toolchain recorded for this build: the
// tag on a tag, a pseudo-version below one, and (devel) outside version
// control. Nothing is injected; the recorded value cannot go stale.
func version() string {
bi, ok := debug.ReadBuildInfo()
if !ok || bi.Main.Version == "" {
return "(devel)"
}
return bi.Main.Version
}
func main() {
if len(os.Args) < 2 {
@@ -56,6 +64,8 @@ func main() {
os.Exit(cmdLint(os.Args[2:]))
case "asm":
os.Exit(cmdAsm(os.Args[2:]))
case "dis":
os.Exit(cmdDis(os.Args[2:]))
case "verify":
os.Exit(cmdVerify(os.Args[2:]))
case "debug":
@@ -81,18 +91,18 @@ func main() {
case "help", "--help", "-h":
usage(os.Stdout)
default:
fmt.Fprintf(os.Stderr, "gasm: unknown command %q — run \"gasm --help\" for usage\n", os.Args[1])
fmt.Fprintf(os.Stderr, "gasm: unknown command %q; run \"gasm --help\" for usage\n", os.Args[1])
os.Exit(2)
}
}
// cmdVersion prints the release version.
// cmdVersion prints the recorded version.
func cmdVersion() int {
fmt.Printf("gasm %s\n", version)
fmt.Printf("gasm %s\n", version())
return 0
}
// ANSI color helpers for terminal output.
// ANSI colour helpers for terminal output.
const (
colorReset = "\033[0m"
colorBold = "\033[1m"
@@ -101,7 +111,7 @@ const (
colorGray = "\033[90m"
)
// isTTY reports whether the writer is a terminal (for color output).
// isTTY reports whether the writer is a terminal (for colour output).
func isTTY(w io.Writer) bool {
if f, ok := w.(*os.File); ok {
stat, _ := f.Stat()
@@ -117,7 +127,7 @@ func usage(w io.Writer) {
bold, cyan, yellow, gray, reset = colorBold, colorCyan, colorYellow, colorGray, colorReset
}
fmt.Fprintf(w, "%sgasm %s%s — developer tooling for Go's Plan 9 assembler (GAsm)%s\n\n", bold, version, reset, reset)
fmt.Fprintf(w, "%sgasm %s%s: developer tooling for Go's Plan 9 assembler (GAsm)%s\n\n", bold, version(), reset, reset)
fmt.Fprintf(w, "gasm bundles a lexer, parser, formatter, linter, standalone assembler and\n")
fmt.Fprintf(w, "language server for Plan 9 assembly into one self-contained binary.\n\n")
@@ -132,6 +142,7 @@ func usage(w io.Writer) {
{"fmt", "canonicalise formatting (gofmt for assembly)"},
{"lint", "run static checks"},
{"asm", "assemble .s files to machine code (amd64, arm64, riscv64, loong64)"},
{"dis", "disassemble machine code (raw bytes or an assembled .s file)"},
{"verify", "JIT-assemble and run dynamic checks (amd64, arm64, riscv64, loong64)"},
{"debug", "interactive source-level debugger (amd64, arm64, riscv64, loong64)"},
{"diff", "compare machine code of two .s files"},
@@ -263,24 +274,34 @@ standard input.
funcs++
}
}
fmt.Printf("%s: OK — %d declarations, %d functions\n", path, len(file.Decls), funcs)
fmt.Printf("%s: OK, %d declarations, %d functions\n", path, len(file.Decls), funcs)
return 0
}
func cmdFmt(args []string) int {
fs := newCommand("fmt", "gasm fmt [-w] [path...]", `
fs := newCommand("fmt", "gasm fmt [-w|-l|-d] [path...]", `
Canonicalise the formatting of Plan 9 assembly sources: indentation, operand
spacing, per-function mnemonic alignment and blank-line layout (exactly one
blank line before each label, TEXT and GLOBL block). Formatting is
idempotent and preserves every line, comments included.
With no paths — or a directory path — every .s file below it is reformatted
With no paths, or a directory path, every .s file below it is reformatted
in place and the changed files are listed, the way go fmt does; "." and "_"
directories are skipped. Explicit file paths print to stdout unless -w is
given.
-l and -d rewrite nothing: -l prints the paths whose formatting differs
from gasm's (empty output means everything is formatted, which is what a CI
check wants), -d prints the diffs. They are mutually exclusive.
`)
write := fs.Bool("w", false, "write result to the source file")
list := fs.Bool("l", false, "list files whose formatting differs from gasm's")
diffMode := fs.Bool("d", false, "print diffs instead of rewriting files")
fs.Parse(args)
if *list && *diffMode {
fmt.Fprintln(os.Stderr, "gasm fmt: -l and -d are mutually exclusive")
return 2
}
// Like go fmt: with no arguments, or with a directory argument, every .s
// file below the directory is formatted in place and the names of the
// changed files are listed; explicit file arguments keep the -w / stdout
@@ -318,6 +339,16 @@ given.
continue
}
out := format.Source(src)
if *list || *diffMode {
if out != src {
if *list {
fmt.Println(path)
} else {
fmt.Print(unifiedDiff(path, strings.Split(src, "\n"), strings.Split(out, "\n")))
}
}
continue
}
if dirMode || *write {
if out != src {
if err := os.WriteFile(path, []byte(out), 0o644); err != nil {
@@ -337,7 +368,7 @@ given.
}
// asmFiles collects the .s files below dir, skipping directories whose name
// starts with "." or "_" — as the go tooling does, which keeps .git and
// starts with "." or "_", as the go tooling does, which keeps .git and
// scratch or reference trees (e.g. _refs) untouched.
func asmFiles(dir string) ([]string, error) {
var out []string
@@ -419,6 +450,7 @@ hover, document symbols, diagnostics and semantic-token highlighting.
`)
fs.Parse(args)
srv := lsp.New(os.Stdin, os.Stdout)
srv.SetVersion(version())
if err := srv.Run(); err != nil {
fmt.Fprintln(os.Stderr, "gasm lsp:", err)
return 1
@@ -765,9 +797,11 @@ gasm verify --fuzz which exercises the code paths.
return 0
}
// cmdVerifyRISCV handles the verify subcommand for RISC-V files.
// JIT requires RISC-V hardware; only ground-truth and profile are available.
func cmdVerifyRISCV(path string, groundTruth, profile bool) int {
// cmdVerifyNonJIT handles the verify subcommand for files whose architecture
// the host cannot execute: only the ground-truth comparison and the static
// profile are available there. Relocation sites are masked before the byte
// comparison, as the toolchain leaves them zero for the linker.
func cmdVerifyNonJIT(path string, targetArch arch.Arch, groundTruth, profile bool) int {
src, err := readSource(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
@@ -780,63 +814,34 @@ func cmdVerifyRISCV(path string, groundTruth, profile bool) int {
if len(errs) > 0 {
return 1
}
img, err := asm.AssembleFileRISCV(f)
img, err := assembleFile(targetArch, f)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
if groundTruth {
gt, err := verify.GroundTruthRISCV(path)
var gt map[string][]byte
switch targetArch {
case arch.RISCV:
gt, err = verify.GroundTruthRISCV(path)
case arch.LOONG64:
gt, err = verify.GroundTruthLOONG64(path)
case arch.ARM64:
gt, err = verify.GroundTruthARM64(path)
default:
gt, err = verify.GroundTruth(path)
}
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: ground truth: %v\n", err)
return 1
}
matched, total := 0, 0
for _, fn := range img.Funcs {
gasmCode := img.Code[fn.Offset : fn.Offset+fn.Size]
goCode, ok := gt[fn.Name]
if !ok {
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", fn.Name)
continue
}
total++
gasmCmp := make([]byte, len(gasmCode))
goCmp := make([]byte, len(goCode))
copy(gasmCmp, gasmCode)
copy(goCmp, goCode)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ {
gasmCmp[j] = 0
}
for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ {
goCmp[j] = 0
}
}
if bytes.Equal(gasmCmp, goCmp) {
matched++
if len(fn.Relocs) > 0 {
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
} else {
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
}
} else {
fmt.Printf(" %s: MISMATCH (%d vs %d bytes)\n", fn.Name, fn.Size, len(goCode))
for i := 0; i < len(gasmCode) || i < len(goCode); i += 16 {
var gb, gs string
for j := i; j < i+16 && j < len(gasmCode); j++ {
gb += fmt.Sprintf(" %02x", gasmCode[j])
}
for j := i; j < i+16 && j < len(goCode); j++ {
gs += fmt.Sprintf(" %02x", goCode[j])
}
fmt.Printf(" %04x: gasm:%s\n", i, gb)
fmt.Printf(" %04x: gt: %s\n", i, gs)
}
}
matched, total, diffs := compareGroundTruth(img, gt)
if diffs > 0 {
printCodeDiff(img, gt)
}
fmt.Printf("%s: %d/%d matched\n", path, matched, total)
if matched < total {
if matched < total || diffs > 0 {
return 1
}
return 0
@@ -856,193 +861,74 @@ func cmdVerifyRISCV(path string, groundTruth, profile bool) int {
return 0
}
// cmdVerifyLOONG64 verifies a loong64 source file against `go tool asm`
// (GOARCH=loong64) — the ground-truth oracle — since gasm cannot JIT-load
// LoongArch code on an amd64 host. Relocation sites are masked before the
// byte comparison, as the toolchain leaves them zero for the linker.
func cmdVerifyLOONG64(path string, groundTruth, profile bool) int {
src, err := readSource(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
f, errs := parser.Parse(path, src)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return 1
}
img, err := asm.AssembleFileLOONG64(f)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
if groundTruth {
gt, err := verify.GroundTruthLOONG64(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: ground truth: %v\n", err)
return 1
}
matched, total := 0, 0
for _, fn := range img.Funcs {
gasmCode := img.Code[fn.Offset : fn.Offset+fn.Size]
goCode, ok := gt[fn.Name]
if !ok {
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", fn.Name)
continue
}
total++
gasmCmp := make([]byte, len(gasmCode))
goCmp := make([]byte, len(goCode))
copy(gasmCmp, gasmCode)
copy(goCmp, goCode)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ {
gasmCmp[j] = 0
}
for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ {
goCmp[j] = 0
}
}
if bytes.Equal(gasmCmp, goCmp) {
matched++
if len(fn.Relocs) > 0 {
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
} else {
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
}
} else {
fmt.Printf(" %s: MISMATCH (%d vs %d bytes)\n", fn.Name, fn.Size, len(goCode))
for i := 0; i < len(gasmCode) || i < len(goCode); i += 16 {
var gb, gs string
for j := i; j < i+16 && j < len(gasmCode); j++ {
gb += fmt.Sprintf(" %02x", gasmCode[j])
}
for j := i; j < i+16 && j < len(goCode); j++ {
gs += fmt.Sprintf(" %02x", goCode[j])
}
fmt.Printf(" %04x: gasm:%s\n", i, gb)
fmt.Printf(" %04x: gt: %s\n", i, gs)
}
}
}
fmt.Printf("%s: %d/%d matched\n", path, matched, total)
if matched < total {
return 1
}
return 0
}
if profile {
for _, fn := range img.Funcs {
fmt.Printf("%s: %d bytes, labels: %v\n", fn.Name, fn.Size, fn.Labels)
}
return 0
}
fmt.Printf("%s: %d functions assembled\n", path, len(img.Funcs))
// compareGroundTruth compares the image's functions against the go tool asm
// output byte-for-byte, masking relocation sites (disp32 fields the Go linker
// fills at link time). It prints one line per function and returns the
// matched and compared counts plus the number of functions with byte diffs.
func compareGroundTruth(img *asm.Image, gt map[string][]byte) (matched, total, diffs int) {
for _, fn := range img.Funcs {
fmt.Printf(" %s: %d bytes\n", fn.Name, fn.Size)
gasmCode := img.Code[fn.Offset : fn.Offset+fn.Size]
goCode, ok := gt[fn.Name]
if !ok {
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", fn.Name)
continue
}
total++
gasmCmp := make([]byte, len(gasmCode))
goCmp := make([]byte, len(goCode))
copy(gasmCmp, gasmCode)
copy(goCmp, goCode)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ {
gasmCmp[j] = 0
}
for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ {
goCmp[j] = 0
}
}
if bytes.Equal(gasmCmp, goCmp) {
matched++
if len(fn.Relocs) > 0 {
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
} else {
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
}
} else {
diffs++
fmt.Printf(" %s: MISMATCH (%d vs %d bytes)\n", fn.Name, fn.Size, len(goCode))
}
}
return 0
return matched, total, diffs
}
func cmdVerifyARM64(path string, groundTruth, profile bool) int {
src, err := readSource(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
f, errs := parser.Parse(path, src)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return 1
}
img, err := asm.AssembleFileARM64(f)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
if groundTruth {
gt, err := verify.GroundTruthARM64(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: ground truth: %v\n", err)
return 1
}
matched, total := 0, 0
for _, fn := range img.Funcs {
gasmCode := img.Code[fn.Offset : fn.Offset+fn.Size]
goCode, ok := gt[fn.Name]
if !ok {
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", fn.Name)
continue
}
total++
gasmCmp := make([]byte, len(gasmCode))
goCmp := make([]byte, len(goCode))
copy(gasmCmp, gasmCode)
copy(goCmp, goCode)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ {
gasmCmp[j] = 0
}
for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ {
goCmp[j] = 0
}
}
if bytes.Equal(gasmCmp, goCmp) {
matched++
if len(fn.Relocs) > 0 {
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
} else {
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
}
} else {
fmt.Printf(" %s: MISMATCH (%d vs %d bytes)\n", fn.Name, fn.Size, len(goCode))
for i := 0; i < len(gasmCode) || i < len(goCode); i += 16 {
var gb, gs string
for j := i; j < i+16 && j < len(gasmCode); j++ {
gb += fmt.Sprintf(" %02x", gasmCode[j])
}
for j := i; j < i+16 && j < len(goCode); j++ {
gs += fmt.Sprintf(" %02x", goCode[j])
}
fmt.Printf(" %04x: gasm:%s\n", i, gb)
fmt.Printf(" %04x: gt: %s\n", i, gs)
}
}
}
fmt.Printf("%s: %d/%d matched\n", path, matched, total)
if matched < total {
return 1
}
return 0
}
if profile {
for _, fn := range img.Funcs {
fmt.Printf("%s: %d bytes, labels: %v\n", fn.Name, fn.Size, fn.Labels)
}
return 0
}
fmt.Printf("%s: %d functions assembled\n", path, len(img.Funcs))
// printCodeDiff shows a 16-byte hex dump per function whose gasm bytes differ
// from the go tool asm output.
func printCodeDiff(img *asm.Image, gt map[string][]byte) {
for _, fn := range img.Funcs {
fmt.Printf(" %s: %d bytes\n", fn.Name, fn.Size)
gasmCode := img.Code[fn.Offset : fn.Offset+fn.Size]
goCode, ok := gt[fn.Name]
if !ok || bytes.Equal(gasmCode, goCode) {
continue
}
for i := 0; i < len(gasmCode) || i < len(goCode); i += 16 {
var gb, gs string
for j := i; j < i+16 && j < len(gasmCode); j++ {
gb += fmt.Sprintf(" %02x", gasmCode[j])
}
for j := i; j < i+16 && j < len(goCode); j++ {
gs += fmt.Sprintf(" %02x", goCode[j])
}
fmt.Printf(" %04x: gasm:%s\n", i, gb)
fmt.Printf(" %04x: gt: %s\n", i, gs)
}
}
return 0
}
func cmdVerify(args []string) int {
set := newCommand("verify", "gasm verify [-smoke] [-abi] [-fuzz] [-ground-truth] [-profile] [-call] <file.s>", `
Assemble FILE (amd64), map it into executable memory and report the available
functions. This confirms the assembled image is self-consistent (no
unresolved external symbols) and executable — the prerequisite for dynamic
unresolved external symbols) and executable, the prerequisite for dynamic
testing.
With -smoke, each NOSPLIT function is called with a zeroed argument block to
@@ -1100,17 +986,12 @@ each entry reproduces.
// under the available loong64 emulators), so those kernels take the
// toolchain-comparison path.
if targetArch != hostArch() || targetArch == arch.LOONG64 {
// No JIT on this host: ground truth and profile remain available.
// (loong64 is ground-truth-only everywhere for now: its trampoline
// is implemented but not yet validated against real hardware.)
switch targetArch {
case arch.RISCV:
// RISC-V: ground-truth only (no JIT on non-RISC-V hosts).
return cmdVerifyRISCV(path, *groundTruth, *profile)
case arch.LOONG64:
// LoongArch: ground-truth only (trampoline not yet
// hardware-validated).
return cmdVerifyLOONG64(path, *groundTruth, *profile)
case arch.ARM64:
// AArch64: ground-truth only (no JIT on non-ARM64 hosts).
return cmdVerifyARM64(path, *groundTruth, *profile)
case arch.RISCV, arch.LOONG64, arch.ARM64:
return cmdVerifyNonJIT(path, targetArch, *groundTruth, *profile)
case arch.AMD64:
fmt.Fprintln(os.Stderr, "gasm verify: JIT-based checks need an amd64 host; use --ground-truth here")
return 1
@@ -1224,50 +1105,13 @@ each entry reproduces.
fmt.Fprintf(os.Stderr, "gasm verify: ground truth: %v\n", err)
return 1
}
matched, total := 0, 0
for _, name := range names {
fl, _ := k.Func(name)
gasmCode := k.Image().Code[fl.Offset : fl.Offset+fl.Size]
goCode, ok := gt[name]
if !ok {
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", name)
continue
}
total++
// Compare, masking relocation sites (disp32 fields that the
// Go linker fills at link time — gasm resolves them internally).
gasmCmp := make([]byte, len(gasmCode))
goCmp := make([]byte, len(goCode))
copy(gasmCmp, gasmCode)
copy(goCmp, goCode)
for _, r := range fl.Relocs {
for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ {
gasmCmp[j] = 0
}
for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ {
goCmp[j] = 0
}
}
if bytes.Equal(gasmCmp, goCmp) {
matched++
if len(fl.Relocs) > 0 {
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", name, fl.Size, len(fl.Relocs))
} else {
fmt.Printf(" %s: MATCH (%d bytes)\n", name, fl.Size)
}
} else {
fmt.Printf(" %s: MISMATCH (gasm %d bytes, go %d bytes)\n", name, fl.Size, len(goCode))
for i := 0; i < len(gasmCmp) && i < len(goCmp); i++ {
if gasmCmp[i] != goCmp[i] {
fmt.Printf(" first diff at byte %d: gasm=%02x go=%02x\n", i, gasmCmp[i], goCmp[i])
break
}
}
rc = 1
}
matched, total, diffs := compareGroundTruth(k.Image(), gt)
if diffs > 0 {
printCodeDiff(k.Image(), gt)
rc = 1
}
fmt.Printf("ground truth: %d/%d functions byte-identical\n", matched, total)
if matched < total {
if matched < total || diffs > 0 {
rc = 1
}
}
@@ -1289,13 +1133,11 @@ each entry reproduces.
sigs := verify.ExtractSignatures(src)
fuzzed := 0
for _, name := range names {
sig, ok := sigs[name]
if !ok {
if _, ok := sigs[name]; !ok {
fmt.Printf(" %s: SKIP (no // func signature)\n", name)
continue
}
goCode, ok := gt[name]
if !ok {
if _, ok := gt[name]; !ok {
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", name)
continue
}
@@ -1312,8 +1154,6 @@ each entry reproduces.
rc = 1
}
}
_ = sig
_ = goCode
fuzzed++
}
fmt.Printf("fuzz: %d functions tested, %d iterations each\n", fuzzed, *fuzzN)
@@ -1338,7 +1178,7 @@ each entry reproduces.
// Run smoke and ABI checks in parallel, each function in its own child
// process: the JIT'd code runs with zeroed or fuzzed arguments, and a
// function that dereferences them faults — the crash is reported as a
// function that dereferences them faults, the crash is reported as a
// CRASH line instead of killing this process (mirrors fuzzInSubprocess).
if *smoke || *abi {
type checkResult struct {
@@ -1479,7 +1319,7 @@ func fuzzInSubprocess(path, funcName string, n int, extra ...string) string {
if exitErr, ok := err.(*exec.ExitError); ok {
ws := exitErr.Sys().(syscall.WaitStatus)
if ws.Signaled() {
return fmt.Sprintf("%s: CRASH (%v — partial function, use --ground-truth)", funcName, ws.Signal())
return fmt.Sprintf("%s: CRASH (%v; partial function, use --ground-truth)", funcName, ws.Signal())
}
}
// Non-zero exit without a signal: the fuzz reported mismatches.
@@ -1504,12 +1344,12 @@ func fuzzInSubprocess(path, funcName string, n int, extra ...string) string {
// sweepInSubprocess runs the smoke/abi checks for a single function in a
// child process. If the child is killed by a signal (e.g. SIGSEGV from a
// function that dereferences its zeroed or fuzzed arguments), it returns a
// CRASH report instead of dying — the same isolation fuzzInSubprocess
// CRASH report instead of dying, the same isolation fuzzInSubprocess
// provides for the fuzz sweep.
func sweepInSubprocess(path, funcName string, smoke, abi bool, abiN int) (string, bool) {
self, err := os.Executable()
if err != nil {
return fmt.Sprintf(" smoke/abi: FAIL — cannot find self: %v", err), true
return fmt.Sprintf(" smoke/abi: FAIL: cannot find self: %v", err), true
}
args := []string{"verify"}
if smoke {
@@ -1526,7 +1366,7 @@ func sweepInSubprocess(path, funcName string, smoke, abi bool, abiN int) (string
if exitErr, ok := err.(*exec.ExitError); ok {
ws, ok := exitErr.Sys().(syscall.WaitStatus)
if ok && ws.Signaled() {
return fmt.Sprintf(" smoke/abi: CRASH (%v — the function faults on zeroed or fuzzed\n arguments; verify it with -call and valid buffers)", ws.Signal()), true
return fmt.Sprintf(" smoke/abi: CRASH (%v: the function faults on zeroed or fuzzed\n arguments; verify it with -call and valid buffers)", ws.Signal()), true
}
}
// Non-zero exit without a signal: the checks themselves failed and
@@ -1550,7 +1390,7 @@ func sweepCheckLines(out []byte) string {
}
// runSweepChecks performs the in-process smoke and ABI checks for one
// function — the child half of sweepInSubprocess.
// function, the child half of sweepInSubprocess.
func runSweepChecks(k *verify.Kernel, path, name string, fl asm.FuncLayout, smoke, abi bool, abiN int) ([]string, bool) {
var msgs []string
failed := false
@@ -1559,7 +1399,7 @@ func runSweepChecks(k *verify.Kernel, path, name string, fl asm.FuncLayout, smok
args := make([]byte, fl.Args)
_, err := k.CallFunc(name, args)
if err != nil {
msgs = append(msgs, fmt.Sprintf(" smoke: FAIL — %v", err))
msgs = append(msgs, fmt.Sprintf(" smoke: FAIL: %v", err))
failed = true
} else {
msgs = append(msgs, " smoke: OK")
@@ -1579,7 +1419,7 @@ func runSweepChecks(k *verify.Kernel, path, name string, fl asm.FuncLayout, smok
args := make([]byte, fl.Args)
_, report, err := k.CallFuncChecked(name, args)
if err != nil {
msgs = append(msgs, fmt.Sprintf(" abi: FAIL — %v", err))
msgs = append(msgs, fmt.Sprintf(" abi: FAIL: %v", err))
failed = true
} else if !report.OK() {
msgs = append(msgs, fmt.Sprintf(" abi: %s", report))
@@ -1669,7 +1509,7 @@ func cmdVerifyCall(k *verify.Kernel, path, funcName, bufSpec, scalarSpec string,
for i := range repeat {
out, err := k.CallFunc(funcName, args)
if err != nil {
fmt.Printf(" call %d: FAIL — %v\n", i+1, err)
fmt.Printf(" call %d: FAIL: %v\n", i+1, err)
rc = 1
continue
}
+4 -3
View File
@@ -219,8 +219,9 @@ func TestCmdVersion(t *testing.T) {
if code != 0 {
t.Fatalf("code = %d", code)
}
if !strings.Contains(out, version) {
t.Errorf("version output %q does not mention %q", out, version)
got := version()
if !strings.Contains(out, got) {
t.Errorf("version output %q does not mention %q", out, got)
}
}
@@ -274,7 +275,7 @@ func TestVerifySmokeCrashIsolation(t *testing.T) {
}
if exitErr, ok := err.(*exec.ExitError); ok {
if ws, ok := exitErr.Sys().(syscall.WaitStatus); ok && ws.Signaled() {
t.Fatalf("verify died from %v — the crash was not isolated:\n%s", ws.Signal(), out)
t.Fatalf("verify died from %v; the crash was not isolated:\n%s", ws.Signal(), out)
}
}
if !strings.Contains(string(out), "CRASH") {
+1 -1
View File
@@ -19,7 +19,7 @@ import (
// file: a Go test that seeds random states, drives both the assembly kernel
// and a caller-provided portable reference, and compares the outputs
// byte-for-byte. The lesson this encodes: a pipeline-level fuzz cannot see
// an unwired kernel — only a direct-call differential against the portable
// an unwired kernel, only a direct-call differential against the portable
// specification can, so every kernel ships with one.
//
// The generated file follows two conventions the caller fills in:
+140
View File
@@ -0,0 +1,140 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"fmt"
"slices"
"strings"
)
// unifiedDiff renders a unified diff with three lines of context between the
// two line slices, in the form `gofmt -d` prints. An empty result means the
// inputs are identical.
func unifiedDiff(name string, a, b []string) string {
if slices.Equal(a, b) {
return ""
}
var out strings.Builder
fmt.Fprintf(&out, "--- %s\n+++ %s\n", name, name)
// Longest common subsequence over the lines (assembly files are small
// enough for the quadratic table).
n, m := len(a), len(b)
lcs := make([][]int, n+1)
for i := range lcs {
lcs[i] = make([]int, m+1)
}
for i := n - 1; i >= 0; i-- {
for j := m - 1; j >= 0; j-- {
if a[i] == b[j] {
lcs[i][j] = lcs[i+1][j+1] + 1
} else if lcs[i+1][j] >= lcs[i][j+1] {
lcs[i][j] = lcs[i+1][j]
} else {
lcs[i][j] = lcs[i][j+1]
}
}
}
// Walk the LCS once, assigning every op its absolute position in both
// files (1-based, the position an insertion sits before).
type op struct {
kind byte // ' ', '-' or '+'
aLine, bLine int
text string
}
var ops []op
aPos, bPos := 0, 0
emit := func(kind byte, text string) {
ops = append(ops, op{kind: kind, aLine: aPos + 1, bLine: bPos + 1, text: text})
switch kind {
case ' ':
aPos++
bPos++
case '-':
aPos++
case '+':
bPos++
}
}
i, j := 0, 0
for i < n && j < m {
switch {
case a[i] == b[j]:
emit(' ', a[i])
i++
j++
case lcs[i+1][j] >= lcs[i][j+1]:
emit('-', a[i])
i++
default:
emit('+', b[j])
j++
}
}
for ; i < n; i++ {
emit('-', a[i])
}
for ; j < m; j++ {
emit('+', b[j])
}
// Group the edits into hunks: consecutive changes separated by more than
// twice the context lines start a new hunk.
const context = 3
var changes []int
for k, o := range ops {
if o.kind != ' ' {
changes = append(changes, k)
}
}
for g := 0; g < len(changes); {
last := g
for last+1 < len(changes) && changes[last+1]-changes[last]-1 <= 2*context {
last++
}
lo := max(0, changes[g]-context)
hi := min(len(ops), changes[last]+1+context)
// The header numbers are the first line of each side actually shown:
// the first context, deletion or insertion line. A hunk that shows
// no old lines is a pure insertion and reports the position it sits
// before (0 at the top of the file); the mirror rule holds for a
// pure deletion.
aStart := ops[lo].aLine - 1
bStart := ops[lo].bLine - 1
countA, countB := 0, 0
for _, o := range ops[lo:hi] {
switch o.kind {
case ' ':
countA++
countB++
case '-':
countA++
case '+':
countB++
}
}
for _, o := range ops[lo:hi] {
if o.kind != '+' {
aStart = o.aLine
break
}
}
for _, o := range ops[lo:hi] {
if o.kind != '-' {
bStart = o.bLine
break
}
}
fmt.Fprintf(&out, "@@ -%d,%d +%d,%d @@\n", aStart, countA, bStart, countB)
for _, o := range ops[lo:hi] {
out.WriteByte(o.kind)
out.WriteString(o.text)
out.WriteByte('\n')
}
g = last + 1
}
return out.String()
}
+94
View File
@@ -0,0 +1,94 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"slices"
"strings"
"testing"
)
func lines(ss ...string) []string { return ss }
func TestUnifiedDiffIdentical(t *testing.T) {
if got := unifiedDiff("f", lines("a", "b"), lines("a", "b")); got != "" {
t.Errorf("identical inputs produced %q, want empty", got)
}
}
func TestUnifiedDiffSingleChange(t *testing.T) {
a := lines("1", "2", "3", "4", "5", "6", "7", "8")
b := lines("1", "2", "3!", "4", "5", "6", "7", "8")
want := "--- f\n+++ f\n" +
"@@ -1,6 +1,6 @@\n" +
" 1\n 2\n-3\n+3!\n 4\n 5\n 6\n"
if got := unifiedDiff("f", a, b); got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func TestUnifiedDiffInsertAtStart(t *testing.T) {
got := unifiedDiff("f", lines("x"), lines("new", "x"))
// The single existing line is shown as trailing context, so the hunk
// covers it.
want := "--- f\n+++ f\n@@ -1,1 +1,2 @@\n+new\n x\n"
if got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func TestUnifiedDiffDeleteAtEnd(t *testing.T) {
got := unifiedDiff("f", lines("x", "y"), lines("x"))
want := "--- f\n+++ f\n@@ -1,2 +1,1 @@\n x\n-y\n"
if got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func TestUnifiedDiffTwoHunks(t *testing.T) {
var a, b []string
for i := 1; i <= 20; i++ {
a = append(a, itoa(i))
b = append(b, itoa(i))
}
b[1] = "2!"
b[17] = "18!"
got := unifiedDiff("f", a, b)
if !strings.Contains(got, "@@ -1,5 +1,5 @@\n 1\n-2\n+2!\n 3\n 4\n 5\n") {
t.Errorf("first hunk wrong:\n%s", got)
}
if !strings.Contains(got, "@@ -15,6 +15,6 @@\n 15\n 16\n 17\n-18\n+18!\n 19\n 20\n") {
t.Errorf("second hunk wrong:\n%s", got)
}
}
// TestUnifiedDiffAdjacentHunks merges changes separated by exactly twice the
// context into one hunk.
func TestUnifiedDiffAdjacentHunks(t *testing.T) {
a := lines("1", "2", "3", "4", "5", "6", "7", "8")
b := slices.Clone(a)
b[0] = "1!"
b[7] = "8!"
got := unifiedDiff("f", a, b)
want := "--- f\n+++ f\n" +
"@@ -1,8 +1,8 @@\n" +
"-1\n+1!\n 2\n 3\n 4\n 5\n 6\n 7\n-8\n+8!\n"
if got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func itoa(n int) string {
if n == 0 {
return "0"
}
var buf [4]byte
i := len(buf)
for n > 0 {
i--
buf[i] = byte('0' + n%10)
n /= 10
}
return string(buf[i:])
}
+4 -5
View File
@@ -36,7 +36,7 @@ type Condition struct {
func (c *Condition) Eval(regs *Regs) bool {
actual, ok := regs.RegValue(c.Reg)
if !ok {
return true // unknown register — don't block
return true // unknown register, don't block
}
var expected uint64
switch {
@@ -48,7 +48,7 @@ func (c *Condition) Eval(regs *Regs) bool {
}
expected = v
case c.MemAddr != 0:
// Register-memory comparison — requires a Session, not available here.
// Register-memory comparison, requires a Session, not available here.
// Fall back to treating as constant (the caller should resolve).
expected = c.Value
default:
@@ -72,8 +72,7 @@ func (c *Condition) Eval(regs *Regs) bool {
}
}
// Breakpoints manages the set of breakpoints for a Session.
// Breakpoints manages software breakpoints for a debuggee.
// Breakpoints manages the software breakpoints of one Session.
type Breakpoints struct {
t tracer
bps map[uint64]*Breakpoint
@@ -200,7 +199,7 @@ func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
}
// Check the condition (if any).
if bp.Cond != nil && !bp.Cond.Eval(regs) {
// Condition not met — restore the byte but do NOT rewind RIP.
// Condition not met, restore the byte but do NOT rewind RIP.
// The process continues from the next instruction (past the INT3).
word, err := bm.t.Peek(trapAddr)
if err == nil {
+4 -5
View File
@@ -275,8 +275,7 @@ func TestBreakpointInfo(t *testing.T) {
}
func TestWatchpointSlotTracking(t *testing.T) {
wpSlots = [4]bool{} // reset
s := &Session{}
s := &Session{} // per-session slots start free
// All four slots are free initially.
for i := range 4 {
@@ -289,8 +288,8 @@ func TestWatchpointSlotTracking(t *testing.T) {
}
// Manually mark slots 0 and 2 as used (simulating successful SetWatchpoint).
wpSlots[0] = true
wpSlots[2] = true
s.wpSlots[0] = true
s.wpSlots[2] = true
if !s.IsWatchpointSlotUsed(0) {
t.Error("slot 0 should be in use")
@@ -318,7 +317,7 @@ func TestWatchpointSlotTracking(t *testing.T) {
// Mark all slots used: FindFreeWatchpointSlot returns -1.
for i := range 4 {
wpSlots[i] = true
s.wpSlots[i] = true
}
if got := s.FindFreeWatchpointSlot(); got != -1 {
t.Errorf("FindFreeWatchpointSlot() with all slots used = %d, want -1", got)
+6 -11
View File
@@ -9,27 +9,22 @@ import (
"fmt"
"strings"
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
// Read up to 15 bytes (max x86 instruction length).
mem, err := s.ReadMemory(addr, 15)
if err != nil {
// Try a shorter read if we're near a page boundary.
mem, err = s.ReadMemory(addr, 1)
if err != nil {
return "", 0, err
}
return "", 0, err
}
inst, err := x86asm.Decode(mem, 64)
ins, err := disasm.Decode(arch.AMD64, mem, addr)
if err != nil {
return "???", 1, nil
return "", 0, err
}
text := x86asm.IntelSyntax(inst, addr, nil)
return text, inst.Len, nil
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr and returns
+5 -5
View File
@@ -8,7 +8,8 @@ package debug
import (
"fmt"
"golang.org/x/arch/arm64/arm64asm"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
@@ -18,12 +19,11 @@ func (s *Session) Disassemble(addr uint64) (string, int, error) {
if err != nil {
return "", 0, err
}
inst, err := arm64asm.Decode(mem)
ins, err := disasm.Decode(arch.ARM64, mem, addr)
if err != nil {
return "???", 4, nil
return "", 0, err
}
text := arm64asm.GoSyntax(inst, addr, nil, nil)
return text, 4, nil
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr.
+5 -5
View File
@@ -8,7 +8,8 @@ package debug
import (
"fmt"
"golang.org/x/arch/loong64/loong64asm"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
@@ -18,12 +19,11 @@ func (s *Session) Disassemble(addr uint64) (string, int, error) {
if err != nil {
return "", 0, err
}
inst, err := loong64asm.Decode(mem)
ins, err := disasm.Decode(arch.LOONG64, mem, addr)
if err != nil {
return "???", 4, nil
return "", 0, err
}
text := loong64asm.GoSyntax(inst, addr, nil)
return text, 4, nil
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr.
+5 -5
View File
@@ -8,7 +8,8 @@ package debug
import (
"fmt"
"golang.org/x/arch/riscv64/riscv64asm"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
@@ -18,12 +19,11 @@ func (s *Session) Disassemble(addr uint64) (string, int, error) {
if err != nil {
return "", 0, err
}
inst, err := riscv64asm.Decode(mem)
ins, err := disasm.Decode(arch.RISCV, mem, addr)
if err != nil {
return "???", 4, nil
return "", 0, err
}
text := riscv64asm.GoSyntax(inst, addr, nil, nil)
return text, inst.Len, nil
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr.
+2 -1
View File
@@ -22,7 +22,8 @@ type Session struct {
cmd *exec.Cmd
stopped bool
exited bool
codeBase uint64 // base address of the JIT code in the debuggee
codeBase uint64 // base address of the JIT code in the debuggee
wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 BADVR0-15)
}
// Launch starts the debuggee subprocess (gasm debug --target ...) and
+7 -10
View File
@@ -20,14 +20,11 @@ const (
WatchRead WatchpointType = 3 // trigger on read or write
)
// wpSlots tracks watchpoint slot occupancy (DR0-DR3).
var wpSlots [4]bool
// FindFreeWatchpointSlot returns the index of the first free watchpoint slot
// (0-3), or -1 if all four hardware watchpoints are in use.
func (s *Session) FindFreeWatchpointSlot() int {
for i := range 4 {
if !wpSlots[i] {
if !s.wpSlots[i] {
return i
}
}
@@ -39,7 +36,7 @@ func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot > 3 {
return false
}
return wpSlots[slot]
return s.wpSlots[slot]
}
// SetWatchpoint installs a hardware watchpoint on the given address.
@@ -47,7 +44,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
if slot < 0 || slot > 3 {
return fmt.Errorf("debug: watchpoint slot must be 0-3")
}
if wpSlots[slot] {
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
@@ -96,7 +93,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
return fmt.Errorf("debug: set DR7: %w", err)
}
wpSlots[slot] = true
s.wpSlots[slot] = true
return nil
}
@@ -105,7 +102,7 @@ func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot > 3 {
return fmt.Errorf("debug: watchpoint slot must be 0-3")
}
if !wpSlots[slot] {
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
dr7, err := ptracePeekUser(s.pid, 0x38)
@@ -116,14 +113,14 @@ func (s *Session) ClearWatchpoint(slot int) error {
if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
return err
}
wpSlots[slot] = false
s.wpSlots[slot] = false
return nil
}
// ClearAllWatchpoints removes all hardware watchpoints.
func (s *Session) ClearAllWatchpoints() error {
for slot := range 4 {
if wpSlots[slot] {
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
}
+7 -10
View File
@@ -22,9 +22,6 @@ const (
WatchRead WatchpointType = 3
)
// wpSlots tracks watchpoint slot occupancy.
var wpSlots [16]bool // arm64 supports up to 16 watchpoints
const maxWatchpoints = 16
// hwBreakState mirrors the kernel's struct user_hwdebug_state.
@@ -45,7 +42,7 @@ const (
func (s *Session) FindFreeWatchpointSlot() int {
for i := range maxWatchpoints {
if !wpSlots[i] {
if !s.wpSlots[i] {
return i
}
}
@@ -56,7 +53,7 @@ func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot >= maxWatchpoints {
return false
}
return wpSlots[slot]
return s.wpSlots[slot]
}
// SetWatchpoint installs a hardware watchpoint on the given address.
@@ -64,7 +61,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
}
if wpSlots[slot] {
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
@@ -105,7 +102,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
return fmt.Errorf("debug: set watchpoint: %w", err)
}
wpSlots[slot] = true
s.wpSlots[slot] = true
return nil
}
@@ -113,7 +110,7 @@ func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
}
if !wpSlots[slot] {
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
@@ -126,13 +123,13 @@ func (s *Session) ClearWatchpoint(slot int) error {
if err := s.setHWBreakState(state); err != nil {
return err
}
wpSlots[slot] = false
s.wpSlots[slot] = false
return nil
}
func (s *Session) ClearAllWatchpoints() error {
for slot := 0; slot < maxWatchpoints; slot++ {
if wpSlots[slot] {
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
}
+7 -10
View File
@@ -21,14 +21,11 @@ const (
WatchRead WatchpointType = 3
)
// wpSlots tracks watchpoint slot occupancy.
var wpSlots [4]bool
const maxWatchpoints = 4
func (s *Session) FindFreeWatchpointSlot() int {
for i := range maxWatchpoints {
if !wpSlots[i] {
if !s.wpSlots[i] {
return i
}
}
@@ -39,7 +36,7 @@ func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot >= maxWatchpoints {
return false
}
return wpSlots[slot]
return s.wpSlots[slot]
}
// SetWatchpoint installs a hardware watchpoint.
@@ -47,7 +44,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
}
if wpSlots[slot] {
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
if size != 1 && size != 2 && size != 4 && size != 8 {
@@ -85,7 +82,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
return fmt.Errorf("debug: set watchpoint control: %w", err)
}
wpSlots[slot] = true
s.wpSlots[slot] = true
return nil
}
@@ -93,20 +90,20 @@ func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
}
if !wpSlots[slot] {
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), 0); err != nil {
return err
}
wpSlots[slot] = false
s.wpSlots[slot] = false
return nil
}
func (s *Session) ClearAllWatchpoints() error {
for slot := 0; slot < maxWatchpoints; slot++ {
if wpSlots[slot] {
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
}
+7 -10
View File
@@ -21,14 +21,11 @@ const (
WatchRead WatchpointType = 3
)
// wpSlots tracks watchpoint slot occupancy.
var wpSlots [4]bool
const maxWatchpoints = 4
func (s *Session) FindFreeWatchpointSlot() int {
for i := range maxWatchpoints {
if !wpSlots[i] {
if !s.wpSlots[i] {
return i
}
}
@@ -39,7 +36,7 @@ func (s *Session) IsWatchpointSlotUsed(slot int) bool {
if slot < 0 || slot >= maxWatchpoints {
return false
}
return wpSlots[slot]
return s.wpSlots[slot]
}
// SetWatchpoint installs a hardware watchpoint.
@@ -47,7 +44,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
}
if wpSlots[slot] {
if s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
}
if size != 1 && size != 2 && size != 4 && size != 8 {
@@ -87,7 +84,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
return fmt.Errorf("debug: set watchpoint control: %w", err)
}
wpSlots[slot] = true
s.wpSlots[slot] = true
return nil
}
@@ -95,7 +92,7 @@ func (s *Session) ClearWatchpoint(slot int) error {
if slot < 0 || slot >= maxWatchpoints {
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
}
if !wpSlots[slot] {
if !s.wpSlots[slot] {
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
}
@@ -103,13 +100,13 @@ func (s *Session) ClearWatchpoint(slot int) error {
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), 0); err != nil {
return err
}
wpSlots[slot] = false
s.wpSlots[slot] = false
return nil
}
func (s *Session) ClearAllWatchpoints() error {
for slot := 0; slot < maxWatchpoints; slot++ {
if wpSlots[slot] {
if s.wpSlots[slot] {
if err := s.ClearWatchpoint(slot); err != nil {
return err
}
+95
View File
@@ -0,0 +1,95 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package disasm decodes machine code back to instruction text for the four
// architectures gasm assembles. It is a thin, platform-independent wrapper
// over golang.org/x/arch and backs both the `gasm dis` command and the live
// debugger views.
package disasm
import (
"fmt"
"golang.org/x/arch/arm64/arm64asm"
"golang.org/x/arch/loong64/loong64asm"
"golang.org/x/arch/riscv64/riscv64asm"
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
)
// Instruction is one decoded instruction: its text form, its length in bytes
// and the address it was decoded at.
type Instruction struct {
Addr uint64
Text string
Len int
}
// Decode decodes the instruction at the start of code, located at addr.
// code needs to hold at least the one instruction being decoded (amd64 may
// consume up to 15 bytes). Undecodable bytes yield the placeholder text "???"
// and a length of one word (four bytes, one on amd64) so that a listing can
// keep making progress, mirroring the debugger's behaviour.
func Decode(a arch.Arch, code []byte, addr uint64) (Instruction, error) {
if len(code) == 0 {
return Instruction{}, fmt.Errorf("disasm: empty input")
}
switch a {
case arch.ARM64:
if len(code) < 4 {
return Instruction{}, fmt.Errorf("disasm: need 4 bytes, have %d", len(code))
}
inst, err := arm64asm.Decode(code)
if err != nil {
return Instruction{Addr: addr, Text: "???", Len: 4}, nil
}
return Instruction{Addr: addr, Text: arm64asm.GoSyntax(inst, addr, nil, nil), Len: 4}, nil
case arch.RISCV:
// The compressed extensions are decoded transparently; a 16-bit
// instruction only needs its two bytes.
inst, err := riscv64asm.Decode(code)
if err != nil {
return Instruction{Addr: addr, Text: "???", Len: 2}, nil
}
return Instruction{Addr: addr, Text: riscv64asm.GoSyntax(inst, addr, nil, nil), Len: inst.Len}, nil
case arch.LOONG64:
if len(code) < 4 {
return Instruction{}, fmt.Errorf("disasm: need 4 bytes, have %d", len(code))
}
inst, err := loong64asm.Decode(code)
if err != nil {
return Instruction{Addr: addr, Text: "???", Len: 4}, nil
}
return Instruction{Addr: addr, Text: loong64asm.GoSyntax(inst, addr, nil), Len: 4}, nil
default: // amd64
inst, err := x86asm.Decode(code, 64)
if err != nil {
return Instruction{Addr: addr, Text: "???", Len: 1}, nil
}
return Instruction{Addr: addr, Text: x86asm.IntelSyntax(inst, addr, nil), Len: inst.Len}, nil
}
}
// Block decodes up to max instructions from code starting at addr and returns
// them in order. Decoding stops at the end of code or once an instruction
// would run past it.
func Block(a arch.Arch, code []byte, addr uint64, max int) []Instruction {
var out []Instruction
pc := 0
for len(out) < max && pc < len(code) {
ins, err := Decode(a, code[pc:], addr+uint64(pc))
if err != nil {
break
}
if ins.Len <= 0 || pc+ins.Len > len(code) {
break
}
out = append(out, ins)
pc += ins.Len
}
return out
}
+141
View File
@@ -0,0 +1,141 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package disasm
import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
func TestDecodeKnownBytes(t *testing.T) {
for _, tt := range []struct {
a arch.Arch
code []byte
text string
want int
}{
{arch.AMD64, []byte{0x55}, "push rbp", 1},
{arch.AMD64, []byte{0x48, 0x89, 0xE5}, "mov rbp, rsp", 3},
{arch.ARM64, []byte{0xc0, 0x03, 0x5f, 0xd6}, "RET", 4},
{arch.RISCV, []byte{0x67, 0x80, 0x00, 0x00}, "RET", 4},
{arch.LOONG64, []byte{0x20, 0x00, 0x00, 0x4c}, "RET", 4},
} {
ins, err := Decode(tt.a, tt.code, 0)
if err != nil {
t.Errorf("%s: %v", tt.a, err)
continue
}
if ins.Text != tt.text || ins.Len != tt.want {
t.Errorf("%s: % x decoded to %q (%d bytes), want %q (%d)",
tt.a, tt.code, ins.Text, ins.Len, tt.text, tt.want)
}
}
}
func TestDecodeUndecodable(t *testing.T) {
// Zero words do not encode a usable instruction on arm64 and loong64; the
// placeholder keeps a listing going. RISC-V is the exception: an all-zero
// word is the defined UNIMP instruction.
for _, a := range []arch.Arch{arch.ARM64, arch.LOONG64} {
ins, err := Decode(a, []byte{0, 0, 0, 0}, 0)
if err != nil {
t.Fatalf("%s: %v", a, err)
}
if ins.Text != "???" {
t.Errorf("%s: text = %q, want ???", a, ins.Text)
}
}
// The compressed quadrant claims the zero halfword first, so the zero
// word decodes as the 2-byte compressed UNIMP.
if ins, err := Decode(arch.RISCV, []byte{0, 0, 0, 0}, 0); err != nil || ins.Text != "UNIMP" || ins.Len != 2 {
t.Errorf("riscv zero word: %q len %d err %v, want UNIMP with 2 bytes", ins.Text, ins.Len, err)
}
if _, err := Decode(arch.ARM64, []byte{0, 0}, 0); err == nil {
t.Error("short input: expected an error")
}
if _, err := Decode(arch.AMD64, nil, 0); err == nil {
t.Error("empty input: expected an error")
}
}
// TestBlockRoundTrip assembles a small kernel with the gasm encoder for every
// architecture and disassembles it back: the listing must cover the whole
// function and end in RET.
func TestBlockRoundTrip(t *testing.T) {
for _, tt := range []struct {
a arch.Arch
name string
}{
{arch.AMD64, "k_amd64.s"},
{arch.ARM64, "k_arm64.s"},
{arch.RISCV, "k_riscv64.s"},
{arch.LOONG64, "k_loong64.s"},
} {
src := "TEXT \u00b7k(SB), NOSPLIT, $0\n\tMOVQ AX, CX\n\tRET\n"
if tt.a != arch.AMD64 {
src = "TEXT \u00b7k(SB), NOSPLIT, $0\n\tRET\n"
}
f, errs := parser.Parse(tt.name, src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.a, errs)
}
img, err := assemble(t, tt.a, f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.a, err)
}
fn := img.Funcs[0]
code := img.Code[fn.Offset : fn.Offset+fn.Size]
ins := Block(tt.a, code, 0, 100)
if len(ins) == 0 {
t.Fatalf("%s: empty listing", tt.a)
}
consumed := 0
for _, in := range ins {
if in.Text == "" || in.Text == "???" {
t.Errorf("%s: undecoded instruction at %#x: %q", tt.a, in.Addr, in.Text)
}
consumed += in.Len
}
if consumed != len(code) {
t.Errorf("%s: listing consumed %d of %d bytes", tt.a, consumed, len(code))
}
if last := ins[len(ins)-1]; !strings.Contains(strings.ToLower(last.Text), "ret") {
t.Errorf("%s: last instruction = %q, want RET", tt.a, last.Text)
}
}
}
func TestBlockLimits(t *testing.T) {
code := []byte{0x55, 0x55, 0x55, 0x55, 0x55}
if got := Block(arch.AMD64, code, 0, 3); len(got) != 3 {
t.Errorf("max=3 produced %d instructions, want 3", len(got))
}
if got := Block(arch.AMD64, code, 0, 100); len(got) != 5 {
t.Errorf("code end produced %d instructions, want 5", len(got))
}
if got := Block(arch.AMD64, nil, 0, 3); len(got) != 0 {
t.Errorf("empty code produced %d instructions, want 0", len(got))
}
}
// assemble assembles the parsed file with the encoder for a.
func assemble(t *testing.T, a arch.Arch, f *ast.File) (*asm.Image, error) {
t.Helper()
switch a {
case arch.ARM64:
return asm.AssembleFileARM64(f)
case arch.RISCV:
return asm.AssembleFileRISCV(f)
case arch.LOONG64:
return asm.AssembleFileLOONG64(f)
default:
return asm.AssembleFile(f)
}
}
+110 -13
View File
@@ -4,7 +4,9 @@ How gasm-devkit is put together and why.
Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
## Design goals
## Overview
Three design goals shape everything below.
1. **A real AST, not a grammar hack.** The linter, analyser, assembler and
language server all need to *reason* about assembly, not just colour it.
@@ -20,10 +22,10 @@ Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrb
through two vendor-neutral interfaces: a CLI and an LSP server. No editor
owns the toolkit; the toolkit is offered to editors on standard terms.
## Pipeline
The components, and how data moves between them:
```mermaid
graph TD
flowchart TD
SRC["source .s"] --> LEX["lexer<br/>token stream"]
LEX --> PAR["parser<br/>AST + diagnostics"]
LEX --> FMT["format<br/>re-space tokens"]
@@ -42,9 +44,34 @@ graph TD
The lexer is the shared foundation: the parser builds the AST from it, the
formatter re-spaces its tokens directly, and the language server uses it for
semantic highlighting.
semantic highlighting. The phases follow a dependency chain: Phase 1 (static
analysis) builds only on the AST, Phase 2 (the standalone assembler) emits
object code, and Phases 3 (dynamic analysis) and 4 (the debugger) both consume
the execution substrate that the assembler provides.
## Components
## Packages
| Package | Responsibility |
|---|---|
| `token` | token kinds and positions |
| `lexer` | hand-written scanner; permissive, and it never panics |
| `ast` | the typed syntax tree: declarations, lines, operands |
| `parser` | line-oriented parser producing the AST and its diagnostics |
| `arch` | register and instruction tables for the four architectures |
| `lint` | static checks over the AST |
| `format` | canonical formatter over the token stream |
| `lsp` | the language server |
| `asm` | standalone assembler: encoders, image layout, object emitters |
| `verify` | JIT execution, ABI checks, differential fuzzing |
| `debug` | interactive ptrace debugger |
| `cmd/gasm` | the CLI |
| `_gen` | rebuilds the `arch` tables from the Go toolchain source |
The boundaries matter as much as the responsibilities: `ast` records syntax
only, and whether a name is a register or a label is left to `arch`, so the
parser stays architecture-agnostic. `asm` and `verify` are the only packages
that touch machine code and executable memory, and `cmd/gasm` owns no logic
beyond flags and output.
### `token` and `lexer`
@@ -206,7 +233,7 @@ The standalone assembler (Phase 2). Its core is an amd64 instruction encoder:
a REX/ModR-M/SIB/displacement/immediate engine plus the scalar instruction set,
with the Plan 9 operand order (source first) mapped onto the x86 encoding.
Every encoding is validated by decoding it again with `golang.org/x/arch`, the
one module dependency, used in tests only and never linked into the binary.
one module dependency, which also backs the `gasm dis` listings.
A **RISC-V encoder** (Phase 5, RV64IMAFDC + RVC compression) encodes the full
integer, atomic, float/double, FMA and CSR instruction sets with the MOV
@@ -237,6 +264,18 @@ STP+SUB for large frames) and SB/global symbol references (ADRP+ADD pairs with
R_ADDRARM64 relocations). Like the other encoders it is validated
byte-for-byte against `GOARCH=arm64 go tool asm`.
On top of the per-architecture encoders, every framed function carries the
**stack-split guard**: the prologue check against `g.stackguard0` (small,
medium and large frame classes, the medium and large classes materialising
their offset through the architecture's temporary register and the large
class adding the SP-underflow branch) and the trailing morestack block
(save the link register, `CALL runtime.morestack_noctxt`, jump back to the
function entry). The auto-NOSPLIT rule, the frame classes, the large-frame
prologue and epilogue forms and the tail calls match the toolchain's
`stacksplit` and `preprocess` output byte for byte; a parity suite
assembles kernel files with gasm and the installed `go tool asm` and diffs
the bytes on all four architectures.
On top of the encoder, `Assemble` walks a parsed `TEXT` body, converts each
operand to an encoder operand, and lays the instructions out so local labels
resolve to relative jump offsets: jumps start in the short (rel8) form and
@@ -371,8 +410,8 @@ the Go ABI fixes across calls (amd64 `BP`/`R14`, arm64 `R29`/`R28`, riscv64
raw return trampoline `leaveJITCheckedRaw` verifies them, restoring the
saved registers before Go code resumes. riscv64 is validated end to
end under qemu-user emulation; arm64 shares the same stack convention and
fix; loong64 stays ground-truth-only until hardware validation (see
docs/DECISIONS.md). `gasm verify` runs the JIT checks when the host
fix; loong64 stays ground-truth-only until hardware validation.
`gasm verify` runs the JIT checks when the host
matches the kernel's architecture and the toolchain comparisons
elsewhere.
@@ -424,14 +463,72 @@ watchdog is armed before the ptrace attach, so a sandboxed debuggee cannot
block it), and `--cover` runs to completion with a breakpoint on every
label and reports which blocks executed.
## Extension points
### Extending the toolkit
- **New architecture:** add an entry to the generator in `_gen`, run
`just gen`, and add a `buildXXX()` register file plus a case in `ForArch`.
- **New lint rule:** add a function in `lint` and a rule-code constant.
- **New LSP feature:** add a method case in `dispatch` and a handler.
The phases follow a dependency chain. Phase 1 (static analysis) builds only on
the AST; Phase 2 (the standalone assembler) emits object code; Phases 3
(dynamic analysis) and 4 (the debugger) both consume the execution substrate
that the assembler provides.
## Data flow
The main operation, assembling one file:
```mermaid
sequenceDiagram
participant User
participant CLI as gasm CLI
participant Parser as parser
participant Asm as asm
participant Go as go toolchain
User->>CLI: gasm asm --format goobj -p pkg -o k.o k_amd64.s
CLI->>Parser: Parse(path, src)
Parser-->>CLI: AST, diagnostics
CLI->>Asm: AssembleFile(AST)
Asm->>Asm: encode operands, settle label offsets, lay out data
Asm-->>CLI: Image, code and data and relocations
CLI->>Asm: GOObject(pkg, path)
Asm->>Go: go list -json -export, externals only
Go-->>Asm: package and symbol indices
Asm-->>CLI: Go object bytes
CLI-->>User: wrote N bytes to k.o
```
Errors are produced where the parse or the encoding fails and become values at
the CLI boundary: the parser returns a diagnostic list and never aborts a file,
`AssembleFile` returns an error, and `cmd/gasm` prints what it has to stderr
and returns a non-zero exit code. The formatter and the linter take the same
AST by a different route: `gasm fmt` re-spaces the token stream and `gasm lint`
walks the parsed file, so neither depends on an encoding.
## State and lifetime
- The analysis packages (`lexer`, `parser`, `format`, `lint`, `arch`) hold only
read-only lookup tables and no mutable state: every call allocates its own
tokens and AST, and any number of goroutines may read the `arch` tables.
- A `verify.Kernel` owns one executable mapping, which `Close` releases. The
JIT trampolines keep the Go stack pointer and the checked-call sentinels in
package globals, so a call is a process-wide, one-at-a-time operation. The
`gasm verify` sweeps therefore run each function in a child process, which
contains a crash and keeps the globals unshared.
- `lsp.Server` is long-lived: it runs a single read and dispatch loop over the
stream and touches its document store only from that loop, so one server
serves one connection.
- A `debug.Session` owns a traced child process and pins its goroutine to the
forking OS thread, because ptrace requests must stay on that thread.
## Dependencies
- **`golang.org/x/arch`** (v0.30.0) is the one module dependency: it is the
disassembler backend (`gasm dis` and the debugger's listings) and the source
of the register metadata the encoder consults (`asm/reg.go`, `asm/vex.go`).
The tests additionally decode through it to validate the encodings.
- **The Go toolchain**, as an oracle and never as a library: `go tool asm`
supplies the object preamble and the ground truth for `gasm verify
--ground-truth`, `go list -json -export` locates the archives of the packages
a GOOBJ object references, and `_gen` parses
`$GOROOT/src/cmd/internal/obj/<arch>/anames.go` to rebuild the tables.
- **Linux process interfaces** for the dynamic work: `mmap` and `mprotect` for
the JIT mapping, ptrace with `/proc/pid/mem` for the debugger. That is why
`verify` runs a JIT check only when the host architecture matches the
kernel's, and why `debug` is Linux-only.
+391 -149
View File
@@ -1,198 +1,440 @@
# CLI Reference
# Command line
Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
The reference below is taken from the program's own `--help`. If the two disagree, the
program is right and this file is a defect.
`gasm` is a single binary with subcommands. Run `gasm --help` for an
overview, or `gasm <command> -h` for a command's usage and flags.
## Synopsis
## Global Flags
```sh
gasm [global flags] <command> [command flags] [arguments]
```
| Flag | Description |
|------|-------------|
| `-h`, `--help` | Show help |
| `-V`, `--version` | Print the version |
## Commands
## `gasm tokens <file>`
| Command | Purpose |
|---|---|
| `tokens` | print the lexical token stream |
| `parse` | parse a file and report syntax errors |
| `fmt` | canonicalise the formatting of `.s` files |
| `lint` | run the static checks |
| `asm` | assemble `.s` files to machine code |
| `dis` | disassemble machine code or an assembled file |
| `verify` | JIT-assemble and run the dynamic checks |
| `debug` | interactive source-level debugger |
| `diff` | compare the machine code of two `.s` files |
| `profile` | show the basic-block structure of the functions |
| `audit-instructions` | diff the encoder against the toolchain's name table |
| `scaffold` | generate a differential test skeleton for a kernel |
| `lsp` | run the language server over stdio |
| `version` | print the version |
Print the lexical token stream of FILE: position, token kind, and text,
one token per line. FILE may be `-` to read standard input.
## tokens
## `gasm parse <file>`
```text
Usage: gasm tokens <file>
```
Parse FILE and report syntax errors on stderr. On success, prints how
many declarations and TEXT functions the file contains.
Print the lexical token stream of FILE: position, token kind and text, one
token per line. FILE may be `-` to read standard input.
## `gasm fmt [-w] [path...]`
```sh
gasm tokens hello_amd64.s
```
Canonicalise the formatting of Plan 9 assembly sources: indentation,
operand spacing, per-function mnemonic alignment, and blank-line layout.
```text
1:1 # "#"
1:2 IDENT "include"
1:10 STRING "\"textflag.h\""
```
| Flag | Description |
|------|-------------|
| `-w` | Write result to the source file (default: print to stdout) |
## parse
With no arguments, or with a directory argument, every `.s` file below
it is reformatted in place and the names of changed files are listed
(`go fmt` style). `.` and `_` directories are skipped.
```text
Usage: gasm parse <file>
```
## `gasm lint <file...>`
Parse FILE and report syntax errors on stderr. On success, print how many
declarations and TEXT functions the file contains.
Run static checks and print diagnostics as
`file:line:col: severity: message [code]`. Exit status is non-zero when
an error-severity diagnostic is found.
```sh
gasm parse hello_amd64.s
```
| Flag | Description |
|------|-------------|
| `-disable` | Comma-separated rule codes to disable |
```text
hello_amd64.s: OK, 2 declarations, 1 functions
```
## fmt
```text
Usage: gasm fmt [-w|-l|-d] [path...]
```
| Flag | Default | Effect |
|---|---|---|
| `-w` | off | write the result back to the source file |
| `-l` | off | list the files whose formatting differs; write nothing |
| `-d` | off | print a unified diff of the canonical formatting instead |
`-l` and `-d` are mutually exclusive. With no arguments, or with a directory
argument, every `.s` file below it is reformatted in place and the names of the
changed files are listed, the way `go fmt` does; `.` and `_` directories are
skipped. Explicit file arguments print to stdout unless `-w` is given.
```sh
gasm fmt -l kernel_amd64.s
```
Empty output means every file is formatted, which is the shape a CI check
wants; `-d` shows what would change:
```sh
gasm fmt -d ugly_amd64.s
```
```text
--- ugly_amd64.s
+++ ugly_amd64.s
@@ -2,8 +2,8 @@
// func add(a, b int) int
TEXT ·add(SB), NOSPLIT, $0-24
- MOVQ a+0(FP), AX
- ADDQ b+8(FP), AX
+ MOVQ a+0(FP), AX
+ ADDQ b+8(FP), AX
```
## lint
```text
Usage: gasm lint <file...>
```
| Flag | Default | Effect |
|---|---|---|
| `-disable` | empty | comma-separated rule codes to disable |
Diagnostics are printed as `file:line:col: severity: message [code]`. The exit
status is non-zero when an error-severity diagnostic is found; warnings (the
register-clobber audit, for example) do not affect it.
Rules: `unknown-instruction`, `operand-count`, `undefined-label`,
`duplicate-label`, `missing-ret`, `missing-textflag-include`,
`abi-argsize`, `unreachable-code`, `register-clobber`,
`funcdata-pcdata`, `unused-label`, `invalid-textflag`,
`stack-imbalance`, `register-width-mismatch`, `abi0-register-args`,
`nonportable-register-name` and `unencodable-instruction`.
`nonportable-register-name`, `unencodable-instruction` and
`reserved-register-write`.
## `gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>`
```sh
gasm lint kernel_amd64.s
```
Assemble FILE to machine code (amd64, arm64, riscv64, loong64).
## asm
| Flag | Description |
|------|-------------|
| `--format` | Output format: `raw` (default), `elf`, `goobj` |
| `-p` | Package path (required for `--format goobj`) |
| `-o` | Write output to file (default: hex dump to stdout) |
```text
Usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>
```
## `gasm verify [flags] <file.s>`
| Flag | Default | Effect |
|---|---|---|
| `-format` | `raw` | output format: `raw` (concatenated image), `elf` or `goobj` (Go object) |
| `-p` | empty | package path for `--format goobj`, qualifying the exported symbols |
| `-o` | empty | write the output to this file instead of a hex dump on stdout |
Assemble FILE, map it into executable memory, and run dynamic checks.
Supported architectures: amd64 (VEX/AVX2 and EVEX/AVX-512 included), arm64,
riscv64 (RV64IMAFDC and RVC) and loong64, selected from the file's `_arch.s`
suffix. `raw` concatenates the functions and the data section into one
self-consistent image; `elf` emits a relocatable object that links with the
system toolchain; `goobj` emits the Go toolchain's own object format, which
`cmd/link` consumes directly.
| Flag | Description |
|------|-------------|
| `--ground-truth` | Compare machine code byte-for-byte against `go tool asm` |
| `--fuzz` | Differential fuzz: JIT both gasm and go-tool-asm, compare outputs |
| `-n` | Fuzz iterations per function (default: 1000) |
| `--abi` | Run ABI-checking calls (sentinel registers + red zone) |
| `--abi-n` | Number of ABI check iterations with varied inputs (default: 100) |
| `--profile` | List basic-block structure per function |
| `--smoke` | Call each NOSPLIT function with zeroed args |
| `--call <func>` | Invoke a single function with `--buf` instead of the sweeps |
| `--buf <spec>` | Buffer spec for `--call`: `name:size:pattern[,name:size:pattern]` |
| `--args <spec>` | Scalar args for `--call`: `name=value[,name=value]` (decimal or `0x` hex) |
| `--repeat <n>` | Number of times to repeat a `--call` invocation (default: 1) |
| `--save-corpus <dir>` | With `--fuzz`: write each failing input to DIR as replayable JSON |
| `--replay <dir>` | Re-run saved corpus entries (JSON in DIR), one child process per entry |
```sh
gasm asm hello_amd64.s
```
The `--fuzz` mode runs each function in a subprocess; a partial function
(e.g. a decoder that faults on malformed input) is reported as
`CRASH` without killing the parent. Use `--call` with `--buf` to invoke
partial functions with valid data instead.
```text
add: 16 bytes
0000: 48 8b 44 24 08 48 03 44 24 10 48 89 44 24 18 c3
```
The `--call` mode parses the `// func` signature, allocates the requested
buffers (`zero`, `ones`, `seq`, or a hex blob), builds the ABI0 argument
block with buffer pointers/lengths/capacities at the matching parameter
offsets, and prints the arg block before and after the call, showing
return values and any output written to the buffers. Scalar parameters
are supplied with `--args` (decimal, or `0x` hex) at their ABI0 offsets.
## dis
The `--save-corpus` mode records the logical arguments (buffer contents and
scalars, not raw pointers) of every failing fuzz input as JSON. `--replay`
rebuilds a live argument block from each entry and calls it in its own child
process, reporting `OK`, `CRASH (reproduced)` or `FAIL` per entry and
exiting non-zero when any entry fails.
```text
Usage: gasm dis [-a arch] <file>
```
## `gasm debug [--func <name>] [--buf spec] [--script file] <file.s>`
| Flag | Default | Effect |
|---|---|---|
| `-a` | empty | architecture for raw input without a `_arch.s` name |
Interactive debugger for JIT-assembled functions (amd64, arm64, riscv64,
loong64). Requires a compiled binary on `$PATH` (not `go run`).
With a `.s` file the file is assembled first and the listing follows the real
layout: one block per `TEXT` function, local labels printed at their offsets.
With any other file, or `-` for standard input, the bytes are disassembled
linearly and `-a` selects the architecture (amd64, arm64, riscv64 or loong64).
| Flag | Description |
|------|-------------|
| `--func` | Function to debug (required) |
| `--buf` | Buffer spec: `name:size:pattern[,name:size:pattern]` |
| `--args <file>` | File containing the ABI0 argument block |
| `--script <file>` | Run REPL commands from a file (one per line) and exit; `-` reads stdin |
| `--timeout <dur>` | Kill the debuggee after this duration (e.g. `30s`); for headless `--script` runs |
| `--cover` | Run to completion with a breakpoint on every instruction; report which executed, how often, and which labels were reached |
```sh
gasm dis hello_amd64.s
```
```text
add: 16 bytes
0000: 48 8b 44 24 08 mov rax, qword ptr [rsp+0x8]
0005: 48 03 44 24 10 add rax, qword ptr [rsp+0x10]
000a: 48 89 44 24 18 mov qword ptr [rsp+0x18], rax
000f: c3 ret
```
## verify
```text
Usage: gasm verify [-smoke] [-abi] [-fuzz] [-ground-truth] [-profile] [-call] <file.s>
```
| Flag | Default | Effect |
|---|---|---|
| `--ground-truth` | off | compare the machine code byte-for-byte against `go tool asm` |
| `--fuzz` | off | differential fuzz against the `go tool asm` build |
| `-n` | 1000 | fuzz iterations per function |
| `--abi` | off | ABI-checking calls: sentinel registers and a red-zone canary |
| `--abi-n` | 100 | ABI check iterations with varied inputs |
| `--profile` | off | list the basic-block structure per function |
| `--smoke` | off | call each NOSPLIT function with zeroed arguments |
| `--call` | empty | invoke a single function with `--buf` instead of the sweeps |
| `--buf` | empty | buffer spec for `--call`: `name:size:pattern[,name:size:pattern]` |
| `--args` | empty | scalar args for `--call`: `name=value[,name=value]` (decimal or `0x` hex) |
| `--repeat` | 1 | number of times to repeat a `--call` invocation |
| `--save-corpus` | empty | with `--fuzz`: write each failing input to this directory as replayable JSON |
| `--replay` | empty | re-run saved corpus entries, one child process per entry |
The JIT checks run when the host matches the file's architecture; the
toolchain comparison works everywhere. `--fuzz`, `--smoke` and `--abi` run each
function in its own child process, so a partial function that faults on random
input is reported as `CRASH` instead of ending the sweep; `--call` with `--buf`
invokes such a function with valid data. loong64 stays on the ground-truth path
until hardware validation.
```sh
gasm verify --ground-truth hello_amd64.s
```
```text
hello_amd64.s: 1 functions JIT-loaded
add: MATCH (16 bytes)
ground truth: 1/1 functions byte-identical
add: 16 bytes, args=24, frame=0 NOSPLIT
```
```sh
gasm verify --call add --args a=2,b=3 hello_amd64.s
```
```text
add: 16 bytes, args=24
signature: func add(a int, b int) int
scalars:
a = 2
b = 3
args before: 02 00 00 00 00 00 00 00 03 00 00 00 00 00 00 00 00 00 00 00 00 00 00 00 (24 bytes)
args after: 02 00 00 00 00 00 00 00 03 00 00 00 00 00 00 00 05 00 00 00 00 00 00 00 (24 bytes)
call 1: OK
```
## debug
```text
Usage: gasm debug <file.s> --func <name>
```
| Flag | Default | Effect |
|---|---|---|
| `-func` | empty | the function to debug, required |
| `-buf` | empty | buffer spec: `name:size:pattern[,name:size:pattern]` (zero, ones, seq or hex) |
| `-args` | empty | file containing the ABI0 argument block |
| `-script` | empty | run REPL commands from a file, one per line, and exit; `-` reads stdin |
| `-timeout` | 0 | kill the debuggee after this duration, for headless `-script` runs |
| `-cover` | off | run to completion with a breakpoint on every instruction and report which executed |
The debugger spawns the debuggee from the `gasm` binary on `$PATH`, so install
it first with `just install`; `go run` does not work for the traced child.
Requires Linux (ptrace) and all four architectures are supported.
REPL commands:
| Command | Description |
|---------|-------------|
| `break <label\|addr> [if <reg> <op> <val>]` | Set a breakpoint, optionally conditional |
| `delete <label\|addr>` | Remove a breakpoint |
| `info break` | List all breakpoints |
| `step [n]`, `s` | Single-step n instructions |
| `next`, `n` | Step over CALL |
| `finish`, `fin` | Run until the function returns |
| `continue`, `c` | Run until breakpoint, watchpoint or exit |
| `disas [n]`, `u` | Disassemble n instructions at PC |
| `regs` | Print general-purpose + vector/FP registers |
| `where` | Show source line and nearest label at PC |
| `stack` | Show stack near RSP (return address + ABI0 args) |
| `bt`, `backtrace` | Backtrace (current frame + return address) |
| `x [addr] [len]` | Hex-dump memory |
| `w <addr> <val...>` | Write bytes to memory |
| `set <reg> <value>` | Set a register |
| `watch <addr> [r\|w] [size]` | Set a hardware watchpoint (write by default) |
| `unwatch [<slot>]` | Clear one or all watchpoints |
| `labels`, `l` | List function labels and offsets |
| `help`, `h`, `?` | Show command help |
| `quit`, `q` | Kill the debuggee and exit |
| Command | Effect |
|---|---|
| `break <label\|addr> [if <reg> <op> <val>]` | set a breakpoint, optionally conditional |
| `delete <label\|addr>` | remove a breakpoint |
| `info break` | list the breakpoints |
| `step [n]`, `s` | single-step n instructions |
| `next`, `n` | step over a CALL |
| `finish`, `fin` | run until the function returns |
| `continue`, `c` | run until a breakpoint, watchpoint or exit |
| `disas [n]`, `u` | disassemble n instructions at the PC |
| `regs` | print the general-purpose and vector/FP registers |
| `where` | show the source line and the nearest label at the PC |
| `stack` | show the stack near RSP, the return address and the ABI0 args |
| `bt`, `backtrace` | backtrace: the current frame and the return address |
| `x [addr] [len]` | hex-dump memory |
| `w <addr> <val...>` | write bytes to memory |
| `set <reg> <value>` | set a register |
| `watch <addr> [r\|w] [size]` | set a hardware watchpoint, write by default |
| `unwatch [<slot>]` | clear one watchpoint or all of them |
| `labels`, `l` | list the function's labels and offsets |
| `help`, `h`, `?` | show the command help |
| `quit`, `q` | kill the debuggee and exit |
## `gasm diff [--map old=new,...] <file1.s> <file2.s>`
```sh
gasm debug --func add --cover hello_amd64.s
```
Compare the machine code produced by assembling two files. Shows which
functions differ and the first few differing bytes. Useful for verifying
that two implementations produce identical code, or for tracking encoding
changes between Go assembler versions.
## diff
| Flag | Description |
|------|-------------|
| `--map` | Comma-separated `old=new` pairs to match functions with different names |
```text
Usage: gasm diff <file1.s> <file2.s>
```
Without `--map`, functions are paired by exact name. With `--map`, a
function named `old` in the first file is compared against the function
named `new` in the second file (e.g. `--map wideCopyAVX2=wideCopyAVX512`
pairs AVX2 and AVX-512 variants regardless of suffix).
| Flag | Default | Effect |
|---|---|---|
| `-map` | empty | comma-separated `old=new` pairs to match functions with different names |
## `gasm profile <file.s>`
Functions are paired by exact name unless `--map` says otherwise, so
`--map wideCopyAVX2=wideCopyAVX512` pairs two variants regardless of suffix.
The exit status is non-zero when anything differs.
Show the basic-block structure of functions in an assembly file. Lists
each function's labels, their offsets, and the block boundaries. This is
the static structure; for runtime execution counts, use `gasm verify
--fuzz` which exercises the code paths.
```sh
gasm diff hello_amd64.s hello_amd64.s
```
## `gasm audit-instructions [amd64|arm64|riscv64|loong64]`
```text
add: identical (16 bytes)
all functions identical
```
Compare the gasm encoder for the given architecture (default amd64)
against the installed `go tool asm` and print the diff: superset
encodings (gasm-only spellings, shippable via `gasm asm --format goobj`),
known-but-unencodable names (the encoder backlog) and go-only names
(feature gaps). The Go side is probed black-box with a battery of operand
shapes per mnemonic, so the audit tracks whatever toolchain
`go env GOROOT` provides. On non-amd64 architectures the backlog is an
over-approximation: a name counts as encodable only when a probe shape
assembles cleanly, so a name whose real forms the battery misses lands
in the backlog.
## profile
## `gasm scaffold differential <file.s>`
```text
Usage: gasm profile <file.s>
```
Print a differential test skeleton for every `// func` signature in
FILE. The generated test seeds random states, drives the kernel and a
portable reference (`<name>Portable`), and compares outputs
byte-for-byte. Write the reference bodies, place the file in the
kernel's package, and run it in CI.
Show the basic-block structure of each function: its labels, their offsets and
the block boundaries. This is the static structure; for runtime execution
counts use `gasm debug --cover`, and for input coverage `gasm verify --fuzz`.
## `gasm lsp`
```sh
gasm profile hello_amd64.s
```
Run the language server over standard input/output (JSON-RPC 2.0 with
Content-Length framing). Point an LSP-capable editor at the binary and
associate it with `.s` files. The target architecture is inferred from
the file-name suffix (`_amd64.s`, `_arm64.s`, `_riscv64.s`,
`_loong64.s`).
```text
add: 16 bytes, args=24, frame=0 NOSPLIT
basic blocks: 1
```
Provides: completion, hover, document symbols, push and pull
diagnostics, semantic tokens, go-to-definition, find references, rename,
document formatting, inlay hints, code actions, signature help, document
highlights, workspace symbol search, #include document links, and
folding ranges for function bodies.
## audit-instructions
```text
Usage: gasm audit-instructions [amd64|arm64|riscv64|loong64]
```
Compare the gasm encoder for the given architecture (default amd64) against the
installed `go tool asm` and print the diff: superset encodings (gasm-only
spellings, shippable via `gasm asm --format goobj`), known-but-unencodable
names (the encoder backlog) and go-only names (feature gaps). The Go side is
probed black-box with a battery of operand shapes per mnemonic, so the audit
tracks whatever toolchain `go env GOROOT` provides. On non-amd64
architectures the backlog is an over-approximation: a name counts as encodable
only when a probe shape assembles cleanly, so a name whose real forms the
battery misses lands in the backlog.
```sh
gasm audit-instructions amd64
```
```text
gasm table (amd64, families excluded): 1542 mnemonics
gasm encodable: 580 go tool asm recognized: 1542
shared: 580
```
## scaffold
```text
Usage: gasm scaffold differential <file.s>
```
Print a differential test skeleton for every `// func` signature in FILE. The
generated test seeds random states, drives the kernel and a portable reference
(`<name>Portable`), and compares the outputs byte-for-byte. Write the reference
bodies, place the file in the kernel's package, and run it in CI.
```sh
gasm scaffold differential kernel_amd64.s > kernel_differential_test.go
```
## lsp
```text
Usage: gasm lsp
```
Run the language server over standard input/output, JSON-RPC 2.0 with
`Content-Length` framing. Point an LSP-capable editor at the binary and
associate it with `.s` files; the target architecture is inferred from the
file-name suffix (`_amd64.s`, `_arm64.s`, `_riscv64.s`, `_loong64.s`).
Provides: completion, hover, document symbols, push and pull diagnostics,
semantic tokens, go-to-definition, find references, rename, document
formatting, inlay hints, code actions, signature help, document highlights,
workspace symbol search, #include document links, and folding ranges for
function bodies. Definition, references and rename work across every open
document.
## version
```text
Usage: gasm version
```
Print the version the toolchain recorded for the build, the same string as
`gasm --version`: the tag on a tagged checkout, a pseudo-version naming the
commit below one, with `+dirty` appended on a dirty tree and `(devel)` outside
version control.
## Global flags
| Flag | Default | Effect |
|---|---|---|
| `-h`, `--help` | off | print the usage |
| `-V`, `--version` | off | print the version |
## Exit codes
| Code | Meaning |
|---|---|
| `0` | success |
| `1` | a failure the program detected: a parse or assembly error, an error-severity lint diagnostic, a mismatch in `verify`, a file that cannot be read |
| `2` | the arguments were wrong: a missing or extra argument, an unknown command or format, an invalid `--map` pair |
## Examples
Assemble a kernel, check it, and run it:
```sh
gasm lint kernel_amd64.s
gasm fmt -l kernel_amd64.s
gasm asm -o kernel.bin kernel_amd64.s
gasm verify --ground-truth kernel_amd64.s
```
Link the kernel into a Go program through the toolchain's own object format:
```sh
gasm asm --format goobj -p example.com/kernel -o kernel.o kernel_amd64.s
```
Find which labels a failing kernel reaches, headlessly:
```sh
gasm debug --func decodeBlockAVX2 --cover --script cmds.txt --timeout 30s kernel_amd64.s
```
-111
View File
@@ -1,111 +0,0 @@
# Deferred decisions
Design decisions deliberately postponed, with enough context to pick them up
again without re-deriving the analysis. Each entry records what is deferred,
why, the options on the table, and the trigger that should reopen it.
---
## GOOBJ external (cross-package) symbol references
**Status:** resolved (v0.29.0+, 2026-08-07).
**Approach taken.** Instead of parsing the compiler's iexport data (which
would have required either `golang.org/x/tools` or an in-house parser), the
resolver reads the **GOOBJ data directly** from the target package's `.a`
archive. The `.a` file contains a `_go_.o` member whose GOOBJ s is the
same one gasm writes; the parser reuses the same layout (`blkSymdef`,
`blkNonpkgdef`, the string table), so no new dependency was needed.
**How it works.**
1. `go list -json -export <pkg>` finds the target package's `.a` file.
2. `extractGOOBJ` reads the ar archive, finds the `_go_.o` member, skips
the `"go object …\n!\n"` preamble and parses the GOOBJ header.
3. `goobjFile.symbols()` walks `blkSymdef` and `blkNonpkgdef` in definition
order (the same order the linker uses) to build the symbol-to-index
mapping.
4. `resolveExternalSymbols` wires the resolved `{PkgIdx, SymIdx}` into the
GOOBJ emission.
The resolver is invoked automatically when `img.Externals` is non-empty; it
runs `go list` as a subprocess (consistent with `toolchainObjectPreamble`
which already calls `go tool asm`). All symbol data is cached per package
for the lifetime of the GOOBJ emission.
## 2026-08-30 non-amd64 JIT execution trampolines
**Status:** resolved for riscv64 (validated end to end under qemu-user)
and arm64 (fix in place, consistent with the observed frame convention);
open for loong64 until hardware validation.
**Root cause (found 2026-08-31).** The trampolines advanced SP past the
leave-address slot after loading it, while the assembled kernels read
their first argument at SP+8 per the frame convention (the amd64 path
already kept SP on that slot). Removing the advance fixed riscv64
immediately (plain and checked ABI tests pass under qemu-user); the
arm64 kernel's pre-fix trace showed exactly the same SP+8 reading. The
apparent arm64/loong64 "crashes in the JIT" turned out to be dominated
by an unrelated instability: the Go 1.26 and 1.27 runtimes crash under
qemu-user arm64 emulation (GC worker start, identical signature with the
JIT tests skipped, both qemu 7.2 and 10.2), and the Go loong64 runtime
does not start at all. `gasm verify` therefore keeps loong64 kernels on
the ground-truth path until hardware validation; the GOARCH-guarded
tests (`verify/jit_arch_test.go`, `verify/abi_arch_test.go`) are the
hardware validation entry point.
**State.** The per-architecture trampolines compile for all targets, the
kernels they execute are byte-for-byte correct against `go tool asm`, and
under `qemu-aarch64` the arm64 kernel demonstrably executes and stores its
result correctly. The failure is on the return path into Go code: arm64
and loong64 take a SIGSEGV after the kernel's RET (the Go-side unwind
through `leaveJIT` and its interposed ABIInternal wrapper is the suspect),
and riscv64 returns cleanly but with an untouched result area. amd64 is
unaffected (the checked trampoline saves and restores BP/R14 and the flow
is validated end to end).
**Evidence harness.** `verify/jit_arch_test.go` (plain call) and
`verify/abi_arch_test.go` (checked call) are GOARCH-guarded tests; build
the test binary per target (`GOARCH=arm64 go test -c -o v.test ./verify/`)
and run it under `qemu-aarch64-static` from the `verify/` directory. A
minimal reproducer pattern lives in the qemu exploration notes: verify
loads, the kernel executes, the fault follows the return.
**Fix direction.** Compare the amd64 checked trampoline (GLOBL/DATA raw
address, explicit SP/BP/R14 save-restore) against the arm64/riscv64/
loong64 `leaveJIT` unwind, in particular the interaction with the
ABIInternal wrapper that `reflect.ValueOf(leaveJIT).Pointer()` returns.
The plain-call path (no sentinels) fails the same way, so the checked
path is not the variable.
---
## 2026-08-29 tooling round
- `lint abi0-register-args`: flags kernels whose `// func` parameters are
never read from the FP frame. Motivated by a real latent bug: kernels
reading arguments from registers pass every test while the autogenerated
`F.abi0` wrapper happens to leave the caller's register values intact, and
break on a toolchain upgrade.
- `lint nonportable-register-name`: the RAX/EAX register spellings are a gasm
extension; go tool asm rejects them, so files using them only link through
the gasm goobj path.
- `lint unencodable-instruction`: a mnemonic in the architecture table that
`asm.Encodable` rejects is flagged at edit time instead of failing at
assembly time.
- `audit-instructions`: black-box diff of the encoder against go tool asm.
As of this round the tables fully overlap on names; the audit exists to
catch drift in both directions (future supersets and future gaps).
- `scaffold differential`: generates the direct-call differential skeleton
(two independent seed sets, output and in-place buffer comparison) that a
pipeline-level fuzz can never replace.
- `verify --args`: scalar arguments for `-call`, closing the repro gap where
only buffers could be supplied.
- `debug --script/--timeout/--cover`: headless debugging with a watchdog
armed before the ptrace attach (untracing sandboxes hang the attach), and
label-level block coverage for the "did my test ever enter that branch"
question.
- Superset policy remains: gasm may accept spellings and encodings go tool
asm lacks, but such kernels ship only via `gasm asm --format goobj`; the
audit reports the superset surface. The register-alias superset is warned
about by lint because the default `go build` path cannot consume it.
+88 -66
View File
@@ -4,79 +4,80 @@ Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrb
## Prerequisites
- **Go** 1.27+ with `toolchain go1.27.0`
- **Go** 1.27.1, the exact version the `go` directive in `go.mod` declares
- **just**, the command runner; every task below is a just recipe
- A Linux host on amd64, arm64, riscv64 or loong64: `gasm debug` needs ptrace
and the JIT checks of `gasm verify` need executable memory
- No external dependencies beyond the Go toolchain
## Quick Start
## Setup
```sh
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
cd gasm-devkit
just install # go mod download
just build # go vet + gofmt, must pass with zero output
just test # full suite, race detector, 80 % coverage gate
just build # compile bin/gasm, zero errors and zero warnings
just gates # build, fmt-check, vet, test, race: the definition of done
```
## Just Recipes
## Recipes
### `just install`
Every recipe in the `justfile`, and what it does.
`go mod download`. The only module dependency, `golang.org/x/arch`, is
used in tests only.
### `just build`
Runs `go vet ./...` and checks `gofmt -l .` produces no output. This is
the minimum bar before any commit.
| Recipe | What it does |
|---|---|
| `just build` | compiles `bin/gasm` with `CGO_ENABLED=0` and stripped symbols; zero errors and zero warnings |
| `just test` | the test gate: the suite with `-count=1`, the coverage profile and the 80 % floor |
| `just race` | the same suite under the race detector; the expensive one, so it runs once, inside `gates` |
| `just unit [packages] [run]` | fast, cached, scoped run for iterating: no race and no coverage, so an unchanged package reports instantly |
| `just fuzz <target> <pkg> [fuzztime]` | time-boxed fuzz of one target; the package is required, because `go test -fuzz` refuses more than one |
| `just bench [packages]` | benchmarks (`-benchmem -count=5`); on an idle machine only |
| `just fmt` | formats the tree in place with `gofmt` |
| `just fmt-check` | zero diff; prints nothing when everything is formatted, which is the shape the CI step wants |
| `just vet` | both static gates: `go vet` and `go fix -diff` |
| `just gates` | `build`, `fmt-check`, `vet`, `test` and `race`, in that order: the definition of done |
| `just clean` | removes the build artefacts, `bin/` and `coverage.out` |
| `just install` | builds, then copies the binary into `bindir` (`~/.local/bin`); `gasm debug` needs an installed binary, because it spawns the debuggee from `$PATH` |
| `just uninstall` | removes the installed binary from `bindir` |
| `just run` | runs the CLI with `go run -buildvcs=true`; the recipe takes no arguments, so flags go through the package instead |
| `just dev` | the same as `run`; the project has no watcher to add |
| `just gen` | regenerates the `arch` instruction tables from the Go toolchain source; not a gate |
### `just test`
```sh
go test -race -count=1 ./...
go test -count=1 -timeout 10m -coverprofile=coverage.out \
./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... \
./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
```
Plus a coverage run over the ten analysable packages (arch, asm, ast,
format, lexer, lint, lsp, parser, token, verify; `debug` and `cmd/gasm`
need hardware or are CLI glue) and an `awk` gate that fails if total
coverage is below 80 %.
The suite runs over the logic packages (`-count=1`, so no cached pass
counts): arch, asm, ast, disasm, format, lexer, lint, lsp, parser,
token, verify. `debug` traces a live process and `cmd/gasm` is thin CLI
glue, so both sit outside the sweep, and a thin `cmd/` in it would drag
the coverage total under the floor. The floor fails if the total is
below 80 %. CI runs the same command with the same ten-minute bound, so
the number is the same everywhere.
### `just fmt`
### `just run`
```sh
gofmt -w .
just run
go run -buildvcs=true ./cmd/gasm lint kernel_amd64.s
go run -buildvcs=true ./cmd/gasm verify --ground-truth kernel_amd64.s
```
Run after editing any Go source. The output must be idempotent.
### `just run -- <args>`
Runs the CLI via `go run` with the version string stamped:
```sh
just run -- lint kernel_amd64.s
just run -- fmt -w kernel_amd64.s
just run -- verify --ground-truth kernel_amd64.s
```
### `just install-bin`
Installs the `gasm` binary into `$GOBIN` with the release version
embedded via `-ldflags "-X main.version=..."`.
The flag on `go run` is there because it does not stamp the build otherwise,
which `--version` would then report as `(devel)`.
### `just gen`
Regenerates the architecture instruction tables in `arch/` by parsing
the Go toolchain's own assembler source
Regenerates the architecture instruction tables in `arch/` by parsing the Go
toolchain's own assembler source
(`$GOROOT/src/cmd/internal/obj/<arch>/anames.go`). Requires a Go
installation. Output is committed, with no runtime dependency on the
toolchain.
### `just uninstall`
Removes `coverage.out`, the `gasm` binary, and `*.test` artefacts.
## Running Individual Tests
## Running a single test
```sh
go test -run TestVexGroundTruth ./asm/
@@ -85,32 +86,53 @@ go test -run TestGOObjectLinkAndRun ./asm/
go test -run TestFuzzWideCopy ./verify/
```
## Debugger Note
Add `-v` for the sub-test names, and `-race` when the change touches
concurrency. `-count=1` defeats the test cache when a result looks stale.
`gasm debug` spawns a child process from the binary on `$PATH`. It does
not work with `go run`; install first:
## Coverage
```sh
just install-bin
gasm debug --func decodeBlockAVX2 path/to/kernel_amd64.s
just test
go tool cover -func=coverage.out
```
## Project Layout
The `total:` line is the number that matters, and it stays at 80 percent or
more.
## Debugging the build
```sh
go build -gcflags='-m' ./... # inlining decisions
go build -gcflags='-S' ./... # what the compiler generated
go tool asm -S kernel_amd64.s # how the toolchain's assembler encodes a kernel
gasm dis kernel_amd64.s # what gasm makes of the same kernel
gasm tokens kernel_amd64.s # the token stream
gasm profile kernel_amd64.s # the basic blocks of each function
```
cmd/gasm/ CLI entry point (subcommands)
token/ Lexical token kinds and positions
lexer/ Hand-written scanner
ast/ Abstract syntax tree
parser/ Line-oriented parser
arch/ Register and instruction tables (generated)
lint/ Static analysis rules
format/ Canonical formatter
lsp/ Language Server Protocol server
asm/ Standalone assembler, encoder, object emitters
verify/ JIT execution, differential testing, ABI checks
debug/ Interactive ptrace debugger (all four architectures)
_gen/ Instruction table generator
testdata/ Test fixtures
docs/ Architecture, development, CLI reference
```
`gasm verify --ground-truth` is the differential check that ties the two
together: it compares gasm's bytes with `go tool asm`'s, with the relocation
sites masked, so an encoding drift shows up as a byte difference rather than a
crash later.
## Continuous integration
Workflows live in `.gitea/workflows/` and run on the project's own runners:
Test on a push or pull request to `development`, race dispatched by hand, and
the release on a `v*` tag. They are written by hand rather than through
`just`, but they enforce the same set of gates minus the race detector, which
the shared runner cannot afford on a push; a green `just gates` locally is
therefore the fastest way to a green pipeline.
## Releases
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`,
which triggers the release workflow: it builds the portable Linux targets,
takes the notes from the matching `CHANGELOG.md` section and uploads the
assets.
The version is never injected. `gasm --version` prints what the
toolchain recorded in the build information: the tag on a tagged
checkout, a pseudo-version naming the commit below one, `+dirty` on a
dirty tree, and `(devel)` outside version control. There is no
`-ldflags "-X"` anywhere and no version constant in the source.
+4 -4
View File
@@ -1,7 +1,7 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package format implements a canonical formatter for GAsm source — the
// Package format implements a canonical formatter for GAsm source, the
// equivalent of gofmt for Plan 9 assembly. It works on the token stream
// rather than the AST so that every line (including comments and blanks) is
// preserved; it only normalises indentation, operand spacing and per-function
@@ -92,7 +92,7 @@ func Source(src string) string {
case kInstr:
out = renderInstr(line, maxWidth[inf.funcID])
// A RET ends the body for indentation purposes: comments that
// follow it — typically the next function's doc comment — belong
// follow it, typically the next function's doc comment, belong
// at column 0, not inside the finished function.
if strings.EqualFold(line[0].Text, "RET") {
inBody = false
@@ -120,8 +120,8 @@ type outLine struct {
}
// normalizeSpacing enforces the canonical blank-line layout: runs of blank
// lines collapse to one, and a new block — a label, or a TEXT or GLOBL
// directive — is preceded by exactly one blank line. Comments immediately
// lines collapse to one, and a new block, a label, or a TEXT or GLOBL
// directive, is preceded by exactly one blank line. Comments immediately
// above a block belong to it, so the blank line is inserted before them. No
// blank line is forced at the top of the file, right after a TEXT (the
// function's first label), or between stacked labels that share an address.
+1 -1
View File
@@ -39,7 +39,7 @@ func TestGolden(t *testing.T) {
}
// TestDocCommentIndent checks that a doc comment preceding a TEXT directive
// sits at column 0 even when another function (ending in RET) precedes it —
// sits at column 0 even when another function (ending in RET) precedes it;
// the RET must terminate the previous body for indentation purposes.
func TestDocCommentIndent(t *testing.T) {
in := "#include \"textflag.h\"\n" +
+1 -3
View File
@@ -1,7 +1,5 @@
module sourcedock.dev/petrbalvin/gasm-devkit
go 1.27
toolchain go1.27.0
go 1.27.1
require golang.org/x/arch v0.30.0
+80 -40
View File
@@ -1,58 +1,98 @@
# Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
# SPDX-License-Identifier: BSD-3-Clause
# gasm-devkit.
#
# Replace gasm-devkit here and the values in the variable block. Everything below the block
# is the standard set and is identical in every repository; see the `justfile` skill.
binary := "gasm"
package := "./cmd/gasm"
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
# What the test and bench recipes sweep. Scope this to the logic packages when a thin
# cmd/ drags the coverage floor down, for example "./internal/... ./pkg/...". Never name
# a directory the project does not have: a pattern that matches nothing is a setup
# failure, not an empty run.
packages := "./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/..."
version := "0.32.0"
bindir := env_var_or_default("BINDIR", env_var("HOME") / ".local" / "bin")
default:
@just --list
# Download module dependencies.
install:
go mod download
# Vet + gofmt check — zero errors, zero warnings.
# Compile. Zero errors, zero warnings.
build:
go vet ./...
@test -z "$(gofmt -l .)" || { echo "gofmt diff:"; gofmt -l .; exit 1; }
CGO_ENABLED=0 go build -ldflags "-s -w" -o bin/{{binary}} {{package}}
# Full test suite + race detector + 80 % coverage gate.
# The coverage gate matches CI: it excludes packages that need hardware or
# are CLI glue (debug, cmd/gasm), so the number is identical locally and in CI.
# The test gate: the suite, no cache, the coverage floor.
test:
go test -race -count=1 ./...
go test -count=1 -coverprofile=coverage.out \
sourcedock.dev/petrbalvin/gasm-devkit/arch \
sourcedock.dev/petrbalvin/gasm-devkit/asm \
sourcedock.dev/petrbalvin/gasm-devkit/ast \
sourcedock.dev/petrbalvin/gasm-devkit/format \
sourcedock.dev/petrbalvin/gasm-devkit/lexer \
sourcedock.dev/petrbalvin/gasm-devkit/lint \
sourcedock.dev/petrbalvin/gasm-devkit/lsp \
sourcedock.dev/petrbalvin/gasm-devkit/parser \
sourcedock.dev/petrbalvin/gasm-devkit/token \
sourcedock.dev/petrbalvin/gasm-devkit/verify
go tool cover -func=coverage.out | awk '/^total:/{gsub("%","",$3);if($3+0<80){print "coverage "$3"% < 80%";exit 1}print "coverage "$3"%"}'
#!/usr/bin/env perl
system(q{go}, q{test}, q{-count=1}, q{-timeout}, q{10m},
q{-coverprofile}, q{coverage.out}, qw({{packages}})) == 0
or die qq{the test suite failed\n};
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
my $total;
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
close($c);
die qq{no total line in coverage.out\n} unless defined $total;
printf qq{Total coverage: %s%%\n}, $total;
exit($total < 80 ? 1 : 0);
# Format all Go sources.
# The same suite under the race detector. The expensive one.
race:
go test -race -count=1 -timeout 10m {{packages}}
# Fast scoped run for iterating. This is the one that runs after every edit.
unit pkgs=packages run=".*":
go test {{pkgs}} -run '{{run}}'
# Time-boxed fuzz of one target in one package. The package is required; never a gate.
fuzz target pkg fuzztime="60s":
go test -run '^$' -fuzz '{{target}}' -fuzztime={{fuzztime}} {{pkg}}
# Benchmarks. On an idle machine only.
bench pkgs=packages:
go test -run '^$' -bench=. -benchmem -count=5 {{pkgs}}
# Format in place.
fmt:
gofmt -w .
# Run the gasm CLI (pass args after --, e.g. `just run -- lint file.s`).
run *ARGS:
go run -ldflags "-X main.version={{version}}" ./cmd/gasm {{ARGS}}
# Zero diff. Prints nothing when everything is formatted.
fmt-check:
#!/usr/bin/env perl
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
my @bad = <$g>;
close($g);
print @bad;
exit(@bad ? 1 : 0);
# Install the gasm binary into $GOBIN (stamped with the release version).
install-bin:
go install -ldflags "-X main.version={{version}}" ./cmd/gasm
# Both static gates: go vet and go fix -diff.
vet:
go vet ./...
go fix -diff ./...
# Regenerate the architecture instruction tables from the Go toolchain source.
# The definition of done, in one command. Once per task, never per edit.
gates: build fmt-check vet test race
# Build artefacts only, not the installed binary.
clean:
rm -rf bin/ coverage.out
# Build, then copy the binary into bindir.
install: build
install -d "{{bindir}}"
install -m 755 bin/{{binary}} "{{bindir}}/{{binary}}"
# Remove the installed binary.
uninstall:
rm -f "{{bindir}}/{{binary}}"
# Run the program. The flag is there because `go run` does not stamp the build otherwise.
run:
go run -buildvcs=true {{package}}
# Run with watch or hot reload, where the project has one.
dev:
go run -buildvcs=true {{package}}
# Regenerate the architecture instruction tables from the Go toolchain source. Not a gate.
gen:
go run _gen/gen.go
gofmt -w arch/
# Remove build artefacts.
uninstall:
rm -f coverage.out gasm
find . -name '*.test' -delete
+3 -3
View File
@@ -12,8 +12,8 @@ import (
// checkFuncdata validates the structure of FUNCDATA and PCDATA directives,
// which carry the GC stack-map information. The checks are deliberately
// shallow — they confirm the operands are well formed and that a literal index
// is within the small range the runtime uses — and never try to interpret a
// shallow, they confirm the operands are well formed and that a literal index
// is within the small range the runtime uses, and never try to interpret a
// named index constant such as $PCDATA_StackMapIndex.
func checkFuncdata(t *ast.Text, cfg Config) []Diagnostic {
var out []Diagnostic
@@ -89,7 +89,7 @@ func checkIndex(op *ast.Operand, directive string) []Diagnostic {
if op.Imm.HasVal && (op.Imm.Val < 0 || op.Imm.Val > 10) {
return []Diagnostic{{
Pos: op.Pos, Severity: Warning, Code: CodeFuncdata,
Message: fmt.Sprintf("%s index %d is outside the valid range 0–10", directive, op.Imm.Val),
Message: fmt.Sprintf("%s index %d is outside the valid range 0-10", directive, op.Imm.Val),
}}
}
return nil
+1 -1
View File
@@ -20,7 +20,7 @@ import (
// architecture's syntax (macros, addressing modes, branch aliases) against
// production assembly. It is skipped when the toolchain source is absent.
//
// The bar is zero parse errors and zero error-severity diagnostics — i.e. no
// The bar is zero parse errors and zero error-severity diagnostics; i.e. no
// false "unknown instruction" / "undefined label" findings on code the real
// assembler accepts. Advisory warnings are reported but not fatal, since they
// are heuristics that may legitimately differ across Go versions.
+22 -21
View File
@@ -10,6 +10,7 @@ package lint
import (
"fmt"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
@@ -84,10 +85,12 @@ const (
CodeReservedRegister = "reserved-register-write"
)
// knownTextFlags are the flags recognised by the Go assembler's textflag.h.
// knownTextFlags are the flags recognised by runtime/textflag.h, plus the
// older REFLECTED spelling of REFLECTMETHOD.
var knownTextFlags = map[string]bool{
"NOSPLIT": true, "DUPOK": true, "RODATA": true, "NOPROF": true,
"WRIT": true, "TLSBSS": true, "NOFRAME": true, "REFLECTED": true,
"NOPTR": true, "WRAPPER": true, "NEEDCTXT": true, "TLSBSS": true,
"NOFRAME": true, "REFLECTED": true, "REFLECTMETHOD": true,
"TOPFRAME": true, "ABIWRAPPER": true,
}
@@ -172,8 +175,9 @@ func File(f *ast.File, cfg Config) []Diagnostic {
}
}
for _, fl := range flags {
// Numeric flags (1, 8, 9) are legacy Go toolchain constants.
if fl >= "0" && fl <= "9" {
// Numeric flags are legacy textflag.h constants (1, 2, 8,
// 9, 10, …); their meaning is decided at assembly time.
if _, err := strconv.Atoi(fl); err == nil {
continue
}
if !knownTextFlags[fl] {
@@ -242,7 +246,7 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
// Unreachable code: a real instruction following a RET/UNDEF and
// before any label, in a function whose control flow is fully
// resolvable. Only RET/UNDEF are treated as terminators here — an
// resolvable. Only RET/UNDEF are treated as terminators here, an
// unconditional jump may be one entry of a hand-arranged branch
// table (e.g. the generated callback tables), so it is not assumed
// to make the following code dead. Pseudo-ops and macro invocations
@@ -373,8 +377,10 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
}
// Missing RET heuristic. Functions that invoke a macro are skipped: the
// macro body (opaque to us) may supply the RET.
if doLabelChecks && !cfg.Disable[CodeMissingRet] && instrCount > 0 && !hasRet && !lastTerminal && !hasMacro {
// macro body (opaque to us) may supply the RET. A TEXT whose symbol is
// missing (already reported by the parser) is skipped too.
if doLabelChecks && !cfg.Disable[CodeMissingRet] && t.Name != nil &&
instrCount > 0 && !hasRet && !lastTerminal && !hasMacro {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
@@ -506,7 +512,7 @@ func reachesRuntime(t *ast.Text) bool {
}
// usesFPArgs reports whether a function references its arguments through the FP
// pseudo-register — i.e. it uses the stack-based ABI0 layout, where the
// pseudo-register, i.e. it uses the stack-based ABI0 layout, where the
// declared argument size must match the signature.
func usesFPArgs(t *ast.Text) bool {
for _, s := range t.Body {
@@ -572,7 +578,7 @@ func isMacroInvocation(mnem string, macros map[string]bool) bool {
// maskedEvex reports whether the instruction is a masked EVEX form: the
// mnemonic carries a .Z suffix, or the operand list contains an opmask
// register (K1–K7). Either way the operand count differs from the unmasked
// register (K1-K7). Either way the operand count differs from the unmasked
// form, so count checks are skipped.
func maskedEvex(mnem string, ops []*ast.Operand) bool {
if strings.Contains(mnem, ".") {
@@ -587,7 +593,7 @@ func maskedEvex(mnem string, ops []*ast.Operand) bool {
return false
}
// isMaskReg reports whether name is an opmask register K0–K7.
// isMaskReg reports whether name is an opmask register K0-K7.
func isMaskReg(name string) bool {
return len(name) == 2 && name[0] == 'K' && name[1] >= '0' && name[1] <= '7'
}
@@ -702,11 +708,6 @@ func stackDelta(t *ast.Text, a arch.Arch) int64 {
}
case arch.ARM64:
switch upper {
case "STP":
// STP with pre-index: STP Xt1, Xt2, [SP, #imm]!
if len(in.Operands) >= 3 && isSPReg(in.Operands[2], a) {
// Could be pre-index decrement; skip for simplicity.
}
case "SUB":
if len(in.Operands) >= 3 && isSPReg(in.Operands[2], a) {
if in.Operands[1].Imm.HasVal {
@@ -770,10 +771,10 @@ func isSPReg(op *ast.Operand, a arch.Arch) bool {
// checkRegisterWidth detects amd64 register-width mismatches. The naming
// truth of the Go assembler governs: AX, BX, CX, DX, SI, DI, BP, SP and
// R8–R15 ARE the 64-bit register names (there are no separate EAX/RAX
// spellings in go tool asm), and AL–DH are the byte forms. The width comes
// R8-R15 ARE the 64-bit register names (there are no separate EAX/RAX
// spellings in go tool asm), and AL-DH are the byte forms. The width comes
// from the opcode suffix, so an L/W operation over a canonical 64-bit name is
// the normal, correct spelling — flagging it is pure noise on real kernels.
// the normal, correct spelling, flagging it is pure noise on real kernels.
// What remains worth flagging: a Q (64-bit) operation over a narrower spelled
// register (EAX under the gasm alias extension, or a byte form), and byte
// registers in L/W operations.
@@ -812,8 +813,8 @@ func checkRegisterWidth(mnem string, ops []*ast.Operand) string {
}
// amd64RegWidth returns the width in bytes of an amd64 register name under
// the Go assembler's naming model: the canonical word names (AX…SP, R8–R15)
// are 64-bit, AL–DH are the 8-bit forms, and the R/E-prefixed spellings are
// the Go assembler's naming model: the canonical word names (AX…SP, R8-R15)
// are 64-bit, AL-DH are the 8-bit forms, and the R/E-prefixed spellings are
// the gasm alias extension with their intuitive widths.
func amd64RegWidth(name string) int {
switch name {
@@ -833,7 +834,7 @@ func countRange(min, max int) string {
if min == max {
return fmt.Sprintf("%d operand(s)", min)
}
return fmt.Sprintf("%d–%d operands", min, max)
return fmt.Sprintf("%d-%d operands", min, max)
}
// sortDiagnostics orders diagnostics by line, then column, then code.
+7 -7
View File
@@ -48,7 +48,7 @@ func TestFixtureIsClean(t *testing.T) {
t.Fatalf("parse: %v", errs)
}
// The fixture mirrors the go-flac kernels, which write the Go ABI0
// scratch registers (BX, R13) without saving them — legal under Go's
// scratch registers (BX, R13) without saving them; legal under Go's
// stack-based ABI, so the register-clobber audit stays silent and the
// fixture must lint entirely clean.
diags := File(f, Config{Arch: arch.AMD64})
@@ -186,8 +186,8 @@ done:
}
}
// TestEvexMaskingRecognised checks that masked EVEX forms — the .Z suffix and
// an explicit K operand — are recognised and exempt from operand-count
// TestEvexMaskingRecognised checks that masked EVEX forms; the .Z suffix and
// an explicit K operand; are recognised and exempt from operand-count
// checks.
func TestEvexMaskingRecognised(t *testing.T) {
diags := lintSrc(t, `
@@ -284,7 +284,7 @@ TEXT ·f(SB), NOSPLIT|NOFRAME|DUPOK, $0
}
func TestStackImbalance(t *testing.T) {
// Function with frame size 16 but only SUB 8, SP — imbalance.
// Function with frame size 16 but only SUB 8, SP; imbalance.
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $16-0
@@ -297,7 +297,7 @@ TEXT ·f(SB), NOSPLIT, $16-0
}
func TestStackBalanced(t *testing.T) {
// Function with frame size 16 and matching SUB/ADD — balanced.
// Function with frame size 16 and matching SUB/ADD; balanced.
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $16-0
@@ -311,7 +311,7 @@ TEXT ·f(SB), NOSPLIT, $16-0
}
func TestRegisterWidthMismatch(t *testing.T) {
// MOVQ with 32-bit register — mismatch.
// MOVQ with 32-bit register; mismatch.
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
@@ -324,7 +324,7 @@ TEXT ·f(SB), NOSPLIT, $0
}
func TestRegisterWidthCorrect(t *testing.T) {
// MOVQ with 64-bit registers — correct.
// MOVQ with 64-bit registers; correct.
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
+2 -2
View File
@@ -370,7 +370,7 @@ func sameSet(a, b map[string]bool) bool {
}
// goFixedGPRs returns the general-purpose registers the Go ABI designates as
// fixed across calls — the ones hand-written assembly must not permanently
// fixed across calls, the ones hand-written assembly must not permanently
// clobber. This follows cmd/compile/abi-internal.md, not the platform ABI:
// Go's stack-based ABI0 (which hand-written assembly uses) has no System V
// style callee-saved registers, so clobbering the argument and scratch
@@ -414,7 +414,7 @@ func gprSet(names ...string) map[string]bool {
// without also saving and restoring them. The first result lists registers
// whose loss is never safe; the second lists the goroutine-pointer class,
// whose loss is reported only when reachesRuntime is true (a non-NOSPLIT
// function, or one that makes calls — the ABI0 transition machinery restores
// function, or one that makes calls, the ABI0 transition machinery restores
// the g pointer only on such paths).
func clobberedGoFixed(l *liveness, a arch.Arch, reachesRuntime bool) (always, runtime []string) {
alwaysSet, runtimeSet := goFixedGPRs(a)
+5 -5
View File
@@ -7,13 +7,13 @@ import "testing"
// TestRegisterClobber checks the register-clobber audit is calibrated to the
// Go ABI (cmd/compile/abi-internal.md), not the platform ABI: Go's
// stack-based ABI0 — which hand-written assembly uses — has no System V
// stack-based ABI0, which hand-written assembly uses, has no System V
// style callee-saved registers, so argument and scratch registers may be
// clobbered freely. Only the registers the ABI fixes across calls (the
// frame pointer, the goroutine pointer, OS-reserved registers) are audited.
func TestRegisterClobber(t *testing.T) {
// amd64: BX, R12, R13 and R15 are argument/permanent-scratch registers in
// Go ABI0 — writing them unsaved is legal (a System V calibration would
// Go ABI0; writing them unsaved is legal (a System V calibration would
// report all of these).
scratch := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
@@ -27,7 +27,7 @@ func TestRegisterClobber(t *testing.T) {
}
// amd64: R14 (the goroutine pointer) in a NOSPLIT function without calls
// is the runtime's own pattern — the ABI0 transition restores it — so it
// is the runtime's own pattern (the ABI0 transition restores it), so it
// is not flagged.
leaf := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
@@ -101,7 +101,7 @@ func TestRegisterClobber(t *testing.T) {
t.Fatalf("arm64 R18 write should be flagged: %+v", armReserved)
}
// riscv64: X27 holds the goroutine; X5–X7 are scratch.
// riscv64: X27 holds the goroutine; X5-X7 are scratch.
riscScratch := lintSrcArch(t, "t_riscv64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOV X5, X6\n"+
@@ -117,7 +117,7 @@ func TestRegisterClobber(t *testing.T) {
t.Fatalf("unsaved riscv64 X27 write should be flagged: %+v", riscG)
}
// loong64: R22 holds the goroutine; R5–R19 are argument/scratch.
// loong64: R22 holds the goroutine; R5-R19 are argument/scratch.
loongScratch := lintSrcArch(t, "t_loong64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVV R5, R6\n"+
+1 -1
View File
@@ -22,7 +22,7 @@ var amd64WordNames = map[string]string{
// nonportableRegister maps a gasm-only register alias to the canonical go
// tool asm spelling. R or E followed by a canonical word name is the only
// alias family; R8–R15 are already canonical.
// alias family; R8-R15 are already canonical.
func nonportableRegister(name string) (string, bool) {
up := strings.ToUpper(name)
if len(up) != 3 {
+125 -65
View File
@@ -22,12 +22,13 @@ import (
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// textflagMacros are the flag names defined by textflag.h; they are highlighted
// as macros and offered as completions after a TEXT/GLOBL directive.
// textflagMacros are the flag names defined by runtime/textflag.h; they are
// highlighted as macros and offered as completions after a TEXT/GLOBL
// directive.
var textflagMacros = map[string]bool{
"NOPROFILE": true, "DUPOK": true, "NOSPLIT": true, "RODATA": true,
"NOPTR": true, "WRAPPER": true, "NEEDCTXT": true, "TOPFRAME": true,
"LEAF": true, "ABI0": true, "REFLECTDATA": true,
"NOPROF": true, "DUPOK": true, "NOSPLIT": true, "RODATA": true,
"NOPTR": true, "WRAPPER": true, "NEEDCTXT": true, "TLSBSS": true,
"NOFRAME": true, "REFLECTMETHOD": true, "TOPFRAME": true, "ABIWRAPPER": true,
}
// completion builds the completion list for a document.
@@ -81,13 +82,13 @@ func (s *Server) hover(p hoverParams) *Hover {
var md string
if in, ok := a.Lookup(word); ok {
md = "**" + in.Name + "** — " + in.Summary
md = "**" + in.Name + "**: " + in.Summary
} else if r, ok := a.Register(word); ok {
md = "**" + r.Name + "** — " + r.Class.String() + " register. " + r.Desc
md = "**" + r.Name + "**: " + r.Class.String() + " register. " + r.Desc
} else if desc, ok := arch.PseudoRegDesc(word); ok {
md = "**" + strings.ToUpper(word) + "** — pseudo-register. " + desc
md = "**" + strings.ToUpper(word) + "**: pseudo-register. " + desc
} else if textflagMacros[strings.ToUpper(word)] {
md = "**" + strings.ToUpper(word) + "** — textflag.h flag"
md = "**" + strings.ToUpper(word) + "**: textflag.h flag"
} else {
return nil
}
@@ -97,85 +98,124 @@ func (s *Server) hover(p hoverParams) *Hover {
}
}
// definition returns the location of the label definition for a label reference.
// openAST is one open document with its parsed file.
type openAST struct {
uri string
file *ast.File
}
// openASTs parses every open document, in URI order for deterministic
// results. Parsing is tolerant: a buffer with syntax errors still
// contributes its usable declarations to the workspace scans.
func (s *Server) openASTs() []openAST {
uris := make([]string, 0, len(s.docs))
for uri := range s.docs {
uris = append(uris, uri)
}
sort.Strings(uris)
out := make([]openAST, 0, len(uris))
for _, uri := range uris {
if f, _ := parser.Parse(uriPath(uri), s.docs[uri]); f != nil {
out = append(out, openAST{uri: uri, file: f})
}
}
return out
}
// definition returns the location of the named label or function: a local
// label in the current document wins, then the TEXT functions of every open
// document are searched, so a `CALL ·helper(SB)` jumps to its definition in
// another file.
func (s *Server) definition(p definitionParams) []Location {
text := s.docs[p.TextDocument.URI]
word, _ := wordAt(text, p.Position)
if word == "" {
return nil
}
name := strings.TrimPrefix(word, "\u00B7")
// Parse the document to find label definitions.
f, errs := parser.Parse(uriPath(p.TextDocument.URI), text)
if f == nil || len(errs) > 0 {
return nil
}
// Find the label definition.
for _, d := range f.Decls {
if t, ok := d.(*ast.Text); ok {
for _, stmt := range t.Body {
if lbl, ok := stmt.(*ast.Label); ok {
if lbl.Name.Text == word {
return []Location{{
URI: p.TextDocument.URI,
Range: Range{
Start: Position{Line: lbl.Name.Pos.Line - 1, Character: lbl.Name.Pos.Column - 1},
End: Position{Line: lbl.Name.Pos.Line - 1, Character: lbl.Name.Pos.Column - 1 + len(word)},
},
}}
// The local label definition.
f, _ := parser.Parse(uriPath(p.TextDocument.URI), text)
if f != nil {
for _, d := range f.Decls {
if t, ok := d.(*ast.Text); ok {
for _, stmt := range t.Body {
if lbl, ok := stmt.(*ast.Label); ok {
if lbl.Name.Text == name || lbl.Name.Text == word {
return []Location{{
URI: p.TextDocument.URI,
Range: Range{
Start: Position{Line: lbl.Name.Pos.Line - 1, Character: lbl.Name.Pos.Column - 1},
End: Position{Line: lbl.Name.Pos.Line - 1, Character: lbl.Name.Pos.Column - 1 + len(word)},
},
}}
}
}
}
}
}
}
// Function definitions across the open workspace.
for _, of := range s.openASTs() {
for _, d := range of.file.Decls {
t, ok := d.(*ast.Text)
if !ok || t.Name == nil {
continue
}
if t.Name.Name == name {
return []Location{{URI: of.uri, Range: symRange(t.Name)}}
}
}
}
return nil
}
// references returns all locations where the symbol under the cursor appears.
// references returns all locations where the symbol under the cursor appears
// across every open document. The current document matches labels and any
// operand name, as before; other documents only match SB-qualified operand
// references and the definition itself, because a bare name is a
// function-local label whose repeats in other files are unrelated.
func (s *Server) references(p referenceParams) []Location {
text := s.docs[p.TextDocument.URI]
word, _ := wordAt(text, p.Position)
if word == "" {
return nil
}
f, errs := parser.Parse(uriPath(p.TextDocument.URI), text)
if f == nil || len(errs) > 0 {
return nil
}
name := strings.TrimPrefix(word, "\u00B7")
uri := p.TextDocument.URI
var out []Location
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
continue
}
// Include the definition if requested.
if p.Context.IncludeDeclaration {
if t.Name.Name == word {
out = append(out, Location{
URI: uri,
Range: symRange(t.Name),
})
for _, of := range s.openASTs() {
sameDoc := of.uri == uri
for _, d := range of.file.Decls {
t, ok := d.(*ast.Text)
if !ok || t.Name == nil {
continue
}
}
for _, stmt := range t.Body {
switch st := stmt.(type) {
case *ast.Label:
if st.Name.Text == word {
out = append(out, Location{
URI: uri,
Range: tokenRange(st.Name),
})
}
case *ast.Instr:
for _, op := range st.Operands {
if op.Addr.Sym != nil && op.Addr.Sym.Name == word {
// Include the definition if requested.
if p.Context.IncludeDeclaration && t.Name.Name == name {
out = append(out, Location{URI: of.uri, Range: symRange(t.Name)})
}
for _, stmt := range t.Body {
switch st := stmt.(type) {
case *ast.Label:
if sameDoc && st.Name.Text == name {
out = append(out, Location{URI: of.uri, Range: tokenRange(st.Name)})
}
case *ast.Instr:
for _, op := range st.Operands {
if op.Addr.Sym == nil || op.Addr.Sym.Name != name {
continue
}
if !sameDoc && op.Addr.Sym.Pseudo != "SB" {
continue
}
out = append(out, Location{
URI: uri,
URI: of.uri,
Range: Range{
Start: Position{Line: op.Pos.Line - 1, Character: op.Pos.Column - 1},
End: Position{Line: op.Pos.Line - 1, Character: op.Pos.Column - 1 + runeLen(word)},
End: Position{Line: op.Pos.Line - 1, Character: op.Pos.Column - 1 + runeLen(name)},
},
})
}
@@ -271,7 +311,9 @@ func (s *Server) codeActions(p codeActionParams) []CodeAction {
for _, diag := range p.Context.Diagnostics {
switch diag.Code {
case "missing-ret":
// Offer to add RET at the end of the function.
// Offer to add RET at the end of the flagged function only: the
// diagnostic's range covers the TEXT keyword, so a line match
// picks the function the diagnostic belongs to.
f, _ := parser.Parse(uriPath(p.TextDocument.URI), text)
if f == nil {
continue
@@ -281,6 +323,9 @@ func (s *Server) codeActions(p codeActionParams) []CodeAction {
if !ok {
continue
}
if t.Keyword.Pos.Line-1 != int(diag.Range.Start.Line) {
continue
}
if len(t.Body) == 0 {
continue
}
@@ -298,7 +343,7 @@ func (s *Server) codeActions(p codeActionParams) []CodeAction {
endLine := len(lines) - 1
endChar := len([]rune(lines[endLine]))
actions = append(actions, CodeAction{
Title: "Add RET",
Title: "Add RET to " + t.Name.Name,
Kind: "quickfix",
Edit: &WorkspaceEdit{
Changes: map[string][]TextEdit{p.TextDocument.URI: {
@@ -792,13 +837,28 @@ func textRange(t *ast.Text) Range {
// diagnosticsFor computes the LSP diagnostics of one document; the push
// (publishDiagnostics) and pull (textDocument/diagnostic) paths share it.
// Parse errors surface as error-severity diagnostics so a malformed line is
// visible in the editor instead of only breaking derived features.
func (s *Server) diagnosticsFor(uri string) []Diagnostic {
text := s.docs[uri]
f, _ := parser.Parse(uriPath(uri), text)
f, errs := parser.Parse(uriPath(uri), text)
cfg := lint.Config{Arch: arch.FromFilename(uriPath(uri))}
diags := lint.File(f, cfg)
out := make([]Diagnostic, 0, len(diags))
out := make([]Diagnostic, 0, len(diags)+len(errs))
for _, e := range errs {
pos := token.Position{Line: 1, Column: 1}
if pe, ok := e.(parser.Error); ok && pe.Pos.IsValid() {
pos = pe.Pos
}
out = append(out, Diagnostic{
Range: toRange(pos.Line, pos.Column, token.Position{}),
Severity: sevError,
Code: "syntax",
Source: "gasm",
Message: e.Error(),
})
}
for _, d := range diags {
out = append(out, Diagnostic{
Range: toRange(d.Pos.Line, d.Pos.Column, d.End),
+2 -1
View File
@@ -4,7 +4,7 @@
// Package lsp implements a Language Server Protocol server for GAsm. It
// speaks JSON-RPC 2.0 over any io.Reader/io.Writer pair (normally standard
// input/output) and provides completion, hover documentation, document
// symbols, diagnostics and semantic-token highlighting — all backed by the
// symbols, diagnostics and semantic-token highlighting, all backed by the
// pure-Go lexer, parser, arch and lint packages. It is the vendor-neutral
// integration point: any LSP-capable editor can use it with no editor-specific
// plugin code.
@@ -31,6 +31,7 @@ type rpcError struct {
const (
errMethodNotFound = -32601
errInternalError = -32603
)
// --- LSP positions and ranges ----------------------------------------------
+31 -6
View File
@@ -8,6 +8,7 @@ import (
"encoding/json"
"fmt"
"io"
"net/url"
"strconv"
"strings"
"sync"
@@ -18,10 +19,11 @@ import (
// Server is a GAsm language server bound to a byte stream.
type Server struct {
in *bufio.Reader
out io.Writer
mu sync.Mutex // guards writes to out
docs map[string]string
in *bufio.Reader
out io.Writer
mu sync.Mutex // guards writes to out
docs map[string]string
version string // reported in the initialize result ("" omits it)
}
// New returns a server reading from in and writing to out.
@@ -33,6 +35,9 @@ func New(in io.Reader, out io.Writer) *Server {
}
}
// SetVersion records the server version reported in the initialize result.
func (s *Server) SetVersion(v string) { s.version = v }
// Run serves requests until the input is exhausted or an exit is requested.
func (s *Server) Run() error {
for {
@@ -109,9 +114,24 @@ func (s *Server) notify(method string, params any) {
}
// dispatch routes one message. It returns true when the server should stop.
// A panic in any handler is recovered and answered as an internal error:
// handlers parse live editor buffers, so malformed input must never take
// the whole server down.
func (s *Server) dispatch(msg *rpcMessage) (exit bool) {
defer func() {
if r := recover(); r != nil {
if msg != nil && msg.ID != nil {
s.respondError(msg.ID, errInternalError, fmt.Sprintf("internal error: %v", r))
}
}
}()
switch msg.Method {
case "initialize":
info := map[string]string{"name": "gasm"}
if s.version != "" {
info["version"] = s.version
}
s.respond(msg.ID, initializeResult{
Capabilities: ServerCapabilities{
TextDocumentSync: 1, // full sync
@@ -138,7 +158,7 @@ func (s *Server) dispatch(msg *rpcMessage) (exit bool) {
DocumentLinkProvider: map[string]any{},
FoldingRangeProvider: true,
},
ServerInfo: map[string]string{"name": "gasm", "version": "0.31.1"},
ServerInfo: info,
})
case "initialized", "textDocument/didSave":
@@ -292,9 +312,14 @@ func lintSeverity(s lint.Severity) int {
}
}
// uriPath strips a file:// scheme and returns the path component.
// uriPath strips a file:// scheme and percent-decodes the path component.
// LSP clients percent-encode URIs, so a raw slice would break every on-disk
// lookup for paths containing spaces or non-ASCII characters.
func uriPath(uri string) string {
if rest, ok := strings.CutPrefix(uri, "file://"); ok {
if decoded, err := url.PathUnescape(rest); err == nil {
return decoded
}
return rest
}
return uri
+184 -3
View File
@@ -381,7 +381,7 @@ func TestCodeActions(t *testing.T) {
frame(10, "textDocument/codeAction", map[string]any{
"textDocument": map[string]any{"uri": "file:///f_amd64.s"},
"range": map[string]any{"start": map[string]any{"line": 0, "character": 0}, "end": map[string]any{"line": 2, "character": 0}},
"context": map[string]any{"diagnostics": []map[string]any{{"code": "missing-ret", "range": map[string]any{"start": map[string]any{"line": 0, "character": 0}, "end": map[string]any{"line": 0, "character": 4}}}}},
"context": map[string]any{"diagnostics": []map[string]any{{"code": "missing-ret", "range": map[string]any{"start": map[string]any{"line": 1, "character": 0}, "end": map[string]any{"line": 1, "character": 4}}}}},
}) + frame(nil, "exit", nil)
msgs := run(t, in)
resp := findByID(msgs, 10)
@@ -395,8 +395,8 @@ func TestCodeActions(t *testing.T) {
if len(actions) == 0 {
t.Fatal("want at least 1 code action for missing-ret")
}
if actions[0].Title != "Add RET" {
t.Errorf("action title = %q, want Add RET", actions[0].Title)
if actions[0].Title != "Add RET to foo" {
t.Errorf("action title = %q, want Add RET to foo", actions[0].Title)
}
}
@@ -571,3 +571,184 @@ func TestFoldingRanges(t *testing.T) {
t.Errorf("folding kind = %q, want region", ranges[0].Kind)
}
}
// TestInitializeVersion checks the version reported in the initialize
// result when the caller stamps one.
func TestInitializeVersion(t *testing.T) {
var out bytes.Buffer
srv := New(strings.NewReader(frame(1, "initialize", map[string]any{})+frame(nil, "exit", nil)), &out)
srv.SetVersion("9.9.9")
if err := srv.Run(); err != nil {
t.Fatalf("server run: %v", err)
}
msgs := readFrames(t, &out)
resp := findByID(msgs, 1)
if resp == nil {
t.Fatal("no initialize response")
}
var res initializeResult
if err := json.Unmarshal(mustResult(t, resp), &res); err != nil {
t.Fatal(err)
}
if res.ServerInfo["version"] != "9.9.9" {
t.Errorf("serverInfo = %v, want version 9.9.9", res.ServerInfo)
}
}
// TestSyntaxDiagnosticsPublished checks that parse errors reach the editor
// as error-severity diagnostics with the syntax code.
func TestSyntaxDiagnosticsPublished(t *testing.T) {
msgs := run(t, session("file:///f_amd64.s", "TEXT $\n")+frame(nil, "exit", nil))
pub := findMethod(msgs, "textDocument/publishDiagnostics")
if pub == nil {
t.Fatal("no publishDiagnostics notification")
}
var p publishDiagnosticsParams
json.Unmarshal(pub.Params, &p)
found := false
for _, d := range p.Diagnostics {
if d.Code == "syntax" && d.Severity == sevError {
found = true
}
}
if !found {
t.Fatalf("expected a syntax error diagnostic, got %+v", p.Diagnostics)
}
}
// TestDispatchRecoversFromPanic exercises the per-message recover: a panic
// inside a handler is answered as an internal error instead of taking the
// server down.
func TestDispatchRecoversFromPanic(t *testing.T) {
var out bytes.Buffer
srv := New(strings.NewReader(""), &out)
srv.docs = nil // force a nil-map write inside didOpen
raw := json.RawMessage(`{"textDocument":{"uri":"file:///x.s","text":"RET"}}`)
srv.dispatch(&rpcMessage{ID: rawID(t, 7), Method: "textDocument/didOpen", Params: raw})
msgs := readFrames(t, &out)
resp := findByID(msgs, 7)
if resp == nil {
t.Fatal("no error response after panic")
}
if resp.Error == nil || resp.Error.Code != errInternalError {
t.Fatalf("error = %+v, want internal error", resp.Error)
}
}
func rawID(t *testing.T, n int) *json.RawMessage {
t.Helper()
b, err := json.Marshal(n)
if err != nil {
t.Fatal(err)
}
raw := json.RawMessage(b)
return &raw
}
// TestURIDecoding pins the percent-decoding of file URIs: clients encode
// non-ASCII paths, and the decoded form is what resolves on disk.
func TestURIDecoding(t *testing.T) {
got := uriPath("file:///home/petrbalvin/Repozit%C3%A1%C5%99e/k.s")
if got != "/home/petrbalvin/Repozitáře/k.s" {
t.Errorf("uriPath = %q", got)
}
if got := uriPath("/plain/path.s"); got != "/plain/path.s" {
t.Errorf("uriPath plain = %q", got)
}
}
// TestCodeActionsTargetsFlaggedFunctionOnly checks that a missing-ret
// diagnostic offers an edit for the flagged function only, even when the
// file defines several functions.
func TestCodeActionsTargetsFlaggedFunctionOnly(t *testing.T) {
doc := "TEXT \u00b7first(SB), NOSPLIT, $0\n" +
"\tMOVQ AX, CX\n" +
"\tRET\n" +
"TEXT \u00b7second(SB), NOSPLIT, $0\n" +
"\tMOVQ AX, CX\n"
in := session("file:///f_amd64.s", doc) +
frame(11, "textDocument/codeAction", map[string]any{
"textDocument": map[string]any{"uri": "file:///f_amd64.s"},
"range": map[string]any{"start": map[string]any{"line": 0, "character": 0}, "end": map[string]any{"line": 4, "character": 0}},
"context": map[string]any{"diagnostics": []map[string]any{{"code": "missing-ret", "range": map[string]any{"start": map[string]any{"line": 3, "character": 0}, "end": map[string]any{"line": 3, "character": 4}}}}},
}) + frame(nil, "exit", nil)
msgs := run(t, in)
resp := findByID(msgs, 11)
if resp == nil {
t.Fatal("no codeAction response")
}
var actions []CodeAction
if err := json.Unmarshal(mustResult(t, resp), &actions); err != nil {
t.Fatal(err)
}
if len(actions) != 1 {
t.Fatalf("actions = %d, want 1", len(actions))
}
if actions[0].Title != "Add RET to second" {
t.Errorf("action title = %q, want Add RET to second", actions[0].Title)
}
if !strings.Contains(actions[0].Edit.Changes["file:///f_amd64.s"][0].NewText, "\tRET\nTEXT \u00b7second") {
t.Errorf("edit does not insert RET at the end of second:\n%s", actions[0].Edit.Changes["file:///f_amd64.s"][0].NewText)
}
}
// TestCrossFileDefinitionAndReferences opens two documents: docA calls
// ·helper(SB), docB defines it. Definition must jump to docB and references
// must collect the call site in docA plus the definition in docB.
func TestCrossFileDefinitionAndReferences(t *testing.T) {
docA := "TEXT \u00b7caller(SB), NOSPLIT, $0\n" +
"\tCALL \u00b7helper(SB)\n" +
"\tRET\n"
docB := "TEXT \u00b7helper(SB), NOSPLIT, $0\n" +
"\tRET\n"
in := frame(1, "initialize", map[string]any{}) +
frame(nil, "initialized", map[string]any{}) +
frame(nil, "textDocument/didOpen", map[string]any{
"textDocument": map[string]any{"uri": "file:///a_amd64.s", "languageId": "gasm", "version": 1, "text": docA},
}) +
frame(nil, "textDocument/didOpen", map[string]any{
"textDocument": map[string]any{"uri": "file:///b_amd64.s", "languageId": "gasm", "version": 1, "text": docB},
}) +
frame(2, "textDocument/definition", map[string]any{
"textDocument": map[string]any{"uri": "file:///a_amd64.s"},
"position": map[string]any{"line": 1, "character": 8}, // on helper in CALL ·helper(SB)
}) +
frame(3, "textDocument/references", map[string]any{
"textDocument": map[string]any{"uri": "file:///b_amd64.s"},
"position": map[string]any{"line": 0, "character": 7}, // on helper in TEXT ·helper(SB)
"context": map[string]any{"includeDeclaration": true},
}) +
frame(nil, "exit", nil)
msgs := run(t, in)
dresp := findByID(msgs, 2)
if dresp == nil {
t.Fatal("no definition response")
}
var locs []Location
if err := json.Unmarshal(mustResult(t, dresp), &locs); err != nil {
t.Fatal(err)
}
if len(locs) != 1 || locs[0].URI != "file:///b_amd64.s" || locs[0].Range.Start.Line != 0 {
t.Fatalf("definition = %+v, want the TEXT in b_amd64.s line 0", locs)
}
rresp := findByID(msgs, 3)
if rresp == nil {
t.Fatal("no references response")
}
locs = nil
if err := json.Unmarshal(mustResult(t, rresp), &locs); err != nil {
t.Fatal(err)
}
if len(locs) != 2 {
t.Fatalf("references = %+v, want the definition in b_amd64.s and the call in a_amd64.s", locs)
}
byURI := map[string]int{}
for _, l := range locs {
byURI[l.URI]++
}
if byURI["file:///a_amd64.s"] != 1 || byURI["file:///b_amd64.s"] != 1 {
t.Errorf("references by uri = %v, want one in each file", byURI)
}
}
+34 -10
View File
@@ -210,17 +210,39 @@ func (p *state) parseText(line []token.Token) {
rest = rest[1:]
}
// Frame: $number ; optional args: -number. Whatever remains after the
// header is the body and is parsed by the caller.
// Frame: $[-]number ; optional args: -number. The Go runtime writes
// zero frames with an explicit sign ("$-0-24"), so the number may carry
// one. Whatever remains after the header is the body and is parsed by
// the caller.
if len(rest) > 0 && rest[0].Kind == token.Dollar {
text.Frame = parseOperand(rest[:2]) // "$" "number"
if len(rest) >= 4 && rest[2].Kind == token.Minus && rest[3].Kind == token.Number {
text.Args = &ast.Operand{
Kind: ast.OpImmediate,
Imm: ast.Immediate{Val: parseInt(rest[3].Text), HasVal: true},
Raw: "-" + rest[3].Text,
Pos: rest[2].Pos,
n := 1
neg := false
if n < len(rest) && (rest[n].Kind == token.Minus || rest[n].Kind == token.Plus) {
neg = rest[n].Kind == token.Minus
n++
}
if n < len(rest) && rest[n].Kind == token.Number {
val := parseInt(rest[n].Text)
if neg {
val = -val
}
text.Frame = &ast.Operand{
Kind: ast.OpImmediate,
Imm: ast.Immediate{Val: val, HasVal: true},
Raw: joinRaw(rest[:n+1]),
Pos: rest[0].Pos,
}
// The argument area: a minus sign followed by a number.
if n+2 < len(rest) && rest[n+1].Kind == token.Minus && rest[n+2].Kind == token.Number {
text.Args = &ast.Operand{
Kind: ast.OpImmediate,
Imm: ast.Immediate{Val: parseInt(rest[n+2].Text), HasVal: true},
Raw: "-" + rest[n+2].Text,
Pos: rest[n+1].Pos,
}
}
} else {
p.errorf(rest[0].Pos, "TEXT frame size must be a number after $")
}
}
@@ -234,8 +256,10 @@ func (p *state) parseGlobl(line []token.Token) *ast.Globl {
sym, n := parseSymbolPrefix(rest)
g.Name = sym
rest = skipComma(rest[n:])
// Flags are identifiers (RODATA, DUPOK) or legacy numeric constants
// (2, 8, 9, 10) from runtime/textflag.h.
for len(rest) > 0 && rest[0].Kind != token.Dollar {
if rest[0].Kind == token.Ident {
if rest[0].Kind == token.Ident || rest[0].Kind == token.Number {
g.Flags = append(g.Flags, rest[0].Text)
}
rest = rest[1:]
+66 -3
View File
@@ -149,7 +149,7 @@ func TestOperandStructure(t *testing.T) {
}
}
// MOVQ swin_base+0(FP), SI — the first MOVQ in the body.
// MOVQ swin_base+0(FP), SI; the first MOVQ in the body.
var mov *ast.Instr
for _, s := range fn.Body {
if in, ok := s.(*ast.Instr); ok && in.Mnemonic.Text == "MOVQ" {
@@ -202,7 +202,7 @@ func TestAVX512Operands(t *testing.T) {
}
}
// VALIGND $15, Z9, Z0, Z1 — four operands.
// VALIGND $15, Z9, Z0, Z1; four operands.
val := byMnem["VALIGND"]
if val == nil {
t.Fatal("VALIGND not found")
@@ -224,7 +224,7 @@ func TestAVX512Operands(t *testing.T) {
t.Errorf("VMOVDQU32 dst = %+v, want 4(SI)(AX*1)", dst)
}
// KTESTW K1, K1 — mask registers parse as bare names.
// KTESTW K1, K1; mask registers parse as bare names.
kt := byMnem["KTESTW"]
if kt == nil || len(kt.Operands) != 2 {
t.Fatalf("KTESTW = %+v, want two operands", kt)
@@ -253,3 +253,66 @@ func TestDataWidthAndStatic(t *testing.T) {
t.Errorf("mask24 DATA should be static, got %+v", datas[2].Name)
}
}
// TestTruncatedFrameDollar is a regression test for a TEXT directive whose
// frame size is missing after the $: the parser used to slice past the end
// of the token slice and panic. It must report a diagnostic instead.
func TestTruncatedFrameDollar(t *testing.T) {
for _, src := range []string{
"TEXT $\n",
"TEXT \u00b7foo(SB), $\n",
"TEXT \u00b7foo(SB), NOSPLIT, $\n",
} {
var file *ast.File
func() {
defer func() {
if r := recover(); r != nil {
t.Fatalf("Parse(%q) panicked: %v", src, r)
}
}()
file, _ = Parse("t.s", src)
}()
if file == nil {
t.Fatalf("Parse(%q) returned no file", src)
}
if len(file.Decls) != 1 {
t.Fatalf("Parse(%q) decls = %d, want 1", src, len(file.Decls))
}
txt := file.Decls[0].(*ast.Text)
if txt.Frame != nil {
t.Errorf("Parse(%q) frame = %v, want nil", src, txt.Frame)
}
}
}
// TestFrameAndArgs parses a well-formed TEXT header and checks that the
// frame and args operands are picked up.
func TestFrameAndArgs(t *testing.T) {
file, errs := Parse("t.s", "TEXT \u00b7foo(SB), $32-16\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse errors: %v", errs)
}
txt := file.Decls[0].(*ast.Text)
if txt.Frame == nil || !txt.Frame.Imm.HasVal || txt.Frame.Imm.Val != 32 {
t.Errorf("frame = %+v, want $32", txt.Frame)
}
if txt.Args == nil || !txt.Args.Imm.HasVal || txt.Args.Imm.Val != 16 {
t.Errorf("args = %+v, want -16", txt.Args)
}
}
// TestSignedZeroFrame covers the Go runtime's "$-0-24" spelling: a zero
// frame with an explicit sign plus the argument area.
func TestSignedZeroFrame(t *testing.T) {
file, errs := Parse("t.s", "TEXT \u00b7foo<ABIInternal>(SB), NOSPLIT, $-0-24\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse errors: %v", errs)
}
txt := file.Decls[0].(*ast.Text)
if txt.Frame == nil || !txt.Frame.Imm.HasVal || txt.Frame.Imm.Val != 0 {
t.Errorf("frame = %+v, want $-0", txt.Frame)
}
if txt.Args == nil || !txt.Args.Imm.HasVal || txt.Args.Imm.Val != 24 {
t.Errorf("args = %+v, want -24", txt.Args)
}
}
+5 -5
View File
@@ -5,8 +5,8 @@
// func add(a, b int64) int64
TEXT ·add(SB), NOSPLIT, $0-24
MOV a+0(FP), X10
MOV b+8(FP), X11
ADD X11, X10, X10
MOV X10, ret+16(FP)
RET
MOV a+0(FP), X10
MOV b+8(FP), X11
ADD X11, X10, X10
MOV X10, ret+16(FP)
RET
+10 -10
View File
@@ -5,16 +5,16 @@
// func atomicAdd(ptr *int64, val int64) int64
TEXT ·atomicAdd(SB), NOSPLIT, $0-24
MOV a+0(FP), X10
MOV b+8(FP), X11
AMOADDD X11, (X10), X12
MOV X12, ret+16(FP)
RET
MOV a+0(FP), X10
MOV b+8(FP), X11
AMOADDD X11, (X10), X12
MOV X12, ret+16(FP)
RET
// func fpAdd(a, b float64) float64
TEXT ·fpAdd(SB), NOSPLIT, $0-24
FLD a+0(FP), F10
FLD b+8(FP), F11
FADDD F10, F11, F12
FSD F12, ret+16(FP)
RET
FLD a+0(FP), F10
FLD b+8(FP), F11
FADDD F10, F11, F12
FSD F12, ret+16(FP)
RET
+13 -13
View File
@@ -5,22 +5,22 @@
// func readCSR(csr int64) int64
TEXT ·readCSR(SB), NOSPLIT, $0-16
MOV a+0(FP), X10
CSRRS $0x300, X0, X11
MOV X11, ret+8(FP)
RET
MOV a+0(FP), X10
CSRRS $0x300, X0, X11
MOV X11, ret+8(FP)
RET
// func setCSRBit(csr, bit int64) int64
TEXT ·setCSRBit(SB), NOSPLIT, $0-24
MOV a+0(FP), X10
MOV b+8(FP), X11
CSRRS $0x304, X11, X12
MOV X12, ret+16(FP)
RET
MOV a+0(FP), X10
MOV b+8(FP), X11
CSRRS $0x304, X11, X12
MOV X12, ret+16(FP)
RET
// func writeCSR(val int64) int64
TEXT ·writeCSR(SB), NOSPLIT, $0-16
MOV a+0(FP), X10
CSRRW $0x305, X10, X11
MOV X11, ret+8(FP)
RET
MOV a+0(FP), X10
CSRRW $0x305, X10, X11
MOV X11, ret+8(FP)
RET
+12 -12
View File
@@ -5,18 +5,18 @@
// func fma(a, b, c float64) float64
TEXT ·fma(SB), NOSPLIT, $0-32
FLD a+0(FP), F10
FLD b+8(FP), F11
FLD c+16(FP), F12
FMADDD F10, F11, F12, F13
FSD F13, ret+24(FP)
RET
FLD a+0(FP), F10
FLD b+8(FP), F11
FLD c+16(FP), F12
FMADDD F10, F11, F12, F13
FSD F13, ret+24(FP)
RET
// func fms(a, b, c float64) float64
TEXT ·fms(SB), NOSPLIT, $0-32
FLD a+0(FP), F10
FLD b+8(FP), F11
FLD c+16(FP), F12
FMSUBD F10, F11, F12, F13
FSD F13, ret+24(FP)
RET
FLD a+0(FP), F10
FLD b+8(FP), F11
FLD c+16(FP), F12
FMSUBD F10, F11, F12, F13
FSD F13, ret+24(FP)
RET
+22 -21
View File
@@ -6,31 +6,32 @@
// func casLoop(ptr *int64, old, new int64) bool
TEXT ·casLoop(SB), NOSPLIT, $0-32
cas_retry:
MOV a+0(FP), X10
LRD (X10), X11
MOV b+8(FP), X12
BNE X11, X12, cas_fail
MOV c+16(FP), X13
SCD X13, (X10), X14
BNE X14, X0, cas_retry
ADDI X0, $1, X15
MOV X15, ret+24(FP)
RET
MOV a+0(FP), X10
LRD (X10), X11
MOV b+8(FP), X12
BNE X11, X12, cas_fail
MOV c+16(FP), X13
SCD X13, (X10), X14
BNE X14, X0, cas_retry
ADDI X0, $1, X15
MOV X15, ret+24(FP)
RET
cas_fail:
MOV X0, ret+24(FP)
RET
MOV X0, ret+24(FP)
RET
// func intToFloat(x int64) float64
TEXT ·intToFloat(SB), NOSPLIT, $0-16
MOV a+0(FP), X10
FCVTDL X10, F10
FSD F10, ret+8(FP)
RET
MOV a+0(FP), X10
FCVTDL X10, F10
FSD F10, ret+8(FP)
RET
// func compare(a, b float64) bool
TEXT ·compare(SB), NOSPLIT, $0-24
FLD a+0(FP), F10
FLD b+8(FP), F11
FLTD F10, F11, X10
MOV X10, ret+16(FP)
RET
FLD a+0(FP), F10
FLD b+8(FP), F11
FLTD F10, F11, X10
MOV X10, ret+16(FP)
RET
+28 -27
View File
@@ -15,43 +15,44 @@ DATA mask24<>+4(SB)/4, $0x80050403
// func analyzeO1RangeAVX2(swin []int32, dstP []uint32, hist *[32]uint16) (partSum uint64, overflow bool)
TEXT ·analyzeO1RangeAVX2(SB), NOSPLIT, $0-65
MOVQ swin_base+0(FP), SI
MOVQ dstP_base+24(FP), DI
MOVQ dstP_len+32(FP), BX
MOVQ hist+48(FP), R13
MOVQ swin_base+0(FP), SI
MOVQ dstP_base+24(FP), DI
MOVQ dstP_len+32(FP), BX
MOVQ hist+48(FP), R13
VPCMPEQD Y0, Y0, Y0
VPSLLD $31, Y0, Y0
VPCMPEQD Y0, Y0, Y0
VPSLLD $31, Y0, Y0
LEAQ (SI)(BX*4), R9
MOVQ BX, R10
ANDQ $-8, R10
LEAQ (SI)(BX*4), R9
MOVQ BX, R10
ANDQ $-8, R10
vec1:
CMPQ SI, R10
JGE vec1done
VMOVDQU (SI), Y1
VMOVDQU 4(SI), Y2
VPSUBD Y1, Y2, Y3
ADDQ $32, SI
JMP vec1
CMPQ SI, R10
JGE vec1done
VMOVDQU (SI), Y1
VMOVDQU 4(SI), Y2
VPSUBD Y1, Y2, Y3
ADDQ $32, SI
JMP vec1
vec1done:
MOVQ AX, partSum+56(FP)
MOVB AL, overflow+64(FP)
MOVQ AX, partSum+56(FP)
MOVB AL, overflow+64(FP)
VZEROUPPER
RET
// func decodeFixedO1AVX512(samples []int32, residual []int32)
TEXT ·decodeFixedO1AVX512(SB), NOSPLIT, $0-48
MOVQ samples_base+0(FP), SI
MOVQ residual_base+16(FP), DI
MOVQ samples_base+0(FP), SI
MOVQ residual_base+16(FP), DI
VPBROADCASTD AX, Z15
VMOVDQU32 (DI)(AX*1), Z0
VALIGND $15, Z9, Z0, Z1
VFMADD231PD Z14, Z12, Z10
VPCMPEQD Z0, Z3, K1
KTESTW K1, K1
VPSRAQ X31, Z8, Z8
VMOVDQU32 Z0, 4(SI)(AX*1)
VMOVDQU32 (DI)(AX*1), Z0
VALIGND $15, Z9, Z0, Z1
VFMADD231PD Z14, Z12, Z10
VPCMPEQD Z0, Z3, K1
KTESTW K1, K1
VPSRAQ X31, Z8, Z8
VMOVDQU32 Z0, 4(SI)(AX*1)
RET
+1 -1
View File
@@ -20,7 +20,7 @@ TEXT ·dirtyBP(SB), NOSPLIT, $0-16
RET
// func dirtyR14(a int64) int64
// Deliberately clobbers R14 (the goroutine pointer — a serious ABI violation).
// Deliberately clobbers R14 (the goroutine pointer; a serious ABI violation).
TEXT ·dirtyR14(SB), NOSPLIT, $0-16
MOVQ $0x5678, R14
MOVQ a+0(FP), AX
+30 -30
View File
@@ -13,55 +13,55 @@ TEXT ·add(SB), NOSPLIT, $0-24
// func sum(data []int64) int64
// Sums all elements of the slice.
TEXT ·sum(SB), NOSPLIT, $0-32
MOVQ data_base+0(FP), SI
MOVQ data_len+8(FP), CX
XORQ AX, AX
MOVQ data_base+0(FP), SI
MOVQ data_len+8(FP), CX
XORQ AX, AX
TESTQ CX, CX
JZ sum_done
JZ sum_done
sum_loop:
ADDQ (SI), AX
ADDQ $8, SI
DECQ CX
JNZ sum_loop
ADDQ (SI), AX
ADDQ $8, SI
DECQ CX
JNZ sum_loop
sum_done:
MOVQ AX, ret+24(FP)
MOVQ AX, ret+24(FP)
RET
// func wideCopy(dst, src []byte)
// Non-overlapping copy of min(len(dst), len(src)) bytes using 32-byte moves.
TEXT ·wideCopy(SB), NOSPLIT, $0-48
MOVQ dst_base+0(FP), DI
MOVQ dst_len+8(FP), BX
MOVQ src_base+24(FP), SI
MOVQ src_len+32(FP), R8
CMPQ BX, R8
JLE wc_have_n
MOVQ R8, BX
MOVQ dst_base+0(FP), DI
MOVQ dst_len+8(FP), BX
MOVQ src_base+24(FP), SI
MOVQ src_len+32(FP), R8
CMPQ BX, R8
JLE wc_have_n
MOVQ R8, BX
wc_have_n:
CMPQ BX, $32
JB wc_small
CMPQ BX, $32
JB wc_small
VMOVDQU (SI), Y0
VMOVDQU Y0, (DI)
VMOVDQU -32(SI)(BX*1), Y0
VMOVDQU Y0, -32(DI)(BX*1)
VMOVDQU (SI), Y0
VMOVDQU Y0, (DI)
VMOVDQU -32(SI)(BX*1), Y0
VMOVDQU Y0,-32(DI)(BX*1)
VZEROUPPER
RET
wc_small:
TESTQ BX, BX
JZ wc_done
TESTQ BX, BX
JZ wc_done
wc_byte:
MOVB (SI), R8B
MOVB R8B, (DI)
INCQ SI
INCQ DI
DECQ BX
JNZ wc_byte
MOVB (SI), R8B
MOVB R8B, (DI)
INCQ SI
INCQ DI
DECQ BX
JNZ wc_byte
wc_done:
RET
+32 -29
View File
@@ -5,53 +5,56 @@
// add returns a + b.
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
// arith exercises the register-register integer set.
TEXT ·arith(SB), NOSPLIT, $0-0
ADD R4, R5, R6
SUB R7, R8, R9
AND R10, R11, R12
ORR R12, R13, R14
EOR R14, R15, R16
CMP R16, R17
ADD R4, R5
SUB R6, R7
ADD R4, R5, R6
SUB R7, R8, R9
AND R10, R11, R12
ORR R12, R13, R14
EOR R14, R15, R16
CMP R16, R17
ADD R4, R5
SUB R6, R7
RET
// branch exercises conditional and unconditional control flow.
TEXT ·branch(SB), NOSPLIT, $0-0
BEQ done
BNE skip
BGE done
BLT done
BGT done
BLE done
BEQ done
BNE skip
BGE done
BLT done
BGT done
BLE done
skip:
B loop
B loop
loop:
ADD R4, R5
ADD R4, R5
RET
done:
RET
// mov exercises the MOV pseudo-instruction.
TEXT ·mov(SB), NOSPLIT, $0-16
MOVD $0, R4
MOVD $1, R5
MOVD $42, R6
MOVD a+0(FP), R7
MOVD R7, ret+0(FP)
MOVW $100, R8
MOVD $0, R4
MOVD $1, R5
MOVD $42, R6
MOVD a+0(FP), R7
MOVD R7, ret+0(FP)
MOVW $100, R8
RET
// frame exercises the prologue/epilogue of a function with a real frame.
TEXT ·frame(SB), NOSPLIT, $32-8
MOVD arg+0(FP), R4
ADD $1, R4, R4
MOVD R4, ret+0(FP)
MOVD arg+0(FP), R4
ADD $1, R4, R4
MOVD R4, ret+0(FP)
RET

Some files were not shown because too many files have changed in this diff Show More