Compare commits
205
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4d01bb3ecf | ||
|
|
332c63e440 | ||
|
|
b306c210c6 | ||
|
|
b9015e1c2e | ||
|
|
26c5008136 | ||
|
|
74d6b90d69 | ||
|
|
7b11c62f53 | ||
|
|
8eed54b3da | ||
|
|
4be16dcdf5 | ||
|
|
ded9cabdf4 | ||
|
|
cf6bc6987e | ||
|
|
ff7b1452b1 | ||
|
|
517c1cea25 | ||
|
|
a3e3010e0f | ||
|
|
057c4eb545 | ||
|
|
f720381d43 | ||
|
|
2c9042d62c | ||
|
|
82ef289d3a | ||
|
|
7246b0e002 | ||
|
|
8cfd40aac8 | ||
|
|
5382c9a8e4 | ||
|
|
53de91b2df | ||
|
|
8a36af7c7d | ||
|
|
e9789ce3f4 | ||
|
|
837231c068 | ||
|
|
95025be1bc | ||
|
|
03a964bb2d | ||
|
|
123a16e346 | ||
|
|
9701812bee | ||
|
|
29ac03468e | ||
|
|
bfb7701db1 | ||
|
|
e8b6ff5d7c | ||
|
|
1456907000 | ||
|
|
ec1c521187 | ||
|
|
a7744c24bd | ||
|
|
522e6f2ae8 | ||
|
|
81d4bd81e4 | ||
|
|
687678a2ea | ||
|
|
b0f9071bf5 | ||
|
|
81e2673923 | ||
|
|
75e9fd771b | ||
|
|
863926abd6 | ||
|
|
241e7256f6 | ||
|
|
6556b85abf | ||
|
|
289cabe993 | ||
|
|
d6cf7cfa44 | ||
|
|
4cc2f0eba5 | ||
|
|
97dfaa7526 | ||
|
|
66aa4dbc8b | ||
|
|
dce5d31462 | ||
|
|
9dc3987e02 | ||
|
|
9b238a525a | ||
|
|
ad82aac663 | ||
|
|
0629f5e2df | ||
|
|
ecb203dcf5 | ||
|
|
6c672567f3 | ||
|
|
cc6e416c59 | ||
|
|
c66a47973a | ||
|
|
9629897202 | ||
|
|
5399a8a724 | ||
|
|
de5d9f358e | ||
|
|
ca3fdce0e0 | ||
|
|
fc2d92eabd | ||
|
|
39d2e80145 | ||
|
|
9f4f949c1f | ||
|
|
f0d5238c47 | ||
|
|
2931bbd6b2 | ||
|
|
63562a503a | ||
|
|
e836d6150d | ||
|
|
8a51b060da | ||
|
|
d3d47db727 | ||
|
|
0758556b7d | ||
|
|
ddb8440340 | ||
|
|
f15ff66fb1 | ||
|
|
187e4856d3 | ||
|
|
d315a998ce | ||
|
|
a6f3828c02 | ||
|
|
e3b35bb817 | ||
|
|
eb0a89e58d | ||
|
|
dd32d9e66e | ||
|
|
3a73acb20a | ||
|
|
7604a9443f | ||
|
|
b3908fc43d | ||
|
|
eb8b0cd316 | ||
|
|
a8bfd54ed2 | ||
|
|
375182ef1f | ||
|
|
87b1081c53 | ||
|
|
f3c8510a58 | ||
|
|
ebdf14939f | ||
|
|
79a2c16bac | ||
|
|
401386956c | ||
|
|
4258131a3a | ||
|
|
94e09e8070 | ||
|
|
7aefe6a42d | ||
|
|
ac1c05c793 | ||
|
|
93c47a312a | ||
|
|
708d0a0a5e | ||
|
|
3c8f7cb411 | ||
|
|
7c5b7a1419 | ||
|
|
bc3f448738 | ||
|
|
f37f183577 | ||
|
|
1e77e58250 | ||
|
|
1d0969ed64 | ||
|
|
23c001be51 | ||
|
|
96e81cc98d | ||
|
|
c834d98210 | ||
|
|
03d6d4da54 | ||
|
|
0b42ce7952 | ||
|
|
288a64ccd2 | ||
|
|
5fddfa704b | ||
|
|
a2bb5eeb4e | ||
|
|
48449b7a7f | ||
|
|
3de043c494 | ||
|
|
0078f7be5c | ||
|
|
6a7317d141 | ||
|
|
d08523caa5 | ||
|
|
20e4b8d9c4 | ||
|
|
61f4247cef | ||
|
|
049872ddff | ||
|
|
3669f64ff6 | ||
|
|
fff9f75595 | ||
|
|
40476546df | ||
|
|
70218e84ba | ||
|
|
db50b98179 | ||
|
|
2e2c0b82a0 | ||
|
|
8dc1e98ca1 | ||
|
|
1d8e68c574 | ||
|
|
89fa6ea15e | ||
|
|
edc20ffa97 | ||
|
|
50db6615b2 | ||
|
|
95f1d6f083 | ||
|
|
f43e791e5a | ||
|
|
1691c81095 | ||
|
|
2db563be07 | ||
|
|
4f190ee1a2 | ||
|
|
909f874797 | ||
|
|
e307bf830f | ||
|
|
953c258d6a | ||
|
|
c6f0286732 | ||
|
|
f5fc22d390 | ||
|
|
22226d59a5 | ||
|
|
a0ae0e4e37 | ||
|
|
94c4756d47 | ||
|
|
a5a59d6503 | ||
|
|
9cb1666b35 | ||
|
|
96cc70731f | ||
|
|
b4da13d0f6 | ||
|
|
41b387e54d | ||
|
|
4171e412b5 | ||
|
|
57c0ca8b09 | ||
|
|
75bd83fd52 | ||
|
|
62f6fb4faf | ||
|
|
8f84dac10b | ||
|
|
6d7f10f13e | ||
|
|
56f8babbce | ||
|
|
6c1c8d9d96 | ||
|
|
93794c02f7 | ||
|
|
56ad158772 | ||
|
|
15e8b88d32 | ||
|
|
9beff4ae85 | ||
|
|
eacf33d0f7 | ||
|
|
9a34733615 | ||
|
|
970df7c32a | ||
|
|
49572efe16 | ||
|
|
568c553986 | ||
|
|
7a30e902fc | ||
|
|
cb398d9498 | ||
|
|
e44162a749 | ||
|
|
6fb9629ab6 | ||
|
|
8bda4066e3 | ||
|
|
685b150ecf | ||
|
|
c92e6bed3a | ||
|
|
d75e6bcae6 | ||
|
|
6699ebd34f | ||
|
|
ba4d961b20 | ||
|
|
78b12dd427 | ||
|
|
19a26e049b | ||
|
|
163480e283 | ||
|
|
0d62818db7 | ||
|
|
be2ceaafb9 | ||
|
|
941f7fa990 | ||
|
|
eb3c79c3c8 | ||
|
|
eade875b53 | ||
|
|
b08005753e | ||
|
|
5e06d6a6aa | ||
|
|
e4c9d78968 | ||
|
|
f5fcaf9fa6 | ||
|
|
94b98468bb | ||
|
|
746cef8100 | ||
|
|
f153be8158 | ||
|
|
1bf95a169e | ||
|
|
f9eb4021d6 | ||
|
|
c4930438fd | ||
|
|
f372e2db75 | ||
|
|
ac02c83a86 | ||
|
|
ce5ec24fa8 | ||
|
|
181d8e508c | ||
|
|
1160c96427 | ||
|
|
de9e211ff1 | ||
|
|
3acbdd6533 | ||
|
|
ba502c9b79 | ||
|
|
874e054ecb | ||
|
|
f1960febdc | ||
|
|
ae550cc05a | ||
|
|
cc50035375 |
@@ -0,0 +1,37 @@
|
||||
# Race, Go. Dispatched by hand, and never a gate on a push or a tag: the release tag is
|
||||
# cut only after `just gates` has already raced the tree, so this workflow is the
|
||||
# explicit second opinion, not a step of the release.
|
||||
#
|
||||
# The race detector roughly doubles both time and memory, which the shared runner box
|
||||
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once per
|
||||
# task; here it is a decision rather than a routine.
|
||||
#
|
||||
# Every step is one command, so the step that fails is the gate that failed.
|
||||
name: Race
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
|
||||
env:
|
||||
# One core: parallelism buys no speed here and costs memory the box does not have.
|
||||
GOFLAGS: -p=1
|
||||
GOMAXPROCS: "2"
|
||||
|
||||
jobs:
|
||||
race:
|
||||
runs-on: fedora
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
- name: Install gcc
|
||||
# The race detector needs cgo and the runner image carries no C compiler.
|
||||
run: dnf install -y gcc
|
||||
|
||||
- name: Race
|
||||
run: go test -race -count=1 -timeout 10m ./...
|
||||
+319
-86
@@ -1,73 +1,222 @@
|
||||
# Release — gasm binaries. Runs on version tags (v0.28.0) pushed to main.
|
||||
# Release, Go binaries. Runs on version tags (v1.2.3) pushed to main.
|
||||
#
|
||||
# The module sits at the repository root: the toolchain records a version only for a root
|
||||
# module, measured on go1.27.1, so a build of a module in a subdirectory reports (devel)
|
||||
# even at its own <module>/vX.Y.Z tag and this workflow's smoke test can never pass for
|
||||
# it. A Go repository is one module at the root.
|
||||
#
|
||||
# The version contract these steps implement: nothing is injected. The toolchain records
|
||||
# the tag into the binary's build information, so the build simply has to happen at the
|
||||
# tag, which the trigger guarantees.
|
||||
#
|
||||
# The gates run in their own job, once, before the matrix, minus the race detector: race
|
||||
# never runs on a push path or a tag, and the local gate raced this tree before the tag
|
||||
# was cut. Putting the gates inside the matrix would run the whole suite once per target
|
||||
# on the box that also hosts the forge. Each job validates the tag for itself rather than
|
||||
# passing a value between jobs, so no workflow feature has to be trusted for the version
|
||||
# to reach the file name.
|
||||
name: Release
|
||||
|
||||
on:
|
||||
push:
|
||||
tags: ["v*"]
|
||||
|
||||
env:
|
||||
# The box is shared with the forge, so parallelism is bounded on purpose. The gates job
|
||||
# needs it most; the build jobs inherit it for their parallel compilation.
|
||||
GOFLAGS: -p=1
|
||||
GOMAXPROCS: "2"
|
||||
|
||||
jobs:
|
||||
gates:
|
||||
runs-on: fedora
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
- name: Install Perl
|
||||
# Perl for the steps below. The install is a no-op where the package
|
||||
# is already present.
|
||||
run: dnf install -y perl
|
||||
|
||||
- name: Validate the tag
|
||||
env:
|
||||
VERSION: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
perl -e '
|
||||
my $v = $ENV{VERSION} // q{};
|
||||
$v =~ m{^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$}
|
||||
or die qq{ERROR: expected a semver tag like v1.2.3, got: $v\n};
|
||||
print qq{tag $v\n};
|
||||
'
|
||||
|
||||
- name: Security policy names this release
|
||||
# The supported-versions table is the one part of SECURITY.md that
|
||||
# carries a version, so it goes stale the moment a tag is cut. Fail
|
||||
# here rather than publish a policy naming the previous release.
|
||||
env:
|
||||
VERSION: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
perl -e '
|
||||
my $v = $ENV{VERSION} // q{};
|
||||
(my $nv = $v) =~ s/^v//;
|
||||
open(my $f, q{<}, q{SECURITY.md}) or die qq{SECURITY.md: $!\n};
|
||||
local $/;
|
||||
my $t = <$f>;
|
||||
close $f;
|
||||
$t =~ m{^\|\s*\Q$nv\E\s*\|\s*yes\s*\|}m
|
||||
or die qq{ERROR: SECURITY.md does not name $nv as supported; update the table before releasing.\n};
|
||||
print qq{SECURITY.md names $nv\n};
|
||||
'
|
||||
|
||||
- name: Build
|
||||
run: go build ./...
|
||||
|
||||
- name: Format
|
||||
run: |
|
||||
perl -e '
|
||||
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
|
||||
my @bad = <$g>;
|
||||
close($g);
|
||||
print @bad;
|
||||
exit(@bad ? 1 : 0);
|
||||
'
|
||||
|
||||
- name: Vet
|
||||
run: go vet ./...
|
||||
|
||||
- name: Modernise
|
||||
run: go fix -diff ./...
|
||||
|
||||
- name: Tests
|
||||
# The same command as in test.yml, so the floor is the same number everywhere.
|
||||
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
|
||||
|
||||
- name: Tests outside the coverage set
|
||||
# The same command as in test.yml: the CLI's exit codes and manual-page guard,
|
||||
# and the debugger's architecture-neutral units, run outside the floor.
|
||||
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
|
||||
|
||||
- name: Coverage floor
|
||||
run: |
|
||||
perl -e '
|
||||
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
|
||||
my $total;
|
||||
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
|
||||
close($c);
|
||||
die qq{no total line in coverage.out\n} unless defined $total;
|
||||
printf qq{Total coverage: %s%%\n}, $total;
|
||||
exit($total < 80 ? 1 : 0);
|
||||
'
|
||||
|
||||
build:
|
||||
runs-on: fedora
|
||||
timeout-minutes: 25
|
||||
needs: gates
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
# Portable targets: amd64, arm64, loong64 and riscv64 on Linux, at the toolchain
|
||||
# default level. No 32-bit, no wasm, no macOS, no Windows. FreeBSD stays out until
|
||||
# verify/jit.go ports off syscall.Mprotect: the Go syscall package defines no
|
||||
# Mprotect for freebsd, and verify/jit.go:50 calls it to drop the write bit from
|
||||
# the JIT mapping, so every freebsd target fails to build with "undefined:
|
||||
# syscall.Mprotect" (verified for amd64, arm64 and riscv64 on go1.27.1).
|
||||
include:
|
||||
- goos: linux
|
||||
goarch: amd64
|
||||
- goos: linux
|
||||
goarch: arm64
|
||||
- goos: linux
|
||||
goarch: riscv64
|
||||
- goos: linux
|
||||
goarch: loong64
|
||||
- goos: linux
|
||||
goarch: riscv64
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.27"
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
- name: Install Perl
|
||||
run: dnf install -y perl
|
||||
|
||||
- name: Validate tag and build
|
||||
id: build
|
||||
- name: Validate the tag
|
||||
id: version
|
||||
env:
|
||||
VERSION: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
perl -e '
|
||||
my $v = $ENV{VERSION} // q{};
|
||||
$v =~ m{^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$}
|
||||
or die qq{ERROR: expected a semver tag like v1.2.3, got: $v\n};
|
||||
(my $nv = $v) =~ s{^v}{};
|
||||
open(my $o, q{>>}, $ENV{GITEA_OUTPUT}) or die qq{GITEA_OUTPUT: $!};
|
||||
print $o qq{version_no_v=$nv\n};
|
||||
close($o);
|
||||
print qq{version $nv\n};
|
||||
'
|
||||
|
||||
if ! echo "$VERSION" | grep -qE '^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$'; then
|
||||
echo "ERROR: expected a semver tag like v1.2.3, got: '$VERSION'"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
VERSION_NO_V="${VERSION#v}"
|
||||
echo "version_no_v=${VERSION_NO_V}" >> "$GITEA_OUTPUT"
|
||||
|
||||
mkdir -p bin
|
||||
GOOS=${{ matrix.goos }} GOARCH=${{ matrix.goarch }} CGO_ENABLED=0 \
|
||||
go build -ldflags "-s -w -X main.version=${VERSION_NO_V}" \
|
||||
-o "bin/gasm-${VERSION_NO_V}-${{ matrix.goos }}-${{ matrix.goarch }}" \
|
||||
./cmd/gasm
|
||||
- name: Build
|
||||
env:
|
||||
VERSION_NO_V: ${{ steps.version.outputs.version_no_v }}
|
||||
GOOS: ${{ matrix.goos }}
|
||||
GOARCH: ${{ matrix.goarch }}
|
||||
CGO_ENABLED: "0"
|
||||
run: |
|
||||
# Nothing is injected. The toolchain records the tag into the binary's build
|
||||
# information, so the version is right because this build happens at the tag, and
|
||||
# there is no path for anyone to get wrong. -s -w only strips symbols.
|
||||
go build -ldflags "-s -w" -o "bin/gasm-${VERSION_NO_V}-${GOOS}-${GOARCH}" ./cmd/gasm
|
||||
|
||||
# Artifacts stay on v3: v4 and later detect Gitea as GHES and abort.
|
||||
- name: Upload artifact
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: gasm-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||
path: bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||
path: bin/gasm-${{ steps.version.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||
if-no-files-found: error
|
||||
|
||||
- name: Smoke test
|
||||
# Only a binary matching the runner can be run here. The check is not that --version
|
||||
# exits cleanly but that it reports the tag and nothing more: a build outside version
|
||||
# control reports (devel), and a build whose tree was dirty reports +dirty, and both
|
||||
# would otherwise be published.
|
||||
if: matrix.goos == 'linux' && matrix.goarch == 'amd64'
|
||||
env:
|
||||
TAG: ${{ gitea.ref_name }}
|
||||
BIN: bin/gasm-${{ steps.version.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||
run: |
|
||||
chmod +x bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||
./bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }} --version
|
||||
perl -e '
|
||||
my $want = $ENV{TAG} // die qq{ERROR: no tag\n};
|
||||
open(my $bin, q{-|}, $ENV{BIN}, q{--version}) or die qq{$ENV{BIN}: $!};
|
||||
my $got = <$bin>;
|
||||
close($bin);
|
||||
$got = defined $got ? $got : q{};
|
||||
chomp $got;
|
||||
index($got, $want) >= 0
|
||||
or die qq{ERROR: the binary printed "$got", which does not contain $want. Version control was disabled, so there is no recorded version.\n};
|
||||
index($got, q{+dirty}) < 0
|
||||
or die qq{ERROR: the binary printed "$got". The tree was dirty at build time, which means the checkout was not the tag, or the build artefacts are not ignored.\n};
|
||||
print qq{$ENV{BIN} reports $got\n};
|
||||
'
|
||||
|
||||
release:
|
||||
runs-on: fedora
|
||||
timeout-minutes: 15
|
||||
needs: build
|
||||
permissions:
|
||||
# contents: read is required for the checkout: a job that declares any
|
||||
# permissions gets a token scoped to exactly those, and releases: write
|
||||
# alone leaves the fetch with no read access, which Gitea answers with
|
||||
# a 404 "Repository not found". Verified on the instance 2026-09-16.
|
||||
contents: read
|
||||
releases: write
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
@@ -77,81 +226,165 @@ jobs:
|
||||
with:
|
||||
path: dist
|
||||
|
||||
- name: Extract CHANGELOG section
|
||||
- name: Install Perl
|
||||
run: dnf install -y perl
|
||||
|
||||
- name: Extract the CHANGELOG section
|
||||
env:
|
||||
VERSION: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
VERSION_NO_V="${VERSION#v}"
|
||||
# Each step derives what it needs from the tag, so no value has to travel between
|
||||
# jobs.
|
||||
perl -e '
|
||||
my $v = $ENV{VERSION} // q{};
|
||||
$v =~ s{^v}{};
|
||||
open(my $vout, q{>}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
|
||||
print $vout $v;
|
||||
close($vout);
|
||||
open(my $in, q{<}, q{CHANGELOG.md}) or die qq{CHANGELOG.md: $!};
|
||||
my @lines = <$in>;
|
||||
close($in);
|
||||
my ($start, $end) = (-1, scalar @lines);
|
||||
for my $i (0 .. $#lines) {
|
||||
if ($start < 0) { $start = $i if $lines[$i] =~ m{^##\s+\[\Q$v\E\]} }
|
||||
elsif ($lines[$i] =~ m{^##\s+\[}) { $end = $i; last }
|
||||
}
|
||||
$start >= 0 or die qq{ERROR: no CHANGELOG section for $v, expected a heading like: ## [$v] - YYYY-MM-DD\n};
|
||||
my @body = grep { m{\S} } @lines[$start + 1 .. $end - 1];
|
||||
@body or die qq{ERROR: the CHANGELOG section for $v is empty\n};
|
||||
open(my $out, q{>}, q{release-body.md}) or die qq{release-body.md: $!};
|
||||
print $out @body;
|
||||
close($out);
|
||||
printf qq{notes for %s: %d lines\n}, $v, scalar @body;
|
||||
'
|
||||
|
||||
sed -n "/^## \[${VERSION_NO_V}\] /,/^## \[/p" CHANGELOG.md \
|
||||
| sed '$d' \
|
||||
| tail -n +2 \
|
||||
> release-body.md
|
||||
- name: Build the release request
|
||||
run: |
|
||||
perl -e '
|
||||
open(my $vin, q{<}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
|
||||
my $v = <$vin>;
|
||||
close($vin);
|
||||
chomp $v;
|
||||
open(my $in, q{<:raw}, q{release-body.md}) or die qq{release-body.md: $!};
|
||||
my $body = do { local $/; <$in> };
|
||||
close($in);
|
||||
# Byte-oriented escaping: JSON is UTF-8, so non-ASCII passes through and only the
|
||||
# characters JSON forbids are rewritten.
|
||||
$body =~ s/([\\"])/\\$1/g;
|
||||
$body =~ s/\t/\\t/g;
|
||||
$body =~ s/\r//g;
|
||||
$body =~ s/\n/\\n/g;
|
||||
$body =~ s/([\x00-\x08\x0b\x0c\x0e-\x1f])/sprintf(q{\u%04x}, ord($1))/ge;
|
||||
my $json = sprintf(qq{{"tag_name":"v%s","name":"v%s","body":"%s","draft":false,"prerelease":false}}, $v, $v, $body);
|
||||
open(my $out, q{>}, q{release.json}) or die qq{release.json: $!};
|
||||
print $out $json;
|
||||
close($out);
|
||||
print qq{release.json written for v$v\n};
|
||||
'
|
||||
|
||||
if [ ! -s release-body.md ]; then
|
||||
echo "ERROR: no CHANGELOG section found for ${VERSION_NO_V}"
|
||||
echo "Expected a heading like: ## [${VERSION_NO_V}] — YYYY-MM-DD"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Create release
|
||||
- name: Create the release
|
||||
env:
|
||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
||||
GITEA_REPOSITORY: ${{ gitea.repository }}
|
||||
GITEA_REF_NAME: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
BODY=$(sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' -e 's/\t/\\t/g' -e 's/\r//g' release-body.md | sed ':a;N;$!ba;s/\n/\\n/g')
|
||||
BODY="\"${BODY}\""
|
||||
|
||||
response=$(curl -sS -w '\n%{http_code}' \
|
||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
||||
-H "Content-Type: application/json" \
|
||||
-X POST \
|
||||
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases" \
|
||||
-d "{\"tag_name\":\"${GITEA_REF_NAME}\",\"name\":\"${GITEA_REF_NAME}\",\"body\":${BODY},\"draft\":false,\"prerelease\":false}")
|
||||
|
||||
http_code=$(echo "$response" | tail -1)
|
||||
payload=$(echo "$response" | sed '$d')
|
||||
|
||||
echo "HTTP ${http_code}"
|
||||
if [ "$http_code" != "201" ]; then
|
||||
echo "Failed to create release: ${payload}"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
RELEASE_ID=$(echo "$payload" | grep -oE '"id"[[:space:]]*:[[:space:]]*[0-9]+' | head -1 | grep -oE '[0-9]+')
|
||||
echo "Created release ID=${RELEASE_ID}"
|
||||
printf '%s' "${RELEASE_ID}" > release-id.txt
|
||||
perl -e '
|
||||
my @cmd = (q{curl}, q{-sS}, q{-o}, q{response.json}, q{-w}, q{%{http_code}},
|
||||
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
|
||||
q{-H}, q{Content-Type: application/json},
|
||||
q{-X}, q{POST},
|
||||
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases},
|
||||
q{--data-binary}, q{@release.json});
|
||||
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
|
||||
my $code = <$curl>;
|
||||
my $ok = close($curl);
|
||||
my $exit = $? >> 8;
|
||||
$code = defined $code ? $code : q{};
|
||||
$ok or die qq{ERROR: curl failed (exit $exit) calling $ENV{GITEA_SERVER_URL}\n};
|
||||
open(my $r, q{<:raw}, q{response.json}) or die qq{response.json: $!};
|
||||
my $body = do { local $/; <$r> };
|
||||
close($r);
|
||||
$code eq q{201} or die qq{ERROR: the release was not created, HTTP $code: $body\n};
|
||||
$body =~ m{"id"\s*:\s*([0-9]+)} or die qq{ERROR: no release id in the response: $body\n};
|
||||
open(my $o, q{>}, q{release-id.txt}) or die qq{release-id.txt: $!};
|
||||
print $o $1;
|
||||
close($o);
|
||||
print qq{release id $1\n};
|
||||
'
|
||||
|
||||
- name: Upload assets
|
||||
env:
|
||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
||||
GITEA_REPOSITORY: ${{ gitea.repository }}
|
||||
GITEA_REF_NAME: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
RELEASE_ID=$(cat release-id.txt)
|
||||
|
||||
for binary in dist/gasm-*/gasm-*; do
|
||||
[ -f "$binary" ] || continue
|
||||
fname=$(basename "$binary")
|
||||
echo "Uploading ${fname}..."
|
||||
http_code=$(curl -sS -o /dev/null -w '%{http_code}' \
|
||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
||||
-H "Content-Type: application/octet-stream" \
|
||||
-X POST \
|
||||
--data-binary "@${binary}" \
|
||||
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases/${RELEASE_ID}/assets?name=${fname}")
|
||||
echo " HTTP ${http_code}"
|
||||
if [ "$http_code" != "201" ]; then
|
||||
echo "Failed to upload ${fname}"
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
|
||||
echo "Release ${GITEA_REF_NAME} is live."
|
||||
perl -e '
|
||||
open(my $f, q{<}, q{release-id.txt}) or die qq{release-id.txt: $!};
|
||||
my $id = <$f>;
|
||||
close($f);
|
||||
chomp $id;
|
||||
my @files = grep { -f $_ } glob(q{dist/*/*});
|
||||
@files or die qq{ERROR: no assets under dist/\n};
|
||||
# A file that arrived empty from the artifact step would be uploaded as an
|
||||
# empty attachment, every status would still be 201, and the run would go
|
||||
# green over a release nobody can install. Refuse it here, before the
|
||||
# upload, and verify what was stored afterwards.
|
||||
my %size;
|
||||
for my $path (@files) {
|
||||
my $n = -s $path // 0;
|
||||
(my $name = $path) =~ s{.*/}{};
|
||||
$n > 0 or die qq{ERROR: $path is empty, so there is nothing to upload\n};
|
||||
$size{$name} = $n;
|
||||
}
|
||||
my $bad = 0;
|
||||
for my $path (@files) {
|
||||
(my $name = $path) =~ s{.*/}{};
|
||||
my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}},
|
||||
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
|
||||
q{-H}, q{Content-Type: application/octet-stream},
|
||||
# The @ must not sit inside a qq{} string: there it starts an
|
||||
# array interpolation and the upload body collapses to empty,
|
||||
# which Gitea stores as a 201-created zero-byte attachment.
|
||||
q{-X}, q{POST}, q{--data-binary}, q{@} . $path,
|
||||
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name});
|
||||
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
|
||||
my $code = <$curl>;
|
||||
my $ok = close($curl);
|
||||
my $exit = $? >> 8;
|
||||
$code = defined $code ? $code : q{};
|
||||
unless ($ok) {
|
||||
printf qq{%s: curl failed (exit %d)\n}, $name, $exit;
|
||||
$bad = 1;
|
||||
next;
|
||||
}
|
||||
printf qq{%s: HTTP %s\n}, $name, $code;
|
||||
$bad = 1 if $code ne q{201};
|
||||
}
|
||||
# Read every asset back through the release download route and require the
|
||||
# served length to be the file that was sent: stored but empty is a broken
|
||||
# release however green the run looks.
|
||||
open(my $v, q{<}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
|
||||
my $v = <$v>;
|
||||
close($v);
|
||||
chomp $v;
|
||||
for my $name (sort keys %size) {
|
||||
my $url = qq{$ENV{GITEA_SERVER_URL}/$ENV{GITEA_REPOSITORY}/releases/download/v$v/$name};
|
||||
my @head = (q{curl}, q{-sS}, q{-I}, q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}}, $url);
|
||||
open(my $h, q{-|}, @head) or die qq{curl: $!};
|
||||
my $len;
|
||||
my $status;
|
||||
while (my $l = <$h>) {
|
||||
$status = $1 if $l =~ m{^HTTP/\S+\s+(\d+)};
|
||||
$len = $1 if $l =~ m{^content-length:\s*(\d+)}i;
|
||||
}
|
||||
my $ok = close($h);
|
||||
$len = defined $len ? $len : 0;
|
||||
if (!$ok || $status != 200 || $len != $size{$name}) {
|
||||
printf qq{ERROR: %s serves %s bytes, expected %d\n}, $name, $len, $size{$name};
|
||||
$bad = 1;
|
||||
next;
|
||||
}
|
||||
printf qq{%s: serves %d bytes\n}, $name, $len;
|
||||
}
|
||||
exit($bad ? 1 : 0);
|
||||
'
|
||||
|
||||
+112
-76
@@ -1,4 +1,17 @@
|
||||
# Test — gasm-devkit. Runs on push and pull request to development.
|
||||
# Test, Go. Push and pull request to development. Never on main.
|
||||
#
|
||||
# The gates are the ones the justfile's `gates` recipe runs, minus race: the shared
|
||||
# runner box cannot afford the race detector on every push, so it lives in race.yml.
|
||||
# The box is one core and 2 GB beside Gitea, so parallelism is bounded on purpose and
|
||||
# everything runs in one job. Extra jobs would duplicate the checkout, the Go setup and
|
||||
# the dependency download three times without buying any parallelism.
|
||||
#
|
||||
# Every step is one command, so the step that fails is the gate that failed, and no shell
|
||||
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
|
||||
# not Python: Perl behaves the same on both runner images, there is no bashism to trip over
|
||||
# on ash, and it is one language instead of two. The Perl uses builtins only, because
|
||||
# Fedora packages the Perl modules separately and nothing beyond `perl` itself may be
|
||||
# assumed present.
|
||||
name: Test
|
||||
|
||||
on:
|
||||
@@ -7,90 +20,113 @@ on:
|
||||
pull_request:
|
||||
branches: [development]
|
||||
|
||||
env:
|
||||
# One core: parallelism buys no speed here and costs memory the box does not have.
|
||||
GOFLAGS: -p=1
|
||||
GOMAXPROCS: "2"
|
||||
|
||||
# A superseded run of the same ref is cancelled instead of queueing behind one that
|
||||
# no longer matters. Verified on Gitea 1.27.1 on 2026-09-17: a queued run whose ref
|
||||
# moved on is cancelled before it ever reaches the runner, while a run already
|
||||
# dispatched there runs to completion.
|
||||
concurrency:
|
||||
group: ${{ gitea.workflow }}-${{ gitea.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
vet:
|
||||
runs-on: fedora
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.27"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
|
||||
- name: gofmt
|
||||
run: |
|
||||
set -euo pipefail
|
||||
unformatted=$(gofmt -l .)
|
||||
if [ -n "$unformatted" ]; then
|
||||
echo "These files need gofmt:"
|
||||
echo "$unformatted"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: go vet
|
||||
run: go vet ./...
|
||||
|
||||
test:
|
||||
runs-on: fedora
|
||||
needs: vet
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.27"
|
||||
# The module is the source of truth for the version, so it cannot drift.
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
|
||||
- name: Install gcc
|
||||
run: dnf install -y gcc
|
||||
|
||||
- name: go test -race
|
||||
run: go test -race -count=1 ./...
|
||||
|
||||
- name: Coverage gate — 80 % minimum
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# Exclude packages inherently untestable without hardware:
|
||||
# debug — interactive ptrace, requires a live process
|
||||
# cmd/gasm — CLI glue, covered by integration tests
|
||||
go test -coverprofile=coverage.out \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/arch \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/asm \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/ast \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/format \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/lexer \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/lint \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/lsp \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/parser \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/token \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/verify
|
||||
coverage=$(go tool cover -func=coverage.out | awk '/^total:/ { gsub("%", "", $3); print $3 }')
|
||||
echo "Total coverage: ${coverage}%"
|
||||
if awk -v c="$coverage" 'BEGIN { exit !(c+0 < 80) }'; then
|
||||
echo "ERROR: coverage ${coverage}% is below the 80% threshold"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
build:
|
||||
runs-on: fedora
|
||||
needs: test
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.27"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
- name: Install Perl
|
||||
# The runner images are minimal and Perl is not guaranteed. The install is a
|
||||
# no-op where it is already present; drop this step once verified on the box.
|
||||
run: dnf install -y perl
|
||||
|
||||
# The steps follow the `gates` order of the justfile contract: build, format,
|
||||
# vet, test. The vet gate is go vet and go fix -diff, two steps here.
|
||||
- name: Build
|
||||
run: go build -ldflags="-s -w" -o bin/gasm ./cmd/gasm
|
||||
run: go build ./...
|
||||
|
||||
- name: Smoke test
|
||||
run: ./bin/gasm --version
|
||||
- name: FreeBSD build (amd64)
|
||||
# The debugger's ptrace surface and the JIT substrate are the two
|
||||
# FreeBSD-portable layers the tree carries; the forge has no FreeBSD
|
||||
# runner, so a push can only compile-gate them. Running the ptrace
|
||||
# suite needs real FreeBSD hardware.
|
||||
env:
|
||||
GOOS: freebsd
|
||||
GOARCH: amd64
|
||||
run: go build ./...
|
||||
|
||||
- name: FreeBSD build (arm64)
|
||||
env:
|
||||
GOOS: freebsd
|
||||
GOARCH: arm64
|
||||
run: go build ./...
|
||||
|
||||
- name: FreeBSD build (riscv64)
|
||||
env:
|
||||
GOOS: freebsd
|
||||
GOARCH: riscv64
|
||||
run: go build ./...
|
||||
|
||||
- name: Format
|
||||
run: |
|
||||
perl -e '
|
||||
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
|
||||
my @bad = <$g>;
|
||||
close($g);
|
||||
print @bad;
|
||||
exit(@bad ? 1 : 0);
|
||||
'
|
||||
|
||||
- name: Vet
|
||||
run: go vet ./...
|
||||
|
||||
- name: Modernise
|
||||
# Exits non-zero when it has something to rewrite, so it needs no output capture.
|
||||
run: go fix -diff ./...
|
||||
|
||||
- name: Tests
|
||||
# The suite must be fast: a push pipeline that cannot finish in a few minutes moves
|
||||
# its heavy part behind a dispatch. The inner timeout matches the job's, so a
|
||||
# hanging test reports its own goroutine dump rather than a silent job kill.
|
||||
# The pattern is `packages` in the project's justfile: the logic packages, since a
|
||||
# thin cmd/ would drag the total under the floor. release.yml runs the same
|
||||
# command, so the floor is the same number everywhere. ./verify/... carries the
|
||||
# live oracle-parity comparison against `go tool asm` (the TestGroundTruth
|
||||
# suites); the runner's Go setup provides both the tool and GOROOT.
|
||||
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
|
||||
|
||||
- name: Tests outside the coverage set
|
||||
# The CLI and the debugger sit outside `packages` because a thin main and a
|
||||
# ptrace-bound package pull the total under the floor, but their tests guard
|
||||
# shipped surfaces: the command exit codes, the manual pages against the
|
||||
# binary's own help, and the debugger's architecture-neutral units. They run
|
||||
# here so the floor stays a product measure and nothing is left untested.
|
||||
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
|
||||
|
||||
- name: Oracle parity
|
||||
# Re-run the live go-tool-asm comparison as its own step so that a parity
|
||||
# regression names the gate that failed instead of hiding inside the suite.
|
||||
run: go test -count=1 -timeout 10m -run 'TestGroundTruth' ./verify/...
|
||||
|
||||
- name: Coverage floor
|
||||
run: |
|
||||
perl -e '
|
||||
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
|
||||
my $total;
|
||||
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
|
||||
close($c);
|
||||
die qq{no total line in coverage.out\n} unless defined $total;
|
||||
printf qq{Total coverage: %s%%\n}, $total;
|
||||
exit($total < 80 ? 1 : 0);
|
||||
'
|
||||
|
||||
+9
-10
@@ -1,14 +1,13 @@
|
||||
# Binaries
|
||||
/gasm
|
||||
/bin/
|
||||
*.exe
|
||||
.idea/
|
||||
.zcode/
|
||||
|
||||
# Test and coverage artefacts
|
||||
# Build output
|
||||
/bin/
|
||||
/gasm
|
||||
coverage.out
|
||||
*.test
|
||||
|
||||
# Scratch / temporary work
|
||||
_scratch/
|
||||
|
||||
# ZCode workspace
|
||||
.zcode
|
||||
# Crash dumps from the emulator runs
|
||||
core
|
||||
core.*
|
||||
*.core
|
||||
|
||||
@@ -1,119 +0,0 @@
|
||||
# AGENTS.md — gasm-devkit
|
||||
|
||||
Repository rules for AI agents and contributors. Read before modifying any
|
||||
code in this repository.
|
||||
|
||||
## AI Contribution Policy
|
||||
|
||||
AI agents may assist with code, documentation, tests, and review in this
|
||||
repository. All AI-assisted changes must:
|
||||
|
||||
- Follow the code style and conventions in this file.
|
||||
- Include the trailer `Assisted-by: <model-name>` in every commit message.
|
||||
- Not commit directly to `main` — work on `development`.
|
||||
- Pass the full Definition of Done before any commit.
|
||||
|
||||
## Workflow
|
||||
|
||||
- **Branching.** `development` is the working branch. `main` is
|
||||
release-only: merge from `development`, then tag. Never commit directly
|
||||
to `main`.
|
||||
- **Release procedure.**
|
||||
1. Bump `version` in `justfile` and `cmd/gasm/main.go`.
|
||||
2. Update `CHANGELOG.md` with a new `## [X.Y.Z] — YYYY-MM-DD` section.
|
||||
3. Update `README.md` and `docs/ARCHITECTURE.md` if user-visible
|
||||
behaviour changed.
|
||||
4. Run the Definition of Done (below).
|
||||
5. Commit on `development`.
|
||||
6. `git checkout main && git merge --ff-only development`.
|
||||
7. `git tag vX.Y.Z`.
|
||||
8. `git checkout development`.
|
||||
9. `GOBIN=~/.local/bin just install-bin`.
|
||||
|
||||
## Commit Messages
|
||||
|
||||
Conventional Commits, subject line only, imperative mood, lowercase after
|
||||
the colon:
|
||||
|
||||
```
|
||||
feat(asm): add EVEX gather and scatter with VSIB addressing
|
||||
```
|
||||
|
||||
Allowed types: `feat`, `fix`, `docs`, `style`, `refactor`, `perf`, `test`,
|
||||
`chore`, `ci`, `build`, `revert`.
|
||||
|
||||
Every commit ends with exactly one trailer, using the model that
|
||||
assisted with the change:
|
||||
|
||||
```
|
||||
Assisted-by: <model-name>
|
||||
```
|
||||
|
||||
Replace `<model-name>` with the actual model (e.g. `DeepSeek V4 Pro`).
|
||||
|
||||
No body, no footers, no trailing period on the subject.
|
||||
|
||||
## Code Style
|
||||
|
||||
Language: Go 1.27 (`toolchain go1.27.0`).
|
||||
|
||||
### Formatter
|
||||
|
||||
`gofmt` — zero diff. Run `just fmt` before committing.
|
||||
|
||||
### Linter
|
||||
|
||||
`go vet` — zero warnings. Run `just build` before committing.
|
||||
|
||||
### Tests
|
||||
|
||||
`go test -race -count=1 ./...` — all green, coverage ≥ 80 % (hard gate,
|
||||
enforced by `just test`).
|
||||
|
||||
### Dependencies
|
||||
|
||||
- **Production code:** standard library only. No third-party imports in
|
||||
shipped code.
|
||||
- **Test code:** `golang.org/x/arch` is the sole test dependency (decode
|
||||
oracle for round-trip validation). It is never linked into the binary.
|
||||
- **No cgo, no C, no external toolchains, no JavaScript.**
|
||||
|
||||
### Error Handling
|
||||
|
||||
Explicit `if err != nil`. Wrap with `fmt.Errorf("context: %w", err)`.
|
||||
No panics outside `main`. The one exception: the JIT trampoline's
|
||||
`recover`-guarded decoder hot path, which converts bounds panics to
|
||||
sentinel errors.
|
||||
|
||||
### Assembly
|
||||
|
||||
Plan 9 syntax (Go's assembler dialect). Hand-written — no code generators
|
||||
except `_gen/gen.go` for instruction tables (which parses the Go
|
||||
toolchain source). Every instruction table is committed; no runtime
|
||||
dependency on the Go toolchain.
|
||||
|
||||
### File Naming
|
||||
|
||||
- `_amd64.s`, `_arm64.s`, `_riscv64.s`, `_loong64.s` for
|
||||
architecture-specific assembly.
|
||||
- `_linux_amd64.go` for platform-specific Go files.
|
||||
- `_test.go` suffix for test files.
|
||||
|
||||
## Definition of Done
|
||||
|
||||
A task is not complete until all of these pass:
|
||||
|
||||
1. `just build` — `go vet` + `gofmt` check, zero errors, zero warnings.
|
||||
2. `just test` — full suite with `-race`, coverage ≥ 80 %.
|
||||
3. `just fmt` — produces no diff.
|
||||
4. Diagnostics — zero warnings across the project.
|
||||
5. Non-trivial changes reviewed.
|
||||
|
||||
## Licence
|
||||
|
||||
BSD-3-Clause. Every source file carries the SPDX header:
|
||||
|
||||
```
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
```
|
||||
+860
-152
File diff suppressed because it is too large
Load Diff
+108
-75
@@ -1,101 +1,134 @@
|
||||
# Contributing to gasm-devkit
|
||||
# Contributing
|
||||
|
||||
## Prerequisites
|
||||
Contributions to **gasm-sdk** are governed by the Contributor terms
|
||||
below; submitting one means you accept them.
|
||||
|
||||
- Go 1.27 or later (`toolchain go1.27.0`)
|
||||
- `just` command runner
|
||||
- A Linux host on amd64, arm64, riscv64 or loong64
|
||||
## Contributor terms
|
||||
|
||||
## Development Setup
|
||||
1. This project belongs to its owner alone. The owner decides what is
|
||||
accepted, in what form and when; the decision is final and needs no
|
||||
justification.
|
||||
2. By submitting a contribution you assign to Petr Balvín
|
||||
<opensource@petrbalvin.org> all present and future copyright and
|
||||
related rights in it, worldwide, for the full term of the rights,
|
||||
with the right to relicense and sublicense without restriction,
|
||||
including under proprietary terms.
|
||||
3. Where that assignment is not effective, it counts as a perpetual,
|
||||
irrevocable, royalty-free licence with the same scope.
|
||||
4. To the fullest extent permitted by law, you waive any right of
|
||||
attribution and integrity in the contribution. The project names no
|
||||
contributors and keeps no credits list.
|
||||
5. By submitting you represent that the work is yours and that you
|
||||
hold the rights to assign it as above.
|
||||
|
||||
## Development setup
|
||||
|
||||
Requirements: Go 1.27.1, the exact version the `go` directive in `go.mod`
|
||||
declares, [just](https://github.com/casey/just) for the recipes, and a C
|
||||
compiler (gcc), because `just gates` includes `just race` and the race
|
||||
detector needs cgo.
|
||||
|
||||
```sh
|
||||
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
|
||||
cd gasm-devkit
|
||||
just install # download module dependencies
|
||||
just build # go vet + gofmt check
|
||||
just test # full test suite with race detector
|
||||
git clone https://sourcedock.dev/petrbalvin/gasm-sdk.git
|
||||
cd gasm-sdk
|
||||
just build
|
||||
just gates
|
||||
```
|
||||
|
||||
## Commands
|
||||
## Workflow
|
||||
|
||||
Every just recipe:
|
||||
1. Branch from `development`. Never commit directly to `main`, which is release-only.
|
||||
2. Commit in [Conventional Commits](https://www.conventionalcommits.org/) form:
|
||||
`type(scope): description`, subject line only, imperative mood, lowercase after the
|
||||
colon, no trailing full stop. Allowed types: `feat`, `fix`, `docs`, `style`,
|
||||
`refactor`, `perf`, `test`, `chore`, `ci`, `build`, `revert`.
|
||||
3. One logical change per commit. A refactor, a behaviour change and a formatting pass
|
||||
are three commits, never one.
|
||||
4. Record every user-visible change in `CHANGELOG.md` under `## [development]`.
|
||||
5. Add or update tests. Coverage stays at 80 percent or more; it is a hard gate.
|
||||
6. Update the documentation when the public API, the configuration or the behaviour
|
||||
changes.
|
||||
7. Open a pull request against `development`.
|
||||
|
||||
| Recipe | What it does |
|
||||
|--------|-------------|
|
||||
| `just` | List all recipes |
|
||||
| `just install` | `go mod download` |
|
||||
| `just build` | `go vet ./...` + `gofmt -l .` check — zero errors required |
|
||||
| `just test` | `go test -race -count=1 -coverprofile=coverage.out ./...` + 80 % coverage gate |
|
||||
| `just fmt` | `gofmt -w .` |
|
||||
| `just run -- lint file.s` | Run the CLI with `go run` (args after `--`) |
|
||||
| `just install-bin` | Install `gasm` into `$GOBIN` with the release version stamped |
|
||||
| `just gen` | Regenerate `arch/*_gen.go` instruction tables from the Go toolchain |
|
||||
| `just uninstall` | Remove build artefacts (`coverage.out`, `gasm`, `*.test`) |
|
||||
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The release
|
||||
workflow builds the assets and publishes the release and its notes.
|
||||
|
||||
## Running a Single Test
|
||||
## Code style
|
||||
|
||||
```sh
|
||||
go test -run TestVexGroundTruth ./asm/
|
||||
go test -run TestDifferentialLZ4Fuzz ./verify/
|
||||
```
|
||||
`gofmt` and `go vet` run through `just fmt` and `just vet`, with zero diff and zero
|
||||
warnings tolerated. `just vet` is two gates, `go vet ./...` and `go fix -diff ./...`,
|
||||
so the modernisation rewrites are enforced too. `just gates` is the definition of done in
|
||||
one command, and the recipe file names what it contains. Errors are checked explicitly,
|
||||
wrapped as `fmt.Errorf("context: %w", err)`, and nothing panics outside `main`. The
|
||||
recipe file holds the commands, and the language and standard-library surface is the one
|
||||
the `go` directive in `go.mod` pins.
|
||||
|
||||
## Testing the Debugger
|
||||
- `golang.org/x/arch` is the one module dependency, and it is linked into the binary:
|
||||
`gasm dis` and the debugger's listings decode through it. Everything else is the
|
||||
standard library.
|
||||
- No cgo and no C. The standalone encoder paths (`gasm asm --format raw` and `--format
|
||||
elf`) need no Go installation; `gasm verify --ground-truth`, `gasm verify --fuzz`,
|
||||
`gasm audit-instructions` and `gasm asm --format goobj` resolve through the installed
|
||||
Go toolchain.
|
||||
- The parser, lexer and formatter are hand-written; the `arch` instruction tables are
|
||||
generated only by `_gen/gen.go` (`just gen`) and never edited by hand.
|
||||
- Assembly committed to the repository goes through `gasm fmt` and `gasm lint`, so a
|
||||
`.s` file that `gasm fmt -l .` lists is unfinished.
|
||||
|
||||
The interactive debugger (`gasm debug`) requires a compiled binary —
|
||||
`go run` does not work for the child process. Install first:
|
||||
New source files open with the project's two-line licence header, whose SPDX
|
||||
identifier matches `LICENSE`. Configuration files, workflows and dotfiles do not carry
|
||||
it.
|
||||
|
||||
```sh
|
||||
just install-bin
|
||||
gasm debug --func add testdata/verify/basic_amd64.s
|
||||
```
|
||||
## AI contribution policy
|
||||
|
||||
## Code Style
|
||||
AI tools are welcome as productivity aids and are a normal part of modern software
|
||||
development. What matters is that the contribution stays understandable, reviewable and
|
||||
genuinely useful.
|
||||
|
||||
See [AGENTS.md](AGENTS.md) for the full style guide. Key points:
|
||||
- **Disclose the assistance.** If AI helped draft any part of a commit, issue, pull
|
||||
request or review, say so.
|
||||
- **Commit messages carry exactly one trailer**, as a git trailer on the line after a
|
||||
blank line that closes the subject:
|
||||
|
||||
- `gofmt` — zero diff.
|
||||
- `go vet` — zero warnings.
|
||||
- Standard library only in production code; `golang.org/x/arch` in tests.
|
||||
- No cgo, no C, no JavaScript.
|
||||
- Hand-written Plan 9 assembly; tables generated only via `_gen/gen.go`.
|
||||
```
|
||||
Assisted-by: MODEL
|
||||
```
|
||||
|
||||
## Branches and Releases
|
||||
Name the model that did the work, spelled the way its maker spells it, for example
|
||||
`GLM 5.3`, `DeepSeek V4.1 Flash` or `Qwen 3.8 Flash`. No `Co-Authored-By`, no `Signed-off-by`,
|
||||
no other trailers, and no prose: the trailer is the disclosure.
|
||||
- **Issues and pull requests** attribute the assistance in a comment, for example
|
||||
`_Assisted-by: GLM 5.3_`. It does not belong in the pull request description.
|
||||
- **Take responsibility.** You are accountable for the accuracy, completeness and
|
||||
intent of everything you submit, whether or not AI produced it.
|
||||
- **Review before marking ready.** Read the diff carefully, run it locally, and add the
|
||||
tests it needs. Do not mark a pull request ready until you can defend every change in
|
||||
it.
|
||||
- **Quality over quantity.** Contributions that look like un-reviewed output, or whose
|
||||
author cannot engage substantively during review, may be closed.
|
||||
- **Preferred models.** Prefer open-weight models with transparent training data and
|
||||
minimal output filtering.
|
||||
|
||||
- `development` is the working branch.
|
||||
- `main` is release-only: `git merge --ff-only development`, then `git tag vX.Y.Z`.
|
||||
- Conventional Commits: `feat(asm): add EVEX gather and scatter`.
|
||||
- Every commit ends with `Assisted-by: <model-name>`.
|
||||
AI assists. It does not replace judgement.
|
||||
|
||||
## CI
|
||||
## Continuous integration
|
||||
|
||||
CI runs on every push to `development` and on pull requests:
|
||||
Workflows live in `.gitea/workflows/` and run on the project's own runners:
|
||||
|
||||
- **Test** (`test.yml`) — `gofmt` check, `go vet`, `go test -race` and the
|
||||
80 % coverage gate.
|
||||
- **Release** (`release.yml`) — cross-compiles release binaries for
|
||||
linux/{amd64,arm64,riscv64,loong64} on version tags and publishes them.
|
||||
| Workflow | Trigger | What it does |
|
||||
|---|---|---|
|
||||
| Test | push or pull request to `development` | build, format check, vet, modernisation, the test suite with the coverage floor, the CLI and debugger tests outside the profile, then the oracle-parity rerun against `go tool asm` |
|
||||
| Release | a `v*` tag | the same gates as Test minus the oracle-parity step, then the matrix build, the version smoke test and the release itself; the race detector runs locally in `just gates` before the tag is cut |
|
||||
|
||||
The Definition of Done (`just build` + `just test` + `just fmt`) must still
|
||||
pass locally before pushing.
|
||||
The local equivalent is `just gates`, which is the same set plus the race detector. The
|
||||
race detector also has its own workflow, dispatched by hand; it never runs on a push or a
|
||||
tag, where it would double the time and the memory a shared runner cannot spare.
|
||||
|
||||
## AI-Assisted Contributions
|
||||
## Reporting bugs
|
||||
|
||||
AI agents may assist with code, documentation, tests, and review. All
|
||||
AI-assisted changes must:
|
||||
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-sdk/issues` with the
|
||||
version, the operating system and architecture, the exact command, the full output,
|
||||
and the expected against the actual behaviour.
|
||||
|
||||
- Include the trailer `Assisted-by: <model-name>` in the commit message
|
||||
(e.g. `Assisted-by: DeepSeek V4 Pro`).
|
||||
- Follow the [AGENTS.md](AGENTS.md) rules.
|
||||
- Pass the Definition of Done before committing.
|
||||
|
||||
Attribute agent authorship in issues and pull requests on one trailing
|
||||
line:
|
||||
|
||||
```
|
||||
_Assisted-by: Qwen 3.8 Max_
|
||||
```
|
||||
|
||||
## Questions
|
||||
|
||||
Open an issue at
|
||||
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit/issues).
|
||||
**Security issues do not go in the issue tracker.** Report them as
|
||||
[SECURITY.md](SECURITY.md) describes, to **opensource@petrbalvin.org**.
|
||||
|
||||
@@ -1,144 +1,338 @@
|
||||
# gasm-devkit
|
||||
# GAsm: Software Development Kit for Plan 9 Assembly
|
||||
|
||||
Developer tooling for **GAsm** — Go's built-in Plan 9 assembler.
|
||||
> **Warning: this is an experiment.** gasm-sdk is under active
|
||||
> development and is not stable. The version is 0.x.x: commands, flags,
|
||||
> output formats and behaviour can change without warning at any time.
|
||||
> A 1.0.0 release is light years away. Nothing in this document is a
|
||||
> stability promise. For all of that, this is not a paper project: gasm
|
||||
> is already in active use and is tested on real assembly work. Only
|
||||
> amd64 is validated on real hardware; the other three architectures run
|
||||
> under emulation ([Validation status](#validation-status)).
|
||||
|
||||
Go ships an assembler but no tooling for it. There is no syntax highlighting,
|
||||
no autocomplete, no linter, no static analyser, no formatter, no standalone
|
||||
assembler and no debugger for `.s` files. Developers write assembly blind,
|
||||
validate it by benchmark, and debug it by print statement.
|
||||
**GAsm** is Go's Plan 9 assembler, and Go ships it without tooling:
|
||||
there is no formatter, no linter and no debugger for `.s` files, and no
|
||||
assembler that works without a Go installation. Developers write
|
||||
assembly blind, validate it by benchmark, and debug it by print
|
||||
statement. gasm-sdk is the missing toolkit: a single, self-contained
|
||||
binary, `gasm`, that serves both purposes.
|
||||
|
||||
gasm-devkit is the missing toolkit. It is a single, self-contained binary —
|
||||
`gasm` — that brings proper developer tooling to Plan 9 assembly:
|
||||
- **Help develop Plan 9 assembly.** Formatting, linting, disassembly,
|
||||
dynamic verification, a source-level debugger and a language server,
|
||||
for `.s` files in Go programs.
|
||||
- **Use Plan 9 assembly outside the Go toolchain.** `gasm asm` encodes
|
||||
on its own and writes raw images or linkable ELF objects with DWARF5
|
||||
debug sections, with no Go installation in the loop; the Go
|
||||
toolchain's own GOOBJ format, which `go build` consumes in place of
|
||||
the toolchain's output, needs the installed toolchain.
|
||||
|
||||
```
|
||||
gasm tokens dump the lexical token stream
|
||||
gasm parse parse and report syntax errors
|
||||
gasm fmt canonicalise formatting (gofmt for assembly)
|
||||
gasm lint static checks
|
||||
gasm lsp language server (completion, hover, symbols, diagnostics, highlighting)
|
||||
gasm asm standalone assembler
|
||||
gasm verify dynamic analysis & verification
|
||||
gasm debug source-level debugger
|
||||
gasm diff compare machine code of two .s files
|
||||
gasm profile show basic-block structure of functions
|
||||
## Why Plan 9 assembly
|
||||
|
||||
Plan 9 assembly is the quiet triumph of the field. One syntax across
|
||||
every architecture Go builds for: the same source-first operand order,
|
||||
the same four pseudo-registers, the same frame convention, whether the
|
||||
target is x86, ARM, RISC-V or LoongArch. Learn it once and you can
|
||||
read a kernel on any of them.
|
||||
|
||||
Compare the alternatives. Intel syntax and AT&T syntax disagree on the
|
||||
one question every instruction answers, which operand is the source
|
||||
and which is the destination, so half the world writes it one way,
|
||||
half the other, and every assembly programmer carries both in their
|
||||
head forever. GNU as settles the argument with directives that switch
|
||||
dialects mid-file (`.intel_syntax noprefix`), a percent sign on every
|
||||
register and a dollar on every immediate: punctuation that carries
|
||||
nothing the operand order did not already say. And the x86 family
|
||||
fragments again underneath: NASM is not MASM is not GAS, each with its
|
||||
own directive zoo and macro language, so every project picks a dialect
|
||||
and every reader learns a different one by accident.
|
||||
|
||||
Plan 9 assembly has none of it. Registers are bare names. Memory is
|
||||
one notation, `offset(base)`, extended by an index and a scale when
|
||||
the instruction needs it. Arguments arrive named and offset-checked:
|
||||
`x+0(FP)` is the argument x, on every architecture, and `go vet`
|
||||
polices the offsets against the Go prototype.
|
||||
|
||||
```text
|
||||
AT&T (GNU as): movq %rax, -16(%rbp)
|
||||
Plan 9 (Go): MOVQ AX, total-16(SP)
|
||||
```
|
||||
|
||||
## Architecture support
|
||||
The same lines, but only one of them tells you what the number is for.
|
||||
The syntax is uppercase, regular and boring, which is the highest
|
||||
compliment a language for machine code can earn. gasm-sdk exists
|
||||
to give that syntax the tooling it deserves.
|
||||
|
||||
gasm-devkit targets every architecture Go's assembler speaks. The instruction
|
||||
tables are **generated from the Go toolchain's own assembler source**
|
||||
(`cmd/internal/obj/<arch>`), so gasm-devkit recognises *every* mnemonic the
|
||||
real assembler accepts — not a hand-maintained subset that drifts and rots.
|
||||
## Features
|
||||
|
||||
| Architecture | GOARCH | File suffix | Instructions recognised |
|
||||
|--------------|-------------|----------------|------------------------------------|
|
||||
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
|
||||
| ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes |
|
||||
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes |
|
||||
| LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes |
|
||||
- **Front end.** A hand-written lexer and an error-tolerant parser produce a
|
||||
typed AST with source positions; `gasm tokens` and `gasm parse` expose them
|
||||
directly.
|
||||
- **Formatter.** `gasm fmt` canonicalises indentation, operand spacing,
|
||||
per-function mnemonic alignment and blank-line layout: `gofmt` for assembly,
|
||||
operating recursively on directories the way `go fmt` does. `-l` lists
|
||||
files whose formatting differs and `-d` prints a unified diff.
|
||||
- **Linter.** `gasm lint` runs 18 conservative static checks, among them
|
||||
`undefined-label`, `abi-argsize` (declared argument area vs the `// func`
|
||||
signature), `register-clobber` (Go ABI register liveness over the
|
||||
control-flow graph), `stack-imbalance`, `abi0-register-args` and
|
||||
`unencodable-instruction`.
|
||||
- **Standalone assembler.** `gasm asm` encodes all four architectures without
|
||||
the Go toolchain and writes raw images or linkable ELF objects (with DWARF5
|
||||
debug sections) with no Go installation needed, or the Go toolchain's own
|
||||
GOOBJ format, which needs the installed toolchain and which `go build`
|
||||
consumes in place of the toolchain's output. Framed functions get the
|
||||
stack-split guard and the morestack block, byte-identical to the
|
||||
toolchain's, so split functions link too. The assembler preprocesses
|
||||
like the toolchain (`#define`, `#include` with `-I`, `#ifdef`), generates
|
||||
`go_asm.h` from the package's Go files, and carries `PCALIGN`, the
|
||||
`LOCK`/`REP` prefixes and the literal-data pseudo-ops.
|
||||
- **Disassembler.** `gasm dis` lists a `.s` file's functions at their real
|
||||
offsets after assembling, or disassembles raw bytes from a file or stdin.
|
||||
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
|
||||
executable memory: smoke calls, ABI checks (sentinel registers, red-zone
|
||||
canary), differential fuzzing against the `go tool asm` build, and
|
||||
byte-for-byte ground-truth comparison of the machine code.
|
||||
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
|
||||
breakpoints (optionally conditional), hardware watchpoints, register and
|
||||
memory inspection, and headless script runs that report instruction and
|
||||
label coverage; it runs on Linux (all four architectures) and FreeBSD
|
||||
(amd64, arm64, riscv64).
|
||||
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
|
||||
push and pull diagnostics, semantic-token highlighting, go-to-definition,
|
||||
find references, rename, formatting, inlay hints, code actions, signature
|
||||
help, document highlights, workspace symbol search, #include document
|
||||
links and folding ranges over stdio; definition, references and rename
|
||||
work across every open document and the indexed workspace files beyond
|
||||
them, and the quick fixes add a missing textflag.h include and set the
|
||||
argument area from the // func signature.
|
||||
- **Comparators and audits.** `gasm diff` compares the machine code of two
|
||||
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
|
||||
`gasm audit-instructions` diffs the encoder against the installed toolchain,
|
||||
and `gasm scaffold` generates a differential test skeleton for a kernel.
|
||||
|
||||
### Architecture support
|
||||
|
||||
Four architectures, the four that matter in practice:
|
||||
|
||||
| Architecture | GOARCH | File suffix | Instructions recognised |
|
||||
|--------------|-------------|--------------|---------------------------------------------|
|
||||
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
|
||||
| ARM64 | `arm64` | `_arm64.s` | 645 + common opcodes |
|
||||
| RISC-V | `riscv64` | `_riscv64.s` | 992 + common opcodes |
|
||||
| LoongArch | `loong64` | `_loong64.s` | 808 + common opcodes |
|
||||
|
||||
"Common opcodes" are the instructions shared by every architecture (`RET`,
|
||||
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, …). AMD64 additionally
|
||||
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
|
||||
carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
|
||||
…) that the assembler accepts as aliases. Regenerating the tables is one
|
||||
command — `just gen` — and requires only a Go installation; the committed
|
||||
output has no runtime dependency on the toolchain.
|
||||
...) that the assembler accepts as aliases. The tables are generated from
|
||||
the Go toolchain's own assembler source (`just gen` refreshes them), so
|
||||
every mnemonic the real assembler accepts is recognised; what the encoder
|
||||
can emit today is narrower, and a recognised but unencodable instruction is
|
||||
reported as an explicit error, never as a wrong byte.
|
||||
|
||||
## Supported Platforms
|
||||
The same measurement runs over GOROOT's whole assembly corpus:
|
||||
`gasm audit-instructions --corpus` reports every real-code GOROOT assembly
|
||||
file (the tree without testdata) assembling for every target its build
|
||||
admits: 250 of 250, 100 %. Over the whole tree including testdata the
|
||||
measure is 271 of 322 attemptable (84.2 %); files named for other Go ports
|
||||
are counted but never attempted, and `//go:build` constraints decide which
|
||||
targets attempt a file at all, exactly as the build does. The number moves
|
||||
with every release.
|
||||
|
||||
The toolkit runs on Linux. All four Linux architectures are supported as
|
||||
hosts — amd64, arm64, riscv64 and loong64 — and the release matrix
|
||||
cross-compiles the same four targets.
|
||||
### Validation status
|
||||
|
||||
**FreeBSD support is planned for a future release.**
|
||||
**Only amd64 is validated on real hardware.** The other three
|
||||
architectures are validated under qemu-user emulation, because the
|
||||
project owns no arm64, riscv64 or loong64 machine, and emulation is the
|
||||
only substitute available for the hardware. The distinction matters and
|
||||
is stated rather than implied: everything below is a claim about what has
|
||||
actually been executed.
|
||||
|
||||
## Principles
|
||||
| Layer | amd64 | arm64, riscv64, loong64 |
|
||||
|---|---|---|
|
||||
| Encoding: byte-for-byte against `go tool asm` | native hardware | native hardware (the toolchain cross-assembles any GOARCH on any host) |
|
||||
| Execution: JIT calls, ABI checks, differential fuzzing | native hardware | qemu-user emulation |
|
||||
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more. FreeBSD (amd64, arm64, riscv64) is in the same position: the port compiles behind the cross-build gate and its integration test is ready, but no FreeBSD machine has executed it |
|
||||
|
||||
- **Pure Go and GAsm only.** No C, no cgo, no external toolchains, no native
|
||||
binaries, no JavaScript runtimes. The parser is hand-written; there is no
|
||||
parser generator.
|
||||
- **Self-contained.** The toolkit's production code depends only on the
|
||||
standard library; one binary, no runtime data files. The single module
|
||||
dependency, `golang.org/x/arch`, is used **only in tests** to validate the
|
||||
instruction encoder by round-trip decoding — it is never linked into the
|
||||
`gasm` binary.
|
||||
- **Linux-only.** Runs natively on amd64, arm64, riscv64 and loong64 Linux
|
||||
hosts; the release matrix cross-compiles the same four targets. Latest
|
||||
stable Go only.
|
||||
- **No vendor lock-in.** The integration surface is the Language Server
|
||||
Protocol and a command-line interface — both open standards. No cloud
|
||||
service, no proprietary API, no dependence on any one editor's internals.
|
||||
- **Complete and verifiable.** Instruction coverage is generated from the
|
||||
assembler's own source and regenerated on demand, so it cannot silently fall
|
||||
behind the toolchain.
|
||||
Consequences, stated plainly. An emulator is a model of a CPU, not the
|
||||
CPU: instruction semantics are implemented in software and can differ
|
||||
from silicon in ways a test suite does not reveal. A kernel that passes
|
||||
under qemu-user is therefore not proven correct on real hardware, and a
|
||||
discrepancy found on real hardware is a defect in gasm, reported like any
|
||||
other. Encoding parity is the exception: the byte comparison against the
|
||||
toolchain runs on the host for every architecture, so no emulator stands
|
||||
between the claim and the evidence. The debugger is the weakest case: on
|
||||
the three emulated architectures its per-architecture ptrace code has
|
||||
been compiled and read, never executed. Its architecture-neutral units
|
||||
run under `go test ./...`, which the race workflow and a manual run
|
||||
perform; the default `just test` gate does not sweep `./debug/...`.
|
||||
|
||||
## Components
|
||||
## The documentation goal
|
||||
|
||||
| Package | Purpose |
|
||||
|---------|---------|
|
||||
| `token` | Lexical token kinds and source positions. |
|
||||
| `lexer` | Hand-written scanner for Plan 9 assembly. |
|
||||
| `ast` | The abstract syntax tree. |
|
||||
| `parser` | Line-oriented, error-tolerant parser producing the AST. |
|
||||
| `arch` | amd64, arm64, riscv64 and loong64 register files and instruction tables. |
|
||||
| `lint` | Conservative static checks. |
|
||||
| `format` | A canonical formatter — `gofmt` for assembly. |
|
||||
| `asm` | The standalone assembler: amd64, RISC-V and LoongArch encoders, linker, object-file emitters (ELF, GOOBJ). |
|
||||
| `verify` | JIT execution substrate for dynamic analysis, combined ABI+fuzz differential testing. |
|
||||
| `debug` | Interactive ptrace debugger with GPR/YMM register display and named buffer allocation. |
|
||||
| `lsp` | Language Server Protocol server. |
|
||||
| `cmd/gasm` | The `gasm` binary tying it all together. |
|
||||
| `_gen` | The generator that rebuilds the instruction tables from the Go toolchain. |
|
||||
The toolkit is the primary goal. The secondary one is documentation: a
|
||||
specification of the Plan 9 assembly language and of the GOOBJ object
|
||||
format that is 100 % complete, detailed enough to implement against,
|
||||
and written to a professional standard. These are the two subjects this
|
||||
project works with every day, and they are the two for which no usable
|
||||
documentation exists.
|
||||
|
||||
See [`docs/ARCHITECTURE.md`](docs/ARCHITECTURE.md) for the design rationale and
|
||||
data flow, and [`docs/DECISIONS.md`](docs/DECISIONS.md) for design decisions
|
||||
deliberately postponed (with the analysis needed to pick them up again).
|
||||
Go documents the language on a single page, "A Quick Guide to Go's
|
||||
Assembler", which carries no section for loong64, one of the four
|
||||
architectures gasm supports, and covers a fraction of what each
|
||||
assembler accepts. What exists beyond it lives as comments inside the
|
||||
toolchain's internal source: per-architecture reference manuals for
|
||||
arm64, ppc64, riscv64 and loong64, written for the toolchain's own
|
||||
maintainers rather than for an outside reader, and none at all for
|
||||
amd64. GOOBJ fares worst of all. The format that `go build` consumes
|
||||
has no specification anywhere: it is described by a comment in an
|
||||
internal package, it is not a stable interface, and it can change with
|
||||
any toolchain release.
|
||||
|
||||
The gap is therefore filled the only way it can be filled: by reverse
|
||||
engineering the toolchain itself, the same work the encoders already
|
||||
perform. Most of the documentation can come from nowhere else, and it
|
||||
is written as that knowledge is produced during development. It is
|
||||
verified the way the code is verified: an encoding documented here is
|
||||
one that differential tests against `go tool asm` confirm
|
||||
byte-for-byte, and a format field documented here is one the linker
|
||||
demonstrably reads. The work has begun: [docs/GOOBJ.md](docs/GOOBJ.md)
|
||||
specifies the object file format completely, and
|
||||
[docs/asm/README.md](docs/asm/README.md) opens the language reference
|
||||
with its common core. The per-architecture pages follow.
|
||||
|
||||
## Direction
|
||||
|
||||
The plan, in the order it is being worked:
|
||||
|
||||
- **Extended instruction support.** Two layers. First, encoding
|
||||
coverage for every mnemonic the Go toolchain itself accepts, closed in
|
||||
order of how often real code needs each instruction;
|
||||
`gasm audit-instructions` measures the gap. Second, the larger work:
|
||||
an extended instruction set the toolchain does not know at all. The
|
||||
toolchain-derived tables stay generated and untouched; only the
|
||||
extended instructions are hand-maintained, with their own spellings
|
||||
and encoders, verified by execution (on real hardware for amd64, under
|
||||
emulation for the rest, per the validation status above) because the
|
||||
toolchain offers no ground truth to compare against. The gaps exist
|
||||
on every architecture, amd64 included.
|
||||
- **Full GOOBJ and ELF compilation.** The destination is a complete,
|
||||
standalone compilation path: linkable ELF objects for consumers outside
|
||||
Go, and GOOBJ objects that `go build` links directly. Through GOOBJ, a
|
||||
Go program will be able to use machine instructions that the Go
|
||||
toolchain itself does not support; through ELF, Plan 9 assembly becomes
|
||||
usable outside Go entirely.
|
||||
- **Platforms: Linux and FreeBSD.** Linux is supported today on all four
|
||||
architectures and is where the binary builds. FreeBSD follows on amd64,
|
||||
arm64 and riscv64: the JIT's executable-memory mapping and the ptrace
|
||||
debugger layer are ported (the debugger's live validation awaits a
|
||||
FreeBSD machine, as the validation status states). Other unix systems
|
||||
may follow those two.
|
||||
- **Four architectures, no more.** amd64, arm64, riscv64 and loong64.
|
||||
No others are planned.
|
||||
|
||||
## Install
|
||||
|
||||
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
|
||||
linux/loong64 are on the
|
||||
[releases page](https://sourcedock.dev/petrbalvin/gasm-sdk/releases).
|
||||
From source (Go 1.27.1):
|
||||
|
||||
```sh
|
||||
go install sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm@latest
|
||||
```
|
||||
|
||||
Or from a repository checkout:
|
||||
|
||||
```sh
|
||||
just install
|
||||
```
|
||||
|
||||
The installed binary reports the version the toolchain recorded: the tag
|
||||
on a tagged checkout, a pseudo-version naming the commit below one.
|
||||
|
||||
## Quick start
|
||||
|
||||
```sh
|
||||
just install # download dependencies (there are none)
|
||||
just build # go vet + gofmt check — zero errors, zero warnings
|
||||
just test # full suite, race detector, 80 % coverage gate
|
||||
just fmt # gofmt the tree
|
||||
just gen # regenerate the instruction tables from the Go toolchain
|
||||
cat > hello_amd64.s <<'EOF'
|
||||
#include "textflag.h"
|
||||
|
||||
// func add(a, b int) int
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVQ a+0(FP), AX
|
||||
ADDQ b+8(FP), AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
EOF
|
||||
|
||||
gasm lint hello_amd64.s # static checks
|
||||
gasm asm -o hello.bin hello_amd64.s # assemble to a raw image
|
||||
gasm verify --call add --args a=2,b=3 hello_amd64.s # JIT-call it with arguments
|
||||
```
|
||||
|
||||
Install the binary and use it:
|
||||
## Usage
|
||||
|
||||
```sh
|
||||
just install-bin # installs gasm into $GOBIN
|
||||
|
||||
gasm --help # overview of commands and flags
|
||||
gasm tokens kernel_amd64.s # dump the token stream
|
||||
gasm parse kernel_amd64.s # parse, report syntax errors
|
||||
gasm fmt -w kernel_amd64.s # canonicalise in place
|
||||
gasm fmt # reformat every .s below here, like go fmt
|
||||
gasm lint *.s # static checks
|
||||
gasm asm --format elf -o k.o k.s # assemble to a linkable ELF object
|
||||
gasm verify kernel_amd64.s # JIT-load and report functions
|
||||
gasm verify --ground-truth k.s # byte-for-byte vs go tool asm
|
||||
gasm verify --call decodeBlockAVX2 --buf src:64:hex...,dst:256:zero k.s
|
||||
gasm debug --func name k.s # interactive debugger
|
||||
gasm diff a.s b.s # compare machine code byte-for-byte
|
||||
gasm fmt # reformat every .s below here, like go fmt
|
||||
gasm fmt -w kernel_amd64.s # canonicalise one file in place
|
||||
gasm fmt -l *.s # list files whose formatting differs
|
||||
gasm fmt -d kernel_amd64.s # print a unified diff instead
|
||||
gasm lint *.s # static checks
|
||||
gasm asm --format elf -o k.o k.s # assemble to a linkable ELF object
|
||||
gasm asm --format goobj -p pkg/path -o k.o k.s # Go object, consumed by go build
|
||||
gasm dis k.s # assemble, then list each function
|
||||
gasm dis -a amd64 - < dump.bin # disassemble raw bytes from stdin
|
||||
gasm verify --ground-truth k.s # byte-for-byte vs go tool asm
|
||||
gasm verify --fuzz k.s # differential fuzz vs the go tool asm build
|
||||
gasm debug --func name k.s # interactive debugger
|
||||
gasm debug --func name --script cmds.txt --timeout 30s k.s # headless run
|
||||
gasm debug --func name --cover k.s # instruction and label coverage
|
||||
gasm diff a.s b.s # compare machine code byte-for-byte
|
||||
gasm diff --map wideCopyAVX2=wideCopyAVX512 avx2.s avx512.s
|
||||
gasm profile k.s # show basic-block structure
|
||||
gasm profile k.s # show basic-block structure
|
||||
gasm audit-instructions # encoder vs go tool asm name diff
|
||||
gasm scaffold differential k.s # generate a differential test skeleton
|
||||
```
|
||||
|
||||
See [CONTRIBUTING.md](CONTRIBUTING.md) for the full development workflow,
|
||||
[docs/CLI.md](docs/CLI.md) for the command reference, and
|
||||
[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for setup and recipes.
|
||||
Run `gasm --help` for the command overview and `gasm <command> -h` for a
|
||||
command's flags. [docs/CLI.md](docs/CLI.md) is the full reference.
|
||||
|
||||
## Editor integration
|
||||
### Editor integration
|
||||
|
||||
`gasm lsp` speaks the Language Server Protocol over standard input/output, so
|
||||
any LSP-capable editor can use it — point your editor's LSP client at the
|
||||
any LSP-capable editor can use it: point your editor's LSP client at the
|
||||
binary and associate it with `.s` files. Syntax highlighting is delivered as
|
||||
**LSP semantic tokens**, so no editor-specific grammar is required. The server
|
||||
LSP semantic tokens, so no editor-specific grammar is required. The server
|
||||
infers the target architecture from the file-name suffix
|
||||
(`_amd64.s` / `_arm64.s` / `_riscv64.s` / `_loong64.s`).
|
||||
|
||||
## License
|
||||
## Development
|
||||
|
||||
```sh
|
||||
just build # compile, zero errors and zero warnings
|
||||
just test # the suite, no cache, the 80 % coverage floor
|
||||
just gates # build, fmt-check, vet, test, race: the definition of done
|
||||
just fmt # gofmt the tree
|
||||
just gen # regenerate the instruction tables from the Go toolchain
|
||||
```
|
||||
|
||||
See [CONTRIBUTING.md](CONTRIBUTING.md) for the development workflow and
|
||||
[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for setup details and every
|
||||
recipe.
|
||||
|
||||
## Documentation
|
||||
|
||||
- [docs/CLI.md](docs/CLI.md): full command reference
|
||||
- man pages: `just install-man` installs gasm(1) and one page per command
|
||||
except `version`, which is documented inside gasm(1) instead, into
|
||||
~/.local/share/man (MANDIR overrides); `just uninstall-man` removes
|
||||
them
|
||||
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
|
||||
- [docs/GOOBJ.md](docs/GOOBJ.md): the GOOBJ object file format specification
|
||||
- [docs/asm/](docs/asm/README.md): the Plan 9 assembly language reference
|
||||
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
|
||||
- [CHANGELOG.md](CHANGELOG.md): release history
|
||||
|
||||
## Licence
|
||||
|
||||
BSD-3-Clause; see [LICENSE](LICENSE).
|
||||
|
||||
BSD-3-Clause — see [LICENSE](LICENSE).
|
||||
Copyright © 2026 [Petr Balvín](https://petrbalvin.org)
|
||||
|
||||
+41
@@ -0,0 +1,41 @@
|
||||
# Security policy
|
||||
|
||||
## Supported versions
|
||||
|
||||
Security fixes go to the newest release and to the `development` branch. Older
|
||||
releases do not receive them.
|
||||
|
||||
| Version | Supported |
|
||||
|---|---|
|
||||
| 0.35.0 | yes |
|
||||
| older releases | no |
|
||||
|
||||
## Reporting a vulnerability
|
||||
|
||||
**Do not open a public issue for a security problem.** A public report tells everyone
|
||||
about the flaw before there is a fix. Report it privately to
|
||||
**opensource@petrbalvin.org**.
|
||||
|
||||
Include:
|
||||
|
||||
- the version or commit you tested, and the platform
|
||||
- what the problem is, and what an attacker gains from it
|
||||
- the smallest reproducer you have, ideally a test or a single command
|
||||
- a suggested fix, if you have one
|
||||
|
||||
## What to expect
|
||||
|
||||
- A human reads the report, and you get an acknowledgement.
|
||||
- You are kept informed while the fix is being made, and told when it ships.
|
||||
- The fix is released before the details are published, and the timing is agreed with
|
||||
you.
|
||||
- The fix ships without naming you: the project keeps no credits list, so the release
|
||||
notes, the changelog and the commits name no reporter.
|
||||
|
||||
## Out of scope
|
||||
|
||||
- Findings that require the attacker to already run code as the user, or to have local
|
||||
access.
|
||||
- Missing hardening with no demonstrated impact.
|
||||
- Flaws in a third-party dependency: report them to that project, and to this one only
|
||||
when this project's use of it makes them reachable.
|
||||
+105
-7
@@ -5,9 +5,13 @@
|
||||
// toolchain's own assembler source. Go's Plan 9 assembler defines the exact,
|
||||
// complete set of mnemonics it accepts for each architecture in
|
||||
// $GOROOT/src/cmd/internal/obj/<arch>/anames.go; this tool extracts those
|
||||
// names so gasm-devkit supports every instruction the real assembler does,
|
||||
// names so gasm-sdk supports every instruction the real assembler does,
|
||||
// with no hand-maintained (and therefore inevitably incomplete) lists.
|
||||
//
|
||||
// The same data feeds the generated instruction appendices of the assembly
|
||||
// language reference, docs/asm/INSTRUCTIONS-<ARCH>.md, so that the reference
|
||||
// cannot drift from the tables it documents.
|
||||
//
|
||||
// Usage (via the justfile):
|
||||
//
|
||||
// just gen
|
||||
@@ -26,9 +30,12 @@ import (
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||
)
|
||||
|
||||
// archDirs maps a gasm-devkit architecture name to its obj sub-directory.
|
||||
// archDirs maps a gasm-sdk architecture name to its obj sub-directory.
|
||||
var archDirs = []struct {
|
||||
arch string
|
||||
sub string
|
||||
@@ -39,11 +46,30 @@ var archDirs = []struct {
|
||||
{"loong64", "loong64"},
|
||||
}
|
||||
|
||||
// docPages maps an architecture to its generated appendix in the language
|
||||
// reference. The amd64 page carries a per-mnemonic encodability column,
|
||||
// decided by asm.Encodable, which mirrors the encoder's own dispatch; the
|
||||
// other targets have no single cheap predicate, so their pages carry the
|
||||
// inventory and point at the live measurement instead.
|
||||
var docPages = []struct {
|
||||
arch arch.Arch
|
||||
title string
|
||||
file string
|
||||
anames string
|
||||
encodable bool
|
||||
}{
|
||||
{arch.AMD64, "AMD64", "INSTRUCTIONS-AMD64.md", "cmd/internal/obj/x86/anames.go", true},
|
||||
{arch.ARM64, "ARM64", "INSTRUCTIONS-ARM64.md", "cmd/internal/obj/arm64/anames.go", false},
|
||||
{arch.RISCV, "RISC-V 64", "INSTRUCTIONS-RISCV64.md", "cmd/internal/obj/riscv/anames.go", false},
|
||||
{arch.LOONG64, "LoongArch 64", "INSTRUCTIONS-LOONG64.md", "cmd/internal/obj/loong64/anames.go", false},
|
||||
}
|
||||
|
||||
func main() {
|
||||
goroot := strings.TrimSpace(runGoEnvGOROOT())
|
||||
if goroot == "" {
|
||||
fatal("could not determine GOROOT")
|
||||
}
|
||||
version := strings.TrimSpace(runGoEnv("GOVERSION"))
|
||||
// The common opcodes shared by every architecture (RET, JMP, NOP, CALL,
|
||||
// TEXT, FUNCDATA, …) live in cmd/internal/obj/util.go.
|
||||
commonPath := filepath.Join(goroot, "src", "cmd", "internal", "obj", "util.go")
|
||||
@@ -57,16 +83,24 @@ func main() {
|
||||
}
|
||||
fmt.Printf("%-8s %4d instructions -> arch/common_gen.go\n", "common", len(common))
|
||||
|
||||
names := map[string][]string{}
|
||||
for _, a := range archDirs {
|
||||
path := filepath.Join(goroot, "src", "cmd", "internal", "obj", a.sub, "anames.go")
|
||||
names, err := extractInstrs(path)
|
||||
names[a.arch], err = extractInstrs(path)
|
||||
if err != nil {
|
||||
fatal("extract %s: %v", a.arch, err)
|
||||
}
|
||||
if err := writeGen(a.arch, a.sub, names); err != nil {
|
||||
if err := writeGen(a.arch, a.sub, names[a.arch]); err != nil {
|
||||
fatal("write %s: %v", a.arch, err)
|
||||
}
|
||||
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names), a.arch)
|
||||
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names[a.arch]), a.arch)
|
||||
}
|
||||
|
||||
for _, p := range docPages {
|
||||
if err := writeDocPage(p.arch, p.title, p.file, p.anames, version, p.encodable); err != nil {
|
||||
fatal("write %s: %v", p.file, err)
|
||||
}
|
||||
fmt.Printf("%-8s -> docs/asm/%s\n", p.arch, p.file)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -85,7 +119,7 @@ func filterCommon(names []string) []string {
|
||||
// writeCommon emits arch/common_gen.go.
|
||||
func writeCommon(names []string) error {
|
||||
var b strings.Builder
|
||||
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
||||
b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
|
||||
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
|
||||
b.WriteString("//\n")
|
||||
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
||||
@@ -156,7 +190,7 @@ func stringLit(elt ast.Expr) string {
|
||||
// writeGen emits arch/<arch>_gen.go.
|
||||
func writeGen(arch, sub string, names []string) error {
|
||||
var b strings.Builder
|
||||
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
||||
b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
|
||||
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
|
||||
b.WriteString("//\n")
|
||||
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
||||
@@ -172,6 +206,61 @@ func writeGen(arch, sub string, names []string) error {
|
||||
return os.WriteFile(filepath.Join("arch", arch+"_gen.go"), []byte(b.String()), 0o644)
|
||||
}
|
||||
|
||||
// writeDocPage emits docs/asm/<file>, the generated instruction appendix of
|
||||
// the language reference for one architecture: every mnemonic the toolchain
|
||||
// accepts, with the curated summary where the architecture table carries one
|
||||
// and, on amd64, a per-mnemonic encodability column.
|
||||
func writeDocPage(a arch.Arch, title, file, anames, version string, encodable bool) error {
|
||||
table := arch.ForArch(a)
|
||||
instrs := table.Instructions()
|
||||
|
||||
var b strings.Builder
|
||||
b.WriteString("# " + title + ": instruction inventory\n\n")
|
||||
b.WriteString("Generated by gasm-sdk's `_gen` from the Go toolchain's instruction table\n")
|
||||
b.WriteString("(`" + anames + "`, " + version + "); DO NOT EDIT. This page lists every mnemonic\n")
|
||||
b.WriteString("`go tool asm` accepts on this target, which is the upper bound of the\n")
|
||||
b.WriteString("language on it: a name absent here is not an instruction of the target,\n")
|
||||
b.WriteString("and a name present here may still be one gasm's encoder cannot emit yet.\n\n")
|
||||
|
||||
encodableCount := 0
|
||||
if encodable {
|
||||
b.WriteString("The `gasm encodes` column reports whether gasm's encoder can emit the\n")
|
||||
b.WriteString("mnemonic today; the gap is the encoder backlog, measured live by\n")
|
||||
b.WriteString("`gasm audit-instructions`.\n\n")
|
||||
b.WriteString("| Mnemonic | gasm encodes | Notes |\n")
|
||||
b.WriteString("|---|---|---|\n")
|
||||
for _, in := range instrs {
|
||||
ok := asm.Encodable(in.Name)
|
||||
if ok {
|
||||
encodableCount++
|
||||
}
|
||||
b.WriteString("| `" + in.Name + "` | " + yesNo(ok) + " | " + in.Summary + " |\n")
|
||||
}
|
||||
b.WriteString("\n")
|
||||
fmt.Fprintf(&b, "Recognised: %d mnemonics. gasm encodes: %d.\n", len(instrs), encodableCount)
|
||||
} else {
|
||||
b.WriteString("The inventory carries no per-mnemonic encoder column: on this target\n")
|
||||
b.WriteString("encodability is decided per operand shape, and the live measured\n")
|
||||
b.WriteString("coverage is reported by `gasm audit-instructions`.\n\n")
|
||||
b.WriteString("| Mnemonic | Notes |\n")
|
||||
b.WriteString("|---|---|\n")
|
||||
for _, in := range instrs {
|
||||
b.WriteString("| `" + in.Name + "` | " + in.Summary + " |\n")
|
||||
}
|
||||
b.WriteString("\n")
|
||||
fmt.Fprintf(&b, "Recognised: %d mnemonics.\n", len(instrs))
|
||||
}
|
||||
return os.WriteFile(filepath.Join("docs", "asm", file), []byte(b.String()), 0o644)
|
||||
}
|
||||
|
||||
// yesNo renders a boolean as the word the appendix tables use.
|
||||
func yesNo(v bool) string {
|
||||
if v {
|
||||
return "yes"
|
||||
}
|
||||
return "no"
|
||||
}
|
||||
|
||||
func runGoEnvGOROOT() string {
|
||||
out, err := exec.Command("go", "env", "GOROOT").Output()
|
||||
if err != nil {
|
||||
@@ -180,6 +269,15 @@ func runGoEnvGOROOT() string {
|
||||
return string(out)
|
||||
}
|
||||
|
||||
// runGoEnv runs `go env` for a single variable.
|
||||
func runGoEnv(name string) string {
|
||||
out, err := exec.Command("go", "env", name).Output()
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
return string(out)
|
||||
}
|
||||
|
||||
func fatal(format string, args ...any) {
|
||||
fmt.Fprintf(os.Stderr, "gen: "+format+"\n", args...)
|
||||
os.Exit(1)
|
||||
|
||||
@@ -70,6 +70,10 @@ func amd64Registers() []Register {
|
||||
for i := 0; i <= 7; i++ {
|
||||
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
|
||||
}
|
||||
// x87 stack registers (FMOVD and the other x87 moves).
|
||||
for i := 0; i <= 7; i++ {
|
||||
add(fmt.Sprintf("F%d", i), Float, "x87 stack register")
|
||||
}
|
||||
return regs
|
||||
}
|
||||
|
||||
@@ -270,6 +274,8 @@ func amd64Curated() []Instr {
|
||||
"VMINPD", "VMINPS", "VMINSD", "VMINSS", "VMAXPD", "VMAXPS", "VMAXSD", "VMAXSS",
|
||||
"VXORPD", "VXORPS", "VANDPD", "VANDPS", "VANDNPD", "VANDNPS", "VORPD", "VORPS",
|
||||
"VUNPCKHPD", "VUNPCKLPD", "VUNPCKHPS", "VUNPCKLPS",
|
||||
"PSHUFD", "PSHUFHW", "PSHUFLW", "SHUFPS", "SHUFPD",
|
||||
"UNPCKLPS", "UNPCKHPS", "UNPCKLPD", "UNPCKHPD",
|
||||
"VSQRTPD", "VSQRTPS", "VSQRTSD", "VSQRTSS", "VRSQRTPS", "VRCPPS",
|
||||
"VCMPPD", "VCMPPS", "VCMPSD", "VCMPSS",
|
||||
} {
|
||||
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
|
||||
//
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
|
||||
@@ -55,6 +55,7 @@ const (
|
||||
Mask // AVX-512 mask register (K)
|
||||
Float // arm64 floating-point register (F)
|
||||
VecARM // arm64 SIMD/vector register (V)
|
||||
VecSIMD // architecture-neutral SIMD/vector register (LoongArch LSX/LASX)
|
||||
Special // architecture-special register
|
||||
)
|
||||
|
||||
@@ -73,6 +74,8 @@ func (c RegClass) String() string {
|
||||
return "float"
|
||||
case VecARM:
|
||||
return "vector (arm64)"
|
||||
case VecSIMD:
|
||||
return "vector"
|
||||
case Special:
|
||||
return "special"
|
||||
default:
|
||||
|
||||
+30
-2
@@ -29,10 +29,11 @@ func arm64Registers() []Register {
|
||||
regs = append(regs, Register{Name: name, Class: class, Desc: desc})
|
||||
}
|
||||
|
||||
// General-purpose integer registers R0–R30.
|
||||
// General-purpose integer registers R0-R30.
|
||||
for i := 0; i <= 30; i++ {
|
||||
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
|
||||
}
|
||||
add("R18_PLATFORM", GPR, "R18 under its toolchain-reserved Windows name (an alias of R18)")
|
||||
add("ZR", Special, "zero register (reads as 0)")
|
||||
add("SP", Special, "stack pointer")
|
||||
add("LR", Special, "link register (alias of R30)")
|
||||
@@ -145,11 +146,38 @@ func arm64Curated() []Instr {
|
||||
for _, op := range []string{
|
||||
"LDAXR", "LDAXRB", "LDAXRH", "LDAXRW", "STXR", "STXRB", "STXRH", "STXRW",
|
||||
"LDAR", "LDARB", "LDARH", "LDARW", "STLR", "STLRB", "STLRH", "STLRW",
|
||||
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL", "CASAL",
|
||||
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL",
|
||||
} {
|
||||
t = append(t, i(op, "Atomic memory operation"))
|
||||
}
|
||||
|
||||
// Register-pair loads and stores.
|
||||
for _, op := range []string{"LDP", "STP", "LDPW", "STPW", "FLDPD", "FSTPD"} {
|
||||
t = append(t, ic(op, "Register-pair load or store", 2, 2))
|
||||
}
|
||||
|
||||
// Cache maintenance and prefetch.
|
||||
t = append(t, i("DC", "Data cache maintenance"))
|
||||
t = append(t, i("PRFM", "Memory prefetch"))
|
||||
for _, op := range []string{"LDADDAL", "LDCLRAL", "LDORAL", "SWPAL"} {
|
||||
t = append(t, i(op, "Atomic memory operation with acquire and release semantics"))
|
||||
}
|
||||
|
||||
// Cryptographic extensions.
|
||||
for _, op := range []string{"AESE", "AESD", "AESMC", "AESIMC"} {
|
||||
t = append(t, i(op, "AES round"))
|
||||
}
|
||||
for _, op := range []string{
|
||||
"SHA1C", "SHA1P", "SHA1M", "SHA1H", "SHA1SU0", "SHA1SU1",
|
||||
"SHA256H", "SHA256H2", "SHA256SU0", "SHA256SU1",
|
||||
"SHA512H", "SHA512H2", "SHA512SU0", "SHA512SU1",
|
||||
} {
|
||||
t = append(t, i(op, "SHA round"))
|
||||
}
|
||||
for _, op := range []string{"VEOR3", "VBCAX", "VXAR", "VRAX1"} {
|
||||
t = append(t, i(op, "Three-way XOR / rotate crypto vector operation"))
|
||||
}
|
||||
|
||||
// Floating-point scalar.
|
||||
for _, op := range []string{
|
||||
"FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX",
|
||||
|
||||
+108
-1
@@ -1,4 +1,4 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
|
||||
//
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
@@ -364,6 +364,8 @@ var arm64GeneratedInstrs = []string{
|
||||
"REVW",
|
||||
"ROR",
|
||||
"RORW",
|
||||
"RPRFM",
|
||||
"SB",
|
||||
"SBC",
|
||||
"SBCS",
|
||||
"SBCSW",
|
||||
@@ -477,23 +479,68 @@ var arm64GeneratedInstrs = []string{
|
||||
"UXTH",
|
||||
"UXTHW",
|
||||
"UXTW",
|
||||
"VABS",
|
||||
"VADD",
|
||||
"VADDP",
|
||||
"VADDV",
|
||||
"VAND",
|
||||
"VBCAX",
|
||||
"VBIC",
|
||||
"VBIF",
|
||||
"VBIT",
|
||||
"VBSL",
|
||||
"VCLS",
|
||||
"VCLZ",
|
||||
"VCMEQ",
|
||||
"VCMGE",
|
||||
"VCMGT",
|
||||
"VCMHI",
|
||||
"VCMHS",
|
||||
"VCMLE",
|
||||
"VCMLT",
|
||||
"VCMTST",
|
||||
"VCNT",
|
||||
"VDUP",
|
||||
"VEOR",
|
||||
"VEOR3",
|
||||
"VEXT",
|
||||
"VFABS",
|
||||
"VFADD",
|
||||
"VFADDP",
|
||||
"VFCMEQ",
|
||||
"VFCMGE",
|
||||
"VFCMGT",
|
||||
"VFCMLE",
|
||||
"VFCMLT",
|
||||
"VFCVTL",
|
||||
"VFCVTL2",
|
||||
"VFCVTN",
|
||||
"VFCVTN2",
|
||||
"VFCVTZS",
|
||||
"VFCVTZU",
|
||||
"VFDIV",
|
||||
"VFMAX",
|
||||
"VFMAXNM",
|
||||
"VFMAXNMP",
|
||||
"VFMAXNMV",
|
||||
"VFMAXP",
|
||||
"VFMAXV",
|
||||
"VFMIN",
|
||||
"VFMINNM",
|
||||
"VFMINNMP",
|
||||
"VFMINNMV",
|
||||
"VFMINP",
|
||||
"VFMINV",
|
||||
"VFMLA",
|
||||
"VFMLS",
|
||||
"VFMUL",
|
||||
"VFNEG",
|
||||
"VFRINTM",
|
||||
"VFRINTN",
|
||||
"VFRINTP",
|
||||
"VFRINTZ",
|
||||
"VFSQRT",
|
||||
"VFSUB",
|
||||
"VLD1",
|
||||
"VLD1R",
|
||||
"VLD2",
|
||||
@@ -502,11 +549,17 @@ var arm64GeneratedInstrs = []string{
|
||||
"VLD3R",
|
||||
"VLD4",
|
||||
"VLD4R",
|
||||
"VMLA",
|
||||
"VMLS",
|
||||
"VMOV",
|
||||
"VMOVD",
|
||||
"VMOVI",
|
||||
"VMOVQ",
|
||||
"VMOVS",
|
||||
"VMUL",
|
||||
"VNEG",
|
||||
"VNOT",
|
||||
"VORN",
|
||||
"VORR",
|
||||
"VPMULL",
|
||||
"VPMULL2",
|
||||
@@ -515,14 +568,47 @@ var arm64GeneratedInstrs = []string{
|
||||
"VREV16",
|
||||
"VREV32",
|
||||
"VREV64",
|
||||
"VSCVTF",
|
||||
"VSHADD",
|
||||
"VSHL",
|
||||
"VSHRN",
|
||||
"VSHRN2",
|
||||
"VSLI",
|
||||
"VSMAX",
|
||||
"VSMAXP",
|
||||
"VSMAXV",
|
||||
"VSMIN",
|
||||
"VSMINP",
|
||||
"VSMINV",
|
||||
"VSMLAL",
|
||||
"VSMLAL2",
|
||||
"VSMLSL",
|
||||
"VSMLSL2",
|
||||
"VSMULL",
|
||||
"VSMULL2",
|
||||
"VSQABS",
|
||||
"VSQADD",
|
||||
"VSQNEG",
|
||||
"VSQSHL",
|
||||
"VSQSUB",
|
||||
"VSQXTN",
|
||||
"VSQXTN2",
|
||||
"VSQXTUN",
|
||||
"VSQXTUN2",
|
||||
"VSRHADD",
|
||||
"VSRI",
|
||||
"VSRSHR",
|
||||
"VSSHL",
|
||||
"VSSHLL",
|
||||
"VSSHLL2",
|
||||
"VSSHR",
|
||||
"VST1",
|
||||
"VST2",
|
||||
"VST3",
|
||||
"VST4",
|
||||
"VSUB",
|
||||
"VSXTL",
|
||||
"VSXTL2",
|
||||
"VTBL",
|
||||
"VTBX",
|
||||
"VTRN1",
|
||||
@@ -530,8 +616,27 @@ var arm64GeneratedInstrs = []string{
|
||||
"VUADDLV",
|
||||
"VUADDW",
|
||||
"VUADDW2",
|
||||
"VUCVTF",
|
||||
"VUHADD",
|
||||
"VUMAX",
|
||||
"VUMAXP",
|
||||
"VUMAXV",
|
||||
"VUMIN",
|
||||
"VUMINP",
|
||||
"VUMINV",
|
||||
"VUMLAL",
|
||||
"VUMLAL2",
|
||||
"VUMLSL",
|
||||
"VUMLSL2",
|
||||
"VUMULL",
|
||||
"VUMULL2",
|
||||
"VUQADD",
|
||||
"VUQSHL",
|
||||
"VUQSUB",
|
||||
"VUQXTN",
|
||||
"VUQXTN2",
|
||||
"VURHADD",
|
||||
"VUSHL",
|
||||
"VUSHLL",
|
||||
"VUSHLL2",
|
||||
"VUSHR",
|
||||
@@ -541,6 +646,8 @@ var arm64GeneratedInstrs = []string{
|
||||
"VUZP1",
|
||||
"VUZP2",
|
||||
"VXAR",
|
||||
"VXTN",
|
||||
"VXTN2",
|
||||
"VZIP1",
|
||||
"VZIP2",
|
||||
"WFE",
|
||||
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/util.go from the Go toolchain.
|
||||
//
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
|
||||
+2
-2
@@ -31,10 +31,10 @@ func loong64Registers() []Register {
|
||||
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
|
||||
}
|
||||
for i := 0; i <= 31; i++ {
|
||||
add(fmt.Sprintf("V%d", i), VecARM, "LSX 128-bit vector register")
|
||||
add(fmt.Sprintf("V%d", i), VecSIMD, "LSX 128-bit vector register")
|
||||
}
|
||||
for i := 0; i <= 31; i++ {
|
||||
add(fmt.Sprintf("X%d", i), VecARM, "LASX 256-bit vector register")
|
||||
add(fmt.Sprintf("X%d", i), VecSIMD, "LASX 256-bit vector register")
|
||||
}
|
||||
return regs
|
||||
}
|
||||
|
||||
+10
-1
@@ -1,4 +1,4 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
|
||||
//
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
@@ -152,6 +152,8 @@ var loong64GeneratedInstrs = []string{
|
||||
"FNMADDF",
|
||||
"FNMSUBD",
|
||||
"FNMSUBF",
|
||||
"FRINTD",
|
||||
"FRINTF",
|
||||
"FSCALEBD",
|
||||
"FSCALEBF",
|
||||
"FSEL",
|
||||
@@ -177,7 +179,10 @@ var loong64GeneratedInstrs = []string{
|
||||
"FTINTWF",
|
||||
"JIRL",
|
||||
"LL",
|
||||
"LLACQV",
|
||||
"LLACQW",
|
||||
"LLV",
|
||||
"LLW",
|
||||
"LU12IW",
|
||||
"LU32ID",
|
||||
"LU52ID",
|
||||
@@ -248,7 +253,11 @@ var loong64GeneratedInstrs = []string{
|
||||
"ROTR",
|
||||
"ROTRV",
|
||||
"SC",
|
||||
"SCQ",
|
||||
"SCRELV",
|
||||
"SCRELW",
|
||||
"SCV",
|
||||
"SCW",
|
||||
"SGT",
|
||||
"SGTU",
|
||||
"SLL",
|
||||
|
||||
+32
-1
@@ -1,4 +1,4 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
|
||||
//
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
@@ -81,6 +81,9 @@ var riscvGeneratedInstrs = []string{
|
||||
"CLD",
|
||||
"CLDSP",
|
||||
"CLI",
|
||||
"CLMUL",
|
||||
"CLMULH",
|
||||
"CLMULR",
|
||||
"CLUI",
|
||||
"CLW",
|
||||
"CLWSP",
|
||||
@@ -95,13 +98,20 @@ var riscvGeneratedInstrs = []string{
|
||||
"CSDSP",
|
||||
"CSLLI",
|
||||
"CSRAI",
|
||||
"CSRC",
|
||||
"CSRCI",
|
||||
"CSRLI",
|
||||
"CSRR",
|
||||
"CSRRC",
|
||||
"CSRRCI",
|
||||
"CSRRS",
|
||||
"CSRRSI",
|
||||
"CSRRW",
|
||||
"CSRRWI",
|
||||
"CSRS",
|
||||
"CSRSI",
|
||||
"CSRW",
|
||||
"CSRWI",
|
||||
"CSUB",
|
||||
"CSUBW",
|
||||
"CSW",
|
||||
@@ -259,6 +269,7 @@ var riscvGeneratedInstrs = []string{
|
||||
"ORCB",
|
||||
"ORI",
|
||||
"ORN",
|
||||
"PAUSE",
|
||||
"RDCYCLE",
|
||||
"RDINSTRET",
|
||||
"RDTIME",
|
||||
@@ -322,6 +333,8 @@ var riscvGeneratedInstrs = []string{
|
||||
"VADDVI",
|
||||
"VADDVV",
|
||||
"VADDVX",
|
||||
"VANDNVV",
|
||||
"VANDNVX",
|
||||
"VANDVI",
|
||||
"VANDVV",
|
||||
"VANDVX",
|
||||
@@ -329,8 +342,17 @@ var riscvGeneratedInstrs = []string{
|
||||
"VASUBUVX",
|
||||
"VASUBVV",
|
||||
"VASUBVX",
|
||||
"VBREV8V",
|
||||
"VBREVV",
|
||||
"VCLMULHVV",
|
||||
"VCLMULHVX",
|
||||
"VCLMULVV",
|
||||
"VCLMULVX",
|
||||
"VCLZV",
|
||||
"VCOMPRESSVM",
|
||||
"VCPOPM",
|
||||
"VCPOPV",
|
||||
"VCTZV",
|
||||
"VDIVUVV",
|
||||
"VDIVUVX",
|
||||
"VDIVVV",
|
||||
@@ -743,10 +765,16 @@ var riscvGeneratedInstrs = []string{
|
||||
"VREMUVX",
|
||||
"VREMVV",
|
||||
"VREMVX",
|
||||
"VREV8V",
|
||||
"VRGATHEREI16VV",
|
||||
"VRGATHERVI",
|
||||
"VRGATHERVV",
|
||||
"VRGATHERVX",
|
||||
"VROLVV",
|
||||
"VROLVX",
|
||||
"VRORVI",
|
||||
"VRORVV",
|
||||
"VRORVX",
|
||||
"VRSUBVI",
|
||||
"VRSUBVX",
|
||||
"VS1RV",
|
||||
@@ -950,6 +978,9 @@ var riscvGeneratedInstrs = []string{
|
||||
"VWMULVX",
|
||||
"VWREDSUMUVS",
|
||||
"VWREDSUMVS",
|
||||
"VWSLLVI",
|
||||
"VWSLLVV",
|
||||
"VWSLLVX",
|
||||
"VWSUBUVV",
|
||||
"VWSUBUVX",
|
||||
"VWSUBUWV",
|
||||
|
||||
+173
-6
@@ -4,13 +4,14 @@
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// TestGOObjectAARCH64Structure checks the basic structure of the emitted
|
||||
@@ -61,6 +62,62 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectAARCH64PairReloc pins the ADRP-pair relocation shape against
|
||||
// the toolchain's own object for the same source: exactly one R_ADDRARM64
|
||||
// of Siz 8 at the ADRP word (cmd/internal/obj/arm64/asm7.go adds a single
|
||||
// Siz-8 relocation per pair and the linker patches both instructions from
|
||||
// it). gasm's assembler records the ADRP+ADD form as two word relocs; the
|
||||
// emitter must coalesce them, not emit two Siz-4 records.
|
||||
func TestGOObjectAARCH64PairReloc(t *testing.T) {
|
||||
f, errs := parser.Parse("gv_arm64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·getv(SB), NOSPLIT, $0-8
|
||||
MOVD $v<>(SB), R4
|
||||
MOVD R4, ret+0(FP)
|
||||
RET
|
||||
|
||||
GLOBL v<>(SB), RODATA, $8
|
||||
DATA v<>+0(SB)/8, $7
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
obj, err := img.GOObjectAARCH64("main", "gv_arm64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObjectAARCH64: %v", err)
|
||||
}
|
||||
v := openGoobj(t, obj)
|
||||
relocs := v.blk(blkReloc)
|
||||
le := binary.LittleEndian
|
||||
// Two DWARF relocs on the lines/DIE symbols, then the code's one pair
|
||||
// relocation.
|
||||
if len(relocs) != 3*23 {
|
||||
t.Fatalf("relocs = %d bytes, want three entries", len(relocs))
|
||||
}
|
||||
cr := relocs[2*23:]
|
||||
if off := int32(le.Uint32(cr[0:])); off != 0 {
|
||||
t.Errorf("pair reloc off = %d, want 0 (the ADRP word)", off)
|
||||
}
|
||||
if siz := cr[4]; siz != 8 {
|
||||
t.Errorf("pair reloc siz = %d, want 8", siz)
|
||||
}
|
||||
if typ := le.Uint16(cr[5:]); typ != relocArm64Addr {
|
||||
t.Errorf("pair reloc type = %d, want %d (R_ADDRARM64)", typ, relocArm64Addr)
|
||||
}
|
||||
if pkg := le.Uint32(cr[15:]); pkg != pkgIdxSelf {
|
||||
t.Errorf("pair reloc PkgIdx = %#x, want pkgIdxSelf", pkg)
|
||||
}
|
||||
// The GLOBL is the first package definition.
|
||||
if sym := le.Uint32(cr[19:]); sym != 0 {
|
||||
t.Errorf("pair reloc SymIdx = %d, want 0 (the GLOBL definition)", sym)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectAARCH64Link does an end-to-end link test: it cross-compiles a
|
||||
// Go program for arm64, substitutes the gasm-produced object into the package
|
||||
// archive, re-links with cmd/link, and verifies the symbol appears in the
|
||||
@@ -79,6 +136,14 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
||||
ADD R5, R4, R4
|
||||
MOVD R4, ret+16(FP)
|
||||
RET
|
||||
|
||||
TEXT ·getv(SB), NOSPLIT, $0-8
|
||||
MOVD $v<>(SB), R4
|
||||
MOVD R4, ret+0(FP)
|
||||
RET
|
||||
|
||||
GLOBL v<>(SB), RODATA, $8
|
||||
DATA v<>+0(SB)/8, $7
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
@@ -86,11 +151,15 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
||||
mainSrc := `package main
|
||||
|
||||
func add(a, b int64) int64
|
||||
func getv() *int64
|
||||
|
||||
func main() {
|
||||
if add(20, 22) != 42 {
|
||||
panic("bad add")
|
||||
}
|
||||
if getv() == nil {
|
||||
panic("bad getv")
|
||||
}
|
||||
}
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||
@@ -109,16 +178,13 @@ func main() {
|
||||
if err != nil {
|
||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||
}
|
||||
var pkgArch, work, linkLine, asmObj string
|
||||
for _, line := range strings.Split(string(buildLog), "\n") {
|
||||
var work, linkLine, asmObj string
|
||||
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
|
||||
asmObj = fieldAfter(line, "-o")
|
||||
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
|
||||
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
|
||||
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
|
||||
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||
linkLine = line
|
||||
}
|
||||
@@ -179,3 +245,104 @@ func main() {
|
||||
t.Error("binary does not contain expected symbol")
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectAARCH64DataSymbolLink does for symbol-valued DATA fields what
|
||||
// the rt0 files do ("DATA _rt0…lib+0(SB)/8, $_rt0…lib(SB)"): the gasm object
|
||||
// carries an R_ADDR against the file's own TEXT symbol, the toolchain links
|
||||
// it, and the binary is checked for the symbol (no arm64 host to run it).
|
||||
func TestGOObjectAARCH64DataSymbolLink(t *testing.T) {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
t.Skip("no Go toolchain available")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
asmSrc := `#include "textflag.h"
|
||||
GLOBL entry(SB), NOPTR, $8
|
||||
DATA entry+0(SB)/8, $·keepme(SB)
|
||||
|
||||
TEXT ·keepme(SB), NOSPLIT, $0-0
|
||||
RET
|
||||
|
||||
TEXT ·entryptr(SB), NOSPLIT, $0-8
|
||||
MOVD entry+0(SB), R4
|
||||
MOVD R4, ret+0(FP)
|
||||
RET
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
mainSrc := `package main
|
||||
|
||||
func keepme()
|
||||
func entryptr() uintptr
|
||||
|
||||
func main() {
|
||||
if entryptr() == 0 {
|
||||
panic("the entry word is empty")
|
||||
}
|
||||
}
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module a64dlink\n\ngo 1.21\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||
build.Dir = dir
|
||||
build.Env = append(os.Environ(), "GOARCH=arm64")
|
||||
buildLog, err := build.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||
}
|
||||
var work, linkLine, asmObj string
|
||||
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
|
||||
asmObj = fieldAfter(line, "-o")
|
||||
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||
linkLine = line
|
||||
}
|
||||
}
|
||||
if work == "" || asmObj == "" || linkLine == "" {
|
||||
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
|
||||
}
|
||||
defer os.RemoveAll(work)
|
||||
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
|
||||
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||
|
||||
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f, errs := parser.Parse("main_arm64.s", string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
gasmObj, err := img.GOObjectAARCH64("a64dlink", "main_arm64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObjectAARCH64: %v", err)
|
||||
}
|
||||
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
|
||||
t.Fatalf("write gasm object: %v", err)
|
||||
}
|
||||
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
|
||||
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
|
||||
if out, err := linkCmd.CombinedOutput(); err != nil {
|
||||
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
|
||||
}
|
||||
binData, err := os.ReadFile(filepath.Join(dir, "prog"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.Contains(string(binData), "keepme") {
|
||||
t.Error("binary does not contain the keepme symbol")
|
||||
}
|
||||
}
|
||||
|
||||
+3198
-242
File diff suppressed because it is too large
Load Diff
+832
-172
File diff suppressed because it is too large
Load Diff
+1177
-7
File diff suppressed because it is too large
Load Diff
+257
-20
@@ -53,7 +53,7 @@ package asm
|
||||
import (
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||
)
|
||||
|
||||
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
|
||||
@@ -63,6 +63,12 @@ type arm64FrameInfo struct {
|
||||
args int // the declared -argsize
|
||||
noSplit bool // the NOSPLIT flag
|
||||
leaf bool // no call instructions in the body
|
||||
|
||||
// Stack-split guard state: needSplit mirrors the toolchain, which skips
|
||||
// the check for NOSPLIT functions and auto-marks leaf functions with an
|
||||
// autosize below StackSmall as NOSPLIT.
|
||||
needSplit bool
|
||||
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
|
||||
}
|
||||
|
||||
// arm64ComputeFrame derives the frame layout for a TEXT function.
|
||||
@@ -80,15 +86,68 @@ func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
|
||||
|
||||
if fi.frame != 0 || !fi.leaf {
|
||||
fi.autosize = fi.frame + 8 // space for the saved LR
|
||||
if fi.autosize%16 != 0 {
|
||||
// The toolchain aligns to 16: if autosize%16 == 8, add 8;
|
||||
// otherwise add whatever is needed.
|
||||
// The toolchain always adds an extrasize: 8 when the total leaves a
|
||||
// 16-byte alignment gap, another 16 when already aligned.
|
||||
switch fi.autosize % 16 {
|
||||
case 8:
|
||||
fi.autosize += 8
|
||||
case 0:
|
||||
fi.autosize += 16
|
||||
default:
|
||||
// The toolchain rejects unaligned frames; round up so such
|
||||
// sources still assemble.
|
||||
fi.autosize += 16 - (fi.autosize % 16)
|
||||
}
|
||||
}
|
||||
switch {
|
||||
case fi.noSplit:
|
||||
case fi.autosize < stackSmall && fi.leaf:
|
||||
// Auto-NOSPLIT, as the toolchain's leaf mark concludes.
|
||||
default:
|
||||
fi.needSplit = true
|
||||
switch {
|
||||
case fi.autosize <= stackSmall:
|
||||
fi.splitClass = 0
|
||||
case fi.autosize <= stackBig:
|
||||
fi.splitClass = 1
|
||||
default:
|
||||
fi.splitClass = 2
|
||||
}
|
||||
}
|
||||
return fi
|
||||
}
|
||||
|
||||
// arm64GuardLen returns the byte length of the stack-split guard prefix
|
||||
// (zero when the function needs no guard). The big class materialises
|
||||
// framesize-StackSmall into REGTMP, whose MOVZ/MOVK sequence length varies.
|
||||
func arm64GuardLen(fi arm64FrameInfo) int {
|
||||
if !fi.needSplit {
|
||||
return 0
|
||||
}
|
||||
switch fi.splitClass {
|
||||
case 0:
|
||||
return 12
|
||||
case 1:
|
||||
return 16
|
||||
default:
|
||||
n, err := arm64LoadImmLen(int64(fi.autosize - stackSmall))
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
return 4 + n + 4 + 4 + 4 + 4
|
||||
}
|
||||
}
|
||||
|
||||
// arm64LoadImmLen returns the byte length of the MOVZ/MOVK sequence that
|
||||
// loads v into a register.
|
||||
func arm64LoadImmLen(v int64) (int, error) {
|
||||
b, err := encodeARM64LoadImm(27, v, "MOVD")
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return len(b), nil
|
||||
}
|
||||
|
||||
// arm64IsLeaf reports whether a function contains no call instructions
|
||||
// (BL/CALL), matching the toolchain's LEAF mark.
|
||||
func arm64IsLeaf(t *ast.Text) bool {
|
||||
@@ -119,12 +178,99 @@ func arm64Prologue(fi arm64FrameInfo) []byte {
|
||||
)
|
||||
}
|
||||
// Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP
|
||||
return a64WordsLE(
|
||||
a64AddSub(1, 1, 0, 0, uint32(fi.autosize), 31, 20), // SUB $autosize, SP, R20
|
||||
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
|
||||
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
|
||||
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
|
||||
ws := arm64SubImmWords(uint32(fi.autosize), 20)
|
||||
ws = append(ws,
|
||||
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
|
||||
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
|
||||
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
|
||||
)
|
||||
return a64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// arm64SplitImm12 reports whether the toolchain decomposes ADD/SUB $imm into
|
||||
// two imm12 instructions instead of materialising it into REGTMP
|
||||
// (asm7.go case 48, the C_ADDCON2 class): the value must fit 24 bits
|
||||
// unsigned and be neither encodable as one imm12 (checked by the callers
|
||||
// first), nor loadable into a register in a single MOVZ/MOVN word, nor a
|
||||
// logical immediate, because conclass tests all three before C_ADDCON2.
|
||||
func arm64SplitImm12(imm uint32) bool {
|
||||
if imm > 0xFFFFFF {
|
||||
return false
|
||||
}
|
||||
if _, _, _, ok := arm64Bitmask(uint64(imm), 1); ok {
|
||||
return false
|
||||
}
|
||||
return arm64Movcon(int64(imm)) < 0 && arm64Movcon(^int64(imm)) < 0
|
||||
}
|
||||
|
||||
// arm64SubImmWords emits SUB $imm, SP, Rd with the toolchain's ladder for an
|
||||
// ADD/SUB constant (asm7.go conclass and cases 2, 48, 62 and 13): the
|
||||
// immediate form when the value fits imm12 (plain, or shifted left by 12
|
||||
// when it is a multiple of 4096); a value with a single 16-bit chunk, a
|
||||
// logical immediate, or one wider than 24 bits is materialised into REGTMP
|
||||
// (R27) and subtracted in the extended-register form; everything else up to
|
||||
// 0xFFFFFF is split into two imm12 instructions:
|
||||
//
|
||||
// SUB $(imm&0xfff), SP, Rd
|
||||
// SUB $((imm&0xfff000)>>12)<<12, Rd, Rd
|
||||
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
||||
if imm <= 0xFFF {
|
||||
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
|
||||
}
|
||||
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
||||
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
|
||||
}
|
||||
if !arm64SplitImm12(imm) {
|
||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||
if err != nil {
|
||||
mov = nil
|
||||
}
|
||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
|
||||
}
|
||||
return []uint32{
|
||||
a64AddSub(1, 1, 0, 0, imm&0xFFF, 31, rd),
|
||||
a64AddSub(1, 1, 0, 1, (imm&0xFFF000)>>12, rd, rd),
|
||||
}
|
||||
}
|
||||
|
||||
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12,
|
||||
// split and REGTMP ladder as arm64SubImmWords.
|
||||
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
||||
if imm <= 0xFFF {
|
||||
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
|
||||
}
|
||||
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
||||
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
|
||||
}
|
||||
if !arm64SplitImm12(imm) {
|
||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||
if err != nil {
|
||||
mov = nil
|
||||
}
|
||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
|
||||
}
|
||||
return []uint32{
|
||||
a64AddSub(1, 0, 0, 0, imm&0xFFF, 31, rd),
|
||||
a64AddSub(1, 0, 0, 1, (imm&0xFFF000)>>12, rd, rd),
|
||||
}
|
||||
}
|
||||
|
||||
// arm64RetAddWords emits the frame deallocation of a non-leaf RET with a
|
||||
// large frame. The toolchain adds the frame back with a single instruction:
|
||||
// a plain imm12 ADD when autosize fits 12 bits, otherwise the value is
|
||||
// materialised into REGTMP and added as a register, so the epilogue never
|
||||
// leaves a partially deallocated frame (obj7.go ARET, issue 73259). The
|
||||
// shifted-imm12 and split-imm12 forms are therefore never used here, unlike
|
||||
// the leaf epilogue's plain ADD instructions.
|
||||
func arm64RetAddWords(autosize uint32) []uint32 {
|
||||
if autosize < 1<<12 {
|
||||
return []uint32{a64AddSub(1, 0, 0, 0, autosize, 31, 31)}
|
||||
}
|
||||
mov, err := encodeARM64LoadImm(27, int64(autosize), "MOVD")
|
||||
if err != nil {
|
||||
mov = nil
|
||||
}
|
||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, 31))
|
||||
}
|
||||
|
||||
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
|
||||
@@ -134,10 +280,8 @@ func arm64Return(fi arm64FrameInfo) []byte {
|
||||
if fi.autosize != 0 {
|
||||
if fi.leaf {
|
||||
// Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP
|
||||
ws = append(ws,
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize-8), 31, 29), // ADD $autosize-8, SP, FP
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
|
||||
)
|
||||
ws = append(ws, arm64AddImmWords(uint32(fi.autosize-8), 29)...)
|
||||
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
|
||||
} else if fi.autosize <= 0xf0 {
|
||||
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize
|
||||
ws = append(ws,
|
||||
@@ -145,11 +289,11 @@ func arm64Return(fi arm64FrameInfo) []byte {
|
||||
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
|
||||
)
|
||||
} else {
|
||||
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
|
||||
// Large frame: LDP -8(SP), (FP, LR), then deallocate.
|
||||
ws = append(ws,
|
||||
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
|
||||
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
||||
)
|
||||
ws = append(ws, arm64RetAddWords(uint32(fi.autosize))...)
|
||||
}
|
||||
}
|
||||
// RET: BR LR (0xd65f03c0)
|
||||
@@ -166,22 +310,32 @@ func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
|
||||
if fi.autosize <= 0xf0 {
|
||||
return 4 // MOVD.W instruction decrements SP
|
||||
}
|
||||
return 8 // SUB + STP + MOVD (3 instructions, SP updated at the MOVD)
|
||||
// Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose
|
||||
// position depends on how many words the SUB itself took (immediate,
|
||||
// shifted immediate, the two-word imm12 split, or a materialised REGTMP
|
||||
// sequence).
|
||||
return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1)
|
||||
}
|
||||
|
||||
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
||||
// (but not including) the final RET instruction.
|
||||
// (but not including) the final RET instruction. The lengths are read from
|
||||
// the same word-emitting helpers the epilogue uses rather than assumed: the
|
||||
// leaf path shares the prologue's immediate ladder, and a materialised
|
||||
// autosize costs its MOV words plus the ADD itself.
|
||||
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
|
||||
if fi.autosize == 0 {
|
||||
return 0
|
||||
}
|
||||
if fi.leaf {
|
||||
return 8 // ADD + ADD
|
||||
return 4 * (len(arm64AddImmWords(uint32(fi.autosize-8), 29)) +
|
||||
len(arm64AddImmWords(uint32(fi.autosize), 31)))
|
||||
}
|
||||
if fi.autosize <= 0xf0 {
|
||||
return 8 // LDR + LDR.P
|
||||
}
|
||||
return 8 // LDP + ADD
|
||||
// LDP + the deallocation emitted by arm64RetAddWords, so the length
|
||||
// tracks whatever the MOVD ladder needs.
|
||||
return 4 + 4*len(arm64RetAddWords(uint32(fi.autosize)))
|
||||
}
|
||||
|
||||
// arm64ResolvePseudo translates a pseudo-register memory reference into a
|
||||
@@ -235,3 +389,86 @@ func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 {
|
||||
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
|
||||
1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
||||
}
|
||||
|
||||
// Data-processing (shifted register) base opcodes for the guard blocks.
|
||||
const (
|
||||
arm64OpAdd = 1<<31 | 0<<30 | 0<<29 | 0x0b<<24
|
||||
arm64OpSub = 1<<31 | 1<<30 | 0<<29 | 0x0b<<24
|
||||
arm64OpSubs = 1<<31 | 1<<30 | 1<<29 | 0x0b<<24
|
||||
)
|
||||
|
||||
// arm64DPSRWords builds one data-processing (shifted register) word:
|
||||
// OP Rm, Rn, Rd in the Go assembler's operand order.
|
||||
func arm64DPSRWords(base uint32, rm, rn, rd uint32) uint32 {
|
||||
return base | rm<<16 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// arm64DPExtWords builds one data-processing (extended register) word, the
|
||||
// form the toolchain picks when a large immediate was materialised into
|
||||
// REGTMP before the operation: base | 1<<21 | Rm<<16 | UXTX<<13 | Rn<<5 | Rd.
|
||||
func arm64DPExtWords(base, rm, rn, rd uint32) uint32 {
|
||||
return base | 1<<21 | rm<<16 | 3<<13 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// wordsOf converts little-endian instruction bytes back to words.
|
||||
func wordsOf(b []byte) []uint32 {
|
||||
ws := make([]uint32, 0, len(b)/4)
|
||||
for i := 0; i+4 <= len(b); i += 4 {
|
||||
ws = append(ws, uint32(b[i])|uint32(b[i+1])<<8|uint32(b[i+2])<<16|uint32(b[i+3])<<24)
|
||||
}
|
||||
return ws
|
||||
}
|
||||
|
||||
// arm64GuardBytes emits the stack-split guard prefix; blockStart is the
|
||||
// function-relative byte address of the morestack block the branches target.
|
||||
func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte {
|
||||
// MOVD 16(R28), R16 (g.stackguard0)
|
||||
ws := []uint32{a64LSU(3, 0, 1, 2, 28, 16)}
|
||||
br := func(from int, cond uint32) uint32 {
|
||||
return a64BranchCond(int32((blockStart-from)>>2), cond)
|
||||
}
|
||||
switch fi.splitClass {
|
||||
case 0:
|
||||
// CMP R16, RSP in the exact encoding go tool asm emits for it.
|
||||
ws = append(ws, 0xeb3063ff)
|
||||
ws = append(ws, br(8, a64CondLS))
|
||||
case 1:
|
||||
ws = append(ws, a64AddSub(1, 1, 0, 0, uint32(fi.autosize-stackSmall), 31, 17))
|
||||
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
|
||||
ws = append(ws, br(12, a64CondLS))
|
||||
default:
|
||||
mov, err := encodeARM64LoadImm(27, int64(fi.autosize-stackSmall), "MOVD")
|
||||
if err != nil {
|
||||
mov = nil
|
||||
}
|
||||
ws = append(ws, wordsOf(mov)...)
|
||||
ml := len(mov) / 4
|
||||
ws = append(ws, arm64DPExtWords(arm64OpSubs, 27, 31, 17)) // SUBS R17, RSP, R27
|
||||
// The branches sit at fixed byte offsets in the guard prefix: after
|
||||
// the LDR (4), the ml MOV words (4*ml) and the SUBS (4) for B.LO,
|
||||
// then a further B.LO word and the CMP for B.LS.
|
||||
ws = append(ws, br(8+4*ml, a64CondLO))
|
||||
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
|
||||
ws = append(ws, br(16+4*ml, a64CondLS))
|
||||
}
|
||||
return a64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// arm64MoreStackBlock emits the trailing block: MOVD R30, R3 (save LR),
|
||||
// BL runtime.morestack_noctxt, B back to the function start. The BL carries
|
||||
// the R_CALLARM64 relocation.
|
||||
func arm64MoreStackBlock(blockStart int) ([]byte, Reloc) {
|
||||
ws := []uint32{
|
||||
1<<31 | 1<<29 | 0x0a<<24 | 30<<16 | 31<<5 | 3, // MOVD R30, R3
|
||||
a64Branch(1, 0), // BL, patched by the linker
|
||||
}
|
||||
bPC := blockStart + 8
|
||||
ws = append(ws, a64Branch(0, int32(-bPC>>2))) // B back to the entry
|
||||
reloc := Reloc{
|
||||
Off: blockStart + 4,
|
||||
After: blockStart + 8,
|
||||
Name: "runtime\u00b7morestack_noctxt",
|
||||
Kind: RelArm64Branch,
|
||||
}
|
||||
return a64WordsLE(ws...), reloc
|
||||
}
|
||||
|
||||
@@ -0,0 +1,126 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// parseArm64File is a helper assembling one arm64 source file.
|
||||
func parseArm64File(t *testing.T, src string) *Image {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse("k_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
return img
|
||||
}
|
||||
|
||||
// TestArm64RelocOffsetsIncludePrologue pins the function-relative relocation
|
||||
// offsets of a framed function: the offsets used to exclude the prologue, so
|
||||
// every relocation landed on a prologue instruction in the GOOBJ/ELF output.
|
||||
// The function calls an external, so it is a non-leaf and carries the
|
||||
// stack-split guard (12 bytes, small class) before the prologue.
|
||||
func TestArm64RelocOffsetsIncludePrologue(t *testing.T) {
|
||||
img := parseArm64File(t, "TEXT \u00b7f(SB), $16-0\n"+
|
||||
"\tBL ext\u00b7foo(SB)\n"+
|
||||
"\tMOVD $gdata(SB), R5\n"+
|
||||
"\tMOVD $extsym(SB), R6\n"+
|
||||
"\tRET\n"+
|
||||
"GLOBL gdata(SB), $8\n")
|
||||
fn := img.Funcs[0]
|
||||
|
||||
// Layout: 12-byte guard, 12-byte prologue, BL (24), ADRP+ADD (28, 32),
|
||||
// ADRP+ADD (36, 40), 12-byte epilogue with RET, 12-byte morestack block.
|
||||
want := []struct {
|
||||
off int
|
||||
after int
|
||||
name string
|
||||
kind RelocKind
|
||||
external bool
|
||||
}{
|
||||
{24, 28, "foo", RelArm64Branch, true},
|
||||
{28, 28, "gdata", RelArm64Addr, false},
|
||||
{32, 32, "gdata", RelArm64Addr, false},
|
||||
{36, 36, "extsym", RelArm64Addr, true},
|
||||
{40, 40, "extsym", RelArm64Addr, true},
|
||||
{60, 64, "runtime\u00b7morestack_noctxt", RelArm64Branch, true},
|
||||
}
|
||||
if len(fn.Relocs) != len(want) {
|
||||
t.Fatalf("relocs = %d, want %d", len(fn.Relocs), len(want))
|
||||
}
|
||||
for i, w := range want {
|
||||
r := fn.Relocs[i]
|
||||
if r.Off != w.off || r.After != w.after || r.Name != w.name || r.Kind != w.kind || r.External != w.external {
|
||||
t.Errorf("reloc %d = {off %d after %d name %q kind %d ext %v}, want {off %d after %d name %q kind %d ext %v}",
|
||||
i, r.Off, r.After, r.Name, r.Kind, r.External, w.off, w.after, w.name, w.kind, w.external)
|
||||
}
|
||||
}
|
||||
|
||||
// The BL with a zero offset sits exactly at the first reloc site.
|
||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
if w := binary.LittleEndian.Uint32(code[24:28]); w != 0x94000000 {
|
||||
t.Errorf("BL word = %08x, want 94000000", w)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64SBLoadStoreMatchesToolchain pins the ADRP scratch register
|
||||
// (REGTMP, R27) and the LDST64 relocation kind for sym loads and stores,
|
||||
// against the bytes go tool asm emits for MOVD sym(SB), R5.
|
||||
func TestArm64SBLoadStoreMatchesToolchain(t *testing.T) {
|
||||
img := parseArm64File(t, "TEXT \u00b7ld(SB), NOSPLIT, $0\n"+
|
||||
"\tMOVD sym(SB), R5\n"+
|
||||
"\tMOVD R5, sym(SB)\n"+
|
||||
"\tRET\n"+
|
||||
"GLOBL sym(SB), $8\n")
|
||||
fn := img.Funcs[0]
|
||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
|
||||
// go tool asm: ADRP 0(PC), R27 (9000001b); MOVD (R27), R5 (f9400365);
|
||||
// ADRP 0(PC), R27; MOVD R5, (R27) (f9000365).
|
||||
for off, want := range map[int]uint32{0: 0x9000001b, 4: 0xf9400365, 8: 0x9000001b, 12: 0xf9000365} {
|
||||
if got := binary.LittleEndian.Uint32(code[off : off+4]); got != want {
|
||||
t.Errorf("word at %d = %08x, want %08x", off, got, want)
|
||||
}
|
||||
}
|
||||
|
||||
if len(fn.Relocs) != 2 {
|
||||
t.Fatalf("relocs = %d, want 2", len(fn.Relocs))
|
||||
}
|
||||
for i, w := range []struct{ off, after int }{{0, 8}, {8, 16}} {
|
||||
r := fn.Relocs[i]
|
||||
if r.Kind != RelArm64LDST64 {
|
||||
t.Errorf("reloc %d kind = %d, want RelArm64LDST64 (%d)", i, r.Kind, RelArm64LDST64)
|
||||
}
|
||||
if r.Off != w.off || r.After != w.after {
|
||||
t.Errorf("reloc %d = {off %d after %d}, want {off %d after %d}", i, r.Off, r.After, w.off, w.after)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64GOObjRelocTypes checks that GOOBJ emission succeeds with the new
|
||||
// relocation kinds in play; the detailed layout is covered by the goobj tests.
|
||||
func TestArm64GOObjRelocTypes(t *testing.T) {
|
||||
img := parseArm64File(t, "TEXT \u00b7ld(SB), NOSPLIT, $0\n"+
|
||||
"\tMOVD sym(SB), R5\n"+
|
||||
"\tMOVD R5, sym(SB)\n"+
|
||||
"\tRET\n"+
|
||||
"GLOBL sym(SB), $8\n")
|
||||
obj, err := img.GOObjectAARCH64("testpkg", "k_arm64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObjectAARCH64: %v", err)
|
||||
}
|
||||
if len(obj) == 0 {
|
||||
t.Fatal("empty object")
|
||||
}
|
||||
// The detailed layout is covered by the goobj tests; here we only pin
|
||||
// that emission succeeds with the new relocation kinds in play.
|
||||
}
|
||||
+927
-43
File diff suppressed because it is too large
Load Diff
+333
-5
@@ -4,13 +4,14 @@
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// firstText parses src and returns its first TEXT function.
|
||||
@@ -62,7 +63,7 @@ TEXT ·f(SB), NOSPLIT, $0
|
||||
XORQ AX, AX
|
||||
loop:
|
||||
ADDQ $1, AX
|
||||
CMPQ $10, AX
|
||||
CMPQ AX, $10
|
||||
JLT loop
|
||||
RET
|
||||
`)
|
||||
@@ -159,6 +160,57 @@ TEXT ·loadarg(SB), NOSPLIT, $0-24
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleFramelessCall verifies the forced base-pointer frame a $0-frame
|
||||
// function containing a CALL receives: the PUSHQ BP prologue with no stack
|
||||
// adjustment and the x+N(FP) → (N+16)(SP) translation, against the bytes the
|
||||
// Go assembler produces. The push is the frame, so the offset must not count
|
||||
// it twice.
|
||||
func TestAssembleFramelessCall(t *testing.T) {
|
||||
f, errs := parser.Parse("frameless_call_amd64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·withcall(SB), NOSPLIT, $0-16
|
||||
MOVQ x+0(FP), AX
|
||||
CALL ·other(SB)
|
||||
MOVQ AX, ret+8(FP)
|
||||
RET
|
||||
TEXT ·other(SB), NOSPLIT, $0-0
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
code := append([]byte(nil), img.Code[img.Funcs[0].Offset:img.Funcs[0].Offset+img.Funcs[0].Size]...)
|
||||
for _, r := range img.Funcs[0].Relocs {
|
||||
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||
code[j] = 0
|
||||
}
|
||||
}
|
||||
// From `go tool objdump` of the Go-assembled function:
|
||||
// PUSHQ BP 55
|
||||
// MOVQ SP, BP 4889e5
|
||||
// MOVQ 0x10(SP), AX 488b442410
|
||||
// CALL other e800000000
|
||||
// MOVQ AX, 0x18(SP) 4889442418
|
||||
// POPQ BP 5d
|
||||
// RET c3
|
||||
want := []byte{
|
||||
0x55,
|
||||
0x48, 0x89, 0xe5,
|
||||
0x48, 0x8b, 0x44, 0x24, 0x10,
|
||||
0xe8, 0x00, 0x00, 0x00, 0x00,
|
||||
0x48, 0x89, 0x44, 0x24, 0x18,
|
||||
0x5d,
|
||||
0xc3,
|
||||
}
|
||||
if hexBytes(code) != hexBytes(want) {
|
||||
t.Errorf("frameless CALL FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style
|
||||
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
|
||||
// the bytes the Go assembler produces.
|
||||
@@ -201,8 +253,8 @@ TEXT ·withframe(SB), NOSPLIT, $16-16
|
||||
}
|
||||
|
||||
// TestAssembleVexKernel assembles the horizontal-sum reduction the go-flac
|
||||
// kernels end with — exercising the VEX moves, shuffle and extract forms
|
||||
// through the full parser → encoder path — and checks the output is
|
||||
// kernels end with; exercising the VEX moves, shuffle and extract forms
|
||||
// through the full parser → encoder path; and checks the output is
|
||||
// byte-identical to the Go assembler's.
|
||||
func TestAssembleVexKernel(t *testing.T) {
|
||||
fn := firstText(t, `
|
||||
@@ -317,3 +369,279 @@ end:
|
||||
t.Errorf("jump-folding mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleNumericPCJumps pins the numeric ±N(PC) branch operands: N
|
||||
// counts instruction statements, skipping labels, in both directions (the
|
||||
// runtime's exit loops write JMP -3(PC)), N = 0 parks on the jump itself.
|
||||
func TestAssembleNumericPCJumps(t *testing.T) {
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·exit(SB), NOSPLIT, $0
|
||||
MOVB $1, AL
|
||||
lab:
|
||||
MOVB $2, AL
|
||||
MOVB $3, AL
|
||||
JMP -3(PC)
|
||||
MOVB $4, AL
|
||||
park:
|
||||
JMP 0(PC)
|
||||
MOVB $5, AL
|
||||
JMP 2(PC)
|
||||
MOVB $6, AL
|
||||
RET
|
||||
`)
|
||||
code, _, err := Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble: %v", err)
|
||||
}
|
||||
// From the Go-assembled function:
|
||||
// MOVB $1, AL b001
|
||||
// MOVB $2, AL b002
|
||||
// MOVB $3, AL b003
|
||||
// JMP -3(PC) ebf8 (three instructions back, past lab:)
|
||||
// MOVB $4, AL b004
|
||||
// JMP 0(PC) ebfe (the park loop)
|
||||
// MOVB $5, AL b005
|
||||
// JMP 2(PC) eb02 (over MOVB $6 to the RET)
|
||||
// MOVB $6, AL b006
|
||||
// RET c3
|
||||
want := []byte{
|
||||
0xb0, 0x01,
|
||||
0xb0, 0x02,
|
||||
0xb0, 0x03,
|
||||
0xeb, 0xf8,
|
||||
0xb0, 0x04,
|
||||
0xeb, 0xfe,
|
||||
0xb0, 0x05,
|
||||
0xeb, 0x02,
|
||||
0xb0, 0x06,
|
||||
0xc3,
|
||||
}
|
||||
if hexBytes(code) != hexBytes(want) {
|
||||
t.Errorf("numeric-PC mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssemblePrefetch(t *testing.T) {
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·pf(SB), NOSPLIT, $0
|
||||
PREFETCHNTA (AX)
|
||||
PREFETCHT0 (BX)
|
||||
PREFETCHT1 8(CX)
|
||||
PREFETCHT2 -1(AX)(R12*1)
|
||||
RET
|
||||
`)
|
||||
code, _, err := Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble: %v", err)
|
||||
}
|
||||
got := strings.Join(disasm(t, code), "\n")
|
||||
want := strings.Join([]string{
|
||||
"prefetchnta zmmword ptr [rax]",
|
||||
"prefetcht0 zmmword ptr [rbx]",
|
||||
"prefetcht1 zmmword ptr [rcx+0x8]",
|
||||
"prefetcht2 zmmword ptr [rax+r12-0x1]",
|
||||
"ret",
|
||||
}, "\n")
|
||||
if got != want {
|
||||
t.Errorf("prefetch disassembly mismatch:\n got:\n%s\n want:\n%s", got, want)
|
||||
}
|
||||
// Byte-level expectations: 0F 18 with the variant in the reg field.
|
||||
if hex := hexBytes(code[:3]); hex != "0f 18 00" {
|
||||
t.Errorf("PREFETCHNTA bytes: got %s, want 0f 18 00", hex)
|
||||
}
|
||||
if hex := hexBytes(code[3:6]); hex != "0f 18 0b" {
|
||||
t.Errorf("PREFETCHT0 bytes: got %s, want 0f 18 0b", hex)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleBareJump checks that a zero-operand jump (which parses, because
|
||||
// the parser does not arity-check mnemonics) is rejected with an error rather
|
||||
// than panicking in the layout loop, which indexes Operands[0] before the
|
||||
// emission pass gets a chance to diagnose the arity.
|
||||
func TestAssembleBareJump(t *testing.T) {
|
||||
for _, mnem := range []string{"JE", "JMP", "JLT", "CALL"} {
|
||||
fn := firstText(t, "TEXT ·bare(SB), $16-0\n\t"+mnem+"\n")
|
||||
if _, _, err := Assemble(fn); err == nil {
|
||||
t.Errorf("%s with no operand: expected an error, got none", mnem)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
|
||||
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
|
||||
// larger. The intermediate 129..255 range used to encode an ADD with a
|
||||
// truncated immediate, moving SP the wrong way.
|
||||
func TestSubSPEncodings(t *testing.T) {
|
||||
for _, tt := range []struct {
|
||||
size int
|
||||
want []byte
|
||||
}{
|
||||
{8, []byte{0x48, 0x83, 0xEC, 0x08}},
|
||||
{127, []byte{0x48, 0x83, 0xEC, 0x7F}},
|
||||
{128, []byte{0x48, 0x81, 0xEC, 0x80, 0x00, 0x00, 0x00}},
|
||||
{200, []byte{0x48, 0x81, 0xEC, 0xC8, 0x00, 0x00, 0x00}},
|
||||
{255, []byte{0x48, 0x81, 0xEC, 0xFF, 0x00, 0x00, 0x00}},
|
||||
{4096, []byte{0x48, 0x81, 0xEC, 0x00, 0x10, 0x00, 0x00}},
|
||||
} {
|
||||
got := subSP(tt.size)
|
||||
if !bytes.Equal(got, tt.want) {
|
||||
t.Errorf("subSP(%d) = %x, want %x", tt.size, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssemblePseudoStatements runs LOCK/REP, BYTE/WORD and END through the
|
||||
// full statement pipeline, pinned against go tool asm (Go 1.27, amd64). It
|
||||
// asserts the three behaviours the toolchain shows: each prefix statement is
|
||||
// a standalone byte with a PC of its own (so a label placed on the LOCK
|
||||
// points at the F0), the data pseudo-ops write their literal bytes inline,
|
||||
// and END terminates nothing (the statements after it still belong to the
|
||||
// function and carry no trace of it).
|
||||
func TestAssemblePseudoStatements(t *testing.T) {
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·pseudo(SB), NOSPLIT, $0-0
|
||||
pfx:
|
||||
LOCK
|
||||
CMPXCHGQ AX, (BX)
|
||||
REP
|
||||
MOVSQ
|
||||
BYTE $0x0f
|
||||
BYTE $0x1f
|
||||
WORD $0x1234
|
||||
END
|
||||
BYTE $0x02
|
||||
RET
|
||||
`)
|
||||
code, labels, err := Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble: %v", err)
|
||||
}
|
||||
// go tool asm: f0 480fb103 f3 48a5 0f 1f 3412 02 c3
|
||||
want := []byte{
|
||||
0xf0,
|
||||
0x48, 0x0f, 0xb1, 0x03,
|
||||
0xf3, 0x48, 0xa5,
|
||||
0x0f, 0x1f, 0x34, 0x12,
|
||||
0x02, 0xc3,
|
||||
}
|
||||
if hexBytes(code) != hexBytes(want) {
|
||||
t.Errorf("pseudo statements:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
// The label sits on the LOCK byte, exactly where the toolchain's PC
|
||||
// listing puts it.
|
||||
if off := labels["pfx"]; off != 0 {
|
||||
t.Errorf("label pfx = %d, want 0 (the LOCK's own byte)", off)
|
||||
}
|
||||
// The trailing BYTE lands where the layout says: after the 8 bytes of
|
||||
// LOCK, CMPXCHGQ, REP and MOVSQ plus the 4 data bytes, END contributing
|
||||
// none.
|
||||
if code[12] != 0x02 {
|
||||
t.Errorf("byte at 12 = %02x, want 02 (the BYTE after END)", code[12])
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleAdjspBalance pins the toolchain's push/pop balance rule over
|
||||
// ADJSP: the straight-line sum of the adjustments must be zero at each
|
||||
// RET, branches in between counting for nothing (verified against go tool
|
||||
// asm: ADJSP $16 before a RET is reported as "unbalanced PUSH/POP", a
|
||||
// $16/$-16 pair with a JMP in between assembles).
|
||||
func TestAssembleAdjspBalance(t *testing.T) {
|
||||
// Balanced pair with a branch in between, bytes pinned from go tool asm.
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·adjsp(SB), NOSPLIT, $0-0
|
||||
ADJSP $16
|
||||
JMP body
|
||||
body:
|
||||
ADJSP $-16
|
||||
RET
|
||||
`)
|
||||
code, _, err := Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble: %v", err)
|
||||
}
|
||||
want := []byte{0x48, 0x83, 0xEC, 0x10, 0xEB, 0x00, 0x48, 0x83, 0xC4, 0x10, 0xC3}
|
||||
if hexBytes(code) != hexBytes(want) {
|
||||
t.Errorf("adjsp pair:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
|
||||
// Unbalanced at the RET: the toolchain diagnoses, so must we.
|
||||
_, _, err = Assemble(firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·unbalanced(SB), NOSPLIT, $0-0
|
||||
ADJSP $16
|
||||
RET
|
||||
`))
|
||||
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
|
||||
t.Errorf("unbalanced ADJSP: err = %v, want unbalanced PUSH/POP", err)
|
||||
}
|
||||
|
||||
// The check runs per RET: a closed pair before the first RET does not
|
||||
// excuse an open adjustment before the second.
|
||||
_, _, err = Assemble(firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·tworet(SB), NOSPLIT, $0-0
|
||||
ADJSP $8
|
||||
ADJSP $-8
|
||||
RET
|
||||
mid:
|
||||
ADJSP $8
|
||||
RET
|
||||
`))
|
||||
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
|
||||
t.Errorf("second RET with open ADJSP: err = %v, want unbalanced PUSH/POP", err)
|
||||
}
|
||||
|
||||
// A framed function: the assembler's own prologue and epilogue
|
||||
// contribute matching deltas, so the pair in the body still balances,
|
||||
// and the bytes match go tool asm end to end.
|
||||
fn = firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·framed(SB), $16-8
|
||||
ADJSP $8
|
||||
ADJSP $-8
|
||||
RET
|
||||
`)
|
||||
code, _, err = Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble framed: %v", err)
|
||||
}
|
||||
want = []byte{
|
||||
0x55, 0x48, 0x89, 0xE5, 0x48, 0x83, 0xEC, 0x10, // prologue
|
||||
0x48, 0x83, 0xEC, 0x08, // ADJSP $8
|
||||
0x48, 0x83, 0xC4, 0x08, // ADJSP $-8
|
||||
0x48, 0x83, 0xC4, 0x10, 0x5D, // epilogue
|
||||
0xC3,
|
||||
}
|
||||
if hexBytes(code) != hexBytes(want) {
|
||||
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleRegRange pins the bracketed register range at the statement
|
||||
// level: exactly four consecutive same-width vector registers assemble, the
|
||||
// toolchain's rejected shapes all report an error.
|
||||
func TestAssembleRegRange(t *testing.T) {
|
||||
asm := func(t *testing.T, op string) ([]byte, error) {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse("f_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tV4FMADDPS 17(SP), "+op+", K2, Z0\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse %s: %v", op, errs)
|
||||
}
|
||||
code, _, err := Assemble(f.Decls[0].(*ast.Text))
|
||||
return code, err
|
||||
}
|
||||
for _, op := range []string{"[Z0-Z3]", "[Z4-Z7]", "[Z28-Z31]"} {
|
||||
if _, err := asm(t, op); err != nil {
|
||||
t.Errorf("%s: %v", op, err)
|
||||
}
|
||||
}
|
||||
for _, op := range []string{"[Z0-Z4]", "[Z0-Z2]", "[Z0-Z0]", "[Z4-Z0]", "[Z1-Z0]", "[AX-Z3]", "[Z0-AX]"} {
|
||||
if _, err := asm(t, op); err == nil {
|
||||
t.Errorf("%s: assembled, want an error", op)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+144
-17
@@ -12,12 +12,11 @@ import (
|
||||
// Image: a .text section holding the function bodies, a .data section
|
||||
// holding the GLOBL initialisers, a symbol table with one symbol per TEXT
|
||||
// and GLOBL (file-local <> symbols are STB_LOCAL, the rest STB_GLOBAL), and
|
||||
// a .rela.text relocation table — one R_X86_64_PC32 entry per static-symbol
|
||||
// a .rela.text relocation table, one R_X86_64_PC32 entry per static-symbol
|
||||
// reference, internal references resolving against the local data symbols
|
||||
// and external ones against undefined globals. The output links with the
|
||||
// system toolchain (cc/ld) the way a hand-assembled .o would.
|
||||
|
||||
// ELF constants (ELF64, little-endian, System V).
|
||||
const (
|
||||
elfClass64 = 2
|
||||
elfDataLSB = 1
|
||||
@@ -36,18 +35,20 @@ const (
|
||||
shfAlloc = 2
|
||||
shfExecInstr = 4
|
||||
|
||||
stbLocal = 0
|
||||
stbGlobal = 1
|
||||
|
||||
sttNotype = 0
|
||||
sttObject = 1
|
||||
sttFunc = 2
|
||||
sttSection = 3
|
||||
stInfoShift = 4
|
||||
|
||||
shnUndef = 0
|
||||
|
||||
rX8664PC32 = 2
|
||||
// R_X86_64_32 (debug/elf): the absolute 32-bit address of a symbol, the
|
||||
// R_ADDR shape a 4-byte DATA field carries.
|
||||
rX8664Abs32 = 10
|
||||
// R_X86_64_TPOFF32 (debug/elf): the local-exec TLS offset the stack
|
||||
// guard loads from FS. 20 is R_X86_64_TLSLD, a different relocation.
|
||||
rX8664TPOFF32 = 23
|
||||
)
|
||||
|
||||
// elfSym is one symbol-table entry in construction.
|
||||
@@ -76,7 +77,7 @@ func (img *Image) ELFObject() ([]byte, error) {
|
||||
|
||||
// Build the symbol table: the null entry and the two section symbols
|
||||
// come first, then the local symbols (static TEXT and GLOBL), then the
|
||||
// globals (exported TEXT and GLOBL, and the undefined externals) — ELF
|
||||
// globals (exported TEXT and GLOBL, and the undefined externals), ELF
|
||||
// requires every local to precede every global, and sh_info records the
|
||||
// boundary. symIdx maps a symbol name to its index for the relocations.
|
||||
var locals, globals []elfSym
|
||||
@@ -130,11 +131,19 @@ func (img *Image) ELFObject() ([]byte, error) {
|
||||
type elfRela struct {
|
||||
off uint64
|
||||
sym int
|
||||
typ uint32
|
||||
addend int64
|
||||
}
|
||||
var relas []elfRela
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
var typ uint32 = rX8664PC32
|
||||
if r.Kind == RelTLSLE {
|
||||
// R_X86_64_TPOFF32 resolves to the local-exec TLS offset and
|
||||
// carries no symbol.
|
||||
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), sym: 0, typ: rX8664TPOFF32})
|
||||
continue
|
||||
}
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||
@@ -142,6 +151,7 @@ func (img *Image) ELFObject() ([]byte, error) {
|
||||
relas = append(relas, elfRela{
|
||||
off: uint64(fn.Offset + r.Off),
|
||||
sym: idx,
|
||||
typ: typ,
|
||||
// R_X86_64_PC32 computes S + A − P with P the patch site; the
|
||||
// assembler measures the symbol from the instruction end,
|
||||
// After − Off bytes past the field, so the addend carries
|
||||
@@ -151,6 +161,50 @@ func (img *Image) ELFObject() ([]byte, error) {
|
||||
}
|
||||
}
|
||||
|
||||
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||
// DATA line's width at the field's data-section offset, S + A with no
|
||||
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||
// cannot hold an address, so they are refused rather than truncated.
|
||||
var dataRelas []elfRela
|
||||
for _, d := range img.DataSyms {
|
||||
for _, r := range d.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
var typ uint32
|
||||
switch r.Siz {
|
||||
case 8:
|
||||
typ = rX8664Abs64
|
||||
case 4:
|
||||
typ = rX8664Abs32
|
||||
default:
|
||||
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||
}
|
||||
dataRelas = append(dataRelas, elfRela{
|
||||
off: uint64(d.Offset + r.Off),
|
||||
sym: idx,
|
||||
typ: typ,
|
||||
addend: r.Addend,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Section presence: .rela.text only when there are code relocations,
|
||||
// .rela.data only when a DATA line holds a symbol value.
|
||||
hasRela := len(relas) > 0
|
||||
hasDataRela := len(dataRelas) > 0
|
||||
nSections := 6 // NULL, .text, .data, .symtab, .strtab, .shstrtab
|
||||
if hasRela {
|
||||
nSections++
|
||||
}
|
||||
if hasDataRela {
|
||||
nSections++
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Serialise the string tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
@@ -160,15 +214,12 @@ func (img *Image) ELFObject() ([]byte, error) {
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
// Section presence: .rela.text only when there are relocations.
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6 // NULL, .text, .data, .symtab, .strtab, .shstrtab
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
if hasDataRela {
|
||||
stSections.add(".rela.data")
|
||||
}
|
||||
for _, n := range dwarfSectionNames {
|
||||
stSections.add(n)
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Lay the file out: header, section data, section headers.
|
||||
var out []byte
|
||||
@@ -204,14 +255,25 @@ func (img *Image) ELFObject() ([]byte, error) {
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
var relaOff, relaDataOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
for _, r := range relas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|rX8664PC32)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
if hasDataRela {
|
||||
align(8)
|
||||
relaDataOff = len(out)
|
||||
for _, r := range dataRelas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
@@ -220,6 +282,34 @@ func (img *Image) ELFObject() ([]byte, error) {
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
|
||||
// DWARF debug sections; the address placeholders they leave are carried
|
||||
// as .rela.debug_info/.rela.debug_line entries the system linker applies.
|
||||
dwAlign := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiAMD64)
|
||||
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
|
||||
if dw != nil {
|
||||
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
|
||||
// .debug_line_str and .debug_frame (the CIE is unconditional, so
|
||||
// the frame section is always present), plus the relocation
|
||||
// sections below when they carry entries.
|
||||
dwarfStart = nSections
|
||||
nSections += 5
|
||||
appendDWARFRelas(&out, dw, rX8664Abs64, dwAlign)
|
||||
if dw.infoRelaCount > 0 {
|
||||
nSections++
|
||||
}
|
||||
if dw.lineRelaCount > 0 {
|
||||
nSections++
|
||||
}
|
||||
if dw.frameRelaCount > 0 {
|
||||
nSections++
|
||||
}
|
||||
}
|
||||
|
||||
align(8)
|
||||
shoff := len(out)
|
||||
|
||||
@@ -246,8 +336,45 @@ func (img *Image) ELFObject() ([]byte, error) {
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
if hasDataRela {
|
||||
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
|
||||
// DWARF section headers; their indices follow the write order.
|
||||
if dw != nil {
|
||||
// secIdx is a running section index: each putSh below emits the
|
||||
// next header, and the sh_info of a .rela section names the index
|
||||
// of the section it relocates.
|
||||
secIdx := dwarfStart
|
||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||
secIdx++
|
||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||
secInfoIdx := secIdx
|
||||
secIdx++
|
||||
if dw.infoRelaCount > 0 {
|
||||
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
|
||||
secIdx++
|
||||
}
|
||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||
secLineIdx := secIdx
|
||||
secIdx++
|
||||
if dw.lineRelaCount > 0 {
|
||||
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
|
||||
secIdx++
|
||||
}
|
||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||
secIdx++
|
||||
if dw.frameSize > 0 {
|
||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||
secFrameIdx := secIdx
|
||||
secIdx++
|
||||
if dw.frameRelaCount > 0 {
|
||||
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The ELF header.
|
||||
hdr := out[:64]
|
||||
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
|
||||
|
||||
@@ -0,0 +1,406 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
)
|
||||
|
||||
// DWARF5 section generation for ELF output. Unlike the GOOBJ path (where
|
||||
// the linker assembles the final DWARF), the ELF path must emit complete,
|
||||
// self-contained sections because the system linker only performs fixup
|
||||
// relocations, not assembly.
|
||||
|
||||
// DWARF5 attribute, form and line-table constants (the values the
|
||||
// toolchain uses, cmd/internal/dwarf/dwarf_defs.go; the DIE streams below
|
||||
// are written against these forms).
|
||||
const (
|
||||
dwAtName = 0x03 // DW_AT_name
|
||||
dwAtStmtList = 0x10 // DW_AT_stmt_list
|
||||
dwAtLowPC = 0x11 // DW_AT_low_pc
|
||||
dwAtHighPC = 0x12 // DW_AT_high_pc
|
||||
dwAtDeclFile = 0x3a // DW_AT_decl_file
|
||||
dwAtDeclLine = 0x3b // DW_AT_decl_line
|
||||
dwAtExternal = 0x3f // DW_AT_external
|
||||
dwAtFrameBase = 0x40 // DW_AT_frame_base
|
||||
dwTagSubprog = 0x2e // DW_TAG_subprogram
|
||||
dwTagCompUnit = 0x11 // DW_TAG_compile_unit
|
||||
dwFormAddr = 0x01 // DW_FORM_addr
|
||||
dwFormData8 = 0x07 // DW_FORM_data8
|
||||
dwFormString = 0x08 // DW_FORM_string
|
||||
dwFormData1 = 0x0b // DW_FORM_data1
|
||||
dwFormUdata = 0x0f // DW_FORM_udata
|
||||
dwFormSecOff = 0x17 // DW_FORM_sec_offset
|
||||
dwFormExprloc = 0x18 // DW_FORM_exprloc
|
||||
dwFormLineStrp = 0x1f // DW_FORM_line_strp
|
||||
dwLnctPath = 0x01 // DW_LNCT_path
|
||||
dwLnctDirIndex = 0x02 // DW_LNCT_directory_index
|
||||
)
|
||||
|
||||
// dwarfAbbrevTable returns the .debug_abbrev content: a single compilation
|
||||
// unit with DW_TAG_compile_unit and DW_TAG_subprogram entries. The
|
||||
// attribute/form pairs must match the DIE streams dwarfBuildInfoSection
|
||||
// writes byte for byte, in the same order, or every consumer's parse of
|
||||
// .debug_info desynchronises.
|
||||
func dwarfAbbrevTable() []byte {
|
||||
var b []byte
|
||||
// Abbrev 1: DW_TAG_compile_unit.
|
||||
b = append(b, 1) // abbreviation code
|
||||
b = appendUleb(b, dwTagCompUnit) // DW_TAG_compile_unit
|
||||
b = append(b, 1) // DW_CHILDREN_yes
|
||||
b = appendUleb(b, dwAtLowPC) // DW_AT_low_pc
|
||||
b = appendUleb(b, dwFormAddr) // DW_FORM_addr
|
||||
b = appendUleb(b, dwAtHighPC) // DW_AT_high_pc
|
||||
b = appendUleb(b, dwFormData8) // DW_FORM_data8
|
||||
b = appendUleb(b, dwAtStmtList) // DW_AT_stmt_list
|
||||
b = appendUleb(b, dwFormSecOff) // DW_FORM_sec_offset (4 bytes here)
|
||||
b = appendUleb(b, dwAtName) // DW_AT_name
|
||||
b = appendUleb(b, dwFormString) // DW_FORM_string
|
||||
b = appendUleb(b, 0) // end of attributes: attr 0
|
||||
b = appendUleb(b, 0) // ... paired with form 0
|
||||
|
||||
// Abbrev 2: DW_TAG_subprogram.
|
||||
b = append(b, 2) // abbreviation code
|
||||
b = appendUleb(b, dwTagSubprog) // DW_TAG_subprogram
|
||||
b = append(b, 0) // DW_CHILDREN_no
|
||||
b = appendUleb(b, dwAtName) // DW_AT_name
|
||||
b = appendUleb(b, dwFormString) // DW_FORM_string
|
||||
b = appendUleb(b, dwAtLowPC) // DW_AT_low_pc
|
||||
b = appendUleb(b, dwFormAddr) // DW_FORM_addr
|
||||
b = appendUleb(b, dwAtHighPC) // DW_AT_high_pc
|
||||
b = appendUleb(b, dwFormData8) // DW_FORM_data8
|
||||
b = appendUleb(b, dwAtFrameBase) // DW_AT_frame_base
|
||||
b = appendUleb(b, dwFormExprloc) // DW_FORM_exprloc
|
||||
b = appendUleb(b, dwAtDeclFile) // DW_AT_decl_file
|
||||
b = appendUleb(b, dwFormData1) // DW_FORM_data1
|
||||
b = appendUleb(b, dwAtDeclLine) // DW_AT_decl_line
|
||||
b = appendUleb(b, dwFormData1) // DW_FORM_data1
|
||||
b = appendUleb(b, dwAtExternal) // DW_AT_external
|
||||
b = appendUleb(b, 0x0c) // DW_FORM_flag (one byte, 0 or 1)
|
||||
b = appendUleb(b, 0) // end of attributes: attr 0
|
||||
b = appendUleb(b, 0) // ... paired with form 0
|
||||
|
||||
// End of table.
|
||||
b = append(b, 0)
|
||||
return b
|
||||
}
|
||||
|
||||
// dwarfSections holds the generated DWARF section payloads and their
|
||||
// relocations (byte offsets within .debug_info and .debug_line that need
|
||||
// fixup against .text symbols).
|
||||
type dwarfSections struct {
|
||||
debugAbbrev []byte
|
||||
debugInfo []byte
|
||||
debugLine []byte
|
||||
debugLineStr []byte
|
||||
debugFrame []byte
|
||||
// Relocations for .debug_info: (offset, symbol name, addend).
|
||||
infoRelocs []dwarfReloc
|
||||
// Relocations for .debug_line: (offset, symbol name, addend).
|
||||
lineRelocs []dwarfReloc
|
||||
// Relocations for .debug_frame: (offset, symbol name, addend), one per
|
||||
// FDE initial_location.
|
||||
frameRelocs []dwarfReloc
|
||||
}
|
||||
|
||||
type dwarfReloc struct {
|
||||
off uint64
|
||||
name string
|
||||
addend int64
|
||||
}
|
||||
|
||||
// emitDWARF generates complete DWARF5 sections for the image. cfi carries
|
||||
// the architecture's .debug_frame register conventions.
|
||||
func emitDWARF(img *Image, srcFile string, cfi cfiArch) *dwarfSections {
|
||||
ds := &dwarfSections{}
|
||||
ds.debugAbbrev = dwarfAbbrevTable()
|
||||
|
||||
// Build the string table for .debug_line_str.
|
||||
lineStr := newElfStrtab()
|
||||
lineStr.add(srcFile)
|
||||
ds.debugLineStr = lineStr.bytes()
|
||||
|
||||
// Build .debug_line; the file table references the source name through
|
||||
// its offset in .debug_line_str.
|
||||
ds.debugLine = dwarfBuildLineSection(img, uint32(lineStr.at(srcFile)), ds)
|
||||
|
||||
// Build .debug_info.
|
||||
ds.debugInfo = dwarfBuildInfoSection(img, srcFile, ds)
|
||||
|
||||
// Build .debug_frame.
|
||||
ds.debugFrame = dwarfBuildFrameSection(img, cfi, ds)
|
||||
return ds
|
||||
}
|
||||
|
||||
// dwarfBuildLineSection builds a complete .debug_line section. srcStrOff is
|
||||
// the source file name's offset in .debug_line_str.
|
||||
func dwarfBuildLineSection(img *Image, srcStrOff uint32, ds *dwarfSections) []byte {
|
||||
var b []byte
|
||||
le := binary.LittleEndian
|
||||
|
||||
// We'll build the header first, then the programs, then patch the length.
|
||||
headerStart := len(b)
|
||||
b = append(b, 0, 0, 0, 0) // unit_length (placeholder)
|
||||
b = le.AppendUint16(b, 5) // version (DWARF5)
|
||||
b = append(b, 8) // address_size
|
||||
b = append(b, 0) // segment_selector_size
|
||||
b = append(b, 0, 0, 0, 0) // header_length (placeholder)
|
||||
|
||||
// Line program parameters.
|
||||
b = append(b, 1) // minimum_instruction_length
|
||||
b = append(b, 1) // maximum_ops_per_instruction
|
||||
b = append(b, 1) // default_is_stmt
|
||||
b = append(b, byte(dwLineBase&0xFF)) // line_base (-4 as unsigned)
|
||||
b = append(b, uint8(dwLineRange)) // line_range
|
||||
b = append(b, uint8(dwOpcodeBase)) // opcode_base
|
||||
// Standard opcode lengths (opcode 1..opcode_base-1).
|
||||
b = append(b, 0, 1, 1, 1, 1, 0, 0, 0, 1, 0)
|
||||
|
||||
// Directory table (DWARF5 §6.2.4): entry format descriptors followed by
|
||||
// the entries. One directory, the compilation directory, whose path is
|
||||
// the empty string at .debug_line_str offset 0.
|
||||
b = append(b, 1) // directory_entry_format_count
|
||||
b = appendUleb(b, dwLnctPath) // DW_LNCT_path
|
||||
b = appendUleb(b, dwFormLineStrp) // DW_FORM_line_strp
|
||||
b = appendUleb(b, 1) // directories_count
|
||||
b = le.AppendUint32(b, 0) // .debug_line_str offset of ""
|
||||
|
||||
// File table (DWARF5 §6.2.5). v5 indexes files from 0, so the source
|
||||
// file is entry 0, matching the DW_AT_decl_file value 0 the DIEs carry.
|
||||
b = append(b, 2) // file_name_entry_format_count
|
||||
b = appendUleb(b, dwLnctPath) // DW_LNCT_path
|
||||
b = appendUleb(b, dwFormLineStrp) // DW_FORM_line_strp
|
||||
b = appendUleb(b, dwLnctDirIndex) // DW_LNCT_directory_index
|
||||
b = appendUleb(b, dwFormUdata) // DW_FORM_udata
|
||||
b = appendUleb(b, 1) // file_names_count
|
||||
b = le.AppendUint32(b, srcStrOff) // .debug_line_str offset of the source name
|
||||
b = appendUleb(b, 0) // directory index 0 (the compilation directory)
|
||||
|
||||
headerEnd := len(b)
|
||||
|
||||
// Per-function line programs.
|
||||
for _, fn := range img.Funcs {
|
||||
// LNE_set_address with the function's offset in .text.
|
||||
b = append(b, 0, 9, 2) // extended opcode, length 9, DW_LNE_set_address
|
||||
addrOff := len(b)
|
||||
b = le.AppendUint64(b, 0) // placeholder for address
|
||||
ds.lineRelocs = append(ds.lineRelocs, dwarfReloc{
|
||||
off: uint64(addrOff),
|
||||
name: fn.Name,
|
||||
addend: 0,
|
||||
})
|
||||
|
||||
// Build the line entries.
|
||||
pts := make([]LineEntry, 0, len(fn.Lines)+1)
|
||||
if len(fn.Lines) == 0 || fn.Lines[0].Offset > 0 {
|
||||
pts = append(pts, LineEntry{Offset: 0, Line: fn.Line})
|
||||
}
|
||||
pts = append(pts, fn.Lines...)
|
||||
|
||||
line := int64(1)
|
||||
pc := uint64(0)
|
||||
for _, p := range pts {
|
||||
if p.Line == 0 || uint64(p.Offset) < pc {
|
||||
continue
|
||||
}
|
||||
if int64(p.Line) == line {
|
||||
continue
|
||||
}
|
||||
deltaPC := uint64(p.Offset) - pc
|
||||
deltaLC := int64(p.Line) - line
|
||||
b = dwPutPCLCDelta(b, deltaPC, deltaLC)
|
||||
line, pc = int64(p.Line), uint64(p.Offset)
|
||||
}
|
||||
|
||||
// Advance to end of function.
|
||||
if end := uint64(fn.Size) - pc; end > 0 {
|
||||
b = append(b, 2) // DW_LNS_advance_pc
|
||||
b = appendUleb(b, end)
|
||||
}
|
||||
b = append(b, 0, 1, 1) // LNE_end_sequence
|
||||
}
|
||||
|
||||
// Patch unit_length.
|
||||
le.PutUint32(b[headerStart:], uint32(len(b)-headerStart-4))
|
||||
// Patch header_length. In the v5 header it follows the one-byte
|
||||
// address_size and segment_selector_size (offset 8, not the DWARF2-4
|
||||
// offset 6), and counts from just past itself to the first program
|
||||
// byte.
|
||||
le.PutUint32(b[headerStart+8:], uint32(headerEnd-headerStart-12))
|
||||
return b
|
||||
}
|
||||
|
||||
// dwarfBuildInfoSection builds a complete .debug_info section.
|
||||
func dwarfBuildInfoSection(img *Image, srcFile string, ds *dwarfSections) []byte {
|
||||
var b []byte
|
||||
le := binary.LittleEndian
|
||||
|
||||
cuStart := len(b)
|
||||
b = append(b, 0, 0, 0, 0) // unit_length (placeholder)
|
||||
b = le.AppendUint16(b, 5) // version (DWARF5)
|
||||
b = append(b, 0x01) // unit_type (DW_UT_compile)
|
||||
b = append(b, 8) // address_size
|
||||
b = le.AppendUint32(b, 0) // debug_abbrev_offset (0 since single CU)
|
||||
|
||||
// DW_TAG_compile_unit (abbrev 1).
|
||||
b = append(b, 1) // abbreviation code
|
||||
// DW_AT_low_pc: address of .text start. A data-only image has no
|
||||
// functions to relocate against; its CU covers no code, so the base
|
||||
// stays zero (the DWARF "no base address" value) with no relocation.
|
||||
b = le.AppendUint64(b, 0) // placeholder
|
||||
if len(img.Funcs) > 0 {
|
||||
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
|
||||
off: uint64(len(b) - 8),
|
||||
name: img.Funcs[0].Name,
|
||||
})
|
||||
}
|
||||
// DW_AT_high_pc: size of .text.
|
||||
b = le.AppendUint64(b, uint64(len(img.Code)))
|
||||
// DW_AT_stmt_list: offset into .debug_line (0).
|
||||
b = le.AppendUint32(b, 0)
|
||||
// DW_AT_name: source file name.
|
||||
b = append(b, srcFile...)
|
||||
b = append(b, 0)
|
||||
|
||||
// DW_TAG_subprogram entries (abbrev 2).
|
||||
for _, fn := range img.Funcs {
|
||||
b = append(b, 2) // abbreviation code
|
||||
// DW_AT_name.
|
||||
b = append(b, fn.Name...)
|
||||
b = append(b, 0)
|
||||
// DW_AT_low_pc.
|
||||
addrOff := len(b)
|
||||
b = le.AppendUint64(b, 0) // placeholder
|
||||
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
|
||||
off: uint64(addrOff),
|
||||
name: fn.Name,
|
||||
addend: 0,
|
||||
})
|
||||
// DW_AT_high_pc: function size.
|
||||
b = le.AppendUint64(b, uint64(fn.Size))
|
||||
// DW_AT_frame_base: DW_OP_call_frame_cfa.
|
||||
b = append(b, 1, 0x9c)
|
||||
// DW_AT_decl_file: the single file-table entry, index 0 (v5 indexes
|
||||
// files from 0).
|
||||
b = append(b, 0)
|
||||
// DW_AT_decl_line.
|
||||
b = append(b, uint8(fn.Line))
|
||||
// DW_AT_external.
|
||||
if fn.Static {
|
||||
b = append(b, 0)
|
||||
} else {
|
||||
b = append(b, 1)
|
||||
}
|
||||
}
|
||||
|
||||
// End of compile unit children.
|
||||
b = append(b, 0)
|
||||
|
||||
// Patch unit_length.
|
||||
le.PutUint32(b[cuStart:], uint32(len(b)-cuStart-4))
|
||||
return b
|
||||
}
|
||||
|
||||
func appendUleb(b []byte, v uint64) []byte {
|
||||
return binary.AppendUvarint(b, v)
|
||||
}
|
||||
|
||||
// appendSleb appends v in signed LEB128, the encoding DWARF specifies:
|
||||
// two's-complement sign extension, which is NOT Go's zigzag varint
|
||||
// (binary.AppendVarint(-8) encodes 15, where DWARF wants 0x78).
|
||||
func appendSleb(b []byte, v int64) []byte {
|
||||
for {
|
||||
c := byte(v & 0x7f)
|
||||
v >>= 7
|
||||
if (v == 0 && c&0x40 == 0) || (v == -1 && c&0x40 != 0) {
|
||||
return append(b, c)
|
||||
}
|
||||
b = append(b, c|0x80)
|
||||
}
|
||||
}
|
||||
|
||||
// cfiArch carries the .debug_frame CIE parameters that differ per
|
||||
// architecture: the DWARF register numbers of the stack pointer the initial
|
||||
// CFA rule names and of the return address. The values are the ones the Go
|
||||
// linker writes into its own CIE (cmd/link/internal/ld/dwarf.go uses
|
||||
// Dwarfregsp and Dwarfreglr; the per-architecture constants live in
|
||||
// cmd/link/internal/<arch>/l.go).
|
||||
type cfiArch struct {
|
||||
name string
|
||||
cfaReg byte // the stack-pointer register the initial CFA rule names
|
||||
raReg byte // the return-address register
|
||||
}
|
||||
|
||||
var (
|
||||
cfiAMD64 = cfiArch{"amd64", 7, 16} // RSP, RIP
|
||||
cfiARM64 = cfiArch{"arm64", 31, 30} // SP (X31), LR (X30)
|
||||
cfiRISCV64 = cfiArch{"riscv64", 2, 1} // X2 (sp), X1 (ra)
|
||||
cfiLOONG64 = cfiArch{"loong64", 3, 1} // $r3 (sp), $r1 (ra)
|
||||
)
|
||||
|
||||
// dwarfBuildFrameSection builds a .debug_frame section with CFI for stack
|
||||
// unwinding. It emits one CIE and one FDE per function, encoding the
|
||||
// CFA (Canonical Frame Address) rule changes at each stack-adjustment
|
||||
// boundary recorded in FuncLayout.Spadj.
|
||||
func dwarfBuildFrameSection(img *Image, cfi cfiArch, ds *dwarfSections) []byte {
|
||||
var b []byte
|
||||
le := binary.LittleEndian
|
||||
|
||||
// CIE (Common Information Entry).
|
||||
cieStart := len(b)
|
||||
b = append(b, 0, 0, 0, 0) // length (placeholder)
|
||||
b = le.AppendUint32(b, 0xFFFFFFFF) // CIE marker
|
||||
b = append(b, 3) // version (DWARF3, widely supported)
|
||||
b = append(b, 0) // augmentation (empty)
|
||||
b = appendUleb(b, 1) // code alignment
|
||||
b = appendSleb(b, -8) // data alignment (-8 for 64-bit)
|
||||
b = appendUleb(b, uint64(cfi.raReg)) // return address register
|
||||
// Initial CFA rule: DW_CFA_def_cfa (SP, 0)
|
||||
b = append(b, 0x0c) // DW_CFA_def_cfa
|
||||
b = appendUleb(b, uint64(cfi.cfaReg)) // the architecture's stack pointer
|
||||
b = appendUleb(b, 0) // offset: 0
|
||||
b = append(b, 0) // DW_CFA_nop (padding)
|
||||
// Patch CIE length.
|
||||
le.PutUint32(b[cieStart:], uint32(len(b)-cieStart-4))
|
||||
|
||||
// FDEs (Frame Description Entries), one per function.
|
||||
for _, fn := range img.Funcs {
|
||||
fdeStart := len(b)
|
||||
b = append(b, 0, 0, 0, 0) // length (placeholder)
|
||||
b = le.AppendUint32(b, uint32(cieStart)) // CIE pointer (offset from start)
|
||||
// Initial location: function offset in .text, referenced through
|
||||
// the function's symbol so the linker relocates it.
|
||||
ds.frameRelocs = append(ds.frameRelocs, dwarfReloc{
|
||||
off: uint64(fdeStart + 8),
|
||||
name: fn.Name,
|
||||
})
|
||||
b = le.AppendUint64(b, uint64(fn.Offset))
|
||||
// Address range: function size.
|
||||
b = le.AppendUint64(b, uint64(fn.Size))
|
||||
|
||||
// Emit CFA rule changes at each Spadj boundary.
|
||||
for _, step := range fn.Spadj {
|
||||
if step.Value == 0 {
|
||||
continue
|
||||
}
|
||||
// DW_CFA_def_cfa_offset: set CFA = SP + |delta|.
|
||||
// The delta is negative (stack grows down), so CFA offset = -delta.
|
||||
offset := -step.Value
|
||||
if offset > 0 {
|
||||
b = append(b, 0x0e) // DW_CFA_def_cfa_offset
|
||||
b = appendUleb(b, uint64(offset))
|
||||
}
|
||||
}
|
||||
|
||||
// Pad to alignment.
|
||||
for len(b)%4 != 0 {
|
||||
b = append(b, 0) // DW_CFA_nop
|
||||
}
|
||||
|
||||
// Patch FDE length.
|
||||
le.PutUint32(b[fdeStart:], uint32(len(b)-fdeStart-4))
|
||||
}
|
||||
|
||||
return b
|
||||
}
|
||||
@@ -0,0 +1,179 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import "encoding/binary"
|
||||
|
||||
// Absolute 64-bit relocation types for the DWARF address fixups, one per
|
||||
// supported architecture (the numbers debug/elf carries).
|
||||
const (
|
||||
rX8664Abs64 = 1 // R_X86_64_64
|
||||
rAARCH64Abs64 = 257 // R_AARCH64_ABS64
|
||||
rRISCVAbs64 = 2 // R_RISCV_64
|
||||
rLarchAbs64 = 2 // R_LARCH_64
|
||||
)
|
||||
|
||||
// dwarfELFSections holds the laid-out DWARF sections ready for inclusion
|
||||
// in an ELF file.
|
||||
type dwarfELFSections struct {
|
||||
abbrevOff, abbrevSize int
|
||||
infoOff, infoSize int
|
||||
lineOff, lineSize int
|
||||
lineStrOff, lineStrSize int
|
||||
frameOff, frameSize int
|
||||
// .rela.debug_info and .rela.debug_line contents: file offsets and
|
||||
// entry counts (zero count: the section is absent).
|
||||
infoRelaOff, infoRelaCount int
|
||||
lineRelaOff, lineRelaCount int
|
||||
frameRelaOff, frameRelaCount int
|
||||
// Relocations for .debug_info address references, offsets relative to
|
||||
// the section start (what an r_offset in .rela.debug_info means).
|
||||
infoRelocs []elfDwarfReloc
|
||||
// Relocations for .debug_line address references, section-relative.
|
||||
lineRelocs []elfDwarfReloc
|
||||
// Relocations for .debug_frame FDE initial locations, section-relative.
|
||||
frameRelocs []elfDwarfReloc
|
||||
}
|
||||
|
||||
type elfDwarfReloc struct {
|
||||
off uint64 // offset within the target section
|
||||
sym int // symbol index in .symtab
|
||||
addend int64
|
||||
}
|
||||
|
||||
// appendDWARFSections generates and appends DWARF5 debug sections to the ELF
|
||||
// output. It returns the section offsets/sizes and relocations for the caller
|
||||
// to emit section headers and relocation records.
|
||||
//
|
||||
// symIdx maps function names to their .symtab indices (needed for relocations
|
||||
// against .text symbols). The map uses objectName format (pkg.name); the
|
||||
// DWARF code uses bare function names, so we build a reverse lookup. cfi
|
||||
// carries the architecture's .debug_frame register conventions.
|
||||
func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[string]int, align func(int), cfi cfiArch) *dwarfELFSections {
|
||||
// Build a lookup from bare function name to symbol index.
|
||||
nameToIdx := make(map[string]int, len(symIdx))
|
||||
for name, idx := range symIdx {
|
||||
// Strip package prefix: "pkg.name" → "name".
|
||||
if i := len(name) - 1; i >= 0 {
|
||||
for j := len(name) - 1; j >= 0; j-- {
|
||||
if name[j] == '.' {
|
||||
nameToIdx[name[j+1:]] = idx
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
nameToIdx[name] = idx
|
||||
}
|
||||
ds := emitDWARF(img, srcFile, cfi)
|
||||
if ds == nil || len(ds.debugAbbrev) == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
result := &dwarfELFSections{}
|
||||
|
||||
// .debug_abbrev
|
||||
align(1)
|
||||
result.abbrevOff = len(*out)
|
||||
result.abbrevSize = len(ds.debugAbbrev)
|
||||
*out = append(*out, ds.debugAbbrev...)
|
||||
|
||||
// .debug_line_str
|
||||
align(1)
|
||||
result.lineStrOff = len(*out)
|
||||
result.lineStrSize = len(ds.debugLineStr)
|
||||
*out = append(*out, ds.debugLineStr...)
|
||||
|
||||
// .debug_line
|
||||
align(1)
|
||||
result.lineOff = len(*out)
|
||||
result.lineSize = len(ds.debugLine)
|
||||
*out = append(*out, ds.debugLine...)
|
||||
for _, dr := range ds.lineRelocs {
|
||||
if idx, ok := nameToIdx[dr.name]; ok {
|
||||
result.lineRelocs = append(result.lineRelocs, elfDwarfReloc{
|
||||
off: dr.off,
|
||||
sym: idx,
|
||||
addend: dr.addend,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// .debug_info
|
||||
align(1)
|
||||
result.infoOff = len(*out)
|
||||
result.infoSize = len(ds.debugInfo)
|
||||
*out = append(*out, ds.debugInfo...)
|
||||
for _, dr := range ds.infoRelocs {
|
||||
if idx, ok := nameToIdx[dr.name]; ok {
|
||||
result.infoRelocs = append(result.infoRelocs, elfDwarfReloc{
|
||||
off: dr.off,
|
||||
sym: idx,
|
||||
addend: dr.addend,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// .debug_frame: the section header declares alignment 8, so the data is
|
||||
// padded to 8, matching it.
|
||||
if len(ds.debugFrame) > 0 {
|
||||
align(8)
|
||||
result.frameOff = len(*out)
|
||||
result.frameSize = len(ds.debugFrame)
|
||||
*out = append(*out, ds.debugFrame...)
|
||||
for _, dr := range ds.frameRelocs {
|
||||
if idx, ok := nameToIdx[dr.name]; ok {
|
||||
result.frameRelocs = append(result.frameRelocs, elfDwarfReloc{
|
||||
off: dr.off,
|
||||
sym: idx,
|
||||
addend: dr.addend,
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return result
|
||||
}
|
||||
|
||||
// appendDWARFRelas writes the .rela.debug_info and .rela.debug_line section
|
||||
// bodies from the relocations appendDWARFSections recorded, with the
|
||||
// architecture's absolute 64-bit relocation type, and records their file
|
||||
// offsets and entry counts on dw. Called after the DWARF sections
|
||||
// themselves so the r_offsets (section-relative) need no adjustment.
|
||||
func appendDWARFRelas(out *[]byte, dw *dwarfELFSections, abs64 uint32, align func(int)) {
|
||||
le := binary.LittleEndian
|
||||
write := func(relas []elfDwarfReloc) (off, count int) {
|
||||
if len(relas) == 0 {
|
||||
return 0, 0
|
||||
}
|
||||
align(8)
|
||||
off = len(*out)
|
||||
for _, r := range relas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(abs64))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
*out = append(*out, b[:]...)
|
||||
}
|
||||
return off, len(relas)
|
||||
}
|
||||
dw.infoRelaOff, dw.infoRelaCount = write(dw.infoRelocs)
|
||||
dw.lineRelaOff, dw.lineRelaCount = write(dw.lineRelocs)
|
||||
dw.frameRelaOff, dw.frameRelaCount = write(dw.frameRelocs)
|
||||
}
|
||||
|
||||
// dwarfSourceName returns the source name the DWARF sections record: the
|
||||
// image's source path when the assembler captured one, "gasm.s" otherwise.
|
||||
func dwarfSourceName(img *Image) string {
|
||||
if img.SourcePath != "" {
|
||||
return img.SourcePath
|
||||
}
|
||||
return "gasm.s"
|
||||
}
|
||||
|
||||
// dwarfSectionNames returns the DWARF section names for the string table.
|
||||
var dwarfSectionNames = []string{
|
||||
".debug_abbrev", ".debug_info", ".debug_line", ".debug_line_str",
|
||||
".debug_frame", ".rela.debug_info", ".rela.debug_line",
|
||||
".rela.debug_frame",
|
||||
}
|
||||
@@ -0,0 +1,372 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// ulebIter reads ULEB128 values, the .debug_abbrev and line-header
|
||||
// encoding.
|
||||
type ulebIter struct {
|
||||
b []byte
|
||||
i int
|
||||
}
|
||||
|
||||
func (r *ulebIter) uleb(t *testing.T) uint64 {
|
||||
t.Helper()
|
||||
v, n := binary.Uvarint(r.b[r.i:])
|
||||
if n <= 0 {
|
||||
t.Fatalf("bad ULEB at %d", r.i)
|
||||
}
|
||||
r.i += n
|
||||
return v
|
||||
}
|
||||
|
||||
func (r *ulebIter) byteAt(t *testing.T) byte {
|
||||
t.Helper()
|
||||
if r.i >= len(r.b) {
|
||||
t.Fatalf("read past end at %d", r.i)
|
||||
}
|
||||
c := r.b[r.i]
|
||||
r.i++
|
||||
return c
|
||||
}
|
||||
|
||||
func (r *ulebIter) uint32At(t *testing.T) uint32 {
|
||||
t.Helper()
|
||||
v := binary.LittleEndian.Uint32(r.b[r.i:])
|
||||
r.i += 4
|
||||
return v
|
||||
}
|
||||
|
||||
// sleb reads a signed LEB128, the DWARF encoding (sign-extended two's
|
||||
// complement, not Go's zigzag varint).
|
||||
func (r *ulebIter) sleb(t *testing.T) int64 {
|
||||
t.Helper()
|
||||
var v int64
|
||||
var shift uint
|
||||
for {
|
||||
c := r.byteAt(t)
|
||||
v |= int64(c&0x7f) << shift
|
||||
shift += 7
|
||||
if c&0x80 == 0 {
|
||||
if c&0x40 != 0 {
|
||||
v |= -1 << shift
|
||||
}
|
||||
return v
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// dwarfAttr is one attribute/form pair of an abbreviation.
|
||||
type dwarfAttr struct{ attr, form uint64 }
|
||||
|
||||
// dwarfAbbrev is one parsed abbreviation declaration.
|
||||
type dwarfAbbrev struct {
|
||||
code uint64
|
||||
tag uint64
|
||||
children bool
|
||||
attrs []dwarfAttr
|
||||
}
|
||||
|
||||
// parseAbbrevs walks a .debug_abbrev table: abbreviation code, tag,
|
||||
// children flag, then attr/form ULEB pairs terminated by a double zero.
|
||||
func parseAbbrevs(t *testing.T, b []byte) map[uint64]dwarfAbbrev {
|
||||
t.Helper()
|
||||
out := map[uint64]dwarfAbbrev{}
|
||||
r := &ulebIter{b: b}
|
||||
for {
|
||||
code := r.uleb(t)
|
||||
if code == 0 {
|
||||
return out
|
||||
}
|
||||
ab := dwarfAbbrev{code: code, tag: r.uleb(t)}
|
||||
ab.children = r.byteAt(t) == 1
|
||||
for {
|
||||
attr := r.uleb(t)
|
||||
form := r.uleb(t)
|
||||
if attr == 0 && form == 0 {
|
||||
break
|
||||
}
|
||||
if attr == 0 || form == 0 {
|
||||
t.Fatalf("abbrev %d: half-terminated attr/form pair (%d, %d)", code, attr, form)
|
||||
}
|
||||
ab.attrs = append(ab.attrs, dwarfAttr{attr, form})
|
||||
}
|
||||
out[code] = ab
|
||||
}
|
||||
}
|
||||
|
||||
func eqAttrs(t *testing.T, ab dwarfAbbrev, want []dwarfAttr) {
|
||||
t.Helper()
|
||||
if len(ab.attrs) != len(want) {
|
||||
t.Fatalf("abbrev %d attrs = %v, want %v", ab.code, ab.attrs, want)
|
||||
}
|
||||
for i, w := range want {
|
||||
if ab.attrs[i] != w {
|
||||
t.Fatalf("abbrev %d attr %d = (%#x, %#x), want (%#x, %#x)", ab.code, i, ab.attrs[i].attr, ab.attrs[i].form, w.attr, w.form)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestDwarfAbbrevTable walks the abbreviation table as a consumer does and
|
||||
// checks the attribute/form sets against the constants the toolchain uses
|
||||
// (cmd/internal/dwarf/dwarf_defs.go). A wrong constant here renames an
|
||||
// attribute (0x1b is comp_dir, not low_pc; 0x29 and 0x37 are bounds and
|
||||
// count) and a wrong form desynchronises the DIE parse: 0x25 is strx1, one
|
||||
// byte, where the writer emits four for a section offset.
|
||||
func TestDwarfAbbrevTable(t *testing.T) {
|
||||
abbrev := dwarfAbbrevTable()
|
||||
if len(abbrev) == 0 {
|
||||
t.Fatal("empty abbrev table")
|
||||
}
|
||||
// Must end with a zero byte (end of table).
|
||||
if abbrev[len(abbrev)-1] != 0 {
|
||||
t.Fatalf("abbrev table last byte = %d, want 0", abbrev[len(abbrev)-1])
|
||||
}
|
||||
abs := parseAbbrevs(t, abbrev)
|
||||
if len(abs) != 2 {
|
||||
t.Fatalf("abbreviations = %d, want 2", len(abs))
|
||||
}
|
||||
cu, ok := abs[1]
|
||||
if !ok {
|
||||
t.Fatal("missing abbreviation 1 (compile unit)")
|
||||
}
|
||||
if cu.tag != dwTagCompUnit || !cu.children {
|
||||
t.Errorf("abbrev 1: tag %#x children %v, want compile unit with children", cu.tag, cu.children)
|
||||
}
|
||||
eqAttrs(t, cu, []dwarfAttr{
|
||||
{dwAtLowPC, dwFormAddr},
|
||||
{dwAtHighPC, dwFormData8},
|
||||
{dwAtStmtList, dwFormSecOff},
|
||||
{dwAtName, dwFormString},
|
||||
})
|
||||
sp, ok := abs[2]
|
||||
if !ok {
|
||||
t.Fatal("missing abbreviation 2 (subprogram)")
|
||||
}
|
||||
if sp.tag != dwTagSubprog || sp.children {
|
||||
t.Errorf("abbrev 2: tag %#x children %v, want subprogram without children", sp.tag, sp.children)
|
||||
}
|
||||
eqAttrs(t, sp, []dwarfAttr{
|
||||
{dwAtName, dwFormString},
|
||||
{dwAtLowPC, dwFormAddr},
|
||||
{dwAtHighPC, dwFormData8},
|
||||
{dwAtFrameBase, dwFormExprloc},
|
||||
{dwAtDeclFile, dwFormData1},
|
||||
{dwAtDeclLine, dwFormData1},
|
||||
{dwAtExternal, 0x0c}, // DW_FORM_flag
|
||||
})
|
||||
}
|
||||
|
||||
// TestDwarfLineHeaderV5 parses the .debug_line header under DWARF5 rules:
|
||||
// the directory and file tables are format-descriptor lists, not the
|
||||
// DWARF2-4 shape of null-terminated strings, and the file entry references
|
||||
// the source name through .debug_line_str.
|
||||
func TestDwarfLineHeaderV5(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVQ a+0(FP), AX
|
||||
MOVQ b+8(FP), BX
|
||||
ADDQ BX, AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_amd64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
|
||||
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
|
||||
r := &ulebIter{b: ds.debugLine}
|
||||
r.uint32At(t) // unit_length
|
||||
if v := binary.LittleEndian.Uint16(ds.debugLine[4:]); v != 5 {
|
||||
t.Fatalf("version = %d, want 5", v)
|
||||
}
|
||||
r.i = 6
|
||||
r.byteAt(t) // address_size
|
||||
r.byteAt(t) // segment_selector_size
|
||||
r.uint32At(t) // header_length
|
||||
r.byteAt(t) // minimum_instruction_length
|
||||
r.byteAt(t) // maximum_ops_per_instruction
|
||||
r.byteAt(t) // default_is_stmt
|
||||
r.byteAt(t) // line_base
|
||||
r.byteAt(t) // line_range
|
||||
opcodeBase := r.byteAt(t)
|
||||
for range int(opcodeBase) - 1 {
|
||||
r.byteAt(t) // standard opcode lengths
|
||||
}
|
||||
|
||||
// Directory table (DWARF5 §6.2.4).
|
||||
if n := r.byteAt(t); n != 1 {
|
||||
t.Fatalf("directory_entry_format_count = %d, want 1", n)
|
||||
}
|
||||
if lnct := r.uleb(t); lnct != dwLnctPath {
|
||||
t.Errorf("directory content type = %#x, want DW_LNCT_path", lnct)
|
||||
}
|
||||
if form := r.uleb(t); form != dwFormLineStrp {
|
||||
t.Errorf("directory form = %#x, want DW_FORM_line_strp", form)
|
||||
}
|
||||
if n := r.uleb(t); n != 1 {
|
||||
t.Fatalf("directories_count = %d, want 1", n)
|
||||
}
|
||||
if off := r.uint32At(t); off != 0 {
|
||||
t.Errorf("compilation directory line_strp = %d, want 0 (the empty string)", off)
|
||||
}
|
||||
|
||||
// File table (DWARF5 §6.2.5).
|
||||
if n := r.byteAt(t); n != 2 {
|
||||
t.Fatalf("file_name_entry_format_count = %d, want 2", n)
|
||||
}
|
||||
if lnct := r.uleb(t); lnct != dwLnctPath {
|
||||
t.Errorf("file content type = %#x, want DW_LNCT_path", lnct)
|
||||
}
|
||||
if form := r.uleb(t); form != dwFormLineStrp {
|
||||
t.Errorf("file path form = %#x, want DW_FORM_line_strp", form)
|
||||
}
|
||||
if lnct := r.uleb(t); lnct != dwLnctDirIndex {
|
||||
t.Errorf("file content type = %#x, want DW_LNCT_directory_index", lnct)
|
||||
}
|
||||
if form := r.uleb(t); form != dwFormUdata {
|
||||
t.Errorf("file dir-index form = %#x, want DW_FORM_udata", form)
|
||||
}
|
||||
if n := r.uleb(t); n != 1 {
|
||||
t.Fatalf("file_names_count = %d, want 1", n)
|
||||
}
|
||||
strOff := r.uint32At(t)
|
||||
if dirIdx := r.uleb(t); dirIdx != 0 {
|
||||
t.Errorf("file directory index = %d, want 0", dirIdx)
|
||||
}
|
||||
|
||||
// The file entry's line_strp must resolve to the source name.
|
||||
end := int(strOff) + len("test_amd64.s")
|
||||
if int(strOff) >= len(ds.debugLineStr) || !bytes.Equal(ds.debugLineStr[strOff:end], []byte("test_amd64.s")) {
|
||||
t.Errorf("file entry line_strp %d does not name the source: %q", strOff, ds.debugLineStr)
|
||||
}
|
||||
|
||||
// The fixed header fields: address_size 8 and a header_length that
|
||||
// points just past the file table (the patch site is offset 8 in the
|
||||
// v5 header, and the field counts from its own end).
|
||||
if ds.debugLine[6] != 8 || ds.debugLine[7] != 0 {
|
||||
t.Errorf("address_size/segment_selector = %d/%d, want 8/0", ds.debugLine[6], ds.debugLine[7])
|
||||
}
|
||||
if hl := binary.LittleEndian.Uint32(ds.debugLine[8:]); hl != uint32(r.i-12) {
|
||||
t.Errorf("header_length = %d, want %d (the byte after the file table is %d)", hl, r.i-12, r.i)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDwarfFrameCIEArch checks the shared CIE carries each architecture's
|
||||
// stack-pointer and return-address registers: the values the Go linker
|
||||
// writes (cmd/link/internal/<arch>/l.go dwarfRegSP/dwarfRegLR).
|
||||
func TestDwarfFrameCIEArch(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
cfi cfiArch
|
||||
}{
|
||||
{"amd64", cfiAMD64},
|
||||
{"arm64", cfiARM64},
|
||||
{"riscv64", cfiRISCV64},
|
||||
{"loong64", cfiLOONG64},
|
||||
} {
|
||||
frame := dwarfBuildFrameSection(&Image{}, tc.cfi, &dwarfSections{})
|
||||
r := &ulebIter{b: frame}
|
||||
r.uint32At(t) // length
|
||||
if cid := r.uint32At(t); cid != 0xFFFFFFFF {
|
||||
t.Errorf("%s: CIE id = %#x, want 0xffffffff", tc.name, cid)
|
||||
}
|
||||
if v := r.byteAt(t); v != 3 {
|
||||
t.Errorf("%s: CIE version = %d, want 3", tc.name, v)
|
||||
}
|
||||
if aug := r.byteAt(t); aug != 0 {
|
||||
t.Errorf("%s: CIE augmentation = %d, want 0", tc.name, aug)
|
||||
}
|
||||
if ca := r.uleb(t); ca != 1 {
|
||||
t.Errorf("%s: code alignment = %d, want 1", tc.name, ca)
|
||||
}
|
||||
if da := r.sleb(t); da != -8 {
|
||||
t.Errorf("%s: data alignment = %d, want -8 (signed LEB128, not zigzag)", tc.name, da)
|
||||
}
|
||||
if ra := r.uleb(t); ra != uint64(tc.cfi.raReg) {
|
||||
t.Errorf("%s: return-address register = %d, want %d", tc.name, ra, tc.cfi.raReg)
|
||||
}
|
||||
if op := r.byteAt(t); op != 0x0c {
|
||||
t.Errorf("%s: expected DW_CFA_def_cfa, got opcode %#x", tc.name, op)
|
||||
}
|
||||
if cfa := r.uleb(t); cfa != uint64(tc.cfi.cfaReg) {
|
||||
t.Errorf("%s: CFA register = %d, want %d", tc.name, cfa, tc.cfi.cfaReg)
|
||||
}
|
||||
if off := r.uleb(t); off != 0 {
|
||||
t.Errorf("%s: CFA offset = %d, want 0", tc.name, off)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEmitDWARF(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVQ a+0(FP), AX
|
||||
MOVQ b+8(FP), BX
|
||||
ADDQ BX, AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_amd64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
|
||||
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
|
||||
|
||||
// .debug_abbrev must not be empty and must start with abbrev code 1.
|
||||
if len(ds.debugAbbrev) == 0 {
|
||||
t.Fatal("empty .debug_abbrev")
|
||||
}
|
||||
if ds.debugAbbrev[0] != 1 {
|
||||
t.Fatalf(".debug_abbrev first byte = %d, want 1", ds.debugAbbrev[0])
|
||||
}
|
||||
|
||||
// .debug_info must have a compile unit header (DWARF5 version 5).
|
||||
if len(ds.debugInfo) < 12 {
|
||||
t.Fatalf(".debug_info too short: %d bytes", len(ds.debugInfo))
|
||||
}
|
||||
// Version field at offset 4 (after unit_length).
|
||||
if ds.debugInfo[4] != 5 || ds.debugInfo[5] != 0 {
|
||||
t.Fatalf(".debug_info version = %d, want 5", uint16(ds.debugInfo[4])|uint16(ds.debugInfo[5])<<8)
|
||||
}
|
||||
|
||||
// .debug_line must have a header.
|
||||
if len(ds.debugLine) < 20 {
|
||||
t.Fatalf(".debug_line too short: %d bytes", len(ds.debugLine))
|
||||
}
|
||||
// Version at offset 4.
|
||||
if ds.debugLine[4] != 5 || ds.debugLine[5] != 0 {
|
||||
t.Fatalf(".debug_line version = %d, want 5", uint16(ds.debugLine[4])|uint16(ds.debugLine[5])<<8)
|
||||
}
|
||||
|
||||
// .debug_line_str must contain the source file name.
|
||||
if len(ds.debugLineStr) == 0 {
|
||||
t.Fatal("empty .debug_line_str")
|
||||
}
|
||||
|
||||
// Relocations must reference the function.
|
||||
if len(ds.lineRelocs) == 0 {
|
||||
t.Fatal("no .debug_line relocations")
|
||||
}
|
||||
if len(ds.infoRelocs) == 0 {
|
||||
t.Fatal("no .debug_info relocations")
|
||||
}
|
||||
}
|
||||
+553
-3
@@ -12,7 +12,8 @@ import (
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// The object-file tests share one source: two exported functions, one
|
||||
@@ -52,7 +53,7 @@ func elfTestImage(t *testing.T) *Image {
|
||||
}
|
||||
|
||||
// TestAssembleFileExternals checks that a reference to a symbol no GLOBL
|
||||
// defines is recorded as an external relocation instead of failing — the
|
||||
// defines is recorded as an external relocation instead of failing; the
|
||||
// raw image leaves the displacement zero, the object emitters carry it.
|
||||
func TestAssembleFileExternals(t *testing.T) {
|
||||
img := elfTestImage(t)
|
||||
@@ -187,7 +188,7 @@ func TestELFObject(t *testing.T) {
|
||||
end := bytes.IndexByte(strtabRaw[stName:], 0)
|
||||
return string(strtabRaw[stName : int(stName)+end])
|
||||
}
|
||||
for i := 0; i < 2; i++ {
|
||||
for i := range 2 {
|
||||
e := raw[i*24 : (i+1)*24]
|
||||
off := binary.LittleEndian.Uint64(e[0:])
|
||||
info := binary.LittleEndian.Uint64(e[8:])
|
||||
@@ -211,6 +212,75 @@ func TestELFObject(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFObjectTLSGuardReloc checks that a non-NOSPLIT function's stack
|
||||
// guard carries an R_X86_64_TPOFF32 relocation against the null symbol in
|
||||
// .rela.text. The serialisation must honour the record's type field: a
|
||||
// hardcoded R_X86_64_PC32 mislinks the TLS load as an ordinary
|
||||
// PC-relative reference.
|
||||
func TestELFObjectTLSGuardReloc(t *testing.T) {
|
||||
f, errs := parser.Parse("g_amd64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·grow(SB), $0
|
||||
CALL ·other(SB)
|
||||
RET
|
||||
|
||||
TEXT ·other(SB), NOSPLIT, $0
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
var haveTLS bool
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
if r.Kind == RelTLSLE {
|
||||
haveTLS = true
|
||||
}
|
||||
}
|
||||
}
|
||||
if !haveTLS {
|
||||
t.Fatal("test source produced no RelTLSLE relocation")
|
||||
}
|
||||
obj, err := img.ELFObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFObject: %v", err)
|
||||
}
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
relaSec := ef.Section(".rela.text")
|
||||
if relaSec == nil {
|
||||
t.Fatal("missing .rela.text")
|
||||
}
|
||||
raw, err := relaSec.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
found := false
|
||||
for i := 0; i+24 <= len(raw); i += 24 {
|
||||
e := raw[i:]
|
||||
info := binary.LittleEndian.Uint64(e[8:])
|
||||
typ := info & 0xffffffff
|
||||
sym := int(info >> 32)
|
||||
if typ == uint64(elf.R_X86_64_TPOFF32) {
|
||||
found = true
|
||||
if sym != 0 {
|
||||
t.Errorf("TPOFF32 relocation against symbol %d, want 0 (the null symbol)", sym)
|
||||
}
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Errorf("no R_X86_64_TPOFF32 relocation in .rela.text (%d bytes)", len(raw))
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFObjectNoRelocations checks a file with no static-symbol references
|
||||
// emits a valid object without a .rela.text section.
|
||||
func TestELFObjectNoRelocations(t *testing.T) {
|
||||
@@ -253,6 +323,238 @@ TEXT ·nop(SB), NOSPLIT, $0
|
||||
}
|
||||
}
|
||||
|
||||
// elfSectionHeaderCount returns the e_shnum the ELF header declares.
|
||||
func elfSectionHeaderCount(t *testing.T, obj []byte) int {
|
||||
t.Helper()
|
||||
return int(binary.LittleEndian.Uint16(obj[60:]))
|
||||
}
|
||||
|
||||
// checkELFSectionAccounting verifies the number of section headers the
|
||||
// writer physically laid out equals e_shnum: every DWARF section written
|
||||
// after .shstrtab must be counted, or the last ones (always .debug_frame)
|
||||
// are invisible to every consumer, debug/elf included.
|
||||
func checkELFSectionAccounting(t *testing.T, obj []byte) {
|
||||
t.Helper()
|
||||
shoff := int(binary.LittleEndian.Uint64(obj[40:]))
|
||||
shentsize := int(binary.LittleEndian.Uint16(obj[58:]))
|
||||
shnum := elfSectionHeaderCount(t, obj)
|
||||
if shentsize != 64 {
|
||||
t.Fatalf("e_shentsize = %d, want 64", shentsize)
|
||||
}
|
||||
if (len(obj)-shoff)%shentsize != 0 {
|
||||
t.Fatalf("section header table is not a whole number of entries: shoff=%d len=%d", shoff, len(obj))
|
||||
}
|
||||
if present := (len(obj) - shoff) / shentsize; present != shnum {
|
||||
t.Errorf("e_shnum = %d but %d section headers are laid out", shnum, present)
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFDWARFSectionAccounting runs the header accounting check over all
|
||||
// four architecture emitters, and additionally checks the .debug_frame
|
||||
// section is visible (its data aligned as its header declares).
|
||||
func TestELFDWARFSectionAccounting(t *testing.T) {
|
||||
parse := func(name, src string) *ast.File {
|
||||
f, errs := parser.Parse(name, src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse %s: %v", name, errs)
|
||||
}
|
||||
return f
|
||||
}
|
||||
cases := []struct {
|
||||
name string
|
||||
img *Image
|
||||
emit func(*Image) ([]byte, error)
|
||||
}{
|
||||
{"amd64", elfTestImage(t), (*Image).ELFObject},
|
||||
{"arm64", mustImage(t, func() (*Image, error) {
|
||||
return AssembleFileARM64(parse("k_arm64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVD a+0(FP), R4
|
||||
MOVD b+8(FP), R5
|
||||
ADD R5, R4, R4
|
||||
MOVD R4, ret+16(FP)
|
||||
RET
|
||||
`))
|
||||
}), (*Image).ELFAARCH64Object},
|
||||
{"riscv64", mustImage(t, func() (*Image, error) {
|
||||
return AssembleFileRISCV(parse("k_riscv64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·sb(SB), NOSPLIT, $0-0
|
||||
MOV $answer<>(SB), X10
|
||||
RET
|
||||
|
||||
GLOBL answer<>(SB), RODATA, $8
|
||||
DATA answer<>+0(SB)/8, $42
|
||||
`))
|
||||
}), (*Image).ELFRISCVObject},
|
||||
{"loong64", mustImage(t, func() (*Image, error) {
|
||||
return AssembleFileLOONG64(parse("k_loong64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVV a+0(FP), R4
|
||||
MOVV b+8(FP), R5
|
||||
ADDV R5, R4, R4
|
||||
MOVV R4, ret+16(FP)
|
||||
RET
|
||||
`))
|
||||
}), (*Image).ELFLOONG64Object},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
obj, err := tc.emit(tc.img)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: emit: %v", tc.name, err)
|
||||
}
|
||||
checkELFSectionAccounting(t, obj)
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("%s: parse emitted object: %v", tc.name, err)
|
||||
}
|
||||
frame := ef.Section(".debug_frame")
|
||||
if frame == nil {
|
||||
t.Errorf("%s: .debug_frame invisible to debug/elf (e_shnum too small?)", tc.name)
|
||||
ef.Close()
|
||||
continue
|
||||
}
|
||||
if frame.Offset%8 != 0 || frame.Addralign != 8 {
|
||||
t.Errorf("%s: .debug_frame offset %d align %d, want offset%%8==0 align 8", tc.name, frame.Offset, frame.Addralign)
|
||||
}
|
||||
ef.Close()
|
||||
}
|
||||
}
|
||||
|
||||
func mustImage(t *testing.T, f func() (*Image, error)) *Image {
|
||||
t.Helper()
|
||||
img, err := f()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return img
|
||||
}
|
||||
|
||||
// TestELFDWARFRelocations checks the .rela.debug_info and .rela.debug_line
|
||||
// sections exist and carry absolute 64-bit relocations against the
|
||||
// function symbols, with r_offsets inside their target sections.
|
||||
func TestELFDWARFRelocations(t *testing.T) {
|
||||
img := elfTestImage(t)
|
||||
obj, err := img.ELFObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFObject: %v", err)
|
||||
}
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
// The DWARF must record the assembled file's path (threaded through
|
||||
// Image.SourcePath), not a placeholder name.
|
||||
info, err := ef.Section(".debug_info").Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if img.SourcePath != "t_amd64.s" || !bytes.Contains(info, []byte(img.SourcePath)) {
|
||||
t.Errorf("DWARF compilation unit does not name the source %q", img.SourcePath)
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
rela string
|
||||
target string
|
||||
want uint32
|
||||
}{
|
||||
{".rela.debug_info", ".debug_info", rX8664Abs64},
|
||||
{".rela.debug_line", ".debug_line", rX8664Abs64},
|
||||
{".rela.debug_frame", ".debug_frame", rX8664Abs64},
|
||||
} {
|
||||
rs := ef.Section(tc.rela)
|
||||
if rs == nil {
|
||||
t.Fatalf("missing %s", tc.rela)
|
||||
}
|
||||
if rs.Type != elf.SHT_RELA {
|
||||
t.Errorf("%s: type %v, want SHT_RELA", tc.rela, rs.Type)
|
||||
}
|
||||
target := ef.Section(tc.target)
|
||||
if target == nil {
|
||||
t.Fatalf("missing %s", tc.target)
|
||||
}
|
||||
if rs.Link == 0 || ef.Sections[rs.Info] != target {
|
||||
t.Errorf("%s: link %d info %d, want the symtab and %s", tc.rela, rs.Link, rs.Info, tc.target)
|
||||
}
|
||||
b, err := rs.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// .debug_line has one address per function; .debug_info adds the
|
||||
// compile unit's own low_pc.
|
||||
want := len(img.Funcs)
|
||||
if tc.target == ".debug_info" {
|
||||
want++
|
||||
}
|
||||
if len(b)/24 != want {
|
||||
t.Errorf("%s: %d entries, want %d", tc.rela, len(b)/24, want)
|
||||
}
|
||||
for i := 0; i+24 <= len(b); i += 24 {
|
||||
r_offset := binary.LittleEndian.Uint64(b[i:])
|
||||
info := binary.LittleEndian.Uint64(b[i+8:])
|
||||
typ := uint32(info)
|
||||
sym := int(info >> 32)
|
||||
if typ != tc.want {
|
||||
t.Errorf("%s entry %d: type %d, want R_X86_64_64 (%d)", tc.rela, i/24, typ, tc.want)
|
||||
}
|
||||
if r_offset >= uint64(target.Size) {
|
||||
t.Errorf("%s entry %d: r_offset %d outside %s (%d bytes)", tc.rela, i/24, r_offset, tc.target, target.Size)
|
||||
}
|
||||
if sym == 0 {
|
||||
t.Errorf("%s entry %d: against the null symbol", tc.rela, i/24)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFDataOnly checks a source with GLOBL data and no TEXT emits a valid
|
||||
// ELF object: the DWARF compilation unit of a code-less image has no
|
||||
// function to relocate against and must not reach for one.
|
||||
func TestELFDataOnly(t *testing.T) {
|
||||
f, errs := parser.Parse("d0_amd64.s", `
|
||||
GLOBL table<>(SB), RODATA, $8
|
||||
DATA table<>+0(SB)/8, $12345
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
obj, err := img.ELFObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFObject: %v", err)
|
||||
}
|
||||
checkELFSectionAccounting(t, obj)
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
found := false
|
||||
for _, s := range syms {
|
||||
if s.Name == "table" && s.Size == 8 {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Errorf("data symbol table missing: %v", syms)
|
||||
}
|
||||
if ef.Section(".rela.debug_info") != nil || ef.Section(".rela.debug_line") != nil {
|
||||
t.Error("data-only image must not emit DWARF address relocations")
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFLinkAndRun is the end-to-end check: assemble the test functions,
|
||||
// link the emitted object with a C driver that defines the external symbol,
|
||||
// and run the result. Skipped when no C compiler is available.
|
||||
@@ -307,4 +609,252 @@ int main(void) {
|
||||
if got := string(run); got != "42 42 7\n" {
|
||||
t.Errorf("output %q, want \"42 42 7\\n\"", got)
|
||||
}
|
||||
|
||||
// The DWARF addresses must have resolved at link time: the .debug_info
|
||||
// placeholders were carried by .rela.debug_info, so every subprogram's
|
||||
// low_pc must now equal its linked symbol address.
|
||||
bin, err := os.ReadFile(appPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
lef, err := elf.NewFile(bytes.NewReader(bin))
|
||||
if err != nil {
|
||||
t.Fatalf("parse linked binary: %v", err)
|
||||
}
|
||||
defer lef.Close()
|
||||
syms, err := lef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
addrByName := map[string]uint64{}
|
||||
for _, s := range syms {
|
||||
if elf.ST_TYPE(s.Info) == elf.STT_FUNC && s.Value != 0 {
|
||||
addrByName[s.Name] = s.Value
|
||||
}
|
||||
}
|
||||
lowPCs := dwarfSubprogramLowPCs(t, lef)
|
||||
if len(lowPCs) == 0 {
|
||||
t.Fatal("no subprogram DW_AT_low_pc parsed from the linked binary")
|
||||
}
|
||||
for name, pc := range lowPCs {
|
||||
addr, ok := addrByName[name]
|
||||
if !ok {
|
||||
t.Errorf("subprogram %q not in the linked symbol table", name)
|
||||
continue
|
||||
}
|
||||
if pc != addr {
|
||||
t.Errorf("subprogram %q: DW_AT_low_pc = %#x, linked address %#x (DWARF relocation unresolved)", name, pc, addr)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// dwarfSubprogramLowPCs walks the linked binary's .debug_info with its own
|
||||
// .debug_abbrev and returns each DW_TAG_subprogram's DW_AT_low_pc by name.
|
||||
func dwarfSubprogramLowPCs(t *testing.T, ef *elf.File) map[string]uint64 {
|
||||
t.Helper()
|
||||
abbrevSec := ef.Section(".debug_abbrev")
|
||||
infoSec := ef.Section(".debug_info")
|
||||
if abbrevSec == nil || infoSec == nil {
|
||||
t.Fatal("linked binary lacks .debug_abbrev or .debug_info")
|
||||
}
|
||||
abbrev, err := abbrevSec.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
info, err := infoSec.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
abs := parseAbbrevs(t, abbrev)
|
||||
le := binary.LittleEndian
|
||||
out := map[string]uint64{}
|
||||
r := &ulebIter{b: info}
|
||||
r.uint32At(t) // unit_length
|
||||
if v := le.Uint16(info[4:]); v != 5 {
|
||||
t.Fatalf(".debug_info version %d, want 5", v)
|
||||
}
|
||||
r.i = 6
|
||||
r.byteAt(t) // unit_type
|
||||
r.byteAt(t) // address_size
|
||||
r.uint32At(t) // debug_abbrev_offset
|
||||
var name string
|
||||
var lowPC uint64
|
||||
for r.i < len(r.b) {
|
||||
code := r.uleb(t)
|
||||
if code == 0 {
|
||||
continue // end of the CU's children
|
||||
}
|
||||
ab, ok := abs[code]
|
||||
if !ok {
|
||||
t.Fatalf("unknown abbreviation code %d", code)
|
||||
}
|
||||
name, lowPC = "", 0
|
||||
for _, a := range ab.attrs {
|
||||
switch a.attr {
|
||||
case dwAtName:
|
||||
readFormKeep(t, r, a.form, &name, nil)
|
||||
case dwAtLowPC:
|
||||
readFormKeep(t, r, a.form, nil, &lowPC)
|
||||
default:
|
||||
readFormSkip(t, r, a.form)
|
||||
}
|
||||
}
|
||||
if ab.tag == dwTagSubprog && name != "" {
|
||||
out[name] = lowPC
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// readFormKeep reads one DIE attribute value, keeping a string or an
|
||||
// address into the pointer it was given (nil keeps nothing).
|
||||
func readFormKeep(t *testing.T, r *ulebIter, form uint64, name *string, addr *uint64) {
|
||||
t.Helper()
|
||||
switch form {
|
||||
case dwFormString:
|
||||
end := r.i
|
||||
for end < len(r.b) && r.b[end] != 0 {
|
||||
end++
|
||||
}
|
||||
if name != nil {
|
||||
*name = string(r.b[r.i:end])
|
||||
}
|
||||
r.i = end + 1
|
||||
case dwFormAddr:
|
||||
if addr != nil {
|
||||
*addr = binary.LittleEndian.Uint64(r.b[r.i:])
|
||||
}
|
||||
r.i += 8
|
||||
default:
|
||||
readFormSkip(t, r, form)
|
||||
}
|
||||
}
|
||||
|
||||
func readFormSkip(t *testing.T, r *ulebIter, form uint64) {
|
||||
t.Helper()
|
||||
switch form {
|
||||
case dwFormString:
|
||||
for r.i < len(r.b) && r.b[r.i] != 0 {
|
||||
r.i++
|
||||
}
|
||||
r.i++
|
||||
case dwFormAddr, dwFormData8:
|
||||
r.i += 8
|
||||
case dwFormSecOff:
|
||||
r.i += 4
|
||||
case dwFormExprloc:
|
||||
r.i += int(r.uleb(t))
|
||||
case dwFormData1, 0x0c:
|
||||
r.i++
|
||||
default:
|
||||
t.Fatalf("unsupported form %#x", form)
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFObjectDataRelocation checks that a symbol-valued DATA field ("DATA
|
||||
// s+0(SB)/8, $other(SB)") reaches the ELF object as a .rela.data entry: an
|
||||
// absolute 64-bit relocation at the field's offset within .data, against
|
||||
// the named symbol, external targets included.
|
||||
func TestELFObjectDataRelocation(t *testing.T) {
|
||||
f, errs := parser.Parse("t_amd64.s", `#include "textflag.h"
|
||||
TEXT ·Keep(SB), NOSPLIT, $0-8
|
||||
RET
|
||||
GLOBL holder(SB), NOPTR, $24
|
||||
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||
DATA holder+8(SB)/8, $holder(SB)
|
||||
DATA holder+16(SB)/8, $extvar(SB)
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
obj, err := img.ELFObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFObject: %v", err)
|
||||
}
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
relaData := ef.Section(".rela.data")
|
||||
if relaData == nil {
|
||||
t.Fatal("missing .rela.data section")
|
||||
}
|
||||
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||
}
|
||||
if ef.Sections[relaData.Info].Name != ".data" {
|
||||
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||
}
|
||||
relas, err := relaData.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var got []struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}
|
||||
for i := 0; i+24 <= len(relas); i += 24 {
|
||||
got = append(got, struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}{
|
||||
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||
// r_info packs the type in the low dword and the symbol index
|
||||
// in the high dword.
|
||||
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||
})
|
||||
}
|
||||
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
name := func(idx uint32) string {
|
||||
if idx >= 1 && int(idx) <= len(syms) {
|
||||
return syms[idx-1].Name
|
||||
}
|
||||
return ""
|
||||
}
|
||||
// The offsets are data-section-relative: the field's DATA offset plus
|
||||
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||
base := uint64(0)
|
||||
for _, d := range img.DataSyms {
|
||||
if d.Name == "holder" {
|
||||
base = uint64(d.Offset)
|
||||
}
|
||||
}
|
||||
want := []struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
addend int64
|
||||
target string
|
||||
}{
|
||||
{off: base + 0, typ: uint32(elf.R_X86_64_64), addend: 5, target: "Keep"},
|
||||
{off: base + 8, typ: uint32(elf.R_X86_64_64), addend: 0, target: "holder"},
|
||||
{off: base + 16, typ: uint32(elf.R_X86_64_64), addend: 0, target: "extvar"},
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i, w := range want {
|
||||
g := got[i]
|
||||
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||
}
|
||||
if n := name(g.sym); n != w.target {
|
||||
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+168
-22
@@ -15,13 +15,19 @@ const (
|
||||
|
||||
// AArch64 relocation types (the ELF psABI).
|
||||
rArm64PrelPgHi21 = 275 // R_AARCH64_ADR_PREL_PG_HI21 (ADRP page)
|
||||
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD/STR/LDR page offset)
|
||||
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset)
|
||||
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
|
||||
rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset)
|
||||
// R_AARCH64_ABS32 (debug/elf 258): the absolute 32-bit address of a
|
||||
// symbol, the R_ADDR shape a 4-byte DATA field carries. ABS64 (257)
|
||||
// lives with the DWARF fixup constants as rAARCH64Abs64.
|
||||
rArm64Abs32 = 258
|
||||
)
|
||||
|
||||
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
|
||||
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
|
||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
|
||||
// optional .rela.text.
|
||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
|
||||
// optional .rela.text and an optional .rela.data.
|
||||
func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
@@ -79,8 +85,19 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
}
|
||||
|
||||
// Build relocations. Each SB reference is an ADRP pair:
|
||||
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21
|
||||
// ADD/LDR/STR → R_AARCH64_ADD_ABS_LO12_NC
|
||||
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21 at the ADRP
|
||||
// ADD → R_AARCH64_ADD_ABS_LO12_NC at the ADD word
|
||||
// LDR/STR X → R_AARCH64_LDST64_ABS_LO12_NC at the LDR/STR word
|
||||
// BL → R_AARCH64_CALL26
|
||||
// cmd/link's own conversion emits the HI21 at sectoff and the LO12 at
|
||||
// sectoff+4 (cmd/link/internal/arm64/asm.go), so the ADD or load word
|
||||
// carries the page-offset relocation, never a second HI21. The
|
||||
// assembler records two RelArm64Addr relocs per ADRP+ADD pair (one per
|
||||
// word), so the second of the pair is consumed here.
|
||||
// Addends stay raw: ADR_PREL_PG_HI21 and the ABS_LO12_NC forms resolve
|
||||
// against S+A, and CALL26 branches take the branch instruction's own
|
||||
// place as the PC-relative base, so subtracting the field width (the
|
||||
// amd64 R_PCREL convention) would misplace every branch by 4 bytes.
|
||||
type elfRela struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
@@ -89,25 +106,81 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
}
|
||||
var relas []elfRela
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
for i := 0; i < len(fn.Relocs); i++ {
|
||||
r := fn.Relocs[i]
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
typ := uint32(rArm64PrelPgHi21)
|
||||
if r.Kind == RelArm64Addr && r.Off%4 == 4 {
|
||||
// The second instruction in an ADRP pair uses ADD_ABS_LO12_NC.
|
||||
typ = rArm64AddAbsLo12NC
|
||||
switch r.Kind {
|
||||
case RelArm64Branch:
|
||||
relas = append(relas, elfRela{
|
||||
off: uint64(fn.Offset + r.Off), typ: rArm64Call26, sym: idx, addend: r.Addend,
|
||||
})
|
||||
case RelArm64Addr:
|
||||
// ADRP+ADD: the pair's second reloc (at Off+4) is the
|
||||
// assembler's twin of the same pair; skip it.
|
||||
relas = append(relas,
|
||||
elfRela{off: uint64(fn.Offset + r.Off), typ: rArm64PrelPgHi21, sym: idx, addend: r.Addend},
|
||||
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rArm64AddAbsLo12NC, sym: idx, addend: r.Addend},
|
||||
)
|
||||
i++
|
||||
case RelArm64LDST64:
|
||||
// ADRP+LDR/STR: one assembler reloc covers the pair.
|
||||
relas = append(relas,
|
||||
elfRela{off: uint64(fn.Offset + r.Off), typ: rArm64PrelPgHi21, sym: idx, addend: r.Addend},
|
||||
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rArm64Ldst64Lo12NC, sym: idx, addend: r.Addend},
|
||||
)
|
||||
default:
|
||||
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
|
||||
}
|
||||
relas = append(relas, elfRela{
|
||||
off: uint64(fn.Offset + r.Off),
|
||||
typ: typ,
|
||||
}
|
||||
}
|
||||
|
||||
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||
// DATA line's width at the field's data-section offset, S + A with no
|
||||
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||
// cannot hold an address, so they are refused rather than truncated.
|
||||
var dataRelas []elfRela
|
||||
for _, d := range img.DataSyms {
|
||||
for _, r := range d.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
var typ uint32
|
||||
switch r.Siz {
|
||||
case 8:
|
||||
typ = rAARCH64Abs64
|
||||
case 4:
|
||||
typ = rArm64Abs32
|
||||
default:
|
||||
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||
}
|
||||
dataRelas = append(dataRelas, elfRela{
|
||||
off: uint64(d.Offset + r.Off),
|
||||
sym: idx,
|
||||
addend: r.Addend - int64(r.After-r.Off),
|
||||
typ: typ,
|
||||
addend: r.Addend,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Section presence: .rela.text only when there are code relocations,
|
||||
// .rela.data only when a DATA line holds a symbol value.
|
||||
hasRela := len(relas) > 0
|
||||
hasDataRela := len(dataRelas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections++
|
||||
}
|
||||
if hasDataRela {
|
||||
nSections++
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// String tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
@@ -117,14 +190,12 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
if hasDataRela {
|
||||
stSections.add(".rela.data")
|
||||
}
|
||||
for _, n := range dwarfSectionNames {
|
||||
stSections.add(n)
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Layout.
|
||||
var out []byte
|
||||
@@ -160,7 +231,7 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
var relaOff, relaDataOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
@@ -172,10 +243,49 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
if hasDataRela {
|
||||
align(8)
|
||||
relaDataOff = len(out)
|
||||
for _, r := range dataRelas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
|
||||
// DWARF debug sections; the address placeholders they leave are carried
|
||||
// as .rela.debug_info/.rela.debug_line entries the system linker applies.
|
||||
dwAlign := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiARM64)
|
||||
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
|
||||
if dw != nil {
|
||||
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
|
||||
// .debug_line_str and .debug_frame (the CIE is unconditional, so
|
||||
// the frame section is always present), plus the relocation
|
||||
// sections below when they carry entries.
|
||||
dwarfStart = nSections
|
||||
nSections += 5
|
||||
appendDWARFRelas(&out, dw, rAARCH64Abs64, dwAlign)
|
||||
if dw.infoRelaCount > 0 {
|
||||
nSections++
|
||||
}
|
||||
if dw.lineRelaCount > 0 {
|
||||
nSections++
|
||||
}
|
||||
if dw.frameRelaCount > 0 {
|
||||
nSections++
|
||||
}
|
||||
}
|
||||
|
||||
align(8)
|
||||
shoff := len(out)
|
||||
|
||||
@@ -201,7 +311,43 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
if hasDataRela {
|
||||
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
// DWARF section headers; their indices follow the write order.
|
||||
if dw != nil {
|
||||
// secIdx is a running section index: each putSh below emits the
|
||||
// next header, and the sh_info of a .rela section names the index
|
||||
// of the section it relocates.
|
||||
secIdx := dwarfStart
|
||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||
secIdx++
|
||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||
secInfoIdx := secIdx
|
||||
secIdx++
|
||||
if dw.infoRelaCount > 0 {
|
||||
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
|
||||
secIdx++
|
||||
}
|
||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||
secLineIdx := secIdx
|
||||
secIdx++
|
||||
if dw.lineRelaCount > 0 {
|
||||
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
|
||||
secIdx++
|
||||
}
|
||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||
secIdx++
|
||||
if dw.frameSize > 0 {
|
||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||
secFrameIdx := secIdx
|
||||
secIdx++
|
||||
if dw.frameRelaCount > 0 {
|
||||
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ELF header.
|
||||
hdr := out[:64]
|
||||
|
||||
+174
-2
@@ -6,9 +6,10 @@ package asm
|
||||
import (
|
||||
"bytes"
|
||||
"debug/elf"
|
||||
"encoding/binary"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
|
||||
@@ -28,6 +29,7 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
||||
|
||||
TEXT ·getanswer(SB), NOSPLIT, $0-8
|
||||
MOVD answer<>(SB), R4
|
||||
MOVD $answer<>(SB), R5
|
||||
MOVD R4, ret+0(FP)
|
||||
RET
|
||||
|
||||
@@ -102,8 +104,63 @@ DATA answer<>+0(SB)/8, $42
|
||||
// Check that .rela.text exists (getanswer has SB reference).
|
||||
relaText := ef.Section(".rela.text")
|
||||
if relaText == nil {
|
||||
t.Error("missing .rela.text section")
|
||||
t.Fatal("missing .rela.text section")
|
||||
}
|
||||
|
||||
// The SB references of getanswer form two ADRP pairs: the load
|
||||
// (MOVD answer<>(SB), R4) is ADRP+LDR carrying HI21 at the ADRP and
|
||||
// LDST64_ABS_LO12_NC at the LDR word, and the address-of
|
||||
// (MOVD $answer<>(SB), R5) is ADRP+ADD carrying HI21 and
|
||||
// ADD_ABS_LO12_NC. cmd/link's own conversion emits exactly this
|
||||
// sectoff / sectoff+4 pairing; a second HI21 at the ADD or LDR word
|
||||
// corrupts the pair.
|
||||
raw, err := relaText.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(raw)%24 != 0 || len(raw)/24 != 4 {
|
||||
t.Fatalf(".rela.text has %d bytes, want four 24-byte entries", len(raw))
|
||||
}
|
||||
wantRela := []struct {
|
||||
typ elf.R_AARCH64
|
||||
off uint64 // relative to the getanswer function start
|
||||
}{
|
||||
{elf.R_AARCH64_ADR_PREL_PG_HI21, 0},
|
||||
{elf.R_AARCH64_LDST64_ABS_LO12_NC, 4},
|
||||
{elf.R_AARCH64_ADR_PREL_PG_HI21, 8},
|
||||
{elf.R_AARCH64_ADD_ABS_LO12_NC, 12},
|
||||
}
|
||||
getanswer := byNameElf(t, ef, "getanswer")
|
||||
for i, w := range wantRela {
|
||||
e := raw[i*24 : (i+1)*24]
|
||||
off := binary.LittleEndian.Uint64(e[0:])
|
||||
info := binary.LittleEndian.Uint64(e[8:])
|
||||
typ := elf.R_AARCH64(info & 0xffffffff)
|
||||
sym := int(info >> 32)
|
||||
if typ != w.typ || off != getanswer.Value+w.off {
|
||||
t.Errorf("reloc %d: type %v off %d, want %v at %d", i, typ, off, w.typ, getanswer.Value+w.off)
|
||||
}
|
||||
if sym != 3 { // NULL, .text, .data, then the first local: answer
|
||||
t.Errorf("reloc %d: symbol index %d, want 3 (answer)", i, sym)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// byNameElf returns the symbol table entry for name from the raw .symtab,
|
||||
// which carries every entry including the null and section symbols in order.
|
||||
func byNameElf(t *testing.T, ef *elf.File, name string) elf.Symbol {
|
||||
t.Helper()
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatalf("symbols: %v", err)
|
||||
}
|
||||
for _, s := range syms {
|
||||
if s.Name == name {
|
||||
return s
|
||||
}
|
||||
}
|
||||
t.Fatalf("symbol %q not found", name)
|
||||
return elf.Symbol{}
|
||||
}
|
||||
|
||||
// TestELFAARCH64ObjectNoRelocations checks the ELF output when there are no
|
||||
@@ -140,3 +197,118 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
||||
t.Error("unexpected .rela.text section when there are no relocations")
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFAARCH64ObjectDataRelocation checks that a symbol-valued DATA field
|
||||
// ("DATA s+0(SB)/8, $other(SB)") reaches the AArch64 ELF object as a
|
||||
// .rela.data entry: an R_AARCH64_ABS64 (ABS32 for a width-4 field) at the
|
||||
// field's offset within .data, against the named symbol, external targets
|
||||
// included.
|
||||
func TestELFAARCH64ObjectDataRelocation(t *testing.T) {
|
||||
f, errs := parser.Parse("t_arm64.s", `#include "textflag.h"
|
||||
TEXT ·Keep(SB), NOSPLIT, $0-0
|
||||
RET
|
||||
GLOBL holder(SB), NOPTR, $32
|
||||
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||
DATA holder+8(SB)/8, $holder(SB)
|
||||
DATA holder+16(SB)/8, $extvar(SB)
|
||||
DATA holder+24(SB)/4, $Keep(SB)
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
obj, err := img.ELFAARCH64Object()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFAARCH64Object: %v", err)
|
||||
}
|
||||
checkELFSectionAccounting(t, obj)
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
relaData := ef.Section(".rela.data")
|
||||
if relaData == nil {
|
||||
t.Fatal("missing .rela.data section")
|
||||
}
|
||||
if relaData.Type != elf.SHT_RELA {
|
||||
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
|
||||
}
|
||||
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||
}
|
||||
if ef.Sections[relaData.Info].Name != ".data" {
|
||||
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||
}
|
||||
relas, err := relaData.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var got []struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}
|
||||
for i := 0; i+24 <= len(relas); i += 24 {
|
||||
got = append(got, struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}{
|
||||
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||
// r_info packs the type in the low dword and the symbol index
|
||||
// in the high dword.
|
||||
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||
})
|
||||
}
|
||||
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
name := func(idx uint32) string {
|
||||
if idx >= 1 && int(idx) <= len(syms) {
|
||||
return syms[idx-1].Name
|
||||
}
|
||||
return ""
|
||||
}
|
||||
// The offsets are data-section-relative: the field's DATA offset plus
|
||||
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||
base := uint64(0)
|
||||
for _, d := range img.DataSyms {
|
||||
if d.Name == "holder" {
|
||||
base = uint64(d.Offset)
|
||||
}
|
||||
}
|
||||
want := []struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
addend int64
|
||||
target string
|
||||
}{
|
||||
{off: base + 0, typ: uint32(elf.R_AARCH64_ABS64), addend: 5, target: "Keep"},
|
||||
{off: base + 8, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "holder"},
|
||||
{off: base + 16, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "extvar"},
|
||||
{off: base + 24, typ: uint32(elf.R_AARCH64_ABS32), addend: 0, target: "Keep"},
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i, w := range want {
|
||||
g := got[i]
|
||||
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||
}
|
||||
if n := name(g.sym); n != w.target {
|
||||
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+142
-13
@@ -13,15 +13,26 @@ import (
|
||||
const (
|
||||
emLOONGARCH = 258 // EM_LOONGARCH
|
||||
|
||||
// EF_LOONGARCH_ABI_DOUBLE_FLOAT | EF_LOONGARCH_OBJABI_V1: the flags the
|
||||
// Go toolchain writes (cmd/link/internal/ld/elf.go: Flags = 0x43 for
|
||||
// Loong64). System linkers refuse to merge ET_REL objects whose float
|
||||
// ABI differs, so 0 (soft-float) would make the object unlinkable.
|
||||
efLarchAbiDoubleObjV1 = 0x43
|
||||
|
||||
// LoongArch relocation types (the ELF psABI).
|
||||
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
|
||||
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
|
||||
rLarchB26 = 66 // R_LARCH_B26 (b/bl, matches the Go linker's mapping)
|
||||
// R_LARCH_32 (debug/elf 1): the absolute 32-bit address of a symbol,
|
||||
// the R_ADDR shape a 4-byte DATA field carries. R_LARCH_64 (2) lives
|
||||
// with the DWARF fixup constants as rLarchAbs64.
|
||||
rLarchAbs32 = 1
|
||||
)
|
||||
|
||||
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
|
||||
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
|
||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
|
||||
// optional .rela.text.
|
||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
|
||||
// optional .rela.text and an optional .rela.data.
|
||||
func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
@@ -95,18 +106,65 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
typ := uint32(rLarchPCALAHI20)
|
||||
if r.Kind == RelLoong64AddrLo {
|
||||
switch r.Kind {
|
||||
case RelLoong64AddrLo:
|
||||
typ = rLarchPCALALO12
|
||||
case RelLoong64Branch:
|
||||
typ = rLarchB26
|
||||
}
|
||||
relas = append(relas, elfRela{
|
||||
off: uint64(fn.Offset + r.Off),
|
||||
typ: typ,
|
||||
sym: idx,
|
||||
addend: r.Addend - int64(r.After-r.Off),
|
||||
addend: r.Addend,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||
// DATA line's width at the field's data-section offset, S + A with no
|
||||
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||
// cannot hold an address, so they are refused rather than truncated.
|
||||
var dataRelas []elfRela
|
||||
for _, d := range img.DataSyms {
|
||||
for _, r := range d.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
var typ uint32
|
||||
switch r.Siz {
|
||||
case 8:
|
||||
typ = rLarchAbs64
|
||||
case 4:
|
||||
typ = rLarchAbs32
|
||||
default:
|
||||
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||
}
|
||||
dataRelas = append(dataRelas, elfRela{
|
||||
off: uint64(d.Offset + r.Off),
|
||||
sym: idx,
|
||||
typ: typ,
|
||||
addend: r.Addend,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Section presence: .rela.text only when there are code relocations,
|
||||
// .rela.data only when a DATA line holds a symbol value.
|
||||
hasRela := len(relas) > 0
|
||||
hasDataRela := len(dataRelas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections++
|
||||
}
|
||||
if hasDataRela {
|
||||
nSections++
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// String tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
@@ -116,14 +174,12 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
if hasDataRela {
|
||||
stSections.add(".rela.data")
|
||||
}
|
||||
for _, n := range dwarfSectionNames {
|
||||
stSections.add(n)
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Layout.
|
||||
var out []byte
|
||||
@@ -159,7 +215,7 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
var relaOff, relaDataOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
@@ -171,10 +227,47 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
if hasDataRela {
|
||||
align(8)
|
||||
relaDataOff = len(out)
|
||||
for _, r := range dataRelas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
|
||||
dwAlign := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiLOONG64)
|
||||
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
|
||||
if dw != nil {
|
||||
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
|
||||
// .debug_line_str and .debug_frame (the CIE is unconditional, so
|
||||
// the frame section is always present), plus the relocation
|
||||
// sections below when they carry entries.
|
||||
dwarfStart = nSections
|
||||
nSections += 5
|
||||
appendDWARFRelas(&out, dw, rLarchAbs64, dwAlign)
|
||||
if dw.infoRelaCount > 0 {
|
||||
nSections++
|
||||
}
|
||||
if dw.lineRelaCount > 0 {
|
||||
nSections++
|
||||
}
|
||||
if dw.frameRelaCount > 0 {
|
||||
nSections++
|
||||
}
|
||||
}
|
||||
|
||||
align(8)
|
||||
shoff := len(out)
|
||||
|
||||
@@ -200,7 +293,43 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
if hasDataRela {
|
||||
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
// DWARF section headers; their indices follow the write order.
|
||||
if dw != nil {
|
||||
// secIdx is a running section index: each putSh below emits the
|
||||
// next header, and the sh_info of a .rela section names the index
|
||||
// of the section it relocates.
|
||||
secIdx := dwarfStart
|
||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||
secIdx++
|
||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||
secInfoIdx := secIdx
|
||||
secIdx++
|
||||
if dw.infoRelaCount > 0 {
|
||||
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
|
||||
secIdx++
|
||||
}
|
||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||
secLineIdx := secIdx
|
||||
secIdx++
|
||||
if dw.lineRelaCount > 0 {
|
||||
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
|
||||
secIdx++
|
||||
}
|
||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||
secIdx++
|
||||
if dw.frameSize > 0 {
|
||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||
secFrameIdx := secIdx
|
||||
secIdx++
|
||||
if dw.frameRelaCount > 0 {
|
||||
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ELF header.
|
||||
hdr := out[:64]
|
||||
@@ -211,7 +340,7 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
le.PutUint64(hdr[24:], 0)
|
||||
le.PutUint64(hdr[32:], 0)
|
||||
le.PutUint64(hdr[40:], uint64(shoff))
|
||||
le.PutUint32(hdr[48:], 0)
|
||||
le.PutUint32(hdr[48:], efLarchAbiDoubleObjV1)
|
||||
le.PutUint16(hdr[52:], 64)
|
||||
le.PutUint16(hdr[54:], 0)
|
||||
le.PutUint16(hdr[56:], 0)
|
||||
|
||||
+164
-2
@@ -9,7 +9,7 @@ import (
|
||||
"encoding/binary"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
|
||||
@@ -55,6 +55,11 @@ DATA answer<>+0(SB)/8, $42
|
||||
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_LOONGARCH {
|
||||
t.Errorf("type/machine = %v/%v, want ET_REL/EM_LOONGARCH", ef.Type, ef.Machine)
|
||||
}
|
||||
// The double-float ABI plus OBJABI_V1 flags the Go toolchain writes;
|
||||
// system linkers refuse ABI-mismatched merges.
|
||||
if flags := binary.LittleEndian.Uint32(obj[48:]); flags != efLarchAbiDoubleObjV1 {
|
||||
t.Errorf("e_flags = %#x, want %#x (double-float, OBJABI_V1)", flags, efLarchAbiDoubleObjV1)
|
||||
}
|
||||
|
||||
text := ef.Section(".text")
|
||||
data := ef.Section(".data")
|
||||
@@ -139,7 +144,7 @@ DATA answer<>+0(SB)/8, $42
|
||||
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
|
||||
}
|
||||
le := binary.LittleEndian
|
||||
for i := 0; i < 2; i++ {
|
||||
for i := range 2 {
|
||||
e := raw[i*24 : (i+1)*24]
|
||||
off := le.Uint64(e[0:])
|
||||
info := le.Uint64(e[8:])
|
||||
@@ -198,3 +203,160 @@ TEXT ·nop(SB), NOSPLIT, $0
|
||||
t.Error("function symbol nop not found")
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFLOONG64BranchRelocation checks that the morestack call and an
|
||||
// internal CALL both carry R_LARCH_B26 in the emitted object, matching the
|
||||
// Go linker's mapping of its call relocation.
|
||||
func TestELFLOONG64BranchRelocation(t *testing.T) {
|
||||
f, errs := parser.Parse("k_loong64.s", "TEXT \u00b7callbig(SB), $8192-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileLOONG64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||
}
|
||||
obj, err := img.ELFLOONG64Object()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFLOONG64Object: %v", err)
|
||||
}
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
relaSec := ef.Section(".rela.text")
|
||||
if relaSec == nil {
|
||||
t.Fatal("missing .rela.text")
|
||||
}
|
||||
raw, err := relaSec.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// The guard's morestack call plus the body's CALL to other.
|
||||
if len(raw)%24 != 0 || len(raw)/24 != 2 {
|
||||
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
|
||||
}
|
||||
le := binary.LittleEndian
|
||||
for i := range 2 {
|
||||
info := le.Uint64(raw[i*24+8:])
|
||||
if elf.R_LARCH(info&0xffffffff) != elf.R_LARCH_B26 {
|
||||
t.Errorf("relocation %d type = %v, want R_LARCH_B26", i, elf.R_LARCH(info&0xffffffff))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFLOONG64ObjectDataRelocation checks that a symbol-valued DATA field
|
||||
// ("DATA s+0(SB)/8, $other(SB)") reaches the LoongArch ELF object as a
|
||||
// .rela.data entry: an R_LARCH_64 (R_LARCH_32 for a width-4 field) at the
|
||||
// field's offset within .data, against the named symbol, external targets
|
||||
// included.
|
||||
func TestELFLOONG64ObjectDataRelocation(t *testing.T) {
|
||||
f, errs := parser.Parse("t_loong64.s", `#include "textflag.h"
|
||||
TEXT ·Keep(SB), NOSPLIT, $0-0
|
||||
RET
|
||||
GLOBL holder(SB), NOPTR, $32
|
||||
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||
DATA holder+8(SB)/8, $holder(SB)
|
||||
DATA holder+16(SB)/8, $extvar(SB)
|
||||
DATA holder+24(SB)/4, $Keep(SB)
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileLOONG64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||
}
|
||||
obj, err := img.ELFLOONG64Object()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFLOONG64Object: %v", err)
|
||||
}
|
||||
checkELFSectionAccounting(t, obj)
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
relaData := ef.Section(".rela.data")
|
||||
if relaData == nil {
|
||||
t.Fatal("missing .rela.data section")
|
||||
}
|
||||
if relaData.Type != elf.SHT_RELA {
|
||||
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
|
||||
}
|
||||
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||
}
|
||||
if ef.Sections[relaData.Info].Name != ".data" {
|
||||
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||
}
|
||||
relas, err := relaData.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var got []struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}
|
||||
for i := 0; i+24 <= len(relas); i += 24 {
|
||||
got = append(got, struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}{
|
||||
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||
// r_info packs the type in the low dword and the symbol index
|
||||
// in the high dword.
|
||||
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||
})
|
||||
}
|
||||
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
name := func(idx uint32) string {
|
||||
if idx >= 1 && int(idx) <= len(syms) {
|
||||
return syms[idx-1].Name
|
||||
}
|
||||
return ""
|
||||
}
|
||||
// The offsets are data-section-relative: the field's DATA offset plus
|
||||
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||
base := uint64(0)
|
||||
for _, d := range img.DataSyms {
|
||||
if d.Name == "holder" {
|
||||
base = uint64(d.Offset)
|
||||
}
|
||||
}
|
||||
want := []struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
addend int64
|
||||
target string
|
||||
}{
|
||||
{off: base + 0, typ: uint32(elf.R_LARCH_64), addend: 5, target: "Keep"},
|
||||
{off: base + 8, typ: uint32(elf.R_LARCH_64), addend: 0, target: "holder"},
|
||||
{off: base + 16, typ: uint32(elf.R_LARCH_64), addend: 0, target: "extvar"},
|
||||
{off: base + 24, typ: uint32(elf.R_LARCH_32), addend: 0, target: "Keep"},
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i, w := range want {
|
||||
g := got[i]
|
||||
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||
}
|
||||
if n := name(g.sym); n != w.target {
|
||||
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+147
-18
@@ -13,17 +13,27 @@ import (
|
||||
const (
|
||||
emRISCV = 243 // EM_RISCV
|
||||
|
||||
// EF_RISCV_FLOAT_ABI_DOUBLE: the double-precision float ABI the Go
|
||||
// toolchain targets (cmd/link/internal/ld/elf.go writes Flags = 0x4 for
|
||||
// RISCV64). System linkers refuse to merge ET_REL objects whose float
|
||||
// ABI differs, so 0 (soft-float) would make the object unlinkable.
|
||||
efRISCVFloatAbiDouble = 0x4
|
||||
|
||||
// RISC-V relocation types.
|
||||
rRISCV32 = 1
|
||||
rRISCVJAL = 17 // R_RISCV_JAL
|
||||
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
|
||||
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
|
||||
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
|
||||
// R_RISCV_32 (debug/elf 1): the absolute 32-bit address of a symbol,
|
||||
// the R_ADDR shape a 4-byte DATA field carries. R_RISCV_64 (2) lives
|
||||
// with the DWARF fixup constants as rRISCVAbs64.
|
||||
rRISVCAbs32 = 1
|
||||
)
|
||||
|
||||
// ELFRISCVObject returns the image as an ELF64 relocatable object file for
|
||||
// RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64
|
||||
// ELF emission: .text, .data, .symtab, .strtab and optional .rela.text.
|
||||
// ELF emission: .text, .data, .symtab, .strtab, an optional .rela.text and
|
||||
// an optional .rela.data.
|
||||
func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
@@ -83,9 +93,14 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
// Build relocations. Each SB reference is an AUIPC + second-instruction
|
||||
// pair carrying a single relocation kind; the ELF writer expands it into
|
||||
// the R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I/S pair the psABI expects.
|
||||
// The HI20 carries the symbol addend; the LO12 addend is zero, matching
|
||||
// cmd/link's own ELF conversion (the LO12 resolves against the HI20's
|
||||
// AUIPC location).
|
||||
// The HI20 carries the symbol and its addend. The LO12's symbol must
|
||||
// denote the AUIPC site the HI20 relocates (psABI §8.4.9: the pair is
|
||||
// resolved against the label of the AUIPC, not the target symbol;
|
||||
// cmd/link generates one local text symbol per AUIPC for exactly this,
|
||||
// cmd/link/internal/riscv64/asm.go). The .text section symbol with the
|
||||
// AUIPC's section-relative offset as addend gives S + A = the AUIPC
|
||||
// address, which is that label.
|
||||
const secSymText = 1 // syms[1], the .text section symbol
|
||||
type elfRela struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
@@ -99,27 +114,70 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
auipc := int64(fn.Offset + r.Off)
|
||||
switch r.Kind {
|
||||
case RelRISCVPCRELIType:
|
||||
relas = append(relas,
|
||||
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
||||
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: idx, addend: 0},
|
||||
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: secSymText, addend: auipc},
|
||||
)
|
||||
case RelRISCVPCRELSType:
|
||||
relas = append(relas,
|
||||
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
||||
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: idx, addend: 0},
|
||||
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: secSymText, addend: auipc},
|
||||
)
|
||||
case RelRISCVJal:
|
||||
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVJAL, sym: idx, addend: r.Addend})
|
||||
case RelPCRelAbs:
|
||||
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCV32, sym: idx, addend: r.Addend})
|
||||
default:
|
||||
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||
// DATA line's width at the field's data-section offset, S + A with no
|
||||
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||
// cannot hold an address, so they are refused rather than truncated.
|
||||
var dataRelas []elfRela
|
||||
for _, d := range img.DataSyms {
|
||||
for _, r := range d.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
var typ uint32
|
||||
switch r.Siz {
|
||||
case 8:
|
||||
typ = rRISCVAbs64
|
||||
case 4:
|
||||
typ = rRISVCAbs32
|
||||
default:
|
||||
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||
}
|
||||
dataRelas = append(dataRelas, elfRela{
|
||||
off: uint64(d.Offset + r.Off),
|
||||
sym: idx,
|
||||
typ: typ,
|
||||
addend: r.Addend,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Section presence: .rela.text only when there are code relocations,
|
||||
// .rela.data only when a DATA line holds a symbol value.
|
||||
hasRela := len(relas) > 0
|
||||
hasDataRela := len(dataRelas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections++
|
||||
}
|
||||
if hasDataRela {
|
||||
nSections++
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// String tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
@@ -129,14 +187,12 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
if hasDataRela {
|
||||
stSections.add(".rela.data")
|
||||
}
|
||||
for _, n := range dwarfSectionNames {
|
||||
stSections.add(n)
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Layout.
|
||||
var out []byte
|
||||
@@ -172,7 +228,7 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
var relaOff, relaDataOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
@@ -184,10 +240,47 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
if hasDataRela {
|
||||
align(8)
|
||||
relaDataOff = len(out)
|
||||
for _, r := range dataRelas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
|
||||
dwAlign := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiRISCV64)
|
||||
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
|
||||
if dw != nil {
|
||||
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
|
||||
// .debug_line_str and .debug_frame (the CIE is unconditional, so
|
||||
// the frame section is always present), plus the relocation
|
||||
// sections below when they carry entries.
|
||||
dwarfStart = nSections
|
||||
nSections += 5
|
||||
appendDWARFRelas(&out, dw, rRISCVAbs64, dwAlign)
|
||||
if dw.infoRelaCount > 0 {
|
||||
nSections++
|
||||
}
|
||||
if dw.lineRelaCount > 0 {
|
||||
nSections++
|
||||
}
|
||||
if dw.frameRelaCount > 0 {
|
||||
nSections++
|
||||
}
|
||||
}
|
||||
|
||||
align(8)
|
||||
shoff := len(out)
|
||||
|
||||
@@ -213,7 +306,43 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
if hasDataRela {
|
||||
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
// DWARF section headers; their indices follow the write order.
|
||||
if dw != nil {
|
||||
// secIdx is a running section index: each putSh below emits the
|
||||
// next header, and the sh_info of a .rela section names the index
|
||||
// of the section it relocates.
|
||||
secIdx := dwarfStart
|
||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||
secIdx++
|
||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||
secInfoIdx := secIdx
|
||||
secIdx++
|
||||
if dw.infoRelaCount > 0 {
|
||||
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
|
||||
secIdx++
|
||||
}
|
||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||
secLineIdx := secIdx
|
||||
secIdx++
|
||||
if dw.lineRelaCount > 0 {
|
||||
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
|
||||
secIdx++
|
||||
}
|
||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||
secIdx++
|
||||
if dw.frameSize > 0 {
|
||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||
secFrameIdx := secIdx
|
||||
secIdx++
|
||||
if dw.frameRelaCount > 0 {
|
||||
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ELF header.
|
||||
hdr := out[:64]
|
||||
@@ -224,7 +353,7 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
le.PutUint64(hdr[24:], 0)
|
||||
le.PutUint64(hdr[32:], 0)
|
||||
le.PutUint64(hdr[40:], uint64(shoff))
|
||||
le.PutUint32(hdr[48:], 0)
|
||||
le.PutUint32(hdr[48:], efRISCVFloatAbiDouble)
|
||||
le.PutUint16(hdr[52:], 64)
|
||||
le.PutUint16(hdr[54:], 0)
|
||||
le.PutUint16(hdr[56:], 0)
|
||||
|
||||
@@ -0,0 +1,128 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"debug/elf"
|
||||
"encoding/binary"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// TestELFRISCVObjectDataRelocation checks that a symbol-valued DATA field
|
||||
// ("DATA s+0(SB)/8, $other(SB)") reaches the RISC-V ELF object as a
|
||||
// .rela.data entry: an R_RISCV_64 (R_RISCV_32 for a width-4 field) at the
|
||||
// field's offset within .data, against the named symbol, external targets
|
||||
// included.
|
||||
func TestELFRISCVObjectDataRelocation(t *testing.T) {
|
||||
f, errs := parser.Parse("t_riscv64.s", `#include "textflag.h"
|
||||
TEXT ·Keep(SB), NOSPLIT, $0-0
|
||||
RET
|
||||
GLOBL holder(SB), NOPTR, $32
|
||||
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||
DATA holder+8(SB)/8, $holder(SB)
|
||||
DATA holder+16(SB)/8, $extvar(SB)
|
||||
DATA holder+24(SB)/4, $Keep(SB)
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
obj, err := img.ELFRISCVObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFRISCVObject: %v", err)
|
||||
}
|
||||
checkELFSectionAccounting(t, obj)
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
relaData := ef.Section(".rela.data")
|
||||
if relaData == nil {
|
||||
t.Fatal("missing .rela.data section")
|
||||
}
|
||||
if relaData.Type != elf.SHT_RELA {
|
||||
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
|
||||
}
|
||||
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||
}
|
||||
if ef.Sections[relaData.Info].Name != ".data" {
|
||||
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||
}
|
||||
relas, err := relaData.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var got []struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}
|
||||
for i := 0; i+24 <= len(relas); i += 24 {
|
||||
got = append(got, struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}{
|
||||
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||
// r_info packs the type in the low dword and the symbol index
|
||||
// in the high dword.
|
||||
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||
})
|
||||
}
|
||||
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
name := func(idx uint32) string {
|
||||
if idx >= 1 && int(idx) <= len(syms) {
|
||||
return syms[idx-1].Name
|
||||
}
|
||||
return ""
|
||||
}
|
||||
// The offsets are data-section-relative: the field's DATA offset plus
|
||||
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||
base := uint64(0)
|
||||
for _, d := range img.DataSyms {
|
||||
if d.Name == "holder" {
|
||||
base = uint64(d.Offset)
|
||||
}
|
||||
}
|
||||
want := []struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
addend int64
|
||||
target string
|
||||
}{
|
||||
{off: base + 0, typ: uint32(elf.R_RISCV_64), addend: 5, target: "Keep"},
|
||||
{off: base + 8, typ: uint32(elf.R_RISCV_64), addend: 0, target: "holder"},
|
||||
{off: base + 16, typ: uint32(elf.R_RISCV_64), addend: 0, target: "extvar"},
|
||||
{off: base + 24, typ: uint32(elf.R_RISCV_32), addend: 0, target: "Keep"},
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i, w := range want {
|
||||
g := got[i]
|
||||
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||
}
|
||||
if n := name(g.sym); n != w.target {
|
||||
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,128 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import "strings"
|
||||
|
||||
// Encodable reports whether the amd64 encoder knows how to encode the
|
||||
// mnemonic. It mirrors the dispatch in (*enc).encode: the fixed-name
|
||||
// instructions, conditional jumps, the CMOV/SET condition families, the
|
||||
// VEX/EVEX/opmask/gather/scatter vector paths, the legacy SSE tables and the
|
||||
// explicit scalar cases. A mnemonic that parses (is in the architecture
|
||||
// table) but is not encodable would otherwise surface only at assembly time,
|
||||
// deep inside a build; the linter uses this predicate to flag it at edit
|
||||
// time.
|
||||
func Encodable(mnemonic string) bool {
|
||||
upper := strings.ToUpper(mnemonic)
|
||||
|
||||
// Fixed-name instructions (no size suffix).
|
||||
switch upper {
|
||||
case "RET", "NOP", "CALL", "JMP",
|
||||
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
|
||||
// The literal-data pseudo-ops, the accepted-and-ignored END and
|
||||
// bookkeeping statements, and the SP adjust.
|
||||
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA":
|
||||
return true
|
||||
}
|
||||
if _, ok := noOperandTable[upper]; ok {
|
||||
return true
|
||||
}
|
||||
if _, ok := condCode(upper); ok {
|
||||
return true
|
||||
}
|
||||
|
||||
// VEX/EVEX and friends: the trailing B/W/L/Q/D is part of the mnemonic.
|
||||
base, _, err := parseEvexSuffix(upper)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
||||
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
|
||||
return true
|
||||
}
|
||||
|
||||
// CMOV carries size then condition (CMOVLGT); SET carries the condition
|
||||
// alone (SETNE). The size letter is checked exactly as encodeCmov does,
|
||||
// so a spelling like CMOVBGT is not reported encodable when Encode
|
||||
// would reject it.
|
||||
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
|
||||
switch rest[0] {
|
||||
case 'W', 'L', 'Q':
|
||||
if _, ok := jccMap[rest[1:]]; ok {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
|
||||
if _, ok := jccMap[rest]; ok {
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// Legacy SSE shuffles and packed binaries dispatch on the full name; so
|
||||
// do the imm8-controlled instructions, the lane extracts and inserts and
|
||||
// the packed integer shifts (their trailing width letters belong to the
|
||||
// mnemonic).
|
||||
if _, ok := sseShufTable[upper]; ok {
|
||||
return true
|
||||
}
|
||||
if _, ok := sseBinTable[upper]; ok {
|
||||
return true
|
||||
}
|
||||
if _, ok := sseImm3Table[upper]; ok {
|
||||
return true
|
||||
}
|
||||
if _, ok := sseExtractTable[upper]; ok {
|
||||
return true
|
||||
}
|
||||
if _, ok := sseInsertTable[upper]; ok {
|
||||
return true
|
||||
}
|
||||
if _, ok := sseShiftImm[upper]; ok {
|
||||
return true
|
||||
}
|
||||
|
||||
// The size-suffix split: retry the tables and the scalar switch on the
|
||||
// base.
|
||||
base2, size := splitSize(upper)
|
||||
if size == 0 {
|
||||
size = 8
|
||||
}
|
||||
_ = size
|
||||
if base2 != upper {
|
||||
if _, ok := sseBinTable[base2]; ok {
|
||||
return true
|
||||
}
|
||||
}
|
||||
switch base2 {
|
||||
case "MOV", "MOVD",
|
||||
"ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB",
|
||||
"TEST",
|
||||
"LEA",
|
||||
"INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV",
|
||||
"SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR",
|
||||
"BT", "BTS", "BTR", "BTC",
|
||||
"XCHG", "CMPXCHG", "XADD", "CRC32", "ADCX", "ADOX",
|
||||
"MOVS", "STOS",
|
||||
"IMUL", "IMUL3",
|
||||
"PUSH", "POP",
|
||||
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
|
||||
"BSWAP",
|
||||
"PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2",
|
||||
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
|
||||
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX",
|
||||
"CVTSL2SD", "CVTSQ2SD",
|
||||
"CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S",
|
||||
"FMOVD",
|
||||
"MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||
return true
|
||||
}
|
||||
// Full-name dispatches the size split would eat (a trailing width
|
||||
// letter that is part of the mnemonic).
|
||||
switch upper {
|
||||
case "PMOVMSKB":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
+468
-17
@@ -5,6 +5,8 @@ package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
@@ -21,6 +23,39 @@ func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
|
||||
type enc struct {
|
||||
out []byte
|
||||
patches []encPatch // disp32 fields awaiting static-symbol resolution
|
||||
|
||||
// FloatPool collects the pooled constants the floating-point
|
||||
// immediates reference, in first-use order.
|
||||
floatPool []floatPoolEntry
|
||||
floatPoolSeen map[string]bool
|
||||
}
|
||||
|
||||
// floatPoolEntry is one pooled floating-point constant: the symbol name
|
||||
// the emitted RIP-relative load refers to and its IEEE-754 bytes.
|
||||
type floatPoolEntry struct {
|
||||
name string
|
||||
data []byte
|
||||
}
|
||||
|
||||
// addFloatPool records a pooled constant, deduplicated by symbol name.
|
||||
func (e *enc) addFloatPool(name string, bits uint64, width int) {
|
||||
if e.floatPoolSeen == nil {
|
||||
e.floatPoolSeen = map[string]bool{}
|
||||
}
|
||||
if e.floatPoolSeen[name] {
|
||||
return
|
||||
}
|
||||
e.floatPoolSeen[name] = true
|
||||
data := make([]byte, width)
|
||||
for i := range width {
|
||||
data[i] = byte(bits >> (8 * i))
|
||||
}
|
||||
e.floatPool = append(e.floatPool, floatPoolEntry{name: name, data: data})
|
||||
}
|
||||
|
||||
// floatPoolList returns the pooled constants in first-use order.
|
||||
func (e *enc) floatPoolList() []floatPoolEntry {
|
||||
return e.floatPool
|
||||
}
|
||||
|
||||
// encPatch marks a 4-byte displacement field in enc.out that must receive the
|
||||
@@ -29,6 +64,7 @@ type encPatch struct {
|
||||
off int
|
||||
name string
|
||||
addend int64
|
||||
tls bool // a TLS slot offset: the patch is R_TLSLE with no symbol
|
||||
}
|
||||
|
||||
func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
@@ -40,14 +76,74 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
return e.encodeRet()
|
||||
case upper == "NOP":
|
||||
return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1})
|
||||
case upper == "CALL":
|
||||
return e.encodeJmpRel(ops, []byte{0xE8})
|
||||
case upper == "JMP":
|
||||
return e.encodeJmpRel(ops, []byte{0xE9})
|
||||
case upper == "CALL" || upper == "JMP":
|
||||
// Through a register or memory: FF /2 (CALL) or FF /4 (JMP).
|
||||
// Anything else is a rel32 against a label resolved by the assembler.
|
||||
if len(ops) == 1 {
|
||||
switch ops[0].(type) {
|
||||
case Reg, Mem:
|
||||
return e.encodeIndirectBranch(upper, ops)
|
||||
}
|
||||
}
|
||||
opcode := []byte{0xE8}
|
||||
if upper == "JMP" {
|
||||
opcode = []byte{0xE9}
|
||||
}
|
||||
return e.encodeJmpRel(ops, opcode)
|
||||
}
|
||||
if cc, ok := condCode(upper); ok {
|
||||
return e.encodeJcc(cc, ops)
|
||||
}
|
||||
// No-operand system and string-control instructions (CPUID, RDTSC,
|
||||
// SYSCALL, the fences, UNDEF, …).
|
||||
if op, ok := noOperandTable[upper]; ok {
|
||||
if len(ops) != 0 {
|
||||
return fmt.Errorf("%s takes no operands, got %d", upper, len(ops))
|
||||
}
|
||||
return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
|
||||
}
|
||||
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
|
||||
// are rejected by go tool asm in 64-bit mode, so they stay unsupported.
|
||||
switch upper {
|
||||
case "POPFQ":
|
||||
if len(ops) != 0 {
|
||||
return fmt.Errorf("POPFQ takes no operands, got %d", len(ops))
|
||||
}
|
||||
return e.emit(&instr{opcode: []byte{0x9D}, modrm: -1, sib: -1})
|
||||
case "PUSHFQ":
|
||||
if len(ops) != 0 {
|
||||
return fmt.Errorf("PUSHFQ takes no operands, got %d", len(ops))
|
||||
}
|
||||
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
|
||||
case "INT":
|
||||
return e.encodeInt(ops)
|
||||
case "LDMXCSR":
|
||||
return e.encodeMxcsr(2, ops)
|
||||
case "STMXCSR":
|
||||
return e.encodeMxcsr(3, ops)
|
||||
// CMPSD is the scalar double compare, whose predicate immediate comes
|
||||
// LAST in Plan 9 order (src, dst, $imm).
|
||||
case "CMPSD":
|
||||
return e.encodeCmpsd(ops)
|
||||
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
|
||||
case "SHA256RNDS2":
|
||||
return e.encodeSha256rnds2(ops)
|
||||
// BYTE, WORD, LONG and QUAD write the immediate into the text stream
|
||||
// itself: 1, 2, 4 or 8 literal bytes, little-endian. END is accepted
|
||||
// and ignored. ADJSP adjusts SP by the immediate, sign-chosen between
|
||||
// the SUBQ and ADDQ forms.
|
||||
case "BYTE", "WORD", "LONG", "QUAD":
|
||||
return e.encodeData(upper, ops)
|
||||
case "END":
|
||||
return e.encodeEnd(ops)
|
||||
case "ADJSP":
|
||||
return e.encodeAdjsp(ops)
|
||||
// The runtime's bookkeeping statements carry no text bytes: go tool asm
|
||||
// records FUNCDATA and PCDATA in the program list only, so the encoded
|
||||
// body shows nothing, on every architecture.
|
||||
case "FUNCDATA", "PCDATA":
|
||||
return e.encodeFuncdata(upper, ops)
|
||||
}
|
||||
|
||||
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
||||
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
|
||||
@@ -57,7 +153,9 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
|
||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
||||
isEvexPrefGather(base) ||
|
||||
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
|
||||
return e.encodeVec(base, ops, sfx)
|
||||
}
|
||||
if sfx.any() {
|
||||
@@ -76,37 +174,345 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
if size == 0 {
|
||||
size = 8 // default operand size in 64-bit mode (e.g. PUSHQ)
|
||||
}
|
||||
// Legacy SSE imm8 shuffles whose names end in W/H (PSHUFLW,
|
||||
// PSHUFHW) must dispatch BEFORE the size-suffix split, and the
|
||||
// others ride along.
|
||||
if m, ok := sseShufTable[upper]; ok {
|
||||
return e.encodeSSEShuf(m, ops)
|
||||
}
|
||||
// Legacy SSE packed binaries dispatch on the full name: the packed
|
||||
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
|
||||
// which the size split must not eat. A floating-point immediate
|
||||
// rewrites into a pooled-constant read on the scalar members.
|
||||
if m, ok := sseBinTable[upper]; ok {
|
||||
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||
return e.encodeSSEFloatBin(upper, m, f, ops)
|
||||
}
|
||||
return e.encodeSSEBin(m, ops)
|
||||
}
|
||||
if m, ok := sseBinTable[base]; ok {
|
||||
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||
return e.encodeSSEFloatBin(upper, m, f, ops)
|
||||
}
|
||||
return e.encodeSSEBin(m, ops)
|
||||
}
|
||||
// The imm8-controlled legacy instructions, the lane extracts and inserts
|
||||
// and the packed integer shifts all dispatch on the full name: a trailing
|
||||
// width letter here belongs to the mnemonic, not to the size split.
|
||||
if m, ok := sseImm3Table[upper]; ok {
|
||||
return e.encodeSSEImm3(m, ops)
|
||||
}
|
||||
if m, ok := sseExtractTable[upper]; ok {
|
||||
return e.encodeSSEExtract(m, ops)
|
||||
}
|
||||
if m, ok := sseInsertTable[upper]; ok {
|
||||
return e.encodeSSEInsert(m, ops)
|
||||
}
|
||||
if _, ok := sseShiftImm[upper]; ok {
|
||||
return e.encodeSSEShift(upper, ops)
|
||||
}
|
||||
// PMOVMSKB ends in a width letter the size split would eat, so it
|
||||
// dispatches on the full name like the packed binaries above.
|
||||
if upper == "PMOVMSKB" {
|
||||
return e.encodePmovmskb(upper, ops)
|
||||
}
|
||||
switch base {
|
||||
case "MOV":
|
||||
return e.encodeMov(ops, size)
|
||||
case "ADD", "SUB", "AND", "OR", "XOR", "CMP":
|
||||
// MOVD is the Go assembler's alias of MOVQ: the same byte forms, 64-bit
|
||||
// REX.W and all.
|
||||
case "MOVD":
|
||||
return e.encodeMov(ops, 8)
|
||||
case "ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB":
|
||||
return e.encodeALU(aluOp[base], ops, size)
|
||||
case "TEST":
|
||||
return e.encodeTest(ops, size)
|
||||
case "LEA":
|
||||
return e.encodeLea(ops, size)
|
||||
case "INC", "DEC", "NEG", "NOT":
|
||||
case "INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV":
|
||||
return e.encodeUnary(unaryOp[base], ops, size)
|
||||
case "SHL", "SHR", "SAR":
|
||||
return e.encodeShift(shiftOp[base], ops, size)
|
||||
case "SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR":
|
||||
return e.encodeShift(base, ops, size)
|
||||
case "BT", "BTS", "BTR", "BTC":
|
||||
return e.encodeBitTest(base, ops, size)
|
||||
case "XCHG":
|
||||
return e.encodeExchange(ops, size)
|
||||
case "CMPXCHG":
|
||||
return e.encodeRegRegOp(0xB0, 0xB1, base, ops, size)
|
||||
case "XADD":
|
||||
return e.encodeRegRegOp(0xC0, 0xC1, base, ops, size)
|
||||
case "CRC32":
|
||||
return e.encodeCrc32(ops, size)
|
||||
case "ADCX":
|
||||
return e.encodeCarryExt(0x66, ops, size)
|
||||
case "ADOX":
|
||||
return e.encodeCarryExt(0xF3, ops, size)
|
||||
case "MOVS", "STOS":
|
||||
return e.encodeStringOp(base, ops, size)
|
||||
case "IMUL", "IMUL3":
|
||||
return e.encodeImul(ops, size)
|
||||
case "PUSH":
|
||||
return e.encodePushPop(ops, true)
|
||||
return e.encodePushPop(ops, size, true)
|
||||
case "POP":
|
||||
return e.encodePushPop(ops, false)
|
||||
case "LZCNT", "TZCNT":
|
||||
return e.encodePushPop(ops, size, false)
|
||||
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
|
||||
return e.encodeCount(base, ops, size)
|
||||
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX":
|
||||
case "BSWAP":
|
||||
return e.encodeBswap(ops, size)
|
||||
case "PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2":
|
||||
return e.encodePrefetch(base, ops)
|
||||
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
|
||||
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX":
|
||||
return e.encodeMovExtend(base, ops)
|
||||
case "CVTSL2SD", "CVTSQ2SD":
|
||||
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
|
||||
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||
case "CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S":
|
||||
return e.encodeCvtInt(base, ops, size)
|
||||
case "FMOVD":
|
||||
return e.encodeFmov(ops)
|
||||
case "MOVSD", "MOVSS":
|
||||
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||
return e.encodeSSEFloatMove(upper, f, ops)
|
||||
}
|
||||
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD":
|
||||
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||
}
|
||||
return fmt.Errorf("unsupported instruction %q", mnem)
|
||||
}
|
||||
|
||||
// encodePrefetch emits the 0F 18 /r prefetch hints: the reg field selects
|
||||
// the locality (NTA=0, T0=1, T1=2, T2=3) and the single operand is memory.
|
||||
func (e *enc) encodePrefetch(base string, ops []Operand) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("%s expects one memory operand", base)
|
||||
}
|
||||
m, ok := ops[0].(Mem)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s requires a memory operand", base)
|
||||
}
|
||||
i := newInstr(0, []byte{0x0F, 0x18})
|
||||
if err := setMem(i, prefetchVariant[base], m); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
var prefetchVariant = map[string]int{
|
||||
"PREFETCHNTA": 0,
|
||||
"PREFETCHT0": 1,
|
||||
"PREFETCHT1": 2,
|
||||
"PREFETCHT2": 3,
|
||||
}
|
||||
|
||||
// dataWidth is the literal byte count of each data-emission pseudo-op.
|
||||
var dataWidth = map[string]int{
|
||||
"BYTE": 1,
|
||||
"WORD": 2,
|
||||
"LONG": 4,
|
||||
"QUAD": 8,
|
||||
}
|
||||
|
||||
// encodeData emits the literal-data pseudo-ops: BYTE, WORD, LONG and QUAD
|
||||
// write the immediate into the text stream as 1, 2, 4 or 8 bytes,
|
||||
// little-endian, with no opcode lookup. The value is truncated to the
|
||||
// width rather than range-checked, exactly as go tool asm behaves (BYTE
|
||||
// $0x1FF emits FF, WORD $0x12345 emits 45 23, both without an error), and
|
||||
// exactly one immediate is accepted: the toolchain rejects a list such as
|
||||
// BYTE $1, $2, $3.
|
||||
func (e *enc) encodeData(mnem string, ops []Operand) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("%s expects 1 immediate operand, got %d", mnem, len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s requires an integer immediate", mnem)
|
||||
}
|
||||
width := dataWidth[mnem]
|
||||
out := make([]byte, width)
|
||||
u := uint64(imm)
|
||||
for i := range width {
|
||||
out[i] = byte(u >> (8 * i))
|
||||
}
|
||||
e.out = append(e.out, out...)
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeFuncdata accepts-and-ignores the runtime bookkeeping statements:
|
||||
// FUNCDATA $n, sym(SB) and PCDATA $n, $m. go tool asm emits no text bytes
|
||||
// for either (the entries live in the object's ancillary tables, not the
|
||||
// function body), and the operand shapes it takes are exactly these: an
|
||||
// integer count first, then a symbol reference for FUNCDATA and an integer
|
||||
// value for PCDATA. The other architectures accept-and-ignore the same
|
||||
// statements; amd64 now matches.
|
||||
func (e *enc) encodeFuncdata(upper string, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
if _, ok := ops[0].(Imm); !ok {
|
||||
return fmt.Errorf("%s: first operand must be an integer immediate", upper)
|
||||
}
|
||||
switch upper {
|
||||
case "FUNCDATA":
|
||||
if _, ok := ops[1].(sbMem); !ok {
|
||||
return fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
|
||||
}
|
||||
case "PCDATA":
|
||||
if _, ok := ops[1].(Imm); !ok {
|
||||
return fmt.Errorf("PCDATA: second operand must be an integer immediate")
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeEnd accepts-and-ignores END. go tool asm drops the statement
|
||||
// entirely: the AEND Prog is skipped when the program list is flushed, so
|
||||
// the statements after an END still belong to the same function and the
|
||||
// encoded body carries no trace of it, whatever operands follow the name
|
||||
// (the toolchain takes END $0 and END AX alike). Zero bytes, no effect.
|
||||
func (e *enc) encodeEnd(ops []Operand) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeAdjsp emits ADJSP $imm: a positive value is SUBQ $imm, SP, a
|
||||
// negative one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude
|
||||
// picks (the same selection subSP and addSP make for the frame). go tool
|
||||
// asm refuses ADJSP $0 outright, so a zero value is an error here too; the
|
||||
// statement's effect on the SP balance is checked by the function-level
|
||||
// assembly (checkAdjspBalance), as the toolchain's push/pop walk does.
|
||||
func (e *enc) encodeAdjsp(ops []Operand) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("ADJSP expects 1 immediate operand, got %d", len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("ADJSP requires an integer immediate")
|
||||
}
|
||||
switch v := int(imm); {
|
||||
case v > 0:
|
||||
e.out = append(e.out, subSP(v)...)
|
||||
case v < 0:
|
||||
e.out = append(e.out, addSP(-v)...)
|
||||
default:
|
||||
return fmt.Errorf("ADJSP $0 has no encoding")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// --- floating-point immediates ----------------------------------------------
|
||||
|
||||
// sseFloatImm lists the mnemonics whose first operand may be a floating-point
|
||||
// immediate, the set go tool asm rewrites into a pooled-constant read: the
|
||||
// scalar moves, the four scalar arithmetic pairs and the scalar compares.
|
||||
// The packed members and the uniform forms (MAXSD, MINSD, SQRTSD, CMPSD)
|
||||
// reject the immediate in the toolchain and are absent here on purpose.
|
||||
var sseFloatImm = map[string]bool{
|
||||
"MOVSD": true, "MOVSS": true,
|
||||
"ADDSD": true, "ADDSS": true,
|
||||
"SUBSD": true, "SUBSS": true,
|
||||
"MULSD": true, "MULSS": true,
|
||||
"DIVSD": true, "DIVSS": true,
|
||||
"COMISD": true, "COMISS": true,
|
||||
"UCOMISD": true, "UCOMISS": true,
|
||||
}
|
||||
|
||||
// floatImmOperand reports whether the operand list opens with a
|
||||
// floating-point immediate in the two-operand spelling (imm, dst).
|
||||
func floatImmOperand(ops []Operand) (FloatImm, bool) {
|
||||
if len(ops) != 2 {
|
||||
return FloatImm{}, false
|
||||
}
|
||||
f, ok := ops[0].(FloatImm)
|
||||
return f, ok
|
||||
}
|
||||
|
||||
// floatPoolValue evaluates a floating-point immediate at the width its
|
||||
// mnemonic encodes and names the pool constant the toolchain synthesises:
|
||||
// $f64.<16 hex> for the doubles, $f32.<8 hex> for the singles (the float32
|
||||
// rounding of the parsed value). The name carries the IEEE-754 bits; the
|
||||
// section holds them little-endian.
|
||||
func floatPoolValue(mnem string, f FloatImm) (bits uint64, name string, err error) {
|
||||
v, err := strconv.ParseFloat(f.Text, 64)
|
||||
if err != nil {
|
||||
return 0, "", fmt.Errorf("invalid floating-point immediate %q", f.Text)
|
||||
}
|
||||
if f.Neg {
|
||||
v = -v
|
||||
}
|
||||
if strings.HasSuffix(mnem, "D") {
|
||||
bits = math.Float64bits(v)
|
||||
return bits, fmt.Sprintf("$f64.%016x", bits), nil
|
||||
}
|
||||
bits = uint64(math.Float32bits(float32(v)))
|
||||
return bits, fmt.Sprintf("$f32.%08x", bits), nil
|
||||
}
|
||||
|
||||
// encodeSSEFloatMove encodes MOVSD/MOVSS with a floating-point immediate
|
||||
// source. A positive zero needs no memory read: the toolchain emits
|
||||
// XORPS dst, dst. Anything else loads the pooled constant RIP-relative
|
||||
// ($f64.<hex>(SB) / $f32.<hex>(SB)), the displacement a patch site the
|
||||
// file-level layout or the linker resolves.
|
||||
func (e *enc) encodeSSEFloatMove(mnem string, f FloatImm, ops []Operand) error {
|
||||
if !sseFloatImm[mnem] {
|
||||
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
|
||||
}
|
||||
dst, ok := ops[1].(Reg)
|
||||
if !ok || !dst.isVec() {
|
||||
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||
}
|
||||
bits, name, err := floatPoolValue(mnem, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
e.addFloatPool(name, bits, mwidth(mnem))
|
||||
if bits == 0 {
|
||||
i := &instr{opcode: []byte{0x0F, 0x57}, modrm: -1, sib: -1} // XORPS
|
||||
if err := setRM(i, dst, dst, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
m := sseMoveTable[mnem]
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.load}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeSSEFloatBin encodes the scalar arithmetic and compare mnemonics with
|
||||
// a floating-point immediate source: the constant is read from the pool into
|
||||
// the instruction's r/m side (reg = destination), the rewrite go tool asm
|
||||
// performs at the source level.
|
||||
func (e *enc) encodeSSEFloatBin(mnem string, m sseBin, f FloatImm, ops []Operand) error {
|
||||
if !sseFloatImm[mnem] {
|
||||
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
|
||||
}
|
||||
dst, ok := ops[1].(Reg)
|
||||
if !ok || !dst.isVec() {
|
||||
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||
}
|
||||
bits, name, err := floatPoolValue(mnem, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
e.addFloatPool(name, bits, mwidth(mnem))
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// mwidth returns the operand width a scalar SSE mnemonic encodes: the double
|
||||
// spellings end in D, the single spellings in S.
|
||||
func mwidth(mnem string) int {
|
||||
if strings.HasSuffix(mnem, "D") {
|
||||
return 8
|
||||
}
|
||||
return 4
|
||||
}
|
||||
|
||||
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
|
||||
func splitSize(upper string) (base string, size int) {
|
||||
if upper == "" {
|
||||
@@ -136,7 +542,7 @@ func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
|
||||
if ss, ok := scatterTable[upper]; ok {
|
||||
return e.encodeScatter(upper, ss, ops, sfx)
|
||||
}
|
||||
if upper == "KMOVW" || upper == "KMOVQ" {
|
||||
if upper == "KMOVW" || upper == "KMOVQ" || upper == "KMOVB" || upper == "KMOVD" {
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
||||
}
|
||||
@@ -173,6 +579,7 @@ type instr struct {
|
||||
disp []byte
|
||||
imm []byte
|
||||
sb *sbRef // static-symbol displacement in disp, awaiting resolution
|
||||
tls bool // the displacement is a TLS slot offset, patched R_TLSLE
|
||||
}
|
||||
|
||||
// sbRef records that an instruction's displacement refers to a static symbol
|
||||
@@ -215,6 +622,9 @@ func (e *enc) emit(i *instr) error {
|
||||
if i.sb != nil {
|
||||
e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend})
|
||||
}
|
||||
if i.tls {
|
||||
e.patches = append(e.patches, encPatch{off: len(e.out), tls: true})
|
||||
}
|
||||
e.out = append(e.out, i.disp...)
|
||||
e.out = append(e.out, i.imm...)
|
||||
return nil
|
||||
@@ -241,7 +651,7 @@ func setRM(i *instr, reg Reg, rm Operand, opSize int) error {
|
||||
}
|
||||
|
||||
// setRMDigit fills in the ModR/M for an instruction whose reg field is an
|
||||
// opcode /digit extension (0–7), which carries none of the register REX rules.
|
||||
// opcode /digit extension (0-7), which carries none of the register REX rules.
|
||||
func setRMDigit(i *instr, digit int, rm Operand, opSize int) error {
|
||||
return setRMReg(i, digit, false, false, rm, opSize)
|
||||
}
|
||||
@@ -269,12 +679,30 @@ func setRMReg(i *instr, regField int, rexR, regForced bool, rm Operand, opSize i
|
||||
i.disp = le32(0)
|
||||
i.sb = &sbRef{name: r.name, addend: r.addend}
|
||||
return nil
|
||||
case TLSMem:
|
||||
// off(TLS): the segment-prefixed absolute access, mod=00 with the
|
||||
// SIB escape's disp32 absolute form. The displacement is the TLS
|
||||
// slot offset, patched by the linker's TLS relocation.
|
||||
i.prefix = r.Seg
|
||||
i.modrm = 0x04 | regField<<3
|
||||
i.sib = 0x25
|
||||
i.disp = le32(r.Disp)
|
||||
i.tls = true
|
||||
return nil
|
||||
case SegAbs:
|
||||
// 0x30(GS): the segment override with the SIB escape's disp32
|
||||
// absolute form, no relocation.
|
||||
setSegAbs(i, regField, r)
|
||||
return nil
|
||||
default:
|
||||
return fmt.Errorf("invalid r/m operand %T", rm)
|
||||
}
|
||||
}
|
||||
|
||||
func setMem(i *instr, regField int, m Mem) error {
|
||||
if m.Seg != 0 {
|
||||
i.prefix = m.Seg
|
||||
}
|
||||
modrm, sib, disp, xBit, bBit, err := memComponents(regField, m)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -287,16 +715,39 @@ func setMem(i *instr, regField int, m Mem) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// setSegAbs assembles a segment-absolute operand, 0x30(GS): the segment
|
||||
// override with the mod=00 SIB escape's disp32 absolute form and no
|
||||
// relocation.
|
||||
func setSegAbs(i *instr, regField int, m SegAbs) {
|
||||
i.prefix = m.Seg
|
||||
i.modrm = 0x04 | regField<<3
|
||||
i.sib = 0x25
|
||||
i.disp = le32(m.Disp)
|
||||
}
|
||||
|
||||
// memComponents computes the ModR/M byte (with the given reg field), the SIB
|
||||
// byte (-1 if none), the displacement bytes, and the high index/base bits, for
|
||||
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
|
||||
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
|
||||
sib = -1
|
||||
// A displacement wider than int32 fits no encoding form; truncating it
|
||||
// would address a different location, and go tool asm reports "offset
|
||||
// too large" for the same operand.
|
||||
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
|
||||
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
|
||||
}
|
||||
// RIP-relative: neither base nor index.
|
||||
if !m.HasBase && !m.HasIndex {
|
||||
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
|
||||
}
|
||||
|
||||
// The SIB scale field only encodes 1/2/4/8; the Go assembler rejects
|
||||
// anything else ("bad scale: 16"), so a silent fallback to scale 1 here
|
||||
// would mis-assemble the operand instead of reporting it.
|
||||
if m.HasIndex && m.Scale != 1 && m.Scale != 2 && m.Scale != 4 && m.Scale != 8 {
|
||||
return 0, -1, nil, 0, 0, fmt.Errorf("bad scale: %d", m.Scale)
|
||||
}
|
||||
|
||||
needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4)
|
||||
|
||||
var mod int
|
||||
@@ -369,7 +820,7 @@ func le16(v int64) []byte {
|
||||
func le64(v int64) []byte {
|
||||
u := uint64(v)
|
||||
b := make([]byte, 8)
|
||||
for i := 0; i < 8; i++ {
|
||||
for i := range 8 {
|
||||
b[i] = byte(u >> (8 * i))
|
||||
}
|
||||
return b
|
||||
|
||||
+931
-4
File diff suppressed because it is too large
Load Diff
+794
-142
File diff suppressed because it is too large
Load Diff
+291
-13
@@ -16,7 +16,7 @@ import (
|
||||
// kernels use: NDS arithmetic, immediate and variable shifts, shuffles with
|
||||
// an immediate, lane extracts, narrowing stores, broadcasts from a GPR or
|
||||
// memory, mask destinations, mask moves, disp8×N compression and the 5-bit
|
||||
// register fields (X/Y 16–31, Z 0–31).
|
||||
// register fields (X/Y 16-31, Z 0-31).
|
||||
func TestEvexGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
@@ -38,6 +38,15 @@ func TestEvexGroundTruth(t *testing.T) {
|
||||
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
|
||||
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
|
||||
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
|
||||
// The qword OR spelling always encodes through EVEX.
|
||||
{"VPORQ Y0,Y1,Y2", "VPORQ", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "62f1f528ebd0"},
|
||||
{"VPORQ X0,X1,X2", "VPORQ", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f1f508ebd0"},
|
||||
// Byte permute and population count.
|
||||
{"VPERMI2B X0,X1,X2", "VPERMI2B", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f2750875d0"},
|
||||
{"VPOPCNTB X0,X1", "VPOPCNTB", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0854c8"},
|
||||
{"VPOPCNTD X0,X1", "VPOPCNTD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0855c8"},
|
||||
{"VPOPCNTD Y0,Y1", "VPOPCNTD", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "62f27d2855c8"},
|
||||
{"VPOPCNTQ X0,X1", "VPOPCNTQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f2fd0855c8"},
|
||||
// Align (NDS + imm8).
|
||||
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
|
||||
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
|
||||
@@ -52,13 +61,23 @@ func TestEvexGroundTruth(t *testing.T) {
|
||||
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
|
||||
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
|
||||
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
|
||||
{"KMOVB K1,K2", "KMOVB", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f990d1"},
|
||||
{"KMOVB AX,K1", "KMOVB", []Operand{AX, vreg(t, "K1")}, "c5f992c8"},
|
||||
{"KMOVB K1,AX", "KMOVB", []Operand{vreg(t, "K1"), AX}, "c5f993c1"},
|
||||
{"KMOVB K1,(AX)", "KMOVB", []Operand{vreg(t, "K1"), Ptr(AX, 0, 1)}, "c5f99108"},
|
||||
{"KMOVD K1,K2", "KMOVD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f990d1"},
|
||||
{"KMOVD AX,K1", "KMOVD", []Operand{AX, vreg(t, "K1")}, "c5fb92c8"},
|
||||
{"KMOVD K1,AX", "KMOVD", []Operand{vreg(t, "K1"), AX}, "c5fb93c1"},
|
||||
{"KMOVD K1,(AX)", "KMOVD", []Operand{vreg(t, "K1"), Ptr(AX, 0, 4)}, "c4e1f99108"},
|
||||
{"KMOVB (AX),K1", "KMOVB", []Operand{Ptr(AX, 0, 1), vreg(t, "K1")}, "c5f99008"},
|
||||
{"KMOVQ (AX),K1", "KMOVQ", []Operand{Ptr(AX, 0, 8), vreg(t, "K1")}, "c4e1f89008"},
|
||||
// Moves, incl. disp8×N (64 for a 512-bit operand).
|
||||
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
|
||||
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
|
||||
{"VMOVDQU32 16(SI)(R15*4),Z4", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 16, 64), vreg(t, "Z4")}, "62b17e486fa4be10000000"},
|
||||
{"VMOVDQU32 Z0,4(SI)(AX*1)", "VMOVDQU32", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f17e487f840604000000"},
|
||||
{"VMOVDQU32 Z3,(DI)(R15*4)", "VMOVDQU32", []Operand{vreg(t, "Z3"), Idx(DI, vreg(t, "R15"), 4, 0, 64)}, "62b17e487f1cbf"},
|
||||
// VMOVDQU64 — the W1 qword variant.
|
||||
// VMOVDQU64; the W1 qword variant.
|
||||
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
|
||||
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
|
||||
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
|
||||
@@ -77,7 +96,7 @@ func TestEvexGroundTruth(t *testing.T) {
|
||||
{"VPSHUFB Z1,Z2,Z3", "VPSHUFB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4800d9"},
|
||||
{"VMOVDQU8 Z1,Z2", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17f487fca"},
|
||||
{"VMOVDQU16 Z1,Z2", "VMOVDQU16", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff487fca"},
|
||||
// Indices 16–31: rm[4] rides in X̄ for register operands.
|
||||
// Indices 16-31: rm[4] rides in X̄ for register operands.
|
||||
{"VPSHUFD $1,X16,X17", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}, "62a17d0870c801"},
|
||||
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
|
||||
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
|
||||
@@ -96,7 +115,7 @@ func TestEvexGroundTruth(t *testing.T) {
|
||||
{"VPBROADCASTD 4(SI),Z10", "VPBROADCASTD", []Operand{Ptr(SI, 4, 4), vreg(t, "Z10")}, "62727d48585601"},
|
||||
{"VPBROADCASTQ R8,X31", "VPBROADCASTQ", []Operand{vreg(t, "R8"), vreg(t, "X31")}, "6242fd087cf8"},
|
||||
{"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"},
|
||||
// Register indices 16–31 exist only in EVEX encodings.
|
||||
// Register indices 16-31 exist only in EVEX encodings.
|
||||
{"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"},
|
||||
// Packed double arithmetic / unpack (EVEX forms carry W=1).
|
||||
{"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"},
|
||||
@@ -107,7 +126,7 @@ func TestEvexGroundTruth(t *testing.T) {
|
||||
{"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"},
|
||||
{"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"},
|
||||
{"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"},
|
||||
// VMOVDDUP — duplicate the low double; disp8×N = 64 at 512 bits, and
|
||||
// VMOVDDUP; duplicate the low double; disp8×N = 64 at 512 bits, and
|
||||
// X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄).
|
||||
{"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"},
|
||||
{"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"},
|
||||
@@ -150,7 +169,7 @@ func TestEvexGroundTruth(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexMasking checks the AVX-512 mask operand (K1–K7, placed freely among
|
||||
// TestEvexMasking checks the AVX-512 mask operand (K1-K7, placed freely among
|
||||
// the operands) and the .Z zeroing suffix, byte for byte against the Go
|
||||
// assembler.
|
||||
func TestEvexMasking(t *testing.T) {
|
||||
@@ -241,11 +260,11 @@ func TestEvexMasking(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set — ternary
|
||||
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set; ternary
|
||||
// logic, lane shuffles/inserts/extracts, compares with a K destination,
|
||||
// permutes, the wider integer families, expand/compress, broadcasts,
|
||||
// rotates and word shifts, the opmask instructions, the EVEX suffixes
|
||||
// (rounding/SAE/broadcast) and the aligned/scalar moves — byte for byte
|
||||
// (rounding/SAE/broadcast) and the aligned/scalar moves; byte for byte
|
||||
// against the Go assembler.
|
||||
func TestEvexExtendedGroundTruth(t *testing.T) {
|
||||
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
|
||||
@@ -275,7 +294,7 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
|
||||
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
|
||||
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
|
||||
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
|
||||
// Packed single arithmetic (same opcodes, no mandatory prefix) —
|
||||
// Packed single arithmetic (same opcodes, no mandatory prefix);
|
||||
// ZMM, YMM and XMM widths, rounding and broadcast.
|
||||
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
|
||||
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
|
||||
@@ -319,6 +338,43 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
|
||||
{"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"},
|
||||
{"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"},
|
||||
{"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"},
|
||||
// Completed opmask families (ANDN, NOT, OR/XOR word+qword, TEST,
|
||||
// word-width shifts; byte-exact against go tool asm).
|
||||
{"KANDNW", "KANDNW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec42d9"},
|
||||
{"KANDNB", "KANDNB", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c5d542f4"},
|
||||
{"KANDND", "KANDND", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ed42d9"},
|
||||
{"KANDNQ", "KANDNQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d442f4"},
|
||||
{"KANDD", "KANDD", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ed41d9"},
|
||||
{"KADDD", "KADDD", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d54af4"},
|
||||
{"KNOTW", "KNOTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f844d1"},
|
||||
{"KNOTD", "KNOTD", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c4e1f944e3"},
|
||||
{"KNOTQ", "KNOTQ", []Operand{vreg(t, "K5"), vreg(t, "K6")}, "c4e1f844f5"},
|
||||
{"KORW", "KORW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec45d9"},
|
||||
{"KORQ", "KORQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d445f4"},
|
||||
{"KXNORB", "KXNORB", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ed46d9"},
|
||||
{"KXORW", "KXORW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec47d9"},
|
||||
{"KXORQ", "KXORQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d447f4"},
|
||||
{"KORTESTW", "KORTESTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f898d1"},
|
||||
{"KORTESTB", "KORTESTB", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c5f998e3"},
|
||||
{"KORTESTQ", "KORTESTQ", []Operand{vreg(t, "K5"), vreg(t, "K6")}, "c4e1f898f5"},
|
||||
{"KTESTW", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f899d1"},
|
||||
{"KTESTD", "KTESTD", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c4e1f999e3"},
|
||||
{"KSHIFTLB", "KSHIFTLB", []Operand{Imm(1), vreg(t, "K1"), vreg(t, "K2")}, "c4e37932d101"},
|
||||
{"KSHIFTLD", "KSHIFTLD", []Operand{Imm(2), vreg(t, "K3"), vreg(t, "K4")}, "c4e37933e302"},
|
||||
{"KSHIFTLQ", "KSHIFTLQ", []Operand{Imm(3), vreg(t, "K5"), vreg(t, "K6")}, "c4e3f933f503"},
|
||||
{"KSHIFTRB", "KSHIFTRB", []Operand{Imm(4), vreg(t, "K1"), vreg(t, "K2")}, "c4e37930d104"},
|
||||
{"KSHIFTRW", "KSHIFTRW", []Operand{Imm(5), vreg(t, "K3"), vreg(t, "K4")}, "c4e3f930e305"},
|
||||
{"KSHIFTRQ", "KSHIFTRQ", []Operand{Imm(6), vreg(t, "K5"), vreg(t, "K6")}, "c4e3f931f506"},
|
||||
// Integer compares with an opmask destination (0F3A map, the
|
||||
// go-bzip2 partition kernel's classify instructions).
|
||||
{"VPCMPUB", "VPCMPUB", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X0"), vreg(t, "K1")}, "62f37d083ec901"},
|
||||
{"VPCMPB", "VPCMPB", []Operand{Imm(2), vreg(t, "Y2"), vreg(t, "Y3"), vreg(t, "K2")}, "62f365283fd202"},
|
||||
{"VPCMPUW", "VPCMPUW", []Operand{Imm(5), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3")}, "62f3ed483ed905"},
|
||||
{"VPCMPW", "VPCMPW", []Operand{Imm(6), vreg(t, "X3"), vreg(t, "X4"), vreg(t, "K4")}, "62f3dd083fe306"},
|
||||
{"VPCMPD", "VPCMPD", []Operand{Imm(0), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "K1")}, "62f36d281fc900"},
|
||||
{"VPCMPUD", "VPCMPUD", []Operand{Imm(1), vreg(t, "Z2"), vreg(t, "Z3"), vreg(t, "K2")}, "62f365481ed201"},
|
||||
{"VPCMPQ", "VPCMPQ", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K3")}, "62f3ed081fd902"},
|
||||
{"VPCMPUQ", "VPCMPUQ", []Operand{Imm(3), vreg(t, "Y3"), vreg(t, "Y4"), vreg(t, "K4")}, "62f3dd281ee303"},
|
||||
// Lane extract / insert.
|
||||
{"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"},
|
||||
{"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"},
|
||||
@@ -362,8 +418,8 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
|
||||
}
|
||||
|
||||
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
|
||||
// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef,
|
||||
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions —
|
||||
// tail of the EVEX set; reciprocals, rsqrt, getexp/getmant, scalef,
|
||||
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions;
|
||||
// plus gather/scatter with VSIB addressing, byte for byte against the Go
|
||||
// assembler.
|
||||
func TestEvexHelperGroundTruth(t *testing.T) {
|
||||
@@ -465,9 +521,9 @@ func TestEvexHelperGroundTruth(t *testing.T) {
|
||||
}
|
||||
|
||||
// TestEvexGprGroundTruth covers the scalar conversions between vector and
|
||||
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
|
||||
// general-purpose registers; the signed and truncated VCVT{,T}S{D,S}2SI
|
||||
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
|
||||
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
|
||||
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv; byte
|
||||
// for byte against the Go assembler, including memory sources and extended
|
||||
// GPRs.
|
||||
func TestEvexGprGroundTruth(t *testing.T) {
|
||||
@@ -638,6 +694,15 @@ func TestEvexErrors(t *testing.T) {
|
||||
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
|
||||
// VEX-only mnemonics reject registers only EVEX can encode.
|
||||
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
|
||||
// The scalar EVEX move matches its VEX twin and the Go assembler:
|
||||
// XMM↔memory only, never reg-reg and never a wider register (the
|
||||
// toolchain rejects every one of these shapes).
|
||||
{"VMOVSS X1,X2", "VMOVSS", []Operand{vreg(t, "X1"), vreg(t, "X2")}},
|
||||
{"VMOVSS X16,X2", "VMOVSS", []Operand{vreg(t, "X16"), vreg(t, "X2")}},
|
||||
{"VMOVSS Y1,(AX)", "VMOVSS", []Operand{vreg(t, "Y1"), Ptr(AX, 0, 4)}},
|
||||
{"VMOVSS Z1,Z2", "VMOVSS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}},
|
||||
{"VMOVSS Z1,(AX)", "VMOVSS", []Operand{vreg(t, "Z1"), Ptr(AX, 0, 4)}},
|
||||
{"VMOVSS (AX),Z2", "VMOVSS", []Operand{Ptr(AX, 0, 4), vreg(t, "Z2")}},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
@@ -656,3 +721,216 @@ func hexCompact(b []byte) string {
|
||||
}
|
||||
return string(out)
|
||||
}
|
||||
|
||||
// TestAvx512CorpusFamilies pins representative encodings of the AVX-512
|
||||
// families the toolchain's avx512enc corpus exercises: the bytes are the
|
||||
// go tool asm output for exactly these operands, and the same families are
|
||||
// covered end to end by the avx512_amd64.s differential kernel.
|
||||
func TestAvx512CorpusFamilies(t *testing.T) {
|
||||
vsib := func(base, idx string, scale int) Operand {
|
||||
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
|
||||
}
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// AES rounds (EVEX NDS, VEX twin routed by operand width).
|
||||
{"VAESDEC Z", "VAESDEC", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d48ded9"},
|
||||
// Integer VNNI and the bit algorithm group.
|
||||
{"VPDPBUSD", "VPDPBUSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f26d4a50d9"},
|
||||
{"VPOPCNTW", "VPOPCNTW", []Operand{vreg(t, "Z1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f2fd4b54d1"},
|
||||
{"VPCONFLICTD", "VPCONFLICTD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d49c4d1"},
|
||||
{"VPLZCNTQ masked", "VPLZCNTQ", []Operand{vreg(t, "Z7"), vreg(t, "K1"), vreg(t, "Z8")}, "6272fd4944c7"},
|
||||
{"VPERMT2B", "VPERMT2B", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f26d497dd9"},
|
||||
{"VPMULTISHIFTQB", "VPMULTISHIFTQB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f2ed4b83e1"},
|
||||
{"VDBPSADBW", "VDBPSADBW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z3")}, "62f36d4b42d903"},
|
||||
{"VPSHUFBITQMB", "VPSHUFBITQMB", []Operand{vreg(t, "Z9"), vreg(t, "Z10"), vreg(t, "K3")}, "62d22d488fd9"},
|
||||
{"VPTESTNMQ", "VPTESTNMQ", []Operand{vreg(t, "Z13"), vreg(t, "Z14"), vreg(t, "K5")}, "62d28e4827ed"},
|
||||
// Permutations: immediate and register counts.
|
||||
{"VALIGNQ", "VALIGNQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f3ed4903d903"},
|
||||
{"VPERMQ imm", "VPERMQ", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f3fd4a00d101"},
|
||||
{"VPERMQ reg", "VPERMQ", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K2"), vreg(t, "Z5")}, "62f2dd4a36eb"},
|
||||
{"VPERMPD reg", "VPERMPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4816d9"},
|
||||
{"VPERMILPS imm", "VPERMILPS", []Operand{Imm(5), vreg(t, "Z9"), vreg(t, "K2"), vreg(t, "Z10")}, "62537d4a04d105"},
|
||||
{"VPERMILPS reg", "VPERMILPS", []Operand{vreg(t, "Z11"), vreg(t, "Z12"), vreg(t, "K2"), vreg(t, "Z13")}, "62521d4a0ceb"},
|
||||
// Shifts: immediate, register-count and memory-count forms; the
|
||||
// count source carries its own XMM tuple width.
|
||||
{"VPSLLW imm mask", "VPSLLW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f16d4a71f103"},
|
||||
{"VPSLLD reg count", "VPSLLD", []Operand{vreg(t, "X1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49f2d9"},
|
||||
{"VPSLLDQ", "VPSLLDQ", []Operand{Imm(9), vreg(t, "Z7"), vreg(t, "Z8")}, "62f13d4873ff09"},
|
||||
{"VPSRLDQ mem", "VPSRLDQ", []Operand{Imm(11), Ptr(SI, 16, 16), vreg(t, "Z4")}, "62f15d48739e100000000b"},
|
||||
{"VPSRLVW", "VPSRLVW", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K1"), vreg(t, "Z5")}, "62f2dd4910eb"},
|
||||
// Conversions and shuffles with the F2 prefix and no prefix.
|
||||
{"VCVTUDQ2PS", "VCVTUDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f17f497ad1"},
|
||||
{"VSHUFPS", "VSHUFPS", []Operand{Imm(2), vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f15449c6f402"},
|
||||
// Gather and scatter prefetch hints (memory-only, /digit in reg).
|
||||
{"VGATHERPF0DPD", "VGATHERPF0DPD", []Operand{vreg(t, "K5"), vsib("R10", "Y29", 8)}, "6292fd45c60cea"},
|
||||
{"VSCATTERPF1DPS", "VSCATTERPF1DPS", []Operand{vreg(t, "K2"), vsib("R10", "Z28", 4)}, "62927d42c634a2"},
|
||||
// Opmask broadcasts and the K logic.
|
||||
{"VPBROADCASTMB2Q", "VPBROADCASTMB2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe482ad1"},
|
||||
{"VPBROADCASTMW2D", "VPBROADCASTMW2D", []Operand{vreg(t, "K3"), vreg(t, "Z4")}, "62f27e483ae3"},
|
||||
{"KUNPCKWD", "KUNPCKWD", []Operand{vreg(t, "K6"), vreg(t, "K4"), vreg(t, "K1")}, "c5dc4bce"},
|
||||
{"KADDB", "KADDB", []Operand{vreg(t, "K2"), vreg(t, "K3"), vreg(t, "K5")}, "c5e54aea"},
|
||||
// Lane extracts to general registers (EVEX and VEX routes).
|
||||
{"VPEXTRB", "VPEXTRB", []Operand{Imm(3), vreg(t, "X26"), AX}, "62637d0814d003"},
|
||||
{"VPEXTRD", "VPEXTRD", []Operand{Imm(1), vreg(t, "X26"), vreg(t, "R9")}, "62437d0816d101"},
|
||||
{"VPEXTRD vex", "VPEXTRD", []Operand{Imm(1), vreg(t, "X2"), DI}, "c4e37916d701"},
|
||||
{"VPINSRQ", "VPINSRQ", []Operand{Imm(1), DI, vreg(t, "X3"), vreg(t, "X4")}, "c4e3e122e701"},
|
||||
// Moves: masked unaligned, masked scalar register form, half moves
|
||||
// and non-temporal stores.
|
||||
{"VMOVUPS mask", "VMOVUPS", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f17c4a11cb"},
|
||||
{"VMOVSD 3op", "VMOVSD", []Operand{vreg(t, "X14"), vreg(t, "X5"), vreg(t, "K3"), vreg(t, "X22")}, "6231d70b11f6"},
|
||||
{"VMOVSS 3op", "VMOVSS", []Operand{vreg(t, "X18"), vreg(t, "X3"), vreg(t, "K2"), vreg(t, "X25")}, "6281660a11d1"},
|
||||
{"VMOVHPS insert", "VMOVHPS", []Operand{Ptr(SI, 0, 8), vreg(t, "X18"), vreg(t, "X19")}, "62e16c00161e"},
|
||||
{"VMOVHPS store", "VMOVHPS", []Operand{vreg(t, "X20"), Ptr(SI, 8, 8)}, "62e17c08176601"},
|
||||
{"VMOVLHPS", "VMOVLHPS", []Operand{vreg(t, "X16"), vreg(t, "X5"), vreg(t, "X17")}, "62a1540816c8"},
|
||||
{"VMOVNTDQ", "VMOVNTDQ", []Operand{vreg(t, "Z7"), Ptr(SI, 0, 64)}, "62f17d48e73e"},
|
||||
{"VMOVNTDQA", "VMOVNTDQA", []Operand{Ptr(SI, 64, 64), vreg(t, "Z8")}, "62727d482a4601"},
|
||||
{"VMOVNTPS", "VMOVNTPS", []Operand{vreg(t, "Z9"), Ptr(SI, 0, 64)}, "62717c482b0e"},
|
||||
// Scalar compares with and without the 66 prefix.
|
||||
{"VCOMISD", "VCOMISD", []Operand{vreg(t, "X5"), vreg(t, "X6")}, "c5f92ff5"},
|
||||
{"VUCOMISS", "VUCOMISS", []Operand{vreg(t, "X7"), vreg(t, "X8")}, "c5782ec7"},
|
||||
// Floating point helpers.
|
||||
{"VSQRTSD", "VSQRTSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f1ef0951d9"},
|
||||
{"VEXP2PD", "VEXP2PD", []Operand{vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f2fd49c8f5"},
|
||||
{"VRCP28SD", "VRCP28SD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "K1"), vreg(t, "X10")}, "6252bd09cbd1"},
|
||||
{"VBROADCASTF32X2", "VBROADCASTF32X2", []Operand{vreg(t, "X1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d4919d1"},
|
||||
{"VPCOMPRESSB", "VPCOMPRESSB", []Operand{vreg(t, "Z1"), vreg(t, "K1"), Ptr(SI, 0, 64)}, "62f27d49630e"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexQuadRegisterGroundTruth pins the quad-register instructions (the
|
||||
// 4FMAPS and 4VNNIW families) byte for byte against go tool asm: the memory
|
||||
// source keeps r/m, the bracketed list's LOW register travels the inverted
|
||||
// 5-bit V'VVVV field, the destination sits in reg, the opmask rides aaa and
|
||||
// the vector length follows the destination (L'L=512 for the ZMM forms,
|
||||
// 128 for the scalar ones) while the disp8×N multiplier stays 16 for every
|
||||
// member. The x86 decoder has no view of these forms, so no decode check
|
||||
// runs.
|
||||
func TestEvexQuadRegisterGroundTruth(t *testing.T) {
|
||||
sp := vreg(t, "RSP")
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"V4FMADDPS 17(SP) [Z0-Z3] K2 Z0", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||
"62f27f4a9a842411000000"},
|
||||
{"V4FMADDPS [Z10-Z13]", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z10"), vreg(t, "Z13")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||
"62f22f4a9a842411000000"},
|
||||
{"V4FMADDPS [Z20-Z23]", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z20"), vreg(t, "Z23")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||
"62f25f429a842411000000"},
|
||||
{"V4FMADDPS Z8 dst", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z8")},
|
||||
"62727f4a9a842411000000"},
|
||||
{"V4FMADDPS disp8x16", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 64, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||
"62f27f4a9a442404"},
|
||||
{"V4FMADDPS unmasked", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
|
||||
"62f27f489a842411000000"},
|
||||
{"V4FMADDSS 7(AX) [X0-X3] K5 X22", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||
"62e27f0d9bb007000000"},
|
||||
{"V4FMADDSS (DI)", "V4FMADDSS",
|
||||
[]Operand{Ptr(DI, 0, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||
"62e27f0d9b37"},
|
||||
{"V4FMADDSS [X10-X13]", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X10"), vreg(t, "X13")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||
"62e22f0d9bb007000000"},
|
||||
{"V4FMADDSS [X20-X23]", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||
"62e25f059bb007000000"},
|
||||
{"V4FMADDSS X30 dst", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X30")},
|
||||
"62627f0d9bb007000000"},
|
||||
{"V4FMADDSS X3 dst", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X3")},
|
||||
"62f27f0d9b9807000000"},
|
||||
{"V4FMADDSS disp8x16", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 16, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X30")},
|
||||
"62625f059b7001"},
|
||||
{"V4FNMADDPS", "V4FNMADDPS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||
"62f27f4aaa842411000000"},
|
||||
{"V4FNMADDSS", "V4FNMADDSS",
|
||||
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||
"62e27f0dabb007000000"},
|
||||
{"VP4DPWSSD", "VP4DPWSSD",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||
"62f27f4a52842411000000"},
|
||||
{"VP4DPWSSDS unmasked", "VP4DPWSSDS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
|
||||
"62f27f4853842411000000"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexQuadRegisterErrors pins the operand shapes the toolchain rejects:
|
||||
// the register class the list and the destination take is fixed per
|
||||
// instruction, the source is memory only, the opmask slot is positional and
|
||||
// the list's low register owns V'VVVV.
|
||||
func TestEvexQuadRegisterErrors(t *testing.T) {
|
||||
sp := vreg(t, "RSP")
|
||||
list := func(lo, hi string) RegList {
|
||||
return RegList{vreg(t, lo), vreg(t, hi)}
|
||||
}
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"X list on the PS form", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 0, 8), list("X0", "X3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||
{"Z list on the SS form", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 0, 8), list("Z0", "Z3"), vreg(t, "K5"), vreg(t, "X22")}},
|
||||
{"Y destination", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Y0")}},
|
||||
{"register source", "V4FMADDPS",
|
||||
[]Operand{vreg(t, "Z1"), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||
{"non-mask third operand", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z4"), vreg(t, "Z0")}},
|
||||
{"k0 mask", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K0"), vreg(t, "Z0")}},
|
||||
{"K after the destination", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0"), vreg(t, "K2")}},
|
||||
{"zeroing without a mask", "V4FMADDPS.Z",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0")}},
|
||||
{"SAE suffix", "V4FMADDPS.SAE",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||
{"high index source", "VP4DPWSSD",
|
||||
[]Operand{Idx(DI, vreg(t, "X16"), 1, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||
{"short operand list", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3")}},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", c.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+166
-25
@@ -14,8 +14,8 @@ import (
|
||||
"sync"
|
||||
)
|
||||
|
||||
// This file emits GOOBJ — the Go toolchain's object format, which cmd/link
|
||||
// consumes directly — so gasm-assembled functions drop into a go build
|
||||
// This file emits GOOBJ, the Go toolchain's object format, which cmd/link
|
||||
// consumes directly, so gasm-assembled functions drop into a go build
|
||||
// without the Go assembler. The layout follows cmd/internal/goobj: a
|
||||
// toolchain preamble ("go object ...\n!\n"), the go120ld header with its
|
||||
// block offsets, a string table, symbol definitions, the relocation /
|
||||
@@ -33,7 +33,10 @@ import (
|
||||
// methods supply the toolchain preamble, the MinLC (pc-value delta unit)
|
||||
// and the relocation-type mapping for code relocations.
|
||||
|
||||
// GOOBJ block indices (cmd/internal/goobj).
|
||||
// GOOBJ block indices (cmd/internal/goobj). These MUST match the real
|
||||
// archive layout: the emitter writes the header offsets per index and the
|
||||
// reader (groundtruth, goobj_resolve) parses real Go archives with them.
|
||||
// blkAutolib is unused by the emitter but still defines index 0.
|
||||
const (
|
||||
blkAutolib = iota
|
||||
blkPkgIdx
|
||||
@@ -65,11 +68,12 @@ const (
|
||||
kindSDWARFLINES = 20
|
||||
)
|
||||
|
||||
// Symbol flags (cmd/internal/goobj).
|
||||
// Symbol flags (cmd/internal/goobj). The linkname flag is set only for
|
||||
// //go:linkname symbols (and main.main); ordinary assembly symbols carry
|
||||
// none, matching cmd/asm's output.
|
||||
const (
|
||||
symFlagDupok = 0x01
|
||||
symFlagNoSplit = 0x10
|
||||
symFlag2Link = 0x10 // asm objects flag every named symbol as linkname
|
||||
symABIStatic = 0xffff
|
||||
)
|
||||
|
||||
@@ -91,10 +95,12 @@ const (
|
||||
)
|
||||
|
||||
// Relocation types (cmd/internal/objabi).
|
||||
// R_PCREL and R_ADDR are stable across Go versions.
|
||||
// R_ADDR, R_CALL, R_PCREL and R_TLS_LE are stable across Go versions.
|
||||
const (
|
||||
relocPCRel = 14 // R_PCREL
|
||||
relocAddr = 1 // R_ADDR
|
||||
relocCall = 7 // R_CALL
|
||||
relocPCRel = 14 // R_PCREL
|
||||
relocTLSLE = 15 // R_TLS_LE
|
||||
)
|
||||
|
||||
// relocDWTXTADDRU4 returns the R_DWTXTADDR_U4 relocation type for the
|
||||
@@ -137,10 +143,30 @@ func isGo127OrLater() bool {
|
||||
|
||||
// Special package indices for symbol references.
|
||||
const (
|
||||
pkgIdxNone = 0x7fffffff
|
||||
pkgIdxSelf = 0x7ffffffb
|
||||
pkgIdxNone = 0x7fffffff
|
||||
pkgIdxSelf = 0x7ffffffb
|
||||
pkgIdxBuiltin = 0x7ffffffc
|
||||
)
|
||||
|
||||
// goobjBuiltinMorestackNoctxt is the index of runtime.morestack_noctxt in
|
||||
// cmd/internal/goobj/builtinlist.go of the toolchain the object targets
|
||||
// (246 since Go 1.25; the list is append-only).
|
||||
const goobjBuiltinMorestackNoctxt = 246
|
||||
|
||||
// goobjBuiltinMorestack is the builtin reference the toolchain emits for the
|
||||
// stack-guard call.
|
||||
var goobjBuiltinMorestack = "runtime\u00b7morestack_noctxt"
|
||||
|
||||
// isCallReloc reports whether k is one of the per-arch call relocations a
|
||||
// direct branch to a TEXT symbol carries.
|
||||
func isCallReloc(k RelocKind) bool {
|
||||
switch k {
|
||||
case RelCall, RelRISCVJal, RelArm64Branch, RelLoong64Branch:
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
const goobjMagic = "\x00go120ld"
|
||||
|
||||
// goSym is one symbol definition under construction.
|
||||
@@ -175,14 +201,24 @@ type dwarfRelocSet struct {
|
||||
// does with its -p flag). srcPath names the source file recorded in the
|
||||
// object's file table and line tables. The toolchain's object preamble is
|
||||
// captured from the installed go tool asm, so the output links with the
|
||||
// toolchain it was produced on — exactly like a real assembly object.
|
||||
// toolchain it was produced on, exactly like a real assembly object.
|
||||
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
|
||||
pre, err := toolchainObjectPreamble()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// amd64: MinLC 1, R_PCREL for the code relocations.
|
||||
return img.emitGOObject(pkgPath, srcPath, pre, 1, func(Reloc) (uint16, uint8) { return relocPCRel, 4 })
|
||||
// amd64: MinLC 1, R_PCREL for displacements, R_CALL for calls and
|
||||
// R_TLS_LE for the stack-guard TLS load.
|
||||
return img.emitGOObject(pkgPath, srcPath, pre, 1, func(r Reloc) (uint16, uint8) {
|
||||
switch r.Kind {
|
||||
case RelCall:
|
||||
return relocCall, 4
|
||||
case RelTLSLE:
|
||||
return relocTLSLE, 4
|
||||
default:
|
||||
return relocPCRel, 4
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// emitGOObject assembles the GOOBJ payload for any architecture. pre is
|
||||
@@ -195,7 +231,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
||||
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
|
||||
}
|
||||
|
||||
// The non-package definitions first — the DWARF symbols reference the
|
||||
// The non-package definitions first, the DWARF symbols reference the
|
||||
// functions by these indices: per function the four pc-value tables
|
||||
// and the function itself, as cmd/asm lays them out.
|
||||
type npSym struct {
|
||||
@@ -244,7 +280,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
||||
// their relocations cover whole AUIPC/pcalau12i pairs, so
|
||||
// zeroing r.Off would erase the opcode/register bits the linker
|
||||
// preserves when it patches only the immediate.
|
||||
if r.Kind != RelPCRel32 {
|
||||
if r.Kind != RelPCRel32 && r.Kind != RelCall {
|
||||
continue
|
||||
}
|
||||
if r.Off >= 0 && r.Off+4 <= len(code) {
|
||||
@@ -252,7 +288,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
||||
}
|
||||
}
|
||||
nps = append(nps, npSym{
|
||||
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
|
||||
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, size: uint32(fn.Size)},
|
||||
data: code,
|
||||
})
|
||||
}
|
||||
@@ -282,7 +318,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
||||
abi = symABIStatic
|
||||
}
|
||||
defIdx[d.Name] = len(defs)
|
||||
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)})
|
||||
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, size: uint32(d.Size)})
|
||||
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
|
||||
}
|
||||
fnFiIdx := make([]int, len(img.Funcs))
|
||||
@@ -317,6 +353,13 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
||||
)
|
||||
}
|
||||
|
||||
// Index the non-package TEXT definitions by short name for the internal
|
||||
// call references.
|
||||
textNpIdx := map[string]int{}
|
||||
for i, fn := range img.Funcs {
|
||||
textNpIdx[fn.Name] = fnNpIdx[i]
|
||||
}
|
||||
|
||||
// Resolve external symbol references (cross-package). Build the
|
||||
// package index table and determine each external symbol's SymIdx
|
||||
// by reading the target package's export data.
|
||||
@@ -324,10 +367,20 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
||||
var extPkgIdx map[string]int
|
||||
var extSymIdx map[string]int
|
||||
if len(img.Externals) > 0 {
|
||||
var err error
|
||||
extPkgTable, extPkgIdx, extSymIdx, err = resolveExternalSymbols(img.Externals)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("GOOBJ emission: resolving external symbols: %w", err)
|
||||
// The morestack call is a builtin reference, not a resolved external.
|
||||
var need []string
|
||||
for _, n := range img.Externals {
|
||||
if n == goobjBuiltinMorestack {
|
||||
continue
|
||||
}
|
||||
need = append(need, n)
|
||||
}
|
||||
if len(need) > 0 {
|
||||
var err error
|
||||
extPkgTable, extPkgIdx, extSymIdx, err = resolveExternalSymbols(need)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("GOOBJ emission: resolving external symbols: %w", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -339,6 +392,31 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
||||
si := len(defs) + fnNpIdx[i]
|
||||
for _, r := range fn.Relocs {
|
||||
typ, size := relocField(r)
|
||||
if r.Kind == RelTLSLE {
|
||||
// The TLS load has no symbol: {0, 0} is the nil ref.
|
||||
var rec [23]byte
|
||||
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
|
||||
rec[4] = size
|
||||
binary.LittleEndian.PutUint16(rec[5:], typ)
|
||||
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
|
||||
binary.LittleEndian.PutUint32(rec[15:], 0)
|
||||
binary.LittleEndian.PutUint32(rec[19:], 0)
|
||||
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
||||
continue
|
||||
}
|
||||
if r.External && r.Name == goobjBuiltinMorestack {
|
||||
// The stack-guard morestack call uses the toolchain's
|
||||
// builtin reference.
|
||||
var rec [23]byte
|
||||
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
|
||||
rec[4] = size
|
||||
binary.LittleEndian.PutUint16(rec[5:], typ)
|
||||
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
|
||||
binary.LittleEndian.PutUint32(rec[15:], pkgIdxBuiltin)
|
||||
binary.LittleEndian.PutUint32(rec[19:], goobjBuiltinMorestackNoctxt)
|
||||
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
||||
continue
|
||||
}
|
||||
if r.External {
|
||||
// Split package-qualified name: "runtime·morestack" → runtime, morestack.
|
||||
pkg, name := splitQualified(r.Name)
|
||||
@@ -363,20 +441,83 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
||||
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
||||
continue
|
||||
}
|
||||
pkg := uint32(pkgIdxSelf)
|
||||
di, ok := defIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
|
||||
// A call to a TEXT function of the same file references the
|
||||
// non-package definition table.
|
||||
ni, isText := textNpIdx[r.Name]
|
||||
if !isText || !isCallReloc(r.Kind) {
|
||||
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
|
||||
}
|
||||
pkg = pkgIdxNone
|
||||
di = ni
|
||||
}
|
||||
var rec [23]byte
|
||||
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
|
||||
rec[4] = size // field width
|
||||
binary.LittleEndian.PutUint16(rec[5:], typ)
|
||||
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
|
||||
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
|
||||
binary.LittleEndian.PutUint32(rec[15:], pkg)
|
||||
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
|
||||
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
||||
}
|
||||
}
|
||||
// The data symbols' own relocations: the symbol-valued DATA fields
|
||||
// ("DATA s+0(SB)/8, $other(SB)"). The toolchain patches each field
|
||||
// with the target's absolute address through an R_ADDR of the DATA
|
||||
// line's width, on every architecture (the code relocations are
|
||||
// per-architecture PC-relative shapes; a data pointer word is not), so
|
||||
// this mapping bypasses relocField. The definitions were appended in
|
||||
// DataSyms order, so data symbol i is definition index i.
|
||||
for i, d := range img.DataSyms {
|
||||
for _, r := range d.Relocs {
|
||||
if r.Kind != RelAddr {
|
||||
return nil, fmt.Errorf("GOOBJ emission: data symbol %q carries a non-data relocation", d.Name)
|
||||
}
|
||||
var rec [23]byte
|
||||
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
|
||||
rec[4] = r.Siz
|
||||
binary.LittleEndian.PutUint16(rec[5:], relocAddr)
|
||||
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
|
||||
switch {
|
||||
case r.External && r.Name == goobjBuiltinMorestack:
|
||||
binary.LittleEndian.PutUint32(rec[15:], pkgIdxBuiltin)
|
||||
binary.LittleEndian.PutUint32(rec[19:], goobjBuiltinMorestackNoctxt)
|
||||
case r.External:
|
||||
pkg, name := splitQualified(r.Name)
|
||||
if pkg == "" {
|
||||
return nil, fmt.Errorf("GOOBJ emission: external symbol %q has no package prefix", r.Name)
|
||||
}
|
||||
pIdx, ok := extPkgIdx[pkg]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("GOOBJ emission: package %q not resolved", pkg)
|
||||
}
|
||||
sIdx, ok := extSymIdx[pkg+"·"+name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("GOOBJ emission: symbol %s·%s not resolved", pkg, name)
|
||||
}
|
||||
binary.LittleEndian.PutUint32(rec[15:], uint32(pIdx))
|
||||
binary.LittleEndian.PutUint32(rec[19:], uint32(sIdx))
|
||||
default:
|
||||
if di, ok := defIdx[r.Name]; ok {
|
||||
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
|
||||
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
|
||||
break
|
||||
}
|
||||
// A DATA field may hold the address of a TEXT function of
|
||||
// the same file (the rt0 lib entry spelling), which is a
|
||||
// non-package definition.
|
||||
ni, isText := textNpIdx[r.Name]
|
||||
if !isText {
|
||||
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
|
||||
}
|
||||
binary.LittleEndian.PutUint32(rec[15:], pkgIdxNone)
|
||||
binary.LittleEndian.PutUint32(rec[19:], uint32(ni))
|
||||
}
|
||||
symRelocs[i] = append(symRelocs[i], rec[:]...)
|
||||
}
|
||||
}
|
||||
// The DWARF symbols' own relocations (the function address references).
|
||||
for _, ds := range dwarfRelocs {
|
||||
for _, r := range ds.relocs {
|
||||
@@ -417,7 +558,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
||||
// The string table. Absolute offsets: it starts right after the
|
||||
// 96-byte header (magic, fingerprint, flags, the 19 block offsets).
|
||||
const headerSize = 8 + 8 + 4 + 4*(blkEnd+1)
|
||||
strTab := []byte{}
|
||||
var strTab []byte
|
||||
strOff := map[string]uint32{}
|
||||
addStr := func(s string) {
|
||||
if _, ok := strOff[s]; ok {
|
||||
@@ -464,7 +605,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
||||
auxIdxBlk := make([]byte, 0, 4*(nsyms+1))
|
||||
dataIdxBlk := make([]byte, 0, 4*(nsyms+1))
|
||||
var nr, na, nd uint32
|
||||
for si := 0; si < nsyms; si++ {
|
||||
for si := range nsyms {
|
||||
relocIdxBlk = binary.LittleEndian.AppendUint32(relocIdxBlk, nr)
|
||||
auxIdxBlk = binary.LittleEndian.AppendUint32(auxIdxBlk, na)
|
||||
dataIdxBlk = binary.LittleEndian.AppendUint32(dataIdxBlk, nd)
|
||||
@@ -505,7 +646,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
||||
// The fingerprint stays zero, as cmd/asm leaves it.
|
||||
binary.LittleEndian.PutUint32(payload[16:], 4) // ObjFlagFromAssembly
|
||||
off := uint32(headerSize + len(strTab))
|
||||
for i := 0; i < blkEnd; i++ {
|
||||
for i := range blkEnd {
|
||||
binary.LittleEndian.PutUint32(payload[20+4*i:], off)
|
||||
off += uint32(len(blocks[i]))
|
||||
}
|
||||
|
||||
+1
-4
@@ -139,10 +139,7 @@ func dwSelectOpcode(deltaPC uint64, deltaLC int64) int64 {
|
||||
return int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*int64(deltaPC)
|
||||
default:
|
||||
if deltaPC <= uint64(dwPCRange) {
|
||||
op := int64(dwOpcodeBase) + (dwLineRange - 1) + dwLineRange*int64(deltaPC)
|
||||
if op > 255 {
|
||||
op = 255
|
||||
}
|
||||
op := min(int64(dwOpcodeBase)+(dwLineRange-1)+dwLineRange*int64(deltaPC), 255)
|
||||
return op
|
||||
}
|
||||
switch deltaPC - uint64(dwPCRange) {
|
||||
|
||||
+76
-54
@@ -12,24 +12,6 @@ import (
|
||||
"strings"
|
||||
)
|
||||
|
||||
// readGOOBJSymbols reads the GOOBJ symbol definitions from a compiled Go
|
||||
// package's export file. The file is an ar archive containing a __.PKGDEF
|
||||
// member whose payload is the "go object ...\n!\n" preamble followed by the
|
||||
// GOOBJ data. The function returns the symbol names in definition order
|
||||
// (the order they appear in blkSymdef), which matches the SymIdx the linker
|
||||
// expects for cross-package references.
|
||||
func readGOOBJSymbols(exportPath string) ([]string, error) {
|
||||
data, err := os.ReadFile(exportPath)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
goobj, err := extractGOOBJ(data)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", exportPath, err)
|
||||
}
|
||||
return goobj.symbols(), nil
|
||||
}
|
||||
|
||||
// exportPath returns the export file path for a given import path by running
|
||||
// "go list -export". The result is cached so repeated calls for the same
|
||||
// package are fast.
|
||||
@@ -58,8 +40,11 @@ func exportPath(importPath string) (string, error) {
|
||||
//
|
||||
// refs maps package import paths to the symbol names referenced from that
|
||||
// package. The returned pkgIdx maps each import path to its position in
|
||||
// the blkPkgIdx table (0-based), and symIdx gives each symbol's index within
|
||||
// its package.
|
||||
// the blkPkgIdx table, which reserves index 0 for the dummy invalid
|
||||
// package (cmd/internal/obj/sym.go: "0 is invalid index"; the loader's
|
||||
// reader loop starts at 1), so package i sits at block index i+1 and its
|
||||
// relocations carry i+1. symIdx gives each symbol's index within its
|
||||
// package.
|
||||
func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symIdx map[string]int, err error) {
|
||||
pkgIdx = make(map[string]int, len(refs))
|
||||
symIdx = make(map[string]int)
|
||||
@@ -68,7 +53,9 @@ func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symI
|
||||
packages := sortedPkgRefs(refs)
|
||||
|
||||
for i, pkg := range packages {
|
||||
pkgIdx[pkg.path] = i
|
||||
// Block index 0 is the dummy invalid package; the first real
|
||||
// package starts at 1.
|
||||
pkgIdx[pkg.path] = i + 1
|
||||
exp, err := exportPath(pkg.path)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
@@ -102,7 +89,7 @@ func sortedPkgRefs(refs map[string][]string) []pkgRef {
|
||||
for pkg, syms := range refs {
|
||||
pkgs = append(pkgs, pkgRef{pkg, syms})
|
||||
}
|
||||
// Simple insertion sort — the list is tiny (usually 1–3 packages).
|
||||
// Simple insertion sort, the list is tiny (usually 1-3 packages).
|
||||
for i := 1; i < len(pkgs); i++ {
|
||||
for j := i; j > 0 && pkgs[j-1].path > pkgs[j].path; j-- {
|
||||
pkgs[j-1], pkgs[j] = pkgs[j], pkgs[j-1]
|
||||
@@ -163,40 +150,65 @@ func parseArDecimal(b []byte) int {
|
||||
}
|
||||
|
||||
// goobjFile is a parsed GOOBJ file: the string table and the symbol-definition
|
||||
// block.
|
||||
// blocks. The hashed blocks are kept raw: their symbols carry no names, only
|
||||
// the loader needs their counts.
|
||||
type goobjFile struct {
|
||||
strTab []byte // string table, at headerSize + n
|
||||
symdef []byte // blkSymdef raw block
|
||||
npdef []byte // blkNonpkgdef raw block
|
||||
strTab []byte // string table, at headerSize + n
|
||||
symdef []byte // blkSymdef raw block
|
||||
hashed64 []byte // blkHashed64def raw block
|
||||
hashed []byte // blkHasheddef raw block
|
||||
npdef []byte // blkNonpkgdef raw block
|
||||
}
|
||||
|
||||
// symbols returns all symbol names in definition order by scanning the
|
||||
// symdef and nonpkgdef blocks and resolving each name through the string
|
||||
// table. Package definitions (blkSymdef) use fully-qualified names like
|
||||
// "runtime.morestack"; non-package definitions (blkNonpkgdef) use bare
|
||||
// names like "morestack". This combined list matches the index the
|
||||
// linker expects for cross-package references.
|
||||
// loaderIndexBase returns the index the first nonpkgdef symbol occupies in the
|
||||
// loader's per-object symbol array. cmd/link lays the definition blocks out as
|
||||
// symdef, hashed64def, hasheddef, nonpkgdef, nonpkgref (loader.go: preloadSyms
|
||||
// fills r.syms in exactly that order, and resolve() indexes PkgIdxNone and
|
||||
// cross-package SymIdx into it), so a symbol found in blkNonpkgdef carries the
|
||||
// three leading blocks' symbol counts as its base.
|
||||
func (f *goobjFile) loaderIndexBase() int {
|
||||
return len(f.symdef)/recSymSize + len(f.hashed64)/recSymSize + len(f.hashed)/recSymSize
|
||||
}
|
||||
|
||||
// symbols returns the names of the symdef and nonpkgdef blocks in
|
||||
// definition order. Package definitions (blkSymdef) use fully-qualified
|
||||
// names like "runtime.morestack"; non-package definitions (blkNonpkgdef)
|
||||
// use bare names like "morestack". For lookups by index prefer
|
||||
// findSymbol: it adds the hashed blocks' count the loader's array
|
||||
// interleaves between the two.
|
||||
func (f *goobjFile) symbols() []string {
|
||||
return append(f.defNames(), f.npdefNames()...)
|
||||
}
|
||||
|
||||
// findSymbol returns the index of a symbol within the combined symbol list,
|
||||
// or -1 if not found. It first tries the fully-qualified name (pkg.name),
|
||||
// then the bare name.
|
||||
// findSymbol returns the index of a symbol within the loader's per-object
|
||||
// symbol array, or -1 if not found. It first tries the fully-qualified
|
||||
// name (pkg.name), then the bare name (assembly objects store dotless
|
||||
// names, e.g. runtime's "gogo", for symbols other packages reach through
|
||||
// a linkname).
|
||||
func (f *goobjFile) findSymbol(pkg, name string) int {
|
||||
base := f.loaderIndexBase()
|
||||
qualified := pkg + "." + name
|
||||
syms := f.symbols()
|
||||
for i, s := range syms {
|
||||
for i, s := range f.defNames() {
|
||||
if s == qualified {
|
||||
return i
|
||||
}
|
||||
}
|
||||
// Try bare name (for non-package definitions).
|
||||
for i, s := range syms {
|
||||
for i, s := range f.npdefNames() {
|
||||
if s == qualified {
|
||||
return base + i
|
||||
}
|
||||
}
|
||||
// Try bare name (for dotless assembly definitions).
|
||||
for i, s := range f.defNames() {
|
||||
if s == name {
|
||||
return i
|
||||
}
|
||||
}
|
||||
for i, s := range f.npdefNames() {
|
||||
if s == name {
|
||||
return base + i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
@@ -210,18 +222,22 @@ func (f *goobjFile) npdefNames() []string {
|
||||
return f.readSymNames(f.npdef)
|
||||
}
|
||||
|
||||
// recSymSize is the size of one Sym record in the definition blocks
|
||||
// (goobj.SymSize: stringRefSize + 2 + 1 + 1 + 1 + 4 + 4).
|
||||
const recSymSize = 21
|
||||
|
||||
// readSymNames reads symbol names from a symdef/nonpkgdef block. Each record
|
||||
// is 21 bytes: nameLen (u32), nameOff (u32), abi (u16), typ, flag, flag2,
|
||||
// size (u32), align (u32). nameOff is an absolute offset into the string
|
||||
// table.
|
||||
func (f *goobjFile) readSymNames(block []byte) []string {
|
||||
const recSize = 21
|
||||
const recSize = recSymSize
|
||||
if len(block) < recSize {
|
||||
return nil
|
||||
}
|
||||
n := len(block) / recSize
|
||||
names := make([]string, 0, n)
|
||||
for i := 0; i < n; i++ {
|
||||
for i := range n {
|
||||
rec := block[i*recSize : (i+1)*recSize]
|
||||
nameLen := binary.LittleEndian.Uint32(rec[0:4])
|
||||
nameOff := binary.LittleEndian.Uint32(rec[4:8])
|
||||
@@ -265,16 +281,18 @@ func parseGOOBJ(data []byte) (*goobjFile, error) {
|
||||
// [16:20] flags
|
||||
// [20:96] 19 × uint32 offsets
|
||||
var offs [blkEnd + 1]uint32
|
||||
for i := 0; i <= blkEnd; i++ {
|
||||
for i := range blkEnd + 1 {
|
||||
offs[i] = binary.LittleEndian.Uint32(payload[20+4*i:])
|
||||
}
|
||||
// The string table lives at headerSize.
|
||||
strTabStart := uint32(goobjHeaderSize)
|
||||
|
||||
f := &goobjFile{
|
||||
strTab: payload[strTabStart:offs[0]],
|
||||
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
|
||||
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
|
||||
strTab: payload[strTabStart:offs[0]],
|
||||
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
|
||||
hashed64: blockSlice(payload, offs, blkHashed64def, blkHashed64def+1),
|
||||
hashed: blockSlice(payload, offs, blkHasheddef, blkHasheddef+1),
|
||||
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
|
||||
}
|
||||
return f, nil
|
||||
}
|
||||
@@ -324,25 +342,29 @@ func resolveExternalSymbols(externals []string) (pkgTable []string, pkgIdxMap ma
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
|
||||
// Build the package table in pkgIdx order.
|
||||
// Build the package table in pkgIdx order. The indices are 1-based
|
||||
// (0 is the dummy invalid package, written by the emitter itself), so
|
||||
// the table without the dummy is indexed one below.
|
||||
pkgTable = make([]string, len(pkgIdx1))
|
||||
for pkg, idx := range pkgIdx1 {
|
||||
pkgTable[idx] = pkg
|
||||
pkgTable[idx-1] = pkg
|
||||
}
|
||||
|
||||
return pkgTable, pkgIdx1, symIdx1, nil
|
||||
}
|
||||
|
||||
// splitQualified splits a qualified Go symbol name (pkgpath·name) into its
|
||||
// package path and local name. The separator is the middle dot (U+00B7).
|
||||
// If no separator is found, the symbol is assumed to be in the current
|
||||
// package (empty pkg).
|
||||
// package path and local name. The separator is the middle dot (U+00B7),
|
||||
// whose UTF-8 encoding is two bytes, so the search must be string-based:
|
||||
// IndexByte would match only the second byte and leave the lead byte on
|
||||
// the package path. If no separator is found, the symbol is assumed to be
|
||||
// in the current package (empty pkg).
|
||||
func splitQualified(full string) (pkg, name string) {
|
||||
if idx := strings.IndexByte(full, '\u00b7'); idx >= 0 {
|
||||
return full[:idx], full[idx+len("\u00b7"):]
|
||||
if before, after, ok := strings.Cut(full, "\u00b7"); ok {
|
||||
return before, after
|
||||
}
|
||||
if idx := strings.IndexByte(full, '.'); idx >= 0 {
|
||||
return full[:idx], full[idx+1:]
|
||||
if before, after, ok := strings.Cut(full, "."); ok {
|
||||
return before, after
|
||||
}
|
||||
return "", full
|
||||
}
|
||||
|
||||
@@ -56,8 +56,12 @@ func TestResolveExternalSymbols(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("resolveExternalGOOBJ: %v", err)
|
||||
}
|
||||
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 0 {
|
||||
t.Errorf("pkgIdx = %v, want runtime→0", pkgIdx)
|
||||
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 1 {
|
||||
// Index 0 is the dummy invalid package in the blkPkgIdx table;
|
||||
// the loader's reader loop starts at 1 (cmd/link/internal/
|
||||
// loader/loader.go: "PkgIdx 0 is a dummy invalid package"), so
|
||||
// the first real package must carry index 1.
|
||||
t.Errorf("pkgIdx = %v, want runtime→1", pkgIdx)
|
||||
}
|
||||
if _, ok := symIdx["runtime·g0"]; !ok {
|
||||
t.Errorf("symIdx missing runtime·g0, got %v", symIdx)
|
||||
|
||||
+195
-8
@@ -12,7 +12,7 @@ import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// goobjView is a minimal parsed view of a GOOBJ payload, enough to check
|
||||
@@ -117,7 +117,9 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
||||
if len(defs) != 7 {
|
||||
t.Fatalf("symdefs = %d, want 7", len(defs))
|
||||
}
|
||||
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
|
||||
// The linkname flag stays clear: the toolchain sets it only for
|
||||
// //go:linkname symbols, and an ordinary static GLOBL is not one.
|
||||
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != 0 {
|
||||
t.Errorf("mask symbol = %+v", defs[0])
|
||||
}
|
||||
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
|
||||
@@ -158,8 +160,8 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
||||
t.Errorf("funcinfo bytes %x", fi)
|
||||
}
|
||||
|
||||
// The pc-value tables of addq (non-package indices 0–3, so global
|
||||
// indices 7–10): pcsp a flat zero over the whole function, pcinline a
|
||||
// The pc-value tables of addq (non-package indices 0-3, so global
|
||||
// indices 7-10): pcsp a flat zero over the whole function, pcinline a
|
||||
// flat -1, both with the pc delta in MinLC (1) units.
|
||||
pcsp := data[le.Uint32(didx[4*7:]):]
|
||||
if got := pcsp[:3]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
|
||||
@@ -294,7 +296,7 @@ TEXT ·framed(SB), NOSPLIT, $8-0
|
||||
}
|
||||
for i := range wantPCs {
|
||||
if pcs[i] != wantPCs[i] || vals[i] != wantVals[i] {
|
||||
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d) — all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
|
||||
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d); all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
|
||||
}
|
||||
}
|
||||
// The last two steps unwind the epilogue to zero.
|
||||
@@ -333,7 +335,7 @@ TEXT ·useext(SB), NOSPLIT, $0-8
|
||||
|
||||
// TestGOObjectLinkAndRun is the end-to-end check: assemble the test
|
||||
// functions to a GOOBJ, swap it into a go build in place of the toolchain's
|
||||
// assembly object, link, and run — the output must match the baseline
|
||||
// assembly object, link, and run; the output must match the baseline
|
||||
// binary the Go assembler produced. Skipped when no Go toolchain is
|
||||
// available.
|
||||
func TestGOObjectLinkAndRun(t *testing.T) {
|
||||
@@ -393,7 +395,7 @@ func main() {
|
||||
}
|
||||
var work string
|
||||
var asmObj, pkgArch, linkLine string
|
||||
for _, line := range strings.Split(string(buildLog), "\n") {
|
||||
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
@@ -461,7 +463,7 @@ func main() {
|
||||
newArch := filepath.Join(dir, "pkg.a")
|
||||
args := []string{"tool", "pack", "c", newArch}
|
||||
seen := map[string]bool{}
|
||||
for _, m := range strings.Fields(string(listOut)) {
|
||||
for m := range strings.FieldsSeq(string(listOut)) {
|
||||
if seen[m] {
|
||||
continue
|
||||
}
|
||||
@@ -511,3 +513,188 @@ func fieldAfter(line, flag string) string {
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// TestGOObjectExternalPackageLink is the cross-package end-to-end check: a
|
||||
// GOOBJ whose code references a real external package symbol (runtime's
|
||||
// morestack, a plain reference rather than the builtin noctxt form) must
|
||||
// carry a package index that points past the blkPkgIdx table's dummy entry
|
||||
// 0, and the object must link against the real runtime. Pre-fix, the
|
||||
// relocations carried block index 0, which the loader never fills, so the
|
||||
// reference resolved against whatever object was loaded first and the link
|
||||
// failed. The binary is not run: morestack returns to the call site's
|
||||
// stack check, which a hand-written caller has none of.
|
||||
func TestGOObjectExternalPackageLink(t *testing.T) {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
t.Skip("no Go toolchain available")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
|
||||
const asmSrc = `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·fn(SB), NOSPLIT, $0-0
|
||||
CALL ·helper(SB)
|
||||
RET
|
||||
|
||||
TEXT ·helper(SB), NOSPLIT, $0-0
|
||||
RET
|
||||
`
|
||||
const mainSrc = `package main
|
||||
|
||||
func fn()
|
||||
func helper()
|
||||
|
||||
func main() {
|
||||
fn()
|
||||
helper()
|
||||
}
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module extlink\n\ngo 1.27\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Capture the build the toolchain performs and re-run only its link
|
||||
// step with our object swapped into the package archive, mirroring
|
||||
// TestGOObjectLinkAndRun.
|
||||
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||
build.Dir = dir
|
||||
buildLog, err := build.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||
}
|
||||
var work, linkLine, asmObj, pkgArch string
|
||||
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
|
||||
asmObj = fieldAfter(line, "-o")
|
||||
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
|
||||
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
|
||||
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
|
||||
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||
linkLine = line
|
||||
}
|
||||
}
|
||||
if work == "" || asmObj == "" || pkgArch == "" || linkLine == "" {
|
||||
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
|
||||
}
|
||||
defer os.RemoveAll(work)
|
||||
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
|
||||
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
|
||||
|
||||
// Assemble the source with gasm, then retarget fn's internal call at
|
||||
// a real external package symbol: the reloc's qualified name drives
|
||||
// the export-data resolution the way a source-level runtime·sym(SB)
|
||||
// reference would.
|
||||
f, errs := parser.Parse("main_amd64.s", asmSrc)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
fn := &img.Funcs[0]
|
||||
for i := range fn.Relocs {
|
||||
fn.Relocs[i].Name = "runtime\u00b7morestack"
|
||||
fn.Relocs[i].External = true
|
||||
}
|
||||
img.Externals = []string{"runtime\u00b7morestack"}
|
||||
obj, err := img.GOObject("main", "main_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObject: %v", err)
|
||||
}
|
||||
|
||||
// Structural check: the blkPkgIdx block reserves entry 0 for the
|
||||
// dummy invalid package and places runtime at entry 1, and fn's call
|
||||
// relocation carries PkgIdx 1.
|
||||
v := openGoobj(t, obj)
|
||||
pkgBlk := v.blk(blkPkgIdx)
|
||||
if len(pkgBlk) != 2*8 {
|
||||
t.Fatalf("blkPkgIdx = %d bytes, want two entries", len(pkgBlk))
|
||||
}
|
||||
le := binary.LittleEndian
|
||||
strEntry := func(i int) string {
|
||||
e := pkgBlk[i*8 : (i+1)*8]
|
||||
return v.str(le.Uint32(e[4:]), le.Uint32(e[0:]))
|
||||
}
|
||||
if s := strEntry(0); s != "" {
|
||||
t.Errorf("blkPkgIdx[0] = %q, want the dummy empty package", s)
|
||||
}
|
||||
if s := strEntry(1); s != "runtime" {
|
||||
t.Errorf("blkPkgIdx[1] = %q, want runtime", s)
|
||||
}
|
||||
relocs := v.blk(blkReloc)
|
||||
// fn is the last non-package symbol (two functions, four pc tables
|
||||
// each); its one reloc is the final record.
|
||||
fnRec := relocs[len(relocs)-23:]
|
||||
if pIdx := le.Uint32(fnRec[15:]); pIdx != 1 {
|
||||
t.Errorf("external reloc PkgIdx = %d, want 1 (runtime)", pIdx)
|
||||
}
|
||||
|
||||
// Swap the object into the package archive and link with cmd/link;
|
||||
// the link line consumes the archive, not the loose object file.
|
||||
membersDir := filepath.Join(dir, "members")
|
||||
if err := os.MkdirAll(membersDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
|
||||
extract.Dir = membersDir
|
||||
if out, err := extract.CombinedOutput(); err != nil {
|
||||
t.Fatalf("pack x: %v\n%s", err, out)
|
||||
}
|
||||
member := filepath.Join(membersDir, filepath.Base(asmObj))
|
||||
if err := os.Chmod(member, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(member, obj, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
|
||||
listOut, err := listCmd.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("pack t: %v\n%s", err, listOut)
|
||||
}
|
||||
newArch := filepath.Join(dir, "pkg.a")
|
||||
args := []string{"tool", "pack", "c", newArch}
|
||||
seen := map[string]bool{}
|
||||
for m := range strings.FieldsSeq(string(listOut)) {
|
||||
if seen[m] {
|
||||
continue
|
||||
}
|
||||
seen[m] = true
|
||||
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
args = append(args, filepath.Join(membersDir, m))
|
||||
}
|
||||
pack := exec.Command(goBin, args...)
|
||||
pack.Dir = membersDir
|
||||
if out, err := pack.CombinedOutput(); err != nil {
|
||||
t.Fatalf("pack c: %v\n%s", err, out)
|
||||
}
|
||||
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||
linkLine = strings.ReplaceAll(linkLine, pkgArch, newArch)
|
||||
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "prog2"))
|
||||
linkCmd := exec.Command("sh", "-c", "cd "+dir+" && "+linkLine)
|
||||
if out, err := linkCmd.CombinedOutput(); err != nil {
|
||||
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
|
||||
}
|
||||
|
||||
// The call must have resolved to the real runtime symbol.
|
||||
dump, err := exec.Command(goBin, "tool", "objdump", "-s", "main.fn", filepath.Join(dir, "prog2")).CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("objdump main.fn: %v\n%s", err, dump)
|
||||
}
|
||||
if !bytes.Contains(dump, []byte("runtime.morestack")) {
|
||||
t.Errorf("main.fn does not call runtime.morestack:\n%s", dump)
|
||||
}
|
||||
}
|
||||
|
||||
+38
-9
@@ -13,25 +13,54 @@ import (
|
||||
)
|
||||
|
||||
// GOObjectAARCH64 emits a GOOBJ object file for AArch64. The layout is
|
||||
// the shared one in goobj.go — the toolchain preamble, the go120ld header
|
||||
// the shared one in goobj.go, the toolchain preamble, the go120ld header
|
||||
// with its block offsets, the string table, the symbol definitions and the
|
||||
// reloc/aux/data index arrays — with the arm64 preamble, the MinLC of 4
|
||||
// for the pc-value deltas, and R_ADDRARM64 relocation types for the
|
||||
// ADRP+ADD/LDR/STR address pairs.
|
||||
// reloc/aux/data index arrays, with the arm64 preamble, the MinLC of 4
|
||||
// for the pc-value deltas, and the arm64 relocation types for the ADRP
|
||||
// pairs and BL calls.
|
||||
//
|
||||
// The toolchain records one relocation per ADRP pair: a single R_ADDRARM64
|
||||
// or R_ARM64_PCREL_LDST64 of Siz 8 at the ADRP word, from which the linker
|
||||
// patches both instructions of the pair (cmd/internal/obj/arm64/asm7.go,
|
||||
// the ADRP cases: one AddRel with Off at the pair's pc and Siz 8). gasm's
|
||||
// assembler records the ADRP+ADD form as two word relocs, so the second
|
||||
// word's twin is dropped here before emission.
|
||||
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
|
||||
pre, err := toolchainObjectPreambleAARCH64()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
|
||||
return relocArm64Addr, 4
|
||||
coalesced := *img
|
||||
coalesced.Funcs = append([]FuncLayout(nil), img.Funcs...)
|
||||
for i := range coalesced.Funcs {
|
||||
rs := coalesced.Funcs[i].Relocs
|
||||
var keep []Reloc
|
||||
for j := 0; j < len(rs); j++ {
|
||||
keep = append(keep, rs[j])
|
||||
if rs[j].Kind == RelArm64Addr && j+1 < len(rs) &&
|
||||
rs[j+1].Kind == RelArm64Addr && rs[j+1].Off == rs[j].Off+4 {
|
||||
j++ // the ADD word's twin: the Siz-8 pair reloc covers it
|
||||
}
|
||||
}
|
||||
coalesced.Funcs[i].Relocs = keep
|
||||
}
|
||||
return coalesced.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
|
||||
switch r.Kind {
|
||||
case RelArm64Branch:
|
||||
return relocArm64Branch, 4
|
||||
case RelArm64LDST64:
|
||||
return relocArm64LDST64, 8
|
||||
default:
|
||||
return relocArm64Addr, 8
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// arm64 relocation types (cmd/internal/objabi). R_ADDRARM64 resolves an
|
||||
// ADRP+ADD/LDR/STR pair to a symbol's address.
|
||||
// arm64 relocation types (cmd/internal/objabi).
|
||||
const (
|
||||
relocArm64Addr = 9 // R_ADDRARM64
|
||||
relocArm64Addr = 3 // R_ADDRARM64, ADRP+ADD pair
|
||||
relocArm64Branch = 9 // R_CALLARM64, BL instruction
|
||||
relocArm64LDST64 = 40 // R_ARM64_PCREL_LDST64, ADRP+LDR/STR pair
|
||||
)
|
||||
|
||||
// toolchainObjectPreambleAARCH64 returns the "go object ...\n!\n" header
|
||||
|
||||
+11
-5
@@ -13,9 +13,9 @@ import (
|
||||
)
|
||||
|
||||
// GOObjectLOONG64 emits a GOOBJ object file for LoongArch. The layout is
|
||||
// the shared one in goobj.go — the toolchain preamble, the go120ld header
|
||||
// the shared one in goobj.go, the toolchain preamble, the go120ld header
|
||||
// with its block offsets, the string table, the symbol definitions and the
|
||||
// reloc/aux/data index arrays — with the loong64 preamble, the MinLC of 4
|
||||
// reloc/aux/data index arrays, with the loong64 preamble, the MinLC of 4
|
||||
// for the pc-value deltas, and R_LOONG64_ADDR_HI/LO relocation types for
|
||||
// the pcalau12i+addi.d address pairs.
|
||||
func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
|
||||
@@ -25,11 +25,16 @@ func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
|
||||
}
|
||||
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
|
||||
// A pcalau12i+addi.d pair: the high part carries
|
||||
// R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO.
|
||||
if r.Kind == RelLoong64AddrLo {
|
||||
// R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO; the guard's
|
||||
// morestack call carries R_CALLLOONG64.
|
||||
switch {
|
||||
case r.Kind == RelLoong64AddrLo:
|
||||
return relocLoong64AddrLo, 4
|
||||
case r.Kind == RelLoong64Branch:
|
||||
return relocCallLoong64, 4
|
||||
default:
|
||||
return relocLoong64AddrHi, 4
|
||||
}
|
||||
return relocLoong64AddrHi, 4
|
||||
})
|
||||
}
|
||||
|
||||
@@ -39,6 +44,7 @@ func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
|
||||
const (
|
||||
relocLoong64AddrHi = 77 // R_LOONG64_ADDR_HI
|
||||
relocLoong64AddrLo = 78 // R_LOONG64_ADDR_LO
|
||||
relocCallLoong64 = 84 // R_CALLLOONG64
|
||||
)
|
||||
|
||||
// toolchainObjectPreambleLOONG64 returns the "go object ...\n!\n" header
|
||||
|
||||
+2
-2
@@ -13,9 +13,9 @@ import (
|
||||
)
|
||||
|
||||
// GOObjectRISCV emits a GOOBJ object file for RISC-V. The layout is the
|
||||
// shared one in goobj.go — the toolchain preamble, the go120ld header with
|
||||
// shared one in goobj.go, the toolchain preamble, the go120ld header with
|
||||
// its block offsets, the string table, the symbol definitions and the
|
||||
// reloc/aux/data index arrays — with the RISC-V preamble, the MinLC of 2 for
|
||||
// reloc/aux/data index arrays, with the RISC-V preamble, the MinLC of 2 for
|
||||
// the pc-value deltas, and the single R_RISCV_PCREL_ITYPE/STYPE relocation
|
||||
// per AUIPC pair, matching `go tool asm`'s model (each pair is one 8-byte
|
||||
// relocation, not the ELF HI20/LO12 pair).
|
||||
|
||||
@@ -0,0 +1,329 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/hex"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64,
|
||||
// verified with go tool objdump): the stack-split guard classes, the morestack
|
||||
// block and the auto-NOSPLIT leaf behaviour.
|
||||
func TestStackGuardBytes(t *testing.T) {
|
||||
for _, tt := range []struct {
|
||||
name string
|
||||
src string
|
||||
want string
|
||||
}{
|
||||
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
|
||||
"554889e54883ec104883c4105dc3"},
|
||||
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
|
||||
"644c8b3425000000004c8da42478ffffff4d3b66107614554889e54881ec000100004881c4000100005dc3e800000000ebce"},
|
||||
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
|
||||
"644c8b3425000000004989e44981ec881f0000721a4d3b66107614554889e54881ec002000004881c4002000005dc3e800000000ebca"},
|
||||
// Class 2 with a body long enough that the underflow JB relaxes to
|
||||
// rel32: its displacement must span the real 6-byte JB, else the
|
||||
// branch lands 4 bytes past the morestack block, inside the CALL
|
||||
// displacement field.
|
||||
{"leafbiglong", "TEXT \u00b7leafbiglong(SB), $8192-0\n" + strings.Repeat("\tMOVQ AX, BX\n", 40) + "\tRET\n",
|
||||
"644c8b3425000000004989e44981ec881f00000f82960000004d3b66100f868c000000554889e54881ec00200000" + strings.Repeat("4889c3", 40) + "4881c4002000005dc3e800000000e947ffffff"},
|
||||
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
|
||||
"644c8b342500000000493b66107613554889e54883ec10e8000000004883c4105dc3e800000000ebd7"},
|
||||
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
|
||||
"554889e54883ec104883c4105dc3"},
|
||||
} {
|
||||
f, errs := parser.Parse("g_amd64.s", tt.src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("%s: parse: %v", tt.name, errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: assemble: %v", tt.name, err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
// The toolchain's object leaves every relocation field zero for the
|
||||
// linker, while the gasm image resolves file-internal references, so
|
||||
// the comparison masks the patch sites the way verify's ground truth
|
||||
// does.
|
||||
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
|
||||
for _, r := range fn.Relocs {
|
||||
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||
code[j] = 0
|
||||
}
|
||||
}
|
||||
got := hex.EncodeToString(code)
|
||||
if got != tt.want {
|
||||
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestStackGuardRelocs checks the guard's patch sites: the TLS slot and the
|
||||
// morestack call.
|
||||
func TestStackGuardRelocs(t *testing.T) {
|
||||
f, errs := parser.Parse("g_amd64.s", "TEXT \u00b7f(SB), $256-0\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
relocs := img.Funcs[0].Relocs
|
||||
if len(relocs) != 2 {
|
||||
t.Fatalf("relocs = %d, want 2", len(relocs))
|
||||
}
|
||||
tls, call := relocs[0], relocs[1]
|
||||
if tls.Kind != RelTLSLE || tls.Off != 5 || tls.Name != "" || tls.External {
|
||||
t.Errorf("tls reloc = %+v, want RelTLSLE at 5 with no symbol", tls)
|
||||
}
|
||||
if call.Kind != RelCall || call.Name != "runtime\u00b7morestack_noctxt" || !call.External {
|
||||
t.Errorf("call reloc = %+v, want RelCall to runtime.morestack_noctxt", call)
|
||||
}
|
||||
}
|
||||
|
||||
// TestStackGuardGOObj emissions succeed with the guard's TLS and builtin
|
||||
// references in play.
|
||||
func TestStackGuardGOObj(t *testing.T) {
|
||||
f, errs := parser.Parse("g_amd64.s", "TEXT \u00b7f(SB), $256-0\n\tCALL \u00b7helper(SB)\n\tRET\nTEXT \u00b7helper(SB), NOSPLIT, $0\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
obj, err := img.GOObject("testpkg", "g_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObject: %v", err)
|
||||
}
|
||||
if !bytes.Contains(obj, []byte("go120ld")) {
|
||||
t.Fatal("object lacks the GOOBJ magic")
|
||||
}
|
||||
}
|
||||
|
||||
// The arm64 stack-split guard, pinned from `go tool asm` (Go 1.27, arm64):
|
||||
// the guard classes, the auto-NOSPLIT leaf behaviour and the morestack
|
||||
// block. Relocation fields are masked: the toolchain's object leaves them
|
||||
// zero for the linker, the gasm image resolves file-internal references.
|
||||
func TestStackGuardBytesARM64(t *testing.T) {
|
||||
for _, tt := range []struct {
|
||||
name string
|
||||
src string
|
||||
want string
|
||||
}{
|
||||
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
|
||||
"fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"},
|
||||
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
|
||||
"900b40f9f14302d13f0210eb09010054f44304d19dfa3fa99f020091fd2300d1fd230491ff430491c0035fd6e3031eaa00000000f3ffff17"},
|
||||
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
|
||||
"900b40f91bf283d2f1633beba30100543f0210eb690100541b0284d2f4633bcb9dfa3fa99f020091fd2300d11b0184d2fd633b8b1b0284d2ff633b8bc0035fd6e3031eaa00000000eeffff17"},
|
||||
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
|
||||
"900b40f9ff6330eb09010054fe0f1ef8fd831ff8fd2300d100000000fd835ff8fe0742f8c0035fd6e3031eaa00000000f4ffff17"},
|
||||
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
|
||||
"fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"},
|
||||
} {
|
||||
f, errs := parser.Parse("g_arm64.s", tt.src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("%s: parse: %v", tt.name, errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: assemble: %v", tt.name, err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
|
||||
for _, r := range fn.Relocs {
|
||||
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||
code[j] = 0
|
||||
}
|
||||
}
|
||||
got := hex.EncodeToString(code)
|
||||
if got != tt.want {
|
||||
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestStackGuardBranchTargetsARM64 checks the class-2 guard's branch
|
||||
// positions for a frame whose guard constant needs two MOV words: the
|
||||
// displacements must be computed from byte offsets (8+4*ml and 16+4*ml), so
|
||||
// both branches land on the morestack block rather than inside the body.
|
||||
// The frame size makes the toolchain switch its own prologue decomposition,
|
||||
// so the assertion is on the branch targets, not pinned bytes.
|
||||
func TestStackGuardBranchTargetsARM64(t *testing.T) {
|
||||
f, errs := parser.Parse("g_arm64.s", "TEXT \u00b7f(SB), $65664-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
if len(code)%4 != 0 {
|
||||
t.Fatalf("function size %d is not a word multiple", len(code))
|
||||
}
|
||||
// autosize = 65680, so the guard materialises 65552 = MOVZ+MOVK: ml = 2
|
||||
// and the branches sit at bytes 16 and 24 of the guard prefix.
|
||||
const morestackBlock = 12 // MOVD R30, R3; BL; B back
|
||||
blockStart := len(code) - morestackBlock
|
||||
check := func(name string, off int) {
|
||||
t.Helper()
|
||||
w := leWord(code[off:])
|
||||
imm19 := int32(w>>5) & 0x7FFFF
|
||||
if imm19&(1<<18) != 0 {
|
||||
imm19 -= 1 << 19
|
||||
}
|
||||
if target := off + int(imm19)*4; target != blockStart {
|
||||
t.Errorf("%s at byte %d targets byte %d, want the morestack block at %d", name, off, target, blockStart)
|
||||
}
|
||||
}
|
||||
check("B.LO", 16)
|
||||
check("B.LS", 24)
|
||||
}
|
||||
|
||||
// The riscv64 stack-split guard, pinned from `go tool asm` (Go 1.27,
|
||||
// riscv64): the morestack call sits between the guard and the body, and the
|
||||
// guard branches forward over it. Relocation fields are masked.
|
||||
func TestStackGuardBytesRISCV64(t *testing.T) {
|
||||
for _, tt := range []struct {
|
||||
name string
|
||||
src string
|
||||
want string
|
||||
}{
|
||||
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
|
||||
"03b30d0163662300000000006ff05fff233411fe211106e08260610167800000"},
|
||||
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
|
||||
"03b30d01930381f763667300000000006ff01fff233c11ee130181ef06e082601301811067800000"},
|
||||
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
|
||||
"03b30d0189639b8383f863697100f97f9b8f8f07b303f10163667300000000006ff01ffef97f8a9f23bc1ffef97fe13f7e9106e08260896fa12f7e9167800000"},
|
||||
{"frameless", "TEXT \u00b7frameless(SB), $0-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
|
||||
"03b30d0163662300000000006ff05fff233c11fe611106e0000000008260210167800000"},
|
||||
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
|
||||
"233411fe211106e08260610167800000"},
|
||||
} {
|
||||
f, errs := parser.Parse("g_riscv64.s", tt.src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("%s: parse: %v", tt.name, errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: assemble: %v", tt.name, err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
|
||||
for _, r := range fn.Relocs {
|
||||
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||
code[j] = 0
|
||||
}
|
||||
}
|
||||
got := hex.EncodeToString(code)
|
||||
if got != tt.want {
|
||||
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The loong64 stack-split guard, pinned from `go tool asm` (Go 1.27,
|
||||
// loong64): every guard class (including the medium class with the
|
||||
// materialised constant and the big class with the ORI-less constants), the
|
||||
// auto-NOSPLIT leaf behaviour, the large-frame R30 prologue/epilogue forms
|
||||
// and the morestack block. Relocation fields are masked.
|
||||
func TestStackGuardBytesLOONG64(t *testing.T) {
|
||||
for _, tt := range []struct {
|
||||
name string
|
||||
src string
|
||||
want string
|
||||
}{
|
||||
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
|
||||
"61a0ff2963a0ff026100c0296360c0022000004c"},
|
||||
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
|
||||
"d442c02878e0fd0294e21200801a004061e0fb2963e0fb026100c0296320c4022000004c3f00150000000000ffd7ff53"},
|
||||
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
|
||||
"61a0ff2963a0ff026100c0296360c0022000004c"},
|
||||
// The LR store leaves the 12-bit store-offset range while the SP
|
||||
// adjust immediate still fits, and the epilogue adjusts through a
|
||||
// single ORI.
|
||||
{"fit2048", "TEXT \u00b7fit2048(SB), $2040-0\n\tRET\n",
|
||||
"d442c0287800e20294e21200802600401e000014de8f1000c103e0296300e0026100c0291e00a00363f810002000004c3f00150000000000ffcbff53"},
|
||||
// Medium class at the materialisation boundary (off = 2048 still
|
||||
// immediate, 2049+ goes through R30).
|
||||
{"med2048off", "TEXT \u00b7med2048off(SB), $2168-0\n\tRET\n",
|
||||
"d442c0287800e00294e21200802e0040feffff15de8f1000c103de29feffff15de039e0363f810006100c0291e00a20363f810002000004c3f00150000000000ffc3ff53"},
|
||||
{"medmat", "TEXT \u00b7medmat(SB), $2176-0\n\tRET\n",
|
||||
"d442c028feffff15dee39f0378f8100094e21200802e0040feffff15de8f1000c1e3dd29feffff15dee39d0363f810006100c0291e20a20363f810002000004c3f00150000000000ffbbff53"},
|
||||
// Big class with the rounding-split store and the floor-split adjust.
|
||||
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
|
||||
"d442c0283e000014de23be0378f8120000470044deffff15dee3810378f8100094e2120080320040deffff15de8f1000c1e3ff29beffff15dee3bf0363f810006100c0295e000014de23800363f810002000004c3f00150000000000ffa7ff53"},
|
||||
// Zero low 12 bits drop the ORI from the store, the adjust and the
|
||||
// epilogue materialisation.
|
||||
{"bigzero", "TEXT \u00b7bigzero(SB), $4088-0\n\tRET\n",
|
||||
"d442c028feffff15de03820378f8100094e21200802a0040feffff15de8f1000c103c029feffff1563f810006100c0293e00001463f810002000004c3f00150000000000ffbfff53"},
|
||||
// Big class whose first constant has a zero high part: a single ORI.
|
||||
{"big3976", "TEXT \u00b7big3976(SB), $4096-0\n\tRET\n",
|
||||
"d442c0281e20be0378f8120000470044feffff15dee3810378f8100094e2120080320040feffff15de8f1000c1e3ff29deffff15dee3bf0363f810006100c0293e000014de23800363f810002000004c3f00150000000000ffabff53"},
|
||||
// Big class at a multiple of 4096: both guard constants lose their
|
||||
// ORI word.
|
||||
{"giantlo0", "TEXT \u00b7giantlo0(SB), $4216-0\n\tRET\n",
|
||||
"d442c0283e00001478f8120000430044feffff1578f8100094e2120080320040feffff15de8f1000c103fe29deffff15de03be0363f810006100c0293e000014de03820363f810002000004c3f00150000000000ffafff53"},
|
||||
// Non-leaf big frame: the body call plus the LR restore epilogue.
|
||||
{"callbig", "TEXT \u00b7callbig(SB), $8192-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
|
||||
"d442c0283e000014de23be0378f81200004f0044deffff15dee3810378f8100094e21200803a0040deffff15de8f1000c1e3ff29beffff15dee3bf0363f810006100c029000000006100c0285e000014de23800363f810002000004c3f00150000000000ff9fff53"},
|
||||
} {
|
||||
f, errs := parser.Parse("g_loong64.s", tt.src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("%s: parse: %v", tt.name, errs)
|
||||
}
|
||||
img, err := AssembleFileLOONG64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: assemble: %v", tt.name, err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
|
||||
for _, r := range fn.Relocs {
|
||||
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||
code[j] = 0
|
||||
}
|
||||
}
|
||||
got := hex.EncodeToString(code)
|
||||
if got != tt.want {
|
||||
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestStackGuardGOObjInternalCall checks that GOOBJ emission succeeds when a
|
||||
// guarded function calls a TEXT symbol of the same file, for every arch's
|
||||
// call relocation kind.
|
||||
func TestStackGuardGOObjInternalCall(t *testing.T) {
|
||||
for _, tt := range []struct {
|
||||
src string
|
||||
assemble func(*ast.File, ...AssembleOption) (*Image, error)
|
||||
}{
|
||||
{"g_amd64.s", AssembleFile},
|
||||
{"g_arm64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileARM64(f) }},
|
||||
{"g_riscv64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileRISCV(f) }},
|
||||
{"g_loong64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileLOONG64(f) }},
|
||||
} {
|
||||
f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("%s: parse: %v", tt.src, errs)
|
||||
}
|
||||
img, err := tt.assemble(f)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: assemble: %v", tt.src, err)
|
||||
}
|
||||
if _, err := img.GOObject("testpkg", tt.src); err != nil {
|
||||
t.Errorf("%s: GOObject: %v", tt.src, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
+1167
-79
File diff suppressed because it is too large
Load Diff
@@ -16,16 +16,16 @@ import (
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel —
|
||||
// all functions plus the file-local mask24 constant — and checks that every
|
||||
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel;
|
||||
// all functions plus the file-local mask24 constant; and checks that every
|
||||
// static-symbol load resolves to the right bytes in the image.
|
||||
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
||||
path := "../../go-libraries/go-flac/avx2_amd64.s"
|
||||
if _, err := os.Stat(path); err != nil {
|
||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
||||
t.Skip("go-libraries repository not present next to gasm-sdk")
|
||||
}
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
@@ -81,12 +81,12 @@ func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
||||
}
|
||||
|
||||
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
|
||||
// kernel — all functions plus the file-global idx16 constant — and checks
|
||||
// kernel, all functions plus the file-global idx16 constant, and checks
|
||||
// that the static-symbol load resolves to the right bytes in the image.
|
||||
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
|
||||
path := "../../go-libraries/go-flac/avx512_amd64.s"
|
||||
if _, err := os.Stat(path); err != nil {
|
||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
||||
t.Skip("go-libraries repository not present next to gasm-sdk")
|
||||
}
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
|
||||
@@ -0,0 +1,219 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// The differential kernels for the DATA-path and front-end gaps are kept in
|
||||
// testdata/verify beside the campaign's other kernels; the verify package's
|
||||
// suites are not open to the asm package, so this test is their runner: each
|
||||
// kernel assembles through gasm and through go tool asm, and the functions'
|
||||
// bytes must agree with the relocation sites masked on both sides.
|
||||
|
||||
// toolAsmObject assembles path with the installed toolchain's assembler for
|
||||
// goarch ("" = the host) and returns the object bytes.
|
||||
func toolAsmObject(t *testing.T, path, goarch string) []byte {
|
||||
t.Helper()
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
t.Skip("no Go toolchain available")
|
||||
}
|
||||
out, err := exec.Command(goBin, "env", "GOROOT").Output()
|
||||
if err != nil {
|
||||
t.Fatalf("go env GOROOT: %v", err)
|
||||
}
|
||||
includeDir := filepath.Join(strings.TrimSpace(string(out)), "pkg", "include")
|
||||
|
||||
pkg := strings.TrimSuffix(filepath.Base(path), ".s")
|
||||
pkg = strings.TrimSuffix(pkg, "_amd64")
|
||||
pkg = strings.TrimSuffix(pkg, "_arm64")
|
||||
|
||||
objPath := filepath.Join(t.TempDir(), "oracle.o")
|
||||
cmd := exec.Command(goBin, "tool", "asm", "-I", includeDir, "-p", pkg, "-o", objPath, path)
|
||||
if goarch != "" {
|
||||
environ := os.Environ()
|
||||
env := make([]string, 0, len(environ)+1)
|
||||
for _, e := range environ {
|
||||
if !strings.HasPrefix(e, "GOARCH=") {
|
||||
env = append(env, e)
|
||||
}
|
||||
}
|
||||
cmd.Env = append(env, "GOARCH="+goarch)
|
||||
}
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
t.Fatalf("go tool asm %s: %v\n%s", filepath.Base(path), err, out)
|
||||
}
|
||||
obj, err := os.ReadFile(objPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return obj
|
||||
}
|
||||
|
||||
// oracleFuncCode extracts the non-package TEXT functions' code bytes from a
|
||||
// toolchain object, keyed by the name the object records (pkg.name). Each
|
||||
// function's span is its own symbol size: a toolchain object that follows
|
||||
// the text with data symbols (the synthesised float-constant pool) would
|
||||
// otherwise fold them into the last function's bytes.
|
||||
func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
|
||||
t.Helper()
|
||||
v := openGoobj(t, obj)
|
||||
le := binary.LittleEndian
|
||||
const symSize = 21
|
||||
nps := v.syms(blkNonpkgdef)
|
||||
data := v.blk(blkData)
|
||||
didx := v.blk(blkDataIdx)
|
||||
preceding := 0
|
||||
for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef} {
|
||||
preceding += len(v.blk(bi)) / symSize
|
||||
}
|
||||
out := make(map[string][]byte, len(nps))
|
||||
for i, s := range nps {
|
||||
if s.typ != kindSTEXT {
|
||||
continue
|
||||
}
|
||||
start := le.Uint32(didx[4*(preceding+i):])
|
||||
out[s.name] = data[start : start+s.size]
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// maskCode zeroes every relocation field, the way the toolchain's object
|
||||
// leaves them for the linker.
|
||||
func maskCode(code []byte, relocs []Reloc) []byte {
|
||||
for _, r := range relocs {
|
||||
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||
code[j] = 0
|
||||
}
|
||||
}
|
||||
return code
|
||||
}
|
||||
|
||||
// code assembles src for amd64 and returns the image's code bytes.
|
||||
func code(path, src string) []byte {
|
||||
f, errs := parser.Parse(path, src)
|
||||
if len(errs) > 0 {
|
||||
return nil
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
return img.Code
|
||||
}
|
||||
|
||||
// TestDifferentialKernels pins the new kernels against the oracle.
|
||||
func TestDifferentialKernels(t *testing.T) {
|
||||
if runtime.GOARCH != "amd64" {
|
||||
t.Skip("the amd64 kernels assume an amd64 host assembler default")
|
||||
}
|
||||
for _, k := range []struct {
|
||||
path string
|
||||
goarch string
|
||||
arm64 bool
|
||||
}{
|
||||
{filepath.Join("..", "testdata", "verify", "datarel_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "divslash_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "semicolons_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "forms_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true},
|
||||
{filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true},
|
||||
} {
|
||||
t.Run(filepath.Base(k.path), func(t *testing.T) {
|
||||
src, err := os.ReadFile(k.path)
|
||||
if err != nil {
|
||||
t.Fatalf("read: %v", err)
|
||||
}
|
||||
f, errs := parser.Parse(k.path, string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
var img *Image
|
||||
if k.arm64 {
|
||||
img, err = AssembleFileARM64(f)
|
||||
} else {
|
||||
img, err = AssembleFile(f)
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
gt := oracleFuncCode(t, toolAsmObject(t, k.path, k.goarch))
|
||||
// The oracle keys its functions by the qualified object name
|
||||
// (pkg.name); match on the local part.
|
||||
byLocal := make(map[string][]byte, len(gt))
|
||||
for name, code := range gt {
|
||||
if _, after, ok := strings.Cut(name, "."); ok {
|
||||
name = after
|
||||
}
|
||||
byLocal[name] = code
|
||||
}
|
||||
|
||||
matched := 0
|
||||
for _, fn := range img.Funcs {
|
||||
gasmCode := maskCode(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
|
||||
goCode, ok := byLocal[fn.Name]
|
||||
if !ok {
|
||||
t.Errorf("%s: not in ground truth (%d functions: %v)", fn.Name, len(gt), keysOf(byLocal))
|
||||
continue
|
||||
}
|
||||
goCode = maskCode(append([]byte(nil), goCode...), fn.Relocs)
|
||||
cmpLen := min(len(goCode), len(gasmCode))
|
||||
if !bytes.Equal(gasmCode[:cmpLen], goCode[:cmpLen]) {
|
||||
t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\ngasm %x\ngo %x", fn.Name, len(gasmCode), len(goCode), gasmCode, goCode)
|
||||
continue
|
||||
}
|
||||
for _, b := range goCode[len(gasmCode):] {
|
||||
if b != 0 {
|
||||
t.Errorf("%s: non-zero trailing bytes in go tool asm output", fn.Name)
|
||||
break
|
||||
}
|
||||
}
|
||||
matched++
|
||||
t.Logf("%s: MATCH (%d bytes)", fn.Name, len(gasmCode))
|
||||
}
|
||||
if matched == 0 {
|
||||
t.Fatal("no functions matched")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func keysOf(m map[string][]byte) []string {
|
||||
out := make([]string, 0, len(m))
|
||||
for k := range m {
|
||||
out = append(out, k)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// TestSemicolonSpellingParity pins that the ';' statement separator changes
|
||||
// nothing about the encoding: the one-line spelling assembles to exactly the
|
||||
// bytes of the same statements written one per line.
|
||||
func TestSemicolonSpellingParity(t *testing.T) {
|
||||
for _, tt := range []struct{ one, two string }{
|
||||
{"\tROLQ $3, DI; ROLQ $13, DI\n", "\tROLQ $3, DI\n\tROLQ $13, DI\n"},
|
||||
{"\tREP; MOVSQ\n", "\tREP\n\tMOVSQ\n"},
|
||||
{"\tXORQ AX, AX; XORQ CX, CX\n", "\tXORQ AX, AX\n\tXORQ CX, CX\n"},
|
||||
} {
|
||||
one := code("t.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n"+tt.one+"\tRET\n")
|
||||
two := code("t.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n"+tt.two+"\tRET\n")
|
||||
if !bytes.Equal(one, two) {
|
||||
t.Errorf("semicolon spelling %q: %x, want the two-line bytes %x", tt.one, one, two)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -12,7 +12,7 @@ import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
|
||||
@@ -94,9 +94,9 @@ DATA ·table<>+0(SB)/8, $0x1122334455667788
|
||||
}
|
||||
|
||||
// The debug_line program: LNE_set_address (the R_ADDR relocation
|
||||
// carries the function address), then one row per line change — the
|
||||
// carries the function address), then one row per line change; the
|
||||
// TEXT is on line 4 (a leading blank line precedes the include), the
|
||||
// instructions on lines 5–9 — an advance to the 20-byte end and an
|
||||
// instructions on lines 5-9; an advance to the 20-byte end and an
|
||||
// end-of-sequence.
|
||||
linesOff := le.Uint32(dataIdx[4*2:])
|
||||
lines := dataBlk[linesOff : linesOff+21]
|
||||
@@ -214,7 +214,7 @@ func main() {
|
||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||
}
|
||||
var pkgArch, work, linkLine, asmObj string
|
||||
for _, line := range strings.Split(string(buildLog), "\n") {
|
||||
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
@@ -278,7 +278,7 @@ func main() {
|
||||
newArch := filepath.Join(dir, "pkg.a")
|
||||
args := []string{"tool", "pack", "c", newArch}
|
||||
seen := map[string]bool{}
|
||||
for _, m := range strings.Fields(string(listOut)) {
|
||||
for m := range strings.FieldsSeq(string(listOut)) {
|
||||
if seen[m] {
|
||||
continue
|
||||
}
|
||||
|
||||
+313
-77
@@ -5,9 +5,11 @@ package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math"
|
||||
"sort"
|
||||
"strconv"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||
)
|
||||
|
||||
// Image is an assembled file: the function bodies laid out in source order,
|
||||
@@ -15,7 +17,7 @@ import (
|
||||
// file-local static symbols are encoded RIP-relative and resolved within the
|
||||
// image, so the raw bytes are self-consistent and executable at any base
|
||||
// address; references to external symbols are recorded as relocations
|
||||
// (Funcs[i].Relocs, Externals) and left unresolved — the object-file
|
||||
// (Funcs[i].Relocs, Externals) and left unresolved, the object-file
|
||||
// emitters turn them into linker relocations.
|
||||
type Image struct {
|
||||
Code []byte // concatenated function bodies
|
||||
@@ -24,6 +26,10 @@ type Image struct {
|
||||
Symbols map[string]int // static symbol → byte offset within the image
|
||||
DataSyms []DataSymbol // GLOBL symbols, in layout order
|
||||
Externals []string // referenced but undefined symbols, sorted
|
||||
// SourcePath is the assembled file's path, recorded in the DWARF
|
||||
// sections in place of a placeholder name. Empty when the image was
|
||||
// not built from a named file.
|
||||
SourcePath string
|
||||
}
|
||||
|
||||
// FuncLayout describes one assembled function within an Image.
|
||||
@@ -80,32 +86,45 @@ func (fl *FuncLayout) LineAt(offset int) int {
|
||||
return 0
|
||||
}
|
||||
|
||||
// Reloc is one static-symbol reference within a function body: the disp32
|
||||
// field at Off (function-relative) must reach the symbol plus Addend,
|
||||
// measured from After, the address just past the instruction. An External
|
||||
// relocation names a symbol no GLOBL in the file defines; the object-file
|
||||
// emitters carry it into the output's relocation table.
|
||||
// RelocKind discriminates the type of relocation needed.
|
||||
// RelocKind discriminates the relocation a static-symbol reference needs;
|
||||
// the encoders record one per SB reference, and the object-file emitters map
|
||||
// it to their format's relocation type.
|
||||
type RelocKind int
|
||||
|
||||
const (
|
||||
RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64)
|
||||
RelCall // R_CALL: CALL to a function symbol (amd64)
|
||||
RelTLSLE // R_TLS_LE: local-exec TLS load, no symbol (amd64 guard)
|
||||
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
|
||||
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
|
||||
RelRISCVJal // R_RISCV_JAL (J-type call)
|
||||
RelPCRelAbs // 32-bit absolute (R_RISCV_32)
|
||||
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
|
||||
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
|
||||
RelArm64Addr // R_ADDRARM64 (ADRP + ADD/LDR/STR pair)
|
||||
RelArm64Addr // R_ADDRARM64 (ADRP + ADD pair)
|
||||
RelArm64Branch // R_CALLARM64 (BL instruction)
|
||||
RelArm64LDST64 // R_ARM64_PCREL_LDST64 (ADRP + 64-bit LDR/STR pair)
|
||||
RelLoong64Branch // R_CALLLOONG64 (BL instruction)
|
||||
RelAddr // R_ADDR: the absolute address of a symbol held in a DATA field
|
||||
)
|
||||
|
||||
type Reloc struct {
|
||||
// Off is the function-relative offset of the field the linker patches
|
||||
// and After the address just past the instruction, the base the
|
||||
// assembler measures PC-relative displacements from. Name plus
|
||||
// Addend select the target: the symbol plus the byte offset. An
|
||||
// External relocation names a symbol no GLOBL in the file defines;
|
||||
// the object-file emitters carry it into the output's relocation
|
||||
// table. Siz is the width of the patched field and is set only for
|
||||
// data-field relocations (RelAddr, Off relative to the data symbol),
|
||||
// whose width is the DATA line's; code relocations take their width
|
||||
// from the architecture's instruction encoding.
|
||||
Off int
|
||||
After int
|
||||
Name string
|
||||
Addend int64
|
||||
External bool
|
||||
Kind RelocKind
|
||||
Siz uint8
|
||||
}
|
||||
|
||||
// DataSymbol describes one GLOBL symbol laid out in the data section.
|
||||
@@ -117,6 +136,11 @@ type DataSymbol struct {
|
||||
Static bool // the <> marker: file-local, not exported
|
||||
Rodata bool // the RODATA flag: read-only data
|
||||
Dupok bool // the DUPOK flag: duplicate-OK
|
||||
// Relocs carries the symbol-valued DATA initialisers ("DATA s+0(SB)/8,
|
||||
// $other(SB)"): fields of this symbol's data that hold another symbol's
|
||||
// address, resolved by the linker. Off is relative to the symbol's
|
||||
// data start.
|
||||
Relocs []Reloc
|
||||
}
|
||||
|
||||
// Bytes returns the whole image: code, then data.
|
||||
@@ -126,14 +150,26 @@ func (img *Image) Bytes() []byte {
|
||||
return append(out, img.Data...)
|
||||
}
|
||||
|
||||
// AssembleOption adjusts the file-level assembly context.
|
||||
type AssembleOption func(*linkInfo)
|
||||
|
||||
// WithGOOS selects the target operating system for the forms that depend on
|
||||
// it, the TLS access shape above all: linux and freebsd take the
|
||||
// one-instruction form, windows and plan9 keep the two-instruction load.
|
||||
func WithGOOS(goos string) AssembleOption {
|
||||
return func(l *linkInfo) {
|
||||
l.goos = goos
|
||||
}
|
||||
}
|
||||
|
||||
// AssembleFile assembles every TEXT function of a parsed file and lays out
|
||||
// its static symbols (GLOBL/DATA) in a data section behind the code. Each
|
||||
// reference to a file-local static symbol becomes a RIP-relative load whose
|
||||
// displacement is resolved against that layout; a reference to a symbol no
|
||||
// GLOBL defines is recorded as an external relocation (Externals) with its
|
||||
// displacement left zero — the object-file emitters resolve it at link
|
||||
// displacement left zero, the object-file emitters resolve it at link
|
||||
// time, while the raw image (Bytes) cannot represent it.
|
||||
func AssembleFile(f *ast.File) (*Image, error) {
|
||||
func AssembleFile(f *ast.File, opts ...AssembleOption) (*Image, error) {
|
||||
dataSyms, err := collectData(f)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -142,9 +178,21 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
||||
for _, d := range dataSyms {
|
||||
known[d.name] = true
|
||||
}
|
||||
// TEXT symbols are file-level definitions too: a symbol immediate
|
||||
// ($fn(SB)) may name one, exactly as a data reference names a GLOBL.
|
||||
for _, d := range f.Decls {
|
||||
if t, ok := d.(*ast.Text); ok {
|
||||
known[t.Name.Name] = true
|
||||
}
|
||||
}
|
||||
link := &linkInfo{symbols: known, allowExternal: true}
|
||||
for _, o := range opts {
|
||||
o(link)
|
||||
}
|
||||
poolSeen := map[string]bool{}
|
||||
|
||||
img := &Image{Symbols: map[string]int{}}
|
||||
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
||||
textOff := map[string]int{}
|
||||
type asmFunc struct {
|
||||
name string
|
||||
patches []sbPatch
|
||||
@@ -155,7 +203,26 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, patches, labels, steps, lines, err := assemble(t, link)
|
||||
code, patches, labels, steps, lines, pool, err := assemble(t, link)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||
}
|
||||
// The pooled floating-point constants join the declared data as
|
||||
// read-only symbols, deduplicated across the file (the toolchain
|
||||
// synthesises the same symbols into its rodata).
|
||||
for _, entry := range pool {
|
||||
if poolSeen[entry.name] {
|
||||
continue
|
||||
}
|
||||
poolSeen[entry.name] = true
|
||||
dataSyms = append(dataSyms, dataSym{
|
||||
name: entry.name,
|
||||
buf: entry.data,
|
||||
size: len(entry.data),
|
||||
rodata: true,
|
||||
dupok: true,
|
||||
})
|
||||
}
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||
}
|
||||
@@ -182,6 +249,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
||||
for _, s := range steps {
|
||||
fl.Spadj = append(fl.Spadj, SpadjStep{PC: s.pc, Value: s.value})
|
||||
}
|
||||
textOff[t.Name.Name] = len(img.Code)
|
||||
img.Funcs = append(img.Funcs, fl)
|
||||
img.Code = append(img.Code, code...)
|
||||
funcs = append(funcs, asmFunc{name: t.Name.Name, patches: patches})
|
||||
@@ -214,13 +282,27 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
||||
base := img.Funcs[i].Offset
|
||||
code := img.Code[base : base+img.Funcs[i].Size]
|
||||
for _, p := range fn.patches {
|
||||
reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend}
|
||||
reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend, Kind: p.kind}
|
||||
if p.kind == RelTLSLE {
|
||||
// The TLS slot has no symbol: the linker fills the offset
|
||||
// from the runtime's TLS layout.
|
||||
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
|
||||
continue
|
||||
}
|
||||
if imgOff, ok := img.Symbols[p.name]; ok {
|
||||
rel := int64(imgOff) + p.addend - int64(base+p.after)
|
||||
if rel < -1<<31 || rel >= 1<<31 {
|
||||
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
|
||||
}
|
||||
copy(code[p.off:p.off+4], le32(rel))
|
||||
} else if imgOff, ok := textOff[p.name]; ok {
|
||||
// A CALL to a TEXT function of the same file: resolve the
|
||||
// displacement against the function's layout position.
|
||||
rel := int64(imgOff) + p.addend - int64(base+p.after)
|
||||
if rel < -1<<31 || rel >= 1<<31 {
|
||||
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
|
||||
}
|
||||
copy(code[p.off:p.off+4], le32(rel))
|
||||
} else {
|
||||
reloc.External = true
|
||||
externals[p.name] = true
|
||||
@@ -228,6 +310,22 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
||||
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
|
||||
}
|
||||
}
|
||||
// The data symbols' symbol-valued DATA fields resolve the same way the
|
||||
// code references do: a name the file defines (GLOBL or TEXT) stays an
|
||||
// internal reference the emitters resolve, anything else is external.
|
||||
// img.DataSyms was laid out in dataSyms order, so the indexes line up.
|
||||
for i := range img.DataSyms {
|
||||
for _, r := range dataSyms[i].relocs {
|
||||
reloc := r
|
||||
if _, ok := img.Symbols[reloc.Name]; !ok {
|
||||
if _, ok := textOff[reloc.Name]; !ok {
|
||||
reloc.External = true
|
||||
externals[reloc.Name] = true
|
||||
}
|
||||
}
|
||||
img.DataSyms[i].Relocs = append(img.DataSyms[i].Relocs, reloc)
|
||||
}
|
||||
}
|
||||
for name := range externals {
|
||||
img.Externals = append(img.Externals, name)
|
||||
}
|
||||
@@ -244,17 +342,34 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// The pooled $i64 constants the wide MOV immediate loads refer to join
|
||||
// the declared data as read-only symbols, deduplicated across the file
|
||||
// (the toolchain synthesises the same symbols into its rodata).
|
||||
litSeen := map[string]bool{}
|
||||
|
||||
img := &Image{Symbols: map[string]int{}}
|
||||
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
||||
for _, d := range f.Decls {
|
||||
t, ok := d.(*ast.Text)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, labels, relocs, lines, spadj, err := assembleRISCV(t)
|
||||
code, labels, relocs, lines, spadj, lits, err := assembleRISCV(t)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||
}
|
||||
for _, lit := range lits {
|
||||
if litSeen[lit.Name] {
|
||||
continue
|
||||
}
|
||||
litSeen[lit.Name] = true
|
||||
dataSyms = append(dataSyms, dataSym{
|
||||
name: lit.Name,
|
||||
buf: lit.Data,
|
||||
size: len(lit.Data),
|
||||
rodata: true,
|
||||
dupok: true,
|
||||
})
|
||||
}
|
||||
fl := FuncLayout{
|
||||
Name: t.Name.Name,
|
||||
Pkg: t.Name.Pkg,
|
||||
@@ -302,6 +417,7 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
|
||||
})
|
||||
}
|
||||
|
||||
markExternals(img, dataSyms)
|
||||
return img, nil
|
||||
}
|
||||
|
||||
@@ -316,7 +432,7 @@ func AssembleFileLOONG64(f *ast.File) (*Image, error) {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
img := &Image{Symbols: map[string]int{}}
|
||||
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
||||
for _, d := range f.Decls {
|
||||
t, ok := d.(*ast.Text)
|
||||
if !ok {
|
||||
@@ -373,9 +489,55 @@ func AssembleFileLOONG64(f *ast.File) (*Image, error) {
|
||||
})
|
||||
}
|
||||
|
||||
markExternals(img, dataSyms)
|
||||
return img, nil
|
||||
}
|
||||
|
||||
// markExternals identifies relocations that reference symbols not defined in
|
||||
// the file (neither a GLOBL/DATA symbol nor a TEXT function) and records them
|
||||
// as external. The non-amd64 architectures emit relocations for every SB
|
||||
// reference; this post-processing step distinguishes file-local from external.
|
||||
func markExternals(img *Image, dataSyms []dataSym) {
|
||||
known := make(map[string]bool, len(dataSyms)+len(img.Funcs))
|
||||
for _, d := range dataSyms {
|
||||
known[d.name] = true
|
||||
}
|
||||
for _, fn := range img.Funcs {
|
||||
known[fn.Name] = true
|
||||
}
|
||||
externals := map[string]bool{}
|
||||
for i := range img.Funcs {
|
||||
for j := range img.Funcs[i].Relocs {
|
||||
r := &img.Funcs[i].Relocs[j]
|
||||
if !known[r.Name] {
|
||||
r.External = true
|
||||
externals[r.Name] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
// The declared data symbols carry the file's own relocations (the
|
||||
// symbol-valued DATA fields); the layouts appended img.DataSyms in
|
||||
// dataSyms order, so the indexes line up. The trailing entries (the
|
||||
// pooled arm64 literals) have no source relocations.
|
||||
for i := range img.DataSyms {
|
||||
if i >= len(dataSyms) {
|
||||
break
|
||||
}
|
||||
for _, r := range dataSyms[i].relocs {
|
||||
reloc := r
|
||||
if !known[reloc.Name] {
|
||||
reloc.External = true
|
||||
externals[reloc.Name] = true
|
||||
}
|
||||
img.DataSyms[i].Relocs = append(img.DataSyms[i].Relocs, reloc)
|
||||
}
|
||||
}
|
||||
for name := range externals {
|
||||
img.Externals = append(img.Externals, name)
|
||||
}
|
||||
sort.Strings(img.Externals)
|
||||
}
|
||||
|
||||
// dataSym is one GLOBL symbol and its DATA initialiser.
|
||||
type dataSym struct {
|
||||
name string
|
||||
@@ -385,81 +547,155 @@ type dataSym struct {
|
||||
static bool
|
||||
rodata bool
|
||||
dupok bool
|
||||
// relocs are the symbol-valued DATA fields, in declaration order; Off
|
||||
// is relative to the symbol's data start.
|
||||
relocs []Reloc
|
||||
}
|
||||
|
||||
// collectData gathers the file's static symbols (GLOBL) and their initial
|
||||
// contents (DATA) into byte buffers, in declaration order.
|
||||
// contents (DATA) into byte buffers. Two passes: the Plan 9 convention puts
|
||||
// every DATA line before its symbol's GLOBL, so the symbols are registered
|
||||
// before the initialisers are applied.
|
||||
func collectData(f *ast.File) ([]dataSym, error) {
|
||||
index := map[string]int{}
|
||||
var syms []dataSym
|
||||
for _, d := range f.Decls {
|
||||
switch dd := d.(type) {
|
||||
case *ast.Globl:
|
||||
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
||||
continue
|
||||
}
|
||||
name := dd.Name.Name
|
||||
if _, dup := index[name]; dup {
|
||||
return nil, fmt.Errorf("duplicate GLOBL %q", name)
|
||||
}
|
||||
size := 0
|
||||
if dd.Size != nil && dd.Size.Imm.HasVal {
|
||||
size = int(dd.Size.Imm.Val)
|
||||
}
|
||||
index[name] = len(syms)
|
||||
ds := dataSym{
|
||||
name: name,
|
||||
pkg: dd.Name.Pkg,
|
||||
buf: make([]byte, size),
|
||||
size: size,
|
||||
static: dd.Name.Static,
|
||||
}
|
||||
for _, f := range dd.Flags {
|
||||
switch f {
|
||||
case "RODATA":
|
||||
ds.rodata = true
|
||||
case "DUPOK":
|
||||
ds.dupok = true
|
||||
case "1":
|
||||
ds.dupok = true
|
||||
case "8":
|
||||
ds.rodata = true
|
||||
case "9":
|
||||
ds.dupok = true
|
||||
ds.rodata = true
|
||||
gd, ok := d.(*ast.Globl)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if gd.Name == nil || gd.Name.Pseudo != "SB" {
|
||||
continue
|
||||
}
|
||||
name := gd.Name.Name
|
||||
if _, dup := index[name]; dup {
|
||||
return nil, fmt.Errorf("duplicate GLOBL %q", name)
|
||||
}
|
||||
size := 0
|
||||
if gd.Size != nil && gd.Size.Imm.HasVal {
|
||||
size = int(gd.Size.Imm.Val)
|
||||
}
|
||||
index[name] = len(syms)
|
||||
ds := dataSym{
|
||||
name: name,
|
||||
pkg: gd.Name.Pkg,
|
||||
buf: make([]byte, size),
|
||||
size: size,
|
||||
static: gd.Name.Static,
|
||||
}
|
||||
for _, f := range gd.Flags {
|
||||
switch f {
|
||||
case "RODATA":
|
||||
ds.rodata = true
|
||||
case "DUPOK":
|
||||
ds.dupok = true
|
||||
default:
|
||||
// Legacy numeric flag constants (runtime/textflag.h):
|
||||
// DUPOK is 2, RODATA is 8; combinations arrive as one
|
||||
// number (e.g. 10 = RODATA|DUPOK).
|
||||
if n, err := strconv.Atoi(f); err == nil {
|
||||
if n&2 != 0 {
|
||||
ds.dupok = true
|
||||
}
|
||||
if n&8 != 0 {
|
||||
ds.rodata = true
|
||||
}
|
||||
}
|
||||
}
|
||||
syms = append(syms, ds)
|
||||
|
||||
case *ast.Data:
|
||||
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
||||
continue
|
||||
}
|
||||
syms = append(syms, ds)
|
||||
}
|
||||
for _, d := range f.Decls {
|
||||
dd, ok := d.(*ast.Data)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
||||
continue
|
||||
}
|
||||
i, ok := index[dd.Name.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
|
||||
}
|
||||
if dd.Value == nil {
|
||||
return nil, fmt.Errorf("DATA %q: missing value", dd.Name.Name)
|
||||
}
|
||||
w := dd.Width
|
||||
off := dd.Name.Offset
|
||||
buf := syms[i].buf
|
||||
if off < 0 || off+int64(w) > int64(len(buf)) {
|
||||
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
|
||||
}
|
||||
// A symbol value ("DATA s+0(SB)/8, $other(SB)", the rt0 spelling)
|
||||
// leaves the field zero and records a relocation against the named
|
||||
// symbol: the linker patches the absolute address at this data
|
||||
// offset. The toolchain emits the same shape, an R_ADDR of the
|
||||
// DATA width with the value's offset as the addend, on every
|
||||
// architecture.
|
||||
if sym := dd.Value.Imm.Sym; !dd.Value.Imm.HasVal && sym != nil {
|
||||
syms[i].relocs = append(syms[i].relocs, Reloc{
|
||||
Off: int(off),
|
||||
Name: sym.Name,
|
||||
Addend: sym.Offset,
|
||||
Kind: RelAddr,
|
||||
Siz: uint8(w),
|
||||
})
|
||||
continue
|
||||
}
|
||||
// A string or rune value ("DATA s+0(SB)/20, $"text"") writes its
|
||||
// bytes into the field and leaves the rest zero, the toolchain's
|
||||
// WriteString: the declared width must hold every byte, and any
|
||||
// width is legal.
|
||||
if s := dd.Value.Imm.Str; s != "" && !dd.Value.Imm.HasVal {
|
||||
text, err := strconv.Unquote(s)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("DATA %q: invalid string value %s", dd.Name.Name, s)
|
||||
}
|
||||
i, ok := index[dd.Name.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
|
||||
if len(text) > w {
|
||||
return nil, fmt.Errorf("DATA %q: string of %d bytes does not fit width %d", dd.Name.Name, len(text), w)
|
||||
}
|
||||
if dd.Value == nil || !dd.Value.Imm.HasVal {
|
||||
return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name)
|
||||
copy(buf[off:], text)
|
||||
continue
|
||||
}
|
||||
// A floating-point value stores its IEEE-754 bits: /4 the float32
|
||||
// rounding of the parsed double, /8 the full 64 bits, the
|
||||
// toolchain's WriteFloat32 and WriteFloat64.
|
||||
if f := dd.Value.Imm.Float; f != "" && !dd.Value.Imm.HasVal {
|
||||
num, err := strconv.ParseFloat(f, 64)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("DATA %q: invalid floating-point value %q", dd.Name.Name, f)
|
||||
}
|
||||
w := dd.Width
|
||||
switch w {
|
||||
case 1, 2, 4, 8:
|
||||
default:
|
||||
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
|
||||
}
|
||||
off := dd.Name.Offset
|
||||
buf := syms[i].buf
|
||||
if off < 0 || off+int64(w) > int64(len(buf)) {
|
||||
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
|
||||
}
|
||||
v := dd.Value.Imm.Val
|
||||
if dd.Value.Imm.Neg {
|
||||
v = -v
|
||||
num = -num
|
||||
}
|
||||
for j := 0; j < w; j++ {
|
||||
var v uint64
|
||||
switch w {
|
||||
case 4:
|
||||
v = uint64(math.Float32bits(float32(num)))
|
||||
case 8:
|
||||
v = math.Float64bits(num)
|
||||
default:
|
||||
return nil, fmt.Errorf("DATA %q: invalid width %d for a float (want 4 or 8)", dd.Name.Name, w)
|
||||
}
|
||||
for j := range w {
|
||||
buf[off+int64(j)] = byte(v >> (8 * j))
|
||||
}
|
||||
continue
|
||||
}
|
||||
if !dd.Value.Imm.HasVal {
|
||||
return nil, fmt.Errorf("DATA %q: value must be an integer immediate or a symbol address", dd.Name.Name)
|
||||
}
|
||||
switch w {
|
||||
case 1, 2, 4, 8:
|
||||
default:
|
||||
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
|
||||
}
|
||||
v := dd.Value.Imm.Val
|
||||
if dd.Value.Imm.Neg {
|
||||
v = -v
|
||||
}
|
||||
for j := range w {
|
||||
buf[off+int64(j)] = byte(v >> (8 * j))
|
||||
}
|
||||
}
|
||||
return syms, nil
|
||||
|
||||
+366
-3
@@ -4,14 +4,19 @@
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// TestAssembleFileStaticData checks the whole-image layout — code, padding
|
||||
// and the data section — and that the RIP-relative displacements of static
|
||||
// TestAssembleFileStaticData checks the whole-image layout; code, padding
|
||||
// and the data section; and that the RIP-relative displacements of static
|
||||
// symbol loads resolve to the right bytes.
|
||||
func TestAssembleFileStaticData(t *testing.T) {
|
||||
f, errs := parser.Parse("d_amd64.s", `
|
||||
@@ -128,3 +133,361 @@ DATA x<>+0(SB)/4, $1
|
||||
t.Errorf("single-function SB: error %v, want a file-level-assembly error", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCollectDataNumericFlags pins the numeric GLOBL flag constants from
|
||||
// runtime/textflag.h: DUPOK is 2, RODATA is 8, and combinations arrive as
|
||||
// one number (9 = NOPROF|RODATA, 10 = RODATA|DUPOK).
|
||||
func TestCollectDataNumericFlags(t *testing.T) {
|
||||
tests := []struct {
|
||||
flags string
|
||||
rodata bool
|
||||
dupok bool
|
||||
}{
|
||||
{"2", false, true},
|
||||
{"8", true, false},
|
||||
{"9", true, false}, // NOPROF|RODATA, not DUPOK
|
||||
{"10", true, true}, // RODATA|DUPOK
|
||||
{"RODATA", true, false},
|
||||
{"DUPOK", false, true},
|
||||
{"RODATA|DUPOK", true, true},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
src := "TEXT \u00b7f(SB), NOSPLIT, $0\n\tRET\nGLOBL sym(SB), " + tt.flags + ", $8\n"
|
||||
f, errs := parser.Parse("f_amd64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse %q: %v", tt.flags, errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble %q: %v", tt.flags, err)
|
||||
}
|
||||
if len(img.DataSyms) != 1 {
|
||||
t.Fatalf("%q: data syms = %d, want 1", tt.flags, len(img.DataSyms))
|
||||
}
|
||||
d := img.DataSyms[0]
|
||||
if d.Rodata != tt.rodata || d.Dupok != tt.dupok {
|
||||
t.Errorf("flags %q: rodata=%v dupok=%v, want rodata=%v dupok=%v",
|
||||
tt.flags, d.Rodata, d.Dupok, tt.rodata, tt.dupok)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestCollectDataSymbolValue covers the symbol-valued DATA field ("DATA
|
||||
// s+0(SB)/8, $other(SB)", the rt0 spelling): the field stays zero in the
|
||||
// image and the relocation is recorded against the named symbol, whatever
|
||||
// the file defines (a TEXT function, a GLOBL) or leaves external.
|
||||
func TestCollectDataSymbolValue(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·Keep(SB), NOSPLIT, $0-8
|
||||
MOVQ target+0(FP), AX
|
||||
RET
|
||||
GLOBL holder(SB), NOPTR, $32
|
||||
DATA holder+0(SB)/8, $·Keep(SB)
|
||||
DATA holder+8(SB)/8, $·Keep+5(SB)
|
||||
DATA holder+16(SB)/8, $holder(SB)
|
||||
GLOBL spare(SB), NOPTR, $8
|
||||
DATA spare+0(SB)/8, $extvar(SB)
|
||||
`
|
||||
f, errs := parser.Parse("f_amd64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
byName := map[string]DataSymbol{}
|
||||
for _, d := range img.DataSyms {
|
||||
byName[d.Name] = d
|
||||
}
|
||||
want := []struct {
|
||||
sym string
|
||||
off int
|
||||
name string
|
||||
addend int64
|
||||
ext bool
|
||||
}{
|
||||
{"holder", 0, "Keep", 0, false},
|
||||
{"holder", 8, "Keep", 5, false},
|
||||
{"holder", 16, "holder", 0, false},
|
||||
{"spare", 0, "extvar", 0, true},
|
||||
}
|
||||
var flat []struct {
|
||||
sym string
|
||||
r Reloc
|
||||
}
|
||||
for _, d := range img.DataSyms {
|
||||
for _, r := range d.Relocs {
|
||||
flat = append(flat, struct {
|
||||
sym string
|
||||
r Reloc
|
||||
}{d.Name, r})
|
||||
}
|
||||
}
|
||||
if len(flat) != len(want) {
|
||||
t.Fatalf("data relocations = %d, want %d", len(flat), len(want))
|
||||
}
|
||||
for i, w := range want {
|
||||
g := flat[i]
|
||||
r := g.r
|
||||
if g.sym != w.sym {
|
||||
t.Errorf("relocation %d sits on %q, want %q", i, g.sym, w.sym)
|
||||
continue
|
||||
}
|
||||
if r.Off != w.off || r.Name != w.name || r.Addend != w.addend || r.External != w.ext {
|
||||
t.Errorf("relocation %d = {+%d %q addend %d ext %v}, want {+%d %q addend %d ext %v}",
|
||||
i, r.Off, r.Name, r.Addend, r.External, w.off, w.name, w.addend, w.ext)
|
||||
}
|
||||
if r.Kind != RelAddr {
|
||||
t.Errorf("relocation %d kind = %v, want RelAddr", i, r.Kind)
|
||||
}
|
||||
if r.Siz != 8 {
|
||||
t.Errorf("relocation %d siz = %d, want 8", i, r.Siz)
|
||||
}
|
||||
}
|
||||
// The fields themselves stay zero: only the linker fills them.
|
||||
for _, b := range img.Data {
|
||||
if b != 0 {
|
||||
t.Fatal("data section is not all zero before relocation")
|
||||
}
|
||||
}
|
||||
if len(img.Externals) != 1 || img.Externals[0] != "extvar" {
|
||||
t.Errorf("Externals = %v, want [extvar]", img.Externals)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectDataSymbolReloc pins the GOOBJ record a symbol-valued DATA
|
||||
// field produces, against the shape the toolchain emits for the same
|
||||
// source: an R_ADDR of the DATA width at the field offset, pkgIdxNone plus
|
||||
// the non-package definition index when the target is the file's own TEXT
|
||||
// function (the rt0 lib entry spelling).
|
||||
func TestGOObjectDataSymbolReloc(t *testing.T) {
|
||||
f, errs := parser.Parse("f_amd64.s", `#include "textflag.h"
|
||||
TEXT ·Keep(SB), NOSPLIT, $0-8
|
||||
RET
|
||||
GLOBL holder(SB), NOPTR, $16
|
||||
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
obj, err := img.GOObject("main", "f_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObject: %v", err)
|
||||
}
|
||||
v := openGoobj(t, obj)
|
||||
// Walk every relocation record; the data record is the one of Siz 8
|
||||
// and type R_ADDR.
|
||||
var off, add int64
|
||||
var pkg, sym uint32
|
||||
found := false
|
||||
for data := v.blk(blkReloc); len(data) >= 23; data = data[23:] {
|
||||
if data[4] != 8 || binary.LittleEndian.Uint16(data[5:]) != relocAddr {
|
||||
continue
|
||||
}
|
||||
found = true
|
||||
off = int64(int32(binary.LittleEndian.Uint32(data[0:])))
|
||||
add = int64(binary.LittleEndian.Uint64(data[7:]))
|
||||
pkg = binary.LittleEndian.Uint32(data[15:])
|
||||
sym = binary.LittleEndian.Uint32(data[19:])
|
||||
break
|
||||
}
|
||||
if !found {
|
||||
t.Fatal("no data relocation record in the object")
|
||||
}
|
||||
if off != 0 || add != 5 {
|
||||
t.Errorf("data reloc = {off %d addend %d}, want {off 0 addend 5}", off, add)
|
||||
}
|
||||
if pkg != pkgIdxNone {
|
||||
t.Errorf("data reloc pkg = %#x, want pkgIdxNone (the TEXT function)", pkg)
|
||||
}
|
||||
// The function's non-package definition index: the four pc tables
|
||||
// precede it, so index 4.
|
||||
if sym != 4 {
|
||||
t.Errorf("data reloc sym = %d, want 4", sym)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectDataSymbolLink is the end-to-end proof for symbol-valued DATA
|
||||
// fields: the gasm object is substituted for the toolchain's and re-linked,
|
||||
// then executed, and the linked data word must hold the real address of the
|
||||
// function the DATA line named (runtime.FuncForPC identifies it).
|
||||
func TestGOObjectDataSymbolLink(t *testing.T) {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
t.Skip("no Go toolchain available")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
asmSrc := `#include "textflag.h"
|
||||
GLOBL entry(SB), NOPTR, $8
|
||||
DATA entry+0(SB)/8, $·keepme(SB)
|
||||
|
||||
TEXT ·keepme(SB), NOSPLIT, $0-0
|
||||
RET
|
||||
|
||||
TEXT ·entryptr(SB), NOSPLIT, $0-8
|
||||
MOVQ entry+0(SB), AX
|
||||
MOVQ AX, ret+0(FP)
|
||||
RET
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
mainSrc := `package main
|
||||
|
||||
import "runtime"
|
||||
|
||||
func keepme()
|
||||
func entryptr() uintptr
|
||||
|
||||
func main() {
|
||||
pc := entryptr()
|
||||
fn := runtime.FuncForPC(pc)
|
||||
if fn == nil {
|
||||
panic("the entry word does not point at a function")
|
||||
}
|
||||
if fn.Name() != "main.keepme" {
|
||||
panic("the entry word points at " + fn.Name())
|
||||
}
|
||||
}
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module dlink\n\ngo 1.21\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Capture the build: the package archive's asm object and the link line.
|
||||
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||
build.Dir = dir
|
||||
buildLog, err := build.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||
}
|
||||
var work, linkLine, asmObj string
|
||||
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
|
||||
asmObj = fieldAfter(line, "-o")
|
||||
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||
linkLine = line
|
||||
}
|
||||
}
|
||||
if work == "" || asmObj == "" || linkLine == "" {
|
||||
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
|
||||
}
|
||||
defer os.RemoveAll(work)
|
||||
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
|
||||
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||
|
||||
// Assemble the same source with gasm and substitute the object.
|
||||
src, err := os.ReadFile(filepath.Join(dir, "main_amd64.s"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f, errs := parser.Parse("main_amd64.s", string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
gasmObj, err := img.GOObject("dlink", "main_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObject: %v", err)
|
||||
}
|
||||
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
|
||||
t.Fatalf("write gasm object: %v", err)
|
||||
}
|
||||
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
|
||||
if out, err := linkCmd.CombinedOutput(); err != nil {
|
||||
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
|
||||
}
|
||||
|
||||
// The linked program must run and find the right function behind the
|
||||
// data word.
|
||||
out, err := exec.Command(filepath.Join(dir, "prog")).CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("linked program failed: %v\n%s", err, out)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCollectDataFloatAndStringValues covers the non-integer DATA values the
|
||||
// runtime's math and asm files use: floating-point initialisers store their
|
||||
// IEEE-754 bits (/4 the float32 rounding, /8 the full double) and string
|
||||
// initialisers write their bytes zero-padded within the declared width.
|
||||
func TestCollectDataFloatAndStringValues(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·Keep(SB), NOSPLIT, $0-8
|
||||
RET
|
||||
GLOBL vals<>(SB), RODATA, $44
|
||||
DATA vals<>+0(SB)/8, $0.5
|
||||
DATA vals<>+8(SB)/8, $-1.0
|
||||
DATA vals<>+16(SB)/4, $1.5
|
||||
DATA vals<>+20(SB)/16, $"call frame too "
|
||||
DATA vals<>+36(SB)/4, $"hi"
|
||||
`
|
||||
f, errs := parser.Parse("fvals_amd64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
byName := map[string]DataSymbol{}
|
||||
for _, d := range img.DataSyms {
|
||||
byName[d.Name] = d
|
||||
}
|
||||
d := byName["vals"]
|
||||
if d.Size != 44 {
|
||||
t.Fatalf("vals size = %d, want 44", d.Size)
|
||||
}
|
||||
buf := img.Data[d.Offset : d.Offset+44]
|
||||
// 0.5 = 0x3FE0000000000000, -1.0 = 0xBFF0000000000000 (float64);
|
||||
// 1.5 = 0x3FC00000 (float32).
|
||||
for _, c := range []struct {
|
||||
off int
|
||||
want []byte
|
||||
}{
|
||||
{0, []byte{0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xE0, 0x3F}},
|
||||
{8, []byte{0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xF0, 0xBF}},
|
||||
{16, []byte{0x00, 0x00, 0xC0, 0x3F}},
|
||||
{20, []byte("call frame too ")},
|
||||
{36, []byte{'h', 'i', 0x00, 0x00}},
|
||||
} {
|
||||
if string(buf[c.off:c.off+len(c.want)]) != string(c.want) {
|
||||
t.Errorf("vals+%d: got % x, want % x", c.off, buf[c.off:c.off+len(c.want)], c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestCollectDataValueErrors pins the value-kind width rules: a float needs
|
||||
// width 4 or 8, a string must fit its declared width, and a bad float
|
||||
// literal is diagnosed rather than stored.
|
||||
func TestCollectDataValueErrors(t *testing.T) {
|
||||
cases := []string{
|
||||
`GLOBL v<>(SB), RODATA, $4
|
||||
DATA v<>+0(SB)/1, $0.5`,
|
||||
`GLOBL v<>(SB), RODATA, $2
|
||||
DATA v<>+0(SB)/2, $"toolarge"`,
|
||||
}
|
||||
for i, src := range cases {
|
||||
full := "#include \"textflag.h\"\nTEXT ·Keep(SB), NOSPLIT, $0-8\n\tRET\n" + src
|
||||
f, errs := parser.Parse(fmt.Sprintf("verr%d_amd64.s", i), full)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("case %d parse: %v", i, errs)
|
||||
}
|
||||
if _, err := AssembleFile(f); err == nil {
|
||||
t.Errorf("case %d: expected an error, got none", i)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+1068
-139
File diff suppressed because it is too large
Load Diff
+628
-24
@@ -9,7 +9,7 @@ package asm
|
||||
// an opcode constant, and the format selects the bit layout. The opcode
|
||||
// constants and formats are transcribed from the Go toolchain's own loong64
|
||||
// backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm`
|
||||
// exactly — the ground-truth oracle for the verify suite.
|
||||
// exactly, the ground-truth oracle for the verify suite.
|
||||
//
|
||||
// All LoongArch instructions are 32 bits, little-endian. The formats used
|
||||
// here (per the LoongArch Volume I specification):
|
||||
@@ -30,9 +30,14 @@ package asm
|
||||
// of the immediate and register fields), mirroring the toolchain's OP_*
|
||||
// helpers, so each l64* function only ORs its fields in.
|
||||
|
||||
import (
|
||||
"maps"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// loong64RegNum returns the 5-bit register number for a LoongArch register
|
||||
// name: R0–R31 (integer), F0–F31 (floating point), FCC0–FCC7 (condition
|
||||
// flags), FCSR0–FCSR31 (control/status) and the ABI aliases the runtime's
|
||||
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
|
||||
// flags), FCSR0-FCSR31 (control/status) and the ABI aliases the runtime's
|
||||
// assembly uses. Returns -1 for an unrecognised name.
|
||||
func loong64RegNum(name string) int {
|
||||
switch name {
|
||||
@@ -101,12 +106,17 @@ func loong64RegNum(name string) int {
|
||||
case "R31", "S8":
|
||||
return 31
|
||||
}
|
||||
// F0–F31, FCC0–FCC7, FCSR0–FCSR31.
|
||||
// F0-F31, FCC0-FCC7, FCSR0-FCSR31. The LSX/LASX vector banks (V0-V31,
|
||||
// X0-X31) are deliberately NOT accepted here: they are a separate
|
||||
// register class, and the toolchain rejects V/X names wherever an
|
||||
// integer or FP register is expected (GOARCH=loong64 go tool asm reports
|
||||
// "unrecognized instruction" for `BEQZ X0`). Vector operands are
|
||||
// resolved only through loong64VecRegNum.
|
||||
if len(name) >= 4 && name[:4] == "FCSR" {
|
||||
return loong64RegSpecial(name[4:], "FCSR", 31)
|
||||
return loong64RegSpecial(name[4:], 31)
|
||||
}
|
||||
if len(name) >= 3 && name[:3] == "FCC" {
|
||||
return loong64RegSpecial(name[3:], "FCC", 7)
|
||||
return loong64RegSpecial(name[3:], 7)
|
||||
}
|
||||
if len(name) < 2 {
|
||||
return -1
|
||||
@@ -129,7 +139,7 @@ func loong64RegNum(name string) int {
|
||||
}
|
||||
|
||||
// loong64RegSpecial parses a numbered FCC/FCSR register.
|
||||
func loong64RegSpecial(digits, prefix string, max int) int {
|
||||
func loong64RegSpecial(digits string, max int) int {
|
||||
if digits == "" {
|
||||
return -1
|
||||
}
|
||||
@@ -146,6 +156,19 @@ func loong64RegSpecial(digits, prefix string, max int) int {
|
||||
return -1
|
||||
}
|
||||
|
||||
// loong64VecRegNum resolves an LSX/LASX vector register name (V0-V31 or
|
||||
// X0-X31) to its 5-bit number, or -1. The vector banks are a register class
|
||||
// of their own: the toolchain accepts them only in the vector operands of the
|
||||
// LSX/LASX instructions (GOARCH=loong64 go tool asm assembles `VADDV V0, V1,
|
||||
// V2` and `XVADDV X0, X1, X2`, and rejects `VADDV R4, R5, R6`), so the V/X
|
||||
// spellings never reach the integer/FP resolver.
|
||||
func loong64VecRegNum(name string) int {
|
||||
if len(name) < 2 || (name[0] != 'V' && name[0] != 'X') {
|
||||
return -1
|
||||
}
|
||||
return loong64RegSpecial(name[1:], 31)
|
||||
}
|
||||
|
||||
// ---- format helpers ----
|
||||
|
||||
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
|
||||
@@ -197,7 +220,9 @@ func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
|
||||
}
|
||||
|
||||
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
|
||||
// The msb/lsb fields are 6 bits wide (0–63) and are validated by the caller.
|
||||
// The msb/lsb fields are 6 bits wide and are inserted unmasked: the caller
|
||||
// must have validated them (0..31 for the .w forms, 0..63 for the .d forms,
|
||||
// lsb <= msb), the same rule the toolchain enforces as "illegal bit number".
|
||||
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
|
||||
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
|
||||
}
|
||||
@@ -243,7 +268,7 @@ const (
|
||||
l64Firr14 // 2RI14 (ldptr/stptr)
|
||||
l64Firr16 // 2RI16 (addu16i.d)
|
||||
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
|
||||
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub)
|
||||
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub, fsel)
|
||||
l64Firir // bstrins/bstrpick
|
||||
l64Firrr // alsl
|
||||
l64Fi15 // syscall/break/dbar
|
||||
@@ -251,6 +276,9 @@ const (
|
||||
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
|
||||
l64Fshift // 2RI12 with a 5/6-bit shift immediate
|
||||
l64Fpreld // preld (2RI12 + 5-bit hint)
|
||||
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
|
||||
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
|
||||
l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd
|
||||
)
|
||||
|
||||
// l64Enc is one instruction's encoding: its bit layout (format) and the
|
||||
@@ -273,12 +301,74 @@ type l64DualEnc struct {
|
||||
var l64DualTable = map[string]l64DualEnc{}
|
||||
|
||||
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
|
||||
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered
|
||||
// yet; the base integer, memory and floating-point ISA is complete.
|
||||
// to their encoding.
|
||||
var l64InstrTable = map[string]l64Enc{}
|
||||
|
||||
// l64Vec3Enc pairs a vector opcode with its register bank: false = LSX
|
||||
// (V0-V31), true = LASX (X0-X31). The toolchain accepts one bank per
|
||||
// spelling: GOARCH=loong64 go tool asm assembles `VADDV V1, V2, V3` and
|
||||
// `XVADDV X1, X2, X3`, and rejects the crossed spellings.
|
||||
type l64Vec3Enc struct {
|
||||
op uint32
|
||||
lasx bool
|
||||
}
|
||||
|
||||
// l64VecImmEnc carries the immediate-form encoding of a vector mnemonic:
|
||||
// the opcode, the bank, the accepted immediate range, the bias the toolchain
|
||||
// adds (vsrai.b encodes imm+8) and the mask of the encoded field (vseqi.b
|
||||
// keeps a 5-bit two's-complement value, vseqi.d a 7-bit one).
|
||||
type l64VecImmEnc struct {
|
||||
op uint32
|
||||
lasx bool
|
||||
min, max int
|
||||
bias int
|
||||
mask int
|
||||
}
|
||||
|
||||
// l64VecBank marks the LSX/LASX mnemonics and records which register bank
|
||||
// each accepts; presence in the map routes the mnemonic through the vector
|
||||
// dispatcher rather than the integer/FP formats.
|
||||
var l64VecBank = map[string]bool{}
|
||||
|
||||
// l64VecImmInfo mirrors l64VecImmTable for the dispatcher.
|
||||
var l64VecImmInfo = map[string]l64VecImmEnc{}
|
||||
|
||||
// l64Vec2R marks the two-operand vector mnemonics (INSTR vj, vd, such as
|
||||
// vpcnt.v).
|
||||
var l64Vec2R = map[string]bool{}
|
||||
|
||||
// l64Vec4R marks the four-operand vector mnemonics (INSTR va, vk, vj, vd,
|
||||
// such as vshuf.b).
|
||||
var l64Vec4R = map[string]bool{}
|
||||
|
||||
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit
|
||||
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels.
|
||||
type l64VmovqEnc struct {
|
||||
ld, st, ldx, stx uint32 // plain and indexed load/store
|
||||
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
|
||||
pickS, pickU uint32 // vpickve2gr.{,u} element extract
|
||||
ins uint32 // vinsgr2vr element insert
|
||||
dup uint32 // vreplgr2vr duplicate (width in [11:10])
|
||||
move uint32 // vori.b/xvori.b $0 register move
|
||||
}
|
||||
|
||||
var l64VmovqTable = map[bool]l64VmovqEnc{
|
||||
false: { // VMOVQ, the LSX (V) bank
|
||||
ld: 0x5800 << 15, st: 0x5880 << 15, ldx: 0x7080 << 15, stx: 0x7088 << 15,
|
||||
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
|
||||
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
|
||||
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
|
||||
},
|
||||
true: { // XVMOVQ, the LASX (X) bank
|
||||
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
|
||||
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
|
||||
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
|
||||
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
|
||||
},
|
||||
}
|
||||
|
||||
func init() {
|
||||
// 3R — integer.
|
||||
// 3R, integer.
|
||||
rrr := map[string]uint32{
|
||||
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
|
||||
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
|
||||
@@ -298,7 +388,7 @@ func init() {
|
||||
"CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15,
|
||||
"CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15,
|
||||
}
|
||||
// 3R — floating point.
|
||||
// 3R, floating point.
|
||||
rrr["MULF"] = 0x209 << 15
|
||||
rrr["MULD"] = 0x20a << 15
|
||||
rrr["DIVF"] = 0x20d << 15
|
||||
@@ -356,6 +446,18 @@ func init() {
|
||||
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
|
||||
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
|
||||
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
|
||||
// LSX: convert a 64-bit integer lane to a double float. The operand
|
||||
// bank is the FP registers (the toolchain spells it `FFINTDV F0, F1`),
|
||||
// so the entry stays on the 2R integer/FP format.
|
||||
"FFINTDV": 0x474a << 10,
|
||||
// The rest of the scalar conversions (all F-bank, 2R).
|
||||
"FFINTFW": 0x4744 << 10, // ffint.s.w
|
||||
"FFINTFV": 0x4746 << 10, // ffint.s.l
|
||||
"FFINTDW": 0x4748 << 10, // ffint.d.w
|
||||
"FTINTWF": 0x46c1 << 10, // ftint.w.s
|
||||
"FTINTWD": 0x46c2 << 10, // ftint.w.d
|
||||
"FTINTVF": 0x46c9 << 10, // ftint.l.s
|
||||
"FTINTVD": 0x46ca << 10, // ftint.l.d
|
||||
}
|
||||
for m, op := range rr {
|
||||
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
|
||||
@@ -368,7 +470,7 @@ func init() {
|
||||
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
|
||||
// selected by the operand kind; the shift mnemonics pair the 3R form
|
||||
// with a 5/6-bit shift immediate.
|
||||
for m, e := range map[string]l64DualEnc{
|
||||
maps.Copy(l64DualTable, map[string]l64DualEnc{
|
||||
"ADD": {rrr: 0x20 << 15, imm: 0x00a << 22},
|
||||
"ADDW": {rrr: 0x20 << 15, imm: 0x00a << 22},
|
||||
"ADDV": {rrr: 0x21 << 15, imm: 0x00b << 22},
|
||||
@@ -386,16 +488,14 @@ func init() {
|
||||
"SRLV": {rrr: 0x32 << 15, imm: 0x0045 << 16, shift: true},
|
||||
"SRAV": {rrr: 0x33 << 15, imm: 0x0049 << 16, shift: true},
|
||||
"ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true},
|
||||
} {
|
||||
l64DualTable[m] = e
|
||||
}
|
||||
})
|
||||
|
||||
// 2RI12 — pure immediate arithmetic (LU52ID has no register form).
|
||||
// 2RI12, pure immediate arithmetic (LU52ID has no register form).
|
||||
l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22}
|
||||
// ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16.
|
||||
l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26}
|
||||
|
||||
// 2RI14 — LL/SC are aliased by the Go assembler to the pointer loads and
|
||||
// 2RI14, LL/SC are aliased by the Go assembler to the pointer loads and
|
||||
// stores (ldptr/stptr), with the offset scaled by 4.
|
||||
l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w
|
||||
l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d
|
||||
@@ -414,18 +514,20 @@ func init() {
|
||||
// LUI is the Plan 9 spelling of lu12i.w.
|
||||
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
||||
|
||||
// 4R — fused multiply-add.
|
||||
// 4R, fused multiply-add, and FSEL (fsel.d: the first operand is a FCC
|
||||
// condition flag, the layout matches the 4R shape).
|
||||
rrrr := map[string]uint32{
|
||||
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
|
||||
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
|
||||
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
|
||||
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
|
||||
"FSEL": 0x340 << 18,
|
||||
}
|
||||
for m, op := range rrrr {
|
||||
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
|
||||
}
|
||||
|
||||
// IRIR — bit-field insert/extract.
|
||||
// IRIR, bit-field insert/extract.
|
||||
irir := map[string]uint32{
|
||||
"BSTRINSW": 0x3<<21 | 0x0<<15,
|
||||
"BSTRINSV": 0x2 << 22,
|
||||
@@ -436,7 +538,7 @@ func init() {
|
||||
l64InstrTable[m] = l64Enc{format: l64Firir, op: op}
|
||||
}
|
||||
|
||||
// 3RI2 — ALSL.
|
||||
// 3RI2, ALSL.
|
||||
irrr := map[string]uint32{
|
||||
"ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17,
|
||||
}
|
||||
@@ -452,7 +554,11 @@ func init() {
|
||||
// PRELD.
|
||||
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
|
||||
|
||||
// Atomics — 3R with the AM field order (rk=value, rj=address, rd=result).
|
||||
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
|
||||
// The toolchain's form is three operands, `AMADDW rk, (rj), rd`
|
||||
// (cmd/asm/internal/asm/testdata/loong64enc1.s and
|
||||
// internal/runtime/atomic/atomic_loong64.s); the two-register spelling
|
||||
// is rejected by the oracle.
|
||||
am := map[string]uint32{
|
||||
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
|
||||
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
|
||||
@@ -470,14 +576,512 @@ func init() {
|
||||
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
|
||||
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
|
||||
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
|
||||
// The _dbar (acquire/release) add, and, or variants: opcodes read off
|
||||
// `go tool objdump` of `AMADDDBW R14, (R13), R12` and friends.
|
||||
"AMADDDBW": 0x070D4 << 15, "AMADDDBV": 0x070D5 << 15,
|
||||
"AMANDDBW": 0x070D6 << 15, "AMANDDBV": 0x070D7 << 15,
|
||||
"AMORDBW": 0x070D8 << 15, "AMORDBV": 0x070D9 << 15,
|
||||
// The remaining _dbar exchange variants (loong64enc1.s).
|
||||
"AMXORDBW": 0x070DA << 15, "AMXORDBV": 0x070DB << 15,
|
||||
"AMMAXDBW": 0x070DC << 15, "AMMAXDBV": 0x070DD << 15,
|
||||
"AMMINDBW": 0x070DE << 15, "AMMINDBV": 0x070DF << 15,
|
||||
"AMMAXDBWU": 0x070E0 << 15, "AMMAXDBVU": 0x070E1 << 15,
|
||||
"AMMINDBWU": 0x070E2 << 15, "AMMINDBVU": 0x070E3 << 15,
|
||||
}
|
||||
for m, op := range am {
|
||||
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
|
||||
}
|
||||
|
||||
// ---- LSX/LASX (V*/XV*) ----
|
||||
// Every opcode below was read off `go tool objdump` of a GOARCH=loong64
|
||||
// `go tool asm` kernel (the toolchain's own loong64enc1.s cross-checks
|
||||
// most of them), not assumed from the LoongArch manual.
|
||||
|
||||
// Three vector registers: INSTR vk, vj, vd (or INSTR vk, vd with
|
||||
// vj = vd). l64Vec3Enc.lasx selects the register bank the toolchain
|
||||
// accepts: LSX spellings take V0-V31, LASX spellings X0-X31.
|
||||
vec3 := map[string]l64Vec3Enc{
|
||||
"VADDW": {0xE016 << 15, false}, "VADDV": {0xE017 << 15, false},
|
||||
"VANDV": {0xE24C << 15, false}, "VXORV": {0xE24E << 15, false},
|
||||
"VSEQB": {0xE000 << 15, false}, "VSEQV": {0xE003 << 15, false},
|
||||
"VSRAB": {0xE1D8 << 15, false}, "VROTRW": {0xE1DE << 15, false},
|
||||
"XVADDV": {0xE817 << 15, true},
|
||||
"XVANDV": {0xEA4C << 15, true}, "XVXORV": {0xEA4E << 15, true},
|
||||
"XVSEQB": {0xE800 << 15, true}, "XVSEQV": {0xE803 << 15, true},
|
||||
}
|
||||
|
||||
// The integer and FP add/subtract families: [X]VADD and [X]VSUB by lane
|
||||
// width, plus the [X]VSADD/[X]VSSUB saturating pairs.
|
||||
// Opcodes transcribed from the toolchain's loong64enc1.s.
|
||||
addsub := map[string]l64Vec3Enc{
|
||||
"VADDB": {0xE014 << 15, false}, "VADDH": {0xE015 << 15, false},
|
||||
"VADDD": {0xE262 << 15, false}, "VADDF": {0xE261 << 15, false},
|
||||
"VADDQ": {0xE25A << 15, false},
|
||||
"VSUBB": {0xE018 << 15, false}, "VSUBH": {0xE019 << 15, false},
|
||||
"VSUBW": {0xE01A << 15, false}, "VSUBV": {0xE01B << 15, false},
|
||||
"VSUBQ": {0xE25B << 15, false},
|
||||
"VSUBF": {0xE265 << 15, false}, "VSUBD": {0xE266 << 15, false},
|
||||
"VSADDB": {0xE08C << 15, false}, "VSADDH": {0xE08D << 15, false},
|
||||
"VSADDW": {0xE08E << 15, false}, "VSADDV": {0xE08F << 15, false},
|
||||
"VSADDBU": {0xE094 << 15, false}, "VSADDHU": {0xE095 << 15, false},
|
||||
"VSADDWU": {0xE096 << 15, false}, "VSADDVU": {0xE097 << 15, false},
|
||||
"VSSUBB": {0xE090 << 15, false}, "VSSUBH": {0xE091 << 15, false},
|
||||
"VSSUBW": {0xE092 << 15, false}, "VSSUBV": {0xE093 << 15, false},
|
||||
"VSSUBBU": {0xE098 << 15, false}, "VSSUBHU": {0xE099 << 15, false},
|
||||
"VSSUBWU": {0xE09A << 15, false}, "VSSUBVU": {0xE09B << 15, false},
|
||||
"XVADDB": {0xE814 << 15, true}, "XVADDH": {0xE815 << 15, true},
|
||||
"XVADDW": {0xE816 << 15, true},
|
||||
"XVADDD": {0xEA62 << 15, true}, "XVADDF": {0xEA61 << 15, true},
|
||||
"XVADDQ": {0xEA5A << 15, true},
|
||||
"XVSUBB": {0xE818 << 15, true}, "XVSUBH": {0xE819 << 15, true},
|
||||
"XVSUBW": {0xE81A << 15, true}, "XVSUBV": {0xE81B << 15, true},
|
||||
"XVSUBQ": {0xEA5B << 15, true},
|
||||
"XVSUBF": {0xEA65 << 15, true}, "XVSUBD": {0xEA66 << 15, true},
|
||||
"XVSADDB": {0xE88C << 15, true}, "XVSADDH": {0xE88D << 15, true},
|
||||
"XVSADDW": {0xE88E << 15, true}, "XVSADDV": {0xE88F << 15, true},
|
||||
"XVSADDBU": {0xE894 << 15, true}, "XVSADDHU": {0xE895 << 15, true},
|
||||
"XVSADDWU": {0xE896 << 15, true}, "XVSADDVU": {0xE897 << 15, true},
|
||||
"XVSSUBB": {0xE890 << 15, true}, "XVSSUBH": {0xE891 << 15, true},
|
||||
"XVSSUBW": {0xE892 << 15, true}, "XVSSUBV": {0xE893 << 15, true},
|
||||
"XVSSUBBU": {0xE898 << 15, true}, "XVSSUBHU": {0xE899 << 15, true},
|
||||
"XVSSUBWU": {0xE89A << 15, true}, "XVSSUBVU": {0xE89B << 15, true},
|
||||
}
|
||||
|
||||
// The multiply families: plain and high-half [X]VMUL/[X]VMUH, the
|
||||
// widening [X]VMULW{EV,OD} ladder and its accumulating [X]VMADDW twins,
|
||||
// plus the [X]VMADD/[X]VMSUB fused multiply-add and the [X]VDIV/[X]VMOD
|
||||
// divide and modulo pairs.
|
||||
muldiv := map[string]l64Vec3Enc{
|
||||
"VMULB": {0xE108 << 15, false}, "VMULH": {0xE109 << 15, false},
|
||||
"VMULW": {0xE10A << 15, false}, "VMULV": {0xE10B << 15, false},
|
||||
"VMUHB": {0xE10C << 15, false}, "VMUHH": {0xE10D << 15, false},
|
||||
"VMUHW": {0xE10E << 15, false}, "VMUHV": {0xE10F << 15, false},
|
||||
"VMUHBU": {0xE110 << 15, false}, "VMUHHU": {0xE111 << 15, false},
|
||||
"VMUHWU": {0xE112 << 15, false}, "VMUHVU": {0xE113 << 15, false},
|
||||
"VMULWEVHB": {0xE120 << 15, false}, "VMULWEVWH": {0xE121 << 15, false},
|
||||
"VMULWEVVW": {0xE122 << 15, false}, "VMULWEVQV": {0xE123 << 15, false},
|
||||
"VMULWODHB": {0xE124 << 15, false}, "VMULWODWH": {0xE125 << 15, false},
|
||||
"VMULWODVW": {0xE126 << 15, false}, "VMULWODQV": {0xE127 << 15, false},
|
||||
"VMULWEVHBU": {0xE130 << 15, false}, "VMULWEVWHU": {0xE131 << 15, false},
|
||||
"VMULWEVVWU": {0xE132 << 15, false}, "VMULWEVQVU": {0xE133 << 15, false},
|
||||
"VMULWODHBU": {0xE134 << 15, false}, "VMULWODWHU": {0xE135 << 15, false},
|
||||
"VMULWODVWU": {0xE136 << 15, false}, "VMULWODQVU": {0xE137 << 15, false},
|
||||
"VMULWEVHBUB": {0xE140 << 15, false}, "VMULWEVWHUH": {0xE141 << 15, false},
|
||||
"VMULWEVVWUW": {0xE142 << 15, false}, "VMULWEVQVUV": {0xE143 << 15, false},
|
||||
"VMULWODHBUB": {0xE144 << 15, false}, "VMULWODWHUH": {0xE145 << 15, false},
|
||||
"VMULWODVWUW": {0xE146 << 15, false}, "VMULWODQVUV": {0xE147 << 15, false},
|
||||
"VMADDB": {0xE150 << 15, false}, "VMADDH": {0xE151 << 15, false},
|
||||
"VMADDW": {0xE152 << 15, false}, "VMADDV": {0xE153 << 15, false},
|
||||
"VMSUBB": {0xE154 << 15, false}, "VMSUBH": {0xE155 << 15, false},
|
||||
"VMSUBW": {0xE156 << 15, false}, "VMSUBV": {0xE157 << 15, false},
|
||||
"VMADDWEVHB": {0xE158 << 15, false}, "VMADDWEVWH": {0xE159 << 15, false},
|
||||
"VMADDWEVVW": {0xE15A << 15, false}, "VMADDWEVQV": {0xE15B << 15, false},
|
||||
"VMADDWODHB": {0xE15C << 15, false}, "VMADDWODWH": {0xE15D << 15, false},
|
||||
"VMADDWODVW": {0xE15E << 15, false}, "VMADDWODQV": {0xE15F << 15, false},
|
||||
"VMADDWEVHBU": {0xE168 << 15, false}, "VMADDWEVWHU": {0xE169 << 15, false},
|
||||
"VMADDWEVVWU": {0xE16A << 15, false}, "VMADDWEVQVU": {0xE16B << 15, false},
|
||||
"VMADDWODHBU": {0xE16C << 15, false}, "VMADDWODWHU": {0xE16D << 15, false},
|
||||
"VMADDWODVWU": {0xE16E << 15, false}, "VMADDWODQVU": {0xE16F << 15, false},
|
||||
"VMADDWEVHBUB": {0xE178 << 15, false}, "VMADDWEVWHUH": {0xE179 << 15, false},
|
||||
"VMADDWEVVWUW": {0xE17A << 15, false}, "VMADDWEVQVUV": {0xE17B << 15, false},
|
||||
"VMADDWODHBUB": {0xE17C << 15, false}, "VMADDWODWHUH": {0xE17D << 15, false},
|
||||
"VMADDWODVWUW": {0xE17E << 15, false}, "VMADDWODQVUV": {0xE17F << 15, false},
|
||||
"VDIVB": {0xE1C0 << 15, false}, "VDIVH": {0xE1C1 << 15, false},
|
||||
"VDIVW": {0xE1C2 << 15, false}, "VDIVV": {0xE1C3 << 15, false},
|
||||
"VMODB": {0xE1C4 << 15, false}, "VMODH": {0xE1C5 << 15, false},
|
||||
"VMODW": {0xE1C6 << 15, false}, "VMODV": {0xE1C7 << 15, false},
|
||||
"VDIVBU": {0xE1C8 << 15, false}, "VDIVHU": {0xE1C9 << 15, false},
|
||||
"VDIVWU": {0xE1CA << 15, false}, "VDIVVU": {0xE1CB << 15, false},
|
||||
"VMODBU": {0xE1CC << 15, false}, "VMODHU": {0xE1CD << 15, false},
|
||||
"VMODWU": {0xE1CE << 15, false}, "VMODVU": {0xE1CF << 15, false},
|
||||
"VMULF": {0xE271 << 15, false}, "VMULD": {0xE272 << 15, false},
|
||||
"VDIVF": {0xE275 << 15, false}, "VDIVD": {0xE276 << 15, false},
|
||||
"XVMULB": {0xE908 << 15, true}, "XVMULH": {0xE909 << 15, true},
|
||||
"XVMULW": {0xE90A << 15, true}, "XVMULV": {0xE90B << 15, true},
|
||||
"XVMUHB": {0xE90C << 15, true}, "XVMUHH": {0xE90D << 15, true},
|
||||
"XVMUHW": {0xE90E << 15, true}, "XVMUHV": {0xE90F << 15, true},
|
||||
"XVMUHBU": {0xE910 << 15, true}, "XVMUHHU": {0xE911 << 15, true},
|
||||
"XVMUHWU": {0xE912 << 15, true}, "XVMUHVU": {0xE913 << 15, true},
|
||||
"XVMULWEVHB": {0xE920 << 15, true}, "XVMULWEVWH": {0xE921 << 15, true},
|
||||
"XVMULWEVVW": {0xE922 << 15, true}, "XVMULWEVQV": {0xE923 << 15, true},
|
||||
"XVMULWODHB": {0xE924 << 15, true}, "XVMULWODWH": {0xE925 << 15, true},
|
||||
"XVMULWODVW": {0xE926 << 15, true}, "XVMULWODQV": {0xE927 << 15, true},
|
||||
"XVMULWEVHBU": {0xE930 << 15, true}, "XVMULWEVWHU": {0xE931 << 15, true},
|
||||
"XVMULWEVVWU": {0xE932 << 15, true}, "XVMULWEVQVU": {0xE933 << 15, true},
|
||||
"XVMULWODHBU": {0xE934 << 15, true}, "XVMULWODWHU": {0xE935 << 15, true},
|
||||
"XVMULWODVWU": {0xE936 << 15, true}, "XVMULWODQVU": {0xE937 << 15, true},
|
||||
"XVMULWEVHBUB": {0xE940 << 15, true}, "XVMULWEVWHUH": {0xE941 << 15, true},
|
||||
"XVMULWEVVWUW": {0xE942 << 15, true}, "XVMULWEVQVUV": {0xE943 << 15, true},
|
||||
"XVMULWODHBUB": {0xE944 << 15, true}, "XVMULWODWHUH": {0xE945 << 15, true},
|
||||
"XVMULWODVWUW": {0xE946 << 15, true}, "XVMULWODQVUV": {0xE947 << 15, true},
|
||||
"XVMADDB": {0xE950 << 15, true}, "XVMADDH": {0xE951 << 15, true},
|
||||
"XVMADDW": {0xE952 << 15, true}, "XVMADDV": {0xE953 << 15, true},
|
||||
"XVMSUBB": {0xE954 << 15, true}, "XVMSUBH": {0xE955 << 15, true},
|
||||
"XVMSUBW": {0xE956 << 15, true}, "XVMSUBV": {0xE957 << 15, true},
|
||||
"XVMADDWEVHB": {0xE958 << 15, true}, "XVMADDWEVWH": {0xE959 << 15, true},
|
||||
"XVMADDWEVVW": {0xE95A << 15, true}, "XVMADDWEVQV": {0xE95B << 15, true},
|
||||
"XVMADDWODHB": {0xE95C << 15, true}, "XVMADDWODWH": {0xE95D << 15, true},
|
||||
"XVMADDWODVW": {0xE95E << 15, true}, "XVMADDWODQV": {0xE95F << 15, true},
|
||||
"XVMADDWEVHBU": {0xE968 << 15, true}, "XVMADDWEVWHU": {0xE969 << 15, true},
|
||||
"XVMADDWEVVWU": {0xE96A << 15, true}, "XVMADDWEVQVU": {0xE96B << 15, true},
|
||||
"XVMADDWODHBU": {0xE96C << 15, true}, "XVMADDWODWHU": {0xE96D << 15, true},
|
||||
"XVMADDWODVWU": {0xE96E << 15, true}, "XVMADDWODQVU": {0xE96F << 15, true},
|
||||
"XVMADDWEVHBUB": {0xE978 << 15, true}, "XVMADDWEVWHUH": {0xE979 << 15, true},
|
||||
"XVMADDWEVVWUW": {0xE97A << 15, true}, "XVMADDWEVQVUV": {0xE97B << 15, true},
|
||||
"XVMADDWODHBUB": {0xE97C << 15, true}, "XVMADDWODWHUH": {0xE97D << 15, true},
|
||||
"XVMADDWODVWUW": {0xE97E << 15, true}, "XVMADDWODQVUV": {0xE97F << 15, true},
|
||||
"XVDIVB": {0xE9C0 << 15, true}, "XVDIVH": {0xE9C1 << 15, true},
|
||||
"XVDIVW": {0xE9C2 << 15, true}, "XVDIVV": {0xE9C3 << 15, true},
|
||||
"XVMODB": {0xE9C4 << 15, true}, "XVMODH": {0xE9C5 << 15, true},
|
||||
"XVMODW": {0xE9C6 << 15, true}, "XVMODV": {0xE9C7 << 15, true},
|
||||
"XVDIVBU": {0xE9C8 << 15, true}, "XVDIVHU": {0xE9C9 << 15, true},
|
||||
"XVDIVWU": {0xE9CA << 15, true}, "XVDIVVU": {0xE9CB << 15, true},
|
||||
"XVMODBU": {0xE9CC << 15, true}, "XVMODHU": {0xE9CD << 15, true},
|
||||
"XVMODWU": {0xE9CE << 15, true}, "XVMODVU": {0xE9CF << 15, true},
|
||||
"XVMULF": {0xEA71 << 15, true}, "XVMULD": {0xEA72 << 15, true},
|
||||
"XVDIVF": {0xEA75 << 15, true}, "XVDIVD": {0xEA76 << 15, true},
|
||||
}
|
||||
|
||||
// The lane-wise shifts and rotates (three-register forms; the immediate
|
||||
// forms live in l64VecImmInfo), the interleave families, the bit
|
||||
// clear/set/rev register forms, the remaining logic and compare
|
||||
// spellings, the widening add/subtract ladder and the vector FP
|
||||
// arithmetic.
|
||||
vecmisc := map[string]l64Vec3Enc{
|
||||
"VSLLB": {0xE1D0 << 15, false}, "VSLLH": {0xE1D1 << 15, false},
|
||||
"VSLLW": {0xE1D2 << 15, false}, "VSLLV": {0xE1D3 << 15, false},
|
||||
"VSRLB": {0xE1D4 << 15, false}, "VSRLH": {0xE1D5 << 15, false},
|
||||
"VSRLW": {0xE1D6 << 15, false}, "VSRLV": {0xE1D7 << 15, false},
|
||||
"VSRAH": {0xE1D9 << 15, false}, "VSRAW": {0xE1DA << 15, false},
|
||||
"VSRAV": {0xE1DB << 15, false},
|
||||
"VROTRB": {0xE1DC << 15, false}, "VROTRH": {0xE1DD << 15, false},
|
||||
"VROTRV": {0xE1DF << 15, false},
|
||||
"VILVLB": {0xE234 << 15, false}, "VILVLH": {0xE235 << 15, false},
|
||||
"VILVLW": {0xE236 << 15, false}, "VILVLV": {0xE237 << 15, false},
|
||||
"VILVHB": {0xE238 << 15, false}, "VILVHH": {0xE239 << 15, false},
|
||||
"VILVHW": {0xE23A << 15, false}, "VILVHV": {0xE23B << 15, false},
|
||||
"VBITCLRB": {0xE218 << 15, false}, "VBITCLRH": {0xE219 << 15, false},
|
||||
"VBITCLRW": {0xE21A << 15, false}, "VBITCLRV": {0xE21B << 15, false},
|
||||
"VBITSETB": {0xE21C << 15, false}, "VBITSETH": {0xE21D << 15, false},
|
||||
"VBITSETW": {0xE21E << 15, false}, "VBITSETV": {0xE21F << 15, false},
|
||||
"VBITREVB": {0xE220 << 15, false}, "VBITREVH": {0xE221 << 15, false},
|
||||
"VBITREVW": {0xE222 << 15, false}, "VBITREVV": {0xE223 << 15, false},
|
||||
"VORV": {0xE24D << 15, false}, "VNORV": {0xE24F << 15, false},
|
||||
"VANDNV": {0xE250 << 15, false}, "VORNV": {0xE251 << 15, false},
|
||||
"VSEQH": {0xE001 << 15, false}, "VSEQW": {0xE002 << 15, false},
|
||||
"VSLTB": {0xE00C << 15, false}, "VSLTH": {0xE00D << 15, false},
|
||||
"VSLTW": {0xE00E << 15, false}, "VSLTV": {0xE00F << 15, false},
|
||||
"VSLTBU": {0xE010 << 15, false}, "VSLTHU": {0xE011 << 15, false},
|
||||
"VSLTWU": {0xE012 << 15, false}, "VSLTVU": {0xE013 << 15, false},
|
||||
"VADDWEVHB": {0xE03C << 15, false}, "VADDWEVWH": {0xE03D << 15, false},
|
||||
"VADDWEVVW": {0xE03E << 15, false}, "VADDWEVQV": {0xE03F << 15, false},
|
||||
"VSUBWEVHB": {0xE040 << 15, false}, "VSUBWEVWH": {0xE041 << 15, false},
|
||||
"VSUBWEVVW": {0xE042 << 15, false}, "VSUBWEVQV": {0xE043 << 15, false},
|
||||
"VADDWODHB": {0xE044 << 15, false}, "VADDWODWH": {0xE045 << 15, false},
|
||||
"VADDWODVW": {0xE046 << 15, false}, "VADDWODQV": {0xE047 << 15, false},
|
||||
"VSUBWODHB": {0xE048 << 15, false}, "VSUBWODWH": {0xE049 << 15, false},
|
||||
"VSUBWODVW": {0xE04A << 15, false}, "VSUBWODQV": {0xE04B << 15, false},
|
||||
"VSUBWEVHBU": {0xE060 << 15, false}, "VSUBWEVWHU": {0xE061 << 15, false},
|
||||
"VSUBWEVVWU": {0xE062 << 15, false}, "VSUBWEVQVU": {0xE063 << 15, false},
|
||||
"VADDWEVHBU": {0xE05C << 15, false}, "VADDWEVWHU": {0xE05D << 15, false},
|
||||
"VADDWEVVWU": {0xE05E << 15, false}, "VADDWEVQVU": {0xE05F << 15, false},
|
||||
"VADDWODHBU": {0xE064 << 15, false}, "VADDWODWHU": {0xE065 << 15, false},
|
||||
"VADDWODVWU": {0xE066 << 15, false}, "VADDWODQVU": {0xE067 << 15, false},
|
||||
"VSUBWODHBU": {0xE068 << 15, false}, "VSUBWODWHU": {0xE069 << 15, false},
|
||||
"VSUBWODVWU": {0xE06A << 15, false}, "VSUBWODQVU": {0xE06B << 15, false},
|
||||
"VSHUFH": {0xE2F5 << 15, false}, "VSHUFW": {0xE2F6 << 15, false},
|
||||
"VSHUFV": {0xE2F7 << 15, false},
|
||||
"XVSLLB": {0xE9D0 << 15, true}, "XVSLLH": {0xE9D1 << 15, true},
|
||||
"XVSLLW": {0xE9D2 << 15, true}, "XVSLLV": {0xE9D3 << 15, true},
|
||||
"XVSRLB": {0xE9D4 << 15, true}, "XVSRLH": {0xE9D5 << 15, true},
|
||||
"XVSRLW": {0xE9D6 << 15, true}, "XVSRLV": {0xE9D7 << 15, true},
|
||||
"XVSRAB": {0xE9D8 << 15, true}, "XVSRAH": {0xE9D9 << 15, true},
|
||||
"XVSRAW": {0xE9DA << 15, true}, "XVSRAV": {0xE9DB << 15, true},
|
||||
"XVROTRB": {0xE9DC << 15, true}, "XVROTRH": {0xE9DD << 15, true},
|
||||
"XVROTRW": {0xE9DE << 15, true}, "XVROTRV": {0xE9DF << 15, true},
|
||||
"XVILVLB": {0xEA34 << 15, true}, "XVILVLH": {0xEA35 << 15, true},
|
||||
"XVILVLW": {0xEA36 << 15, true}, "XVILVLV": {0xEA37 << 15, true},
|
||||
"XVILVHB": {0xEA38 << 15, true}, "XVILVHH": {0xEA39 << 15, true},
|
||||
"XVILVHW": {0xEA3A << 15, true}, "XVILVHV": {0xEA3B << 15, true},
|
||||
"XVBITCLRB": {0xEA18 << 15, true}, "XVBITCLRH": {0xEA19 << 15, true},
|
||||
"XVBITCLRW": {0xEA1A << 15, true}, "XVBITCLRV": {0xEA1B << 15, true},
|
||||
"XVBITSETB": {0xEA1C << 15, true}, "XVBITSETH": {0xEA1D << 15, true},
|
||||
"XVBITSETW": {0xEA1E << 15, true}, "XVBITSETV": {0xEA1F << 15, true},
|
||||
"XVBITREVB": {0xEA20 << 15, true}, "XVBITREVH": {0xEA21 << 15, true},
|
||||
"XVBITREVW": {0xEA22 << 15, true}, "XVBITREVV": {0xEA23 << 15, true},
|
||||
"XVORV": {0xEA4D << 15, true}, "XVNORV": {0xEA4F << 15, true},
|
||||
"XVANDNV": {0xEA50 << 15, true}, "XVORNV": {0xEA51 << 15, true},
|
||||
"XVSEQH": {0xE801 << 15, true}, "XVSEQW": {0xE802 << 15, true},
|
||||
"XVSLTB": {0xE80C << 15, true}, "XVSLTH": {0xE80D << 15, true},
|
||||
"XVSLTW": {0xE80E << 15, true}, "XVSLTV": {0xE80F << 15, true},
|
||||
"XVSLTBU": {0xE810 << 15, true}, "XVSLTHU": {0xE811 << 15, true},
|
||||
"XVSLTWU": {0xE812 << 15, true}, "XVSLTVU": {0xE813 << 15, true},
|
||||
"XVADDWEVHB": {0xE83C << 15, true}, "XVADDWEVWH": {0xE83D << 15, true},
|
||||
"XVADDWEVVW": {0xE83E << 15, true}, "XVADDWEVQV": {0xE83F << 15, true},
|
||||
"XVSUBWEVHB": {0xE840 << 15, true}, "XVSUBWEVWH": {0xE841 << 15, true},
|
||||
"XVSUBWEVVW": {0xE842 << 15, true}, "XVSUBWEVQV": {0xE843 << 15, true},
|
||||
"XVADDWODHB": {0xE844 << 15, true}, "XVADDWODWH": {0xE845 << 15, true},
|
||||
"XVADDWODVW": {0xE846 << 15, true}, "XVADDWODQV": {0xE847 << 15, true},
|
||||
"XVSUBWODHB": {0xE848 << 15, true}, "XVSUBWODWH": {0xE849 << 15, true},
|
||||
"XVSUBWODVW": {0xE84A << 15, true}, "XVSUBWODQV": {0xE84B << 15, true},
|
||||
"XVADDWEVHBU": {0xE85C << 15, true}, "XVADDWEVWHU": {0xE85D << 15, true},
|
||||
"XVADDWEVVWU": {0xE85E << 15, true}, "XVADDWEVQVU": {0xE85F << 15, true},
|
||||
"XVSUBWEVHBU": {0xE860 << 15, true}, "XVSUBWEVWHU": {0xE861 << 15, true},
|
||||
"XVSUBWEVVWU": {0xE862 << 15, true}, "XVSUBWEVQVU": {0xE863 << 15, true},
|
||||
"XVADDWODHBU": {0xE864 << 15, true}, "XVADDWODWHU": {0xE865 << 15, true},
|
||||
"XVADDWODVWU": {0xE866 << 15, true}, "XVADDWODQVU": {0xE867 << 15, true},
|
||||
"XVSUBWODHBU": {0xE868 << 15, true}, "XVSUBWODWHU": {0xE869 << 15, true},
|
||||
"XVSUBWODVWU": {0xE86A << 15, true}, "XVSUBWODQVU": {0xE86B << 15, true},
|
||||
"XVSHUFH": {0xEAF5 << 15, true}, "XVSHUFW": {0xEAF6 << 15, true},
|
||||
"XVSHUFV": {0xEAF7 << 15, true},
|
||||
}
|
||||
for _, tab := range []map[string]l64Vec3Enc{addsub, muldiv, vecmisc} {
|
||||
for m, e := range tab {
|
||||
if _, dup := vec3[m]; dup {
|
||||
panic("loong64: duplicate vector mnemonic " + m)
|
||||
}
|
||||
vec3[m] = e
|
||||
}
|
||||
}
|
||||
for m, e := range vec3 {
|
||||
l64InstrTable[m] = l64Enc{format: l64Fvvv, op: e.op}
|
||||
l64VecBank[m] = e.lasx
|
||||
}
|
||||
|
||||
// Immediate forms: INSTR $imm, vj, vd (or INSTR $imm, vd). The immediate
|
||||
// range, bias and field mask are the ones the toolchain encodes: vandi.b
|
||||
// stores the raw 8-bit constant, vsrari.b stores imm+8 (lane-width
|
||||
// bias), the si5 compares store 5-bit two's-complement values and vseqi.d
|
||||
// a 7-bit field the toolchain range-checks down to si5.
|
||||
// The mnemonics that also have a register form (the shifts, the bit
|
||||
// clear/set/rev families, VSEQ and the logic immediates) keep their
|
||||
// three-register entry in l64InstrTable; the dispatcher picks the
|
||||
// immediate opcode from l64VecImmInfo by operand kind, so the immediate
|
||||
// entries must not overwrite the table.
|
||||
vecImm := map[string]l64VecImmEnc{
|
||||
"VANDB": {0xE7A0 << 15, false, 0, 255, 0, 0xFF},
|
||||
"XVANDB": {0xEFA0 << 15, true, 0, 255, 0, 0xFF},
|
||||
"VORB": {0xE7A8 << 15, false, 0, 255, 0, 0xFF},
|
||||
"XVORB": {0xEFA8 << 15, true, 0, 255, 0, 0xFF},
|
||||
"VXORB": {0xE7B0 << 15, false, 0, 255, 0, 0xFF},
|
||||
"XVXORB": {0xEFB0 << 15, true, 0, 255, 0, 0xFF},
|
||||
"VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF},
|
||||
"XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF},
|
||||
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
|
||||
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F},
|
||||
// vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a
|
||||
// 7-bit field, but the toolchain range-checks it down to si5 as well
|
||||
// (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64).
|
||||
"VSEQH": {0xE501 << 15, false, -16, 15, 0, 0x1F},
|
||||
"XVSEQH": {0xED01 << 15, true, -16, 15, 0, 0x1F},
|
||||
"VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F},
|
||||
"XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F},
|
||||
"VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F},
|
||||
"XVSEQV": {0xE903 << 15, true, -16, 15, 0, 0x7F},
|
||||
// vslti compares against a signed (or, in the U spellings, unsigned)
|
||||
// si5/ui5 constant.
|
||||
"VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F},
|
||||
"XVSLTB": {0xED0C << 15, true, -16, 15, 0, 0x1F},
|
||||
"VSLTH": {0xE50D << 15, false, -16, 15, 0, 0x1F},
|
||||
"XVSLTH": {0xED0D << 15, true, -16, 15, 0, 0x1F},
|
||||
"VSLTW": {0xE50E << 15, false, -16, 15, 0, 0x1F},
|
||||
"XVSLTW": {0xED0E << 15, true, -16, 15, 0, 0x1F},
|
||||
"VSLTV": {0xE50F << 15, false, -16, 15, 0, 0x1F},
|
||||
"XVSLTV": {0xED0F << 15, true, -16, 15, 0, 0x1F},
|
||||
"VSLTBU": {0xE510 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVSLTBU": {0xED10 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VSLTHU": {0xE511 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVSLTHU": {0xED11 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VSLTWU": {0xE512 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVSLTWU": {0xED12 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VSLTVU": {0xE513 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVSLTVU": {0xED13 << 15, true, 0, 31, 0, 0x1F},
|
||||
// vaddi/vsubi take ui5 constants for every width on this toolchain
|
||||
// (VADDVU $32 is rejected by the oracle although the field is ui8).
|
||||
"VADDBU": {0xE514 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVADDBU": {0xED14 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VADDHU": {0xE515 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVADDHU": {0xED15 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VADDWU": {0xE516 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVADDWU": {0xED16 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VADDVU": {0xE517 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVADDVU": {0xED17 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VSUBBU": {0xE518 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVSUBBU": {0xED18 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VSUBHU": {0xE519 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVSUBHU": {0xED19 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VSUBWU": {0xE51A << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVSUBWU": {0xED1A << 15, true, 0, 31, 0, 0x1F},
|
||||
"VSUBVU": {0xE51B << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVSUBVU": {0xED1B << 15, true, 0, 31, 0, 0x1F},
|
||||
// The shift/rotate immediates ride in a width-sized field whose upper
|
||||
// bits carry the lane-width code: vslli.b stores ui3 at [12:0] with
|
||||
// bits [14:13] inside the opcode, vslli.h ui4 under a 4 bit mask, and
|
||||
// the .w/.d spellings a raw ui5/ui6.
|
||||
"VSLLB": {0x732C2000, false, 0, 7, 0, 0x7},
|
||||
"XVSLLB": {0x772C2000, true, 0, 7, 0, 0x7},
|
||||
"VSLLH": {0x732C4000, false, 0, 15, 0, 0xF},
|
||||
"XVSLLH": {0x772C4000, true, 0, 15, 0, 0xF},
|
||||
"VSLLW": {0xE659 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVSLLW": {0xEE59 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VSLLV": {0xE65A << 15, false, 0, 63, 0, 0x3F},
|
||||
"XVSLLV": {0xEE5A << 15, true, 0, 63, 0, 0x3F},
|
||||
"VSRLB": {0x73302000, false, 0, 7, 0, 0x7},
|
||||
"XVSRLB": {0x77302000, true, 0, 7, 0, 0x7},
|
||||
"VSRLH": {0x73304000, false, 0, 15, 0, 0xF},
|
||||
"XVSRLH": {0x77304000, true, 0, 15, 0, 0xF},
|
||||
"VSRLW": {0xE661 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVSRLW": {0xEE61 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VSRLV": {0xE662 << 15, false, 0, 63, 0, 0x3F},
|
||||
"XVSRLV": {0xEE62 << 15, true, 0, 63, 0, 0x3F},
|
||||
// vsrari/vrotri bias the field so the lane-width code rides above the
|
||||
// shift amount (.b adds 8, .h 16, .w 32; .d is a raw ui6).
|
||||
"VSRAB": {0xE668 << 15, false, 0, 7, 8, 0x1F},
|
||||
"XVSRAB": {0xEE68 << 15, true, 0, 7, 8, 0x1F},
|
||||
"VSRAH": {0x73344000, false, 0, 15, 0, 0xF},
|
||||
"XVSRAH": {0x77344000, true, 0, 15, 0, 0xF},
|
||||
"VSRAW": {0xE669 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVSRAW": {0xEE69 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VSRAV": {0xE66A << 15, false, 0, 63, 0, 0x3F},
|
||||
"XVSRAV": {0xEE6A << 15, true, 0, 63, 0, 0x3F},
|
||||
"VROTRB": {0x72A02000, false, 0, 7, 0, 0x7},
|
||||
"XVROTRB": {0x76A02000, true, 0, 7, 0, 0x7},
|
||||
"VROTRH": {0x72A04000, false, 0, 15, 0, 0xF},
|
||||
"XVROTRH": {0x76A04000, true, 0, 15, 0, 0xF},
|
||||
"VROTRW": {0xE541 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVROTRW": {0xED41 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VROTRV": {0xE542 << 15, false, 0, 63, 0, 0x3F},
|
||||
"XVROTRV": {0xED42 << 15, true, 0, 63, 0, 0x3F},
|
||||
// vbitclri/vbitseti/vbitrevi follow the same width-coded layout.
|
||||
"VBITCLRB": {0x73102000, false, 0, 7, 0, 0x7},
|
||||
"XVBITCLRB": {0x77102000, true, 0, 7, 0, 0x7},
|
||||
"VBITCLRH": {0x73104000, false, 0, 15, 0, 0xF},
|
||||
"XVBITCLRH": {0x77104000, true, 0, 15, 0, 0xF},
|
||||
"VBITCLRW": {0xE621 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVBITCLRW": {0xEE21 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VBITCLRV": {0xE622 << 15, false, 0, 63, 0, 0x3F},
|
||||
"XVBITCLRV": {0xEE22 << 15, true, 0, 63, 0, 0x3F},
|
||||
"VBITSETB": {0x73142000, false, 0, 7, 0, 0x7},
|
||||
"XVBITSETB": {0x77142000, true, 0, 7, 0, 0x7},
|
||||
"VBITSETH": {0x73144000, false, 0, 15, 0, 0xF},
|
||||
"XVBITSETH": {0x77144000, true, 0, 15, 0, 0xF},
|
||||
"VBITSETW": {0xE629 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVBITSETW": {0xEE29 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VBITSETV": {0xE62A << 15, false, 0, 63, 0, 0x3F},
|
||||
"XVBITSETV": {0xEE2A << 15, true, 0, 63, 0, 0x3F},
|
||||
"VBITREVB": {0x73182000, false, 0, 7, 0, 0x7},
|
||||
"XVBITREVB": {0x77182000, true, 0, 7, 0, 0x7},
|
||||
"VBITREVH": {0x73184000, false, 0, 15, 0, 0xF},
|
||||
"XVBITREVH": {0x77184000, true, 0, 15, 0, 0xF},
|
||||
"VBITREVW": {0xE631 << 15, false, 0, 31, 0, 0x1F},
|
||||
"XVBITREVW": {0xEE31 << 15, true, 0, 31, 0, 0x1F},
|
||||
"VBITREVV": {0xE632 << 15, false, 0, 63, 0, 0x3F},
|
||||
"XVBITREVV": {0xEE32 << 15, true, 0, 63, 0, 0x3F},
|
||||
// The 4-bit-select shuffles and the byte-extract/insert permutations
|
||||
// take ui8 (the .d shuffle ui4 range-checked to 0..15 by the
|
||||
// toolchain) packing both position nibbles.
|
||||
"VSHUF4IB": {0xE720 << 15, false, 0, 255, 0, 0xFF},
|
||||
"XVSHUF4IB": {0xEF20 << 15, true, 0, 255, 0, 0xFF},
|
||||
"VSHUF4IH": {0xE728 << 15, false, 0, 255, 0, 0xFF},
|
||||
"XVSHUF4IH": {0xEF28 << 15, true, 0, 255, 0, 0xFF},
|
||||
"VSHUF4IW": {0xE730 << 15, false, 0, 255, 0, 0xFF},
|
||||
"XVSHUF4IW": {0xEF30 << 15, true, 0, 255, 0, 0xFF},
|
||||
"VSHUF4IV": {0xE738 << 15, false, 0, 15, 0, 0xFF},
|
||||
"XVSHUF4IV": {0xEF38 << 15, true, 0, 15, 0, 0xFF},
|
||||
"VPERMIW": {0xE7C8 << 15, false, 0, 255, 0, 0xFF},
|
||||
"XVPERMIW": {0xEFC8 << 15, true, 0, 255, 0, 0xFF},
|
||||
"XVPERMIV": {0xEFD0 << 15, true, 0, 255, 0, 0xFF},
|
||||
"XVPERMIQ": {0xEFD8 << 15, true, 0, 255, 0, 0xFF},
|
||||
"VEXTRINSB": {0xE718 << 15, false, 0, 255, 0, 0xFF},
|
||||
"XVEXTRINSB": {0xEF18 << 15, true, 0, 255, 0, 0xFF},
|
||||
"VEXTRINSH": {0xE710 << 15, false, 0, 255, 0, 0xFF},
|
||||
"XVEXTRINSH": {0xEF10 << 15, true, 0, 255, 0, 0xFF},
|
||||
"VEXTRINSW": {0xE708 << 15, false, 0, 255, 0, 0xFF},
|
||||
"XVEXTRINSW": {0xEF08 << 15, true, 0, 255, 0, 0xFF},
|
||||
"VEXTRINSV": {0xE700 << 15, false, 0, 255, 0, 0xFF},
|
||||
"XVEXTRINSV": {0xEF00 << 15, true, 0, 255, 0, 0xFF},
|
||||
}
|
||||
for m, e := range vecImm {
|
||||
l64VecImmInfo[m] = e
|
||||
l64VecBank[m] = e.lasx
|
||||
}
|
||||
|
||||
// Vector-to-condition flag: INSTR vj, FCCn (vsetnez.v, vsetanyeqz.*,
|
||||
// vsetallnez.*): the sub-op rides in the rk field.
|
||||
vecCf := map[string]uint32{
|
||||
"VSETNEV": 0xE539<<15 | 7<<10, "XVSETNEV": 0xED39<<15 | 7<<10,
|
||||
"VSETANYEQB": 0xE539<<15 | 8<<10, "XVSETANYEQB": 0xED39<<15 | 8<<10,
|
||||
"VSETANYEQV": 0xE539<<15 | 11<<10, "XVSETANYEQV": 0xED39<<15 | 11<<10,
|
||||
"VSETALLNEV": 0xE539<<15 | 15<<10, "XVSETALLNEV": 0xED39<<15 | 15<<10,
|
||||
"VSETEQV": 0xE539<<15 | 6<<10, "XVSETEQV": 0xED39<<15 | 6<<10,
|
||||
"VSETANYEQH": 0xE539<<15 | 9<<10, "XVSETANYEQH": 0xED39<<15 | 9<<10,
|
||||
"VSETANYEQW": 0xE539<<15 | 10<<10, "XVSETANYEQW": 0xED39<<15 | 10<<10,
|
||||
"VSETALLNEB": 0xE539<<15 | 12<<10, "XVSETALLNEB": 0xED39<<15 | 12<<10,
|
||||
"VSETALLNEH": 0xE539<<15 | 13<<10, "XVSETALLNEH": 0xED39<<15 | 13<<10,
|
||||
"VSETALLNEW": 0xE539<<15 | 14<<10, "XVSETALLNEW": 0xED39<<15 | 14<<10,
|
||||
}
|
||||
for m, op := range vecCf {
|
||||
l64InstrTable[m] = l64Enc{format: l64Fvcf, op: op}
|
||||
l64VecBank[m] = strings.HasPrefix(m, "XV")
|
||||
}
|
||||
|
||||
// Lane popcount and the two-operand vector FP/unary spellings: INSTR vj,
|
||||
// vd (the 2R layout with the opcode extending over the unused vk field;
|
||||
// the low byte of each constant is the instruction's own sub-op).
|
||||
vec2r := map[string]l64Vec3Enc{
|
||||
"VPCNTV": {0x1CA70B << 10, false}, "XVPCNTV": {0x1DA70B << 10, true},
|
||||
}
|
||||
// The rest of the lane popcounts, the vector negations and the vector FP
|
||||
// unary conversions (loong64enc1.s).
|
||||
vec2rMore := map[string]l64Vec3Enc{
|
||||
"VPCNTB": {0x1CA708 << 10, false}, "VPCNTH": {0x1CA709 << 10, false},
|
||||
"VPCNTW": {0x1CA70A << 10, false},
|
||||
"VNEGB": {0x1CA70C << 10, false}, "VNEGH": {0x1CA70D << 10, false},
|
||||
"VNEGW": {0x1CA70E << 10, false}, "VNEGV": {0x1CA70F << 10, false},
|
||||
"VFCLASSF": {0x1CA735 << 10, false}, "VFCLASSD": {0x1CA736 << 10, false},
|
||||
"VFSQRTF": {0x1CA739 << 10, false}, "VFSQRTD": {0x1CA73A << 10, false},
|
||||
"VFRECIPF": {0x1CA73D << 10, false}, "VFRECIPD": {0x1CA73E << 10, false},
|
||||
"VFRSQRTF": {0x1CA741 << 10, false}, "VFRSQRTD": {0x1CA742 << 10, false},
|
||||
"VFRINTF": {0x1CA74D << 10, false}, "VFRINTD": {0x1CA74E << 10, false},
|
||||
"VFRINTRMF": {0x1CA751 << 10, false}, "VFRINTRMD": {0x1CA752 << 10, false},
|
||||
"VFRINTRPF": {0x1CA755 << 10, false}, "VFRINTRPD": {0x1CA756 << 10, false},
|
||||
"VFRINTRZF": {0x1CA759 << 10, false}, "VFRINTRZD": {0x1CA75A << 10, false},
|
||||
"VFRINTRNEF": {0x1CA75D << 10, false}, "VFRINTRNED": {0x1CA75E << 10, false},
|
||||
"XVPCNTB": {0x1DA708 << 10, true}, "XVPCNTH": {0x1DA709 << 10, true},
|
||||
"XVPCNTW": {0x1DA70A << 10, true},
|
||||
"XVNEGB": {0x1DA70C << 10, true}, "XVNEGH": {0x1DA70D << 10, true},
|
||||
"XVNEGW": {0x1DA70E << 10, true}, "XVNEGV": {0x1DA70F << 10, true},
|
||||
"XVFCLASSF": {0x1DA735 << 10, true}, "XVFCLASSD": {0x1DA736 << 10, true},
|
||||
"XVFSQRTF": {0x1DA739 << 10, true}, "XVFSQRTD": {0x1DA73A << 10, true},
|
||||
"XVFRECIPF": {0x1DA73D << 10, true}, "XVFRECIPD": {0x1DA73E << 10, true},
|
||||
"XVFRSQRTF": {0x1DA741 << 10, true}, "XVFRSQRTD": {0x1DA742 << 10, true},
|
||||
"XVFRINTF": {0x1DA74D << 10, true}, "XVFRINTD": {0x1DA74E << 10, true},
|
||||
"XVFRINTRMF": {0x1DA751 << 10, true}, "XVFRINTRMD": {0x1DA752 << 10, true},
|
||||
"XVFRINTRPF": {0x1DA755 << 10, true}, "XVFRINTRPD": {0x1DA756 << 10, true},
|
||||
"XVFRINTRZF": {0x1DA759 << 10, true}, "XVFRINTRZD": {0x1DA75A << 10, true},
|
||||
"XVFRINTRNEF": {0x1DA75D << 10, true}, "XVFRINTRNED": {0x1DA75E << 10, true},
|
||||
}
|
||||
maps.Copy(vec2r, vec2rMore)
|
||||
for m, e := range vec2r {
|
||||
l64InstrTable[m] = l64Enc{format: l64Frr, op: e.op}
|
||||
l64VecBank[m] = e.lasx
|
||||
l64Vec2R[m] = true
|
||||
}
|
||||
|
||||
// The four-register byte shuffle: INSTR va, vk, vj, vd (the operand the
|
||||
// table reads in each field position, va at bits [19:15]).
|
||||
vec4r := map[string]l64Vec3Enc{
|
||||
"VSHUFB": {0x0D50 << 16, false}, "XVSHUFB": {0x0D60 << 16, true},
|
||||
}
|
||||
for m, e := range vec4r {
|
||||
l64InstrTable[m] = l64Enc{format: l64Fvvvv, op: e.op}
|
||||
l64VecBank[m] = e.lasx
|
||||
l64Vec4R[m] = true
|
||||
}
|
||||
}
|
||||
|
||||
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
|
||||
// register move between the integer and floating-point register banks — the
|
||||
// register move between the integer and floating-point register banks, the
|
||||
// MOVW/MOVV specials the Go assembler accepts.
|
||||
var l64FpMovTable = map[string]uint32{
|
||||
"MOVV.R.F": 0x452a << 10, // movgr2fr.d
|
||||
|
||||
+542
-2
@@ -8,8 +8,8 @@ import (
|
||||
"encoding/binary"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// firstTextLOONG64 parses assembly source and returns the first TEXT body.
|
||||
@@ -263,6 +263,20 @@ func TestLOONG64_regNames(t *testing.T) {
|
||||
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
|
||||
}
|
||||
}
|
||||
// The X/V spellings name the LSX/LASX vector banks, a register class of
|
||||
// their own: the oracle (GOARCH=loong64 go tool asm) rejects `BEQZ X0`
|
||||
// with "unrecognized instruction" while assembling `VADDV V0, V1, V2`
|
||||
// and `XVADDV X0, X1, X2`, so loong64RegNum stays strict and the vector
|
||||
// operands resolve through loong64VecRegNum only.
|
||||
vecCases := map[string]int{
|
||||
"V0": 0, "V31": 31, "X0": 0, "X31": 31,
|
||||
"R4": -1, "F0": -1, "FCC0": -1, "V32": -1, "X32": -1, "V": -1, "X": -1,
|
||||
}
|
||||
for name, want := range vecCases {
|
||||
if got := loong64VecRegNum(name); got != want {
|
||||
t.Errorf("loong64VecRegNum(%q) = %d, want %d", name, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
|
||||
@@ -291,3 +305,529 @@ done:
|
||||
t.Errorf("code = % x\nwant % x", code, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestLOONG64IndirectBranch pins the indirect branch encodings: JMP (Rj) and
|
||||
// JAL (Rj) lower to jirl, and the raw JIRL spelling encodes the written
|
||||
// offset (the Go loong64 assembler deletes raw JIRL instructions entirely,
|
||||
// so this form is a gasm-only superset with faithful semantics).
|
||||
func TestLOONG64IndirectBranch(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
JMP (R4)
|
||||
JIRL R0, R4, 8
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x4C000080, // jirl r0, r4, 0
|
||||
0x4C002080, // jirl r0, r4, 8
|
||||
0x4C000020, // jirl r0, r1, 0 (RET)
|
||||
)
|
||||
|
||||
// JAL (R5) links, so the toolchain gives the function its autosize-8
|
||||
// prologue and epilogue around the call and the closing RET.
|
||||
fn = firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
JAL (R5)
|
||||
RET
|
||||
`)
|
||||
code = assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x29FFE061, // st.d r1, -8(r3) (prologue saves RA below the new SP)
|
||||
0x02FFE063, // addi.d r3, r3, -8 (prologue opens the frame)
|
||||
0x29C00061, // st.d r1, 0(r3) (prologue saves RA at SP)
|
||||
0x4C0000A1, // jirl r1, r5, 0
|
||||
0x28C00061, // ld.d r1, 0(r3) (epilogue restores RA)
|
||||
0x02C02063, // addi.d r3, r3, 8
|
||||
0x4C000020, // jirl r0, r1, 0 (RET)
|
||||
)
|
||||
}
|
||||
|
||||
// TestLOONG64_vector pins the LSX/LASX slice against words read off
|
||||
// GOARCH=loong64 go tool asm (cross-checked against the toolchain's own
|
||||
// loong64enc1.s): the three-register forms, the immediate forms with their
|
||||
// biases, the vector-to-condition forms, lane popcount, the FP conversion,
|
||||
// FSEL and the VMOVQ move family.
|
||||
func TestLOONG64_vector(t *testing.T) {
|
||||
t.Run("three-register and immediate forms", func(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·v(SB), NOSPLIT, $0
|
||||
VADDV V1, V2, V3
|
||||
VADDW V1, V2, V3
|
||||
VADDV V2, V1
|
||||
VANDV V1, V2
|
||||
VXORV V1, V2, V3
|
||||
VSEQB V1, V2, V3
|
||||
VSEQV V1, V2, V3
|
||||
VSRAB V1, V2, V3
|
||||
VROTRW V1, V2, V3
|
||||
VANDB $0, V2, V3
|
||||
VANDB $255, V2
|
||||
VSEQB $3, V2, V3
|
||||
VSEQV $15, V2, V3
|
||||
VSEQV $-15, V2, V3
|
||||
VSRAB $7, V1, V2
|
||||
VROTRW $16, V1, V2
|
||||
VPCNTV V1, V2
|
||||
XVADDV X1, X2, X3
|
||||
XVXORV X1, X2, X3
|
||||
XVSEQB X1, X2, X3
|
||||
XVPCNTV X1, X2
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x700B8443, // vadd.v v3, v2, v1
|
||||
0x700B0443, // vadd.w
|
||||
0x700B8821, // vadd.v v1, v1, v2 (two-operand form)
|
||||
0x71260442, // vand.v v2, v2, v1
|
||||
0x71270443, // vxor.v
|
||||
0x70000443, // vseq.b
|
||||
0x70018443, // vseq.d
|
||||
0x70EC0443, // vsra.b
|
||||
0x70EF0443, // vrotr.w
|
||||
0x73D00043, // vandi.b v3, v2, 0
|
||||
0x73D3FC42, // vandi.b v2, v2, 255 (two-operand form)
|
||||
0x72800C43, // vseqi.b v3, v2, 3
|
||||
0x7281BC43, // vseqi.d v3, v2, 15
|
||||
0x7281C443, // vseqi.d v3, v2, -15 (7-bit two's complement)
|
||||
0x73343C22, // vsrai.b v2, v1, 7 (encoded as 7+8)
|
||||
0x72A0C022, // vrotri.w v2, v1, 16
|
||||
0x729C2C22, // vpcnt.d v2, v1
|
||||
0x740B8443, // xvadd.d x3, x2, x1
|
||||
0x75270443, // xvxor.d
|
||||
0x74000443, // xvseq.b
|
||||
0x769C2C22, // xvpcnt.d x2, x1
|
||||
0x4C000020,
|
||||
)
|
||||
})
|
||||
|
||||
t.Run("vector-to-condition", func(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·v(SB), NOSPLIT, $0
|
||||
VSETNEV V1, FCC0
|
||||
VSETANYEQB V1, FCC0
|
||||
VSETANYEQV V2, FCC0
|
||||
VSETALLNEV V0, FCC0
|
||||
XVSETNEV X1, FCC0
|
||||
XVSETALLNEV X1, FCC0
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x729C9C20, // vsetnez.d fcc0, v1
|
||||
0x729CA020, // vsetanyeqz.b
|
||||
0x729CAC40, // vsetanyeqz.d
|
||||
0x729CBC00, // vsetallnez.d
|
||||
0x769C9C20, // xvsetnez.d
|
||||
0x769CBC20, // xvsetallnez.d
|
||||
0x4C000020,
|
||||
)
|
||||
})
|
||||
|
||||
t.Run("FP convert and FSEL", func(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·v(SB), NOSPLIT, $0
|
||||
FFINTDV F0, F1
|
||||
FSEL FCC0, F3, F4, F3
|
||||
FSEL FCC1, F1, F2
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x011D2801, // ffint.d.v f1, f0
|
||||
0x0D000C83, // fsel f3, f4, f3, fcc0
|
||||
0x0D008442, // fsel f2, f2, f1, fcc1
|
||||
0x4C000020,
|
||||
)
|
||||
})
|
||||
|
||||
t.Run("VMOVQ move family", func(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·v(SB), NOSPLIT, $0
|
||||
VMOVQ V1, V9
|
||||
VMOVQ (R4), V2
|
||||
VMOVQ 16(R4), V2
|
||||
VMOVQ V0, (R4)
|
||||
VMOVQ V0, 32(R4)
|
||||
VMOVQ (R4)(R7), V3
|
||||
VMOVQ V3, (R4)(R7)
|
||||
VMOVQ R6, V0.B16
|
||||
VMOVQ R6, V12.W4
|
||||
VMOVQ (R4), V4.W4
|
||||
XVMOVQ X3, X7
|
||||
XVMOVQ (R4), X2
|
||||
XVMOVQ X0, (R4)
|
||||
XVMOVQ (R4)(R7), X4
|
||||
XVMOVQ X0, (R4)(R7)
|
||||
XVMOVQ R6, X0.B32
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x732D0029, // vori.b v9, v1, 0 (register move)
|
||||
0x2C000082, // vld v2, r4, 0
|
||||
0x2C004082, // vld v2, r4, 16
|
||||
0x2C400080, // vst v0, r4, 0
|
||||
0x2C408080, // vst v0, r4, 32
|
||||
0x38401C83, // vldx v3, r4, r7
|
||||
0x38441C83, // vstx v3, r4, r7
|
||||
0x729F00C0, // vreplgr2vr.b v0, r6
|
||||
0x729F08CC, // vreplgr2vr.w v12, r6
|
||||
0x30200084, // vldrepl.w v4, r4, 0
|
||||
0x772D0067, // xvori.b x7, x3, 0
|
||||
0x2C800082, // xvld x2, r4, 0
|
||||
0x2CC00080, // xvst x0, r4, 0
|
||||
0x38481C84, // xvldx x4, r4, r7
|
||||
0x384C1C80, // xvstx x0, r4, r7
|
||||
0x769F00C0, // xvreplgr2vr.b x0, r6
|
||||
0x4C000020,
|
||||
)
|
||||
})
|
||||
|
||||
t.Run("element extract and insert", func(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·v(SB), NOSPLIT, $0
|
||||
VMOVQ V0.V[0], R10
|
||||
VMOVQ V6.V[1], R8
|
||||
VMOVQ R9, V1.V[0]
|
||||
XVMOVQ X0.V[0], R10
|
||||
XVMOVQ X5.W[7], R7
|
||||
XVMOVQ R4, X7.V[3]
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x72EFF00A, // vpickve2gr.d r10, v0, 0
|
||||
0x72EFF4C8, // vpickve2gr.d r8, v6, 1
|
||||
0x72EBF121, // vinsgr2vr.d v1, r9, 0
|
||||
0x76EFE00A, // xvpickve2gr.d r10, x0, 0
|
||||
0x76EFDCA7, // xvpickve2gr.w r7, x5, 7
|
||||
0x76EBEC87, // xvinsgr2vr.d x7, r4, 3
|
||||
0x4C000020,
|
||||
)
|
||||
})
|
||||
|
||||
// The integer and FP add/subtract families with their saturating pairs
|
||||
// and immediate spellings (loong64enc1.s words).
|
||||
t.Run("add and subtract families", func(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·v(SB), NOSPLIT, $0
|
||||
VADDB V1, V2, V3
|
||||
VADDF V1, V2, V3
|
||||
VADDD V1, V2, V3
|
||||
VSUBD V1, V2, V3
|
||||
VSADDV V1, V2, V3
|
||||
VSSUBVU V1, V2, V3
|
||||
VADDBU $1, V2, V1
|
||||
VADDBU $1, V2
|
||||
VSUBVU $31, V2
|
||||
XVSADDV X3, X2, X1
|
||||
XVSUBD X1, X2, X3
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x700A0443, // vadd.b
|
||||
0x71308443, // vadd.f
|
||||
0x71310443, // vadd.d
|
||||
0x71330443, // vsub.d
|
||||
0x70478443, // vsadd.v
|
||||
0x704D8443, // vssub.u.d
|
||||
0x728A0441, // vaddi.bu v1, v2, 1
|
||||
0x728A0442, // vaddi.bu v2, v2, 1 (two-operand form)
|
||||
0x728DFC42, // vsubi.du v2, v2, 31 (two-operand form)
|
||||
0x74478C41, // xvsadd.d x1, x2, x3
|
||||
0x75330443, // xvsub.d x3, x2, x1
|
||||
0x4C000020,
|
||||
)
|
||||
})
|
||||
|
||||
// The multiply, divide and accumulate families.
|
||||
t.Run("multiply and divide families", func(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·v(SB), NOSPLIT, $0
|
||||
VMULV V1, V2, V3
|
||||
VMUHHU V1, V2, V3
|
||||
VDIVBU V1, V2, V3
|
||||
VMODV V1, V2, V3
|
||||
VMADDB V1, V2, V3
|
||||
VMSUBV V1, V2, V3
|
||||
VMULWEVHB V1, V2, V3
|
||||
VMULWODQV V1, V2, V3
|
||||
VMADDWEVHBUB V1, V2, V3
|
||||
XVDIVD X1, X2, X3
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x70858443, // vmul.v
|
||||
0x70888443, // vmuh.u.d
|
||||
0x70E40443, // vdiv.u.b
|
||||
0x70E38443, // vmod.d
|
||||
0x70A80443, // vmadd.b
|
||||
0x70AB8443, // vmsub.d
|
||||
0x70900443, // vmulwev.h.b
|
||||
0x70938443, // vmulwod.q.d
|
||||
0x70BC0443, // vmaddwev.h.bu.b
|
||||
0x753B0443, // xvdiv.d
|
||||
0x4C000020,
|
||||
)
|
||||
})
|
||||
|
||||
// The shift, bit and interleave families in register and immediate
|
||||
// spellings, with the width-coded shift immediates.
|
||||
t.Run("shift, bit and interleave families", func(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·v(SB), NOSPLIT, $0
|
||||
VSLLV V1, V2, V3
|
||||
VROTRB V1, V2, V3
|
||||
VBITCLRV V1, V2, V3
|
||||
VBITSETW V1, V2, V3
|
||||
VBITREVV V1, V2, V3
|
||||
VILVLB V1, V2, V3
|
||||
VILVHV V1, V2, V3
|
||||
VSLLB $7, V1, V2
|
||||
VSLLB $5, V1
|
||||
VSRLH $15, V1, V2
|
||||
VSRAW $31, V1, V2
|
||||
VSRAV $63, V1, V2
|
||||
VROTRV $63, V1, V2
|
||||
VBITCLRB $7, V2, V3
|
||||
VBITREVV $63, V2, V3
|
||||
VSEQH $-16, V2, V3
|
||||
VSLTB $1, V2, V3
|
||||
VSLTHU $31, V2, V3
|
||||
XVILVLV X3, X2, X1
|
||||
XVSLLB $7, X2, X1
|
||||
XVSRAV $63, X2, X1
|
||||
XVBITREVV $63, X2, X1
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x70E98443, // vsll.d
|
||||
0x70EE0443, // vrotr.b
|
||||
0x710D8443, // vbitclr.d
|
||||
0x710F0443, // vbitset.w
|
||||
0x71118443, // vbitrev.d
|
||||
0x711A0443, // vilvl.b
|
||||
0x711D8443, // vilvh.d
|
||||
0x732C3C22, // vslli.b v2, v1, 7
|
||||
0x732C3421, // vslli.b v1, v1, 5 (two-operand form)
|
||||
0x73307C22, // vsrli.h v2, v1, 15
|
||||
0x7334FC22, // vsrai.w v2, v1, 31
|
||||
0x7335FC22, // vsrai.d v2, v1, 63
|
||||
0x72A1FC22, // vrotri.d v2, v1, 63
|
||||
0x73103C43, // vbitclri.b v3, v2, 7
|
||||
0x7319FC43, // vbitrevi.d v3, v2, 63
|
||||
0x7280C043, // vseqi.h v3, v2, -16
|
||||
0x72860443, // vslti.b v3, v2, 1
|
||||
0x7288FC43, // vslti.hu v3, v2, 31
|
||||
0x751B8C41, // xvilvl.d x1, x2, x3
|
||||
0x772C3C41, // xvslli.b x1, x2, 7
|
||||
0x7735FC41, // xvsrai.d x1, x2, 63
|
||||
0x7719FC41, // xvbitrevi.d x1, x2, 63
|
||||
0x4C000020,
|
||||
)
|
||||
})
|
||||
|
||||
// The shuffle, select and permutation families, including the
|
||||
// four-register byte shuffle.
|
||||
t.Run("shuffle and permutation families", func(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·v(SB), NOSPLIT, $0
|
||||
VSHUFH V1, V2, V3
|
||||
VSHUFW V1, V2, V3
|
||||
VSHUFV V1, V2, V3
|
||||
VSHUFB V1, V2, V3, V4
|
||||
XVSHUFB X1, X2, X3, X4
|
||||
VSHUF4IB $255, V2, V1
|
||||
VSHUF4IV $15, V2, V1
|
||||
XVSHUF4IV $15, X1, X2
|
||||
VEXTRINSB $0x18, V1, V2
|
||||
XVEXTRINSV $0x81, X1, X2
|
||||
VPERMIW $0x1B, V1, V2
|
||||
XVPERMIQ $0x4B, X1, X2
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x717A8443, // vshuf.h
|
||||
0x717B0443, // vshuf.w
|
||||
0x717B8443, // vshuf.d
|
||||
0x0D508864, // vshuf.b v4, v3, v2, v1
|
||||
0x0D608864, // xvshuf.b
|
||||
0x7393FC41, // vshuf4i.b v1, v2, 255
|
||||
0x739C3C41, // vshuf4i.d v1, v2, 15
|
||||
0x779C3C22, // xvshuf4i.d x2, x1, 15
|
||||
0x738C6022, // vextrins.b v2, v1, 0x18
|
||||
0x77820422, // xvextrins.d x2, x1, 0x81
|
||||
0x73E46C22, // vpermi.w v2, v1, 0x1b
|
||||
0x77ED2C22, // xvpermi.q x2, x1, 0x4b
|
||||
0x4C000020,
|
||||
)
|
||||
})
|
||||
|
||||
// The vector FP families, the unary spellings, the compare-to-flag
|
||||
// additions and the scalar int/float conversions.
|
||||
t.Run("FP and conversion families", func(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·v(SB), NOSPLIT, $0
|
||||
VADDF V1, V2, V3
|
||||
VMULF V1, V2, V3
|
||||
VFCLASSD V1, V2
|
||||
VFSQRTF V1, V2
|
||||
VFRECIPD V1, V2
|
||||
VFRSQRTF V1, V2
|
||||
VFRINTF V1, V2
|
||||
VFRINTRNED V1, V2
|
||||
VNEGB V1, V2
|
||||
VPCNTB V1, V2
|
||||
XVNEGV X2, X1
|
||||
XVPCNTW X3, X2
|
||||
XVFRINTRNEF X1, X2
|
||||
VSETEQV V1, FCC0
|
||||
VSETANYEQH V1, FCC0
|
||||
VSETALLNEB V1, FCC0
|
||||
XVSETALLNEW X1, FCC0
|
||||
FFINTFW F0, F1
|
||||
FTINTVD F0, F1
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x71308443, // vfadd.s
|
||||
0x71388443, // vfmul.s
|
||||
0x729CD822, // vfclass.d
|
||||
0x729CE422, // vfsqrt.s
|
||||
0x729CF822, // vfrecip.d
|
||||
0x729D0422, // vfrsqrt.s
|
||||
0x729D3422, // vfrint.s
|
||||
0x729D7822, // vfrintne.s
|
||||
0x729C3022, // vneg.b
|
||||
0x729C2022, // vpcnt.b
|
||||
0x769C3C41, // xvneg.d x1, x2
|
||||
0x769C2862, // xvpcnt.w x2, x3
|
||||
0x769D7422, // xvfrintne.s x2, x1
|
||||
0x729C9820, // vseteqz.d fcc0, v1
|
||||
0x729CA420, // vsetanyeqz.h
|
||||
0x729CB020, // vsetallnez.b
|
||||
0x769CB820, // xvsetallnez.w
|
||||
0x011D1001, // ffint.s.w f1, f0
|
||||
0x011B2801, // ftint.l.d f1, f0
|
||||
0x4C000020,
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
// TestLOONG64_vectorErrors pins the register-class and range diagnostics of
|
||||
// the vector slice; each shape is rejected by the oracle as well
|
||||
// (GOARCH=loong64 go tool asm).
|
||||
func TestLOONG64_vectorErrors(t *testing.T) {
|
||||
cases := []string{
|
||||
// Integer registers in vector positions.
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
VADDV R4, R5, R6
|
||||
RET
|
||||
`,
|
||||
// Crossed banks: LSX spellings take V, LASX spellings X.
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
VADDV X1, X2, X3
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
XVADDV V1, V2, V3
|
||||
RET
|
||||
`,
|
||||
// The LASX bank has no .b/.h element forms.
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
XVMOVQ R4, X2.B[0]
|
||||
RET
|
||||
`,
|
||||
// Immediate ranges.
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
VANDB $256, V2
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
VSEQB $16, V2, V3
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
VROTRW $32, V1, V2
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
VADDVU $32, V2
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
VSEQV $32, V2, V3
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
VSHUF4IV $16, V2, V1
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
VEXTRINSB $256, V1, V2
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
VSLTV $-17, V2, V3
|
||||
RET
|
||||
`,
|
||||
// VSHUFB wants four vector registers.
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
VSHUFB V1, V2, V3
|
||||
RET
|
||||
`,
|
||||
// The FCC forms still refuse vector registers.
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
VSETEQV V1, V2
|
||||
RET
|
||||
`,
|
||||
// VSET* wants an FCC flag, not a vector register.
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
VSETNEV V1, V2
|
||||
RET
|
||||
`,
|
||||
}
|
||||
for i, src := range cases {
|
||||
fn := firstTextLOONG64(t, src)
|
||||
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||
t.Errorf("case %d: expected an error, got none", i)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestLOONG64_dbarAtomics pins the _dbar (acquire/release) AMO variants.
|
||||
// The oracle words come from GOARCH=loong64 go tool objdump of kernels
|
||||
// assembled with go tool asm, and match the toolchain's loong64enc1.s.
|
||||
func TestLOONG64_dbarAtomics(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·atoms(SB), NOSPLIT, $0
|
||||
AMADDDBW R14, (R13), R12
|
||||
AMADDDBV R14, (R13), R12
|
||||
AMANDDBW R5, (R4), R6
|
||||
AMANDDBV R5, (R4), R6
|
||||
AMORDBW R5, (R4), R0
|
||||
AMORDBV R5, (R4), R6
|
||||
AMSWAPDBW R5, (R4), R6
|
||||
AMCASDBV R6, (R4), R5
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x386A39AC, // amadd_db.w r12, r13, r14
|
||||
0x386AB9AC, // amadd_db.d
|
||||
0x386B1486, // amand_db.w r6, r4, r5
|
||||
0x386B9486, // amand_db.d
|
||||
0x386C1480, // amor_db.w r0, r4, r5
|
||||
0x386C9486, // amor_db.d
|
||||
0x38691486, // amswap_db.w
|
||||
0x385B9885, // amcas_db.w
|
||||
0x4C000020,
|
||||
)
|
||||
}
|
||||
|
||||
+224
-13
@@ -6,7 +6,7 @@ package asm
|
||||
import (
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||
)
|
||||
|
||||
// Loong64 frame mapping, matching the Go toolchain's loong64 backend.
|
||||
@@ -20,14 +20,20 @@ import (
|
||||
// (the toolchain aligns frames with `if autosize&4 != 0 { autosize += 4 }`).
|
||||
// A leaf function (no calls) with a zero frame gets no prologue at all.
|
||||
//
|
||||
// Prologue (autosize > 0), byte-identical to the toolchain:
|
||||
// Prologue (autosize > 0, small), byte-identical to the toolchain:
|
||||
//
|
||||
// MOVV R1, -autosize(R3) // save LR below the new SP (traceback-safe)
|
||||
// ADDV $-autosize, R3 // open the frame
|
||||
// MOVV R1, 0(R3) // save LR again at SP (signal-safety)
|
||||
//
|
||||
// Large frames (autosize past the 12-bit offset or immediate ranges) expand
|
||||
// the store and the adjust through REGTMP (R30) exactly as the toolchain's
|
||||
// assembler does: the store via the rounding LU12IW split, the adjust via
|
||||
// the floor LU12IW/ORI split.
|
||||
//
|
||||
// Epilogue: MOVV 0(R3), R1; ADDV $autosize, R3 (non-leaf only for the LR
|
||||
// restore); the RET's jirl r0, r1, 0 follows.
|
||||
// restore; the adjust materialised when the immediate does not fit); the
|
||||
// RET's jirl r0, r1, 0 follows.
|
||||
|
||||
// loong64FrameInfo holds the frame layout derived from a TEXT directive.
|
||||
type loong64FrameInfo struct {
|
||||
@@ -36,6 +42,11 @@ type loong64FrameInfo struct {
|
||||
args int // the declared -argsize
|
||||
noSplit bool // the NOSPLIT flag
|
||||
leaf bool // no call instructions in the body
|
||||
|
||||
// Stack-split guard state: like amd64 and arm64, a leaf function with a
|
||||
// small autosize is auto-marked NOSPLIT by the toolchain.
|
||||
needSplit bool
|
||||
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
|
||||
}
|
||||
|
||||
// loong64ComputeFrame derives the frame layout for a TEXT function.
|
||||
@@ -59,9 +70,157 @@ func loong64ComputeFrame(t *ast.Text) loong64FrameInfo {
|
||||
// A zero-frame non-leaf function still opens an 8-byte frame for LR.
|
||||
fi.autosize = 8
|
||||
}
|
||||
switch {
|
||||
case fi.noSplit:
|
||||
case fi.autosize < stackSmall && fi.leaf:
|
||||
// Auto-NOSPLIT, as the toolchain's leaf mark concludes.
|
||||
default:
|
||||
fi.needSplit = true
|
||||
switch {
|
||||
case fi.autosize <= stackSmall:
|
||||
fi.splitClass = 0
|
||||
case fi.autosize <= stackBig:
|
||||
fi.splitClass = 1
|
||||
default:
|
||||
fi.splitClass = 2
|
||||
}
|
||||
}
|
||||
return fi
|
||||
}
|
||||
|
||||
// loong64GuardLen returns the byte length of the stack-split guard prefix
|
||||
// (zero when the function needs no guard). The big class materialises two
|
||||
// constants through R30; each materialisation shrinks by one word when the
|
||||
// constant's low 12 bits are zero.
|
||||
func loong64GuardLen(fi loong64FrameInfo) int {
|
||||
if !fi.needSplit {
|
||||
return 0
|
||||
}
|
||||
off := int64(fi.autosize - stackSmall)
|
||||
switch fi.splitClass {
|
||||
case 0:
|
||||
return 12
|
||||
case 1:
|
||||
if off <= 2048 {
|
||||
return 16 // ADDV $-off fits the signed 12-bit immediate
|
||||
}
|
||||
return 24 // MOVV + LU12IW + ORI + ADDV + SGTU + BEQ
|
||||
default:
|
||||
// MOVV + [mat] + SGTU + BNE + [mat] + ADDV + SGTU + BEQ
|
||||
return (6 + loong64MatLen(off) + loong64MatLen(-off)) * 4
|
||||
}
|
||||
}
|
||||
|
||||
// loong64MatLen reports the word count of materialising v in R30: a value
|
||||
// with a zero high part needs only the ORI (the toolchain's MOVW $v, R30),
|
||||
// one with a zero low part only the LU12IW.
|
||||
func loong64MatLen(v int64) int {
|
||||
if v>>12 == 0 || v&0xFFF == 0 {
|
||||
return 1
|
||||
}
|
||||
return 2
|
||||
}
|
||||
|
||||
// loong64MatWords appends the words that materialise v in R30, splitting it
|
||||
// as v>>12 plus the zero-extended low 12 bits.
|
||||
func loong64MatWords(ws []uint32, v int64) []uint32 {
|
||||
hi := v >> 12
|
||||
lo := v & 0xFFF
|
||||
if hi == 0 {
|
||||
return append(ws, l64irr(l64OriOp, int(v), 0, 30))
|
||||
}
|
||||
ws = append(ws, l64ir(l64Lu12iwOp, int(hi), 30))
|
||||
if lo != 0 {
|
||||
ws = append(ws, l64irr(l64OriOp, int(lo), 30, 30))
|
||||
}
|
||||
return ws
|
||||
}
|
||||
|
||||
// The LU12IW and ORI opcode bases (2RI20 and 2RI12 formats); the ORI reads
|
||||
// and writes rd itself.
|
||||
const (
|
||||
l64Lu12iwOp = 0x0a << 25
|
||||
l64OriOp = 0x0e << 22
|
||||
)
|
||||
|
||||
// loong64Imm12 reports whether v fits a signed 12-bit immediate.
|
||||
func loong64Imm12(v int64) bool { return v >= -2048 && v <= 2047 }
|
||||
|
||||
// loong64GuardBytes emits the stack-split guard prefix. blockStart is the
|
||||
// function-relative address of the morestack call at the end of the function;
|
||||
// branch displacements are in instructions and are computed from each
|
||||
// branch's own position.
|
||||
func loong64GuardBytes(fi loong64FrameInfo, blockStart int) []byte {
|
||||
// MOVV 16(g), R20 (g.stackguard0), g = R22.
|
||||
ws := []uint32{l64irr(l64loadStoreTable["MOVV"].ld, 16, 22, 20)}
|
||||
off := int64(fi.autosize - stackSmall)
|
||||
// beq appends BEQ R20, blockStart from the branch's own position.
|
||||
beq := func() {
|
||||
ws = append(ws, loong64Beqz(20, int32((blockStart-len(ws)*4)>>2)))
|
||||
}
|
||||
switch fi.splitClass {
|
||||
case 0:
|
||||
// SGTU SP, R20, R20; BEQ R20, more
|
||||
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 3, 20, 20))
|
||||
beq()
|
||||
case 1:
|
||||
ws = append(ws, loong64MediumWords(off)...)
|
||||
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 24, 20, 20))
|
||||
beq()
|
||||
default:
|
||||
// SGTU $off, SP, R24 catches the SP underflow a huge frame would
|
||||
// cause; BNE jumps to morestack in that case.
|
||||
ws = append(ws, loong64MatWords(nil, off)...)
|
||||
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 30, 3, 24))
|
||||
ws = append(ws, loong64Bnez(24, int32((blockStart-len(ws)*4)>>2)))
|
||||
ws = append(ws, loong64MatWords(nil, -off)...)
|
||||
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 24))
|
||||
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 24, 20, 20))
|
||||
beq()
|
||||
}
|
||||
return l64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// loong64MediumWords emits the medium-class stack check for offset off: the
|
||||
// ADDV immediate when it fits, otherwise the same sequence with the constant
|
||||
// materialised in R30.
|
||||
func loong64MediumWords(off int64) []uint32 {
|
||||
if off <= 2048 {
|
||||
return []uint32{l64irr(l64DualTable["ADDV"].imm, int(-off), 3, 24)}
|
||||
}
|
||||
ws := loong64MatWords(nil, -off)
|
||||
return append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 24))
|
||||
}
|
||||
|
||||
// loong64Beqz/loong64Bnez build the 21-bit conditional branches against R0
|
||||
// that the toolchain emits for its guard compares.
|
||||
func loong64Beqz(rj int, dispInstr int32) uint32 {
|
||||
return l64ir21(l64branch21Table["BEQZ"], int(dispInstr), rj)
|
||||
}
|
||||
|
||||
func loong64Bnez(rj int, dispInstr int32) uint32 {
|
||||
return l64ir21(l64branch21Table["BNEZ"], int(dispInstr), rj)
|
||||
}
|
||||
|
||||
// loong64MoreStackBlock emits the trailing block: MOVV R1, R31 (save LR, the
|
||||
// toolchain's OR R1, R0, R31 expansion), BL runtime.morestack_noctxt, B back
|
||||
// to the function entry.
|
||||
func loong64MoreStackBlock(blockStart int) ([]byte, Reloc) {
|
||||
ws := []uint32{
|
||||
l64rrr(l64DualTable["OR"].rrr, 0, 1, 31), // MOVV R1, R31 (OR R1, R0, R31)
|
||||
l64bbl(l64jumpTable["BL"], 0), // BL, patched by the linker
|
||||
}
|
||||
disp := (-(blockStart + 8)) >> 2
|
||||
ws = append(ws, l64bbl(l64jumpTable["B"], int(disp)))
|
||||
reloc := Reloc{
|
||||
Off: blockStart + 4,
|
||||
After: blockStart + 8,
|
||||
Name: "runtime\u00b7morestack_noctxt",
|
||||
Kind: RelLoong64Branch,
|
||||
}
|
||||
return l64WordsLE(ws...), reloc
|
||||
}
|
||||
|
||||
// loong64IsLeaf reports whether a function contains no call instructions
|
||||
// (JAL/BL/CALL), matching the toolchain's LEAF mark, which drives the frame
|
||||
// and the epilogue shape.
|
||||
@@ -79,17 +238,37 @@ func loong64IsLeaf(t *ast.Text) bool {
|
||||
return true
|
||||
}
|
||||
|
||||
// loong64Prologue returns the prologue bytes for a loong64 function.
|
||||
// loong64Prologue returns the prologue bytes for a loong64 function. When
|
||||
// the LR store offset leaves the toolchain's 12-bit store range ([-2046,
|
||||
// 2045], BIG_12 = 2046) or the SP adjust immediate its 12-bit immediate
|
||||
// range, each switches to the R30 materialisation the assembler expands it
|
||||
// to: the store uses the rounding %hi/%lo split (LU12IW of (v+2048)>>12,
|
||||
// REGTMP += SP, store at the raw offset), the adjust the floor split
|
||||
// (LU12IW, ORI when the low part is non-zero, REGTMP += SP).
|
||||
func loong64Prologue(fi loong64FrameInfo) []byte {
|
||||
if fi.autosize == 0 {
|
||||
return nil
|
||||
}
|
||||
addiD := l64DualTable["ADDV"].imm
|
||||
return l64WordsLE(
|
||||
l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, 3, 1), // MOVV R1, -autosize(R3)
|
||||
l64irr(addiD, -fi.autosize, 3, 3), // ADDV $-autosize, R3
|
||||
l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1), // MOVV R1, 0(R3)
|
||||
)
|
||||
var ws []uint32
|
||||
storeBase := 3
|
||||
if fi.autosize > 2046 {
|
||||
// The store goes through REGTMP: LU12IW of the rounding split,
|
||||
// REGTMP += SP, then the store at REGTMP with the truncated offset.
|
||||
v := -int64(fi.autosize)
|
||||
ws = append(ws, l64ir(l64Lu12iwOp, int((v+2048)>>12), 30))
|
||||
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 3, 30, 30))
|
||||
storeBase = 30
|
||||
}
|
||||
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, storeBase, 1)) // MOVV R1, -autosize(base)
|
||||
if loong64Imm12(-int64(fi.autosize)) {
|
||||
ws = append(ws, l64irr(addiD, -fi.autosize, 3, 3)) // ADDV $-autosize, R3
|
||||
} else {
|
||||
ws = append(ws, loong64MatWords(nil, -int64(fi.autosize))...)
|
||||
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 3))
|
||||
}
|
||||
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1)) // MOVV R1, 0(R3)
|
||||
return l64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// loong64Return returns the bytes for a RET: the epilogue (restore LR and
|
||||
@@ -98,17 +277,49 @@ func loong64Return(fi loong64FrameInfo) []byte {
|
||||
var ws []uint32
|
||||
if fi.autosize != 0 {
|
||||
if !fi.leaf {
|
||||
// MOVV 0(R3), R1 — restore the link register.
|
||||
// MOVV 0(R3), R1, restore the link register.
|
||||
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1))
|
||||
}
|
||||
// ADDV $autosize, R3 — close the frame.
|
||||
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
|
||||
// ADDV $autosize, R3, close the frame (materialised when the
|
||||
// immediate does not fit).
|
||||
if loong64Imm12(int64(fi.autosize)) {
|
||||
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
|
||||
} else {
|
||||
ws = append(ws, loong64MatWords(nil, int64(fi.autosize))...)
|
||||
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 3))
|
||||
}
|
||||
}
|
||||
// jirl r0, r1, 0 — return.
|
||||
// jirl r0, r1, 0, return.
|
||||
ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0))
|
||||
return l64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// loong64StoreWords reports the prologue word count of the LR store, and
|
||||
// loong64AdjustWords the word count of an SP adjust of v: the immediate
|
||||
// forms when they fit, otherwise the R30 materialisation sequences.
|
||||
func loong64StoreWords(autosize int) int {
|
||||
if autosize > 2046 {
|
||||
return 3
|
||||
}
|
||||
return 1
|
||||
}
|
||||
|
||||
func loong64AdjustWords(v int64) int {
|
||||
if loong64Imm12(v) {
|
||||
return 1
|
||||
}
|
||||
return loong64MatLen(v) + 1
|
||||
}
|
||||
|
||||
// loong64EpilogueWords reports the epilogue word count the RET expands to.
|
||||
func loong64EpilogueWords(fi loong64FrameInfo) int {
|
||||
n := loong64AdjustWords(int64(fi.autosize))
|
||||
if !fi.leaf {
|
||||
n++
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// loong64ResolvePseudo translates a pseudo-register memory reference into a
|
||||
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
|
||||
// x-N(SP) → (autosize - N)(SP). Returns base = -1 for an unresolvable
|
||||
|
||||
@@ -7,7 +7,7 @@ import (
|
||||
"bytes"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// TestLOONG64_sys exercises the no-operand system instructions and the
|
||||
@@ -232,7 +232,10 @@ DATA ·table+0(SB)/8, $42
|
||||
}
|
||||
|
||||
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
|
||||
// invalid register operands and operand-count mismatches.
|
||||
// invalid register operands and operand-count mismatches. The X0 and
|
||||
// AMADDW cases follow the oracle: GOARCH=loong64 go tool asm rejects
|
||||
// `BEQZ X0` (the X bank is not an integer register) and the two-register
|
||||
// `AMADDW R4, R5` (the AM* family is strictly `val, (addr), result`).
|
||||
func TestLOONG64_errors(t *testing.T) {
|
||||
cases := []string{
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
@@ -280,7 +283,7 @@ done:
|
||||
// TestLOONG64_pcsp checks the stack-adjustment table of a framed function:
|
||||
// the prologue raises the SP delta by autosize (in effect from the third
|
||||
// instruction) and the RET's epilogue restores it to zero, with the pc deltas
|
||||
// in MinLC (4) units — byte-identical to `go tool asm`.
|
||||
// in MinLC (4) units; byte-identical to `go tool asm`.
|
||||
func TestLOONG64_pcsp(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
@@ -347,21 +350,94 @@ TEXT ·sb(SB), NOSPLIT, $0
|
||||
}
|
||||
}
|
||||
|
||||
// TestLOONG64_movImmToFp checks the immediate-to-FP move forms.
|
||||
// TestLOONG64_movImmToFp checks the immediate-to-FP move: MOVW $c, Fd is the
|
||||
// only spelling the toolchain accepts, expanding to ori (or addi.w for the
|
||||
// negative span) into R30 plus movgr2fr.w. The pinned words are the
|
||||
// toolchain's own bytes; the other widths and out-of-range constants are
|
||||
// illegal combinations there and are diagnosed here.
|
||||
func TestLOONG64_movImmToFp(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·fpmov(SB), NOSPLIT, $0
|
||||
MOVV $0x1, F0
|
||||
MOVW $0x1, F0
|
||||
MOVW $0x2, F4
|
||||
MOVW $-1, F4
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
want := []byte{
|
||||
0x00, 0x04, 0x80, 0x03, // ori f0, r0, 1
|
||||
0x04, 0x08, 0x80, 0x03, // ori f4, r0, 2
|
||||
0x1e, 0x04, 0x80, 0x03, // ori r30, r0, 1
|
||||
0xc0, 0xa7, 0x14, 0x01, // movgr2fr.w f0, r30
|
||||
0x1e, 0x08, 0x80, 0x03, // ori r30, r0, 2
|
||||
0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30
|
||||
0x1e, 0xfc, 0xbf, 0x02, // addi.w r30, r0, -1
|
||||
0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30
|
||||
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
|
||||
}
|
||||
if !bytes.Equal(code, want) {
|
||||
t.Errorf("code = % x\nwant % x", code, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestLOONG64_movImmToFpErrors checks the immediate-to-FP diagnostics: the
|
||||
// widths the toolchain rejects as illegal combinations, and constants beyond
|
||||
// the 12-bit ori/addi.w span (the toolchain never materialises a wider
|
||||
// constant on this path).
|
||||
func TestLOONG64_movImmToFpErrors(t *testing.T) {
|
||||
cases := []string{
|
||||
"MOVV $1, F0",
|
||||
"MOVF $2, F4",
|
||||
"MOVD $2, F4",
|
||||
"MOVW $100000, F1",
|
||||
"MOVW $-2049, F1",
|
||||
"MOVW $4096, F1",
|
||||
}
|
||||
for _, src := range cases {
|
||||
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
|
||||
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", src)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestLOONG64_branch16Unsigned pins the unsigned two-operand branches: with
|
||||
// one register BLTU/BGEU keep the register-register form against R0 (never
|
||||
// taken), the toolchain's encoding, where a beqz would test the wrong
|
||||
// condition; the three-operand forms are unchanged.
|
||||
func TestLOONG64_branch16Unsigned(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·u(SB), NOSPLIT, $0
|
||||
BLTU R4, done
|
||||
BGEU R5, done
|
||||
BLTU R6, R7, done
|
||||
BGEU R8, R9, done
|
||||
done:
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x68001080, // bltu r4, r0, +4
|
||||
0x6C000CA0, // bgeu r5, r0, +3
|
||||
0x680008C7, // bltu r6, r7, +2
|
||||
0x6C000509, // bgeu r8, r9, +1
|
||||
0x4C000020, // jirl r0, r1, 0
|
||||
)
|
||||
}
|
||||
|
||||
// TestLOONG64_bitFieldRange checks the BSTRINS/BSTRPICK bit-number
|
||||
// validation, mirroring the toolchain's "illegal bit number" rule: 0..31 for
|
||||
// the .w forms, 0..63 for the .d forms, and lsb <= msb.
|
||||
func TestLOONG64_bitFieldRange(t *testing.T) {
|
||||
cases := []string{
|
||||
"BSTRINSW $32, R4, $0, R5",
|
||||
"BSTRPICKW $31, R4, $32, R5",
|
||||
"BSTRINSV $64, R4, $0, R5",
|
||||
"BSTRPICKV $3, R4, $4, R5",
|
||||
"BSTRINSW $-1, R4, $0, R5",
|
||||
}
|
||||
for _, src := range cases {
|
||||
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
|
||||
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", src)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// TestLOONG64RelocOffsetsIncludePrologue pins the function-relative
|
||||
// relocation offsets of a framed loong64 function: the offsets used to
|
||||
// exclude the prologue, so every relocation landed on a prologue
|
||||
// instruction in the GOOBJ/ELF output.
|
||||
func TestLOONG64RelocOffsetsIncludePrologue(t *testing.T) {
|
||||
f, errs := parser.Parse("k_loong64.s", "TEXT \u00b7f(SB), $16-0\n"+
|
||||
"\tMOVV $gdata(SB), R4\n"+
|
||||
"\tRET\n"+
|
||||
"GLOBL gdata(SB), $8\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileLOONG64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
|
||||
// Layout: 12-byte prologue (autosize 32), pcalau12i+addi.d (12, 16),
|
||||
// epilogue with RET.
|
||||
if len(fn.Relocs) != 2 {
|
||||
t.Fatalf("relocs = %d, want 2", len(fn.Relocs))
|
||||
}
|
||||
hi, lo := fn.Relocs[0], fn.Relocs[1]
|
||||
if hi.Kind != RelLoong64AddrHi || hi.Off != 12 || hi.After != 12 {
|
||||
t.Errorf("hi reloc = {off %d after %d kind %d}, want {off 12 after 12 kind RelLoong64AddrHi}", hi.Off, hi.After, hi.Kind)
|
||||
}
|
||||
if lo.Kind != RelLoong64AddrLo || lo.Off != 16 || lo.After != 16 {
|
||||
t.Errorf("lo reloc = {off %d after %d kind %d}, want {off 16 after 16 kind RelLoong64AddrLo}", lo.Off, lo.After, lo.Kind)
|
||||
}
|
||||
}
|
||||
+59
-5
@@ -14,6 +14,51 @@ type Imm int64
|
||||
|
||||
func (Imm) isOperand() {}
|
||||
|
||||
// RegList is a bracketed register range, [Z0-Z3]: the four-register source
|
||||
// of the 4FMAPS and 4VNNIW families. The EVEX emit path carries the list's
|
||||
// low register through the inverted 5-bit V'VVVV field; the three higher
|
||||
// registers are implied by the instruction, so only the pair travels here.
|
||||
type RegList struct {
|
||||
Lo Reg
|
||||
Hi Reg // implied by the encoding; Lo.idx+3 by construction
|
||||
}
|
||||
|
||||
func (RegList) isOperand() {}
|
||||
|
||||
// FloatImm is a floating-point immediate ($-1.0). The SSE mnemonics whose
|
||||
// encoding takes an XMM/memory source at that position rewrite it as a read
|
||||
// from a read-only pool constant ($f64.<hex> or $f32.<hex>), the toolchain's
|
||||
// own behaviour; every other instruction rejects it.
|
||||
type FloatImm struct {
|
||||
Text string // the numeric text as written, sign excluded
|
||||
Neg bool // a leading minus
|
||||
}
|
||||
|
||||
func (FloatImm) isOperand() {}
|
||||
|
||||
// TLSMem is a thread-local access, the source form off(base)(TLS*1) with the
|
||||
// base dropped: the toolchain's one-instruction TLS rewrite assembles it as
|
||||
// the segment-prefixed absolute whose disp32 carries an R_TLS_LE patch site
|
||||
// (the linker fills the TLS slot offset).
|
||||
type TLSMem struct {
|
||||
Disp int64
|
||||
Size int
|
||||
Seg byte // the segment override: FS (0x64) or GS (0x65) on windows
|
||||
}
|
||||
|
||||
func (TLSMem) isOperand() {}
|
||||
|
||||
// SegAbs is a segment-absolute access, 0x30(GS): the segment override
|
||||
// prefixes a disp32 absolute reference with no relocation. The base
|
||||
// register spellings GS and FS produce it.
|
||||
type SegAbs struct {
|
||||
Disp int64
|
||||
Size int
|
||||
Seg byte // 0x64 FS, 0x65 GS
|
||||
}
|
||||
|
||||
func (SegAbs) isOperand() {}
|
||||
|
||||
// Mem is a memory operand of the form disp(base)(index*scale).
|
||||
type Mem struct {
|
||||
Base Reg
|
||||
@@ -23,6 +68,7 @@ type Mem struct {
|
||||
Size int // operand width in bytes
|
||||
HasBase bool
|
||||
HasIndex bool
|
||||
Seg byte // segment override prefix (0x64 FS, 0x65 GS); 0 = none
|
||||
}
|
||||
|
||||
func (Mem) isOperand() {}
|
||||
@@ -37,11 +83,6 @@ func Idx(base, index Reg, scale int, disp int64, size int) Mem {
|
||||
return Mem{Base: base, Index: index, Scale: scale, Disp: disp, Size: size, HasBase: true, HasIndex: true}
|
||||
}
|
||||
|
||||
// Rip builds a RIP-relative memory operand (RIP)+disp.
|
||||
func Rip(disp int64, size int) Mem {
|
||||
return Mem{Disp: disp, Size: size}
|
||||
}
|
||||
|
||||
// sbMem is a memory operand that references a static (SB) symbol. It encodes
|
||||
// as a RIP-relative reference with a placeholder displacement; the encoder
|
||||
// records a patch site so the file-level layout can fill in the true rel32
|
||||
@@ -53,3 +94,16 @@ type sbMem struct {
|
||||
}
|
||||
|
||||
func (sbMem) isOperand() {}
|
||||
|
||||
// isX86Mem reports whether the operand is an amd64 memory reference: a base
|
||||
// or indexed Mem, or an SB-relative sbMem. Encoders that gate on "memory in
|
||||
// this position" must accept both; the r/m emitters distinguish the two
|
||||
// themselves.
|
||||
func isX86Mem(o Operand) bool {
|
||||
switch o.(type) {
|
||||
case Mem, sbMem:
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
+36
-31
@@ -7,36 +7,39 @@
|
||||
// by round-tripping through golang.org/x/arch's decoder in the tests.
|
||||
package asm
|
||||
|
||||
import "maps"
|
||||
|
||||
import "strings"
|
||||
|
||||
// Reg is an x86-64 register. In Plan 9 assembly the classic names (AX, BX, …)
|
||||
// are size-agnostic — the instruction suffix (MOVQ vs MOVL) fixes the width —
|
||||
// are size-agnostic, the instruction suffix (MOVQ vs MOVL) fixes the width
|
||||
// so the encoder keys off the register's index and lets the mnemonic supply the
|
||||
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
|
||||
// occupy indices 4–7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
|
||||
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
|
||||
// those indices but require one. The mask flag marks the AVX-512 opmask
|
||||
// registers K0–K7.
|
||||
// registers K0-K7, the fp flag the x87 stack registers F0-F7.
|
||||
type Reg struct {
|
||||
idx int
|
||||
size int // informational width implied by the name; the mnemonic decides
|
||||
high bool // AH/CH/DH/BH
|
||||
mask bool // K0–K7 opmask register
|
||||
mask bool // K0-K7 opmask register
|
||||
fp bool // F0-F7 x87 stack register
|
||||
}
|
||||
|
||||
// Index returns the register number (0–15 for GPRs, 0–31 for vectors).
|
||||
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
|
||||
func (r Reg) Index() int { return r.idx }
|
||||
|
||||
// Size returns the width in bytes implied by the register's name.
|
||||
func (r Reg) Size() int { return r.size }
|
||||
|
||||
// IsMask reports whether r is an AVX-512 opmask register (K0–K7).
|
||||
// IsMask reports whether r is an AVX-512 opmask register (K0-K7).
|
||||
func (r Reg) IsMask() bool { return r.mask }
|
||||
|
||||
func (r Reg) isOperand() {}
|
||||
|
||||
// needsREX reports whether this register forces a REX prefix at the given
|
||||
// operand size: the extended registers R8–R15 always do, and at byte size the
|
||||
// low registers SPL/BPL/SIL/DIL (indices 4–7, not high) do as well.
|
||||
// operand size: the extended registers R8-R15 always do, and at byte size the
|
||||
// low registers SPL/BPL/SIL/DIL (indices 4-7, not high) do as well.
|
||||
func (r Reg) needsREX(opSize int) bool {
|
||||
if r.idx >= 8 {
|
||||
return true
|
||||
@@ -63,28 +66,28 @@ var (
|
||||
CX = Reg{idx: 1, size: 2}
|
||||
DX = Reg{idx: 2, size: 2}
|
||||
BX = Reg{idx: 3, size: 2}
|
||||
SP = Reg{idx: 4, size: 2}
|
||||
BP = Reg{idx: 5, size: 2}
|
||||
_ = Reg{idx: 4, size: 2}
|
||||
_ = Reg{idx: 5, size: 2}
|
||||
SI = Reg{idx: 6, size: 2}
|
||||
DI = Reg{idx: 7, size: 2}
|
||||
|
||||
EAX = Reg{idx: 0, size: 4}
|
||||
ECX = Reg{idx: 1, size: 4}
|
||||
EDX = Reg{idx: 2, size: 4}
|
||||
EBX = Reg{idx: 3, size: 4}
|
||||
ESP = Reg{idx: 4, size: 4}
|
||||
EBP = Reg{idx: 5, size: 4}
|
||||
ESI = Reg{idx: 6, size: 4}
|
||||
EDI = Reg{idx: 7, size: 4}
|
||||
_ = Reg{idx: 0, size: 4}
|
||||
_ = Reg{idx: 1, size: 4}
|
||||
_ = Reg{idx: 2, size: 4}
|
||||
_ = Reg{idx: 3, size: 4}
|
||||
_ = Reg{idx: 4, size: 4}
|
||||
_ = Reg{idx: 5, size: 4}
|
||||
_ = Reg{idx: 6, size: 4}
|
||||
_ = Reg{idx: 7, size: 4}
|
||||
|
||||
RAX = Reg{idx: 0, size: 8}
|
||||
RCX = Reg{idx: 1, size: 8}
|
||||
RDX = Reg{idx: 2, size: 8}
|
||||
RBX = Reg{idx: 3, size: 8}
|
||||
RSP = Reg{idx: 4, size: 8}
|
||||
RBP = Reg{idx: 5, size: 8}
|
||||
RSI = Reg{idx: 6, size: 8}
|
||||
RDI = Reg{idx: 7, size: 8}
|
||||
_ = Reg{idx: 0, size: 8}
|
||||
_ = Reg{idx: 1, size: 8}
|
||||
_ = Reg{idx: 2, size: 8}
|
||||
_ = Reg{idx: 3, size: 8}
|
||||
_ = Reg{idx: 4, size: 8}
|
||||
_ = Reg{idx: 5, size: 8}
|
||||
_ = Reg{idx: 6, size: 8}
|
||||
_ = Reg{idx: 7, size: 8}
|
||||
)
|
||||
|
||||
// regByName maps an assembly register name (case-insensitive) to a Reg.
|
||||
@@ -121,19 +124,17 @@ func buildRegByName() map[string]Reg {
|
||||
}
|
||||
|
||||
// 8-bit: AL..BH, SPL..DIL, R8B..R15B.
|
||||
for n, r := range map[string]Reg{
|
||||
maps.Copy(m, map[string]Reg{
|
||||
"AL": AL, "CL": CL, "DL": DL, "BL": BL,
|
||||
"AH": AH, "CH": CH, "DH": DH, "BH": BH,
|
||||
"SPL": SPL, "BPL": BPL, "SIL": SIL, "DIL": DIL,
|
||||
} {
|
||||
m[n] = r
|
||||
}
|
||||
})
|
||||
for i := 8; i <= 15; i++ {
|
||||
m["R"+itoa(i)+"B"] = Reg{idx: i, size: 1}
|
||||
}
|
||||
|
||||
// Vector: X0..X31 (128-bit, size 16), Y0..Y31 (256-bit, size 32),
|
||||
// Z0..Z31 (512-bit, size 64). Indices 16–31 are only encodable in EVEX
|
||||
// Z0..Z31 (512-bit, size 64). Indices 16-31 are only encodable in EVEX
|
||||
// (AVX-512) instructions; the encoder validates that through its tables.
|
||||
for i := 0; i <= 31; i++ {
|
||||
m["X"+itoa(i)] = Reg{idx: i, size: 16}
|
||||
@@ -144,6 +145,10 @@ func buildRegByName() map[string]Reg {
|
||||
for i := 0; i <= 7; i++ {
|
||||
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
|
||||
}
|
||||
// x87 stack: F0..F7.
|
||||
for i := 0; i <= 7; i++ {
|
||||
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
|
||||
+1777
-98
File diff suppressed because it is too large
Load Diff
+174
-50
@@ -63,9 +63,9 @@ func riscvRegNum(name string) int {
|
||||
return 24
|
||||
case "X25", "S9":
|
||||
return 25
|
||||
case "X26", "S10":
|
||||
case "X26", "S10", "CTXT":
|
||||
return 26
|
||||
case "X27", "S11":
|
||||
case "X27", "S11", "g":
|
||||
return 27
|
||||
case "X28", "T3":
|
||||
return 28
|
||||
@@ -141,10 +141,36 @@ func riscvRegNum(name string) int {
|
||||
case "F31", "FT11":
|
||||
return 31
|
||||
default:
|
||||
// Vector registers V0-V31 (the "V" extension). They share the
|
||||
// register numbering with the integer file: a bare number 0-31.
|
||||
if len(name) >= 2 && name[0] == 'V' {
|
||||
if n, ok := parseRegDigits(name[1:], 31); ok {
|
||||
return n
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
}
|
||||
|
||||
// parseRegDigits parses a decimal register suffix and reports whether it is
|
||||
// within [0, max].
|
||||
func parseRegDigits(digits string, max int) (int, bool) {
|
||||
if digits == "" {
|
||||
return 0, false
|
||||
}
|
||||
n := 0
|
||||
for i := 0; i < len(digits); i++ {
|
||||
if digits[i] < '0' || digits[i] > '9' {
|
||||
return 0, false
|
||||
}
|
||||
n = n*10 + int(digits[i]-'0')
|
||||
if n > max {
|
||||
return 0, false
|
||||
}
|
||||
}
|
||||
return n, true
|
||||
}
|
||||
|
||||
// RISC-V instruction encoding parameters.
|
||||
type riscvEnc struct {
|
||||
opcode uint32 // bits [6:0]
|
||||
@@ -154,7 +180,7 @@ type riscvEnc struct {
|
||||
|
||||
// riscvInstrTable maps RISC-V mnemonics to their encoding.
|
||||
var riscvInstrTable = map[string]riscvEnc{
|
||||
// RV64I — R-type arithmetic/logic.
|
||||
// RV64I, R-type arithmetic/logic.
|
||||
"ADD": {0x33, 0x0, 0x00},
|
||||
"SUB": {0x33, 0x0, 0x20},
|
||||
"SLL": {0x33, 0x1, 0x00},
|
||||
@@ -165,20 +191,20 @@ var riscvInstrTable = map[string]riscvEnc{
|
||||
"SRA": {0x33, 0x5, 0x20},
|
||||
"OR": {0x33, 0x6, 0x00},
|
||||
"AND": {0x33, 0x7, 0x00},
|
||||
// RV64I — 32-bit variants (W suffix).
|
||||
// RV64I, 32-bit variants (W suffix).
|
||||
"ADDW": {0x3B, 0x0, 0x00},
|
||||
"SUBW": {0x3B, 0x0, 0x20},
|
||||
"SLLW": {0x3B, 0x1, 0x00},
|
||||
"SRLW": {0x3B, 0x5, 0x00},
|
||||
"SRAW": {0x3B, 0x5, 0x20},
|
||||
// RV64I — I-type shift-immediate (shamt in rs2 field).
|
||||
// RV64I, I-type shift-immediate (shamt in rs2 field).
|
||||
"SLLI": {0x13, 0x1, 0x00},
|
||||
"SRLI": {0x13, 0x5, 0x00},
|
||||
"SRAI": {0x13, 0x5, 0x20},
|
||||
"SLLIW": {0x1B, 0x1, 0x00},
|
||||
"SRLIW": {0x1B, 0x5, 0x00},
|
||||
"SRAIW": {0x1B, 0x5, 0x20},
|
||||
// RV64M — multiply/divide.
|
||||
// RV64M, multiply/divide.
|
||||
"MUL": {0x33, 0x0, 0x01},
|
||||
"MULH": {0x33, 0x1, 0x01},
|
||||
"MULHSU": {0x33, 0x2, 0x01},
|
||||
@@ -187,13 +213,16 @@ var riscvInstrTable = map[string]riscvEnc{
|
||||
"DIVU": {0x33, 0x5, 0x01},
|
||||
"REM": {0x33, 0x6, 0x01},
|
||||
"REMU": {0x33, 0x7, 0x01},
|
||||
// RV64M — 32-bit variants.
|
||||
// RV64M, 32-bit variants.
|
||||
"MULW": {0x3B, 0x0, 0x01},
|
||||
"DIVW": {0x3B, 0x4, 0x01},
|
||||
"DIVUW": {0x3B, 0x5, 0x01},
|
||||
"REMW": {0x3B, 0x6, 0x01},
|
||||
"REMUW": {0x3B, 0x7, 0x01},
|
||||
// RV64I — I-type arithmetic.
|
||||
// Zicond conditional zeroing.
|
||||
"CZEROEQZ": {0x33, 0x5, 0x07},
|
||||
"CZERONEZ": {0x33, 0x7, 0x07},
|
||||
// RV64I, I-type arithmetic.
|
||||
"ADDI": {0x13, 0x0, 0x00},
|
||||
"ADDIW": {0x1B, 0x0, 0x00},
|
||||
"SLTI": {0x13, 0x2, 0x00},
|
||||
@@ -221,38 +250,49 @@ var riscvInstrTable = map[string]riscvEnc{
|
||||
"BGE": {0x63, 0x5, 0x00},
|
||||
"BLTU": {0x63, 0x6, 0x00},
|
||||
"BGEU": {0x63, 0x7, 0x00},
|
||||
// The swapped-spelling comparison forms: encoded as BLT/BGE/BLTU/BGEU
|
||||
// with the register operands swapped.
|
||||
"BGT": {0x63, 0x4, 0x00},
|
||||
"BLE": {0x63, 0x5, 0x00},
|
||||
"BGTU": {0x63, 0x6, 0x00},
|
||||
"BLEU": {0x63, 0x7, 0x00},
|
||||
// U-type.
|
||||
"LUI": {0x37, 0x0, 0x00},
|
||||
"AUIPC": {0x17, 0x0, 0x00},
|
||||
// System.
|
||||
"ECALL": {0x73, 0x0, 0x00},
|
||||
"EBREAK": {0x73, 0x0, 0x00},
|
||||
"FENCE": {0x0F, 0x0, 0x00},
|
||||
// JALR — indirect jump/call (I-type).
|
||||
"ECALL": {0x73, 0x0, 0x00},
|
||||
"EBREAK": {0x73, 0x0, 0x00},
|
||||
"FENCE": {0x0F, 0x0, 0x00},
|
||||
"FENCE.TSO": {0x0F, 0x0, 0x00},
|
||||
"PAUSE": {0x0F, 0x0, 0x00},
|
||||
// JALR, indirect jump/call (I-type).
|
||||
"JALR": {0x67, 0x0, 0x00},
|
||||
|
||||
// RV64A — atomics (AMO opcode 0x2F).
|
||||
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
|
||||
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
|
||||
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
|
||||
"AMOADDW": {0x2F, 0x2, 0x00 << 2},
|
||||
"AMOADDD": {0x2F, 0x3, 0x00 << 2},
|
||||
"AMOANDW": {0x2F, 0x2, 0x0C << 2},
|
||||
"AMOANDD": {0x2F, 0x3, 0x0C << 2},
|
||||
"AMOORW": {0x2F, 0x2, 0x06 << 2},
|
||||
"AMOORD": {0x2F, 0x3, 0x06 << 2},
|
||||
"AMOXORW": {0x2F, 0x2, 0x04 << 2},
|
||||
"AMOXORD": {0x2F, 0x3, 0x04 << 2},
|
||||
"AMOMAXW": {0x2F, 0x2, 0x14 << 2},
|
||||
"AMOMAXD": {0x2F, 0x3, 0x14 << 2},
|
||||
"AMOMINW": {0x2F, 0x2, 0x10 << 2},
|
||||
"AMOMIND": {0x2F, 0x3, 0x10 << 2},
|
||||
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2},
|
||||
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2},
|
||||
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
|
||||
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
|
||||
// RV64A, atomics (AMO opcode 0x2F).
|
||||
// funct3: 0x2 = word, 0x3 = doubleword. The stored funct7 is the full
|
||||
// 7-bit field: funct5 in the upper five bits and the aq/rl ordering bits in
|
||||
// the lower two, exactly as the toolchain writes them: every AMO sets both
|
||||
// aq and rl (funct7 |= 3).
|
||||
"AMOSWAPW": {0x2F, 0x2, 0x01<<2 | 0x3},
|
||||
"AMOSWAPD": {0x2F, 0x3, 0x01<<2 | 0x3},
|
||||
"AMOADDW": {0x2F, 0x2, 0x00<<2 | 0x3},
|
||||
"AMOADDD": {0x2F, 0x3, 0x00<<2 | 0x3},
|
||||
"AMOANDW": {0x2F, 0x2, 0x0C<<2 | 0x3},
|
||||
"AMOANDD": {0x2F, 0x3, 0x0C<<2 | 0x3},
|
||||
"AMOORW": {0x2F, 0x2, 0x08<<2 | 0x3},
|
||||
"AMOORD": {0x2F, 0x3, 0x08<<2 | 0x3},
|
||||
"AMOXORW": {0x2F, 0x2, 0x04<<2 | 0x3},
|
||||
"AMOXORD": {0x2F, 0x3, 0x04<<2 | 0x3},
|
||||
"AMOMAXW": {0x2F, 0x2, 0x14<<2 | 0x3},
|
||||
"AMOMAXD": {0x2F, 0x3, 0x14<<2 | 0x3},
|
||||
"AMOMINW": {0x2F, 0x2, 0x10<<2 | 0x3},
|
||||
"AMOMIND": {0x2F, 0x3, 0x10<<2 | 0x3},
|
||||
"AMOMAXUW": {0x2F, 0x2, 0x1C<<2 | 0x3},
|
||||
"AMOMAXUD": {0x2F, 0x3, 0x1C<<2 | 0x3},
|
||||
"AMOMINUW": {0x2F, 0x2, 0x18<<2 | 0x3},
|
||||
"AMOMINUD": {0x2F, 0x3, 0x18<<2 | 0x3},
|
||||
|
||||
// RV64F/D — floating-point arithmetic.
|
||||
// RV64F/D, floating-point arithmetic.
|
||||
"FADDS": {0x53, 0x0, 0x00},
|
||||
"FSUBS": {0x53, 0x0, 0x04},
|
||||
"FMULS": {0x53, 0x0, 0x08},
|
||||
@@ -273,14 +313,25 @@ var riscvInstrTable = map[string]riscvEnc{
|
||||
"FMAXS": {0x53, 0x1, 0x14},
|
||||
"FMIND": {0x53, 0x0, 0x15},
|
||||
"FMAXD": {0x53, 0x1, 0x15},
|
||||
// FP sign injection (double): rs2 carries the sign source.
|
||||
"FSGNJD": {0x53, 0x0, 0x11},
|
||||
"FSGNJS": {0x53, 0x0, 0x10},
|
||||
"FSGNJX": {0x53, 0x0, 0x14},
|
||||
"FSGNJXD": {0x53, 0x0, 0x15},
|
||||
"FSGNJXS": {0x53, 0x0, 0x14},
|
||||
"FSGNJND": {0x53, 0x1, 0x11},
|
||||
"FSGNJNS": {0x53, 0x1, 0x10},
|
||||
"FSGNJNX": {0x53, 0x1, 0x14},
|
||||
|
||||
// RV64A — load-reserved / store-conditional (funct5 0x02 / 0x03).
|
||||
"LRW": {0x2F, 0x2, 0x02 << 2},
|
||||
"LRD": {0x2F, 0x3, 0x02 << 2},
|
||||
"SCW": {0x2F, 0x2, 0x03 << 2},
|
||||
"SCD": {0x2F, 0x3, 0x03 << 2},
|
||||
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
|
||||
// The toolchain gives LR acquire ordering (aq = 1) and SC release
|
||||
// ordering (rl = 1).
|
||||
"LRW": {0x2F, 0x2, 0x02<<2 | 0x2},
|
||||
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
|
||||
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
|
||||
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
|
||||
|
||||
// FP compare — result in integer register (funct7 0x50/0x51).
|
||||
// FP compare, result in integer register (funct7 0x50/0x51).
|
||||
"FEQS": {0x53, 0x2, 0x50},
|
||||
"FLTS": {0x53, 0x1, 0x50},
|
||||
"FLES": {0x53, 0x0, 0x50},
|
||||
@@ -296,11 +347,11 @@ func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||
}
|
||||
|
||||
// riscvAMOType encodes an atomic (AMO) instruction.
|
||||
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode.
|
||||
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2).
|
||||
// Layout: funct7 | rs2 | rs1 | funct3 | rd | opcode, where funct7 carries the
|
||||
// funct5 in its upper five bits and the aq/rl ordering bits in the lower two
|
||||
// (the table stores the full field, so the word needs no reassembly).
|
||||
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value
|
||||
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
@@ -328,6 +379,8 @@ var riscvCvtTable = map[string]riscvCvtEnc{
|
||||
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
|
||||
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
|
||||
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
|
||||
"FCLASSS": {0x70, 0x0, 0x53}, // classify float32 → GPR mask
|
||||
"FCLASSD": {0x70, 0x0, 0x53}, // classify float64 → GPR mask
|
||||
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
|
||||
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
|
||||
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
|
||||
@@ -340,6 +393,10 @@ var riscvCvtTable = map[string]riscvCvtEnc{
|
||||
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
|
||||
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
|
||||
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
|
||||
// The toolchain's W/D suffix spellings of the same moves.
|
||||
"FMVXS": {0x70, 0x0, 0x53},
|
||||
"FMVFS": {0x78, 0x0, 0x53},
|
||||
"FMVSX": {0x79, 0x0, 0x53},
|
||||
}
|
||||
|
||||
// riscvCvtType encodes an FP conversion instruction.
|
||||
@@ -371,7 +428,7 @@ var riscvFmaTable = map[string]riscvFmaEnc{
|
||||
// riscvFmaType encodes an R4-type fused multiply-add instruction.
|
||||
func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 {
|
||||
return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) |
|
||||
(uint32(rs1) << 15) | (0x0 << 12) /* rm=dynamic */ | (uint32(rd) << 7) | enc.opcode
|
||||
(uint32(rs1) << 15) | (0x0 << 12) /* rm=RNE */ | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// CSR (Control and Status Register) instructions.
|
||||
@@ -439,13 +496,78 @@ func riscvJType(rd int, offset int32) uint32 {
|
||||
0x6F // JAL opcode
|
||||
}
|
||||
|
||||
// ---- RVV ("V" extension) encoding helpers ----
|
||||
|
||||
// The OP-V major opcode and its funct3 subclasses.
|
||||
const (
|
||||
riscvOpV = 0x57 // the vector operation opcode (also OPcfg for vset*)
|
||||
// funct3 values: 0 OPIVV, 1 OPFVV, 2 OPMVV, 3 OPIVI, 4 OPIVX,
|
||||
// 5 OPFVF, 6 OPMVX, 7 vsetvli.
|
||||
riscvVf3VV = 0x0 // vector-vector
|
||||
riscvVf3MV = 0x2 // vector mask
|
||||
riscvVf3VI = 0x3 // vector-immediate
|
||||
riscvVf3VX = 0x4 // vector-scalar
|
||||
riscvVf3Cfg = 0x7 // vsetvli
|
||||
)
|
||||
|
||||
// riscvVType composes the vsetvli/vsetivli vtype immediate: the register
|
||||
// group multiplier in [2:0], the selected element width in [5:3] and the
|
||||
// tail-agnostic and mask-agnostic policies in bits 6 and 7.
|
||||
func riscvVType(vsew, vlmul, vta, vma int) int {
|
||||
return vlmul | vsew<<3 | vta<<6 | vma<<7
|
||||
}
|
||||
|
||||
// riscvVSetEnc encodes VSETVLI and VSETIVLI: imm[31:20] = vtype, rs1 = the
|
||||
// avl register or 5-bit uimm, rd = the destination. Both carry funct3 7; a
|
||||
// vsetivli is distinguished by bits [31:30] set in the immediate (the 0xC00
|
||||
// the toolchain writes above its 10-bit vtype).
|
||||
func riscvVSetEnc(vsetivli bool, avl, vtype, rd int) uint32 {
|
||||
imm := vtype & 0x3FF
|
||||
if vsetivli {
|
||||
imm |= 0xC00
|
||||
}
|
||||
return uint32(imm)<<20 | uint32(avl&0x1F)<<15 | uint32(riscvVf3Cfg)<<12 |
|
||||
uint32(rd)<<7 | riscvOpV
|
||||
}
|
||||
|
||||
// riscvVLSType encodes a vector load or store: the full 32-bit word with the
|
||||
// segment count in bits [31:29], the addressing mode in bits [28:26], the
|
||||
// unmasked bit at 25 and the width in funct3. width follows the load
|
||||
// convention (0 = 8-bit, 5 = 16-bit, 6 = 32-bit, 7 = 64-bit).
|
||||
func riscvVLSType(op uint32, nf, mop, width int, rs2 int32, rs1, rd int) uint32 {
|
||||
return uint32(nf&0x7)<<29 | uint32(mop&0x7)<<26 | 1<<25 |
|
||||
uint32(rs2)<<20 | uint32(rs1)<<15 | uint32(width&0x7)<<12 |
|
||||
uint32(rd)<<7 | op
|
||||
}
|
||||
|
||||
// riscvVVInstr encodes an OP-V instruction with the six-bit operation code in
|
||||
// funct7's upper bits, bit 25 as the unmasked flag and the three registers in
|
||||
// the standard positions. vs1 may name an integer register for the *VX forms
|
||||
// (the scalar sits in the rs1 field) or an immediate for the *VI forms.
|
||||
func riscvVVInstr(funct6, funct3 int, vs1 int32, vs2, vd int) uint32 {
|
||||
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs1)<<15 |
|
||||
uint32(funct3)<<12 | uint32(vs2)<<20 | uint32(vd)<<7 | riscvOpV
|
||||
}
|
||||
|
||||
// riscvVUnaryInstr encodes a one-vector-operand OP-V instruction whose fixed
|
||||
// fields live where the second source register would be: rs1Field and vs2 are
|
||||
// written verbatim (the oracle writes fixed non-zero constants there for some
|
||||
// instructions, such as 0x11 in the rs1 field of vmfirst.m and vid.v).
|
||||
func riscvVUnaryInstr(funct6, funct3 int, rs1Field int32, vs2, vd int) uint32 {
|
||||
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs2&0x1F)<<20 |
|
||||
uint32(rs1Field&0x1F)<<15 | uint32(funct3&0x7)<<12 | uint32(vd&0x1F)<<7 | riscvOpV
|
||||
}
|
||||
|
||||
// riscvSegNF maps a segment count to the 3-bit nf field (count - 1).
|
||||
func riscvSegNF(n int) int32 { return int32(n - 1) }
|
||||
|
||||
// ---- RVC (compressed) encoding helpers ----
|
||||
|
||||
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
|
||||
// prime register field used by compressed instructions (x8–x15).
|
||||
// prime register field used by compressed instructions (x8-x15).
|
||||
func isRVCIntReg(r int) bool { return r >= 8 && r <= 15 }
|
||||
|
||||
// rvcReg3 returns the 3-bit encoding for registers x8–x15 (0–7).
|
||||
// rvcReg3 returns the 3-bit encoding for registers x8-x15 (0-7).
|
||||
func rvcReg3(r int) uint32 { return uint32(r - 8) }
|
||||
|
||||
// rvcCR encodes a CR-type (register) compressed instruction.
|
||||
@@ -455,7 +577,7 @@ func rvcCR(funct4, rd, rs2 uint32) uint16 {
|
||||
}
|
||||
|
||||
// rvcCI encodes a CI-type (immediate) compressed instruction.
|
||||
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW — linear 6-bit immediate.
|
||||
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW, linear 6-bit immediate.
|
||||
func rvcCI(funct3, rd uint32, imm uint32) uint16 {
|
||||
return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x1)
|
||||
}
|
||||
@@ -520,11 +642,13 @@ func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 {
|
||||
|
||||
// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
|
||||
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
|
||||
// offset; the immediate bits are extracted per the RISC-V CS format.
|
||||
// offset; the immediate bits are extracted per the RISC-V CS format, with the
|
||||
// same five-bit patterns as the load side ({5,4,3,7,6} and {5,4,3,2,6},
|
||||
// matching the toolchain's encodeCS).
|
||||
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
|
||||
pattern := []int{5, 3, 7, 6}
|
||||
pattern := []int{5, 4, 3, 7, 6}
|
||||
if funct3 == 0x6 {
|
||||
pattern = []int{5, 3, 2, 6}
|
||||
pattern = []int{5, 4, 3, 2, 6}
|
||||
}
|
||||
packed := encodeRVCPattern(imm, pattern)
|
||||
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
|
||||
|
||||
+591
-12
@@ -5,10 +5,13 @@ package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"encoding/hex"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// firstTextRISCV parses assembly source and returns the first TEXT function body.
|
||||
@@ -30,7 +33,7 @@ func firstTextRISCV(t *testing.T, src string) *ast.Text {
|
||||
// assembleRISCVHelper assembles one TEXT function and returns its code bytes.
|
||||
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
|
||||
t.Helper()
|
||||
code, _, _, _, _, err := assembleRISCV(fn)
|
||||
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
@@ -289,7 +292,7 @@ TEXT ·cmp(SB), NOSPLIT, $0
|
||||
}
|
||||
|
||||
func TestRISCV_forwardBranch(t *testing.T) {
|
||||
// Forward label reference — must not fail.
|
||||
// Forward label reference; must not fail.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·fwd(SB), NOSPLIT, $0
|
||||
ADDI $1, X10, X10
|
||||
@@ -691,6 +694,81 @@ DATA answer<>+0(SB)/8, $42
|
||||
}
|
||||
}
|
||||
|
||||
// TestRISCV_RVC_StorePatterns pins the register-relative compressed store
|
||||
// encodings for offsets with immediate bits 4 and 5 set, byte-identical to
|
||||
// the toolchain's encodeCS (patterns {5,4,3,7,6} and {5,4,3,2,6}).
|
||||
// Regression: the store-side patterns dropped imm[4], so every such store
|
||||
// silently encoded the wrong address while the loads stayed correct.
|
||||
func TestRISCV_RVC_StorePatterns(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·csstores(SB), NOSPLIT, $0
|
||||
SD X9, 24(X8)
|
||||
SW X10, 16(X11)
|
||||
FSD F8, 40(X12)
|
||||
LD 24(X8), X9
|
||||
LW 16(X11), X10
|
||||
FLD 40(X12), F8
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
want := []byte{
|
||||
0x04, 0xec, // c.sd x9, 24(x8)
|
||||
0x88, 0xc9, // c.sw x10, 16(x11)
|
||||
0x00, 0xb6, // c.fsd f8, 40(x12)
|
||||
0x04, 0x6c, // c.ld x9, 24(x8)
|
||||
0x88, 0x49, // c.lw x10, 16(x11)
|
||||
0x00, 0x36, // c.fld f8, 40(x12)
|
||||
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
|
||||
}
|
||||
if !bytes.Equal(code, want) {
|
||||
t.Errorf("code = % x\nwant % x", code, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRISCV_FENCE pins the FENCE encoding: the toolchain expands the bare
|
||||
// mnemonic to fence iorw, iorw (0x0FF0000F), not fence 0,0.
|
||||
func TestRISCV_FENCE(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·fence(SB), NOSPLIT, $0
|
||||
FENCE
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
want := []byte{
|
||||
0x0f, 0x00, 0xf0, 0x0f, // fence iorw, iorw
|
||||
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
|
||||
}
|
||||
if !bytes.Equal(code, want) {
|
||||
t.Errorf("code = % x\nwant % x", code, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRISCV_RVC_WidthSpellings pins the compression of the GOROOT width
|
||||
// spellings: MOVW and MOVD lower to their base load/store and compress
|
||||
// exactly like LW/SW/FLD/FSD would (the toolchain compresses these shapes;
|
||||
// before the normalisation they stayed 4 bytes).
|
||||
func TestRISCV_RVC_WidthSpellings(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·widths(SB), NOSPLIT, $0-16
|
||||
MOVW w+0(FP), X9
|
||||
MOVW X9, v+4(FP)
|
||||
MOVD d+0(FP), F8
|
||||
MOVD F8, r+8(FP)
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
want := []byte{
|
||||
0xa2, 0x44, // c.lwsp x9, 8
|
||||
0x26, 0xc6, // c.swsp x9, 12
|
||||
0x22, 0x24, // c.fldsp f8, 8
|
||||
0x22, 0xa8, // c.fsdsp f8, 16
|
||||
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
|
||||
}
|
||||
if !bytes.Equal(code, want) {
|
||||
t.Errorf("code = % x\nwant % x", code, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_system_instrs(t *testing.T) {
|
||||
// Test FENCE, ECALL, EBREAK encoding.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
@@ -707,16 +785,140 @@ TEXT ·sys(SB), NOSPLIT, $0
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_MOV_sym_FP_error(t *testing.T) {
|
||||
// MOV $sym(FP), rd should return an error (unsupported).
|
||||
func TestRISCV_MOV_sym_FP(t *testing.T) {
|
||||
// MOV $sym(FP), rd lowers to the frame-adjusted ADDI against SP: the
|
||||
// toolchain's argframe spelling. A zero frame leaves the offset at the
|
||||
// 8-byte link slot, compressed to C.ADDI4SPN.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·badfp(SB), NOSPLIT, $0
|
||||
TEXT ·argfp(SB), NOSPLIT, $0
|
||||
MOV $arg(FP), X10
|
||||
RET
|
||||
`)
|
||||
_, _, _, _, _, err := assembleRISCV(fn)
|
||||
if err == nil {
|
||||
t.Error("expected error for MOV $arg(FP), got nil")
|
||||
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
// prologue (0: leaf, zero frame) + C.ADDI4SPN (2) + RET (4) = 6
|
||||
want := []byte{0x28, 0x00, 0x67, 0x80, 0x00, 0x00}
|
||||
if string(code) != string(want) {
|
||||
t.Errorf("got % x, want % x", code, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_Bookkeeping(t *testing.T) {
|
||||
// FUNCDATA and PCDATA contribute no bytes; UNDEF is the toolchain's
|
||||
// ebreak, compressed to C.EBREAK under RVC.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·book(SB), NOSPLIT, $0-8
|
||||
FUNCDATA $0, marks<>(SB)
|
||||
PCDATA $1, $1
|
||||
UNDEF
|
||||
MOV $1, X10
|
||||
MOV X10, ret+0(FP)
|
||||
RET
|
||||
`)
|
||||
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
// C.EBREAK (2) + C.LI X10, 1 (2) + C.SWSP (2) + RET (4) = 10: the
|
||||
// FUNCDATA and PCDATA statements contribute nothing.
|
||||
want := []byte{0x02, 0x90, 0x05, 0x45, 0x2a, 0xe4, 0x67, 0x80, 0x00, 0x00}
|
||||
if string(code) != string(want) {
|
||||
t.Errorf("got % x, want % x", code, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_JMPPCRel(t *testing.T) {
|
||||
// JMP N(PC): the displacement tracks the instruction N source slots
|
||||
// away in the final layout (0 the jump itself, negative backwards).
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·slots(SB), NOSPLIT, $0-0
|
||||
JMP 2(PC)
|
||||
MOV $1, X11
|
||||
MOV $2, X12
|
||||
MOV X12, X11
|
||||
JMP -3(PC)
|
||||
RET
|
||||
`)
|
||||
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
// JMP 2(PC) lands on the C.MV six bytes ahead; JMP -3(PC) lands back on
|
||||
// the first C.LI, six bytes behind.
|
||||
want := []byte{
|
||||
0x6f, 0x00, 0x60, 0x00, // JAL X0, 6
|
||||
0x85, 0x45, // C.LI X11, 1
|
||||
0x09, 0x46, // C.LI X12, 2
|
||||
0xb2, 0x85, // C.MV X11, X12
|
||||
0x6f, 0xf0, 0xbf, 0xff, // JAL X0, -6
|
||||
0x67, 0x80, 0x00, 0x00, // RET
|
||||
}
|
||||
if string(code) != string(want) {
|
||||
t.Errorf("got % x, want % x", code, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_MOVWideImm(t *testing.T) {
|
||||
// Shift-sequence constants compress like the toolchain's expansion.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·wide(SB), NOSPLIT, $0-0
|
||||
MOV $0x8000000000000000, X5
|
||||
MOV $0x100000000, X5
|
||||
MOV $0x000fffffffffffda, X5
|
||||
RET
|
||||
`)
|
||||
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
// C.LI -1, C.SLLI 63; C.LI 1, C.SLLI 32; C.LI -19, C.SLLI 13, SRLI 12.
|
||||
want := []byte{
|
||||
0xfd, 0x52, 0xfe, 0x12,
|
||||
0x85, 0x42, 0x82, 0x12,
|
||||
0xb5, 0x52, 0xb6, 0x02, 0x93, 0xd2, 0xc2, 0x00,
|
||||
0x67, 0x80, 0x00, 0x00,
|
||||
}
|
||||
if string(code) != string(want) {
|
||||
t.Errorf("got % x, want % x", code, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_MOVImmPool(t *testing.T) {
|
||||
// A constant outside the shift shapes loads from the pooled $i64 data
|
||||
// symbol via AUIPC+LD, named like the toolchain's pool.
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·pool(SB), NOSPLIT, $0-8
|
||||
MOV $0x0101010101010101, X16
|
||||
MOV X16, ret+0(FP)
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("pool_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
// AUIPC X16, 0 + LD X16, 0(X16): the relocation pair carries the symbol.
|
||||
wantCode := []byte{0x17, 0x08, 0x00, 0x00, 0x03, 0x38, 0x08, 0x00}
|
||||
if string(img.Code[0:8]) != string(wantCode) {
|
||||
t.Errorf("pool load: got % x", img.Code[0:8])
|
||||
}
|
||||
var lit *DataSymbol
|
||||
for i := range img.DataSyms {
|
||||
if img.DataSyms[i].Name == "$i64.0101010101010101" {
|
||||
lit = &img.DataSyms[i]
|
||||
}
|
||||
}
|
||||
if lit == nil {
|
||||
t.Fatalf("pool symbol missing: %v", img.DataSyms)
|
||||
}
|
||||
wantData := []byte{0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01}
|
||||
if string(img.Data[lit.Offset:lit.Offset+8]) != string(wantData) {
|
||||
t.Errorf("pool bytes: got % x", img.Data[lit.Offset:lit.Offset+8])
|
||||
}
|
||||
}
|
||||
|
||||
@@ -727,7 +929,7 @@ TEXT ·calltest(SB), NOSPLIT, $0
|
||||
CALL ext(SB)
|
||||
RET
|
||||
`)
|
||||
code, _, relocs, _, _, err := assembleRISCV(fn)
|
||||
code, _, relocs, _, _, _, err := assembleRISCV(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
@@ -756,8 +958,385 @@ TEXT ·calllocal(SB), NOSPLIT, $0
|
||||
sub:
|
||||
RET
|
||||
`)
|
||||
_, _, _, _, _, err := assembleRISCV(fn)
|
||||
_, _, _, _, _, _, err := assembleRISCV(fn)
|
||||
if err == nil {
|
||||
t.Error("expected error for CALL to local label, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
// TestRISCVIndirectBranch pins the indirect branch encodings: JMP (X5) is the
|
||||
// toolchain's JALR X0, 0(X5), and the trampoline form JALR rd, offset(rs1)
|
||||
// takes its destination from the first operand (regression: the base
|
||||
// register was once read as the destination, silently jumping to X0).
|
||||
func TestRISCVIndirectBranch(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
JMP (X5)
|
||||
JALR X0, 0(X6)
|
||||
JALR X28, 0(X9)
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x00028067, // jalr x0, 5(x0), 0
|
||||
0x00030067, // jalr x0, 6(x0), 0
|
||||
0x00048e67, // jalr x28, 9(x0), 0
|
||||
0x00008067, // jalr x0, 1(x0), 0 (RET)
|
||||
)
|
||||
}
|
||||
|
||||
// encodeOneInstrRISCV encodes a single parsed instruction against a synthetic
|
||||
// offsets map, the smallest honest harness for the branch-range diagnostics:
|
||||
// the spans are far larger than any source a test would want to spell out.
|
||||
func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]int) ([]byte, error) {
|
||||
t.Helper()
|
||||
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
|
||||
instr := fn.Body[0].(*ast.Instr)
|
||||
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil, nil, nil)
|
||||
}
|
||||
|
||||
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span
|
||||
// [-4096, 4094] and the J-type span [-1048576, 1048574] are diagnosed instead
|
||||
// of wrapping silently to a wrong target.
|
||||
func TestRISCVBranchJumpRange(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
src string
|
||||
off int // the target's function-relative offset (pc 0)
|
||||
ok bool
|
||||
}{
|
||||
{"branch max", "BEQ X10, X11, tgt\nRET\n", 4094, true},
|
||||
{"branch past max", "BEQ X10, X11, tgt\nRET\n", 4096, false},
|
||||
{"branch back max", "BEQ X10, X11, tgt\nRET\n", -4096, true},
|
||||
{"branch back past max", "BEQ X10, X11, tgt\nRET\n", -4098, false},
|
||||
{"branchz past max", "BEQZ X10, tgt\nRET\n", 4096, false},
|
||||
{"jump max", "JMP tgt\nRET\n", 1048574, true},
|
||||
{"jump past max", "JMP tgt\nRET\n", 1048576, false},
|
||||
{"jump back max", "JMP tgt\nRET\n", -1048576, true},
|
||||
{"jump back past max", "JMP tgt\nRET\n", -1048578, false},
|
||||
{"jal past max", "JAL tgt\nRET\n", 1048576, false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
_, err := encodeOneInstrRISCV(t, "TEXT ·f(SB), NOSPLIT, $0\n\t"+c.src, 0, map[string]int{"tgt": c.off})
|
||||
if c.ok && err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
if !c.ok && err == nil {
|
||||
t.Fatal("expected an out-of-range diagnostic, got none")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestRISCVBranchFarBody drives the relaxation pass through the full
|
||||
// assembler: a forward branch over a body larger than the B-type span is
|
||||
// rewritten as an inverted branch over an inserted JMP, the same layout the
|
||||
// toolchain produces, instead of wrapping to a wrong target.
|
||||
func TestRISCVBranchFarBody(t *testing.T) {
|
||||
var sb strings.Builder
|
||||
sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n")
|
||||
for range 1100 {
|
||||
sb.WriteString("\tADD X10, X11, X12\n")
|
||||
}
|
||||
sb.WriteString("done:\n\tRET\n")
|
||||
fn := firstTextRISCV(t, sb.String())
|
||||
out, _, _, _, _, _, err := assembleRISCV(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
// The relaxed branch at offset 0 targets the inserted JMP at 4 (bne
|
||||
// x10, x11, +4); the JMP at 4 carries the far forward displacement.
|
||||
wantBranch := wordLE(riscvBType(riscvEnc{0x63, 0x1, 0x00}, 10, 11, 4))
|
||||
if !bytes.Equal(out[0:4], wantBranch) {
|
||||
t.Errorf("relaxed branch = %x, want %x", out[0:4], wantBranch)
|
||||
}
|
||||
// done sits after 1100 ADDs: 4 + 4400, i.e. offset 4404 from the JMP at 4.
|
||||
wantJmp := wordLE(riscvJType(0, 4404))
|
||||
if !bytes.Equal(out[4:8], wantJmp) {
|
||||
t.Errorf("inserted JMP = %x, want %x", out[4:8], wantJmp)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRISCV_CSRRange checks the CSR address range: the 12-bit field is
|
||||
// diagnosed rather than masked, so CSRRW $4096 does not silently address
|
||||
// CSR 0.
|
||||
func TestRISCV_CSRRange(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·csrhi(SB), NOSPLIT, $0
|
||||
CSRRW $4096, X10, X11
|
||||
RET
|
||||
`)
|
||||
if _, _, _, _, _, _, err := assembleRISCV(fn); err == nil {
|
||||
t.Error("expected an out-of-range error for CSR $4096, got none")
|
||||
}
|
||||
fn = firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·csrmax(SB), NOSPLIT, $0
|
||||
CSRRW $4095, X10, X11
|
||||
RET
|
||||
`)
|
||||
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
|
||||
t.Errorf("CSR $4095 must assemble: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRISCV_Imm64Rejected checks that immediates outside the signed 32-bit
|
||||
// span are diagnosed instead of silently truncated to their low 32 bits for
|
||||
// the I-type arithmetic; the MOV forms materialise the wide constant instead
|
||||
// (shift sequence or pooled load), like the toolchain.
|
||||
func TestRISCV_Imm64Rejected(t *testing.T) {
|
||||
cases := []string{
|
||||
"ADDI $0x100000000, X10, X11",
|
||||
"ANDI $-0x800000001, X10, X11",
|
||||
"SUB $0x100000000, X10, X11",
|
||||
}
|
||||
for _, src := range cases {
|
||||
fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·wide(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
|
||||
if _, _, _, _, _, _, err := assembleRISCV(fn); err == nil {
|
||||
t.Errorf("%s: expected an out-of-range error, got none", src)
|
||||
}
|
||||
}
|
||||
// The full signed 32-bit span still assembles, including the SUB form
|
||||
// whose negated immediate only just fits.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·edge(SB), NOSPLIT, $0
|
||||
MOV $2147483647, X10
|
||||
MOV $-2147483648, X11
|
||||
SUB $0x80000000, X12, X13
|
||||
RET
|
||||
`)
|
||||
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
|
||||
t.Errorf("int32-span immediates must assemble: %v", err)
|
||||
}
|
||||
// Beyond the span the MOV forms materialise the constant like the
|
||||
// toolchain instead of diagnosing it.
|
||||
fn = firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·pool(SB), NOSPLIT, $0
|
||||
MOV $0x123456789, X10
|
||||
RET
|
||||
`)
|
||||
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
|
||||
t.Errorf("MOV with a 64-bit immediate must assemble: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// riscvWants decodes code as little-endian words and pins each one; the
|
||||
// expected values below were read off GOARCH=riscv64 go tool objdump of
|
||||
// kernels assembled with go tool asm (the toolchain's riscv64.s testdata
|
||||
// cross-checks the same words).
|
||||
func riscvWants(t *testing.T, code []byte, want ...uint32) {
|
||||
t.Helper()
|
||||
got := make([]uint32, 0, len(code)/4)
|
||||
for i := 0; i+4 <= len(code); i += 4 {
|
||||
got = append(got, binary.LittleEndian.Uint32(code[i:]))
|
||||
}
|
||||
if len(got) < len(want) {
|
||||
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
|
||||
}
|
||||
// The RET (JALR) ends the sequence; only the pinned prefix is compared.
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// riscvWantsHex pins the exact hex encoding of a function's instruction
|
||||
// bytes, including any 2-byte compressed instructions in the stream; the
|
||||
// expected strings were read off GOARCH=riscv64 go tool objdump of kernels
|
||||
// assembled with go tool asm (the toolchain's riscv64.s testdata
|
||||
// cross-checks the same words).
|
||||
func riscvWantsHex(t *testing.T, code []byte, wantHex string) {
|
||||
t.Helper()
|
||||
got := hex.EncodeToString(code)
|
||||
if got != wantHex {
|
||||
t.Errorf("code = %s, want %s", got, wantHex)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRISCV_extendedPseudos pins the toolchain-synthesised instructions:
|
||||
// ANDN/ORN (XORI + AND/OR through the destination or TMP), the five-word
|
||||
// MIN/MAX expansion, the four-word rotate, ROR's compressed reverse shift
|
||||
// (C.SLLI when rd == rs1, both non-zero, 1 <= sll <= 63), the identical-
|
||||
// input MIN/MAX fold to C.MV, FABSD (FSGNJX.D), SEQZ and RDTIME (csrrs with
|
||||
// the time CSR).
|
||||
func TestRISCV_extendedPseudos(t *testing.T) {
|
||||
t.Run("logic and minmax", func(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·l(SB), NOSPLIT, $0
|
||||
ANDN X19, X20, X21
|
||||
ANDN X19, X20
|
||||
ORN X20, X19
|
||||
MAX X26, X28, X29
|
||||
MIN X29, X30, X5
|
||||
MAX X5, X5
|
||||
MAX X5, X5, X6
|
||||
SEQZ X5, X6
|
||||
NEG X5, X6
|
||||
NOT X5
|
||||
RDTIME X5
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// Words 0-10 up to the folded C.MV pair (halfwords 96 82 and 16 83),
|
||||
// then SEQZ, NEG, NOT and RDTIME.
|
||||
riscvWantsHex(t, code,
|
||||
"93caf9ffb37a5a01"+"93cff9ff337afa01"+"934ffaffb3e9f901"+
|
||||
"b32fae01b30ff041b34eae01b3fedf01b34ede01"+
|
||||
"b3afee01b30ff041b342df01b3f25f00b3425f00"+
|
||||
"9682"+"1683"+
|
||||
"13b31200"+"33035040"+"93c2f2ff"+"f32210c0"+"67800000")
|
||||
})
|
||||
|
||||
t.Run("rotate", func(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·r(SB), NOSPLIT, $0
|
||||
ROR X10, X11, X12
|
||||
ROR X10, X11
|
||||
ROR $63, X11
|
||||
RORIW $31, X13, X14
|
||||
RORIW $1, X14, X15
|
||||
RORIW $3, X14
|
||||
RORW X15, X16, X17
|
||||
RORW $31, X13
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// The third ROR carries the compressed C.SLLI (05 86) in mid-stream.
|
||||
riscvWantsHex(t, code,
|
||||
"b30fa040b39ff50133d6a50033e6cf00"+
|
||||
"b30fa040b39ff501b3d5a500b3e5bf00"+
|
||||
"93dff5038605b3e5bf00"+
|
||||
"9bdff6011b97160033e7ef00"+
|
||||
"9b5f17009b17f701b3e7ff00"+
|
||||
"9b5f37001b17d70133e7ef00"+
|
||||
"b30ff040bb1ff801bb58f800b3e81f01"+
|
||||
"9bdff6019b961600b3e6df00"+"67800000")
|
||||
})
|
||||
|
||||
t.Run("fp and branches", func(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
FABSD F1, F2
|
||||
FSGNJD F1, F0, F2
|
||||
FMADDD F1, F2, F3, F4
|
||||
FMSUBD F1, F2, F3, F4
|
||||
FNMSUBD F1, F2, F3, F4
|
||||
BGT X5, X6, tgt
|
||||
BLE X5, X6, tgt
|
||||
BGTU X5, X6, tgt
|
||||
BLEU X5, X6, tgt
|
||||
tgt:
|
||||
RDTIME X5
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
riscvWantsHex(t, code,
|
||||
"53a11022"+"53011022"+"4382201a4782201a4b82201a"+
|
||||
"63485300635653006364530063725300"+ // blt/bge/bltu/bgeu x6, x5
|
||||
"f32210c0"+"67800000")
|
||||
})
|
||||
}
|
||||
|
||||
// TestRISCV_amoWords pins the full AMO family: every AMO carries aq and rl
|
||||
// (funct7 |= 3), LR is acquire (funct7 |= 2) and SC release (funct7 |= 1),
|
||||
// exactly as GOARCH=riscv64 go tool asm encodes them.
|
||||
func TestRISCV_amoWords(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·amo(SB), NOSPLIT, $0
|
||||
AMOSWAPW X5, (X6), X7
|
||||
AMOSWAPD X5, (X6), X7
|
||||
AMOADDW X5, (X6), X7
|
||||
AMOADDD X5, (X6), X7
|
||||
AMOANDW X5, (X6), X7
|
||||
AMOANDD X5, (X6), X7
|
||||
AMOORW X5, (X6), X7
|
||||
AMOORD X5, (X6), X7
|
||||
AMOXORW X5, (X6), X7
|
||||
AMOXORD X5, (X6), X7
|
||||
AMOMAXW X5, (X6), X7
|
||||
AMOMAXD X5, (X6), X7
|
||||
AMOMAXUW X5, (X6), X7
|
||||
AMOMAXUD X5, (X6), X7
|
||||
AMOMINUW X5, (X6), X7
|
||||
AMOMINUD X5, (X6), X7
|
||||
LRW (X5), X6
|
||||
LRD (X5), X6
|
||||
SCW X5, (X6), X7
|
||||
SCD X5, (X6), X7
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
riscvWants(t, code,
|
||||
0x0E5323AF, // amoswap.w
|
||||
0x0E5333AF, // amoswap.d
|
||||
0x065323AF, // amoaddd.w
|
||||
0x065333AF, // amoadd.d
|
||||
0x665323AF, // amoand.w
|
||||
0x665333AF, // amoand.d
|
||||
0x465323AF, // amoor.w
|
||||
0x465333AF, // amoor.d
|
||||
0x265323AF, // amoxor.w
|
||||
0x265333AF, // amoxor.d
|
||||
0xA65323AF, // amomax.w
|
||||
0xA65333AF, // amomax.d
|
||||
0xE65323AF, // amomaxu.w
|
||||
0xE65333AF, // amomaxu.d
|
||||
0xC65323AF, // amominu.w
|
||||
0xC65333AF, // amominu.d
|
||||
0x1402A32F, // lr.w (aq)
|
||||
0x1402B32F, // lr.d
|
||||
0x1A5323AF, // sc.w (rl)
|
||||
0x1A5333AF, // sc.d
|
||||
)
|
||||
}
|
||||
|
||||
// TestRISCV_vectorWords pins the RVV slice and the VSET* encodings. The
|
||||
// toolchain canonicalises an immediate avl to vsetivli even under the
|
||||
// VSETVLI spelling (`VSETVLI $15` and `VSETIVLI $15` come out byte-
|
||||
// identical), which is what the 0xC00 bit of the first word carries.
|
||||
func TestRISCV_vectorWords(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·v(SB), NOSPLIT, $0
|
||||
VSETVLI X5, E8, M8, TA, MA, X6
|
||||
VSETIVLI $4, E32, M1, TA, MA, X0
|
||||
VSETVLI $15, E32, M1, TA, MA, X12
|
||||
VADDVV V1, V2, V3
|
||||
VADDVX X12, V12, V12
|
||||
VXORVV V8, V16, V24
|
||||
VMSEQVX X12, V8, V0
|
||||
VMSNEVV V8, V16, V0
|
||||
VSLLVI $8, V28, V30
|
||||
VSRLVI $25, V29, V29
|
||||
VFIRSTM V0, X6
|
||||
VIDV V12
|
||||
VMV4RV V8, V24
|
||||
VLE8V (X10), V8
|
||||
VSE8V V24, (X10)
|
||||
VSE32V V9, (X11)
|
||||
VLSSEG4E32V (X14), X0, V0
|
||||
VLSSEG8E32V (X10), X0, V4
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
riscvWants(t, code,
|
||||
0x0C32F357, // vsetvli x6, x5, vtype 0xc3 (E8, M8, TA, MA)
|
||||
0xCD027057, // vsetivli x0, 4
|
||||
0xCD07F657, // vsetivli x12, 15: VSETVLI $15 canonicalises to the same word
|
||||
0x022081D7, // vadd.vv v3, v2, v1
|
||||
0x02C64657, // vadd.vx v12, v12, x12
|
||||
0x2F040C57, // vxor.vv v24, v16, v8
|
||||
0x62864057, // vmseq.vx v0, v8, x12
|
||||
0x67040057, // vmsne.vv v0, v16, v8
|
||||
0x97C43F57, // vsll.vi v30, v28, 8
|
||||
0xA3DCBED7, // vsrl.vi v29, v29, 25
|
||||
0x4208A357, // vmfirst.m x6, v0
|
||||
0x5208A657, // vid.v v12
|
||||
0x9E81BC57, // vmv4r.v v24, v8
|
||||
0x02050407, // vle8.v v8, (x10)
|
||||
0x02050C27, // vse8.v v24, (x10)
|
||||
0x0205E4A7, // vse32.v v9, (x11)
|
||||
0x6A076007, // vlsseg4e32.v v0, (x14), x0
|
||||
0xEA056207, // vlsseg8e32.v v4, (x10), x0
|
||||
)
|
||||
}
|
||||
|
||||
+212
-30
@@ -4,9 +4,10 @@
|
||||
package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||
)
|
||||
|
||||
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
|
||||
@@ -32,6 +33,13 @@ import (
|
||||
// riscvFrameInfo holds the frame layout derived from a TEXT directive.
|
||||
type riscvFrameInfo struct {
|
||||
autosize int // the real SP adjustment (locals + saved LR)
|
||||
|
||||
// Stack-split guard state: the toolchain emits the check for every
|
||||
// non-NOSPLIT function whose autosize is nonzero (a zero autosize is
|
||||
// "effectively NOSPLIT"); unlike amd64 and arm64 there is no leaf
|
||||
// auto-NOSPLIT.
|
||||
needSplit bool
|
||||
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
|
||||
}
|
||||
|
||||
// riscvComputeFrame derives the frame layout for a TEXT function.
|
||||
@@ -40,11 +48,34 @@ func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
|
||||
if frame != 0 || !riscvIsLeaf(t) {
|
||||
// FixedFrameSize = 8: space for the saved link register. A
|
||||
// zero-frame non-leaf function still opens an 8-byte frame for LR.
|
||||
return riscvFrameInfo{autosize: frame + 8}
|
||||
autosize := frame + 8
|
||||
fi := riscvFrameInfo{autosize: autosize}
|
||||
if !hasNoSplitFlag(t) {
|
||||
fi.needSplit = true
|
||||
switch {
|
||||
case autosize <= stackSmall:
|
||||
fi.splitClass = 0
|
||||
case autosize <= stackBig:
|
||||
fi.splitClass = 1
|
||||
default:
|
||||
fi.splitClass = 2
|
||||
}
|
||||
}
|
||||
return fi
|
||||
}
|
||||
return riscvFrameInfo{}
|
||||
}
|
||||
|
||||
// hasNoSplitFlag reports whether the TEXT directive carries NOSPLIT.
|
||||
func hasNoSplitFlag(t *ast.Text) bool {
|
||||
for _, f := range t.Flags {
|
||||
if strings.EqualFold(f, "NOSPLIT") {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// riscvIsLeaf reports whether a function contains no call instructions.
|
||||
// CALL always links; JAL/JALR link only when their destination register is
|
||||
// the link register (X1), matching cmd/internal/obj/riscv's containsCall.
|
||||
@@ -58,17 +89,24 @@ func riscvIsLeaf(t *ast.Text) bool {
|
||||
case "CALL":
|
||||
return false
|
||||
case "JAL":
|
||||
// JAL rd, target — a call only when rd is the link register.
|
||||
// JAL rd, target, a call only when rd is the link register.
|
||||
if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 {
|
||||
return false
|
||||
}
|
||||
case "JALR":
|
||||
// JALR rs1, rd — a call when rd is X1; JALR offset(rs1) always
|
||||
// links to X1.
|
||||
// JALR rd, offset(rs1) links when the destination register (the
|
||||
// first operand) is X1; JALR rs1, rd links when the second
|
||||
// register is X1; JALR offset(rs1) always links to X1.
|
||||
if len(in.Operands) == 1 {
|
||||
return false
|
||||
}
|
||||
if len(in.Operands) >= 2 && regFromOperand(in.Operands[1]) == 1 {
|
||||
if isMemOperand(in.Operands[1]) {
|
||||
if regFromOperand(in.Operands[0]) == 1 {
|
||||
return false
|
||||
}
|
||||
continue
|
||||
}
|
||||
if regFromOperand(in.Operands[1]) == 1 {
|
||||
return false
|
||||
}
|
||||
}
|
||||
@@ -84,28 +122,99 @@ func riscvPrologue(fi riscvFrameInfo) []byte {
|
||||
return nil
|
||||
}
|
||||
var out []byte
|
||||
// MOV LR, -autosize(SP) — SD X1, -autosize(X2). The negative offset is
|
||||
// not compressible to C.SDSP (unsigned), so it stays 4 bytes.
|
||||
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
|
||||
// ADDI $-autosize, SP, SP — open the frame (C.ADDI when it fits).
|
||||
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
|
||||
// MOV LR, 0(SP) — SD X1, 0(X2) → C.SDSP X1, 0.
|
||||
// MOV LR, -autosize(SP), SD X1, -autosize(X2). The negative offset is
|
||||
// not compressible to C.SDSP (unsigned), so it stays 4 bytes. Beyond
|
||||
// the imm12 range the toolchain materialises the address in X31.
|
||||
if fits12(int32(-fi.autosize)) {
|
||||
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
|
||||
} else {
|
||||
out = append(out, riscvAddressInX31(int32(-fi.autosize))...)
|
||||
lo := int32(-fi.autosize) - (splitHi(int32(-fi.autosize)) << 12)
|
||||
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 1, lo))...)
|
||||
}
|
||||
// ADDI $-autosize, SP, SP, open the frame (C.ADDI when it fits; X31
|
||||
// materialisation beyond imm12).
|
||||
if fits12(int32(-fi.autosize)) {
|
||||
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
|
||||
} else {
|
||||
out = append(out, riscvAddToSP(int32(-fi.autosize))...)
|
||||
}
|
||||
// MOV LR, 0(SP), SD X1, 0(X2) → C.SDSP X1, 0.
|
||||
c := rvcSSP(0x7, 1, 0)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
return out
|
||||
}
|
||||
|
||||
func fits12(v int32) bool { return v >= -2048 && v <= 2047 }
|
||||
|
||||
// splitHi returns the LUI half of the hi/lo split of v (what remains is the
|
||||
// sign-extended 12-bit low part).
|
||||
func splitHi(v int32) int32 {
|
||||
_, high := splitRISCV32Imm(v)
|
||||
return high
|
||||
}
|
||||
|
||||
// riscvAddressInX31 materialises hi(v) into X31 against the stack pointer,
|
||||
// matching the toolchain's large-frame addressing: C.LUI (or LUI) X31, hi;
|
||||
// C.ADD (or ADD) X31, SP.
|
||||
func riscvAddressInX31(v int32) []byte {
|
||||
return riscvAddressInX31WithBase(v, 2)
|
||||
}
|
||||
|
||||
// riscvAddressInX31WithBase materialises hi(v) into X31 against an arbitrary
|
||||
// base register: LUI (or C.LUI) X31, hi; C.ADD X31, rs1. The CR rs2 field
|
||||
// carries the full 5-bit register, so the compressed form is always
|
||||
// available.
|
||||
func riscvAddressInX31WithBase(v int32, rs1 int) []byte {
|
||||
hi := splitHi(v)
|
||||
var out []byte
|
||||
if hi >= -32 && hi <= 31 {
|
||||
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
} else {
|
||||
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
|
||||
}
|
||||
c := rvcCR(0x9, 31, uint32(rs1))
|
||||
return append(out, byte(c), byte(c>>8))
|
||||
}
|
||||
|
||||
// riscvAddToSP adds v to SP through X31 for the values imm12 cannot carry:
|
||||
// C.LUI X31, hi; C.ADDIW X31, lo; C.ADD SP, X31 (the toolchain's form).
|
||||
func riscvAddToSP(v int32) []byte {
|
||||
hi := splitHi(v)
|
||||
lo := v - (hi << 12)
|
||||
var out []byte
|
||||
if hi >= -32 && hi <= 31 {
|
||||
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
} else {
|
||||
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
|
||||
}
|
||||
if lo >= -32 && lo <= 31 {
|
||||
c := rvcCI(0x1, 31, uint32(lo)&0x3F)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
} else {
|
||||
out = append(out, wordLE(riscvIType(riscvEnc{0x1b, 0x0, 0x00}, 31, 31, lo))...)
|
||||
}
|
||||
c := rvcCR(0x9, 2, 31)
|
||||
return append(out, byte(c), byte(c>>8))
|
||||
}
|
||||
|
||||
// riscvReturn returns the bytes for a RET: the epilogue (restore LR and
|
||||
// deallocate the frame when present) followed by the uncompressed JALR X0,
|
||||
// 0(X1) the toolchain emits for RET (it never compresses RET to C.JR).
|
||||
func riscvReturn(fi riscvFrameInfo) []byte {
|
||||
var out []byte
|
||||
if fi.autosize != 0 {
|
||||
// MOV 0(SP), LR — LD X1, 0(X2) → C.LDSP X1, 0.
|
||||
// MOV 0(SP), LR, LD X1, 0(X2) → C.LDSP X1, 0.
|
||||
c := rvcLSP(0x3, 1, 0)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
// ADDI $autosize, SP, SP — close the frame (C.ADDI when it fits).
|
||||
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
|
||||
// ADDI $autosize, SP, SP, close the frame (C.ADDI when it fits).
|
||||
if fits12(int32(fi.autosize)) {
|
||||
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
|
||||
} else {
|
||||
out = append(out, riscvAddToSP(int32(fi.autosize))...)
|
||||
}
|
||||
}
|
||||
// JALR X0, 0(X1).
|
||||
return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...)
|
||||
@@ -133,33 +242,36 @@ func riscvFitsCAddi(imm int32) bool {
|
||||
}
|
||||
|
||||
// riscvPrologueSpadjPC returns the function-relative byte offset where the
|
||||
// prologue has finished decrementing SP (the delta becomes autosize).
|
||||
// prologue has finished decrementing SP (the delta becomes autosize). It is
|
||||
// computed from the same expansion functions the prologue emits, so the
|
||||
// large-frame X31 materialisations are counted: C.LUI + C.ADD before the SD,
|
||||
// C.LUI + ADDIW + C.ADD for the SP adjust.
|
||||
func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
|
||||
if fi.autosize == 0 {
|
||||
return 0
|
||||
}
|
||||
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
||||
return 4 + riscvSPAdjustLen(int32(-fi.autosize))
|
||||
adj := int32(-fi.autosize)
|
||||
if fits12(adj) {
|
||||
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
||||
return 4 + len(riscvSPAdjust(adj))
|
||||
}
|
||||
return len(riscvAddressInX31(adj)) + 4 + len(riscvAddToSP(adj))
|
||||
}
|
||||
|
||||
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
||||
// (but not including) the final JALR — the point where SP is restored.
|
||||
// (but not including) the final JALR, the point where SP is restored. The
|
||||
// small frame closes with C.LDSP + ADDI/C.ADDI; the large frame materialises
|
||||
// the adjustment through X31 (C.LUI + ADDIW + C.ADD).
|
||||
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
|
||||
if fi.autosize == 0 {
|
||||
return 0
|
||||
}
|
||||
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
||||
return 2 + riscvSPAdjustLen(int32(fi.autosize))
|
||||
}
|
||||
|
||||
func riscvSPAdjustLen(imm int32) int {
|
||||
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
|
||||
return 2
|
||||
adj := int32(fi.autosize)
|
||||
if fits12(adj) {
|
||||
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
||||
return 2 + len(riscvSPAdjust(adj))
|
||||
}
|
||||
if riscvFitsCAddi(imm) {
|
||||
return 2
|
||||
}
|
||||
return 4
|
||||
return 2 + len(riscvAddToSP(adj))
|
||||
}
|
||||
|
||||
// riscvResolvePseudo translates a pseudo-register memory reference into a
|
||||
@@ -180,3 +292,73 @@ func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32
|
||||
}
|
||||
return -1, 0
|
||||
}
|
||||
|
||||
// riscvGuardLen returns the byte length of the stack-split guard prefix
|
||||
// including the inline morestack call (zero when the function needs no
|
||||
// guard). Unlike amd64 and arm64, the toolchain places the morestack call
|
||||
// between the guard and the body: the guard branches forward over it.
|
||||
func riscvGuardLen(fi riscvFrameInfo) (int, error) {
|
||||
g, _, err := riscvGuard(fi)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return len(g), nil
|
||||
}
|
||||
|
||||
// riscvGuard emits the stack-split guard prefix with the inline morestack
|
||||
// call: the branch skips forward over JAL X5 and JAL X0 straight into the
|
||||
// body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are
|
||||
// relative to the guard itself, which sits at function offset 0.
|
||||
func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc, error) {
|
||||
if !fi.needSplit {
|
||||
return nil, Reloc{}, nil
|
||||
}
|
||||
// MOV 16(g), X6 (g.stackguard0), g = X27.
|
||||
out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16))
|
||||
jalBack := func() []byte {
|
||||
// JAL X0 back to the function start: it sits right after the JAL X5,
|
||||
// so its displacement is minus the current offset.
|
||||
return wordLE(riscvJType(0, int32(-len(out))))
|
||||
}
|
||||
var reloc Reloc
|
||||
switch fi.splitClass {
|
||||
case 0:
|
||||
// BLTU X6, SP, done (+12: over the CALL and the JMP back)
|
||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...)
|
||||
call := len(out)
|
||||
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
|
||||
out = append(out, wordLE(riscvJType(5, 0))...)
|
||||
out = append(out, jalBack()...)
|
||||
case 1:
|
||||
// ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+12)
|
||||
off := int32(fi.autosize - stackSmall)
|
||||
out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...)
|
||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
|
||||
call := len(out)
|
||||
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
|
||||
out = append(out, wordLE(riscvJType(5, 0))...)
|
||||
out = append(out, jalBack()...)
|
||||
default:
|
||||
// MOV $(framesize-StackSmall), X7; BLTU SP, X7, call;
|
||||
// ADD $-(framesize-StackSmall), SP, X7; BLTU X6, X7, call
|
||||
off := int32(fi.autosize - stackSmall)
|
||||
mov := encodeRISCVLoadImm(7, off)
|
||||
out = append(out, mov...)
|
||||
addiLen := riscvItypeImmediateSize("ADDI", -off)
|
||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...)
|
||||
addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off)
|
||||
if err != nil {
|
||||
// The ADDI expansion failed: the SP adjustment this class
|
||||
// depends on is not emittable, and silently dropping it would
|
||||
// corrupt every stack reference in the body.
|
||||
return nil, Reloc{}, fmt.Errorf("stack-split guard: %w", err)
|
||||
}
|
||||
out = append(out, addi...)
|
||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
|
||||
call := len(out)
|
||||
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
|
||||
out = append(out, wordLE(riscvJType(5, 0))...)
|
||||
out = append(out, jalBack()...)
|
||||
}
|
||||
return out, reloc, nil
|
||||
}
|
||||
|
||||
+42
-1
@@ -6,7 +6,7 @@ package asm
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// TestRISCVFrameSpadjAndLines checks that a framed function records its
|
||||
@@ -65,6 +65,47 @@ TEXT ·framed(SB), NOSPLIT, $16-16
|
||||
}
|
||||
}
|
||||
|
||||
// TestRISCVFrameSpadjLargeFrame checks the stack-adjustment boundaries of a
|
||||
// frame past the imm12 range: the prologue materialises the LR-store address
|
||||
// and the SP adjustment through X31 (C.LUI + C.ADD + SD, then C.LUI + ADDIW +
|
||||
// C.ADD), so the SP boundary lands at PC 16, and the RET closes with
|
||||
// C.LDSP plus the same X31 adjustment, 10 bytes. Regression: both helpers
|
||||
// assumed the small-frame prologue and reported 8 and 6.
|
||||
func TestRISCVFrameSpadjLargeFrame(t *testing.T) {
|
||||
f, errs := parser.Parse("bigframe_riscv64.s", `#include "textflag.h"
|
||||
|
||||
TEXT ·big(SB), NOSPLIT, $9000-8
|
||||
MOV a+0(FP), X10
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
|
||||
// autosize = 9008. Prologue: C.LUI X31 + C.ADD X31,SP (4) + SD (4) +
|
||||
// C.LUI X31 + ADDIW X31 + C.ADD SP,X31 (8) = 16 bytes to the SP boundary;
|
||||
// C.SDSP X1 (2) follows, so the body starts at 18.
|
||||
wantSpadj := []SpadjStep{{PC: 16, Value: 9008}, {PC: 36, Value: 0}}
|
||||
if len(fn.Spadj) != len(wantSpadj) {
|
||||
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
|
||||
}
|
||||
for i := range wantSpadj {
|
||||
if fn.Spadj[i] != wantSpadj[i] {
|
||||
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
|
||||
}
|
||||
}
|
||||
// The FP load materialises its 9016-byte offset through X31 as well
|
||||
// (8 bytes), then RET's epilogue (C.LDSP + X31 adjust = 10) plus JALR.
|
||||
if fn.Size != 18+8+14 {
|
||||
t.Errorf("size = %d, want %d", fn.Size, 18+8+14)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRISCVRegAliases checks the Go ABI register aliases that the toolchain
|
||||
// defines: LR is the link register (X1) and TMP is the assembler scratch
|
||||
// register (X31/T6).
|
||||
|
||||
+87
-4
@@ -13,7 +13,7 @@ import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
|
||||
@@ -74,6 +74,89 @@ DATA callee<>+0(SB)/8, $42
|
||||
t.Error("ELF object missing R_RISCV_JAL relocation")
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFRISCVPCRELLO12Anchor checks the psABI's LO12 pairing rule: the
|
||||
// R_RISCV_PCREL_LO12_I/S relocation must reference a symbol whose value is
|
||||
// the AUIPC site of its HI20 partner (psABI §8.4.9; cmd/link generates one
|
||||
// local text symbol per AUIPC for exactly this). The emitter pairs each
|
||||
// HI20 (against the target symbol) with a LO12 against the .text section
|
||||
// symbol whose addend is the AUIPC's section-relative offset, so S + A is
|
||||
// the AUIPC address.
|
||||
func TestELFRISCVPCRELLO12Anchor(t *testing.T) {
|
||||
f, errs := parser.Parse("k_riscv64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·sb(SB), NOSPLIT, $0-0
|
||||
MOV $answer<>(SB), X10
|
||||
MOV answer<>(SB), X11
|
||||
MOV X12, answer<>(SB)
|
||||
RET
|
||||
|
||||
GLOBL answer<>(SB), RODATA, $8
|
||||
DATA answer<>+0(SB)/8, $42
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
obj, err := img.ELFRISCVObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFRISCVObject: %v", err)
|
||||
}
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse ELF: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
if flags := binary.LittleEndian.Uint32(obj[48:]); flags != efRISCVFloatAbiDouble {
|
||||
t.Errorf("e_flags = %#x, want %#x (EF_RISCV_FLOAT_ABI_DOUBLE)", flags, efRISCVFloatAbiDouble)
|
||||
}
|
||||
rela := ef.Section(".rela.text")
|
||||
if rela == nil {
|
||||
t.Fatal("missing .rela.text")
|
||||
}
|
||||
b, err := rela.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(b) != 6*24 {
|
||||
t.Fatalf(".rela.text holds %d entries, want six (three HI20/LO12 pairs)", len(b)/24)
|
||||
}
|
||||
le := binary.LittleEndian
|
||||
wantLo := []uint32{rRISCVPCRELLO12I, rRISCVPCRELLO12I, rRISCVPCRELLO12S}
|
||||
for p := range 3 {
|
||||
auipc := 8 * p
|
||||
hi := b[p*2*24:]
|
||||
lo := b[(p*2+1)*24:]
|
||||
if off := le.Uint64(hi[0:]); off != uint64(auipc) {
|
||||
t.Errorf("pair %d: HI20 r_offset = %d, want %d (the AUIPC)", p, off, auipc)
|
||||
}
|
||||
if typ := uint32(le.Uint64(hi[8:])); typ != rRISCVPCRELHI20 {
|
||||
t.Errorf("pair %d: HI20 type = %d, want %d", p, typ, rRISCVPCRELHI20)
|
||||
}
|
||||
if sym := int(le.Uint64(hi[8:]) >> 32); sym == 0 || sym == 1 {
|
||||
t.Errorf("pair %d: HI20 against symbol %d, want the target", p, sym)
|
||||
}
|
||||
if off := le.Uint64(lo[0:]); off != uint64(auipc+4) {
|
||||
t.Errorf("pair %d: LO12 r_offset = %d, want %d", p, off, auipc+4)
|
||||
}
|
||||
if typ := uint32(le.Uint64(lo[8:])); typ != wantLo[p] {
|
||||
t.Errorf("pair %d: LO12 type = %d, want %d", p, typ, wantLo[p])
|
||||
}
|
||||
// The LO12 must denote the AUIPC site: the .text section symbol
|
||||
// (index 1) plus the AUIPC's section-relative offset as addend.
|
||||
if sym := int(le.Uint64(lo[8:]) >> 32); sym != 1 {
|
||||
t.Errorf("pair %d: LO12 against symbol %d, want 1 (the .text section symbol)", p, sym)
|
||||
}
|
||||
if add := int64(le.Uint64(lo[16:])); add != int64(auipc) {
|
||||
t.Errorf("pair %d: LO12 addend = %d, want %d (S + A = the AUIPC address)", p, add, auipc)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestGOObjectRISCVStructure(t *testing.T) {
|
||||
f, errs := parser.Parse("k_riscv64.s", `
|
||||
#include "textflag.h"
|
||||
@@ -141,7 +224,7 @@ DATA answer<>+0(SB)/8, $42
|
||||
first := int(le.Uint32(relocIdx[4*(4+4):]))
|
||||
wantType := []uint16{relocRISCVPcrelItype, relocRISCVPcrelItype, relocRISCVPcrelStype}
|
||||
wantOffAbs := []int{0, 8, 16}
|
||||
for i := 0; i < 3; i++ {
|
||||
for i := range 3 {
|
||||
e := relocs[(first+i)*23:]
|
||||
if int32(le.Uint32(e[0:])) != int32(wantOffAbs[i]) || e[4] != 8 || le.Uint16(e[5:]) != wantType[i] ||
|
||||
le.Uint32(e[15:]) != pkgIdxSelf || le.Uint32(e[19:]) != 0 {
|
||||
@@ -221,7 +304,7 @@ func main() {
|
||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||
}
|
||||
var pkgArch, work, linkLine, asmObj string
|
||||
for _, line := range strings.Split(string(buildLog), "\n") {
|
||||
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
@@ -279,7 +362,7 @@ func main() {
|
||||
newArch := filepath.Join(dir, "pkg.a")
|
||||
args := []string{"tool", "pack", "c", newArch}
|
||||
seen := map[string]bool{}
|
||||
for _, m := range strings.Fields(string(listOut)) {
|
||||
for m := range strings.FieldsSeq(string(listOut)) {
|
||||
if seen[m] {
|
||||
continue
|
||||
}
|
||||
|
||||
+407
-48
@@ -36,12 +36,12 @@ const (
|
||||
vexNDS3Imm
|
||||
// vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg =
|
||||
// ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM
|
||||
// source lives in the reg field, the destination in r/m — the PEXTR-style
|
||||
// source lives in the reg field, the destination in r/m, the PEXTR-style
|
||||
// layout. VEXTRACTI128 and VEXTRACTF128 use this shape.
|
||||
vexExtract
|
||||
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
||||
// in ModRM.reg and the destination in r/m — the layout of the EVEX
|
||||
// narrowing stores (VPMOVDW, VPMOVQD).
|
||||
// in ModRM.reg and the destination in r/m, the layout of the EVEX
|
||||
// narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ.
|
||||
vexRMRev
|
||||
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
||||
// vector length follows the source: the packed-double → dword
|
||||
@@ -52,6 +52,30 @@ const (
|
||||
vexRMSrcLen
|
||||
// vexZero is the no-operand form (VZEROUPPER).
|
||||
vexZero
|
||||
// vexZeroAll is the no-operand form that zeroes the full upper state
|
||||
// (VZEROALL, the L = 1 twin of VZEROUPPER).
|
||||
vexZeroAll
|
||||
// vexNDS3GPR is the three-operand NDS form over general-purpose
|
||||
// registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0.
|
||||
vexNDS3GPR
|
||||
// vexImmRMGPR is the immediate form over general-purpose registers
|
||||
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
|
||||
vexImmRMGPR
|
||||
// vexRMOpGPR is the two-operand /digit form over general-purpose
|
||||
// registers (BLSI, BLSMSK, BLSR): ModRM.reg = /digit, ModRM.rm = src
|
||||
// (op0), VEX.vvvv = dst (op1), L = 0.
|
||||
vexRMOpGPR
|
||||
// vexCountGPR is the three-operand count form over general-purpose
|
||||
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): the first operand rides
|
||||
// VEX.vvvv and the second is r/m, the opposite pairing of the ANDN
|
||||
// family, with reg = dst (op2), L = 0.
|
||||
vexCountGPR
|
||||
// vexExtractGPR is the lane-extract-to-GPR form `OP $imm, xsrc, GPR/mem
|
||||
// dst`: ModRM.reg = xsrc (op1), ModRM.rm = destination (op2), imm8 =
|
||||
// op0, the VPEXTRB/W/D/Q layout. EVEX only; the destination never
|
||||
// carries a vector length, so the register the L'L field follows is the
|
||||
// XMM source.
|
||||
vexExtractGPR
|
||||
)
|
||||
|
||||
// vexSpec describes one VEX instruction's encoding parameters.
|
||||
@@ -68,7 +92,7 @@ type vexSpec struct {
|
||||
// incrementally; every entry is covered by a byte-for-byte ground-truth test
|
||||
// against the Go assembler.
|
||||
var vexTable = map[string]vexSpec{
|
||||
// VEX.128/256.66.0F.WIG — integer arithmetic / logic / compare.
|
||||
// VEX.128/256.66.0F.WIG, integer arithmetic / logic / compare.
|
||||
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3},
|
||||
"VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3},
|
||||
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3},
|
||||
@@ -82,7 +106,7 @@ var vexTable = map[string]vexSpec{
|
||||
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3},
|
||||
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3},
|
||||
// VEX.256.66.0F38.W0 — dword permute (three-operand NDS form).
|
||||
// VEX.256.66.0F38.W0, dword permute (three-operand NDS form).
|
||||
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3},
|
||||
// VEX.128/256.66.0F38.WIG.
|
||||
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3},
|
||||
@@ -90,14 +114,14 @@ var vexTable = map[string]vexSpec{
|
||||
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3},
|
||||
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
|
||||
|
||||
// VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic.
|
||||
// VEX.128/256.66.0F.WIG, packed double-precision arithmetic / logic.
|
||||
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
|
||||
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
|
||||
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
|
||||
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
|
||||
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
|
||||
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
|
||||
// VEX.128/256.0F.WIG — packed single-precision arithmetic.
|
||||
// VEX.128/256.0F.WIG, packed single-precision arithmetic.
|
||||
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
|
||||
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
|
||||
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
|
||||
@@ -107,7 +131,7 @@ var vexTable = map[string]vexSpec{
|
||||
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
|
||||
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
|
||||
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
|
||||
// VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed
|
||||
// VEX.128.F2.0F.WIG, scalar double-precision arithmetic (the packed
|
||||
// opcodes with an F2 pp).
|
||||
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
|
||||
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
|
||||
@@ -115,7 +139,7 @@ var vexTable = map[string]vexSpec{
|
||||
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
|
||||
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
|
||||
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
|
||||
// VEX.128.F3.0F.WIG — scalar single-precision arithmetic (the packed
|
||||
// VEX.128.F3.0F.WIG, scalar single-precision arithmetic (the packed
|
||||
// opcodes with an F3 pp).
|
||||
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
|
||||
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
|
||||
@@ -123,10 +147,16 @@ var vexTable = map[string]vexSpec{
|
||||
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
|
||||
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
|
||||
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
||||
// VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form).
|
||||
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
|
||||
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
||||
// Scalar fused multiply-add (NDS form). The Go assembler carries the
|
||||
// same 66 prefix as the packed forms on every FMA row, and W1 on the
|
||||
// double-precision spellings, so SD shares PD's prefix/W pair and the
|
||||
// scalar width rides on the W bit.
|
||||
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3},
|
||||
|
||||
// VEX.128/256.66.0F38.WIG — sign/zero extend and broadcast (reg=dst, rm=src,
|
||||
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
|
||||
// no vvvv).
|
||||
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
||||
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
||||
@@ -141,70 +171,131 @@ var vexTable = map[string]vexSpec{
|
||||
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
|
||||
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
||||
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
||||
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
|
||||
"VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM},
|
||||
"VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM},
|
||||
// VEX.128/256.F3.0F.WIG, signed dword to packed double conversion
|
||||
// (reg=dst, rm=src, no vvvv; the length follows the destination).
|
||||
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
|
||||
// VEX.128/256.0F.WIG — signed dword to packed single conversion
|
||||
// VEX.128/256.0F.WIG, signed dword to packed single conversion
|
||||
// (reg=dst, rm=src, no vvvv, no mandatory prefix).
|
||||
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
|
||||
// VEX.128/256.0F.WIG — packed single to packed double conversion
|
||||
// VEX.128/256.0F.WIG, packed single to packed double conversion
|
||||
// (reg=dst, rm=src; the destination is the wide operand and sets the
|
||||
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
|
||||
// the Go assembler emits the instruction with pp = 00, and gasm follows
|
||||
// the Go assembler's bytes — its machine code is the oracle, not the
|
||||
// the Go assembler's bytes, its machine code is the oracle, not the
|
||||
// manual.
|
||||
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
|
||||
// VEX.128.F2.0F.WIG — duplicate the low double of each 128-bit lane
|
||||
// VEX.128.F2.0F.WIG, duplicate the low double of each 128-bit lane
|
||||
// (reg=dst, rm=src, no vvvv; the length follows the destination).
|
||||
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
|
||||
// VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src).
|
||||
// VEX.128/256.66.0F.WIG, move mask to a GPR (reg=gpr dst, rm=vec src).
|
||||
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
|
||||
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
|
||||
|
||||
// VEX.128/256.66.0F.WIG — immediate shifts (opdigit selects the shift).
|
||||
// VEX.128/256.66.0F.WIG, immediate shifts (opdigit selects the shift).
|
||||
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm},
|
||||
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm},
|
||||
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm},
|
||||
"VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm},
|
||||
"VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm},
|
||||
|
||||
// VEX.128/256.66.0F.WIG — immediate shuffle (reg=dst, rm=src, imm8).
|
||||
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
|
||||
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
|
||||
// VEX.256.66.0F3A.W1 — qword permute (reg=dst, rm=src, imm8).
|
||||
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8).
|
||||
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
|
||||
|
||||
// VEX.128/256.66.0F.WIG — two-source shuffle (reg=dst, vvvv=src1, rm=src2,
|
||||
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
|
||||
// imm8).
|
||||
"VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm},
|
||||
// VEX.256.66.0F3A.W0 — permute / insert (same shape; VINSERTI128's rm is
|
||||
// VEX.256.66.0F3A.W0, permute / insert (same shape; VINSERTI128's rm is
|
||||
// the XMM or memory source).
|
||||
"VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm},
|
||||
"VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm},
|
||||
|
||||
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
|
||||
// VEX.256.66.0F3A.W0, lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
|
||||
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
|
||||
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
|
||||
// VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst:
|
||||
// reg=src, rm=XMM/memory dst, imm8 — the extract layout).
|
||||
// VEX.128/256.66.0F3A.W0, half-precision convert back ($imm, src, dst:
|
||||
// reg=src, rm=XMM/memory dst, imm8, the extract layout).
|
||||
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
|
||||
|
||||
// VEX.128.0F.W0 — no operands.
|
||||
// VEX.128.0F.W0, no operands.
|
||||
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
||||
// VEX.256.0F.W0, zero all vector registers (the L = 1 twin).
|
||||
"VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll},
|
||||
// VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare.
|
||||
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm},
|
||||
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm},
|
||||
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3},
|
||||
// VEX.128/256.0F.WIG, packed single XOR (NDS form).
|
||||
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3},
|
||||
// VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control.
|
||||
"VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm},
|
||||
"VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm},
|
||||
// VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling
|
||||
// falls through to the EVEX table.
|
||||
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm},
|
||||
// VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst).
|
||||
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm},
|
||||
// VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8).
|
||||
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm},
|
||||
// BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR).
|
||||
"ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR},
|
||||
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
|
||||
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
|
||||
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
|
||||
// VEX.NDS.LZ.0F38, the BMI2 three-operand bit ops: BEXTR and BZHI
|
||||
// share the F7/F5 opcodes across W, the variable shifts carry their
|
||||
// direction in the prefix (SHLX 66, SHRX F2, SARX F3) and PDEP/PEXT
|
||||
// in F2/F3.
|
||||
"BEXTRL": {2, 0xF7, 0, 0, -1, vexCountGPR},
|
||||
"BEXTRQ": {2, 0xF7, 1, 0, -1, vexCountGPR},
|
||||
"BZHIL": {2, 0xF5, 0, 0, -1, vexCountGPR},
|
||||
"BZHIQ": {2, 0xF5, 1, 0, -1, vexCountGPR},
|
||||
"SARXL": {2, 0xF7, 0, 2, -1, vexCountGPR},
|
||||
"SARXQ": {2, 0xF7, 1, 2, -1, vexCountGPR},
|
||||
"SHLXL": {2, 0xF7, 0, 1, -1, vexCountGPR},
|
||||
"SHLXQ": {2, 0xF7, 1, 1, -1, vexCountGPR},
|
||||
"SHRXL": {2, 0xF7, 0, 3, -1, vexCountGPR},
|
||||
"SHRXQ": {2, 0xF7, 1, 3, -1, vexCountGPR},
|
||||
"PDEPL": {2, 0xF5, 0, 3, -1, vexNDS3GPR},
|
||||
"PDEPQ": {2, 0xF5, 1, 3, -1, vexNDS3GPR},
|
||||
"PEXTL": {2, 0xF5, 0, 2, -1, vexNDS3GPR},
|
||||
"PEXTQ": {2, 0xF5, 1, 2, -1, vexNDS3GPR},
|
||||
// VEX.LZ.0F38.W, the BMI1 unary bit ops (src, dst: ModRM.reg = /digit,
|
||||
// rm = src, vvvv = dst).
|
||||
"BLSIL": {2, 0xF3, 0, 0, 3, vexRMOpGPR},
|
||||
"BLSIQ": {2, 0xF3, 1, 0, 3, vexRMOpGPR},
|
||||
"BLSMSKL": {2, 0xF3, 0, 0, 2, vexRMOpGPR},
|
||||
"BLSMSKQ": {2, 0xF3, 1, 0, 2, vexRMOpGPR},
|
||||
"BLSRL": {2, 0xF3, 0, 0, 1, vexRMOpGPR},
|
||||
"BLSRQ": {2, 0xF3, 1, 0, 1, vexRMOpGPR},
|
||||
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
|
||||
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
|
||||
|
||||
// VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
||||
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
||||
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
||||
|
||||
// VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst,
|
||||
// VEX.66.0F38.W0, broadcast a single/double to all lanes (reg=dst,
|
||||
// rm=scalar memory; SD is 256-bit only).
|
||||
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
||||
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
||||
// VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width
|
||||
// VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a
|
||||
// YMM (the encoder rejects an XMM destination, as go tool asm does).
|
||||
"VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM},
|
||||
// VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg,
|
||||
// memory destination in rm).
|
||||
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev},
|
||||
// VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv).
|
||||
"VPTEST": {2, 0x17, 0, 1, -1, vexRM},
|
||||
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
|
||||
// source).
|
||||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
||||
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
|
||||
// VEX.F3.0F.WIG, replicate even/odd singles (reg=dst, rm=src).
|
||||
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
||||
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
||||
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y
|
||||
// VEX.66.0F.WIG, packed double to packed single conversion, the X/Y
|
||||
// spellings: the destination is always XMM and the spelling fixes the
|
||||
// source length (X = 128, Y = 256).
|
||||
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||
@@ -228,18 +319,140 @@ var vexTable = map[string]vexSpec{
|
||||
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
|
||||
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
|
||||
|
||||
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
|
||||
// VEX.128/256.66.0F.WIG, word shifts (opdigit selects the shift).
|
||||
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
||||
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
|
||||
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
|
||||
|
||||
// VEX.F2.0F — packed double to packed dword conversions, truncating and
|
||||
// VEX.F2.0F, packed double to packed dword conversions, truncating and
|
||||
// non-truncating. The destination is always XMM; the X/Y spellings fix
|
||||
// the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen.
|
||||
// the source length (XMM/YMM), and VEX.L follows it, see vexSrcLen.
|
||||
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
||||
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
||||
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||
|
||||
// --- the VEX forms the avx512enc corpus exercises alongside the EVEX
|
||||
// spellings, read off the toolchain opcode tables ---
|
||||
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3},
|
||||
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3},
|
||||
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3},
|
||||
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3},
|
||||
"VANDNPD": {1, 0x55, 0, 1, -1, vexNDS3},
|
||||
"VANDPD": {1, 0x54, 0, 1, -1, vexNDS3},
|
||||
"VCOMISD": {1, 0x2F, 0, 1, -1, vexRM},
|
||||
"VCVTSD2SS": {1, 0x5A, 0, 3, -1, vexNDS3},
|
||||
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3},
|
||||
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3},
|
||||
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3},
|
||||
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3},
|
||||
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3},
|
||||
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3},
|
||||
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3},
|
||||
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3},
|
||||
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3},
|
||||
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3},
|
||||
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3},
|
||||
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3},
|
||||
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3},
|
||||
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3},
|
||||
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3},
|
||||
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3},
|
||||
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3},
|
||||
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3},
|
||||
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3},
|
||||
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3},
|
||||
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3},
|
||||
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3},
|
||||
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3},
|
||||
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm},
|
||||
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3},
|
||||
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM},
|
||||
"VMOVNTPD": {1, 0x2B, 0, 1, -1, vexRMRev},
|
||||
"VORPD": {1, 0x56, 0, 1, -1, vexNDS3},
|
||||
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3},
|
||||
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3},
|
||||
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3},
|
||||
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3},
|
||||
"VPCMPEQQ": {2, 0x29, 0, 1, -1, vexNDS3},
|
||||
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3},
|
||||
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3},
|
||||
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3},
|
||||
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3},
|
||||
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3},
|
||||
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtract},
|
||||
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtract},
|
||||
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtract},
|
||||
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm},
|
||||
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm},
|
||||
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3},
|
||||
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3},
|
||||
"VPMULUDQ": {1, 0xF4, 0, 1, -1, vexNDS3},
|
||||
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3},
|
||||
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3},
|
||||
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3},
|
||||
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3},
|
||||
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKHQDQ": {1, 0x6D, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3},
|
||||
"VSQRTPD": {1, 0x51, 0, 1, -1, vexRM},
|
||||
"VSQRTSD": {1, 0x51, 0, 3, -1, vexNDS3},
|
||||
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3},
|
||||
"VUCOMISD": {1, 0x2E, 0, 1, -1, vexRM},
|
||||
|
||||
// VEX.0F.WIG, the plain-prefix single/double arithmetic and unpack
|
||||
// spellings (no 66 prefix; WIG, so W = 0).
|
||||
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3},
|
||||
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3},
|
||||
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3},
|
||||
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3},
|
||||
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3},
|
||||
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM},
|
||||
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev},
|
||||
// VEX.128.66.0F, the scalar and packed compare forms.
|
||||
"VCOMISS": {1, 0x2F, 0, 1, -1, vexRM},
|
||||
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM},
|
||||
// VEX.128.0F.F3/F2.W0, the high/low word shuffles ($imm, src, dst).
|
||||
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM},
|
||||
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM},
|
||||
}
|
||||
|
||||
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
|
||||
@@ -255,7 +468,7 @@ var vexSrcLen = map[string]int{
|
||||
"VCVTPD2PSY": 1,
|
||||
}
|
||||
|
||||
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
|
||||
// vexVarShift maps the shift mnemonics to their variable-count opcode, the
|
||||
// form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8),
|
||||
// an ordinary NDS encoding rather than the /digit immediate form above.
|
||||
var vexVarShift = map[string]byte{
|
||||
@@ -286,20 +499,22 @@ type vexMoveSpec struct {
|
||||
|
||||
// vexMoveTable maps an upper-case move mnemonic to its encoding.
|
||||
var vexMoveTable = map[string]vexMoveSpec{
|
||||
// VEX.128/256.F3.0F.WIG — unaligned integer move.
|
||||
// VEX.128/256.F3.0F.WIG, unaligned integer move.
|
||||
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
||||
// VEX.128/256.66.0F.WIG — unaligned packed double move.
|
||||
// VEX.128/256.66.0F.WIG, aligned integer move.
|
||||
"VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
||||
// VEX.128/256.66.0F.WIG, unaligned packed double move.
|
||||
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
|
||||
// VEX.128.66.0F.W0 — 32-bit GPR/memory ↔ XMM.
|
||||
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
|
||||
"VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true},
|
||||
// VMOVQ — 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
|
||||
// VMOVQ, 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
|
||||
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true},
|
||||
// VEX.128.F2.0F.WIG — scalar double move, memory operands only (the
|
||||
// VEX.128.F2.0F.WIG, scalar double move, memory operands only (the
|
||||
// register form takes three operands and is not supported yet).
|
||||
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
|
||||
// VEX.128.F3.0F.WIG — scalar single move, memory operands only.
|
||||
// VEX.128.F3.0F.WIG, scalar single move, memory operands only.
|
||||
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
|
||||
// VEX.128/256 — aligned packed moves.
|
||||
// VEX.128/256, aligned packed moves.
|
||||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
|
||||
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
|
||||
}
|
||||
@@ -315,13 +530,21 @@ func isVex(mnemUpper string) bool {
|
||||
|
||||
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
|
||||
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
||||
// Vector register indices 16–31 exist only in EVEX encodings; fail
|
||||
// Vector register indices 16-31 exist only in EVEX encodings; fail
|
||||
// loudly rather than silently truncating the index.
|
||||
for _, op := range ops {
|
||||
if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 {
|
||||
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
||||
}
|
||||
}
|
||||
// VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination
|
||||
// only; an XMM destination is rejected exactly as go tool asm does.
|
||||
if mnemUpper == "VBROADCASTI128" {
|
||||
dstReg, ok := ops[len(ops)-1].(Reg)
|
||||
if len(ops) != 2 || !ok || dstReg.size != 32 {
|
||||
return fmt.Errorf("VBROADCASTI128 requires a YMM destination")
|
||||
}
|
||||
}
|
||||
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
||||
return e.encodeVexMove(mnemUpper, ms, ops)
|
||||
}
|
||||
@@ -354,6 +577,18 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
||||
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
||||
case vexZero:
|
||||
return e.encodeVexZero(mnemUpper, spec, ops)
|
||||
case vexZeroAll:
|
||||
return e.encodeVexZeroAll(mnemUpper, spec, ops)
|
||||
case vexNDS3GPR:
|
||||
return e.encodeVexNDS3GPR(spec, ops)
|
||||
case vexImmRMGPR:
|
||||
return e.encodeVexImmRMGPR(spec, ops)
|
||||
case vexRMOpGPR:
|
||||
return e.encodeVexRMOpGPR(spec, ops)
|
||||
case vexCountGPR:
|
||||
return e.encodeVexCountGPR(spec, ops)
|
||||
case vexRMRev:
|
||||
return e.encodeVexRMRev(spec, ops)
|
||||
}
|
||||
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
|
||||
}
|
||||
@@ -418,7 +653,7 @@ func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
|
||||
}
|
||||
|
||||
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
|
||||
// the destination always XMM and the VEX.L bit following the source — fixed
|
||||
// the destination always XMM and the VEX.L bit following the source, fixed
|
||||
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
|
||||
// the source is memory.
|
||||
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
|
||||
@@ -455,9 +690,10 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
|
||||
if !ok {
|
||||
return fmt.Errorf("shift count must be an immediate")
|
||||
}
|
||||
srcReg, ok := src.(Reg)
|
||||
if !ok || !srcReg.isVec() {
|
||||
return fmt.Errorf("shift source must be a vector register")
|
||||
// The count source is a vector register or memory; the VEX length
|
||||
// follows the destination register either way.
|
||||
if !vecOrMem(src) {
|
||||
return fmt.Errorf("shift source must be a vector register or memory")
|
||||
}
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
@@ -465,7 +701,7 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
|
||||
}
|
||||
|
||||
vvvvBar := 15 - (dstReg.idx & 15)
|
||||
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, srcReg); err != nil {
|
||||
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, src); err != nil {
|
||||
return err
|
||||
}
|
||||
immByte, err := imm8(int64(immVal))
|
||||
@@ -605,6 +841,129 @@ func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1
|
||||
// twin of VZEROUPPER.
|
||||
func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 0 {
|
||||
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
|
||||
}
|
||||
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1.
|
||||
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode)
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose
|
||||
// registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1,
|
||||
// rm = src2 and L = 0.
|
||||
func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
|
||||
}
|
||||
src2, src1, dst := ops[0], ops[1], ops[2]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || dstReg.isVec() {
|
||||
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||
}
|
||||
vvvvReg, ok := src1.(Reg)
|
||||
if !ok || vvvvReg.isVec() {
|
||||
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
|
||||
}
|
||||
rBit := 0
|
||||
if dstReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2)
|
||||
}
|
||||
|
||||
// encodeVexImmRMGPR encodes the immediate form over general-purpose
|
||||
// registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0.
|
||||
func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||
}
|
||||
imm, src, dst := ops[0], ops[1], ops[2]
|
||||
immVal, ok := imm.(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("shift control must be an immediate")
|
||||
}
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || dstReg.isVec() {
|
||||
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||
}
|
||||
immByte, err := imm8(int64(immVal))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src); err != nil {
|
||||
return err
|
||||
}
|
||||
e.out = append(e.out, immByte)
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeVexRMOpGPR encodes the two-operand /digit form over general-purpose
|
||||
// registers (BLSI, BLSMSK, BLSR): OP src, dst with ModRM.reg = /digit,
|
||||
// ModRM.rm = src and VEX.vvvv = dst.
|
||||
func (e *enc) encodeVexRMOpGPR(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("instruction expects 2 operands (src, dst), got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || dstReg.isVec() {
|
||||
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||
}
|
||||
return e.emitVexFields(spec, 0, spec.opdigit, 0, 15-(dstReg.idx&15), src)
|
||||
}
|
||||
|
||||
// encodeVexCountGPR encodes the three-operand count form over general-purpose
|
||||
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): OP src, count, dst with
|
||||
// VEX.vvvv = src (op0), ModRM.rm = count (op1), ModRM.reg = dst (op2).
|
||||
func (e *enc) encodeVexCountGPR(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("VEX count instruction expects 3 operands, got %d", len(ops))
|
||||
}
|
||||
src, count, dst := ops[0], ops[1], ops[2]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || dstReg.isVec() {
|
||||
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||
}
|
||||
countReg, ok := count.(Reg)
|
||||
if !ok || countReg.isVec() {
|
||||
return fmt.Errorf("VEX count operand must be a general-purpose register")
|
||||
}
|
||||
srcReg, ok := src.(Reg)
|
||||
if !ok || srcReg.isVec() {
|
||||
return fmt.Errorf("VEX count source must be a general-purpose register")
|
||||
}
|
||||
rBit := 0
|
||||
if dstReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(srcReg.idx&15), count)
|
||||
}
|
||||
|
||||
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
|
||||
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
|
||||
// a store with no register-destination form).
|
||||
func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("store expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
srcReg, ok := ops[0].(Reg)
|
||||
if !ok || !srcReg.isVec() {
|
||||
return fmt.Errorf("store source must be a vector register")
|
||||
}
|
||||
if !memOperand(ops[1]) {
|
||||
return fmt.Errorf("store destination must be memory")
|
||||
}
|
||||
rBit := 0
|
||||
if srcReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
|
||||
}
|
||||
|
||||
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
|
||||
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
|
||||
// move uses the store-form layout (reg = source, rm = destination), matching
|
||||
|
||||
+121
-5
@@ -19,6 +19,65 @@ func vreg(t *testing.T, name string) Reg {
|
||||
return r
|
||||
}
|
||||
|
||||
// x86asmUnrecognised lists the VEX mnemonics whose machine code the
|
||||
// golang.org/x/arch decoder cannot resolve; their bytes are verified against
|
||||
// go tool asm in the ground-truth tests instead.
|
||||
var x86asmUnrecognised = map[string]bool{
|
||||
"ANDNL": true,
|
||||
"ANDNQ": true,
|
||||
"MULXL": true,
|
||||
"MULXQ": true,
|
||||
"RORXL": true,
|
||||
"RORXQ": true,
|
||||
"VFMADD213SD": true,
|
||||
"VFNMADD231SD": true,
|
||||
// The scalar FMA spellings the decoder's tables lack entirely.
|
||||
"VFMADD132SD": true,
|
||||
"VFMADD132SS": true,
|
||||
"VFMADD213SS": true,
|
||||
"VFMADD231SD": true,
|
||||
"VFMADD231SS": true,
|
||||
"VFMSUB132SD": true,
|
||||
"VFMSUB132SS": true,
|
||||
"VFMSUB213SD": true,
|
||||
"VFMSUB213SS": true,
|
||||
"VFMSUB231SD": true,
|
||||
"VFMSUB231SS": true,
|
||||
"VFNMADD132SD": true,
|
||||
"VFNMADD132SS": true,
|
||||
"VFNMADD213SD": true,
|
||||
"VFNMADD213SS": true,
|
||||
"VFNMADD231SS": true,
|
||||
"VFNMSUB132SD": true,
|
||||
"VFNMSUB132SS": true,
|
||||
"VFNMSUB213SD": true,
|
||||
"VFNMSUB213SS": true,
|
||||
"VFNMSUB231SD": true,
|
||||
"VFNMSUB231SS": true,
|
||||
// The BMI1 unary bit ops the decoder's AVX tables lack.
|
||||
"BLSIL": true,
|
||||
"BLSIQ": true,
|
||||
"BLSMSKL": true,
|
||||
"BLSMSKQ": true,
|
||||
"BLSRL": true,
|
||||
"BLSRQ": true,
|
||||
// The BMI2 bit ops whose W1/LZ rows the decoder misses.
|
||||
"BEXTRL": true,
|
||||
"BEXTRQ": true,
|
||||
"BZHIL": true,
|
||||
"BZHIQ": true,
|
||||
"PDEPL": true,
|
||||
"PDEPQ": true,
|
||||
"PEXTL": true,
|
||||
"PEXTQ": true,
|
||||
"SARXL": true,
|
||||
"SARXQ": true,
|
||||
"SHLXL": true,
|
||||
"SHLXQ": true,
|
||||
"SHRXL": true,
|
||||
"SHRXQ": true,
|
||||
}
|
||||
|
||||
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
|
||||
// instruction and verifies it round-trips through the x86 decoder to the same
|
||||
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
|
||||
@@ -37,8 +96,15 @@ func TestVexNDS3(t *testing.T) {
|
||||
t.Errorf("%s: Encode: %v", mnem, err)
|
||||
continue
|
||||
}
|
||||
// The x86 decoder's table lacks a handful of rows the Go assembler
|
||||
// emits (the scalar 213/231 FMA spellings among them); those are
|
||||
// pinned byte for byte against go tool asm in TestVexGroundTruth
|
||||
// instead of round-tripped here.
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
if strings.Contains(err.Error(), "unrecognized instruction") && x86asmUnrecognised[mnem] {
|
||||
continue
|
||||
}
|
||||
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
||||
continue
|
||||
}
|
||||
@@ -173,7 +239,7 @@ func TestVexGroundTruth(t *testing.T) {
|
||||
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""},
|
||||
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""},
|
||||
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""},
|
||||
// Floating point (packed and scalar) and FMA — same NDS form, the pp
|
||||
// Floating point (packed and scalar) and FMA; same NDS form, the pp
|
||||
// bits and map select the operation.
|
||||
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""},
|
||||
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""},
|
||||
@@ -184,6 +250,50 @@ func TestVexGroundTruth(t *testing.T) {
|
||||
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
|
||||
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
|
||||
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
|
||||
{"VFMADD213SD X0,X1,X2", "VFMADD213SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1a9d0", ""},
|
||||
{"VFNMADD231SD X0,X1,X2", "VFNMADD231SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1bdd0", ""},
|
||||
// Packed single XOR and byte compare (NDS form).
|
||||
{"VXORPS Y0,Y1,Y2", "VXORPS", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f457d0", ""},
|
||||
{"VPCMPEQB Y0,Y1,Y2", "VPCMPEQB", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f574d0", ""},
|
||||
// Octa byte shifts (vvvv carries the destination).
|
||||
{"VPSLLDQ $2,X0,X1", "VPSLLDQ", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "c5f173f802", ""},
|
||||
{"VPSRLDQ $2,Y0,Y1", "VPSRLDQ", []Operand{Imm(2), vreg(t, "Y0"), vreg(t, "Y1")}, "c5f573d802", ""},
|
||||
// Two-source shuffle, blend and carry-less multiply (NDS + imm8).
|
||||
{"VPERM2F128 $3,Y0,Y1,Y2", "VPERM2F128", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37506d003", ""},
|
||||
{"VPBLENDD $3,X0,X1,X2", "VPBLENDD", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37102d003", ""},
|
||||
{"VPBLENDD $3,Y0,Y1,Y2", "VPBLENDD", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37502d003", ""},
|
||||
{"VPCLMULQDQ $0,X0,X1,X2", "VPCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37144d000", ""},
|
||||
{"VGF2P8AFFINEQB $0,X0,X1,X2", "VGF2P8AFFINEQB", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e3f1ced000", ""},
|
||||
// Two-operand test and the non-temporal and broadcast stores.
|
||||
{"VPTEST X0,X1", "VPTEST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c4e27917c8", ""},
|
||||
{"VPTEST Y0,Y1", "VPTEST", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c4e27d17c8", ""},
|
||||
{"VMOVNTDQ Y0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "Y0"), Ptr(AX, 0, 32)}, "c5fde700", ""},
|
||||
{"VMOVNTDQ X0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "c5f9e700", ""},
|
||||
{"VBROADCASTI128 (AX),Y1", "VBROADCASTI128", []Operand{Ptr(AX, 0, 16), vreg(t, "Y1")}, "c4e27d5a08", ""},
|
||||
// Aligned integer move and the full zeroing form.
|
||||
{"VMOVDQA X0,X1", "VMOVDQA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c5f97fc1", ""},
|
||||
{"VMOVDQA (AX),X1", "VMOVDQA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "c5f96f08", ""},
|
||||
{"VMOVDQA Y0,Y1", "VMOVDQA", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c5fd7fc1", ""},
|
||||
{"VZEROALL", "VZEROALL", []Operand{}, "c5fc77", ""},
|
||||
// BMI1/BMI2 general-register VEX forms.
|
||||
{"ANDNL AX,BX,CX", "ANDNL", []Operand{AX, BX, CX}, "c4e260f2c8", ""},
|
||||
{"ANDNQ AX,BX,CX", "ANDNQ", []Operand{AX, BX, CX}, "c4e2e0f2c8", ""},
|
||||
{"MULXL AX,BX,CX", "MULXL", []Operand{AX, BX, CX}, "c4e263f6c8", ""},
|
||||
{"MULXQ AX,BX,CX", "MULXQ", []Operand{AX, BX, CX}, "c4e2e3f6c8", ""},
|
||||
{"RORXL $3,AX,CX", "RORXL", []Operand{Imm(3), AX, CX}, "c4e37bf0c803", ""},
|
||||
{"RORXQ $3,AX,CX", "RORXQ", []Operand{Imm(3), AX, CX}, "c4e3fbf0c803", ""},
|
||||
// BMI2 variable shifts and bit ops (three general registers).
|
||||
{"SHLXL AX,CX,R15", "SHLXL", []Operand{AX, CX, vreg(t, "R15")}, "c46279f7f9", ""},
|
||||
{"SHRXQ R8,DX,AX", "SHRXQ", []Operand{vreg(t, "R8"), DX, AX}, "c4e2bbf7c2", ""},
|
||||
{"SARXQ AX,DX,R9", "SARXQ", []Operand{AX, DX, vreg(t, "R9")}, "c462faf7ca", ""},
|
||||
{"BEXTRL AX,CX,R15", "BEXTRL", []Operand{AX, CX, vreg(t, "R15")}, "c46278f7f9", ""},
|
||||
{"BZHIQ AX,CX,R15", "BZHIQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f8f5f9", ""},
|
||||
{"PDEPQ AX,CX,R15", "PDEPQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f3f5f8", ""},
|
||||
{"PEXTQ AX,CX,R15", "PEXTQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f2f5f8", ""},
|
||||
// BMI1 unary bit ops (src, dst: /digit in ModRM.reg, dst in vvvv).
|
||||
{"BLSIL AX,CX", "BLSIL", []Operand{AX, CX}, "c4e270f3d8", ""},
|
||||
{"BLSRQ AX,CX", "BLSRQ", []Operand{AX, CX}, "c4e2f0f3c8", ""},
|
||||
{"BLSMSKQ AX,CX", "BLSMSKQ", []Operand{AX, CX}, "c4e2f0f3d0", ""},
|
||||
// Two-operand reg/rm form (v̄vvv must be 1111).
|
||||
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
|
||||
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
|
||||
@@ -217,7 +327,7 @@ func TestVexGroundTruth(t *testing.T) {
|
||||
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""},
|
||||
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""},
|
||||
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""},
|
||||
// Moves — each direction picks its own opcode and VEX.W.
|
||||
// Moves; each direction picks its own opcode and VEX.W.
|
||||
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""},
|
||||
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""},
|
||||
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""},
|
||||
@@ -234,7 +344,7 @@ func TestVexGroundTruth(t *testing.T) {
|
||||
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""},
|
||||
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""},
|
||||
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""},
|
||||
// Packed double arithmetic and unpack — the NDS form, the opcode
|
||||
// Packed double arithmetic and unpack; the NDS form, the opcode
|
||||
// selects the operation.
|
||||
{"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""},
|
||||
{"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""},
|
||||
@@ -255,12 +365,12 @@ func TestVexGroundTruth(t *testing.T) {
|
||||
{"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""},
|
||||
{"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""},
|
||||
{"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""},
|
||||
// VMOVDDUP — duplicate the low double (reg=dst, rm=src, F2 pp).
|
||||
// VMOVDDUP; duplicate the low double (reg=dst, rm=src, F2 pp).
|
||||
{"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""},
|
||||
{"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""},
|
||||
{"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""},
|
||||
// Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3
|
||||
// prefix — see the table comment), DQ→PD.
|
||||
// prefix; see the table comment), DQ→PD.
|
||||
{"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""},
|
||||
{"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""},
|
||||
{"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""},
|
||||
@@ -287,6 +397,12 @@ func TestVexGroundTruth(t *testing.T) {
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
// The decoder's AVX/BMI table lacks a few rows the Go
|
||||
// assembler emits (the GPR VEX forms and the scalar FMA
|
||||
// spellings); their bytes are the ground truth here.
|
||||
if x86asmUnrecognised[c.mnem] {
|
||||
continue
|
||||
}
|
||||
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
|
||||
+21
-12
@@ -8,7 +8,7 @@
|
||||
// to the arch package; the AST records syntax only.
|
||||
package ast
|
||||
|
||||
import "sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
import "sourcedock.dev/petrbalvin/gasm-sdk/token"
|
||||
|
||||
// File is the parsed representation of one .s source file.
|
||||
type File struct {
|
||||
@@ -17,7 +17,7 @@ type File struct {
|
||||
Orphans []Stmt // labels/instructions seen before any TEXT directive
|
||||
// Macros holds the names introduced by #define directives in this file.
|
||||
// The linter uses it to avoid flagging macro invocations as unknown
|
||||
// instructions (macro expansion itself is out of scope — see the docs).
|
||||
// instructions (macro expansion itself is out of scope, see the docs).
|
||||
Macros map[string]bool
|
||||
}
|
||||
|
||||
@@ -114,6 +114,7 @@ type Symbol struct {
|
||||
Pkg string // package prefix before the middle dot ("" = current package)
|
||||
Name string // identifier without the middle dot or <>
|
||||
Static bool // the <> marker is present
|
||||
ABI string // the <NAME> ABI marker, e.g. ABIInternal ("" when absent)
|
||||
Pseudo string // FP, SP, SB or PC ("" for a bare name)
|
||||
Offset int64
|
||||
HasOff bool
|
||||
@@ -123,10 +124,8 @@ type Symbol struct {
|
||||
// OpKind classifies an operand syntactically.
|
||||
type OpKind int
|
||||
|
||||
// Operand kinds.
|
||||
const (
|
||||
OpInvalid OpKind = iota
|
||||
OpImmediate // $value
|
||||
OpImmediate = iota // $value
|
||||
OpAddr // register, memory reference, symbol or label
|
||||
)
|
||||
|
||||
@@ -152,11 +151,21 @@ type Immediate struct {
|
||||
// Address is a non-immediate operand: a register, a memory reference, a symbol
|
||||
// reference or a label. Fields are populated best-effort from the syntax.
|
||||
type Address struct {
|
||||
Sym *Symbol // name reference (bare ident, or name+off(pseudo))
|
||||
Base string // base register, from (base)
|
||||
Index string // index register, from (index*scale)
|
||||
Scale int // index scale; 0 when absent
|
||||
Offset int64 // leading displacement, from off(base)
|
||||
HasOff bool // a leading displacement is present
|
||||
Shift string // verbatim arm64 shift suffix, e.g. "<<2"
|
||||
Sym *Symbol // name reference (bare ident, or name+off(pseudo))
|
||||
Base string // base register, from (base)
|
||||
Index string // index register, from (index*scale)
|
||||
Scale int // index scale; 0 when absent
|
||||
Offset int64 // leading displacement, from off(base)
|
||||
HasOff bool // a leading displacement is present
|
||||
Shift string // verbatim arm64 shift suffix, e.g. "<< 2"
|
||||
Range *RegRange // bracketed register range; nil for every other form
|
||||
}
|
||||
|
||||
// RegRange is a bracketed register range, [Z0-Z3]: the amd64 spelling of
|
||||
// the four-register source of the 4FMAPS/4VNNIW families. Lo and Hi carry
|
||||
// the verbatim register spellings; the range is inclusive at both ends.
|
||||
type RegRange struct {
|
||||
Lo string
|
||||
Hi string
|
||||
Pos token.Position
|
||||
}
|
||||
|
||||
+3
-3
@@ -6,7 +6,7 @@ package ast
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/token"
|
||||
)
|
||||
|
||||
func pos(line, col int) token.Position { return token.Position{Line: line, Column: col} }
|
||||
@@ -54,11 +54,11 @@ func TestStmtPositions(t *testing.T) {
|
||||
// TestInterfaces confirms the node types satisfy their interfaces, so callers
|
||||
// can range over Decls and Stmts.
|
||||
func TestInterfaces(t *testing.T) {
|
||||
var decls []Decl = []Decl{&Include{}, &Preproc{}, &Text{}, &Globl{}, &Data{}}
|
||||
var decls = []Decl{&Include{}, &Preproc{}, &Text{}, &Globl{}, &Data{}}
|
||||
if len(decls) != 5 {
|
||||
t.Fatal("decl interface set")
|
||||
}
|
||||
var stmts []Stmt = []Stmt{&Label{}, &Instr{}}
|
||||
var stmts = []Stmt{&Label{}, &Instr{}}
|
||||
if len(stmts) != 2 {
|
||||
t.Fatal("stmt interface set")
|
||||
}
|
||||
|
||||
@@ -0,0 +1,367 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"go/ast"
|
||||
"go/build"
|
||||
"go/constant"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"go/types"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||
)
|
||||
|
||||
// go_asm.h is the header the Go compiler writes for every package that
|
||||
// carries assembly (the compiler's -asmhdr output): "#define const_NAME
|
||||
// value" for each package constant, and for each named struct type
|
||||
// "#define TYPE__size size" plus one "#define TYPE_field offset" per field.
|
||||
// GOROOT assembly includes it, and a standalone assembler has no compiler
|
||||
// to have produced it, so gasm generates the equivalent itself: the package
|
||||
// the .s file lives in is parsed and type-checked here, with the target
|
||||
// architecture's own sizes, and the same defines are written out. The
|
||||
// type-checking GOOS is selected by the caller: a GOOS-specific file
|
||||
// (sys_darwin_arm64.s) needs its platform's defines, which a header from
|
||||
// the ambient GOOS silently omits.
|
||||
//
|
||||
// The emitter mirrors cmd/compile's dumpasmhdr exactly: constants come out
|
||||
// as "const_NAME", struct entries as "NAME__size" followed by the fields in
|
||||
// declaration order, blank names are skipped, and float and complex
|
||||
// constants are omitted (the assembler carries integers, bools and strings
|
||||
// only). Aliases to structs are emitted, generic types are not: they have
|
||||
// no fixed size. A define the assembly references but this header does not
|
||||
// carry surfaces later as the assembler's own "undefined" diagnostic naming
|
||||
// the define, which is the honest failure.
|
||||
|
||||
// goAsmInclude matches the #include "go_asm.h" directive, tolerant of
|
||||
// whitespace, so the wiring knows which files need a generated header
|
||||
// before the preprocessor runs and would report the header as missing.
|
||||
var goAsmInclude = regexp.MustCompile(`(?m)^\s*#\s*include\s+"go_asm\.h"`)
|
||||
|
||||
// needsGoAsmHeader reports whether src includes go_asm.h.
|
||||
func needsGoAsmHeader(src string) bool {
|
||||
return goAsmInclude.MatchString(src)
|
||||
}
|
||||
|
||||
// goAsmHeaderResolved reports whether the include of go_asm.h from a file in
|
||||
// asmDir already resolves: to a header in the package directory itself, or
|
||||
// in one of the -I directories, the way the preprocessor searches. Only an
|
||||
// unresolved include is generated for; a header someone placed by hand is
|
||||
// the tool the author chose, and it also wins the preprocessor's own search
|
||||
// order, so generating a second copy would be dead weight at best.
|
||||
func goAsmHeaderResolved(asmDir string, dirs []string) bool {
|
||||
candidates := []string{filepath.Join(asmDir, "go_asm.h")}
|
||||
for _, d := range dirs {
|
||||
candidates = append(candidates, filepath.Join(d, "go_asm.h"))
|
||||
}
|
||||
for _, candidate := range candidates {
|
||||
if st, err := os.Stat(candidate); err == nil && !st.IsDir() {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// generateGoAsmHeader type-checks the Go package in pkgDir for goos and
|
||||
// goarch, writes its go_asm.h equivalent into dir, and returns dir. An
|
||||
// empty goos means the ambient one. The caller owns the directory and its
|
||||
// removal.
|
||||
func generateGoAsmHeader(pkgDir, goos, goarch, dir string) (string, error) {
|
||||
if goos == "" {
|
||||
goos = build.Default.GOOS
|
||||
}
|
||||
imp := newSourceImporter(goos, goarch)
|
||||
if imp.sizes == nil {
|
||||
return "", fmt.Errorf("go_asm.h: unknown GOARCH %q", goarch)
|
||||
}
|
||||
bp, err := imp.ctxt.ImportDir(pkgDir, 0)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %w", goarch, pkgDir, err)
|
||||
}
|
||||
files, errs := imp.parse(bp)
|
||||
if len(errs) > 0 {
|
||||
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %s", goarch, pkgDir, errorList(errs))
|
||||
}
|
||||
_, info, errs := imp.checkPackage(bp, files)
|
||||
if len(errs) > 0 {
|
||||
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: package does not type-check: %s", goarch, pkgDir, errorList(errs))
|
||||
}
|
||||
|
||||
var b strings.Builder
|
||||
fmt.Fprintf(&b, "// generated by gasm from package %s (GOOS %s, GOARCH %s)\n\n", bp.Name, goos, goarch)
|
||||
// Files in the build's own order and declarations in source order: the
|
||||
// same walk the compiler's reader makes, so the header reads the same
|
||||
// way the toolchain's does. Order carries no meaning to the assembler
|
||||
// (defines form a table), only to a human diffing against one.
|
||||
for _, f := range files {
|
||||
for _, decl := range f.Decls {
|
||||
gd, ok := decl.(*ast.GenDecl)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
for _, spec := range gd.Specs {
|
||||
switch gd.Tok {
|
||||
case token.CONST:
|
||||
vs, ok := spec.(*ast.ValueSpec)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
for _, name := range vs.Names {
|
||||
emitConst(&b, info.Defs[name], name.Name)
|
||||
}
|
||||
case token.TYPE:
|
||||
ts, ok := spec.(*ast.TypeSpec)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
emitStruct(&b, imp.sizes, info.Defs[ts.Name], ts.Name.Name)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %w", goarch, pkgDir, err)
|
||||
}
|
||||
out := filepath.Join(dir, "go_asm.h")
|
||||
if err := os.WriteFile(out, []byte(b.String()), 0o644); err != nil {
|
||||
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %w", goarch, pkgDir, err)
|
||||
}
|
||||
return dir, nil
|
||||
}
|
||||
|
||||
// emitConst writes one const define, skipping what the toolchain skips:
|
||||
// blank names, and float and complex values the assembler has no syntax for.
|
||||
func emitConst(b *strings.Builder, obj types.Object, name string) {
|
||||
c, ok := obj.(*types.Const)
|
||||
if !ok || name == "_" {
|
||||
return
|
||||
}
|
||||
switch c.Val().Kind() {
|
||||
case constant.Float, constant.Complex, constant.Unknown:
|
||||
return
|
||||
}
|
||||
fmt.Fprintf(b, "#define const_%s %s\n", name, c.Val().ExactString())
|
||||
}
|
||||
|
||||
// emitStruct writes one named struct type's size and field offsets,
|
||||
// skipping what the toolchain skips: blank names, non-struct types, and
|
||||
// generic types, whose size depends on their instantiation.
|
||||
func emitStruct(b *strings.Builder, sizes types.Sizes, obj types.Object, name string) {
|
||||
tn, ok := obj.(*types.TypeName)
|
||||
if !ok || name == "_" {
|
||||
return
|
||||
}
|
||||
t := types.Unalias(tn.Type())
|
||||
// Generic types are spelled *types.Named with a type-parameter list;
|
||||
// a plain struct type or an instantiated one carries none.
|
||||
if named, ok := t.(*types.Named); ok && named.TypeParams().Len() > 0 {
|
||||
return
|
||||
}
|
||||
st, ok := t.Underlying().(*types.Struct)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
fmt.Fprintf(b, "#define %s__size %d\n", name, sizes.Sizeof(t))
|
||||
fields := make([]*types.Var, st.NumFields())
|
||||
for i := range st.NumFields() {
|
||||
fields[i] = st.Field(i)
|
||||
}
|
||||
for i, off := range sizes.Offsetsof(fields) {
|
||||
fld := fields[i]
|
||||
if fld.Name() == "_" {
|
||||
continue
|
||||
}
|
||||
fmt.Fprintf(b, "#define %s_%s %d\n", name, fld.Name(), off)
|
||||
}
|
||||
}
|
||||
|
||||
// errorList renders at most three errors, enough to say what is wrong
|
||||
// without burying the diagnostic the caller actually reads.
|
||||
func errorList(errs []error) string {
|
||||
if len(errs) > 3 {
|
||||
errs = errs[:3]
|
||||
}
|
||||
msgs := make([]string, len(errs))
|
||||
for i, err := range errs {
|
||||
msgs[i] = err.Error()
|
||||
}
|
||||
return strings.Join(msgs, "; ")
|
||||
}
|
||||
|
||||
// sourceImporter type-checks imported packages from source with the target
|
||||
// architecture's sizes. go/importer's "source" importer pins the host
|
||||
// GOARCH, which would lay out imported types (internal/cpu, internal/abi)
|
||||
// for the wrong target on a cross-architecture header, so the recursion is
|
||||
// carried here with one build context and one sizes instance per
|
||||
// architecture.
|
||||
type sourceImporter struct {
|
||||
fset *token.FileSet
|
||||
ctxt *build.Context
|
||||
sizes types.Sizes
|
||||
pkgs map[string]*types.Package
|
||||
}
|
||||
|
||||
// newSourceImporter returns the importer for one target GOOS and GOARCH.
|
||||
// Cgo is disabled so the file set is deterministic and independent of the
|
||||
// host's C toolchain: cgo-tagged files drop out of the build exactly as
|
||||
// they do from a CGO_ENABLED=0 build, whose assembly is what gasm targets.
|
||||
func newSourceImporter(goos, goarch string) *sourceImporter {
|
||||
ctxt := new(build.Context)
|
||||
*ctxt = build.Default
|
||||
ctxt.GOOS = goos
|
||||
ctxt.GOARCH = goarch
|
||||
ctxt.CgoEnabled = false
|
||||
return &sourceImporter{
|
||||
fset: token.NewFileSet(),
|
||||
ctxt: ctxt,
|
||||
sizes: types.SizesFor("gc", goarch),
|
||||
pkgs: map[string]*types.Package{},
|
||||
}
|
||||
}
|
||||
|
||||
// Import type-checks one imported package and memoises it. "unsafe" must
|
||||
// resolve to go/types' own package, never to the source in GOROOT/src/unsafe:
|
||||
// the source declares Sizeof and Offsetof as ordinary functions over
|
||||
// ArbitraryType, and checking against that signature rejects half the
|
||||
// unsafe arithmetic the gc compiler accepts, which is exactly the divergence
|
||||
// srcimporter guards against the same way.
|
||||
func (im *sourceImporter) Import(path string) (*types.Package, error) {
|
||||
if path == "unsafe" {
|
||||
return types.Unsafe, nil
|
||||
}
|
||||
if p, ok := im.pkgs[path]; ok {
|
||||
return p, nil
|
||||
}
|
||||
bp, err := im.ctxt.Import(path, "", 0)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
files, errs := im.parse(bp)
|
||||
if len(errs) > 0 {
|
||||
return nil, errors.New(errorList(errs))
|
||||
}
|
||||
pkg, _, _ := im.checkPackage(bp, files)
|
||||
im.pkgs[path] = pkg
|
||||
return pkg, nil
|
||||
}
|
||||
|
||||
// parse reads the build package's Go files. Import-level failures (no Go
|
||||
// files for the target, unreadable files) come back as errors, and the
|
||||
// type-check decides the rest.
|
||||
func (im *sourceImporter) parse(bp *build.Package) ([]*ast.File, []error) {
|
||||
if len(bp.GoFiles) == 0 {
|
||||
return nil, []error{fmt.Errorf("no Go source files for GOOS=%s GOARCH=%s", im.ctxt.GOOS, im.ctxt.GOARCH)}
|
||||
}
|
||||
var (
|
||||
files []*ast.File
|
||||
errs []error
|
||||
)
|
||||
for _, name := range bp.GoFiles {
|
||||
f, err := parser.ParseFile(im.fset, filepath.Join(bp.Dir, name), nil, parser.SkipObjectResolution)
|
||||
if err != nil {
|
||||
errs = append(errs, err)
|
||||
continue
|
||||
}
|
||||
files = append(files, f)
|
||||
}
|
||||
return files, errs
|
||||
}
|
||||
|
||||
// checkPackage type-checks one package's files with the importer's sizes,
|
||||
// recording every error: a header from a package that does not type-check
|
||||
// could silently mis-state an offset, so the caller refuses the header
|
||||
// rather than trusting it. The returned Defs map backs the root package's
|
||||
// emission walk; imports only need the checked package itself.
|
||||
func (im *sourceImporter) checkPackage(bp *build.Package, files []*ast.File) (*types.Package, *types.Info, []error) {
|
||||
var errs []error
|
||||
conf := &types.Config{
|
||||
Importer: im,
|
||||
Sizes: im.sizes,
|
||||
Error: func(err error) { errs = append(errs, err) },
|
||||
}
|
||||
info := &types.Info{Defs: map[*ast.Ident]types.Object{}}
|
||||
pkg, _ := conf.Check(bp.ImportPath, im.fset, files, info)
|
||||
return pkg, info, errs
|
||||
}
|
||||
|
||||
// asmhdrCache generates one go_asm.h per package directory and target
|
||||
// architecture under one temp root, for callers that assemble many files
|
||||
// (the corpus audit). Failures are cached too: a package that does not
|
||||
// type-check must not be re-checked once per file.
|
||||
type asmhdrCache struct {
|
||||
root string
|
||||
dirs map[string]string // "pkgDir\x00goos\x00goarch" -> directory holding go_asm.h
|
||||
errs map[string]error
|
||||
}
|
||||
|
||||
func newAsmhdrCache() (*asmhdrCache, error) {
|
||||
root, err := os.MkdirTemp("", "gasm-asmhdr")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &asmhdrCache{root: root, dirs: map[string]string{}, errs: map[string]error{}}, nil
|
||||
}
|
||||
|
||||
// dirFor returns the directory holding the generated go_asm.h for pkgDir
|
||||
// under goos and goarch, generating it on first use. An empty goos means
|
||||
// the ambient one, resolved here so that one package cannot generate twice
|
||||
// under an explicit and an implicit spelling of the same GOOS.
|
||||
func (c *asmhdrCache) dirFor(pkgDir, goos, goarch string) (string, error) {
|
||||
if goos == "" {
|
||||
goos = build.Default.GOOS
|
||||
}
|
||||
key := pkgDir + "\x00" + goos + "\x00" + goarch
|
||||
if dir, ok := c.dirs[key]; ok {
|
||||
return dir, nil
|
||||
}
|
||||
if err, ok := c.errs[key]; ok {
|
||||
return "", err
|
||||
}
|
||||
dir := filepath.Join(c.root, fmt.Sprintf("h%d_%s_%s", len(c.dirs), goos, goarch))
|
||||
if _, err := generateGoAsmHeader(pkgDir, goos, goarch, dir); err != nil {
|
||||
c.errs[key] = err
|
||||
return "", err
|
||||
}
|
||||
c.dirs[key] = dir
|
||||
return dir, nil
|
||||
}
|
||||
|
||||
// close removes the temp root.
|
||||
func (c *asmhdrCache) close() { os.RemoveAll(c.root) }
|
||||
|
||||
// ensureGoAsmHeader prepares the include directory a file that includes
|
||||
// go_asm.h needs: the generated header for the package in path's directory,
|
||||
// for the file's target GOOS and architecture. It reports a usage error
|
||||
// when the architecture cannot be determined, and passes through the
|
||||
// generator's diagnostics, which name the package.
|
||||
func ensureGoAsmHeader(path string, target arch.Arch, goos string, cache *asmhdrCache) (string, func(), error) {
|
||||
if path == "-" {
|
||||
return "", nil, errors.New("cannot generate go_asm.h for standard input (no package directory)")
|
||||
}
|
||||
if target == arch.Unknown {
|
||||
return "", nil, errors.New("a file that includes go_asm.h needs a target architecture: name the file _<arch>.s or pass -GOARCH")
|
||||
}
|
||||
if cache != nil {
|
||||
dir, err := cache.dirFor(filepath.Dir(path), goos, goarchName(target))
|
||||
return dir, func() {}, err
|
||||
}
|
||||
root, err := os.MkdirTemp("", "gasm-asmhdr")
|
||||
if err != nil {
|
||||
return "", nil, err
|
||||
}
|
||||
dir, err := generateGoAsmHeader(filepath.Dir(path), goos, goarchName(target), root)
|
||||
if err != nil {
|
||||
os.RemoveAll(root)
|
||||
return "", nil, err
|
||||
}
|
||||
return dir, func() { os.RemoveAll(root) }, nil
|
||||
}
|
||||
@@ -0,0 +1,476 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// writePkg lays out a minimal Go package in a temp directory.
|
||||
func writePkg(t *testing.T, files map[string]string) string {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
for name, src := range files {
|
||||
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
return dir
|
||||
}
|
||||
|
||||
// generateFor generates the header for dir and returns its text. An empty
|
||||
// goos means the ambient one.
|
||||
func generateFor(t *testing.T, dir, goos, goarch string) string {
|
||||
t.Helper()
|
||||
hdrDir, err := generateGoAsmHeader(dir, goos, goarch, t.TempDir())
|
||||
if err != nil {
|
||||
t.Fatalf("generateGoAsmHeader(%q, %s, %s): %v", dir, goos, goarch, err)
|
||||
}
|
||||
b, err := os.ReadFile(filepath.Join(hdrDir, "go_asm.h"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
func TestGenerateGoAsmHeaderShape(t *testing.T) {
|
||||
dir := writePkg(t, map[string]string{"sample.go": `package sample
|
||||
|
||||
const bufSize = 1024
|
||||
|
||||
const (
|
||||
a = iota * 8
|
||||
b
|
||||
c
|
||||
)
|
||||
|
||||
const (
|
||||
strConst = "hello"
|
||||
boolConst = true
|
||||
floatConst = 1.5
|
||||
_ = "the blank identifier is skipped"
|
||||
)
|
||||
|
||||
const shift = 1 << 20
|
||||
|
||||
type reader struct {
|
||||
r int64
|
||||
w int64
|
||||
_ [4]byte
|
||||
name string
|
||||
}
|
||||
|
||||
type scalar int
|
||||
|
||||
type aliased struct {
|
||||
k uint32
|
||||
v uint32
|
||||
}
|
||||
|
||||
type alias = aliased
|
||||
`})
|
||||
hdr := generateFor(t, dir, "", "amd64")
|
||||
want := []string{
|
||||
"#define const_bufSize 1024",
|
||||
// iota resolves through go/types, one define per name.
|
||||
"#define const_a 0",
|
||||
"#define const_b 8",
|
||||
"#define const_c 16",
|
||||
`#define const_strConst "hello"`,
|
||||
"#define const_boolConst true",
|
||||
// Floats are the toolchain's own skip, as are blank names.
|
||||
"#define const_shift 1048576",
|
||||
// The blank field still occupies its bytes: the pad after w runs to
|
||||
// the string's 8-byte alignment.
|
||||
"#define reader__size 40",
|
||||
"#define reader_r 0",
|
||||
"#define reader_w 8",
|
||||
"#define reader_name 24",
|
||||
// Non-struct named types carry no defines; aliases to structs do.
|
||||
"#define aliased__size 8",
|
||||
"#define aliased_k 0",
|
||||
"#define aliased_v 4",
|
||||
"#define alias__size 8",
|
||||
"#define alias_k 0",
|
||||
"#define alias_v 4",
|
||||
}
|
||||
for _, w := range want {
|
||||
if !strings.Contains(hdr, w+"\n") {
|
||||
t.Errorf("header misses %q\ngot:\n%s", w, hdr)
|
||||
}
|
||||
}
|
||||
for _, banned := range []string{"#define const_floatConst", "#define _ ", "#define scalar"} {
|
||||
if strings.Contains(hdr, banned) {
|
||||
t.Errorf("header must not carry %s\ngot:\n%s", banned, hdr)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestGenerateGoAsmHeaderPerArch(t *testing.T) {
|
||||
dir := writePkg(t, map[string]string{
|
||||
"common.go": `package perarch
|
||||
|
||||
type layout struct {
|
||||
a int32
|
||||
p uintptr
|
||||
}
|
||||
`,
|
||||
// The build-tagged file set is part of the contract: a per-arch
|
||||
// package is exactly how internal/cpu declares its layouts.
|
||||
"const_amd64.go": `//go:build amd64
|
||||
|
||||
package perarch
|
||||
|
||||
const flavour = 1
|
||||
`,
|
||||
"const_arm64.go": `//go:build arm64
|
||||
|
||||
package perarch
|
||||
|
||||
const flavour = 2
|
||||
`,
|
||||
})
|
||||
amd64 := generateFor(t, dir, "", "amd64")
|
||||
arm64 := generateFor(t, dir, "", "arm64")
|
||||
if !strings.Contains(amd64, "#define const_flavour 1\n") {
|
||||
t.Errorf("amd64 header misses const_flavour 1:\n%s", amd64)
|
||||
}
|
||||
if !strings.Contains(arm64, "#define const_flavour 2\n") {
|
||||
t.Errorf("arm64 header misses const_flavour 2:\n%s", arm64)
|
||||
}
|
||||
if strings.Contains(arm64, "#define const_flavour 1\n") {
|
||||
t.Errorf("arm64 header must not carry the amd64 file's value")
|
||||
}
|
||||
// SizesFor makes the layout the target's: uintptr is 4 bytes wide on
|
||||
// 386 and 8 on amd64, which must move p and grow the struct.
|
||||
if !strings.Contains(amd64, "#define layout__size 16\n") || !strings.Contains(amd64, "#define layout_p 8\n") {
|
||||
t.Errorf("amd64 layout wrong:\n%s", amd64)
|
||||
}
|
||||
w386 := generateFor(t, dir, "", "386")
|
||||
if !strings.Contains(w386, "#define layout__size 8\n") || !strings.Contains(w386, "#define layout_p 4\n") {
|
||||
t.Errorf("386 layout wrong:\n%s", w386)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGenerateGoAsmHeaderGOOS pins the GOOS half of the target: only the
|
||||
// platform's own files type-check into the header, which is why
|
||||
// sys_darwin_arm64.s cannot assemble against a linux-generated one.
|
||||
func TestGenerateGoAsmHeaderGOOS(t *testing.T) {
|
||||
dir := writePkg(t, map[string]string{
|
||||
"common.go": `package goosaware
|
||||
|
||||
type shared struct {
|
||||
a int32
|
||||
}
|
||||
`,
|
||||
"plat_darwin.go": `//go:build darwin
|
||||
|
||||
package goosaware
|
||||
|
||||
type platform struct {
|
||||
trampoline_numer int64
|
||||
}
|
||||
`,
|
||||
"plat_windows.go": `//go:build windows
|
||||
|
||||
package goosaware
|
||||
|
||||
type platform struct {
|
||||
callbackArgs__size int32
|
||||
}
|
||||
`,
|
||||
})
|
||||
darwin := generateFor(t, dir, "darwin", "arm64")
|
||||
if !strings.Contains(darwin, "#define platform__size 8\n") || !strings.Contains(darwin, "#define platform_trampoline_numer 0\n") {
|
||||
t.Errorf("darwin header misses the darwin layout:\n%s", darwin)
|
||||
}
|
||||
if strings.Contains(darwin, "callbackArgs") {
|
||||
t.Errorf("darwin header must not carry the windows layout:\n%s", darwin)
|
||||
}
|
||||
windows := generateFor(t, dir, "windows", "arm64")
|
||||
if !strings.Contains(windows, "#define platform_callbackArgs__size 0\n") {
|
||||
t.Errorf("windows header misses the windows layout:\n%s", windows)
|
||||
}
|
||||
if strings.Contains(windows, "trampoline_numer") {
|
||||
t.Errorf("windows header must not carry the darwin layout:\n%s", windows)
|
||||
}
|
||||
// The ambient GOOS is neither of the two, so only shared's defines are
|
||||
// emitted; the shared type keeps its layout there.
|
||||
ambient := generateFor(t, dir, "", "arm64")
|
||||
if !strings.Contains(ambient, "#define shared__size 4\n") {
|
||||
t.Errorf("ambient header misses the shared layout:\n%s", ambient)
|
||||
}
|
||||
if strings.Contains(ambient, "#define platform_") {
|
||||
t.Errorf("ambient header must not carry either platform layout:\n%s", ambient)
|
||||
}
|
||||
}
|
||||
|
||||
func TestGoosFromFilename(t *testing.T) {
|
||||
for path, want := range map[string]string{
|
||||
"/x/sys_darwin_arm64.s": "darwin",
|
||||
"/x/sys_windows_arm64.s": "windows",
|
||||
"/x/asm_linux_amd64.s": "linux",
|
||||
"/x/rt0_darwin_arm64.s": "darwin",
|
||||
"/x/vgetrandom_zos_s390x.s": "zos",
|
||||
"/x/rt0_js_wasm.s": "js",
|
||||
"/x/memmove_amd64.s": "",
|
||||
"/x/vlop_arm.s": "",
|
||||
"/x/stubs.s": "",
|
||||
} {
|
||||
if got := goosFromFilename(path); got != want {
|
||||
t.Errorf("goosFromFilename(%q) = %q, want %q", path, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestGenerateGoAsmHeaderErrors(t *testing.T) {
|
||||
t.Run("type error", func(t *testing.T) {
|
||||
dir := writePkg(t, map[string]string{"bad.go": `package bad
|
||||
|
||||
const x = undefinedIdent
|
||||
`})
|
||||
_, err := generateGoAsmHeader(dir, "", "amd64", t.TempDir())
|
||||
if err == nil {
|
||||
t.Fatal("generation must fail for a package that does not type-check")
|
||||
}
|
||||
if !strings.Contains(err.Error(), dir) {
|
||||
t.Errorf("error must name the package directory: %v", err)
|
||||
}
|
||||
if !strings.Contains(err.Error(), "type-check") {
|
||||
t.Errorf("error must say the package does not type-check: %v", err)
|
||||
}
|
||||
})
|
||||
t.Run("no go files", func(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
_, err := generateGoAsmHeader(dir, "", "amd64", t.TempDir())
|
||||
if err == nil {
|
||||
t.Fatal("generation must fail without Go files")
|
||||
}
|
||||
if !strings.Contains(err.Error(), dir) {
|
||||
t.Errorf("error must name the package directory: %v", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestNeedsGoAsmHeader(t *testing.T) {
|
||||
yes := "#include \"go_asm.h\"\n#include \"textflag.h\"\n"
|
||||
no := "#include \"textflag.h\"\n#include \"funcdata.h\"\n"
|
||||
if !needsGoAsmHeader(yes) {
|
||||
t.Error("needsGoAsmHeader(missing on a go_asm.h include)")
|
||||
}
|
||||
if needsGoAsmHeader(no) {
|
||||
t.Error("needsGoAsmHeader claims other headers need generation")
|
||||
}
|
||||
}
|
||||
|
||||
func TestGoAsmHeaderResolved(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
if goAsmHeaderResolved(dir, nil) {
|
||||
t.Error("resolved with no header anywhere")
|
||||
}
|
||||
other := t.TempDir()
|
||||
if goAsmHeaderResolved(dir, []string{other}) {
|
||||
t.Error("resolved with an empty -I directory")
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "go_asm.h"), nil, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !goAsmHeaderResolved(dir, nil) {
|
||||
t.Error("not resolved with the header in the package directory")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOtherGOOSFile(t *testing.T) {
|
||||
for path, want := range map[string]bool{
|
||||
"/x/sys_windows_amd64.s": true,
|
||||
"/x/rt0_js_wasm.s": true,
|
||||
"/x/sys_darwin_arm64.s": true,
|
||||
"/x/sys_linux_amd64.s": false,
|
||||
"/x/time_linux_amd64.s": false,
|
||||
"/x/memmove_amd64.s": false,
|
||||
"/x/generic.s": false,
|
||||
} {
|
||||
if got := otherGOOSFile(path); got != want {
|
||||
t.Errorf("otherGOOSFile(%q) = %v, want %v", path, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestRunCorpusAuditGoAsm covers the audit wiring end to end: a package
|
||||
// beside its kernel, the kernel living off the generated defines, and the
|
||||
// histogram recording a generation failure as its own reason.
|
||||
func TestRunCorpusAuditGoAsm(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
write := func(name, src string) {
|
||||
t.Helper()
|
||||
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
write("pkg.go", `package corpus
|
||||
|
||||
const pageSize = 4096
|
||||
|
||||
type header struct {
|
||||
magic uint64
|
||||
flags uint64
|
||||
}
|
||||
`)
|
||||
write("kern_amd64.s", "#include \"go_asm.h\"\nTEXT \xc2\xb7f(SB), NOSPLIT, $0-16\n\tMOVQ\t$const_pageSize, AX\n\tMOVQ\t$header__size, BX\n\tRET\n")
|
||||
// The defines live in the file's own package; a kernel in a directory
|
||||
// without Go files has no package to generate from.
|
||||
if err := os.MkdirAll(filepath.Join(dir, "sub"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
write(filepath.Join("sub", "lonely_arm64.s"), "#include \"go_asm.h\"\nTEXT \xc2\xb7g(SB), NOSPLIT, $0-0\n\tRET\n")
|
||||
|
||||
stats, err := runCorpusAudit(dir, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("runCorpusAudit: %v", err)
|
||||
}
|
||||
get := func(name string) *corpusTally {
|
||||
for i, tg := range stats.targets {
|
||||
if tg.name == name {
|
||||
return stats.tallies[i]
|
||||
}
|
||||
}
|
||||
t.Fatalf("no tally for %s", name)
|
||||
return nil
|
||||
}
|
||||
if a := get("amd64"); a.attempted != 1 || a.assembled != 1 {
|
||||
t.Errorf("amd64 = %d/%d, want 1/1", a.assembled, a.attempted)
|
||||
}
|
||||
// lonely_arm64.s is an arm64 file whose package cannot be generated.
|
||||
if a := get("arm64"); a.attempted != 1 || a.assembled != 0 {
|
||||
t.Errorf("arm64 = %d/%d, want 0/1", a.assembled, a.attempted)
|
||||
}
|
||||
if r := get("arm64").reasons["go_asm.h generation failed"]; r != 1 {
|
||||
t.Errorf("arm64 go_asm.h failure count = %d, want 1", r)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRunCorpusAuditGOOS covers the filename-derived GOOS end to end: a
|
||||
// kernel whose name names darwin must have its header type-checked with
|
||||
// GOOS=darwin, so the darwin-only constant it offsets with is defined. The
|
||||
// operand mirrors sys_darwin_arm64.s's trampoline, where a missing define
|
||||
// leaves an unexpanded symbol in the offset and fails.
|
||||
func TestRunCorpusAuditGOOS(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
write := func(name, src string) {
|
||||
t.Helper()
|
||||
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
write("pkg.go", "package corpus\n")
|
||||
write("plat_darwin.go", "//go:build darwin\n\npackage corpus\n\nconst trampolineNumer = 8\n")
|
||||
write("kern_darwin_arm64.s", "#include \"go_asm.h\"\n"+
|
||||
"GLOBL timebase<>(SB), NOPTR, $16\n"+
|
||||
"TEXT \xc2\xb7g(SB), NOSPLIT, $0-0\n"+
|
||||
"\tMOVD\ttimebase<>+const_trampolineNumer(SB), R0\n"+
|
||||
"\tRET\n")
|
||||
|
||||
stats, err := runCorpusAudit(dir, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("runCorpusAudit: %v", err)
|
||||
}
|
||||
var arm *corpusTally
|
||||
for i, tg := range stats.targets {
|
||||
if tg.name == "arm64" {
|
||||
arm = stats.tallies[i]
|
||||
}
|
||||
}
|
||||
if arm == nil {
|
||||
t.Fatal("no arm64 tally")
|
||||
}
|
||||
if arm.attempted != 1 || arm.assembled != 1 {
|
||||
t.Errorf("arm64 = %d/%d, want 1/1; reasons: %v", arm.assembled, arm.attempted, arm.reasons)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRunCorpusAuditBuildConstraint covers the //go:build classification end
|
||||
// to end: a generic-named file whose constraint admits one target is
|
||||
// attempted there alone (cpu_x86.s on amd64), and a file whose constraint
|
||||
// admits none of the four targets is never attempted (the msan and
|
||||
// goexperiment trees).
|
||||
func TestRunCorpusAuditBuildConstraint(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
write := func(name, src string) {
|
||||
t.Helper()
|
||||
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
write("x86.s", "//go:build 386 || amd64\n\nTEXT \xc2\xb7f(SB), NOSPLIT, $0\n\tRET\n")
|
||||
write("racey.s", "//go:build race\n\nTEXT \xc2\xb7r(SB), NOSPLIT, $0\n\tRET\n")
|
||||
write("plain.s", "TEXT \xc2\xb7p(SB), NOSPLIT, $0\n\tRET\n")
|
||||
|
||||
stats, err := runCorpusAudit(dir, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("runCorpusAudit: %v", err)
|
||||
}
|
||||
tally := func(name string) *corpusTally {
|
||||
for i, tg := range stats.targets {
|
||||
if tg.name == name {
|
||||
return stats.tallies[i]
|
||||
}
|
||||
}
|
||||
t.Fatalf("no tally for %s", name)
|
||||
return nil
|
||||
}
|
||||
if stats.narrowed != 1 || stats.excluded != 1 || stats.generic != 1 {
|
||||
t.Errorf("buckets = narrowed %d, excluded %d, generic %d; want 1, 1, 1", stats.narrowed, stats.excluded, stats.generic)
|
||||
}
|
||||
if a := tally("amd64"); a.attempted != 2 || a.assembled != 2 {
|
||||
t.Errorf("amd64 = %d/%d, want 2/2 (x86.s and plain.s)", a.assembled, a.attempted)
|
||||
}
|
||||
for _, name := range []string{"arm64", "riscv64", "loong64"} {
|
||||
if a := tally(name); a.attempted != 1 || a.assembled != 1 {
|
||||
t.Errorf("%s = %d/%d, want 1/1 (plain.s only)", name, a.assembled, a.attempted)
|
||||
}
|
||||
}
|
||||
if stats.full != 2 {
|
||||
t.Errorf("full = %d, want 2 (x86.s over its one target, plain.s over all four)", stats.full)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGenerateGoAsmHeaderRuntime pins the generator against the real thing:
|
||||
// the runtime package of the ambient toolchain, whose header the toolchain's
|
||||
// own -asmhdr output was sampled from. Skipped in short mode: it type-checks
|
||||
// the whole package. The GOROOT comes from the go command itself, so the
|
||||
// test follows whatever toolchain the host provides.
|
||||
func TestGenerateGoAsmHeaderRuntime(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("type-checks the whole runtime package")
|
||||
}
|
||||
out, err := exec.Command("go", "env", "GOROOT").Output()
|
||||
if err != nil {
|
||||
t.Skipf("no Go toolchain: %v", err)
|
||||
}
|
||||
runtimeDir := filepath.Join(strings.TrimSpace(string(out)), "src", "runtime")
|
||||
dir, err := generateGoAsmHeader(runtimeDir, "", "amd64", t.TempDir())
|
||||
if err != nil {
|
||||
t.Fatalf("generateGoAsmHeader(runtime): %v", err)
|
||||
}
|
||||
b, err := os.ReadFile(dir + "/go_asm.h")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
hdr := string(b)
|
||||
for _, want := range []string{
|
||||
"#define const_hashSize 8\n",
|
||||
"#define const_avxSupported 1\n",
|
||||
"#define const_pageSize 8192\n",
|
||||
"#define g_stackguard0 16\n",
|
||||
"#define m__size ",
|
||||
} {
|
||||
if !strings.Contains(hdr, want) {
|
||||
t.Errorf("runtime header misses %q", want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,912 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"go/build/constraint"
|
||||
"maps"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"runtime"
|
||||
"slices"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// cmdAuditInstructions cross-checks a gasm encoder against the Go toolchain's
|
||||
// own assembler, probed black-box: every mnemonic in the gasm table is offered
|
||||
// to go tool asm in its bare form, and a mnemonic counts as known to Go when
|
||||
// the error is anything but "unrecognized instruction" (a wrong-shape error
|
||||
// still proves the mnemonic exists in Go's tables). The audit answers three
|
||||
// questions at a glance:
|
||||
//
|
||||
// - which mnemonics gasm can encode that go tool asm does not know
|
||||
// (superset encodings, usable only through the gasm goobj path);
|
||||
// - which mnemonics the architecture table knows but the encoder cannot
|
||||
// emit yet (the implementation backlog);
|
||||
// - which mnemonics go tool asm knows that gasm cannot encode (feature
|
||||
// gaps).
|
||||
//
|
||||
// The amd64 derived families (Jcc, CMOVcc, SETcc) exist on both sides by
|
||||
// construction and are excluded from the diff; the other architectures list
|
||||
// their conditional branches outright.
|
||||
func cmdAuditInstructions(args []string) error {
|
||||
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [--list] [-I dir] [amd64|arm64|riscv64|loong64]", `
|
||||
Compare the gasm encoder for the given architecture (default amd64) against
|
||||
go tool asm and print the diff: superset encodings (gasm-only, shippable via
|
||||
gasm asm --format goobj) and known-but-unencodable names (the backlog). The
|
||||
Go side is probed black-box one bare mnemonic at a time, so the audit tracks
|
||||
whatever toolchain `+"`go env GOROOT`"+` provides; the gasm side answers from
|
||||
the encoder table on amd64 and from trial assembly over a battery of operand
|
||||
shapes elsewhere. Names go tool asm knows and gasm does not cannot be
|
||||
enumerated by probing, because Go's table is visible only through names
|
||||
already in the gasm table; the report closes with a note saying so.
|
||||
|
||||
With --corpus the audit changes shape: it assembles every .s file under the
|
||||
given directory (default GOROOT/src) with the gasm encoder only, no
|
||||
toolchain probing. A file whose name carries a recognisable _arch suffix is
|
||||
attempted for that architecture; a file without one is attempted for all
|
||||
four, exactly as a GOARCH build would compile it. The report gives the
|
||||
per-architecture pass rates and the most common failure reasons, which drive
|
||||
the encodability backlog by frequency rather than by table order. With
|
||||
-list the report also prints every failing file with its reason, per
|
||||
architecture.
|
||||
`)
|
||||
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
|
||||
list := fs.Bool("list", false, "with --corpus, list every failing file with its reason, per architecture")
|
||||
var dirs includeDirs
|
||||
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return err
|
||||
}
|
||||
if *corpus {
|
||||
return cmdAuditCorpus(fs.Args(), dirs, *list)
|
||||
}
|
||||
archName := "amd64"
|
||||
switch n := len(fs.Args()); {
|
||||
case n > 1:
|
||||
return &usageError{fmt.Errorf("audit-instructions takes at most one architecture argument")}
|
||||
case n == 1:
|
||||
archName = strings.ToLower(fs.Arg(0))
|
||||
}
|
||||
a, err := auditArch(archName)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
tab := arch.ForArch(a)
|
||||
var names []string
|
||||
seen := map[string]bool{}
|
||||
for _, in := range tab.Instructions() {
|
||||
name := strings.ToUpper(in.Name)
|
||||
if a == arch.AMD64 && derivedFamily(name) || seen[name] {
|
||||
continue
|
||||
}
|
||||
seen[name] = true
|
||||
names = append(names, name)
|
||||
}
|
||||
|
||||
goKnown, err := probeGoAsm(goarchName(a), names)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
var superset, backlog, shared []string
|
||||
for _, name := range names {
|
||||
switch {
|
||||
case !gasmEncodable(a, name):
|
||||
backlog = append(backlog, name)
|
||||
case !goKnown[name]:
|
||||
superset = append(superset, name)
|
||||
default:
|
||||
shared = append(shared, name)
|
||||
}
|
||||
}
|
||||
// GO-ONLY is not enumerable by probing: Go's table is only visible
|
||||
// through names we already know, so nothing can be reported there.
|
||||
|
||||
slices.Sort(superset)
|
||||
slices.Sort(backlog)
|
||||
slices.Sort(shared)
|
||||
|
||||
w := os.Stdout
|
||||
fmt.Fprintf(w, "gasm table (%s, families excluded): %d mnemonics\n", archName, len(names))
|
||||
fmt.Fprintf(w, "gasm encodable: %d go tool asm recognised: %d\n", len(shared)+len(superset), countTrue(goKnown))
|
||||
fmt.Fprintf(w, "shared: %d\n", len(shared))
|
||||
fmt.Fprintf(w, "\nSuperset encodings (gasm-only; ship via gasm asm --format goobj):\n")
|
||||
for _, n := range superset {
|
||||
fmt.Fprintf(w, " %s\n", n)
|
||||
}
|
||||
fmt.Fprintf(w, "\nKnown but not encodable (backlog):\n")
|
||||
for _, n := range backlog {
|
||||
fmt.Fprintf(w, " %s\n", n)
|
||||
}
|
||||
fmt.Fprintf(w, "\nGo-only names cannot be enumerated by probing; extend the gasm\n")
|
||||
fmt.Fprintf(w, "table from the Go release notes when a new instruction family ships.\n")
|
||||
return nil
|
||||
}
|
||||
|
||||
// auditArch resolves the audit's architecture argument.
|
||||
func auditArch(name string) (arch.Arch, error) {
|
||||
switch strings.ToLower(name) {
|
||||
case "amd64":
|
||||
return arch.AMD64, nil
|
||||
case "arm64":
|
||||
return arch.ARM64, nil
|
||||
case "riscv64", "riscv":
|
||||
return arch.RISCV, nil
|
||||
case "loong64", "loong":
|
||||
return arch.LOONG64, nil
|
||||
}
|
||||
return arch.Unknown, &usageError{fmt.Errorf("unknown architecture %q: want amd64, arm64, riscv64 or loong64", name)}
|
||||
}
|
||||
|
||||
// goarchName maps an arch identifier onto its GOARCH spelling.
|
||||
func goarchName(a arch.Arch) string {
|
||||
switch a {
|
||||
case arch.ARM64:
|
||||
return "arm64"
|
||||
case arch.RISCV:
|
||||
return "riscv64"
|
||||
case arch.LOONG64:
|
||||
return "loong64"
|
||||
}
|
||||
return "amd64"
|
||||
}
|
||||
|
||||
func countTrue(m map[string]bool) int {
|
||||
n := 0
|
||||
for _, v := range m {
|
||||
if v {
|
||||
n++
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// derivedFamily reports whether a mnemonic belongs to a family both
|
||||
// assemblers construct from condition codes rather than list exhaustively
|
||||
// (JEQ/CMOVLGT/SETNE and friends). Such names never probe cleanly, so
|
||||
// including them in the diff would be noise. amd64 only: the other
|
||||
// architectures list their conditional branches outright.
|
||||
func derivedFamily(name string) bool {
|
||||
if strings.HasPrefix(name, "J") && name != "JMP" && name != "JMPQ" {
|
||||
return true
|
||||
}
|
||||
if strings.HasPrefix(name, "CMOV") || strings.HasPrefix(name, "SET") {
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
var unrecognizedRe = regexp.MustCompile(`unrecognized instruction`)
|
||||
|
||||
// probeGoAsm feeds every mnemonic to go tool asm in one generated file and
|
||||
// classifies the diagnostics. "Unrecognized instruction" is a parse-stage
|
||||
// verdict on the mnemonic alone, so a single bare-instruction probe per
|
||||
// mnemonic decides recognition; the combined file still reports every line's
|
||||
// error even when others fail.
|
||||
func probeGoAsm(goarch string, names []string) (map[string]bool, error) {
|
||||
dir, err := os.MkdirTemp("", "gasm-audit")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
|
||||
var sb strings.Builder
|
||||
sb.WriteString("TEXT ·probe(SB), 4, $0\n\tRET\n")
|
||||
lineMnemonic := map[int]string{}
|
||||
line := 3
|
||||
for _, name := range names {
|
||||
fmt.Fprintf(&sb, "TEXT ·p%s%d(SB), 4, $0\n", sanitize(name), line)
|
||||
sb.WriteString("\t" + name + "\n\tRET\n")
|
||||
lineMnemonic[line+1] = name // the instruction line, after TEXT
|
||||
line += 3
|
||||
}
|
||||
probePath := filepath.Join(dir, "probe.s")
|
||||
if err := os.WriteFile(probePath, []byte(sb.String()), 0o644); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
toolDir, err := exec.Command("go", "env", "GOTOOLDIR").Output()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("go env GOTOOLDIR: %w", err)
|
||||
}
|
||||
asmBin := filepath.Join(strings.TrimSpace(string(toolDir)), "asm")
|
||||
if _, err := os.Stat(asmBin); err != nil {
|
||||
return nil, fmt.Errorf("go tool asm not found at %s", asmBin)
|
||||
}
|
||||
cmd := exec.Command(asmBin, "-p", "probe", "-o", filepath.Join(dir, "probe.o"), probePath)
|
||||
cmd.Env = append(os.Environ(), "GOARCH="+goarch, "GOOS="+runtime.GOOS)
|
||||
out, _ := cmd.CombinedOutput()
|
||||
// The expected failure mode is a non-zero exit with compiler diagnostics
|
||||
// on stdout; empty output means the probe broke at the exec level (a
|
||||
// killed child, a tool that would not start), and seeding every name as
|
||||
// recognized on that silence would fake a clean audit.
|
||||
if len(out) == 0 {
|
||||
return nil, fmt.Errorf("go tool asm probe for GOARCH=%s produced no output", goarch)
|
||||
}
|
||||
|
||||
result := map[string]bool{}
|
||||
for _, name := range names {
|
||||
result[name] = true // no news = the name parsed fine
|
||||
}
|
||||
reParse := regexp.MustCompile(`probe\.s:(\d+):`)
|
||||
for l := range strings.SplitSeq(string(out), "\n") {
|
||||
m := reParse.FindStringSubmatch(l)
|
||||
if m == nil {
|
||||
continue
|
||||
}
|
||||
lineNo, err := strconv.Atoi(m[1])
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if name, ok := lineMnemonic[lineNo]; ok && unrecognizedRe.MatchString(l) {
|
||||
result[name] = false
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// probeShapes lists representative operand shapes for the encodability
|
||||
// probe. The assemblers report an unknown mnemonic and a known mnemonic
|
||||
// with no supported form alike ("unsupported <arch> instruction"), so only
|
||||
// a shape that assembles cleanly counts, and the backlog over-approximates:
|
||||
// a name whose real forms the battery misses lands there. amd64 keeps its
|
||||
// exact table-driven check.
|
||||
func probeShapes(a arch.Arch) []string {
|
||||
switch a {
|
||||
case arch.ARM64:
|
||||
return []string{
|
||||
"X0, X1, X2", "X0, X1", "X0", "$1, X0", "X0, (X1)", "(X0), X1",
|
||||
"X0, (X1, 8)", "(SP), X0", "F0, F1, F2", "F0, F1", "F0",
|
||||
"V0.B16, V1.B16, V2.B16", "p2", "X0, p2", "X0, X1, p2",
|
||||
// The conditional select family spells the condition first
|
||||
// and takes R register spellings.
|
||||
"EQ, R0, R1, R2", "EQ, R0, R1", "EQ, R0",
|
||||
"GE, F0, F1, F2", "NE, F0, F1, $0",
|
||||
// Pairs, acquire/release and exclusive atomics, LSE-AL forms.
|
||||
"(R0), R1", "R0, (R1)", "R1, (R2), R3", "(R2, R3), 8(R1)",
|
||||
"8(R1), (R2, R3)", "R1, R2, (R3)", "(R0)",
|
||||
// System operations and their register/operand names.
|
||||
"$4, R1, p2", "$35943", "$1", "$1, SPSel", "SPSel, R0",
|
||||
"IVAC, R0", "(R0), PLDL1KEEP", "R1, R2, R3, R4",
|
||||
// SIMD element, structure and literal-pool forms.
|
||||
"(R0), [V1.B16]", "[V1.B16], (R0)", "V13.S[0], R1",
|
||||
"R1, V2.B[3]", "$4, V1.B16, V2.B16", "V1.B16, (R0)",
|
||||
"(R0), V1.B16", "",
|
||||
// The spellings GOROOT's own kernels use, from the
|
||||
// differential kernels this table was proven against.
|
||||
"R0, p2", "R0, R1", "F0, F1, F2, F3", "$4, V1.B16, V2.B16, V3.B16, V4.B16",
|
||||
"(R0), [V0.B8, V1.B8, V2.B8, V3.B8]", "$1, $2, V1",
|
||||
"R0, R1, p2", "p2, R1", "$1234, R1", "DCZID_EL0, R1",
|
||||
"$0", "R1, $4, EQ", "$33, R1, $25, R2", "$4, R1, p2",
|
||||
"$4, V1.B8, V2.B8, V3.B8", "$63, V1.D2, V2.D2, V3.D2",
|
||||
"V1.B16, [V2.B16], V3.B16", "V1.B8, [V2.B16, V3.B16], V4.B8",
|
||||
"$4, V1.B16, V2.B16, V3.B16", "$15, V1", "V1, V2, p2",
|
||||
"R0, R1, $1, $4, p2",
|
||||
// The landing-pad kind, the compiler's PCDATA
|
||||
// bookkeeping and the four-operand bitfield
|
||||
// insert/extract family, as the toolchain's own
|
||||
// testdata spells them.
|
||||
"C", "$1, $0", "$0, R1, $1, R2",
|
||||
}
|
||||
case arch.RISCV:
|
||||
return []string{
|
||||
"X5, X6, X7", "X5, X6", "X5", "$1, X5", "X5, (X6)", "$1, X5, X6",
|
||||
"(X5), X6", "F0, F1, F2", "F0, F1", "p2", "X1, p2", "X0, p2",
|
||||
"X5, X6, p2", "p2(SB)",
|
||||
// AMO atomics: destination, base, source.
|
||||
"R5, (R4), R6", "X5, (X4), X6",
|
||||
// Segment stores take the first vector register aligned
|
||||
// to the segment count, as the toolchain requires.
|
||||
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
|
||||
// The FP multiply-add family takes four registers.
|
||||
"F0, F1, F2, F3",
|
||||
// The RVV slice: register, vector-register and vtype forms.
|
||||
"V1, V2, V3", "V1, X5, V2", "V1", "V1, (X5)", "(X5), V1",
|
||||
"$15, V1", "$15", "V1, V2", "V1, X5",
|
||||
"X5, X6, p2", "R5, R6, p2",
|
||||
"X5, E8, M8, TA, MA, X6", "$4, E32, M1, TA, MA, X1",
|
||||
"(X5), X6, V1, V2",
|
||||
// The CSR immediate forms the toolchain's testdata spells:
|
||||
// immediate, CSR name, destination.
|
||||
"$2, TIME, X5",
|
||||
"",
|
||||
}
|
||||
case arch.LOONG64:
|
||||
return []string{
|
||||
"R4, R5, R6", "R4, R5", "R4", "$1, R4", "R4, (R5)", "(R4), R5",
|
||||
"F0, F1, F2", "F0, F1", "p2", "R1, p2", "R4, p2",
|
||||
"$1, R4, R5, R6", "$65536, R4", "R4, R5, p2", "p2(SB)",
|
||||
// AMO atomics: destination, base, source.
|
||||
"R5, (R4), R6", "X5, (X4), X6",
|
||||
// Segment stores take the first vector register aligned
|
||||
// to the segment count, as the toolchain requires.
|
||||
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
|
||||
// The LSX and LASX banks share the 5-bit numbering with F.
|
||||
"V1, V2, V3", "X1, X2, X3", "V1, V2", "X1, X2", "V1", "X1",
|
||||
// The vector compare-to-flag forms land in an FCC register.
|
||||
"V1, FCC0", "X1, FCC0",
|
||||
// The compiler's bookkeeping pair and the raw spellings the
|
||||
// toolchain's own testdata carries: JIRL rd, rj, offset (the
|
||||
// form RET lowers to), the prefetch with a 32-bit address and
|
||||
// hint, and the byte-shuffle quads.
|
||||
"$1, $0", "R1, R5, 0", "0(R7), $5, $0", "(R7), $5, $0",
|
||||
"V1, V2, V3, V4", "X1, X2, X3, X4",
|
||||
"",
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// gasmEncodable reports whether the gasm encoder for a can emit the
|
||||
// mnemonic, decided by trial assembly over the shape battery.
|
||||
func gasmEncodable(a arch.Arch, name string) bool {
|
||||
switch a {
|
||||
case arch.ARM64, arch.RISCV, arch.LOONG64:
|
||||
default:
|
||||
return asm.Encodable(name)
|
||||
}
|
||||
for _, shape := range probeShapes(a) {
|
||||
if gasmAssembles(a, name, shape) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// gasmAssembles reports whether a one-instruction probe file containing name
|
||||
// with the given operand shape assembles without error.
|
||||
func gasmAssembles(a arch.Arch, name, shape string) bool {
|
||||
src := "TEXT ·p(SB), NOSPLIT, $0\n\t" + name
|
||||
if shape != "" {
|
||||
src += " " + shape
|
||||
}
|
||||
src += "\n\tRET\np2:\n\tRET\n"
|
||||
f, errs := parser.Parse("probe.s", src)
|
||||
if len(errs) > 0 {
|
||||
return false
|
||||
}
|
||||
var err error
|
||||
switch a {
|
||||
case arch.ARM64:
|
||||
_, err = asm.AssembleFileARM64(f)
|
||||
case arch.RISCV:
|
||||
_, err = asm.AssembleFileRISCV(f)
|
||||
case arch.LOONG64:
|
||||
_, err = asm.AssembleFileLOONG64(f)
|
||||
}
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// sanitize makes a mnemonic safe for use in a Go symbol name.
|
||||
func sanitize(name string) string {
|
||||
return strings.NewReplacer(".", "_", "$", "_").Replace(name)
|
||||
}
|
||||
|
||||
// --- corpus audit -----------------------------------------------------------
|
||||
|
||||
// corpusTarget is one architecture row of the corpus report.
|
||||
type corpusTarget struct {
|
||||
a arch.Arch
|
||||
name string
|
||||
}
|
||||
|
||||
// corpusTally accumulates one architecture's attempts over the corpus.
|
||||
type corpusTally struct {
|
||||
attempted int
|
||||
assembled int
|
||||
reasons map[string]int // failure reason → count
|
||||
example map[string]string // failure reason → one representative file
|
||||
fails []corpusFailure // every failure, in file order, for --list
|
||||
}
|
||||
|
||||
// corpusFailure is one failed attempt, recorded for the --list report.
|
||||
type corpusFailure struct {
|
||||
path string
|
||||
reason string
|
||||
detail string
|
||||
}
|
||||
|
||||
func (t *corpusTally) fail(path string, err error) {
|
||||
reason := corpusReason(err)
|
||||
t.reasons[reason]++
|
||||
if t.example[reason] == "" {
|
||||
t.example[reason] = path
|
||||
}
|
||||
t.fails = append(t.fails, corpusFailure{path: path, reason: reason, detail: firstLine(err.Error())})
|
||||
}
|
||||
|
||||
// cmdAuditCorpus implements audit-instructions --corpus. The include
|
||||
// directories carry #include resolution over a corpus whose files refer to
|
||||
// headers such as GOROOT/pkg/include, the same -I a toolchain comparison
|
||||
// needs.
|
||||
func cmdAuditCorpus(args []string, dirs includeDirs, list bool) error {
|
||||
if len(args) > 1 {
|
||||
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
|
||||
}
|
||||
root := ""
|
||||
if len(args) == 1 {
|
||||
root = args[0]
|
||||
} else {
|
||||
out, err := exec.Command("go", "env", "GOROOT").Output()
|
||||
if err != nil {
|
||||
return fmt.Errorf("locate GOROOT: %w", err)
|
||||
}
|
||||
root = filepath.Join(strings.TrimSpace(string(out)), "src")
|
||||
}
|
||||
// The toolchain's shipped headers (funcdata.h and friends) define the
|
||||
// macros GOROOT files include; a corpus audit measures those files, so
|
||||
// the header directory joins the search path automatically. go_asm.h
|
||||
// is compiler-generated per package, so it is not resolved from here:
|
||||
// files that include it get one generated per target architecture,
|
||||
// which runCorpusAudit arranges.
|
||||
if out, err := exec.Command("go", "env", "GOROOT").Output(); err == nil {
|
||||
pkgInclude := filepath.Join(strings.TrimSpace(string(out)), "pkg", "include")
|
||||
if fi, err := os.Stat(pkgInclude); err == nil && fi.IsDir() {
|
||||
seen := false
|
||||
for _, d := range dirs {
|
||||
if d == pkgInclude {
|
||||
seen = true
|
||||
}
|
||||
}
|
||||
if !seen {
|
||||
dirs = append(dirs, pkgInclude)
|
||||
}
|
||||
}
|
||||
}
|
||||
stats, err := runCorpusAudit(root, dirs)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printCorpusStats(stats, list)
|
||||
return nil
|
||||
}
|
||||
|
||||
// corpusStats is the outcome of one corpus audit run.
|
||||
type corpusStats struct {
|
||||
root string
|
||||
files int
|
||||
generic int // files attempted for all four architectures
|
||||
narrowed int // files whose //go:build admits a proper subset of the four
|
||||
excluded int // files whose //go:build admits none of the four: never compiled
|
||||
otherPort int // files named for another Go port: never attempted
|
||||
full int // files that assembled for every applicable target architecture
|
||||
targets []corpusTarget
|
||||
tallies []*corpusTally
|
||||
}
|
||||
|
||||
// runCorpusAudit assembles every .s file under root and returns the stats.
|
||||
// goPortSuffixes lists every architecture the Go project ports to. A file
|
||||
// named for one of them belongs to that port's build, not to the generic
|
||||
// set, even when gasm does not support the architecture.
|
||||
var goPortSuffixes = []string{
|
||||
"386", "amd64", "arm", "arm64", "loong64", "mips", "mips64",
|
||||
"mips64le", "mipsle", "mips64x", "mipsx", "ppc64", "ppc64le",
|
||||
"ppc64x", "riscv", "riscv64", "s390x", "wasm",
|
||||
}
|
||||
|
||||
// otherPortFile reports whether the file belongs to a build no supported
|
||||
// target ever compiles: either its name carries a Go-architecture suffix
|
||||
// gasm does not support, or, for a file with no architecture suffix at all,
|
||||
// it names another GOOS, which go/build drops from the file set
|
||||
// (rt0_js_wasm.s is a javascript build, not a generic one).
|
||||
func otherPortFile(path string) bool {
|
||||
if otherGOOSFile(path) {
|
||||
return true
|
||||
}
|
||||
base := path
|
||||
if i := strings.LastIndexByte(base, '/'); i >= 0 {
|
||||
base = base[i+1:]
|
||||
}
|
||||
for _, sfx := range goPortSuffixes {
|
||||
if strings.HasSuffix(base, "_"+sfx+".s") {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// goOSNames are the GOOS values go/build recognises in file names.
|
||||
var goOSNames = map[string]bool{
|
||||
"aix": true, "android": true, "darwin": true, "dragonfly": true,
|
||||
"freebsd": true, "hurd": true, "illumos": true, "ios": true,
|
||||
"js": true, "linux": true, "nacl": true, "netbsd": true,
|
||||
"openbsd": true, "plan9": true, "solaris": true, "wasip1": true,
|
||||
"windows": true, "zos": true,
|
||||
}
|
||||
|
||||
// resolveGOOS validates a -GOOS flag value, mirroring the architecture
|
||||
// check's surface: a usage error naming what the tool accepts.
|
||||
func resolveGOOS(name string) (string, error) {
|
||||
lower := strings.ToLower(name)
|
||||
if goOSNames[lower] {
|
||||
return lower, nil
|
||||
}
|
||||
return "", &usageError{fmt.Errorf("unknown GOOS %q: want one of %s", name, strings.Join(slices.Sorted(maps.Keys(goOSNames)), ", "))}
|
||||
}
|
||||
|
||||
// goosFromFilename returns the GOOS the file's name carries, by go/build's
|
||||
// goodOSArchFile rule: the GOOS segment sits last, or last before the
|
||||
// architecture segment (sys_darwin_arm64.s, vlop_arm.s carries none). An
|
||||
// empty result means the name names no GOOS and the ambient one applies.
|
||||
func goosFromFilename(path string) string {
|
||||
base := path
|
||||
if i := strings.LastIndexByte(base, '/'); i >= 0 {
|
||||
base = base[i+1:]
|
||||
}
|
||||
base = strings.TrimSuffix(base, ".s")
|
||||
// go/build ignores everything before the first underscore, so a GOOS
|
||||
// segment is only ever looked for from there on.
|
||||
i := strings.IndexByte(base, '_')
|
||||
if i < 0 {
|
||||
return ""
|
||||
}
|
||||
segs := strings.Split(base[i:], "_")
|
||||
if n := len(segs); n >= 2 && goOSNames[segs[n-2]] && slices.Contains(goPortSuffixes, segs[n-1]) {
|
||||
return segs[n-2]
|
||||
}
|
||||
if goOSNames[segs[len(segs)-1]] {
|
||||
return segs[len(segs)-1]
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// otherGOOSFile reports whether the file's name names a GOOS other than the
|
||||
// host's, by go/build's file-name rules.
|
||||
func otherGOOSFile(path string) bool {
|
||||
base := path
|
||||
if i := strings.LastIndexByte(base, '/'); i >= 0 {
|
||||
base = base[i+1:]
|
||||
}
|
||||
for seg := range strings.SplitSeq(strings.TrimSuffix(base, ".s"), "_") {
|
||||
if goOSNames[seg] && seg != runtime.GOOS {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// buildConstraint returns the file's leading //go:build expression, or nil
|
||||
// when the file carries none. The constraint governs the same header block
|
||||
// go/build reads: blank lines and comments may precede it, and the first
|
||||
// line that is neither ends the block. A constraint that does not parse
|
||||
// narrows nothing, so the file stays in the attempted set: the audit must
|
||||
// never exclude a file the toolchain would compile.
|
||||
func buildConstraint(src string) constraint.Expr {
|
||||
for line := range strings.SplitSeq(src, "\n") {
|
||||
t := strings.TrimSpace(line)
|
||||
switch {
|
||||
case t == "":
|
||||
continue
|
||||
case strings.HasPrefix(t, "//"):
|
||||
if constraint.IsGoBuild(t) {
|
||||
e, err := constraint.Parse(t)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
return e
|
||||
}
|
||||
continue
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// unixOS is go/build's unixOS set: the GOOSes the unix build tag admits.
|
||||
var unixOS = map[string]bool{
|
||||
"aix": true, "android": true, "darwin": true, "dragonfly": true,
|
||||
"freebsd": true, "hurd": true, "illumos": true, "ios": true,
|
||||
"linux": true, "netbsd": true, "openbsd": true, "solaris": true,
|
||||
}
|
||||
|
||||
// constraintTags answers the build tags a plain `go build` sets for a
|
||||
// target: the GOOS and GOARCH, gc, and unix on the unix-like GOOSes. No
|
||||
// experiment, sanitiser or cgo tag is ever true: the audit models the
|
||||
// default build, and no GOROOT assembly file's constraint hinges on cgo.
|
||||
func constraintTags(goarch, goos string) func(string) bool {
|
||||
return func(tag string) bool {
|
||||
switch tag {
|
||||
case goarch, goos, "gc":
|
||||
return true
|
||||
case "unix":
|
||||
return unixOS[goos]
|
||||
}
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
|
||||
files, err := asmFiles(root)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
targets := []corpusTarget{
|
||||
{arch.AMD64, "amd64"},
|
||||
{arch.ARM64, "arm64"},
|
||||
{arch.RISCV, "riscv64"},
|
||||
{arch.LOONG64, "loong64"},
|
||||
}
|
||||
tallies := make([]*corpusTally, len(targets))
|
||||
for i := range tallies {
|
||||
tallies[i] = &corpusTally{reasons: map[string]int{}, example: map[string]string{}}
|
||||
}
|
||||
// full is the north-star number: a file counts when every architecture
|
||||
// its build admits assembles it.
|
||||
full, generic, otherPort, narrowedCount, excluded := 0, 0, 0, 0, 0
|
||||
|
||||
// Header generation is created on first use, so a corpus with no
|
||||
// go_asm.h includes never pays for a temp directory.
|
||||
var hdr *asmhdrCache
|
||||
defer func() {
|
||||
if hdr != nil {
|
||||
hdr.close()
|
||||
}
|
||||
}()
|
||||
|
||||
for _, path := range files {
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// The GOOS the header generation type-checks under follows the
|
||||
// file's name when the name carries one; the ambient GOOS is the
|
||||
// honest guess otherwise (a build tag naming another GOOS is
|
||||
// invisible to a file-name rule).
|
||||
goos := goosFromFilename(path)
|
||||
|
||||
// The GOOS the header generation type-checks under follows the
|
||||
// file's name when the name carries one; the ambient GOOS is the
|
||||
// honest guess otherwise.
|
||||
namedArch := arch.FromFilename(path)
|
||||
var wanted []int // indexes into targets
|
||||
other := false
|
||||
switch {
|
||||
case namedArch != arch.Unknown:
|
||||
for i, tg := range targets {
|
||||
if tg.a == namedArch {
|
||||
wanted = append(wanted, i)
|
||||
}
|
||||
}
|
||||
case otherPortFile(path):
|
||||
// A file named for a Go port gasm does not support (arm,
|
||||
// 386, s390x, ...) or for another GOOS is compiled by no
|
||||
// supported-arch build, so it is neither generic nor a
|
||||
// per-arch attempt: counting it as generic would make the
|
||||
// headline unreachably low for reasons no supported target
|
||||
// can fix.
|
||||
other = true
|
||||
otherPort++
|
||||
default:
|
||||
for i := range targets {
|
||||
wanted = append(wanted, i)
|
||||
}
|
||||
}
|
||||
|
||||
// A //go:build constraint narrows the set of targets the file is
|
||||
// assembled for, the way the go command compiles the file only for
|
||||
// the targets the expression admits: cpu_x86.s belongs to the x86
|
||||
// build alone, and a file whose constraint admits none of the four
|
||||
// targets (the goexperiment.runtimesecret and msan trees) is
|
||||
// compiled by no supported build. The tags mirror what a plain
|
||||
// `go build` sets: the GOOS and GOARCH, gc, and unix on the
|
||||
// unix-like GOOSes; no experiment, sanitiser or cgo tag is ever
|
||||
// true. The GOOS is the file's own when the name carries one,
|
||||
// else the ambient one.
|
||||
goosForEval := goos
|
||||
if goosForEval == "" {
|
||||
goosForEval = runtime.GOOS
|
||||
}
|
||||
narrowed := false
|
||||
if len(wanted) > 0 {
|
||||
if ce := buildConstraint(src); ce != nil {
|
||||
kept := make([]int, 0, len(wanted))
|
||||
for _, i := range wanted {
|
||||
tg := targets[i]
|
||||
if ce.Eval(constraintTags(goarchName(tg.a), goosForEval)) {
|
||||
kept = append(kept, i)
|
||||
}
|
||||
}
|
||||
if len(kept) < len(wanted) {
|
||||
narrowed = true
|
||||
}
|
||||
wanted = kept
|
||||
}
|
||||
}
|
||||
|
||||
switch {
|
||||
case other:
|
||||
// already tallied above
|
||||
case len(wanted) == 0:
|
||||
excluded++
|
||||
case namedArch != arch.Unknown:
|
||||
// a per-arch attempt over the constraint's subset
|
||||
case narrowed:
|
||||
narrowedCount++
|
||||
default:
|
||||
generic++
|
||||
}
|
||||
|
||||
// A file that includes go_asm.h parses against a per-target header:
|
||||
// the defines differ per architecture (internal/cpu's layout, for
|
||||
// one) and per GOOS (sys_darwin_arm64.s's trampoline constants,
|
||||
// for another), so the parse cannot be shared the way a
|
||||
// header-free file's can. A generation failure is a failure for
|
||||
// every target, named for the package rather than a bare "include
|
||||
// not found". A header already resolvable in the package
|
||||
// directory or the -I list is left alone.
|
||||
if len(wanted) > 0 && needsGoAsmHeader(src) && !goAsmHeaderResolved(filepath.Dir(path), dirs) {
|
||||
if hdr == nil {
|
||||
if hdr, err = newAsmhdrCache(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
pkgDir := filepath.Dir(path)
|
||||
ok := true
|
||||
for _, i := range wanted {
|
||||
tg, t := targets[i], tallies[i]
|
||||
t.attempted++
|
||||
hdrDir, err := hdr.dirFor(pkgDir, goos, goarchName(tg.a))
|
||||
if err != nil {
|
||||
ok = false
|
||||
t.fail(path, err)
|
||||
continue
|
||||
}
|
||||
f, errs := parser.ParseWithOptions(path, src, parser.Options{
|
||||
Expand: true,
|
||||
IncludeDirs: append(slices.Clone(dirs), hdrDir),
|
||||
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
|
||||
})
|
||||
if len(errs) > 0 {
|
||||
ok = false
|
||||
t.fail(path, errs[0])
|
||||
continue
|
||||
}
|
||||
if _, err := assembleFile(tg.a, f, goos); err != nil {
|
||||
ok = false
|
||||
t.fail(path, err)
|
||||
continue
|
||||
}
|
||||
t.assembled++
|
||||
}
|
||||
if ok && len(wanted) > 0 {
|
||||
full++
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
ok := true
|
||||
for _, i := range wanted {
|
||||
tg, t := targets[i], tallies[i]
|
||||
t.attempted++
|
||||
// The parse carries the target's platform predefines, so it
|
||||
// cannot be shared across targets the way a header-free file's
|
||||
// could: a #ifdef GOARCH_arm block must be live on arm64 and
|
||||
// dead everywhere else.
|
||||
f, errs := parser.ParseWithOptions(path, src, parser.Options{
|
||||
Expand: true,
|
||||
IncludeDirs: dirs,
|
||||
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
|
||||
})
|
||||
var err error
|
||||
if len(errs) > 0 {
|
||||
err = errs[0] // a parse failure is a failure for every target
|
||||
} else {
|
||||
_, err = assembleFile(tg.a, f, goos)
|
||||
}
|
||||
if err != nil {
|
||||
ok = false
|
||||
t.fail(path, err)
|
||||
continue
|
||||
}
|
||||
t.assembled++
|
||||
}
|
||||
if ok && len(wanted) > 0 {
|
||||
full++
|
||||
}
|
||||
}
|
||||
|
||||
return &corpusStats{
|
||||
root: root,
|
||||
files: len(files),
|
||||
generic: generic,
|
||||
narrowed: narrowedCount,
|
||||
excluded: excluded,
|
||||
otherPort: otherPort,
|
||||
full: full,
|
||||
targets: targets,
|
||||
tallies: tallies,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// printCorpusStats renders the corpus audit report.
|
||||
func printCorpusStats(s *corpusStats, list bool) {
|
||||
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d narrowed by //go:build; %d excluded by //go:build; %d named for other Go ports, never attempted)\n",
|
||||
s.root, s.files, s.generic, s.narrowed, s.excluded, s.otherPort)
|
||||
// The rate is over the files a supported build would attempt: the
|
||||
// other ports' files and the ones no supported target compiles sit in
|
||||
// the count for completeness but can never assemble, so counting them
|
||||
// in the denominator would report the gap of platforms gasm
|
||||
// deliberately does not target.
|
||||
attemptable := max(s.files-s.otherPort-s.excluded, 1)
|
||||
fmt.Printf(" assemble for every applicable target: %d of %d attemptable (%.1f%%)\n", s.full, attemptable, 100*float64(s.full)/float64(attemptable))
|
||||
for i, tg := range s.targets {
|
||||
t := s.tallies[i]
|
||||
fmt.Printf(" %s: %d/%d attempted\n", tg.name, t.assembled, t.attempted)
|
||||
for _, r := range topReasons(t) {
|
||||
fmt.Printf(" %4d %s\n", t.reasons[r], r)
|
||||
fmt.Printf(" e.g. %s\n", t.example[r])
|
||||
}
|
||||
if !list {
|
||||
continue
|
||||
}
|
||||
for _, f := range t.fails {
|
||||
fmt.Printf(" FAIL %s\n", f.path)
|
||||
fmt.Printf(" %s: %s\n", f.reason, f.detail)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// corpusReason buckets an assembly or parse failure for the histogram.
|
||||
func corpusReason(err error) string {
|
||||
msg := err.Error()
|
||||
switch {
|
||||
case strings.Contains(msg, "go_asm.h for GOARCH"):
|
||||
return "go_asm.h generation failed"
|
||||
case strings.Contains(msg, "unsupported"), strings.Contains(msg, "cannot encode"):
|
||||
return "instruction not encodable"
|
||||
case strings.Contains(msg, "undefined label"):
|
||||
return "undefined label"
|
||||
case strings.Contains(msg, "undefined symbol"), strings.Contains(msg, "external symbol"), strings.Contains(msg, "file-level assembly"):
|
||||
return "undefined symbol or external"
|
||||
case strings.Contains(msg, "operand"), strings.Contains(msg, "operand form"):
|
||||
return "unsupported operand form"
|
||||
default:
|
||||
return "other: " + firstLine(msg)
|
||||
}
|
||||
}
|
||||
|
||||
// topReasons returns at most five reasons, most frequent first.
|
||||
func topReasons(t *corpusTally) []string {
|
||||
type kv struct {
|
||||
k string
|
||||
n int
|
||||
}
|
||||
var kvs []kv
|
||||
for k, n := range t.reasons {
|
||||
kvs = append(kvs, kv{k, n})
|
||||
}
|
||||
slices.SortFunc(kvs, func(a, b kv) int { return b.n - a.n })
|
||||
if len(kvs) > 5 {
|
||||
kvs = kvs[:5]
|
||||
}
|
||||
out := make([]string, len(kvs))
|
||||
for i, kv := range kvs {
|
||||
out[i] = kv.k
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// firstLine returns the first line of an error message, truncated.
|
||||
func firstLine(msg string) string {
|
||||
if i := strings.IndexByte(msg, '\n'); i >= 0 {
|
||||
msg = msg[:i]
|
||||
}
|
||||
if len(msg) > 80 {
|
||||
msg = msg[:80]
|
||||
}
|
||||
return msg
|
||||
}
|
||||
@@ -0,0 +1,126 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"runtime"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||
)
|
||||
|
||||
func TestDerivedFamily(t *testing.T) {
|
||||
for _, n := range []string{"JEQ", "JLT", "JCC", "CMOVLGT", "SETNE", "SETA"} {
|
||||
if !derivedFamily(n) {
|
||||
t.Errorf("derivedFamily(%q) = false, want true", n)
|
||||
}
|
||||
}
|
||||
for _, n := range []string{"JMP", "ADDQ", "VPGATHERDD", "MOVBE", "PSHUFB"} {
|
||||
if derivedFamily(n) {
|
||||
t.Errorf("derivedFamily(%q) = true, want false", n)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestSanitize(t *testing.T) {
|
||||
if got := sanitize("VPCMP.UB"); got != "VPCMP_UB" {
|
||||
t.Errorf("sanitize: got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAuditArch(t *testing.T) {
|
||||
for in, want := range map[string]arch.Arch{
|
||||
"amd64": arch.AMD64, "arm64": arch.ARM64,
|
||||
"riscv64": arch.RISCV, "riscv": arch.RISCV,
|
||||
"loong64": arch.LOONG64, "LOONG": arch.LOONG64,
|
||||
} {
|
||||
got, err := auditArch(in)
|
||||
if err != nil || got != want {
|
||||
t.Errorf("auditArch(%q) = %v, %v; want %v", in, got, err, want)
|
||||
}
|
||||
}
|
||||
if _, err := auditArch("mips"); err == nil {
|
||||
t.Error("auditArch(mips) must fail")
|
||||
}
|
||||
}
|
||||
|
||||
func TestGasmEncodable(t *testing.T) {
|
||||
cases := []struct {
|
||||
a arch.Arch
|
||||
yes string
|
||||
no string
|
||||
}{
|
||||
{arch.AMD64, "ADDQ", "NOSUCHMNEMONIC"},
|
||||
{arch.ARM64, "ADD", "NOSUCHMNEMONIC"},
|
||||
{arch.RISCV, "ADD", "NOSUCHMNEMONIC"},
|
||||
{arch.LOONG64, "ADDV", "NOSUCHMNEMONIC"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if !gasmEncodable(c.a, c.yes) {
|
||||
t.Errorf("%s: %s should be encodable", c.a, c.yes)
|
||||
}
|
||||
if gasmEncodable(c.a, c.no) {
|
||||
t.Errorf("%s: %s should not be encodable", c.a, c.no)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestBuildConstraint pins the //go:build reader: the constraint governs the
|
||||
// leading comment block, the first non-comment line ends it (a tag below a
|
||||
// #include governs nothing, exactly as go/build drops it), and a file
|
||||
// without one admits every target.
|
||||
func TestBuildConstraint(t *testing.T) {
|
||||
admits := func(src, goarch, goos string) bool {
|
||||
t.Helper()
|
||||
e := buildConstraint(src)
|
||||
if e == nil {
|
||||
return true
|
||||
}
|
||||
return e.Eval(constraintTags(goarch, goos))
|
||||
}
|
||||
const ret = "TEXT \xc2\xb7f(SB), NOSPLIT, $0\n\tRET\n"
|
||||
cases := []struct {
|
||||
name string
|
||||
src string
|
||||
amd64, arm64 bool
|
||||
}{
|
||||
{"no constraint", ret, true, true},
|
||||
{"x86 only", "//go:build 386 || amd64\n\n" + ret, true, false},
|
||||
{"arm64 and linux", "//go:build arm64 && linux\n\n" + ret, false, true},
|
||||
{"msan never", "//go:build msan\n\n" + ret, false, false},
|
||||
{"experiment never", "//go:build goexperiment.runtimesecret\n\n" + ret, false, false},
|
||||
{"below an include governs nothing", "#include \"textflag.h\"\n//go:build amd64\n" + ret, true, true},
|
||||
{"unparsable narrows nothing", "//go:build (amd64\n" + ret, true, true},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
if got := admits(c.src, "amd64", runtime.GOOS); got != c.amd64 {
|
||||
t.Errorf("amd64 admission = %v, want %v", got, c.amd64)
|
||||
}
|
||||
if got := admits(c.src, "arm64", runtime.GOOS); got != c.arm64 {
|
||||
t.Errorf("arm64 admission = %v, want %v", got, c.arm64)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestConstraintTags pins the tag set a plain `go build` sets: the GOOS and
|
||||
// GOARCH, gc, unix on the unix-like GOOSes; nothing else is ever true.
|
||||
func TestConstraintTags(t *testing.T) {
|
||||
ok := constraintTags("amd64", "linux")
|
||||
for _, tag := range []string{"amd64", "linux", "gc", "unix"} {
|
||||
if !ok(tag) {
|
||||
t.Errorf("tag %q = false, want true", tag)
|
||||
}
|
||||
}
|
||||
for _, tag := range []string{"arm64", "freebsd", "darwin", "cgo", "race", "msan", "goexperiment.runtimesecret"} {
|
||||
if ok(tag) {
|
||||
t.Errorf("tag %q = true, want false", tag)
|
||||
}
|
||||
}
|
||||
fb := constraintTags("arm64", "freebsd")
|
||||
if !fb("unix") {
|
||||
t.Error("unix on freebsd = false, want true")
|
||||
}
|
||||
}
|
||||
@@ -1,193 +0,0 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/debug"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
||||
)
|
||||
|
||||
func cmdDebug(args []string) int {
|
||||
fs := newCommand("debug", "gasm debug <file.s> --func <name>", `
|
||||
Interactive debugger for JIT-assembled amd64 functions. Launches the
|
||||
function in a traced subprocess (ptrace), then provides a REPL for
|
||||
single-stepping, breakpoints, register and memory inspection.
|
||||
|
||||
REPL commands:
|
||||
break <label|addr> set a breakpoint at a label or absolute address
|
||||
step [n] single-step n instructions (default 1)
|
||||
continue run until next breakpoint or exit
|
||||
regs print general-purpose registers
|
||||
x [addr] [len] hex-dump memory (default: current PC, 64 bytes)
|
||||
labels list function labels and offsets
|
||||
quit kill the debuggee and exit
|
||||
`)
|
||||
target := fs.Bool("target", false, "") // hidden: debuggee subprocess mode
|
||||
funcName := fs.String("func", "", "function to debug")
|
||||
argsFile := fs.String("args", "", "file containing the ABI0 argument block")
|
||||
bufSpec := fs.String("buf", "", "buffer specification: name:size:pattern[,name:size:pattern...] where pattern is zero, ones, seq, or hex")
|
||||
fs.Parse(args)
|
||||
|
||||
// --- Debuggee mode (internal, spawned by the debugger) ---
|
||||
if *target {
|
||||
tmpDir := os.Getenv("GASM_DEBUG_TMP")
|
||||
if tmpDir == "" || fs.NArg() < 1 || *funcName == "" || *argsFile == "" {
|
||||
fmt.Fprintln(os.Stderr, "gasm debug --target: internal mode")
|
||||
return 2
|
||||
}
|
||||
if err := debug.RunTarget(fs.Arg(0), *funcName, *argsFile, tmpDir); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// --- Debugger mode (interactive REPL) ---
|
||||
if fs.NArg() < 1 || *funcName == "" {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm debug <file.s> --func <name>")
|
||||
return 2
|
||||
}
|
||||
path := fs.Arg(0)
|
||||
|
||||
// Load the kernel to extract function metadata and labels.
|
||||
k, err := verify.Load(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer k.Close()
|
||||
|
||||
fl, err := k.Func(*funcName)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
// Build the label list for the REPL.
|
||||
var labels []debug.Label
|
||||
for name, off := range fl.Labels {
|
||||
labels = append(labels, debug.Label{Name: name, Offset: off})
|
||||
}
|
||||
sort.Slice(labels, func(i, j int) bool { return labels[i].Offset < labels[j].Offset })
|
||||
|
||||
// Launch the debuggee with the argument block.
|
||||
var argBlock []byte
|
||||
var bufAddrs []uint64
|
||||
var sess *debug.Session
|
||||
if *bufSpec != "" {
|
||||
// Parse the function signature to determine argument layout.
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
sig, ok := verify.ExtractFuncSig(src, *funcName)
|
||||
if !ok {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: no // func signature found for %s\n", *funcName)
|
||||
return 1
|
||||
}
|
||||
layout := verify.ArgLayout(sig)
|
||||
|
||||
// Parse the buffer spec to get buffer names.
|
||||
bufNames := parseBufNames(*bufSpec)
|
||||
|
||||
// Allocate buffers in the debuggee.
|
||||
argBlock = make([]byte, fl.Args)
|
||||
sess, bufAddrs, err = debug.LaunchWithBuffers("", path, *funcName, argBlock, *bufSpec)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
// Construct the argument block with buffer pointers at the correct positions.
|
||||
bufIdx := 0
|
||||
for _, arg := range layout {
|
||||
if !arg.IsPtr {
|
||||
continue
|
||||
}
|
||||
// Find the buffer that matches this argument.
|
||||
for i, name := range bufNames {
|
||||
if i < len(bufAddrs) && (name == arg.Name || strings.HasPrefix(arg.Name, name)) {
|
||||
addr := bufAddrs[i]
|
||||
off := arg.Offset
|
||||
if off+8 <= len(argBlock) {
|
||||
argBlock[off] = byte(addr)
|
||||
argBlock[off+1] = byte(addr >> 8)
|
||||
argBlock[off+2] = byte(addr >> 16)
|
||||
argBlock[off+3] = byte(addr >> 24)
|
||||
argBlock[off+4] = byte(addr >> 32)
|
||||
argBlock[off+5] = byte(addr >> 40)
|
||||
argBlock[off+6] = byte(addr >> 48)
|
||||
argBlock[off+7] = byte(addr >> 56)
|
||||
}
|
||||
// For slices, also set the length and capacity.
|
||||
if strings.HasPrefix(arg.Typ, "[]") && off+24 <= len(argBlock) {
|
||||
// Find the buffer size from the spec.
|
||||
size := parseBufSize(*bufSpec, name)
|
||||
// Length at offset+8, capacity at offset+16.
|
||||
for j := 0; j < 8; j++ {
|
||||
argBlock[off+8+j] = byte(size >> (j * 8))
|
||||
argBlock[off+16+j] = byte(size >> (j * 8))
|
||||
}
|
||||
}
|
||||
bufIdx++
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
_ = bufIdx
|
||||
} else {
|
||||
argBlock = make([]byte, fl.Args)
|
||||
sess, err = debug.Launch("", path, *funcName, argBlock)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
}
|
||||
defer sess.Kill()
|
||||
|
||||
bm := debug.NewBreakpoints(sess)
|
||||
fmt.Printf("gasm debug: %s in %s (pid %d)\n", *funcName, path, sess.Pid())
|
||||
|
||||
// Convert the line table for the REPL.
|
||||
var srcLines []debug.SourceLine
|
||||
for _, le := range fl.Lines {
|
||||
srcLines = append(srcLines, debug.SourceLine{Offset: le.Offset, Line: le.Line})
|
||||
}
|
||||
debug.REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, labels, srcLines)
|
||||
return 0
|
||||
}
|
||||
|
||||
// parseBufNames extracts buffer names from a buffer specification.
|
||||
// Format: name:size:pattern[,name:size:pattern...]
|
||||
func parseBufNames(spec string) []string {
|
||||
var names []string
|
||||
for _, part := range strings.Split(spec, ",") {
|
||||
fields := strings.SplitN(part, ":", 3)
|
||||
if len(fields) >= 1 && fields[0] != "" {
|
||||
names = append(names, fields[0])
|
||||
}
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
// parseBufSize extracts the size of a named buffer from a buffer specification.
|
||||
func parseBufSize(spec, name string) int {
|
||||
for _, part := range strings.Split(spec, ",") {
|
||||
fields := strings.SplitN(part, ":", 3)
|
||||
if len(fields) >= 2 && fields[0] == name {
|
||||
var size int
|
||||
fmt.Sscanf(fields[1], "%d", &size)
|
||||
return size
|
||||
}
|
||||
}
|
||||
return 0
|
||||
}
|
||||
@@ -1,7 +1,7 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build !(linux && amd64)
|
||||
//go:build !(linux || (freebsd && (amd64 || arm64 || riscv64)))
|
||||
|
||||
package main
|
||||
|
||||
@@ -11,6 +11,6 @@ import (
|
||||
)
|
||||
|
||||
func cmdDebug(args []string) int {
|
||||
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires linux/amd64 (ptrace)")
|
||||
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux or FreeBSD (ptrace)")
|
||||
return 1
|
||||
}
|
||||
|
||||
@@ -0,0 +1,336 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux || (freebsd && (amd64 || arm64 || riscv64))
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/debug"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
|
||||
)
|
||||
|
||||
func cmdDebug(args []string) int {
|
||||
fs := newCommand("debug", "gasm debug <file.s> --func <name>", `
|
||||
Interactive debugger for JIT-assembled functions. Launches the
|
||||
function in a traced subprocess (ptrace), then provides a REPL for
|
||||
single-stepping, breakpoints, register and memory inspection.
|
||||
|
||||
REPL commands:
|
||||
break <label|addr|line> [if <reg> <op> <val|reg|*addr>]
|
||||
set a breakpoint, optionally conditional on a
|
||||
comparison of one register against a constant,
|
||||
another register, or the 8-byte word at *addr
|
||||
delete <label|addr> remove a breakpoint
|
||||
info break list all breakpoints
|
||||
step [n], s single-step n instructions (default 1)
|
||||
next, n step over a CALL
|
||||
finish, fin run until the function returns
|
||||
continue, c run until a breakpoint, watchpoint or exit
|
||||
disas [n], u disassemble n instructions at PC
|
||||
regs print general-purpose and vector registers
|
||||
where show source line and nearest label at PC
|
||||
stack show stack near RSP (return address + ABI0 args)
|
||||
bt, backtrace backtrace (current frame + return address)
|
||||
x [addr] [len] hex-dump memory (default: current PC, 64 bytes)
|
||||
w <addr> <val...> write bytes to memory
|
||||
set <reg> <value> set a register
|
||||
watch <addr> [r|w] [size]
|
||||
set a hardware watchpoint (write by default)
|
||||
unwatch [<slot>] clear one watchpoint, or all without an argument
|
||||
labels, l list function labels and offsets
|
||||
help, h, ? show command help
|
||||
quit, q kill the debuggee and exit
|
||||
`)
|
||||
funcName := fs.String("func", "", "function to debug")
|
||||
argsFile := fs.String("args", "", "file containing the ABI0 argument block")
|
||||
bufSpec := fs.String("buf", "", "buffer specification: name:size:pattern[,name:size:pattern...] where pattern is zero, ones, seq, or hex")
|
||||
script := fs.String("script", "", "run REPL commands from a file (one per line) and exit; '-' reads stdin")
|
||||
cover := fs.Bool("cover", false, "run to completion with a breakpoint on every instruction and report which executed and how often")
|
||||
timeout := fs.Duration("timeout", 0, "kill the debuggee after this duration (e.g. 30s); for headless --script runs; a timeout exits 3")
|
||||
fs.Parse(args)
|
||||
|
||||
// --- Debuggee mode (internal, spawned by the debugger) ---
|
||||
if os.Getenv("GASM_DEBUG_TARGET") != "" {
|
||||
tmpDir := os.Getenv("GASM_DEBUG_TMP")
|
||||
if tmpDir == "" || fs.NArg() < 1 || *funcName == "" || *argsFile == "" {
|
||||
fmt.Fprintln(os.Stderr, "gasm debug: internal debuggee mode")
|
||||
return 2
|
||||
}
|
||||
if err := debug.RunTarget(fs.Arg(0), *funcName, *argsFile, tmpDir); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// --- Debugger mode (interactive REPL) ---
|
||||
if fs.NArg() < 1 || *funcName == "" {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm debug <file.s> --func <name>")
|
||||
return 2
|
||||
}
|
||||
path := fs.Arg(0)
|
||||
|
||||
// The watchdog is armed before anything can block: ptrace attach and a
|
||||
// continued kernel loop both hang the run when the environment forbids
|
||||
// tracing or the kernel loops forever, and neither is interruptible from
|
||||
// the inside.
|
||||
if *timeout > 0 {
|
||||
go func() {
|
||||
time.Sleep(*timeout)
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: timeout (%s), killing the debuggee\n", *timeout)
|
||||
os.Exit(3)
|
||||
}()
|
||||
}
|
||||
|
||||
// Load the kernel to extract function metadata and labels.
|
||||
k, err := verify.Load(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer k.Close()
|
||||
|
||||
fl, err := k.Func(*funcName)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
// Build the label list for the REPL.
|
||||
var labels []debug.Label
|
||||
for name, off := range fl.Labels {
|
||||
labels = append(labels, debug.Label{Name: name, Offset: off})
|
||||
}
|
||||
sort.Slice(labels, func(i, j int) bool { return labels[i].Offset < labels[j].Offset })
|
||||
|
||||
// Launch the debuggee with the argument block.
|
||||
var argBlock []byte
|
||||
var bufAddrs []uint64
|
||||
var sess *debug.Session
|
||||
if *bufSpec != "" {
|
||||
// Parse the function signature to determine argument layout.
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
sig, ok := verify.ExtractFuncSig(src, *funcName)
|
||||
if !ok {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: no // func signature found for %s\n", *funcName)
|
||||
return 1
|
||||
}
|
||||
layout := verify.ArgLayout(sig)
|
||||
|
||||
// Parse the buffer spec to get buffer names.
|
||||
bufNames := parseBufNames(*bufSpec)
|
||||
|
||||
// Allocate buffers in the debuggee.
|
||||
argBlock = make([]byte, fl.Args)
|
||||
sess, bufAddrs, err = debug.LaunchWithBuffers("", path, *funcName, argBlock, *bufSpec)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
// Construct the argument block with buffer pointers at the correct positions.
|
||||
for _, arg := range layout {
|
||||
if !arg.IsPtr {
|
||||
continue
|
||||
}
|
||||
// Find the buffer that matches this argument.
|
||||
for i, name := range bufNames {
|
||||
if i < len(bufAddrs) && (name == arg.Name || strings.HasPrefix(arg.Name, name)) {
|
||||
addr := bufAddrs[i]
|
||||
off := arg.Offset
|
||||
if off+8 <= len(argBlock) {
|
||||
argBlock[off] = byte(addr)
|
||||
argBlock[off+1] = byte(addr >> 8)
|
||||
argBlock[off+2] = byte(addr >> 16)
|
||||
argBlock[off+3] = byte(addr >> 24)
|
||||
argBlock[off+4] = byte(addr >> 32)
|
||||
argBlock[off+5] = byte(addr >> 40)
|
||||
argBlock[off+6] = byte(addr >> 48)
|
||||
argBlock[off+7] = byte(addr >> 56)
|
||||
}
|
||||
// For slices, also set the length and capacity.
|
||||
if strings.HasPrefix(arg.Typ, "[]") && off+24 <= len(argBlock) {
|
||||
// Find the buffer size from the spec.
|
||||
size := parseBufSize(*bufSpec, name)
|
||||
// Length at offset+8, capacity at offset+16.
|
||||
for j := range 8 {
|
||||
argBlock[off+8+j] = byte(size >> (j * 8))
|
||||
argBlock[off+16+j] = byte(size >> (j * 8))
|
||||
}
|
||||
}
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
argBlock = make([]byte, fl.Args)
|
||||
sess, err = debug.Launch("", path, *funcName, argBlock)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
}
|
||||
defer sess.Kill()
|
||||
|
||||
bm := debug.NewBreakpoints(sess)
|
||||
fmt.Printf("gasm debug: %s in %s (pid %d)\n", *funcName, path, sess.Pid())
|
||||
|
||||
// Convert the line table for the command loop.
|
||||
var srcLines []debug.SourceLine
|
||||
for _, le := range fl.Lines {
|
||||
srcLines = append(srcLines, debug.SourceLine{Offset: le.Offset, Line: le.Line})
|
||||
}
|
||||
|
||||
// Coverage mode: pre-register a breakpoint on every instruction (walked
|
||||
// by length through the function body while the debuggee is stopped) and
|
||||
// let the kernel run to completion. Each trap counts a hit for that
|
||||
// instruction, so the final report shows exactly which instructions
|
||||
// executed and how often, with the label-level view derived from it.
|
||||
// Expect the run to slow to ptrace speed: one trap per executed
|
||||
// instruction.
|
||||
if *cover {
|
||||
base := sess.CodeBase() + uint64(fl.Offset)
|
||||
type coverInstr struct {
|
||||
off uint64
|
||||
text string
|
||||
}
|
||||
var instrs []coverInstr
|
||||
for off := uint64(0); off < uint64(fl.Size); {
|
||||
text, ln, err := sess.Disassemble(base + off)
|
||||
if err != nil || ln == 0 {
|
||||
break
|
||||
}
|
||||
instrs = append(instrs, coverInstr{off: off, text: text})
|
||||
off += uint64(ln)
|
||||
}
|
||||
for _, in := range instrs {
|
||||
if _, err := bm.SetWithCond(base+in.off, fmt.Sprintf("func+%#x", in.off), nil); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: cover: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
}
|
||||
fmt.Printf("gasm debug: coverage run over %d instructions\n", len(instrs))
|
||||
for {
|
||||
for _, bp := range bm.All() {
|
||||
bm.Reinsert(bp.Addr)
|
||||
}
|
||||
if err := sess.Continue(); err != nil {
|
||||
break // debuggee finished or died
|
||||
}
|
||||
if sess.Exited() {
|
||||
break
|
||||
}
|
||||
// A genuine signal-delivery-stop (a fault in the kernel): the
|
||||
// run cannot make progress, because resuming would restart the
|
||||
// faulting instruction and fault forever. Report and stop.
|
||||
if sig := sess.LastSignal(); sig != 0 {
|
||||
fmt.Printf("gasm debug: cover: stopped on signal %v\n", sig)
|
||||
break
|
||||
}
|
||||
regs, rerr := sess.GetRegs()
|
||||
if rerr != nil {
|
||||
break
|
||||
}
|
||||
// HandleTrap restores the original byte, rewinds PC and counts
|
||||
// the hit on the breakpoint itself. Single-step over the
|
||||
// restored instruction so the reinsertion at the top of the
|
||||
// loop cannot re-trap on the same breakpoint.
|
||||
if bp := bm.HandleTrap(®s); bp != nil {
|
||||
if err := sess.Step(); err != nil {
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
hits := map[uint64]int{}
|
||||
traps := 0
|
||||
for _, bp := range bm.All() {
|
||||
if n := bp.Hits(); n > 0 {
|
||||
hits[bp.Addr-base] = n
|
||||
traps += n
|
||||
}
|
||||
}
|
||||
var hit []string
|
||||
var missed []string
|
||||
for _, l := range labels {
|
||||
if hits[uint64(l.Offset)] > 0 {
|
||||
hit = append(hit, l.Name)
|
||||
} else {
|
||||
missed = append(missed, l.Name)
|
||||
}
|
||||
}
|
||||
sort.Strings(hit)
|
||||
sort.Strings(missed)
|
||||
fmt.Printf("coverage: %d/%d instructions executed (%d traps)\n", len(hits), len(instrs), traps)
|
||||
fmt.Printf("coverage: %d/%d labels reached\n", len(hit), len(labels))
|
||||
for _, l := range hit {
|
||||
fmt.Printf(" covered %s\n", l)
|
||||
}
|
||||
for _, l := range missed {
|
||||
fmt.Printf(" MISSED %s\n", l)
|
||||
}
|
||||
fmt.Println("executed instructions:")
|
||||
for _, in := range instrs {
|
||||
if n := hits[in.off]; n > 0 {
|
||||
fmt.Printf(" func+%#04x %4dx %s\n", in.off, n, in.text)
|
||||
}
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// Headless mode: run the script through the normal command loop and
|
||||
// exit. The watchdog armed above covers launch, continue and step.
|
||||
var in io.Reader = os.Stdin
|
||||
if *script != "" {
|
||||
if *script == "-" {
|
||||
in = os.Stdin
|
||||
} else {
|
||||
f, err := os.Open(*script)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer f.Close()
|
||||
in = f
|
||||
}
|
||||
}
|
||||
debug.REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, labels, srcLines, in)
|
||||
return 0
|
||||
}
|
||||
|
||||
// parseBufNames extracts buffer names from a buffer specification.
|
||||
// Format: name:size:pattern[,name:size:pattern...]
|
||||
func parseBufNames(spec string) []string {
|
||||
var names []string
|
||||
for part := range strings.SplitSeq(spec, ",") {
|
||||
fields := strings.SplitN(part, ":", 3)
|
||||
if len(fields) >= 1 && fields[0] != "" {
|
||||
names = append(names, fields[0])
|
||||
}
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
// parseBufSize extracts the size of a named buffer from a buffer specification.
|
||||
func parseBufSize(spec, name string) int {
|
||||
for part := range strings.SplitSeq(spec, ",") {
|
||||
fields := strings.SplitN(part, ":", 3)
|
||||
if len(fields) >= 2 && fields[0] == name {
|
||||
var size int
|
||||
fmt.Sscanf(fields[1], "%d", &size)
|
||||
return size
|
||||
}
|
||||
}
|
||||
return 0
|
||||
}
|
||||
+148
@@ -0,0 +1,148 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// cmdDis disassembles machine code: either a raw binary (standard input with
|
||||
// "-") whose architecture is given with -a, or a .s file, which is assembled
|
||||
// first so the listing shows the real function and label layout.
|
||||
func cmdDis(args []string) int {
|
||||
fs := newCommand("dis", "gasm dis [-a arch] <file>", `
|
||||
Disassemble machine code to instruction text (via golang.org/x/arch).
|
||||
|
||||
With a .s file, the file is assembled first and the listing follows the
|
||||
real layout: one block per TEXT function, local labels printed at their
|
||||
offsets. The architecture comes from the file name suffix, or from -a.
|
||||
|
||||
With any other file, or "-" for standard input, the bytes are disassembled
|
||||
linearly and -a selects the architecture (amd64, arm64, riscv64 or
|
||||
loong64).
|
||||
`)
|
||||
archName := fs.String("a", "", "architecture for raw input: amd64, arm64, riscv64 or loong64")
|
||||
fs.Parse(args)
|
||||
if fs.NArg() != 1 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm dis [-a arch] <file>")
|
||||
return 2
|
||||
}
|
||||
path := fs.Arg(0)
|
||||
var target arch.Arch
|
||||
if *archName != "" {
|
||||
var err error
|
||||
target, err = auditArch(*archName)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
}
|
||||
|
||||
if strings.HasSuffix(path, ".s") {
|
||||
if target == arch.Unknown {
|
||||
target = arch.FromFilename(path)
|
||||
}
|
||||
if target == arch.Unknown {
|
||||
fmt.Fprintln(os.Stderr, "gasm dis: cannot infer the architecture from the file name; use -a")
|
||||
return 2
|
||||
}
|
||||
return disSource(path, target)
|
||||
}
|
||||
|
||||
if target == arch.Unknown {
|
||||
fmt.Fprintln(os.Stderr, "gasm dis: raw input needs -a (amd64, arm64, riscv64 or loong64)")
|
||||
return 2
|
||||
}
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm dis:", err)
|
||||
return 1
|
||||
}
|
||||
printListing(target, []byte(src), 0, nil)
|
||||
return 0
|
||||
}
|
||||
|
||||
// disSource assembles a .s file and prints one listing block per function.
|
||||
func disSource(path string, target arch.Arch) int {
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm dis:", err)
|
||||
return 1
|
||||
}
|
||||
f, errs := parser.Parse(path, src)
|
||||
for _, e := range errs {
|
||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
||||
}
|
||||
if len(errs) > 0 {
|
||||
return 1
|
||||
}
|
||||
img, err := assembleFile(target, f, "")
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
if len(img.Funcs) == 0 {
|
||||
fmt.Fprintln(os.Stderr, "gasm dis: no assemblable TEXT functions found")
|
||||
return 1
|
||||
}
|
||||
for _, fn := range img.Funcs {
|
||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
fmt.Printf("%s: %d bytes\n", fn.Name, fn.Size)
|
||||
labels := make(map[int][]string, len(fn.Labels))
|
||||
for name, off := range fn.Labels {
|
||||
labels[off] = append(labels[off], name)
|
||||
}
|
||||
for off := range labels {
|
||||
sort.Strings(labels[off])
|
||||
}
|
||||
printListing(target, code, uint64(fn.Offset), labels)
|
||||
}
|
||||
if len(img.Data) > 0 {
|
||||
fmt.Printf("data: %d bytes at 0x%x\n", len(img.Data), len(img.Code))
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// printListing decodes code linearly from offset base, printing label lines
|
||||
// (label name to offset within the block) as they are reached.
|
||||
func printListing(a arch.Arch, code []byte, base uint64, labels map[int][]string) {
|
||||
pc := 0
|
||||
for pc < len(code) {
|
||||
for _, name := range labels[pc] {
|
||||
fmt.Printf("%s:\n", name)
|
||||
}
|
||||
ins, err := disasm.Decode(a, code[pc:], base+uint64(pc))
|
||||
if err != nil {
|
||||
break
|
||||
}
|
||||
end := min(pc+ins.Len, len(code))
|
||||
fmt.Printf(" %04x: %-16s %s\n", base+uint64(pc), hexBytes(code[pc:end]), ins.Text)
|
||||
if ins.Len <= 0 {
|
||||
break
|
||||
}
|
||||
pc += ins.Len
|
||||
}
|
||||
}
|
||||
|
||||
// hexBytes renders up to 8 bytes as contiguous hex.
|
||||
func hexBytes(b []byte) string {
|
||||
var sb strings.Builder
|
||||
for i, c := range b {
|
||||
if i == 8 {
|
||||
break
|
||||
}
|
||||
if i > 0 {
|
||||
sb.WriteByte(' ')
|
||||
}
|
||||
fmt.Fprintf(&sb, "%02x", c)
|
||||
}
|
||||
return sb.String()
|
||||
}
|
||||
+767
-476
File diff suppressed because it is too large
Load Diff
+225
-2
@@ -7,9 +7,15 @@ import (
|
||||
"bytes"
|
||||
"io"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strings"
|
||||
"syscall"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||
)
|
||||
|
||||
const clean = "#include \"textflag.h\"\n" +
|
||||
@@ -216,8 +222,9 @@ func TestCmdVersion(t *testing.T) {
|
||||
if code != 0 {
|
||||
t.Fatalf("code = %d", code)
|
||||
}
|
||||
if !strings.Contains(out, version) {
|
||||
t.Errorf("version output %q does not mention %q", out, version)
|
||||
got := version()
|
||||
if !strings.Contains(out, got) {
|
||||
t.Errorf("version output %q does not mention %q", out, got)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -237,3 +244,219 @@ func TestCmdArgErrors(t *testing.T) {
|
||||
t.Errorf("cmdParse() code = %d, want 2", code)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUsageExitCodes pins the exit-code contract for the commands whose main
|
||||
// dispatches on a returned error: a wrong argument set exits 2, the same as
|
||||
// the commands that count their arguments themselves, while a runtime
|
||||
// failure (an unreadable file) keeps exit 1.
|
||||
func TestUsageExitCodes(t *testing.T) {
|
||||
for name, err := range map[string]error{
|
||||
"audit-instructions extra argument": cmdAuditInstructions([]string{"amd64", "extra"}),
|
||||
"audit-instructions unknown arch": cmdAuditInstructions([]string{"mips"}),
|
||||
"audit-instructions corpus extra": cmdAuditInstructions([]string{"--corpus", "a", "b"}),
|
||||
"scaffold no arguments": cmdScaffold(nil),
|
||||
"scaffold extra arguments": cmdScaffold([]string{"differential", "a.s", "b.s"}),
|
||||
} {
|
||||
if err == nil {
|
||||
t.Errorf("%s: expected an error", name)
|
||||
continue
|
||||
}
|
||||
if code := exitCodeFor(err); code != 2 {
|
||||
t.Errorf("%s: exit code = %d, want 2 (err: %v)", name, code, err)
|
||||
}
|
||||
}
|
||||
if err := cmdScaffold([]string{"differential", "/nonexistent/file.s"}); err == nil {
|
||||
t.Error("scaffold on a missing file should fail")
|
||||
} else if code := exitCodeFor(err); code != 1 {
|
||||
t.Errorf("scaffold on a missing file: exit code = %d, want 1", code)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCmdAsmFormatValidation checks that an unknown --format exits 2 with
|
||||
// and without -o, instead of assembling and silently dumping a raw image.
|
||||
func TestCmdAsmFormatValidation(t *testing.T) {
|
||||
path := writeTemp(t, "f_amd64.s", clean)
|
||||
out := filepath.Join(t.TempDir(), "f.bin")
|
||||
if _, _, code := capture(func() int { return cmdAsm([]string{"--format", "bogus", path}) }); code != 2 {
|
||||
t.Errorf("asm --format bogus without -o: code = %d, want 2", code)
|
||||
}
|
||||
if _, _, code := capture(func() int { return cmdAsm([]string{"--format", "bogus", "-o", out, path}) }); code != 2 {
|
||||
t.Errorf("asm --format bogus with -o: code = %d, want 2", code)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCmdAsmOutputFile pins the documented -o behaviour: the output goes to
|
||||
// the file and stdout carries no hex dump; without -o the dump is the output.
|
||||
func TestCmdAsmOutputFile(t *testing.T) {
|
||||
path := writeTemp(t, "f_amd64.s", clean)
|
||||
out := filepath.Join(t.TempDir(), "f.bin")
|
||||
stdout, _, code := capture(func() int { return cmdAsm([]string{"-o", out, path}) })
|
||||
if code != 0 {
|
||||
t.Fatalf("code = %d", code)
|
||||
}
|
||||
if strings.Contains(stdout, "0000:") {
|
||||
t.Errorf("stdout carries a hex dump despite -o:\n%s", stdout)
|
||||
}
|
||||
if !strings.Contains(stdout, "wrote ") {
|
||||
t.Errorf("stdout misses the wrote line:\n%s", stdout)
|
||||
}
|
||||
b, err := os.ReadFile(out)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(b) == 0 {
|
||||
t.Error("the output file is empty")
|
||||
}
|
||||
|
||||
stdout, _, code = capture(func() int { return cmdAsm([]string{path}) })
|
||||
if code != 0 {
|
||||
t.Fatalf("without -o: code = %d", code)
|
||||
}
|
||||
if !strings.Contains(stdout, "0000:") {
|
||||
t.Errorf("without -o the hex dump is missing:\n%s", stdout)
|
||||
}
|
||||
}
|
||||
|
||||
// TestVerifyNonJITAMD64GroundTruth drives the cross-architecture
|
||||
// ground-truth path for an amd64 kernel: the path a host of any other
|
||||
// architecture takes, which must compare against the toolchain rather than
|
||||
// refuse to run.
|
||||
func TestVerifyNonJITAMD64GroundTruth(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("runs go tool asm")
|
||||
}
|
||||
path := writeTemp(t, "f_amd64.s", clean)
|
||||
out, _, code := capture(func() int { return cmdVerifyNonJIT(path, arch.AMD64, true, false) })
|
||||
if code != 0 {
|
||||
t.Fatalf("code = %d (%s)", code, out)
|
||||
}
|
||||
if !strings.Contains(out, "1/1 matched") {
|
||||
t.Errorf("output misses the matched report:\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
// TestVerifySmokeCrashIsolation checks that a function faulting on its
|
||||
// zeroed smoke arguments is reported as CRASH by a child process instead of
|
||||
// killing `gasm verify` itself.
|
||||
func TestVerifySmokeCrashIsolation(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("builds the gasm binary")
|
||||
}
|
||||
if runtime.GOARCH != "amd64" {
|
||||
t.Skip("amd64 JIT only")
|
||||
}
|
||||
bin := filepath.Join(t.TempDir(), "gasm")
|
||||
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
|
||||
t.Fatalf("build gasm: %v\n%s", err, out)
|
||||
}
|
||||
src := filepath.Join(t.TempDir(), "crash_amd64.s")
|
||||
kernel := "#include \"textflag.h\"\n" +
|
||||
"\n" +
|
||||
"// func Fault(x []byte) int\n" +
|
||||
"TEXT ·Fault(SB), NOSPLIT, $0-32\n" +
|
||||
"\tMOVQ\tx+0(FP), AX\n" +
|
||||
"\tMOVQ\t(AX), AX // faults on the zeroed nil pointer\n" +
|
||||
"\tMOVQ\tAX, ret+24(FP)\n" +
|
||||
"\tRET\n"
|
||||
if err := os.WriteFile(src, []byte(kernel), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
cmd := exec.Command(bin, "verify", "-smoke", src)
|
||||
out, err := cmd.CombinedOutput()
|
||||
if err == nil {
|
||||
t.Fatalf("expected a failure report, got success:\n%s", out)
|
||||
}
|
||||
if exitErr, ok := err.(*exec.ExitError); ok {
|
||||
if ws, ok := exitErr.Sys().(syscall.WaitStatus); ok && ws.Signaled() {
|
||||
t.Fatalf("verify died from %v; the crash was not isolated:\n%s", ws.Signal(), out)
|
||||
}
|
||||
}
|
||||
if !strings.Contains(string(out), "CRASH") {
|
||||
t.Errorf("output does not report CRASH:\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSweepCheckLines(t *testing.T) {
|
||||
out := []byte("crash_amd64.s: 1 functions JIT-loaded\n" +
|
||||
" Fault: 21 bytes, args=32, frame=0 NOSPLIT\n" +
|
||||
" smoke: OK\n" +
|
||||
" abi: clean (10 varied inputs)\n")
|
||||
want := " smoke: OK\n abi: clean (10 varied inputs)"
|
||||
if got := sweepCheckLines(out); got != want {
|
||||
t.Errorf("sweepCheckLines = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRunCorpusAudit drives the corpus audit over a small fixture tree: one
|
||||
// suffixed amd64 file, one suffixed arm64 file whose body is not arm64, one
|
||||
// generic file, and one file that does not parse.
|
||||
func TestRunCorpusAudit(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
write := func(name, src string) {
|
||||
t.Helper()
|
||||
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
write("good_amd64.s", "#include \"textflag.h\"\nTEXT ·add(SB), NOSPLIT, $0-0\n\tMOVQ AX, BX\n\tRET\n")
|
||||
write("bad_arm64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0-0\n\tMOVQ AX, BX\n\tRET\n")
|
||||
write("generic.s", "#include \"textflag.h\"\nTEXT ·g(SB), NOSPLIT, $0-0\n\tRET\n")
|
||||
write("broken.s", "#include \"textflag.h\"\nTEXT ·b(SB), NOSPLIT, $0-0\n\tJMP nowhere\n\tRET\n")
|
||||
|
||||
stats, err := runCorpusAudit(dir, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("runCorpusAudit: %v", err)
|
||||
}
|
||||
if stats.files != 4 {
|
||||
t.Errorf("files = %d, want 4", stats.files)
|
||||
}
|
||||
if stats.generic != 2 {
|
||||
t.Errorf("generic = %d, want 2 (generic.s and broken.s)", stats.generic)
|
||||
}
|
||||
// good_amd64 and generic.s assemble everywhere they are attempted.
|
||||
if stats.full != 2 {
|
||||
t.Errorf("full = %d, want 2", stats.full)
|
||||
}
|
||||
get := func(name string) *corpusTally {
|
||||
for i, tg := range stats.targets {
|
||||
if tg.name == name {
|
||||
return stats.tallies[i]
|
||||
}
|
||||
}
|
||||
t.Fatalf("no tally for %s", name)
|
||||
return nil
|
||||
}
|
||||
// amd64: good_amd64 + generic.s + broken.s; the broken file fails to parse.
|
||||
if a := get("amd64"); a.attempted != 3 || a.assembled != 2 {
|
||||
t.Errorf("amd64 = %d/%d, want 2/3", a.assembled, a.attempted)
|
||||
}
|
||||
// arm64: bad_arm64 (MOVQ is not arm64) + generic.s + broken.s.
|
||||
if a := get("arm64"); a.attempted != 3 || a.assembled != 1 {
|
||||
t.Errorf("arm64 = %d/%d, want 1/3", a.assembled, a.attempted)
|
||||
}
|
||||
if r := get("amd64").reasons["instruction not encodable"]; r != 0 {
|
||||
t.Errorf("amd64 unexpected unencodable reason: %d", r)
|
||||
}
|
||||
if r := get("arm64").reasons["instruction not encodable"]; r != 1 {
|
||||
t.Errorf("arm64 unencodable reasons = %d, want 1", r)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCompareGroundTruthPadding pins the padding-aware ground-truth
|
||||
// comparison: the toolchain pads text symbols to 16-byte boundaries, so
|
||||
// trailing zeros in the reference must not read as a mismatch, while any
|
||||
// non-zero tail still must.
|
||||
func TestCompareGroundTruthPadding(t *testing.T) {
|
||||
code := []byte{0x48, 0x8b, 0x07, 0xc3} // 4 bytes, not a multiple of 16
|
||||
img := &asm.Image{Code: code, Funcs: []asm.FuncLayout{{Name: "f", Offset: 0, Size: len(code)}}}
|
||||
padded := append(append([]byte(nil), code...), 0, 0, 0)
|
||||
matched, total, diffs := compareGroundTruth(img, map[string][]byte{"f": padded})
|
||||
if matched != 1 || total != 1 || diffs != 0 {
|
||||
t.Fatalf("zero padding should match: matched=%d total=%d diffs=%d", matched, total, diffs)
|
||||
}
|
||||
dirty := append(append([]byte(nil), code...), 0, 0x90, 0)
|
||||
matched, _, diffs = compareGroundTruth(img, map[string][]byte{"f": dirty})
|
||||
if matched != 0 || diffs != 1 {
|
||||
t.Fatalf("non-zero padding must mismatch: matched=%d diffs=%d", matched, diffs)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,241 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// TestManPagesTrackTheCLI builds the binary once, then compares every
|
||||
// command's live `-h` output with its docs/man/gasm-<command>.1 page: the
|
||||
// flag sets must agree both ways, and the page's SYNOPSIS line must carry
|
||||
// the command's usage line. A flag or a usage change that skips the man
|
||||
// page fails here, so the pages cannot drift from the binary.
|
||||
func TestManPagesTrackTheCLI(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("builds the gasm binary")
|
||||
}
|
||||
bin := filepath.Join(t.TempDir(), "gasm")
|
||||
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
|
||||
t.Fatalf("build gasm: %v\n%s", err, out)
|
||||
}
|
||||
|
||||
for _, cmd := range []string{
|
||||
"tokens", "parse", "fmt", "lint", "asm", "dis", "verify",
|
||||
"debug", "diff", "profile", "audit-instructions", "scaffold", "lsp",
|
||||
} {
|
||||
t.Run(cmd, func(t *testing.T) {
|
||||
raw, err := os.ReadFile(filepath.Join("..", "..", "docs", "man", "gasm-"+cmd+".1"))
|
||||
if err != nil {
|
||||
t.Fatalf("read man page: %v", err)
|
||||
}
|
||||
page := string(raw)
|
||||
|
||||
out, _ := exec.Command(bin, cmd, "-h").CombinedOutput()
|
||||
help := string(out)
|
||||
|
||||
binFlags := helpFlags(help)
|
||||
pageFlags := roffFlags(page)
|
||||
for f := range binFlags {
|
||||
if !pageFlags[f] {
|
||||
t.Errorf("flag -%s is in the binary's help but missing from the man page", f)
|
||||
}
|
||||
}
|
||||
for f := range pageFlags {
|
||||
if !binFlags[f] {
|
||||
t.Errorf("flag -%s is in the man page but the binary does not accept it", f)
|
||||
}
|
||||
}
|
||||
|
||||
want := helpUsage(help)
|
||||
got := roffSynopsis(page)
|
||||
if want != "" && got != want {
|
||||
t.Errorf("SYNOPSIS drift:\n page: %s\nbinary: %s", got, want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestManCommandsTrackHelp compares the gasm(1) COMMANDS list with the
|
||||
// top-level help output, so a subcommand added to the binary cannot miss
|
||||
// its man entry and a stale entry cannot outlive its command.
|
||||
func TestManCommandsTrackHelp(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("builds the gasm binary")
|
||||
}
|
||||
bin := filepath.Join(t.TempDir(), "gasm")
|
||||
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
|
||||
t.Fatalf("build gasm: %v\n%s", err, out)
|
||||
}
|
||||
|
||||
raw, err := os.ReadFile(filepath.Join("..", "..", "docs", "man", "gasm.1"))
|
||||
if err != nil {
|
||||
t.Fatalf("read man page: %v", err)
|
||||
}
|
||||
|
||||
helpOut, err := exec.Command(bin, "--help").Output()
|
||||
if err != nil {
|
||||
t.Fatalf("gasm --help: %v", err)
|
||||
}
|
||||
|
||||
binCmds := helpCommands(string(helpOut))
|
||||
pageCmds := roffCommands(string(raw))
|
||||
for c := range binCmds {
|
||||
if !pageCmds[c] {
|
||||
t.Errorf("command %q is in the binary's help but missing from gasm(1) COMMANDS", c)
|
||||
}
|
||||
}
|
||||
for c := range pageCmds {
|
||||
if !binCmds[c] {
|
||||
t.Errorf("command %q is in gasm(1) COMMANDS but the binary does not list it", c)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// helpFlags extracts the flag names from a `gasm <cmd> -h` output.
|
||||
func helpFlags(help string) map[string]bool {
|
||||
m := map[string]bool{}
|
||||
inFlags := false
|
||||
for line := range strings.SplitSeq(help, "\n") {
|
||||
if strings.TrimRight(line, " \t") == "Flags:" {
|
||||
inFlags = true
|
||||
continue
|
||||
}
|
||||
if !inFlags {
|
||||
continue
|
||||
}
|
||||
if !strings.HasPrefix(line, " -") {
|
||||
continue
|
||||
}
|
||||
token := strings.FieldsFunc(strings.TrimLeft(line, " "), func(r rune) bool {
|
||||
return r == ' ' || r == '\t'
|
||||
})
|
||||
if len(token) == 0 {
|
||||
continue
|
||||
}
|
||||
m[strings.TrimLeft(token[0], "-")] = true
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
var roffEscape = regexp.MustCompile(`\\f[BIRP]`)
|
||||
|
||||
// roffFlags extracts the flag names from a man page's OPTIONS section.
|
||||
func roffFlags(page string) map[string]bool {
|
||||
m := map[string]bool{}
|
||||
inOptions := false
|
||||
for line := range strings.SplitSeq(page, "\n") {
|
||||
if strings.HasPrefix(line, ".SH ") {
|
||||
inOptions = strings.HasPrefix(line, ".SH OPTIONS")
|
||||
continue
|
||||
}
|
||||
if !inOptions {
|
||||
continue
|
||||
}
|
||||
// Flag entries are written as either `.B \-flag` or `\fB\-flag`.
|
||||
var body string
|
||||
switch {
|
||||
case strings.HasPrefix(line, `.B \-`):
|
||||
body = line[3:]
|
||||
case strings.HasPrefix(line, `\fB\-`):
|
||||
body = line[1:]
|
||||
default:
|
||||
continue
|
||||
}
|
||||
name := roffEscape.ReplaceAllString(body, "")
|
||||
name = strings.ReplaceAll(name, `\-`, "-")
|
||||
name = strings.TrimSpace(name)
|
||||
if i := strings.IndexAny(name, " \t"); i >= 0 {
|
||||
name = name[:i]
|
||||
}
|
||||
m[strings.TrimLeft(name, "-")] = true
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// helpCommands extracts the command names from the top-level help output's
|
||||
// Commands section.
|
||||
func helpCommands(help string) map[string]bool {
|
||||
m := map[string]bool{}
|
||||
inCmds := false
|
||||
for line := range strings.SplitSeq(help, "\n") {
|
||||
if strings.TrimSpace(line) == "Commands:" {
|
||||
inCmds = true
|
||||
continue
|
||||
}
|
||||
if !inCmds {
|
||||
continue
|
||||
}
|
||||
t := strings.TrimSpace(line)
|
||||
if t == "" {
|
||||
break
|
||||
}
|
||||
name, _, _ := strings.Cut(t, " ")
|
||||
m[name] = true
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// roffCommands extracts the command names from gasm(1)'s COMMANDS section,
|
||||
// where each entry is written as `.B gasm\-<name>(1)` or `.B gasm <name>`.
|
||||
func roffCommands(page string) map[string]bool {
|
||||
m := map[string]bool{}
|
||||
inCmds := false
|
||||
for line := range strings.SplitSeq(page, "\n") {
|
||||
if strings.HasPrefix(line, ".SH ") {
|
||||
inCmds = strings.HasPrefix(line, ".SH COMMANDS")
|
||||
continue
|
||||
}
|
||||
if !inCmds || !strings.HasPrefix(line, ".B gasm") {
|
||||
continue
|
||||
}
|
||||
entry := strings.ReplaceAll(strings.TrimPrefix(line, ".B "), `\-`, "-")
|
||||
entry = strings.TrimSuffix(entry, "(1)")
|
||||
switch {
|
||||
case strings.HasPrefix(entry, "gasm-"):
|
||||
m[strings.TrimPrefix(entry, "gasm-")] = true
|
||||
case strings.HasPrefix(entry, "gasm "):
|
||||
m[strings.TrimPrefix(entry, "gasm ")] = true
|
||||
}
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// helpUsage returns the command's usage line without the "Usage: " prefix.
|
||||
func helpUsage(help string) string {
|
||||
for line := range strings.SplitSeq(help, "\n") {
|
||||
if strings.HasPrefix(line, "Usage: ") {
|
||||
return normaliseUsage(line[len("Usage: "):])
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// roffSynopsis returns the page's SYNOPSIS usage line, unescaped.
|
||||
func roffSynopsis(page string) string {
|
||||
inSyn := false
|
||||
for line := range strings.SplitSeq(page, "\n") {
|
||||
if strings.HasPrefix(line, ".SH ") {
|
||||
inSyn = strings.HasPrefix(line, ".SH SYNOPSIS")
|
||||
continue
|
||||
}
|
||||
if !inSyn || !strings.HasPrefix(line, ".B ") {
|
||||
continue
|
||||
}
|
||||
return normaliseUsage(strings.ReplaceAll(line[3:], `\-`, "-"))
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// normaliseUsage flattens whitespace and drops the roff font escapes so that
|
||||
// the binary's usage line and the page's SYNOPSIS line compare equal.
|
||||
func normaliseUsage(s string) string {
|
||||
s = roffEscape.ReplaceAllString(s, "")
|
||||
return strings.Join(strings.Fields(s), " ")
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// writeTree writes a directory of files and returns its root.
|
||||
func writeTree(t *testing.T, files map[string]string) string {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
for name, content := range files {
|
||||
path := filepath.Join(dir, name)
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
return dir
|
||||
}
|
||||
|
||||
// TestAsmMacroAndIncludeEndToEnd drives `gasm asm` over a source with an
|
||||
// in-file parameterised macro and an include resolved through -I, and checks
|
||||
// the assembled bytes came from the expansion (the loop body counts six
|
||||
// increments, two per expanded iteration).
|
||||
func TestAsmMacroAndIncludeEndToEnd(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("runs the assembler end to end")
|
||||
}
|
||||
dir := writeTree(t, map[string]string{
|
||||
"inc/consts.h": "#define NITER 3\n",
|
||||
"main_amd64.s": "#include \"textflag.h\"\n" +
|
||||
"#include \"consts.h\"\n" +
|
||||
"#define STEP(r) ADDQ $1, r; ADDQ $1, r\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0-8\n" +
|
||||
"\tXORQ AX, AX\n" +
|
||||
"\tMOVQ $NITER, CX\n" +
|
||||
"loop:\n" +
|
||||
"\tSTEP(AX)\n" +
|
||||
"\tDECQ CX\n" +
|
||||
"\tJNZ loop\n" +
|
||||
"\tMOVQ AX, ret+0(FP)\n" +
|
||||
"\tRET\n",
|
||||
})
|
||||
stdout, stderr, code := capture(func() int {
|
||||
return cmdAsm([]string{"-I", filepath.Join(dir, "inc"), "-GOARCH", "amd64", filepath.Join(dir, "main_amd64.s")})
|
||||
})
|
||||
if code != 0 {
|
||||
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
|
||||
}
|
||||
// The macro expanded to two ADDQ $1 encodings in the static body; the
|
||||
// iteration count lives in the runtime loop.
|
||||
if n := strings.Count(stdout, "83 c0 01"); n != 2 {
|
||||
t.Errorf("found %d ADDQ $1 encodings in the image, want 2:\n%s", n, stdout)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAsmIncludeResolutionOrder pins the -I search order end to end: the
|
||||
// including file's directory wins over the -I directories.
|
||||
func TestAsmIncludeResolutionOrder(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("runs the assembler end to end")
|
||||
}
|
||||
dir := writeTree(t, map[string]string{
|
||||
"src/main_amd64.s": "#include \"textflag.h\"\n" +
|
||||
"#include \"vals.h\"\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||
"\tMOVQ $VAL, AX\n" +
|
||||
"\tRET\n",
|
||||
"src/vals.h": "#define VAL 1\n",
|
||||
"late/vals.h": "#define VAL 2\n",
|
||||
"early/vals.h": "#define VAL 3\n",
|
||||
})
|
||||
stdout, stderr, code := capture(func() int {
|
||||
return cmdAsm([]string{"-I", filepath.Join(dir, "early"), "-I", filepath.Join(dir, "late"),
|
||||
"-GOARCH", "amd64", filepath.Join(dir, "src", "main_amd64.s")})
|
||||
})
|
||||
if code != 0 {
|
||||
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
|
||||
}
|
||||
// VAL came from src/vals.h, not from either -I directory: the image
|
||||
// loads the immediate 1.
|
||||
if !strings.Contains(stdout, "b8 01 00 00 00") {
|
||||
t.Errorf("expected the source-directory VAL (immediate 1) in:\n%s", stdout)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAsmMissingIncludeIsAnError pins the diagnostic for an include that
|
||||
// resolves nowhere on the assembly path.
|
||||
func TestAsmMissingIncludeIsAnError(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("runs the assembler end to end")
|
||||
}
|
||||
path := writeTemp(t, "main_amd64.s", "#include \"textflag.h\"\n#include \"nothere.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
|
||||
_, stderr, code := capture(func() int { return cmdAsm([]string{"-GOARCH", "amd64", path}) })
|
||||
if code == 0 {
|
||||
t.Fatal("gasm asm accepted a file whose include resolves nowhere")
|
||||
}
|
||||
if !strings.Contains(stderr, `#include "nothere.h"`) {
|
||||
t.Errorf("stderr does not name the failing include: %s", stderr)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,345 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
gasmast "sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||
gasmparser "sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||
)
|
||||
|
||||
// cmdScaffold generates a differential test skeleton for every kernel in a
|
||||
// file: a Go test that seeds random states, drives both the assembly kernel
|
||||
// and a caller-provided portable reference, and compares the outputs
|
||||
// byte-for-byte. The lesson this encodes: a pipeline-level fuzz cannot see
|
||||
// an unwired kernel, only a direct-call differential against the portable
|
||||
// specification can, so every kernel ships with one.
|
||||
//
|
||||
// The generated file follows two conventions the caller fills in:
|
||||
// - the assembly symbols resolve because the test lives in the kernel's
|
||||
// own package (the //go:noescape declarations reference them);
|
||||
// - each kernel gets a <name>Portable Go function the author implements as
|
||||
// the specification, and the test fails on the first divergent byte.
|
||||
func cmdScaffold(args []string) error {
|
||||
fs := newCommand("scaffold", "gasm scaffold differential <file.s>", `
|
||||
Print a differential test skeleton for every // func signature in FILE.
|
||||
The test seeds random states, drives the kernel and a portable reference
|
||||
(<name>Portable), and compares outputs byte-for-byte. Write the reference
|
||||
bodies, place the file in the kernel's package, and run it in CI.
|
||||
`)
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return err
|
||||
}
|
||||
rest := fs.Args()
|
||||
// The first positional word is the scaffold style; "differential" is the
|
||||
// only one today.
|
||||
if len(rest) > 0 && rest[0] == "differential" {
|
||||
rest = rest[1:]
|
||||
}
|
||||
if len(rest) != 1 {
|
||||
return &usageError{fmt.Errorf("usage: gasm scaffold differential <file.s>")}
|
||||
}
|
||||
path := rest[0]
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
f, errs := gasmparser.Parse(path, string(src))
|
||||
if len(errs) > 0 {
|
||||
return fmt.Errorf("parse: %v", errs[0])
|
||||
}
|
||||
|
||||
var out strings.Builder
|
||||
out.WriteString(headerComment)
|
||||
out.WriteString("package " + packageName + "\n\n")
|
||||
out.WriteString("import (\n\t\"bytes\"\n\t\"math/rand\"\n\t\"testing\"\n)\n\n")
|
||||
out.WriteString(generatedHelpers)
|
||||
|
||||
kernels := 0
|
||||
for _, d := range f.Decls {
|
||||
txt, ok := d.(*gasmast.Text)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
params, results, ok := parseSig(txt.Doc)
|
||||
if !ok || len(params) == 0 {
|
||||
continue
|
||||
}
|
||||
kernels++
|
||||
name := txt.Name.Name
|
||||
fmt.Fprintf(&out, "// %sPortable is the specification %s is pinned against:\n", name, name)
|
||||
fmt.Fprintf(&out, "// fill in a straightforward implementation of the same contract.\n")
|
||||
fmt.Fprintf(&out, "func %sPortable(%s) (%s) {\n\tpanic(\"implement the portable specification\")\n}\n\n", name, paramDecl(params), resultDecl(results))
|
||||
|
||||
fmt.Fprintf(&out, "func Test%sDifferential(t *testing.T) {\n", strings.ToUpper(name[:1])+name[1:])
|
||||
fmt.Fprintf(&out, "\trng := rand.New(rand.NewSource(1))\n")
|
||||
fmt.Fprintf(&out, "\tfor range 1000 {\n")
|
||||
// Seed two independent argument sets per iteration: the kernel runs
|
||||
// on set A, the portable reference on set B, so in-place writes
|
||||
// through pointer/slice arguments cannot contaminate the other side.
|
||||
var sliceNames []string
|
||||
seen := map[string]bool{}
|
||||
aArgs := make([]string, 0, len(params))
|
||||
bArgs := make([]string, 0, len(params))
|
||||
for _, p := range params {
|
||||
a, b, slices := genParamSeed(&out, p, seen)
|
||||
aArgs = append(aArgs, a)
|
||||
bArgs = append(bArgs, b)
|
||||
sliceNames = append(sliceNames, slices...)
|
||||
}
|
||||
fmt.Fprintf(&out, "\t\tgot := %s(%s)\n", name, strings.Join(aArgs, ", "))
|
||||
fmt.Fprintf(&out, "\t\twant := %sPortable(%s)\n", name, strings.Join(bArgs, ", "))
|
||||
fmt.Fprintf(&out, "\t\tif !bytes.Equal(outputBytes(got), outputBytes(want)) {\n")
|
||||
fmt.Fprintf(&out, "\t\t\tt.Fatalf(\"kernel diverges from the portable spec (seed 1, deterministic)\")\n")
|
||||
fmt.Fprintf(&out, "\t\t}\n")
|
||||
for _, s := range sliceNames {
|
||||
fmt.Fprintf(&out, "\t\tif !bytes.Equal(outputBytes(%sA), outputBytes(%sB)) {\n", s, s)
|
||||
fmt.Fprintf(&out, "\t\t\tt.Fatalf(\"kernel mutated %%q differently (seed 1, deterministic)\", %q)\n", s)
|
||||
fmt.Fprintf(&out, "\t\t}\n")
|
||||
}
|
||||
fmt.Fprintf(&out, "\t}\n}\n\n")
|
||||
}
|
||||
if kernels == 0 {
|
||||
return fmt.Errorf("%s: no // func signatures found; add one doc comment per kernel", path)
|
||||
}
|
||||
os.Stdout.WriteString(out.String())
|
||||
return nil
|
||||
}
|
||||
|
||||
const packageName = "yourpkg"
|
||||
|
||||
const headerComment = `// Code generated by gasm scaffold differential; EDIT THE PANICS.
|
||||
// Each Test*Differential drives the assembly kernel and its portable
|
||||
// reference over the same random states and compares the outputs.
|
||||
// Place this file in the kernel's own package so the symbols resolve.
|
||||
|
||||
`
|
||||
|
||||
// sigParam is one parsed // func parameter.
|
||||
type sigParam struct {
|
||||
Names []string
|
||||
Type string
|
||||
}
|
||||
|
||||
type sigResult struct {
|
||||
Names []string
|
||||
Type string
|
||||
}
|
||||
|
||||
// parseSig parses the // func signature of a doc comment.
|
||||
func parseSig(doc string) ([]sigParam, []sigResult, bool) {
|
||||
var line string
|
||||
for l := range strings.SplitSeq(doc, "\n") {
|
||||
if t := strings.TrimSpace(l); strings.HasPrefix(t, "func ") {
|
||||
line = t
|
||||
break
|
||||
}
|
||||
}
|
||||
if line == "" {
|
||||
return nil, nil, false
|
||||
}
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "sig.go", "package p\n"+line+" {}\n", 0)
|
||||
if err != nil {
|
||||
return nil, nil, false
|
||||
}
|
||||
fd, ok := f.Decls[0].(*ast.FuncDecl)
|
||||
if !ok || fd.Type == nil {
|
||||
return nil, nil, false
|
||||
}
|
||||
var params []sigParam
|
||||
for _, field := range fd.Type.Params.List {
|
||||
typ := exprString(field.Type)
|
||||
if len(field.Names) == 0 {
|
||||
params = append(params, sigParam{Names: []string{""}, Type: typ})
|
||||
continue
|
||||
}
|
||||
// Shared names (`L, result *byte`) expand to one entry per name:
|
||||
// every name is a separate argument at the call site.
|
||||
for _, n := range field.Names {
|
||||
params = append(params, sigParam{Names: []string{n.Name}, Type: typ})
|
||||
}
|
||||
}
|
||||
var results []sigResult
|
||||
if fd.Type.Results != nil {
|
||||
for _, field := range fd.Type.Results.List {
|
||||
results = append(results, sigResult{Names: identNames(field.Names), Type: exprString(field.Type)})
|
||||
}
|
||||
}
|
||||
return params, results, true
|
||||
}
|
||||
|
||||
func identNames(idents []*ast.Ident) []string {
|
||||
var out []string
|
||||
for _, id := range idents {
|
||||
out = append(out, id.Name)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func exprString(e ast.Expr) string {
|
||||
switch t := e.(type) {
|
||||
case *ast.Ident:
|
||||
return t.Name
|
||||
case *ast.StarExpr:
|
||||
return "*" + exprString(t.X)
|
||||
case *ast.SelectorExpr:
|
||||
return exprString(t.X) + "." + t.Sel.Name
|
||||
case *ast.ArrayType:
|
||||
if t.Len == nil {
|
||||
return "[]" + exprString(t.Elt)
|
||||
}
|
||||
return "[N]" + exprString(t.Elt)
|
||||
}
|
||||
return "interface{}"
|
||||
}
|
||||
|
||||
// paramDecl renders a parameter list for the portable reference signature.
|
||||
func paramDecl(params []sigParam) string {
|
||||
var parts []string
|
||||
for _, p := range params {
|
||||
if len(p.Names) == 0 {
|
||||
parts = append(parts, p.Type)
|
||||
continue
|
||||
}
|
||||
for _, n := range p.Names {
|
||||
parts = append(parts, n+" "+p.Type)
|
||||
}
|
||||
}
|
||||
return strings.Join(parts, ", ")
|
||||
}
|
||||
|
||||
// resultDecl renders a result list; unnamed results keep bare types.
|
||||
func resultDecl(results []sigResult) string {
|
||||
if len(results) == 0 {
|
||||
return ""
|
||||
}
|
||||
var parts []string
|
||||
for _, r := range results {
|
||||
parts = append(parts, r.Type)
|
||||
}
|
||||
return strings.Join(parts, ", ")
|
||||
}
|
||||
|
||||
// genParamSeed emits the seeding statements for one parameter and returns
|
||||
// the kernel-side (A) and reference-side (B) argument expressions, plus the
|
||||
// names of any slice variables written in place (compared after the calls).
|
||||
func genParamSeed(out *strings.Builder, p sigParam, seen map[string]bool) (aArg, bArg string, slices []string) {
|
||||
name := p.Names[0]
|
||||
elem := strings.TrimPrefix(p.Type, "*")
|
||||
isSlice := strings.HasPrefix(p.Type, "[]")
|
||||
if isSlice {
|
||||
elem = strings.TrimPrefix(p.Type, "[]")
|
||||
}
|
||||
switch {
|
||||
case isSlice:
|
||||
v := uniqueName(seen, name)
|
||||
fmt.Fprintf(out, "\t\t%sA := make([]%s, 1+rng.Intn(512))\n", v, elem)
|
||||
fmt.Fprintf(out, "\t\t%sB := make([]%s, len(%sA))\n", v, elem, v)
|
||||
fmt.Fprintf(out, "\t\tfor i := range %sA {\n", v)
|
||||
fmt.Fprintf(out, "\t\t\tw%s := %s(rng.Intn(256))\n", v, goCast(elem))
|
||||
fmt.Fprintf(out, "\t\t\t%sA[i] = w%s\n", v, v)
|
||||
fmt.Fprintf(out, "\t\t\t%sB[i] = w%s\n", v, v)
|
||||
fmt.Fprintf(out, "\t\t}\n")
|
||||
return v, v, []string{v}
|
||||
case strings.HasPrefix(p.Type, "*"):
|
||||
v := uniqueName(seen, name)
|
||||
fmt.Fprintf(out, "\t\tvar %sA, %sB %s\n", v, v, elem)
|
||||
fmt.Fprintf(out, "\t\tw%s := %s(rng.Intn(256))\n", v, goCast(elem))
|
||||
fmt.Fprintf(out, "\t\t%sA = w%s\n", v, v)
|
||||
fmt.Fprintf(out, "\t\t%sB = w%s\n", v, v)
|
||||
return "&" + v + "A", "&" + v + "B", nil
|
||||
default:
|
||||
v := uniqueName(seen, name)
|
||||
fmt.Fprintf(out, "\t\tw%s := %s(rng.Intn(512))\n", v, goCast(""))
|
||||
fmt.Fprintf(out, "\t\tvar %sA, %sB %s = w%s, w%s\n", v, v, p.Type, v, v)
|
||||
return v + "A", v + "B", nil
|
||||
}
|
||||
}
|
||||
|
||||
// uniqueName de-duplicates seeded variable names when one kernel takes two
|
||||
// parameters of the same name (impossible in Go) or a name repeats across
|
||||
// kernels in one file.
|
||||
func uniqueName(seen map[string]bool, base string) string {
|
||||
if base == "" {
|
||||
base = "arg"
|
||||
}
|
||||
if !seen[base] {
|
||||
seen[base] = true
|
||||
return base
|
||||
}
|
||||
for i := 2; ; i++ {
|
||||
cand := fmt.Sprintf("%s%d", base, i)
|
||||
if !seen[cand] {
|
||||
seen[cand] = true
|
||||
return cand
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// goCast returns the conversion turning rng.Intn into the element type.
|
||||
func goCast(elem string) string {
|
||||
switch elem {
|
||||
case "byte", "uint8":
|
||||
return "byte"
|
||||
case "int8":
|
||||
return "int8"
|
||||
case "uint16":
|
||||
return "uint16"
|
||||
case "int16":
|
||||
return "int16"
|
||||
case "uint32":
|
||||
return "uint32"
|
||||
case "int32":
|
||||
return "int32"
|
||||
case "uint64":
|
||||
return "uint64"
|
||||
default:
|
||||
return "int"
|
||||
}
|
||||
}
|
||||
|
||||
// generatedHelpers is emitted into every generated test file: outputBytes
|
||||
// narrows returned slices and scalars to a byte form for the comparison.
|
||||
// It lives in the template, not in this binary, because only the generated
|
||||
// file ever calls it.
|
||||
const generatedHelpers = `// outputBytes narrows a returned slice or scalar to bytes for the
|
||||
// comparison; extend the switch when a kernel returns a wider type.
|
||||
func outputBytes(v any) []byte {
|
||||
switch t := v.(type) {
|
||||
case []byte:
|
||||
return t
|
||||
case []int32:
|
||||
b := make([]byte, 4*len(t))
|
||||
for i, x := range t {
|
||||
b[i*4] = byte(x)
|
||||
b[i*4+1] = byte(x >> 8)
|
||||
b[i*4+2] = byte(x >> 16)
|
||||
b[i*4+3] = byte(x >> 24)
|
||||
}
|
||||
return b
|
||||
case []uint16:
|
||||
b := make([]byte, 2*len(t))
|
||||
for i, x := range t {
|
||||
b[i*2] = byte(x)
|
||||
b[i*2+1] = byte(x >> 8)
|
||||
}
|
||||
return b
|
||||
case int:
|
||||
b := make([]byte, 8)
|
||||
for i := range 8 {
|
||||
b[i] = byte(uint64(t) >> (8 * i))
|
||||
}
|
||||
return b
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
`
|
||||
@@ -0,0 +1,140 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"slices"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// unifiedDiff renders a unified diff with three lines of context between the
|
||||
// two line slices, in the form `gofmt -d` prints. An empty result means the
|
||||
// inputs are identical.
|
||||
func unifiedDiff(name string, a, b []string) string {
|
||||
if slices.Equal(a, b) {
|
||||
return ""
|
||||
}
|
||||
var out strings.Builder
|
||||
fmt.Fprintf(&out, "--- %s\n+++ %s\n", name, name)
|
||||
|
||||
// Longest common subsequence over the lines (assembly files are small
|
||||
// enough for the quadratic table).
|
||||
n, m := len(a), len(b)
|
||||
lcs := make([][]int, n+1)
|
||||
for i := range lcs {
|
||||
lcs[i] = make([]int, m+1)
|
||||
}
|
||||
for i := n - 1; i >= 0; i-- {
|
||||
for j := m - 1; j >= 0; j-- {
|
||||
if a[i] == b[j] {
|
||||
lcs[i][j] = lcs[i+1][j+1] + 1
|
||||
} else if lcs[i+1][j] >= lcs[i][j+1] {
|
||||
lcs[i][j] = lcs[i+1][j]
|
||||
} else {
|
||||
lcs[i][j] = lcs[i][j+1]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Walk the LCS once, assigning every op its absolute position in both
|
||||
// files (1-based, the position an insertion sits before).
|
||||
type op struct {
|
||||
kind byte // ' ', '-' or '+'
|
||||
aLine, bLine int
|
||||
text string
|
||||
}
|
||||
var ops []op
|
||||
aPos, bPos := 0, 0
|
||||
emit := func(kind byte, text string) {
|
||||
ops = append(ops, op{kind: kind, aLine: aPos + 1, bLine: bPos + 1, text: text})
|
||||
switch kind {
|
||||
case ' ':
|
||||
aPos++
|
||||
bPos++
|
||||
case '-':
|
||||
aPos++
|
||||
case '+':
|
||||
bPos++
|
||||
}
|
||||
}
|
||||
i, j := 0, 0
|
||||
for i < n && j < m {
|
||||
switch {
|
||||
case a[i] == b[j]:
|
||||
emit(' ', a[i])
|
||||
i++
|
||||
j++
|
||||
case lcs[i+1][j] >= lcs[i][j+1]:
|
||||
emit('-', a[i])
|
||||
i++
|
||||
default:
|
||||
emit('+', b[j])
|
||||
j++
|
||||
}
|
||||
}
|
||||
for ; i < n; i++ {
|
||||
emit('-', a[i])
|
||||
}
|
||||
for ; j < m; j++ {
|
||||
emit('+', b[j])
|
||||
}
|
||||
|
||||
// Group the edits into hunks: consecutive changes separated by more than
|
||||
// twice the context lines start a new hunk.
|
||||
const context = 3
|
||||
var changes []int
|
||||
for k, o := range ops {
|
||||
if o.kind != ' ' {
|
||||
changes = append(changes, k)
|
||||
}
|
||||
}
|
||||
for g := 0; g < len(changes); {
|
||||
last := g
|
||||
for last+1 < len(changes) && changes[last+1]-changes[last]-1 <= 2*context {
|
||||
last++
|
||||
}
|
||||
lo := max(0, changes[g]-context)
|
||||
hi := min(len(ops), changes[last]+1+context)
|
||||
// The header numbers are the first line of each side actually shown:
|
||||
// the first context, deletion or insertion line. A hunk that shows
|
||||
// no old lines is a pure insertion and reports the position it sits
|
||||
// before (0 at the top of the file); the mirror rule holds for a
|
||||
// pure deletion.
|
||||
aStart := ops[lo].aLine - 1
|
||||
bStart := ops[lo].bLine - 1
|
||||
countA, countB := 0, 0
|
||||
for _, o := range ops[lo:hi] {
|
||||
switch o.kind {
|
||||
case ' ':
|
||||
countA++
|
||||
countB++
|
||||
case '-':
|
||||
countA++
|
||||
case '+':
|
||||
countB++
|
||||
}
|
||||
}
|
||||
for _, o := range ops[lo:hi] {
|
||||
if o.kind != '+' {
|
||||
aStart = o.aLine
|
||||
break
|
||||
}
|
||||
}
|
||||
for _, o := range ops[lo:hi] {
|
||||
if o.kind != '-' {
|
||||
bStart = o.bLine
|
||||
break
|
||||
}
|
||||
}
|
||||
fmt.Fprintf(&out, "@@ -%d,%d +%d,%d @@\n", aStart, countA, bStart, countB)
|
||||
for _, o := range ops[lo:hi] {
|
||||
out.WriteByte(o.kind)
|
||||
out.WriteString(o.text)
|
||||
out.WriteByte('\n')
|
||||
}
|
||||
g = last + 1
|
||||
}
|
||||
return out.String()
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func lines(ss ...string) []string { return ss }
|
||||
|
||||
func TestUnifiedDiffIdentical(t *testing.T) {
|
||||
if got := unifiedDiff("f", lines("a", "b"), lines("a", "b")); got != "" {
|
||||
t.Errorf("identical inputs produced %q, want empty", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnifiedDiffSingleChange(t *testing.T) {
|
||||
a := lines("1", "2", "3", "4", "5", "6", "7", "8")
|
||||
b := lines("1", "2", "3!", "4", "5", "6", "7", "8")
|
||||
want := "--- f\n+++ f\n" +
|
||||
"@@ -1,6 +1,6 @@\n" +
|
||||
" 1\n 2\n-3\n+3!\n 4\n 5\n 6\n"
|
||||
if got := unifiedDiff("f", a, b); got != want {
|
||||
t.Errorf("diff = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnifiedDiffInsertAtStart(t *testing.T) {
|
||||
got := unifiedDiff("f", lines("x"), lines("new", "x"))
|
||||
// The single existing line is shown as trailing context, so the hunk
|
||||
// covers it.
|
||||
want := "--- f\n+++ f\n@@ -1,1 +1,2 @@\n+new\n x\n"
|
||||
if got != want {
|
||||
t.Errorf("diff = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnifiedDiffDeleteAtEnd(t *testing.T) {
|
||||
got := unifiedDiff("f", lines("x", "y"), lines("x"))
|
||||
want := "--- f\n+++ f\n@@ -1,2 +1,1 @@\n x\n-y\n"
|
||||
if got != want {
|
||||
t.Errorf("diff = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnifiedDiffTwoHunks(t *testing.T) {
|
||||
var a, b []string
|
||||
for i := 1; i <= 20; i++ {
|
||||
a = append(a, itoa(i))
|
||||
b = append(b, itoa(i))
|
||||
}
|
||||
b[1] = "2!"
|
||||
b[17] = "18!"
|
||||
got := unifiedDiff("f", a, b)
|
||||
if !strings.Contains(got, "@@ -1,5 +1,5 @@\n 1\n-2\n+2!\n 3\n 4\n 5\n") {
|
||||
t.Errorf("first hunk wrong:\n%s", got)
|
||||
}
|
||||
if !strings.Contains(got, "@@ -15,6 +15,6 @@\n 15\n 16\n 17\n-18\n+18!\n 19\n 20\n") {
|
||||
t.Errorf("second hunk wrong:\n%s", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUnifiedDiffAdjacentHunks merges changes separated by exactly twice the
|
||||
// context into one hunk.
|
||||
func TestUnifiedDiffAdjacentHunks(t *testing.T) {
|
||||
a := lines("1", "2", "3", "4", "5", "6", "7", "8")
|
||||
b := slices.Clone(a)
|
||||
b[0] = "1!"
|
||||
b[7] = "8!"
|
||||
got := unifiedDiff("f", a, b)
|
||||
want := "--- f\n+++ f\n" +
|
||||
"@@ -1,8 +1,8 @@\n" +
|
||||
"-1\n+1!\n 2\n 3\n 4\n 5\n 6\n 7\n-8\n+8!\n"
|
||||
if got != want {
|
||||
t.Errorf("diff = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func itoa(n int) string {
|
||||
if n == 0 {
|
||||
return "0"
|
||||
}
|
||||
var buf [4]byte
|
||||
i := len(buf)
|
||||
for n > 0 {
|
||||
i--
|
||||
buf[i] = byte('0' + n%10)
|
||||
n /= 10
|
||||
}
|
||||
return string(buf[i:])
|
||||
}
|
||||
+155
-96
@@ -1,89 +1,101 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux || (freebsd && (amd64 || arm64 || riscv64))
|
||||
|
||||
package debug
|
||||
|
||||
import "strings"
|
||||
|
||||
import "fmt"
|
||||
|
||||
// Breakpoint is one INT3 breakpoint in the debuggee.
|
||||
// Breakpoint is one software breakpoint in the debuggee.
|
||||
type Breakpoint struct {
|
||||
Addr uint64 // absolute address in the debuggee
|
||||
Label string // source label ("" for raw addresses)
|
||||
Orig byte // original byte at Addr (restored on removal)
|
||||
Orig []byte // original bytes at Addr (restored on removal)
|
||||
Enabled bool
|
||||
Cond *Condition // optional condition (nil = unconditional)
|
||||
hits int
|
||||
}
|
||||
|
||||
// Condition is a simple register-comparison condition evaluated when a
|
||||
// breakpoint is hit. Format: <reg> <op> <value>.
|
||||
// Condition is a register-comparison condition evaluated when a breakpoint
|
||||
// is hit. Supports three forms:
|
||||
// - register vs constant: <reg> <op> <value>
|
||||
// - register vs register: <reg> <op> <reg2>
|
||||
// - register vs memory: <reg> <op> *<addr>
|
||||
type Condition struct {
|
||||
Reg string // register name (rax, rbx, rip, rsp, ...)
|
||||
Op string // comparison operator: ==, !=, <, >, <=, >=
|
||||
Value uint64
|
||||
Reg string // register name (rax, rbx, rip, rsp, ...)
|
||||
Op string // comparison operator: ==, !=, <, >, <=, >=
|
||||
Value uint64 // constant value (when Reg2 == "" and MemAddr == 0)
|
||||
Reg2 string // second register name (for register-register comparison)
|
||||
MemAddr uint64 // memory address (for register-memory comparison, prefixed with *)
|
||||
}
|
||||
|
||||
// Eval checks the condition against the current registers.
|
||||
func (c *Condition) Eval(regs *Regs) bool {
|
||||
var actual uint64
|
||||
switch c.Reg {
|
||||
case "rax", "eax", "ax", "al":
|
||||
actual = regs.RAX
|
||||
case "rbx", "ebx", "bx", "bl":
|
||||
actual = regs.RBX
|
||||
case "rcx", "ecx", "cx", "cl":
|
||||
actual = regs.RCX
|
||||
case "rdx", "edx", "dx", "dl":
|
||||
actual = regs.RDX
|
||||
case "rsi", "esi", "si":
|
||||
actual = regs.RSI
|
||||
case "rdi", "edi", "di":
|
||||
actual = regs.RDI
|
||||
case "rbp", "ebp", "bp":
|
||||
actual = regs.RBP
|
||||
case "rsp", "esp", "sp":
|
||||
actual = regs.RSP
|
||||
case "r8":
|
||||
actual = regs.R8
|
||||
case "r9":
|
||||
actual = regs.R9
|
||||
case "r10":
|
||||
actual = regs.R10
|
||||
case "r11":
|
||||
actual = regs.R11
|
||||
case "r12":
|
||||
actual = regs.R12
|
||||
case "r13":
|
||||
actual = regs.R13
|
||||
case "r14":
|
||||
actual = regs.R14
|
||||
case "r15":
|
||||
actual = regs.R15
|
||||
case "rip", "eip":
|
||||
actual = regs.RIP
|
||||
// Eval checks the condition against the current registers. For the
|
||||
// register-memory form, mem reads an 8-byte little-endian word from the
|
||||
// debuggee; it may be nil when no reader is available. Anything that cannot
|
||||
// be decided (unknown register or operator, unreadable memory) does not
|
||||
// block the breakpoint.
|
||||
func (c *Condition) Eval(regs *Regs, mem func(addr uint64) (uint64, bool)) bool {
|
||||
actual, ok := regs.RegValue(c.Reg)
|
||||
if !ok {
|
||||
return true // unknown register, don't block
|
||||
}
|
||||
var expected uint64
|
||||
switch {
|
||||
case c.Reg2 != "":
|
||||
// Register-register comparison.
|
||||
v, ok := regs.RegValue(c.Reg2)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
expected = v
|
||||
case c.MemAddr != 0:
|
||||
// Register-memory comparison, resolved in the debuggee at
|
||||
// evaluation time.
|
||||
if mem == nil {
|
||||
return true
|
||||
}
|
||||
v, ok := mem(c.MemAddr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
expected = v
|
||||
default:
|
||||
return true // unknown register — don't block
|
||||
expected = c.Value
|
||||
}
|
||||
switch c.Op {
|
||||
case "==", "=":
|
||||
return actual == c.Value
|
||||
return actual == expected
|
||||
case "!=":
|
||||
return actual != c.Value
|
||||
return actual != expected
|
||||
case "<":
|
||||
return actual < c.Value
|
||||
return actual < expected
|
||||
case ">":
|
||||
return actual > c.Value
|
||||
return actual > expected
|
||||
case "<=":
|
||||
return actual <= c.Value
|
||||
return actual <= expected
|
||||
case ">=":
|
||||
return actual >= c.Value
|
||||
return actual >= expected
|
||||
default:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// Breakpoints manages the set of breakpoints for a Session.
|
||||
// Breakpoints manages software breakpoints for a debuggee.
|
||||
// String renders the condition for display.
|
||||
func (c *Condition) String() string {
|
||||
switch {
|
||||
case c.Reg2 != "":
|
||||
return fmt.Sprintf("%s %s %s", c.Reg, c.Op, c.Reg2)
|
||||
case c.MemAddr != 0:
|
||||
return fmt.Sprintf("%s %s *%#x", c.Reg, c.Op, c.MemAddr)
|
||||
default:
|
||||
return fmt.Sprintf("%s %s %#x", c.Reg, c.Op, c.Value)
|
||||
}
|
||||
}
|
||||
|
||||
// Breakpoints manages the software breakpoints of one Session.
|
||||
type Breakpoints struct {
|
||||
t tracer
|
||||
bps map[uint64]*Breakpoint
|
||||
@@ -94,6 +106,18 @@ func NewBreakpoints(t tracer) *Breakpoints {
|
||||
return &Breakpoints{t: t, bps: make(map[uint64]*Breakpoint)}
|
||||
}
|
||||
|
||||
// breakpointMask is the byte mask of the breakpoint instruction inside a
|
||||
// peeked word: the low len(breakpointInsn) bytes, because every supported
|
||||
// architecture is little-endian and patches the instruction at the lowest
|
||||
// address of the word.
|
||||
func breakpointMask() uint64 {
|
||||
var mask uint64
|
||||
for range breakpointInsn {
|
||||
mask = (mask << 8) | 0xFF
|
||||
}
|
||||
return mask
|
||||
}
|
||||
|
||||
// Set installs a breakpoint at addr (replaces any existing one).
|
||||
func (bm *Breakpoints) Set(addr uint64, label string) (*Breakpoint, error) {
|
||||
return bm.SetWithCond(addr, label, nil)
|
||||
@@ -106,14 +130,17 @@ func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (
|
||||
bp.Cond = cond
|
||||
return bp, nil
|
||||
}
|
||||
// Read the original byte.
|
||||
// Read the original bytes.
|
||||
word, err := bm.t.Peek(addr)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
orig := byte(word)
|
||||
// Patch with INT3 (0xCC), preserving the rest of the word.
|
||||
patched := (word &^ 0xFF) | 0xCC
|
||||
orig := make([]byte, len(breakpointInsn))
|
||||
for i := range orig {
|
||||
orig[i] = byte(word >> (8 * i))
|
||||
}
|
||||
// Patch with the breakpoint instruction, preserving the rest of the word.
|
||||
patched := (word &^ breakpointMask()) | breakpointWord(breakpointInsn)
|
||||
if err := bm.t.Poke(addr, patched); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -122,17 +149,12 @@ func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (
|
||||
return bp, nil
|
||||
}
|
||||
|
||||
// Hits returns the number of times the breakpoint has been hit.
|
||||
func (bp *Breakpoint) Hits() int {
|
||||
return bp.hits
|
||||
}
|
||||
|
||||
// Info returns a formatted list of all breakpoints.
|
||||
func (bm *Breakpoints) Info() string {
|
||||
if len(bm.bps) == 0 {
|
||||
return "no breakpoints set\n"
|
||||
}
|
||||
result := ""
|
||||
var result strings.Builder
|
||||
i := 0
|
||||
for _, bp := range bm.bps {
|
||||
i++
|
||||
@@ -146,26 +168,40 @@ func (bm *Breakpoints) Info() string {
|
||||
}
|
||||
cond := ""
|
||||
if bp.Cond != nil {
|
||||
cond = fmt.Sprintf(" if %s %s %#x", bp.Cond.Reg, bp.Cond.Op, bp.Cond.Value)
|
||||
cond = " if " + bp.Cond.String()
|
||||
}
|
||||
result += fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond)
|
||||
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond))
|
||||
}
|
||||
return result
|
||||
return result.String()
|
||||
}
|
||||
|
||||
// Clear removes the breakpoint at addr, restoring the original byte.
|
||||
// restore writes the saved original bytes back over the breakpoint
|
||||
// instruction, preserving the rest of the peeked word. It reports whether
|
||||
// both the peek and the poke succeeded.
|
||||
func (bm *Breakpoints) restore(addr uint64, bp *Breakpoint) bool {
|
||||
word, err := bm.t.Peek(addr)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
orig := uint64(0)
|
||||
for i, b := range bp.Orig {
|
||||
orig |= uint64(b) << (8 * i)
|
||||
}
|
||||
return bm.t.Poke(addr, (word&^breakpointMask())|orig) == nil
|
||||
}
|
||||
|
||||
// Clear removes the breakpoint at addr, restoring the original bytes.
|
||||
func (bm *Breakpoints) Clear(addr uint64) error {
|
||||
bp, ok := bm.bps[addr]
|
||||
if !ok {
|
||||
return fmt.Errorf("debug: no breakpoint at %#x", addr)
|
||||
}
|
||||
word, err := bm.t.Peek(addr)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
||||
if err := bm.t.Poke(addr, restored); err != nil {
|
||||
return err
|
||||
if !bm.restore(addr, bp) {
|
||||
word, err := bm.t.Peek(addr)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return fmt.Errorf("debug: restore breakpoint at %#x failed, word is %#x", addr, word)
|
||||
}
|
||||
delete(bm.bps, addr)
|
||||
return nil
|
||||
@@ -196,41 +232,55 @@ func (bm *Breakpoints) All() []*Breakpoint {
|
||||
}
|
||||
|
||||
// HandleTrap is called after the debuggee stops on SIGTRAP. It checks
|
||||
// whether the trap was caused by one of our breakpoints (RIP-1 matches
|
||||
// a breakpoint address), restores the original byte, rewinds RIP, and
|
||||
// whether the trap was caused by one of our breakpoints (PC-adjust matches
|
||||
// a breakpoint address), restores the original bytes, rewinds PC, and
|
||||
// returns the breakpoint that was hit (or nil if it was a single-step).
|
||||
// Hits returns how many times the breakpoint has been hit.
|
||||
func (bp *Breakpoint) Hits() int { return bp.hits }
|
||||
|
||||
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
|
||||
// After INT3, RIP points to the byte AFTER the 0xCC.
|
||||
trapAddr := regs.RIP - 1
|
||||
// On amd64 the kernel reports the trap with RIP past the INT3; on the
|
||||
// other supported architectures the PC still stands on the trap
|
||||
// instruction, which breakpointPCAdjust encodes per architecture.
|
||||
trapAddr := regs.GetPC() - uint64(breakpointPCAdjust)
|
||||
bp, ok := bm.bps[trapAddr]
|
||||
if !ok || !bp.Enabled {
|
||||
return nil // single-step trap or unknown
|
||||
}
|
||||
// Check the condition (if any).
|
||||
if bp.Cond != nil && !bp.Cond.Eval(regs) {
|
||||
// Condition not met — restore the byte but do NOT rewind RIP.
|
||||
// The process continues from the next instruction (past the INT3).
|
||||
word, err := bm.t.Peek(trapAddr)
|
||||
if err == nil {
|
||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
||||
bm.t.Poke(trapAddr, restored)
|
||||
if bp.Cond != nil && !bp.Cond.Eval(regs, bm.peekValue) {
|
||||
// Condition not met: step the original instruction and re-arm the
|
||||
// breakpoint, leaving the debuggee stopped just past it, ready to
|
||||
// resume silently. The PC must be rewound first: on architectures
|
||||
// that report the trap past the instruction (amd64) it would
|
||||
// otherwise sit on the second byte of the replaced instruction.
|
||||
if !bm.restore(trapAddr, bp) {
|
||||
return nil
|
||||
}
|
||||
// RIP is already past the INT3 (trapAddr + 1). Don't rewind.
|
||||
regs.SetPC(trapAddr)
|
||||
if err := bm.t.SetRegs(regs); err != nil {
|
||||
return nil
|
||||
}
|
||||
if err := bm.t.Step(); err != nil {
|
||||
return nil
|
||||
}
|
||||
bm.Reinsert(trapAddr)
|
||||
return nil
|
||||
}
|
||||
bp.hits++
|
||||
// Restore the original byte.
|
||||
word, err := bm.t.Peek(trapAddr)
|
||||
if err == nil {
|
||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
||||
bm.t.Poke(trapAddr, restored)
|
||||
}
|
||||
// Rewind RIP to re-execute the original instruction.
|
||||
regs.RIP = trapAddr
|
||||
// Restore the original bytes and rewind PC to re-execute them.
|
||||
bm.restore(trapAddr, bp)
|
||||
regs.SetPC(trapAddr)
|
||||
bm.t.SetRegs(regs)
|
||||
return bp
|
||||
}
|
||||
|
||||
// peekValue adapts tracer.Peek to the Condition value reader.
|
||||
func (bm *Breakpoints) peekValue(addr uint64) (uint64, bool) {
|
||||
v, err := bm.t.Peek(addr)
|
||||
return v, err == nil
|
||||
}
|
||||
|
||||
// Reinsert re-inserts the breakpoint at addr after a single-step past it.
|
||||
// Called after Step() when we want the breakpoint to fire again on the
|
||||
// next Continue().
|
||||
@@ -243,6 +293,15 @@ func (bm *Breakpoints) Reinsert(addr uint64) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
patched := (word &^ 0xFF) | 0xCC
|
||||
patched := (word &^ breakpointMask()) | breakpointWord(breakpointInsn)
|
||||
return bm.t.Poke(addr, patched)
|
||||
}
|
||||
|
||||
// breakpointWord converts the breakpoint instruction bytes to a uint64.
|
||||
func breakpointWord(insn []byte) uint64 {
|
||||
var w uint64
|
||||
for i, b := range insn {
|
||||
w |= uint64(b) << (i * 8)
|
||||
}
|
||||
return w
|
||||
}
|
||||
|
||||
@@ -0,0 +1,265 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux
|
||||
|
||||
package debug
|
||||
|
||||
// Architecture-neutral tests: label and line tables, and the breakpoint
|
||||
// manager against the mock tracer. These do not launch a debuggee, so they
|
||||
// build on every supported linux architecture.
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestLineAt(t *testing.T) {
|
||||
lines := []SourceLine{
|
||||
{Offset: 0, Line: 5},
|
||||
{Offset: 5, Line: 6},
|
||||
{Offset: 10, Line: 7},
|
||||
{Offset: 15, Line: 8},
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
offset int
|
||||
want int
|
||||
}{
|
||||
{0, 5},
|
||||
{1, 5},
|
||||
{4, 5},
|
||||
{5, 6},
|
||||
{7, 6},
|
||||
{10, 7},
|
||||
{12, 7},
|
||||
{15, 8},
|
||||
{20, 8},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := lineAt(lines, tt.offset)
|
||||
if got != tt.want {
|
||||
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
|
||||
}
|
||||
}
|
||||
|
||||
// Empty table.
|
||||
if lineAt(nil, 5) != 0 {
|
||||
t.Error("lineAt(nil, 5) should return 0")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOffsetForLine(t *testing.T) {
|
||||
lines := []SourceLine{
|
||||
{Offset: 0, Line: 5},
|
||||
{Offset: 5, Line: 6},
|
||||
{Offset: 10, Line: 7},
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
line int
|
||||
want int
|
||||
}{
|
||||
{5, 0},
|
||||
{6, 5},
|
||||
{7, 10},
|
||||
{99, -1}, // not found
|
||||
{0, -1}, // not found
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := offsetForLine(lines, tt.line)
|
||||
if got != tt.want {
|
||||
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestNearestLabel(t *testing.T) {
|
||||
labels := []Label{
|
||||
{Name: "start", Offset: 0},
|
||||
{Name: "loop", Offset: 10},
|
||||
{Name: "done", Offset: 20},
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
offset int
|
||||
want string
|
||||
}{
|
||||
{0, "start"},
|
||||
{5, "start"},
|
||||
{10, "loop"},
|
||||
{15, "loop"},
|
||||
{20, "done"},
|
||||
{25, "done"},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := nearestLabel(labels, tt.offset)
|
||||
if got != tt.want {
|
||||
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointsSetAndClear(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
|
||||
// Set a breakpoint at address 0x1000.
|
||||
bp, err := bm.Set(0x1000, "test")
|
||||
if err != nil {
|
||||
t.Fatalf("Set: %v", err)
|
||||
}
|
||||
if !bp.Enabled {
|
||||
t.Error("breakpoint not enabled")
|
||||
}
|
||||
if bp.Label != "test" {
|
||||
t.Errorf("label = %q, want test", bp.Label)
|
||||
}
|
||||
|
||||
// Verify Peek was called.
|
||||
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
|
||||
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
|
||||
}
|
||||
|
||||
// Verify Poke wrote the breakpoint instruction's bytes.
|
||||
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
|
||||
t.Errorf("pokes = %v", tr.pokes)
|
||||
}
|
||||
if got := tr.pokes[0].val & breakpointMask(); got != breakpointWord(breakpointInsn) {
|
||||
t.Errorf("patched bytes %#x, want %#x", got, breakpointWord(breakpointInsn))
|
||||
}
|
||||
|
||||
// At should find it.
|
||||
if bm.At(0x1000) == nil {
|
||||
t.Error("At(0x1000) returned nil")
|
||||
}
|
||||
|
||||
// All should return it.
|
||||
all := bm.All()
|
||||
if len(all) != 1 {
|
||||
t.Errorf("All() = %d breakpoints, want 1", len(all))
|
||||
}
|
||||
|
||||
// Clear it.
|
||||
if err := bm.Clear(0x1000); err != nil {
|
||||
t.Fatalf("Clear: %v", err)
|
||||
}
|
||||
if bm.At(0x1000) != nil {
|
||||
t.Error("At(0x1000) after Clear should be nil")
|
||||
}
|
||||
}
|
||||
|
||||
// TestBreakpointRestoreWidth proves the restore path writes back every
|
||||
// byte of the breakpoint instruction's width, not just the first byte: on
|
||||
// arm64, riscv64 and loong64 the instruction is four bytes, and restoring
|
||||
// one byte would leave three bytes of the trap instruction in place.
|
||||
func TestBreakpointRestoreWidth(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
tr.mem[0x3000] = 0x11
|
||||
tr.mem[0x3001] = 0x22
|
||||
tr.mem[0x3002] = 0x33
|
||||
tr.mem[0x3003] = 0x44
|
||||
|
||||
if _, err := bm.Set(0x3000, "width"); err != nil {
|
||||
t.Fatalf("Set: %v", err)
|
||||
}
|
||||
for i, b := range breakpointInsn {
|
||||
if tr.mem[0x3000+uint64(i)] != b {
|
||||
t.Fatalf("byte %d after Set = %#x, want the breakpoint byte %#x", i, tr.mem[0x3000+uint64(i)], b)
|
||||
}
|
||||
}
|
||||
if len(bm.At(0x3000).Orig) != len(breakpointInsn) {
|
||||
t.Fatalf("Orig holds %d bytes, want %d", len(bm.At(0x3000).Orig), len(breakpointInsn))
|
||||
}
|
||||
|
||||
if err := bm.Clear(0x3000); err != nil {
|
||||
t.Fatalf("Clear: %v", err)
|
||||
}
|
||||
want := []byte{0x11, 0x22, 0x33, 0x44}
|
||||
for i, b := range want {
|
||||
if tr.mem[0x3000+uint64(i)] != b {
|
||||
t.Errorf("byte %d after Clear = %#x, want %#x (restore must cover the full instruction width)", i, tr.mem[0x3000+uint64(i)], b)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointsSetWithCond(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
|
||||
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
|
||||
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
|
||||
if err != nil {
|
||||
t.Fatalf("SetWithCond: %v", err)
|
||||
}
|
||||
if bp.Cond == nil || bp.Cond.Value != 42 {
|
||||
t.Error("condition not set")
|
||||
}
|
||||
|
||||
// Re-setting the same address should update the condition.
|
||||
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
|
||||
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
|
||||
if err != nil {
|
||||
t.Fatalf("SetWithCond (update): %v", err)
|
||||
}
|
||||
if bp2.Cond.Value != 100 {
|
||||
t.Error("condition not updated")
|
||||
}
|
||||
// Should have only 1 Peek (first Set), second is update (no Peek needed).
|
||||
if len(tr.peeks) != 1 {
|
||||
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointsClearAll(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
|
||||
bm.Set(0x1000, "a")
|
||||
bm.Set(0x2000, "b")
|
||||
bm.Set(0x3000, "c")
|
||||
|
||||
if len(bm.All()) != 3 {
|
||||
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
|
||||
}
|
||||
|
||||
bm.ClearAll()
|
||||
if len(bm.All()) != 0 {
|
||||
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointInfo(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
bm.Set(0x4000, "info_test")
|
||||
|
||||
info := bm.Info()
|
||||
if info == "" {
|
||||
t.Error("Info returned empty string")
|
||||
}
|
||||
if !strings.Contains(info, "info_test") {
|
||||
t.Errorf("Info %q does not contain label", info)
|
||||
}
|
||||
}
|
||||
|
||||
// TestConditionString covers the display of all three condition forms.
|
||||
func TestConditionString(t *testing.T) {
|
||||
tests := []struct {
|
||||
cond Condition
|
||||
want string
|
||||
}{
|
||||
{Condition{Reg: "rax", Op: "==", Value: 42}, "rax == 0x2a"},
|
||||
{Condition{Reg: "rax", Op: "!=", Reg2: "rbx"}, "rax != rbx"},
|
||||
{Condition{Reg: "rax", Op: "<", MemAddr: 0x5000}, "rax < *0x5000"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
if got := tt.cond.String(); got != tt.want {
|
||||
t.Errorf("Condition.String() = %q, want %q", got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
+43
-191
@@ -1,10 +1,11 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -46,7 +47,7 @@ func TestConditionEval(t *testing.T) {
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := tt.cond.Eval(regs)
|
||||
got := tt.cond.Eval(regs, nil)
|
||||
if got != tt.want {
|
||||
t.Errorf("Condition{%q %q %d}.Eval() = %v, want %v",
|
||||
tt.cond.Reg, tt.cond.Op, tt.cond.Value, got, tt.want)
|
||||
@@ -54,65 +55,33 @@ func TestConditionEval(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestLineAt(t *testing.T) {
|
||||
lines := []SourceLine{
|
||||
{Offset: 0, Line: 5},
|
||||
{Offset: 5, Line: 6},
|
||||
{Offset: 10, Line: 7},
|
||||
{Offset: 15, Line: 8},
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
offset int
|
||||
want int
|
||||
}{
|
||||
{0, 5},
|
||||
{1, 5},
|
||||
{4, 5},
|
||||
{5, 6},
|
||||
{7, 6},
|
||||
{10, 7},
|
||||
{12, 7},
|
||||
{15, 8},
|
||||
{20, 8},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := lineAt(lines, tt.offset)
|
||||
if got != tt.want {
|
||||
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
|
||||
// TestConditionEvalMem covers the register-memory form: the value is read
|
||||
// through the supplied reader, and a missing or failing reader must not
|
||||
// block the breakpoint.
|
||||
func TestConditionEvalMem(t *testing.T) {
|
||||
regs := &Regs{RAX: 7}
|
||||
mem := func(addr uint64) (uint64, bool) {
|
||||
if addr == 0x5000 {
|
||||
return 7, true
|
||||
}
|
||||
return 0, false
|
||||
}
|
||||
|
||||
// Empty table.
|
||||
if lineAt(nil, 5) != 0 {
|
||||
t.Error("lineAt(nil, 5) should return 0")
|
||||
eq := Condition{Reg: "rax", Op: "==", MemAddr: 0x5000}
|
||||
if !eq.Eval(regs, mem) {
|
||||
t.Error("register-memory comparison with matching word should hold")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOffsetForLine(t *testing.T) {
|
||||
lines := []SourceLine{
|
||||
{Offset: 0, Line: 5},
|
||||
{Offset: 5, Line: 6},
|
||||
{Offset: 10, Line: 7},
|
||||
ne := Condition{Reg: "rax", Op: "!=", MemAddr: 0x5000}
|
||||
if ne.Eval(regs, mem) {
|
||||
t.Error("register-memory comparison with mismatching word should not hold")
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
line int
|
||||
want int
|
||||
}{
|
||||
{5, 0},
|
||||
{6, 5},
|
||||
{7, 10},
|
||||
{99, -1}, // not found
|
||||
{0, -1}, // not found
|
||||
bad := Condition{Reg: "rax", Op: "==", MemAddr: 0x6000}
|
||||
if !bad.Eval(regs, mem) {
|
||||
t.Error("unreadable memory must not block the breakpoint")
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := offsetForLine(lines, tt.line)
|
||||
if got != tt.want {
|
||||
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
|
||||
}
|
||||
noReader := Condition{Reg: "rax", Op: "==", MemAddr: 0x5000}
|
||||
if !noReader.Eval(regs, nil) {
|
||||
t.Error("missing memory reader must not block the breakpoint")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -139,144 +108,11 @@ func TestDecodeRflags(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestNearestLabel(t *testing.T) {
|
||||
labels := []Label{
|
||||
{Name: "start", Offset: 0},
|
||||
{Name: "loop", Offset: 10},
|
||||
{Name: "done", Offset: 20},
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
offset int
|
||||
want string
|
||||
}{
|
||||
{0, "start"},
|
||||
{5, "start"},
|
||||
{10, "loop"},
|
||||
{15, "loop"},
|
||||
{20, "done"},
|
||||
{25, "done"},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := nearestLabel(labels, tt.offset)
|
||||
if got != tt.want {
|
||||
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointsSetAndClear(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
|
||||
// Set a breakpoint at address 0x1000.
|
||||
bp, err := bm.Set(0x1000, "test")
|
||||
if err != nil {
|
||||
t.Fatalf("Set: %v", err)
|
||||
}
|
||||
if !bp.Enabled {
|
||||
t.Error("breakpoint not enabled")
|
||||
}
|
||||
if bp.Label != "test" {
|
||||
t.Errorf("label = %q, want test", bp.Label)
|
||||
}
|
||||
|
||||
// Verify Peek was called.
|
||||
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
|
||||
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
|
||||
}
|
||||
|
||||
// Verify Poke wrote INT3.
|
||||
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
|
||||
t.Errorf("pokes = %v", tr.pokes)
|
||||
}
|
||||
|
||||
// At should find it.
|
||||
if bm.At(0x1000) == nil {
|
||||
t.Error("At(0x1000) returned nil")
|
||||
}
|
||||
|
||||
// All should return it.
|
||||
all := bm.All()
|
||||
if len(all) != 1 {
|
||||
t.Errorf("All() = %d breakpoints, want 1", len(all))
|
||||
}
|
||||
|
||||
// Clear it.
|
||||
if err := bm.Clear(0x1000); err != nil {
|
||||
t.Fatalf("Clear: %v", err)
|
||||
}
|
||||
if bm.At(0x1000) != nil {
|
||||
t.Error("At(0x1000) after Clear should be nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointsSetWithCond(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
|
||||
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
|
||||
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
|
||||
if err != nil {
|
||||
t.Fatalf("SetWithCond: %v", err)
|
||||
}
|
||||
if bp.Cond == nil || bp.Cond.Value != 42 {
|
||||
t.Error("condition not set")
|
||||
}
|
||||
|
||||
// Re-setting the same address should update the condition.
|
||||
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
|
||||
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
|
||||
if err != nil {
|
||||
t.Fatalf("SetWithCond (update): %v", err)
|
||||
}
|
||||
if bp2.Cond.Value != 100 {
|
||||
t.Error("condition not updated")
|
||||
}
|
||||
// Should have only 1 Peek (first Set), second is update (no Peek needed).
|
||||
if len(tr.peeks) != 1 {
|
||||
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointsClearAll(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
|
||||
bm.Set(0x1000, "a")
|
||||
bm.Set(0x2000, "b")
|
||||
bm.Set(0x3000, "c")
|
||||
|
||||
if len(bm.All()) != 3 {
|
||||
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
|
||||
}
|
||||
|
||||
bm.ClearAll()
|
||||
if len(bm.All()) != 0 {
|
||||
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointInfo(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
bm.Set(0x4000, "info_test")
|
||||
|
||||
info := bm.Info()
|
||||
if info == "" {
|
||||
t.Error("Info returned empty string")
|
||||
}
|
||||
if !strings.Contains(info, "info_test") {
|
||||
t.Errorf("Info %q does not contain label", info)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWatchpointSlotTracking(t *testing.T) {
|
||||
s := &Session{}
|
||||
s := &Session{} // per-session slots start free
|
||||
|
||||
// All four slots are free initially.
|
||||
for i := 0; i < 4; i++ {
|
||||
for i := range 4 {
|
||||
if s.IsWatchpointSlotUsed(i) {
|
||||
t.Errorf("slot %d should be free initially", i)
|
||||
}
|
||||
@@ -314,10 +150,26 @@ func TestWatchpointSlotTracking(t *testing.T) {
|
||||
}
|
||||
|
||||
// Mark all slots used: FindFreeWatchpointSlot returns -1.
|
||||
for i := 0; i < 4; i++ {
|
||||
for i := range 4 {
|
||||
s.wpSlots[i] = true
|
||||
}
|
||||
if got := s.FindFreeWatchpointSlot(); got != -1 {
|
||||
t.Errorf("FindFreeWatchpointSlot() with all slots used = %d, want -1", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUnwatchSlotBound checks the bound the REPL parses against: it must
|
||||
// cover the architecture's whole slot range, not a hardcoded 0-3.
|
||||
func TestUnwatchSlotBound(t *testing.T) {
|
||||
max := maxWatchpoints()
|
||||
if max < 4 {
|
||||
t.Fatalf("maxWatchpoints() = %d, want at least 4", max)
|
||||
}
|
||||
s := &Session{}
|
||||
if s.IsWatchpointSlotUsed(max - 1) {
|
||||
t.Errorf("slot %d should be free initially", max-1)
|
||||
}
|
||||
if s.IsWatchpointSlotUsed(max) {
|
||||
t.Errorf("slot %d must be out of range", max)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build freebsd && amd64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||
)
|
||||
|
||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||
// memory and returns its text representation and length in bytes.
|
||||
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||
mem, err := s.ReadMemory(addr, 15)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
ins, err := disasm.Decode(arch.AMD64, mem, addr)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
return ins.Text, ins.Len, nil
|
||||
}
|
||||
|
||||
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||
// them as a formatted string with addresses and byte offsets.
|
||||
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||
var result strings.Builder
|
||||
pc := addr
|
||||
for range n {
|
||||
text, length, err := s.Disassemble(pc)
|
||||
if err != nil {
|
||||
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
|
||||
break
|
||||
}
|
||||
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
|
||||
if length == 0 {
|
||||
length = 1
|
||||
}
|
||||
pc += uint64(length)
|
||||
}
|
||||
return result.String()
|
||||
}
|
||||
|
||||
// isCallInsn reports whether disassembled text (x86asm.IntelSyntax) is a
|
||||
// call. The first token must match exactly: a prefix test would also catch
|
||||
// unrelated mnemonics.
|
||||
func isCallInsn(text string) bool {
|
||||
m, _, _ := strings.Cut(text, " ")
|
||||
return strings.ToLower(m) == "call"
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build freebsd && arm64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||
)
|
||||
|
||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||
// memory and returns its text representation and length in bytes.
|
||||
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||
mem, err := s.ReadMemory(addr, 4)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
ins, err := disasm.Decode(arch.ARM64, mem, addr)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
return ins.Text, ins.Len, nil
|
||||
}
|
||||
|
||||
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||
// them as a formatted string with addresses and byte offsets.
|
||||
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||
var result strings.Builder
|
||||
pc := addr
|
||||
for range n {
|
||||
text, length, err := s.Disassemble(pc)
|
||||
if err != nil {
|
||||
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
|
||||
break
|
||||
}
|
||||
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
|
||||
if length == 0 {
|
||||
length = 1
|
||||
}
|
||||
pc += uint64(length)
|
||||
}
|
||||
return result.String()
|
||||
}
|
||||
|
||||
// isCallInsn reports whether disassembled text (arm64asm.GoSyntax) is a
|
||||
// call. GoSyntax renders bl as CALL; the native mnemonic is accepted too.
|
||||
// The first token must match exactly so branches never match.
|
||||
func isCallInsn(text string) bool {
|
||||
m, _, _ := strings.Cut(text, " ")
|
||||
switch strings.ToLower(m) {
|
||||
case "call", "bl":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build freebsd && riscv64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||
)
|
||||
|
||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||
// memory and returns its text representation and length in bytes.
|
||||
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||
mem, err := s.ReadMemory(addr, 4)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
ins, err := disasm.Decode(arch.RISCV, mem, addr)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
return ins.Text, ins.Len, nil
|
||||
}
|
||||
|
||||
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||
// them as a formatted string with addresses and byte offsets.
|
||||
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||
var result strings.Builder
|
||||
pc := addr
|
||||
for range n {
|
||||
text, length, err := s.Disassemble(pc)
|
||||
if err != nil {
|
||||
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
|
||||
break
|
||||
}
|
||||
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
|
||||
if length == 0 {
|
||||
length = 1
|
||||
}
|
||||
pc += uint64(length)
|
||||
}
|
||||
return result.String()
|
||||
}
|
||||
|
||||
// isCallInsn reports whether disassembled text (riscv64asm.GoSyntax) is a
|
||||
// call. GoSyntax renders jal and jalr calls as CALL; the native mnemonics
|
||||
// are accepted too. The first token must match exactly: a prefix test on
|
||||
// "bl" would catch branches on other architectures, and jalr as ret prints
|
||||
// RET, which must not be stepped over.
|
||||
func isCallInsn(text string) bool {
|
||||
m, _, _ := strings.Cut(text, " ")
|
||||
switch strings.ToLower(m) {
|
||||
case "call", "jal", "jalr":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
+20
-16
@@ -7,46 +7,50 @@ package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||
)
|
||||
|
||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||
// memory and returns its text representation and length in bytes.
|
||||
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||
// Read up to 15 bytes (max x86 instruction length).
|
||||
mem, err := s.ReadMemory(addr, 15)
|
||||
if err != nil {
|
||||
// Try a shorter read if we're near a page boundary.
|
||||
mem, err = s.ReadMemory(addr, 1)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
return "", 0, err
|
||||
}
|
||||
inst, err := x86asm.Decode(mem, 64)
|
||||
ins, err := disasm.Decode(arch.AMD64, mem, addr)
|
||||
if err != nil {
|
||||
return "???", 1, nil
|
||||
return "", 0, err
|
||||
}
|
||||
text := x86asm.IntelSyntax(inst, addr, nil)
|
||||
return text, inst.Len, nil
|
||||
return ins.Text, ins.Len, nil
|
||||
}
|
||||
|
||||
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||
// them as a formatted string with addresses and byte offsets.
|
||||
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||
var result string
|
||||
var result strings.Builder
|
||||
pc := addr
|
||||
for i := 0; i < n; i++ {
|
||||
for range n {
|
||||
text, length, err := s.Disassemble(pc)
|
||||
if err != nil {
|
||||
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
|
||||
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
|
||||
break
|
||||
}
|
||||
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
|
||||
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
|
||||
if length == 0 {
|
||||
length = 1
|
||||
}
|
||||
pc += uint64(length)
|
||||
}
|
||||
return result
|
||||
return result.String()
|
||||
}
|
||||
|
||||
// isCallInsn reports whether disassembled text (x86asm.IntelSyntax) is a
|
||||
// call. The first token must match exactly: a prefix test would also catch
|
||||
// unrelated mnemonics.
|
||||
func isCallInsn(text string) bool {
|
||||
m, _, _ := strings.Cut(text, " ")
|
||||
return strings.ToLower(m) == "call"
|
||||
}
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && arm64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||
)
|
||||
|
||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||
// memory and returns its text representation and length in bytes.
|
||||
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||
mem, err := s.ReadMemory(addr, 4)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
ins, err := disasm.Decode(arch.ARM64, mem, addr)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
return ins.Text, ins.Len, nil
|
||||
}
|
||||
|
||||
// DisassembleN decodes up to n instructions starting at addr.
|
||||
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||
var result string
|
||||
pc := addr
|
||||
for range n {
|
||||
text, length, err := s.Disassemble(pc)
|
||||
if err != nil {
|
||||
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
|
||||
break
|
||||
}
|
||||
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
|
||||
if length == 0 {
|
||||
length = 4
|
||||
}
|
||||
pc += uint64(length)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// isCallInsn reports whether disassembled text (arm64asm.GoSyntax) is a
|
||||
// call. GoSyntax renders bl as CALL; the native mnemonic is accepted too.
|
||||
// The first token must match exactly so branches never match.
|
||||
func isCallInsn(text string) bool {
|
||||
m, _, _ := strings.Cut(text, " ")
|
||||
switch strings.ToLower(m) {
|
||||
case "call", "bl":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && loong64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||
)
|
||||
|
||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||
// memory and returns its text representation and length in bytes.
|
||||
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||
mem, err := s.ReadMemory(addr, 4)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
ins, err := disasm.Decode(arch.LOONG64, mem, addr)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
return ins.Text, ins.Len, nil
|
||||
}
|
||||
|
||||
// DisassembleN decodes up to n instructions starting at addr.
|
||||
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||
var result string
|
||||
pc := addr
|
||||
for range n {
|
||||
text, length, err := s.Disassemble(pc)
|
||||
if err != nil {
|
||||
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
|
||||
break
|
||||
}
|
||||
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
|
||||
if length == 0 {
|
||||
length = 4
|
||||
}
|
||||
pc += uint64(length)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// isCallInsn reports whether disassembled text (loong64asm.GoSyntax) is a
|
||||
// call. GoSyntax renders bl and jirl calls as CALL (jirl returns print
|
||||
// RET); the native mnemonics are accepted too. The first token must match
|
||||
// exactly: a "bl" prefix would catch bltz and other branches.
|
||||
func isCallInsn(text string) bool {
|
||||
m, _, _ := strings.Cut(text, " ")
|
||||
switch strings.ToLower(m) {
|
||||
case "call", "bl", "jirl":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user