Compare commits
62
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1529eba9ce | ||
|
|
9cbd31e00b | ||
|
|
924e0013eb | ||
|
|
56376a0e59 | ||
|
|
70eb8fb8bd | ||
|
|
39d2e80145 | ||
|
|
9f4f949c1f | ||
|
|
f0d5238c47 | ||
|
|
2931bbd6b2 | ||
|
|
63562a503a | ||
|
|
e836d6150d | ||
|
|
8a51b060da | ||
|
|
d3d47db727 | ||
|
|
0758556b7d | ||
|
|
ddb8440340 | ||
|
|
f15ff66fb1 | ||
|
|
187e4856d3 | ||
|
|
d315a998ce | ||
|
|
a6f3828c02 | ||
|
|
e3b35bb817 | ||
|
|
eb0a89e58d | ||
|
|
dd32d9e66e | ||
|
|
3a73acb20a | ||
|
|
7604a9443f | ||
|
|
b3908fc43d | ||
|
|
eb8b0cd316 | ||
|
|
a8bfd54ed2 | ||
|
|
375182ef1f | ||
|
|
87b1081c53 | ||
|
|
f3c8510a58 | ||
|
|
ebdf14939f | ||
|
|
79a2c16bac | ||
|
|
401386956c | ||
|
|
4258131a3a | ||
|
|
94e09e8070 | ||
|
|
7aefe6a42d | ||
|
|
ac1c05c793 | ||
|
|
93c47a312a | ||
|
|
708d0a0a5e | ||
|
|
3c8f7cb411 | ||
|
|
7c5b7a1419 | ||
|
|
bc3f448738 | ||
|
|
f37f183577 | ||
|
|
1e77e58250 | ||
|
|
1d0969ed64 | ||
|
|
23c001be51 | ||
|
|
96e81cc98d | ||
|
|
c834d98210 | ||
|
|
03d6d4da54 | ||
|
|
0b42ce7952 | ||
|
|
288a64ccd2 | ||
|
|
5fddfa704b | ||
|
|
a2bb5eeb4e | ||
|
|
48449b7a7f | ||
|
|
3de043c494 | ||
|
|
0078f7be5c | ||
|
|
6a7317d141 | ||
|
|
d08523caa5 | ||
|
|
20e4b8d9c4 | ||
|
|
61f4247cef | ||
|
|
049872ddff | ||
|
|
3669f64ff6 |
@@ -0,0 +1,37 @@
|
|||||||
|
# Race, Go. Dispatched by hand, and never a gate on a push or a tag: the release tag is
|
||||||
|
# cut only after `just gates` has already raced the tree, so this workflow is the
|
||||||
|
# explicit second opinion, not a step of the release.
|
||||||
|
#
|
||||||
|
# The race detector roughly doubles both time and memory, which the shared runner box
|
||||||
|
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once per
|
||||||
|
# task; here it is a decision rather than a routine.
|
||||||
|
#
|
||||||
|
# Every step is one command, so the step that fails is the gate that failed.
|
||||||
|
name: Race
|
||||||
|
|
||||||
|
on:
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
env:
|
||||||
|
# One core: parallelism buys no speed here and costs memory the box does not have.
|
||||||
|
GOFLAGS: -p=1
|
||||||
|
GOMAXPROCS: "2"
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
race:
|
||||||
|
runs-on: fedora
|
||||||
|
timeout-minutes: 20
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
|
- uses: actions/setup-go@v6
|
||||||
|
with:
|
||||||
|
go-version-file: go.mod
|
||||||
|
cache: true
|
||||||
|
|
||||||
|
- name: Install gcc
|
||||||
|
# The race detector needs cgo and the runner image carries no C compiler.
|
||||||
|
run: dnf install -y gcc
|
||||||
|
|
||||||
|
- name: Race
|
||||||
|
run: go test -race -count=1 -timeout 10m ./...
|
||||||
+316
-86
@@ -1,73 +1,222 @@
|
|||||||
# Release — gasm binaries. Runs on version tags (v0.28.0) pushed to main.
|
# Release, Go binaries. Runs on version tags (v1.2.3) pushed to main.
|
||||||
|
#
|
||||||
|
# The module sits at the repository root: the toolchain records a version only for a root
|
||||||
|
# module, measured on go1.27.1, so a build of a module in a subdirectory reports (devel)
|
||||||
|
# even at its own <module>/vX.Y.Z tag and this workflow's smoke test can never pass for
|
||||||
|
# it. A Go repository is one module at the root.
|
||||||
|
#
|
||||||
|
# The version contract these steps implement: nothing is injected. The toolchain records
|
||||||
|
# the tag into the binary's build information, so the build simply has to happen at the
|
||||||
|
# tag, which the trigger guarantees.
|
||||||
|
#
|
||||||
|
# The gates run in their own job, once, before the matrix, minus the race detector: race
|
||||||
|
# never runs on a push path or a tag, and the local gate raced this tree before the tag
|
||||||
|
# was cut. Putting the gates inside the matrix would run the whole suite once per target
|
||||||
|
# on the box that also hosts the forge. Each job validates the tag for itself rather than
|
||||||
|
# passing a value between jobs, so no workflow feature has to be trusted for the version
|
||||||
|
# to reach the file name.
|
||||||
name: Release
|
name: Release
|
||||||
|
|
||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
tags: ["v*"]
|
tags: ["v*"]
|
||||||
|
|
||||||
|
env:
|
||||||
|
# The box is shared with the forge, so parallelism is bounded on purpose. The gates job
|
||||||
|
# needs it most; the build jobs inherit it for their parallel compilation.
|
||||||
|
GOFLAGS: -p=1
|
||||||
|
GOMAXPROCS: "2"
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
|
gates:
|
||||||
|
runs-on: fedora
|
||||||
|
timeout-minutes: 10
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
|
- uses: actions/setup-go@v6
|
||||||
|
with:
|
||||||
|
go-version-file: go.mod
|
||||||
|
cache: true
|
||||||
|
|
||||||
|
- name: Install Perl
|
||||||
|
# Perl for the steps below. The install is a no-op where the package
|
||||||
|
# is already present.
|
||||||
|
run: dnf install -y perl
|
||||||
|
|
||||||
|
- name: Validate the tag
|
||||||
|
env:
|
||||||
|
VERSION: ${{ gitea.ref_name }}
|
||||||
|
run: |
|
||||||
|
perl -e '
|
||||||
|
my $v = $ENV{VERSION} // q{};
|
||||||
|
$v =~ m{^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$}
|
||||||
|
or die qq{ERROR: expected a semver tag like v1.2.3, got: $v\n};
|
||||||
|
print qq{tag $v\n};
|
||||||
|
'
|
||||||
|
|
||||||
|
- name: Security policy names this release
|
||||||
|
# The supported-versions table is the one part of SECURITY.md that
|
||||||
|
# carries a version, so it goes stale the moment a tag is cut. Fail
|
||||||
|
# here rather than publish a policy naming the previous release.
|
||||||
|
env:
|
||||||
|
VERSION: ${{ gitea.ref_name }}
|
||||||
|
run: |
|
||||||
|
perl -e '
|
||||||
|
my $v = $ENV{VERSION} // q{};
|
||||||
|
(my $nv = $v) =~ s/^v//;
|
||||||
|
open(my $f, q{<}, q{SECURITY.md}) or die qq{SECURITY.md: $!\n};
|
||||||
|
local $/;
|
||||||
|
my $t = <$f>;
|
||||||
|
close $f;
|
||||||
|
$t =~ m{^\|\s*\Q$nv\E\s*\|\s*yes\s*\|}m
|
||||||
|
or die qq{ERROR: SECURITY.md does not name $nv as supported; update the table before releasing.\n};
|
||||||
|
print qq{SECURITY.md names $nv\n};
|
||||||
|
'
|
||||||
|
|
||||||
|
- name: Build
|
||||||
|
run: go build ./...
|
||||||
|
|
||||||
|
- name: Format
|
||||||
|
run: |
|
||||||
|
perl -e '
|
||||||
|
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
|
||||||
|
my @bad = <$g>;
|
||||||
|
close($g);
|
||||||
|
print @bad;
|
||||||
|
exit(@bad ? 1 : 0);
|
||||||
|
'
|
||||||
|
|
||||||
|
- name: Vet
|
||||||
|
run: go vet ./...
|
||||||
|
|
||||||
|
- name: Modernise
|
||||||
|
run: go fix -diff ./...
|
||||||
|
|
||||||
|
- name: Tests
|
||||||
|
# The same command as in test.yml, so the floor is the same number everywhere.
|
||||||
|
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
|
||||||
|
|
||||||
|
- name: Tests outside the coverage set
|
||||||
|
# The same command as in test.yml: the CLI's exit codes and manual-page guard,
|
||||||
|
# and the debugger's architecture-neutral units, run outside the floor.
|
||||||
|
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
|
||||||
|
|
||||||
|
- name: Coverage floor
|
||||||
|
run: |
|
||||||
|
perl -e '
|
||||||
|
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
|
||||||
|
my $total;
|
||||||
|
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
|
||||||
|
close($c);
|
||||||
|
die qq{no total line in coverage.out\n} unless defined $total;
|
||||||
|
printf qq{Total coverage: %s%%\n}, $total;
|
||||||
|
exit($total < 80 ? 1 : 0);
|
||||||
|
'
|
||||||
|
|
||||||
build:
|
build:
|
||||||
runs-on: fedora
|
runs-on: fedora
|
||||||
|
timeout-minutes: 25
|
||||||
|
needs: gates
|
||||||
strategy:
|
strategy:
|
||||||
fail-fast: false
|
fail-fast: false
|
||||||
matrix:
|
matrix:
|
||||||
|
# Portable targets: amd64, arm64, loong64 and riscv64 on Linux, at the toolchain
|
||||||
|
# default level. No 32-bit, no wasm, no macOS, no Windows. FreeBSD stays out until
|
||||||
|
# verify/jit.go ports off syscall.Mprotect: the Go syscall package defines no
|
||||||
|
# Mprotect for freebsd, and verify/jit.go:50 calls it to drop the write bit from
|
||||||
|
# the JIT mapping, so every freebsd target fails to build with "undefined:
|
||||||
|
# syscall.Mprotect" (verified for amd64, arm64 and riscv64 on go1.27.1).
|
||||||
include:
|
include:
|
||||||
- goos: linux
|
- goos: linux
|
||||||
goarch: amd64
|
goarch: amd64
|
||||||
- goos: linux
|
- goos: linux
|
||||||
goarch: arm64
|
goarch: arm64
|
||||||
- goos: linux
|
|
||||||
goarch: riscv64
|
|
||||||
- goos: linux
|
- goos: linux
|
||||||
goarch: loong64
|
goarch: loong64
|
||||||
|
- goos: linux
|
||||||
|
goarch: riscv64
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v7
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
- uses: actions/setup-go@v6
|
- uses: actions/setup-go@v6
|
||||||
with:
|
with:
|
||||||
go-version: "1.27"
|
go-version-file: go.mod
|
||||||
|
cache: true
|
||||||
|
|
||||||
- name: Download dependencies
|
- name: Install Perl
|
||||||
run: go mod download
|
run: dnf install -y perl
|
||||||
|
|
||||||
- name: Validate tag and build
|
- name: Validate the tag
|
||||||
id: build
|
id: version
|
||||||
env:
|
env:
|
||||||
VERSION: ${{ gitea.ref_name }}
|
VERSION: ${{ gitea.ref_name }}
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
perl -e '
|
||||||
|
my $v = $ENV{VERSION} // q{};
|
||||||
|
$v =~ m{^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$}
|
||||||
|
or die qq{ERROR: expected a semver tag like v1.2.3, got: $v\n};
|
||||||
|
(my $nv = $v) =~ s{^v}{};
|
||||||
|
open(my $o, q{>>}, $ENV{GITEA_OUTPUT}) or die qq{GITEA_OUTPUT: $!};
|
||||||
|
print $o qq{version_no_v=$nv\n};
|
||||||
|
close($o);
|
||||||
|
print qq{version $nv\n};
|
||||||
|
'
|
||||||
|
|
||||||
if ! echo "$VERSION" | grep -qE '^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$'; then
|
- name: Build
|
||||||
echo "ERROR: expected a semver tag like v1.2.3, got: '$VERSION'"
|
env:
|
||||||
exit 1
|
VERSION_NO_V: ${{ steps.version.outputs.version_no_v }}
|
||||||
fi
|
GOOS: ${{ matrix.goos }}
|
||||||
|
GOARCH: ${{ matrix.goarch }}
|
||||||
VERSION_NO_V="${VERSION#v}"
|
CGO_ENABLED: "0"
|
||||||
echo "version_no_v=${VERSION_NO_V}" >> "$GITEA_OUTPUT"
|
run: |
|
||||||
|
# Nothing is injected. The toolchain records the tag into the binary's build
|
||||||
mkdir -p bin
|
# information, so the version is right because this build happens at the tag, and
|
||||||
GOOS=${{ matrix.goos }} GOARCH=${{ matrix.goarch }} CGO_ENABLED=0 \
|
# there is no path for anyone to get wrong. -s -w only strips symbols.
|
||||||
go build -ldflags "-s -w -X main.version=${VERSION_NO_V}" \
|
go build -ldflags "-s -w" -o "bin/gasm-${VERSION_NO_V}-${GOOS}-${GOARCH}" ./cmd/gasm
|
||||||
-o "bin/gasm-${VERSION_NO_V}-${{ matrix.goos }}-${{ matrix.goarch }}" \
|
|
||||||
./cmd/gasm
|
|
||||||
|
|
||||||
|
# Artifacts stay on v3: v4 and later detect Gitea as GHES and abort.
|
||||||
- name: Upload artifact
|
- name: Upload artifact
|
||||||
uses: actions/upload-artifact@v3
|
uses: actions/upload-artifact@v3
|
||||||
with:
|
with:
|
||||||
name: gasm-${{ matrix.goos }}-${{ matrix.goarch }}
|
name: gasm-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||||
path: bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
path: bin/gasm-${{ steps.version.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||||
if-no-files-found: error
|
if-no-files-found: error
|
||||||
|
|
||||||
- name: Smoke test
|
- name: Smoke test
|
||||||
|
# Only a binary matching the runner can be run here. The check is not that --version
|
||||||
|
# exits cleanly but that it reports the tag and nothing more: a build outside version
|
||||||
|
# control reports (devel), and a build whose tree was dirty reports +dirty, and both
|
||||||
|
# would otherwise be published.
|
||||||
if: matrix.goos == 'linux' && matrix.goarch == 'amd64'
|
if: matrix.goos == 'linux' && matrix.goarch == 'amd64'
|
||||||
|
env:
|
||||||
|
TAG: ${{ gitea.ref_name }}
|
||||||
|
BIN: bin/gasm-${{ steps.version.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||||
run: |
|
run: |
|
||||||
chmod +x bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
perl -e '
|
||||||
./bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }} --version
|
my $want = $ENV{TAG} // die qq{ERROR: no tag\n};
|
||||||
|
open(my $bin, q{-|}, $ENV{BIN}, q{--version}) or die qq{$ENV{BIN}: $!};
|
||||||
|
my $got = <$bin>;
|
||||||
|
close($bin);
|
||||||
|
$got = defined $got ? $got : q{};
|
||||||
|
chomp $got;
|
||||||
|
index($got, $want) >= 0
|
||||||
|
or die qq{ERROR: the binary printed "$got", which does not contain $want. Version control was disabled, so there is no recorded version.\n};
|
||||||
|
index($got, q{+dirty}) < 0
|
||||||
|
or die qq{ERROR: the binary printed "$got". The tree was dirty at build time, which means the checkout was not the tag, or the build artefacts are not ignored.\n};
|
||||||
|
print qq{$ENV{BIN} reports $got\n};
|
||||||
|
'
|
||||||
|
|
||||||
release:
|
release:
|
||||||
runs-on: fedora
|
runs-on: fedora
|
||||||
|
timeout-minutes: 15
|
||||||
needs: build
|
needs: build
|
||||||
permissions:
|
permissions:
|
||||||
|
# contents: read is required for the checkout: a job that declares any
|
||||||
|
# permissions gets a token scoped to exactly those, and releases: write
|
||||||
|
# alone leaves the fetch with no read access, which Gitea answers with
|
||||||
|
# a 404 "Repository not found". Verified on the instance 2026-09-16.
|
||||||
|
contents: read
|
||||||
releases: write
|
releases: write
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v7
|
- uses: actions/checkout@v7
|
||||||
@@ -77,81 +226,162 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
path: dist
|
path: dist
|
||||||
|
|
||||||
- name: Extract CHANGELOG section
|
- name: Install Perl
|
||||||
|
run: dnf install -y perl
|
||||||
|
|
||||||
|
- name: Extract the CHANGELOG section
|
||||||
env:
|
env:
|
||||||
VERSION: ${{ gitea.ref_name }}
|
VERSION: ${{ gitea.ref_name }}
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
# Each step derives what it needs from the tag, so no value has to travel between
|
||||||
VERSION_NO_V="${VERSION#v}"
|
# jobs.
|
||||||
|
perl -e '
|
||||||
|
my $v = $ENV{VERSION} // q{};
|
||||||
|
$v =~ s{^v}{};
|
||||||
|
open(my $vout, q{>}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
|
||||||
|
print $vout $v;
|
||||||
|
close($vout);
|
||||||
|
open(my $in, q{<}, q{CHANGELOG.md}) or die qq{CHANGELOG.md: $!};
|
||||||
|
my @lines = <$in>;
|
||||||
|
close($in);
|
||||||
|
my ($start, $end) = (-1, scalar @lines);
|
||||||
|
for my $i (0 .. $#lines) {
|
||||||
|
if ($start < 0) { $start = $i if $lines[$i] =~ m{^##\s+\[\Q$v\E\]} }
|
||||||
|
elsif ($lines[$i] =~ m{^##\s+\[}) { $end = $i; last }
|
||||||
|
}
|
||||||
|
$start >= 0 or die qq{ERROR: no CHANGELOG section for $v, expected a heading like: ## [$v] - YYYY-MM-DD\n};
|
||||||
|
my @body = grep { m{\S} } @lines[$start + 1 .. $end - 1];
|
||||||
|
@body or die qq{ERROR: the CHANGELOG section for $v is empty\n};
|
||||||
|
open(my $out, q{>}, q{release-body.md}) or die qq{release-body.md: $!};
|
||||||
|
print $out @body;
|
||||||
|
close($out);
|
||||||
|
printf qq{notes for %s: %d lines\n}, $v, scalar @body;
|
||||||
|
'
|
||||||
|
|
||||||
sed -n "/^## \[${VERSION_NO_V}\] /,/^## \[/p" CHANGELOG.md \
|
- name: Build the release request
|
||||||
| sed '$d' \
|
run: |
|
||||||
| tail -n +2 \
|
perl -e '
|
||||||
> release-body.md
|
open(my $vin, q{<}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
|
||||||
|
my $v = <$vin>;
|
||||||
|
close($vin);
|
||||||
|
chomp $v;
|
||||||
|
open(my $in, q{<:raw}, q{release-body.md}) or die qq{release-body.md: $!};
|
||||||
|
my $body = do { local $/; <$in> };
|
||||||
|
close($in);
|
||||||
|
# Byte-oriented escaping: JSON is UTF-8, so non-ASCII passes through and only the
|
||||||
|
# characters JSON forbids are rewritten.
|
||||||
|
$body =~ s/([\\"])/\\$1/g;
|
||||||
|
$body =~ s/\t/\\t/g;
|
||||||
|
$body =~ s/\r//g;
|
||||||
|
$body =~ s/\n/\\n/g;
|
||||||
|
$body =~ s/([\x00-\x08\x0b\x0c\x0e-\x1f])/sprintf(q{\u%04x}, ord($1))/ge;
|
||||||
|
my $json = sprintf(qq{{"tag_name":"v%s","name":"v%s","body":"%s","draft":false,"prerelease":false}}, $v, $v, $body);
|
||||||
|
open(my $out, q{>}, q{release.json}) or die qq{release.json: $!};
|
||||||
|
print $out $json;
|
||||||
|
close($out);
|
||||||
|
print qq{release.json written for v$v\n};
|
||||||
|
'
|
||||||
|
|
||||||
if [ ! -s release-body.md ]; then
|
- name: Create the release
|
||||||
echo "ERROR: no CHANGELOG section found for ${VERSION_NO_V}"
|
|
||||||
echo "Expected a heading like: ## [${VERSION_NO_V}] — YYYY-MM-DD"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
- name: Create release
|
|
||||||
env:
|
env:
|
||||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||||
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
||||||
GITEA_REPOSITORY: ${{ gitea.repository }}
|
GITEA_REPOSITORY: ${{ gitea.repository }}
|
||||||
GITEA_REF_NAME: ${{ gitea.ref_name }}
|
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
perl -e '
|
||||||
|
my @cmd = (q{curl}, q{-sS}, q{-o}, q{response.json}, q{-w}, q{%{http_code}},
|
||||||
BODY=$(sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' -e 's/\t/\\t/g' -e 's/\r//g' release-body.md | sed ':a;N;$!ba;s/\n/\\n/g')
|
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
|
||||||
BODY="\"${BODY}\""
|
q{-H}, q{Content-Type: application/json},
|
||||||
|
q{-X}, q{POST},
|
||||||
response=$(curl -sS -w '\n%{http_code}' \
|
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases},
|
||||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
q{--data-binary}, q{@release.json});
|
||||||
-H "Content-Type: application/json" \
|
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
|
||||||
-X POST \
|
my $code = <$curl>;
|
||||||
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases" \
|
my $ok = close($curl);
|
||||||
-d "{\"tag_name\":\"${GITEA_REF_NAME}\",\"name\":\"${GITEA_REF_NAME}\",\"body\":${BODY},\"draft\":false,\"prerelease\":false}")
|
my $exit = $? >> 8;
|
||||||
|
$code = defined $code ? $code : q{};
|
||||||
http_code=$(echo "$response" | tail -1)
|
$ok or die qq{ERROR: curl failed (exit $exit) calling $ENV{GITEA_SERVER_URL}\n};
|
||||||
payload=$(echo "$response" | sed '$d')
|
open(my $r, q{<:raw}, q{response.json}) or die qq{response.json: $!};
|
||||||
|
my $body = do { local $/; <$r> };
|
||||||
echo "HTTP ${http_code}"
|
close($r);
|
||||||
if [ "$http_code" != "201" ]; then
|
$code eq q{201} or die qq{ERROR: the release was not created, HTTP $code: $body\n};
|
||||||
echo "Failed to create release: ${payload}"
|
$body =~ m{"id"\s*:\s*([0-9]+)} or die qq{ERROR: no release id in the response: $body\n};
|
||||||
exit 1
|
open(my $o, q{>}, q{release-id.txt}) or die qq{release-id.txt: $!};
|
||||||
fi
|
print $o $1;
|
||||||
|
close($o);
|
||||||
RELEASE_ID=$(echo "$payload" | grep -oE '"id"[[:space:]]*:[[:space:]]*[0-9]+' | head -1 | grep -oE '[0-9]+')
|
print qq{release id $1\n};
|
||||||
echo "Created release ID=${RELEASE_ID}"
|
'
|
||||||
printf '%s' "${RELEASE_ID}" > release-id.txt
|
|
||||||
|
|
||||||
- name: Upload assets
|
- name: Upload assets
|
||||||
env:
|
env:
|
||||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||||
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
||||||
GITEA_REPOSITORY: ${{ gitea.repository }}
|
GITEA_REPOSITORY: ${{ gitea.repository }}
|
||||||
GITEA_REF_NAME: ${{ gitea.ref_name }}
|
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
perl -e '
|
||||||
RELEASE_ID=$(cat release-id.txt)
|
open(my $f, q{<}, q{release-id.txt}) or die qq{release-id.txt: $!};
|
||||||
|
my $id = <$f>;
|
||||||
for binary in dist/gasm-*/gasm-*; do
|
close($f);
|
||||||
[ -f "$binary" ] || continue
|
chomp $id;
|
||||||
fname=$(basename "$binary")
|
my @files = grep { -f $_ } glob(q{dist/*/*});
|
||||||
echo "Uploading ${fname}..."
|
@files or die qq{ERROR: no assets under dist/\n};
|
||||||
http_code=$(curl -sS -o /dev/null -w '%{http_code}' \
|
# A file that arrived empty from the artifact step would be uploaded as an
|
||||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
# empty attachment, every status would still be 201, and the run would go
|
||||||
-H "Content-Type: application/octet-stream" \
|
# green over a release nobody can install. Refuse it here, before the
|
||||||
-X POST \
|
# upload, and verify what was stored afterwards.
|
||||||
--data-binary "@${binary}" \
|
my %size;
|
||||||
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases/${RELEASE_ID}/assets?name=${fname}")
|
for my $path (@files) {
|
||||||
echo " HTTP ${http_code}"
|
my $n = -s $path // 0;
|
||||||
if [ "$http_code" != "201" ]; then
|
(my $name = $path) =~ s{.*/}{};
|
||||||
echo "Failed to upload ${fname}"
|
$n > 0 or die qq{ERROR: $path is empty, so there is nothing to upload\n};
|
||||||
exit 1
|
$size{$name} = $n;
|
||||||
fi
|
}
|
||||||
done
|
my $bad = 0;
|
||||||
|
for my $path (@files) {
|
||||||
echo "Release ${GITEA_REF_NAME} is live."
|
(my $name = $path) =~ s{.*/}{};
|
||||||
|
my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}},
|
||||||
|
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
|
||||||
|
q{-H}, q{Content-Type: application/octet-stream},
|
||||||
|
q{-X}, q{POST}, q{--data-binary}, qq{@$path},
|
||||||
|
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name});
|
||||||
|
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
|
||||||
|
my $code = <$curl>;
|
||||||
|
my $ok = close($curl);
|
||||||
|
my $exit = $? >> 8;
|
||||||
|
$code = defined $code ? $code : q{};
|
||||||
|
unless ($ok) {
|
||||||
|
printf qq{%s: curl failed (exit %d)\n}, $name, $exit;
|
||||||
|
$bad = 1;
|
||||||
|
next;
|
||||||
|
}
|
||||||
|
printf qq{%s: HTTP %s\n}, $name, $code;
|
||||||
|
$bad = 1 if $code ne q{201};
|
||||||
|
}
|
||||||
|
# Read every asset back through the release download route and require the
|
||||||
|
# served length to be the file that was sent: stored but empty is a broken
|
||||||
|
# release however green the run looks.
|
||||||
|
open(my $v, q{<}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
|
||||||
|
my $v = <$v>;
|
||||||
|
close($v);
|
||||||
|
chomp $v;
|
||||||
|
for my $name (sort keys %size) {
|
||||||
|
my $url = qq{$ENV{GITEA_SERVER_URL}/$ENV{GITEA_REPOSITORY}/releases/download/v$v/$name};
|
||||||
|
my @head = (q{curl}, q{-sS}, q{-I}, q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}}, $url);
|
||||||
|
open(my $h, q{-|}, @head) or die qq{curl: $!};
|
||||||
|
my $len;
|
||||||
|
my $status;
|
||||||
|
while (my $l = <$h>) {
|
||||||
|
$status = $1 if $l =~ m{^HTTP/\S+\s+(\d+)};
|
||||||
|
$len = $1 if $l =~ m{^content-length:\s*(\d+)}i;
|
||||||
|
}
|
||||||
|
my $ok = close($h);
|
||||||
|
$len = defined $len ? $len : 0;
|
||||||
|
if (!$ok || $status != 200 || $len != $size{$name}) {
|
||||||
|
printf qq{ERROR: %s serves %s bytes, expected %d\n}, $name, $len, $size{$name};
|
||||||
|
$bad = 1;
|
||||||
|
next;
|
||||||
|
}
|
||||||
|
printf qq{%s: serves %d bytes\n}, $name, $len;
|
||||||
|
}
|
||||||
|
exit($bad ? 1 : 0);
|
||||||
|
'
|
||||||
|
|||||||
+90
-76
@@ -1,4 +1,17 @@
|
|||||||
# Test — gasm-devkit. Runs on push and pull request to development.
|
# Test, Go. Push and pull request to development. Never on main.
|
||||||
|
#
|
||||||
|
# The gates are the ones the justfile's `gates` recipe runs, minus race: the shared
|
||||||
|
# runner box cannot afford the race detector on every push, so it lives in race.yml.
|
||||||
|
# The box is one core and 2 GB beside Gitea, so parallelism is bounded on purpose and
|
||||||
|
# everything runs in one job. Extra jobs would duplicate the checkout, the Go setup and
|
||||||
|
# the dependency download three times without buying any parallelism.
|
||||||
|
#
|
||||||
|
# Every step is one command, so the step that fails is the gate that failed, and no shell
|
||||||
|
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
|
||||||
|
# not Python: Perl behaves the same on both runner images, there is no bashism to trip over
|
||||||
|
# on ash, and it is one language instead of two. The Perl uses builtins only, because
|
||||||
|
# Fedora packages the Perl modules separately and nothing beyond `perl` itself may be
|
||||||
|
# assumed present.
|
||||||
name: Test
|
name: Test
|
||||||
|
|
||||||
on:
|
on:
|
||||||
@@ -7,90 +20,91 @@ on:
|
|||||||
pull_request:
|
pull_request:
|
||||||
branches: [development]
|
branches: [development]
|
||||||
|
|
||||||
|
env:
|
||||||
|
# One core: parallelism buys no speed here and costs memory the box does not have.
|
||||||
|
GOFLAGS: -p=1
|
||||||
|
GOMAXPROCS: "2"
|
||||||
|
|
||||||
|
# A superseded run of the same ref is cancelled instead of queueing behind one that
|
||||||
|
# no longer matters. Verified on Gitea 1.27.1 on 2026-09-17: a queued run whose ref
|
||||||
|
# moved on is cancelled before it ever reaches the runner, while a run already
|
||||||
|
# dispatched there runs to completion.
|
||||||
|
concurrency:
|
||||||
|
group: ${{ gitea.workflow }}-${{ gitea.ref }}
|
||||||
|
cancel-in-progress: true
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
vet:
|
|
||||||
runs-on: fedora
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v7
|
|
||||||
|
|
||||||
- uses: actions/setup-go@v6
|
|
||||||
with:
|
|
||||||
go-version: "1.27"
|
|
||||||
|
|
||||||
- name: Download dependencies
|
|
||||||
run: go mod download
|
|
||||||
|
|
||||||
- name: gofmt
|
|
||||||
run: |
|
|
||||||
set -euo pipefail
|
|
||||||
unformatted=$(gofmt -l .)
|
|
||||||
if [ -n "$unformatted" ]; then
|
|
||||||
echo "These files need gofmt:"
|
|
||||||
echo "$unformatted"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
- name: go vet
|
|
||||||
run: go vet ./...
|
|
||||||
|
|
||||||
test:
|
test:
|
||||||
runs-on: fedora
|
runs-on: fedora
|
||||||
needs: vet
|
timeout-minutes: 10
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v7
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
- uses: actions/setup-go@v6
|
- uses: actions/setup-go@v6
|
||||||
with:
|
with:
|
||||||
go-version: "1.27"
|
# The module is the source of truth for the version, so it cannot drift.
|
||||||
|
go-version-file: go.mod
|
||||||
|
cache: true
|
||||||
|
|
||||||
- name: Download dependencies
|
- name: Install Perl
|
||||||
run: go mod download
|
# The runner images are minimal and Perl is not guaranteed. The install is a
|
||||||
|
# no-op where it is already present; drop this step once verified on the box.
|
||||||
- name: Install gcc
|
run: dnf install -y perl
|
||||||
run: dnf install -y gcc
|
|
||||||
|
|
||||||
- name: go test -race
|
|
||||||
run: go test -race -count=1 ./...
|
|
||||||
|
|
||||||
- name: Coverage gate — 80 % minimum
|
|
||||||
run: |
|
|
||||||
set -euo pipefail
|
|
||||||
# Exclude packages inherently untestable without hardware:
|
|
||||||
# debug — interactive ptrace, requires a live process
|
|
||||||
# cmd/gasm — CLI glue, covered by integration tests
|
|
||||||
go test -coverprofile=coverage.out \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/arch \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/asm \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/ast \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/format \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/lexer \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/lint \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/lsp \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/parser \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/token \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/verify
|
|
||||||
coverage=$(go tool cover -func=coverage.out | awk '/^total:/ { gsub("%", "", $3); print $3 }')
|
|
||||||
echo "Total coverage: ${coverage}%"
|
|
||||||
if awk -v c="$coverage" 'BEGIN { exit !(c+0 < 80) }'; then
|
|
||||||
echo "ERROR: coverage ${coverage}% is below the 80% threshold"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
build:
|
|
||||||
runs-on: fedora
|
|
||||||
needs: test
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v7
|
|
||||||
|
|
||||||
- uses: actions/setup-go@v6
|
|
||||||
with:
|
|
||||||
go-version: "1.27"
|
|
||||||
|
|
||||||
- name: Download dependencies
|
|
||||||
run: go mod download
|
|
||||||
|
|
||||||
|
# The steps follow the `gates` order of the justfile contract: build, format,
|
||||||
|
# vet, test. The vet gate is go vet and go fix -diff, two steps here.
|
||||||
- name: Build
|
- name: Build
|
||||||
run: go build -ldflags="-s -w" -o bin/gasm ./cmd/gasm
|
run: go build ./...
|
||||||
|
|
||||||
- name: Smoke test
|
- name: Format
|
||||||
run: ./bin/gasm --version
|
run: |
|
||||||
|
perl -e '
|
||||||
|
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
|
||||||
|
my @bad = <$g>;
|
||||||
|
close($g);
|
||||||
|
print @bad;
|
||||||
|
exit(@bad ? 1 : 0);
|
||||||
|
'
|
||||||
|
|
||||||
|
- name: Vet
|
||||||
|
run: go vet ./...
|
||||||
|
|
||||||
|
- name: Modernise
|
||||||
|
# Exits non-zero when it has something to rewrite, so it needs no output capture.
|
||||||
|
run: go fix -diff ./...
|
||||||
|
|
||||||
|
- name: Tests
|
||||||
|
# The suite must be fast: a push pipeline that cannot finish in a few minutes moves
|
||||||
|
# its heavy part behind a dispatch. The inner timeout matches the job's, so a
|
||||||
|
# hanging test reports its own goroutine dump rather than a silent job kill.
|
||||||
|
# The pattern is `packages` in the project's justfile: the logic packages, since a
|
||||||
|
# thin cmd/ would drag the total under the floor. release.yml runs the same
|
||||||
|
# command, so the floor is the same number everywhere. ./verify/... carries the
|
||||||
|
# live oracle-parity comparison against `go tool asm` (the TestGroundTruth
|
||||||
|
# suites); the runner's Go setup provides both the tool and GOROOT.
|
||||||
|
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
|
||||||
|
|
||||||
|
- name: Tests outside the coverage set
|
||||||
|
# The CLI and the debugger sit outside `packages` because a thin main and a
|
||||||
|
# ptrace-bound package pull the total under the floor, but their tests guard
|
||||||
|
# shipped surfaces: the command exit codes, the manual pages against the
|
||||||
|
# binary's own help, and the debugger's architecture-neutral units. They run
|
||||||
|
# here so the floor stays a product measure and nothing is left untested.
|
||||||
|
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
|
||||||
|
|
||||||
|
- name: Oracle parity
|
||||||
|
# Re-run the live go-tool-asm comparison as its own step so that a parity
|
||||||
|
# regression names the gate that failed instead of hiding inside the suite.
|
||||||
|
run: go test -count=1 -timeout 10m -run 'TestGroundTruth' ./verify/...
|
||||||
|
|
||||||
|
- name: Coverage floor
|
||||||
|
run: |
|
||||||
|
perl -e '
|
||||||
|
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
|
||||||
|
my $total;
|
||||||
|
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
|
||||||
|
close($c);
|
||||||
|
die qq{no total line in coverage.out\n} unless defined $total;
|
||||||
|
printf qq{Total coverage: %s%%\n}, $total;
|
||||||
|
exit($total < 80 ? 1 : 0);
|
||||||
|
'
|
||||||
|
|||||||
+3
-15
@@ -1,25 +1,13 @@
|
|||||||
# Metadata (always first, per repo convention)
|
|
||||||
.idea/
|
.idea/
|
||||||
.zcode/
|
.zcode/
|
||||||
.qwen/
|
|
||||||
.mimocode/
|
|
||||||
|
|
||||||
# Binaries
|
# Build output
|
||||||
/gasm
|
|
||||||
/bin/
|
/bin/
|
||||||
*.exe
|
/gasm
|
||||||
|
|
||||||
# Test and coverage artefacts
|
|
||||||
coverage.out
|
coverage.out
|
||||||
*.test
|
*.test
|
||||||
|
|
||||||
# Crash dumps
|
# Crash dumps from the emulator runs
|
||||||
core
|
core
|
||||||
core.*
|
core.*
|
||||||
*.core
|
*.core
|
||||||
|
|
||||||
# Scratch / temporary work
|
|
||||||
_scratch/
|
|
||||||
|
|
||||||
# ZCode workspace
|
|
||||||
.zcode
|
|
||||||
|
|||||||
+478
-148
File diff suppressed because it is too large
Load Diff
+105
-78
@@ -1,107 +1,134 @@
|
|||||||
# Contributing to gasm-devkit
|
# Contributing
|
||||||
|
|
||||||
Thanks for contributing to gasm-devkit.
|
Contributions to **gasm-devkit** are governed by the Contributor terms
|
||||||
|
below; submitting one means you accept them.
|
||||||
|
|
||||||
|
## Contributor terms
|
||||||
|
|
||||||
|
1. This project belongs to its owner alone. The owner decides what is
|
||||||
|
accepted, in what form and when; the decision is final and needs no
|
||||||
|
justification.
|
||||||
|
2. By submitting a contribution you assign to Petr Balvín
|
||||||
|
<opensource@petrbalvin.org> all present and future copyright and
|
||||||
|
related rights in it, worldwide, for the full term of the rights,
|
||||||
|
with the right to relicense and sublicense without restriction,
|
||||||
|
including under proprietary terms.
|
||||||
|
3. Where that assignment is not effective, it counts as a perpetual,
|
||||||
|
irrevocable, royalty-free licence with the same scope.
|
||||||
|
4. To the fullest extent permitted by law, you waive any right of
|
||||||
|
attribution and integrity in the contribution. The project names no
|
||||||
|
contributors and keeps no credits list.
|
||||||
|
5. By submitting you represent that the work is yours and that you
|
||||||
|
hold the rights to assign it as above.
|
||||||
|
|
||||||
## Development setup
|
## Development setup
|
||||||
|
|
||||||
Requirements: Go 1.27 or later, the [just](https://github.com/casey/just)
|
Requirements: Go 1.27.1, the exact version the `go` directive in `go.mod`
|
||||||
command runner, and a Linux host on amd64, arm64, riscv64 or loong64.
|
declares, [just](https://github.com/casey/just) for the recipes, and a C
|
||||||
|
compiler (gcc), because `just gates` includes `just race` and the race
|
||||||
|
detector needs cgo.
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
|
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
|
||||||
cd gasm-devkit
|
cd gasm-devkit
|
||||||
just install # download module dependencies
|
just build
|
||||||
just build # go vet + gofmt check
|
just gates
|
||||||
just test # full suite, race detector, 80 % coverage gate
|
|
||||||
```
|
```
|
||||||
|
|
||||||
## Workflow
|
## Workflow
|
||||||
|
|
||||||
1. Branch from `development`; never commit directly to `main` (`main` is
|
1. Branch from `development`. Never commit directly to `main`, which is release-only.
|
||||||
release-only: merge from `development`, then tag).
|
2. Commit in [Conventional Commits](https://www.conventionalcommits.org/) form:
|
||||||
2. Commit with [Conventional Commits](https://www.conventionalcommits.org/):
|
`type(scope): description`, subject line only, imperative mood, lowercase after the
|
||||||
`type(scope): description`: subject line only, imperative mood,
|
colon, no trailing full stop. Allowed types: `feat`, `fix`, `docs`, `style`,
|
||||||
lowercase after the colon, no trailing dot. Allowed types: `feat`,
|
`refactor`, `perf`, `test`, `chore`, `ci`, `build`, `revert`.
|
||||||
`fix`, `docs`, `style`, `refactor`, `perf`, `test`, `chore`, `ci`,
|
3. One logical change per commit. A refactor, a behaviour change and a formatting pass
|
||||||
`build`, `revert`. The only line after the subject is the trailer:
|
are three commits, never one.
|
||||||
`Assisted-by: <model-name>`. No `Co-Authored-By`, no `Signed-off-by`,
|
4. Record every user-visible change in `CHANGELOG.md` under `## [development]`.
|
||||||
no other trailers.
|
5. Add or update tests. Coverage stays at 80 percent or more; it is a hard gate.
|
||||||
3. Record every user-visible change in `CHANGELOG.md` under
|
6. Update the documentation when the public API, the configuration or the behaviour
|
||||||
`## [development]` (categories: Added, Changed, Fixed, Removed,
|
changes.
|
||||||
Security).
|
7. Open a pull request against `development`.
|
||||||
4. Add or update tests; coverage must stay **at or above 80 %** (hard
|
|
||||||
gate, enforced by CI).
|
|
||||||
5. Update the documentation when behaviour, flags or the public surface
|
|
||||||
change.
|
|
||||||
6. Open a pull request against `development`.
|
|
||||||
|
|
||||||
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`;
|
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The release
|
||||||
CI builds and publishes the binaries for all four architectures.
|
workflow builds the assets and publishes the release and its notes.
|
||||||
|
|
||||||
## Code style
|
## Code style
|
||||||
|
|
||||||
`gofmt` and `go vet` via `just fmt` / `just build`; both must pass with
|
`gofmt` and `go vet` run through `just fmt` and `just vet`, with zero diff and zero
|
||||||
zero output; `go fix -diff ./...` must report nothing on touched packages.
|
warnings tolerated. `just vet` is two gates, `go vet ./...` and `go fix -diff ./...`,
|
||||||
|
so the modernisation rewrites are enforced too. `just gates` is the definition of done in
|
||||||
|
one command, and the recipe file names what it contains. Errors are checked explicitly,
|
||||||
|
wrapped as `fmt.Errorf("context: %w", err)`, and nothing panics outside `main`. The
|
||||||
|
recipe file holds the commands, and the language and standard-library surface is the one
|
||||||
|
the `go` directive in `go.mod` pins.
|
||||||
|
|
||||||
- Standard library only in production code; `golang.org/x/arch` is used
|
- `golang.org/x/arch` is the one module dependency, and it is linked into the binary:
|
||||||
in tests only (round-trip decoding) and is never linked into the `gasm`
|
`gasm dis` and the debugger's listings decode through it. Everything else is the
|
||||||
binary.
|
standard library.
|
||||||
- No cgo, no C, no external toolchains at runtime.
|
- No cgo and no C. The standalone encoder paths (`gasm asm --format raw` and `--format
|
||||||
- Explicit `if err != nil`; errors wrapped with
|
elf`) need no Go installation; `gasm verify --ground-truth`, `gasm verify --fuzz`,
|
||||||
`fmt.Errorf("context: %w", err)`; no panics outside `main`.
|
`gasm audit-instructions` and `gasm asm --format goobj` resolve through the installed
|
||||||
- The parser, lexer and formatter are hand-written; the `arch` instruction
|
Go toolchain.
|
||||||
tables are generated only via `_gen/gen.go` (`just gen`), never edited.
|
- The parser, lexer and formatter are hand-written; the `arch` instruction tables are
|
||||||
|
generated only by `_gen/gen.go` (`just gen`) and never edited by hand.
|
||||||
|
- Assembly committed to the repository goes through `gasm fmt` and `gasm lint`, so a
|
||||||
|
`.s` file that `gasm fmt -l .` lists is unfinished.
|
||||||
|
|
||||||
## Running a single test
|
New source files open with the project's two-line licence header, whose SPDX
|
||||||
|
identifier matches `LICENSE`. Configuration files, workflows and dotfiles do not carry
|
||||||
|
it.
|
||||||
|
|
||||||
```sh
|
## AI contribution policy
|
||||||
go test -run TestVexGroundTruth ./asm/
|
|
||||||
go test -run TestGroundTruthBasic ./verify/
|
|
||||||
go test -run TestGOObjectLinkAndRun ./asm/
|
|
||||||
go test -run TestFuzzWideCopy ./verify/
|
|
||||||
```
|
|
||||||
|
|
||||||
The interactive debugger (`gasm debug`) requires a compiled binary on
|
AI tools are welcome as productivity aids and are a normal part of modern software
|
||||||
`$PATH`; `go run` does not work for the traced child process. Install
|
development. What matters is that the contribution stays understandable, reviewable and
|
||||||
first with `just install-bin`.
|
genuinely useful.
|
||||||
|
|
||||||
## CI (Gitea Actions)
|
- **Disclose the assistance.** If AI helped draft any part of a commit, issue, pull
|
||||||
|
request or review, say so.
|
||||||
|
- **Commit messages carry exactly one trailer**, as a git trailer on the line after a
|
||||||
|
blank line that closes the subject:
|
||||||
|
|
||||||
Workflows live in `.gitea/workflows/` and run on self-hosted runners:
|
```
|
||||||
|
Assisted-by: MODEL
|
||||||
|
```
|
||||||
|
|
||||||
|
Name the model that did the work, spelled the way its maker spells it, for example
|
||||||
|
`GLM 5.3`, `DeepSeek V4.1 Flash` or `Qwen 3.8 Flash`. No `Co-Authored-By`, no `Signed-off-by`,
|
||||||
|
no other trailers, and no prose: the trailer is the disclosure.
|
||||||
|
- **Issues and pull requests** attribute the assistance in a comment, for example
|
||||||
|
`_Assisted-by: GLM 5.3_`. It does not belong in the pull request description.
|
||||||
|
- **Take responsibility.** You are accountable for the accuracy, completeness and
|
||||||
|
intent of everything you submit, whether or not AI produced it.
|
||||||
|
- **Review before marking ready.** Read the diff carefully, run it locally, and add the
|
||||||
|
tests it needs. Do not mark a pull request ready until you can defend every change in
|
||||||
|
it.
|
||||||
|
- **Quality over quantity.** Contributions that look like un-reviewed output, or whose
|
||||||
|
author cannot engage substantively during review, may be closed.
|
||||||
|
- **Preferred models.** Prefer open-weight models with transparent training data and
|
||||||
|
minimal output filtering.
|
||||||
|
|
||||||
|
AI assists. It does not replace judgement.
|
||||||
|
|
||||||
|
## Continuous integration
|
||||||
|
|
||||||
|
Workflows live in `.gitea/workflows/` and run on the project's own runners:
|
||||||
|
|
||||||
| Workflow | Trigger | What it does |
|
| Workflow | Trigger | What it does |
|
||||||
|----------|---------|--------------|
|
|---|---|---|
|
||||||
| Test | push / PR to `development` | gofmt check, `go vet`, `go test -race`, 80 % coverage gate |
|
| Test | push or pull request to `development` | build, format check, vet, modernisation, the test suite with the coverage floor, the CLI and debugger tests outside the profile, then the oracle-parity rerun against `go tool asm` |
|
||||||
| Release | tag `v*` | cross-compiles binaries for linux/{amd64,arm64,riscv64,loong64} and publishes the Gitea release |
|
| Release | a `v*` tag | the same gates as Test minus the oracle-parity step, then the matrix build, the version smoke test and the release itself; the race detector runs locally in `just gates` before the tag is cut |
|
||||||
|
|
||||||
The Definition of Done (`just build` + `just test` + `just fmt`) must
|
The local equivalent is `just gates`, which is the same set plus the race detector. The
|
||||||
still pass locally before pushing.
|
race detector also has its own workflow, dispatched by hand; it never runs on a push or a
|
||||||
|
tag, where it would double the time and the memory a shared runner cannot spare.
|
||||||
## AI Contribution Policy
|
|
||||||
|
|
||||||
AI tools are welcome as productivity aids. What matters is that
|
|
||||||
contributions remain understandable, reviewable, and genuinely useful.
|
|
||||||
|
|
||||||
- **Disclose AI use.** If you used AI to draft or generate any part of a
|
|
||||||
commit, issue, pull request, or code review, say so clearly.
|
|
||||||
- **Commit messages:** end every commit with exactly one trailer:
|
|
||||||
`Assisted-by: <model-name>` (e.g. `Assisted-by: GLM 5.3`).
|
|
||||||
- **Pull requests and issues:** attribute AI assistance in one trailing
|
|
||||||
line, e.g. `_Assisted-by: GLM 5.3_`. Do not paste it into the PR
|
|
||||||
description as a section.
|
|
||||||
- **Take responsibility.** You remain accountable for the accuracy,
|
|
||||||
completeness, and intent of everything you submit.
|
|
||||||
- **Review before marking ready.** Read AI-generated diffs carefully, run
|
|
||||||
them locally, and add or update tests where appropriate.
|
|
||||||
- **Preferred models.** Prefer open-weight models with transparent
|
|
||||||
training data: **GLM**, **DeepSeek**, and **MiMo**.
|
|
||||||
|
|
||||||
## Reporting bugs
|
## Reporting bugs
|
||||||
|
|
||||||
Open an issue at
|
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-devkit/issues` with the
|
||||||
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit/issues)
|
version, the operating system and architecture, the exact command, the full output,
|
||||||
with the version (`gasm --version`), OS and architecture, the exact
|
and the expected against the actual behaviour.
|
||||||
command, the full output, and the expected versus actual behaviour.
|
|
||||||
|
|
||||||
**Security issues:** email **opensource@petrbalvin.org** instead of opening
|
**Security issues do not go in the issue tracker.** Report them as
|
||||||
a public issue.
|
[SECURITY.md](SECURITY.md) describes, to **opensource@petrbalvin.org**.
|
||||||
|
|||||||
@@ -1,13 +1,65 @@
|
|||||||
# gasm-devkit
|
# Plan 9 assembly tooling, inside and outside Go
|
||||||
|
|
||||||
Developer tooling for **GAsm**, Go's built-in Plan 9 assembler.
|
> **Warning: this is an experiment.** gasm-devkit is under active
|
||||||
|
> development and is not stable. The version is 0.x.x: commands, flags,
|
||||||
|
> output formats and behaviour can change without warning at any time.
|
||||||
|
> A 1.0.0 release is light years away. Nothing in this document is a
|
||||||
|
> stability promise. For all of that, this is not a paper project: gasm
|
||||||
|
> is already in active use and is tested on real assembly work. Only
|
||||||
|
> amd64 is validated on real hardware; the other three architectures run
|
||||||
|
> under emulation ([Validation status](#validation-status)).
|
||||||
|
|
||||||
Go ships an assembler but no tooling for it: there is no syntax highlighting,
|
**GAsm** is Go's Plan 9 assembler, and Go ships it without tooling:
|
||||||
no autocomplete, no linter, no static analyser, no formatter, no standalone
|
there is no formatter, no linter and no debugger for `.s` files, and no
|
||||||
assembler and no debugger for `.s` files. Developers write assembly blind,
|
assembler that works without a Go installation. Developers write
|
||||||
validate it by benchmark, and debug it by print statement. gasm-devkit is the
|
assembly blind, validate it by benchmark, and debug it by print
|
||||||
missing toolkit: a single, self-contained binary, `gasm`, that brings proper
|
statement. gasm-devkit is the missing toolkit: a single, self-contained
|
||||||
developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
|
binary, `gasm`, that serves both purposes.
|
||||||
|
|
||||||
|
- **Help develop Plan 9 assembly.** Formatting, linting, disassembly,
|
||||||
|
dynamic verification, a source-level debugger and a language server,
|
||||||
|
for `.s` files in Go programs.
|
||||||
|
- **Use Plan 9 assembly outside the Go toolchain.** `gasm asm` encodes
|
||||||
|
on its own and writes raw images or linkable ELF objects with DWARF5
|
||||||
|
debug sections, with no Go installation in the loop; the Go
|
||||||
|
toolchain's own GOOBJ format, which `go build` consumes in place of
|
||||||
|
the toolchain's output, needs the installed toolchain.
|
||||||
|
|
||||||
|
## Why Plan 9 assembly
|
||||||
|
|
||||||
|
Plan 9 assembly is the quiet triumph of the field. One syntax across
|
||||||
|
every architecture Go builds for: the same source-first operand order,
|
||||||
|
the same four pseudo-registers, the same frame convention, whether the
|
||||||
|
target is x86, ARM, RISC-V or LoongArch. Learn it once and you can
|
||||||
|
read a kernel on any of them.
|
||||||
|
|
||||||
|
Compare the alternatives. Intel syntax and AT&T syntax disagree on the
|
||||||
|
one question every instruction answers, which operand is the source
|
||||||
|
and which is the destination, so half the world writes it one way,
|
||||||
|
half the other, and every assembly programmer carries both in their
|
||||||
|
head forever. GNU as settles the argument with directives that switch
|
||||||
|
dialects mid-file (`.intel_syntax noprefix`), a percent sign on every
|
||||||
|
register and a dollar on every immediate: punctuation that carries
|
||||||
|
nothing the operand order did not already say. And the x86 family
|
||||||
|
fragments again underneath: NASM is not MASM is not GAS, each with its
|
||||||
|
own directive zoo and macro language, so every project picks a dialect
|
||||||
|
and every reader learns a different one by accident.
|
||||||
|
|
||||||
|
Plan 9 assembly has none of it. Registers are bare names. Memory is
|
||||||
|
one notation, `offset(base)`, extended by an index and a scale when
|
||||||
|
the instruction needs it. Arguments arrive named and offset-checked:
|
||||||
|
`x+0(FP)` is the argument x, on every architecture, and `go vet`
|
||||||
|
polices the offsets against the Go prototype.
|
||||||
|
|
||||||
|
```text
|
||||||
|
AT&T (GNU as): movq %rax, -16(%rbp)
|
||||||
|
Plan 9 (Go): MOVQ AX, total-16(SP)
|
||||||
|
```
|
||||||
|
|
||||||
|
The same lines, but only one of them tells you what the number is for.
|
||||||
|
The syntax is uppercase, regular and boring, which is the highest
|
||||||
|
compliment a language for machine code can earn. gasm-devkit exists
|
||||||
|
to give that syntax the tooling it deserves.
|
||||||
|
|
||||||
## Features
|
## Features
|
||||||
|
|
||||||
@@ -19,12 +71,14 @@ developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
|
|||||||
operating recursively on directories the way `go fmt` does. `-l` lists
|
operating recursively on directories the way `go fmt` does. `-l` lists
|
||||||
files whose formatting differs and `-d` prints a unified diff.
|
files whose formatting differs and `-d` prints a unified diff.
|
||||||
- **Linter.** `gasm lint` runs 18 conservative static checks, among them
|
- **Linter.** `gasm lint` runs 18 conservative static checks, among them
|
||||||
`undefined-label`, `abi-argsize` (declared frame vs the `// func` signature),
|
`undefined-label`, `abi-argsize` (declared argument area vs the `// func`
|
||||||
`register-clobber` (Go ABI register liveness over the control-flow graph),
|
signature), `register-clobber` (Go ABI register liveness over the
|
||||||
`stack-imbalance`, `abi0-register-args` and `unencodable-instruction`.
|
control-flow graph), `stack-imbalance`, `abi0-register-args` and
|
||||||
|
`unencodable-instruction`.
|
||||||
- **Standalone assembler.** `gasm asm` encodes all four architectures without
|
- **Standalone assembler.** `gasm asm` encodes all four architectures without
|
||||||
the Go toolchain and writes raw images, linkable ELF objects (with DWARF5
|
the Go toolchain and writes raw images or linkable ELF objects (with DWARF5
|
||||||
debug sections) or the Go toolchain's own GOOBJ format, which `go build`
|
debug sections) with no Go installation needed, or the Go toolchain's own
|
||||||
|
GOOBJ format, which needs the installed toolchain and which `go build`
|
||||||
consumes in place of the toolchain's output. Framed functions get the
|
consumes in place of the toolchain's output. Framed functions get the
|
||||||
stack-split guard and the morestack block, byte-identical to the
|
stack-split guard and the morestack block, byte-identical to the
|
||||||
toolchain's, so split functions link too.
|
toolchain's, so split functions link too.
|
||||||
@@ -36,7 +90,8 @@ developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
|
|||||||
byte-for-byte ground-truth comparison of the machine code.
|
byte-for-byte ground-truth comparison of the machine code.
|
||||||
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
|
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
|
||||||
breakpoints (optionally conditional), hardware watchpoints, register and
|
breakpoints (optionally conditional), hardware watchpoints, register and
|
||||||
memory inspection, and headless script runs with label-level coverage.
|
memory inspection, and headless script runs that report instruction and
|
||||||
|
label coverage.
|
||||||
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
|
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
|
||||||
push and pull diagnostics, semantic-token highlighting, go-to-definition,
|
push and pull diagnostics, semantic-token highlighting, go-to-definition,
|
||||||
find references, rename, formatting, inlay hints, code actions, signature
|
find references, rename, formatting, inlay hints, code actions, signature
|
||||||
@@ -47,12 +102,11 @@ developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
|
|||||||
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
|
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
|
||||||
`gasm audit-instructions` diffs the encoder against the installed toolchain,
|
`gasm audit-instructions` diffs the encoder against the installed toolchain,
|
||||||
and `gasm scaffold` generates a differential test skeleton for a kernel.
|
and `gasm scaffold` generates a differential test skeleton for a kernel.
|
||||||
- **Complete instruction coverage.** The instruction tables are generated
|
|
||||||
from the Go toolchain's own assembler source, so the toolkit recognises
|
|
||||||
every mnemonic the real assembler accepts; `just gen` refreshes them.
|
|
||||||
|
|
||||||
### Architecture support
|
### Architecture support
|
||||||
|
|
||||||
|
Four architectures, the four that matter in practice:
|
||||||
|
|
||||||
| Architecture | GOARCH | File suffix | Instructions recognised |
|
| Architecture | GOARCH | File suffix | Instructions recognised |
|
||||||
|--------------|-------------|--------------|---------------------------------------------|
|
|--------------|-------------|--------------|---------------------------------------------|
|
||||||
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
|
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
|
||||||
@@ -63,27 +117,94 @@ developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
|
|||||||
"Common opcodes" are the instructions shared by every architecture (`RET`,
|
"Common opcodes" are the instructions shared by every architecture (`RET`,
|
||||||
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
|
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
|
||||||
carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
|
carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
|
||||||
...) that the assembler accepts as aliases. Regenerating the tables is one
|
...) that the assembler accepts as aliases. The tables are generated from
|
||||||
command (`just gen`) and requires only a Go installation; the committed output
|
the Go toolchain's own assembler source (`just gen` refreshes them), so
|
||||||
has no runtime dependency on the toolchain.
|
every mnemonic the real assembler accepts is recognised; what the encoder
|
||||||
|
can emit today is narrower, and a recognised but unencodable instruction is
|
||||||
|
reported as an explicit error, never as a wrong byte.
|
||||||
|
|
||||||
|
The same measurement runs over GOROOT's whole assembly corpus:
|
||||||
|
`gasm audit-instructions --corpus` reports 136 of 433 attemptable files
|
||||||
|
(31.4 %) assembling for every target architecture today (files named for
|
||||||
|
other Go ports are counted but never attempted), with the top failure
|
||||||
|
reasons per architecture; the number moves with every release.
|
||||||
|
|
||||||
|
### Validation status
|
||||||
|
|
||||||
|
**Only amd64 is validated on real hardware.** The other three
|
||||||
|
architectures are validated under qemu-user emulation, because the
|
||||||
|
project owns no arm64, riscv64 or loong64 machine, and emulation is the
|
||||||
|
only substitute available for the hardware. The distinction matters and
|
||||||
|
is stated rather than implied: everything below is a claim about what has
|
||||||
|
actually been executed.
|
||||||
|
|
||||||
|
| Layer | amd64 | arm64, riscv64, loong64 |
|
||||||
|
|---|---|---|
|
||||||
|
| Encoding: byte-for-byte against `go tool asm` | native hardware | native hardware (the toolchain cross-assembles any GOARCH on any host) |
|
||||||
|
| Execution: JIT calls, ABI checks, differential fuzzing | native hardware | qemu-user emulation |
|
||||||
|
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more |
|
||||||
|
|
||||||
|
Consequences, stated plainly. An emulator is a model of a CPU, not the
|
||||||
|
CPU: instruction semantics are implemented in software and can differ
|
||||||
|
from silicon in ways a test suite does not reveal. A kernel that passes
|
||||||
|
under qemu-user is therefore not proven correct on real hardware, and a
|
||||||
|
discrepancy found on real hardware is a defect in gasm, reported like any
|
||||||
|
other. Encoding parity is the exception: the byte comparison against the
|
||||||
|
toolchain runs on the host for every architecture, so no emulator stands
|
||||||
|
between the claim and the evidence. The debugger is the weakest case: on
|
||||||
|
the three emulated architectures its per-architecture ptrace code has
|
||||||
|
been compiled and read, never executed. Its architecture-neutral units
|
||||||
|
run under `go test ./...`, which the race workflow and a manual run
|
||||||
|
perform; the default `just test` gate does not sweep `./debug/...`.
|
||||||
|
|
||||||
|
## Direction
|
||||||
|
|
||||||
|
The plan, in the order it is being worked:
|
||||||
|
|
||||||
|
- **Extended instruction support.** Two layers. First, encoding
|
||||||
|
coverage for every mnemonic the Go toolchain itself accepts, closed in
|
||||||
|
order of how often real code needs each instruction;
|
||||||
|
`gasm audit-instructions` measures the gap. Second, the larger work:
|
||||||
|
an extended instruction set the toolchain does not know at all. The
|
||||||
|
toolchain-derived tables stay generated and untouched; only the
|
||||||
|
extended instructions are hand-maintained, with their own spellings
|
||||||
|
and encoders, verified by execution (on real hardware for amd64, under
|
||||||
|
emulation for the rest, per the validation status above) because the
|
||||||
|
toolchain offers no ground truth to compare against. The gaps exist
|
||||||
|
on every architecture, amd64 included.
|
||||||
|
- **Full GOOBJ and ELF compilation.** The destination is a complete,
|
||||||
|
standalone compilation path: linkable ELF objects for consumers outside
|
||||||
|
Go, and GOOBJ objects that `go build` links directly. Through GOOBJ, a
|
||||||
|
Go program will be able to use machine instructions that the Go
|
||||||
|
toolchain itself does not support; through ELF, Plan 9 assembly becomes
|
||||||
|
usable outside Go entirely.
|
||||||
|
- **Platforms: Linux and FreeBSD.** Linux is supported today on all four
|
||||||
|
architectures and is where the binary builds. FreeBSD follows: the
|
||||||
|
JIT's executable-memory mapping and the ptrace debugger layer are the
|
||||||
|
two pieces of porting work. Other unix systems may follow those two.
|
||||||
|
- **Four architectures, no more.** amd64, arm64, riscv64 and loong64.
|
||||||
|
No others are planned.
|
||||||
|
|
||||||
## Install
|
## Install
|
||||||
|
|
||||||
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
|
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
|
||||||
linux/loong64 are on the
|
linux/loong64 are on the
|
||||||
[releases page](https://sourcedock.dev/petrbalvin/gasm-devkit/releases).
|
[releases page](https://sourcedock.dev/petrbalvin/gasm-devkit/releases).
|
||||||
From source (Go 1.27 or later):
|
From source (Go 1.27.1):
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
go install sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm@latest
|
go install sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm@latest
|
||||||
```
|
```
|
||||||
|
|
||||||
Or from a repository checkout, with the development version stamped:
|
Or from a repository checkout:
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
just install-bin
|
just install
|
||||||
```
|
```
|
||||||
|
|
||||||
|
The installed binary reports the version the toolchain recorded: the tag
|
||||||
|
on a tagged checkout, a pseudo-version naming the commit below one.
|
||||||
|
|
||||||
## Quick start
|
## Quick start
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
@@ -119,7 +240,7 @@ gasm verify --ground-truth k.s # byte-for-byte vs go tool asm
|
|||||||
gasm verify --fuzz k.s # differential fuzz vs the go tool asm build
|
gasm verify --fuzz k.s # differential fuzz vs the go tool asm build
|
||||||
gasm debug --func name k.s # interactive debugger
|
gasm debug --func name k.s # interactive debugger
|
||||||
gasm debug --func name --script cmds.txt --timeout 30s k.s # headless run
|
gasm debug --func name --script cmds.txt --timeout 30s k.s # headless run
|
||||||
gasm debug --func name --cover k.s # which labels did execution reach?
|
gasm debug --func name --cover k.s # instruction and label coverage
|
||||||
gasm diff a.s b.s # compare machine code byte-for-byte
|
gasm diff a.s b.s # compare machine code byte-for-byte
|
||||||
gasm diff --map wideCopyAVX2=wideCopyAVX512 avx2.s avx512.s
|
gasm diff --map wideCopyAVX2=wideCopyAVX512 avx2.s avx512.s
|
||||||
gasm profile k.s # show basic-block structure
|
gasm profile k.s # show basic-block structure
|
||||||
@@ -142,9 +263,9 @@ infers the target architecture from the file-name suffix
|
|||||||
## Development
|
## Development
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
just install # download module dependencies
|
just build # compile, zero errors and zero warnings
|
||||||
just build # go vet + gofmt check, zero errors and zero warnings
|
just test # the suite, no cache, the 80 % coverage floor
|
||||||
just test # full suite, race detector, 80 % coverage gate
|
just gates # build, fmt-check, vet, test, race: the definition of done
|
||||||
just fmt # gofmt the tree
|
just fmt # gofmt the tree
|
||||||
just gen # regenerate the instruction tables from the Go toolchain
|
just gen # regenerate the instruction tables from the Go toolchain
|
||||||
```
|
```
|
||||||
@@ -155,14 +276,17 @@ recipe.
|
|||||||
|
|
||||||
## Documentation
|
## Documentation
|
||||||
|
|
||||||
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
|
|
||||||
- [docs/CLI.md](docs/CLI.md): full command reference
|
- [docs/CLI.md](docs/CLI.md): full command reference
|
||||||
|
- man pages: `just install-man` installs gasm(1) and one page per command
|
||||||
|
except `version`, which is documented inside gasm(1) instead, into
|
||||||
|
~/.local/share/man (MANDIR overrides); `just uninstall-man` removes
|
||||||
|
them
|
||||||
|
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
|
||||||
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
|
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
|
||||||
- [docs/DECISIONS.md](docs/DECISIONS.md): deferred design decisions
|
|
||||||
- [CHANGELOG.md](CHANGELOG.md): release history
|
- [CHANGELOG.md](CHANGELOG.md): release history
|
||||||
|
|
||||||
## Licence
|
## Licence
|
||||||
|
|
||||||
BSD-3-Clause — see [LICENSE](LICENSE).
|
BSD-3-Clause; see [LICENSE](LICENSE).
|
||||||
|
|
||||||
Copyright © 2026 [Petr Balvín](https://petrbalvin.org)
|
Copyright © 2026 [Petr Balvín](https://petrbalvin.org)
|
||||||
|
|||||||
+41
@@ -0,0 +1,41 @@
|
|||||||
|
# Security policy
|
||||||
|
|
||||||
|
## Supported versions
|
||||||
|
|
||||||
|
Security fixes go to the newest release and to the `development` branch. Older
|
||||||
|
releases do not receive them.
|
||||||
|
|
||||||
|
| Version | Supported |
|
||||||
|
|---|---|
|
||||||
|
| 0.34.0 | yes |
|
||||||
|
| older releases | no |
|
||||||
|
|
||||||
|
## Reporting a vulnerability
|
||||||
|
|
||||||
|
**Do not open a public issue for a security problem.** A public report tells everyone
|
||||||
|
about the flaw before there is a fix. Report it privately to
|
||||||
|
**opensource@petrbalvin.org**.
|
||||||
|
|
||||||
|
Include:
|
||||||
|
|
||||||
|
- the version or commit you tested, and the platform
|
||||||
|
- what the problem is, and what an attacker gains from it
|
||||||
|
- the smallest reproducer you have, ideally a test or a single command
|
||||||
|
- a suggested fix, if you have one
|
||||||
|
|
||||||
|
## What to expect
|
||||||
|
|
||||||
|
- A human reads the report, and you get an acknowledgement.
|
||||||
|
- You are kept informed while the fix is being made, and told when it ships.
|
||||||
|
- The fix is released before the details are published, and the timing is agreed with
|
||||||
|
you.
|
||||||
|
- The fix ships without naming you: the project keeps no credits list, so the release
|
||||||
|
notes, the changelog and the commits name no reporter.
|
||||||
|
|
||||||
|
## Out of scope
|
||||||
|
|
||||||
|
- Findings that require the attacker to already run code as the user, or to have local
|
||||||
|
access.
|
||||||
|
- Missing hardening with no demonstrated impact.
|
||||||
|
- Flaws in a third-party dependency: report them to that project, and to this one only
|
||||||
|
when this project's use of it makes them reachable.
|
||||||
@@ -70,6 +70,10 @@ func amd64Registers() []Register {
|
|||||||
for i := 0; i <= 7; i++ {
|
for i := 0; i <= 7; i++ {
|
||||||
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
|
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
|
||||||
}
|
}
|
||||||
|
// x87 stack registers (FMOVD and the other x87 moves).
|
||||||
|
for i := 0; i <= 7; i++ {
|
||||||
|
add(fmt.Sprintf("F%d", i), Float, "x87 stack register")
|
||||||
|
}
|
||||||
return regs
|
return regs
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -55,6 +55,7 @@ const (
|
|||||||
Mask // AVX-512 mask register (K)
|
Mask // AVX-512 mask register (K)
|
||||||
Float // arm64 floating-point register (F)
|
Float // arm64 floating-point register (F)
|
||||||
VecARM // arm64 SIMD/vector register (V)
|
VecARM // arm64 SIMD/vector register (V)
|
||||||
|
VecSIMD // architecture-neutral SIMD/vector register (LoongArch LSX/LASX)
|
||||||
Special // architecture-special register
|
Special // architecture-special register
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -73,6 +74,8 @@ func (c RegClass) String() string {
|
|||||||
return "float"
|
return "float"
|
||||||
case VecARM:
|
case VecARM:
|
||||||
return "vector (arm64)"
|
return "vector (arm64)"
|
||||||
|
case VecSIMD:
|
||||||
|
return "vector"
|
||||||
case Special:
|
case Special:
|
||||||
return "special"
|
return "special"
|
||||||
default:
|
default:
|
||||||
|
|||||||
+28
-1
@@ -145,11 +145,38 @@ func arm64Curated() []Instr {
|
|||||||
for _, op := range []string{
|
for _, op := range []string{
|
||||||
"LDAXR", "LDAXRB", "LDAXRH", "LDAXRW", "STXR", "STXRB", "STXRH", "STXRW",
|
"LDAXR", "LDAXRB", "LDAXRH", "LDAXRW", "STXR", "STXRB", "STXRH", "STXRW",
|
||||||
"LDAR", "LDARB", "LDARH", "LDARW", "STLR", "STLRB", "STLRH", "STLRW",
|
"LDAR", "LDARB", "LDARH", "LDARW", "STLR", "STLRB", "STLRH", "STLRW",
|
||||||
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL", "CASAL",
|
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL",
|
||||||
} {
|
} {
|
||||||
t = append(t, i(op, "Atomic memory operation"))
|
t = append(t, i(op, "Atomic memory operation"))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Register-pair loads and stores.
|
||||||
|
for _, op := range []string{"LDP", "STP", "LDPW", "STPW", "FLDPD", "FSTPD"} {
|
||||||
|
t = append(t, ic(op, "Register-pair load or store", 2, 2))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Cache maintenance and prefetch.
|
||||||
|
t = append(t, i("DC", "Data cache maintenance"))
|
||||||
|
t = append(t, i("PRFM", "Memory prefetch"))
|
||||||
|
for _, op := range []string{"LDADDAL", "LDCLRAL", "LDORAL", "SWPAL"} {
|
||||||
|
t = append(t, i(op, "Atomic memory operation with acquire and release semantics"))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Cryptographic extensions.
|
||||||
|
for _, op := range []string{"AESE", "AESD", "AESMC", "AESIMC"} {
|
||||||
|
t = append(t, i(op, "AES round"))
|
||||||
|
}
|
||||||
|
for _, op := range []string{
|
||||||
|
"SHA1C", "SHA1P", "SHA1M", "SHA1H", "SHA1SU0", "SHA1SU1",
|
||||||
|
"SHA256H", "SHA256H2", "SHA256SU0", "SHA256SU1",
|
||||||
|
"SHA512H", "SHA512H2", "SHA512SU0", "SHA512SU1",
|
||||||
|
} {
|
||||||
|
t = append(t, i(op, "SHA round"))
|
||||||
|
}
|
||||||
|
for _, op := range []string{"VEOR3", "VBCAX", "VXAR", "VRAX1"} {
|
||||||
|
t = append(t, i(op, "Three-way XOR / rotate crypto vector operation"))
|
||||||
|
}
|
||||||
|
|
||||||
// Floating-point scalar.
|
// Floating-point scalar.
|
||||||
for _, op := range []string{
|
for _, op := range []string{
|
||||||
"FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX",
|
"FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX",
|
||||||
|
|||||||
+2
-2
@@ -31,10 +31,10 @@ func loong64Registers() []Register {
|
|||||||
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
|
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
|
||||||
}
|
}
|
||||||
for i := 0; i <= 31; i++ {
|
for i := 0; i <= 31; i++ {
|
||||||
add(fmt.Sprintf("V%d", i), VecARM, "LSX 128-bit vector register")
|
add(fmt.Sprintf("V%d", i), VecSIMD, "LSX 128-bit vector register")
|
||||||
}
|
}
|
||||||
for i := 0; i <= 31; i++ {
|
for i := 0; i <= 31; i++ {
|
||||||
add(fmt.Sprintf("X%d", i), VecARM, "LASX 256-bit vector register")
|
add(fmt.Sprintf("X%d", i), VecSIMD, "LASX 256-bit vector register")
|
||||||
}
|
}
|
||||||
return regs
|
return regs
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,6 +4,7 @@
|
|||||||
package asm
|
package asm
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"encoding/binary"
|
||||||
"os"
|
"os"
|
||||||
"os/exec"
|
"os/exec"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
@@ -61,6 +62,62 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestGOObjectAARCH64PairReloc pins the ADRP-pair relocation shape against
|
||||||
|
// the toolchain's own object for the same source: exactly one R_ADDRARM64
|
||||||
|
// of Siz 8 at the ADRP word (cmd/internal/obj/arm64/asm7.go adds a single
|
||||||
|
// Siz-8 relocation per pair and the linker patches both instructions from
|
||||||
|
// it). gasm's assembler records the ADRP+ADD form as two word relocs; the
|
||||||
|
// emitter must coalesce them, not emit two Siz-4 records.
|
||||||
|
func TestGOObjectAARCH64PairReloc(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("gv_arm64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·getv(SB), NOSPLIT, $0-8
|
||||||
|
MOVD $v<>(SB), R4
|
||||||
|
MOVD R4, ret+0(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
GLOBL v<>(SB), RODATA, $8
|
||||||
|
DATA v<>+0(SB)/8, $7
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.GOObjectAARCH64("main", "gv_arm64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObjectAARCH64: %v", err)
|
||||||
|
}
|
||||||
|
v := openGoobj(t, obj)
|
||||||
|
relocs := v.blk(blkReloc)
|
||||||
|
le := binary.LittleEndian
|
||||||
|
// Two DWARF relocs on the lines/DIE symbols, then the code's one pair
|
||||||
|
// relocation.
|
||||||
|
if len(relocs) != 3*23 {
|
||||||
|
t.Fatalf("relocs = %d bytes, want three entries", len(relocs))
|
||||||
|
}
|
||||||
|
cr := relocs[2*23:]
|
||||||
|
if off := int32(le.Uint32(cr[0:])); off != 0 {
|
||||||
|
t.Errorf("pair reloc off = %d, want 0 (the ADRP word)", off)
|
||||||
|
}
|
||||||
|
if siz := cr[4]; siz != 8 {
|
||||||
|
t.Errorf("pair reloc siz = %d, want 8", siz)
|
||||||
|
}
|
||||||
|
if typ := le.Uint16(cr[5:]); typ != relocArm64Addr {
|
||||||
|
t.Errorf("pair reloc type = %d, want %d (R_ADDRARM64)", typ, relocArm64Addr)
|
||||||
|
}
|
||||||
|
if pkg := le.Uint32(cr[15:]); pkg != pkgIdxSelf {
|
||||||
|
t.Errorf("pair reloc PkgIdx = %#x, want pkgIdxSelf", pkg)
|
||||||
|
}
|
||||||
|
// The GLOBL is the first package definition.
|
||||||
|
if sym := le.Uint32(cr[19:]); sym != 0 {
|
||||||
|
t.Errorf("pair reloc SymIdx = %d, want 0 (the GLOBL definition)", sym)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestGOObjectAARCH64Link does an end-to-end link test: it cross-compiles a
|
// TestGOObjectAARCH64Link does an end-to-end link test: it cross-compiles a
|
||||||
// Go program for arm64, substitutes the gasm-produced object into the package
|
// Go program for arm64, substitutes the gasm-produced object into the package
|
||||||
// archive, re-links with cmd/link, and verifies the symbol appears in the
|
// archive, re-links with cmd/link, and verifies the symbol appears in the
|
||||||
@@ -79,6 +136,14 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
ADD R5, R4, R4
|
ADD R5, R4, R4
|
||||||
MOVD R4, ret+16(FP)
|
MOVD R4, ret+16(FP)
|
||||||
RET
|
RET
|
||||||
|
|
||||||
|
TEXT ·getv(SB), NOSPLIT, $0-8
|
||||||
|
MOVD $v<>(SB), R4
|
||||||
|
MOVD R4, ret+0(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
GLOBL v<>(SB), RODATA, $8
|
||||||
|
DATA v<>+0(SB)/8, $7
|
||||||
`
|
`
|
||||||
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
|
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
@@ -86,11 +151,15 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
mainSrc := `package main
|
mainSrc := `package main
|
||||||
|
|
||||||
func add(a, b int64) int64
|
func add(a, b int64) int64
|
||||||
|
func getv() *int64
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
if add(20, 22) != 42 {
|
if add(20, 22) != 42 {
|
||||||
panic("bad add")
|
panic("bad add")
|
||||||
}
|
}
|
||||||
|
if getv() == nil {
|
||||||
|
panic("bad getv")
|
||||||
|
}
|
||||||
}
|
}
|
||||||
`
|
`
|
||||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||||
|
|||||||
+1440
-118
File diff suppressed because it is too large
Load Diff
+499
-76
@@ -27,6 +27,14 @@ package asm
|
|||||||
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
|
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
|
||||||
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
|
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
|
||||||
|
|
||||||
|
import (
|
||||||
|
"maps"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
|
)
|
||||||
|
|
||||||
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
|
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
|
||||||
// R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
|
// R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
|
||||||
// runtime's assembly uses. Returns -1 for an unrecognised name.
|
// runtime's assembly uses. Returns -1 for an unrecognised name.
|
||||||
@@ -96,8 +104,12 @@ func arm64RegNum(name string) int {
|
|||||||
return 30
|
return 30
|
||||||
case "R31", "ZR":
|
case "R31", "ZR":
|
||||||
return 31
|
return 31
|
||||||
case "SP":
|
case "SP", "RSP":
|
||||||
return 31 // SP and ZR share encoding 31; context determines meaning
|
// RSP is the toolchain's spelling for register 31 (it rejects
|
||||||
|
// R31 in an operand); SP stays for sources that spell it the
|
||||||
|
// amd64 way. SP and ZR share encoding 31; context determines
|
||||||
|
// the meaning.
|
||||||
|
return 31
|
||||||
}
|
}
|
||||||
// F0-F31.
|
// F0-F31.
|
||||||
if len(name) >= 1 && name[0] == 'F' {
|
if len(name) >= 1 && name[0] == 'F' {
|
||||||
@@ -262,19 +274,15 @@ type a64Format uint8
|
|||||||
|
|
||||||
const (
|
const (
|
||||||
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
|
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
|
||||||
a64FDPIR // data-processing (immediate): ADD/SUB $imm
|
|
||||||
a64FLogImm // logical (immediate): AND/ORR/EOR $imm
|
|
||||||
a64FMovWide // move wide: MOVZ, MOVN, MOVK
|
a64FMovWide // move wide: MOVZ, MOVN, MOVK
|
||||||
a64FLSU // load/store (unsigned immediate, scaled)
|
|
||||||
a64FLSUnscaled // load/store (unscaled immediate)
|
|
||||||
a64FLSPair // load/store pair
|
|
||||||
a64FBranch // unconditional branch (B/BL)
|
a64FBranch // unconditional branch (B/BL)
|
||||||
a64FBranchCond // conditional branch (B.cond)
|
a64FBranchCond // conditional branch (B.cond)
|
||||||
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
|
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
|
||||||
a64FADR // ADR/ADRP
|
a64FADR // ADR/ADRP
|
||||||
a64FEXTR // EXTR
|
a64FEXTR // EXTR
|
||||||
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
||||||
a64FSystem // system: NOP, BRK, etc.
|
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
|
||||||
|
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
|
||||||
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
||||||
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
|
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
|
||||||
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
|
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
|
||||||
@@ -282,12 +290,29 @@ const (
|
|||||||
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
|
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
|
||||||
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
|
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
|
||||||
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
||||||
a64FFMovGR // FMOV between GP and FP registers
|
|
||||||
a64FCRC32 // CRC32
|
a64FCRC32 // CRC32
|
||||||
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
||||||
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR
|
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
|
||||||
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
||||||
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
|
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
|
||||||
|
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
|
||||||
|
a64FCondCmp // conditional compare: CCMP, CCMN
|
||||||
|
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
|
||||||
|
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
|
||||||
|
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
|
||||||
|
a64FAcqRel // acquire/release: LDAR family, STLR family
|
||||||
|
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
|
||||||
|
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
|
||||||
|
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
|
||||||
|
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
|
||||||
|
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
|
||||||
|
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
|
||||||
|
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
|
||||||
|
a64FVTBL // SIMD table lookup: VTBL
|
||||||
|
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
|
||||||
|
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
|
||||||
|
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
|
||||||
|
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
|
||||||
)
|
)
|
||||||
|
|
||||||
// a64Enc is one instruction's encoding: its bit layout (format) and the
|
// a64Enc is one instruction's encoding: its bit layout (format) and the
|
||||||
@@ -332,30 +357,14 @@ func init() {
|
|||||||
"ANDSW": 0<<31 | 3<<29 | 0x0a<<24,
|
"ANDSW": 0<<31 | 3<<29 | 0x0a<<24,
|
||||||
"BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21,
|
"BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21,
|
||||||
"BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21,
|
"BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21,
|
||||||
// Shift
|
// Divide (data-processing 2 source): the opcode occupies bits 15:10
|
||||||
"LSL": 1<<31 | 0<<29 | 0x0a<<24, // alias of UBFM
|
// of the 0xd6<<21 fixed field, UDIV=0b0010 and SDIV=0b0011 (ARM ARM
|
||||||
"LSLW": 0<<31 | 0<<29 | 0x0a<<24,
|
// "Data-processing (2 source)"; the toolchain spells them OPDP2(2)
|
||||||
"LSR": 1<<31 | 0<<29 | 0x0a<<24,
|
// and OPDP2(3)). sf=1 selects the X forms.
|
||||||
"LSRW": 0<<31 | 0<<29 | 0x0a<<24,
|
"SDIV": 1<<31 | 0xd6<<21 | 3<<10,
|
||||||
"ASR": 1<<31 | 0<<29 | 0x0a<<24,
|
"SDIVW": 0<<31 | 0xd6<<21 | 3<<10,
|
||||||
"ASRW": 0<<31 | 0<<29 | 0x0a<<24,
|
"UDIV": 1<<31 | 0xd6<<21 | 2<<10,
|
||||||
"ROR": 1<<31 | 0<<29 | 0x0a<<24,
|
"UDIVW": 0<<31 | 0xd6<<21 | 2<<10,
|
||||||
"RORW": 0<<31 | 0<<29 | 0x0a<<24,
|
|
||||||
// Multiply
|
|
||||||
"MADD": 1<<31 | 0<<29 | 0x1b<<24 | 0<<21,
|
|
||||||
"MADDW": 0<<31 | 0<<29 | 0x1b<<24 | 0<<21,
|
|
||||||
"MSUB": 1<<31 | 0<<29 | 0x1b<<24 | 1<<21,
|
|
||||||
"MSUBW": 0<<31 | 0<<29 | 0x1b<<24 | 1<<21,
|
|
||||||
// Divide
|
|
||||||
"SDIV": 1<<31 | 0<<29 | 0x0d<<24,
|
|
||||||
"SDIVW": 0<<31 | 0<<29 | 0x0d<<24,
|
|
||||||
"UDIV": 1<<31 | 0<<29 | 0x0d<<24 | 1<<10,
|
|
||||||
"UDIVW": 0<<31 | 0<<29 | 0x0d<<24 | 1<<10,
|
|
||||||
// CRC
|
|
||||||
"CRC32B": 0<<31 | 0<<29 | 0x1b<<24 | 4<<10,
|
|
||||||
"CRC32H": 0<<31 | 0<<29 | 0x1b<<24 | 5<<10,
|
|
||||||
"CRC32W": 0<<31 | 0<<29 | 0x1b<<24 | 6<<10,
|
|
||||||
"CRC32X": 1<<31 | 0<<29 | 0x1b<<24 | 7<<10,
|
|
||||||
// Conditional select
|
// Conditional select
|
||||||
"CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10,
|
"CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10,
|
||||||
"CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10,
|
"CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10,
|
||||||
@@ -385,14 +394,37 @@ func init() {
|
|||||||
a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]}
|
a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]}
|
||||||
a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]}
|
a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]}
|
||||||
|
|
||||||
// ---- data-processing (immediate) ----
|
// ---- shifts ----
|
||||||
// ADD/SUB $imm, Rn, Rd
|
// The mnemonic serves both forms: with an immediate the aliases of the
|
||||||
a64InstrTable["ADDImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 0<<29 | 0x11<<24}
|
// data-processing (immediate) group apply (ARM ARM "Shifts"), with a
|
||||||
a64InstrTable["ADDWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 0<<30 | 0<<29 | 0x11<<24}
|
// register the data-processing (2 source) LSLV/LSRV/ASRV/RORV. The op
|
||||||
a64InstrTable["SUBImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 0<<29 | 0x11<<24}
|
// field carries the immediate-alias base; encodeARM64Shift derives both
|
||||||
a64InstrTable["SUBWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 1<<30 | 0<<29 | 0x11<<24}
|
// it and the two-source opcode. Identities, W = 64 (X) or 32 (W):
|
||||||
a64InstrTable["ADDSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 1<<29 | 0x11<<24}
|
//
|
||||||
a64InstrTable["SUBSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 1<<29 | 0x11<<24}
|
// LSL $sh, Rn, Rd = UBFM Rd, Rn, #(-sh) mod W, #(W-1)-sh
|
||||||
|
// LSR $sh, Rn, Rd = UBFM Rd, Rn, #sh, #(W-1)
|
||||||
|
// ASR $sh, Rn, Rd = SBFM Rd, Rn, #sh, #(W-1)
|
||||||
|
// ROR $sh, Rn, Rd = EXTR Rd, Rn, Rn, #sh
|
||||||
|
shifts := map[string]a64Enc{
|
||||||
|
"LSL": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X
|
||||||
|
"LSLW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W
|
||||||
|
"LSR": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X
|
||||||
|
"LSRW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W
|
||||||
|
"ASR": {format: a64FShift, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}, // SBFM X
|
||||||
|
"ASRW": {format: a64FShift, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}, // SBFM W
|
||||||
|
"ROR": {format: a64FShift, op: 1<<31 | 0x27<<23 | 1<<22}, // EXTR X
|
||||||
|
"RORW": {format: a64FShift, op: 0<<31 | 0x27<<23 | 0<<22}, // EXTR W
|
||||||
|
}
|
||||||
|
maps.Copy(a64InstrTable, shifts)
|
||||||
|
|
||||||
|
// ---- multiply accumulate ----
|
||||||
|
// MADD/MSUB Rm, Ra, Rn, Rd: sf 00 11011 o0(15) Rm Ra Rn Rd. The
|
||||||
|
// toolchain's optab has no shorter row, so all four operands are
|
||||||
|
// mandatory, and Ra is the SECOND operand.
|
||||||
|
a64InstrTable["MADD"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24}
|
||||||
|
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
|
||||||
|
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
|
||||||
|
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
|
||||||
|
|
||||||
// ---- move wide ----
|
// ---- move wide ----
|
||||||
// MOVZ/MOVN/MOVK
|
// MOVZ/MOVN/MOVK
|
||||||
@@ -407,22 +439,9 @@ func init() {
|
|||||||
a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0}
|
a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0}
|
||||||
a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1}
|
a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1}
|
||||||
|
|
||||||
// ---- load/store (unsigned immediate) ----
|
// Load/store mnemonics never enter this table: the MOV pseudo-instruction
|
||||||
a64InstrTable["MOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<22} // LDR 64-bit
|
// dispatch handles them through a64LoadTable, which also carries the store
|
||||||
a64InstrTable["MOVWU"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<22} // LDR 32-bit unsigned
|
// opcode (integer and FP stores both use opc=00, differing only in V).
|
||||||
a64InstrTable["MOVHU"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 1<<22} // LDRH unsigned
|
|
||||||
a64InstrTable["MOVBU"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 1<<22} // LDRB unsigned
|
|
||||||
a64InstrTable["MOVW"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 2<<22} // LDRSW (signed 32→64)
|
|
||||||
a64InstrTable["MOVH"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 2<<22} // LDRSH (signed half)
|
|
||||||
a64InstrTable["MOVB"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 2<<22} // LDRSB (signed byte)
|
|
||||||
a64InstrTable["FMOVS"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 32-bit FP
|
|
||||||
a64InstrTable["FMOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 64-bit FP
|
|
||||||
|
|
||||||
// Store opcodes (load ^ (1<<22)):
|
|
||||||
// STR 64-bit: size=3, V=0, opc=00 → 3<<30 | 7<<27 | 0<<22
|
|
||||||
// STR 32-bit: size=2, V=0, opc=00 → 2<<30 | 7<<27 | 0<<22
|
|
||||||
// STRH: size=1, V=0, opc=00 → 1<<30 | 7<<27 | 0<<22
|
|
||||||
// STRB: size=0, V=0, opc=00 → 0<<30 | 7<<27 | 0<<22
|
|
||||||
|
|
||||||
// ---- branches ----
|
// ---- branches ----
|
||||||
a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26}
|
a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26}
|
||||||
@@ -445,10 +464,8 @@ func init() {
|
|||||||
a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21}
|
a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21}
|
||||||
|
|
||||||
// ---- system ----
|
// ---- system ----
|
||||||
a64InstrTable["NOP"] = a64Enc{format: a64FSystem, op: a64NOP}
|
// NOP/NOOP/UNDEF are spelled out in encodeARM64Instr's pseudo switch,
|
||||||
a64InstrTable["NOOP"] = a64Enc{format: a64FSystem, op: a64NOP}
|
// so they carry no table entry; a64NOP and a64BRK are the encoders.
|
||||||
a64InstrTable["BRK"] = a64Enc{format: a64FSystem, op: 0xd4200000}
|
|
||||||
a64InstrTable["UNDEF"] = a64Enc{format: a64FSystem, op: a64BRK(0)}
|
|
||||||
|
|
||||||
// ---- EXTR ----
|
// ---- EXTR ----
|
||||||
a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22}
|
a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22}
|
||||||
@@ -549,8 +566,8 @@ func init() {
|
|||||||
a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op}
|
a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---- FMOV between GP and FP registers ----
|
// FMOV between GP and FP registers needs no table entry: the MOV
|
||||||
a64InstrTable["FMOVGR"] = a64Enc{format: a64FFMovGR, op: 0x1e260000} // placeholder, actual encoding depends on direction
|
// pseudo-instruction dispatches it by operand class (encodeARM64RegMove).
|
||||||
|
|
||||||
// ---- conditional select: CSEL, CSINC, CSINV, CSNEG ----
|
// ---- conditional select: CSEL, CSINC, CSINV, CSNEG ----
|
||||||
csel := map[string]uint32{
|
csel := map[string]uint32{
|
||||||
@@ -586,6 +603,10 @@ func init() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// ---- exclusive load/store ----
|
// ---- exclusive load/store ----
|
||||||
|
// Single-register forms pre-set the unused Rs and Rt2 fields to 31 (the
|
||||||
|
// 0x7c00/0x1f0000 halves of the constants below); the register-pair
|
||||||
|
// forms carry a real Rt2 in bits 14:10, so their opcodes pre-set
|
||||||
|
// neither field.
|
||||||
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
|
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
|
||||||
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
|
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
|
||||||
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
|
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
|
||||||
@@ -594,6 +615,12 @@ func init() {
|
|||||||
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
|
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
|
||||||
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
|
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
|
||||||
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
|
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
|
||||||
|
// Pair loads, LDSTX(sz, 0, l=1, o1=1, o0) in asm7.go: LDXP/ LDXPW have
|
||||||
|
// o0=0, LDAXP/LDAXPW o0=1 (bit 15). Rs (bits 20:16) stays 31.
|
||||||
|
a64InstrTable["LDXP"] = a64Enc{format: a64FExcl, op: 0xc8600000}
|
||||||
|
a64InstrTable["LDXPW"] = a64Enc{format: a64FExcl, op: 0x88600000}
|
||||||
|
a64InstrTable["LDAXP"] = a64Enc{format: a64FExcl, op: 0xc8608000}
|
||||||
|
a64InstrTable["LDAXPW"] = a64Enc{format: a64FExcl, op: 0x88608000}
|
||||||
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
|
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
|
||||||
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
|
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
|
||||||
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
|
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
|
||||||
@@ -602,6 +629,12 @@ func init() {
|
|||||||
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
|
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
|
||||||
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
|
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
|
||||||
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
|
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
|
||||||
|
// Pair stores, LDSTX(sz, 0, l=0, o1=1, o0): STXP/STXPW have o0=0,
|
||||||
|
// STLXP/STLXPW o0=1 (bit 15). Both Rs and Rt2 are real fields.
|
||||||
|
a64InstrTable["STXP"] = a64Enc{format: a64FExcl, op: 0xc8200000}
|
||||||
|
a64InstrTable["STXPW"] = a64Enc{format: a64FExcl, op: 0x88200000}
|
||||||
|
a64InstrTable["STLXP"] = a64Enc{format: a64FExcl, op: 0xc8208000}
|
||||||
|
a64InstrTable["STLXPW"] = a64Enc{format: a64FExcl, op: 0x88208000}
|
||||||
|
|
||||||
// ---- LSE atomics ----
|
// ---- LSE atomics ----
|
||||||
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
|
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
|
||||||
@@ -613,10 +646,403 @@ func init() {
|
|||||||
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
|
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
|
||||||
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
|
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
|
||||||
|
|
||||||
// ---- SIMD basics ----
|
// ---- SIMD: the arrangement-aware tables in this file carry VADD,
|
||||||
a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400}
|
// VSUB, VMUL and every other three-register vector op. ----
|
||||||
a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400}
|
|
||||||
a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00}
|
// ---- data-processing (1 source): sf 10 11010110 opcode 00000 Rn Rd ----
|
||||||
|
dp1 := map[string]uint32{
|
||||||
|
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
|
||||||
|
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
|
||||||
|
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
|
||||||
|
}
|
||||||
|
for m, op := range dp1 {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- bitfield extract: the UBFM/SBFM bases, immediate operands wrap ----
|
||||||
|
a64InstrTable["UBFX"] = a64Enc{format: a64FBitfield2, op: 0xd3400000}
|
||||||
|
a64InstrTable["SBFX"] = a64Enc{format: a64FBitfield2, op: 0x93400000}
|
||||||
|
a64InstrTable["UBFXW"] = a64Enc{format: a64FBitfield2, op: 0x53000000}
|
||||||
|
a64InstrTable["SBFXW"] = a64Enc{format: a64FBitfield2, op: 0x13000000}
|
||||||
|
|
||||||
|
// ---- conditional compare: sf 1 1 101001 0 imm5/Rm cond op2 Rn nzcv ----
|
||||||
|
a64InstrTable["CCMP"] = a64Enc{format: a64FCondCmp, op: 0xfa400000}
|
||||||
|
a64InstrTable["CCMN"] = a64Enc{format: a64FCondCmp, op: 0xba400000}
|
||||||
|
a64InstrTable["CCMPW"] = a64Enc{format: a64FCondCmp, op: 0x7a400000}
|
||||||
|
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
|
||||||
|
|
||||||
|
// ---- system operations ----
|
||||||
|
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "DC", "MRS", "MSR", "PRFM"} {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FSys}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- compare/test and branch ----
|
||||||
|
a64InstrTable["CBZ"] = a64Enc{format: a64FBranch19, op: 0xb4000000}
|
||||||
|
a64InstrTable["CBZW"] = a64Enc{format: a64FBranch19, op: 0x34000000}
|
||||||
|
a64InstrTable["CBNZ"] = a64Enc{format: a64FBranch19, op: 0xb5000000}
|
||||||
|
a64InstrTable["CBNZW"] = a64Enc{format: a64FBranch19, op: 0x35000000}
|
||||||
|
a64InstrTable["TBZ"] = a64Enc{format: a64FTestBranch, op: 0x36000000}
|
||||||
|
a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000}
|
||||||
|
|
||||||
|
// ---- load/store pair (signed offset) ----
|
||||||
|
a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000}
|
||||||
|
a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000}
|
||||||
|
a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000}
|
||||||
|
a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000}
|
||||||
|
a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000}
|
||||||
|
a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000}
|
||||||
|
|
||||||
|
// ---- acquire/release loads and stores ----
|
||||||
|
a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00}
|
||||||
|
a64InstrTable["LDARB"] = a64Enc{format: a64FAcqRel, op: 0x08dffc00}
|
||||||
|
a64InstrTable["LDARH"] = a64Enc{format: a64FAcqRel, op: 0x48dffc00}
|
||||||
|
a64InstrTable["LDARW"] = a64Enc{format: a64FAcqRel, op: 0x88dffc00}
|
||||||
|
a64InstrTable["STLR"] = a64Enc{format: a64FAcqRel, op: 0xc89ffc00}
|
||||||
|
a64InstrTable["STLRB"] = a64Enc{format: a64FAcqRel, op: 0x089ffc00}
|
||||||
|
a64InstrTable["STLRH"] = a64Enc{format: a64FAcqRel, op: 0x489ffc00}
|
||||||
|
a64InstrTable["STLRW"] = a64Enc{format: a64FAcqRel, op: 0x889ffc00}
|
||||||
|
|
||||||
|
// ---- LSE atomics with acquire and release semantics ----
|
||||||
|
// CAS carries a preset fixed op field and a real Rs; the LDADD/LDCLR/
|
||||||
|
// LDOR/SWP families leave Rs free for the returned value.
|
||||||
|
lse := map[string]uint32{
|
||||||
|
"CASALD": 0xc8e0fc00,
|
||||||
|
"CASALW": 0x88e0fc00,
|
||||||
|
"LDADDALD": 0xf8e00000,
|
||||||
|
"LDADDALW": 0xb8e00000,
|
||||||
|
"LDCLRALB": 0x38e01000,
|
||||||
|
"LDCLRALW": 0xb8e01000,
|
||||||
|
"LDCLRALD": 0xf8e01000,
|
||||||
|
"LDORALB": 0x38e03000,
|
||||||
|
"LDORALW": 0xb8e03000,
|
||||||
|
"LDORALD": 0xf8e03000,
|
||||||
|
"SWPALB": 0x38e08000,
|
||||||
|
"SWPALW": 0xb8e08000,
|
||||||
|
"SWPALD": 0xf8e08000,
|
||||||
|
}
|
||||||
|
for m, op := range lse {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- carry-setting/carry-using arithmetic and widening multiply ----
|
||||||
|
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
|
||||||
|
// register preset to ZR (bits 14:10 = 11111).
|
||||||
|
dpsrExtra := map[string]uint32{
|
||||||
|
"ADC": 0x9a000000, "ADCW": 0x1a000000,
|
||||||
|
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
|
||||||
|
"SBC": 0xda000000, "SBCW": 0x5a000000,
|
||||||
|
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
|
||||||
|
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
|
||||||
|
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
|
||||||
|
}
|
||||||
|
for m, op := range dpsrExtra {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- crypto, 2-register (Rn, Rd) and 3-register (Rm, Rn, Rd) forms ----
|
||||||
|
crypto2 := map[string]uint32{
|
||||||
|
"AESD": 0x4e285800, "AESE": 0x4e284800,
|
||||||
|
"AESIMC": 0x4e287800, "AESMC": 0x4e286800,
|
||||||
|
"SHA1H": 0x5e280800, "SHA1SU1": 0x5e281800,
|
||||||
|
"SHA256SU0": 0x5e282800, "SHA512SU0": 0xcec08000,
|
||||||
|
}
|
||||||
|
for m, op := range crypto2 {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FCrypto2, op: op}
|
||||||
|
}
|
||||||
|
crypto3 := map[string]uint32{
|
||||||
|
"SHA1C": 0x5e000000, "SHA1P": 0x5e001000,
|
||||||
|
"SHA1M": 0x5e002000, "SHA1SU0": 0x5e003000,
|
||||||
|
"SHA256H": 0x5e004000, "SHA256H2": 0x5e005000,
|
||||||
|
"SHA256SU1": 0x5e006000, "SHA512H": 0xce608000,
|
||||||
|
"SHA512H2": 0xce608400, "SHA512SU1": 0xce608800,
|
||||||
|
}
|
||||||
|
for m, op := range crypto3 {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FCrypto3, op: op}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- arrangement-aware SIMD, see a64SimdVTable and a64SimdV2Table ----
|
||||||
|
a64InstrTable["VEOR3"] = a64Enc{format: a64FSIMDV4, op: 0xce000000}
|
||||||
|
a64InstrTable["VBCAX"] = a64Enc{format: a64FSIMDV4, op: 0xce200000}
|
||||||
|
a64InstrTable["VXAR"] = a64Enc{format: a64FSIMDV4, op: 0xce800000}
|
||||||
|
a64InstrTable["VEXT"] = a64Enc{format: a64FSIMDV4, op: 0x2e000000}
|
||||||
|
a64InstrTable["VTBL"] = a64Enc{format: a64FVTBL}
|
||||||
|
a64InstrTable["VDUP"] = a64Enc{format: a64FDUP}
|
||||||
|
a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000}
|
||||||
|
a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000}
|
||||||
|
a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000}
|
||||||
|
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
|
||||||
|
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
|
||||||
|
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
|
||||||
|
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
|
||||||
|
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||||
|
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
|
||||||
|
a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||||
|
a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST}
|
||||||
|
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
|
||||||
|
// the set of arrangements it accepts as a bitmask over the a64Arr index and,
|
||||||
|
// for instructions that exist at a single arrangement and carry that
|
||||||
|
// arrangement's bits inside the base already, the fixed flag.
|
||||||
|
type a64SimdVSpec struct {
|
||||||
|
base uint32
|
||||||
|
arrs uint16
|
||||||
|
fixed bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64Arr names the vector arrangements the encoders deal with, indexed by
|
||||||
|
// a64Arr. The source spellings put the element letter first: B8, H4, S2,
|
||||||
|
// D1 and the 128-bit halves B16, H8, S4, D2.
|
||||||
|
const (
|
||||||
|
a64Arr8B = iota
|
||||||
|
a64Arr16B
|
||||||
|
a64Arr4H
|
||||||
|
a64Arr8H
|
||||||
|
a64Arr2S
|
||||||
|
a64Arr4S
|
||||||
|
a64Arr2D
|
||||||
|
a64ArrD1
|
||||||
|
a64ArrQ1
|
||||||
|
a64ArrCount
|
||||||
|
)
|
||||||
|
|
||||||
|
// a64ArrNames maps an arrangement to its source spelling (element letter
|
||||||
|
// first, as the toolchain writes it).
|
||||||
|
var a64ArrNames = [a64ArrCount]string{
|
||||||
|
a64Arr8B: "B8", a64Arr16B: "B16", a64Arr4H: "H4", a64Arr8H: "H8",
|
||||||
|
a64Arr2S: "S2", a64Arr4S: "S4", a64Arr2D: "D2", a64ArrD1: "D1", a64ArrQ1: "Q1",
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64ArrIndex resolves a source spelling to its a64Arr index, -1 when
|
||||||
|
// unknown.
|
||||||
|
func a64ArrIndex(s string) int {
|
||||||
|
for i, n := range a64ArrNames {
|
||||||
|
if n == s {
|
||||||
|
return i
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64ElemLetter reports whether s is a bare element spelling (B, H, S, D, Q)
|
||||||
|
// as it appears in element operands such as V13.S[0].
|
||||||
|
func a64ElemLetter(s string) bool {
|
||||||
|
switch s {
|
||||||
|
case "B", "H", "S", "D", "Q":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64ArrBits carries the fixed bits an arrangement contributes to the
|
||||||
|
// three-same word shape: the element size at bits 23:22 and the 128-bit
|
||||||
|
// flag at bit 30. Bit 29 belongs to the instruction's own base.
|
||||||
|
var a64ArrBits = [a64ArrCount]uint32{
|
||||||
|
a64Arr8B: 0,
|
||||||
|
a64Arr16B: 1 << 30,
|
||||||
|
a64Arr4H: 1 << 22,
|
||||||
|
a64Arr8H: 1<<30 | 1<<22,
|
||||||
|
a64Arr2S: 1 << 23,
|
||||||
|
a64Arr4S: 1<<30 | 1<<23,
|
||||||
|
a64Arr2D: 1<<30 | 1<<23 | 1<<22,
|
||||||
|
a64ArrD1: 1<<23 | 1<<22,
|
||||||
|
a64ArrQ1: 0,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64SimdVTable holds the arrangement-aware three-register SIMD
|
||||||
|
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
|
||||||
|
// word and arrangement bit was read off go tool asm.
|
||||||
|
var a64SimdVTable = map[string]a64SimdVSpec{
|
||||||
|
"VADD": {0x0e208400, 0x7f, false},
|
||||||
|
"VSUB": {0x2e208400, 0x7f, false},
|
||||||
|
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
|
||||||
|
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
|
||||||
|
"VEOR": {0x2e201c00, 0x03, false},
|
||||||
|
"VORR": {0x0ea01c00, 0x03, false},
|
||||||
|
"VADDP": {0x0e20bc00, 0x7f, false},
|
||||||
|
"VZIP1": {0x0e003800, 0x7f, false},
|
||||||
|
"VZIP2": {0x0e007800, 0x7f, false},
|
||||||
|
"VCMEQ": {0x2e208c00, 0x7f, false},
|
||||||
|
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
|
||||||
|
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
|
||||||
|
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
|
||||||
|
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
|
||||||
|
// the register pair spelling ORR Vd, Vn, Vm.
|
||||||
|
var a64SimdV2Table = map[string]a64SimdVSpec{
|
||||||
|
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
|
||||||
|
"VREV64": {0x0e200800, 0x3f, false},
|
||||||
|
"VUADDLV": {0x2e303800, 0x3f, false},
|
||||||
|
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64CryptoArr is the arrangement each crypto instruction's operands must
|
||||||
|
// carry when they spell one at all; a bare V/F spelling is accepted as is.
|
||||||
|
var a64CryptoArr = map[string]int{
|
||||||
|
"AESD": a64Arr16B, "AESE": a64Arr16B, "AESIMC": a64Arr16B, "AESMC": a64Arr16B,
|
||||||
|
"SHA1H": a64Arr4S, "SHA1SU1": a64Arr4S, "SHA256SU0": a64Arr4S, "SHA512SU0": a64Arr2D,
|
||||||
|
"SHA1C": a64Arr4S, "SHA1P": a64Arr4S, "SHA1M": a64Arr4S, "SHA1SU0": a64Arr4S,
|
||||||
|
"SHA256H": a64Arr4S, "SHA256H2": a64Arr4S, "SHA256SU1": a64Arr4S,
|
||||||
|
"SHA512H": a64Arr2D, "SHA512H2": a64Arr2D, "SHA512SU1": a64Arr2D,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64DCOps maps the data-cache maintenance operation names to their fixed
|
||||||
|
// word (the register rides bits 4:0).
|
||||||
|
var a64DCOps = map[string]uint32{
|
||||||
|
"IVAC": 0xd5087620, "ZVA": 0xd50b7420,
|
||||||
|
"CVAC": 0xd50b7a20, "CVAU": 0xd50b7b20, "CIVAC": 0xd50b7e20,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64MRSOps maps the system register names GOROOT reads to their fixed word
|
||||||
|
// (the destination register rides bits 4:0).
|
||||||
|
var a64MRSOps = map[string]uint32{
|
||||||
|
"ELR_EL1": 0xd5384020, "MIDR_EL1": 0xd5380000,
|
||||||
|
"ID_AA64PFR0_EL1": 0xd5380400, "ID_AA64ISAR0_EL1": 0xd5380600,
|
||||||
|
"ID_AA64ISAR1_EL1": 0xd5380620, "CNTFRQ_EL0": 0xd53be000,
|
||||||
|
"CNTPCT_EL0": 0xd53be020, "CNTVCT_EL0": 0xd53be040,
|
||||||
|
"DCZID_EL0": 0xd53b00e0, "DIT": 0xd53b42a0, "ID_AA64ZFR0_EL1": 0xd5380480,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64MSROps maps the system register names GOROOT writes to their fixed
|
||||||
|
// word; the immediate rides CRm at bits 11:8 and Rt is the fixed 11111.
|
||||||
|
var a64MSROps = map[string]uint32{
|
||||||
|
"SPSel": 0xd50040a0, "DAIFSet": 0xd50340c0, "DAIFClr": 0xd50340e0, "DIT": 0xd5034040,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64PRFOps maps the prefetch operation names to their prfop immediate
|
||||||
|
// (word = 0xf9800000 | Rn<<5 | prfop).
|
||||||
|
var a64PRFOps = map[string]int{
|
||||||
|
"PLDL1KEEP": 0x00, "PLDL1STRM": 0x01, "PLDL2KEEP": 0x02, "PLDL2STRM": 0x03,
|
||||||
|
"PLDL3KEEP": 0x04, "PLDL3STRM": 0x05,
|
||||||
|
"PLIL1KEEP": 0x08, "PLIL1STRM": 0x09, "PLIL2KEEP": 0x0a, "PLIL2STRM": 0x0b,
|
||||||
|
"PLIL3KEEP": 0x0c, "PLIL3STRM": 0x0d,
|
||||||
|
"PSTL1KEEP": 0x10, "PSTL1STRM": 0x11, "PSTL2KEEP": 0x12, "PSTL2STRM": 0x13,
|
||||||
|
"PSTL3KEEP": 0x14, "PSTL3STRM": 0x15,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64VLD1Base holds the fixed words of the multi-register structure
|
||||||
|
// accesses, indexed by register count 1..4, before the Q and size bits.
|
||||||
|
// Post-index spellings add 0x9f0000 (post bit and Rm = 11111).
|
||||||
|
var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000}
|
||||||
|
var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000}
|
||||||
|
|
||||||
|
// a64Vec is a parsed vector operand: the register number, the arrangement
|
||||||
|
// ("" when the operand spells none) and, for element forms, the lane index.
|
||||||
|
type a64Vec struct {
|
||||||
|
reg int
|
||||||
|
arr string
|
||||||
|
idx int
|
||||||
|
hasIdx bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64VecReg parses a vector register operand: V0..V31 (F0..F31 as an alias,
|
||||||
|
// the same architectural registers the scalar floating-point spellings use),
|
||||||
|
// optionally with an arrangement suffix such as V0.B16 and, for element
|
||||||
|
// forms, a lane index such as V13.S[0]. It reports ok=false for anything
|
||||||
|
// else, including X/W and R spellings, which the toolchain's vector
|
||||||
|
// operands reject as well.
|
||||||
|
func a64VecReg(name string) (v a64Vec, ok bool) {
|
||||||
|
s := strings.TrimSpace(name)
|
||||||
|
if i := strings.IndexByte(s, '.'); i >= 0 {
|
||||||
|
v.arr = strings.TrimSpace(s[i+1:])
|
||||||
|
s = s[:i]
|
||||||
|
}
|
||||||
|
if v.arr != "" {
|
||||||
|
// Element form: B[3], S[2] and friends.
|
||||||
|
if j := strings.IndexByte(v.arr, '['); j >= 0 {
|
||||||
|
k := strings.LastIndexByte(v.arr, ']')
|
||||||
|
if k < j {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
n, err := strconv.Atoi(strings.TrimSpace(v.arr[j+1 : k]))
|
||||||
|
if err != nil || n < 0 {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
v.idx, v.hasIdx = n, true
|
||||||
|
v.arr = strings.TrimSpace(v.arr[:j])
|
||||||
|
}
|
||||||
|
if a64ArrIndex(v.arr) < 0 && !a64ElemLetter(v.arr) {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(s) < 2 || (s[0] != 'V' && s[0] != 'F') {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
n := 0
|
||||||
|
for i := 1; i < len(s); i++ {
|
||||||
|
if s[i] < '0' || s[i] > '9' {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
n = n*10 + int(s[i]-'0')
|
||||||
|
}
|
||||||
|
if n > 31 {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
v.reg = n
|
||||||
|
return v, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64ElemField encodes a lane index for the copy/insert group: imm5 = the
|
||||||
|
// index shifted by the element scale, with the scale's own bit set. B gets
|
||||||
|
// shift 1 (the Q bit rides elsewhere), H shift 2, S shift 3 and D shift 4.
|
||||||
|
func a64ElemField(arr string, idx int) (uint32, bool) {
|
||||||
|
var shift, low uint32
|
||||||
|
switch arr {
|
||||||
|
case "B8", "B16", "B":
|
||||||
|
shift, low = 1, 1
|
||||||
|
case "H4", "H8", "H":
|
||||||
|
shift, low = 2, 2
|
||||||
|
case "S2", "S4", "S":
|
||||||
|
shift, low = 3, 4
|
||||||
|
case "D1", "D2", "D":
|
||||||
|
shift, low = 4, 8
|
||||||
|
default:
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
if idx < 0 || idx >= 1<<(5-shift) {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
return uint32(idx)<<shift | low, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64VecListOf recovers the register list of a VLD1/VST1/VTBL operand run.
|
||||||
|
// The parser keeps parenthesised groups whole but splits bracketed lists on
|
||||||
|
// the commas, so a list arrives as one operand run whose first Raw starts
|
||||||
|
// with "[" and whose last Raw ends with "]". It returns the parsed
|
||||||
|
// registers with the brackets and spaces removed.
|
||||||
|
func a64VecListOf(ops []*ast.Operand, start int) (vs []a64Vec, end int, ok bool) {
|
||||||
|
if start >= len(ops) || !strings.HasPrefix(strings.TrimSpace(ops[start].Raw), "[") {
|
||||||
|
return nil, 0, false
|
||||||
|
}
|
||||||
|
end = start
|
||||||
|
for end < len(ops) {
|
||||||
|
if strings.HasSuffix(strings.TrimSpace(ops[end].Raw), "]") {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
end++
|
||||||
|
}
|
||||||
|
if end >= len(ops) {
|
||||||
|
return nil, 0, false
|
||||||
|
}
|
||||||
|
for i := start; i <= end; i++ {
|
||||||
|
s := strings.TrimSpace(ops[i].Raw)
|
||||||
|
s = strings.TrimPrefix(s, "[")
|
||||||
|
s = strings.TrimSuffix(s, "]")
|
||||||
|
if s == "" && len(ops) > start+1 {
|
||||||
|
return nil, 0, false
|
||||||
|
}
|
||||||
|
for part := range strings.SplitSeq(s, ",") {
|
||||||
|
v, ok := a64VecReg(part)
|
||||||
|
if !ok {
|
||||||
|
return nil, 0, false
|
||||||
|
}
|
||||||
|
vs = append(vs, v)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return vs, end, true
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---- load/store helper tables ----
|
// ---- load/store helper tables ----
|
||||||
@@ -642,14 +1068,11 @@ var a64LoadTable = map[string]a64LSType{
|
|||||||
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
|
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
|
||||||
}
|
}
|
||||||
|
|
||||||
// a64StoreOpc returns the store opc for a given load type.
|
// a64StoreOpc returns the store opc for a given load type: integer and FP
|
||||||
// For integer: store opc = 00 (the load opc bits cleared).
|
// stores both encode opc=00 (the load's signedness bit sits in opc[1], which
|
||||||
// For FP: store opc = 00 (same pattern).
|
// the store form clears; FP registers are selected by V, not opc).
|
||||||
func a64StoreOpc(t a64LSType) int {
|
func a64StoreOpc(t a64LSType) int {
|
||||||
if t.V == 1 {
|
return 0
|
||||||
return 0 // FP store
|
|
||||||
}
|
|
||||||
return 0 // integer store
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64RegClass discriminates integer (R), floating-point (F) registers for
|
// arm64RegClass discriminates integer (R), floating-point (F) registers for
|
||||||
|
|||||||
+865
-5
@@ -472,12 +472,456 @@ TEXT ·f(SB), NOSPLIT, $0-0
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestArm64SIMD tests SIMD encoding (via the instruction table).
|
// TestArm64SIMD tests SIMD encoding (via the arrangement-aware table).
|
||||||
func TestArm64SIMD(t *testing.T) {
|
func TestArm64SIMD(t *testing.T) {
|
||||||
// Verify SIMD instructions are in the table.
|
// Verify SIMD instructions are in the arrangement table.
|
||||||
for _, mnem := range []string{"VADD", "VSUB", "VMUL"} {
|
for _, mnem := range []string{"VADD", "VSUB", "VMUL", "VAND", "VEOR", "VORR", "VCMEQ", "VZIP1", "VZIP2"} {
|
||||||
if _, ok := a64InstrTable[mnem]; !ok {
|
if _, ok := a64SimdVTable[mnem]; !ok {
|
||||||
t.Errorf("%s not in instruction table", mnem)
|
t.Errorf("%s not in the SIMD arrangement table", mnem)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64CarryAndBitOps pins the carry-setting arithmetic, the widening
|
||||||
|
// multiplies and the data-processing (1 source) group against go tool asm.
|
||||||
|
func TestArm64CarryAndBitOps(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tADC R0, R2, R12\n\tADCS $0, R1\n\tSBCS R5, R9, R5\n\tSBC R25, R10, R26\n"+
|
||||||
|
"\tMUL R4, R3, R0\n\tUMULH R24, R20, R24\n\tSMULH R1, R2, R3\n\tMSUB R19, R16, R26, R2\n"+
|
||||||
|
"\tRBIT R11, R4\n\tREV R1, R2\n\tCLZ R21, R9\n\tREVW R1, R2\n\tCLSW R1, R2\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x9a00004c, // ADC R12, R2, R0
|
||||||
|
0xba1f0021, // ADCS R1, R1, ZR
|
||||||
|
0xfa050125, // SBCS R5, R9, R5
|
||||||
|
0xda19015a, // SBC R26, R10, R25
|
||||||
|
0x9b047c60, // MUL R0, R3, R4
|
||||||
|
0x9bd87e98, // UMULH R24, R20, R24
|
||||||
|
0x9b417c43, // SMULH R3, R2, R1
|
||||||
|
0x9b13c342, // MSUB R2, R26, R19, R16
|
||||||
|
0xdac00164, // RBIT R4, R11
|
||||||
|
0xdac00c22, // REV R2, R1
|
||||||
|
0xdac012a9, // CLZ R9, R21
|
||||||
|
0x5ac00822, // REVW R2, R1
|
||||||
|
0x5ac01422, // CLSW R2, R1
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64BitfieldExtract pins UBFX/SBFX: immr wraps to the register
|
||||||
|
// width, an out-of-range imms is an error.
|
||||||
|
func TestArm64BitfieldExtract(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tUBFX $33, R17, $25, R5\n\tUBFXW $4, R1, $9, R2\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xd361e625, // UBFX immr=1 (33 wrapped), imms=25
|
||||||
|
0x53043022, // UBFXW immr=4, imms=9
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, body := range []string{"\tUBFX $33, R17, $70, R5\n", "\tUBFX $-1, R17, $3, R5\n"} {
|
||||||
|
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFileARM64(f); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", body)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64CondCompare pins CCMP/CCMN.
|
||||||
|
func TestArm64CondCompare(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tCCMP LE, R7, $19, $3\n\tCCMP LT, R30, R6, $7\n\tCCMN EQ, R1, R2, $3\n\tCCMPW LE, R7, $19, $3\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xfa53d8e3, // CCMP imm form
|
||||||
|
0xfa46b3c7, // CCMP register form
|
||||||
|
0xba420023, // CCMN register form
|
||||||
|
0x7a53d8e3, // CCMPW
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64CompareBranch pins CBZ/CBNZ/TBZ/TBNZ against a label five and
|
||||||
|
// six words ahead, matching go tool asm's own offsets.
|
||||||
|
func TestArm64CompareBranch(t *testing.T) {
|
||||||
|
// Layout: CBZ(0) TBZ(4) TBNZ(8) CBNZ(12) NOP(16) NOP(17th word...) done.
|
||||||
|
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
|
||||||
|
"\tCBZ R1, done\n\tTBZ $4, R7, done\n\tTBNZ $33, R7, done\n\tCBNZW R2, done\n" +
|
||||||
|
"\tNOP\n\tNOP\n\tdone:\tNOP\n\tRET\n"
|
||||||
|
f, errs := parser.Parse("test_arm64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
got := leWords(img.Code)
|
||||||
|
// done sits at word 6 from each branch's own pc: CBZ rel 6, TBZ rel 5,
|
||||||
|
// TBNZ rel 4, CBNZW rel 3.
|
||||||
|
want := []uint32{
|
||||||
|
0xb40000c1, // CBZ R1, +6
|
||||||
|
0x362000a7, // TBZ $4, R7, +5
|
||||||
|
0xb7080087, // TBNZ $33, R7, +4
|
||||||
|
0x35000062, // CBNZW R2, +3
|
||||||
|
0xd503201f, 0xd503201f, 0xd503201f,
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64ADR pins ADR against a forward label.
|
||||||
|
func TestArm64ADR(t *testing.T) {
|
||||||
|
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
|
||||||
|
"\tADR done, R10\n\tNOP\n\tNOP\n\tdone:\tNOP\n\tRET\n"
|
||||||
|
f, errs := parser.Parse("test_arm64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
got := leWords(img.Code)
|
||||||
|
// rel = 12 bytes: immlo 0, immhi 3.
|
||||||
|
want := []uint32{0x1000006a, 0xd503201f, 0xd503201f, 0xd503201f, 0xd65f03c0}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64PairLoadStore pins LDP/STP/LDPW/FLDPD/FSTPD.
|
||||||
|
func TestArm64PairLoadStore(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tSTP (R2, R3), 8(R5)\n\tLDP -8(R5), (R2, R3)\n\tLDPW 4(R0), (R1, R2)\n\tSTPW (R1, R2), 4(R0)\n"+
|
||||||
|
"\tFLDPD 8(R0), (F1, F2)\n\tFSTPD (F3, F4), -8(R5)\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xa9008ca2, // STP (R2, R3), 8(R5)
|
||||||
|
0xa97f8ca2, // LDP -8(R5), (R2, R3)
|
||||||
|
0x29408801, // LDPW 4(R0), (R1, R2)
|
||||||
|
0x29008801, // STPW (R1, R2), 4(R0)
|
||||||
|
0x6d408801, // FLDPD 8(R0), (F1, F2)
|
||||||
|
0x6d3f90a3, // FSTPD (F3, F4), -8(R5)
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64AcquireRelease pins LDAR/STLR and the acquire/release LSE
|
||||||
|
// families.
|
||||||
|
func TestArm64AcquireRelease(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tLDAR (R27), R22\n\tLDARB (R25), R2\n\tLDARW (R12), R29\n\tSTLR R3, (R24)\n\tSTLRB R11, (R22)\n"+
|
||||||
|
"\tCASALD R5, (R6), R7\n\tLDADDALD R5, (R6), R7\n\tLDCLRALB R5, (R6), R7\n\tLDORALD R5, (RSP), R7\n\tSWPALW R5, (R6), R7\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xc8dfff76, // LDAR R22, (R27)
|
||||||
|
0x08dfff22, // LDARB R2, (R25)
|
||||||
|
0x88dffd9d, // LDARW R29, (R12)
|
||||||
|
0xc89fff03, // STLR R3, (R24)
|
||||||
|
0x089ffecb, // STLRB R11, (R22)
|
||||||
|
0xc8e5fcc7, // CASALD R7, (R6), R5
|
||||||
|
0xf8e500c7, // LDADDALD R7, (R6), R5
|
||||||
|
0x38e510c7, // LDCLRALB R7, (R6), R5
|
||||||
|
0xf8e533e7, // LDORALD R7, (RSP), R5
|
||||||
|
0xb8e580c7, // SWPALW R7, (R6), R5
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64System pins BRK, SVC, the barriers, cache maintenance and the
|
||||||
|
// system register accesses.
|
||||||
|
func TestArm64System(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tBRK $35943\n\tBRK\n\tSVC $7165\n\tDMB $1\n\tDSB $1\n\tISB $15\n"+
|
||||||
|
"\tDC ZVA, R4\n\tDC IVAC, R1\n\tMRS DCZID_EL0, R3\n\tMRS CNTVCT_EL0, R0\n\tMSR $9, DAIFSet\n\tMSR $3, SPSel\n"+
|
||||||
|
"\tPRFM (R0), PLDL1KEEP\n\tPRFM (R3), PLDL3KEEP\n\tPRFM (R2), $25\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xd4318ce0, // BRK $35943
|
||||||
|
0xd4200000, // BRK
|
||||||
|
0xd4037fa1, // SVC $7165
|
||||||
|
0xd50331bf, // DMB $1
|
||||||
|
0xd503319f, // DSB $1
|
||||||
|
0xd5033fdf, // ISB $15
|
||||||
|
0xd50b7424, // DC ZVA, R4
|
||||||
|
0xd5087621, // DC IVAC, R1
|
||||||
|
0xd53b00e3, // MRS DCZID_EL0, R3
|
||||||
|
0xd53be040, // MRS CNTVCT_EL0, R0
|
||||||
|
0xd50349df, // MSR $9, DAIFSet
|
||||||
|
0xd50043bf, // MSR $3, SPSel
|
||||||
|
0xf9800000, // PRFM (R0), PLDL1KEEP
|
||||||
|
0xf9800064, // PRFM (R3), PLDL3KEEP
|
||||||
|
0xf9800059, // PRFM (R2), $25
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64Crypto pins the AES and SHA families.
|
||||||
|
func TestArm64Crypto(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tAESE V31.B16, V29.B16\n\tAESD V22.B16, V19.B16\n\tAESIMC V12.B16, V27.B16\n\tAESMC V14.B16, V28.B16\n"+
|
||||||
|
"\tSHA1C V8.S4, V8, V2\n\tSHA1H V17, V25\n\tSHA1P V3.S4, V20, V27\n\tSHA1SU0 V17.S4, V13.S4, V16.S4\n\tSHA1SU1 V24.S4, V23.S4\n"+
|
||||||
|
"\tSHA256H V4.S4, V2, V11\n\tSHA256H2 V6.S4, V16, V11\n\tSHA256SU0 V0.S4, V16.S4\n\tSHA256SU1 V31.S4, V3.S4, V15.S4\n"+
|
||||||
|
"\tSHA512H V2.D2, V1, V0\n\tSHA512H2 V4.D2, V3, V2\n\tSHA512SU0 V9.D2, V8.D2\n\tSHA512SU1 V7.D2, V6.D2, V5.D2\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x4e284bfd, // AESE
|
||||||
|
0x4e285ad3, // AESD
|
||||||
|
0x4e28799b, // AESIMC
|
||||||
|
0x4e2869dc, // AESMC
|
||||||
|
0x5e080102, // SHA1C
|
||||||
|
0x5e280a39, // SHA1H
|
||||||
|
0x5e03129b, // SHA1P
|
||||||
|
0x5e1131b0, // SHA1SU0
|
||||||
|
0x5e281b17, // SHA1SU1
|
||||||
|
0x5e04404b, // SHA256H
|
||||||
|
0x5e06520b, // SHA256H2
|
||||||
|
0x5e282810, // SHA256SU0
|
||||||
|
0x5e1f606f, // SHA256SU1
|
||||||
|
0xce628020, // SHA512H
|
||||||
|
0xce648462, // SHA512H2
|
||||||
|
0xcec08128, // SHA512SU0
|
||||||
|
0xce6788c5, // SHA512SU1
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64SIMDLogical pins the arrangement-aware three- and two-register
|
||||||
|
// SIMD paths.
|
||||||
|
func TestArm64SIMDLogical(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tVADD V1.B16, V2.B16, V3.B16\n\tVAND V4.B16, V4.B16, V9.B16\n\tVEOR V0.B16, V1.B16, V0.B16\n"+
|
||||||
|
"\tVORR V5.B16, V4.B16, V3.B16\n\tVADDP V1.H8, V2.H8, V3.H8\n\tVZIP1 V16.H8, V3.H8, V19.H8\n\tVZIP2 V22.D2, V25.D2, V21.D2\n"+
|
||||||
|
"\tVCMEQ V24.S4, V13.S4, V12.S4\n\tVCMEQ $0, V2.H4, V3.H4\n\tVREV32 V2.H8, V1.H8\n\tVREV64 V2.S4, V3.S4\n\tVUADDLV V31.S4, V11\n"+
|
||||||
|
"\tVPMULL V2.D1, V1.D1, V3.Q1\n\tVPMULL2 V2.B16, V1.B16, V4.H8\n\tVRAX1 V26.D2, V29.D2, V30.D2\n\tVMOV V2.B16, V4.B16\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x4e218443, // VADD 16B
|
||||||
|
0x4e241c89, // VAND
|
||||||
|
0x6e201c20, // VEOR
|
||||||
|
0x4ea51c83, // VORR
|
||||||
|
0x4e61bc43, // VADDP 8H
|
||||||
|
0x4e503873, // VZIP1 8H
|
||||||
|
0x4ed67b35, // VZIP2 2D
|
||||||
|
0x6eb88dac, // VCMEQ 4S
|
||||||
|
0x0e609843, // VCMEQ $0, 4H
|
||||||
|
0x6e600841, // VREV32 8H
|
||||||
|
0x4ea00843, // VREV64 4S
|
||||||
|
0x6eb03beb, // VUADDLV 4S
|
||||||
|
0x0ee2e023, // VPMULL D1
|
||||||
|
0x4e22e024, // VPMULL2 16B
|
||||||
|
0xce7a8fbe, // VRAX1 2D
|
||||||
|
0x4ea21c44, // VMOV 16B pair
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64SIMDWide pins the four-register crypto group, VXAR, VEXT and the
|
||||||
|
// shift-by-immediate encodings.
|
||||||
|
func TestArm64SIMDWide(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tVEOR3 V2.B16, V7.B16, V12.B16, V25.B16\n\tVBCAX V1.B16, V2.B16, V26.B16, V31.B16\n"+
|
||||||
|
"\tVXAR $63, V27.D2, V21.D2, V26.D2\n\tVEXT $4, V2.B8, V1.B8, V3.B8\n\tVEXT $8, V2.B16, V1.B16, V3.B16\n"+
|
||||||
|
"\tVSHL $7, V22.D2, V25.D2\n\tVUSHR $6, V22.H8, V23.H8\n\tVSRI $24, V1.S4, V2.S4\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xce070999, // VEOR3
|
||||||
|
0xce22075f, // VBCAX
|
||||||
|
0xce9bfeba, // VXAR
|
||||||
|
0x2e022023, // VEXT B8
|
||||||
|
0x6e024023, // VEXT B16
|
||||||
|
0x4f4756d9, // VSHL D2 $7
|
||||||
|
0x6f1a06d7, // VUSHR H8 $6
|
||||||
|
0x6f284422, // VSRI S4 $24
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64SIMDElement pins VDUP and the VMOV element forms.
|
||||||
|
func TestArm64SIMDElement(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tVDUP V31.B[15], V18\n\tVDUP V19.S[3], V18.S4\n\tVDUP V1.D[1], V2.D2\n"+
|
||||||
|
"\tVMOV V13.S[0], R20\n\tVMOV V11.B[11], V16.B[12]\n\tVMOV R20, V21.B[2]\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x5e1f07f2, // VDUP element to register
|
||||||
|
0x4e1c0672, // VDUP element across S4
|
||||||
|
0x4e180422, // VDUP element across D2
|
||||||
|
0x0e043db4, // VMOV element to register
|
||||||
|
0x6e195d70, // VMOV element to element
|
||||||
|
0x4e051e95, // VMOV register into element
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64SIMDLoadStore pins the structure loads and stores.
|
||||||
|
func TestArm64SIMDLoadStore(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+
|
||||||
|
"\tVLD1.P 32(R1), [V2.B16, V3.B16]\n\tVST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)\n\tVST1.P [V2.B16], (R1)\n"+
|
||||||
|
"\tVLD1R (R1), [V9.B8]\n\tVLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x4c407055, // VLD1 one register
|
||||||
|
0x4c40a022, // VLD1 two registers
|
||||||
|
0x0c402fae, // VLD1 four registers D1
|
||||||
|
0x4cdfa022, // VLD1.P two registers
|
||||||
|
0x4c0029c2, // VST1 four registers S4
|
||||||
|
0x4c9f7022, // VST1.P one register
|
||||||
|
0x0d40c029, // VLD1R
|
||||||
|
0x0d60e000, // VLD4R
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64MoviLiteral pins the VMOVS/VMOVD/VMOVQ constant loads: three
|
||||||
|
// words each (ADRP, ADD, wide load) plus the pooled literal in the data
|
||||||
|
// section.
|
||||||
|
func TestArm64MoviLiteral(t *testing.T) {
|
||||||
|
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
|
||||||
|
"\tVMOVS $0x80402010, V11\n\tVMOVD $0x8040201008040201, V20\n" +
|
||||||
|
"\tVMOVQ $0x7040201008040201, $0x8040201008040201, V10\n\tRET\n"
|
||||||
|
f, errs := parser.Parse("test_arm64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
if img.Funcs[0].Size != 12*3+4 {
|
||||||
|
t.Errorf("func size = %d, want %d", img.Funcs[0].Size, 12*3+4)
|
||||||
|
}
|
||||||
|
want := []uint32{
|
||||||
|
0x9000001b, 0x9100037b, 0xbd40036b, // VMOVS: ADRP, ADD, LDR S
|
||||||
|
0x9000001b, 0x9100037b, 0xfd400374, // VMOVD: ADRP, ADD, LDR D
|
||||||
|
0x9000001b, 0x9100037b, 0x3dc0036a, // VMOVQ: ADRP, ADD, LDR Q
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
got := leWords(img.Code)
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The literals sit in the data section.
|
||||||
|
var found32, found64, found128 bool
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
switch d.Name {
|
||||||
|
case "$i32.80402010":
|
||||||
|
found32 = d.Size == 4
|
||||||
|
case "$i64.8040201008040201":
|
||||||
|
found64 = d.Size == 8
|
||||||
|
case "$i128.80402010080402017040201008040201":
|
||||||
|
found128 = d.Size == 16
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !found32 || !found64 || !found128 {
|
||||||
|
t.Errorf("literals missing: i32=%v i64=%v i128=%v", found32, found64, found128)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64MOVK pins standalone MOVK with the hw field derived from the
|
||||||
|
// chunk position.
|
||||||
|
func TestArm64MOVK(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tMOVK $1234, R5\n\tMOVK $305397760, R5\n\tMOVKW $1234, R5\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xf2809a45, // MOVK hw=0
|
||||||
|
0xf2a24685, // MOVK hw=1
|
||||||
|
0x72809a45, // MOVKW hw=0
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -572,3 +1016,419 @@ func leWords(b []byte) []uint32 {
|
|||||||
}
|
}
|
||||||
return w
|
return w
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestArm64IndirectBranch pins the indirect branch forms in a leaf function:
|
||||||
|
// JMP (Rn) lowers to BR Rn, matching the toolchain's spelling, and the raw
|
||||||
|
// BR/BLR mnemonics encode directly (a gasm superset the toolchain's front
|
||||||
|
// end does not accept). CALL (Rn) shares the BLR path and its non-leaf
|
||||||
|
// prologue parity is covered by the ground-truth kernel.
|
||||||
|
func TestArm64IndirectBranch(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·f(SB), NOSPLIT, $0-0
|
||||||
|
JMP (R0)
|
||||||
|
BR R5
|
||||||
|
BLR R6
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("test_arm64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
want := []uint32{
|
||||||
|
0xd61f0000, // BR R0
|
||||||
|
0xd61f00a0, // BR R5
|
||||||
|
0xd63f00c0, // BLR R6
|
||||||
|
0xd65f03c0, // RET (BR LR)
|
||||||
|
}
|
||||||
|
got := leWords(img.Code)
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// arm64Words assembles a single NOSPLIT leaf body and returns its words.
|
||||||
|
func arm64Words(t *testing.T, body string) []uint32 {
|
||||||
|
t.Helper()
|
||||||
|
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
return leWords(img.Code)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64ShiftEncodings pins the shift words against `go tool asm -S`
|
||||||
|
// output (Go 1.27, arm64): immediate forms alias SBFM/UBFM with ROR as EXTR,
|
||||||
|
// register forms are the two-source LSLV/LSRV/ASRV/RORV.
|
||||||
|
func TestArm64ShiftEncodings(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tLSL $4, R0, R1\n\tLSR $8, R0, R2\n\tASR $4, R0, R3\n\tROR $12, R0, R4\n"+
|
||||||
|
"\tLSLW $4, R0, R5\n\tLSRW $8, R0, R6\n\tASRW $4, R0, R7\n\tRORW $12, R0, R8\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xd37cec01, // LSL $4 = UBFM X1, X0, #60, #59
|
||||||
|
0xd348fc02, // LSR $8 = UBFM X2, X0, #8, #63
|
||||||
|
0x9344fc03, // ASR $4 = SBFM X3, X0, #4, #63
|
||||||
|
0x93c03004, // ROR $12 = EXTR X4, X0, X0, #12
|
||||||
|
0x531c6c05, // LSLW $4 = UBFM W5, W0, #28, #27
|
||||||
|
0x53087c06, // LSRW $8 = UBFM W6, W0, #8, #31
|
||||||
|
0x13047c07, // ASRW $4 = SBFM W7, W0, #4, #31
|
||||||
|
0x13803008, // RORW $12 = EXTR W8, W0, W0, #12
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("imm shift word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
got = arm64Words(t, "\tLSL R9, R0, R10\n\tLSR R9, R0, R11\n\tASR R9, R0, R12\n\tROR R9, R0, R13\n"+
|
||||||
|
"\tLSLW R9, R0, R14\n\tLSRW R9, R0, R15\n\tASRW R9, R0, R16\n\tRORW R9, R0, R17\n")
|
||||||
|
want = []uint32{
|
||||||
|
0x9ac9200a, // LSLV X10, X0, X9
|
||||||
|
0x9ac9240b, // LSRV X11, X0, X9
|
||||||
|
0x9ac9280c, // ASRV X12, X0, X9
|
||||||
|
0x9ac92c0d, // RORV X13, X0, X9
|
||||||
|
0x1ac9200e, // LSLV W14, W0, W9
|
||||||
|
0x1ac9240f, // LSRV W15, W0, W9
|
||||||
|
0x1ac92810, // ASRV W16, W0, W9
|
||||||
|
0x1ac92c11, // RORV W17, W0, W9
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("reg shift word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Two-operand spellings fold to Rn = Rd.
|
||||||
|
got = arm64Words(t, "\tLSL $4, R1\n\tLSR R9, R1\n\tASR $4, R1\n\tROR R9, R1\n\tLSLW $4, R1\n\tRORW R9, R1\n")
|
||||||
|
want = []uint32{
|
||||||
|
0xd37cec21, // LSL $4, R1 = UBFM X1, X1, #60, #59
|
||||||
|
0x9ac92421, // LSRV X1, X1, X9
|
||||||
|
0x9344fc21, // ASR $4, R1 = SBFM X1, X1, #4, #63
|
||||||
|
0x9ac92c21, // RORV X1, X1, X9
|
||||||
|
0x531c6c21, // LSLW $4, R1 = UBFM W1, W1, #28, #27
|
||||||
|
0x1ac92c21, // RORV W1, W1, W9
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("2op shift word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64ShiftRangeErrors: the toolchain reports "illegal bit number" for
|
||||||
|
// shift amounts at or above the operand width.
|
||||||
|
func TestArm64ShiftRangeErrors(t *testing.T) {
|
||||||
|
for _, src := range []string{
|
||||||
|
"\tLSL $64, R0, R1\n",
|
||||||
|
"\tLSRW $32, R0, R1\n",
|
||||||
|
"\tRORW $32, R0, R1\n",
|
||||||
|
"\tASR $-1, R0, R1\n",
|
||||||
|
} {
|
||||||
|
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+src+"\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFileARM64(f); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", src)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64DivEncodings pins SDIV/UDIV in both widths: the 2-source opcode
|
||||||
|
// field (bits 15:10 of the 0xd6<<21 fixed field) is UDIV=0b0010, SDIV=0b0011.
|
||||||
|
func TestArm64DivEncodings(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tSDIV R1, R2, R3\n\tUDIV R1, R2, R3\n\tSDIVW R1, R2, R3\n\tUDIVW R1, R2, R3\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x9ac10c43, // SDIV X3, X2, X1
|
||||||
|
0x9ac10843, // UDIV X3, X2, X1
|
||||||
|
0x1ac10c43, // SDIV W3, W2, W1
|
||||||
|
0x1ac10843, // UDIV W3, W2, W1
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("div word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64MAddSub pins the four-operand MADD/MSUB words (Rm, Ra, Rn, Rd,
|
||||||
|
// with Ra in bits 14:10) and rejects the shorter spellings the toolchain
|
||||||
|
// also rejects.
|
||||||
|
func TestArm64MAddSub(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tMADD R1, R2, R3, R4\n\tMSUB R1, R2, R3, R4\n\tMADDW R1, R2, R3, R5\n\tMSUBW R1, R2, R3, R5\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x9b010864, // MADD X4, X3, X1, X2 (Rm=1, Ra=2, Rn=3)
|
||||||
|
0x9b018864, // MSUB X4, X3, X1, X2
|
||||||
|
0x1b010865, // MADD W5, W3, W1, W2
|
||||||
|
0x1b018865, // MSUB W5, W3, W1, W2
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("madd word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The accumulate operand is mandatory: 2- and 3-operand forms error
|
||||||
|
// rather than silently reading R0 or ZR as the accumulator.
|
||||||
|
for _, body := range []string{
|
||||||
|
"\tMADD R1, R2\n",
|
||||||
|
"\tMADD R1, R2, R3\n",
|
||||||
|
"\tMSUBW R1, R2, R3\n",
|
||||||
|
} {
|
||||||
|
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFileARM64(f); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", body)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64MovImmWidth pins the immediate classifications whose size pass
|
||||||
|
// once disagreed with the encoder: negative and 0xFFFFFFFF W values go
|
||||||
|
// through MOVN after 32-bit truncation, and 3- to 4-chunk constants expand
|
||||||
|
// to one word per non-zero chunk.
|
||||||
|
func TestArm64MovImmWidth(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tMOVW $-1, R0\n\tMOVW $0xFFFFFFFF, R3\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x12800000, // MOVN W0, #0
|
||||||
|
0x12800003, // MOVN W3, #0
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("movw word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range []struct {
|
||||||
|
body string
|
||||||
|
words int
|
||||||
|
}{
|
||||||
|
{"\tMOVD $0x0001000200030000, R2\n", 3}, // three chunks
|
||||||
|
{"\tMOVD $0x0001000200030004, R1\n", 4}, // four chunks
|
||||||
|
{"\tMOVW $-1, R0\n", 1}, // MOVN after truncation
|
||||||
|
} {
|
||||||
|
if got := arm64Words(t, tt.body); len(got) != tt.words+1 {
|
||||||
|
t.Errorf("%s: %d words, want %d (including RET)", tt.body, len(got), tt.words+1)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64ExclOffsetErrors: exclusive and atomic encodings carry no
|
||||||
|
// immediate field, so a non-zero offset is rejected the way the toolchain
|
||||||
|
// reports "illegal combination" for it, never silently dropped.
|
||||||
|
func TestArm64ExclOffsetErrors(t *testing.T) {
|
||||||
|
for _, body := range []string{
|
||||||
|
"\tLDXR 8(R1), R2\n",
|
||||||
|
"\tLDAXR 8(R1), R2\n",
|
||||||
|
"\tSTXR R3, 8(R1), R4\n",
|
||||||
|
"\tSTLXR R3, 8(R1), R4\n",
|
||||||
|
"\tCASD R3, 8(R1), R4\n",
|
||||||
|
"\tLDADDD R3, 8(R1), R4\n",
|
||||||
|
} {
|
||||||
|
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFileARM64(f); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", body)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64ExclNoOffset pins the plain (Rn) forms, byte-for-byte against
|
||||||
|
// go tool asm. The toolchain parses the FIRST register of a store as the
|
||||||
|
// data register and the LAST as the status register (asm7.go case 59), and
|
||||||
|
// the pair forms as (Rt1, Rt2) (case 58/59):
|
||||||
|
//
|
||||||
|
// STXR R3, (R1), R4 → c8047c23 (Rt=3, Rn=1, Rs=4)
|
||||||
|
// STXP (R3, R4), (R1), R5 → c8251023 (Rt=3, Rt2=4, Rn=1, Rs=5)
|
||||||
|
// LDXP (R1), (R3, R4) → c87f1023 (Rn=1, Rt=3, Rt2=4)
|
||||||
|
func TestArm64ExclNoOffset(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tLDXR (R1), R2\n\tSTXR R3, (R1), R4\n"+
|
||||||
|
"\tSTXP (R3, R4), (R1), R5\n\tSTXPW (R3, R4), (R1), R5\n"+
|
||||||
|
"\tLDXP (R1), (R3, R4)\n\tLDXPW (R1), (R3, R4)\n"+
|
||||||
|
"\tSTXR R3, (RSP), R4\n\tLDXR (RSP), R2\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xc85f7c22, // LDXR X2, [X1]
|
||||||
|
0xc8047c23, // STXR W3, [X1], W4 with Rt = R3, Rs = R4
|
||||||
|
0xc8251023, // STXP (R3, R4), [X1], R5
|
||||||
|
0x88251023, // STXPW (R3, R4), [X1], R5
|
||||||
|
0xc87f1023, // LDXP [X1], (R3, R4)
|
||||||
|
0x887f1023, // LDXPW [X1], (R3, R4)
|
||||||
|
0xc8047fe3, // STXR R3, [SP], R4
|
||||||
|
0xc85f7fe2, // LDXR [SP], R2
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("excl word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64AddSubImmRange: immediates that cannot ride the imm12 field are
|
||||||
|
// rejected instead of wrapping through int32.
|
||||||
|
func TestArm64AddSubImmRange(t *testing.T) {
|
||||||
|
for _, body := range []string{
|
||||||
|
"\tADD $0x100000000, R0, R1\n",
|
||||||
|
"\tSUB $-0x100000000, R0, R1\n",
|
||||||
|
"\tCMP $0x100000000, R0\n",
|
||||||
|
} {
|
||||||
|
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFileARM64(f); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", body)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64LargeRegisterOffset pins the large-offset path for a register
|
||||||
|
// base: the ADD offsets from the operand's own base, not from SP, matching
|
||||||
|
// the toolchain's `ADD $(256<<12), R2, R27; MOVD (R27), R3`.
|
||||||
|
func TestArm64LargeRegisterOffset(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tMOVD 0x100000(R2), R3\n\tMOVD R3, 0x100000(R2)\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x9144005b, // ADD $(256<<12), R2, R27
|
||||||
|
0xf9400363, // MOVD (R27), R3
|
||||||
|
0x9144005b, // ADD $(256<<12), R2, R27
|
||||||
|
0xf9000363, // MOVD R3, (R27)
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("large offset word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64LargeFrameSpadj checks the stack-adjustment boundaries of a frame
|
||||||
|
// whose autosize must be materialised into REGTMP: $5000 rounds the autosize
|
||||||
|
// to 5024, so the prologue is [MOVD $5024, R27][SUB R27, RSP, R20][STP][ADD
|
||||||
|
// R20, SP][SUB $8] and SP moves only at its fourth word, while the RET's
|
||||||
|
// epilogue is [LDP][MOVD $5024, R27][ADD R27, RSP, RSP] before the final
|
||||||
|
// RET. These PCs feed the DWARF CFA rules and the goobj stack maps.
|
||||||
|
func TestArm64LargeFrameSpadj(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("frame_arm64.s", "#include \"textflag.h\"\n\nTEXT ·framed(SB), $5000-0\n\tCALL ·other(SB)\n\tRET\n\nTEXT ·other(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
fn := img.Funcs[0]
|
||||||
|
// autosize 5024: class-2 guard of 6 words (24 bytes), a 5-word prologue
|
||||||
|
// whose ADD R20, SP sits at byte 8 inside it, a one-instruction body,
|
||||||
|
// then a 3-word epilogue before the final RET.
|
||||||
|
wantSpadj := []SpadjStep{{PC: 24 + 12, Value: 5024}, {PC: 24 + 20 + 4 + 12, Value: 0}}
|
||||||
|
if len(fn.Spadj) != len(wantSpadj) {
|
||||||
|
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
|
||||||
|
}
|
||||||
|
for i := range wantSpadj {
|
||||||
|
if fn.Spadj[i] != wantSpadj[i] {
|
||||||
|
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The words those PCs point between: the prologue's ADD R20, SP at byte
|
||||||
|
// 36, and the epilogue's materialised ADD R27, RSP, RSP right before the
|
||||||
|
// final RET at byte 60.
|
||||||
|
words := leWords(img.Code[fn.Offset : fn.Offset+fn.Size])
|
||||||
|
if got := words[(24+12)/4]; got != 0x9100029f {
|
||||||
|
t.Errorf("prologue word at byte 36 = %08x, want 9100029f (ADD R20, SP)", got)
|
||||||
|
}
|
||||||
|
if got := words[(24+20+4+8)/4]; got != 0x8b3b63ff {
|
||||||
|
t.Errorf("epilogue word at byte 56 = %08x, want 8b3b63ff (ADD R27, RSP, RSP)", got)
|
||||||
|
}
|
||||||
|
if got := words[(24+20+4+12)/4]; got != 0xd65f03c0 {
|
||||||
|
t.Errorf("final RET word at byte 60 = %08x, want d65f03c0", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64SplitFrameSpadj pins the addcon2 band, where neither imm12 form
|
||||||
|
// nor a single MOVZ carries the autosize and the toolchain splits the
|
||||||
|
// prologue SUB into two imm12 instructions (asm7.go case 48) while the
|
||||||
|
// non-leaf RET still materialises the value into REGTMP (obj7.go ARET,
|
||||||
|
// issue 73259). $65664 rounds the autosize to 65680 = 144 + 16<<12:
|
||||||
|
//
|
||||||
|
// [SUB $144, RSP, R20][SUB $(16<<12), R20, R20][STP][MOVD R20, SP][SUB $8]
|
||||||
|
// [CALL]
|
||||||
|
// [LDP][MOVD $144, R27][MOVK $(1<<16), R27][ADD R27, RSP, RSP][RET]
|
||||||
|
//
|
||||||
|
// SP moves at the fourth word (byte 12) and returns to zero at the final
|
||||||
|
// RET (byte 40); the words are go tool asm's own for the same source.
|
||||||
|
func TestArm64SplitFrameSpadj(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("frame_arm64.s", "#include \"textflag.h\"\n\nTEXT ·framed(SB), NOSPLIT, $65664-0\n\tCALL ·other(SB)\n\tRET\n\nTEXT ·other(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
fn := img.Funcs[0]
|
||||||
|
wantSpadj := []SpadjStep{{PC: 12, Value: 65680}, {PC: 40, Value: 0}}
|
||||||
|
if len(fn.Spadj) != len(wantSpadj) {
|
||||||
|
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
|
||||||
|
}
|
||||||
|
for i := range wantSpadj {
|
||||||
|
if fn.Spadj[i] != wantSpadj[i] {
|
||||||
|
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
want := []uint32{
|
||||||
|
0xd10243f4, // SUB $144, RSP, R20
|
||||||
|
0xd1404294, // SUB $(16<<12), R20, R20
|
||||||
|
0xa93ffa9d, // STP (R29, R30), -8(R20)
|
||||||
|
0x9100029f, // MOVD R20, RSP
|
||||||
|
0xd10023fd, // SUB $8, RSP, R29
|
||||||
|
0x94000000, // CALL (relocation masked at link time)
|
||||||
|
0xa97ffbfd, // LDP -8(RSP), (R29, R30)
|
||||||
|
0xd280121b, // MOVD $144, R27
|
||||||
|
0xf2a0003b, // MOVK $(1<<16), R27
|
||||||
|
0x8b3b63ff, // ADD R27, RSP, RSP
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
words := leWords(img.Code[fn.Offset : fn.Offset+fn.Size])
|
||||||
|
if len(words) != len(want) {
|
||||||
|
t.Fatalf("framed = %d words, want %d", len(words), len(want))
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
if words[i] != w {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, words[i], w)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+85
-20
@@ -187,10 +187,32 @@ func arm64Prologue(fi arm64FrameInfo) []byte {
|
|||||||
return a64WordsLE(ws...)
|
return a64WordsLE(ws...)
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value
|
// arm64SplitImm12 reports whether the toolchain decomposes ADD/SUB $imm into
|
||||||
// fits the imm12 field (plain, or shifted left by 12 when it is a multiple
|
// two imm12 instructions instead of materialising it into REGTMP
|
||||||
// of 4096); otherwise the toolchain materialises it into REGTMP (R27) and
|
// (asm7.go case 48, the C_ADDCON2 class): the value must fit 24 bits
|
||||||
// subtracts the register in the extended-register form.
|
// unsigned and be neither encodable as one imm12 (checked by the callers
|
||||||
|
// first), nor loadable into a register in a single MOVZ/MOVN word, nor a
|
||||||
|
// logical immediate, because conclass tests all three before C_ADDCON2.
|
||||||
|
func arm64SplitImm12(imm uint32) bool {
|
||||||
|
if imm > 0xFFFFFF {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
if _, _, _, ok := arm64Bitmask(uint64(imm), 1); ok {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
return arm64Movcon(int64(imm)) < 0 && arm64Movcon(^int64(imm)) < 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// arm64SubImmWords emits SUB $imm, SP, Rd with the toolchain's ladder for an
|
||||||
|
// ADD/SUB constant (asm7.go conclass and cases 2, 48, 62 and 13): the
|
||||||
|
// immediate form when the value fits imm12 (plain, or shifted left by 12
|
||||||
|
// when it is a multiple of 4096); a value with a single 16-bit chunk, a
|
||||||
|
// logical immediate, or one wider than 24 bits is materialised into REGTMP
|
||||||
|
// (R27) and subtracted in the extended-register form; everything else up to
|
||||||
|
// 0xFFFFFF is split into two imm12 instructions:
|
||||||
|
//
|
||||||
|
// SUB $(imm&0xfff), SP, Rd
|
||||||
|
// SUB $((imm&0xfff000)>>12)<<12, Rd, Rd
|
||||||
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
||||||
if imm <= 0xFFF {
|
if imm <= 0xFFF {
|
||||||
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
|
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
|
||||||
@@ -198,15 +220,21 @@ func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
|||||||
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
||||||
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
|
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
|
||||||
}
|
}
|
||||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
if !arm64SplitImm12(imm) {
|
||||||
if err != nil {
|
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||||
mov = nil
|
if err != nil {
|
||||||
|
mov = nil
|
||||||
|
}
|
||||||
|
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
|
||||||
|
}
|
||||||
|
return []uint32{
|
||||||
|
a64AddSub(1, 1, 0, 0, imm&0xFFF, 31, rd),
|
||||||
|
a64AddSub(1, 1, 0, 1, (imm&0xFFF000)>>12, rd, rd),
|
||||||
}
|
}
|
||||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12
|
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12,
|
||||||
// and REGTMP fallback ladder.
|
// split and REGTMP ladder as arm64SubImmWords.
|
||||||
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
||||||
if imm <= 0xFFF {
|
if imm <= 0xFFF {
|
||||||
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
|
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
|
||||||
@@ -214,11 +242,35 @@ func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
|||||||
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
||||||
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
|
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
|
||||||
}
|
}
|
||||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
if !arm64SplitImm12(imm) {
|
||||||
|
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||||
|
if err != nil {
|
||||||
|
mov = nil
|
||||||
|
}
|
||||||
|
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
|
||||||
|
}
|
||||||
|
return []uint32{
|
||||||
|
a64AddSub(1, 0, 0, 0, imm&0xFFF, 31, rd),
|
||||||
|
a64AddSub(1, 0, 0, 1, (imm&0xFFF000)>>12, rd, rd),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// arm64RetAddWords emits the frame deallocation of a non-leaf RET with a
|
||||||
|
// large frame. The toolchain adds the frame back with a single instruction:
|
||||||
|
// a plain imm12 ADD when autosize fits 12 bits, otherwise the value is
|
||||||
|
// materialised into REGTMP and added as a register, so the epilogue never
|
||||||
|
// leaves a partially deallocated frame (obj7.go ARET, issue 73259). The
|
||||||
|
// shifted-imm12 and split-imm12 forms are therefore never used here, unlike
|
||||||
|
// the leaf epilogue's plain ADD instructions.
|
||||||
|
func arm64RetAddWords(autosize uint32) []uint32 {
|
||||||
|
if autosize < 1<<12 {
|
||||||
|
return []uint32{a64AddSub(1, 0, 0, 0, autosize, 31, 31)}
|
||||||
|
}
|
||||||
|
mov, err := encodeARM64LoadImm(27, int64(autosize), "MOVD")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
mov = nil
|
mov = nil
|
||||||
}
|
}
|
||||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
|
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, 31))
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
|
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
|
||||||
@@ -237,11 +289,11 @@ func arm64Return(fi arm64FrameInfo) []byte {
|
|||||||
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
|
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
|
||||||
)
|
)
|
||||||
} else {
|
} else {
|
||||||
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
|
// Large frame: LDP -8(SP), (FP, LR), then deallocate.
|
||||||
ws = append(ws,
|
ws = append(ws,
|
||||||
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
||||||
)
|
)
|
||||||
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
|
ws = append(ws, arm64RetAddWords(uint32(fi.autosize))...)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// RET: BR LR (0xd65f03c0)
|
// RET: BR LR (0xd65f03c0)
|
||||||
@@ -258,22 +310,32 @@ func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
|
|||||||
if fi.autosize <= 0xf0 {
|
if fi.autosize <= 0xf0 {
|
||||||
return 4 // MOVD.W instruction decrements SP
|
return 4 // MOVD.W instruction decrements SP
|
||||||
}
|
}
|
||||||
return 8 // SUB + STP + MOVD (3 instructions, SP updated at the MOVD)
|
// Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose
|
||||||
|
// position depends on how many words the SUB itself took (immediate,
|
||||||
|
// shifted immediate, the two-word imm12 split, or a materialised REGTMP
|
||||||
|
// sequence).
|
||||||
|
return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1)
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
||||||
// (but not including) the final RET instruction.
|
// (but not including) the final RET instruction. The lengths are read from
|
||||||
|
// the same word-emitting helpers the epilogue uses rather than assumed: the
|
||||||
|
// leaf path shares the prologue's immediate ladder, and a materialised
|
||||||
|
// autosize costs its MOV words plus the ADD itself.
|
||||||
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
|
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
|
||||||
if fi.autosize == 0 {
|
if fi.autosize == 0 {
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
if fi.leaf {
|
if fi.leaf {
|
||||||
return 8 // ADD + ADD
|
return 4 * (len(arm64AddImmWords(uint32(fi.autosize-8), 29)) +
|
||||||
|
len(arm64AddImmWords(uint32(fi.autosize), 31)))
|
||||||
}
|
}
|
||||||
if fi.autosize <= 0xf0 {
|
if fi.autosize <= 0xf0 {
|
||||||
return 8 // LDR + LDR.P
|
return 8 // LDR + LDR.P
|
||||||
}
|
}
|
||||||
return 8 // LDP + ADD
|
// LDP + the deallocation emitted by arm64RetAddWords, so the length
|
||||||
|
// tracks whatever the MOVD ladder needs.
|
||||||
|
return 4 + 4*len(arm64RetAddWords(uint32(fi.autosize)))
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64ResolvePseudo translates a pseudo-register memory reference into a
|
// arm64ResolvePseudo translates a pseudo-register memory reference into a
|
||||||
@@ -382,9 +444,12 @@ func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte {
|
|||||||
ws = append(ws, wordsOf(mov)...)
|
ws = append(ws, wordsOf(mov)...)
|
||||||
ml := len(mov) / 4
|
ml := len(mov) / 4
|
||||||
ws = append(ws, arm64DPExtWords(arm64OpSubs, 27, 31, 17)) // SUBS R17, RSP, R27
|
ws = append(ws, arm64DPExtWords(arm64OpSubs, 27, 31, 17)) // SUBS R17, RSP, R27
|
||||||
ws = append(ws, br(8+ml, a64CondLO))
|
// The branches sit at fixed byte offsets in the guard prefix: after
|
||||||
|
// the LDR (4), the ml MOV words (4*ml) and the SUBS (4) for B.LO,
|
||||||
|
// then a further B.LO word and the CMP for B.LS.
|
||||||
|
ws = append(ws, br(8+4*ml, a64CondLO))
|
||||||
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
|
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
|
||||||
ws = append(ws, br(8+ml+8, a64CondLS))
|
ws = append(ws, br(16+4*ml, a64CondLS))
|
||||||
}
|
}
|
||||||
return a64WordsLE(ws...)
|
return a64WordsLE(ws...)
|
||||||
}
|
}
|
||||||
|
|||||||
+93
-28
@@ -114,6 +114,11 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] {
|
if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
// A zero-operand jump parses; its arity is reported during
|
||||||
|
// emission (encodeJump), so the layout must not index Operands.
|
||||||
|
if len(s.Operands) != 1 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
name, ok := labelName(s.Operands[0])
|
name, ok := labelName(s.Operands[0])
|
||||||
if !ok {
|
if !ok {
|
||||||
continue // reported during emission
|
continue // reported during emission
|
||||||
@@ -137,28 +142,22 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
}
|
}
|
||||||
if fi.splitClass == 2 && !guardJBlong {
|
if fi.splitClass == 2 && !guardJBlong {
|
||||||
// The underflow JB sits before the CMPQ; its displacement spans
|
// The underflow JB sits before the CMPQ; its displacement spans
|
||||||
// the rest of the guard plus the prologue and the body.
|
// the rest of the guard plus the prologue and the body. The JB
|
||||||
jbLen := 2
|
// is still the short form this branch tests (relaxing it is this
|
||||||
if guardJBlong {
|
// branch's job), so guardLen is taken with a short JB and the
|
||||||
jbLen = 6
|
// subtraction drops the prefix and the JB's own 2 bytes.
|
||||||
}
|
rest := fi.guardLen(false, guardJBElong) - (9 + 3 + 7 + 2)
|
||||||
rest := fi.guardLen(guardJBlong, guardJBElong) - (9 + 3 + 7 + jbLen)
|
|
||||||
if !fits8(int64(rest + len(fi.prologue) + bodyLen)) {
|
if !fits8(int64(rest + len(fi.prologue) + bodyLen)) {
|
||||||
guardJBlong = true
|
guardJBlong = true
|
||||||
changed = true
|
changed = true
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// The morestack JMP returns to the function start, so its
|
// The morestack JMP returns to the function start, so its
|
||||||
// displacement is the negated distance from its own end.
|
// displacement is the negated distance from its own end; while it is
|
||||||
if !moreJMPlong {
|
// still short, its own length is 2 bytes.
|
||||||
jmpLen := 2
|
if !moreJMPlong && !fits8(-int64(guard+len(fi.prologue)+bodyLen+5+2)) {
|
||||||
if moreJMPlong {
|
moreJMPlong = true
|
||||||
jmpLen = 5
|
changed = true
|
||||||
}
|
|
||||||
if !fits8(-int64(guard + len(fi.prologue) + bodyLen + 5 + jmpLen)) {
|
|
||||||
moreJMPlong = true
|
|
||||||
changed = true
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
if !changed {
|
if !changed {
|
||||||
break
|
break
|
||||||
@@ -181,7 +180,15 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
var out []byte
|
var out []byte
|
||||||
var patches []sbPatch
|
var patches []sbPatch
|
||||||
if fi.needSplit {
|
if fi.needSplit {
|
||||||
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+2)+len(fi.prologue)+bodyLen))
|
// The JBE ends the guard, so its displacement is the prologue plus
|
||||||
|
// the body; the underflow JB additionally spans the trailing CMPQ and
|
||||||
|
// JBE, whose combined length is guardLen minus the prefix and the
|
||||||
|
// JB's own length (2 short, 6 long).
|
||||||
|
jbLen := 2
|
||||||
|
if guardJBlong {
|
||||||
|
jbLen = 6
|
||||||
|
}
|
||||||
|
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+jbLen)+len(fi.prologue)+bodyLen))
|
||||||
out = append(out, guard...)
|
out = append(out, guard...)
|
||||||
patches = append(patches, tlsPatch)
|
patches = append(patches, tlsPatch)
|
||||||
}
|
}
|
||||||
@@ -337,7 +344,21 @@ func computeFrame(t *ast.Text) frameInfo {
|
|||||||
if t.Frame != nil && t.Frame.Imm.HasVal {
|
if t.Frame != nil && t.Frame.Imm.HasVal {
|
||||||
fi.size = int(t.Frame.Imm.Val)
|
fi.size = int(t.Frame.Imm.Val)
|
||||||
}
|
}
|
||||||
if fi.size > 0 {
|
if fi.size == 0 && hasCall(t) {
|
||||||
|
// The toolchain gives a frameless function containing a CALL an
|
||||||
|
// 8-byte frame for the pushed base pointer: the prologue saves BP
|
||||||
|
// with no stack adjustment, every RET pops it back, FP references
|
||||||
|
// pass one extra slot, and the virtual SP is the hardware SP.
|
||||||
|
fi.size = 8
|
||||||
|
fi.useFP = true
|
||||||
|
// The push is the frame: the saved BP sits at SP+0 and the
|
||||||
|
// return address at SP+8, so arguments begin at SP+16. Unlike
|
||||||
|
// a SUBQ frame, the 8-byte size must not be added again.
|
||||||
|
fi.fpAdjust = 16
|
||||||
|
fi.spAdjust = 0
|
||||||
|
fi.prologue = []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
|
||||||
|
fi.epilogue = []byte{0x5D} // POPQ BP
|
||||||
|
} else if fi.size > 0 {
|
||||||
fi.useFP = true
|
fi.useFP = true
|
||||||
fi.fpAdjust = int64(fi.size) + 16 // frame + saved BP + return address
|
fi.fpAdjust = int64(fi.size) + 16 // frame + saved BP + return address
|
||||||
fi.spAdjust = int64(fi.size)
|
fi.spAdjust = int64(fi.size)
|
||||||
@@ -415,16 +436,6 @@ func (fi frameInfo) guardLen(jbLong, jbeLong bool) int {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// moreLen returns the byte length of the trailing morestack block: the CALL
|
|
||||||
// (always rel32) plus the JMP back to the function start.
|
|
||||||
func moreLen(jmpLong bool) int {
|
|
||||||
jmp := 2
|
|
||||||
if jmpLong {
|
|
||||||
jmp = 5
|
|
||||||
}
|
|
||||||
return 5 + jmp
|
|
||||||
}
|
|
||||||
|
|
||||||
// buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the
|
// buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the
|
||||||
// already-computed displacements of the conditional branches that jump to the
|
// already-computed displacements of the conditional branches that jump to the
|
||||||
// morestack block (unused in classes without them). The TLS load carries a
|
// morestack block (unused in classes without them). The TLS load carries a
|
||||||
@@ -518,6 +529,13 @@ func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, erro
|
|||||||
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
|
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
|
||||||
return 5, nil // opcode + rel32, always the long form
|
return 5, nil // opcode + rel32, always the long form
|
||||||
}
|
}
|
||||||
|
if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) {
|
||||||
|
code, err := encodeIndirectJump(s, mnem)
|
||||||
|
if err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
return len(code), nil
|
||||||
|
}
|
||||||
return jumpSize(mnem, long), nil
|
return jumpSize(mnem, long), nil
|
||||||
}
|
}
|
||||||
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
|
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
|
||||||
@@ -585,6 +603,15 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
|||||||
}
|
}
|
||||||
return append(prefix, code...), ps, nil
|
return append(prefix, code...), ps, nil
|
||||||
}
|
}
|
||||||
|
if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) {
|
||||||
|
// JMP/CALL through a register or memory: no relocation and no
|
||||||
|
// label to resolve, the operand fully determines the bytes.
|
||||||
|
code, err = encodeIndirectJump(s, mnem)
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, err
|
||||||
|
}
|
||||||
|
return append(prefix, code...), nil, nil
|
||||||
|
}
|
||||||
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
|
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
|
||||||
} else {
|
} else {
|
||||||
code, ps, err = encodeNormal(s, fi, link)
|
code, ps, err = encodeNormal(s, fi, link)
|
||||||
@@ -706,6 +733,44 @@ func labelName(op *ast.Operand) (string, bool) {
|
|||||||
return "", false
|
return "", false
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// indirectJumpTarget reports whether the JMP/CALL operand addresses a
|
||||||
|
// register or a memory location rather than a label or a static symbol.
|
||||||
|
// A bare identifier is a register when the register table knows the name and
|
||||||
|
// a label otherwise, which is exactly how the parser cannot distinguish them.
|
||||||
|
func indirectJumpTarget(s *ast.Instr) bool {
|
||||||
|
if len(s.Operands) != 1 || s.Operands[0].Kind != ast.OpAddr {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
a := s.Operands[0].Addr
|
||||||
|
if a.Base != "" || a.Index != "" {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" {
|
||||||
|
if _, ok := ParseReg(a.Sym.Name); ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeIndirectJump assembles a JMP/CALL through a register or memory
|
||||||
|
// operand, which carries no relocation and no label to resolve.
|
||||||
|
func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
|
||||||
|
ops := make([]Operand, len(s.Operands))
|
||||||
|
for i, op := range s.Operands {
|
||||||
|
o, err := operandFromAST(op, 8, frameInfo{}, nil)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
ops[i] = o
|
||||||
|
}
|
||||||
|
e := &enc{}
|
||||||
|
if err := e.encodeIndirectBranch(mnem, ops); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return e.out, nil
|
||||||
|
}
|
||||||
|
|
||||||
// spReg is the hardware stack pointer used to realise FP/SP pseudo-operands.
|
// spReg is the hardware stack pointer used to realise FP/SP pseudo-operands.
|
||||||
var spReg = Reg{idx: 4, size: 8}
|
var spReg = Reg{idx: 4, size: 8}
|
||||||
|
|
||||||
|
|||||||
+66
-2
@@ -160,6 +160,57 @@ TEXT ·loadarg(SB), NOSPLIT, $0-24
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAssembleFramelessCall verifies the forced base-pointer frame a $0-frame
|
||||||
|
// function containing a CALL receives: the PUSHQ BP prologue with no stack
|
||||||
|
// adjustment and the x+N(FP) → (N+16)(SP) translation, against the bytes the
|
||||||
|
// Go assembler produces. The push is the frame, so the offset must not count
|
||||||
|
// it twice.
|
||||||
|
func TestAssembleFramelessCall(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("frameless_call_amd64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·withcall(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ x+0(FP), AX
|
||||||
|
CALL ·other(SB)
|
||||||
|
MOVQ AX, ret+8(FP)
|
||||||
|
RET
|
||||||
|
TEXT ·other(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
code := append([]byte(nil), img.Code[img.Funcs[0].Offset:img.Funcs[0].Offset+img.Funcs[0].Size]...)
|
||||||
|
for _, r := range img.Funcs[0].Relocs {
|
||||||
|
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||||
|
code[j] = 0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// From `go tool objdump` of the Go-assembled function:
|
||||||
|
// PUSHQ BP 55
|
||||||
|
// MOVQ SP, BP 4889e5
|
||||||
|
// MOVQ 0x10(SP), AX 488b442410
|
||||||
|
// CALL other e800000000
|
||||||
|
// MOVQ AX, 0x18(SP) 4889442418
|
||||||
|
// POPQ BP 5d
|
||||||
|
// RET c3
|
||||||
|
want := []byte{
|
||||||
|
0x55,
|
||||||
|
0x48, 0x89, 0xe5,
|
||||||
|
0x48, 0x8b, 0x44, 0x24, 0x10,
|
||||||
|
0xe8, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x48, 0x89, 0x44, 0x24, 0x18,
|
||||||
|
0x5d,
|
||||||
|
0xc3,
|
||||||
|
}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("frameless CALL FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style
|
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style
|
||||||
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
|
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
|
||||||
// the bytes the Go assembler produces.
|
// the bytes the Go assembler produces.
|
||||||
@@ -202,8 +253,8 @@ TEXT ·withframe(SB), NOSPLIT, $16-16
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestAssembleVexKernel assembles the horizontal-sum reduction the go-flac
|
// TestAssembleVexKernel assembles the horizontal-sum reduction the go-flac
|
||||||
// kernels end with — exercising the VEX moves, shuffle and extract forms
|
// kernels end with; exercising the VEX moves, shuffle and extract forms
|
||||||
// through the full parser → encoder path — and checks the output is
|
// through the full parser → encoder path; and checks the output is
|
||||||
// byte-identical to the Go assembler's.
|
// byte-identical to the Go assembler's.
|
||||||
func TestAssembleVexKernel(t *testing.T) {
|
func TestAssembleVexKernel(t *testing.T) {
|
||||||
fn := firstText(t, `
|
fn := firstText(t, `
|
||||||
@@ -353,6 +404,19 @@ TEXT ·pf(SB), NOSPLIT, $0
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAssembleBareJump checks that a zero-operand jump (which parses, because
|
||||||
|
// the parser does not arity-check mnemonics) is rejected with an error rather
|
||||||
|
// than panicking in the layout loop, which indexes Operands[0] before the
|
||||||
|
// emission pass gets a chance to diagnose the arity.
|
||||||
|
func TestAssembleBareJump(t *testing.T) {
|
||||||
|
for _, mnem := range []string{"JE", "JMP", "JLT", "CALL"} {
|
||||||
|
fn := firstText(t, "TEXT ·bare(SB), $16-0\n\t"+mnem+"\n")
|
||||||
|
if _, _, err := Assemble(fn); err == nil {
|
||||||
|
t.Errorf("%s with no operand: expected an error, got none", mnem)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
|
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
|
||||||
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
|
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
|
||||||
// larger. The intermediate 129..255 range used to encode an ADD with a
|
// larger. The intermediate 129..255 range used to encode an ADD with a
|
||||||
|
|||||||
+49
-7
@@ -42,8 +42,10 @@ const (
|
|||||||
sttSection = 3
|
sttSection = 3
|
||||||
stInfoShift = 4
|
stInfoShift = 4
|
||||||
|
|
||||||
rX8664PC32 = 2
|
rX8664PC32 = 2
|
||||||
rX8664TPOFF32 = 20
|
// R_X86_64_TPOFF32 (debug/elf): the local-exec TLS offset the stack
|
||||||
|
// guard loads from FS. 20 is R_X86_64_TLSLD, a different relocation.
|
||||||
|
rX8664TPOFF32 = 23
|
||||||
)
|
)
|
||||||
|
|
||||||
// elfSym is one symbol-table entry in construction.
|
// elfSym is one symbol-table entry in construction.
|
||||||
@@ -219,7 +221,7 @@ func (img *Image) ELFObject() ([]byte, error) {
|
|||||||
for _, r := range relas {
|
for _, r := range relas {
|
||||||
var b [24]byte
|
var b [24]byte
|
||||||
le.PutUint64(b[0:], r.off)
|
le.PutUint64(b[0:], r.off)
|
||||||
le.PutUint64(b[8:], uint64(r.sym)<<32|rX8664PC32)
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||||
le.PutUint64(b[16:], uint64(r.addend))
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
out = append(out, b[:]...)
|
out = append(out, b[:]...)
|
||||||
}
|
}
|
||||||
@@ -228,15 +230,32 @@ func (img *Image) ELFObject() ([]byte, error) {
|
|||||||
shstrOff := len(out)
|
shstrOff := len(out)
|
||||||
out = append(out, stSections.bytes()...)
|
out = append(out, stSections.bytes()...)
|
||||||
|
|
||||||
// DWARF debug sections (no relocations, the linker resolves DWARF fixups).
|
// DWARF debug sections; the address placeholders they leave are carried
|
||||||
|
// as .rela.debug_info/.rela.debug_line entries the system linker applies.
|
||||||
dwAlign := func(n int) {
|
dwAlign := func(n int) {
|
||||||
for len(out)%n != 0 {
|
for len(out)%n != 0 {
|
||||||
out = append(out, 0)
|
out = append(out, 0)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
|
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiAMD64)
|
||||||
|
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
nSections += 4 // .debug_abbrev, .debug_info, .debug_line, .debug_line_str
|
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
|
||||||
|
// .debug_line_str and .debug_frame (the CIE is unconditional, so
|
||||||
|
// the frame section is always present), plus the relocation
|
||||||
|
// sections below when they carry entries.
|
||||||
|
dwarfStart = nSections
|
||||||
|
nSections += 5
|
||||||
|
appendDWARFRelas(&out, dw, rX8664Abs64, dwAlign)
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
align(8)
|
align(8)
|
||||||
@@ -267,14 +286,37 @@ func (img *Image) ELFObject() ([]byte, error) {
|
|||||||
}
|
}
|
||||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
|
|
||||||
// DWARF section headers.
|
// DWARF section headers; their indices follow the write order.
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
|
// secIdx is a running section index: each putSh below emits the
|
||||||
|
// next header, and the sh_info of a .rela section names the index
|
||||||
|
// of the section it relocates.
|
||||||
|
secIdx := dwarfStart
|
||||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||||
|
secInfoIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||||
|
secLineIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
if dw.frameSize > 0 {
|
if dw.frameSize > 0 {
|
||||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||||
|
secFrameIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+161
-74
@@ -12,43 +12,74 @@ import (
|
|||||||
// self-contained sections because the system linker only performs fixup
|
// self-contained sections because the system linker only performs fixup
|
||||||
// relocations, not assembly.
|
// relocations, not assembly.
|
||||||
|
|
||||||
|
// DWARF5 attribute, form and line-table constants (the values the
|
||||||
|
// toolchain uses, cmd/internal/dwarf/dwarf_defs.go; the DIE streams below
|
||||||
|
// are written against these forms).
|
||||||
|
const (
|
||||||
|
dwAtName = 0x03 // DW_AT_name
|
||||||
|
dwAtStmtList = 0x10 // DW_AT_stmt_list
|
||||||
|
dwAtLowPC = 0x11 // DW_AT_low_pc
|
||||||
|
dwAtHighPC = 0x12 // DW_AT_high_pc
|
||||||
|
dwAtDeclFile = 0x3a // DW_AT_decl_file
|
||||||
|
dwAtDeclLine = 0x3b // DW_AT_decl_line
|
||||||
|
dwAtExternal = 0x3f // DW_AT_external
|
||||||
|
dwAtFrameBase = 0x40 // DW_AT_frame_base
|
||||||
|
dwTagSubprog = 0x2e // DW_TAG_subprogram
|
||||||
|
dwTagCompUnit = 0x11 // DW_TAG_compile_unit
|
||||||
|
dwFormAddr = 0x01 // DW_FORM_addr
|
||||||
|
dwFormData8 = 0x07 // DW_FORM_data8
|
||||||
|
dwFormString = 0x08 // DW_FORM_string
|
||||||
|
dwFormData1 = 0x0b // DW_FORM_data1
|
||||||
|
dwFormUdata = 0x0f // DW_FORM_udata
|
||||||
|
dwFormSecOff = 0x17 // DW_FORM_sec_offset
|
||||||
|
dwFormExprloc = 0x18 // DW_FORM_exprloc
|
||||||
|
dwFormLineStrp = 0x1f // DW_FORM_line_strp
|
||||||
|
dwLnctPath = 0x01 // DW_LNCT_path
|
||||||
|
dwLnctDirIndex = 0x02 // DW_LNCT_directory_index
|
||||||
|
)
|
||||||
|
|
||||||
// dwarfAbbrevTable returns the .debug_abbrev content: a single compilation
|
// dwarfAbbrevTable returns the .debug_abbrev content: a single compilation
|
||||||
// unit with DW_TAG_compile_unit and DW_TAG_subprogram entries.
|
// unit with DW_TAG_compile_unit and DW_TAG_subprogram entries. The
|
||||||
|
// attribute/form pairs must match the DIE streams dwarfBuildInfoSection
|
||||||
|
// writes byte for byte, in the same order, or every consumer's parse of
|
||||||
|
// .debug_info desynchronises.
|
||||||
func dwarfAbbrevTable() []byte {
|
func dwarfAbbrevTable() []byte {
|
||||||
var b []byte
|
var b []byte
|
||||||
// Abbrev 1: DW_TAG_compile_unit
|
// Abbrev 1: DW_TAG_compile_unit.
|
||||||
b = append(b, 1) // abbreviation code
|
b = append(b, 1) // abbreviation code
|
||||||
b = append(b, 0x11) // DW_TAG_compile_unit
|
b = appendUleb(b, dwTagCompUnit) // DW_TAG_compile_unit
|
||||||
b = append(b, 1) // DW_CHILDREN_yes
|
b = append(b, 1) // DW_CHILDREN_yes
|
||||||
b = appendUleb(b, 0x1b) // DW_AT_low_pc
|
b = appendUleb(b, dwAtLowPC) // DW_AT_low_pc
|
||||||
b = appendUleb(b, 0x01) // DW_FORM_addr
|
b = appendUleb(b, dwFormAddr) // DW_FORM_addr
|
||||||
b = appendUleb(b, 0x29) // DW_AT_high_pc
|
b = appendUleb(b, dwAtHighPC) // DW_AT_high_pc
|
||||||
b = appendUleb(b, 0x07) // DW_FORM_data8
|
b = appendUleb(b, dwFormData8) // DW_FORM_data8
|
||||||
b = appendUleb(b, 0x10) // DW_AT_stmt_list
|
b = appendUleb(b, dwAtStmtList) // DW_AT_stmt_list
|
||||||
b = appendUleb(b, 0x25) // DW_FORM_sec_offset
|
b = appendUleb(b, dwFormSecOff) // DW_FORM_sec_offset (4 bytes here)
|
||||||
b = appendUleb(b, 0x01) // DW_AT_name
|
b = appendUleb(b, dwAtName) // DW_AT_name
|
||||||
b = appendUleb(b, 0x08) // DW_FORM_string
|
b = appendUleb(b, dwFormString) // DW_FORM_string
|
||||||
b = appendUleb(b, 0) // end of attributes
|
b = appendUleb(b, 0) // end of attributes: attr 0
|
||||||
|
b = appendUleb(b, 0) // ... paired with form 0
|
||||||
|
|
||||||
// Abbrev 2: DW_TAG_subprogram
|
// Abbrev 2: DW_TAG_subprogram.
|
||||||
b = append(b, 2) // abbreviation code
|
b = append(b, 2) // abbreviation code
|
||||||
b = append(b, 0x2e) // DW_TAG_subprogram
|
b = appendUleb(b, dwTagSubprog) // DW_TAG_subprogram
|
||||||
b = append(b, 0) // DW_CHILDREN_no
|
b = append(b, 0) // DW_CHILDREN_no
|
||||||
b = appendUleb(b, 0x03) // DW_AT_name
|
b = appendUleb(b, dwAtName) // DW_AT_name
|
||||||
b = appendUleb(b, 0x08) // DW_FORM_string
|
b = appendUleb(b, dwFormString) // DW_FORM_string
|
||||||
b = appendUleb(b, 0x11) // DW_AT_low_pc
|
b = appendUleb(b, dwAtLowPC) // DW_AT_low_pc
|
||||||
b = appendUleb(b, 0x01) // DW_FORM_addr
|
b = appendUleb(b, dwFormAddr) // DW_FORM_addr
|
||||||
b = appendUleb(b, 0x29) // DW_AT_high_pc
|
b = appendUleb(b, dwAtHighPC) // DW_AT_high_pc
|
||||||
b = appendUleb(b, 0x07) // DW_FORM_data8
|
b = appendUleb(b, dwFormData8) // DW_FORM_data8
|
||||||
b = appendUleb(b, 0x3f) // DW_AT_frame_base
|
b = appendUleb(b, dwAtFrameBase) // DW_AT_frame_base
|
||||||
b = appendUleb(b, 0x18) // DW_FORM_exprloc
|
b = appendUleb(b, dwFormExprloc) // DW_FORM_exprloc
|
||||||
b = appendUleb(b, 0x3b) // DW_AT_decl_file
|
b = appendUleb(b, dwAtDeclFile) // DW_AT_decl_file
|
||||||
b = appendUleb(b, 0x0b) // DW_FORM_data1
|
b = appendUleb(b, dwFormData1) // DW_FORM_data1
|
||||||
b = appendUleb(b, 0x37) // DW_AT_decl_line
|
b = appendUleb(b, dwAtDeclLine) // DW_AT_decl_line
|
||||||
b = appendUleb(b, 0x0b) // DW_FORM_data1
|
b = appendUleb(b, dwFormData1) // DW_FORM_data1
|
||||||
b = appendUleb(b, 0x63) // DW_AT_external
|
b = appendUleb(b, dwAtExternal) // DW_AT_external
|
||||||
b = appendUleb(b, 0x0b) // DW_FORM_flag
|
b = appendUleb(b, 0x0c) // DW_FORM_flag (one byte, 0 or 1)
|
||||||
b = appendUleb(b, 0) // end of attributes
|
b = appendUleb(b, 0) // end of attributes: attr 0
|
||||||
|
b = appendUleb(b, 0) // ... paired with form 0
|
||||||
|
|
||||||
// End of table.
|
// End of table.
|
||||||
b = append(b, 0)
|
b = append(b, 0)
|
||||||
@@ -68,6 +99,9 @@ type dwarfSections struct {
|
|||||||
infoRelocs []dwarfReloc
|
infoRelocs []dwarfReloc
|
||||||
// Relocations for .debug_line: (offset, symbol name, addend).
|
// Relocations for .debug_line: (offset, symbol name, addend).
|
||||||
lineRelocs []dwarfReloc
|
lineRelocs []dwarfReloc
|
||||||
|
// Relocations for .debug_frame: (offset, symbol name, addend), one per
|
||||||
|
// FDE initial_location.
|
||||||
|
frameRelocs []dwarfReloc
|
||||||
}
|
}
|
||||||
|
|
||||||
type dwarfReloc struct {
|
type dwarfReloc struct {
|
||||||
@@ -76,8 +110,9 @@ type dwarfReloc struct {
|
|||||||
addend int64
|
addend int64
|
||||||
}
|
}
|
||||||
|
|
||||||
// emitDWARF generates complete DWARF5 sections for the image.
|
// emitDWARF generates complete DWARF5 sections for the image. cfi carries
|
||||||
func emitDWARF(img *Image, srcFile string) *dwarfSections {
|
// the architecture's .debug_frame register conventions.
|
||||||
|
func emitDWARF(img *Image, srcFile string, cfi cfiArch) *dwarfSections {
|
||||||
ds := &dwarfSections{}
|
ds := &dwarfSections{}
|
||||||
ds.debugAbbrev = dwarfAbbrevTable()
|
ds.debugAbbrev = dwarfAbbrevTable()
|
||||||
|
|
||||||
@@ -86,19 +121,21 @@ func emitDWARF(img *Image, srcFile string) *dwarfSections {
|
|||||||
lineStr.add(srcFile)
|
lineStr.add(srcFile)
|
||||||
ds.debugLineStr = lineStr.bytes()
|
ds.debugLineStr = lineStr.bytes()
|
||||||
|
|
||||||
// Build .debug_line.
|
// Build .debug_line; the file table references the source name through
|
||||||
ds.debugLine = dwarfBuildLineSection(img, ds)
|
// its offset in .debug_line_str.
|
||||||
|
ds.debugLine = dwarfBuildLineSection(img, uint32(lineStr.at(srcFile)), ds)
|
||||||
|
|
||||||
// Build .debug_info.
|
// Build .debug_info.
|
||||||
ds.debugInfo = dwarfBuildInfoSection(img, srcFile, ds)
|
ds.debugInfo = dwarfBuildInfoSection(img, srcFile, ds)
|
||||||
|
|
||||||
// Build .debug_frame.
|
// Build .debug_frame.
|
||||||
ds.debugFrame = dwarfBuildFrameSection(img)
|
ds.debugFrame = dwarfBuildFrameSection(img, cfi, ds)
|
||||||
return ds
|
return ds
|
||||||
}
|
}
|
||||||
|
|
||||||
// dwarfBuildLineSection builds a complete .debug_line section.
|
// dwarfBuildLineSection builds a complete .debug_line section. srcStrOff is
|
||||||
func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
|
// the source file name's offset in .debug_line_str.
|
||||||
|
func dwarfBuildLineSection(img *Image, srcStrOff uint32, ds *dwarfSections) []byte {
|
||||||
var b []byte
|
var b []byte
|
||||||
le := binary.LittleEndian
|
le := binary.LittleEndian
|
||||||
|
|
||||||
@@ -120,15 +157,25 @@ func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
|
|||||||
// Standard opcode lengths (opcode 1..opcode_base-1).
|
// Standard opcode lengths (opcode 1..opcode_base-1).
|
||||||
b = append(b, 0, 1, 1, 1, 1, 0, 0, 0, 1, 0)
|
b = append(b, 0, 1, 1, 1, 1, 0, 0, 0, 1, 0)
|
||||||
|
|
||||||
// Directory table (DWARF5 format).
|
// Directory table (DWARF5 §6.2.4): entry format descriptors followed by
|
||||||
b = append(b, 0) // one directory entry (index 0 = empty)
|
// the entries. One directory, the compilation directory, whose path is
|
||||||
// File table.
|
// the empty string at .debug_line_str offset 0.
|
||||||
b = appendUleb(b, 1) // file count
|
b = append(b, 1) // directory_entry_format_count
|
||||||
// File 1: name index into .debug_line_str, dir index, time, size.
|
b = appendUleb(b, dwLnctPath) // DW_LNCT_path
|
||||||
b = appendUleb(b, 0) // name (index 0 in line_str)
|
b = appendUleb(b, dwFormLineStrp) // DW_FORM_line_strp
|
||||||
b = appendUleb(b, 0) // directory index
|
b = appendUleb(b, 1) // directories_count
|
||||||
b = appendUleb(b, 0) // last modification time
|
b = le.AppendUint32(b, 0) // .debug_line_str offset of ""
|
||||||
b = appendUleb(b, 0) // file size
|
|
||||||
|
// File table (DWARF5 §6.2.5). v5 indexes files from 0, so the source
|
||||||
|
// file is entry 0, matching the DW_AT_decl_file value 0 the DIEs carry.
|
||||||
|
b = append(b, 2) // file_name_entry_format_count
|
||||||
|
b = appendUleb(b, dwLnctPath) // DW_LNCT_path
|
||||||
|
b = appendUleb(b, dwFormLineStrp) // DW_FORM_line_strp
|
||||||
|
b = appendUleb(b, dwLnctDirIndex) // DW_LNCT_directory_index
|
||||||
|
b = appendUleb(b, dwFormUdata) // DW_FORM_udata
|
||||||
|
b = appendUleb(b, 1) // file_names_count
|
||||||
|
b = le.AppendUint32(b, srcStrOff) // .debug_line_str offset of the source name
|
||||||
|
b = appendUleb(b, 0) // directory index 0 (the compilation directory)
|
||||||
|
|
||||||
headerEnd := len(b)
|
headerEnd := len(b)
|
||||||
|
|
||||||
@@ -176,8 +223,11 @@ func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
|
|||||||
|
|
||||||
// Patch unit_length.
|
// Patch unit_length.
|
||||||
le.PutUint32(b[headerStart:], uint32(len(b)-headerStart-4))
|
le.PutUint32(b[headerStart:], uint32(len(b)-headerStart-4))
|
||||||
// Patch header_length.
|
// Patch header_length. In the v5 header it follows the one-byte
|
||||||
le.PutUint32(b[headerStart+6:], uint32(headerEnd-headerStart-10))
|
// address_size and segment_selector_size (offset 8, not the DWARF2-4
|
||||||
|
// offset 6), and counts from just past itself to the first program
|
||||||
|
// byte.
|
||||||
|
le.PutUint32(b[headerStart+8:], uint32(headerEnd-headerStart-12))
|
||||||
return b
|
return b
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -195,14 +245,16 @@ func dwarfBuildInfoSection(img *Image, srcFile string, ds *dwarfSections) []byte
|
|||||||
|
|
||||||
// DW_TAG_compile_unit (abbrev 1).
|
// DW_TAG_compile_unit (abbrev 1).
|
||||||
b = append(b, 1) // abbreviation code
|
b = append(b, 1) // abbreviation code
|
||||||
// DW_AT_low_pc: address of .text start.
|
// DW_AT_low_pc: address of .text start. A data-only image has no
|
||||||
infoRelocBase := len(b)
|
// functions to relocate against; its CU covers no code, so the base
|
||||||
|
// stays zero (the DWARF "no base address" value) with no relocation.
|
||||||
b = le.AppendUint64(b, 0) // placeholder
|
b = le.AppendUint64(b, 0) // placeholder
|
||||||
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
|
if len(img.Funcs) > 0 {
|
||||||
off: uint64(infoRelocBase),
|
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
|
||||||
name: img.Funcs[0].Name,
|
off: uint64(len(b) - 8),
|
||||||
addend: 0,
|
name: img.Funcs[0].Name,
|
||||||
})
|
})
|
||||||
|
}
|
||||||
// DW_AT_high_pc: size of .text.
|
// DW_AT_high_pc: size of .text.
|
||||||
b = le.AppendUint64(b, uint64(len(img.Code)))
|
b = le.AppendUint64(b, uint64(len(img.Code)))
|
||||||
// DW_AT_stmt_list: offset into .debug_line (0).
|
// DW_AT_stmt_list: offset into .debug_line (0).
|
||||||
@@ -229,8 +281,9 @@ func dwarfBuildInfoSection(img *Image, srcFile string, ds *dwarfSections) []byte
|
|||||||
b = le.AppendUint64(b, uint64(fn.Size))
|
b = le.AppendUint64(b, uint64(fn.Size))
|
||||||
// DW_AT_frame_base: DW_OP_call_frame_cfa.
|
// DW_AT_frame_base: DW_OP_call_frame_cfa.
|
||||||
b = append(b, 1, 0x9c)
|
b = append(b, 1, 0x9c)
|
||||||
// DW_AT_decl_file: file index 1.
|
// DW_AT_decl_file: the single file-table entry, index 0 (v5 indexes
|
||||||
b = append(b, 1)
|
// files from 0).
|
||||||
|
b = append(b, 0)
|
||||||
// DW_AT_decl_line.
|
// DW_AT_decl_line.
|
||||||
b = append(b, uint8(fn.Line))
|
b = append(b, uint8(fn.Line))
|
||||||
// DW_AT_external.
|
// DW_AT_external.
|
||||||
@@ -253,32 +306,61 @@ func appendUleb(b []byte, v uint64) []byte {
|
|||||||
return binary.AppendUvarint(b, v)
|
return binary.AppendUvarint(b, v)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// appendSleb appends v in signed LEB128, the encoding DWARF specifies:
|
||||||
|
// two's-complement sign extension, which is NOT Go's zigzag varint
|
||||||
|
// (binary.AppendVarint(-8) encodes 15, where DWARF wants 0x78).
|
||||||
func appendSleb(b []byte, v int64) []byte {
|
func appendSleb(b []byte, v int64) []byte {
|
||||||
return binary.AppendVarint(b, v)
|
for {
|
||||||
|
c := byte(v & 0x7f)
|
||||||
|
v >>= 7
|
||||||
|
if (v == 0 && c&0x40 == 0) || (v == -1 && c&0x40 != 0) {
|
||||||
|
return append(b, c)
|
||||||
|
}
|
||||||
|
b = append(b, c|0x80)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// cfiArch carries the .debug_frame CIE parameters that differ per
|
||||||
|
// architecture: the DWARF register numbers of the stack pointer the initial
|
||||||
|
// CFA rule names and of the return address. The values are the ones the Go
|
||||||
|
// linker writes into its own CIE (cmd/link/internal/ld/dwarf.go uses
|
||||||
|
// Dwarfregsp and Dwarfreglr; the per-architecture constants live in
|
||||||
|
// cmd/link/internal/<arch>/l.go).
|
||||||
|
type cfiArch struct {
|
||||||
|
name string
|
||||||
|
cfaReg byte // the stack-pointer register the initial CFA rule names
|
||||||
|
raReg byte // the return-address register
|
||||||
|
}
|
||||||
|
|
||||||
|
var (
|
||||||
|
cfiAMD64 = cfiArch{"amd64", 7, 16} // RSP, RIP
|
||||||
|
cfiARM64 = cfiArch{"arm64", 31, 30} // SP (X31), LR (X30)
|
||||||
|
cfiRISCV64 = cfiArch{"riscv64", 2, 1} // X2 (sp), X1 (ra)
|
||||||
|
cfiLOONG64 = cfiArch{"loong64", 3, 1} // $r3 (sp), $r1 (ra)
|
||||||
|
)
|
||||||
|
|
||||||
// dwarfBuildFrameSection builds a .debug_frame section with CFI for stack
|
// dwarfBuildFrameSection builds a .debug_frame section with CFI for stack
|
||||||
// unwinding. It emits one CIE and one FDE per function, encoding the
|
// unwinding. It emits one CIE and one FDE per function, encoding the
|
||||||
// CFA (Canonical Frame Address) rule changes at each stack-adjustment
|
// CFA (Canonical Frame Address) rule changes at each stack-adjustment
|
||||||
// boundary recorded in FuncLayout.Spadj.
|
// boundary recorded in FuncLayout.Spadj.
|
||||||
func dwarfBuildFrameSection(img *Image) []byte {
|
func dwarfBuildFrameSection(img *Image, cfi cfiArch, ds *dwarfSections) []byte {
|
||||||
var b []byte
|
var b []byte
|
||||||
le := binary.LittleEndian
|
le := binary.LittleEndian
|
||||||
|
|
||||||
// CIE (Common Information Entry).
|
// CIE (Common Information Entry).
|
||||||
cieStart := len(b)
|
cieStart := len(b)
|
||||||
b = append(b, 0, 0, 0, 0) // length (placeholder)
|
b = append(b, 0, 0, 0, 0) // length (placeholder)
|
||||||
b = le.AppendUint32(b, 0xFFFFFFFF) // CIE marker
|
b = le.AppendUint32(b, 0xFFFFFFFF) // CIE marker
|
||||||
b = append(b, 3) // version (DWARF3, widely supported)
|
b = append(b, 3) // version (DWARF3, widely supported)
|
||||||
b = append(b, 0) // augmentation (empty)
|
b = append(b, 0) // augmentation (empty)
|
||||||
b = appendUleb(b, 1) // code alignment
|
b = appendUleb(b, 1) // code alignment
|
||||||
b = appendSleb(b, -8) // data alignment (-8 for 64-bit)
|
b = appendSleb(b, -8) // data alignment (-8 for 64-bit)
|
||||||
b = appendUleb(b, 16) // return address register (LR on arm64, RIP on amd64)
|
b = appendUleb(b, uint64(cfi.raReg)) // return address register
|
||||||
// Initial CFA rule: DW_CFA_def_cfa (SP, 0)
|
// Initial CFA rule: DW_CFA_def_cfa (SP, 0)
|
||||||
b = append(b, 0x0c) // DW_CFA_def_cfa
|
b = append(b, 0x0c) // DW_CFA_def_cfa
|
||||||
b = appendUleb(b, 31) // register: SP (RSP=7 on amd64, SP=31 on arm64)
|
b = appendUleb(b, uint64(cfi.cfaReg)) // the architecture's stack pointer
|
||||||
b = appendUleb(b, 0) // offset: 0
|
b = appendUleb(b, 0) // offset: 0
|
||||||
b = append(b, 0) // DW_CFA_nop (padding)
|
b = append(b, 0) // DW_CFA_nop (padding)
|
||||||
// Patch CIE length.
|
// Patch CIE length.
|
||||||
le.PutUint32(b[cieStart:], uint32(len(b)-cieStart-4))
|
le.PutUint32(b[cieStart:], uint32(len(b)-cieStart-4))
|
||||||
|
|
||||||
@@ -287,7 +369,12 @@ func dwarfBuildFrameSection(img *Image) []byte {
|
|||||||
fdeStart := len(b)
|
fdeStart := len(b)
|
||||||
b = append(b, 0, 0, 0, 0) // length (placeholder)
|
b = append(b, 0, 0, 0, 0) // length (placeholder)
|
||||||
b = le.AppendUint32(b, uint32(cieStart)) // CIE pointer (offset from start)
|
b = le.AppendUint32(b, uint32(cieStart)) // CIE pointer (offset from start)
|
||||||
// Initial location: function offset in .text (relocated by linker).
|
// Initial location: function offset in .text, referenced through
|
||||||
|
// the function's symbol so the linker relocates it.
|
||||||
|
ds.frameRelocs = append(ds.frameRelocs, dwarfReloc{
|
||||||
|
off: uint64(fdeStart + 8),
|
||||||
|
name: fn.Name,
|
||||||
|
})
|
||||||
b = le.AppendUint64(b, uint64(fn.Offset))
|
b = le.AppendUint64(b, uint64(fn.Offset))
|
||||||
// Address range: function size.
|
// Address range: function size.
|
||||||
b = le.AppendUint64(b, uint64(fn.Size))
|
b = le.AppendUint64(b, uint64(fn.Size))
|
||||||
|
|||||||
+84
-24
@@ -3,6 +3,17 @@
|
|||||||
|
|
||||||
package asm
|
package asm
|
||||||
|
|
||||||
|
import "encoding/binary"
|
||||||
|
|
||||||
|
// Absolute 64-bit relocation types for the DWARF address fixups, one per
|
||||||
|
// supported architecture (the numbers debug/elf carries).
|
||||||
|
const (
|
||||||
|
rX8664Abs64 = 1 // R_X86_64_64
|
||||||
|
rAARCH64Abs64 = 257 // R_AARCH64_ABS64
|
||||||
|
rRISCVAbs64 = 2 // R_RISCV_64
|
||||||
|
rLarchAbs64 = 2 // R_LARCH_64
|
||||||
|
)
|
||||||
|
|
||||||
// dwarfELFSections holds the laid-out DWARF sections ready for inclusion
|
// dwarfELFSections holds the laid-out DWARF sections ready for inclusion
|
||||||
// in an ELF file.
|
// in an ELF file.
|
||||||
type dwarfELFSections struct {
|
type dwarfELFSections struct {
|
||||||
@@ -11,15 +22,23 @@ type dwarfELFSections struct {
|
|||||||
lineOff, lineSize int
|
lineOff, lineSize int
|
||||||
lineStrOff, lineStrSize int
|
lineStrOff, lineStrSize int
|
||||||
frameOff, frameSize int
|
frameOff, frameSize int
|
||||||
// Relocations for .debug_info address references.
|
// .rela.debug_info and .rela.debug_line contents: file offsets and
|
||||||
|
// entry counts (zero count: the section is absent).
|
||||||
|
infoRelaOff, infoRelaCount int
|
||||||
|
lineRelaOff, lineRelaCount int
|
||||||
|
frameRelaOff, frameRelaCount int
|
||||||
|
// Relocations for .debug_info address references, offsets relative to
|
||||||
|
// the section start (what an r_offset in .rela.debug_info means).
|
||||||
infoRelocs []elfDwarfReloc
|
infoRelocs []elfDwarfReloc
|
||||||
// Relocations for .debug_line address references.
|
// Relocations for .debug_line address references, section-relative.
|
||||||
lineRelocs []elfDwarfReloc
|
lineRelocs []elfDwarfReloc
|
||||||
|
// Relocations for .debug_frame FDE initial locations, section-relative.
|
||||||
|
frameRelocs []elfDwarfReloc
|
||||||
}
|
}
|
||||||
|
|
||||||
type elfDwarfReloc struct {
|
type elfDwarfReloc struct {
|
||||||
off uint64
|
off uint64 // offset within the target section
|
||||||
sym int // symbol index in .symtab
|
sym int // symbol index in .symtab
|
||||||
addend int64
|
addend int64
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -29,8 +48,9 @@ type elfDwarfReloc struct {
|
|||||||
//
|
//
|
||||||
// symIdx maps function names to their .symtab indices (needed for relocations
|
// symIdx maps function names to their .symtab indices (needed for relocations
|
||||||
// against .text symbols). The map uses objectName format (pkg.name); the
|
// against .text symbols). The map uses objectName format (pkg.name); the
|
||||||
// DWARF code uses bare function names, so we build a reverse lookup.
|
// DWARF code uses bare function names, so we build a reverse lookup. cfi
|
||||||
func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[string]int, align func(int)) *dwarfELFSections {
|
// carries the architecture's .debug_frame register conventions.
|
||||||
|
func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[string]int, align func(int), cfi cfiArch) *dwarfELFSections {
|
||||||
// Build a lookup from bare function name to symbol index.
|
// Build a lookup from bare function name to symbol index.
|
||||||
nameToIdx := make(map[string]int, len(symIdx))
|
nameToIdx := make(map[string]int, len(symIdx))
|
||||||
for name, idx := range symIdx {
|
for name, idx := range symIdx {
|
||||||
@@ -45,7 +65,7 @@ func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[str
|
|||||||
}
|
}
|
||||||
nameToIdx[name] = idx
|
nameToIdx[name] = idx
|
||||||
}
|
}
|
||||||
ds := emitDWARF(img, srcFile)
|
ds := emitDWARF(img, srcFile, cfi)
|
||||||
if ds == nil || len(ds.debugAbbrev) == 0 {
|
if ds == nil || len(ds.debugAbbrev) == 0 {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
@@ -68,15 +88,11 @@ func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[str
|
|||||||
align(1)
|
align(1)
|
||||||
result.lineOff = len(*out)
|
result.lineOff = len(*out)
|
||||||
result.lineSize = len(ds.debugLine)
|
result.lineSize = len(ds.debugLine)
|
||||||
lineBase := len(*out)
|
|
||||||
*out = append(*out, ds.debugLine...)
|
*out = append(*out, ds.debugLine...)
|
||||||
|
|
||||||
// Patch .debug_line relocations: replace placeholder addresses with
|
|
||||||
// actual .text offsets via symbol lookup.
|
|
||||||
for _, dr := range ds.lineRelocs {
|
for _, dr := range ds.lineRelocs {
|
||||||
if idx, ok := nameToIdx[dr.name]; ok {
|
if idx, ok := nameToIdx[dr.name]; ok {
|
||||||
result.lineRelocs = append(result.lineRelocs, elfDwarfReloc{
|
result.lineRelocs = append(result.lineRelocs, elfDwarfReloc{
|
||||||
off: uint64(lineBase) + dr.off,
|
off: dr.off,
|
||||||
sym: idx,
|
sym: idx,
|
||||||
addend: dr.addend,
|
addend: dr.addend,
|
||||||
})
|
})
|
||||||
@@ -87,33 +103,77 @@ func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[str
|
|||||||
align(1)
|
align(1)
|
||||||
result.infoOff = len(*out)
|
result.infoOff = len(*out)
|
||||||
result.infoSize = len(ds.debugInfo)
|
result.infoSize = len(ds.debugInfo)
|
||||||
infoBase := len(*out)
|
|
||||||
*out = append(*out, ds.debugInfo...)
|
*out = append(*out, ds.debugInfo...)
|
||||||
|
|
||||||
// .debug_frame
|
|
||||||
if len(ds.debugFrame) > 0 {
|
|
||||||
align(1)
|
|
||||||
result.frameOff = len(*out)
|
|
||||||
result.frameSize = len(ds.debugFrame)
|
|
||||||
*out = append(*out, ds.debugFrame...)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Patch .debug_info relocations.
|
|
||||||
for _, dr := range ds.infoRelocs {
|
for _, dr := range ds.infoRelocs {
|
||||||
if idx, ok := nameToIdx[dr.name]; ok {
|
if idx, ok := nameToIdx[dr.name]; ok {
|
||||||
result.infoRelocs = append(result.infoRelocs, elfDwarfReloc{
|
result.infoRelocs = append(result.infoRelocs, elfDwarfReloc{
|
||||||
off: uint64(infoBase) + dr.off,
|
off: dr.off,
|
||||||
sym: idx,
|
sym: idx,
|
||||||
addend: dr.addend,
|
addend: dr.addend,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// .debug_frame: the section header declares alignment 8, so the data is
|
||||||
|
// padded to 8, matching it.
|
||||||
|
if len(ds.debugFrame) > 0 {
|
||||||
|
align(8)
|
||||||
|
result.frameOff = len(*out)
|
||||||
|
result.frameSize = len(ds.debugFrame)
|
||||||
|
*out = append(*out, ds.debugFrame...)
|
||||||
|
for _, dr := range ds.frameRelocs {
|
||||||
|
if idx, ok := nameToIdx[dr.name]; ok {
|
||||||
|
result.frameRelocs = append(result.frameRelocs, elfDwarfReloc{
|
||||||
|
off: dr.off,
|
||||||
|
sym: idx,
|
||||||
|
addend: dr.addend,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
return result
|
return result
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// appendDWARFRelas writes the .rela.debug_info and .rela.debug_line section
|
||||||
|
// bodies from the relocations appendDWARFSections recorded, with the
|
||||||
|
// architecture's absolute 64-bit relocation type, and records their file
|
||||||
|
// offsets and entry counts on dw. Called after the DWARF sections
|
||||||
|
// themselves so the r_offsets (section-relative) need no adjustment.
|
||||||
|
func appendDWARFRelas(out *[]byte, dw *dwarfELFSections, abs64 uint32, align func(int)) {
|
||||||
|
le := binary.LittleEndian
|
||||||
|
write := func(relas []elfDwarfReloc) (off, count int) {
|
||||||
|
if len(relas) == 0 {
|
||||||
|
return 0, 0
|
||||||
|
}
|
||||||
|
align(8)
|
||||||
|
off = len(*out)
|
||||||
|
for _, r := range relas {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint64(b[0:], r.off)
|
||||||
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(abs64))
|
||||||
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
|
*out = append(*out, b[:]...)
|
||||||
|
}
|
||||||
|
return off, len(relas)
|
||||||
|
}
|
||||||
|
dw.infoRelaOff, dw.infoRelaCount = write(dw.infoRelocs)
|
||||||
|
dw.lineRelaOff, dw.lineRelaCount = write(dw.lineRelocs)
|
||||||
|
dw.frameRelaOff, dw.frameRelaCount = write(dw.frameRelocs)
|
||||||
|
}
|
||||||
|
|
||||||
|
// dwarfSourceName returns the source name the DWARF sections record: the
|
||||||
|
// image's source path when the assembler captured one, "gasm.s" otherwise.
|
||||||
|
func dwarfSourceName(img *Image) string {
|
||||||
|
if img.SourcePath != "" {
|
||||||
|
return img.SourcePath
|
||||||
|
}
|
||||||
|
return "gasm.s"
|
||||||
|
}
|
||||||
|
|
||||||
// dwarfSectionNames returns the DWARF section names for the string table.
|
// dwarfSectionNames returns the DWARF section names for the string table.
|
||||||
var dwarfSectionNames = []string{
|
var dwarfSectionNames = []string{
|
||||||
".debug_abbrev", ".debug_info", ".debug_line", ".debug_line_str",
|
".debug_abbrev", ".debug_info", ".debug_line", ".debug_line_str",
|
||||||
".debug_frame", ".rela.debug_info", ".rela.debug_line",
|
".debug_frame", ".rela.debug_info", ".rela.debug_line",
|
||||||
|
".rela.debug_frame",
|
||||||
}
|
}
|
||||||
|
|||||||
+303
-12
@@ -4,11 +4,313 @@
|
|||||||
package asm
|
package asm
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/binary"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// ulebIter reads ULEB128 values, the .debug_abbrev and line-header
|
||||||
|
// encoding.
|
||||||
|
type ulebIter struct {
|
||||||
|
b []byte
|
||||||
|
i int
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *ulebIter) uleb(t *testing.T) uint64 {
|
||||||
|
t.Helper()
|
||||||
|
v, n := binary.Uvarint(r.b[r.i:])
|
||||||
|
if n <= 0 {
|
||||||
|
t.Fatalf("bad ULEB at %d", r.i)
|
||||||
|
}
|
||||||
|
r.i += n
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *ulebIter) byteAt(t *testing.T) byte {
|
||||||
|
t.Helper()
|
||||||
|
if r.i >= len(r.b) {
|
||||||
|
t.Fatalf("read past end at %d", r.i)
|
||||||
|
}
|
||||||
|
c := r.b[r.i]
|
||||||
|
r.i++
|
||||||
|
return c
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *ulebIter) uint32At(t *testing.T) uint32 {
|
||||||
|
t.Helper()
|
||||||
|
v := binary.LittleEndian.Uint32(r.b[r.i:])
|
||||||
|
r.i += 4
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
|
||||||
|
// sleb reads a signed LEB128, the DWARF encoding (sign-extended two's
|
||||||
|
// complement, not Go's zigzag varint).
|
||||||
|
func (r *ulebIter) sleb(t *testing.T) int64 {
|
||||||
|
t.Helper()
|
||||||
|
var v int64
|
||||||
|
var shift uint
|
||||||
|
for {
|
||||||
|
c := r.byteAt(t)
|
||||||
|
v |= int64(c&0x7f) << shift
|
||||||
|
shift += 7
|
||||||
|
if c&0x80 == 0 {
|
||||||
|
if c&0x40 != 0 {
|
||||||
|
v |= -1 << shift
|
||||||
|
}
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// dwarfAttr is one attribute/form pair of an abbreviation.
|
||||||
|
type dwarfAttr struct{ attr, form uint64 }
|
||||||
|
|
||||||
|
// dwarfAbbrev is one parsed abbreviation declaration.
|
||||||
|
type dwarfAbbrev struct {
|
||||||
|
code uint64
|
||||||
|
tag uint64
|
||||||
|
children bool
|
||||||
|
attrs []dwarfAttr
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseAbbrevs walks a .debug_abbrev table: abbreviation code, tag,
|
||||||
|
// children flag, then attr/form ULEB pairs terminated by a double zero.
|
||||||
|
func parseAbbrevs(t *testing.T, b []byte) map[uint64]dwarfAbbrev {
|
||||||
|
t.Helper()
|
||||||
|
out := map[uint64]dwarfAbbrev{}
|
||||||
|
r := &ulebIter{b: b}
|
||||||
|
for {
|
||||||
|
code := r.uleb(t)
|
||||||
|
if code == 0 {
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
ab := dwarfAbbrev{code: code, tag: r.uleb(t)}
|
||||||
|
ab.children = r.byteAt(t) == 1
|
||||||
|
for {
|
||||||
|
attr := r.uleb(t)
|
||||||
|
form := r.uleb(t)
|
||||||
|
if attr == 0 && form == 0 {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
if attr == 0 || form == 0 {
|
||||||
|
t.Fatalf("abbrev %d: half-terminated attr/form pair (%d, %d)", code, attr, form)
|
||||||
|
}
|
||||||
|
ab.attrs = append(ab.attrs, dwarfAttr{attr, form})
|
||||||
|
}
|
||||||
|
out[code] = ab
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func eqAttrs(t *testing.T, ab dwarfAbbrev, want []dwarfAttr) {
|
||||||
|
t.Helper()
|
||||||
|
if len(ab.attrs) != len(want) {
|
||||||
|
t.Fatalf("abbrev %d attrs = %v, want %v", ab.code, ab.attrs, want)
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
if ab.attrs[i] != w {
|
||||||
|
t.Fatalf("abbrev %d attr %d = (%#x, %#x), want (%#x, %#x)", ab.code, i, ab.attrs[i].attr, ab.attrs[i].form, w.attr, w.form)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDwarfAbbrevTable walks the abbreviation table as a consumer does and
|
||||||
|
// checks the attribute/form sets against the constants the toolchain uses
|
||||||
|
// (cmd/internal/dwarf/dwarf_defs.go). A wrong constant here renames an
|
||||||
|
// attribute (0x1b is comp_dir, not low_pc; 0x29 and 0x37 are bounds and
|
||||||
|
// count) and a wrong form desynchronises the DIE parse: 0x25 is strx1, one
|
||||||
|
// byte, where the writer emits four for a section offset.
|
||||||
|
func TestDwarfAbbrevTable(t *testing.T) {
|
||||||
|
abbrev := dwarfAbbrevTable()
|
||||||
|
if len(abbrev) == 0 {
|
||||||
|
t.Fatal("empty abbrev table")
|
||||||
|
}
|
||||||
|
// Must end with a zero byte (end of table).
|
||||||
|
if abbrev[len(abbrev)-1] != 0 {
|
||||||
|
t.Fatalf("abbrev table last byte = %d, want 0", abbrev[len(abbrev)-1])
|
||||||
|
}
|
||||||
|
abs := parseAbbrevs(t, abbrev)
|
||||||
|
if len(abs) != 2 {
|
||||||
|
t.Fatalf("abbreviations = %d, want 2", len(abs))
|
||||||
|
}
|
||||||
|
cu, ok := abs[1]
|
||||||
|
if !ok {
|
||||||
|
t.Fatal("missing abbreviation 1 (compile unit)")
|
||||||
|
}
|
||||||
|
if cu.tag != dwTagCompUnit || !cu.children {
|
||||||
|
t.Errorf("abbrev 1: tag %#x children %v, want compile unit with children", cu.tag, cu.children)
|
||||||
|
}
|
||||||
|
eqAttrs(t, cu, []dwarfAttr{
|
||||||
|
{dwAtLowPC, dwFormAddr},
|
||||||
|
{dwAtHighPC, dwFormData8},
|
||||||
|
{dwAtStmtList, dwFormSecOff},
|
||||||
|
{dwAtName, dwFormString},
|
||||||
|
})
|
||||||
|
sp, ok := abs[2]
|
||||||
|
if !ok {
|
||||||
|
t.Fatal("missing abbreviation 2 (subprogram)")
|
||||||
|
}
|
||||||
|
if sp.tag != dwTagSubprog || sp.children {
|
||||||
|
t.Errorf("abbrev 2: tag %#x children %v, want subprogram without children", sp.tag, sp.children)
|
||||||
|
}
|
||||||
|
eqAttrs(t, sp, []dwarfAttr{
|
||||||
|
{dwAtName, dwFormString},
|
||||||
|
{dwAtLowPC, dwFormAddr},
|
||||||
|
{dwAtHighPC, dwFormData8},
|
||||||
|
{dwAtFrameBase, dwFormExprloc},
|
||||||
|
{dwAtDeclFile, dwFormData1},
|
||||||
|
{dwAtDeclLine, dwFormData1},
|
||||||
|
{dwAtExternal, 0x0c}, // DW_FORM_flag
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDwarfLineHeaderV5 parses the .debug_line header under DWARF5 rules:
|
||||||
|
// the directory and file tables are format-descriptor lists, not the
|
||||||
|
// DWARF2-4 shape of null-terminated strings, and the file entry references
|
||||||
|
// the source name through .debug_line_str.
|
||||||
|
func TestDwarfLineHeaderV5(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ a+0(FP), AX
|
||||||
|
MOVQ b+8(FP), BX
|
||||||
|
ADDQ BX, AX
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("test_amd64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
|
||||||
|
r := &ulebIter{b: ds.debugLine}
|
||||||
|
r.uint32At(t) // unit_length
|
||||||
|
if v := binary.LittleEndian.Uint16(ds.debugLine[4:]); v != 5 {
|
||||||
|
t.Fatalf("version = %d, want 5", v)
|
||||||
|
}
|
||||||
|
r.i = 6
|
||||||
|
r.byteAt(t) // address_size
|
||||||
|
r.byteAt(t) // segment_selector_size
|
||||||
|
r.uint32At(t) // header_length
|
||||||
|
r.byteAt(t) // minimum_instruction_length
|
||||||
|
r.byteAt(t) // maximum_ops_per_instruction
|
||||||
|
r.byteAt(t) // default_is_stmt
|
||||||
|
r.byteAt(t) // line_base
|
||||||
|
r.byteAt(t) // line_range
|
||||||
|
opcodeBase := r.byteAt(t)
|
||||||
|
for range int(opcodeBase) - 1 {
|
||||||
|
r.byteAt(t) // standard opcode lengths
|
||||||
|
}
|
||||||
|
|
||||||
|
// Directory table (DWARF5 §6.2.4).
|
||||||
|
if n := r.byteAt(t); n != 1 {
|
||||||
|
t.Fatalf("directory_entry_format_count = %d, want 1", n)
|
||||||
|
}
|
||||||
|
if lnct := r.uleb(t); lnct != dwLnctPath {
|
||||||
|
t.Errorf("directory content type = %#x, want DW_LNCT_path", lnct)
|
||||||
|
}
|
||||||
|
if form := r.uleb(t); form != dwFormLineStrp {
|
||||||
|
t.Errorf("directory form = %#x, want DW_FORM_line_strp", form)
|
||||||
|
}
|
||||||
|
if n := r.uleb(t); n != 1 {
|
||||||
|
t.Fatalf("directories_count = %d, want 1", n)
|
||||||
|
}
|
||||||
|
if off := r.uint32At(t); off != 0 {
|
||||||
|
t.Errorf("compilation directory line_strp = %d, want 0 (the empty string)", off)
|
||||||
|
}
|
||||||
|
|
||||||
|
// File table (DWARF5 §6.2.5).
|
||||||
|
if n := r.byteAt(t); n != 2 {
|
||||||
|
t.Fatalf("file_name_entry_format_count = %d, want 2", n)
|
||||||
|
}
|
||||||
|
if lnct := r.uleb(t); lnct != dwLnctPath {
|
||||||
|
t.Errorf("file content type = %#x, want DW_LNCT_path", lnct)
|
||||||
|
}
|
||||||
|
if form := r.uleb(t); form != dwFormLineStrp {
|
||||||
|
t.Errorf("file path form = %#x, want DW_FORM_line_strp", form)
|
||||||
|
}
|
||||||
|
if lnct := r.uleb(t); lnct != dwLnctDirIndex {
|
||||||
|
t.Errorf("file content type = %#x, want DW_LNCT_directory_index", lnct)
|
||||||
|
}
|
||||||
|
if form := r.uleb(t); form != dwFormUdata {
|
||||||
|
t.Errorf("file dir-index form = %#x, want DW_FORM_udata", form)
|
||||||
|
}
|
||||||
|
if n := r.uleb(t); n != 1 {
|
||||||
|
t.Fatalf("file_names_count = %d, want 1", n)
|
||||||
|
}
|
||||||
|
strOff := r.uint32At(t)
|
||||||
|
if dirIdx := r.uleb(t); dirIdx != 0 {
|
||||||
|
t.Errorf("file directory index = %d, want 0", dirIdx)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The file entry's line_strp must resolve to the source name.
|
||||||
|
end := int(strOff) + len("test_amd64.s")
|
||||||
|
if int(strOff) >= len(ds.debugLineStr) || !bytes.Equal(ds.debugLineStr[strOff:end], []byte("test_amd64.s")) {
|
||||||
|
t.Errorf("file entry line_strp %d does not name the source: %q", strOff, ds.debugLineStr)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The fixed header fields: address_size 8 and a header_length that
|
||||||
|
// points just past the file table (the patch site is offset 8 in the
|
||||||
|
// v5 header, and the field counts from its own end).
|
||||||
|
if ds.debugLine[6] != 8 || ds.debugLine[7] != 0 {
|
||||||
|
t.Errorf("address_size/segment_selector = %d/%d, want 8/0", ds.debugLine[6], ds.debugLine[7])
|
||||||
|
}
|
||||||
|
if hl := binary.LittleEndian.Uint32(ds.debugLine[8:]); hl != uint32(r.i-12) {
|
||||||
|
t.Errorf("header_length = %d, want %d (the byte after the file table is %d)", hl, r.i-12, r.i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDwarfFrameCIEArch checks the shared CIE carries each architecture's
|
||||||
|
// stack-pointer and return-address registers: the values the Go linker
|
||||||
|
// writes (cmd/link/internal/<arch>/l.go dwarfRegSP/dwarfRegLR).
|
||||||
|
func TestDwarfFrameCIEArch(t *testing.T) {
|
||||||
|
for _, tc := range []struct {
|
||||||
|
name string
|
||||||
|
cfi cfiArch
|
||||||
|
}{
|
||||||
|
{"amd64", cfiAMD64},
|
||||||
|
{"arm64", cfiARM64},
|
||||||
|
{"riscv64", cfiRISCV64},
|
||||||
|
{"loong64", cfiLOONG64},
|
||||||
|
} {
|
||||||
|
frame := dwarfBuildFrameSection(&Image{}, tc.cfi, &dwarfSections{})
|
||||||
|
r := &ulebIter{b: frame}
|
||||||
|
r.uint32At(t) // length
|
||||||
|
if cid := r.uint32At(t); cid != 0xFFFFFFFF {
|
||||||
|
t.Errorf("%s: CIE id = %#x, want 0xffffffff", tc.name, cid)
|
||||||
|
}
|
||||||
|
if v := r.byteAt(t); v != 3 {
|
||||||
|
t.Errorf("%s: CIE version = %d, want 3", tc.name, v)
|
||||||
|
}
|
||||||
|
if aug := r.byteAt(t); aug != 0 {
|
||||||
|
t.Errorf("%s: CIE augmentation = %d, want 0", tc.name, aug)
|
||||||
|
}
|
||||||
|
if ca := r.uleb(t); ca != 1 {
|
||||||
|
t.Errorf("%s: code alignment = %d, want 1", tc.name, ca)
|
||||||
|
}
|
||||||
|
if da := r.sleb(t); da != -8 {
|
||||||
|
t.Errorf("%s: data alignment = %d, want -8 (signed LEB128, not zigzag)", tc.name, da)
|
||||||
|
}
|
||||||
|
if ra := r.uleb(t); ra != uint64(tc.cfi.raReg) {
|
||||||
|
t.Errorf("%s: return-address register = %d, want %d", tc.name, ra, tc.cfi.raReg)
|
||||||
|
}
|
||||||
|
if op := r.byteAt(t); op != 0x0c {
|
||||||
|
t.Errorf("%s: expected DW_CFA_def_cfa, got opcode %#x", tc.name, op)
|
||||||
|
}
|
||||||
|
if cfa := r.uleb(t); cfa != uint64(tc.cfi.cfaReg) {
|
||||||
|
t.Errorf("%s: CFA register = %d, want %d", tc.name, cfa, tc.cfi.cfaReg)
|
||||||
|
}
|
||||||
|
if off := r.uleb(t); off != 0 {
|
||||||
|
t.Errorf("%s: CFA offset = %d, want 0", tc.name, off)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestEmitDWARF(t *testing.T) {
|
func TestEmitDWARF(t *testing.T) {
|
||||||
src := `#include "textflag.h"
|
src := `#include "textflag.h"
|
||||||
TEXT ·add(SB), NOSPLIT, $0-24
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
@@ -27,7 +329,7 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
t.Fatalf("assemble: %v", err)
|
t.Fatalf("assemble: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
ds := emitDWARF(img, "test_amd64.s")
|
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
|
||||||
|
|
||||||
// .debug_abbrev must not be empty and must start with abbrev code 1.
|
// .debug_abbrev must not be empty and must start with abbrev code 1.
|
||||||
if len(ds.debugAbbrev) == 0 {
|
if len(ds.debugAbbrev) == 0 {
|
||||||
@@ -68,14 +370,3 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
t.Fatal("no .debug_info relocations")
|
t.Fatal("no .debug_info relocations")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestDwarfAbbrevTable(t *testing.T) {
|
|
||||||
abbrev := dwarfAbbrevTable()
|
|
||||||
if len(abbrev) == 0 {
|
|
||||||
t.Fatal("empty abbrev table")
|
|
||||||
}
|
|
||||||
// Must end with a zero byte (end of table).
|
|
||||||
if abbrev[len(abbrev)-1] != 0 {
|
|
||||||
t.Fatalf("abbrev table last byte = %d, want 0", abbrev[len(abbrev)-1])
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
+443
-1
@@ -12,6 +12,7 @@ import (
|
|||||||
"path/filepath"
|
"path/filepath"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -52,7 +53,7 @@ func elfTestImage(t *testing.T) *Image {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestAssembleFileExternals checks that a reference to a symbol no GLOBL
|
// TestAssembleFileExternals checks that a reference to a symbol no GLOBL
|
||||||
// defines is recorded as an external relocation instead of failing — the
|
// defines is recorded as an external relocation instead of failing; the
|
||||||
// raw image leaves the displacement zero, the object emitters carry it.
|
// raw image leaves the displacement zero, the object emitters carry it.
|
||||||
func TestAssembleFileExternals(t *testing.T) {
|
func TestAssembleFileExternals(t *testing.T) {
|
||||||
img := elfTestImage(t)
|
img := elfTestImage(t)
|
||||||
@@ -211,6 +212,75 @@ func TestELFObject(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestELFObjectTLSGuardReloc checks that a non-NOSPLIT function's stack
|
||||||
|
// guard carries an R_X86_64_TPOFF32 relocation against the null symbol in
|
||||||
|
// .rela.text. The serialisation must honour the record's type field: a
|
||||||
|
// hardcoded R_X86_64_PC32 mislinks the TLS load as an ordinary
|
||||||
|
// PC-relative reference.
|
||||||
|
func TestELFObjectTLSGuardReloc(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("g_amd64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·grow(SB), $0
|
||||||
|
CALL ·other(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·other(SB), NOSPLIT, $0
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
var haveTLS bool
|
||||||
|
for _, fn := range img.Funcs {
|
||||||
|
for _, r := range fn.Relocs {
|
||||||
|
if r.Kind == RelTLSLE {
|
||||||
|
haveTLS = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !haveTLS {
|
||||||
|
t.Fatal("test source produced no RelTLSLE relocation")
|
||||||
|
}
|
||||||
|
obj, err := img.ELFObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFObject: %v", err)
|
||||||
|
}
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
relaSec := ef.Section(".rela.text")
|
||||||
|
if relaSec == nil {
|
||||||
|
t.Fatal("missing .rela.text")
|
||||||
|
}
|
||||||
|
raw, err := relaSec.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
found := false
|
||||||
|
for i := 0; i+24 <= len(raw); i += 24 {
|
||||||
|
e := raw[i:]
|
||||||
|
info := binary.LittleEndian.Uint64(e[8:])
|
||||||
|
typ := info & 0xffffffff
|
||||||
|
sym := int(info >> 32)
|
||||||
|
if typ == uint64(elf.R_X86_64_TPOFF32) {
|
||||||
|
found = true
|
||||||
|
if sym != 0 {
|
||||||
|
t.Errorf("TPOFF32 relocation against symbol %d, want 0 (the null symbol)", sym)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !found {
|
||||||
|
t.Errorf("no R_X86_64_TPOFF32 relocation in .rela.text (%d bytes)", len(raw))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestELFObjectNoRelocations checks a file with no static-symbol references
|
// TestELFObjectNoRelocations checks a file with no static-symbol references
|
||||||
// emits a valid object without a .rela.text section.
|
// emits a valid object without a .rela.text section.
|
||||||
func TestELFObjectNoRelocations(t *testing.T) {
|
func TestELFObjectNoRelocations(t *testing.T) {
|
||||||
@@ -253,6 +323,238 @@ TEXT ·nop(SB), NOSPLIT, $0
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// elfSectionHeaderCount returns the e_shnum the ELF header declares.
|
||||||
|
func elfSectionHeaderCount(t *testing.T, obj []byte) int {
|
||||||
|
t.Helper()
|
||||||
|
return int(binary.LittleEndian.Uint16(obj[60:]))
|
||||||
|
}
|
||||||
|
|
||||||
|
// checkELFSectionAccounting verifies the number of section headers the
|
||||||
|
// writer physically laid out equals e_shnum: every DWARF section written
|
||||||
|
// after .shstrtab must be counted, or the last ones (always .debug_frame)
|
||||||
|
// are invisible to every consumer, debug/elf included.
|
||||||
|
func checkELFSectionAccounting(t *testing.T, obj []byte) {
|
||||||
|
t.Helper()
|
||||||
|
shoff := int(binary.LittleEndian.Uint64(obj[40:]))
|
||||||
|
shentsize := int(binary.LittleEndian.Uint16(obj[58:]))
|
||||||
|
shnum := elfSectionHeaderCount(t, obj)
|
||||||
|
if shentsize != 64 {
|
||||||
|
t.Fatalf("e_shentsize = %d, want 64", shentsize)
|
||||||
|
}
|
||||||
|
if (len(obj)-shoff)%shentsize != 0 {
|
||||||
|
t.Fatalf("section header table is not a whole number of entries: shoff=%d len=%d", shoff, len(obj))
|
||||||
|
}
|
||||||
|
if present := (len(obj) - shoff) / shentsize; present != shnum {
|
||||||
|
t.Errorf("e_shnum = %d but %d section headers are laid out", shnum, present)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestELFDWARFSectionAccounting runs the header accounting check over all
|
||||||
|
// four architecture emitters, and additionally checks the .debug_frame
|
||||||
|
// section is visible (its data aligned as its header declares).
|
||||||
|
func TestELFDWARFSectionAccounting(t *testing.T) {
|
||||||
|
parse := func(name, src string) *ast.File {
|
||||||
|
f, errs := parser.Parse(name, src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse %s: %v", name, errs)
|
||||||
|
}
|
||||||
|
return f
|
||||||
|
}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
img *Image
|
||||||
|
emit func(*Image) ([]byte, error)
|
||||||
|
}{
|
||||||
|
{"amd64", elfTestImage(t), (*Image).ELFObject},
|
||||||
|
{"arm64", mustImage(t, func() (*Image, error) {
|
||||||
|
return AssembleFileARM64(parse("k_arm64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOVD a+0(FP), R4
|
||||||
|
MOVD b+8(FP), R5
|
||||||
|
ADD R5, R4, R4
|
||||||
|
MOVD R4, ret+16(FP)
|
||||||
|
RET
|
||||||
|
`))
|
||||||
|
}), (*Image).ELFAARCH64Object},
|
||||||
|
{"riscv64", mustImage(t, func() (*Image, error) {
|
||||||
|
return AssembleFileRISCV(parse("k_riscv64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·sb(SB), NOSPLIT, $0-0
|
||||||
|
MOV $answer<>(SB), X10
|
||||||
|
RET
|
||||||
|
|
||||||
|
GLOBL answer<>(SB), RODATA, $8
|
||||||
|
DATA answer<>+0(SB)/8, $42
|
||||||
|
`))
|
||||||
|
}), (*Image).ELFRISCVObject},
|
||||||
|
{"loong64", mustImage(t, func() (*Image, error) {
|
||||||
|
return AssembleFileLOONG64(parse("k_loong64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOVV a+0(FP), R4
|
||||||
|
MOVV b+8(FP), R5
|
||||||
|
ADDV R5, R4, R4
|
||||||
|
MOVV R4, ret+16(FP)
|
||||||
|
RET
|
||||||
|
`))
|
||||||
|
}), (*Image).ELFLOONG64Object},
|
||||||
|
}
|
||||||
|
for _, tc := range cases {
|
||||||
|
obj, err := tc.emit(tc.img)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("%s: emit: %v", tc.name, err)
|
||||||
|
}
|
||||||
|
checkELFSectionAccounting(t, obj)
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("%s: parse emitted object: %v", tc.name, err)
|
||||||
|
}
|
||||||
|
frame := ef.Section(".debug_frame")
|
||||||
|
if frame == nil {
|
||||||
|
t.Errorf("%s: .debug_frame invisible to debug/elf (e_shnum too small?)", tc.name)
|
||||||
|
ef.Close()
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if frame.Offset%8 != 0 || frame.Addralign != 8 {
|
||||||
|
t.Errorf("%s: .debug_frame offset %d align %d, want offset%%8==0 align 8", tc.name, frame.Offset, frame.Addralign)
|
||||||
|
}
|
||||||
|
ef.Close()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func mustImage(t *testing.T, f func() (*Image, error)) *Image {
|
||||||
|
t.Helper()
|
||||||
|
img, err := f()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return img
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestELFDWARFRelocations checks the .rela.debug_info and .rela.debug_line
|
||||||
|
// sections exist and carry absolute 64-bit relocations against the
|
||||||
|
// function symbols, with r_offsets inside their target sections.
|
||||||
|
func TestELFDWARFRelocations(t *testing.T) {
|
||||||
|
img := elfTestImage(t)
|
||||||
|
obj, err := img.ELFObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFObject: %v", err)
|
||||||
|
}
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
// The DWARF must record the assembled file's path (threaded through
|
||||||
|
// Image.SourcePath), not a placeholder name.
|
||||||
|
info, err := ef.Section(".debug_info").Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if img.SourcePath != "t_amd64.s" || !bytes.Contains(info, []byte(img.SourcePath)) {
|
||||||
|
t.Errorf("DWARF compilation unit does not name the source %q", img.SourcePath)
|
||||||
|
}
|
||||||
|
for _, tc := range []struct {
|
||||||
|
rela string
|
||||||
|
target string
|
||||||
|
want uint32
|
||||||
|
}{
|
||||||
|
{".rela.debug_info", ".debug_info", rX8664Abs64},
|
||||||
|
{".rela.debug_line", ".debug_line", rX8664Abs64},
|
||||||
|
{".rela.debug_frame", ".debug_frame", rX8664Abs64},
|
||||||
|
} {
|
||||||
|
rs := ef.Section(tc.rela)
|
||||||
|
if rs == nil {
|
||||||
|
t.Fatalf("missing %s", tc.rela)
|
||||||
|
}
|
||||||
|
if rs.Type != elf.SHT_RELA {
|
||||||
|
t.Errorf("%s: type %v, want SHT_RELA", tc.rela, rs.Type)
|
||||||
|
}
|
||||||
|
target := ef.Section(tc.target)
|
||||||
|
if target == nil {
|
||||||
|
t.Fatalf("missing %s", tc.target)
|
||||||
|
}
|
||||||
|
if rs.Link == 0 || ef.Sections[rs.Info] != target {
|
||||||
|
t.Errorf("%s: link %d info %d, want the symtab and %s", tc.rela, rs.Link, rs.Info, tc.target)
|
||||||
|
}
|
||||||
|
b, err := rs.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
// .debug_line has one address per function; .debug_info adds the
|
||||||
|
// compile unit's own low_pc.
|
||||||
|
want := len(img.Funcs)
|
||||||
|
if tc.target == ".debug_info" {
|
||||||
|
want++
|
||||||
|
}
|
||||||
|
if len(b)/24 != want {
|
||||||
|
t.Errorf("%s: %d entries, want %d", tc.rela, len(b)/24, want)
|
||||||
|
}
|
||||||
|
for i := 0; i+24 <= len(b); i += 24 {
|
||||||
|
r_offset := binary.LittleEndian.Uint64(b[i:])
|
||||||
|
info := binary.LittleEndian.Uint64(b[i+8:])
|
||||||
|
typ := uint32(info)
|
||||||
|
sym := int(info >> 32)
|
||||||
|
if typ != tc.want {
|
||||||
|
t.Errorf("%s entry %d: type %d, want R_X86_64_64 (%d)", tc.rela, i/24, typ, tc.want)
|
||||||
|
}
|
||||||
|
if r_offset >= uint64(target.Size) {
|
||||||
|
t.Errorf("%s entry %d: r_offset %d outside %s (%d bytes)", tc.rela, i/24, r_offset, tc.target, target.Size)
|
||||||
|
}
|
||||||
|
if sym == 0 {
|
||||||
|
t.Errorf("%s entry %d: against the null symbol", tc.rela, i/24)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestELFDataOnly checks a source with GLOBL data and no TEXT emits a valid
|
||||||
|
// ELF object: the DWARF compilation unit of a code-less image has no
|
||||||
|
// function to relocate against and must not reach for one.
|
||||||
|
func TestELFDataOnly(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("d0_amd64.s", `
|
||||||
|
GLOBL table<>(SB), RODATA, $8
|
||||||
|
DATA table<>+0(SB)/8, $12345
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFObject: %v", err)
|
||||||
|
}
|
||||||
|
checkELFSectionAccounting(t, obj)
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
found := false
|
||||||
|
for _, s := range syms {
|
||||||
|
if s.Name == "table" && s.Size == 8 {
|
||||||
|
found = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !found {
|
||||||
|
t.Errorf("data symbol table missing: %v", syms)
|
||||||
|
}
|
||||||
|
if ef.Section(".rela.debug_info") != nil || ef.Section(".rela.debug_line") != nil {
|
||||||
|
t.Error("data-only image must not emit DWARF address relocations")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestELFLinkAndRun is the end-to-end check: assemble the test functions,
|
// TestELFLinkAndRun is the end-to-end check: assemble the test functions,
|
||||||
// link the emitted object with a C driver that defines the external symbol,
|
// link the emitted object with a C driver that defines the external symbol,
|
||||||
// and run the result. Skipped when no C compiler is available.
|
// and run the result. Skipped when no C compiler is available.
|
||||||
@@ -307,4 +609,144 @@ int main(void) {
|
|||||||
if got := string(run); got != "42 42 7\n" {
|
if got := string(run); got != "42 42 7\n" {
|
||||||
t.Errorf("output %q, want \"42 42 7\\n\"", got)
|
t.Errorf("output %q, want \"42 42 7\\n\"", got)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The DWARF addresses must have resolved at link time: the .debug_info
|
||||||
|
// placeholders were carried by .rela.debug_info, so every subprogram's
|
||||||
|
// low_pc must now equal its linked symbol address.
|
||||||
|
bin, err := os.ReadFile(appPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
lef, err := elf.NewFile(bytes.NewReader(bin))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse linked binary: %v", err)
|
||||||
|
}
|
||||||
|
defer lef.Close()
|
||||||
|
syms, err := lef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
addrByName := map[string]uint64{}
|
||||||
|
for _, s := range syms {
|
||||||
|
if elf.ST_TYPE(s.Info) == elf.STT_FUNC && s.Value != 0 {
|
||||||
|
addrByName[s.Name] = s.Value
|
||||||
|
}
|
||||||
|
}
|
||||||
|
lowPCs := dwarfSubprogramLowPCs(t, lef)
|
||||||
|
if len(lowPCs) == 0 {
|
||||||
|
t.Fatal("no subprogram DW_AT_low_pc parsed from the linked binary")
|
||||||
|
}
|
||||||
|
for name, pc := range lowPCs {
|
||||||
|
addr, ok := addrByName[name]
|
||||||
|
if !ok {
|
||||||
|
t.Errorf("subprogram %q not in the linked symbol table", name)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if pc != addr {
|
||||||
|
t.Errorf("subprogram %q: DW_AT_low_pc = %#x, linked address %#x (DWARF relocation unresolved)", name, pc, addr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// dwarfSubprogramLowPCs walks the linked binary's .debug_info with its own
|
||||||
|
// .debug_abbrev and returns each DW_TAG_subprogram's DW_AT_low_pc by name.
|
||||||
|
func dwarfSubprogramLowPCs(t *testing.T, ef *elf.File) map[string]uint64 {
|
||||||
|
t.Helper()
|
||||||
|
abbrevSec := ef.Section(".debug_abbrev")
|
||||||
|
infoSec := ef.Section(".debug_info")
|
||||||
|
if abbrevSec == nil || infoSec == nil {
|
||||||
|
t.Fatal("linked binary lacks .debug_abbrev or .debug_info")
|
||||||
|
}
|
||||||
|
abbrev, err := abbrevSec.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
info, err := infoSec.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
abs := parseAbbrevs(t, abbrev)
|
||||||
|
le := binary.LittleEndian
|
||||||
|
out := map[string]uint64{}
|
||||||
|
r := &ulebIter{b: info}
|
||||||
|
r.uint32At(t) // unit_length
|
||||||
|
if v := le.Uint16(info[4:]); v != 5 {
|
||||||
|
t.Fatalf(".debug_info version %d, want 5", v)
|
||||||
|
}
|
||||||
|
r.i = 6
|
||||||
|
r.byteAt(t) // unit_type
|
||||||
|
r.byteAt(t) // address_size
|
||||||
|
r.uint32At(t) // debug_abbrev_offset
|
||||||
|
var name string
|
||||||
|
var lowPC uint64
|
||||||
|
for r.i < len(r.b) {
|
||||||
|
code := r.uleb(t)
|
||||||
|
if code == 0 {
|
||||||
|
continue // end of the CU's children
|
||||||
|
}
|
||||||
|
ab, ok := abs[code]
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("unknown abbreviation code %d", code)
|
||||||
|
}
|
||||||
|
name, lowPC = "", 0
|
||||||
|
for _, a := range ab.attrs {
|
||||||
|
switch a.attr {
|
||||||
|
case dwAtName:
|
||||||
|
readFormKeep(t, r, a.form, &name, nil)
|
||||||
|
case dwAtLowPC:
|
||||||
|
readFormKeep(t, r, a.form, nil, &lowPC)
|
||||||
|
default:
|
||||||
|
readFormSkip(t, r, a.form)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if ab.tag == dwTagSubprog && name != "" {
|
||||||
|
out[name] = lowPC
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// readFormKeep reads one DIE attribute value, keeping a string or an
|
||||||
|
// address into the pointer it was given (nil keeps nothing).
|
||||||
|
func readFormKeep(t *testing.T, r *ulebIter, form uint64, name *string, addr *uint64) {
|
||||||
|
t.Helper()
|
||||||
|
switch form {
|
||||||
|
case dwFormString:
|
||||||
|
end := r.i
|
||||||
|
for end < len(r.b) && r.b[end] != 0 {
|
||||||
|
end++
|
||||||
|
}
|
||||||
|
if name != nil {
|
||||||
|
*name = string(r.b[r.i:end])
|
||||||
|
}
|
||||||
|
r.i = end + 1
|
||||||
|
case dwFormAddr:
|
||||||
|
if addr != nil {
|
||||||
|
*addr = binary.LittleEndian.Uint64(r.b[r.i:])
|
||||||
|
}
|
||||||
|
r.i += 8
|
||||||
|
default:
|
||||||
|
readFormSkip(t, r, form)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func readFormSkip(t *testing.T, r *ulebIter, form uint64) {
|
||||||
|
t.Helper()
|
||||||
|
switch form {
|
||||||
|
case dwFormString:
|
||||||
|
for r.i < len(r.b) && r.b[r.i] != 0 {
|
||||||
|
r.i++
|
||||||
|
}
|
||||||
|
r.i++
|
||||||
|
case dwFormAddr, dwFormData8:
|
||||||
|
r.i += 8
|
||||||
|
case dwFormSecOff:
|
||||||
|
r.i += 4
|
||||||
|
case dwFormExprloc:
|
||||||
|
r.i += int(r.uleb(t))
|
||||||
|
case dwFormData1, 0x0c:
|
||||||
|
r.i++
|
||||||
|
default:
|
||||||
|
t.Fatalf("unsupported form %#x", form)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+74
-22
@@ -81,10 +81,15 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Build relocations. Each SB reference is an ADRP pair:
|
// Build relocations. Each SB reference is an ADRP pair:
|
||||||
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21
|
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21 at the ADRP
|
||||||
// ADD → R_AARCH64_ADD_ABS_LO12_NC
|
// ADD → R_AARCH64_ADD_ABS_LO12_NC at the ADD word
|
||||||
// LDR/STR X → R_AARCH64_LDST64_ABS_LO12_NC
|
// LDR/STR X → R_AARCH64_LDST64_ABS_LO12_NC at the LDR/STR word
|
||||||
// BL → R_AARCH64_CALL26
|
// BL → R_AARCH64_CALL26
|
||||||
|
// cmd/link's own conversion emits the HI21 at sectoff and the LO12 at
|
||||||
|
// sectoff+4 (cmd/link/internal/arm64/asm.go), so the ADD or load word
|
||||||
|
// carries the page-offset relocation, never a second HI21. The
|
||||||
|
// assembler records two RelArm64Addr relocs per ADRP+ADD pair (one per
|
||||||
|
// word), so the second of the pair is consumed here.
|
||||||
// Addends stay raw: ADR_PREL_PG_HI21 and the ABS_LO12_NC forms resolve
|
// Addends stay raw: ADR_PREL_PG_HI21 and the ABS_LO12_NC forms resolve
|
||||||
// against S+A, and CALL26 branches take the branch instruction's own
|
// against S+A, and CALL26 branches take the branch instruction's own
|
||||||
// place as the PC-relative base, so subtracting the field width (the
|
// place as the PC-relative base, so subtracting the field width (the
|
||||||
@@ -97,28 +102,34 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
}
|
}
|
||||||
var relas []elfRela
|
var relas []elfRela
|
||||||
for _, fn := range img.Funcs {
|
for _, fn := range img.Funcs {
|
||||||
for _, r := range fn.Relocs {
|
for i := 0; i < len(fn.Relocs); i++ {
|
||||||
|
r := fn.Relocs[i]
|
||||||
idx, ok := symIdx[r.Name]
|
idx, ok := symIdx[r.Name]
|
||||||
if !ok {
|
if !ok {
|
||||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||||
}
|
}
|
||||||
var typ uint32
|
switch r.Kind {
|
||||||
switch {
|
case RelArm64Branch:
|
||||||
case r.Kind == RelArm64Branch:
|
relas = append(relas, elfRela{
|
||||||
typ = rArm64Call26
|
off: uint64(fn.Offset + r.Off), typ: rArm64Call26, sym: idx, addend: r.Addend,
|
||||||
case r.Kind == RelArm64LDST64 && r.Off%4 == 4:
|
})
|
||||||
typ = rArm64Ldst64Lo12NC
|
case RelArm64Addr:
|
||||||
case r.Kind == RelArm64Addr && r.Off%4 == 4:
|
// ADRP+ADD: the pair's second reloc (at Off+4) is the
|
||||||
typ = rArm64AddAbsLo12NC
|
// assembler's twin of the same pair; skip it.
|
||||||
|
relas = append(relas,
|
||||||
|
elfRela{off: uint64(fn.Offset + r.Off), typ: rArm64PrelPgHi21, sym: idx, addend: r.Addend},
|
||||||
|
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rArm64AddAbsLo12NC, sym: idx, addend: r.Addend},
|
||||||
|
)
|
||||||
|
i++
|
||||||
|
case RelArm64LDST64:
|
||||||
|
// ADRP+LDR/STR: one assembler reloc covers the pair.
|
||||||
|
relas = append(relas,
|
||||||
|
elfRela{off: uint64(fn.Offset + r.Off), typ: rArm64PrelPgHi21, sym: idx, addend: r.Addend},
|
||||||
|
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rArm64Ldst64Lo12NC, sym: idx, addend: r.Addend},
|
||||||
|
)
|
||||||
default:
|
default:
|
||||||
typ = rArm64PrelPgHi21
|
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
|
||||||
}
|
}
|
||||||
relas = append(relas, elfRela{
|
|
||||||
off: uint64(fn.Offset + r.Off),
|
|
||||||
typ: typ,
|
|
||||||
sym: idx,
|
|
||||||
addend: r.Addend,
|
|
||||||
})
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -193,15 +204,32 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
shstrOff := len(out)
|
shstrOff := len(out)
|
||||||
out = append(out, stSections.bytes()...)
|
out = append(out, stSections.bytes()...)
|
||||||
|
|
||||||
// DWARF debug sections.
|
// DWARF debug sections; the address placeholders they leave are carried
|
||||||
|
// as .rela.debug_info/.rela.debug_line entries the system linker applies.
|
||||||
dwAlign := func(n int) {
|
dwAlign := func(n int) {
|
||||||
for len(out)%n != 0 {
|
for len(out)%n != 0 {
|
||||||
out = append(out, 0)
|
out = append(out, 0)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
|
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiARM64)
|
||||||
|
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
nSections += 4
|
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
|
||||||
|
// .debug_line_str and .debug_frame (the CIE is unconditional, so
|
||||||
|
// the frame section is always present), plus the relocation
|
||||||
|
// sections below when they carry entries.
|
||||||
|
dwarfStart = nSections
|
||||||
|
nSections += 5
|
||||||
|
appendDWARFRelas(&out, dw, rAARCH64Abs64, dwAlign)
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
align(8)
|
align(8)
|
||||||
@@ -230,13 +258,37 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||||
}
|
}
|
||||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
|
// DWARF section headers; their indices follow the write order.
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
|
// secIdx is a running section index: each putSh below emits the
|
||||||
|
// next header, and the sh_info of a .rela section names the index
|
||||||
|
// of the section it relocates.
|
||||||
|
secIdx := dwarfStart
|
||||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||||
|
secInfoIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||||
|
secLineIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
if dw.frameSize > 0 {
|
if dw.frameSize > 0 {
|
||||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||||
|
secFrameIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+58
-1
@@ -6,6 +6,7 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
"debug/elf"
|
"debug/elf"
|
||||||
|
"encoding/binary"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
@@ -28,6 +29,7 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
|
|
||||||
TEXT ·getanswer(SB), NOSPLIT, $0-8
|
TEXT ·getanswer(SB), NOSPLIT, $0-8
|
||||||
MOVD answer<>(SB), R4
|
MOVD answer<>(SB), R4
|
||||||
|
MOVD $answer<>(SB), R5
|
||||||
MOVD R4, ret+0(FP)
|
MOVD R4, ret+0(FP)
|
||||||
RET
|
RET
|
||||||
|
|
||||||
@@ -102,8 +104,63 @@ DATA answer<>+0(SB)/8, $42
|
|||||||
// Check that .rela.text exists (getanswer has SB reference).
|
// Check that .rela.text exists (getanswer has SB reference).
|
||||||
relaText := ef.Section(".rela.text")
|
relaText := ef.Section(".rela.text")
|
||||||
if relaText == nil {
|
if relaText == nil {
|
||||||
t.Error("missing .rela.text section")
|
t.Fatal("missing .rela.text section")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The SB references of getanswer form two ADRP pairs: the load
|
||||||
|
// (MOVD answer<>(SB), R4) is ADRP+LDR carrying HI21 at the ADRP and
|
||||||
|
// LDST64_ABS_LO12_NC at the LDR word, and the address-of
|
||||||
|
// (MOVD $answer<>(SB), R5) is ADRP+ADD carrying HI21 and
|
||||||
|
// ADD_ABS_LO12_NC. cmd/link's own conversion emits exactly this
|
||||||
|
// sectoff / sectoff+4 pairing; a second HI21 at the ADD or LDR word
|
||||||
|
// corrupts the pair.
|
||||||
|
raw, err := relaText.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if len(raw)%24 != 0 || len(raw)/24 != 4 {
|
||||||
|
t.Fatalf(".rela.text has %d bytes, want four 24-byte entries", len(raw))
|
||||||
|
}
|
||||||
|
wantRela := []struct {
|
||||||
|
typ elf.R_AARCH64
|
||||||
|
off uint64 // relative to the getanswer function start
|
||||||
|
}{
|
||||||
|
{elf.R_AARCH64_ADR_PREL_PG_HI21, 0},
|
||||||
|
{elf.R_AARCH64_LDST64_ABS_LO12_NC, 4},
|
||||||
|
{elf.R_AARCH64_ADR_PREL_PG_HI21, 8},
|
||||||
|
{elf.R_AARCH64_ADD_ABS_LO12_NC, 12},
|
||||||
|
}
|
||||||
|
getanswer := byNameElf(t, ef, "getanswer")
|
||||||
|
for i, w := range wantRela {
|
||||||
|
e := raw[i*24 : (i+1)*24]
|
||||||
|
off := binary.LittleEndian.Uint64(e[0:])
|
||||||
|
info := binary.LittleEndian.Uint64(e[8:])
|
||||||
|
typ := elf.R_AARCH64(info & 0xffffffff)
|
||||||
|
sym := int(info >> 32)
|
||||||
|
if typ != w.typ || off != getanswer.Value+w.off {
|
||||||
|
t.Errorf("reloc %d: type %v off %d, want %v at %d", i, typ, off, w.typ, getanswer.Value+w.off)
|
||||||
|
}
|
||||||
|
if sym != 3 { // NULL, .text, .data, then the first local: answer
|
||||||
|
t.Errorf("reloc %d: symbol index %d, want 3 (answer)", i, sym)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// byNameElf returns the symbol table entry for name from the raw .symtab,
|
||||||
|
// which carries every entry including the null and section symbols in order.
|
||||||
|
func byNameElf(t *testing.T, ef *elf.File, name string) elf.Symbol {
|
||||||
|
t.Helper()
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("symbols: %v", err)
|
||||||
|
}
|
||||||
|
for _, s := range syms {
|
||||||
|
if s.Name == name {
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatalf("symbol %q not found", name)
|
||||||
|
return elf.Symbol{}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestELFAARCH64ObjectNoRelocations checks the ELF output when there are no
|
// TestELFAARCH64ObjectNoRelocations checks the ELF output when there are no
|
||||||
|
|||||||
+49
-3
@@ -13,6 +13,12 @@ import (
|
|||||||
const (
|
const (
|
||||||
emLOONGARCH = 258 // EM_LOONGARCH
|
emLOONGARCH = 258 // EM_LOONGARCH
|
||||||
|
|
||||||
|
// EF_LOONGARCH_ABI_DOUBLE_FLOAT | EF_LOONGARCH_OBJABI_V1: the flags the
|
||||||
|
// Go toolchain writes (cmd/link/internal/ld/elf.go: Flags = 0x43 for
|
||||||
|
// Loong64). System linkers refuse to merge ET_REL objects whose float
|
||||||
|
// ABI differs, so 0 (soft-float) would make the object unlinkable.
|
||||||
|
efLarchAbiDoubleObjV1 = 0x43
|
||||||
|
|
||||||
// LoongArch relocation types (the ELF psABI).
|
// LoongArch relocation types (the ELF psABI).
|
||||||
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
|
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
|
||||||
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
|
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
|
||||||
@@ -187,9 +193,25 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
out = append(out, 0)
|
out = append(out, 0)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
|
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiLOONG64)
|
||||||
|
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
nSections += 4
|
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
|
||||||
|
// .debug_line_str and .debug_frame (the CIE is unconditional, so
|
||||||
|
// the frame section is always present), plus the relocation
|
||||||
|
// sections below when they carry entries.
|
||||||
|
dwarfStart = nSections
|
||||||
|
nSections += 5
|
||||||
|
appendDWARFRelas(&out, dw, rLarchAbs64, dwAlign)
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
align(8)
|
align(8)
|
||||||
@@ -218,13 +240,37 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||||
}
|
}
|
||||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
|
// DWARF section headers; their indices follow the write order.
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
|
// secIdx is a running section index: each putSh below emits the
|
||||||
|
// next header, and the sh_info of a .rela section names the index
|
||||||
|
// of the section it relocates.
|
||||||
|
secIdx := dwarfStart
|
||||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||||
|
secInfoIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||||
|
secLineIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
if dw.frameSize > 0 {
|
if dw.frameSize > 0 {
|
||||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||||
|
secFrameIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -237,7 +283,7 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
le.PutUint64(hdr[24:], 0)
|
le.PutUint64(hdr[24:], 0)
|
||||||
le.PutUint64(hdr[32:], 0)
|
le.PutUint64(hdr[32:], 0)
|
||||||
le.PutUint64(hdr[40:], uint64(shoff))
|
le.PutUint64(hdr[40:], uint64(shoff))
|
||||||
le.PutUint32(hdr[48:], 0)
|
le.PutUint32(hdr[48:], efLarchAbiDoubleObjV1)
|
||||||
le.PutUint16(hdr[52:], 64)
|
le.PutUint16(hdr[52:], 64)
|
||||||
le.PutUint16(hdr[54:], 0)
|
le.PutUint16(hdr[54:], 0)
|
||||||
le.PutUint16(hdr[56:], 0)
|
le.PutUint16(hdr[56:], 0)
|
||||||
|
|||||||
@@ -55,6 +55,11 @@ DATA answer<>+0(SB)/8, $42
|
|||||||
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_LOONGARCH {
|
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_LOONGARCH {
|
||||||
t.Errorf("type/machine = %v/%v, want ET_REL/EM_LOONGARCH", ef.Type, ef.Machine)
|
t.Errorf("type/machine = %v/%v, want ET_REL/EM_LOONGARCH", ef.Type, ef.Machine)
|
||||||
}
|
}
|
||||||
|
// The double-float ABI plus OBJABI_V1 flags the Go toolchain writes;
|
||||||
|
// system linkers refuse ABI-mismatched merges.
|
||||||
|
if flags := binary.LittleEndian.Uint32(obj[48:]); flags != efLarchAbiDoubleObjV1 {
|
||||||
|
t.Errorf("e_flags = %#x, want %#x (double-float, OBJABI_V1)", flags, efLarchAbiDoubleObjV1)
|
||||||
|
}
|
||||||
|
|
||||||
text := ef.Section(".text")
|
text := ef.Section(".text")
|
||||||
data := ef.Section(".data")
|
data := ef.Section(".data")
|
||||||
|
|||||||
+60
-11
@@ -13,8 +13,13 @@ import (
|
|||||||
const (
|
const (
|
||||||
emRISCV = 243 // EM_RISCV
|
emRISCV = 243 // EM_RISCV
|
||||||
|
|
||||||
|
// EF_RISCV_FLOAT_ABI_DOUBLE: the double-precision float ABI the Go
|
||||||
|
// toolchain targets (cmd/link/internal/ld/elf.go writes Flags = 0x4 for
|
||||||
|
// RISCV64). System linkers refuse to merge ET_REL objects whose float
|
||||||
|
// ABI differs, so 0 (soft-float) would make the object unlinkable.
|
||||||
|
efRISCVFloatAbiDouble = 0x4
|
||||||
|
|
||||||
// RISC-V relocation types.
|
// RISC-V relocation types.
|
||||||
rRISCV32 = 1
|
|
||||||
rRISCVJAL = 17 // R_RISCV_JAL
|
rRISCVJAL = 17 // R_RISCV_JAL
|
||||||
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
|
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
|
||||||
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
|
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
|
||||||
@@ -83,9 +88,14 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
// Build relocations. Each SB reference is an AUIPC + second-instruction
|
// Build relocations. Each SB reference is an AUIPC + second-instruction
|
||||||
// pair carrying a single relocation kind; the ELF writer expands it into
|
// pair carrying a single relocation kind; the ELF writer expands it into
|
||||||
// the R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I/S pair the psABI expects.
|
// the R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I/S pair the psABI expects.
|
||||||
// The HI20 carries the symbol addend; the LO12 addend is zero, matching
|
// The HI20 carries the symbol and its addend. The LO12's symbol must
|
||||||
// cmd/link's own ELF conversion (the LO12 resolves against the HI20's
|
// denote the AUIPC site the HI20 relocates (psABI §8.4.9: the pair is
|
||||||
// AUIPC location).
|
// resolved against the label of the AUIPC, not the target symbol;
|
||||||
|
// cmd/link generates one local text symbol per AUIPC for exactly this,
|
||||||
|
// cmd/link/internal/riscv64/asm.go). The .text section symbol with the
|
||||||
|
// AUIPC's section-relative offset as addend gives S + A = the AUIPC
|
||||||
|
// address, which is that label.
|
||||||
|
const secSymText = 1 // syms[1], the .text section symbol
|
||||||
type elfRela struct {
|
type elfRela struct {
|
||||||
off uint64
|
off uint64
|
||||||
typ uint32
|
typ uint32
|
||||||
@@ -99,21 +109,20 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||||
}
|
}
|
||||||
|
auipc := int64(fn.Offset + r.Off)
|
||||||
switch r.Kind {
|
switch r.Kind {
|
||||||
case RelRISCVPCRELIType:
|
case RelRISCVPCRELIType:
|
||||||
relas = append(relas,
|
relas = append(relas,
|
||||||
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
||||||
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: idx, addend: 0},
|
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: secSymText, addend: auipc},
|
||||||
)
|
)
|
||||||
case RelRISCVPCRELSType:
|
case RelRISCVPCRELSType:
|
||||||
relas = append(relas,
|
relas = append(relas,
|
||||||
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
||||||
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: idx, addend: 0},
|
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: secSymText, addend: auipc},
|
||||||
)
|
)
|
||||||
case RelRISCVJal:
|
case RelRISCVJal:
|
||||||
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVJAL, sym: idx, addend: r.Addend})
|
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVJAL, sym: idx, addend: r.Addend})
|
||||||
case RelPCRelAbs:
|
|
||||||
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCV32, sym: idx, addend: r.Addend})
|
|
||||||
default:
|
default:
|
||||||
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
|
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
|
||||||
}
|
}
|
||||||
@@ -196,9 +205,25 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
out = append(out, 0)
|
out = append(out, 0)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
|
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiRISCV64)
|
||||||
|
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
nSections += 4
|
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
|
||||||
|
// .debug_line_str and .debug_frame (the CIE is unconditional, so
|
||||||
|
// the frame section is always present), plus the relocation
|
||||||
|
// sections below when they carry entries.
|
||||||
|
dwarfStart = nSections
|
||||||
|
nSections += 5
|
||||||
|
appendDWARFRelas(&out, dw, rRISCVAbs64, dwAlign)
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
align(8)
|
align(8)
|
||||||
@@ -227,13 +252,37 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||||
}
|
}
|
||||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
|
// DWARF section headers; their indices follow the write order.
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
|
// secIdx is a running section index: each putSh below emits the
|
||||||
|
// next header, and the sh_info of a .rela section names the index
|
||||||
|
// of the section it relocates.
|
||||||
|
secIdx := dwarfStart
|
||||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||||
|
secInfoIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||||
|
secLineIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
if dw.frameSize > 0 {
|
if dw.frameSize > 0 {
|
||||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||||
|
secFrameIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -246,7 +295,7 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
le.PutUint64(hdr[24:], 0)
|
le.PutUint64(hdr[24:], 0)
|
||||||
le.PutUint64(hdr[32:], 0)
|
le.PutUint64(hdr[32:], 0)
|
||||||
le.PutUint64(hdr[40:], uint64(shoff))
|
le.PutUint64(hdr[40:], uint64(shoff))
|
||||||
le.PutUint32(hdr[48:], 0)
|
le.PutUint32(hdr[48:], efRISCVFloatAbiDouble)
|
||||||
le.PutUint16(hdr[52:], 64)
|
le.PutUint16(hdr[52:], 64)
|
||||||
le.PutUint16(hdr[54:], 0)
|
le.PutUint16(hdr[54:], 0)
|
||||||
le.PutUint16(hdr[56:], 0)
|
le.PutUint16(hdr[56:], 0)
|
||||||
|
|||||||
+47
-11
@@ -18,7 +18,11 @@ func Encodable(mnemonic string) bool {
|
|||||||
|
|
||||||
// Fixed-name instructions (no size suffix).
|
// Fixed-name instructions (no size suffix).
|
||||||
switch upper {
|
switch upper {
|
||||||
case "RET", "NOP", "CALL", "JMP":
|
case "RET", "NOP", "CALL", "JMP",
|
||||||
|
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := noOperandTable[upper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
if _, ok := condCode(upper); ok {
|
if _, ok := condCode(upper); ok {
|
||||||
@@ -31,15 +35,20 @@ func Encodable(mnemonic string) bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
||||||
base == "KMOVW" || base == "KMOVQ" {
|
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
// CMOV carries size then condition (CMOVLGT); SET carries the condition
|
// CMOV carries size then condition (CMOVLGT); SET carries the condition
|
||||||
// alone (SETNE).
|
// alone (SETNE). The size letter is checked exactly as encodeCmov does,
|
||||||
|
// so a spelling like CMOVBGT is not reported encodable when Encode
|
||||||
|
// would reject it.
|
||||||
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
|
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
|
||||||
if _, ok := jccMap[rest[1:]]; ok {
|
switch rest[0] {
|
||||||
return true
|
case 'W', 'L', 'Q':
|
||||||
|
if _, ok := jccMap[rest[1:]]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
|
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
|
||||||
@@ -48,13 +57,28 @@ func Encodable(mnemonic string) bool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Legacy SSE shuffles and packed binaries dispatch on the full name.
|
// Legacy SSE shuffles and packed binaries dispatch on the full name; so
|
||||||
|
// do the imm8-controlled instructions, the lane extracts and inserts and
|
||||||
|
// the packed integer shifts (their trailing width letters belong to the
|
||||||
|
// mnemonic).
|
||||||
if _, ok := sseShufTable[upper]; ok {
|
if _, ok := sseShufTable[upper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
if _, ok := sseBinTable[upper]; ok {
|
if _, ok := sseBinTable[upper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
if _, ok := sseImm3Table[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sseExtractTable[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sseInsertTable[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sseShiftImm[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
// The size-suffix split: retry the tables and the scalar switch on the
|
// The size-suffix split: retry the tables and the scalar switch on the
|
||||||
// base.
|
// base.
|
||||||
@@ -69,20 +93,32 @@ func Encodable(mnemonic string) bool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
switch base2 {
|
switch base2 {
|
||||||
case "MOV",
|
case "MOV", "MOVD",
|
||||||
"ADD", "SUB", "AND", "OR", "XOR", "CMP",
|
"ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB",
|
||||||
"TEST",
|
"TEST",
|
||||||
"LEA",
|
"LEA",
|
||||||
"INC", "DEC", "NEG", "NOT",
|
"INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV",
|
||||||
"SHL", "SHR", "SAR",
|
"SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR",
|
||||||
|
"BT", "BTS", "BTR", "BTC",
|
||||||
|
"XCHG", "CMPXCHG", "XADD", "CRC32", "ADCX", "ADOX",
|
||||||
|
"MOVS", "STOS",
|
||||||
"IMUL", "IMUL3",
|
"IMUL", "IMUL3",
|
||||||
"PUSH", "POP",
|
"PUSH", "POP",
|
||||||
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
|
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
|
||||||
"BSWAP",
|
"BSWAP",
|
||||||
"PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2",
|
"PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2",
|
||||||
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
|
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
|
||||||
|
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX",
|
||||||
"CVTSL2SD", "CVTSQ2SD",
|
"CVTSL2SD", "CVTSQ2SD",
|
||||||
"MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
"CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S",
|
||||||
|
"FMOVD",
|
||||||
|
"MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
// Full-name dispatches the size split would eat (a trailing width
|
||||||
|
// letter that is part of the mnemonic).
|
||||||
|
switch upper {
|
||||||
|
case "PMOVMSKB":
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
return false
|
return false
|
||||||
|
|||||||
+110
-13
@@ -40,14 +40,59 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
return e.encodeRet()
|
return e.encodeRet()
|
||||||
case upper == "NOP":
|
case upper == "NOP":
|
||||||
return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1})
|
return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1})
|
||||||
case upper == "CALL":
|
case upper == "CALL" || upper == "JMP":
|
||||||
return e.encodeJmpRel(ops, []byte{0xE8})
|
// Through a register or memory: FF /2 (CALL) or FF /4 (JMP).
|
||||||
case upper == "JMP":
|
// Anything else is a rel32 against a label resolved by the assembler.
|
||||||
return e.encodeJmpRel(ops, []byte{0xE9})
|
if len(ops) == 1 {
|
||||||
|
switch ops[0].(type) {
|
||||||
|
case Reg, Mem:
|
||||||
|
return e.encodeIndirectBranch(upper, ops)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
opcode := []byte{0xE8}
|
||||||
|
if upper == "JMP" {
|
||||||
|
opcode = []byte{0xE9}
|
||||||
|
}
|
||||||
|
return e.encodeJmpRel(ops, opcode)
|
||||||
}
|
}
|
||||||
if cc, ok := condCode(upper); ok {
|
if cc, ok := condCode(upper); ok {
|
||||||
return e.encodeJcc(cc, ops)
|
return e.encodeJcc(cc, ops)
|
||||||
}
|
}
|
||||||
|
// No-operand system and string-control instructions (CPUID, RDTSC,
|
||||||
|
// SYSCALL, the fences, UNDEF, …).
|
||||||
|
if op, ok := noOperandTable[upper]; ok {
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("%s takes no operands, got %d", upper, len(ops))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
|
||||||
|
}
|
||||||
|
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
|
||||||
|
// are rejected by go tool asm in 64-bit mode, so they stay unsupported.
|
||||||
|
switch upper {
|
||||||
|
case "POPFQ":
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("POPFQ takes no operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: []byte{0x9D}, modrm: -1, sib: -1})
|
||||||
|
case "PUSHFQ":
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("PUSHFQ takes no operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
|
||||||
|
case "INT":
|
||||||
|
return e.encodeInt(ops)
|
||||||
|
case "LDMXCSR":
|
||||||
|
return e.encodeMxcsr(2, ops)
|
||||||
|
case "STMXCSR":
|
||||||
|
return e.encodeMxcsr(3, ops)
|
||||||
|
// CMPSD is the scalar double compare, whose predicate immediate comes
|
||||||
|
// LAST in Plan 9 order (src, dst, $imm).
|
||||||
|
case "CMPSD":
|
||||||
|
return e.encodeCmpsd(ops)
|
||||||
|
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
|
||||||
|
case "SHA256RNDS2":
|
||||||
|
return e.encodeSha256rnds2(ops)
|
||||||
|
}
|
||||||
|
|
||||||
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
||||||
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
|
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
|
||||||
@@ -57,7 +102,8 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
|
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
||||||
|
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
|
||||||
return e.encodeVec(base, ops, sfx)
|
return e.encodeVec(base, ops, sfx)
|
||||||
}
|
}
|
||||||
if sfx.any() {
|
if sfx.any() {
|
||||||
@@ -91,36 +137,81 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
if m, ok := sseBinTable[base]; ok {
|
if m, ok := sseBinTable[base]; ok {
|
||||||
return e.encodeSSEBin(m, ops)
|
return e.encodeSSEBin(m, ops)
|
||||||
}
|
}
|
||||||
|
// The imm8-controlled legacy instructions, the lane extracts and inserts
|
||||||
|
// and the packed integer shifts all dispatch on the full name: a trailing
|
||||||
|
// width letter here belongs to the mnemonic, not to the size split.
|
||||||
|
if m, ok := sseImm3Table[upper]; ok {
|
||||||
|
return e.encodeSSEImm3(m, ops)
|
||||||
|
}
|
||||||
|
if m, ok := sseExtractTable[upper]; ok {
|
||||||
|
return e.encodeSSEExtract(m, ops)
|
||||||
|
}
|
||||||
|
if m, ok := sseInsertTable[upper]; ok {
|
||||||
|
return e.encodeSSEInsert(m, ops)
|
||||||
|
}
|
||||||
|
if _, ok := sseShiftImm[upper]; ok {
|
||||||
|
return e.encodeSSEShift(upper, ops)
|
||||||
|
}
|
||||||
|
// PMOVMSKB ends in a width letter the size split would eat, so it
|
||||||
|
// dispatches on the full name like the packed binaries above.
|
||||||
|
if upper == "PMOVMSKB" {
|
||||||
|
return e.encodePmovmskb(upper, ops)
|
||||||
|
}
|
||||||
switch base {
|
switch base {
|
||||||
case "MOV":
|
case "MOV":
|
||||||
return e.encodeMov(ops, size)
|
return e.encodeMov(ops, size)
|
||||||
case "ADD", "SUB", "AND", "OR", "XOR", "CMP":
|
// MOVD is the Go assembler's alias of MOVQ: the same byte forms, 64-bit
|
||||||
|
// REX.W and all.
|
||||||
|
case "MOVD":
|
||||||
|
return e.encodeMov(ops, 8)
|
||||||
|
case "ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB":
|
||||||
return e.encodeALU(aluOp[base], ops, size)
|
return e.encodeALU(aluOp[base], ops, size)
|
||||||
case "TEST":
|
case "TEST":
|
||||||
return e.encodeTest(ops, size)
|
return e.encodeTest(ops, size)
|
||||||
case "LEA":
|
case "LEA":
|
||||||
return e.encodeLea(ops, size)
|
return e.encodeLea(ops, size)
|
||||||
case "INC", "DEC", "NEG", "NOT":
|
case "INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV":
|
||||||
return e.encodeUnary(unaryOp[base], ops, size)
|
return e.encodeUnary(unaryOp[base], ops, size)
|
||||||
case "SHL", "SHR", "SAR":
|
case "SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR":
|
||||||
return e.encodeShift(shiftOp[base], ops, size)
|
return e.encodeShift(shiftOp[base], ops, size)
|
||||||
|
case "BT", "BTS", "BTR", "BTC":
|
||||||
|
return e.encodeBitTest(base, ops, size)
|
||||||
|
case "XCHG":
|
||||||
|
return e.encodeExchange(ops, size)
|
||||||
|
case "CMPXCHG":
|
||||||
|
return e.encodeRegRegOp(0xB0, 0xB1, base, ops, size)
|
||||||
|
case "XADD":
|
||||||
|
return e.encodeRegRegOp(0xC0, 0xC1, base, ops, size)
|
||||||
|
case "CRC32":
|
||||||
|
return e.encodeCrc32(ops, size)
|
||||||
|
case "ADCX":
|
||||||
|
return e.encodeCarryExt(0x66, ops, size)
|
||||||
|
case "ADOX":
|
||||||
|
return e.encodeCarryExt(0xF3, ops, size)
|
||||||
|
case "MOVS", "STOS":
|
||||||
|
return e.encodeStringOp(base, ops, size)
|
||||||
case "IMUL", "IMUL3":
|
case "IMUL", "IMUL3":
|
||||||
return e.encodeImul(ops, size)
|
return e.encodeImul(ops, size)
|
||||||
case "PUSH":
|
case "PUSH":
|
||||||
return e.encodePushPop(ops, true)
|
return e.encodePushPop(ops, size, true)
|
||||||
case "POP":
|
case "POP":
|
||||||
return e.encodePushPop(ops, false)
|
return e.encodePushPop(ops, size, false)
|
||||||
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
|
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
|
||||||
return e.encodeCount(base, ops, size)
|
return e.encodeCount(base, ops, size)
|
||||||
case "BSWAP":
|
case "BSWAP":
|
||||||
return e.encodeBswap(ops, size)
|
return e.encodeBswap(ops, size)
|
||||||
case "PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2":
|
case "PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2":
|
||||||
return e.encodePrefetch(base, ops)
|
return e.encodePrefetch(base, ops)
|
||||||
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX":
|
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
|
||||||
|
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX":
|
||||||
return e.encodeMovExtend(base, ops)
|
return e.encodeMovExtend(base, ops)
|
||||||
case "CVTSL2SD", "CVTSQ2SD":
|
case "CVTSL2SD", "CVTSQ2SD":
|
||||||
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
|
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
|
||||||
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
case "CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S":
|
||||||
|
return e.encodeCvtInt(base, ops, size)
|
||||||
|
case "FMOVD":
|
||||||
|
return e.encodeFmov(ops)
|
||||||
|
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||||
return e.encodeSSEMove(sseMoveTable[base], ops)
|
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("unsupported instruction %q", mnem)
|
return fmt.Errorf("unsupported instruction %q", mnem)
|
||||||
@@ -179,7 +270,7 @@ func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
|
|||||||
if ss, ok := scatterTable[upper]; ok {
|
if ss, ok := scatterTable[upper]; ok {
|
||||||
return e.encodeScatter(upper, ss, ops, sfx)
|
return e.encodeScatter(upper, ss, ops, sfx)
|
||||||
}
|
}
|
||||||
if upper == "KMOVW" || upper == "KMOVQ" {
|
if upper == "KMOVW" || upper == "KMOVQ" || upper == "KMOVB" || upper == "KMOVD" {
|
||||||
if sfx.any() {
|
if sfx.any() {
|
||||||
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
||||||
}
|
}
|
||||||
@@ -335,6 +426,12 @@ func setMem(i *instr, regField int, m Mem) error {
|
|||||||
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
|
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
|
||||||
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
|
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
|
||||||
sib = -1
|
sib = -1
|
||||||
|
// A displacement wider than int32 fits no encoding form; truncating it
|
||||||
|
// would address a different location, and go tool asm reports "offset
|
||||||
|
// too large" for the same operand.
|
||||||
|
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
|
||||||
|
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
|
||||||
|
}
|
||||||
// RIP-relative: neither base nor index.
|
// RIP-relative: neither base nor index.
|
||||||
if !m.HasBase && !m.HasIndex {
|
if !m.HasBase && !m.HasIndex {
|
||||||
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
|
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
|
||||||
|
|||||||
+426
-2
@@ -149,6 +149,45 @@ func TestPushPop(t *testing.T) {
|
|||||||
checkSyntax(t, "push rbx", "PUSHQ", BX)
|
checkSyntax(t, "push rbx", "PUSHQ", BX)
|
||||||
checkSyntax(t, "pop r12", "POPQ", Reg{idx: 12, size: 8})
|
checkSyntax(t, "pop r12", "POPQ", Reg{idx: 12, size: 8})
|
||||||
checkSyntax(t, "push 0x5", "PUSHQ", Imm(5))
|
checkSyntax(t, "push 0x5", "PUSHQ", Imm(5))
|
||||||
|
// The W spelling carries the 0x66 operand-size prefix, byte for byte
|
||||||
|
// with go tool asm; the L and B spellings are illegal in 64-bit mode
|
||||||
|
// there and rejected here rather than silently widened.
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"PUSHW AX", "PUSHW", []Operand{AX}, "6650"},
|
||||||
|
{"POPW AX", "POPW", []Operand{AX}, "6658"},
|
||||||
|
{"PUSHW $5", "PUSHW", []Operand{Imm(5)}, "666a05"},
|
||||||
|
{"PUSHW (AX)", "PUSHW", []Operand{Ptr(AX, 0, 2)}, "66ff30"},
|
||||||
|
{"PUSHQ AX", "PUSHQ", []Operand{AX}, "50"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, c := range []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
}{
|
||||||
|
{"PUSHL AX", "PUSHL", []Operand{AX}},
|
||||||
|
{"PUSHL R8", "PUSHL", []Operand{Reg{idx: 8, size: 8}}},
|
||||||
|
{"POPL BX", "POPL", []Operand{BX}},
|
||||||
|
{"PUSHB AX", "PUSHB", []Operand{AX}},
|
||||||
|
} {
|
||||||
|
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", c.name)
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestUnary(t *testing.T) {
|
func TestUnary(t *testing.T) {
|
||||||
@@ -181,6 +220,39 @@ func TestControl(t *testing.T) {
|
|||||||
checkOp(t, x86asm.JBE, "JLS", Imm(0))
|
checkOp(t, x86asm.JBE, "JLS", Imm(0))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestIndirectControlFlow pins the indirect JMP/CALL forms: FF /4 for JMP and
|
||||||
|
// FF /2 for CALL through a register or memory. A REX appears only for the
|
||||||
|
// extended registers, never REX.W: the branch operand size is fixed at 64
|
||||||
|
// bits in long mode.
|
||||||
|
func TestIndirectControlFlow(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"JMP AX", "JMP", []Operand{AX}, "ffe0"},
|
||||||
|
{"CALL AX", "CALL", []Operand{AX}, "ffd0"},
|
||||||
|
{"JMP (BX)", "JMP", []Operand{Ptr(BX, 0, 8)}, "ff23"},
|
||||||
|
{"CALL (BX)", "CALL", []Operand{Ptr(BX, 0, 8)}, "ff13"},
|
||||||
|
{"JMP 8(BX)", "JMP", []Operand{Ptr(BX, 8, 8)}, "ff6308"},
|
||||||
|
{"CALL -16(BX)", "CALL", []Operand{Ptr(BX, -16, 8)}, "ff53f0"},
|
||||||
|
{"JMP R8", "JMP", []Operand{Reg{idx: 8, size: 2}}, "41ffe0"},
|
||||||
|
{"CALL R9", "CALL", []Operand{Reg{idx: 9, size: 2}}, "41ffd1"},
|
||||||
|
{"JMP R15", "JMP", []Operand{Reg{idx: 15, size: 2}}, "41ffe7"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestSSEMoveGroundTruth checks the legacy (non-VEX) SSE moves byte for byte
|
// TestSSEMoveGroundTruth checks the legacy (non-VEX) SSE moves byte for byte
|
||||||
// against the Go assembler. wantOp is the decoder's name, which differs from
|
// against the Go assembler. wantOp is the decoder's name, which differs from
|
||||||
// the Plan 9 spelling for the octa moves (MOVOU = MOVDQU, MOVO = MOVDQA).
|
// the Plan 9 spelling for the octa moves (MOVOU = MOVDQU, MOVO = MOVDQA).
|
||||||
@@ -230,7 +302,7 @@ func TestSSEMoveGroundTruth(t *testing.T) {
|
|||||||
// TestGoFlacScalarTail encodes the scalar tail of an analyze kernel to confirm
|
// TestGoFlacScalarTail encodes the scalar tail of an analyze kernel to confirm
|
||||||
// the encoder handles a realistic instruction sequence.
|
// the encoder handles a realistic instruction sequence.
|
||||||
func TestGoFlacScalarTail(t *testing.T) {
|
func TestGoFlacScalarTail(t *testing.T) {
|
||||||
// MOVQ swin_base+0(FP), SI — modelled as MOVQ disp(reg), reg.
|
// MOVQ swin_base+0(FP), SI; modelled as MOVQ disp(reg), reg.
|
||||||
checkSyntax(t, "mov rsi, qword ptr [rax+0x10]", "MOVQ", Ptr(AX, 0x10, 8), SI)
|
checkSyntax(t, "mov rsi, qword ptr [rax+0x10]", "MOVQ", Ptr(AX, 0x10, 8), SI)
|
||||||
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
|
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
|
||||||
checkSyntax(t, "and r10, -0x8", "ANDQ", Imm(-8), Reg{idx: 10, size: 8})
|
checkSyntax(t, "and r10, -0x8", "ANDQ", Imm(-8), Reg{idx: 10, size: 8})
|
||||||
@@ -285,6 +357,18 @@ func TestScalarGroundTruth(t *testing.T) {
|
|||||||
{"MOVBQZX AL,R8", "MOVBQZX", []Operand{AL, r8}, "4c0fb6c0", "MOVZX"},
|
{"MOVBQZX AL,R8", "MOVBQZX", []Operand{AL, r8}, "4c0fb6c0", "MOVZX"},
|
||||||
{"MOVWLZX AX,CX", "MOVWLZX", []Operand{AX, CX}, "0fb7c8", "MOVZX"},
|
{"MOVWLZX AX,CX", "MOVWLZX", []Operand{AX, CX}, "0fb7c8", "MOVZX"},
|
||||||
{"MOVWQZX AX,R8", "MOVWQZX", []Operand{AX, r8}, "4c0fb7c0", "MOVZX"},
|
{"MOVWQZX AX,R8", "MOVWQZX", []Operand{AX, r8}, "4c0fb7c0", "MOVZX"},
|
||||||
|
// The width pairs the toolchain accepts and GOROOT uses; bytes
|
||||||
|
// pinned from go tool asm (see testdata/verify/widen_amd64.s).
|
||||||
|
{"MOVBWZX (BX),R11W", "MOVBWZX", []Operand{Ptr(BX, 0, 1), Reg{idx: 11, size: 2}}, "66440fb61b", "MOVZX"},
|
||||||
|
{"MOVBWSX (BX),R11W", "MOVBWSX", []Operand{Ptr(BX, 0, 1), Reg{idx: 11, size: 2}}, "66440fbe1b", "MOVSX"},
|
||||||
|
{"MOVBLSX (BX),AX", "MOVBLSX", []Operand{Ptr(BX, 0, 1), AX}, "0fbe03", "MOVSX"},
|
||||||
|
{"MOVBQSX (BX),R8", "MOVBQSX", []Operand{Ptr(BX, 0, 1), r8}, "4c0fbe03", "MOVSX"},
|
||||||
|
{"MOVWQSX (BX),R9", "MOVWQSX", []Operand{Ptr(BX, 0, 2), r9}, "4c0fbf0b", "MOVSX"},
|
||||||
|
// A long to quad zero-extend is a plain 32-bit move.
|
||||||
|
{"MOVLQZX (BX),DX", "MOVLQZX", []Operand{Ptr(BX, 0, 4), DX}, "8b13", "MOV"},
|
||||||
|
{"MOVLQZX AX,DX", "MOVLQZX", []Operand{AX, DX}, "8bd0", "MOV"},
|
||||||
|
{"PMOVMSKB X1,AX", "PMOVMSKB", []Operand{vreg(t, "X1"), AX}, "660fd7c1", "PMOVMSKB"},
|
||||||
|
{"PMOVMSKB X11,CX", "PMOVMSKB", []Operand{vreg(t, "X11"), CX}, "66410fd7cb", "PMOVMSKB"},
|
||||||
{"CVTSL2SD R8,X13", "CVTSL2SD", []Operand{r8, vreg(t, "X13")}, "f2450f2ae8", "CVTSI2SD"},
|
{"CVTSL2SD R8,X13", "CVTSL2SD", []Operand{r8, vreg(t, "X13")}, "f2450f2ae8", "CVTSI2SD"},
|
||||||
{"CVTSL2SD AX,X0", "CVTSL2SD", []Operand{AX, vreg(t, "X0")}, "f20f2ac0", "CVTSI2SD"},
|
{"CVTSL2SD AX,X0", "CVTSL2SD", []Operand{AX, vreg(t, "X0")}, "f20f2ac0", "CVTSI2SD"},
|
||||||
{"CVTSQ2SD R8,X13", "CVTSQ2SD", []Operand{r8, vreg(t, "X13")}, "f24d0f2ae8", "CVTSI2SD"},
|
{"CVTSQ2SD R8,X13", "CVTSQ2SD", []Operand{r8, vreg(t, "X13")}, "f24d0f2ae8", "CVTSI2SD"},
|
||||||
@@ -364,6 +448,346 @@ func TestScalarErrors(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestImmediateOutOfRange pins the go-tool-asm parity of the immediate and
|
||||||
|
// displacement spans: a scalar immediate must fit a signed or unsigned 32-bit
|
||||||
|
// word (only MOVQ reg, $imm takes the full int64), a scalar shift count must
|
||||||
|
// be an unsigned byte, and a displacement must fit int32. Every rejected
|
||||||
|
// shape here is rejected by `go tool asm` too; every accepted one encodes the
|
||||||
|
// same bytes.
|
||||||
|
func TestImmediateOutOfRange(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
}{
|
||||||
|
{"SHLQ count 300", "SHLQ", []Operand{Imm(300), AX}},
|
||||||
|
{"SHLQ count -1", "SHLQ", []Operand{Imm(-1), AX}},
|
||||||
|
{"SHLW count 256", "SHLW", []Operand{Imm(256), DX}},
|
||||||
|
{"SHLB count 300", "SHLB", []Operand{Imm(300), BL}},
|
||||||
|
{"MOVL imm32+", "MOVL", []Operand{Imm(4294967296), AX}},
|
||||||
|
{"MOVL imm32-", "MOVL", []Operand{Imm(-2147483649), AX}},
|
||||||
|
{"MOVW imm32+", "MOVW", []Operand{Imm(4294967296), AX}},
|
||||||
|
{"MOVB imm32+", "MOVB", []Operand{Imm(4294967296), AL}},
|
||||||
|
{"ADDB imm32+", "ADDB", []Operand{Imm(4294967296), AL}},
|
||||||
|
{"ADDL imm32+", "ADDL", []Operand{Imm(4294967296), AX}},
|
||||||
|
{"ADDQ imm32+", "ADDQ", []Operand{Imm(8589934592), AX}},
|
||||||
|
{"CMPQ imm32+", "CMPQ", []Operand{AX, Imm(4294967296)}},
|
||||||
|
{"CMPQ imm32-", "CMPQ", []Operand{AX, Imm(-2147483649)}},
|
||||||
|
{"TESTL imm32+", "TESTL", []Operand{Imm(4294967296), AX}},
|
||||||
|
{"IMUL3L imm32+", "IMUL3L", []Operand{Imm(4294967296), CX, DX}},
|
||||||
|
{"PUSHQ imm32+", "PUSHQ", []Operand{Imm(4294967296)}},
|
||||||
|
{"MOVQ mem imm32+", "MOVQ", []Operand{Imm(4294967296), Ptr(AX, 0, 8)}},
|
||||||
|
{"disp32+", "MOVQ", []Operand{Ptr(AX, 4294967296, 8), BX}},
|
||||||
|
{"disp32+ max", "MOVQ", []Operand{Ptr(AX, 2147483648, 8), BX}},
|
||||||
|
{"disp32-", "MOVQ", []Operand{Ptr(AX, -2147483649, 8), BX}},
|
||||||
|
{"VEX disp32+", "VMOVDQU", []Operand{Ptr(AX, 4294967296, 32), vreg(t, "Y1")}},
|
||||||
|
{"EVEX disp32+", "VMOVDQU32", []Operand{Ptr(AX, 4294967296, 64), vreg(t, "Z1")}},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", c.name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestImmediateTruncation pins the toolchain-matching truncations inside the
|
||||||
|
// accepted 32-bit span: the narrower fields take the low bits silently, byte
|
||||||
|
// for byte with `go tool asm` (which rejects none of these).
|
||||||
|
func TestImmediateTruncation(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"ADDB $256,BL", "ADDB", []Operand{Imm(256), BL}, "80c300"},
|
||||||
|
{"ADDB $1000,BL", "ADDB", []Operand{Imm(1000), BL}, "80c3e8"},
|
||||||
|
{"MOVB $256,AL", "MOVB", []Operand{Imm(256), AL}, "b000"},
|
||||||
|
{"MOVB $-129,AL", "MOVB", []Operand{Imm(-129), AL}, "b07f"},
|
||||||
|
{"MOVW $65536,AX", "MOVW", []Operand{Imm(65536), AX}, "66b80000"},
|
||||||
|
{"MOVW $65535,AX", "MOVW", []Operand{Imm(65535), AX}, "66b8ffff"},
|
||||||
|
{"MOVW $-32769,AX", "MOVW", []Operand{Imm(-32769), AX}, "66b8ff7f"},
|
||||||
|
{"MOVL $4294967295,AX", "MOVL", []Operand{Imm(4294967295), AX}, "b8ffffffff"},
|
||||||
|
{"ADDQ $4294967295,AX", "ADDQ", []Operand{Imm(4294967295), AX}, "4805ffffffff"},
|
||||||
|
{"CMPB BL,$255", "CMPB", []Operand{BL, Imm(255)}, "80fbff"},
|
||||||
|
{"CMPQ AX,$4294967295", "CMPQ", []Operand{AX, Imm(4294967295)}, "483dffffffff"},
|
||||||
|
{"MOVQ $4294967295,0(AX)", "MOVQ", []Operand{Imm(4294967295), Ptr(AX, 0, 8)}, "48c700ffffffff"},
|
||||||
|
{"SHLQ $255,AX", "SHLQ", []Operand{Imm(255), AX}, "48c1e0ff"},
|
||||||
|
{"SHLQ $0,AX", "SHLQ", []Operand{Imm(0), AX}, "48c1e000"},
|
||||||
|
// The one form beyond the 32-bit span: the imm64 MOVQ register move.
|
||||||
|
{"MOVQ $4294967296,AX", "MOVQ", []Operand{Imm(4294967296), AX}, "48b80000000001000000"},
|
||||||
|
{"MOVQ disp32 max", "MOVQ", []Operand{Ptr(AX, 2147483647, 8), BX}, "488b98ffffff7f"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEncodableCmovSize pins the linter contract for CMOVcc: Encodable must
|
||||||
|
// reject the spellings Encode rejects, so a mnemonic like CMOVBGT (no size
|
||||||
|
// letter) is not reported as encodable.
|
||||||
|
func TestEncodableCmovSize(t *testing.T) {
|
||||||
|
for _, m := range []string{"CMOVBGT", "CMOVXEQ", "CMOVB", "CMOV", "CMOVWXX"} {
|
||||||
|
if Encodable(m) {
|
||||||
|
t.Errorf("Encodable(%q) = true, want false", m)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, m := range []string{"CMOVLGT", "CMOVQGT", "CMOVWLS", "CMOVLEQ"} {
|
||||||
|
if !Encodable(m) {
|
||||||
|
t.Errorf("Encodable(%q) = false, want true", m)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCarryShiftMulGroundTruth pins the carry-flag ALU family (ADC/SBB with
|
||||||
|
// their accumulator immediate forms), the rotate family, MUL/DIV/IDIV and the
|
||||||
|
// bit-test family byte for byte against go tool asm (see
|
||||||
|
// testdata/verify/scalar_amd64.s).
|
||||||
|
func TestCarryShiftMulGroundTruth(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"ADCQ AX,BX", "ADCQ", []Operand{AX, BX}, "4811c3"},
|
||||||
|
{"ADCL AX,BX", "ADCL", []Operand{AX, BX}, "11c3"},
|
||||||
|
{"ADCB AL,BL", "ADCB", []Operand{AL, BL}, "10c3"},
|
||||||
|
{"ADCW AX,BX", "ADCW", []Operand{AX, BX}, "6611c3"},
|
||||||
|
{"SBBQ AX,BX", "SBBQ", []Operand{AX, BX}, "4819c3"},
|
||||||
|
{"ADCQ $5,BX", "ADCQ", []Operand{Imm(5), BX}, "4883d305"},
|
||||||
|
{"ADCQ $300,BX", "ADCQ", []Operand{Imm(300), BX}, "4881d32c010000"},
|
||||||
|
{"ADCQ $300,AX", "ADCQ", []Operand{Imm(300), AX}, "48152c010000"},
|
||||||
|
{"ADCB $5,AL", "ADCB", []Operand{Imm(5), AL}, "1405"},
|
||||||
|
{"SBBQ $300,AX", "SBBQ", []Operand{Imm(300), AX}, "481d2c010000"},
|
||||||
|
{"ADCQ AX,(BX)", "ADCQ", []Operand{AX, Ptr(BX, 0, 8)}, "481103"},
|
||||||
|
{"ROLQ $3,AX", "ROLQ", []Operand{Imm(3), AX}, "48c1c003"},
|
||||||
|
{"ROLL CX,BX", "ROLL", []Operand{CL, BX}, "d3c3"},
|
||||||
|
{"RORQ CL,AX", "RORQ", []Operand{CL, AX}, "48d3c8"},
|
||||||
|
{"RCRQ $1,BX", "RCRQ", []Operand{Imm(1), BX}, "48d1db"},
|
||||||
|
{"RCLQ $3,AX", "RCLQ", []Operand{Imm(3), AX}, "48c1d003"},
|
||||||
|
{"RORB CL,BL", "RORB", []Operand{CL, BL}, "d2cb"},
|
||||||
|
{"SALQ $2,AX", "SALQ", []Operand{Imm(2), AX}, "48c1e002"},
|
||||||
|
{"ROLW $1,AX", "ROLW", []Operand{Imm(1), AX}, "66d1c0"},
|
||||||
|
{"MULQ CX", "MULQ", []Operand{CX}, "48f7e1"},
|
||||||
|
{"MULL CX", "MULL", []Operand{CX}, "f7e1"},
|
||||||
|
{"MULB CL", "MULB", []Operand{CL}, "f6e1"},
|
||||||
|
{"DIVL CX", "DIVL", []Operand{CX}, "f7f1"},
|
||||||
|
{"IDIVQ CX", "IDIVQ", []Operand{CX}, "48f7f9"},
|
||||||
|
{"MULW CX", "MULW", []Operand{CX}, "66f7e1"},
|
||||||
|
{"BTQ AX,DX", "BTQ", []Operand{AX, DX}, "480fa3c2"},
|
||||||
|
{"BTL AX,DX", "BTL", []Operand{AX, DX}, "0fa3c2"},
|
||||||
|
{"BTW AX,DX", "BTW", []Operand{AX, DX}, "660fa3c2"},
|
||||||
|
{"BTQ $3,BX", "BTQ", []Operand{Imm(3), BX}, "480fbae303"},
|
||||||
|
{"BTQ $3,(AX)", "BTQ", []Operand{Imm(3), Ptr(AX, 0, 8)}, "480fba2003"},
|
||||||
|
{"BTSQ $5,BX", "BTSQ", []Operand{Imm(5), BX}, "480fbaeb05"},
|
||||||
|
{"BTCQ AX,BX", "BTCQ", []Operand{AX, BX}, "480fbbc3"},
|
||||||
|
{"BTRQ $7,BX", "BTRQ", []Operand{Imm(7), BX}, "480fbaf307"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The bit-test immediate is an unsigned bit index with the negative
|
||||||
|
// spelling accepted, the shuffle convention: BTQ $300 must be rejected.
|
||||||
|
if _, err := Encode("BTQ", Imm(300), AX); err == nil {
|
||||||
|
t.Errorf("BTQ $300: expected an error, got none")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAtomicSystemGroundTruth pins the exchange/compare-exchange/accumulate
|
||||||
|
// family, the string primitives, the flag and system instructions, the MXCSR
|
||||||
|
// pair, the scalar float-to-int conversions and the x87 FMOVD byte for byte
|
||||||
|
// against go tool asm (see testdata/verify/atomics_amd64.s and
|
||||||
|
// testdata/verify/system_amd64.s).
|
||||||
|
func TestAtomicSystemGroundTruth(t *testing.T) {
|
||||||
|
r8 := Reg{idx: 8, size: 8}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"XCHGQ AX,BX", "XCHGQ", []Operand{AX, BX}, "4893"},
|
||||||
|
{"XCHGQ BX,AX", "XCHGQ", []Operand{BX, AX}, "4893"},
|
||||||
|
{"XCHGL AX,BX", "XCHGL", []Operand{AX, BX}, "93"},
|
||||||
|
{"XCHGB AL,BL", "XCHGB", []Operand{AL, BL}, "86c3"},
|
||||||
|
{"XCHGW AX,BX", "XCHGW", []Operand{AX, BX}, "6693"},
|
||||||
|
{"XCHGQ R8,R9", "XCHGQ", []Operand{r8, Reg{idx: 9, size: 8}}, "4d87c1"},
|
||||||
|
{"XCHGQ BX,(AX)", "XCHGQ", []Operand{BX, Ptr(AX, 0, 8)}, "488718"},
|
||||||
|
{"XCHGQ (AX),BX", "XCHGQ", []Operand{Ptr(AX, 0, 8), BX}, "488718"},
|
||||||
|
{"XCHGQ AX,(BX)", "XCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "488703"},
|
||||||
|
{"CMPXCHGL AX,BX", "CMPXCHGL", []Operand{AX, BX}, "0fb1c3"},
|
||||||
|
{"CMPXCHGQ AX,(BX)", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
|
||||||
|
{"CMPXCHGB AL,(BX)", "CMPXCHGB", []Operand{AL, Ptr(BX, 0, 1)}, "0fb003"},
|
||||||
|
{"CMPXCHGW AX,BX", "CMPXCHGW", []Operand{AX, BX}, "660fb1c3"},
|
||||||
|
{"XADDL AX,BX", "XADDL", []Operand{AX, BX}, "0fc1c3"},
|
||||||
|
{"XADDQ AX,(BX)", "XADDQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fc103"},
|
||||||
|
{"XADDB AL,(BX)", "XADDB", []Operand{AL, Ptr(BX, 0, 1)}, "0fc003"},
|
||||||
|
{"XADDW AX,BX", "XADDW", []Operand{AX, BX}, "660fc1c3"},
|
||||||
|
{"ADCXL AX,CX", "ADCXL", []Operand{AX, CX}, "660f38f6c8"},
|
||||||
|
{"ADCXQ AX,CX", "ADCXQ", []Operand{AX, CX}, "66480f38f6c8"},
|
||||||
|
{"ADOXL AX,CX", "ADOXL", []Operand{AX, CX}, "f30f38f6c8"},
|
||||||
|
{"ADOXQ AX,CX", "ADOXQ", []Operand{AX, CX}, "f3480f38f6c8"},
|
||||||
|
{"CRC32B AX,CX", "CRC32B", []Operand{AX, CX}, "f20f38f0c8"},
|
||||||
|
{"CRC32W AX,CX", "CRC32W", []Operand{AX, CX}, "66f20f38f1c8"},
|
||||||
|
{"CRC32L AX,CX", "CRC32L", []Operand{AX, CX}, "f20f38f1c8"},
|
||||||
|
{"CRC32Q AX,CX", "CRC32Q", []Operand{AX, CX}, "f2480f38f1c8"},
|
||||||
|
{"CRC32L (AX),CX", "CRC32L", []Operand{Ptr(AX, 0, 4), CX}, "f20f38f108"},
|
||||||
|
{"MOVSQ", "MOVSQ", []Operand{}, "48a5"},
|
||||||
|
{"MOVSL", "MOVSL", []Operand{}, "a5"},
|
||||||
|
{"MOVSB", "MOVSB", []Operand{}, "a4"},
|
||||||
|
{"MOVSW", "MOVSW", []Operand{}, "66a5"},
|
||||||
|
{"STOSB", "STOSB", []Operand{}, "aa"},
|
||||||
|
{"STOSQ", "STOSQ", []Operand{}, "48ab"},
|
||||||
|
{"STOSL", "STOSL", []Operand{}, "ab"},
|
||||||
|
{"STOSW", "STOSW", []Operand{}, "66ab"},
|
||||||
|
{"CLD", "CLD", []Operand{}, "fc"},
|
||||||
|
{"STD", "STD", []Operand{}, "fd"},
|
||||||
|
{"POPFQ", "POPFQ", []Operand{}, "9d"},
|
||||||
|
{"PUSHFQ", "PUSHFQ", []Operand{}, "9c"},
|
||||||
|
{"CPUID", "CPUID", []Operand{}, "0fa2"},
|
||||||
|
{"RDTSC", "RDTSC", []Operand{}, "0f31"},
|
||||||
|
{"RDTSCP", "RDTSCP", []Operand{}, "0f01f9"},
|
||||||
|
{"SYSCALL", "SYSCALL", []Operand{}, "0f05"},
|
||||||
|
{"XGETBV", "XGETBV", []Operand{}, "0f01d0"},
|
||||||
|
{"PAUSE", "PAUSE", []Operand{}, "f390"},
|
||||||
|
{"LFENCE", "LFENCE", []Operand{}, "0faee8"},
|
||||||
|
{"MFENCE", "MFENCE", []Operand{}, "0faef0"},
|
||||||
|
{"SFENCE", "SFENCE", []Operand{}, "0faef8"},
|
||||||
|
{"UNDEF", "UNDEF", []Operand{}, "0f0b"},
|
||||||
|
{"INT $3", "INT", []Operand{Imm(3)}, "cd03"},
|
||||||
|
{"LDMXCSR (AX)", "LDMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae10"},
|
||||||
|
{"STMXCSR (AX)", "STMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae18"},
|
||||||
|
{"CVTSD2SL X0,AX", "CVTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2dc0"},
|
||||||
|
{"CVTTSD2SQ X0,AX", "CVTTSD2SQ", []Operand{vreg(t, "X0"), AX}, "f2480f2cc0"},
|
||||||
|
{"CVTTSD2SL X0,AX", "CVTTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2cc0"},
|
||||||
|
{"CVTSS2SQ X0,AX", "CVTSS2SQ", []Operand{vreg(t, "X0"), AX}, "f3480f2dc0"},
|
||||||
|
{"FMOVD (AX),F0", "FMOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "F0")}, "dd00"},
|
||||||
|
{"FMOVD F0,(AX)", "FMOVD", []Operand{vreg(t, "F0"), Ptr(AX, 0, 8)}, "dd10"},
|
||||||
|
{"FMOVD F0,F1", "FMOVD", []Operand{vreg(t, "F0"), vreg(t, "F1")}, "ddd1"},
|
||||||
|
{"MOVD AX,X0", "MOVD", []Operand{AX, vreg(t, "X0")}, "66480f6ec0"},
|
||||||
|
{"MOVD X0,AX", "MOVD", []Operand{vreg(t, "X0"), AX}, "66480f7ec0"},
|
||||||
|
{"MOVD X0,X1", "MOVD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f30f7ec8"},
|
||||||
|
{"MOVD (AX),X0", "MOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "X0")}, "f30f7e00"},
|
||||||
|
{"MOVD X0,(AX)", "MOVD", []Operand{vreg(t, "X0"), Ptr(AX, 0, 8)}, "660fd600"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// LDMXCSR/STMXCSR take a memory operand only.
|
||||||
|
if _, err := Encode("LDMXCSR", AX); err == nil {
|
||||||
|
t.Errorf("LDMXCSR AX: expected an error, got none")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSSEGapsGroundTruth pins the legacy SSE gap families: the scalar
|
||||||
|
// compare and square root, the Plan 9 packed spellings, the imm8-controlled
|
||||||
|
// shuffles, the lane extracts and inserts, the packed integer shifts and the
|
||||||
|
// AES/SHA round instructions, byte for byte against go tool asm (see
|
||||||
|
// testdata/verify/crypto_amd64.s and testdata/verify/sse_amd64.s).
|
||||||
|
func TestSSEGapsGroundTruth(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"ANDNPD X0,X1", "ANDNPD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f55c8"},
|
||||||
|
{"ANDNPS X0,X1", "ANDNPS", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f55c8"},
|
||||||
|
{"COMISD X0,X1", "COMISD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f2fc8"},
|
||||||
|
{"SQRTSD X0,X1", "SQRTSD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f20f51c8"},
|
||||||
|
{"PSHUFL $3,X0,X1", "PSHUFL", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f70c803"},
|
||||||
|
{"PALIGNR $2,X0,X1", "PALIGNR", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0fc802"},
|
||||||
|
{"PBLENDW $3,X0,X1", "PBLENDW", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0ec803"},
|
||||||
|
{"PCMPESTRI $1,X0,X1", "PCMPESTRI", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}, "660f3a61c801"},
|
||||||
|
{"PCLMULQDQ $0,X0,X1", "PCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3a44c800"},
|
||||||
|
{"PCLMULQDQ $0,(AX),X1", "PCLMULQDQ", []Operand{Imm(0), Ptr(AX, 0, 16), vreg(t, "X1")}, "660f3a440800"},
|
||||||
|
{"PEXTRB $1,X0,AX", "PEXTRB", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a14c001"},
|
||||||
|
{"PEXTRD $1,X0,AX", "PEXTRD", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a16c001"},
|
||||||
|
{"PEXTRQ $1,X0,AX", "PEXTRQ", []Operand{Imm(1), vreg(t, "X0"), AX}, "66480f3a16c001"},
|
||||||
|
{"PEXTRW $1,X0,AX", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), AX}, "660fc5c001"},
|
||||||
|
{"PEXTRW $1,X0,(AX)", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), Ptr(AX, 0, 2)}, "660f3a150001"},
|
||||||
|
{"PINSRB $1,AX,X0", "PINSRB", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a20c001"},
|
||||||
|
{"PINSRD $1,AX,X0", "PINSRD", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a22c001"},
|
||||||
|
{"PINSRQ $1,AX,X0", "PINSRQ", []Operand{Imm(1), AX, vreg(t, "X0")}, "66480f3a22c001"},
|
||||||
|
{"PINSRW $1,AX,X0", "PINSRW", []Operand{Imm(1), AX, vreg(t, "X0")}, "660fc4c001"},
|
||||||
|
{"PINSRW $1,(AX),X0", "PINSRW", []Operand{Imm(1), Ptr(AX, 0, 2), vreg(t, "X0")}, "660fc40001"},
|
||||||
|
{"PSLLL $2,X0", "PSLLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72f002"},
|
||||||
|
{"PSRAL $2,X0", "PSRAL", []Operand{Imm(2), vreg(t, "X0")}, "660f72e002"},
|
||||||
|
{"PSRLL $2,X0", "PSRLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72d002"},
|
||||||
|
{"PSRLQ $2,X0", "PSRLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d002"},
|
||||||
|
{"PSLLQ $2,X0", "PSLLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f002"},
|
||||||
|
{"PSLLW $2,X0", "PSLLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71f002"},
|
||||||
|
{"PSRLW $2,X0", "PSRLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71d002"},
|
||||||
|
{"PSRAW $2,X0", "PSRAW", []Operand{Imm(2), vreg(t, "X0")}, "660f71e002"},
|
||||||
|
{"PSLLDQ $2,X0", "PSLLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f802"},
|
||||||
|
{"PSRLDQ $2,X0", "PSRLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d802"},
|
||||||
|
{"PSLLL X0,X1", "PSLLL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ff2c8"},
|
||||||
|
{"PSRLQ X0,X1", "PSRLQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660fd3c8"},
|
||||||
|
{"PSLLL (AX),X1", "PSLLL", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660ff208"},
|
||||||
|
{"PSUBL X0,X1", "PSUBL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffac8"},
|
||||||
|
{"PADDL X0,X1", "PADDL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffec8"},
|
||||||
|
{"PCMPEQL X0,X1", "PCMPEQL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f76c8"},
|
||||||
|
{"PUNPCKLBW X0,X1", "PUNPCKLBW", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f60c8"},
|
||||||
|
{"MOVOA X0,X1", "MOVOA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f6fc8"},
|
||||||
|
{"MOVOA (AX),X1", "MOVOA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f6f08"},
|
||||||
|
{"MOVOA X0,(AX)", "MOVOA", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "660f7f00"},
|
||||||
|
{"AESIMC X0,X1", "AESIMC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dbc8"},
|
||||||
|
{"AESIMC (AX),X1", "AESIMC", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f38db08"},
|
||||||
|
{"AESENC X0,X1", "AESENC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dcc8"},
|
||||||
|
{"AESENCLAST X0,X1", "AESENCLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38ddc8"},
|
||||||
|
{"AESDEC X0,X1", "AESDEC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dec8"},
|
||||||
|
{"AESDECLAST X0,X1", "AESDECLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dfc8"},
|
||||||
|
{"AESKEYGENASSIST $0,X0,X1", "AESKEYGENASSIST", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3adfc800"},
|
||||||
|
{"SHA1MSG1 X0,X1", "SHA1MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c9c8"},
|
||||||
|
{"SHA1MSG2 X0,X1", "SHA1MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cac8"},
|
||||||
|
{"SHA1NEXTE X0,X1", "SHA1NEXTE", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c8c8"},
|
||||||
|
{"SHA1RNDS4 $0,X0,X1", "SHA1RNDS4", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "0f3accc800"},
|
||||||
|
{"SHA256MSG1 X0,X1", "SHA256MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38ccc8"},
|
||||||
|
{"SHA256MSG2 X0,X1", "SHA256MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cdc8"},
|
||||||
|
{"SHA256RNDS2 X0,X1,X2", "SHA256RNDS2", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "0f38cbd1"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// SHA256RNDS2's first operand must be the literal X0.
|
||||||
|
if _, err := Encode("SHA256RNDS2", vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")); err == nil {
|
||||||
|
t.Errorf("SHA256RNDS2 X1,...: expected an error, got none")
|
||||||
|
}
|
||||||
|
// PSLLDQ has no variable-count form.
|
||||||
|
if _, err := Encode("PSLLDQ", vreg(t, "X0"), vreg(t, "X1")); err == nil {
|
||||||
|
t.Errorf("PSLLDQ X0,X1: expected an error, got none")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
|
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
|
||||||
// byte for byte (no prefix / 66 / F2 / F3 variants).
|
// byte for byte (no prefix / 66 / F2 / F3 variants).
|
||||||
func TestSSEBinGroundTruth(t *testing.T) {
|
func TestSSEBinGroundTruth(t *testing.T) {
|
||||||
@@ -421,7 +845,7 @@ func TestSSEShuffleGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestMOVQXMMGroundTruth pins the SSE2 packed-quadword move encodings:
|
// TestMOVQXMMGroundTruth pins the SSE2 packed-quadword move encodings:
|
||||||
// loads and register moves on F3 0F 7E, stores on 66 0F D6 — the forms
|
// loads and register moves on F3 0F 7E, stores on 66 0F D6; the forms
|
||||||
// the GPR-move fallback silently corrupted.
|
// the GPR-move fallback silently corrupted.
|
||||||
func TestMOVQXMMGroundTruth(t *testing.T) {
|
func TestMOVQXMMGroundTruth(t *testing.T) {
|
||||||
cases := []struct {
|
cases := []struct {
|
||||||
|
|||||||
+72
-38
@@ -180,10 +180,19 @@ var evexTable = map[string]evexSpec{
|
|||||||
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
// EVEX.66.0F38, permutes (NDS form).
|
// EVEX.66.0F38, permutes (NDS form).
|
||||||
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMI2B": {2, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F38, population count (reg=dst, rm=src; W selects byte/word
|
||||||
|
// against dword/qword).
|
||||||
|
"VPOPCNTB": {2, 0x54, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPOPCNTD": {2, 0x55, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPOPCNTQ": {2, 0x55, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F.W1, the qword spelling of the packed OR (VPORQ has no VEX
|
||||||
|
// form in the Go assembler: it always encodes through EVEX).
|
||||||
|
"VPORQ": {1, 0xEB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
@@ -509,40 +518,43 @@ var evexBcastTable = map[string]evexBcastSpec{
|
|||||||
}
|
}
|
||||||
|
|
||||||
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
|
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
|
||||||
// move table).
|
// move table). vecOK and xmmOnly mirror the VEX twin's operand rules: a
|
||||||
|
// scalar move (vecOK false, xmmOnly true) takes XMM↔memory operands only.
|
||||||
type evexMoveSpec struct {
|
type evexMoveSpec struct {
|
||||||
mapSel int
|
mapSel int
|
||||||
pp int
|
pp int
|
||||||
load byte // r/m → vector
|
load byte // r/m → vector
|
||||||
store byte // vector → r/m
|
store byte // vector → r/m
|
||||||
w int
|
w int
|
||||||
n [3]int
|
n [3]int
|
||||||
|
vecOK bool // the non-memory operand may be a vector register
|
||||||
|
xmmOnly bool // wider than XMM registers are rejected
|
||||||
}
|
}
|
||||||
|
|
||||||
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
|
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
|
||||||
var evexMoveTable = map[string]evexMoveSpec{
|
var evexMoveTable = map[string]evexMoveSpec{
|
||||||
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
|
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
|
||||||
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
|
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
|
||||||
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
|
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
|
||||||
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
|
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
|
||||||
// semantics).
|
// semantics).
|
||||||
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
|
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
|
||||||
// encoding).
|
// encoding).
|
||||||
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
|
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
|
||||||
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
|
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128/256/512, aligned packed moves.
|
// EVEX.128/256/512, aligned packed moves.
|
||||||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}},
|
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false},
|
||||||
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}},
|
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128/256/512.66.0F, aligned integer moves.
|
// EVEX.128/256/512.66.0F, aligned integer moves.
|
||||||
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
||||||
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
|
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
|
||||||
// three-operand register form is not supported).
|
// three-operand register form is not supported).
|
||||||
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}},
|
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true},
|
||||||
}
|
}
|
||||||
|
|
||||||
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
||||||
@@ -1022,6 +1034,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
|
|||||||
var rm Operand
|
var rm Operand
|
||||||
switch {
|
switch {
|
||||||
case srcIsVec && dstIsVec:
|
case srcIsVec && dstIsVec:
|
||||||
|
// A store-form reg-reg move, the layout the Go assembler uses; a
|
||||||
|
// scalar move has no two-register form at all (the register form
|
||||||
|
// takes three operands), matching the VEX twin's vecOK rule.
|
||||||
|
if !ms.vecOK {
|
||||||
|
return fmt.Errorf("%s does not take two vector registers", mnem)
|
||||||
|
}
|
||||||
reg, rm = srcReg, dst
|
reg, rm = srcReg, dst
|
||||||
case srcIsVec:
|
case srcIsVec:
|
||||||
if !memOperand(dst) {
|
if !memOperand(dst) {
|
||||||
@@ -1037,6 +1055,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
|
|||||||
default:
|
default:
|
||||||
return fmt.Errorf("%s needs a vector register operand", mnem)
|
return fmt.Errorf("%s needs a vector register operand", mnem)
|
||||||
}
|
}
|
||||||
|
// The scalar move is 128-bit only, so the register the length follows
|
||||||
|
// must be an XMM (the VEX twin's xmmOnly rule; EVEX also reaches ZMM,
|
||||||
|
// hence the inequality rather than a YMM test).
|
||||||
|
if ms.xmmOnly && reg.size != 16 {
|
||||||
|
return fmt.Errorf("%s operates on XMM registers only", mnem)
|
||||||
|
}
|
||||||
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
|
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
|
||||||
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx)
|
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx)
|
||||||
}
|
}
|
||||||
@@ -1171,9 +1195,6 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
|
|||||||
if r.idx&16 != 0 {
|
if r.idx&16 != 0 {
|
||||||
xBar = 0
|
xBar = 0
|
||||||
}
|
}
|
||||||
if r.idx&16 != 0 {
|
|
||||||
xBar = 0
|
|
||||||
}
|
|
||||||
case Mem:
|
case Mem:
|
||||||
var err error
|
var err error
|
||||||
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
|
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
|
||||||
@@ -1232,6 +1253,11 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
|
|||||||
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
|
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
|
||||||
sib = -1
|
sib = -1
|
||||||
xBar, bBar = 1, 1 // inverted bits: 1 = no extension
|
xBar, bBar = 1, 1 // inverted bits: 1 = no extension
|
||||||
|
// The disp32 fallback bounds the displacement by int32, and the
|
||||||
|
// compressed disp8 form reaches at most ±127×64, well inside it.
|
||||||
|
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
|
||||||
|
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
|
||||||
|
}
|
||||||
if !m.HasBase && !m.HasIndex {
|
if !m.HasBase && !m.HasIndex {
|
||||||
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
|
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
|
||||||
}
|
}
|
||||||
@@ -1418,18 +1444,20 @@ var evexKOperand = map[string]bool{
|
|||||||
}
|
}
|
||||||
|
|
||||||
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
||||||
// direction, kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
// direction, kk (k → k), kmem (k → mem), gprk (GPR/mem → k) and kgpr
|
||||||
// gprk (GPR/mem → K), kgpr (K → GPR), and the GPR forms carry a mandatory
|
// (k → GPR). Each direction group carries its own mandatory prefix and W:
|
||||||
// prefix and W for the wider widths.
|
// the k-destination/source forms share one pair, the GPR forms another.
|
||||||
type kmovSpec struct {
|
type kmovSpec struct {
|
||||||
kk, kmem, gprk, kgpr byte
|
kk, kmem, gprk, kgpr byte
|
||||||
gprPP int
|
kPP, kW int // prefix and VEX.W for the k forms
|
||||||
w int
|
gprPP, gprW int // prefix and VEX.W for the GPR forms
|
||||||
}
|
}
|
||||||
|
|
||||||
var kmovTable = map[string]kmovSpec{
|
var kmovTable = map[string]kmovSpec{
|
||||||
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0},
|
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0, 0, 0},
|
||||||
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 3, 1},
|
"KMOVB": {0x90, 0x91, 0x92, 0x93, 1, 0, 1, 0},
|
||||||
|
"KMOVD": {0x90, 0x91, 0x92, 0x93, 1, 1, 3, 0},
|
||||||
|
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 0, 1, 3, 1},
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeKmov encodes a KMOV width, selecting the opcode by direction.
|
// encodeKmov encodes a KMOV width, selecting the opcode by direction.
|
||||||
@@ -1443,14 +1471,14 @@ func (e *enc) encodeKmov(upper string, ops []Operand) error {
|
|||||||
dstReg, dstIsReg := dst.(Reg)
|
dstReg, dstIsReg := dst.(Reg)
|
||||||
srcK := srcIsReg && srcReg.mask
|
srcK := srcIsReg && srcReg.mask
|
||||||
dstK := dstIsReg && dstReg.mask
|
dstK := dstIsReg && dstReg.mask
|
||||||
spec := vexSpec{mapSel: 1, w: ks.w, pp: 0, opdigit: -1}
|
|
||||||
switch {
|
switch {
|
||||||
case srcK && dstK:
|
case srcK && dstK:
|
||||||
spec.opcode = ks.kk // k ← k: reg = dst, rm = src
|
// k ← k: reg = dst, rm = src.
|
||||||
|
spec := vexSpec{mapSel: 1, opcode: ks.kk, w: ks.kW, pp: ks.kPP, opdigit: -1}
|
||||||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
||||||
case srcK && dstIsReg:
|
case srcK && dstIsReg:
|
||||||
spec.opcode = ks.kgpr // GPR ← k: reg = dst, rm = src
|
// GPR ← k: reg = dst, rm = src.
|
||||||
spec.pp = ks.gprPP
|
spec := vexSpec{mapSel: 1, opcode: ks.kgpr, w: ks.gprW, pp: ks.gprPP, opdigit: -1}
|
||||||
rBit := 0
|
rBit := 0
|
||||||
if dstReg.idx >= 8 {
|
if dstReg.idx >= 8 {
|
||||||
rBit = 1
|
rBit = 1
|
||||||
@@ -1460,11 +1488,17 @@ func (e *enc) encodeKmov(upper string, ops []Operand) error {
|
|||||||
if _, ok := dst.(Mem); !ok {
|
if _, ok := dst.(Mem); !ok {
|
||||||
return fmt.Errorf("%s: invalid destination operand", upper)
|
return fmt.Errorf("%s: invalid destination operand", upper)
|
||||||
}
|
}
|
||||||
spec.opcode = ks.kmem // mem ← k: reg = src, rm = dst
|
// mem ← k: reg = src, rm = dst.
|
||||||
|
spec := vexSpec{mapSel: 1, opcode: ks.kmem, w: ks.kW, pp: ks.kPP, opdigit: -1}
|
||||||
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
|
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
|
||||||
case dstK:
|
case dstK:
|
||||||
spec.opcode = ks.gprk // k ← GPR/mem: reg = dst, rm = src
|
// k ← GPR: reg = dst, rm = src. A memory source shares the k ← k
|
||||||
spec.pp = ks.gprPP
|
// opcode and prefix group (the ykmovb layout the Go assembler uses).
|
||||||
|
opcode, w, pp := ks.gprk, ks.gprW, ks.gprPP
|
||||||
|
if memOperand(src) {
|
||||||
|
opcode, w, pp = ks.kk, ks.kW, ks.kPP
|
||||||
|
}
|
||||||
|
spec := vexSpec{mapSel: 1, opcode: opcode, w: w, pp: pp, opdigit: -1}
|
||||||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("%s requires a K register operand", upper)
|
return fmt.Errorf("%s requires a K register operand", upper)
|
||||||
|
|||||||
+41
-13
@@ -16,7 +16,7 @@ import (
|
|||||||
// kernels use: NDS arithmetic, immediate and variable shifts, shuffles with
|
// kernels use: NDS arithmetic, immediate and variable shifts, shuffles with
|
||||||
// an immediate, lane extracts, narrowing stores, broadcasts from a GPR or
|
// an immediate, lane extracts, narrowing stores, broadcasts from a GPR or
|
||||||
// memory, mask destinations, mask moves, disp8×N compression and the 5-bit
|
// memory, mask destinations, mask moves, disp8×N compression and the 5-bit
|
||||||
// register fields (X/Y 16–31, Z 0–31).
|
// register fields (X/Y 16-31, Z 0-31).
|
||||||
func TestEvexGroundTruth(t *testing.T) {
|
func TestEvexGroundTruth(t *testing.T) {
|
||||||
cases := []struct {
|
cases := []struct {
|
||||||
name string
|
name string
|
||||||
@@ -38,6 +38,15 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
|
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
|
||||||
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
|
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
|
||||||
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
|
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
|
||||||
|
// The qword OR spelling always encodes through EVEX.
|
||||||
|
{"VPORQ Y0,Y1,Y2", "VPORQ", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "62f1f528ebd0"},
|
||||||
|
{"VPORQ X0,X1,X2", "VPORQ", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f1f508ebd0"},
|
||||||
|
// Byte permute and population count.
|
||||||
|
{"VPERMI2B X0,X1,X2", "VPERMI2B", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f2750875d0"},
|
||||||
|
{"VPOPCNTB X0,X1", "VPOPCNTB", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0854c8"},
|
||||||
|
{"VPOPCNTD X0,X1", "VPOPCNTD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0855c8"},
|
||||||
|
{"VPOPCNTD Y0,Y1", "VPOPCNTD", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "62f27d2855c8"},
|
||||||
|
{"VPOPCNTQ X0,X1", "VPOPCNTQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f2fd0855c8"},
|
||||||
// Align (NDS + imm8).
|
// Align (NDS + imm8).
|
||||||
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
|
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
|
||||||
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
|
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
|
||||||
@@ -52,13 +61,23 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
|
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
|
||||||
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
|
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
|
||||||
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
|
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
|
||||||
|
{"KMOVB K1,K2", "KMOVB", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f990d1"},
|
||||||
|
{"KMOVB AX,K1", "KMOVB", []Operand{AX, vreg(t, "K1")}, "c5f992c8"},
|
||||||
|
{"KMOVB K1,AX", "KMOVB", []Operand{vreg(t, "K1"), AX}, "c5f993c1"},
|
||||||
|
{"KMOVB K1,(AX)", "KMOVB", []Operand{vreg(t, "K1"), Ptr(AX, 0, 1)}, "c5f99108"},
|
||||||
|
{"KMOVD K1,K2", "KMOVD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f990d1"},
|
||||||
|
{"KMOVD AX,K1", "KMOVD", []Operand{AX, vreg(t, "K1")}, "c5fb92c8"},
|
||||||
|
{"KMOVD K1,AX", "KMOVD", []Operand{vreg(t, "K1"), AX}, "c5fb93c1"},
|
||||||
|
{"KMOVD K1,(AX)", "KMOVD", []Operand{vreg(t, "K1"), Ptr(AX, 0, 4)}, "c4e1f99108"},
|
||||||
|
{"KMOVB (AX),K1", "KMOVB", []Operand{Ptr(AX, 0, 1), vreg(t, "K1")}, "c5f99008"},
|
||||||
|
{"KMOVQ (AX),K1", "KMOVQ", []Operand{Ptr(AX, 0, 8), vreg(t, "K1")}, "c4e1f89008"},
|
||||||
// Moves, incl. disp8×N (64 for a 512-bit operand).
|
// Moves, incl. disp8×N (64 for a 512-bit operand).
|
||||||
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
|
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
|
||||||
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
|
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
|
||||||
{"VMOVDQU32 16(SI)(R15*4),Z4", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 16, 64), vreg(t, "Z4")}, "62b17e486fa4be10000000"},
|
{"VMOVDQU32 16(SI)(R15*4),Z4", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 16, 64), vreg(t, "Z4")}, "62b17e486fa4be10000000"},
|
||||||
{"VMOVDQU32 Z0,4(SI)(AX*1)", "VMOVDQU32", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f17e487f840604000000"},
|
{"VMOVDQU32 Z0,4(SI)(AX*1)", "VMOVDQU32", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f17e487f840604000000"},
|
||||||
{"VMOVDQU32 Z3,(DI)(R15*4)", "VMOVDQU32", []Operand{vreg(t, "Z3"), Idx(DI, vreg(t, "R15"), 4, 0, 64)}, "62b17e487f1cbf"},
|
{"VMOVDQU32 Z3,(DI)(R15*4)", "VMOVDQU32", []Operand{vreg(t, "Z3"), Idx(DI, vreg(t, "R15"), 4, 0, 64)}, "62b17e487f1cbf"},
|
||||||
// VMOVDQU64 — the W1 qword variant.
|
// VMOVDQU64; the W1 qword variant.
|
||||||
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
|
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
|
||||||
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
|
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
|
||||||
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
|
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
|
||||||
@@ -77,7 +96,7 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"VPSHUFB Z1,Z2,Z3", "VPSHUFB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4800d9"},
|
{"VPSHUFB Z1,Z2,Z3", "VPSHUFB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4800d9"},
|
||||||
{"VMOVDQU8 Z1,Z2", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17f487fca"},
|
{"VMOVDQU8 Z1,Z2", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17f487fca"},
|
||||||
{"VMOVDQU16 Z1,Z2", "VMOVDQU16", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff487fca"},
|
{"VMOVDQU16 Z1,Z2", "VMOVDQU16", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff487fca"},
|
||||||
// Indices 16–31: rm[4] rides in X̄ for register operands.
|
// Indices 16-31: rm[4] rides in X̄ for register operands.
|
||||||
{"VPSHUFD $1,X16,X17", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}, "62a17d0870c801"},
|
{"VPSHUFD $1,X16,X17", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}, "62a17d0870c801"},
|
||||||
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
|
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
|
||||||
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
|
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
|
||||||
@@ -96,7 +115,7 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"VPBROADCASTD 4(SI),Z10", "VPBROADCASTD", []Operand{Ptr(SI, 4, 4), vreg(t, "Z10")}, "62727d48585601"},
|
{"VPBROADCASTD 4(SI),Z10", "VPBROADCASTD", []Operand{Ptr(SI, 4, 4), vreg(t, "Z10")}, "62727d48585601"},
|
||||||
{"VPBROADCASTQ R8,X31", "VPBROADCASTQ", []Operand{vreg(t, "R8"), vreg(t, "X31")}, "6242fd087cf8"},
|
{"VPBROADCASTQ R8,X31", "VPBROADCASTQ", []Operand{vreg(t, "R8"), vreg(t, "X31")}, "6242fd087cf8"},
|
||||||
{"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"},
|
{"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"},
|
||||||
// Register indices 16–31 exist only in EVEX encodings.
|
// Register indices 16-31 exist only in EVEX encodings.
|
||||||
{"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"},
|
{"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"},
|
||||||
// Packed double arithmetic / unpack (EVEX forms carry W=1).
|
// Packed double arithmetic / unpack (EVEX forms carry W=1).
|
||||||
{"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"},
|
{"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"},
|
||||||
@@ -107,7 +126,7 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"},
|
{"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"},
|
||||||
{"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"},
|
{"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"},
|
||||||
{"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"},
|
{"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"},
|
||||||
// VMOVDDUP — duplicate the low double; disp8×N = 64 at 512 bits, and
|
// VMOVDDUP; duplicate the low double; disp8×N = 64 at 512 bits, and
|
||||||
// X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄).
|
// X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄).
|
||||||
{"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"},
|
{"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"},
|
||||||
{"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"},
|
{"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"},
|
||||||
@@ -150,7 +169,7 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestEvexMasking checks the AVX-512 mask operand (K1–K7, placed freely among
|
// TestEvexMasking checks the AVX-512 mask operand (K1-K7, placed freely among
|
||||||
// the operands) and the .Z zeroing suffix, byte for byte against the Go
|
// the operands) and the .Z zeroing suffix, byte for byte against the Go
|
||||||
// assembler.
|
// assembler.
|
||||||
func TestEvexMasking(t *testing.T) {
|
func TestEvexMasking(t *testing.T) {
|
||||||
@@ -241,11 +260,11 @@ func TestEvexMasking(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set — ternary
|
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set; ternary
|
||||||
// logic, lane shuffles/inserts/extracts, compares with a K destination,
|
// logic, lane shuffles/inserts/extracts, compares with a K destination,
|
||||||
// permutes, the wider integer families, expand/compress, broadcasts,
|
// permutes, the wider integer families, expand/compress, broadcasts,
|
||||||
// rotates and word shifts, the opmask instructions, the EVEX suffixes
|
// rotates and word shifts, the opmask instructions, the EVEX suffixes
|
||||||
// (rounding/SAE/broadcast) and the aligned/scalar moves — byte for byte
|
// (rounding/SAE/broadcast) and the aligned/scalar moves; byte for byte
|
||||||
// against the Go assembler.
|
// against the Go assembler.
|
||||||
func TestEvexExtendedGroundTruth(t *testing.T) {
|
func TestEvexExtendedGroundTruth(t *testing.T) {
|
||||||
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
|
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
|
||||||
@@ -275,7 +294,7 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
|
|||||||
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
|
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
|
||||||
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
|
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
|
||||||
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
|
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
|
||||||
// Packed single arithmetic (same opcodes, no mandatory prefix) —
|
// Packed single arithmetic (same opcodes, no mandatory prefix);
|
||||||
// ZMM, YMM and XMM widths, rounding and broadcast.
|
// ZMM, YMM and XMM widths, rounding and broadcast.
|
||||||
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
|
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
|
||||||
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
|
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
|
||||||
@@ -399,8 +418,8 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
|
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
|
||||||
// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef,
|
// tail of the EVEX set; reciprocals, rsqrt, getexp/getmant, scalef,
|
||||||
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions —
|
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions;
|
||||||
// plus gather/scatter with VSIB addressing, byte for byte against the Go
|
// plus gather/scatter with VSIB addressing, byte for byte against the Go
|
||||||
// assembler.
|
// assembler.
|
||||||
func TestEvexHelperGroundTruth(t *testing.T) {
|
func TestEvexHelperGroundTruth(t *testing.T) {
|
||||||
@@ -502,9 +521,9 @@ func TestEvexHelperGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestEvexGprGroundTruth covers the scalar conversions between vector and
|
// TestEvexGprGroundTruth covers the scalar conversions between vector and
|
||||||
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
|
// general-purpose registers; the signed and truncated VCVT{,T}S{D,S}2SI
|
||||||
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
|
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
|
||||||
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
|
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv; byte
|
||||||
// for byte against the Go assembler, including memory sources and extended
|
// for byte against the Go assembler, including memory sources and extended
|
||||||
// GPRs.
|
// GPRs.
|
||||||
func TestEvexGprGroundTruth(t *testing.T) {
|
func TestEvexGprGroundTruth(t *testing.T) {
|
||||||
@@ -675,6 +694,15 @@ func TestEvexErrors(t *testing.T) {
|
|||||||
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
|
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
|
||||||
// VEX-only mnemonics reject registers only EVEX can encode.
|
// VEX-only mnemonics reject registers only EVEX can encode.
|
||||||
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
|
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
|
||||||
|
// The scalar EVEX move matches its VEX twin and the Go assembler:
|
||||||
|
// XMM↔memory only, never reg-reg and never a wider register (the
|
||||||
|
// toolchain rejects every one of these shapes).
|
||||||
|
{"VMOVSS X1,X2", "VMOVSS", []Operand{vreg(t, "X1"), vreg(t, "X2")}},
|
||||||
|
{"VMOVSS X16,X2", "VMOVSS", []Operand{vreg(t, "X16"), vreg(t, "X2")}},
|
||||||
|
{"VMOVSS Y1,(AX)", "VMOVSS", []Operand{vreg(t, "Y1"), Ptr(AX, 0, 4)}},
|
||||||
|
{"VMOVSS Z1,Z2", "VMOVSS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}},
|
||||||
|
{"VMOVSS Z1,(AX)", "VMOVSS", []Operand{vreg(t, "Z1"), Ptr(AX, 0, 4)}},
|
||||||
|
{"VMOVSS (AX),Z2", "VMOVSS", []Operand{Ptr(AX, 0, 4), vreg(t, "Z2")}},
|
||||||
}
|
}
|
||||||
for _, c := range cases {
|
for _, c := range cases {
|
||||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||||
|
|||||||
+5
-4
@@ -68,11 +68,12 @@ const (
|
|||||||
kindSDWARFLINES = 20
|
kindSDWARFLINES = 20
|
||||||
)
|
)
|
||||||
|
|
||||||
// Symbol flags (cmd/internal/goobj).
|
// Symbol flags (cmd/internal/goobj). The linkname flag is set only for
|
||||||
|
// //go:linkname symbols (and main.main); ordinary assembly symbols carry
|
||||||
|
// none, matching cmd/asm's output.
|
||||||
const (
|
const (
|
||||||
symFlagDupok = 0x01
|
symFlagDupok = 0x01
|
||||||
symFlagNoSplit = 0x10
|
symFlagNoSplit = 0x10
|
||||||
symFlag2Link = 0x10 // asm objects flag every named symbol as linkname
|
|
||||||
symABIStatic = 0xffff
|
symABIStatic = 0xffff
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -287,7 +288,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
nps = append(nps, npSym{
|
nps = append(nps, npSym{
|
||||||
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
|
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, size: uint32(fn.Size)},
|
||||||
data: code,
|
data: code,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
@@ -317,7 +318,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
|||||||
abi = symABIStatic
|
abi = symABIStatic
|
||||||
}
|
}
|
||||||
defIdx[d.Name] = len(defs)
|
defIdx[d.Name] = len(defs)
|
||||||
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)})
|
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, size: uint32(d.Size)})
|
||||||
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
|
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
|
||||||
}
|
}
|
||||||
fnFiIdx := make([]int, len(img.Funcs))
|
fnFiIdx := make([]int, len(img.Funcs))
|
||||||
|
|||||||
+72
-32
@@ -40,8 +40,11 @@ func exportPath(importPath string) (string, error) {
|
|||||||
//
|
//
|
||||||
// refs maps package import paths to the symbol names referenced from that
|
// refs maps package import paths to the symbol names referenced from that
|
||||||
// package. The returned pkgIdx maps each import path to its position in
|
// package. The returned pkgIdx maps each import path to its position in
|
||||||
// the blkPkgIdx table (0-based), and symIdx gives each symbol's index within
|
// the blkPkgIdx table, which reserves index 0 for the dummy invalid
|
||||||
// its package.
|
// package (cmd/internal/obj/sym.go: "0 is invalid index"; the loader's
|
||||||
|
// reader loop starts at 1), so package i sits at block index i+1 and its
|
||||||
|
// relocations carry i+1. symIdx gives each symbol's index within its
|
||||||
|
// package.
|
||||||
func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symIdx map[string]int, err error) {
|
func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symIdx map[string]int, err error) {
|
||||||
pkgIdx = make(map[string]int, len(refs))
|
pkgIdx = make(map[string]int, len(refs))
|
||||||
symIdx = make(map[string]int)
|
symIdx = make(map[string]int)
|
||||||
@@ -50,7 +53,9 @@ func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symI
|
|||||||
packages := sortedPkgRefs(refs)
|
packages := sortedPkgRefs(refs)
|
||||||
|
|
||||||
for i, pkg := range packages {
|
for i, pkg := range packages {
|
||||||
pkgIdx[pkg.path] = i
|
// Block index 0 is the dummy invalid package; the first real
|
||||||
|
// package starts at 1.
|
||||||
|
pkgIdx[pkg.path] = i + 1
|
||||||
exp, err := exportPath(pkg.path)
|
exp, err := exportPath(pkg.path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, nil, err
|
return nil, nil, err
|
||||||
@@ -145,40 +150,65 @@ func parseArDecimal(b []byte) int {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// goobjFile is a parsed GOOBJ file: the string table and the symbol-definition
|
// goobjFile is a parsed GOOBJ file: the string table and the symbol-definition
|
||||||
// block.
|
// blocks. The hashed blocks are kept raw: their symbols carry no names, only
|
||||||
|
// the loader needs their counts.
|
||||||
type goobjFile struct {
|
type goobjFile struct {
|
||||||
strTab []byte // string table, at headerSize + n
|
strTab []byte // string table, at headerSize + n
|
||||||
symdef []byte // blkSymdef raw block
|
symdef []byte // blkSymdef raw block
|
||||||
npdef []byte // blkNonpkgdef raw block
|
hashed64 []byte // blkHashed64def raw block
|
||||||
|
hashed []byte // blkHasheddef raw block
|
||||||
|
npdef []byte // blkNonpkgdef raw block
|
||||||
}
|
}
|
||||||
|
|
||||||
// symbols returns all symbol names in definition order by scanning the
|
// loaderIndexBase returns the index the first nonpkgdef symbol occupies in the
|
||||||
// symdef and nonpkgdef blocks and resolving each name through the string
|
// loader's per-object symbol array. cmd/link lays the definition blocks out as
|
||||||
// table. Package definitions (blkSymdef) use fully-qualified names like
|
// symdef, hashed64def, hasheddef, nonpkgdef, nonpkgref (loader.go: preloadSyms
|
||||||
// "runtime.morestack"; non-package definitions (blkNonpkgdef) use bare
|
// fills r.syms in exactly that order, and resolve() indexes PkgIdxNone and
|
||||||
// names like "morestack". This combined list matches the index the
|
// cross-package SymIdx into it), so a symbol found in blkNonpkgdef carries the
|
||||||
// linker expects for cross-package references.
|
// three leading blocks' symbol counts as its base.
|
||||||
|
func (f *goobjFile) loaderIndexBase() int {
|
||||||
|
return len(f.symdef)/recSymSize + len(f.hashed64)/recSymSize + len(f.hashed)/recSymSize
|
||||||
|
}
|
||||||
|
|
||||||
|
// symbols returns the names of the symdef and nonpkgdef blocks in
|
||||||
|
// definition order. Package definitions (blkSymdef) use fully-qualified
|
||||||
|
// names like "runtime.morestack"; non-package definitions (blkNonpkgdef)
|
||||||
|
// use bare names like "morestack". For lookups by index prefer
|
||||||
|
// findSymbol: it adds the hashed blocks' count the loader's array
|
||||||
|
// interleaves between the two.
|
||||||
func (f *goobjFile) symbols() []string {
|
func (f *goobjFile) symbols() []string {
|
||||||
return append(f.defNames(), f.npdefNames()...)
|
return append(f.defNames(), f.npdefNames()...)
|
||||||
}
|
}
|
||||||
|
|
||||||
// findSymbol returns the index of a symbol within the combined symbol list,
|
// findSymbol returns the index of a symbol within the loader's per-object
|
||||||
// or -1 if not found. It first tries the fully-qualified name (pkg.name),
|
// symbol array, or -1 if not found. It first tries the fully-qualified
|
||||||
// then the bare name.
|
// name (pkg.name), then the bare name (assembly objects store dotless
|
||||||
|
// names, e.g. runtime's "gogo", for symbols other packages reach through
|
||||||
|
// a linkname).
|
||||||
func (f *goobjFile) findSymbol(pkg, name string) int {
|
func (f *goobjFile) findSymbol(pkg, name string) int {
|
||||||
|
base := f.loaderIndexBase()
|
||||||
qualified := pkg + "." + name
|
qualified := pkg + "." + name
|
||||||
syms := f.symbols()
|
for i, s := range f.defNames() {
|
||||||
for i, s := range syms {
|
|
||||||
if s == qualified {
|
if s == qualified {
|
||||||
return i
|
return i
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Try bare name (for non-package definitions).
|
for i, s := range f.npdefNames() {
|
||||||
for i, s := range syms {
|
if s == qualified {
|
||||||
|
return base + i
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Try bare name (for dotless assembly definitions).
|
||||||
|
for i, s := range f.defNames() {
|
||||||
if s == name {
|
if s == name {
|
||||||
return i
|
return i
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
for i, s := range f.npdefNames() {
|
||||||
|
if s == name {
|
||||||
|
return base + i
|
||||||
|
}
|
||||||
|
}
|
||||||
return -1
|
return -1
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -192,12 +222,16 @@ func (f *goobjFile) npdefNames() []string {
|
|||||||
return f.readSymNames(f.npdef)
|
return f.readSymNames(f.npdef)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// recSymSize is the size of one Sym record in the definition blocks
|
||||||
|
// (goobj.SymSize: stringRefSize + 2 + 1 + 1 + 1 + 4 + 4).
|
||||||
|
const recSymSize = 21
|
||||||
|
|
||||||
// readSymNames reads symbol names from a symdef/nonpkgdef block. Each record
|
// readSymNames reads symbol names from a symdef/nonpkgdef block. Each record
|
||||||
// is 21 bytes: nameLen (u32), nameOff (u32), abi (u16), typ, flag, flag2,
|
// is 21 bytes: nameLen (u32), nameOff (u32), abi (u16), typ, flag, flag2,
|
||||||
// size (u32), align (u32). nameOff is an absolute offset into the string
|
// size (u32), align (u32). nameOff is an absolute offset into the string
|
||||||
// table.
|
// table.
|
||||||
func (f *goobjFile) readSymNames(block []byte) []string {
|
func (f *goobjFile) readSymNames(block []byte) []string {
|
||||||
const recSize = 21
|
const recSize = recSymSize
|
||||||
if len(block) < recSize {
|
if len(block) < recSize {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
@@ -247,16 +281,18 @@ func parseGOOBJ(data []byte) (*goobjFile, error) {
|
|||||||
// [16:20] flags
|
// [16:20] flags
|
||||||
// [20:96] 19 × uint32 offsets
|
// [20:96] 19 × uint32 offsets
|
||||||
var offs [blkEnd + 1]uint32
|
var offs [blkEnd + 1]uint32
|
||||||
for i := 0; i <= blkEnd; i++ {
|
for i := range blkEnd + 1 {
|
||||||
offs[i] = binary.LittleEndian.Uint32(payload[20+4*i:])
|
offs[i] = binary.LittleEndian.Uint32(payload[20+4*i:])
|
||||||
}
|
}
|
||||||
// The string table lives at headerSize.
|
// The string table lives at headerSize.
|
||||||
strTabStart := uint32(goobjHeaderSize)
|
strTabStart := uint32(goobjHeaderSize)
|
||||||
|
|
||||||
f := &goobjFile{
|
f := &goobjFile{
|
||||||
strTab: payload[strTabStart:offs[0]],
|
strTab: payload[strTabStart:offs[0]],
|
||||||
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
|
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
|
||||||
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
|
hashed64: blockSlice(payload, offs, blkHashed64def, blkHashed64def+1),
|
||||||
|
hashed: blockSlice(payload, offs, blkHasheddef, blkHasheddef+1),
|
||||||
|
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
|
||||||
}
|
}
|
||||||
return f, nil
|
return f, nil
|
||||||
}
|
}
|
||||||
@@ -306,22 +342,26 @@ func resolveExternalSymbols(externals []string) (pkgTable []string, pkgIdxMap ma
|
|||||||
return nil, nil, nil, err
|
return nil, nil, nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
// Build the package table in pkgIdx order.
|
// Build the package table in pkgIdx order. The indices are 1-based
|
||||||
|
// (0 is the dummy invalid package, written by the emitter itself), so
|
||||||
|
// the table without the dummy is indexed one below.
|
||||||
pkgTable = make([]string, len(pkgIdx1))
|
pkgTable = make([]string, len(pkgIdx1))
|
||||||
for pkg, idx := range pkgIdx1 {
|
for pkg, idx := range pkgIdx1 {
|
||||||
pkgTable[idx] = pkg
|
pkgTable[idx-1] = pkg
|
||||||
}
|
}
|
||||||
|
|
||||||
return pkgTable, pkgIdx1, symIdx1, nil
|
return pkgTable, pkgIdx1, symIdx1, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// splitQualified splits a qualified Go symbol name (pkgpath·name) into its
|
// splitQualified splits a qualified Go symbol name (pkgpath·name) into its
|
||||||
// package path and local name. The separator is the middle dot (U+00B7).
|
// package path and local name. The separator is the middle dot (U+00B7),
|
||||||
// If no separator is found, the symbol is assumed to be in the current
|
// whose UTF-8 encoding is two bytes, so the search must be string-based:
|
||||||
// package (empty pkg).
|
// IndexByte would match only the second byte and leave the lead byte on
|
||||||
|
// the package path. If no separator is found, the symbol is assumed to be
|
||||||
|
// in the current package (empty pkg).
|
||||||
func splitQualified(full string) (pkg, name string) {
|
func splitQualified(full string) (pkg, name string) {
|
||||||
if idx := strings.IndexByte(full, '\u00b7'); idx >= 0 {
|
if before, after, ok := strings.Cut(full, "\u00b7"); ok {
|
||||||
return full[:idx], full[idx+len("\u00b7"):]
|
return before, after
|
||||||
}
|
}
|
||||||
if before, after, ok := strings.Cut(full, "."); ok {
|
if before, after, ok := strings.Cut(full, "."); ok {
|
||||||
return before, after
|
return before, after
|
||||||
|
|||||||
@@ -56,8 +56,12 @@ func TestResolveExternalSymbols(t *testing.T) {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("resolveExternalGOOBJ: %v", err)
|
t.Fatalf("resolveExternalGOOBJ: %v", err)
|
||||||
}
|
}
|
||||||
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 0 {
|
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 1 {
|
||||||
t.Errorf("pkgIdx = %v, want runtime→0", pkgIdx)
|
// Index 0 is the dummy invalid package in the blkPkgIdx table;
|
||||||
|
// the loader's reader loop starts at 1 (cmd/link/internal/
|
||||||
|
// loader/loader.go: "PkgIdx 0 is a dummy invalid package"), so
|
||||||
|
// the first real package must carry index 1.
|
||||||
|
t.Errorf("pkgIdx = %v, want runtime→1", pkgIdx)
|
||||||
}
|
}
|
||||||
if _, ok := symIdx["runtime·g0"]; !ok {
|
if _, ok := symIdx["runtime·g0"]; !ok {
|
||||||
t.Errorf("symIdx missing runtime·g0, got %v", symIdx)
|
t.Errorf("symIdx missing runtime·g0, got %v", symIdx)
|
||||||
|
|||||||
+192
-5
@@ -117,7 +117,9 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
|||||||
if len(defs) != 7 {
|
if len(defs) != 7 {
|
||||||
t.Fatalf("symdefs = %d, want 7", len(defs))
|
t.Fatalf("symdefs = %d, want 7", len(defs))
|
||||||
}
|
}
|
||||||
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
|
// The linkname flag stays clear: the toolchain sets it only for
|
||||||
|
// //go:linkname symbols, and an ordinary static GLOBL is not one.
|
||||||
|
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != 0 {
|
||||||
t.Errorf("mask symbol = %+v", defs[0])
|
t.Errorf("mask symbol = %+v", defs[0])
|
||||||
}
|
}
|
||||||
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
|
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
|
||||||
@@ -158,8 +160,8 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
|||||||
t.Errorf("funcinfo bytes %x", fi)
|
t.Errorf("funcinfo bytes %x", fi)
|
||||||
}
|
}
|
||||||
|
|
||||||
// The pc-value tables of addq (non-package indices 0–3, so global
|
// The pc-value tables of addq (non-package indices 0-3, so global
|
||||||
// indices 7–10): pcsp a flat zero over the whole function, pcinline a
|
// indices 7-10): pcsp a flat zero over the whole function, pcinline a
|
||||||
// flat -1, both with the pc delta in MinLC (1) units.
|
// flat -1, both with the pc delta in MinLC (1) units.
|
||||||
pcsp := data[le.Uint32(didx[4*7:]):]
|
pcsp := data[le.Uint32(didx[4*7:]):]
|
||||||
if got := pcsp[:3]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
|
if got := pcsp[:3]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
|
||||||
@@ -294,7 +296,7 @@ TEXT ·framed(SB), NOSPLIT, $8-0
|
|||||||
}
|
}
|
||||||
for i := range wantPCs {
|
for i := range wantPCs {
|
||||||
if pcs[i] != wantPCs[i] || vals[i] != wantVals[i] {
|
if pcs[i] != wantPCs[i] || vals[i] != wantVals[i] {
|
||||||
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d) — all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
|
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d); all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// The last two steps unwind the epilogue to zero.
|
// The last two steps unwind the epilogue to zero.
|
||||||
@@ -333,7 +335,7 @@ TEXT ·useext(SB), NOSPLIT, $0-8
|
|||||||
|
|
||||||
// TestGOObjectLinkAndRun is the end-to-end check: assemble the test
|
// TestGOObjectLinkAndRun is the end-to-end check: assemble the test
|
||||||
// functions to a GOOBJ, swap it into a go build in place of the toolchain's
|
// functions to a GOOBJ, swap it into a go build in place of the toolchain's
|
||||||
// assembly object, link, and run — the output must match the baseline
|
// assembly object, link, and run; the output must match the baseline
|
||||||
// binary the Go assembler produced. Skipped when no Go toolchain is
|
// binary the Go assembler produced. Skipped when no Go toolchain is
|
||||||
// available.
|
// available.
|
||||||
func TestGOObjectLinkAndRun(t *testing.T) {
|
func TestGOObjectLinkAndRun(t *testing.T) {
|
||||||
@@ -511,3 +513,188 @@ func fieldAfter(line, flag string) string {
|
|||||||
}
|
}
|
||||||
return ""
|
return ""
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestGOObjectExternalPackageLink is the cross-package end-to-end check: a
|
||||||
|
// GOOBJ whose code references a real external package symbol (runtime's
|
||||||
|
// morestack, a plain reference rather than the builtin noctxt form) must
|
||||||
|
// carry a package index that points past the blkPkgIdx table's dummy entry
|
||||||
|
// 0, and the object must link against the real runtime. Pre-fix, the
|
||||||
|
// relocations carried block index 0, which the loader never fills, so the
|
||||||
|
// reference resolved against whatever object was loaded first and the link
|
||||||
|
// failed. The binary is not run: morestack returns to the call site's
|
||||||
|
// stack check, which a hand-written caller has none of.
|
||||||
|
func TestGOObjectExternalPackageLink(t *testing.T) {
|
||||||
|
goBin, err := exec.LookPath("go")
|
||||||
|
if err != nil {
|
||||||
|
t.Skip("no Go toolchain available")
|
||||||
|
}
|
||||||
|
dir := t.TempDir()
|
||||||
|
|
||||||
|
const asmSrc = `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·fn(SB), NOSPLIT, $0-0
|
||||||
|
CALL ·helper(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·helper(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
const mainSrc = `package main
|
||||||
|
|
||||||
|
func fn()
|
||||||
|
func helper()
|
||||||
|
|
||||||
|
func main() {
|
||||||
|
fn()
|
||||||
|
helper()
|
||||||
|
}
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module extlink\n\ngo 1.27\n"), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Capture the build the toolchain performs and re-run only its link
|
||||||
|
// step with our object swapped into the package archive, mirroring
|
||||||
|
// TestGOObjectLinkAndRun.
|
||||||
|
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||||
|
build.Dir = dir
|
||||||
|
buildLog, err := build.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||||
|
}
|
||||||
|
var work, linkLine, asmObj, pkgArch string
|
||||||
|
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(line, "WORK="):
|
||||||
|
work = strings.TrimPrefix(line, "WORK=")
|
||||||
|
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
|
||||||
|
asmObj = fieldAfter(line, "-o")
|
||||||
|
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
|
||||||
|
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
|
||||||
|
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
|
||||||
|
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||||
|
linkLine = line
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if work == "" || asmObj == "" || pkgArch == "" || linkLine == "" {
|
||||||
|
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
|
||||||
|
}
|
||||||
|
defer os.RemoveAll(work)
|
||||||
|
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
|
||||||
|
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
|
||||||
|
|
||||||
|
// Assemble the source with gasm, then retarget fn's internal call at
|
||||||
|
// a real external package symbol: the reloc's qualified name drives
|
||||||
|
// the export-data resolution the way a source-level runtime·sym(SB)
|
||||||
|
// reference would.
|
||||||
|
f, errs := parser.Parse("main_amd64.s", asmSrc)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
fn := &img.Funcs[0]
|
||||||
|
for i := range fn.Relocs {
|
||||||
|
fn.Relocs[i].Name = "runtime\u00b7morestack"
|
||||||
|
fn.Relocs[i].External = true
|
||||||
|
}
|
||||||
|
img.Externals = []string{"runtime\u00b7morestack"}
|
||||||
|
obj, err := img.GOObject("main", "main_amd64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObject: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Structural check: the blkPkgIdx block reserves entry 0 for the
|
||||||
|
// dummy invalid package and places runtime at entry 1, and fn's call
|
||||||
|
// relocation carries PkgIdx 1.
|
||||||
|
v := openGoobj(t, obj)
|
||||||
|
pkgBlk := v.blk(blkPkgIdx)
|
||||||
|
if len(pkgBlk) != 2*8 {
|
||||||
|
t.Fatalf("blkPkgIdx = %d bytes, want two entries", len(pkgBlk))
|
||||||
|
}
|
||||||
|
le := binary.LittleEndian
|
||||||
|
strEntry := func(i int) string {
|
||||||
|
e := pkgBlk[i*8 : (i+1)*8]
|
||||||
|
return v.str(le.Uint32(e[4:]), le.Uint32(e[0:]))
|
||||||
|
}
|
||||||
|
if s := strEntry(0); s != "" {
|
||||||
|
t.Errorf("blkPkgIdx[0] = %q, want the dummy empty package", s)
|
||||||
|
}
|
||||||
|
if s := strEntry(1); s != "runtime" {
|
||||||
|
t.Errorf("blkPkgIdx[1] = %q, want runtime", s)
|
||||||
|
}
|
||||||
|
relocs := v.blk(blkReloc)
|
||||||
|
// fn is the last non-package symbol (two functions, four pc tables
|
||||||
|
// each); its one reloc is the final record.
|
||||||
|
fnRec := relocs[len(relocs)-23:]
|
||||||
|
if pIdx := le.Uint32(fnRec[15:]); pIdx != 1 {
|
||||||
|
t.Errorf("external reloc PkgIdx = %d, want 1 (runtime)", pIdx)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Swap the object into the package archive and link with cmd/link;
|
||||||
|
// the link line consumes the archive, not the loose object file.
|
||||||
|
membersDir := filepath.Join(dir, "members")
|
||||||
|
if err := os.MkdirAll(membersDir, 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
|
||||||
|
extract.Dir = membersDir
|
||||||
|
if out, err := extract.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("pack x: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
member := filepath.Join(membersDir, filepath.Base(asmObj))
|
||||||
|
if err := os.Chmod(member, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(member, obj, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
|
||||||
|
listOut, err := listCmd.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("pack t: %v\n%s", err, listOut)
|
||||||
|
}
|
||||||
|
newArch := filepath.Join(dir, "pkg.a")
|
||||||
|
args := []string{"tool", "pack", "c", newArch}
|
||||||
|
seen := map[string]bool{}
|
||||||
|
for m := range strings.FieldsSeq(string(listOut)) {
|
||||||
|
if seen[m] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
seen[m] = true
|
||||||
|
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
args = append(args, filepath.Join(membersDir, m))
|
||||||
|
}
|
||||||
|
pack := exec.Command(goBin, args...)
|
||||||
|
pack.Dir = membersDir
|
||||||
|
if out, err := pack.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("pack c: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, pkgArch, newArch)
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "prog2"))
|
||||||
|
linkCmd := exec.Command("sh", "-c", "cd "+dir+" && "+linkLine)
|
||||||
|
if out, err := linkCmd.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The call must have resolved to the real runtime symbol.
|
||||||
|
dump, err := exec.Command(goBin, "tool", "objdump", "-s", "main.fn", filepath.Join(dir, "prog2")).CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("objdump main.fn: %v\n%s", err, dump)
|
||||||
|
}
|
||||||
|
if !bytes.Contains(dump, []byte("runtime.morestack")) {
|
||||||
|
t.Errorf("main.fn does not call runtime.morestack:\n%s", dump)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+24
-3
@@ -18,19 +18,40 @@ import (
|
|||||||
// reloc/aux/data index arrays, with the arm64 preamble, the MinLC of 4
|
// reloc/aux/data index arrays, with the arm64 preamble, the MinLC of 4
|
||||||
// for the pc-value deltas, and the arm64 relocation types for the ADRP
|
// for the pc-value deltas, and the arm64 relocation types for the ADRP
|
||||||
// pairs and BL calls.
|
// pairs and BL calls.
|
||||||
|
//
|
||||||
|
// The toolchain records one relocation per ADRP pair: a single R_ADDRARM64
|
||||||
|
// or R_ARM64_PCREL_LDST64 of Siz 8 at the ADRP word, from which the linker
|
||||||
|
// patches both instructions of the pair (cmd/internal/obj/arm64/asm7.go,
|
||||||
|
// the ADRP cases: one AddRel with Off at the pair's pc and Siz 8). gasm's
|
||||||
|
// assembler records the ADRP+ADD form as two word relocs, so the second
|
||||||
|
// word's twin is dropped here before emission.
|
||||||
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
|
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
|
||||||
pre, err := toolchainObjectPreambleAARCH64()
|
pre, err := toolchainObjectPreambleAARCH64()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
|
coalesced := *img
|
||||||
|
coalesced.Funcs = append([]FuncLayout(nil), img.Funcs...)
|
||||||
|
for i := range coalesced.Funcs {
|
||||||
|
rs := coalesced.Funcs[i].Relocs
|
||||||
|
var keep []Reloc
|
||||||
|
for j := 0; j < len(rs); j++ {
|
||||||
|
keep = append(keep, rs[j])
|
||||||
|
if rs[j].Kind == RelArm64Addr && j+1 < len(rs) &&
|
||||||
|
rs[j+1].Kind == RelArm64Addr && rs[j+1].Off == rs[j].Off+4 {
|
||||||
|
j++ // the ADD word's twin: the Siz-8 pair reloc covers it
|
||||||
|
}
|
||||||
|
}
|
||||||
|
coalesced.Funcs[i].Relocs = keep
|
||||||
|
}
|
||||||
|
return coalesced.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
|
||||||
switch r.Kind {
|
switch r.Kind {
|
||||||
case RelArm64Branch:
|
case RelArm64Branch:
|
||||||
return relocArm64Branch, 4
|
return relocArm64Branch, 4
|
||||||
case RelArm64LDST64:
|
case RelArm64LDST64:
|
||||||
return relocArm64LDST64, 4
|
return relocArm64LDST64, 8
|
||||||
default:
|
default:
|
||||||
return relocArm64Addr, 4
|
return relocArm64Addr, 8
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
"encoding/hex"
|
"encoding/hex"
|
||||||
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
@@ -27,6 +28,12 @@ func TestStackGuardBytes(t *testing.T) {
|
|||||||
"644c8b3425000000004c8da42478ffffff4d3b66107614554889e54881ec000100004881c4000100005dc3e800000000ebce"},
|
"644c8b3425000000004c8da42478ffffff4d3b66107614554889e54881ec000100004881c4000100005dc3e800000000ebce"},
|
||||||
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
|
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
|
||||||
"644c8b3425000000004989e44981ec881f0000721a4d3b66107614554889e54881ec002000004881c4002000005dc3e800000000ebca"},
|
"644c8b3425000000004989e44981ec881f0000721a4d3b66107614554889e54881ec002000004881c4002000005dc3e800000000ebca"},
|
||||||
|
// Class 2 with a body long enough that the underflow JB relaxes to
|
||||||
|
// rel32: its displacement must span the real 6-byte JB, else the
|
||||||
|
// branch lands 4 bytes past the morestack block, inside the CALL
|
||||||
|
// displacement field.
|
||||||
|
{"leafbiglong", "TEXT \u00b7leafbiglong(SB), $8192-0\n" + strings.Repeat("\tMOVQ AX, BX\n", 40) + "\tRET\n",
|
||||||
|
"644c8b3425000000004989e44981ec881f00000f82960000004d3b66100f868c000000554889e54881ec00200000" + strings.Repeat("4889c3", 40) + "4881c4002000005dc3e800000000e947ffffff"},
|
||||||
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
|
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
|
||||||
"644c8b342500000000493b66107613554889e54883ec10e8000000004883c4105dc3e800000000ebd7"},
|
"644c8b342500000000493b66107613554889e54883ec10e8000000004883c4105dc3e800000000ebd7"},
|
||||||
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
|
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
|
||||||
@@ -145,6 +152,45 @@ func TestStackGuardBytesARM64(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestStackGuardBranchTargetsARM64 checks the class-2 guard's branch
|
||||||
|
// positions for a frame whose guard constant needs two MOV words: the
|
||||||
|
// displacements must be computed from byte offsets (8+4*ml and 16+4*ml), so
|
||||||
|
// both branches land on the morestack block rather than inside the body.
|
||||||
|
// The frame size makes the toolchain switch its own prologue decomposition,
|
||||||
|
// so the assertion is on the branch targets, not pinned bytes.
|
||||||
|
func TestStackGuardBranchTargetsARM64(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("g_arm64.s", "TEXT \u00b7f(SB), $65664-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
fn := img.Funcs[0]
|
||||||
|
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||||
|
if len(code)%4 != 0 {
|
||||||
|
t.Fatalf("function size %d is not a word multiple", len(code))
|
||||||
|
}
|
||||||
|
// autosize = 65680, so the guard materialises 65552 = MOVZ+MOVK: ml = 2
|
||||||
|
// and the branches sit at bytes 16 and 24 of the guard prefix.
|
||||||
|
const morestackBlock = 12 // MOVD R30, R3; BL; B back
|
||||||
|
blockStart := len(code) - morestackBlock
|
||||||
|
check := func(name string, off int) {
|
||||||
|
t.Helper()
|
||||||
|
w := leWord(code[off:])
|
||||||
|
imm19 := int32(w>>5) & 0x7FFFF
|
||||||
|
if imm19&(1<<18) != 0 {
|
||||||
|
imm19 -= 1 << 19
|
||||||
|
}
|
||||||
|
if target := off + int(imm19)*4; target != blockStart {
|
||||||
|
t.Errorf("%s at byte %d targets byte %d, want the morestack block at %d", name, off, target, blockStart)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
check("B.LO", 16)
|
||||||
|
check("B.LS", 24)
|
||||||
|
}
|
||||||
|
|
||||||
// The riscv64 stack-split guard, pinned from `go tool asm` (Go 1.27,
|
// The riscv64 stack-split guard, pinned from `go tool asm` (Go 1.27,
|
||||||
// riscv64): the morestack call sits between the guard and the body, and the
|
// riscv64): the morestack call sits between the guard and the body, and the
|
||||||
// guard branches forward over it. Relocation fields are masked.
|
// guard branches forward over it. Relocation fields are masked.
|
||||||
|
|||||||
+774
-43
@@ -14,30 +14,70 @@ var aluOp = map[string]struct {
|
|||||||
}{
|
}{
|
||||||
"ADD": {0x01, 0},
|
"ADD": {0x01, 0},
|
||||||
"OR": {0x09, 1},
|
"OR": {0x09, 1},
|
||||||
|
"ADC": {0x11, 2},
|
||||||
|
"SBB": {0x19, 3},
|
||||||
"AND": {0x21, 4},
|
"AND": {0x21, 4},
|
||||||
"SUB": {0x29, 5},
|
"SUB": {0x29, 5},
|
||||||
"XOR": {0x31, 6},
|
"XOR": {0x31, 6},
|
||||||
"CMP": {0x39, 7},
|
"CMP": {0x39, 7},
|
||||||
}
|
}
|
||||||
|
|
||||||
// unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use
|
// unaryOp maps INC/DEC/NEG/NOT/MUL/DIV/IDIV to their /digit and base opcode.
|
||||||
// the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes in 64-bit
|
// INC/DEC use the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes
|
||||||
// mode); NEG/NOT use the 0xF6/0xF7 group.
|
// in 64-bit mode); NEG/NOT/MUL/DIV/IDIV use the 0xF6/0xF7 group (MUL /4,
|
||||||
|
// DIV /6, IDIV /7; the accumulator is the implicit other operand).
|
||||||
var unaryOp = map[string]struct {
|
var unaryOp = map[string]struct {
|
||||||
digit int
|
digit int
|
||||||
op byte
|
op byte
|
||||||
}{
|
}{
|
||||||
"INC": {0, 0xFF},
|
"INC": {0, 0xFF},
|
||||||
"DEC": {1, 0xFF},
|
"DEC": {1, 0xFF},
|
||||||
"NOT": {2, 0xF7},
|
"NOT": {2, 0xF7},
|
||||||
"NEG": {3, 0xF7},
|
"NEG": {3, 0xF7},
|
||||||
|
"MUL": {4, 0xF7},
|
||||||
|
"DIV": {6, 0xF7},
|
||||||
|
"IDIV": {7, 0xF7},
|
||||||
}
|
}
|
||||||
|
|
||||||
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0-0xD3 group.
|
// shiftOp maps SHL/SAL/SHR/SAR/ROL/ROR/RCL/RCR to their /digit in the
|
||||||
|
// 0xC0/0xC1/0xD0-0xD3 group. SAL is the same encoding as SHL (/4).
|
||||||
var shiftOp = map[string]int{
|
var shiftOp = map[string]int{
|
||||||
"SHL": 4,
|
"SHL": 4,
|
||||||
|
"SAL": 4,
|
||||||
"SHR": 5,
|
"SHR": 5,
|
||||||
"SAR": 7,
|
"SAR": 7,
|
||||||
|
"ROL": 0,
|
||||||
|
"ROR": 1,
|
||||||
|
"RCL": 2,
|
||||||
|
"RCR": 3,
|
||||||
|
}
|
||||||
|
|
||||||
|
// bitTestOp maps BT/BTS/BTR/BTC to their /digit in the 0F BA immediate form;
|
||||||
|
// the register form is 0F A3/AB/B3/BB, the same digit in the low nibble's
|
||||||
|
// opcode row.
|
||||||
|
var bitTestOp = map[string]int{
|
||||||
|
"BT": 4,
|
||||||
|
"BTS": 5,
|
||||||
|
"BTR": 6,
|
||||||
|
"BTC": 7,
|
||||||
|
}
|
||||||
|
|
||||||
|
// noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The
|
||||||
|
// fence names carry their opcode inside the 0F AE /digit group spelled out in
|
||||||
|
// full (E8/F0/F8), and PAUSE is F3 90.
|
||||||
|
var noOperandTable = map[string][]byte{
|
||||||
|
"CPUID": {0x0F, 0xA2},
|
||||||
|
"RDTSC": {0x0F, 0x31},
|
||||||
|
"RDTSCP": {0x0F, 0x01, 0xF9},
|
||||||
|
"SYSCALL": {0x0F, 0x05},
|
||||||
|
"XGETBV": {0x0F, 0x01, 0xD0},
|
||||||
|
"CLD": {0xFC},
|
||||||
|
"STD": {0xFD},
|
||||||
|
"PAUSE": {0xF3, 0x90},
|
||||||
|
"LFENCE": {0x0F, 0xAE, 0xE8},
|
||||||
|
"MFENCE": {0x0F, 0xAE, 0xF0},
|
||||||
|
"SFENCE": {0x0F, 0xAE, 0xF8},
|
||||||
|
"UNDEF": {0x0F, 0x0B},
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- MOV --------------------------------------------------------------------
|
// --- MOV --------------------------------------------------------------------
|
||||||
@@ -173,7 +213,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
|||||||
if dstReg.needsREX(size) {
|
if dstReg.needsREX(size) {
|
||||||
i.rexForced = true
|
i.rexForced = true
|
||||||
}
|
}
|
||||||
i.imm = immediate(v, size, true)
|
imm, err := immediate(v, size, true)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = imm
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
|
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
|
||||||
@@ -185,7 +229,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
|||||||
if err := setRMDigit(i, 0, dst, size); err != nil {
|
if err := setRMDigit(i, 0, dst, size); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
i.imm = immediate(int64(src), size, false)
|
imm, err := immediate(int64(src), size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = imm
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("MOV: invalid operands")
|
return fmt.Errorf("MOV: invalid operands")
|
||||||
@@ -297,11 +345,22 @@ func (e *enc) encodeALU(op struct {
|
|||||||
|
|
||||||
func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
||||||
if size == 1 {
|
if size == 1 {
|
||||||
|
immBytes, err := immediate(imm, 1, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
// The byte accumulator short form (0x04+digit*8, no ModR/M) when
|
||||||
|
// the destination is AL, the form the Go assembler prefers here.
|
||||||
|
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
||||||
|
i := &instr{opcode: []byte{byte(0x04 + digit*8)}, modrm: -1, sib: -1}
|
||||||
|
i.imm = immBytes
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
i := newInstr(1, []byte{0x80})
|
i := newInstr(1, []byte{0x80})
|
||||||
if err := setRMDigit(i, digit, dst, 1); err != nil {
|
if err := setRMDigit(i, digit, dst, 1); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
i.imm = []byte{byte(int8(imm))}
|
i.imm = immBytes
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
if fits8(imm) {
|
if fits8(imm) {
|
||||||
@@ -319,7 +378,11 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
|||||||
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
||||||
accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit]
|
accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit]
|
||||||
i := newInstr(size, []byte{accOp})
|
i := newInstr(size, []byte{accOp})
|
||||||
i.imm = immediate(imm, size, false)
|
immBytes, err := immediate(imm, size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = immBytes
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
// 0x81 /digit, imm16/imm32.
|
// 0x81 /digit, imm16/imm32.
|
||||||
@@ -327,7 +390,11 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
|||||||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
i.imm = immediate(imm, size, false)
|
immBytes, err := immediate(imm, size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = immBytes
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -348,7 +415,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
|
|||||||
op = 0xA8
|
op = 0xA8
|
||||||
}
|
}
|
||||||
i := newInstr(size, []byte{op})
|
i := newInstr(size, []byte{op})
|
||||||
i.imm = immediate(int64(imm), size, false)
|
immBytes, err := immediate(int64(imm), size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = immBytes
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
op := byte(0xF7)
|
op := byte(0xF7)
|
||||||
@@ -359,7 +430,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
|
|||||||
if err := setRMDigit(i, 0, dst, size); err != nil {
|
if err := setRMDigit(i, 0, dst, size); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
i.imm = immediate(int64(imm), size, false)
|
immBytes, err := immediate(int64(imm), size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = immBytes
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
srcReg, ok := src.(Reg)
|
srcReg, ok := src.(Reg)
|
||||||
@@ -457,7 +532,13 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
|
|||||||
}
|
}
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
// 0xC0 (8-bit) / 0xC1, imm8.
|
// 0xC0 (8-bit) / 0xC1, imm8. The count is an unsigned byte: go tool asm
|
||||||
|
// rejects negative and ≥256 counts, and the hardware masks the count, so
|
||||||
|
// a silent truncation ($300 encoding 44) would shift by a different
|
||||||
|
// amount than the source states.
|
||||||
|
if imm < 0 || imm > 255 {
|
||||||
|
return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm))
|
||||||
|
}
|
||||||
op := byte(0xC1)
|
op := byte(0xC1)
|
||||||
if size == 1 {
|
if size == 1 {
|
||||||
op = 0xC0
|
op = 0xC0
|
||||||
@@ -466,7 +547,7 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
|
|||||||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
i.imm = []byte{byte(int8(imm))}
|
i.imm = []byte{byte(imm)}
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -508,7 +589,11 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
|
|||||||
if err := setRM(i, dstReg, ops[1], size); err != nil {
|
if err := setRM(i, dstReg, ops[1], size); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
i.imm = immediate(int64(imm), size, false)
|
immBytes, err := immediate(int64(imm), size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = immBytes
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
|
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
|
||||||
@@ -516,10 +601,21 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
|
|||||||
|
|
||||||
// --- PUSH / POP -------------------------------------------------------------
|
// --- PUSH / POP -------------------------------------------------------------
|
||||||
|
|
||||||
func (e *enc) encodePushPop(ops []Operand, push bool) error {
|
func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
|
||||||
if len(ops) != 1 {
|
if len(ops) != 1 {
|
||||||
return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops))
|
return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops))
|
||||||
}
|
}
|
||||||
|
// In 64-bit mode go tool asm knows the 64-bit push (the default, with or
|
||||||
|
// without the Q suffix) and the 16-bit W form with its 0x66 operand-size
|
||||||
|
// prefix, and rejects the B and L spellings outright ("illegal in 64-bit
|
||||||
|
// mode"); silently widening those would push a different width than the
|
||||||
|
// source states.
|
||||||
|
switch size {
|
||||||
|
case 0, 8, 2:
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("PUSH/POP size suffix is illegal in 64-bit mode")
|
||||||
|
}
|
||||||
|
w16 := size == 2
|
||||||
switch op := ops[0].(type) {
|
switch op := ops[0].(type) {
|
||||||
case Reg:
|
case Reg:
|
||||||
base := byte(0x50) // PUSH r; POP is 0x58
|
base := byte(0x50) // PUSH r; POP is 0x58
|
||||||
@@ -527,7 +623,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
|
|||||||
base = 0x58
|
base = 0x58
|
||||||
}
|
}
|
||||||
// PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed.
|
// PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed.
|
||||||
i := &instr{opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
|
i := &instr{opSize16: w16, opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
|
||||||
i.rexB = op.idx >= 8
|
i.rexB = op.idx >= 8
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
case Mem:
|
case Mem:
|
||||||
@@ -537,7 +633,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
|
|||||||
opc = 0x8F // POP r/m: /0
|
opc = 0x8F // POP r/m: /0
|
||||||
digit = 0
|
digit = 0
|
||||||
}
|
}
|
||||||
i := &instr{opcode: []byte{opc}, modrm: -1, sib: -1}
|
i := &instr{opSize16: w16, opcode: []byte{opc}, modrm: -1, sib: -1}
|
||||||
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
|
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
@@ -547,10 +643,17 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
|
|||||||
return fmt.Errorf("POP does not take an immediate")
|
return fmt.Errorf("POP does not take an immediate")
|
||||||
}
|
}
|
||||||
if fits8(int64(op)) {
|
if fits8(int64(op)) {
|
||||||
i := &instr{opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
|
i := &instr{opSize16: w16, opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
i := &instr{opSize16: false, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: le32(int64(op))}
|
// PUSH imm32, sign-extended to 64 bits; go tool asm bounds the
|
||||||
|
// immediate by the same signed/unsigned 32-bit span as every other
|
||||||
|
// scalar immediate.
|
||||||
|
immBytes, err := immediate(int64(op), 8, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i := &instr{opSize16: w16, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: immBytes}
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("PUSH/POP: invalid operand")
|
return fmt.Errorf("PUSH/POP: invalid operand")
|
||||||
@@ -575,6 +678,24 @@ func (e *enc) encodeJmpRel(ops []Operand, opcode []byte) error {
|
|||||||
return e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, imm: le32(int64(imm))})
|
return e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, imm: le32(int64(imm))})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// encodeIndirectBranch encodes JMP/CALL through a register or memory operand:
|
||||||
|
// FF /4 for JMP, FF /2 for CALL. The operand size is fixed at 64 bits in
|
||||||
|
// 64-bit mode, so no REX.W is emitted; a REX appears only for R8-R15 bases.
|
||||||
|
func (e *enc) encodeIndirectBranch(mnem string, ops []Operand) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
|
||||||
|
}
|
||||||
|
digit := 4 // JMP r/m64
|
||||||
|
if mnem == "CALL" {
|
||||||
|
digit = 2 // CALL r/m64
|
||||||
|
}
|
||||||
|
i := &instr{opcode: []byte{0xFF}, modrm: -1, sib: -1}
|
||||||
|
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
// condCode maps a Plan 9 conditional-jump mnemonic to its x86 condition code.
|
// condCode maps a Plan 9 conditional-jump mnemonic to its x86 condition code.
|
||||||
func condCode(upper string) (int, bool) {
|
func condCode(upper string) (int, bool) {
|
||||||
if len(upper) < 2 || upper[0] != 'J' || upper == "JMP" {
|
if len(upper) < 2 || upper[0] != 'J' || upper == "JMP" {
|
||||||
@@ -621,19 +742,28 @@ func (e *enc) encodeJcc(cc int, ops []Operand) error {
|
|||||||
// immediate encodes an immediate of the given operand size. full64 selects the
|
// immediate encodes an immediate of the given operand size. full64 selects the
|
||||||
// 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit
|
// 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit
|
||||||
// sign-extended immediate is used for 64-bit operands.
|
// sign-extended immediate is used for 64-bit operands.
|
||||||
func immediate(v int64, size int, full64 bool) []byte {
|
//
|
||||||
|
// The span mirrors go tool asm: every scalar immediate must fit a signed or
|
||||||
|
// unsigned 32-bit word, and the narrower fields then take the low bits
|
||||||
|
// silently (ADDB $256, AL encodes imm8 0, MOVW $65536, AX imm16 0). Only the
|
||||||
|
// imm64 form may exceed the span; anything wider elsewhere is an error rather
|
||||||
|
// than a truncation the source never asked for.
|
||||||
|
func immediate(v int64, size int, full64 bool) ([]byte, error) {
|
||||||
|
if !(size == 8 && full64) && (v < -(1<<31) || v > (1<<32)-1) {
|
||||||
|
return nil, fmt.Errorf("immediate $%d does not fit in 32 bits", v)
|
||||||
|
}
|
||||||
switch size {
|
switch size {
|
||||||
case 1:
|
case 1:
|
||||||
return []byte{byte(int8(v))}
|
return []byte{byte(int8(v))}, nil
|
||||||
case 2:
|
case 2:
|
||||||
return le16(v)
|
return le16(v), nil
|
||||||
case 4:
|
case 4:
|
||||||
return le32(v)
|
return le32(v), nil
|
||||||
default: // 8
|
default: // 8
|
||||||
if full64 {
|
if full64 {
|
||||||
return le64(v)
|
return le64(v), nil
|
||||||
}
|
}
|
||||||
return le32(v) // sign-extended imm32
|
return le32(v), nil // sign-extended imm32
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -755,15 +885,24 @@ func (e *enc) encodeBswap(ops []Operand, size int) error {
|
|||||||
// width. The source is narrower than the destination, so the plain size-suffix
|
// width. The source is narrower than the destination, so the plain size-suffix
|
||||||
// convention does not apply to these names.
|
// convention does not apply to these names.
|
||||||
var movExtendOp = map[string]struct {
|
var movExtendOp = map[string]struct {
|
||||||
op []byte
|
op []byte
|
||||||
dst64 bool
|
dstSize int
|
||||||
}{
|
}{
|
||||||
"MOVBLZX": {[]byte{0x0F, 0xB6}, false}, // byte → long, zero-extend
|
"MOVBLZX": {[]byte{0x0F, 0xB6}, 4}, // byte → long, zero-extend
|
||||||
"MOVBQZX": {[]byte{0x0F, 0xB6}, true}, // byte → quad, zero-extend
|
"MOVBQZX": {[]byte{0x0F, 0xB6}, 8}, // byte → quad, zero-extend
|
||||||
"MOVWLZX": {[]byte{0x0F, 0xB7}, false}, // word → long, zero-extend
|
"MOVWLZX": {[]byte{0x0F, 0xB7}, 4}, // word → long, zero-extend
|
||||||
"MOVWQZX": {[]byte{0x0F, 0xB7}, true}, // word → quad, zero-extend
|
"MOVWQZX": {[]byte{0x0F, 0xB7}, 8}, // word → quad, zero-extend
|
||||||
"MOVWLSX": {[]byte{0x0F, 0xBF}, false}, // word → long, sign-extend
|
"MOVWLSX": {[]byte{0x0F, 0xBF}, 4}, // word → long, sign-extend
|
||||||
"MOVLQSX": {[]byte{0x63}, true}, // long → quad, sign-extend (MOVSXD)
|
"MOVLQSX": {[]byte{0x63}, 8}, // long → quad, sign-extend (MOVSXD)
|
||||||
|
"MOVBWZX": {[]byte{0x0F, 0xB6}, 2}, // byte → word, zero-extend
|
||||||
|
"MOVBWSX": {[]byte{0x0F, 0xBE}, 2}, // byte → word, sign-extend
|
||||||
|
"MOVBLSX": {[]byte{0x0F, 0xBE}, 4}, // byte → long, sign-extend
|
||||||
|
"MOVBQSX": {[]byte{0x0F, 0xBE}, 8}, // byte → quad, sign-extend
|
||||||
|
"MOVWQSX": {[]byte{0x0F, 0xBF}, 8}, // word → quad, sign-extend
|
||||||
|
// A long → quad zero-extend is a plain 32-bit move: every 32-bit
|
||||||
|
// operation zero-extends its result into the full register, so the
|
||||||
|
// toolchain lowers MOVLQZX to the plain MOVL encoding.
|
||||||
|
"MOVLQZX": {[]byte{0x8B}, 4},
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeMovExtend encodes a mixed-width extending move: reg = dst (the wider
|
// encodeMovExtend encodes a mixed-width extending move: reg = dst (the wider
|
||||||
@@ -777,12 +916,30 @@ func (e *enc) encodeMovExtend(base string, ops []Operand) error {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return fmt.Errorf("%s destination must be a register", base)
|
return fmt.Errorf("%s destination must be a register", base)
|
||||||
}
|
}
|
||||||
size := 4
|
i := newInstr(spec.dstSize, spec.op)
|
||||||
if spec.dst64 {
|
if err := setRM(i, dstReg, ops[0], spec.dstSize); err != nil {
|
||||||
size = 8
|
return err
|
||||||
}
|
}
|
||||||
i := newInstr(size, spec.op)
|
return e.emit(i)
|
||||||
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
}
|
||||||
|
|
||||||
|
// encodePmovmskb encodes PMOVMSKB, the legacy SSE2 byte mask extract: the
|
||||||
|
// XMM source's sign bytes pack into a GP destination, 66 0F D7 /r.
|
||||||
|
func (e *enc) encodePmovmskb(base string, ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||||||
|
}
|
||||||
|
srcReg, srcVec := vecReg(ops[0])
|
||||||
|
if !srcVec {
|
||||||
|
return fmt.Errorf("%s source must be an XMM register", base)
|
||||||
|
}
|
||||||
|
dstReg, ok := ops[1].(Reg)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s destination must be a register", base)
|
||||||
|
}
|
||||||
|
i := newInstr(4, []byte{0x0F, 0xD7})
|
||||||
|
i.prefix = 0x66
|
||||||
|
if err := setRM(i, dstReg, srcReg, 4); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
@@ -803,6 +960,7 @@ type sseMove struct {
|
|||||||
var sseMoveTable = map[string]sseMove{
|
var sseMoveTable = map[string]sseMove{
|
||||||
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
|
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
|
||||||
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
|
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
|
||||||
|
"MOVOA": {0x66, 0x6F, 0x7F}, // MOVDQA, the aligned octa alias
|
||||||
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
|
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
|
||||||
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
|
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
|
||||||
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
|
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
|
||||||
@@ -890,10 +1048,120 @@ var sseBinTable = map[string]sseBin{
|
|||||||
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
|
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
|
||||||
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
|
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
|
||||||
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
|
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
|
||||||
"PCMPEQD": {0x66, 0x76, false},
|
"PCMPEQD": {0x66, 0x76, false}, "PCMPEQL": {0x66, 0x76, false},
|
||||||
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
|
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
|
||||||
"PCMPGTD": {0x66, 0x66, false},
|
"PCMPGTD": {0x66, 0x66, false},
|
||||||
"PSHUFB": {0x66, 0x00, true},
|
"PSHUFB": {0x66, 0x00, true},
|
||||||
|
// Scalar compares and square root, packed adds/subtracts and the byte
|
||||||
|
// unpack, the spellings the Plan 9 table uses (COMISD orders the
|
||||||
|
// operands like every other two-operand form).
|
||||||
|
"ANDNPD": {0x66, 0x55, false},
|
||||||
|
"ANDNPS": {0x00, 0x55, false},
|
||||||
|
"COMISD": {0x66, 0x2F, false},
|
||||||
|
"SQRTSD": {0xF2, 0x51, false},
|
||||||
|
"PADDL": {0x66, 0xFE, false},
|
||||||
|
"PSUBL": {0x66, 0xFA, false},
|
||||||
|
"PUNPCKLBW": {0x66, 0x60, false},
|
||||||
|
// AES round functions (66 0F38) and the SHA message schedule helpers
|
||||||
|
// (no prefix, 0F38).
|
||||||
|
"AESENC": {0x66, 0xDC, true},
|
||||||
|
"AESENCLAST": {0x66, 0xDD, true},
|
||||||
|
"AESDEC": {0x66, 0xDE, true},
|
||||||
|
"AESDECLAST": {0x66, 0xDF, true},
|
||||||
|
"AESIMC": {0x66, 0xDB, true},
|
||||||
|
"SHA1MSG1": {0x00, 0xC9, true},
|
||||||
|
"SHA1MSG2": {0x00, 0xCA, true},
|
||||||
|
"SHA1NEXTE": {0x00, 0xC8, true},
|
||||||
|
"SHA256MSG1": {0x00, 0xCC, true},
|
||||||
|
"SHA256MSG2": {0x00, 0xCD, true},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseImm3 describes a legacy SSE instruction taking a leading imm8 and two
|
||||||
|
// further operands: OP $imm, src, dst with reg = dst, rm = src. map38 and
|
||||||
|
// map3A select the opcode map the same way as sseBin's.
|
||||||
|
type sseImm3 struct {
|
||||||
|
prefix byte
|
||||||
|
op byte
|
||||||
|
map3A bool // opcode lives under 0F3A instead of 0F38
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseImm3Table covers the imm8-controlled legacy instructions: the SSSE3
|
||||||
|
// align/blend shuffles, the string compare, carry-less multiply and the AES
|
||||||
|
// key assistant. SHA1RNDS4 carries no prefix, unlike its 0F3A siblings.
|
||||||
|
var sseImm3Table = map[string]sseImm3{
|
||||||
|
"PALIGNR": {0x66, 0x0F, true},
|
||||||
|
"PBLENDW": {0x66, 0x0E, true},
|
||||||
|
"PCMPESTRI": {0x66, 0x61, true},
|
||||||
|
"PCLMULQDQ": {0x66, 0x44, true},
|
||||||
|
"AESKEYGENASSIST": {0x66, 0xDF, true},
|
||||||
|
"SHA1RNDS4": {0x00, 0xCC, true},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseExtract describes a lane extract: OP $imm, xsrc, dst with reg = the XMM
|
||||||
|
// source and rm = the destination (GPR or memory). PEXTRW's GPR destination
|
||||||
|
// uses the older 0F C5 form; its memory destination the SSE4.1 0F3A 15 one,
|
||||||
|
// so it carries both opcodes.
|
||||||
|
type sseExtract struct {
|
||||||
|
op []byte
|
||||||
|
opMem []byte // used when the destination is memory; nil shares op
|
||||||
|
rexW bool // PEXTRQ's REX.W
|
||||||
|
}
|
||||||
|
|
||||||
|
var sseExtractTable = map[string]sseExtract{
|
||||||
|
"PEXTRB": {[]byte{0x0F, 0x3A, 0x14}, nil, false},
|
||||||
|
"PEXTRD": {[]byte{0x0F, 0x3A, 0x16}, nil, false},
|
||||||
|
"PEXTRQ": {[]byte{0x0F, 0x3A, 0x16}, nil, true},
|
||||||
|
"PEXTRW": {[]byte{0x0F, 0xC5}, []byte{0x0F, 0x3A, 0x15}, false},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseInsert describes a lane insert: OP $imm, src, xdst with reg = the XMM
|
||||||
|
// destination and rm = the source (GPR or memory).
|
||||||
|
type sseInsert struct {
|
||||||
|
op []byte
|
||||||
|
rexW bool // PINSRQ's REX.W
|
||||||
|
}
|
||||||
|
|
||||||
|
var sseInsertTable = map[string]sseInsert{
|
||||||
|
"PINSRB": {[]byte{0x0F, 0x3A, 0x20}, false},
|
||||||
|
"PINSRD": {[]byte{0x0F, 0x3A, 0x22}, false},
|
||||||
|
"PINSRQ": {[]byte{0x0F, 0x3A, 0x22}, true},
|
||||||
|
"PINSRW": {[]byte{0x0F, 0xC4}, false},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseShiftImm maps the legacy packed integer shifts' immediate form:
|
||||||
|
// OP $imm, dst (66 0F 71/72/73 /digit). The Plan 9 dword spellings end in L
|
||||||
|
// (PSLLL/PSRAL/PSRLL) and the octa byte shifts are PSLLDQ/PSRLDQ.
|
||||||
|
var sseShiftImm = map[string]sseShift{
|
||||||
|
"PSLLW": {0x71, 6},
|
||||||
|
"PSRLW": {0x71, 2},
|
||||||
|
"PSRAW": {0x71, 4},
|
||||||
|
"PSLLL": {0x72, 6},
|
||||||
|
"PSRLL": {0x72, 2},
|
||||||
|
"PSRAL": {0x72, 4},
|
||||||
|
"PSLLQ": {0x73, 6},
|
||||||
|
"PSRLQ": {0x73, 2},
|
||||||
|
"PSLLDQ": {0x73, 7},
|
||||||
|
"PSRLDQ": {0x73, 3},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseShiftVar maps the variable-count forms (the count comes from an XMM
|
||||||
|
// register or memory): OP count, dst (66 0F D1-F3). PSLLDQ/PSRLDQ have no
|
||||||
|
// variable form.
|
||||||
|
var sseShiftVar = map[string]byte{
|
||||||
|
"PSLLW": 0xF1,
|
||||||
|
"PSRLW": 0xD1,
|
||||||
|
"PSRAW": 0xE1,
|
||||||
|
"PSLLL": 0xF2,
|
||||||
|
"PSRLL": 0xD2,
|
||||||
|
"PSRAL": 0xE2,
|
||||||
|
"PSLLQ": 0xF3,
|
||||||
|
"PSRLQ": 0xD3,
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseShift is one /digit selector in the 0F 71/72/73 immediate group.
|
||||||
|
type sseShift struct {
|
||||||
|
op byte
|
||||||
|
digit int
|
||||||
}
|
}
|
||||||
|
|
||||||
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
|
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
|
||||||
@@ -906,6 +1174,7 @@ type sseShuf struct {
|
|||||||
var sseShufTable = map[string]sseShuf{
|
var sseShufTable = map[string]sseShuf{
|
||||||
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
|
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
|
||||||
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
|
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
|
||||||
|
"PSHUFL": {0x66, 0x70},
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
|
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
|
||||||
@@ -980,3 +1249,465 @@ func (e *enc) encodeCvtsi2sd(quad bool, ops []Operand) error {
|
|||||||
}
|
}
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- carry, bit test, exchange and accumulate -------------------------------
|
||||||
|
|
||||||
|
// encodeBitTest encodes BT/BTS/BTR/BTC. The bit index goes first in Plan 9
|
||||||
|
// order (BTQ AX, BX tests BX at the offset in AX, encoding 0F A3 with
|
||||||
|
// reg = index, rm = target); an immediate index uses 0F BA /digit with imm8.
|
||||||
|
func (e *enc) encodeBitTest(name string, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||||||
|
}
|
||||||
|
digit := bitTestOp[name]
|
||||||
|
index, target := ops[0], ops[1]
|
||||||
|
if reg, ok := index.(Reg); ok {
|
||||||
|
// Register index: 0F A3 (BT) / 0F AB (BTS) / 0F B3 (BTR) / 0F BB (BTC),
|
||||||
|
// the /digit base plus eight per step.
|
||||||
|
i := newInstr(size, []byte{0x0F, 0xA3 + byte(digit-4)<<3})
|
||||||
|
if err := setRM(i, reg, target, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
imm, ok := index.(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s index must be a register or an immediate", name)
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x0F, 0xBA})
|
||||||
|
if err := setRMDigit(i, digit, target, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeExchange encodes XCHG. A register-to-register exchange where either
|
||||||
|
// operand is AX uses the 0x90+r accumulator form (with REX.W for the quad
|
||||||
|
// form, as the Go assembler emits it); everything else uses 0x86/0x87 with
|
||||||
|
// the register operand in ModRM.reg, the memory (or second register) in r/m.
|
||||||
|
func (e *enc) encodeExchange(ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("XCHG expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
src, dst := ops[0], ops[1]
|
||||||
|
srcReg, srcIsReg := src.(Reg)
|
||||||
|
dstReg, dstIsReg := dst.(Reg)
|
||||||
|
if srcIsReg && dstIsReg && size > 1 && (srcReg.idx == 0 || dstReg.idx == 0) {
|
||||||
|
// 0x90+r: r is the non-AX register, whichever side it sits on.
|
||||||
|
r := dstReg
|
||||||
|
if srcReg.idx == 0 {
|
||||||
|
r = dstReg
|
||||||
|
} else {
|
||||||
|
r = srcReg
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x90 + byte(r.idx&7)})
|
||||||
|
i.rexB = r.idx >= 8
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
op := byte(0x87)
|
||||||
|
if size == 1 {
|
||||||
|
op = 0x86
|
||||||
|
}
|
||||||
|
switch {
|
||||||
|
case srcIsReg:
|
||||||
|
i := newInstr(size, []byte{op})
|
||||||
|
if err := setRM(i, srcReg, dst, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
case dstIsReg:
|
||||||
|
i := newInstr(size, []byte{op})
|
||||||
|
if err := setRM(i, dstReg, src, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
return fmt.Errorf("XCHG: at least one operand must be a register")
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeRegRegOp encodes the two-operand read-modify-write pair CMPXCHG
|
||||||
|
// (0F B0/B1) and XADD (0F C0/C1): reg = source, rm = destination, with the
|
||||||
|
// destination writable (register or memory).
|
||||||
|
func (e *enc) encodeRegRegOp(op8, op byte, name string, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||||||
|
}
|
||||||
|
srcReg, ok := ops[0].(Reg)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s source must be a register", name)
|
||||||
|
}
|
||||||
|
opc := op
|
||||||
|
if size == 1 {
|
||||||
|
opc = op8
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x0F, opc})
|
||||||
|
if err := setRM(i, srcReg, ops[1], size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeCrc32 encodes the CRC32 family: F2 0F38 F0 for the byte form, F1 for
|
||||||
|
// the rest; the word form carries a 0x66 operand-size prefix (66 F2, the
|
||||||
|
// prefix order the Go assembler emits) and the quad form REX.W. reg = GPR
|
||||||
|
// accumulator, rm = the data source.
|
||||||
|
func (e *enc) encodeCrc32(ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("CRC32 expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
dstReg, ok := ops[1].(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("CRC32 destination must be a general register")
|
||||||
|
}
|
||||||
|
i := &instr{opSize16: size == 2, prefix: 0xF2, opcode: []byte{0x0F, 0x38, 0xF0}, modrm: -1, sib: -1}
|
||||||
|
if size > 1 {
|
||||||
|
i.opcode[2] = 0xF1
|
||||||
|
}
|
||||||
|
i.rexW = size == 8
|
||||||
|
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeCarryExt encodes ADCX (66 0F38 F6) and ADOX (F3 0F38 F6): reg =
|
||||||
|
// destination, rm = source, the carry/overflow flag as the carry-in.
|
||||||
|
func (e *enc) encodeCarryExt(prefix byte, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("ADCX/ADOX expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
dstReg, ok := ops[1].(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("ADCX/ADOX destination must be a general register")
|
||||||
|
}
|
||||||
|
i := &instr{prefix: prefix, opcode: []byte{0x0F, 0x38, 0xF6}, modrm: -1, sib: -1, rexW: size == 8}
|
||||||
|
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- string primitives, flags and INT ----------------------------------------
|
||||||
|
|
||||||
|
// encodeStringOp encodes the no-operand string primitives MOVS (A4/A5) and
|
||||||
|
// STOS (AA/AB); the size suffix picks the byte form and supplies the 0x66 or
|
||||||
|
// REX.W prefix.
|
||||||
|
func (e *enc) encodeStringOp(base string, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("%s takes no operands, got %d", base, len(ops))
|
||||||
|
}
|
||||||
|
var op byte
|
||||||
|
switch base {
|
||||||
|
case "MOVS":
|
||||||
|
op = 0xA5
|
||||||
|
if size == 1 {
|
||||||
|
op = 0xA4
|
||||||
|
}
|
||||||
|
case "STOS":
|
||||||
|
op = 0xAB
|
||||||
|
if size == 1 {
|
||||||
|
op = 0xAA
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("unsupported string instruction %q", base)
|
||||||
|
}
|
||||||
|
return e.emit(newInstr(size, []byte{op}))
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeInt encodes INT with its single imm8 operand. The field takes the
|
||||||
|
// low byte silently inside the 32-bit span, matching the scalar convention
|
||||||
|
// (go tool asm encodes INT $256 as CD 00).
|
||||||
|
func (e *enc) encodeInt(ops []Operand) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("INT expects 1 operand, got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("INT operand must be an immediate")
|
||||||
|
}
|
||||||
|
if imm < -(1<<31) || imm > (1<<32)-1 {
|
||||||
|
return fmt.Errorf("immediate $%d does not fit in 32 bits", int64(imm))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: []byte{0xCD}, modrm: -1, sib: -1, imm: []byte{byte(imm)}})
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeMxcsr encodes LDMXCSR (0F AE /2) and STMXCSR (0F AE /3); both take a
|
||||||
|
// single 32-bit memory operand.
|
||||||
|
func (e *enc) encodeMxcsr(digit int, ops []Operand) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("MXCSR instruction expects 1 operand, got %d", len(ops))
|
||||||
|
}
|
||||||
|
m, ok := ops[0].(Mem)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("MXCSR instruction requires a memory operand")
|
||||||
|
}
|
||||||
|
i := &instr{opcode: []byte{0x0F, 0xAE}, modrm: -1, sib: -1}
|
||||||
|
if err := setMem(i, digit, m); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// cvtIntOp maps the scalar float-to-integer conversions to their mandatory
|
||||||
|
// prefix and opcode: 0F 2D (CVTSD2S, CVTSS2S) and 0F 2C (their truncating
|
||||||
|
// CVTT forms). The mnemonic's Q/L suffix fixes the GPR destination width.
|
||||||
|
var cvtIntOp = map[string]struct {
|
||||||
|
prefix byte
|
||||||
|
op byte
|
||||||
|
}{
|
||||||
|
"CVTSD2S": {0xF2, 0x2D},
|
||||||
|
"CVTTSD2S": {0xF2, 0x2C},
|
||||||
|
"CVTSS2S": {0xF3, 0x2D},
|
||||||
|
"CVTTSS2S": {0xF3, 0x2C},
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeCvtInt encodes a scalar float-to-integer conversion: F2/F3 0F 2D/2C
|
||||||
|
// with reg = GPR destination, rm = XMM (or memory) source; REX.W follows the
|
||||||
|
// quad spellings.
|
||||||
|
func (e *enc) encodeCvtInt(base string, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||||||
|
}
|
||||||
|
spec := cvtIntOp[base]
|
||||||
|
src, dst := ops[0], ops[1]
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("%s destination must be a general register", base)
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x0F, spec.op})
|
||||||
|
i.prefix = spec.prefix
|
||||||
|
if err := setRM(i, dstReg, src, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeFmov encodes the x87 double move. The memory forms are DD /0
|
||||||
|
// (FMOVD mem, F: load) and DD /2 (FMOVD F, mem: store); a register-to-register
|
||||||
|
// move is DD C0+dst (FLD st(dst)), the form the Go assembler emits.
|
||||||
|
func (e *enc) encodeFmov(ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("FMOVD expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
src, dst := ops[0], ops[1]
|
||||||
|
srcReg, srcIsF := src.(Reg)
|
||||||
|
dstReg, dstIsF := dst.(Reg)
|
||||||
|
srcF := srcIsF && srcReg.fp
|
||||||
|
dstF := dstIsF && dstReg.fp
|
||||||
|
switch {
|
||||||
|
case srcF && dstF:
|
||||||
|
// The register form is DD /2 with rm = the destination (FST st(dst)).
|
||||||
|
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||||||
|
if err := setRMDigit(i, 2, dstReg, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
case dstF:
|
||||||
|
m, ok := src.(Mem)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("FMOVD: invalid source operand")
|
||||||
|
}
|
||||||
|
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||||||
|
if err := setMem(i, 0, m); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
case srcF:
|
||||||
|
m, ok := dst.(Mem)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("FMOVD: invalid destination operand")
|
||||||
|
}
|
||||||
|
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||||||
|
if err := setMem(i, 2, m); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
return fmt.Errorf("FMOVD needs an x87 register operand")
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- legacy SSE imm8, extract, insert and packed shift families --------------
|
||||||
|
|
||||||
|
// encodeSSEImm3 encodes an imm8-controlled three-operand form: OP $imm, src,
|
||||||
|
// dst with reg = dst, rm = src and the immediate appended last (PALIGNR,
|
||||||
|
// PBLENDW, PCMPESTRI, PCLMULQDQ, AESKEYGENASSIST, SHA1RNDS4).
|
||||||
|
func (e *enc) encodeSSEImm3(m sseImm3, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("SSE imm8 instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("SSE imm8 instruction needs an immediate first operand")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
src, dst := ops[1], ops[2]
|
||||||
|
dstReg, ok2 := dst.(Reg)
|
||||||
|
if !ok2 || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("SSE imm8 instruction destination must be a vector register")
|
||||||
|
}
|
||||||
|
opcode := []byte{0x0F, 0x38, m.op}
|
||||||
|
if m.map3A {
|
||||||
|
opcode = []byte{0x0F, 0x3A, m.op}
|
||||||
|
}
|
||||||
|
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dstReg, src, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSSEExtract encodes a lane extract: OP $imm, xsrc, dst with reg = the
|
||||||
|
// XMM source, rm = the GPR or memory destination (PEXTRB/PEXTRD/PEXTRQ and
|
||||||
|
// PEXTRW, whose GPR form is the older 0F C5 opcode and whose memory form the
|
||||||
|
// SSE4.1 0F3A 15 one).
|
||||||
|
func (e *enc) encodeSSEExtract(m sseExtract, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("extract expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("extract needs an immediate first operand")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
srcReg, srcVec := vecReg(ops[1])
|
||||||
|
if !srcVec {
|
||||||
|
return fmt.Errorf("extract source must be an XMM register")
|
||||||
|
}
|
||||||
|
opcode := m.op
|
||||||
|
if m.opMem != nil && memOperand(ops[2]) {
|
||||||
|
opcode = m.opMem
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0x66, opcode: opcode, modrm: -1, sib: -1, rexW: m.rexW}
|
||||||
|
if err := setRM(i, srcReg, ops[2], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSSEInsert encodes a lane insert: OP $imm, src, xdst with reg = the
|
||||||
|
// XMM destination and rm = the GPR or memory source (PINSRB/PINSRD/PINSRQ and
|
||||||
|
// PINSRW).
|
||||||
|
func (e *enc) encodeSSEInsert(m sseInsert, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("insert expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("insert needs an immediate first operand")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
dstReg, dstVec := vecReg(ops[2])
|
||||||
|
if !dstVec {
|
||||||
|
return fmt.Errorf("insert destination must be an XMM register")
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0x66, opcode: m.op, modrm: -1, sib: -1, rexW: m.rexW}
|
||||||
|
if err := setRM(i, dstReg, ops[1], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSSEShift encodes the legacy packed integer shifts. The immediate
|
||||||
|
// form is OP $imm, dst (66 0F 71/72/73 /digit); the variable form
|
||||||
|
// OP count, dst carries the count in an XMM register (or memory) on the
|
||||||
|
// 66 0F D1-F3 opcodes. The destination is always the register written.
|
||||||
|
func (e *enc) encodeSSEShift(name string, ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||||||
|
}
|
||||||
|
dstReg, ok := ops[1].(Reg)
|
||||||
|
if !ok || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("%s destination must be the second, vector operand", name)
|
||||||
|
}
|
||||||
|
if imm, isImm := ops[0].(Imm); isImm {
|
||||||
|
spec := sseShiftImm[name]
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0x66, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
|
||||||
|
if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
if !vecOrMem(ops[0]) {
|
||||||
|
return fmt.Errorf("%s count must be an immediate, a vector register or memory", name)
|
||||||
|
}
|
||||||
|
op, ok := sseShiftVar[name]
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s has no variable-count form", name)
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0x66, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeCmpsd encodes CMPSD, the scalar double compare with its predicate
|
||||||
|
// immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family:
|
||||||
|
// F2 0F C2 with reg = dst, rm = src.
|
||||||
|
func (e *enc) encodeCmpsd(ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[2].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("CMPSD predicate must be an immediate")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
dstReg, ok2 := ops[1].(Reg)
|
||||||
|
if !ok2 || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("CMPSD destination must be a vector register")
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSha256rnds2 encodes SHA256RNDS2, whose first operand must be the
|
||||||
|
// literal X0 carrying the round constant: OP X0, src, dst (0F38 CB, no
|
||||||
|
// prefix, reg = dst, rm = src; X0 is implicit on the wire).
|
||||||
|
func (e *enc) encodeSha256rnds2(ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("SHA256RNDS2 expects 3 operands (X0, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
x0, ok := ops[0].(Reg)
|
||||||
|
if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 {
|
||||||
|
return fmt.Errorf("SHA256RNDS2 first operand must be X0")
|
||||||
|
}
|
||||||
|
dstReg, ok2 := ops[2].(Reg)
|
||||||
|
if !ok2 || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("SHA256RNDS2 destination must be a vector register")
|
||||||
|
}
|
||||||
|
i := &instr{opcode: []byte{0x0F, 0x38, 0xCB}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dstReg, ops[1], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|||||||
@@ -19,8 +19,8 @@ import (
|
|||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel —
|
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel;
|
||||||
// all functions plus the file-local mask24 constant — and checks that every
|
// all functions plus the file-local mask24 constant; and checks that every
|
||||||
// static-symbol load resolves to the right bytes in the image.
|
// static-symbol load resolves to the right bytes in the image.
|
||||||
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
||||||
path := "../../go-libraries/go-flac/avx2_amd64.s"
|
path := "../../go-libraries/go-flac/avx2_amd64.s"
|
||||||
@@ -81,7 +81,7 @@ func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
|
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
|
||||||
// kernel — all functions plus the file-global idx16 constant — and checks
|
// kernel, all functions plus the file-global idx16 constant, and checks
|
||||||
// that the static-symbol load resolves to the right bytes in the image.
|
// that the static-symbol load resolves to the right bytes in the image.
|
||||||
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
|
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
|
||||||
path := "../../go-libraries/go-flac/avx512_amd64.s"
|
path := "../../go-libraries/go-flac/avx512_amd64.s"
|
||||||
|
|||||||
@@ -94,9 +94,9 @@ DATA ·table<>+0(SB)/8, $0x1122334455667788
|
|||||||
}
|
}
|
||||||
|
|
||||||
// The debug_line program: LNE_set_address (the R_ADDR relocation
|
// The debug_line program: LNE_set_address (the R_ADDR relocation
|
||||||
// carries the function address), then one row per line change — the
|
// carries the function address), then one row per line change; the
|
||||||
// TEXT is on line 4 (a leading blank line precedes the include), the
|
// TEXT is on line 4 (a leading blank line precedes the include), the
|
||||||
// instructions on lines 5–9 — an advance to the 20-byte end and an
|
// instructions on lines 5-9; an advance to the 20-byte end and an
|
||||||
// end-of-sequence.
|
// end-of-sequence.
|
||||||
linesOff := le.Uint32(dataIdx[4*2:])
|
linesOff := le.Uint32(dataIdx[4*2:])
|
||||||
lines := dataBlk[linesOff : linesOff+21]
|
lines := dataBlk[linesOff : linesOff+21]
|
||||||
|
|||||||
+94
-80
@@ -25,6 +25,10 @@ type Image struct {
|
|||||||
Symbols map[string]int // static symbol → byte offset within the image
|
Symbols map[string]int // static symbol → byte offset within the image
|
||||||
DataSyms []DataSymbol // GLOBL symbols, in layout order
|
DataSyms []DataSymbol // GLOBL symbols, in layout order
|
||||||
Externals []string // referenced but undefined symbols, sorted
|
Externals []string // referenced but undefined symbols, sorted
|
||||||
|
// SourcePath is the assembled file's path, recorded in the DWARF
|
||||||
|
// sections in place of a placeholder name. Empty when the image was
|
||||||
|
// not built from a named file.
|
||||||
|
SourcePath string
|
||||||
}
|
}
|
||||||
|
|
||||||
// FuncLayout describes one assembled function within an Image.
|
// FuncLayout describes one assembled function within an Image.
|
||||||
@@ -81,12 +85,9 @@ func (fl *FuncLayout) LineAt(offset int) int {
|
|||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
// RelocKind Reloc is one static-symbol reference within a function body: the disp32
|
// RelocKind discriminates the relocation a static-symbol reference needs;
|
||||||
// field at Off (function-relative) must reach the symbol plus Addend,
|
// the encoders record one per SB reference, and the object-file emitters map
|
||||||
// measured from After, the address just past the instruction. An External
|
// it to their format's relocation type.
|
||||||
// relocation names a symbol no GLOBL in the file defines; the object-file
|
|
||||||
// emitters carry it into the output's relocation table.
|
|
||||||
// RelocKind discriminates the type of relocation needed.
|
|
||||||
type RelocKind int
|
type RelocKind int
|
||||||
|
|
||||||
const (
|
const (
|
||||||
@@ -96,7 +97,6 @@ const (
|
|||||||
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
|
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
|
||||||
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
|
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
|
||||||
RelRISCVJal // R_RISCV_JAL (J-type call)
|
RelRISCVJal // R_RISCV_JAL (J-type call)
|
||||||
RelPCRelAbs // 32-bit absolute (R_RISCV_32)
|
|
||||||
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
|
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
|
||||||
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
|
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
|
||||||
RelArm64Addr // R_ADDRARM64 (ADRP + ADD pair)
|
RelArm64Addr // R_ADDRARM64 (ADRP + ADD pair)
|
||||||
@@ -106,6 +106,13 @@ const (
|
|||||||
)
|
)
|
||||||
|
|
||||||
type Reloc struct {
|
type Reloc struct {
|
||||||
|
// Off is the function-relative offset of the field the linker patches
|
||||||
|
// and After the address just past the instruction, the base the
|
||||||
|
// assembler measures PC-relative displacements from. Name plus
|
||||||
|
// Addend select the target: the symbol plus the byte offset. An
|
||||||
|
// External relocation names a symbol no GLOBL in the file defines;
|
||||||
|
// the object-file emitters carry it into the output's relocation
|
||||||
|
// table.
|
||||||
Off int
|
Off int
|
||||||
After int
|
After int
|
||||||
Name string
|
Name string
|
||||||
@@ -150,7 +157,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
|||||||
}
|
}
|
||||||
link := &linkInfo{symbols: known, allowExternal: true}
|
link := &linkInfo{symbols: known, allowExternal: true}
|
||||||
|
|
||||||
img := &Image{Symbols: map[string]int{}}
|
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
||||||
textOff := map[string]int{}
|
textOff := map[string]int{}
|
||||||
type asmFunc struct {
|
type asmFunc struct {
|
||||||
name string
|
name string
|
||||||
@@ -267,7 +274,7 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
|
|||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
img := &Image{Symbols: map[string]int{}}
|
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
||||||
for _, d := range f.Decls {
|
for _, d := range f.Decls {
|
||||||
t, ok := d.(*ast.Text)
|
t, ok := d.(*ast.Text)
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -339,7 +346,7 @@ func AssembleFileLOONG64(f *ast.File) (*Image, error) {
|
|||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
img := &Image{Symbols: map[string]int{}}
|
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
||||||
for _, d := range f.Decls {
|
for _, d := range f.Decls {
|
||||||
t, ok := d.(*ast.Text)
|
t, ok := d.(*ast.Text)
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -440,83 +447,90 @@ type dataSym struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// collectData gathers the file's static symbols (GLOBL) and their initial
|
// collectData gathers the file's static symbols (GLOBL) and their initial
|
||||||
// contents (DATA) into byte buffers, in declaration order.
|
// contents (DATA) into byte buffers. Two passes: the Plan 9 convention puts
|
||||||
|
// every DATA line before its symbol's GLOBL, so the symbols are registered
|
||||||
|
// before the initialisers are applied.
|
||||||
func collectData(f *ast.File) ([]dataSym, error) {
|
func collectData(f *ast.File) ([]dataSym, error) {
|
||||||
index := map[string]int{}
|
index := map[string]int{}
|
||||||
var syms []dataSym
|
var syms []dataSym
|
||||||
for _, d := range f.Decls {
|
for _, d := range f.Decls {
|
||||||
switch dd := d.(type) {
|
gd, ok := d.(*ast.Globl)
|
||||||
case *ast.Globl:
|
if !ok {
|
||||||
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
continue
|
||||||
continue
|
}
|
||||||
}
|
if gd.Name == nil || gd.Name.Pseudo != "SB" {
|
||||||
name := dd.Name.Name
|
continue
|
||||||
if _, dup := index[name]; dup {
|
}
|
||||||
return nil, fmt.Errorf("duplicate GLOBL %q", name)
|
name := gd.Name.Name
|
||||||
}
|
if _, dup := index[name]; dup {
|
||||||
size := 0
|
return nil, fmt.Errorf("duplicate GLOBL %q", name)
|
||||||
if dd.Size != nil && dd.Size.Imm.HasVal {
|
}
|
||||||
size = int(dd.Size.Imm.Val)
|
size := 0
|
||||||
}
|
if gd.Size != nil && gd.Size.Imm.HasVal {
|
||||||
index[name] = len(syms)
|
size = int(gd.Size.Imm.Val)
|
||||||
ds := dataSym{
|
}
|
||||||
name: name,
|
index[name] = len(syms)
|
||||||
pkg: dd.Name.Pkg,
|
ds := dataSym{
|
||||||
buf: make([]byte, size),
|
name: name,
|
||||||
size: size,
|
pkg: gd.Name.Pkg,
|
||||||
static: dd.Name.Static,
|
buf: make([]byte, size),
|
||||||
}
|
size: size,
|
||||||
for _, f := range dd.Flags {
|
static: gd.Name.Static,
|
||||||
switch f {
|
}
|
||||||
case "RODATA":
|
for _, f := range gd.Flags {
|
||||||
ds.rodata = true
|
switch f {
|
||||||
case "DUPOK":
|
case "RODATA":
|
||||||
ds.dupok = true
|
ds.rodata = true
|
||||||
default:
|
case "DUPOK":
|
||||||
// Legacy numeric flag constants (runtime/textflag.h):
|
ds.dupok = true
|
||||||
// DUPOK is 2, RODATA is 8; combinations arrive as one
|
default:
|
||||||
// number (e.g. 10 = RODATA|DUPOK).
|
// Legacy numeric flag constants (runtime/textflag.h):
|
||||||
if n, err := strconv.Atoi(f); err == nil {
|
// DUPOK is 2, RODATA is 8; combinations arrive as one
|
||||||
if n&2 != 0 {
|
// number (e.g. 10 = RODATA|DUPOK).
|
||||||
ds.dupok = true
|
if n, err := strconv.Atoi(f); err == nil {
|
||||||
}
|
if n&2 != 0 {
|
||||||
if n&8 != 0 {
|
ds.dupok = true
|
||||||
ds.rodata = true
|
}
|
||||||
}
|
if n&8 != 0 {
|
||||||
|
ds.rodata = true
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
syms = append(syms, ds)
|
}
|
||||||
|
syms = append(syms, ds)
|
||||||
case *ast.Data:
|
}
|
||||||
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
for _, d := range f.Decls {
|
||||||
continue
|
dd, ok := d.(*ast.Data)
|
||||||
}
|
if !ok {
|
||||||
i, ok := index[dd.Name.Name]
|
continue
|
||||||
if !ok {
|
}
|
||||||
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
|
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
||||||
}
|
continue
|
||||||
if dd.Value == nil || !dd.Value.Imm.HasVal {
|
}
|
||||||
return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name)
|
i, ok := index[dd.Name.Name]
|
||||||
}
|
if !ok {
|
||||||
w := dd.Width
|
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
|
||||||
switch w {
|
}
|
||||||
case 1, 2, 4, 8:
|
if dd.Value == nil || !dd.Value.Imm.HasVal {
|
||||||
default:
|
return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name)
|
||||||
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
|
}
|
||||||
}
|
w := dd.Width
|
||||||
off := dd.Name.Offset
|
switch w {
|
||||||
buf := syms[i].buf
|
case 1, 2, 4, 8:
|
||||||
if off < 0 || off+int64(w) > int64(len(buf)) {
|
default:
|
||||||
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
|
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
|
||||||
}
|
}
|
||||||
v := dd.Value.Imm.Val
|
off := dd.Name.Offset
|
||||||
if dd.Value.Imm.Neg {
|
buf := syms[i].buf
|
||||||
v = -v
|
if off < 0 || off+int64(w) > int64(len(buf)) {
|
||||||
}
|
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
|
||||||
for j := range w {
|
}
|
||||||
buf[off+int64(j)] = byte(v >> (8 * j))
|
v := dd.Value.Imm.Val
|
||||||
}
|
if dd.Value.Imm.Neg {
|
||||||
|
v = -v
|
||||||
|
}
|
||||||
|
for j := range w {
|
||||||
|
buf[off+int64(j)] = byte(v >> (8 * j))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return syms, nil
|
return syms, nil
|
||||||
|
|||||||
+2
-2
@@ -10,8 +10,8 @@ import (
|
|||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestAssembleFileStaticData checks the whole-image layout — code, padding
|
// TestAssembleFileStaticData checks the whole-image layout; code, padding
|
||||||
// and the data section — and that the RIP-relative displacements of static
|
// and the data section; and that the RIP-relative displacements of static
|
||||||
// symbol loads resolve to the right bytes.
|
// symbol loads resolve to the right bytes.
|
||||||
func TestAssembleFileStaticData(t *testing.T) {
|
func TestAssembleFileStaticData(t *testing.T) {
|
||||||
f, errs := parser.Parse("d_amd64.s", `
|
f, errs := parser.Parse("d_amd64.s", `
|
||||||
|
|||||||
+510
-17
@@ -6,6 +6,7 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
"math/bits"
|
"math/bits"
|
||||||
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
@@ -258,6 +259,9 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
|
|||||||
|
|
||||||
// 16-bit branches (BEQ/BNE/BLT/BGE/BLTU/BGEU) and JIRL.
|
// 16-bit branches (BEQ/BNE/BLT/BGE/BLTU/BGEU) and JIRL.
|
||||||
if op, ok := l64branchTable[mnem]; ok {
|
if op, ok := l64branchTable[mnem]; ok {
|
||||||
|
if mnem == "JIRL" {
|
||||||
|
return encodeLOONG64Jirl(op, ops)
|
||||||
|
}
|
||||||
return encodeLOONG64Branch16(mnem, op, ops, pc, offsets, resolve)
|
return encodeLOONG64Branch16(mnem, op, ops, pc, offsets, resolve)
|
||||||
}
|
}
|
||||||
// Single-register branches with 21-bit offsets (BLTZ/BGEZ/BLEZ/BGTZ,
|
// Single-register branches with 21-bit offsets (BLTZ/BGEZ/BLEZ/BGTZ,
|
||||||
@@ -317,6 +321,16 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
|
|||||||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The LSX/LASX vector slice and the VMOVQ/XVMOVQ move family, before
|
||||||
|
// the integer/FP table (their mnemonics overlap the table's 2R format
|
||||||
|
// but resolve vector-bank registers).
|
||||||
|
if code, handled, err := encodeLOONG64Vector(instr, mnem, fi); handled {
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return code, nil
|
||||||
|
}
|
||||||
|
|
||||||
enc, ok := l64InstrTable[mnem]
|
enc, ok := l64InstrTable[mnem]
|
||||||
if !ok {
|
if !ok {
|
||||||
return nil, fmt.Errorf("unsupported loong64 instruction %q", mnem)
|
return nil, fmt.Errorf("unsupported loong64 instruction %q", mnem)
|
||||||
@@ -438,6 +452,15 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
|
|||||||
if rj < 0 || rd < 0 {
|
if rj < 0 || rd < 0 {
|
||||||
return nil, fmt.Errorf("invalid register operand")
|
return nil, fmt.Errorf("invalid register operand")
|
||||||
}
|
}
|
||||||
|
// The toolchain validates the bit numbers ("illegal bit number"):
|
||||||
|
// 0..31 for the .w forms, 0..63 for the .d forms, lsb <= msb.
|
||||||
|
b := 64
|
||||||
|
if strings.HasSuffix(mnem, "W") {
|
||||||
|
b = 32
|
||||||
|
}
|
||||||
|
if msb < 0 || msb >= b || lsb < 0 || lsb >= b || lsb > msb {
|
||||||
|
return nil, fmt.Errorf("%s: illegal bit number (msb %d, lsb %d)", mnem, msb, lsb)
|
||||||
|
}
|
||||||
return l64wordLE(l64irir(enc.op, msb, rj, lsb, rd)), nil
|
return l64wordLE(l64irir(enc.op, msb, rj, lsb, rd)), nil
|
||||||
|
|
||||||
case l64Firrr:
|
case l64Firrr:
|
||||||
@@ -554,6 +577,46 @@ func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[stri
|
|||||||
return l64wordLE(l64bbl(opc, v)), nil
|
return l64wordLE(l64bbl(opc, v)), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// encodeLOONG64Jirl encodes the raw JIRL spelling, JIRL rd, rj, offset, the
|
||||||
|
// form the verify trampolines use. The (rj) indirect form without an offset
|
||||||
|
// is handled by encodeLOONG64Branch.
|
||||||
|
func encodeLOONG64Jirl(op uint32, ops []*ast.Operand) ([]byte, error) {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return nil, fmt.Errorf("JIRL expects 3 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
rd := l64Reg(ops[0])
|
||||||
|
rj := l64Reg(ops[1])
|
||||||
|
if rd < 0 || rj < 0 {
|
||||||
|
return nil, fmt.Errorf("invalid register operand")
|
||||||
|
}
|
||||||
|
off, ok := l64offsetOperand(ops[2])
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("JIRL expects an immediate offset, got %q", ops[2].Raw)
|
||||||
|
}
|
||||||
|
if (int64(off)<<16)>>16 != int64(off) {
|
||||||
|
return nil, fmt.Errorf("JIRL offset %d out of the 16-bit range", off)
|
||||||
|
}
|
||||||
|
return l64wordLE(l64irr16(op, int(off), rj, rd)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64offsetOperand reads a bare numeric branch offset: an immediate ($n) or a
|
||||||
|
// plain number, which parses as an empty address carrying the digits in Raw.
|
||||||
|
func l64offsetOperand(op *ast.Operand) (int32, bool) {
|
||||||
|
if op.Imm.HasVal {
|
||||||
|
v := op.Imm.Val
|
||||||
|
if op.Imm.Neg {
|
||||||
|
v = -v
|
||||||
|
}
|
||||||
|
return int32(v), true
|
||||||
|
}
|
||||||
|
if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "" && op.Addr.Index == "" {
|
||||||
|
if v, err := strconv.ParseInt(op.Raw, 0, 64); err == nil {
|
||||||
|
return int32(v), true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
|
||||||
// encodeLOONG64Branch16 encodes a 16-bit branch (BEQ/BNE/BLT/BGE/BLTU/BGEU):
|
// encodeLOONG64Branch16 encodes a 16-bit branch (BEQ/BNE/BLT/BGE/BLTU/BGEU):
|
||||||
// INSTR rj, rd, label, or INSTR rj, label with rd = R0, which the toolchain
|
// INSTR rj, rd, label, or INSTR rj, label with rd = R0, which the toolchain
|
||||||
// turns into the 21-bit BEQZ/BNEZ form when the register is the only operand.
|
// turns into the 21-bit BEQZ/BNEZ form when the register is the only operand.
|
||||||
@@ -574,6 +637,15 @@ func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, o
|
|||||||
if rj < 0 {
|
if rj < 0 {
|
||||||
return nil, fmt.Errorf("invalid register operand")
|
return nil, fmt.Errorf("invalid register operand")
|
||||||
}
|
}
|
||||||
|
if mnem == "BLTU" || mnem == "BGEU" {
|
||||||
|
// The unsigned compares have no single-register pseudo: the
|
||||||
|
// toolchain keeps the register-register form with rd = R0
|
||||||
|
// (bltu rj, r0 is never taken), not a sometimes-taken beqz.
|
||||||
|
if (v<<16)>>16 != v {
|
||||||
|
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
|
||||||
|
}
|
||||||
|
return l64wordLE(l64irr16(op, v, rj, 0)), nil
|
||||||
|
}
|
||||||
if (v<<11)>>11 != v {
|
if (v<<11)>>11 != v {
|
||||||
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
|
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
|
||||||
}
|
}
|
||||||
@@ -813,9 +885,16 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs
|
|||||||
if rd < 0 {
|
if rd < 0 {
|
||||||
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
|
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
|
||||||
}
|
}
|
||||||
// MOVF/MOVD $imm, Fd → materialise in R30, then movgr2fr.{w,d}.
|
// MOVW $imm, Fd is the only immediate-to-F form the toolchain's optab
|
||||||
if (mnem == "MOVF" || mnem == "MOVD") && loong64RegClass(operandRegName(dst)) == l64ClsFP {
|
// accepts (AMOVW's C_12CON against C_FREG): it materialises the
|
||||||
return encodeLOONG64ImmToFp(rd, l64Imm64(src), mnem), nil
|
// constant in R30 and moves it across with movgr2fr.w. MOVV/MOVF/
|
||||||
|
// MOVD are illegal combinations there, and are diagnosed here rather
|
||||||
|
// than silently written into the GPR of the register's number.
|
||||||
|
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
|
||||||
|
if mnem != "MOVW" {
|
||||||
|
return nil, fmt.Errorf("%s $imm: illegal combination with an F register destination (only MOVW $c, Fd is supported)", mnem)
|
||||||
|
}
|
||||||
|
return encodeLOONG64ImmToFp(rd, l64Imm64(src))
|
||||||
}
|
}
|
||||||
return encodeLOONG64LoadImm(rd, l64Imm64(src), mnem), nil
|
return encodeLOONG64LoadImm(rd, l64Imm64(src), mnem), nil
|
||||||
}
|
}
|
||||||
@@ -896,8 +975,8 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
|
|||||||
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
|
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
|
||||||
return 8 // pcalau12i + addi.d
|
return 8 // pcalau12i + addi.d
|
||||||
}
|
}
|
||||||
if (mnem == "MOVF" || mnem == "MOVD") && loong64RegClass(operandRegName(dst)) == l64ClsFP {
|
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
|
||||||
return 8 // addi/ori r30 + movgr2fr
|
return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid)
|
||||||
}
|
}
|
||||||
v := l64Imm64(src)
|
v := l64Imm64(src)
|
||||||
if v == 0 {
|
if v == 0 {
|
||||||
@@ -938,22 +1017,24 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeLOONG64ImmToFp materialises a 12-bit immediate in R30 and moves it to
|
// encodeLOONG64ImmToFp materialises a 12-bit immediate in R30 and moves it
|
||||||
// an F register (the toolchain's case 34: movgr2fr.w/movgr2fr.d).
|
// to an F register, the toolchain's expansion of MOVW $c, Fd: ori (which
|
||||||
func encodeLOONG64ImmToFp(fd int, v int64, mnem string) []byte {
|
// zero-extends) for the positive span, addi.w for zero and the negative
|
||||||
// ori for positive constants, addi.d for zero/negative.
|
// span, then movgr2fr.w. The toolchain's optab accepts no wider constant on
|
||||||
op := uint32(0x00b << 22)
|
// this path (it never materialises one fully first), so values outside
|
||||||
if v > 0 {
|
// [-2048, 4095] are diagnosed rather than masked into si12.
|
||||||
op = 0x00e << 22
|
func encodeLOONG64ImmToFp(fd int, v int64) ([]byte, error) {
|
||||||
|
if v < -2048 || v > 4095 {
|
||||||
|
return nil, fmt.Errorf("MOVW $%d: immediate out of the [-2048, 4095] range for an F register destination", v)
|
||||||
}
|
}
|
||||||
mov := uint32(0x452a << 10) // movgr2fr.d
|
op := uint32(0x00a << 22) // addi.w r30, r0, v (sign-extends)
|
||||||
if mnem == "MOVF" {
|
if v > 0 {
|
||||||
mov = 0x4529 << 10 // movgr2fr.w
|
op = 0x00e << 22 // ori r30, r0, v (zero-extends)
|
||||||
}
|
}
|
||||||
return l64WordsLE(
|
return l64WordsLE(
|
||||||
l64irr(op, int(v), 0, 30),
|
l64irr(op, int(v), 0, 30),
|
||||||
l64rr(mov, 30, fd),
|
l64rr(0x4529<<10, 30, fd), // movgr2fr.w fd, r30
|
||||||
)
|
), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---- 64-bit immediate classification ----
|
// ---- 64-bit immediate classification ----
|
||||||
@@ -1419,3 +1500,415 @@ func l64Label(op *ast.Operand) string {
|
|||||||
}
|
}
|
||||||
return op.Raw
|
return op.Raw
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- LSX/LASX (V*/XV*) vector dispatch ----
|
||||||
|
|
||||||
|
// l64VecOperand describes a vector register operand: the 5-bit register
|
||||||
|
// number, its bank and an optional width or element suffix (V0.B16,
|
||||||
|
// V1.V[0], X3.WU[2]). The parser hands suffixed operands over verbatim
|
||||||
|
// (the element index survives only in the raw text), so the suffix is
|
||||||
|
// scanned from op.Raw.
|
||||||
|
type l64VecOperand struct {
|
||||||
|
num int // 5-bit register number
|
||||||
|
lasx bool // X bank (LASX) rather than V (LSX)
|
||||||
|
width byte // suffix width letter (B/H/W/V), 0 on a bare register
|
||||||
|
lanes int // lane count of a width suffix (B16 → 16)
|
||||||
|
elem int // element index of a .T[i] suffix
|
||||||
|
hasEl bool // the suffix names an element (.T[i])
|
||||||
|
unsig bool // the suffix carries the U marker (.BU[0])
|
||||||
|
hasSuf bool // any suffix present
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64ParseVecOperand parses a vector register operand with an optional
|
||||||
|
// width or element suffix. ok reports whether the operand names a vector
|
||||||
|
// register at all (V or X bank, with or without a suffix).
|
||||||
|
func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) {
|
||||||
|
if op.Kind == ast.OpImmediate {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
name := strings.ReplaceAll(op.Raw, " ", "")
|
||||||
|
if name == "" || (name[0] != 'V' && name[0] != 'X') {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
i := 1
|
||||||
|
num := 0
|
||||||
|
for i < len(name) && name[i] >= '0' && name[i] <= '9' {
|
||||||
|
num = num*10 + int(name[i]-'0')
|
||||||
|
if num > 31 {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
i++
|
||||||
|
}
|
||||||
|
if i == 1 {
|
||||||
|
return v, false // no register digits
|
||||||
|
}
|
||||||
|
v.num, v.lasx = num, name[0] == 'X'
|
||||||
|
if i == len(name) {
|
||||||
|
return v, true
|
||||||
|
}
|
||||||
|
if name[i] != '.' || i+2 > len(name) {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
i++
|
||||||
|
w := name[i]
|
||||||
|
if w != 'B' && w != 'H' && w != 'W' && w != 'V' {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
v.width, v.hasSuf = w, true
|
||||||
|
i++
|
||||||
|
if i < len(name) && name[i] == 'U' {
|
||||||
|
v.unsig = true
|
||||||
|
i++
|
||||||
|
}
|
||||||
|
if i < len(name) && name[i] == '[' {
|
||||||
|
// Element form .T[i]: the closing bracket ends the operand.
|
||||||
|
if name[len(name)-1] != ']' || i+2 > len(name)-1 {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
idx := 0
|
||||||
|
for _, c := range name[i+1 : len(name)-1] {
|
||||||
|
if c < '0' || c > '9' {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
idx = idx*10 + int(c-'0')
|
||||||
|
if idx > 31 {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
v.elem, v.hasEl = idx, true
|
||||||
|
return v, true
|
||||||
|
}
|
||||||
|
// Width form .T<lanes>: the trailing digits give the lane count.
|
||||||
|
lanes := 0
|
||||||
|
if i >= len(name) {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
for ; i < len(name); i++ {
|
||||||
|
if name[i] < '0' || name[i] > '9' {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
lanes = lanes*10 + int(name[i]-'0')
|
||||||
|
if lanes > 64 {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
v.lanes = lanes
|
||||||
|
return v, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64VecSuffixWidth validates a width suffix against the bank (LSX:
|
||||||
|
// B16/H8/W4/V2, LASX: B32/H16/W8/V4) and returns the encoded 2-bit width
|
||||||
|
// selector of vreplgr2vr and vldrepl.
|
||||||
|
func l64VecSuffixWidth(lasx bool, v l64VecOperand) (int, bool) {
|
||||||
|
want := map[byte]int{'B': 16, 'H': 8, 'W': 4, 'V': 2}
|
||||||
|
if lasx {
|
||||||
|
want = map[byte]int{'B': 32, 'H': 16, 'W': 8, 'V': 4}
|
||||||
|
}
|
||||||
|
lanes, ok := want[v.width]
|
||||||
|
if !ok || lanes != v.lanes {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
switch v.width {
|
||||||
|
case 'B':
|
||||||
|
return 0, true
|
||||||
|
case 'H':
|
||||||
|
return 1, true
|
||||||
|
case 'W':
|
||||||
|
return 2, true
|
||||||
|
default:
|
||||||
|
return 3, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64VecElementBase validates an element suffix against the bank and
|
||||||
|
// returns the encoded index field: the index rides in the rk field above a
|
||||||
|
// per-width base (vpickve2gr/vinsgr2vr give ui4 to .b, ui3 to .h, ui2 to .w
|
||||||
|
// and ui1 to .d). The LASX bank has no .b/.h element forms: the toolchain
|
||||||
|
// rejects `XVMOVQ R4, X2.B[0]` and `XVMOVQ X3.B[31], R5`.
|
||||||
|
func l64VecElementBase(lasx bool, v l64VecOperand) (int, bool) {
|
||||||
|
limit, base := 0, 0
|
||||||
|
switch v.width {
|
||||||
|
case 'B':
|
||||||
|
if lasx {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
limit, base = 15, 0
|
||||||
|
case 'H':
|
||||||
|
if lasx {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
limit, base = 7, 16
|
||||||
|
case 'W':
|
||||||
|
limit, base = 3, 24
|
||||||
|
if lasx {
|
||||||
|
limit, base = 7, 16
|
||||||
|
}
|
||||||
|
case 'V':
|
||||||
|
limit, base = 1, 28
|
||||||
|
if lasx {
|
||||||
|
limit, base = 3, 24
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
if v.elem > limit {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
return base + v.elem, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeLOONG64Vector encodes the LSX/LASX mnemonics the table marks as
|
||||||
|
// vector plus the VMOVQ/XVMOVQ move family. handled reports whether the
|
||||||
|
// mnemonic belongs to the vector slice; the operand shapes and opcode
|
||||||
|
// constants reproduce GOARCH=loong64 `go tool asm` exactly.
|
||||||
|
func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]byte, bool, error) {
|
||||||
|
if mnem == "VMOVQ" || mnem == "XVMOVQ" {
|
||||||
|
code, err := encodeLOONG64Vmovq(mnem == "XVMOVQ", instr.Operands, fi)
|
||||||
|
return code, true, err
|
||||||
|
}
|
||||||
|
lasx, ok := l64VecBank[mnem]
|
||||||
|
if !ok {
|
||||||
|
return nil, false, nil
|
||||||
|
}
|
||||||
|
ops := instr.Operands
|
||||||
|
bank := "V"
|
||||||
|
if lasx {
|
||||||
|
bank = "X"
|
||||||
|
}
|
||||||
|
vec := func(op *ast.Operand) (int, error) {
|
||||||
|
v, isVec := l64ParseVecOperand(op)
|
||||||
|
if !isVec || v.lasx != lasx || v.hasSuf {
|
||||||
|
return -1, fmt.Errorf("%s: expected a bare %s0-%s31 vector register, got %q", mnem, bank, bank, op.Raw)
|
||||||
|
}
|
||||||
|
return v.num, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Two-operand forms (vpcnt.v): INSTR vj, vd.
|
||||||
|
if l64Vec2R[mnem] {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||||||
|
}
|
||||||
|
vj, err := vec(ops[0])
|
||||||
|
if err != nil {
|
||||||
|
return nil, true, err
|
||||||
|
}
|
||||||
|
vd, err := vec(ops[1])
|
||||||
|
if err != nil {
|
||||||
|
return nil, true, err
|
||||||
|
}
|
||||||
|
return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, vd)), true, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Immediate forms: INSTR $imm, vd or INSTR $imm, vj, vd.
|
||||||
|
if e, imm := l64VecImmInfo[mnem]; imm && len(ops) >= 2 && isImmOperand(ops[0]) {
|
||||||
|
if len(ops) > 3 {
|
||||||
|
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||||||
|
}
|
||||||
|
imm := int(immFromOperand(ops[0]))
|
||||||
|
if imm < e.min || imm > e.max {
|
||||||
|
return nil, true, fmt.Errorf("%s: immediate out of range [%d, %d]", mnem, e.min, e.max)
|
||||||
|
}
|
||||||
|
vd, err := vec(ops[len(ops)-1])
|
||||||
|
if err != nil {
|
||||||
|
return nil, true, err
|
||||||
|
}
|
||||||
|
vj := vd
|
||||||
|
if len(ops) == 3 {
|
||||||
|
if vj, err = vec(ops[1]); err != nil {
|
||||||
|
return nil, true, err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return l64wordLE(l64irr(e.op, (imm+e.bias)&e.mask, vj, vd)), true, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Vector-to-condition forms: INSTR vj, FCCn.
|
||||||
|
if l64InstrTable[mnem].format == l64Fvcf {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||||||
|
}
|
||||||
|
vj, err := vec(ops[0])
|
||||||
|
if err != nil {
|
||||||
|
return nil, true, err
|
||||||
|
}
|
||||||
|
if loong64RegClass(operandRegName(ops[1])) != l64ClsFCC {
|
||||||
|
return nil, true, fmt.Errorf("%s: expected an FCC condition flag, got %q", mnem, ops[1].Raw)
|
||||||
|
}
|
||||||
|
fcc := loong64RegNum(operandRegName(ops[1]))
|
||||||
|
return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, fcc)), true, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Three-register forms: INSTR vk, vj, vd or INSTR vk, vd (vj = vd).
|
||||||
|
if len(ops) != 2 && len(ops) != 3 {
|
||||||
|
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||||||
|
}
|
||||||
|
vk, err := vec(ops[0])
|
||||||
|
if err != nil {
|
||||||
|
return nil, true, err
|
||||||
|
}
|
||||||
|
vd, err := vec(ops[len(ops)-1])
|
||||||
|
if err != nil {
|
||||||
|
return nil, true, err
|
||||||
|
}
|
||||||
|
vj := vd
|
||||||
|
if len(ops) == 3 {
|
||||||
|
if vj, err = vec(ops[1]); err != nil {
|
||||||
|
return nil, true, err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return l64wordLE(l64rrr(l64InstrTable[mnem].op, vk, vj, vd)), true, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeLOONG64Vmovq encodes the VMOVQ/XVMOVQ move family. One mnemonic
|
||||||
|
// covers the whole LSX/LASX transfer surface, dispatched by operand shape
|
||||||
|
// exactly as the toolchain's table does:
|
||||||
|
//
|
||||||
|
// VMOVQ vd, off(rj) vst VMOVQ off(rj), vd vld
|
||||||
|
// VMOVQ vd, (rj)(rk) vstx VMOVQ (rj)(rk), vd vldx
|
||||||
|
// VMOVQ off(rj), vd.T vldrepl (load and replicate one element)
|
||||||
|
// VMOVQ vj, vd vori.b $0 (a register move)
|
||||||
|
// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register)
|
||||||
|
// VMOVQ vj.T[i], rd vpickve2gr (extract one element)
|
||||||
|
// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element)
|
||||||
|
func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) {
|
||||||
|
enc := l64VmovqTable[lasx]
|
||||||
|
bank := "V"
|
||||||
|
if lasx {
|
||||||
|
bank = "X"
|
||||||
|
}
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return nil, fmt.Errorf("VMOVQ expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
src, srcVec := l64ParseVecOperand(ops[0])
|
||||||
|
dst, dstVec := l64ParseVecOperand(ops[1])
|
||||||
|
srcMem := isMemOperand(ops[0])
|
||||||
|
dstMem := isMemOperand(ops[1])
|
||||||
|
srcIdx := srcMem && ops[0].Addr.Index != ""
|
||||||
|
dstIdx := dstMem && ops[1].Addr.Index != ""
|
||||||
|
intReg := func(op *ast.Operand) (int, error) {
|
||||||
|
if isMemOperand(op) {
|
||||||
|
return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw)
|
||||||
|
}
|
||||||
|
name := operandRegName(op)
|
||||||
|
if loong64RegClass(name) != l64ClsGR {
|
||||||
|
return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw)
|
||||||
|
}
|
||||||
|
return loong64RegNum(name), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant),
|
||||||
|
// both operands bare registers of the same bank.
|
||||||
|
if srcVec && dstVec {
|
||||||
|
if src.hasSuf || dst.hasSuf {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: a register move takes bare %s registers", bank)
|
||||||
|
}
|
||||||
|
if src.lasx != lasx || dst.lasx != lasx {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
|
||||||
|
}
|
||||||
|
return l64wordLE(l64rr(enc.move, src.num, dst.num)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Store: VMOVQ vd, off(rj) or VMOVQ vd, (rj)(rk).
|
||||||
|
if srcVec && dstMem {
|
||||||
|
if src.hasSuf || src.lasx != lasx {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: expected a bare %s0-%s31 register as the stored value", bank, bank)
|
||||||
|
}
|
||||||
|
if dstIdx {
|
||||||
|
rj, rk := loong64RegNum(ops[1].Addr.Base), loong64RegNum(ops[1].Addr.Index)
|
||||||
|
if rj < 0 || rk < 0 {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: invalid register operand")
|
||||||
|
}
|
||||||
|
return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil
|
||||||
|
}
|
||||||
|
rj, off := l64MemWithFrame(ops[1], fi)
|
||||||
|
if rj < 0 || off < -2048 || off > 2047 {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]")
|
||||||
|
}
|
||||||
|
return l64wordLE(l64irr(enc.st, int(off), rj, src.num)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Load: VMOVQ off(rj), vd, the indexed VMOVQ (rj)(rk), vd, and the
|
||||||
|
// load-and-replicate form VMOVQ off(rj), vd.T.
|
||||||
|
if srcMem && dstVec {
|
||||||
|
if dst.lasx != lasx {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
|
||||||
|
}
|
||||||
|
if srcIdx {
|
||||||
|
if dst.hasSuf {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: an indexed load takes a bare %s register", bank)
|
||||||
|
}
|
||||||
|
rj, rk := loong64RegNum(ops[0].Addr.Base), loong64RegNum(ops[0].Addr.Index)
|
||||||
|
if rj < 0 || rk < 0 {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: invalid register operand")
|
||||||
|
}
|
||||||
|
return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil
|
||||||
|
}
|
||||||
|
rj, off := l64MemWithFrame(ops[0], fi)
|
||||||
|
if rj < 0 || off < -2048 || off > 2047 {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
|
||||||
|
}
|
||||||
|
op := enc.ld
|
||||||
|
if dst.hasSuf {
|
||||||
|
w, ok := l64VecSuffixWidth(lasx, dst)
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw)
|
||||||
|
}
|
||||||
|
switch w {
|
||||||
|
case 0:
|
||||||
|
op = enc.replB
|
||||||
|
case 1:
|
||||||
|
op = enc.replH
|
||||||
|
case 2:
|
||||||
|
op = enc.replW
|
||||||
|
default:
|
||||||
|
op = enc.replD
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Element extract: VMOVQ vj.T[i], rd (vpickve2gr, signed or unsigned).
|
||||||
|
if srcVec && src.hasEl && !dstVec && !dstMem {
|
||||||
|
if src.lasx != lasx {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
|
||||||
|
}
|
||||||
|
idx, ok := l64VecElementBase(lasx, src)
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[0].Raw)
|
||||||
|
}
|
||||||
|
rd, err := intReg(ops[1])
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
op := enc.pickS
|
||||||
|
if src.unsig {
|
||||||
|
op = enc.pickU
|
||||||
|
}
|
||||||
|
return l64wordLE(l64irr(op, idx, src.num, rd)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Insert and duplicate: VMOVQ rj, vd.T[i] (vinsgr2vr) and
|
||||||
|
// VMOVQ rj, vd.T (vreplgr2vr).
|
||||||
|
if !srcVec && !srcMem && dstVec && dst.hasSuf {
|
||||||
|
if dst.lasx != lasx {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
|
||||||
|
}
|
||||||
|
rs, err := intReg(ops[0])
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if dst.hasEl {
|
||||||
|
idx, ok := l64VecElementBase(lasx, dst)
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[1].Raw)
|
||||||
|
}
|
||||||
|
return l64wordLE(l64irr(enc.ins, idx, rs, dst.num)), nil
|
||||||
|
}
|
||||||
|
w, ok := l64VecSuffixWidth(lasx, dst)
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("VMOVQ: invalid width suffix %q", ops[1].Raw)
|
||||||
|
}
|
||||||
|
return l64wordLE(l64irr(enc.dup, w, rs, dst.num)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
return nil, fmt.Errorf("VMOVQ: unsupported operand combination %q, %q", ops[0].Raw, ops[1].Raw)
|
||||||
|
}
|
||||||
|
|||||||
+174
-7
@@ -30,7 +30,10 @@ package asm
|
|||||||
// of the immediate and register fields), mirroring the toolchain's OP_*
|
// of the immediate and register fields), mirroring the toolchain's OP_*
|
||||||
// helpers, so each l64* function only ORs its fields in.
|
// helpers, so each l64* function only ORs its fields in.
|
||||||
|
|
||||||
import "maps"
|
import (
|
||||||
|
"maps"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
// loong64RegNum returns the 5-bit register number for a LoongArch register
|
// loong64RegNum returns the 5-bit register number for a LoongArch register
|
||||||
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
|
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
|
||||||
@@ -103,7 +106,12 @@ func loong64RegNum(name string) int {
|
|||||||
case "R31", "S8":
|
case "R31", "S8":
|
||||||
return 31
|
return 31
|
||||||
}
|
}
|
||||||
// F0-F31, FCC0-FCC7, FCSR0-FCSR31.
|
// F0-F31, FCC0-FCC7, FCSR0-FCSR31. The LSX/LASX vector banks (V0-V31,
|
||||||
|
// X0-X31) are deliberately NOT accepted here: they are a separate
|
||||||
|
// register class, and the toolchain rejects V/X names wherever an
|
||||||
|
// integer or FP register is expected (GOARCH=loong64 go tool asm reports
|
||||||
|
// "unrecognized instruction" for `BEQZ X0`). Vector operands are
|
||||||
|
// resolved only through loong64VecRegNum.
|
||||||
if len(name) >= 4 && name[:4] == "FCSR" {
|
if len(name) >= 4 && name[:4] == "FCSR" {
|
||||||
return loong64RegSpecial(name[4:], 31)
|
return loong64RegSpecial(name[4:], 31)
|
||||||
}
|
}
|
||||||
@@ -148,6 +156,19 @@ func loong64RegSpecial(digits string, max int) int {
|
|||||||
return -1
|
return -1
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// loong64VecRegNum resolves an LSX/LASX vector register name (V0-V31 or
|
||||||
|
// X0-X31) to its 5-bit number, or -1. The vector banks are a register class
|
||||||
|
// of their own: the toolchain accepts them only in the vector operands of the
|
||||||
|
// LSX/LASX instructions (GOARCH=loong64 go tool asm assembles `VADDV V0, V1,
|
||||||
|
// V2` and `XVADDV X0, X1, X2`, and rejects `VADDV R4, R5, R6`), so the V/X
|
||||||
|
// spellings never reach the integer/FP resolver.
|
||||||
|
func loong64VecRegNum(name string) int {
|
||||||
|
if len(name) < 2 || (name[0] != 'V' && name[0] != 'X') {
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
return loong64RegSpecial(name[1:], 31)
|
||||||
|
}
|
||||||
|
|
||||||
// ---- format helpers ----
|
// ---- format helpers ----
|
||||||
|
|
||||||
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
|
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
|
||||||
@@ -199,7 +220,9 @@ func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
|
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
|
||||||
// The msb/lsb fields are 6 bits wide (0-63) and are validated by the caller.
|
// The msb/lsb fields are 6 bits wide and are inserted unmasked: the caller
|
||||||
|
// must have validated them (0..31 for the .w forms, 0..63 for the .d forms,
|
||||||
|
// lsb <= msb), the same rule the toolchain enforces as "illegal bit number".
|
||||||
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
|
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
|
||||||
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
|
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
|
||||||
}
|
}
|
||||||
@@ -245,7 +268,7 @@ const (
|
|||||||
l64Firr14 // 2RI14 (ldptr/stptr)
|
l64Firr14 // 2RI14 (ldptr/stptr)
|
||||||
l64Firr16 // 2RI16 (addu16i.d)
|
l64Firr16 // 2RI16 (addu16i.d)
|
||||||
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
|
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
|
||||||
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub)
|
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub, fsel)
|
||||||
l64Firir // bstrins/bstrpick
|
l64Firir // bstrins/bstrpick
|
||||||
l64Firrr // alsl
|
l64Firrr // alsl
|
||||||
l64Fi15 // syscall/break/dbar
|
l64Fi15 // syscall/break/dbar
|
||||||
@@ -253,6 +276,8 @@ const (
|
|||||||
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
|
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
|
||||||
l64Fshift // 2RI12 with a 5/6-bit shift immediate
|
l64Fshift // 2RI12 with a 5/6-bit shift immediate
|
||||||
l64Fpreld // preld (2RI12 + 5-bit hint)
|
l64Fpreld // preld (2RI12 + 5-bit hint)
|
||||||
|
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
|
||||||
|
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
|
||||||
)
|
)
|
||||||
|
|
||||||
// l64Enc is one instruction's encoding: its bit layout (format) and the
|
// l64Enc is one instruction's encoding: its bit layout (format) and the
|
||||||
@@ -275,10 +300,68 @@ type l64DualEnc struct {
|
|||||||
var l64DualTable = map[string]l64DualEnc{}
|
var l64DualTable = map[string]l64DualEnc{}
|
||||||
|
|
||||||
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
|
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
|
||||||
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered
|
// to their encoding.
|
||||||
// yet; the base integer, memory and floating-point ISA is complete.
|
|
||||||
var l64InstrTable = map[string]l64Enc{}
|
var l64InstrTable = map[string]l64Enc{}
|
||||||
|
|
||||||
|
// l64Vec3Enc pairs a vector opcode with its register bank: false = LSX
|
||||||
|
// (V0-V31), true = LASX (X0-X31). The toolchain accepts one bank per
|
||||||
|
// spelling: GOARCH=loong64 go tool asm assembles `VADDV V1, V2, V3` and
|
||||||
|
// `XVADDV X1, X2, X3`, and rejects the crossed spellings.
|
||||||
|
type l64Vec3Enc struct {
|
||||||
|
op uint32
|
||||||
|
lasx bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64VecImmEnc carries the immediate-form encoding of a vector mnemonic:
|
||||||
|
// the opcode, the bank, the accepted immediate range, the bias the toolchain
|
||||||
|
// adds (vsrai.b encodes imm+8) and the mask of the encoded field (vseqi.b
|
||||||
|
// keeps a 5-bit two's-complement value, vseqi.d a 7-bit one).
|
||||||
|
type l64VecImmEnc struct {
|
||||||
|
op uint32
|
||||||
|
lasx bool
|
||||||
|
min, max int
|
||||||
|
bias int
|
||||||
|
mask int
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64VecBank marks the LSX/LASX mnemonics and records which register bank
|
||||||
|
// each accepts; presence in the map routes the mnemonic through the vector
|
||||||
|
// dispatcher rather than the integer/FP formats.
|
||||||
|
var l64VecBank = map[string]bool{}
|
||||||
|
|
||||||
|
// l64VecImmInfo mirrors l64VecImmTable for the dispatcher.
|
||||||
|
var l64VecImmInfo = map[string]l64VecImmEnc{}
|
||||||
|
|
||||||
|
// l64Vec2R marks the two-operand vector mnemonics (INSTR vj, vd, such as
|
||||||
|
// vpcnt.v).
|
||||||
|
var l64Vec2R = map[string]bool{}
|
||||||
|
|
||||||
|
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit
|
||||||
|
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels.
|
||||||
|
type l64VmovqEnc struct {
|
||||||
|
ld, st, ldx, stx uint32 // plain and indexed load/store
|
||||||
|
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
|
||||||
|
pickS, pickU uint32 // vpickve2gr.{,u} element extract
|
||||||
|
ins uint32 // vinsgr2vr element insert
|
||||||
|
dup uint32 // vreplgr2vr duplicate (width in [11:10])
|
||||||
|
move uint32 // vori.b/xvori.b $0 register move
|
||||||
|
}
|
||||||
|
|
||||||
|
var l64VmovqTable = map[bool]l64VmovqEnc{
|
||||||
|
false: { // VMOVQ, the LSX (V) bank
|
||||||
|
ld: 0x5800 << 15, st: 0x5880 << 15, ldx: 0x7080 << 15, stx: 0x7088 << 15,
|
||||||
|
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
|
||||||
|
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
|
||||||
|
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
|
||||||
|
},
|
||||||
|
true: { // XVMOVQ, the LASX (X) bank
|
||||||
|
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
|
||||||
|
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
|
||||||
|
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
|
||||||
|
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
func init() {
|
func init() {
|
||||||
// 3R, integer.
|
// 3R, integer.
|
||||||
rrr := map[string]uint32{
|
rrr := map[string]uint32{
|
||||||
@@ -358,6 +441,10 @@ func init() {
|
|||||||
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
|
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
|
||||||
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
|
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
|
||||||
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
|
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
|
||||||
|
// LSX: convert a 64-bit integer lane to a double float. The operand
|
||||||
|
// bank is the FP registers (the toolchain spells it `FFINTDV F0, F1`),
|
||||||
|
// so the entry stays on the 2R integer/FP format.
|
||||||
|
"FFINTDV": 0x474a << 10,
|
||||||
}
|
}
|
||||||
for m, op := range rr {
|
for m, op := range rr {
|
||||||
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
|
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
|
||||||
@@ -414,12 +501,14 @@ func init() {
|
|||||||
// LUI is the Plan 9 spelling of lu12i.w.
|
// LUI is the Plan 9 spelling of lu12i.w.
|
||||||
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
||||||
|
|
||||||
// 4R, fused multiply-add.
|
// 4R, fused multiply-add, and FSEL (fsel.d: the first operand is a FCC
|
||||||
|
// condition flag, the layout matches the 4R shape).
|
||||||
rrrr := map[string]uint32{
|
rrrr := map[string]uint32{
|
||||||
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
|
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
|
||||||
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
|
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
|
||||||
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
|
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
|
||||||
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
|
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
|
||||||
|
"FSEL": 0x340 << 18,
|
||||||
}
|
}
|
||||||
for m, op := range rrrr {
|
for m, op := range rrrr {
|
||||||
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
|
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
|
||||||
@@ -453,6 +542,10 @@ func init() {
|
|||||||
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
|
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
|
||||||
|
|
||||||
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
|
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
|
||||||
|
// The toolchain's form is three operands, `AMADDW rk, (rj), rd`
|
||||||
|
// (cmd/asm/internal/asm/testdata/loong64enc1.s and
|
||||||
|
// internal/runtime/atomic/atomic_loong64.s); the two-register spelling
|
||||||
|
// is rejected by the oracle.
|
||||||
am := map[string]uint32{
|
am := map[string]uint32{
|
||||||
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
|
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
|
||||||
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
|
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
|
||||||
@@ -470,10 +563,84 @@ func init() {
|
|||||||
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
|
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
|
||||||
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
|
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
|
||||||
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
|
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
|
||||||
|
// The _dbar (acquire/release) add, and, or variants: opcodes read off
|
||||||
|
// `go tool objdump` of `AMADDDBW R14, (R13), R12` and friends.
|
||||||
|
"AMADDDBW": 0x070D4 << 15, "AMADDDBV": 0x070D5 << 15,
|
||||||
|
"AMANDDBW": 0x070D6 << 15, "AMANDDBV": 0x070D7 << 15,
|
||||||
|
"AMORDBW": 0x070D8 << 15, "AMORDBV": 0x070D9 << 15,
|
||||||
}
|
}
|
||||||
for m, op := range am {
|
for m, op := range am {
|
||||||
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
|
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- LSX/LASX (V*/XV*) ----
|
||||||
|
// Every opcode below was read off `go tool objdump` of a GOARCH=loong64
|
||||||
|
// `go tool asm` kernel (the toolchain's own loong64enc1.s cross-checks
|
||||||
|
// most of them), not assumed from the LoongArch manual.
|
||||||
|
|
||||||
|
// Three vector registers: INSTR vk, vj, vd (or INSTR vk, vd with
|
||||||
|
// vj = vd). l64Vec3Enc.lasx selects the register bank the toolchain
|
||||||
|
// accepts: LSX spellings take V0-V31, LASX spellings X0-X31.
|
||||||
|
vec3 := map[string]l64Vec3Enc{
|
||||||
|
"VADDW": {0xE016 << 15, false}, "VADDV": {0xE017 << 15, false},
|
||||||
|
"VANDV": {0xE24C << 15, false}, "VXORV": {0xE24E << 15, false},
|
||||||
|
"VSEQB": {0xE000 << 15, false}, "VSEQV": {0xE003 << 15, false},
|
||||||
|
"VSRAB": {0xE1D8 << 15, false}, "VROTRW": {0xE1DE << 15, false},
|
||||||
|
"XVADDV": {0xE817 << 15, true},
|
||||||
|
"XVANDV": {0xEA4C << 15, true}, "XVXORV": {0xEA4E << 15, true},
|
||||||
|
"XVSEQB": {0xE800 << 15, true}, "XVSEQV": {0xE803 << 15, true},
|
||||||
|
}
|
||||||
|
for m, e := range vec3 {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Fvvv, op: e.op}
|
||||||
|
l64VecBank[m] = e.lasx
|
||||||
|
}
|
||||||
|
|
||||||
|
// Immediate forms: INSTR $imm, vj, vd (or INSTR $imm, vd). The immediate
|
||||||
|
// range, bias and field mask are the ones the toolchain encodes: vandi.b
|
||||||
|
// stores the raw 8-bit constant, vsrai.b stores imm+8 (byte-lane bias),
|
||||||
|
// vseqi.b and vseqi.d store 5-bit and 7-bit two's-complement values.
|
||||||
|
// The mnemonics that also have a register form (VSEQB, VSEQV, VSRAB,
|
||||||
|
// VROTRW) keep their three-register entry in l64InstrTable; the
|
||||||
|
// dispatcher picks the immediate opcode from l64VecImmInfo by operand
|
||||||
|
// kind, so the immediate entries must not overwrite the table.
|
||||||
|
vecImm := map[string]l64VecImmEnc{
|
||||||
|
"VANDB": {0xE7A0 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVANDB": {0xEFA0 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
"VSEQV": {0xE503 << 15, false, -64, 63, 0, 0x7F},
|
||||||
|
"XVSEQV": {0xE903 << 15, true, -64, 63, 0, 0x7F},
|
||||||
|
"VSRAB": {0xE668 << 15, false, 0, 7, 8, 0x1F},
|
||||||
|
"VROTRW": {0xE541 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
}
|
||||||
|
for m, e := range vecImm {
|
||||||
|
l64VecImmInfo[m] = e
|
||||||
|
l64VecBank[m] = e.lasx
|
||||||
|
}
|
||||||
|
|
||||||
|
// Vector-to-condition flag: INSTR vj, FCCn (vsetnez.v, vsetanyeqz.*,
|
||||||
|
// vsetallnez.*): the sub-op rides in the rk field.
|
||||||
|
vecCf := map[string]uint32{
|
||||||
|
"VSETNEV": 0xE539<<15 | 7<<10, "XVSETNEV": 0xED39<<15 | 7<<10,
|
||||||
|
"VSETANYEQB": 0xE539<<15 | 8<<10, "XVSETANYEQB": 0xED39<<15 | 8<<10,
|
||||||
|
"VSETANYEQV": 0xE539<<15 | 11<<10, "XVSETANYEQV": 0xED39<<15 | 11<<10,
|
||||||
|
"VSETALLNEV": 0xE539<<15 | 15<<10, "XVSETALLNEV": 0xED39<<15 | 15<<10,
|
||||||
|
}
|
||||||
|
for m, op := range vecCf {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Fvcf, op: op}
|
||||||
|
l64VecBank[m] = strings.HasPrefix(m, "XV")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Lane popcount: INSTR vj, vd (the 2R layout with the opcode extending
|
||||||
|
// over the unused vk field).
|
||||||
|
vec2r := map[string]l64Vec3Enc{
|
||||||
|
"VPCNTV": {0x1CA70B << 10, false}, "XVPCNTV": {0x1DA70B << 10, true},
|
||||||
|
}
|
||||||
|
for m, e := range vec2r {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Frr, op: e.op}
|
||||||
|
l64VecBank[m] = e.lasx
|
||||||
|
l64Vec2R[m] = true
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
|
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
|
||||||
|
|||||||
@@ -263,6 +263,20 @@ func TestLOONG64_regNames(t *testing.T) {
|
|||||||
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
|
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// The X/V spellings name the LSX/LASX vector banks, a register class of
|
||||||
|
// their own: the oracle (GOARCH=loong64 go tool asm) rejects `BEQZ X0`
|
||||||
|
// with "unrecognized instruction" while assembling `VADDV V0, V1, V2`
|
||||||
|
// and `XVADDV X0, X1, X2`, so loong64RegNum stays strict and the vector
|
||||||
|
// operands resolve through loong64VecRegNum only.
|
||||||
|
vecCases := map[string]int{
|
||||||
|
"V0": 0, "V31": 31, "X0": 0, "X31": 31,
|
||||||
|
"R4": -1, "F0": -1, "FCC0": -1, "V32": -1, "X32": -1, "V": -1, "X": -1,
|
||||||
|
}
|
||||||
|
for name, want := range vecCases {
|
||||||
|
if got := loong64VecRegNum(name); got != want {
|
||||||
|
t.Errorf("loong64VecRegNum(%q) = %d, want %d", name, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
|
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
|
||||||
@@ -291,3 +305,287 @@ done:
|
|||||||
t.Errorf("code = % x\nwant % x", code, want)
|
t.Errorf("code = % x\nwant % x", code, want)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestLOONG64IndirectBranch pins the indirect branch encodings: JMP (Rj) and
|
||||||
|
// JAL (Rj) lower to jirl, and the raw JIRL spelling encodes the written
|
||||||
|
// offset (the Go loong64 assembler deletes raw JIRL instructions entirely,
|
||||||
|
// so this form is a gasm-only superset with faithful semantics).
|
||||||
|
func TestLOONG64IndirectBranch(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·f(SB), NOSPLIT, $0-0
|
||||||
|
JMP (R4)
|
||||||
|
JIRL R0, R4, 8
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x4C000080, // jirl r0, r4, 0
|
||||||
|
0x4C002080, // jirl r0, r4, 8
|
||||||
|
0x4C000020, // jirl r0, r1, 0 (RET)
|
||||||
|
)
|
||||||
|
|
||||||
|
// JAL (R5) links, so the toolchain gives the function its autosize-8
|
||||||
|
// prologue and epilogue around the call and the closing RET.
|
||||||
|
fn = firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·f(SB), NOSPLIT, $0-0
|
||||||
|
JAL (R5)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code = assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x29FFE061, // st.d r1, -8(r3) (prologue saves RA below the new SP)
|
||||||
|
0x02FFE063, // addi.d r3, r3, -8 (prologue opens the frame)
|
||||||
|
0x29C00061, // st.d r1, 0(r3) (prologue saves RA at SP)
|
||||||
|
0x4C0000A1, // jirl r1, r5, 0
|
||||||
|
0x28C00061, // ld.d r1, 0(r3) (epilogue restores RA)
|
||||||
|
0x02C02063, // addi.d r3, r3, 8
|
||||||
|
0x4C000020, // jirl r0, r1, 0 (RET)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_vector pins the LSX/LASX slice against words read off
|
||||||
|
// GOARCH=loong64 go tool asm (cross-checked against the toolchain's own
|
||||||
|
// loong64enc1.s): the three-register forms, the immediate forms with their
|
||||||
|
// biases, the vector-to-condition forms, lane popcount, the FP conversion,
|
||||||
|
// FSEL and the VMOVQ move family.
|
||||||
|
func TestLOONG64_vector(t *testing.T) {
|
||||||
|
t.Run("three-register and immediate forms", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VADDV V1, V2, V3
|
||||||
|
VADDW V1, V2, V3
|
||||||
|
VADDV V2, V1
|
||||||
|
VANDV V1, V2
|
||||||
|
VXORV V1, V2, V3
|
||||||
|
VSEQB V1, V2, V3
|
||||||
|
VSEQV V1, V2, V3
|
||||||
|
VSRAB V1, V2, V3
|
||||||
|
VROTRW V1, V2, V3
|
||||||
|
VANDB $0, V2, V3
|
||||||
|
VANDB $255, V2
|
||||||
|
VSEQB $3, V2, V3
|
||||||
|
VSEQV $15, V2, V3
|
||||||
|
VSEQV $-15, V2, V3
|
||||||
|
VSRAB $7, V1, V2
|
||||||
|
VROTRW $16, V1, V2
|
||||||
|
VPCNTV V1, V2
|
||||||
|
XVADDV X1, X2, X3
|
||||||
|
XVXORV X1, X2, X3
|
||||||
|
XVSEQB X1, X2, X3
|
||||||
|
XVPCNTV X1, X2
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x700B8443, // vadd.v v3, v2, v1
|
||||||
|
0x700B0443, // vadd.w
|
||||||
|
0x700B8821, // vadd.v v1, v1, v2 (two-operand form)
|
||||||
|
0x71260442, // vand.v v2, v2, v1
|
||||||
|
0x71270443, // vxor.v
|
||||||
|
0x70000443, // vseq.b
|
||||||
|
0x70018443, // vseq.d
|
||||||
|
0x70EC0443, // vsra.b
|
||||||
|
0x70EF0443, // vrotr.w
|
||||||
|
0x73D00043, // vandi.b v3, v2, 0
|
||||||
|
0x73D3FC42, // vandi.b v2, v2, 255 (two-operand form)
|
||||||
|
0x72800C43, // vseqi.b v3, v2, 3
|
||||||
|
0x7281BC43, // vseqi.d v3, v2, 15
|
||||||
|
0x7281C443, // vseqi.d v3, v2, -15 (7-bit two's complement)
|
||||||
|
0x73343C22, // vsrai.b v2, v1, 7 (encoded as 7+8)
|
||||||
|
0x72A0C022, // vrotri.w v2, v1, 16
|
||||||
|
0x729C2C22, // vpcnt.d v2, v1
|
||||||
|
0x740B8443, // xvadd.d x3, x2, x1
|
||||||
|
0x75270443, // xvxor.d
|
||||||
|
0x74000443, // xvseq.b
|
||||||
|
0x769C2C22, // xvpcnt.d x2, x1
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("vector-to-condition", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VSETNEV V1, FCC0
|
||||||
|
VSETANYEQB V1, FCC0
|
||||||
|
VSETANYEQV V2, FCC0
|
||||||
|
VSETALLNEV V0, FCC0
|
||||||
|
XVSETNEV X1, FCC0
|
||||||
|
XVSETALLNEV X1, FCC0
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x729C9C20, // vsetnez.d fcc0, v1
|
||||||
|
0x729CA020, // vsetanyeqz.b
|
||||||
|
0x729CAC40, // vsetanyeqz.d
|
||||||
|
0x729CBC00, // vsetallnez.d
|
||||||
|
0x769C9C20, // xvsetnez.d
|
||||||
|
0x769CBC20, // xvsetallnez.d
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("FP convert and FSEL", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
FFINTDV F0, F1
|
||||||
|
FSEL FCC0, F3, F4, F3
|
||||||
|
FSEL FCC1, F1, F2
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x011D2801, // ffint.d.v f1, f0
|
||||||
|
0x0D000C83, // fsel f3, f4, f3, fcc0
|
||||||
|
0x0D008442, // fsel f2, f2, f1, fcc1
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("VMOVQ move family", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VMOVQ V1, V9
|
||||||
|
VMOVQ (R4), V2
|
||||||
|
VMOVQ 16(R4), V2
|
||||||
|
VMOVQ V0, (R4)
|
||||||
|
VMOVQ V0, 32(R4)
|
||||||
|
VMOVQ (R4)(R7), V3
|
||||||
|
VMOVQ V3, (R4)(R7)
|
||||||
|
VMOVQ R6, V0.B16
|
||||||
|
VMOVQ R6, V12.W4
|
||||||
|
VMOVQ (R4), V4.W4
|
||||||
|
XVMOVQ X3, X7
|
||||||
|
XVMOVQ (R4), X2
|
||||||
|
XVMOVQ X0, (R4)
|
||||||
|
XVMOVQ (R4)(R7), X4
|
||||||
|
XVMOVQ X0, (R4)(R7)
|
||||||
|
XVMOVQ R6, X0.B32
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x732D0029, // vori.b v9, v1, 0 (register move)
|
||||||
|
0x2C000082, // vld v2, r4, 0
|
||||||
|
0x2C004082, // vld v2, r4, 16
|
||||||
|
0x2C400080, // vst v0, r4, 0
|
||||||
|
0x2C408080, // vst v0, r4, 32
|
||||||
|
0x38401C83, // vldx v3, r4, r7
|
||||||
|
0x38441C83, // vstx v3, r4, r7
|
||||||
|
0x729F00C0, // vreplgr2vr.b v0, r6
|
||||||
|
0x729F08CC, // vreplgr2vr.w v12, r6
|
||||||
|
0x30200084, // vldrepl.w v4, r4, 0
|
||||||
|
0x772D0067, // xvori.b x7, x3, 0
|
||||||
|
0x2C800082, // xvld x2, r4, 0
|
||||||
|
0x2CC00080, // xvst x0, r4, 0
|
||||||
|
0x38481C84, // xvldx x4, r4, r7
|
||||||
|
0x384C1C80, // xvstx x0, r4, r7
|
||||||
|
0x769F00C0, // xvreplgr2vr.b x0, r6
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("element extract and insert", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VMOVQ V0.V[0], R10
|
||||||
|
VMOVQ V6.V[1], R8
|
||||||
|
VMOVQ R9, V1.V[0]
|
||||||
|
XVMOVQ X0.V[0], R10
|
||||||
|
XVMOVQ X5.W[7], R7
|
||||||
|
XVMOVQ R4, X7.V[3]
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x72EFF00A, // vpickve2gr.d r10, v0, 0
|
||||||
|
0x72EFF4C8, // vpickve2gr.d r8, v6, 1
|
||||||
|
0x72EBF121, // vinsgr2vr.d v1, r9, 0
|
||||||
|
0x76EFE00A, // xvpickve2gr.d r10, x0, 0
|
||||||
|
0x76EFDCA7, // xvpickve2gr.w r7, x5, 7
|
||||||
|
0x76EBEC87, // xvinsgr2vr.d x7, r4, 3
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_vectorErrors pins the register-class and range diagnostics of
|
||||||
|
// the vector slice; each shape is rejected by the oracle as well
|
||||||
|
// (GOARCH=loong64 go tool asm).
|
||||||
|
func TestLOONG64_vectorErrors(t *testing.T) {
|
||||||
|
cases := []string{
|
||||||
|
// Integer registers in vector positions.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VADDV R4, R5, R6
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// Crossed banks: LSX spellings take V, LASX spellings X.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VADDV X1, X2, X3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
XVADDV V1, V2, V3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// The LASX bank has no .b/.h element forms.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
XVMOVQ R4, X2.B[0]
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// Immediate ranges.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VANDB $256, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSEQB $16, V2, V3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VROTRW $32, V1, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// VSET* wants an FCC flag, not a vector register.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSETNEV V1, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
}
|
||||||
|
for i, src := range cases {
|
||||||
|
fn := firstTextLOONG64(t, src)
|
||||||
|
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||||
|
t.Errorf("case %d: expected an error, got none", i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_dbarAtomics pins the _dbar (acquire/release) AMO variants.
|
||||||
|
// The oracle words come from GOARCH=loong64 go tool objdump of kernels
|
||||||
|
// assembled with go tool asm, and match the toolchain's loong64enc1.s.
|
||||||
|
func TestLOONG64_dbarAtomics(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·atoms(SB), NOSPLIT, $0
|
||||||
|
AMADDDBW R14, (R13), R12
|
||||||
|
AMADDDBV R14, (R13), R12
|
||||||
|
AMANDDBW R5, (R4), R6
|
||||||
|
AMANDDBV R5, (R4), R6
|
||||||
|
AMORDBW R5, (R4), R0
|
||||||
|
AMORDBV R5, (R4), R6
|
||||||
|
AMSWAPDBW R5, (R4), R6
|
||||||
|
AMCASDBV R6, (R4), R5
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x386A39AC, // amadd_db.w r12, r13, r14
|
||||||
|
0x386AB9AC, // amadd_db.d
|
||||||
|
0x386B1486, // amand_db.w r6, r4, r5
|
||||||
|
0x386B9486, // amand_db.d
|
||||||
|
0x386C1480, // amor_db.w r0, r4, r5
|
||||||
|
0x386C9486, // amor_db.d
|
||||||
|
0x38691486, // amswap_db.w
|
||||||
|
0x385B9885, // amcas_db.w
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|||||||
@@ -232,7 +232,10 @@ DATA ·table+0(SB)/8, $42
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
|
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
|
||||||
// invalid register operands and operand-count mismatches.
|
// invalid register operands and operand-count mismatches. The X0 and
|
||||||
|
// AMADDW cases follow the oracle: GOARCH=loong64 go tool asm rejects
|
||||||
|
// `BEQZ X0` (the X bank is not an integer register) and the two-register
|
||||||
|
// `AMADDW R4, R5` (the AM* family is strictly `val, (addr), result`).
|
||||||
func TestLOONG64_errors(t *testing.T) {
|
func TestLOONG64_errors(t *testing.T) {
|
||||||
cases := []string{
|
cases := []string{
|
||||||
`TEXT ·e(SB), NOSPLIT, $0
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
@@ -280,7 +283,7 @@ done:
|
|||||||
// TestLOONG64_pcsp checks the stack-adjustment table of a framed function:
|
// TestLOONG64_pcsp checks the stack-adjustment table of a framed function:
|
||||||
// the prologue raises the SP delta by autosize (in effect from the third
|
// the prologue raises the SP delta by autosize (in effect from the third
|
||||||
// instruction) and the RET's epilogue restores it to zero, with the pc deltas
|
// instruction) and the RET's epilogue restores it to zero, with the pc deltas
|
||||||
// in MinLC (4) units — byte-identical to `go tool asm`.
|
// in MinLC (4) units; byte-identical to `go tool asm`.
|
||||||
func TestLOONG64_pcsp(t *testing.T) {
|
func TestLOONG64_pcsp(t *testing.T) {
|
||||||
cases := []struct {
|
cases := []struct {
|
||||||
name string
|
name string
|
||||||
@@ -347,21 +350,94 @@ TEXT ·sb(SB), NOSPLIT, $0
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestLOONG64_movImmToFp checks the immediate-to-FP move forms.
|
// TestLOONG64_movImmToFp checks the immediate-to-FP move: MOVW $c, Fd is the
|
||||||
|
// only spelling the toolchain accepts, expanding to ori (or addi.w for the
|
||||||
|
// negative span) into R30 plus movgr2fr.w. The pinned words are the
|
||||||
|
// toolchain's own bytes; the other widths and out-of-range constants are
|
||||||
|
// illegal combinations there and are diagnosed here.
|
||||||
func TestLOONG64_movImmToFp(t *testing.T) {
|
func TestLOONG64_movImmToFp(t *testing.T) {
|
||||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
TEXT ·fpmov(SB), NOSPLIT, $0
|
TEXT ·fpmov(SB), NOSPLIT, $0
|
||||||
MOVV $0x1, F0
|
MOVW $0x1, F0
|
||||||
MOVW $0x2, F4
|
MOVW $0x2, F4
|
||||||
|
MOVW $-1, F4
|
||||||
RET
|
RET
|
||||||
`)
|
`)
|
||||||
code := assembleLOONG64Helper(t, fn)
|
code := assembleLOONG64Helper(t, fn)
|
||||||
want := []byte{
|
want := []byte{
|
||||||
0x00, 0x04, 0x80, 0x03, // ori f0, r0, 1
|
0x1e, 0x04, 0x80, 0x03, // ori r30, r0, 1
|
||||||
0x04, 0x08, 0x80, 0x03, // ori f4, r0, 2
|
0xc0, 0xa7, 0x14, 0x01, // movgr2fr.w f0, r30
|
||||||
|
0x1e, 0x08, 0x80, 0x03, // ori r30, r0, 2
|
||||||
|
0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30
|
||||||
|
0x1e, 0xfc, 0xbf, 0x02, // addi.w r30, r0, -1
|
||||||
|
0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30
|
||||||
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
|
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
|
||||||
}
|
}
|
||||||
if !bytes.Equal(code, want) {
|
if !bytes.Equal(code, want) {
|
||||||
t.Errorf("code = % x\nwant % x", code, want)
|
t.Errorf("code = % x\nwant % x", code, want)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_movImmToFpErrors checks the immediate-to-FP diagnostics: the
|
||||||
|
// widths the toolchain rejects as illegal combinations, and constants beyond
|
||||||
|
// the 12-bit ori/addi.w span (the toolchain never materialises a wider
|
||||||
|
// constant on this path).
|
||||||
|
func TestLOONG64_movImmToFpErrors(t *testing.T) {
|
||||||
|
cases := []string{
|
||||||
|
"MOVV $1, F0",
|
||||||
|
"MOVF $2, F4",
|
||||||
|
"MOVD $2, F4",
|
||||||
|
"MOVW $100000, F1",
|
||||||
|
"MOVW $-2049, F1",
|
||||||
|
"MOVW $4096, F1",
|
||||||
|
}
|
||||||
|
for _, src := range cases {
|
||||||
|
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
|
||||||
|
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", src)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_branch16Unsigned pins the unsigned two-operand branches: with
|
||||||
|
// one register BLTU/BGEU keep the register-register form against R0 (never
|
||||||
|
// taken), the toolchain's encoding, where a beqz would test the wrong
|
||||||
|
// condition; the three-operand forms are unchanged.
|
||||||
|
func TestLOONG64_branch16Unsigned(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·u(SB), NOSPLIT, $0
|
||||||
|
BLTU R4, done
|
||||||
|
BGEU R5, done
|
||||||
|
BLTU R6, R7, done
|
||||||
|
BGEU R8, R9, done
|
||||||
|
done:
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x68001080, // bltu r4, r0, +4
|
||||||
|
0x6C000CA0, // bgeu r5, r0, +3
|
||||||
|
0x680008C7, // bltu r6, r7, +2
|
||||||
|
0x6C000509, // bgeu r8, r9, +1
|
||||||
|
0x4C000020, // jirl r0, r1, 0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_bitFieldRange checks the BSTRINS/BSTRPICK bit-number
|
||||||
|
// validation, mirroring the toolchain's "illegal bit number" rule: 0..31 for
|
||||||
|
// the .w forms, 0..63 for the .d forms, and lsb <= msb.
|
||||||
|
func TestLOONG64_bitFieldRange(t *testing.T) {
|
||||||
|
cases := []string{
|
||||||
|
"BSTRINSW $32, R4, $0, R5",
|
||||||
|
"BSTRPICKW $31, R4, $32, R5",
|
||||||
|
"BSTRINSV $64, R4, $0, R5",
|
||||||
|
"BSTRPICKV $3, R4, $4, R5",
|
||||||
|
"BSTRINSW $-1, R4, $0, R5",
|
||||||
|
}
|
||||||
|
for _, src := range cases {
|
||||||
|
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
|
||||||
|
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", src)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+6
-1
@@ -17,12 +17,13 @@ import "strings"
|
|||||||
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
|
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
|
||||||
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
|
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
|
||||||
// those indices but require one. The mask flag marks the AVX-512 opmask
|
// those indices but require one. The mask flag marks the AVX-512 opmask
|
||||||
// registers K0-K7.
|
// registers K0-K7, the fp flag the x87 stack registers F0-F7.
|
||||||
type Reg struct {
|
type Reg struct {
|
||||||
idx int
|
idx int
|
||||||
size int // informational width implied by the name; the mnemonic decides
|
size int // informational width implied by the name; the mnemonic decides
|
||||||
high bool // AH/CH/DH/BH
|
high bool // AH/CH/DH/BH
|
||||||
mask bool // K0-K7 opmask register
|
mask bool // K0-K7 opmask register
|
||||||
|
fp bool // F0-F7 x87 stack register
|
||||||
}
|
}
|
||||||
|
|
||||||
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
|
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
|
||||||
@@ -144,6 +145,10 @@ func buildRegByName() map[string]Reg {
|
|||||||
for i := 0; i <= 7; i++ {
|
for i := 0; i <= 7; i++ {
|
||||||
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
|
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
|
||||||
}
|
}
|
||||||
|
// x87 stack: F0..F7.
|
||||||
|
for i := 0; i <= 7; i++ {
|
||||||
|
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
|
||||||
|
}
|
||||||
return m
|
return m
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+879
-30
File diff suppressed because it is too large
Load Diff
+134
-32
@@ -65,7 +65,7 @@ func riscvRegNum(name string) int {
|
|||||||
return 25
|
return 25
|
||||||
case "X26", "S10":
|
case "X26", "S10":
|
||||||
return 26
|
return 26
|
||||||
case "X27", "S11":
|
case "X27", "S11", "g":
|
||||||
return 27
|
return 27
|
||||||
case "X28", "T3":
|
case "X28", "T3":
|
||||||
return 28
|
return 28
|
||||||
@@ -141,10 +141,36 @@ func riscvRegNum(name string) int {
|
|||||||
case "F31", "FT11":
|
case "F31", "FT11":
|
||||||
return 31
|
return 31
|
||||||
default:
|
default:
|
||||||
|
// Vector registers V0-V31 (the "V" extension). They share the
|
||||||
|
// register numbering with the integer file: a bare number 0-31.
|
||||||
|
if len(name) >= 2 && name[0] == 'V' {
|
||||||
|
if n, ok := parseRegDigits(name[1:], 31); ok {
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
}
|
||||||
return -1
|
return -1
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// parseRegDigits parses a decimal register suffix and reports whether it is
|
||||||
|
// within [0, max].
|
||||||
|
func parseRegDigits(digits string, max int) (int, bool) {
|
||||||
|
if digits == "" {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
n := 0
|
||||||
|
for i := 0; i < len(digits); i++ {
|
||||||
|
if digits[i] < '0' || digits[i] > '9' {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
n = n*10 + int(digits[i]-'0')
|
||||||
|
if n > max {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n, true
|
||||||
|
}
|
||||||
|
|
||||||
// RISC-V instruction encoding parameters.
|
// RISC-V instruction encoding parameters.
|
||||||
type riscvEnc struct {
|
type riscvEnc struct {
|
||||||
opcode uint32 // bits [6:0]
|
opcode uint32 // bits [6:0]
|
||||||
@@ -232,25 +258,28 @@ var riscvInstrTable = map[string]riscvEnc{
|
|||||||
"JALR": {0x67, 0x0, 0x00},
|
"JALR": {0x67, 0x0, 0x00},
|
||||||
|
|
||||||
// RV64A, atomics (AMO opcode 0x2F).
|
// RV64A, atomics (AMO opcode 0x2F).
|
||||||
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
|
// funct3: 0x2 = word, 0x3 = doubleword. The stored funct7 is the full
|
||||||
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
|
// 7-bit field: funct5 in the upper five bits and the aq/rl ordering bits in
|
||||||
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
|
// the lower two, exactly as the toolchain writes them: every AMO sets both
|
||||||
"AMOADDW": {0x2F, 0x2, 0x00 << 2},
|
// aq and rl (funct7 |= 3).
|
||||||
"AMOADDD": {0x2F, 0x3, 0x00 << 2},
|
"AMOSWAPW": {0x2F, 0x2, 0x01<<2 | 0x3},
|
||||||
"AMOANDW": {0x2F, 0x2, 0x0C << 2},
|
"AMOSWAPD": {0x2F, 0x3, 0x01<<2 | 0x3},
|
||||||
"AMOANDD": {0x2F, 0x3, 0x0C << 2},
|
"AMOADDW": {0x2F, 0x2, 0x00<<2 | 0x3},
|
||||||
"AMOORW": {0x2F, 0x2, 0x06 << 2},
|
"AMOADDD": {0x2F, 0x3, 0x00<<2 | 0x3},
|
||||||
"AMOORD": {0x2F, 0x3, 0x06 << 2},
|
"AMOANDW": {0x2F, 0x2, 0x0C<<2 | 0x3},
|
||||||
"AMOXORW": {0x2F, 0x2, 0x04 << 2},
|
"AMOANDD": {0x2F, 0x3, 0x0C<<2 | 0x3},
|
||||||
"AMOXORD": {0x2F, 0x3, 0x04 << 2},
|
"AMOORW": {0x2F, 0x2, 0x08<<2 | 0x3},
|
||||||
"AMOMAXW": {0x2F, 0x2, 0x14 << 2},
|
"AMOORD": {0x2F, 0x3, 0x08<<2 | 0x3},
|
||||||
"AMOMAXD": {0x2F, 0x3, 0x14 << 2},
|
"AMOXORW": {0x2F, 0x2, 0x04<<2 | 0x3},
|
||||||
"AMOMINW": {0x2F, 0x2, 0x10 << 2},
|
"AMOXORD": {0x2F, 0x3, 0x04<<2 | 0x3},
|
||||||
"AMOMIND": {0x2F, 0x3, 0x10 << 2},
|
"AMOMAXW": {0x2F, 0x2, 0x14<<2 | 0x3},
|
||||||
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2},
|
"AMOMAXD": {0x2F, 0x3, 0x14<<2 | 0x3},
|
||||||
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2},
|
"AMOMINW": {0x2F, 0x2, 0x10<<2 | 0x3},
|
||||||
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
|
"AMOMIND": {0x2F, 0x3, 0x10<<2 | 0x3},
|
||||||
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
|
"AMOMAXUW": {0x2F, 0x2, 0x1C<<2 | 0x3},
|
||||||
|
"AMOMAXUD": {0x2F, 0x3, 0x1C<<2 | 0x3},
|
||||||
|
"AMOMINUW": {0x2F, 0x2, 0x18<<2 | 0x3},
|
||||||
|
"AMOMINUD": {0x2F, 0x3, 0x18<<2 | 0x3},
|
||||||
|
|
||||||
// RV64F/D, floating-point arithmetic.
|
// RV64F/D, floating-point arithmetic.
|
||||||
"FADDS": {0x53, 0x0, 0x00},
|
"FADDS": {0x53, 0x0, 0x00},
|
||||||
@@ -273,12 +302,16 @@ var riscvInstrTable = map[string]riscvEnc{
|
|||||||
"FMAXS": {0x53, 0x1, 0x14},
|
"FMAXS": {0x53, 0x1, 0x14},
|
||||||
"FMIND": {0x53, 0x0, 0x15},
|
"FMIND": {0x53, 0x0, 0x15},
|
||||||
"FMAXD": {0x53, 0x1, 0x15},
|
"FMAXD": {0x53, 0x1, 0x15},
|
||||||
|
// FP sign injection (double): rs2 carries the sign source.
|
||||||
|
"FSGNJD": {0x53, 0x0, 0x11},
|
||||||
|
|
||||||
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
|
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
|
||||||
"LRW": {0x2F, 0x2, 0x02 << 2},
|
// The toolchain gives LR acquire ordering (aq = 1) and SC release
|
||||||
"LRD": {0x2F, 0x3, 0x02 << 2},
|
// ordering (rl = 1).
|
||||||
"SCW": {0x2F, 0x2, 0x03 << 2},
|
"LRW": {0x2F, 0x2, 0x02<<2 | 0x2},
|
||||||
"SCD": {0x2F, 0x3, 0x03 << 2},
|
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
|
||||||
|
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
|
||||||
|
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
|
||||||
|
|
||||||
// FP compare, result in integer register (funct7 0x50/0x51).
|
// FP compare, result in integer register (funct7 0x50/0x51).
|
||||||
"FEQS": {0x53, 0x2, 0x50},
|
"FEQS": {0x53, 0x2, 0x50},
|
||||||
@@ -296,11 +329,11 @@ func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// riscvAMOType encodes an atomic (AMO) instruction.
|
// riscvAMOType encodes an atomic (AMO) instruction.
|
||||||
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode.
|
// Layout: funct7 | rs2 | rs1 | funct3 | rd | opcode, where funct7 carries the
|
||||||
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2).
|
// funct5 in its upper five bits and the aq/rl ordering bits in the lower two
|
||||||
|
// (the table stores the full field, so the word needs no reassembly).
|
||||||
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||||
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value
|
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||||
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
|
||||||
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -328,6 +361,8 @@ var riscvCvtTable = map[string]riscvCvtEnc{
|
|||||||
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
|
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
|
||||||
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
|
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
|
||||||
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
|
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
|
||||||
|
"FCLASSS": {0x70, 0x0, 0x53}, // classify float32 → GPR mask
|
||||||
|
"FCLASSD": {0x70, 0x0, 0x53}, // classify float64 → GPR mask
|
||||||
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
|
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
|
||||||
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
|
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
|
||||||
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
|
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
|
||||||
@@ -371,7 +406,7 @@ var riscvFmaTable = map[string]riscvFmaEnc{
|
|||||||
// riscvFmaType encodes an R4-type fused multiply-add instruction.
|
// riscvFmaType encodes an R4-type fused multiply-add instruction.
|
||||||
func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 {
|
func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 {
|
||||||
return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) |
|
return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) |
|
||||||
(uint32(rs1) << 15) | (0x0 << 12) /* rm=dynamic */ | (uint32(rd) << 7) | enc.opcode
|
(uint32(rs1) << 15) | (0x0 << 12) /* rm=RNE */ | (uint32(rd) << 7) | enc.opcode
|
||||||
}
|
}
|
||||||
|
|
||||||
// CSR (Control and Status Register) instructions.
|
// CSR (Control and Status Register) instructions.
|
||||||
@@ -439,6 +474,71 @@ func riscvJType(rd int, offset int32) uint32 {
|
|||||||
0x6F // JAL opcode
|
0x6F // JAL opcode
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- RVV ("V" extension) encoding helpers ----
|
||||||
|
|
||||||
|
// The OP-V major opcode and its funct3 subclasses.
|
||||||
|
const (
|
||||||
|
riscvOpV = 0x57 // the vector operation opcode (also OPcfg for vset*)
|
||||||
|
// funct3 values: 0 OPIVV, 1 OPFVV, 2 OPMVV, 3 OPIVI, 4 OPIVX,
|
||||||
|
// 5 OPFVF, 6 OPMVX, 7 vsetvli.
|
||||||
|
riscvVf3VV = 0x0 // vector-vector
|
||||||
|
riscvVf3MV = 0x2 // vector mask
|
||||||
|
riscvVf3VI = 0x3 // vector-immediate
|
||||||
|
riscvVf3VX = 0x4 // vector-scalar
|
||||||
|
riscvVf3Cfg = 0x7 // vsetvli
|
||||||
|
)
|
||||||
|
|
||||||
|
// riscvVType composes the vsetvli/vsetivli vtype immediate: the register
|
||||||
|
// group multiplier in [2:0], the selected element width in [5:3] and the
|
||||||
|
// tail-agnostic and mask-agnostic policies in bits 6 and 7.
|
||||||
|
func riscvVType(vsew, vlmul, vta, vma int) int {
|
||||||
|
return vlmul | vsew<<3 | vta<<6 | vma<<7
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvVSetEnc encodes VSETVLI and VSETIVLI: imm[31:20] = vtype, rs1 = the
|
||||||
|
// avl register or 5-bit uimm, rd = the destination. Both carry funct3 7; a
|
||||||
|
// vsetivli is distinguished by bits [31:30] set in the immediate (the 0xC00
|
||||||
|
// the toolchain writes above its 10-bit vtype).
|
||||||
|
func riscvVSetEnc(vsetivli bool, avl, vtype, rd int) uint32 {
|
||||||
|
imm := vtype & 0x3FF
|
||||||
|
if vsetivli {
|
||||||
|
imm |= 0xC00
|
||||||
|
}
|
||||||
|
return uint32(imm)<<20 | uint32(avl&0x1F)<<15 | uint32(riscvVf3Cfg)<<12 |
|
||||||
|
uint32(rd)<<7 | riscvOpV
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvVLSType encodes a vector load or store: the full 32-bit word with the
|
||||||
|
// segment count in bits [31:29], the addressing mode in bits [28:26], the
|
||||||
|
// unmasked bit at 25 and the width in funct3. width follows the load
|
||||||
|
// convention (0 = 8-bit, 5 = 16-bit, 6 = 32-bit, 7 = 64-bit).
|
||||||
|
func riscvVLSType(op uint32, nf, mop, width int, rs2 int32, rs1, rd int) uint32 {
|
||||||
|
return uint32(nf&0x7)<<29 | uint32(mop&0x7)<<26 | 1<<25 |
|
||||||
|
uint32(rs2)<<20 | uint32(rs1)<<15 | uint32(width&0x7)<<12 |
|
||||||
|
uint32(rd)<<7 | op
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvVVInstr encodes an OP-V instruction with the six-bit operation code in
|
||||||
|
// funct7's upper bits, bit 25 as the unmasked flag and the three registers in
|
||||||
|
// the standard positions. vs1 may name an integer register for the *VX forms
|
||||||
|
// (the scalar sits in the rs1 field) or an immediate for the *VI forms.
|
||||||
|
func riscvVVInstr(funct6, funct3 int, vs1 int32, vs2, vd int) uint32 {
|
||||||
|
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs1)<<15 |
|
||||||
|
uint32(funct3)<<12 | uint32(vs2)<<20 | uint32(vd)<<7 | riscvOpV
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvVUnaryInstr encodes a one-vector-operand OP-V instruction whose fixed
|
||||||
|
// fields live where the second source register would be: rs1Field and vs2 are
|
||||||
|
// written verbatim (the oracle writes fixed non-zero constants there for some
|
||||||
|
// instructions, such as 0x11 in the rs1 field of vmfirst.m and vid.v).
|
||||||
|
func riscvVUnaryInstr(funct6, funct3 int, rs1Field int32, vs2, vd int) uint32 {
|
||||||
|
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs2&0x1F)<<20 |
|
||||||
|
uint32(rs1Field&0x1F)<<15 | uint32(funct3&0x7)<<12 | uint32(vd&0x1F)<<7 | riscvOpV
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvSegNF maps a segment count to the 3-bit nf field (count - 1).
|
||||||
|
func riscvSegNF(n int) int32 { return int32(n - 1) }
|
||||||
|
|
||||||
// ---- RVC (compressed) encoding helpers ----
|
// ---- RVC (compressed) encoding helpers ----
|
||||||
|
|
||||||
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
|
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
|
||||||
@@ -520,11 +620,13 @@ func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 {
|
|||||||
|
|
||||||
// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
|
// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
|
||||||
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
|
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
|
||||||
// offset; the immediate bits are extracted per the RISC-V CS format.
|
// offset; the immediate bits are extracted per the RISC-V CS format, with the
|
||||||
|
// same five-bit patterns as the load side ({5,4,3,7,6} and {5,4,3,2,6},
|
||||||
|
// matching the toolchain's encodeCS).
|
||||||
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
|
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
|
||||||
pattern := []int{5, 3, 7, 6}
|
pattern := []int{5, 4, 3, 7, 6}
|
||||||
if funct3 == 0x6 {
|
if funct3 == 0x6 {
|
||||||
pattern = []int{5, 3, 2, 6}
|
pattern = []int{5, 4, 3, 2, 6}
|
||||||
}
|
}
|
||||||
packed := encodeRVCPattern(imm, pattern)
|
packed := encodeRVCPattern(imm, pattern)
|
||||||
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
|
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
|
||||||
|
|||||||
+434
-1
@@ -5,6 +5,9 @@ package asm
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
|
"encoding/binary"
|
||||||
|
"encoding/hex"
|
||||||
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
@@ -289,7 +292,7 @@ TEXT ·cmp(SB), NOSPLIT, $0
|
|||||||
}
|
}
|
||||||
|
|
||||||
func TestRISCV_forwardBranch(t *testing.T) {
|
func TestRISCV_forwardBranch(t *testing.T) {
|
||||||
// Forward label reference — must not fail.
|
// Forward label reference; must not fail.
|
||||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
TEXT ·fwd(SB), NOSPLIT, $0
|
TEXT ·fwd(SB), NOSPLIT, $0
|
||||||
ADDI $1, X10, X10
|
ADDI $1, X10, X10
|
||||||
@@ -691,6 +694,81 @@ DATA answer<>+0(SB)/8, $42
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestRISCV_RVC_StorePatterns pins the register-relative compressed store
|
||||||
|
// encodings for offsets with immediate bits 4 and 5 set, byte-identical to
|
||||||
|
// the toolchain's encodeCS (patterns {5,4,3,7,6} and {5,4,3,2,6}).
|
||||||
|
// Regression: the store-side patterns dropped imm[4], so every such store
|
||||||
|
// silently encoded the wrong address while the loads stayed correct.
|
||||||
|
func TestRISCV_RVC_StorePatterns(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·csstores(SB), NOSPLIT, $0
|
||||||
|
SD X9, 24(X8)
|
||||||
|
SW X10, 16(X11)
|
||||||
|
FSD F8, 40(X12)
|
||||||
|
LD 24(X8), X9
|
||||||
|
LW 16(X11), X10
|
||||||
|
FLD 40(X12), F8
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
want := []byte{
|
||||||
|
0x04, 0xec, // c.sd x9, 24(x8)
|
||||||
|
0x88, 0xc9, // c.sw x10, 16(x11)
|
||||||
|
0x00, 0xb6, // c.fsd f8, 40(x12)
|
||||||
|
0x04, 0x6c, // c.ld x9, 24(x8)
|
||||||
|
0x88, 0x49, // c.lw x10, 16(x11)
|
||||||
|
0x00, 0x36, // c.fld f8, 40(x12)
|
||||||
|
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
|
||||||
|
}
|
||||||
|
if !bytes.Equal(code, want) {
|
||||||
|
t.Errorf("code = % x\nwant % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_FENCE pins the FENCE encoding: the toolchain expands the bare
|
||||||
|
// mnemonic to fence iorw, iorw (0x0FF0000F), not fence 0,0.
|
||||||
|
func TestRISCV_FENCE(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·fence(SB), NOSPLIT, $0
|
||||||
|
FENCE
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
want := []byte{
|
||||||
|
0x0f, 0x00, 0xf0, 0x0f, // fence iorw, iorw
|
||||||
|
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
|
||||||
|
}
|
||||||
|
if !bytes.Equal(code, want) {
|
||||||
|
t.Errorf("code = % x\nwant % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_RVC_WidthSpellings pins the compression of the GOROOT width
|
||||||
|
// spellings: MOVW and MOVD lower to their base load/store and compress
|
||||||
|
// exactly like LW/SW/FLD/FSD would (the toolchain compresses these shapes;
|
||||||
|
// before the normalisation they stayed 4 bytes).
|
||||||
|
func TestRISCV_RVC_WidthSpellings(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·widths(SB), NOSPLIT, $0-16
|
||||||
|
MOVW w+0(FP), X9
|
||||||
|
MOVW X9, v+4(FP)
|
||||||
|
MOVD d+0(FP), F8
|
||||||
|
MOVD F8, r+8(FP)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
want := []byte{
|
||||||
|
0xa2, 0x44, // c.lwsp x9, 8
|
||||||
|
0x26, 0xc6, // c.swsp x9, 12
|
||||||
|
0x22, 0x24, // c.fldsp f8, 8
|
||||||
|
0x22, 0xa8, // c.fsdsp f8, 16
|
||||||
|
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
|
||||||
|
}
|
||||||
|
if !bytes.Equal(code, want) {
|
||||||
|
t.Errorf("code = % x\nwant % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestRISCV_system_instrs(t *testing.T) {
|
func TestRISCV_system_instrs(t *testing.T) {
|
||||||
// Test FENCE, ECALL, EBREAK encoding.
|
// Test FENCE, ECALL, EBREAK encoding.
|
||||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
@@ -761,3 +839,358 @@ sub:
|
|||||||
t.Error("expected error for CALL to local label, got nil")
|
t.Error("expected error for CALL to local label, got nil")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestRISCVIndirectBranch pins the indirect branch encodings: JMP (X5) is the
|
||||||
|
// toolchain's JALR X0, 0(X5), and the trampoline form JALR rd, offset(rs1)
|
||||||
|
// takes its destination from the first operand (regression: the base
|
||||||
|
// register was once read as the destination, silently jumping to X0).
|
||||||
|
func TestRISCVIndirectBranch(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·f(SB), NOSPLIT, $0-0
|
||||||
|
JMP (X5)
|
||||||
|
JALR X0, 0(X6)
|
||||||
|
JALR X28, 0(X9)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x00028067, // jalr x0, 5(x0), 0
|
||||||
|
0x00030067, // jalr x0, 6(x0), 0
|
||||||
|
0x00048e67, // jalr x28, 9(x0), 0
|
||||||
|
0x00008067, // jalr x0, 1(x0), 0 (RET)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeOneInstrRISCV encodes a single parsed instruction against a synthetic
|
||||||
|
// offsets map, the smallest honest harness for the branch-range diagnostics:
|
||||||
|
// the spans are far larger than any source a test would want to spell out.
|
||||||
|
func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]int) ([]byte, error) {
|
||||||
|
t.Helper()
|
||||||
|
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
|
||||||
|
instr := fn.Body[0].(*ast.Instr)
|
||||||
|
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span
|
||||||
|
// [-4096, 4094] and the J-type span [-1048576, 1048574] are diagnosed instead
|
||||||
|
// of wrapping silently to a wrong target.
|
||||||
|
func TestRISCVBranchJumpRange(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
src string
|
||||||
|
off int // the target's function-relative offset (pc 0)
|
||||||
|
ok bool
|
||||||
|
}{
|
||||||
|
{"branch max", "BEQ X10, X11, tgt\nRET\n", 4094, true},
|
||||||
|
{"branch past max", "BEQ X10, X11, tgt\nRET\n", 4096, false},
|
||||||
|
{"branch back max", "BEQ X10, X11, tgt\nRET\n", -4096, true},
|
||||||
|
{"branch back past max", "BEQ X10, X11, tgt\nRET\n", -4098, false},
|
||||||
|
{"branchz past max", "BEQZ X10, tgt\nRET\n", 4096, false},
|
||||||
|
{"jump max", "JMP tgt\nRET\n", 1048574, true},
|
||||||
|
{"jump past max", "JMP tgt\nRET\n", 1048576, false},
|
||||||
|
{"jump back max", "JMP tgt\nRET\n", -1048576, true},
|
||||||
|
{"jump back past max", "JMP tgt\nRET\n", -1048578, false},
|
||||||
|
{"jal past max", "JAL tgt\nRET\n", 1048576, false},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
_, err := encodeOneInstrRISCV(t, "TEXT ·f(SB), NOSPLIT, $0\n\t"+c.src, 0, map[string]int{"tgt": c.off})
|
||||||
|
if c.ok && err != nil {
|
||||||
|
t.Fatalf("unexpected error: %v", err)
|
||||||
|
}
|
||||||
|
if !c.ok && err == nil {
|
||||||
|
t.Fatal("expected an out-of-range diagnostic, got none")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCVBranchFarBody drives the range check through the full two-pass
|
||||||
|
// assembler: a forward branch over a body larger than the B-type span must
|
||||||
|
// error rather than wrap.
|
||||||
|
func TestRISCVBranchFarBody(t *testing.T) {
|
||||||
|
var sb strings.Builder
|
||||||
|
sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n")
|
||||||
|
for range 1100 {
|
||||||
|
sb.WriteString("\tADD X10, X11, X12\n")
|
||||||
|
}
|
||||||
|
sb.WriteString("done:\n\tRET\n")
|
||||||
|
fn := firstTextRISCV(t, sb.String())
|
||||||
|
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
|
||||||
|
t.Error("expected a branch-out-of-range error, got none")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_CSRRange checks the CSR address range: the 12-bit field is
|
||||||
|
// diagnosed rather than masked, so CSRRW $4096 does not silently address
|
||||||
|
// CSR 0.
|
||||||
|
func TestRISCV_CSRRange(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·csrhi(SB), NOSPLIT, $0
|
||||||
|
CSRRW $4096, X10, X11
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
|
||||||
|
t.Error("expected an out-of-range error for CSR $4096, got none")
|
||||||
|
}
|
||||||
|
fn = firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·csrmax(SB), NOSPLIT, $0
|
||||||
|
CSRRW $4095, X10, X11
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if _, _, _, _, _, err := assembleRISCV(fn); err != nil {
|
||||||
|
t.Errorf("CSR $4095 must assemble: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_Imm64Rejected checks that immediates outside the signed 32-bit
|
||||||
|
// span are diagnosed instead of silently truncated to their low 32 bits (the
|
||||||
|
// toolchain materialises such constants via SLLI expansion, which this
|
||||||
|
// assembler does not implement).
|
||||||
|
func TestRISCV_Imm64Rejected(t *testing.T) {
|
||||||
|
cases := []string{
|
||||||
|
"MOV $0x123456789, X10",
|
||||||
|
"ADDI $0x100000000, X10, X11",
|
||||||
|
"ANDI $-0x800000001, X10, X11",
|
||||||
|
"SUB $0x100000000, X10, X11",
|
||||||
|
}
|
||||||
|
for _, src := range cases {
|
||||||
|
fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·wide(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
|
||||||
|
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
|
||||||
|
t.Errorf("%s: expected an out-of-range error, got none", src)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The full signed 32-bit span still assembles, including the SUB form
|
||||||
|
// whose negated immediate only just fits.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·edge(SB), NOSPLIT, $0
|
||||||
|
MOV $2147483647, X10
|
||||||
|
MOV $-2147483648, X11
|
||||||
|
SUB $0x80000000, X12, X13
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if _, _, _, _, _, err := assembleRISCV(fn); err != nil {
|
||||||
|
t.Errorf("int32-span immediates must assemble: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvWants decodes code as little-endian words and pins each one; the
|
||||||
|
// expected values below were read off GOARCH=riscv64 go tool objdump of
|
||||||
|
// kernels assembled with go tool asm (the toolchain's riscv64.s testdata
|
||||||
|
// cross-checks the same words).
|
||||||
|
func riscvWants(t *testing.T, code []byte, want ...uint32) {
|
||||||
|
t.Helper()
|
||||||
|
got := make([]uint32, 0, len(code)/4)
|
||||||
|
for i := 0; i+4 <= len(code); i += 4 {
|
||||||
|
got = append(got, binary.LittleEndian.Uint32(code[i:]))
|
||||||
|
}
|
||||||
|
if len(got) < len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
|
||||||
|
}
|
||||||
|
// The RET (JALR) ends the sequence; only the pinned prefix is compared.
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvWantsHex pins the exact hex encoding of a function's instruction
|
||||||
|
// bytes, including any 2-byte compressed instructions in the stream; the
|
||||||
|
// expected strings were read off GOARCH=riscv64 go tool objdump of kernels
|
||||||
|
// assembled with go tool asm (the toolchain's riscv64.s testdata
|
||||||
|
// cross-checks the same words).
|
||||||
|
func riscvWantsHex(t *testing.T, code []byte, wantHex string) {
|
||||||
|
t.Helper()
|
||||||
|
got := hex.EncodeToString(code)
|
||||||
|
if got != wantHex {
|
||||||
|
t.Errorf("code = %s, want %s", got, wantHex)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_extendedPseudos pins the toolchain-synthesised instructions:
|
||||||
|
// ANDN/ORN (XORI + AND/OR through the destination or TMP), the five-word
|
||||||
|
// MIN/MAX expansion, the four-word rotate, ROR's compressed reverse shift
|
||||||
|
// (C.SLLI when rd == rs1, both non-zero, 1 <= sll <= 63), the identical-
|
||||||
|
// input MIN/MAX fold to C.MV, FABSD (FSGNJX.D), SEQZ and RDTIME (csrrs with
|
||||||
|
// the time CSR).
|
||||||
|
func TestRISCV_extendedPseudos(t *testing.T) {
|
||||||
|
t.Run("logic and minmax", func(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·l(SB), NOSPLIT, $0
|
||||||
|
ANDN X19, X20, X21
|
||||||
|
ANDN X19, X20
|
||||||
|
ORN X20, X19
|
||||||
|
MAX X26, X28, X29
|
||||||
|
MIN X29, X30, X5
|
||||||
|
MAX X5, X5
|
||||||
|
MAX X5, X5, X6
|
||||||
|
SEQZ X5, X6
|
||||||
|
NEG X5, X6
|
||||||
|
NOT X5
|
||||||
|
RDTIME X5
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// Words 0-10 up to the folded C.MV pair (halfwords 96 82 and 16 83),
|
||||||
|
// then SEQZ, NEG, NOT and RDTIME.
|
||||||
|
riscvWantsHex(t, code,
|
||||||
|
"93caf9ffb37a5a01"+"93cff9ff337afa01"+"934ffaffb3e9f901"+
|
||||||
|
"b32fae01b30ff041b34eae01b3fedf01b34ede01"+
|
||||||
|
"b3afee01b30ff041b342df01b3f25f00b3425f00"+
|
||||||
|
"9682"+"1683"+
|
||||||
|
"13b31200"+"33035040"+"93c2f2ff"+"f32210c0"+"67800000")
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("rotate", func(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·r(SB), NOSPLIT, $0
|
||||||
|
ROR X10, X11, X12
|
||||||
|
ROR X10, X11
|
||||||
|
ROR $63, X11
|
||||||
|
RORIW $31, X13, X14
|
||||||
|
RORIW $1, X14, X15
|
||||||
|
RORIW $3, X14
|
||||||
|
RORW X15, X16, X17
|
||||||
|
RORW $31, X13
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// The third ROR carries the compressed C.SLLI (05 86) in mid-stream.
|
||||||
|
riscvWantsHex(t, code,
|
||||||
|
"b30fa040b39ff50133d6a50033e6cf00"+
|
||||||
|
"b30fa040b39ff501b3d5a500b3e5bf00"+
|
||||||
|
"93dff5038605b3e5bf00"+
|
||||||
|
"9bdff6011b97160033e7ef00"+
|
||||||
|
"9b5f17009b17f701b3e7ff00"+
|
||||||
|
"9b5f37001b17d70133e7ef00"+
|
||||||
|
"b30ff040bb1ff801bb58f800b3e81f01"+
|
||||||
|
"9bdff6019b961600b3e6df00"+"67800000")
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("fp and branches", func(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·f(SB), NOSPLIT, $0
|
||||||
|
FABSD F1, F2
|
||||||
|
FSGNJD F1, F0, F2
|
||||||
|
FMADDD F1, F2, F3, F4
|
||||||
|
FMSUBD F1, F2, F3, F4
|
||||||
|
FNMSUBD F1, F2, F3, F4
|
||||||
|
BGT X5, X6, tgt
|
||||||
|
BLE X5, X6, tgt
|
||||||
|
BGTU X5, X6, tgt
|
||||||
|
BLEU X5, X6, tgt
|
||||||
|
tgt:
|
||||||
|
RDTIME X5
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
riscvWantsHex(t, code,
|
||||||
|
"53a11022"+"53011022"+"4382201a4782201a4b82201a"+
|
||||||
|
"63485300635653006364530063725300"+ // blt/bge/bltu/bgeu x6, x5
|
||||||
|
"f32210c0"+"67800000")
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_amoWords pins the full AMO family: every AMO carries aq and rl
|
||||||
|
// (funct7 |= 3), LR is acquire (funct7 |= 2) and SC release (funct7 |= 1),
|
||||||
|
// exactly as GOARCH=riscv64 go tool asm encodes them.
|
||||||
|
func TestRISCV_amoWords(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·amo(SB), NOSPLIT, $0
|
||||||
|
AMOSWAPW X5, (X6), X7
|
||||||
|
AMOSWAPD X5, (X6), X7
|
||||||
|
AMOADDW X5, (X6), X7
|
||||||
|
AMOADDD X5, (X6), X7
|
||||||
|
AMOANDW X5, (X6), X7
|
||||||
|
AMOANDD X5, (X6), X7
|
||||||
|
AMOORW X5, (X6), X7
|
||||||
|
AMOORD X5, (X6), X7
|
||||||
|
AMOXORW X5, (X6), X7
|
||||||
|
AMOXORD X5, (X6), X7
|
||||||
|
AMOMAXW X5, (X6), X7
|
||||||
|
AMOMAXD X5, (X6), X7
|
||||||
|
AMOMAXUW X5, (X6), X7
|
||||||
|
AMOMAXUD X5, (X6), X7
|
||||||
|
AMOMINUW X5, (X6), X7
|
||||||
|
AMOMINUD X5, (X6), X7
|
||||||
|
LRW (X5), X6
|
||||||
|
LRD (X5), X6
|
||||||
|
SCW X5, (X6), X7
|
||||||
|
SCD X5, (X6), X7
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
riscvWants(t, code,
|
||||||
|
0x0E5323AF, // amoswap.w
|
||||||
|
0x0E5333AF, // amoswap.d
|
||||||
|
0x065323AF, // amoaddd.w
|
||||||
|
0x065333AF, // amoadd.d
|
||||||
|
0x665323AF, // amoand.w
|
||||||
|
0x665333AF, // amoand.d
|
||||||
|
0x465323AF, // amoor.w
|
||||||
|
0x465333AF, // amoor.d
|
||||||
|
0x265323AF, // amoxor.w
|
||||||
|
0x265333AF, // amoxor.d
|
||||||
|
0xA65323AF, // amomax.w
|
||||||
|
0xA65333AF, // amomax.d
|
||||||
|
0xE65323AF, // amomaxu.w
|
||||||
|
0xE65333AF, // amomaxu.d
|
||||||
|
0xC65323AF, // amominu.w
|
||||||
|
0xC65333AF, // amominu.d
|
||||||
|
0x1402A32F, // lr.w (aq)
|
||||||
|
0x1402B32F, // lr.d
|
||||||
|
0x1A5323AF, // sc.w (rl)
|
||||||
|
0x1A5333AF, // sc.d
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_vectorWords pins the RVV slice and the VSET* encodings. The
|
||||||
|
// toolchain canonicalises an immediate avl to vsetivli even under the
|
||||||
|
// VSETVLI spelling (`VSETVLI $15` and `VSETIVLI $15` come out byte-
|
||||||
|
// identical), which is what the 0xC00 bit of the first word carries.
|
||||||
|
func TestRISCV_vectorWords(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VSETVLI X5, E8, M8, TA, MA, X6
|
||||||
|
VSETIVLI $4, E32, M1, TA, MA, X0
|
||||||
|
VSETVLI $15, E32, M1, TA, MA, X12
|
||||||
|
VADDVV V1, V2, V3
|
||||||
|
VADDVX X12, V12, V12
|
||||||
|
VXORVV V8, V16, V24
|
||||||
|
VMSEQVX X12, V8, V0
|
||||||
|
VMSNEVV V8, V16, V0
|
||||||
|
VSLLVI $8, V28, V30
|
||||||
|
VSRLVI $25, V29, V29
|
||||||
|
VFIRSTM V0, X6
|
||||||
|
VIDV V12
|
||||||
|
VMV4RV V8, V24
|
||||||
|
VLE8V (X10), V8
|
||||||
|
VSE8V V24, (X10)
|
||||||
|
VSE32V V9, (X11)
|
||||||
|
VLSSEG4E32V (X14), X0, V0
|
||||||
|
VLSSEG8E32V (X10), X0, V4
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
riscvWants(t, code,
|
||||||
|
0x0C32F357, // vsetvli x6, x5, vtype 0xc3 (E8, M8, TA, MA)
|
||||||
|
0xCD027057, // vsetivli x0, 4
|
||||||
|
0xCD07F657, // vsetivli x12, 15: VSETVLI $15 canonicalises to the same word
|
||||||
|
0x022081D7, // vadd.vv v3, v2, v1
|
||||||
|
0x02C64657, // vadd.vx v12, v12, x12
|
||||||
|
0x2F040C57, // vxor.vv v24, v16, v8
|
||||||
|
0x62864057, // vmseq.vx v0, v8, x12
|
||||||
|
0x67040057, // vmsne.vv v0, v16, v8
|
||||||
|
0x97C43F57, // vsll.vi v30, v28, 8
|
||||||
|
0xA3DCBED7, // vsrl.vi v29, v29, 25
|
||||||
|
0x4208A357, // vmfirst.m x6, v0
|
||||||
|
0x5208A657, // vid.v v12
|
||||||
|
0x9E81BC57, // vmv4r.v v24, v8
|
||||||
|
0x02050407, // vle8.v v8, (x10)
|
||||||
|
0x02050C27, // vse8.v v24, (x10)
|
||||||
|
0x0205E4A7, // vse32.v v9, (x11)
|
||||||
|
0x6A076007, // vlsseg4e32.v v0, (x14), x0
|
||||||
|
0xEA056207, // vlsseg8e32.v v4, (x10), x0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|||||||
+44
-34
@@ -4,6 +4,7 @@
|
|||||||
package asm
|
package asm
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"fmt"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
@@ -93,12 +94,19 @@ func riscvIsLeaf(t *ast.Text) bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
case "JALR":
|
case "JALR":
|
||||||
// JALR rs1, rd, a call when rd is X1; JALR offset(rs1) always
|
// JALR rd, offset(rs1) links when the destination register (the
|
||||||
// links to X1.
|
// first operand) is X1; JALR rs1, rd links when the second
|
||||||
|
// register is X1; JALR offset(rs1) always links to X1.
|
||||||
if len(in.Operands) == 1 {
|
if len(in.Operands) == 1 {
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
if len(in.Operands) >= 2 && regFromOperand(in.Operands[1]) == 1 {
|
if isMemOperand(in.Operands[1]) {
|
||||||
|
if regFromOperand(in.Operands[0]) == 1 {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if regFromOperand(in.Operands[1]) == 1 {
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -234,33 +242,36 @@ func riscvFitsCAddi(imm int32) bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// riscvPrologueSpadjPC returns the function-relative byte offset where the
|
// riscvPrologueSpadjPC returns the function-relative byte offset where the
|
||||||
// prologue has finished decrementing SP (the delta becomes autosize).
|
// prologue has finished decrementing SP (the delta becomes autosize). It is
|
||||||
|
// computed from the same expansion functions the prologue emits, so the
|
||||||
|
// large-frame X31 materialisations are counted: C.LUI + C.ADD before the SD,
|
||||||
|
// C.LUI + ADDIW + C.ADD for the SP adjust.
|
||||||
func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
|
func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
|
||||||
if fi.autosize == 0 {
|
if fi.autosize == 0 {
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
adj := int32(-fi.autosize)
|
||||||
return 4 + riscvSPAdjustLen(int32(-fi.autosize))
|
if fits12(adj) {
|
||||||
|
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
||||||
|
return 4 + len(riscvSPAdjust(adj))
|
||||||
|
}
|
||||||
|
return len(riscvAddressInX31(adj)) + 4 + len(riscvAddToSP(adj))
|
||||||
}
|
}
|
||||||
|
|
||||||
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
||||||
// (but not including) the final JALR, the point where SP is restored.
|
// (but not including) the final JALR, the point where SP is restored. The
|
||||||
|
// small frame closes with C.LDSP + ADDI/C.ADDI; the large frame materialises
|
||||||
|
// the adjustment through X31 (C.LUI + ADDIW + C.ADD).
|
||||||
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
|
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
|
||||||
if fi.autosize == 0 {
|
if fi.autosize == 0 {
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
adj := int32(fi.autosize)
|
||||||
return 2 + riscvSPAdjustLen(int32(fi.autosize))
|
if fits12(adj) {
|
||||||
}
|
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
||||||
|
return 2 + len(riscvSPAdjust(adj))
|
||||||
func riscvSPAdjustLen(imm int32) int {
|
|
||||||
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
|
|
||||||
return 2
|
|
||||||
}
|
}
|
||||||
if riscvFitsCAddi(imm) {
|
return 2 + len(riscvAddToSP(adj))
|
||||||
return 2
|
|
||||||
}
|
|
||||||
return 4
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// riscvResolvePseudo translates a pseudo-register memory reference into a
|
// riscvResolvePseudo translates a pseudo-register memory reference into a
|
||||||
@@ -286,19 +297,21 @@ func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32
|
|||||||
// including the inline morestack call (zero when the function needs no
|
// including the inline morestack call (zero when the function needs no
|
||||||
// guard). Unlike amd64 and arm64, the toolchain places the morestack call
|
// guard). Unlike amd64 and arm64, the toolchain places the morestack call
|
||||||
// between the guard and the body: the guard branches forward over it.
|
// between the guard and the body: the guard branches forward over it.
|
||||||
func riscvGuardLen(fi riscvFrameInfo) int {
|
func riscvGuardLen(fi riscvFrameInfo) (int, error) {
|
||||||
_, reloc := riscvGuard(fi)
|
g, _, err := riscvGuard(fi)
|
||||||
_ = reloc
|
if err != nil {
|
||||||
return len(riscvGuardBytes(fi))
|
return 0, err
|
||||||
|
}
|
||||||
|
return len(g), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// riscvGuard emits the stack-split guard prefix with the inline morestack
|
// riscvGuard emits the stack-split guard prefix with the inline morestack
|
||||||
// call: the branch skips forward over JAL X5 and JAL X0 straight into the
|
// call: the branch skips forward over JAL X5 and JAL X0 straight into the
|
||||||
// body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are
|
// body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are
|
||||||
// relative to the guard itself, which sits at function offset 0.
|
// relative to the guard itself, which sits at function offset 0.
|
||||||
func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
|
func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc, error) {
|
||||||
if !fi.needSplit {
|
if !fi.needSplit {
|
||||||
return nil, Reloc{}
|
return nil, Reloc{}, nil
|
||||||
}
|
}
|
||||||
// MOV 16(g), X6 (g.stackguard0), g = X27.
|
// MOV 16(g), X6 (g.stackguard0), g = X27.
|
||||||
out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16))
|
out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16))
|
||||||
@@ -310,14 +323,14 @@ func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
|
|||||||
var reloc Reloc
|
var reloc Reloc
|
||||||
switch fi.splitClass {
|
switch fi.splitClass {
|
||||||
case 0:
|
case 0:
|
||||||
// BLTU X6, SP, done (+8: over the CALL and the JMP back)
|
// BLTU X6, SP, done (+12: over the CALL and the JMP back)
|
||||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...)
|
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...)
|
||||||
call := len(out)
|
call := len(out)
|
||||||
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
|
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
|
||||||
out = append(out, wordLE(riscvJType(5, 0))...)
|
out = append(out, wordLE(riscvJType(5, 0))...)
|
||||||
out = append(out, jalBack()...)
|
out = append(out, jalBack()...)
|
||||||
case 1:
|
case 1:
|
||||||
// ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+8)
|
// ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+12)
|
||||||
off := int32(fi.autosize - stackSmall)
|
off := int32(fi.autosize - stackSmall)
|
||||||
out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...)
|
out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...)
|
||||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
|
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
|
||||||
@@ -335,7 +348,10 @@ func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
|
|||||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...)
|
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...)
|
||||||
addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off)
|
addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
addi = nil
|
// The ADDI expansion failed: the SP adjustment this class
|
||||||
|
// depends on is not emittable, and silently dropping it would
|
||||||
|
// corrupt every stack reference in the body.
|
||||||
|
return nil, Reloc{}, fmt.Errorf("stack-split guard: %w", err)
|
||||||
}
|
}
|
||||||
out = append(out, addi...)
|
out = append(out, addi...)
|
||||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
|
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
|
||||||
@@ -344,11 +360,5 @@ func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
|
|||||||
out = append(out, wordLE(riscvJType(5, 0))...)
|
out = append(out, wordLE(riscvJType(5, 0))...)
|
||||||
out = append(out, jalBack()...)
|
out = append(out, jalBack()...)
|
||||||
}
|
}
|
||||||
return out, reloc
|
return out, reloc, nil
|
||||||
}
|
|
||||||
|
|
||||||
// riscvGuardBytes emits the guard prefix bytes alone (sizing helper).
|
|
||||||
func riscvGuardBytes(fi riscvFrameInfo) []byte {
|
|
||||||
g, _ := riscvGuard(fi)
|
|
||||||
return g
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -65,6 +65,47 @@ TEXT ·framed(SB), NOSPLIT, $16-16
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestRISCVFrameSpadjLargeFrame checks the stack-adjustment boundaries of a
|
||||||
|
// frame past the imm12 range: the prologue materialises the LR-store address
|
||||||
|
// and the SP adjustment through X31 (C.LUI + C.ADD + SD, then C.LUI + ADDIW +
|
||||||
|
// C.ADD), so the SP boundary lands at PC 16, and the RET closes with
|
||||||
|
// C.LDSP plus the same X31 adjustment, 10 bytes. Regression: both helpers
|
||||||
|
// assumed the small-frame prologue and reported 8 and 6.
|
||||||
|
func TestRISCVFrameSpadjLargeFrame(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("bigframe_riscv64.s", `#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·big(SB), NOSPLIT, $9000-8
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
fn := img.Funcs[0]
|
||||||
|
|
||||||
|
// autosize = 9008. Prologue: C.LUI X31 + C.ADD X31,SP (4) + SD (4) +
|
||||||
|
// C.LUI X31 + ADDIW X31 + C.ADD SP,X31 (8) = 16 bytes to the SP boundary;
|
||||||
|
// C.SDSP X1 (2) follows, so the body starts at 18.
|
||||||
|
wantSpadj := []SpadjStep{{PC: 16, Value: 9008}, {PC: 36, Value: 0}}
|
||||||
|
if len(fn.Spadj) != len(wantSpadj) {
|
||||||
|
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
|
||||||
|
}
|
||||||
|
for i := range wantSpadj {
|
||||||
|
if fn.Spadj[i] != wantSpadj[i] {
|
||||||
|
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The FP load materialises its 9016-byte offset through X31 as well
|
||||||
|
// (8 bytes), then RET's epilogue (C.LDSP + X31 adjust = 10) plus JALR.
|
||||||
|
if fn.Size != 18+8+14 {
|
||||||
|
t.Errorf("size = %d, want %d", fn.Size, 18+8+14)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestRISCVRegAliases checks the Go ABI register aliases that the toolchain
|
// TestRISCVRegAliases checks the Go ABI register aliases that the toolchain
|
||||||
// defines: LR is the link register (X1) and TMP is the assembler scratch
|
// defines: LR is the link register (X1) and TMP is the assembler scratch
|
||||||
// register (X31/T6).
|
// register (X31/T6).
|
||||||
|
|||||||
@@ -74,6 +74,89 @@ DATA callee<>+0(SB)/8, $42
|
|||||||
t.Error("ELF object missing R_RISCV_JAL relocation")
|
t.Error("ELF object missing R_RISCV_JAL relocation")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestELFRISCVPCRELLO12Anchor checks the psABI's LO12 pairing rule: the
|
||||||
|
// R_RISCV_PCREL_LO12_I/S relocation must reference a symbol whose value is
|
||||||
|
// the AUIPC site of its HI20 partner (psABI §8.4.9; cmd/link generates one
|
||||||
|
// local text symbol per AUIPC for exactly this). The emitter pairs each
|
||||||
|
// HI20 (against the target symbol) with a LO12 against the .text section
|
||||||
|
// symbol whose addend is the AUIPC's section-relative offset, so S + A is
|
||||||
|
// the AUIPC address.
|
||||||
|
func TestELFRISCVPCRELLO12Anchor(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("k_riscv64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·sb(SB), NOSPLIT, $0-0
|
||||||
|
MOV $answer<>(SB), X10
|
||||||
|
MOV answer<>(SB), X11
|
||||||
|
MOV X12, answer<>(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
GLOBL answer<>(SB), RODATA, $8
|
||||||
|
DATA answer<>+0(SB)/8, $42
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFRISCVObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFRISCVObject: %v", err)
|
||||||
|
}
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse ELF: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
if flags := binary.LittleEndian.Uint32(obj[48:]); flags != efRISCVFloatAbiDouble {
|
||||||
|
t.Errorf("e_flags = %#x, want %#x (EF_RISCV_FLOAT_ABI_DOUBLE)", flags, efRISCVFloatAbiDouble)
|
||||||
|
}
|
||||||
|
rela := ef.Section(".rela.text")
|
||||||
|
if rela == nil {
|
||||||
|
t.Fatal("missing .rela.text")
|
||||||
|
}
|
||||||
|
b, err := rela.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if len(b) != 6*24 {
|
||||||
|
t.Fatalf(".rela.text holds %d entries, want six (three HI20/LO12 pairs)", len(b)/24)
|
||||||
|
}
|
||||||
|
le := binary.LittleEndian
|
||||||
|
wantLo := []uint32{rRISCVPCRELLO12I, rRISCVPCRELLO12I, rRISCVPCRELLO12S}
|
||||||
|
for p := range 3 {
|
||||||
|
auipc := 8 * p
|
||||||
|
hi := b[p*2*24:]
|
||||||
|
lo := b[(p*2+1)*24:]
|
||||||
|
if off := le.Uint64(hi[0:]); off != uint64(auipc) {
|
||||||
|
t.Errorf("pair %d: HI20 r_offset = %d, want %d (the AUIPC)", p, off, auipc)
|
||||||
|
}
|
||||||
|
if typ := uint32(le.Uint64(hi[8:])); typ != rRISCVPCRELHI20 {
|
||||||
|
t.Errorf("pair %d: HI20 type = %d, want %d", p, typ, rRISCVPCRELHI20)
|
||||||
|
}
|
||||||
|
if sym := int(le.Uint64(hi[8:]) >> 32); sym == 0 || sym == 1 {
|
||||||
|
t.Errorf("pair %d: HI20 against symbol %d, want the target", p, sym)
|
||||||
|
}
|
||||||
|
if off := le.Uint64(lo[0:]); off != uint64(auipc+4) {
|
||||||
|
t.Errorf("pair %d: LO12 r_offset = %d, want %d", p, off, auipc+4)
|
||||||
|
}
|
||||||
|
if typ := uint32(le.Uint64(lo[8:])); typ != wantLo[p] {
|
||||||
|
t.Errorf("pair %d: LO12 type = %d, want %d", p, typ, wantLo[p])
|
||||||
|
}
|
||||||
|
// The LO12 must denote the AUIPC site: the .text section symbol
|
||||||
|
// (index 1) plus the AUIPC's section-relative offset as addend.
|
||||||
|
if sym := int(le.Uint64(lo[8:]) >> 32); sym != 1 {
|
||||||
|
t.Errorf("pair %d: LO12 against symbol %d, want 1 (the .text section symbol)", p, sym)
|
||||||
|
}
|
||||||
|
if add := int64(le.Uint64(lo[16:])); add != int64(auipc) {
|
||||||
|
t.Errorf("pair %d: LO12 addend = %d, want %d (S + A = the AUIPC address)", p, add, auipc)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestGOObjectRISCVStructure(t *testing.T) {
|
func TestGOObjectRISCVStructure(t *testing.T) {
|
||||||
f, errs := parser.Parse("k_riscv64.s", `
|
f, errs := parser.Parse("k_riscv64.s", `
|
||||||
#include "textflag.h"
|
#include "textflag.h"
|
||||||
|
|||||||
+144
-1
@@ -41,7 +41,7 @@ const (
|
|||||||
vexExtract
|
vexExtract
|
||||||
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
||||||
// in ModRM.reg and the destination in r/m, the layout of the EVEX
|
// in ModRM.reg and the destination in r/m, the layout of the EVEX
|
||||||
// narrowing stores (VPMOVDW, VPMOVQD).
|
// narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ.
|
||||||
vexRMRev
|
vexRMRev
|
||||||
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
||||||
// vector length follows the source: the packed-double → dword
|
// vector length follows the source: the packed-double → dword
|
||||||
@@ -52,6 +52,15 @@ const (
|
|||||||
vexRMSrcLen
|
vexRMSrcLen
|
||||||
// vexZero is the no-operand form (VZEROUPPER).
|
// vexZero is the no-operand form (VZEROUPPER).
|
||||||
vexZero
|
vexZero
|
||||||
|
// vexZeroAll is the no-operand form that zeroes the full upper state
|
||||||
|
// (VZEROALL, the L = 1 twin of VZEROUPPER).
|
||||||
|
vexZeroAll
|
||||||
|
// vexNDS3GPR is the three-operand NDS form over general-purpose
|
||||||
|
// registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0.
|
||||||
|
vexNDS3GPR
|
||||||
|
// vexImmRMGPR is the immediate form over general-purpose registers
|
||||||
|
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
|
||||||
|
vexImmRMGPR
|
||||||
)
|
)
|
||||||
|
|
||||||
// vexSpec describes one VEX instruction's encoding parameters.
|
// vexSpec describes one VEX instruction's encoding parameters.
|
||||||
@@ -125,6 +134,12 @@ var vexTable = map[string]vexSpec{
|
|||||||
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
||||||
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
|
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
|
||||||
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
||||||
|
// Scalar fused multiply-add (NDS form). The Go assembler carries the
|
||||||
|
// same 66 prefix as the packed forms on every FMA row, and W1 on the
|
||||||
|
// double-precision spellings, so SD shares PD's prefix/W pair and the
|
||||||
|
// scalar width rides on the W bit.
|
||||||
|
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3},
|
||||||
|
|
||||||
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
|
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
|
||||||
// no vvvv).
|
// no vvvv).
|
||||||
@@ -192,6 +207,31 @@ var vexTable = map[string]vexSpec{
|
|||||||
|
|
||||||
// VEX.128.0F.W0, no operands.
|
// VEX.128.0F.W0, no operands.
|
||||||
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
||||||
|
// VEX.256.0F.W0, zero all vector registers (the L = 1 twin).
|
||||||
|
"VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll},
|
||||||
|
// VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare.
|
||||||
|
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm},
|
||||||
|
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm},
|
||||||
|
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3},
|
||||||
|
// VEX.128/256.0F.WIG, packed single XOR (NDS form).
|
||||||
|
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3},
|
||||||
|
// VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control.
|
||||||
|
"VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm},
|
||||||
|
"VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm},
|
||||||
|
// VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling
|
||||||
|
// falls through to the EVEX table.
|
||||||
|
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm},
|
||||||
|
// VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst).
|
||||||
|
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm},
|
||||||
|
// VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8).
|
||||||
|
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm},
|
||||||
|
// BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR).
|
||||||
|
"ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR},
|
||||||
|
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
|
||||||
|
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
|
||||||
|
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
|
||||||
|
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
|
||||||
|
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
|
||||||
|
|
||||||
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
||||||
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
||||||
@@ -200,6 +240,14 @@ var vexTable = map[string]vexSpec{
|
|||||||
// rm=scalar memory; SD is 256-bit only).
|
// rm=scalar memory; SD is 256-bit only).
|
||||||
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
||||||
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
||||||
|
// VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a
|
||||||
|
// YMM (the encoder rejects an XMM destination, as go tool asm does).
|
||||||
|
"VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM},
|
||||||
|
// VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg,
|
||||||
|
// memory destination in rm).
|
||||||
|
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev},
|
||||||
|
// VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv).
|
||||||
|
"VPTEST": {2, 0x17, 0, 1, -1, vexRM},
|
||||||
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
|
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
|
||||||
// source).
|
// source).
|
||||||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
||||||
@@ -290,6 +338,8 @@ type vexMoveSpec struct {
|
|||||||
var vexMoveTable = map[string]vexMoveSpec{
|
var vexMoveTable = map[string]vexMoveSpec{
|
||||||
// VEX.128/256.F3.0F.WIG, unaligned integer move.
|
// VEX.128/256.F3.0F.WIG, unaligned integer move.
|
||||||
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
||||||
|
// VEX.128/256.66.0F.WIG, aligned integer move.
|
||||||
|
"VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
||||||
// VEX.128/256.66.0F.WIG, unaligned packed double move.
|
// VEX.128/256.66.0F.WIG, unaligned packed double move.
|
||||||
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
|
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
|
||||||
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
|
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
|
||||||
@@ -324,6 +374,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
|||||||
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination
|
||||||
|
// only; an XMM destination is rejected exactly as go tool asm does.
|
||||||
|
if mnemUpper == "VBROADCASTI128" {
|
||||||
|
dstReg, ok := ops[len(ops)-1].(Reg)
|
||||||
|
if len(ops) != 2 || !ok || dstReg.size != 32 {
|
||||||
|
return fmt.Errorf("VBROADCASTI128 requires a YMM destination")
|
||||||
|
}
|
||||||
|
}
|
||||||
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
||||||
return e.encodeVexMove(mnemUpper, ms, ops)
|
return e.encodeVexMove(mnemUpper, ms, ops)
|
||||||
}
|
}
|
||||||
@@ -356,6 +414,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
|||||||
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
||||||
case vexZero:
|
case vexZero:
|
||||||
return e.encodeVexZero(mnemUpper, spec, ops)
|
return e.encodeVexZero(mnemUpper, spec, ops)
|
||||||
|
case vexZeroAll:
|
||||||
|
return e.encodeVexZeroAll(mnemUpper, spec, ops)
|
||||||
|
case vexNDS3GPR:
|
||||||
|
return e.encodeVexNDS3GPR(spec, ops)
|
||||||
|
case vexImmRMGPR:
|
||||||
|
return e.encodeVexImmRMGPR(spec, ops)
|
||||||
|
case vexRMRev:
|
||||||
|
return e.encodeVexRMRev(spec, ops)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
|
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
|
||||||
}
|
}
|
||||||
@@ -607,6 +673,83 @@ func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1
|
||||||
|
// twin of VZEROUPPER.
|
||||||
|
func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
|
||||||
|
}
|
||||||
|
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1.
|
||||||
|
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose
|
||||||
|
// registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1,
|
||||||
|
// rm = src2 and L = 0.
|
||||||
|
func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
src2, src1, dst := ops[0], ops[1], ops[2]
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||||
|
}
|
||||||
|
vvvvReg, ok := src1.(Reg)
|
||||||
|
if !ok || vvvvReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15-(vvvvReg.idx&15), src2)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexImmRMGPR encodes the immediate form over general-purpose
|
||||||
|
// registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0.
|
||||||
|
func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, src, dst := ops[0], ops[1], ops[2]
|
||||||
|
immVal, ok := imm.(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("shift control must be an immediate")
|
||||||
|
}
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(immVal))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if err := e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
e.out = append(e.out, immByte)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
|
||||||
|
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
|
||||||
|
// a store with no register-destination form).
|
||||||
|
func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("store expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
srcReg, ok := ops[0].(Reg)
|
||||||
|
if !ok || !srcReg.isVec() {
|
||||||
|
return fmt.Errorf("store source must be a vector register")
|
||||||
|
}
|
||||||
|
if !memOperand(ops[1]) {
|
||||||
|
return fmt.Errorf("store destination must be memory")
|
||||||
|
}
|
||||||
|
rBit := 0
|
||||||
|
if srcReg.idx >= 8 {
|
||||||
|
rBit = 1
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
|
||||||
|
}
|
||||||
|
|
||||||
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
|
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
|
||||||
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
|
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
|
||||||
// move uses the store-form layout (reg = source, rm = destination), matching
|
// move uses the store-form layout (reg = source, rm = destination), matching
|
||||||
|
|||||||
+64
-5
@@ -19,6 +19,20 @@ func vreg(t *testing.T, name string) Reg {
|
|||||||
return r
|
return r
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// x86asmUnrecognised lists the VEX mnemonics whose machine code the
|
||||||
|
// golang.org/x/arch decoder cannot resolve; their bytes are verified against
|
||||||
|
// go tool asm in the ground-truth tests instead.
|
||||||
|
var x86asmUnrecognised = map[string]bool{
|
||||||
|
"ANDNL": true,
|
||||||
|
"ANDNQ": true,
|
||||||
|
"MULXL": true,
|
||||||
|
"MULXQ": true,
|
||||||
|
"RORXL": true,
|
||||||
|
"RORXQ": true,
|
||||||
|
"VFMADD213SD": true,
|
||||||
|
"VFNMADD231SD": true,
|
||||||
|
}
|
||||||
|
|
||||||
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
|
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
|
||||||
// instruction and verifies it round-trips through the x86 decoder to the same
|
// instruction and verifies it round-trips through the x86 decoder to the same
|
||||||
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
|
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
|
||||||
@@ -37,8 +51,15 @@ func TestVexNDS3(t *testing.T) {
|
|||||||
t.Errorf("%s: Encode: %v", mnem, err)
|
t.Errorf("%s: Encode: %v", mnem, err)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
// The x86 decoder's table lacks a handful of rows the Go assembler
|
||||||
|
// emits (the scalar 213/231 FMA spellings among them); those are
|
||||||
|
// pinned byte for byte against go tool asm in TestVexGroundTruth
|
||||||
|
// instead of round-tripped here.
|
||||||
inst, err := x86asm.Decode(code, 64)
|
inst, err := x86asm.Decode(code, 64)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
if strings.Contains(err.Error(), "unrecognized instruction") && x86asmUnrecognised[mnem] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
@@ -173,7 +194,7 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""},
|
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""},
|
||||||
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""},
|
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""},
|
||||||
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""},
|
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""},
|
||||||
// Floating point (packed and scalar) and FMA — same NDS form, the pp
|
// Floating point (packed and scalar) and FMA; same NDS form, the pp
|
||||||
// bits and map select the operation.
|
// bits and map select the operation.
|
||||||
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""},
|
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""},
|
||||||
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""},
|
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""},
|
||||||
@@ -184,6 +205,38 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
|
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
|
||||||
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
|
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
|
||||||
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
|
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
|
||||||
|
{"VFMADD213SD X0,X1,X2", "VFMADD213SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1a9d0", ""},
|
||||||
|
{"VFNMADD231SD X0,X1,X2", "VFNMADD231SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1bdd0", ""},
|
||||||
|
// Packed single XOR and byte compare (NDS form).
|
||||||
|
{"VXORPS Y0,Y1,Y2", "VXORPS", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f457d0", ""},
|
||||||
|
{"VPCMPEQB Y0,Y1,Y2", "VPCMPEQB", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f574d0", ""},
|
||||||
|
// Octa byte shifts (vvvv carries the destination).
|
||||||
|
{"VPSLLDQ $2,X0,X1", "VPSLLDQ", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "c5f173f802", ""},
|
||||||
|
{"VPSRLDQ $2,Y0,Y1", "VPSRLDQ", []Operand{Imm(2), vreg(t, "Y0"), vreg(t, "Y1")}, "c5f573d802", ""},
|
||||||
|
// Two-source shuffle, blend and carry-less multiply (NDS + imm8).
|
||||||
|
{"VPERM2F128 $3,Y0,Y1,Y2", "VPERM2F128", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37506d003", ""},
|
||||||
|
{"VPBLENDD $3,X0,X1,X2", "VPBLENDD", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37102d003", ""},
|
||||||
|
{"VPBLENDD $3,Y0,Y1,Y2", "VPBLENDD", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37502d003", ""},
|
||||||
|
{"VPCLMULQDQ $0,X0,X1,X2", "VPCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37144d000", ""},
|
||||||
|
{"VGF2P8AFFINEQB $0,X0,X1,X2", "VGF2P8AFFINEQB", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e3f1ced000", ""},
|
||||||
|
// Two-operand test and the non-temporal and broadcast stores.
|
||||||
|
{"VPTEST X0,X1", "VPTEST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c4e27917c8", ""},
|
||||||
|
{"VPTEST Y0,Y1", "VPTEST", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c4e27d17c8", ""},
|
||||||
|
{"VMOVNTDQ Y0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "Y0"), Ptr(AX, 0, 32)}, "c5fde700", ""},
|
||||||
|
{"VMOVNTDQ X0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "c5f9e700", ""},
|
||||||
|
{"VBROADCASTI128 (AX),Y1", "VBROADCASTI128", []Operand{Ptr(AX, 0, 16), vreg(t, "Y1")}, "c4e27d5a08", ""},
|
||||||
|
// Aligned integer move and the full zeroing form.
|
||||||
|
{"VMOVDQA X0,X1", "VMOVDQA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c5f97fc1", ""},
|
||||||
|
{"VMOVDQA (AX),X1", "VMOVDQA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "c5f96f08", ""},
|
||||||
|
{"VMOVDQA Y0,Y1", "VMOVDQA", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c5fd7fc1", ""},
|
||||||
|
{"VZEROALL", "VZEROALL", []Operand{}, "c5fc77", ""},
|
||||||
|
// BMI1/BMI2 general-register VEX forms.
|
||||||
|
{"ANDNL AX,BX,CX", "ANDNL", []Operand{AX, BX, CX}, "c4e260f2c8", ""},
|
||||||
|
{"ANDNQ AX,BX,CX", "ANDNQ", []Operand{AX, BX, CX}, "c4e2e0f2c8", ""},
|
||||||
|
{"MULXL AX,BX,CX", "MULXL", []Operand{AX, BX, CX}, "c4e263f6c8", ""},
|
||||||
|
{"MULXQ AX,BX,CX", "MULXQ", []Operand{AX, BX, CX}, "c4e2e3f6c8", ""},
|
||||||
|
{"RORXL $3,AX,CX", "RORXL", []Operand{Imm(3), AX, CX}, "c4e37bf0c803", ""},
|
||||||
|
{"RORXQ $3,AX,CX", "RORXQ", []Operand{Imm(3), AX, CX}, "c4e3fbf0c803", ""},
|
||||||
// Two-operand reg/rm form (v̄vvv must be 1111).
|
// Two-operand reg/rm form (v̄vvv must be 1111).
|
||||||
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
|
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
|
||||||
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
|
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
|
||||||
@@ -217,7 +270,7 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""},
|
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""},
|
||||||
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""},
|
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""},
|
||||||
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""},
|
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""},
|
||||||
// Moves — each direction picks its own opcode and VEX.W.
|
// Moves; each direction picks its own opcode and VEX.W.
|
||||||
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""},
|
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""},
|
||||||
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""},
|
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""},
|
||||||
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""},
|
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""},
|
||||||
@@ -234,7 +287,7 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""},
|
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""},
|
||||||
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""},
|
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""},
|
||||||
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""},
|
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""},
|
||||||
// Packed double arithmetic and unpack — the NDS form, the opcode
|
// Packed double arithmetic and unpack; the NDS form, the opcode
|
||||||
// selects the operation.
|
// selects the operation.
|
||||||
{"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""},
|
{"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""},
|
||||||
{"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""},
|
{"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""},
|
||||||
@@ -255,12 +308,12 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
{"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""},
|
{"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""},
|
||||||
{"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""},
|
{"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""},
|
||||||
{"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""},
|
{"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""},
|
||||||
// VMOVDDUP — duplicate the low double (reg=dst, rm=src, F2 pp).
|
// VMOVDDUP; duplicate the low double (reg=dst, rm=src, F2 pp).
|
||||||
{"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""},
|
{"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""},
|
||||||
{"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""},
|
{"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""},
|
||||||
{"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""},
|
{"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""},
|
||||||
// Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3
|
// Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3
|
||||||
// prefix — see the table comment), DQ→PD.
|
// prefix; see the table comment), DQ→PD.
|
||||||
{"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""},
|
{"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""},
|
||||||
{"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""},
|
{"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""},
|
||||||
{"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""},
|
{"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""},
|
||||||
@@ -287,6 +340,12 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
inst, err := x86asm.Decode(code, 64)
|
inst, err := x86asm.Decode(code, 64)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
// The decoder's AVX/BMI table lacks a few rows the Go
|
||||||
|
// assembler emits (the GPR VEX forms and the scalar FMA
|
||||||
|
// spellings); their bytes are the ground truth here.
|
||||||
|
if x86asmUnrecognised[c.mnem] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
|
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|||||||
+2
-1
@@ -114,6 +114,7 @@ type Symbol struct {
|
|||||||
Pkg string // package prefix before the middle dot ("" = current package)
|
Pkg string // package prefix before the middle dot ("" = current package)
|
||||||
Name string // identifier without the middle dot or <>
|
Name string // identifier without the middle dot or <>
|
||||||
Static bool // the <> marker is present
|
Static bool // the <> marker is present
|
||||||
|
ABI string // the <NAME> ABI marker, e.g. ABIInternal ("" when absent)
|
||||||
Pseudo string // FP, SP, SB or PC ("" for a bare name)
|
Pseudo string // FP, SP, SB or PC ("" for a bare name)
|
||||||
Offset int64
|
Offset int64
|
||||||
HasOff bool
|
HasOff bool
|
||||||
@@ -156,5 +157,5 @@ type Address struct {
|
|||||||
Scale int // index scale; 0 when absent
|
Scale int // index scale; 0 when absent
|
||||||
Offset int64 // leading displacement, from off(base)
|
Offset int64 // leading displacement, from off(base)
|
||||||
HasOff bool // a leading displacement is present
|
HasOff bool // a leading displacement is present
|
||||||
Shift string // verbatim arm64 shift suffix, e.g. "<<2"
|
Shift string // verbatim arm64 shift suffix, e.g. "<< 2"
|
||||||
}
|
}
|
||||||
|
|||||||
+305
-8
@@ -37,21 +37,36 @@ import (
|
|||||||
// construction and are excluded from the diff; the other architectures list
|
// construction and are excluded from the diff; the other architectures list
|
||||||
// their conditional branches outright.
|
// their conditional branches outright.
|
||||||
func cmdAuditInstructions(args []string) error {
|
func cmdAuditInstructions(args []string) error {
|
||||||
fs := newCommand("audit-instructions", "gasm audit-instructions [amd64|arm64|riscv64|loong64]", `
|
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64]", `
|
||||||
Compare the gasm encoder for the given architecture (default amd64) against
|
Compare the gasm encoder for the given architecture (default amd64) against
|
||||||
go tool asm and print the diff: superset encodings (gasm-only, shippable via
|
go tool asm and print the diff: superset encodings (gasm-only, shippable via
|
||||||
gasm asm --format goobj), known-but-unencodable names (the backlog) and go-
|
gasm asm --format goobj) and known-but-unencodable names (the backlog). The
|
||||||
only names (feature gaps). The Go side is probed black-box with a battery
|
Go side is probed black-box one bare mnemonic at a time, so the audit tracks
|
||||||
of bare mnemonics, so the audit tracks whatever toolchain `+"`go env GOROOT`"+`
|
whatever toolchain `+"`go env GOROOT`"+` provides; the gasm side answers from
|
||||||
provides.
|
the encoder table on amd64 and from trial assembly over a battery of operand
|
||||||
|
shapes elsewhere. Names go tool asm knows and gasm does not cannot be
|
||||||
|
enumerated by probing, because Go's table is visible only through names
|
||||||
|
already in the gasm table; the report closes with a note saying so.
|
||||||
|
|
||||||
|
With --corpus the audit changes shape: it assembles every .s file under the
|
||||||
|
given directory (default GOROOT/src) with the gasm encoder only, no
|
||||||
|
toolchain probing. A file whose name carries a recognisable _arch suffix is
|
||||||
|
attempted for that architecture; a file without one is attempted for all
|
||||||
|
four, exactly as a GOARCH build would compile it. The report gives the
|
||||||
|
per-architecture pass rates and the most common failure reasons, which drive
|
||||||
|
the encodability backlog by frequency rather than by table order.
|
||||||
`)
|
`)
|
||||||
|
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
|
||||||
if err := fs.Parse(args); err != nil {
|
if err := fs.Parse(args); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
if *corpus {
|
||||||
|
return cmdAuditCorpus(fs.Args())
|
||||||
|
}
|
||||||
archName := "amd64"
|
archName := "amd64"
|
||||||
switch n := len(fs.Args()); {
|
switch n := len(fs.Args()); {
|
||||||
case n > 1:
|
case n > 1:
|
||||||
return fmt.Errorf("audit-instructions takes at most one architecture argument")
|
return &usageError{fmt.Errorf("audit-instructions takes at most one architecture argument")}
|
||||||
case n == 1:
|
case n == 1:
|
||||||
archName = strings.ToLower(fs.Arg(0))
|
archName = strings.ToLower(fs.Arg(0))
|
||||||
}
|
}
|
||||||
@@ -97,7 +112,7 @@ provides.
|
|||||||
|
|
||||||
w := os.Stdout
|
w := os.Stdout
|
||||||
fmt.Fprintf(w, "gasm table (%s, families excluded): %d mnemonics\n", archName, len(names))
|
fmt.Fprintf(w, "gasm table (%s, families excluded): %d mnemonics\n", archName, len(names))
|
||||||
fmt.Fprintf(w, "gasm encodable: %d go tool asm recognized: %d\n", len(shared)+len(superset), countTrue(goKnown))
|
fmt.Fprintf(w, "gasm encodable: %d go tool asm recognised: %d\n", len(shared)+len(superset), countTrue(goKnown))
|
||||||
fmt.Fprintf(w, "shared: %d\n", len(shared))
|
fmt.Fprintf(w, "shared: %d\n", len(shared))
|
||||||
fmt.Fprintf(w, "\nSuperset encodings (gasm-only; ship via gasm asm --format goobj):\n")
|
fmt.Fprintf(w, "\nSuperset encodings (gasm-only; ship via gasm asm --format goobj):\n")
|
||||||
for _, n := range superset {
|
for _, n := range superset {
|
||||||
@@ -124,7 +139,7 @@ func auditArch(name string) (arch.Arch, error) {
|
|||||||
case "loong64", "loong":
|
case "loong64", "loong":
|
||||||
return arch.LOONG64, nil
|
return arch.LOONG64, nil
|
||||||
}
|
}
|
||||||
return arch.Unknown, fmt.Errorf("unknown architecture %q: want amd64, arm64, riscv64 or loong64", name)
|
return arch.Unknown, &usageError{fmt.Errorf("unknown architecture %q: want amd64, arm64, riscv64 or loong64", name)}
|
||||||
}
|
}
|
||||||
|
|
||||||
// goarchName maps an arch identifier onto its GOARCH spelling.
|
// goarchName maps an arch identifier onto its GOARCH spelling.
|
||||||
@@ -205,6 +220,13 @@ func probeGoAsm(goarch string, names []string) (map[string]bool, error) {
|
|||||||
cmd := exec.Command(asmBin, "-p", "probe", "-o", filepath.Join(dir, "probe.o"), probePath)
|
cmd := exec.Command(asmBin, "-p", "probe", "-o", filepath.Join(dir, "probe.o"), probePath)
|
||||||
cmd.Env = append(os.Environ(), "GOARCH="+goarch, "GOOS="+runtime.GOOS)
|
cmd.Env = append(os.Environ(), "GOARCH="+goarch, "GOOS="+runtime.GOOS)
|
||||||
out, _ := cmd.CombinedOutput()
|
out, _ := cmd.CombinedOutput()
|
||||||
|
// The expected failure mode is a non-zero exit with compiler diagnostics
|
||||||
|
// on stdout; empty output means the probe broke at the exec level (a
|
||||||
|
// killed child, a tool that would not start), and seeding every name as
|
||||||
|
// recognized on that silence would fake a clean audit.
|
||||||
|
if len(out) == 0 {
|
||||||
|
return nil, fmt.Errorf("go tool asm probe for GOARCH=%s produced no output", goarch)
|
||||||
|
}
|
||||||
|
|
||||||
result := map[string]bool{}
|
result := map[string]bool{}
|
||||||
for _, name := range names {
|
for _, name := range names {
|
||||||
@@ -244,18 +266,62 @@ func probeShapes(a arch.Arch) []string {
|
|||||||
// and takes R register spellings.
|
// and takes R register spellings.
|
||||||
"EQ, R0, R1, R2", "EQ, R0, R1", "EQ, R0",
|
"EQ, R0, R1, R2", "EQ, R0, R1", "EQ, R0",
|
||||||
"GE, F0, F1, F2", "NE, F0, F1, $0",
|
"GE, F0, F1, F2", "NE, F0, F1, $0",
|
||||||
|
// Pairs, acquire/release and exclusive atomics, LSE-AL forms.
|
||||||
|
"(R0), R1", "R0, (R1)", "R1, (R2), R3", "(R2, R3), 8(R1)",
|
||||||
|
"8(R1), (R2, R3)", "R1, R2, (R3)", "(R0)",
|
||||||
|
// System operations and their register/operand names.
|
||||||
|
"$4, R1, p2", "$35943", "$1", "$1, SPSel", "SPSel, R0",
|
||||||
|
"IVAC, R0", "(R0), PLDL1KEEP", "R1, R2, R3, R4",
|
||||||
|
// SIMD element, structure and literal-pool forms.
|
||||||
|
"(R0), [V1.B16]", "[V1.B16], (R0)", "V13.S[0], R1",
|
||||||
|
"R1, V2.B[3]", "$4, V1.B16, V2.B16", "V1.B16, (R0)",
|
||||||
|
"(R0), V1.B16", "",
|
||||||
|
// The spellings GOROOT's own kernels use, from the
|
||||||
|
// differential kernels this table was proven against.
|
||||||
|
"R0, p2", "R0, R1", "F0, F1, F2, F3", "$4, V1.B16, V2.B16, V3.B16, V4.B16",
|
||||||
|
"(R0), [V0.B8, V1.B8, V2.B8, V3.B8]", "$1, $2, V1",
|
||||||
|
"R0, R1, p2", "p2, R1", "$1234, R1", "DCZID_EL0, R1",
|
||||||
|
"$0", "R1, $4, EQ", "$33, R1, $25, R2", "$4, R1, p2",
|
||||||
|
"$4, V1.B8, V2.B8, V3.B8", "$63, V1.D2, V2.D2, V3.D2",
|
||||||
|
"V1.B16, [V2.B16], V3.B16", "V1.B8, [V2.B16, V3.B16], V4.B8",
|
||||||
|
"$4, V1.B16, V2.B16, V3.B16", "$15, V1", "V1, V2, p2",
|
||||||
|
"R0, R1, $1, $4, p2",
|
||||||
}
|
}
|
||||||
case arch.RISCV:
|
case arch.RISCV:
|
||||||
return []string{
|
return []string{
|
||||||
"X5, X6, X7", "X5, X6", "X5", "$1, X5", "X5, (X6)", "$1, X5, X6",
|
"X5, X6, X7", "X5, X6", "X5", "$1, X5", "X5, (X6)", "$1, X5, X6",
|
||||||
"(X5), X6", "F0, F1, F2", "F0, F1", "p2", "X1, p2", "X0, p2",
|
"(X5), X6", "F0, F1, F2", "F0, F1", "p2", "X1, p2", "X0, p2",
|
||||||
"X5, X6, p2", "p2(SB)",
|
"X5, X6, p2", "p2(SB)",
|
||||||
|
// AMO atomics: destination, base, source.
|
||||||
|
"R5, (R4), R6", "X5, (X4), X6",
|
||||||
|
// Segment stores take the first vector register aligned
|
||||||
|
// to the segment count, as the toolchain requires.
|
||||||
|
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
|
||||||
|
// The FP multiply-add family takes four registers.
|
||||||
|
"F0, F1, F2, F3",
|
||||||
|
// The RVV slice: register, vector-register and vtype forms.
|
||||||
|
"V1, V2, V3", "V1, X5, V2", "V1", "V1, (X5)", "(X5), V1",
|
||||||
|
"$15, V1", "$15", "V1, V2", "V1, X5",
|
||||||
|
"X5, X6, p2", "R5, R6, p2",
|
||||||
|
"X5, E8, M8, TA, MA, X6", "$4, E32, M1, TA, MA, X1",
|
||||||
|
"(X5), X6, V1, V2",
|
||||||
|
"",
|
||||||
}
|
}
|
||||||
case arch.LOONG64:
|
case arch.LOONG64:
|
||||||
return []string{
|
return []string{
|
||||||
"R4, R5, R6", "R4, R5", "R4", "$1, R4", "R4, (R5)", "(R4), R5",
|
"R4, R5, R6", "R4, R5", "R4", "$1, R4", "R4, (R5)", "(R4), R5",
|
||||||
"F0, F1, F2", "F0, F1", "p2", "R1, p2", "R4, p2",
|
"F0, F1, F2", "F0, F1", "p2", "R1, p2", "R4, p2",
|
||||||
"$1, R4, R5, R6", "$65536, R4", "R4, R5, p2", "p2(SB)",
|
"$1, R4, R5, R6", "$65536, R4", "R4, R5, p2", "p2(SB)",
|
||||||
|
// AMO atomics: destination, base, source.
|
||||||
|
"R5, (R4), R6", "X5, (X4), X6",
|
||||||
|
// Segment stores take the first vector register aligned
|
||||||
|
// to the segment count, as the toolchain requires.
|
||||||
|
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
|
||||||
|
// The LSX and LASX banks share the 5-bit numbering with F.
|
||||||
|
"V1, V2, V3", "X1, X2, X3", "V1, V2", "X1, X2", "V1", "X1",
|
||||||
|
// The vector compare-to-flag forms land in an FCC register.
|
||||||
|
"V1, FCC0", "X1, FCC0",
|
||||||
|
"",
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
@@ -305,3 +371,234 @@ func gasmAssembles(a arch.Arch, name, shape string) bool {
|
|||||||
func sanitize(name string) string {
|
func sanitize(name string) string {
|
||||||
return strings.NewReplacer(".", "_", "$", "_").Replace(name)
|
return strings.NewReplacer(".", "_", "$", "_").Replace(name)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- corpus audit -----------------------------------------------------------
|
||||||
|
|
||||||
|
// corpusTarget is one architecture row of the corpus report.
|
||||||
|
type corpusTarget struct {
|
||||||
|
a arch.Arch
|
||||||
|
name string
|
||||||
|
}
|
||||||
|
|
||||||
|
// corpusTally accumulates one architecture's attempts over the corpus.
|
||||||
|
type corpusTally struct {
|
||||||
|
attempted int
|
||||||
|
assembled int
|
||||||
|
reasons map[string]int // failure reason → count
|
||||||
|
example map[string]string // failure reason → one representative file
|
||||||
|
}
|
||||||
|
|
||||||
|
func (t *corpusTally) fail(path, reason string) {
|
||||||
|
t.reasons[reason]++
|
||||||
|
if t.example[reason] == "" {
|
||||||
|
t.example[reason] = path
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// cmdAuditCorpus implements audit-instructions --corpus.
|
||||||
|
func cmdAuditCorpus(args []string) error {
|
||||||
|
if len(args) > 1 {
|
||||||
|
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
|
||||||
|
}
|
||||||
|
root := ""
|
||||||
|
if len(args) == 1 {
|
||||||
|
root = args[0]
|
||||||
|
} else {
|
||||||
|
out, err := exec.Command("go", "env", "GOROOT").Output()
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("locate GOROOT: %w", err)
|
||||||
|
}
|
||||||
|
root = filepath.Join(strings.TrimSpace(string(out)), "src")
|
||||||
|
}
|
||||||
|
stats, err := runCorpusAudit(root)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
printCorpusStats(stats)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// corpusStats is the outcome of one corpus audit run.
|
||||||
|
type corpusStats struct {
|
||||||
|
root string
|
||||||
|
files int
|
||||||
|
generic int // files attempted for all four architectures
|
||||||
|
otherPort int // files named for another Go port: never attempted
|
||||||
|
full int // files that assembled for every target architecture
|
||||||
|
targets []corpusTarget
|
||||||
|
tallies []*corpusTally
|
||||||
|
}
|
||||||
|
|
||||||
|
// runCorpusAudit assembles every .s file under root and returns the stats.
|
||||||
|
// goPortSuffixes lists every architecture the Go project ports to. A file
|
||||||
|
// named for one of them belongs to that port's build, not to the generic
|
||||||
|
// set, even when gasm does not support the architecture.
|
||||||
|
var goPortSuffixes = []string{
|
||||||
|
"386", "amd64", "arm", "arm64", "loong64", "mips", "mips64",
|
||||||
|
"mips64le", "mipsle", "ppc64", "ppc64le", "riscv", "riscv64",
|
||||||
|
"s390x", "wasm",
|
||||||
|
}
|
||||||
|
|
||||||
|
// otherPortFile reports whether the file's name carries a Go-architecture
|
||||||
|
// suffix gasm does not support.
|
||||||
|
func otherPortFile(path string) bool {
|
||||||
|
base := path
|
||||||
|
if i := strings.LastIndexByte(base, '/'); i >= 0 {
|
||||||
|
base = base[i+1:]
|
||||||
|
}
|
||||||
|
for _, sfx := range goPortSuffixes {
|
||||||
|
if strings.HasSuffix(base, "_"+sfx+".s") {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
func runCorpusAudit(root string) (*corpusStats, error) {
|
||||||
|
files, err := asmFiles(root)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
targets := []corpusTarget{
|
||||||
|
{arch.AMD64, "amd64"},
|
||||||
|
{arch.ARM64, "arm64"},
|
||||||
|
{arch.RISCV, "riscv64"},
|
||||||
|
{arch.LOONG64, "loong64"},
|
||||||
|
}
|
||||||
|
tallies := make([]*corpusTally, len(targets))
|
||||||
|
for i := range tallies {
|
||||||
|
tallies[i] = &corpusTally{reasons: map[string]int{}, example: map[string]string{}}
|
||||||
|
}
|
||||||
|
// full is the north-star number: a file counts when every architecture
|
||||||
|
// its name allows assembles it.
|
||||||
|
full, generic, otherPort := 0, 0, 0
|
||||||
|
|
||||||
|
for _, path := range files {
|
||||||
|
src, err := readSource(path)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
f, errs := parser.Parse(path, src)
|
||||||
|
|
||||||
|
var wanted []int // indexes into targets
|
||||||
|
if a := arch.FromFilename(path); a != arch.Unknown {
|
||||||
|
for i, tg := range targets {
|
||||||
|
if tg.a == a {
|
||||||
|
wanted = append(wanted, i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} else if otherPortFile(path) {
|
||||||
|
// A file named for a Go port gasm does not support (arm,
|
||||||
|
// 386, s390x, ...) is compiled by no supported-arch build,
|
||||||
|
// so it is neither generic nor a per-arch attempt: counting
|
||||||
|
// it as generic would make the headline unreachably low
|
||||||
|
// for reasons no supported target can fix.
|
||||||
|
otherPort++
|
||||||
|
} else {
|
||||||
|
generic++
|
||||||
|
for i := range targets {
|
||||||
|
wanted = append(wanted, i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
ok := true
|
||||||
|
for _, i := range wanted {
|
||||||
|
tg, t := targets[i], tallies[i]
|
||||||
|
t.attempted++
|
||||||
|
var err error
|
||||||
|
if len(errs) > 0 {
|
||||||
|
err = errs[0] // a parse failure is a failure for every target
|
||||||
|
} else {
|
||||||
|
_, err = assembleFile(tg.a, f)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
ok = false
|
||||||
|
t.fail(path, corpusReason(err))
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
t.assembled++
|
||||||
|
}
|
||||||
|
if ok && len(wanted) > 0 {
|
||||||
|
full++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return &corpusStats{
|
||||||
|
root: root,
|
||||||
|
files: len(files),
|
||||||
|
generic: generic,
|
||||||
|
otherPort: otherPort,
|
||||||
|
full: full,
|
||||||
|
targets: targets,
|
||||||
|
tallies: tallies,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// printCorpusStats renders the corpus audit report.
|
||||||
|
func printCorpusStats(s *corpusStats) {
|
||||||
|
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d named for other Go ports, never attempted)\n", s.root, s.files, s.generic, s.otherPort)
|
||||||
|
// The rate is over the files a supported build would attempt: the
|
||||||
|
// other ports' files sit in the count for completeness but can never
|
||||||
|
// assemble, so counting them in the denominator would report the gap
|
||||||
|
// of architectures gasm deliberately does not target.
|
||||||
|
attemptable := max(s.files-s.otherPort, 1)
|
||||||
|
fmt.Printf(" assemble for every target architecture: %d of %d attemptable (%.1f%%)\n", s.full, attemptable, 100*float64(s.full)/float64(attemptable))
|
||||||
|
for i, tg := range s.targets {
|
||||||
|
t := s.tallies[i]
|
||||||
|
fmt.Printf(" %s: %d/%d attempted\n", tg.name, t.assembled, t.attempted)
|
||||||
|
for _, r := range topReasons(t) {
|
||||||
|
fmt.Printf(" %4d %s\n", t.reasons[r], r)
|
||||||
|
fmt.Printf(" e.g. %s\n", t.example[r])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// corpusReason buckets an assembly or parse failure for the histogram.
|
||||||
|
func corpusReason(err error) string {
|
||||||
|
msg := err.Error()
|
||||||
|
switch {
|
||||||
|
case strings.Contains(msg, "unsupported"), strings.Contains(msg, "cannot encode"):
|
||||||
|
return "instruction not encodable"
|
||||||
|
case strings.Contains(msg, "undefined label"):
|
||||||
|
return "undefined label"
|
||||||
|
case strings.Contains(msg, "undefined symbol"), strings.Contains(msg, "external symbol"), strings.Contains(msg, "file-level assembly"):
|
||||||
|
return "undefined symbol or external"
|
||||||
|
case strings.Contains(msg, "operand"), strings.Contains(msg, "operand form"):
|
||||||
|
return "unsupported operand form"
|
||||||
|
default:
|
||||||
|
return "other: " + firstLine(msg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// topReasons returns at most five reasons, most frequent first.
|
||||||
|
func topReasons(t *corpusTally) []string {
|
||||||
|
type kv struct {
|
||||||
|
k string
|
||||||
|
n int
|
||||||
|
}
|
||||||
|
var kvs []kv
|
||||||
|
for k, n := range t.reasons {
|
||||||
|
kvs = append(kvs, kv{k, n})
|
||||||
|
}
|
||||||
|
slices.SortFunc(kvs, func(a, b kv) int { return b.n - a.n })
|
||||||
|
if len(kvs) > 5 {
|
||||||
|
kvs = kvs[:5]
|
||||||
|
}
|
||||||
|
out := make([]string, len(kvs))
|
||||||
|
for i, kv := range kvs {
|
||||||
|
out[i] = kv.k
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// firstLine returns the first line of an error message, truncated.
|
||||||
|
func firstLine(msg string) string {
|
||||||
|
if i := strings.IndexByte(msg, '\n'); i >= 0 {
|
||||||
|
msg = msg[:i]
|
||||||
|
}
|
||||||
|
if len(msg) > 80 {
|
||||||
|
msg = msg[:80]
|
||||||
|
}
|
||||||
|
return msg
|
||||||
|
}
|
||||||
|
|||||||
+11
-3
@@ -24,9 +24,10 @@ function in a traced subprocess (ptrace), then provides a REPL for
|
|||||||
single-stepping, breakpoints, register and memory inspection.
|
single-stepping, breakpoints, register and memory inspection.
|
||||||
|
|
||||||
REPL commands:
|
REPL commands:
|
||||||
break <label|addr> [if <reg> <op> <val>]
|
break <label|addr|line> [if <reg> <op> <val|reg|*addr>]
|
||||||
set a breakpoint, optionally conditional on a
|
set a breakpoint, optionally conditional on a
|
||||||
register comparison (reg-reg or reg-immediate)
|
comparison of one register against a constant,
|
||||||
|
another register, or the 8-byte word at *addr
|
||||||
delete <label|addr> remove a breakpoint
|
delete <label|addr> remove a breakpoint
|
||||||
info break list all breakpoints
|
info break list all breakpoints
|
||||||
step [n], s single-step n instructions (default 1)
|
step [n], s single-step n instructions (default 1)
|
||||||
@@ -53,7 +54,7 @@ REPL commands:
|
|||||||
bufSpec := fs.String("buf", "", "buffer specification: name:size:pattern[,name:size:pattern...] where pattern is zero, ones, seq, or hex")
|
bufSpec := fs.String("buf", "", "buffer specification: name:size:pattern[,name:size:pattern...] where pattern is zero, ones, seq, or hex")
|
||||||
script := fs.String("script", "", "run REPL commands from a file (one per line) and exit; '-' reads stdin")
|
script := fs.String("script", "", "run REPL commands from a file (one per line) and exit; '-' reads stdin")
|
||||||
cover := fs.Bool("cover", false, "run to completion with a breakpoint on every instruction and report which executed and how often")
|
cover := fs.Bool("cover", false, "run to completion with a breakpoint on every instruction and report which executed and how often")
|
||||||
timeout := fs.Duration("timeout", 0, "kill the debuggee after this duration (e.g. 30s); for headless --script runs")
|
timeout := fs.Duration("timeout", 0, "kill the debuggee after this duration (e.g. 30s); for headless --script runs; a timeout exits 3")
|
||||||
fs.Parse(args)
|
fs.Parse(args)
|
||||||
|
|
||||||
// --- Debuggee mode (internal, spawned by the debugger) ---
|
// --- Debuggee mode (internal, spawned by the debugger) ---
|
||||||
@@ -231,6 +232,13 @@ REPL commands:
|
|||||||
if sess.Exited() {
|
if sess.Exited() {
|
||||||
break
|
break
|
||||||
}
|
}
|
||||||
|
// A genuine signal-delivery-stop (a fault in the kernel): the
|
||||||
|
// run cannot make progress, because resuming would restart the
|
||||||
|
// faulting instruction and fault forever. Report and stop.
|
||||||
|
if sig := sess.LastSignal(); sig != 0 {
|
||||||
|
fmt.Printf("gasm debug: cover: stopped on signal %v\n", sig)
|
||||||
|
break
|
||||||
|
}
|
||||||
regs, rerr := sess.GetRegs()
|
regs, rerr := sess.GetRegs()
|
||||||
if rerr != nil {
|
if rerr != nil {
|
||||||
break
|
break
|
||||||
|
|||||||
+177
-88
@@ -10,6 +10,7 @@ package main
|
|||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
"flag"
|
"flag"
|
||||||
"fmt"
|
"fmt"
|
||||||
"io"
|
"io"
|
||||||
@@ -18,6 +19,7 @@ import (
|
|||||||
"os/exec"
|
"os/exec"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"runtime"
|
"runtime"
|
||||||
|
"runtime/debug"
|
||||||
"slices"
|
"slices"
|
||||||
"sort"
|
"sort"
|
||||||
"strconv"
|
"strconv"
|
||||||
@@ -36,9 +38,32 @@ import (
|
|||||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
||||||
)
|
)
|
||||||
|
|
||||||
// version is the release version, stamped at build time via
|
// version reports the release the toolchain recorded for this build: the
|
||||||
// -ldflags "-X main.version=…" (defaulting to the current release).
|
// tag on a tag, a pseudo-version below one, and (devel) outside version
|
||||||
var version = "0.33.0"
|
// control. Nothing is injected; the recorded value cannot go stale.
|
||||||
|
func version() string {
|
||||||
|
bi, ok := debug.ReadBuildInfo()
|
||||||
|
if !ok || bi.Main.Version == "" {
|
||||||
|
return "(devel)"
|
||||||
|
}
|
||||||
|
return bi.Main.Version
|
||||||
|
}
|
||||||
|
|
||||||
|
// usageError marks an error the caller's arguments caused, which exits 2
|
||||||
|
// instead of the 1 a runtime failure gets.
|
||||||
|
type usageError struct{ err error }
|
||||||
|
|
||||||
|
func (e *usageError) Error() string { return e.err.Error() }
|
||||||
|
func (e *usageError) Unwrap() error { return e.err }
|
||||||
|
|
||||||
|
// exitCodeFor maps an error onto the process exit status: 2 for a usage
|
||||||
|
// error, 1 for anything else.
|
||||||
|
func exitCodeFor(err error) int {
|
||||||
|
if _, ok := errors.AsType[*usageError](err); ok {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
if len(os.Args) < 2 {
|
if len(os.Args) < 2 {
|
||||||
@@ -69,12 +94,12 @@ func main() {
|
|||||||
case "audit-instructions":
|
case "audit-instructions":
|
||||||
if err := cmdAuditInstructions(os.Args[2:]); err != nil {
|
if err := cmdAuditInstructions(os.Args[2:]); err != nil {
|
||||||
fmt.Fprintln(os.Stderr, err)
|
fmt.Fprintln(os.Stderr, err)
|
||||||
os.Exit(1)
|
os.Exit(exitCodeFor(err))
|
||||||
}
|
}
|
||||||
case "scaffold":
|
case "scaffold":
|
||||||
if err := cmdScaffold(os.Args[2:]); err != nil {
|
if err := cmdScaffold(os.Args[2:]); err != nil {
|
||||||
fmt.Fprintln(os.Stderr, err)
|
fmt.Fprintln(os.Stderr, err)
|
||||||
os.Exit(1)
|
os.Exit(exitCodeFor(err))
|
||||||
}
|
}
|
||||||
case "lsp":
|
case "lsp":
|
||||||
os.Exit(cmdLSP(os.Args[2:]))
|
os.Exit(cmdLSP(os.Args[2:]))
|
||||||
@@ -88,22 +113,22 @@ func main() {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// cmdVersion prints the release version.
|
// cmdVersion prints the recorded version.
|
||||||
func cmdVersion() int {
|
func cmdVersion() int {
|
||||||
fmt.Printf("gasm %s\n", version)
|
fmt.Printf("gasm %s\n", version())
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
// ANSI color helpers for terminal output.
|
// ANSI colour helpers for terminal output.
|
||||||
const (
|
const (
|
||||||
colorReset = "\033[0m"
|
colourReset = "\033[0m"
|
||||||
colorBold = "\033[1m"
|
colourBold = "\033[1m"
|
||||||
colorCyan = "\033[36m"
|
colourCyan = "\033[36m"
|
||||||
colorYellow = "\033[33m"
|
colourYellow = "\033[33m"
|
||||||
colorGray = "\033[90m"
|
colourGrey = "\033[90m"
|
||||||
)
|
)
|
||||||
|
|
||||||
// isTTY reports whether the writer is a terminal (for color output).
|
// isTTY reports whether the writer is a terminal (for colour output).
|
||||||
func isTTY(w io.Writer) bool {
|
func isTTY(w io.Writer) bool {
|
||||||
if f, ok := w.(*os.File); ok {
|
if f, ok := w.(*os.File); ok {
|
||||||
stat, _ := f.Stat()
|
stat, _ := f.Stat()
|
||||||
@@ -114,12 +139,12 @@ func isTTY(w io.Writer) bool {
|
|||||||
|
|
||||||
func usage(w io.Writer) {
|
func usage(w io.Writer) {
|
||||||
useColor := isTTY(w)
|
useColor := isTTY(w)
|
||||||
bold, cyan, yellow, gray, reset := "", "", "", "", ""
|
bold, cyan, yellow, grey, reset := "", "", "", "", ""
|
||||||
if useColor {
|
if useColor {
|
||||||
bold, cyan, yellow, gray, reset = colorBold, colorCyan, colorYellow, colorGray, colorReset
|
bold, cyan, yellow, grey, reset = colourBold, colourCyan, colourYellow, colourGrey, colourReset
|
||||||
}
|
}
|
||||||
|
|
||||||
fmt.Fprintf(w, "%sgasm %s%s: developer tooling for Go's Plan 9 assembler (GAsm)%s\n\n", bold, version, reset, reset)
|
fmt.Fprintf(w, "%sgasm %s%s: developer tooling for Go's Plan 9 assembler (GAsm)%s\n\n", bold, version(), reset, reset)
|
||||||
fmt.Fprintf(w, "gasm bundles a lexer, parser, formatter, linter, standalone assembler and\n")
|
fmt.Fprintf(w, "gasm bundles a lexer, parser, formatter, linter, standalone assembler and\n")
|
||||||
fmt.Fprintf(w, "language server for Plan 9 assembly into one self-contained binary.\n\n")
|
fmt.Fprintf(w, "language server for Plan 9 assembly into one self-contained binary.\n\n")
|
||||||
|
|
||||||
@@ -145,12 +170,12 @@ func usage(w io.Writer) {
|
|||||||
{"version", "print the version (same as --version)"},
|
{"version", "print the version (same as --version)"},
|
||||||
}
|
}
|
||||||
for _, c := range commands {
|
for _, c := range commands {
|
||||||
fmt.Fprintf(w, " %s%-10s%s %s%s%s\n", cyan, c.name, reset, gray, c.desc, reset)
|
fmt.Fprintf(w, " %s%-10s%s %s%s%s\n", cyan, c.name, reset, grey, c.desc, reset)
|
||||||
}
|
}
|
||||||
|
|
||||||
fmt.Fprintf(w, "\n%sFlags:%s\n", yellow, reset)
|
fmt.Fprintf(w, "\n%sFlags:%s\n", yellow, reset)
|
||||||
fmt.Fprintf(w, " %s-h, --help%s %sshow this help%s\n", cyan, reset, gray, reset)
|
fmt.Fprintf(w, " %s-h, --help%s %sshow this help%s\n", cyan, reset, grey, reset)
|
||||||
fmt.Fprintf(w, " %s-V, --version%s %sprint the version%s\n", cyan, reset, gray, reset)
|
fmt.Fprintf(w, " %s-V, --version%s %sprint the version%s\n", cyan, reset, grey, reset)
|
||||||
|
|
||||||
fmt.Fprintf(w, "\nRun \"gasm <command> -h\" for a command's usage and flags.\n\n")
|
fmt.Fprintf(w, "\nRun \"gasm <command> -h\" for a command's usage and flags.\n\n")
|
||||||
|
|
||||||
@@ -164,7 +189,7 @@ func usage(w io.Writer) {
|
|||||||
}
|
}
|
||||||
for _, e := range examples {
|
for _, e := range examples {
|
||||||
if e.desc != "" {
|
if e.desc != "" {
|
||||||
fmt.Fprintf(w, " %s%s%s %s%s%s\n", cyan, e.cmd, reset, gray, e.desc, reset)
|
fmt.Fprintf(w, " %s%s%s %s%s%s\n", cyan, e.cmd, reset, grey, e.desc, reset)
|
||||||
} else {
|
} else {
|
||||||
fmt.Fprintf(w, " %s%s%s\n", cyan, e.cmd, reset)
|
fmt.Fprintf(w, " %s%s%s\n", cyan, e.cmd, reset)
|
||||||
}
|
}
|
||||||
@@ -442,7 +467,7 @@ hover, document symbols, diagnostics and semantic-token highlighting.
|
|||||||
`)
|
`)
|
||||||
fs.Parse(args)
|
fs.Parse(args)
|
||||||
srv := lsp.New(os.Stdin, os.Stdout)
|
srv := lsp.New(os.Stdin, os.Stdout)
|
||||||
srv.SetVersion(version)
|
srv.SetVersion(version())
|
||||||
if err := srv.Run(); err != nil {
|
if err := srv.Run(); err != nil {
|
||||||
fmt.Fprintln(os.Stderr, "gasm lsp:", err)
|
fmt.Fprintln(os.Stderr, "gasm lsp:", err)
|
||||||
return 1
|
return 1
|
||||||
@@ -451,7 +476,7 @@ hover, document symbols, diagnostics and semantic-token highlighting.
|
|||||||
}
|
}
|
||||||
|
|
||||||
func cmdAsm(args []string) int {
|
func cmdAsm(args []string) int {
|
||||||
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>", `
|
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>", `
|
||||||
Assemble FILE without the Go toolchain: every TEXT function is encoded to
|
Assemble FILE without the Go toolchain: every TEXT function is encoded to
|
||||||
machine code and printed as a hex dump. Supported architectures: amd64
|
machine code and printed as a hex dump. Supported architectures: amd64
|
||||||
(including VEX/AVX2 and EVEX/AVX-512), arm64 (AArch64 integer, FP,
|
(including VEX/AVX2 and EVEX/AVX-512), arm64 (AArch64 integer, FP,
|
||||||
@@ -461,21 +486,41 @@ func cmdAsm(args []string) int {
|
|||||||
With -o the output is written to a file instead. The --format flag selects
|
With -o the output is written to a file instead. The --format flag selects
|
||||||
what is written: raw (the default) concatenates the functions and the data
|
what is written: raw (the default) concatenates the functions and the data
|
||||||
section into one self-consistent image; elf emits a relocatable object
|
section into one self-consistent image; elf emits a relocatable object
|
||||||
(.text/.data sections, a symbol table and one PC32 relocation per
|
(.text/.data sections, a symbol table and one relocation per static-symbol
|
||||||
static-symbol reference) that links with the system toolchain; goobj emits
|
reference, in the architecture's own form: R_X86_64_PC32 on amd64,
|
||||||
the Go toolchain's own object format, which cmd/link consumes directly (it
|
R_AARCH64_*, R_RISCV_* or R_LARCH_* on the others) that links with the
|
||||||
requires -p, the package path, and the installed Go toolchain).
|
system toolchain; goobj emits the Go toolchain's own object format, which
|
||||||
|
cmd/link consumes directly (it requires -p, the package path, and the
|
||||||
|
installed Go toolchain: the object preamble is captured from go tool asm
|
||||||
|
and the format version from go version).
|
||||||
`)
|
`)
|
||||||
out := fs.String("o", "", "write the output to this file")
|
out := fs.String("o", "", "write the output to this file")
|
||||||
format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
|
format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
|
||||||
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
|
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
|
||||||
|
archName := fs.String("GOARCH", "", "target architecture: amd64, arm64, riscv64 or loong64 (overrides the file-name suffix)")
|
||||||
fs.Parse(args)
|
fs.Parse(args)
|
||||||
if fs.NArg() != 1 {
|
if fs.NArg() != 1 {
|
||||||
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>")
|
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>")
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
// The format is validated before anything else, so a bogus value exits 2
|
||||||
|
// with or without -o instead of silently dumping the hex of a raw image.
|
||||||
|
switch *format {
|
||||||
|
case "raw", "elf", "goobj":
|
||||||
|
default:
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm asm: unknown format %q (want raw, elf or goobj)\n", *format)
|
||||||
return 2
|
return 2
|
||||||
}
|
}
|
||||||
path := fs.Arg(0)
|
path := fs.Arg(0)
|
||||||
targetArch := arch.FromFilename(path)
|
targetArch := arch.FromFilename(path)
|
||||||
|
if *archName != "" {
|
||||||
|
a, err := auditArch(*archName)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm asm: %v\n", err)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
targetArch = a
|
||||||
|
}
|
||||||
src, err := readSource(path)
|
src, err := readSource(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||||
@@ -494,42 +539,48 @@ requires -p, the package path, and the installed Go toolchain).
|
|||||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
|
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
if len(img.Funcs) == 0 {
|
if len(img.Funcs) == 0 && len(img.Data) == 0 {
|
||||||
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions found")
|
// A file with neither code nor data assembles to nothing, which is
|
||||||
|
// almost always a wrong architecture rather than an intent.
|
||||||
|
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions or GLOBL data found")
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
for _, fn := range img.Funcs {
|
// Without -o the hex dump on stdout is the output; with -o the file is,
|
||||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
// and the dump is skipped, as the -o help text promises.
|
||||||
fmt.Printf("%s: %d bytes\n", fn.Name, fn.Size)
|
if *out == "" {
|
||||||
for i := 0; i < len(code); i += 16 {
|
for _, fn := range img.Funcs {
|
||||||
end := min(i+16, len(code))
|
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||||
fmt.Printf(" %04x:", i)
|
fmt.Printf("%s: %d bytes\n", fn.Name, fn.Size)
|
||||||
for _, b := range code[i:end] {
|
for i := 0; i < len(code); i += 16 {
|
||||||
fmt.Printf(" %02x", b)
|
end := min(i+16, len(code))
|
||||||
|
fmt.Printf(" %04x:", i)
|
||||||
|
for _, b := range code[i:end] {
|
||||||
|
fmt.Printf(" %02x", b)
|
||||||
|
}
|
||||||
|
fmt.Println()
|
||||||
}
|
}
|
||||||
fmt.Println()
|
|
||||||
}
|
}
|
||||||
}
|
if len(img.Data) > 0 {
|
||||||
if len(img.Data) > 0 {
|
fmt.Printf("data: %d bytes at 0x%x\n", len(img.Data), len(img.Code))
|
||||||
fmt.Printf("data: %d bytes at 0x%x\n", len(img.Data), len(img.Code))
|
for _, d := range f.Decls {
|
||||||
for _, d := range f.Decls {
|
g, ok := d.(*ast.Globl)
|
||||||
g, ok := d.(*ast.Globl)
|
if !ok || g.Name == nil || g.Name.Pseudo != "SB" {
|
||||||
if !ok || g.Name == nil || g.Name.Pseudo != "SB" {
|
continue
|
||||||
continue
|
}
|
||||||
|
size := 0
|
||||||
|
if g.Size != nil && g.Size.Imm.HasVal {
|
||||||
|
size = int(g.Size.Imm.Val)
|
||||||
|
}
|
||||||
|
fmt.Printf(" %s: %d bytes at 0x%x\n", g.Name.Name, size, img.Symbols[g.Name.Name])
|
||||||
}
|
}
|
||||||
size := 0
|
for i := 0; i < len(img.Data); i += 16 {
|
||||||
if g.Size != nil && g.Size.Imm.HasVal {
|
end := min(i+16, len(img.Data))
|
||||||
size = int(g.Size.Imm.Val)
|
fmt.Printf(" %04x:", len(img.Code)+i)
|
||||||
|
for _, b := range img.Data[i:end] {
|
||||||
|
fmt.Printf(" %02x", b)
|
||||||
|
}
|
||||||
|
fmt.Println()
|
||||||
}
|
}
|
||||||
fmt.Printf(" %s: %d bytes at 0x%x\n", g.Name.Name, size, img.Symbols[g.Name.Name])
|
|
||||||
}
|
|
||||||
for i := 0; i < len(img.Data); i += 16 {
|
|
||||||
end := min(i+16, len(img.Data))
|
|
||||||
fmt.Printf(" %04x:", len(img.Code)+i)
|
|
||||||
for _, b := range img.Data[i:end] {
|
|
||||||
fmt.Printf(" %02x", b)
|
|
||||||
}
|
|
||||||
fmt.Println()
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if *out != "" {
|
if *out != "" {
|
||||||
@@ -567,9 +618,6 @@ requires -p, the package path, and the installed Go toolchain).
|
|||||||
obj, err = img.GOObject(*pkg, path)
|
obj, err = img.GOObject(*pkg, path)
|
||||||
}
|
}
|
||||||
kind = "Go object"
|
kind = "Go object"
|
||||||
default:
|
|
||||||
fmt.Fprintf(os.Stderr, "gasm asm: unknown format %q (want raw, elf or goobj)\n", *format)
|
|
||||||
return 2
|
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintln(os.Stderr, "gasm asm:", err)
|
fmt.Fprintln(os.Stderr, "gasm asm:", err)
|
||||||
@@ -586,7 +634,7 @@ requires -p, the package path, and the installed Go toolchain).
|
|||||||
|
|
||||||
// cmdDiff compares the machine code of two assembly files.
|
// cmdDiff compares the machine code of two assembly files.
|
||||||
func cmdDiff(args []string) int {
|
func cmdDiff(args []string) int {
|
||||||
set := newCommand("diff", "gasm diff <file1.s> <file2.s>", `
|
set := newCommand("diff", "gasm diff [-GOARCH arch] <file1.s> <file2.s>", `
|
||||||
Compare the machine code produced by assembling two files.
|
Compare the machine code produced by assembling two files.
|
||||||
Shows which functions differ and the byte-level differences.
|
Shows which functions differ and the byte-level differences.
|
||||||
Useful for verifying that two implementations produce identical code,
|
Useful for verifying that two implementations produce identical code,
|
||||||
@@ -596,12 +644,22 @@ Use --map to compare functions whose names differ between the files,
|
|||||||
e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
|
e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
|
||||||
`)
|
`)
|
||||||
mapSpec := set.String("map", "", "comma-separated old=new pairs to match functions with different names")
|
mapSpec := set.String("map", "", "comma-separated old=new pairs to match functions with different names")
|
||||||
|
archName := set.String("GOARCH", "", "target architecture for both files: amd64, arm64, riscv64 or loong64")
|
||||||
set.Parse(args)
|
set.Parse(args)
|
||||||
if set.NArg() != 2 {
|
if set.NArg() != 2 {
|
||||||
fmt.Fprintln(os.Stderr, "usage: gasm diff <file1.s> <file2.s>")
|
fmt.Fprintln(os.Stderr, "usage: gasm diff [-GOARCH arch] <file1.s> <file2.s>")
|
||||||
return 2
|
return 2
|
||||||
}
|
}
|
||||||
path1, path2 := set.Arg(0), set.Arg(1)
|
path1, path2 := set.Arg(0), set.Arg(1)
|
||||||
|
forced := arch.Unknown
|
||||||
|
if *archName != "" {
|
||||||
|
a, err := auditArch(*archName)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm diff: %v\n", err)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
forced = a
|
||||||
|
}
|
||||||
|
|
||||||
// Parse the name mapping (file1 name → file2 name).
|
// Parse the name mapping (file1 name → file2 name).
|
||||||
nameMap := make(map[string]string)
|
nameMap := make(map[string]string)
|
||||||
@@ -617,12 +675,12 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Assemble both files.
|
// Assemble both files.
|
||||||
img1, err := assemblePath(path1)
|
img1, err := assemblePath(path1, forced)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err)
|
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
img2, err := assemblePath(path2)
|
img2, err := assemblePath(path2, forced)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err)
|
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err)
|
||||||
return 1
|
return 1
|
||||||
@@ -697,8 +755,9 @@ func assembleFile(targetArch arch.Arch, f *ast.File) (*asm.Image, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// assemblePath reads, parses and assembles a file (used by cmdDiff).
|
// assemblePath reads, parses and assembles a file (used by cmdDiff). A
|
||||||
func assemblePath(path string) (*asm.Image, error) {
|
// non-Unknown forced architecture overrides the file-name suffix.
|
||||||
|
func assemblePath(path string, forced arch.Arch) (*asm.Image, error) {
|
||||||
src, err := readSource(path)
|
src, err := readSource(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
@@ -710,7 +769,11 @@ func assemblePath(path string) (*asm.Image, error) {
|
|||||||
if len(errs) > 0 {
|
if len(errs) > 0 {
|
||||||
return nil, fmt.Errorf("parse errors")
|
return nil, fmt.Errorf("parse errors")
|
||||||
}
|
}
|
||||||
return assembleFile(arch.FromFilename(path), f)
|
target := forced
|
||||||
|
if target == arch.Unknown {
|
||||||
|
target = arch.FromFilename(path)
|
||||||
|
}
|
||||||
|
return assembleFile(target, f)
|
||||||
}
|
}
|
||||||
|
|
||||||
// printByteDiff shows the first few byte differences between two code blocks.
|
// printByteDiff shows the first few byte differences between two code blocks.
|
||||||
@@ -733,8 +796,8 @@ func cmdProfile(args []string) int {
|
|||||||
flagSet := newCommand("profile", "gasm profile <file.s>", `
|
flagSet := newCommand("profile", "gasm profile <file.s>", `
|
||||||
Show the basic-block structure of functions in an assembly file.
|
Show the basic-block structure of functions in an assembly file.
|
||||||
Lists each function's labels, their offsets, and the block boundaries.
|
Lists each function's labels, their offsets, and the block boundaries.
|
||||||
This is the static structure; for runtime execution counts, use
|
This is the static structure; for runtime execution counts use
|
||||||
gasm verify --fuzz which exercises the code paths.
|
gasm debug --cover, and for input coverage gasm verify --fuzz.
|
||||||
`)
|
`)
|
||||||
flagSet.Parse(args)
|
flagSet.Parse(args)
|
||||||
if flagSet.NArg() != 1 {
|
if flagSet.NArg() != 1 {
|
||||||
@@ -878,11 +941,40 @@ func compareGroundTruth(img *asm.Image, gt map[string][]byte) (matched, total, d
|
|||||||
goCmp[j] = 0
|
goCmp[j] = 0
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if bytes.Equal(gasmCmp, goCmp) {
|
// The toolchain pads text symbols to 16-byte boundaries with
|
||||||
|
// zeros, so a function whose size is not a multiple of 16
|
||||||
|
// carries trailing zeros in the ground truth that are not part
|
||||||
|
// of the encoding. Compare up to the shorter side and require
|
||||||
|
// the remainder of whichever is longer to be zero, so padding
|
||||||
|
// never masks a real difference.
|
||||||
|
cmpLen := min(len(gasmCmp), len(goCmp))
|
||||||
|
equal := bytes.Equal(gasmCmp[:cmpLen], goCmp[:cmpLen])
|
||||||
|
if equal {
|
||||||
|
for _, b := range gasmCmp[cmpLen:] {
|
||||||
|
if b != 0 {
|
||||||
|
equal = false
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if equal {
|
||||||
|
for _, b := range goCmp[cmpLen:] {
|
||||||
|
if b != 0 {
|
||||||
|
equal = false
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if equal {
|
||||||
matched++
|
matched++
|
||||||
if len(fn.Relocs) > 0 {
|
switch {
|
||||||
|
case len(fn.Relocs) > 0 && len(goCmp) > cmpLen:
|
||||||
|
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked, %d padding)\n", fn.Name, fn.Size, len(fn.Relocs), len(goCmp)-cmpLen)
|
||||||
|
case len(fn.Relocs) > 0:
|
||||||
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
|
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
|
||||||
} else {
|
case len(goCmp) > cmpLen:
|
||||||
|
fmt.Printf(" %s: MATCH (%d bytes, %d padding)\n", fn.Name, fn.Size, len(goCmp)-cmpLen)
|
||||||
|
default:
|
||||||
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
|
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
@@ -930,8 +1022,8 @@ that tolerate nil pointers and zero lengths in their arguments.
|
|||||||
With -abi, each function is called with sentinel values in the registers
|
With -abi, each function is called with sentinel values in the registers
|
||||||
the Go ABI fixes across calls (the frame pointer and the goroutine
|
the Go ABI fixes across calls (the frame pointer and the goroutine
|
||||||
pointer) plus a canary below SP; violations are reported. JIT-based
|
pointer) plus a canary below SP; violations are reported. JIT-based
|
||||||
checks run when the host matches the file's architecture (all but
|
checks run when the host matches the file's architecture, on all four
|
||||||
loong64, which is ground-truth only for now).
|
architectures.
|
||||||
|
|
||||||
With -fuzz, each function with a // func signature is differentially fuzzed
|
With -fuzz, each function with a // func signature is differentially fuzzed
|
||||||
against the go-tool-asm version in a subprocess (so a crash on a partial
|
against the go-tool-asm version in a subprocess (so a crash on a partial
|
||||||
@@ -944,7 +1036,8 @@ With -profile, the static basic-block structure is listed for each function.
|
|||||||
|
|
||||||
With -call, a single function is invoked with user-supplied buffers (-buf)
|
With -call, a single function is invoked with user-supplied buffers (-buf)
|
||||||
instead of the smoke/abi/fuzz sweeps. Useful for partial functions (e.g.
|
instead of the smoke/abi/fuzz sweeps. Useful for partial functions (e.g.
|
||||||
decoders) that crash on random input but should succeed on valid data.
|
decoders) that crash on random input but should succeed on valid data. The
|
||||||
|
function named must be NOSPLIT: a function with a stack frame is refused.
|
||||||
|
|
||||||
With -save-corpus (and -fuzz), every input that crashes or mismatches is
|
With -save-corpus (and -fuzz), every input that crashes or mismatches is
|
||||||
written to the directory as replayable JSON. -replay re-runs saved
|
written to the directory as replayable JSON. -replay re-runs saved
|
||||||
@@ -973,20 +1066,16 @@ each entry reproduces.
|
|||||||
path := set.Arg(0)
|
path := set.Arg(0)
|
||||||
targetArch := arch.FromFilename(path)
|
targetArch := arch.FromFilename(path)
|
||||||
// JIT execution runs when the host CPU matches the kernel's
|
// JIT execution runs when the host CPU matches the kernel's
|
||||||
// architecture, except loong64: its trampoline is implemented but not
|
// architecture; every trampoline is validated end to end under
|
||||||
// yet validated against real hardware (the Go runtime cannot start
|
// qemu-user emulation (the loong64 one included, via the raw-address
|
||||||
// under the available loong64 emulators), so those kernels take the
|
// leave handoff).
|
||||||
// toolchain-comparison path.
|
if targetArch != hostArch() {
|
||||||
if targetArch != hostArch() || targetArch == arch.LOONG64 {
|
// No JIT on this host: ground truth and profile remain available for
|
||||||
// No JIT on this host: ground truth and profile remain available.
|
// every architecture, because cmdVerifyNonJIT assembles and compares
|
||||||
// (loong64 is ground-truth-only everywhere for now: its trampoline
|
// against the toolchain without executing anything.
|
||||||
// is implemented but not yet validated against real hardware.)
|
|
||||||
switch targetArch {
|
switch targetArch {
|
||||||
case arch.RISCV, arch.LOONG64, arch.ARM64:
|
case arch.AMD64, arch.RISCV, arch.LOONG64, arch.ARM64:
|
||||||
return cmdVerifyNonJIT(path, targetArch, *groundTruth, *profile)
|
return cmdVerifyNonJIT(path, targetArch, *groundTruth, *profile)
|
||||||
case arch.AMD64:
|
|
||||||
fmt.Fprintln(os.Stderr, "gasm verify: JIT-based checks need an amd64 host; use --ground-truth here")
|
|
||||||
return 1
|
|
||||||
default:
|
default:
|
||||||
fmt.Fprintln(os.Stderr, "gasm verify: unsupported architecture")
|
fmt.Fprintln(os.Stderr, "gasm verify: unsupported architecture")
|
||||||
return 1
|
return 1
|
||||||
|
|||||||
+171
-3
@@ -13,6 +13,9 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"syscall"
|
"syscall"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||||
)
|
)
|
||||||
|
|
||||||
const clean = "#include \"textflag.h\"\n" +
|
const clean = "#include \"textflag.h\"\n" +
|
||||||
@@ -219,8 +222,9 @@ func TestCmdVersion(t *testing.T) {
|
|||||||
if code != 0 {
|
if code != 0 {
|
||||||
t.Fatalf("code = %d", code)
|
t.Fatalf("code = %d", code)
|
||||||
}
|
}
|
||||||
if !strings.Contains(out, version) {
|
got := version()
|
||||||
t.Errorf("version output %q does not mention %q", out, version)
|
if !strings.Contains(out, got) {
|
||||||
|
t.Errorf("version output %q does not mention %q", out, got)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -241,6 +245,96 @@ func TestCmdArgErrors(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestUsageExitCodes pins the exit-code contract for the commands whose main
|
||||||
|
// dispatches on a returned error: a wrong argument set exits 2, the same as
|
||||||
|
// the commands that count their arguments themselves, while a runtime
|
||||||
|
// failure (an unreadable file) keeps exit 1.
|
||||||
|
func TestUsageExitCodes(t *testing.T) {
|
||||||
|
for name, err := range map[string]error{
|
||||||
|
"audit-instructions extra argument": cmdAuditInstructions([]string{"amd64", "extra"}),
|
||||||
|
"audit-instructions unknown arch": cmdAuditInstructions([]string{"mips"}),
|
||||||
|
"audit-instructions corpus extra": cmdAuditInstructions([]string{"--corpus", "a", "b"}),
|
||||||
|
"scaffold no arguments": cmdScaffold(nil),
|
||||||
|
"scaffold extra arguments": cmdScaffold([]string{"differential", "a.s", "b.s"}),
|
||||||
|
} {
|
||||||
|
if err == nil {
|
||||||
|
t.Errorf("%s: expected an error", name)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if code := exitCodeFor(err); code != 2 {
|
||||||
|
t.Errorf("%s: exit code = %d, want 2 (err: %v)", name, code, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if err := cmdScaffold([]string{"differential", "/nonexistent/file.s"}); err == nil {
|
||||||
|
t.Error("scaffold on a missing file should fail")
|
||||||
|
} else if code := exitCodeFor(err); code != 1 {
|
||||||
|
t.Errorf("scaffold on a missing file: exit code = %d, want 1", code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCmdAsmFormatValidation checks that an unknown --format exits 2 with
|
||||||
|
// and without -o, instead of assembling and silently dumping a raw image.
|
||||||
|
func TestCmdAsmFormatValidation(t *testing.T) {
|
||||||
|
path := writeTemp(t, "f_amd64.s", clean)
|
||||||
|
out := filepath.Join(t.TempDir(), "f.bin")
|
||||||
|
if _, _, code := capture(func() int { return cmdAsm([]string{"--format", "bogus", path}) }); code != 2 {
|
||||||
|
t.Errorf("asm --format bogus without -o: code = %d, want 2", code)
|
||||||
|
}
|
||||||
|
if _, _, code := capture(func() int { return cmdAsm([]string{"--format", "bogus", "-o", out, path}) }); code != 2 {
|
||||||
|
t.Errorf("asm --format bogus with -o: code = %d, want 2", code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCmdAsmOutputFile pins the documented -o behaviour: the output goes to
|
||||||
|
// the file and stdout carries no hex dump; without -o the dump is the output.
|
||||||
|
func TestCmdAsmOutputFile(t *testing.T) {
|
||||||
|
path := writeTemp(t, "f_amd64.s", clean)
|
||||||
|
out := filepath.Join(t.TempDir(), "f.bin")
|
||||||
|
stdout, _, code := capture(func() int { return cmdAsm([]string{"-o", out, path}) })
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("code = %d", code)
|
||||||
|
}
|
||||||
|
if strings.Contains(stdout, "0000:") {
|
||||||
|
t.Errorf("stdout carries a hex dump despite -o:\n%s", stdout)
|
||||||
|
}
|
||||||
|
if !strings.Contains(stdout, "wrote ") {
|
||||||
|
t.Errorf("stdout misses the wrote line:\n%s", stdout)
|
||||||
|
}
|
||||||
|
b, err := os.ReadFile(out)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if len(b) == 0 {
|
||||||
|
t.Error("the output file is empty")
|
||||||
|
}
|
||||||
|
|
||||||
|
stdout, _, code = capture(func() int { return cmdAsm([]string{path}) })
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("without -o: code = %d", code)
|
||||||
|
}
|
||||||
|
if !strings.Contains(stdout, "0000:") {
|
||||||
|
t.Errorf("without -o the hex dump is missing:\n%s", stdout)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestVerifyNonJITAMD64GroundTruth drives the cross-architecture
|
||||||
|
// ground-truth path for an amd64 kernel: the path a host of any other
|
||||||
|
// architecture takes, which must compare against the toolchain rather than
|
||||||
|
// refuse to run.
|
||||||
|
func TestVerifyNonJITAMD64GroundTruth(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("runs go tool asm")
|
||||||
|
}
|
||||||
|
path := writeTemp(t, "f_amd64.s", clean)
|
||||||
|
out, _, code := capture(func() int { return cmdVerifyNonJIT(path, arch.AMD64, true, false) })
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("code = %d (%s)", code, out)
|
||||||
|
}
|
||||||
|
if !strings.Contains(out, "1/1 matched") {
|
||||||
|
t.Errorf("output misses the matched report:\n%s", out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestVerifySmokeCrashIsolation checks that a function faulting on its
|
// TestVerifySmokeCrashIsolation checks that a function faulting on its
|
||||||
// zeroed smoke arguments is reported as CRASH by a child process instead of
|
// zeroed smoke arguments is reported as CRASH by a child process instead of
|
||||||
// killing `gasm verify` itself.
|
// killing `gasm verify` itself.
|
||||||
@@ -274,7 +368,7 @@ func TestVerifySmokeCrashIsolation(t *testing.T) {
|
|||||||
}
|
}
|
||||||
if exitErr, ok := err.(*exec.ExitError); ok {
|
if exitErr, ok := err.(*exec.ExitError); ok {
|
||||||
if ws, ok := exitErr.Sys().(syscall.WaitStatus); ok && ws.Signaled() {
|
if ws, ok := exitErr.Sys().(syscall.WaitStatus); ok && ws.Signaled() {
|
||||||
t.Fatalf("verify died from %v — the crash was not isolated:\n%s", ws.Signal(), out)
|
t.Fatalf("verify died from %v; the crash was not isolated:\n%s", ws.Signal(), out)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if !strings.Contains(string(out), "CRASH") {
|
if !strings.Contains(string(out), "CRASH") {
|
||||||
@@ -292,3 +386,77 @@ func TestSweepCheckLines(t *testing.T) {
|
|||||||
t.Errorf("sweepCheckLines = %q, want %q", got, want)
|
t.Errorf("sweepCheckLines = %q, want %q", got, want)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestRunCorpusAudit drives the corpus audit over a small fixture tree: one
|
||||||
|
// suffixed amd64 file, one suffixed arm64 file whose body is not arm64, one
|
||||||
|
// generic file, and one file that does not parse.
|
||||||
|
func TestRunCorpusAudit(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
write := func(name, src string) {
|
||||||
|
t.Helper()
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
write("good_amd64.s", "#include \"textflag.h\"\nTEXT ·add(SB), NOSPLIT, $0-0\n\tMOVQ AX, BX\n\tRET\n")
|
||||||
|
write("bad_arm64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0-0\n\tMOVQ AX, BX\n\tRET\n")
|
||||||
|
write("generic.s", "#include \"textflag.h\"\nTEXT ·g(SB), NOSPLIT, $0-0\n\tRET\n")
|
||||||
|
write("broken.s", "#include \"textflag.h\"\nTEXT ·b(SB), NOSPLIT, $0-0\n\tJMP nowhere\n\tRET\n")
|
||||||
|
|
||||||
|
stats, err := runCorpusAudit(dir)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("runCorpusAudit: %v", err)
|
||||||
|
}
|
||||||
|
if stats.files != 4 {
|
||||||
|
t.Errorf("files = %d, want 4", stats.files)
|
||||||
|
}
|
||||||
|
if stats.generic != 2 {
|
||||||
|
t.Errorf("generic = %d, want 2 (generic.s and broken.s)", stats.generic)
|
||||||
|
}
|
||||||
|
// good_amd64 and generic.s assemble everywhere they are attempted.
|
||||||
|
if stats.full != 2 {
|
||||||
|
t.Errorf("full = %d, want 2", stats.full)
|
||||||
|
}
|
||||||
|
get := func(name string) *corpusTally {
|
||||||
|
for i, tg := range stats.targets {
|
||||||
|
if tg.name == name {
|
||||||
|
return stats.tallies[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatalf("no tally for %s", name)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
// amd64: good_amd64 + generic.s + broken.s; the broken file fails to parse.
|
||||||
|
if a := get("amd64"); a.attempted != 3 || a.assembled != 2 {
|
||||||
|
t.Errorf("amd64 = %d/%d, want 2/3", a.assembled, a.attempted)
|
||||||
|
}
|
||||||
|
// arm64: bad_arm64 (MOVQ is not arm64) + generic.s + broken.s.
|
||||||
|
if a := get("arm64"); a.attempted != 3 || a.assembled != 1 {
|
||||||
|
t.Errorf("arm64 = %d/%d, want 1/3", a.assembled, a.attempted)
|
||||||
|
}
|
||||||
|
if r := get("amd64").reasons["instruction not encodable"]; r != 0 {
|
||||||
|
t.Errorf("amd64 unexpected unencodable reason: %d", r)
|
||||||
|
}
|
||||||
|
if r := get("arm64").reasons["instruction not encodable"]; r != 1 {
|
||||||
|
t.Errorf("arm64 unencodable reasons = %d, want 1", r)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCompareGroundTruthPadding pins the padding-aware ground-truth
|
||||||
|
// comparison: the toolchain pads text symbols to 16-byte boundaries, so
|
||||||
|
// trailing zeros in the reference must not read as a mismatch, while any
|
||||||
|
// non-zero tail still must.
|
||||||
|
func TestCompareGroundTruthPadding(t *testing.T) {
|
||||||
|
code := []byte{0x48, 0x8b, 0x07, 0xc3} // 4 bytes, not a multiple of 16
|
||||||
|
img := &asm.Image{Code: code, Funcs: []asm.FuncLayout{{Name: "f", Offset: 0, Size: len(code)}}}
|
||||||
|
padded := append(append([]byte(nil), code...), 0, 0, 0)
|
||||||
|
matched, total, diffs := compareGroundTruth(img, map[string][]byte{"f": padded})
|
||||||
|
if matched != 1 || total != 1 || diffs != 0 {
|
||||||
|
t.Fatalf("zero padding should match: matched=%d total=%d diffs=%d", matched, total, diffs)
|
||||||
|
}
|
||||||
|
dirty := append(append([]byte(nil), code...), 0, 0x90, 0)
|
||||||
|
matched, _, diffs = compareGroundTruth(img, map[string][]byte{"f": dirty})
|
||||||
|
if matched != 0 || diffs != 1 {
|
||||||
|
t.Fatalf("non-zero padding must mismatch: matched=%d diffs=%d", matched, diffs)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,241 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"regexp"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestManPagesTrackTheCLI builds the binary once, then compares every
|
||||||
|
// command's live `-h` output with its docs/man/gasm-<command>.1 page: the
|
||||||
|
// flag sets must agree both ways, and the page's SYNOPSIS line must carry
|
||||||
|
// the command's usage line. A flag or a usage change that skips the man
|
||||||
|
// page fails here, so the pages cannot drift from the binary.
|
||||||
|
func TestManPagesTrackTheCLI(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("builds the gasm binary")
|
||||||
|
}
|
||||||
|
bin := filepath.Join(t.TempDir(), "gasm")
|
||||||
|
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("build gasm: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, cmd := range []string{
|
||||||
|
"tokens", "parse", "fmt", "lint", "asm", "dis", "verify",
|
||||||
|
"debug", "diff", "profile", "audit-instructions", "scaffold", "lsp",
|
||||||
|
} {
|
||||||
|
t.Run(cmd, func(t *testing.T) {
|
||||||
|
raw, err := os.ReadFile(filepath.Join("..", "..", "docs", "man", "gasm-"+cmd+".1"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read man page: %v", err)
|
||||||
|
}
|
||||||
|
page := string(raw)
|
||||||
|
|
||||||
|
out, _ := exec.Command(bin, cmd, "-h").CombinedOutput()
|
||||||
|
help := string(out)
|
||||||
|
|
||||||
|
binFlags := helpFlags(help)
|
||||||
|
pageFlags := roffFlags(page)
|
||||||
|
for f := range binFlags {
|
||||||
|
if !pageFlags[f] {
|
||||||
|
t.Errorf("flag -%s is in the binary's help but missing from the man page", f)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for f := range pageFlags {
|
||||||
|
if !binFlags[f] {
|
||||||
|
t.Errorf("flag -%s is in the man page but the binary does not accept it", f)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
want := helpUsage(help)
|
||||||
|
got := roffSynopsis(page)
|
||||||
|
if want != "" && got != want {
|
||||||
|
t.Errorf("SYNOPSIS drift:\n page: %s\nbinary: %s", got, want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestManCommandsTrackHelp compares the gasm(1) COMMANDS list with the
|
||||||
|
// top-level help output, so a subcommand added to the binary cannot miss
|
||||||
|
// its man entry and a stale entry cannot outlive its command.
|
||||||
|
func TestManCommandsTrackHelp(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("builds the gasm binary")
|
||||||
|
}
|
||||||
|
bin := filepath.Join(t.TempDir(), "gasm")
|
||||||
|
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("build gasm: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
raw, err := os.ReadFile(filepath.Join("..", "..", "docs", "man", "gasm.1"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read man page: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
helpOut, err := exec.Command(bin, "--help").Output()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("gasm --help: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
binCmds := helpCommands(string(helpOut))
|
||||||
|
pageCmds := roffCommands(string(raw))
|
||||||
|
for c := range binCmds {
|
||||||
|
if !pageCmds[c] {
|
||||||
|
t.Errorf("command %q is in the binary's help but missing from gasm(1) COMMANDS", c)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for c := range pageCmds {
|
||||||
|
if !binCmds[c] {
|
||||||
|
t.Errorf("command %q is in gasm(1) COMMANDS but the binary does not list it", c)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// helpFlags extracts the flag names from a `gasm <cmd> -h` output.
|
||||||
|
func helpFlags(help string) map[string]bool {
|
||||||
|
m := map[string]bool{}
|
||||||
|
inFlags := false
|
||||||
|
for line := range strings.SplitSeq(help, "\n") {
|
||||||
|
if strings.TrimRight(line, " \t") == "Flags:" {
|
||||||
|
inFlags = true
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !inFlags {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !strings.HasPrefix(line, " -") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
token := strings.FieldsFunc(strings.TrimLeft(line, " "), func(r rune) bool {
|
||||||
|
return r == ' ' || r == '\t'
|
||||||
|
})
|
||||||
|
if len(token) == 0 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
m[strings.TrimLeft(token[0], "-")] = true
|
||||||
|
}
|
||||||
|
return m
|
||||||
|
}
|
||||||
|
|
||||||
|
var roffEscape = regexp.MustCompile(`\\f[BIRP]`)
|
||||||
|
|
||||||
|
// roffFlags extracts the flag names from a man page's OPTIONS section.
|
||||||
|
func roffFlags(page string) map[string]bool {
|
||||||
|
m := map[string]bool{}
|
||||||
|
inOptions := false
|
||||||
|
for line := range strings.SplitSeq(page, "\n") {
|
||||||
|
if strings.HasPrefix(line, ".SH ") {
|
||||||
|
inOptions = strings.HasPrefix(line, ".SH OPTIONS")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !inOptions {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Flag entries are written as either `.B \-flag` or `\fB\-flag`.
|
||||||
|
var body string
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(line, `.B \-`):
|
||||||
|
body = line[3:]
|
||||||
|
case strings.HasPrefix(line, `\fB\-`):
|
||||||
|
body = line[1:]
|
||||||
|
default:
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
name := roffEscape.ReplaceAllString(body, "")
|
||||||
|
name = strings.ReplaceAll(name, `\-`, "-")
|
||||||
|
name = strings.TrimSpace(name)
|
||||||
|
if i := strings.IndexAny(name, " \t"); i >= 0 {
|
||||||
|
name = name[:i]
|
||||||
|
}
|
||||||
|
m[strings.TrimLeft(name, "-")] = true
|
||||||
|
}
|
||||||
|
return m
|
||||||
|
}
|
||||||
|
|
||||||
|
// helpCommands extracts the command names from the top-level help output's
|
||||||
|
// Commands section.
|
||||||
|
func helpCommands(help string) map[string]bool {
|
||||||
|
m := map[string]bool{}
|
||||||
|
inCmds := false
|
||||||
|
for line := range strings.SplitSeq(help, "\n") {
|
||||||
|
if strings.TrimSpace(line) == "Commands:" {
|
||||||
|
inCmds = true
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !inCmds {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
t := strings.TrimSpace(line)
|
||||||
|
if t == "" {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
name, _, _ := strings.Cut(t, " ")
|
||||||
|
m[name] = true
|
||||||
|
}
|
||||||
|
return m
|
||||||
|
}
|
||||||
|
|
||||||
|
// roffCommands extracts the command names from gasm(1)'s COMMANDS section,
|
||||||
|
// where each entry is written as `.B gasm\-<name>(1)` or `.B gasm <name>`.
|
||||||
|
func roffCommands(page string) map[string]bool {
|
||||||
|
m := map[string]bool{}
|
||||||
|
inCmds := false
|
||||||
|
for line := range strings.SplitSeq(page, "\n") {
|
||||||
|
if strings.HasPrefix(line, ".SH ") {
|
||||||
|
inCmds = strings.HasPrefix(line, ".SH COMMANDS")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !inCmds || !strings.HasPrefix(line, ".B gasm") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
entry := strings.ReplaceAll(strings.TrimPrefix(line, ".B "), `\-`, "-")
|
||||||
|
entry = strings.TrimSuffix(entry, "(1)")
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(entry, "gasm-"):
|
||||||
|
m[strings.TrimPrefix(entry, "gasm-")] = true
|
||||||
|
case strings.HasPrefix(entry, "gasm "):
|
||||||
|
m[strings.TrimPrefix(entry, "gasm ")] = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return m
|
||||||
|
}
|
||||||
|
|
||||||
|
// helpUsage returns the command's usage line without the "Usage: " prefix.
|
||||||
|
func helpUsage(help string) string {
|
||||||
|
for line := range strings.SplitSeq(help, "\n") {
|
||||||
|
if strings.HasPrefix(line, "Usage: ") {
|
||||||
|
return normaliseUsage(line[len("Usage: "):])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// roffSynopsis returns the page's SYNOPSIS usage line, unescaped.
|
||||||
|
func roffSynopsis(page string) string {
|
||||||
|
inSyn := false
|
||||||
|
for line := range strings.SplitSeq(page, "\n") {
|
||||||
|
if strings.HasPrefix(line, ".SH ") {
|
||||||
|
inSyn = strings.HasPrefix(line, ".SH SYNOPSIS")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !inSyn || !strings.HasPrefix(line, ".B ") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
return normaliseUsage(strings.ReplaceAll(line[3:], `\-`, "-"))
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// normaliseUsage flattens whitespace and drops the roff font escapes so that
|
||||||
|
// the binary's usage line and the page's SYNOPSIS line compare equal.
|
||||||
|
func normaliseUsage(s string) string {
|
||||||
|
s = roffEscape.ReplaceAllString(s, "")
|
||||||
|
return strings.Join(strings.Fields(s), " ")
|
||||||
|
}
|
||||||
@@ -44,7 +44,7 @@ bodies, place the file in the kernel's package, and run it in CI.
|
|||||||
rest = rest[1:]
|
rest = rest[1:]
|
||||||
}
|
}
|
||||||
if len(rest) != 1 {
|
if len(rest) != 1 {
|
||||||
return fmt.Errorf("usage: gasm scaffold differential <file.s>")
|
return &usageError{fmt.Errorf("usage: gasm scaffold differential <file.s>")}
|
||||||
}
|
}
|
||||||
path := rest[0]
|
path := rest[0]
|
||||||
src, err := os.ReadFile(path)
|
src, err := os.ReadFile(path)
|
||||||
|
|||||||
+99
-44
@@ -9,11 +9,11 @@ import "strings"
|
|||||||
|
|
||||||
import "fmt"
|
import "fmt"
|
||||||
|
|
||||||
// Breakpoint is one INT3 breakpoint in the debuggee.
|
// Breakpoint is one software breakpoint in the debuggee.
|
||||||
type Breakpoint struct {
|
type Breakpoint struct {
|
||||||
Addr uint64 // absolute address in the debuggee
|
Addr uint64 // absolute address in the debuggee
|
||||||
Label string // source label ("" for raw addresses)
|
Label string // source label ("" for raw addresses)
|
||||||
Orig byte // original byte at Addr (restored on removal)
|
Orig []byte // original bytes at Addr (restored on removal)
|
||||||
Enabled bool
|
Enabled bool
|
||||||
Cond *Condition // optional condition (nil = unconditional)
|
Cond *Condition // optional condition (nil = unconditional)
|
||||||
hits int
|
hits int
|
||||||
@@ -32,8 +32,12 @@ type Condition struct {
|
|||||||
MemAddr uint64 // memory address (for register-memory comparison, prefixed with *)
|
MemAddr uint64 // memory address (for register-memory comparison, prefixed with *)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Eval checks the condition against the current registers.
|
// Eval checks the condition against the current registers. For the
|
||||||
func (c *Condition) Eval(regs *Regs) bool {
|
// register-memory form, mem reads an 8-byte little-endian word from the
|
||||||
|
// debuggee; it may be nil when no reader is available. Anything that cannot
|
||||||
|
// be decided (unknown register or operator, unreadable memory) does not
|
||||||
|
// block the breakpoint.
|
||||||
|
func (c *Condition) Eval(regs *Regs, mem func(addr uint64) (uint64, bool)) bool {
|
||||||
actual, ok := regs.RegValue(c.Reg)
|
actual, ok := regs.RegValue(c.Reg)
|
||||||
if !ok {
|
if !ok {
|
||||||
return true // unknown register, don't block
|
return true // unknown register, don't block
|
||||||
@@ -48,9 +52,16 @@ func (c *Condition) Eval(regs *Regs) bool {
|
|||||||
}
|
}
|
||||||
expected = v
|
expected = v
|
||||||
case c.MemAddr != 0:
|
case c.MemAddr != 0:
|
||||||
// Register-memory comparison, requires a Session, not available here.
|
// Register-memory comparison, resolved in the debuggee at
|
||||||
// Fall back to treating as constant (the caller should resolve).
|
// evaluation time.
|
||||||
expected = c.Value
|
if mem == nil {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
v, ok := mem(c.MemAddr)
|
||||||
|
if !ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
expected = v
|
||||||
default:
|
default:
|
||||||
expected = c.Value
|
expected = c.Value
|
||||||
}
|
}
|
||||||
@@ -72,6 +83,18 @@ func (c *Condition) Eval(regs *Regs) bool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// String renders the condition for display.
|
||||||
|
func (c *Condition) String() string {
|
||||||
|
switch {
|
||||||
|
case c.Reg2 != "":
|
||||||
|
return fmt.Sprintf("%s %s %s", c.Reg, c.Op, c.Reg2)
|
||||||
|
case c.MemAddr != 0:
|
||||||
|
return fmt.Sprintf("%s %s *%#x", c.Reg, c.Op, c.MemAddr)
|
||||||
|
default:
|
||||||
|
return fmt.Sprintf("%s %s %#x", c.Reg, c.Op, c.Value)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Breakpoints manages the software breakpoints of one Session.
|
// Breakpoints manages the software breakpoints of one Session.
|
||||||
type Breakpoints struct {
|
type Breakpoints struct {
|
||||||
t tracer
|
t tracer
|
||||||
@@ -83,6 +106,18 @@ func NewBreakpoints(t tracer) *Breakpoints {
|
|||||||
return &Breakpoints{t: t, bps: make(map[uint64]*Breakpoint)}
|
return &Breakpoints{t: t, bps: make(map[uint64]*Breakpoint)}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// breakpointMask is the byte mask of the breakpoint instruction inside a
|
||||||
|
// peeked word: the low len(breakpointInsn) bytes, because every supported
|
||||||
|
// architecture is little-endian and patches the instruction at the lowest
|
||||||
|
// address of the word.
|
||||||
|
func breakpointMask() uint64 {
|
||||||
|
var mask uint64
|
||||||
|
for range breakpointInsn {
|
||||||
|
mask = (mask << 8) | 0xFF
|
||||||
|
}
|
||||||
|
return mask
|
||||||
|
}
|
||||||
|
|
||||||
// Set installs a breakpoint at addr (replaces any existing one).
|
// Set installs a breakpoint at addr (replaces any existing one).
|
||||||
func (bm *Breakpoints) Set(addr uint64, label string) (*Breakpoint, error) {
|
func (bm *Breakpoints) Set(addr uint64, label string) (*Breakpoint, error) {
|
||||||
return bm.SetWithCond(addr, label, nil)
|
return bm.SetWithCond(addr, label, nil)
|
||||||
@@ -100,13 +135,12 @@ func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
orig := byte(word)
|
orig := make([]byte, len(breakpointInsn))
|
||||||
// Patch with the breakpoint instruction, preserving the rest of the word.
|
for i := range orig {
|
||||||
mask := uint64(0)
|
orig[i] = byte(word >> (8 * i))
|
||||||
for range breakpointInsn {
|
|
||||||
mask = (mask << 8) | 0xFF
|
|
||||||
}
|
}
|
||||||
patched := (word &^ mask) | breakpointWord(breakpointInsn)
|
// Patch with the breakpoint instruction, preserving the rest of the word.
|
||||||
|
patched := (word &^ breakpointMask()) | breakpointWord(breakpointInsn)
|
||||||
if err := bm.t.Poke(addr, patched); err != nil {
|
if err := bm.t.Poke(addr, patched); err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
@@ -134,26 +168,40 @@ func (bm *Breakpoints) Info() string {
|
|||||||
}
|
}
|
||||||
cond := ""
|
cond := ""
|
||||||
if bp.Cond != nil {
|
if bp.Cond != nil {
|
||||||
cond = fmt.Sprintf(" if %s %s %#x", bp.Cond.Reg, bp.Cond.Op, bp.Cond.Value)
|
cond = " if " + bp.Cond.String()
|
||||||
}
|
}
|
||||||
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond))
|
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond))
|
||||||
}
|
}
|
||||||
return result.String()
|
return result.String()
|
||||||
}
|
}
|
||||||
|
|
||||||
// Clear removes the breakpoint at addr, restoring the original byte.
|
// restore writes the saved original bytes back over the breakpoint
|
||||||
|
// instruction, preserving the rest of the peeked word. It reports whether
|
||||||
|
// both the peek and the poke succeeded.
|
||||||
|
func (bm *Breakpoints) restore(addr uint64, bp *Breakpoint) bool {
|
||||||
|
word, err := bm.t.Peek(addr)
|
||||||
|
if err != nil {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
orig := uint64(0)
|
||||||
|
for i, b := range bp.Orig {
|
||||||
|
orig |= uint64(b) << (8 * i)
|
||||||
|
}
|
||||||
|
return bm.t.Poke(addr, (word&^breakpointMask())|orig) == nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Clear removes the breakpoint at addr, restoring the original bytes.
|
||||||
func (bm *Breakpoints) Clear(addr uint64) error {
|
func (bm *Breakpoints) Clear(addr uint64) error {
|
||||||
bp, ok := bm.bps[addr]
|
bp, ok := bm.bps[addr]
|
||||||
if !ok {
|
if !ok {
|
||||||
return fmt.Errorf("debug: no breakpoint at %#x", addr)
|
return fmt.Errorf("debug: no breakpoint at %#x", addr)
|
||||||
}
|
}
|
||||||
word, err := bm.t.Peek(addr)
|
if !bm.restore(addr, bp) {
|
||||||
if err != nil {
|
word, err := bm.t.Peek(addr)
|
||||||
return err
|
if err != nil {
|
||||||
}
|
return err
|
||||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
}
|
||||||
if err := bm.t.Poke(addr, restored); err != nil {
|
return fmt.Errorf("debug: restore breakpoint at %#x failed, word is %#x", addr, word)
|
||||||
return err
|
|
||||||
}
|
}
|
||||||
delete(bm.bps, addr)
|
delete(bm.bps, addr)
|
||||||
return nil
|
return nil
|
||||||
@@ -185,43 +233,54 @@ func (bm *Breakpoints) All() []*Breakpoint {
|
|||||||
|
|
||||||
// HandleTrap is called after the debuggee stops on SIGTRAP. It checks
|
// HandleTrap is called after the debuggee stops on SIGTRAP. It checks
|
||||||
// whether the trap was caused by one of our breakpoints (PC-adjust matches
|
// whether the trap was caused by one of our breakpoints (PC-adjust matches
|
||||||
// a breakpoint address), restores the original byte, rewinds PC, and
|
// a breakpoint address), restores the original bytes, rewinds PC, and
|
||||||
// returns the breakpoint that was hit (or nil if it was a single-step).
|
// returns the breakpoint that was hit (or nil if it was a single-step).
|
||||||
// Hits returns how many times the breakpoint has been hit.
|
// Hits returns how many times the breakpoint has been hit.
|
||||||
func (bp *Breakpoint) Hits() int { return bp.hits }
|
func (bp *Breakpoint) Hits() int { return bp.hits }
|
||||||
|
|
||||||
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
|
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
|
||||||
// After a breakpoint trap, PC points past the breakpoint instruction.
|
// On amd64 the kernel reports the trap with RIP past the INT3; on the
|
||||||
|
// other supported architectures the PC still stands on the trap
|
||||||
|
// instruction, which breakpointPCAdjust encodes per architecture.
|
||||||
trapAddr := regs.GetPC() - uint64(breakpointPCAdjust)
|
trapAddr := regs.GetPC() - uint64(breakpointPCAdjust)
|
||||||
bp, ok := bm.bps[trapAddr]
|
bp, ok := bm.bps[trapAddr]
|
||||||
if !ok || !bp.Enabled {
|
if !ok || !bp.Enabled {
|
||||||
return nil // single-step trap or unknown
|
return nil // single-step trap or unknown
|
||||||
}
|
}
|
||||||
// Check the condition (if any).
|
// Check the condition (if any).
|
||||||
if bp.Cond != nil && !bp.Cond.Eval(regs) {
|
if bp.Cond != nil && !bp.Cond.Eval(regs, bm.peekValue) {
|
||||||
// Condition not met, restore the byte but do NOT rewind RIP.
|
// Condition not met: step the original instruction and re-arm the
|
||||||
// The process continues from the next instruction (past the INT3).
|
// breakpoint, leaving the debuggee stopped just past it, ready to
|
||||||
word, err := bm.t.Peek(trapAddr)
|
// resume silently. The PC must be rewound first: on architectures
|
||||||
if err == nil {
|
// that report the trap past the instruction (amd64) it would
|
||||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
// otherwise sit on the second byte of the replaced instruction.
|
||||||
bm.t.Poke(trapAddr, restored)
|
if !bm.restore(trapAddr, bp) {
|
||||||
|
return nil
|
||||||
}
|
}
|
||||||
// RIP is already past the INT3 (trapAddr + 1). Don't rewind.
|
regs.SetPC(trapAddr)
|
||||||
|
if err := bm.t.SetRegs(regs); err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if err := bm.t.Step(); err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
bm.Reinsert(trapAddr)
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
bp.hits++
|
bp.hits++
|
||||||
// Restore the original byte.
|
// Restore the original bytes and rewind PC to re-execute them.
|
||||||
word, err := bm.t.Peek(trapAddr)
|
bm.restore(trapAddr, bp)
|
||||||
if err == nil {
|
|
||||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
|
||||||
bm.t.Poke(trapAddr, restored)
|
|
||||||
}
|
|
||||||
// Rewind PC to re-execute the original instruction.
|
|
||||||
regs.SetPC(trapAddr)
|
regs.SetPC(trapAddr)
|
||||||
bm.t.SetRegs(regs)
|
bm.t.SetRegs(regs)
|
||||||
return bp
|
return bp
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// peekValue adapts tracer.Peek to the Condition value reader.
|
||||||
|
func (bm *Breakpoints) peekValue(addr uint64) (uint64, bool) {
|
||||||
|
v, err := bm.t.Peek(addr)
|
||||||
|
return v, err == nil
|
||||||
|
}
|
||||||
|
|
||||||
// Reinsert re-inserts the breakpoint at addr after a single-step past it.
|
// Reinsert re-inserts the breakpoint at addr after a single-step past it.
|
||||||
// Called after Step() when we want the breakpoint to fire again on the
|
// Called after Step() when we want the breakpoint to fire again on the
|
||||||
// next Continue().
|
// next Continue().
|
||||||
@@ -234,11 +293,7 @@ func (bm *Breakpoints) Reinsert(addr uint64) error {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
mask := uint64(0)
|
patched := (word &^ breakpointMask()) | breakpointWord(breakpointInsn)
|
||||||
for range breakpointInsn {
|
|
||||||
mask = (mask << 8) | 0xFF
|
|
||||||
}
|
|
||||||
patched := (word &^ mask) | breakpointWord(breakpointInsn)
|
|
||||||
return bm.t.Poke(addr, patched)
|
return bm.t.Poke(addr, patched)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,265 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build linux
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
// Architecture-neutral tests: label and line tables, and the breakpoint
|
||||||
|
// manager against the mock tracer. These do not launch a debuggee, so they
|
||||||
|
// build on every supported linux architecture.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestLineAt(t *testing.T) {
|
||||||
|
lines := []SourceLine{
|
||||||
|
{Offset: 0, Line: 5},
|
||||||
|
{Offset: 5, Line: 6},
|
||||||
|
{Offset: 10, Line: 7},
|
||||||
|
{Offset: 15, Line: 8},
|
||||||
|
}
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
offset int
|
||||||
|
want int
|
||||||
|
}{
|
||||||
|
{0, 5},
|
||||||
|
{1, 5},
|
||||||
|
{4, 5},
|
||||||
|
{5, 6},
|
||||||
|
{7, 6},
|
||||||
|
{10, 7},
|
||||||
|
{12, 7},
|
||||||
|
{15, 8},
|
||||||
|
{20, 8},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
got := lineAt(lines, tt.offset)
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Empty table.
|
||||||
|
if lineAt(nil, 5) != 0 {
|
||||||
|
t.Error("lineAt(nil, 5) should return 0")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestOffsetForLine(t *testing.T) {
|
||||||
|
lines := []SourceLine{
|
||||||
|
{Offset: 0, Line: 5},
|
||||||
|
{Offset: 5, Line: 6},
|
||||||
|
{Offset: 10, Line: 7},
|
||||||
|
}
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
line int
|
||||||
|
want int
|
||||||
|
}{
|
||||||
|
{5, 0},
|
||||||
|
{6, 5},
|
||||||
|
{7, 10},
|
||||||
|
{99, -1}, // not found
|
||||||
|
{0, -1}, // not found
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
got := offsetForLine(lines, tt.line)
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestNearestLabel(t *testing.T) {
|
||||||
|
labels := []Label{
|
||||||
|
{Name: "start", Offset: 0},
|
||||||
|
{Name: "loop", Offset: 10},
|
||||||
|
{Name: "done", Offset: 20},
|
||||||
|
}
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
offset int
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{0, "start"},
|
||||||
|
{5, "start"},
|
||||||
|
{10, "loop"},
|
||||||
|
{15, "loop"},
|
||||||
|
{20, "done"},
|
||||||
|
{25, "done"},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
got := nearestLabel(labels, tt.offset)
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBreakpointsSetAndClear(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
|
||||||
|
// Set a breakpoint at address 0x1000.
|
||||||
|
bp, err := bm.Set(0x1000, "test")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Set: %v", err)
|
||||||
|
}
|
||||||
|
if !bp.Enabled {
|
||||||
|
t.Error("breakpoint not enabled")
|
||||||
|
}
|
||||||
|
if bp.Label != "test" {
|
||||||
|
t.Errorf("label = %q, want test", bp.Label)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Verify Peek was called.
|
||||||
|
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
|
||||||
|
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Verify Poke wrote the breakpoint instruction's bytes.
|
||||||
|
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
|
||||||
|
t.Errorf("pokes = %v", tr.pokes)
|
||||||
|
}
|
||||||
|
if got := tr.pokes[0].val & breakpointMask(); got != breakpointWord(breakpointInsn) {
|
||||||
|
t.Errorf("patched bytes %#x, want %#x", got, breakpointWord(breakpointInsn))
|
||||||
|
}
|
||||||
|
|
||||||
|
// At should find it.
|
||||||
|
if bm.At(0x1000) == nil {
|
||||||
|
t.Error("At(0x1000) returned nil")
|
||||||
|
}
|
||||||
|
|
||||||
|
// All should return it.
|
||||||
|
all := bm.All()
|
||||||
|
if len(all) != 1 {
|
||||||
|
t.Errorf("All() = %d breakpoints, want 1", len(all))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Clear it.
|
||||||
|
if err := bm.Clear(0x1000); err != nil {
|
||||||
|
t.Fatalf("Clear: %v", err)
|
||||||
|
}
|
||||||
|
if bm.At(0x1000) != nil {
|
||||||
|
t.Error("At(0x1000) after Clear should be nil")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestBreakpointRestoreWidth proves the restore path writes back every
|
||||||
|
// byte of the breakpoint instruction's width, not just the first byte: on
|
||||||
|
// arm64, riscv64 and loong64 the instruction is four bytes, and restoring
|
||||||
|
// one byte would leave three bytes of the trap instruction in place.
|
||||||
|
func TestBreakpointRestoreWidth(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
tr.mem[0x3000] = 0x11
|
||||||
|
tr.mem[0x3001] = 0x22
|
||||||
|
tr.mem[0x3002] = 0x33
|
||||||
|
tr.mem[0x3003] = 0x44
|
||||||
|
|
||||||
|
if _, err := bm.Set(0x3000, "width"); err != nil {
|
||||||
|
t.Fatalf("Set: %v", err)
|
||||||
|
}
|
||||||
|
for i, b := range breakpointInsn {
|
||||||
|
if tr.mem[0x3000+uint64(i)] != b {
|
||||||
|
t.Fatalf("byte %d after Set = %#x, want the breakpoint byte %#x", i, tr.mem[0x3000+uint64(i)], b)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(bm.At(0x3000).Orig) != len(breakpointInsn) {
|
||||||
|
t.Fatalf("Orig holds %d bytes, want %d", len(bm.At(0x3000).Orig), len(breakpointInsn))
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := bm.Clear(0x3000); err != nil {
|
||||||
|
t.Fatalf("Clear: %v", err)
|
||||||
|
}
|
||||||
|
want := []byte{0x11, 0x22, 0x33, 0x44}
|
||||||
|
for i, b := range want {
|
||||||
|
if tr.mem[0x3000+uint64(i)] != b {
|
||||||
|
t.Errorf("byte %d after Clear = %#x, want %#x (restore must cover the full instruction width)", i, tr.mem[0x3000+uint64(i)], b)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBreakpointsSetWithCond(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
|
||||||
|
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
|
||||||
|
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("SetWithCond: %v", err)
|
||||||
|
}
|
||||||
|
if bp.Cond == nil || bp.Cond.Value != 42 {
|
||||||
|
t.Error("condition not set")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Re-setting the same address should update the condition.
|
||||||
|
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
|
||||||
|
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("SetWithCond (update): %v", err)
|
||||||
|
}
|
||||||
|
if bp2.Cond.Value != 100 {
|
||||||
|
t.Error("condition not updated")
|
||||||
|
}
|
||||||
|
// Should have only 1 Peek (first Set), second is update (no Peek needed).
|
||||||
|
if len(tr.peeks) != 1 {
|
||||||
|
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBreakpointsClearAll(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
|
||||||
|
bm.Set(0x1000, "a")
|
||||||
|
bm.Set(0x2000, "b")
|
||||||
|
bm.Set(0x3000, "c")
|
||||||
|
|
||||||
|
if len(bm.All()) != 3 {
|
||||||
|
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
|
||||||
|
}
|
||||||
|
|
||||||
|
bm.ClearAll()
|
||||||
|
if len(bm.All()) != 0 {
|
||||||
|
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBreakpointInfo(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
bm.Set(0x4000, "info_test")
|
||||||
|
|
||||||
|
info := bm.Info()
|
||||||
|
if info == "" {
|
||||||
|
t.Error("Info returned empty string")
|
||||||
|
}
|
||||||
|
if !strings.Contains(info, "info_test") {
|
||||||
|
t.Errorf("Info %q does not contain label", info)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestConditionString covers the display of all three condition forms.
|
||||||
|
func TestConditionString(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
cond Condition
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{Condition{Reg: "rax", Op: "==", Value: 42}, "rax == 0x2a"},
|
||||||
|
{Condition{Reg: "rax", Op: "!=", Reg2: "rbx"}, "rax != rbx"},
|
||||||
|
{Condition{Reg: "rax", Op: "<", MemAddr: 0x5000}, "rax < *0x5000"},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
if got := tt.cond.String(); got != tt.want {
|
||||||
|
t.Errorf("Condition.String() = %q, want %q", got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
+38
-188
@@ -6,7 +6,6 @@
|
|||||||
package debug
|
package debug
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"strings"
|
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -48,7 +47,7 @@ func TestConditionEval(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
for _, tt := range tests {
|
for _, tt := range tests {
|
||||||
got := tt.cond.Eval(regs)
|
got := tt.cond.Eval(regs, nil)
|
||||||
if got != tt.want {
|
if got != tt.want {
|
||||||
t.Errorf("Condition{%q %q %d}.Eval() = %v, want %v",
|
t.Errorf("Condition{%q %q %d}.Eval() = %v, want %v",
|
||||||
tt.cond.Reg, tt.cond.Op, tt.cond.Value, got, tt.want)
|
tt.cond.Reg, tt.cond.Op, tt.cond.Value, got, tt.want)
|
||||||
@@ -56,65 +55,33 @@ func TestConditionEval(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestLineAt(t *testing.T) {
|
// TestConditionEvalMem covers the register-memory form: the value is read
|
||||||
lines := []SourceLine{
|
// through the supplied reader, and a missing or failing reader must not
|
||||||
{Offset: 0, Line: 5},
|
// block the breakpoint.
|
||||||
{Offset: 5, Line: 6},
|
func TestConditionEvalMem(t *testing.T) {
|
||||||
{Offset: 10, Line: 7},
|
regs := &Regs{RAX: 7}
|
||||||
{Offset: 15, Line: 8},
|
mem := func(addr uint64) (uint64, bool) {
|
||||||
}
|
if addr == 0x5000 {
|
||||||
|
return 7, true
|
||||||
tests := []struct {
|
|
||||||
offset int
|
|
||||||
want int
|
|
||||||
}{
|
|
||||||
{0, 5},
|
|
||||||
{1, 5},
|
|
||||||
{4, 5},
|
|
||||||
{5, 6},
|
|
||||||
{7, 6},
|
|
||||||
{10, 7},
|
|
||||||
{12, 7},
|
|
||||||
{15, 8},
|
|
||||||
{20, 8},
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, tt := range tests {
|
|
||||||
got := lineAt(lines, tt.offset)
|
|
||||||
if got != tt.want {
|
|
||||||
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
|
|
||||||
}
|
}
|
||||||
|
return 0, false
|
||||||
}
|
}
|
||||||
|
|
||||||
// Empty table.
|
eq := Condition{Reg: "rax", Op: "==", MemAddr: 0x5000}
|
||||||
if lineAt(nil, 5) != 0 {
|
if !eq.Eval(regs, mem) {
|
||||||
t.Error("lineAt(nil, 5) should return 0")
|
t.Error("register-memory comparison with matching word should hold")
|
||||||
}
|
}
|
||||||
}
|
ne := Condition{Reg: "rax", Op: "!=", MemAddr: 0x5000}
|
||||||
|
if ne.Eval(regs, mem) {
|
||||||
func TestOffsetForLine(t *testing.T) {
|
t.Error("register-memory comparison with mismatching word should not hold")
|
||||||
lines := []SourceLine{
|
|
||||||
{Offset: 0, Line: 5},
|
|
||||||
{Offset: 5, Line: 6},
|
|
||||||
{Offset: 10, Line: 7},
|
|
||||||
}
|
}
|
||||||
|
bad := Condition{Reg: "rax", Op: "==", MemAddr: 0x6000}
|
||||||
tests := []struct {
|
if !bad.Eval(regs, mem) {
|
||||||
line int
|
t.Error("unreadable memory must not block the breakpoint")
|
||||||
want int
|
|
||||||
}{
|
|
||||||
{5, 0},
|
|
||||||
{6, 5},
|
|
||||||
{7, 10},
|
|
||||||
{99, -1}, // not found
|
|
||||||
{0, -1}, // not found
|
|
||||||
}
|
}
|
||||||
|
noReader := Condition{Reg: "rax", Op: "==", MemAddr: 0x5000}
|
||||||
for _, tt := range tests {
|
if !noReader.Eval(regs, nil) {
|
||||||
got := offsetForLine(lines, tt.line)
|
t.Error("missing memory reader must not block the breakpoint")
|
||||||
if got != tt.want {
|
|
||||||
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -141,139 +108,6 @@ func TestDecodeRflags(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestNearestLabel(t *testing.T) {
|
|
||||||
labels := []Label{
|
|
||||||
{Name: "start", Offset: 0},
|
|
||||||
{Name: "loop", Offset: 10},
|
|
||||||
{Name: "done", Offset: 20},
|
|
||||||
}
|
|
||||||
|
|
||||||
tests := []struct {
|
|
||||||
offset int
|
|
||||||
want string
|
|
||||||
}{
|
|
||||||
{0, "start"},
|
|
||||||
{5, "start"},
|
|
||||||
{10, "loop"},
|
|
||||||
{15, "loop"},
|
|
||||||
{20, "done"},
|
|
||||||
{25, "done"},
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, tt := range tests {
|
|
||||||
got := nearestLabel(labels, tt.offset)
|
|
||||||
if got != tt.want {
|
|
||||||
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestBreakpointsSetAndClear(t *testing.T) {
|
|
||||||
tr := newMockTracer()
|
|
||||||
bm := NewBreakpoints(tr)
|
|
||||||
|
|
||||||
// Set a breakpoint at address 0x1000.
|
|
||||||
bp, err := bm.Set(0x1000, "test")
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("Set: %v", err)
|
|
||||||
}
|
|
||||||
if !bp.Enabled {
|
|
||||||
t.Error("breakpoint not enabled")
|
|
||||||
}
|
|
||||||
if bp.Label != "test" {
|
|
||||||
t.Errorf("label = %q, want test", bp.Label)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Verify Peek was called.
|
|
||||||
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
|
|
||||||
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Verify Poke wrote INT3.
|
|
||||||
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
|
|
||||||
t.Errorf("pokes = %v", tr.pokes)
|
|
||||||
}
|
|
||||||
|
|
||||||
// At should find it.
|
|
||||||
if bm.At(0x1000) == nil {
|
|
||||||
t.Error("At(0x1000) returned nil")
|
|
||||||
}
|
|
||||||
|
|
||||||
// All should return it.
|
|
||||||
all := bm.All()
|
|
||||||
if len(all) != 1 {
|
|
||||||
t.Errorf("All() = %d breakpoints, want 1", len(all))
|
|
||||||
}
|
|
||||||
|
|
||||||
// Clear it.
|
|
||||||
if err := bm.Clear(0x1000); err != nil {
|
|
||||||
t.Fatalf("Clear: %v", err)
|
|
||||||
}
|
|
||||||
if bm.At(0x1000) != nil {
|
|
||||||
t.Error("At(0x1000) after Clear should be nil")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestBreakpointsSetWithCond(t *testing.T) {
|
|
||||||
tr := newMockTracer()
|
|
||||||
bm := NewBreakpoints(tr)
|
|
||||||
|
|
||||||
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
|
|
||||||
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("SetWithCond: %v", err)
|
|
||||||
}
|
|
||||||
if bp.Cond == nil || bp.Cond.Value != 42 {
|
|
||||||
t.Error("condition not set")
|
|
||||||
}
|
|
||||||
|
|
||||||
// Re-setting the same address should update the condition.
|
|
||||||
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
|
|
||||||
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("SetWithCond (update): %v", err)
|
|
||||||
}
|
|
||||||
if bp2.Cond.Value != 100 {
|
|
||||||
t.Error("condition not updated")
|
|
||||||
}
|
|
||||||
// Should have only 1 Peek (first Set), second is update (no Peek needed).
|
|
||||||
if len(tr.peeks) != 1 {
|
|
||||||
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestBreakpointsClearAll(t *testing.T) {
|
|
||||||
tr := newMockTracer()
|
|
||||||
bm := NewBreakpoints(tr)
|
|
||||||
|
|
||||||
bm.Set(0x1000, "a")
|
|
||||||
bm.Set(0x2000, "b")
|
|
||||||
bm.Set(0x3000, "c")
|
|
||||||
|
|
||||||
if len(bm.All()) != 3 {
|
|
||||||
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
|
|
||||||
}
|
|
||||||
|
|
||||||
bm.ClearAll()
|
|
||||||
if len(bm.All()) != 0 {
|
|
||||||
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestBreakpointInfo(t *testing.T) {
|
|
||||||
tr := newMockTracer()
|
|
||||||
bm := NewBreakpoints(tr)
|
|
||||||
bm.Set(0x4000, "info_test")
|
|
||||||
|
|
||||||
info := bm.Info()
|
|
||||||
if info == "" {
|
|
||||||
t.Error("Info returned empty string")
|
|
||||||
}
|
|
||||||
if !strings.Contains(info, "info_test") {
|
|
||||||
t.Errorf("Info %q does not contain label", info)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestWatchpointSlotTracking(t *testing.T) {
|
func TestWatchpointSlotTracking(t *testing.T) {
|
||||||
s := &Session{} // per-session slots start free
|
s := &Session{} // per-session slots start free
|
||||||
|
|
||||||
@@ -323,3 +157,19 @@ func TestWatchpointSlotTracking(t *testing.T) {
|
|||||||
t.Errorf("FindFreeWatchpointSlot() with all slots used = %d, want -1", got)
|
t.Errorf("FindFreeWatchpointSlot() with all slots used = %d, want -1", got)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestUnwatchSlotBound checks the bound the REPL parses against: it must
|
||||||
|
// cover the architecture's whole slot range, not a hardcoded 0-3.
|
||||||
|
func TestUnwatchSlotBound(t *testing.T) {
|
||||||
|
max := maxWatchpoints()
|
||||||
|
if max < 4 {
|
||||||
|
t.Fatalf("maxWatchpoints() = %d, want at least 4", max)
|
||||||
|
}
|
||||||
|
s := &Session{}
|
||||||
|
if s.IsWatchpointSlotUsed(max - 1) {
|
||||||
|
t.Errorf("slot %d should be free initially", max-1)
|
||||||
|
}
|
||||||
|
if s.IsWatchpointSlotUsed(max) {
|
||||||
|
t.Errorf("slot %d must be out of range", max)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -46,3 +46,11 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
|
|||||||
}
|
}
|
||||||
return result.String()
|
return result.String()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// isCallInsn reports whether disassembled text (x86asm.IntelSyntax) is a
|
||||||
|
// call. The first token must match exactly: a prefix test would also catch
|
||||||
|
// unrelated mnemonics.
|
||||||
|
func isCallInsn(text string) bool {
|
||||||
|
m, _, _ := strings.Cut(text, " ")
|
||||||
|
return strings.ToLower(m) == "call"
|
||||||
|
}
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ package debug
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
||||||
@@ -44,3 +45,15 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
|
|||||||
}
|
}
|
||||||
return result
|
return result
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// isCallInsn reports whether disassembled text (arm64asm.GoSyntax) is a
|
||||||
|
// call. GoSyntax renders bl as CALL; the native mnemonic is accepted too.
|
||||||
|
// The first token must match exactly so branches never match.
|
||||||
|
func isCallInsn(text string) bool {
|
||||||
|
m, _, _ := strings.Cut(text, " ")
|
||||||
|
switch strings.ToLower(m) {
|
||||||
|
case "call", "bl":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ package debug
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
||||||
@@ -44,3 +45,16 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
|
|||||||
}
|
}
|
||||||
return result
|
return result
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// isCallInsn reports whether disassembled text (loong64asm.GoSyntax) is a
|
||||||
|
// call. GoSyntax renders bl and jirl calls as CALL (jirl returns print
|
||||||
|
// RET); the native mnemonics are accepted too. The first token must match
|
||||||
|
// exactly: a "bl" prefix would catch bltz and other branches.
|
||||||
|
func isCallInsn(text string) bool {
|
||||||
|
m, _, _ := strings.Cut(text, " ")
|
||||||
|
switch strings.ToLower(m) {
|
||||||
|
case "call", "bl", "jirl":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ package debug
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
||||||
@@ -44,3 +45,17 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
|
|||||||
}
|
}
|
||||||
return result
|
return result
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// isCallInsn reports whether disassembled text (riscv64asm.GoSyntax) is a
|
||||||
|
// call. GoSyntax renders jal and jalr calls as CALL; the native mnemonics
|
||||||
|
// are accepted too. The first token must match exactly: a prefix test on
|
||||||
|
// "bl" would catch branches on other architectures, and jalr as ret prints
|
||||||
|
// RET, which must not be stepped over.
|
||||||
|
func isCallInsn(text string) bool {
|
||||||
|
m, _, _ := strings.Cut(text, " ")
|
||||||
|
switch strings.ToLower(m) {
|
||||||
|
case "call", "jal", "jalr":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|||||||
@@ -73,9 +73,29 @@ func decodeRflags(f uint64) string {
|
|||||||
return flags[:len(flags)-1]
|
return flags[:len(flags)-1]
|
||||||
}
|
}
|
||||||
|
|
||||||
// archReturnAddr reads the return address from the stack (amd64 ABI0 convention).
|
// archReturnAddr reads the return address of the current frame (amd64
|
||||||
|
// ABI0 convention). A function that contains a CALL (or has a frame) is
|
||||||
|
// assembled with the prologue PUSHQ BP; MOVQ SP, BP, so mid-function the
|
||||||
|
// word at SP is the saved caller BP, a stack address, and the return
|
||||||
|
// address sits further up. Walk the stack from SP and take the first word
|
||||||
|
// that lies in an executable mapping: stack and data words never do, a
|
||||||
|
// return address always does.
|
||||||
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
|
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
|
||||||
return s.Peek(regs.GetSP())
|
ranges := execRanges(s.pid)
|
||||||
|
for off := uint64(0); off < 512; off += 8 {
|
||||||
|
word, err := s.Peek(regs.RSP + off)
|
||||||
|
if err != nil {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
for _, r := range ranges {
|
||||||
|
if word >= r.lo && word < r.hi {
|
||||||
|
return word, nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// No mapping available or nothing code-like on the stack: fall back to
|
||||||
|
// the raw entry convention, [SP] before any push.
|
||||||
|
return s.Peek(regs.RSP)
|
||||||
}
|
}
|
||||||
|
|
||||||
// archSPLabel returns the SP register name for display.
|
// archSPLabel returns the SP register name for display.
|
||||||
|
|||||||
@@ -5,7 +5,10 @@
|
|||||||
|
|
||||||
package debug
|
package debug
|
||||||
|
|
||||||
import "fmt"
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
)
|
||||||
|
|
||||||
func printRegs(regs *Regs, codeBase, funcOff uint64) {
|
func printRegs(regs *Regs, codeBase, funcOff uint64) {
|
||||||
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
|
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
|
||||||
@@ -31,8 +34,8 @@ func printRegs(regs *Regs, codeBase, funcOff uint64) {
|
|||||||
func printVectorRegs(v *VectorRegs) {
|
func printVectorRegs(v *VectorRegs) {
|
||||||
fmt.Println("\n Vector registers (V0-V31):")
|
fmt.Println("\n Vector registers (V0-V31):")
|
||||||
for i := 0; i < 32; i += 2 {
|
for i := 0; i < 32; i += 2 {
|
||||||
fmt.Printf(" V%-2d = %016x%016x\n", i, v.V[i][8], v.V[i][0])
|
fmt.Printf(" V%-2d = %016x%016x\n", i, binary.LittleEndian.Uint64(v.V[i][8:16]), binary.LittleEndian.Uint64(v.V[i][0:8]))
|
||||||
fmt.Printf(" V%-2d = %016x%016x\n", i+1, v.V[i+1][8], v.V[i+1][0])
|
fmt.Printf(" V%-2d = %016x%016x\n", i+1, binary.LittleEndian.Uint64(v.V[i+1][8:16]), binary.LittleEndian.Uint64(v.V[i+1][0:8]))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,427 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build linux && amd64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"runtime"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
"unsafe"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Integration tests beyond the basic entry breakpoint: hardware watchpoints,
|
||||||
|
// conditional breakpoints, next/finish over a CALL, faulting kernels and the
|
||||||
|
// xstate vector-register readout. All drive a real ptrace session, so they
|
||||||
|
// run on amd64 hosts only.
|
||||||
|
|
||||||
|
// writeKernel writes an assembly source to a temporary file with the
|
||||||
|
// architecture suffix the assembler dispatcher expects.
|
||||||
|
func writeKernel(t *testing.T, src string) string {
|
||||||
|
t.Helper()
|
||||||
|
path := filepath.Join(t.TempDir(), "kernel_amd64.s")
|
||||||
|
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatalf("write kernel: %v", err)
|
||||||
|
}
|
||||||
|
return path
|
||||||
|
}
|
||||||
|
|
||||||
|
// launchKernel launches a session for the kernel source and returns the
|
||||||
|
// session, its breakpoint manager and the function layout.
|
||||||
|
func launchKernel(t *testing.T, bin, path, funcName string, args []byte) (*Session, *Breakpoints, asm.FuncLayout) {
|
||||||
|
t.Helper()
|
||||||
|
k, err := verify.Load(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
t.Cleanup(k.Close)
|
||||||
|
fl, err := k.Func(funcName)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Func: %v", err)
|
||||||
|
}
|
||||||
|
if len(args) < fl.Args {
|
||||||
|
padded := make([]byte, fl.Args)
|
||||||
|
copy(padded, args)
|
||||||
|
args = padded
|
||||||
|
}
|
||||||
|
sess, err := Launch(bin, path, funcName, args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Launch: %v", err)
|
||||||
|
}
|
||||||
|
t.Cleanup(sess.Kill)
|
||||||
|
bm := NewBreakpoints(sess)
|
||||||
|
return sess, bm, fl
|
||||||
|
}
|
||||||
|
|
||||||
|
// runToEntry resumes the freshly launched debuggee until the breakpoint at
|
||||||
|
// the function entry traps, mirroring the REPL continue loop: the debuggee
|
||||||
|
// SIGSTOPs twice (launch barrier and entry barrier) before entering the JIT
|
||||||
|
// call.
|
||||||
|
func runToEntry(t *testing.T, sess *Session, bm *Breakpoints, entry uint64) {
|
||||||
|
t.Helper()
|
||||||
|
for range 50 {
|
||||||
|
for _, bp := range bm.All() {
|
||||||
|
bm.Reinsert(bp.Addr)
|
||||||
|
}
|
||||||
|
if err := sess.Continue(); err != nil {
|
||||||
|
t.Fatalf("Continue: %v", err)
|
||||||
|
}
|
||||||
|
if sess.Exited() {
|
||||||
|
t.Fatal("debuggee exited before the entry breakpoint trapped")
|
||||||
|
}
|
||||||
|
regs, err := sess.GetRegs()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetRegs: %v", err)
|
||||||
|
}
|
||||||
|
if bm.HandleTrap(®s) != nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatal("no entry breakpoint trap after 50 resumes")
|
||||||
|
}
|
||||||
|
|
||||||
|
// captureStdout runs fn with os.Stdout redirected to a pipe and returns
|
||||||
|
// what it printed (the REPL writes its reports to stdout).
|
||||||
|
func captureStdout(t *testing.T, fn func()) string {
|
||||||
|
t.Helper()
|
||||||
|
r, w, err := os.Pipe()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("pipe: %v", err)
|
||||||
|
}
|
||||||
|
old := os.Stdout
|
||||||
|
os.Stdout = w
|
||||||
|
done := make(chan string, 1)
|
||||||
|
go func() {
|
||||||
|
b, _ := io.ReadAll(r)
|
||||||
|
done <- string(b)
|
||||||
|
}()
|
||||||
|
defer func() { os.Stdout = old }()
|
||||||
|
fn()
|
||||||
|
w.Close()
|
||||||
|
return <-done
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestWatchpointArmRunHit proves the debug-register offsets: the watchpoint
|
||||||
|
// must fire on the store, with si_addr naming the watched address. The
|
||||||
|
// kernel writes its return value to ret+0(FP), which is the 8-byte word
|
||||||
|
// right above the stack pointer at entry.
|
||||||
|
func TestWatchpointArmRunHit(t *testing.T) {
|
||||||
|
runtime.LockOSThread()
|
||||||
|
defer runtime.UnlockOSThread()
|
||||||
|
bin := buildGasm(t)
|
||||||
|
|
||||||
|
const kernel = `#include "textflag.h"
|
||||||
|
|
||||||
|
// func wpret() int64
|
||||||
|
TEXT ·wpret(SB), NOSPLIT, $0-8
|
||||||
|
MOVQ $0x5a5a5a5a5a5a5a5a, AX
|
||||||
|
MOVQ AX, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
path := writeKernel(t, kernel)
|
||||||
|
sess, bm, fl := launchKernel(t, bin, path, "wpret", nil)
|
||||||
|
|
||||||
|
entry := sess.CodeBase() + uint64(fl.Offset)
|
||||||
|
if _, err := bm.Set(entry, "entry"); err != nil {
|
||||||
|
t.Fatalf("Set: %v", err)
|
||||||
|
}
|
||||||
|
runToEntry(t, sess, bm, entry)
|
||||||
|
|
||||||
|
regs, err := sess.GetRegs()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetRegs: %v", err)
|
||||||
|
}
|
||||||
|
watched := regs.RSP + 8 // ret+0(FP): the store target
|
||||||
|
|
||||||
|
slot := sess.FindFreeWatchpointSlot()
|
||||||
|
if slot < 0 {
|
||||||
|
t.Fatal("no free watchpoint slot")
|
||||||
|
}
|
||||||
|
if err := sess.SetWatchpoint(slot, watched, WatchWrite, 8); err != nil {
|
||||||
|
t.Fatalf("SetWatchpoint: %v (wrong debug-register offsets?)", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := sess.Continue(); err != nil {
|
||||||
|
t.Fatalf("Continue: %v", err)
|
||||||
|
}
|
||||||
|
reason, addr := sess.StopInfo()
|
||||||
|
if reason != StopWatchpoint {
|
||||||
|
t.Fatalf("stop reason = %v, want StopWatchpoint (DR0-DR3/DR7 offsets are wrong)", reason)
|
||||||
|
}
|
||||||
|
if addr != watched {
|
||||||
|
t.Fatalf("watchpoint address = %#x, want %#x", addr, watched)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The watched word holds the stored value: x86 data breakpoints are
|
||||||
|
// reported with the access complete.
|
||||||
|
if word, err := sess.Peek(watched); err != nil || word != 0x5a5a5a5a5a5a5a5a {
|
||||||
|
t.Errorf("watched word = %#x (err %v), want 0x5a5a5a5a5a5a5a5a", word, err)
|
||||||
|
}
|
||||||
|
if err := sess.ClearWatchpoint(slot); err != nil {
|
||||||
|
t.Fatalf("ClearWatchpoint: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestConditionalBreakpointFalseThenTrue proves the false-condition path:
|
||||||
|
// the breakpoint steps over the original instruction, re-arms itself and
|
||||||
|
// keeps running silently, and the true condition stops exactly once with the
|
||||||
|
// register in the expected state.
|
||||||
|
func TestConditionalBreakpointFalseThenTrue(t *testing.T) {
|
||||||
|
runtime.LockOSThread()
|
||||||
|
defer runtime.UnlockOSThread()
|
||||||
|
bin := buildGasm(t)
|
||||||
|
|
||||||
|
const kernel = `#include "textflag.h"
|
||||||
|
|
||||||
|
// func countdown(n int64) int64
|
||||||
|
TEXT ·countdown(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ n+0(FP), CX
|
||||||
|
loop:
|
||||||
|
DECQ CX
|
||||||
|
CMPQ CX, $0
|
||||||
|
JNE loop
|
||||||
|
MOVQ CX, ret+8(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
path := writeKernel(t, kernel)
|
||||||
|
sess, bm, fl := launchKernel(t, bin, path, "countdown", []byte{8})
|
||||||
|
|
||||||
|
loopAddr := sess.CodeBase() + uint64(fl.Offset) + uint64(fl.Labels["loop"])
|
||||||
|
// The length of the breakpointed instruction, from a disassembly taken
|
||||||
|
// before the INT3 is patched in.
|
||||||
|
_, insnLen, err := sess.Disassemble(loopAddr)
|
||||||
|
if err != nil || insnLen <= 0 {
|
||||||
|
t.Fatalf("Disassemble at %#x: len=%d err=%v", loopAddr, insnLen, err)
|
||||||
|
}
|
||||||
|
cond := &Condition{Reg: "rcx", Op: "==", Value: 1}
|
||||||
|
bp, err := bm.SetWithCond(loopAddr, "loop", cond)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("SetWithCond: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
hits := 0
|
||||||
|
exited := false
|
||||||
|
for range 200 {
|
||||||
|
for _, b := range bm.All() {
|
||||||
|
bm.Reinsert(b.Addr)
|
||||||
|
}
|
||||||
|
if err := sess.Continue(); err != nil {
|
||||||
|
exited = true
|
||||||
|
break // the debuggee finished
|
||||||
|
}
|
||||||
|
if sess.Exited() {
|
||||||
|
exited = true
|
||||||
|
break
|
||||||
|
}
|
||||||
|
if sig := sess.LastSignal(); sig != 0 {
|
||||||
|
t.Fatalf("unexpected signal stop %v", sig)
|
||||||
|
}
|
||||||
|
regs, err := sess.GetRegs()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetRegs: %v", err)
|
||||||
|
}
|
||||||
|
if hit := bm.HandleTrap(®s); hit != nil {
|
||||||
|
hits++
|
||||||
|
if regs.RCX != 1 {
|
||||||
|
t.Fatalf("hit with RCX=%d, want 1", regs.RCX)
|
||||||
|
}
|
||||||
|
// Park after the instruction, as the REPL does.
|
||||||
|
if err := sess.Step(); err != nil {
|
||||||
|
t.Fatalf("Step: %v", err)
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
// A false evaluation must leave the debuggee past the whole
|
||||||
|
// original instruction: a PC inside it (trapAddr+1 on amd64)
|
||||||
|
// means the resume happens mid-instruction.
|
||||||
|
fresh, err := sess.GetRegs()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetRegs: %v", err)
|
||||||
|
}
|
||||||
|
if fresh.RIP > loopAddr && fresh.RIP < loopAddr+uint64(insnLen) {
|
||||||
|
t.Fatalf("false evaluation left the PC at %#x, inside the %d-byte instruction at %#x",
|
||||||
|
fresh.RIP, insnLen, loopAddr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if hits != 1 {
|
||||||
|
t.Fatalf("conditional breakpoint hit %d times, want exactly 1 (false evaluations must run through silently)", hits)
|
||||||
|
}
|
||||||
|
if bp.Hits() != 1 {
|
||||||
|
t.Errorf("bp.Hits() = %d, want 1", bp.Hits())
|
||||||
|
}
|
||||||
|
if !exited || !sess.Exited() {
|
||||||
|
t.Fatal("debuggee did not run to completion after the conditional hit")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestNextAndFinishOverCall proves next and finish evaluate the trap with
|
||||||
|
// registers fetched after the stop: next lands exactly on the instruction
|
||||||
|
// after the CALL, and finish stops exactly on the return address.
|
||||||
|
func TestNextAndFinishOverCall(t *testing.T) {
|
||||||
|
runtime.LockOSThread()
|
||||||
|
defer runtime.UnlockOSThread()
|
||||||
|
bin := buildGasm(t)
|
||||||
|
|
||||||
|
const kernel = `#include "textflag.h"
|
||||||
|
|
||||||
|
// func caller(x int64) int64
|
||||||
|
// The argument travels in AX: FP argument slots of CALL-bearing functions
|
||||||
|
// are an assembler concern outside this test's scope.
|
||||||
|
TEXT ·caller(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ $5, AX
|
||||||
|
CALL ·bump(SB)
|
||||||
|
aftercall:
|
||||||
|
MOVQ AX, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func bump(x int64) int64
|
||||||
|
TEXT ·bump(SB), NOSPLIT, $0-0
|
||||||
|
ADDQ $3, AX
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
path := writeKernel(t, kernel)
|
||||||
|
|
||||||
|
// next: step the prologue and the constant load (3 instructions), then
|
||||||
|
// step over the CALL and check the landing address and RAX.
|
||||||
|
sess, bm, fl := launchKernel(t, bin, path, "caller", nil)
|
||||||
|
entry := sess.CodeBase() + uint64(fl.Offset)
|
||||||
|
if _, err := bm.Set(entry, "entry"); err != nil {
|
||||||
|
t.Fatalf("Set: %v", err)
|
||||||
|
}
|
||||||
|
runToEntry(t, sess, bm, entry)
|
||||||
|
afterOff := uint64(fl.Labels["aftercall"])
|
||||||
|
|
||||||
|
out := captureStdout(t, func() {
|
||||||
|
REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, nil, nil,
|
||||||
|
strings.NewReader("step 3\nnext\nregs\nquit\n"))
|
||||||
|
})
|
||||||
|
if !strings.Contains(out, fmt.Sprintf("func+%#x", afterOff)) {
|
||||||
|
t.Errorf("next did not land on the instruction after the CALL (func+%#x); output:\n%s", afterOff, out)
|
||||||
|
}
|
||||||
|
if !strings.Contains(out, "RAX = 0x0000000000000008") {
|
||||||
|
t.Errorf("callee did not run exactly once under next (want RAX=8); output:\n%s", out)
|
||||||
|
}
|
||||||
|
|
||||||
|
// finish: run to the return address read off the stack at entry.
|
||||||
|
sess2, bm2, fl2 := launchKernel(t, bin, path, "caller", nil)
|
||||||
|
entry2 := sess2.CodeBase() + uint64(fl2.Offset)
|
||||||
|
if _, err := bm2.Set(entry2, "entry"); err != nil {
|
||||||
|
t.Fatalf("Set: %v", err)
|
||||||
|
}
|
||||||
|
runToEntry(t, sess2, bm2, entry2)
|
||||||
|
regs, err := sess2.GetRegs()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetRegs: %v", err)
|
||||||
|
}
|
||||||
|
retAddr, err := sess2.Peek(regs.RSP)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Peek return address: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
out2 := captureStdout(t, func() {
|
||||||
|
REPL(sess2, bm2, sess2.CodeBase(), fl2.Offset, fl2.Size, fl2.Args, nil, nil,
|
||||||
|
strings.NewReader("step 1\nfinish\nquit\n"))
|
||||||
|
})
|
||||||
|
want := fmt.Sprintf("finished, now at %#x\n", retAddr)
|
||||||
|
if !strings.Contains(out2, want) {
|
||||||
|
t.Errorf("finish stopped at the wrong PC; want %q in output:\n%s", want, out2)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSignalStopSurfaced proves a faulting kernel surfaces as a reported
|
||||||
|
// stop instead of an infinite fault loop. A regression here hangs, so a
|
||||||
|
// watchdog fails the run rather than letting CI stall.
|
||||||
|
func TestSignalStopSurfaced(t *testing.T) {
|
||||||
|
runtime.LockOSThread()
|
||||||
|
defer runtime.UnlockOSThread()
|
||||||
|
bin := buildGasm(t)
|
||||||
|
|
||||||
|
const kernel = `#include "textflag.h"
|
||||||
|
|
||||||
|
// func crash() int64
|
||||||
|
TEXT ·crash(SB), NOSPLIT, $0-8
|
||||||
|
XORQ AX, AX
|
||||||
|
MOVQ (AX), AX
|
||||||
|
MOVQ AX, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
path := writeKernel(t, kernel)
|
||||||
|
sess, bm, _ := launchKernel(t, bin, path, "crash", nil)
|
||||||
|
|
||||||
|
timer := time.AfterFunc(time.Minute, func() {
|
||||||
|
panic("watchdog: the debugger hung on the faulting kernel instead of reporting the signal stop")
|
||||||
|
})
|
||||||
|
defer timer.Stop()
|
||||||
|
|
||||||
|
out := captureStdout(t, func() {
|
||||||
|
REPL(sess, bm, sess.CodeBase(), 0, 0, 0, nil, nil,
|
||||||
|
strings.NewReader("continue\nquit\n"))
|
||||||
|
})
|
||||||
|
if !strings.Contains(out, "stopped on signal") {
|
||||||
|
t.Errorf("SIGSEGV did not surface as a reported stop; output:\n%s", out)
|
||||||
|
}
|
||||||
|
if !sess.Exited() {
|
||||||
|
t.Error("debuggee should be killed by quit after the signal stop")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGetVectorRegsXState proves the NT_X86_XSTATE readout: the request
|
||||||
|
// succeeds on a normal process and the XMM halves agree with
|
||||||
|
// PTRACE_GETFPREGS.
|
||||||
|
func TestGetVectorRegsXState(t *testing.T) {
|
||||||
|
// The FPRegs layout must mirror the kernel's user_fpregs_struct
|
||||||
|
// exactly: PTRACE_GETFPREGS fills all 512 bytes, so a short struct
|
||||||
|
// overflows the caller's memory.
|
||||||
|
if got := unsafe.Sizeof(FPRegs{}); got != 512 {
|
||||||
|
t.Fatalf("sizeof(FPRegs) = %d, want 512", got)
|
||||||
|
}
|
||||||
|
if got := unsafe.Offsetof(FPRegs{}.XMM); got != 160 {
|
||||||
|
t.Fatalf("offsetof(FPRegs.XMM) = %d, want 160", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
runtime.LockOSThread()
|
||||||
|
defer runtime.UnlockOSThread()
|
||||||
|
bin := buildGasm(t)
|
||||||
|
|
||||||
|
const kernel = `#include "textflag.h"
|
||||||
|
|
||||||
|
// func vprobe() int64
|
||||||
|
TEXT ·vprobe(SB), NOSPLIT, $0-8
|
||||||
|
MOVQ $1, AX
|
||||||
|
MOVQ AX, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
path := writeKernel(t, kernel)
|
||||||
|
sess, bm, fl := launchKernel(t, bin, path, "vprobe", nil)
|
||||||
|
|
||||||
|
entry := sess.CodeBase() + uint64(fl.Offset)
|
||||||
|
if _, err := bm.Set(entry, "entry"); err != nil {
|
||||||
|
t.Fatalf("Set: %v", err)
|
||||||
|
}
|
||||||
|
runToEntry(t, sess, bm, entry)
|
||||||
|
|
||||||
|
v, err := sess.GetVectorRegs()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetVectorRegs: %v", err)
|
||||||
|
}
|
||||||
|
fp, err := sess.GetFPRegs()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetFPRegs: %v", err)
|
||||||
|
}
|
||||||
|
for i := range 16 {
|
||||||
|
if !bytes.Equal(v.YMM[i][:16], fp.XMM[i][:]) {
|
||||||
|
t.Errorf("YMM%d low half %x, want the FPRegs XMM half %x", i, v.YMM[i][:16], fp.XMM[i][:])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
+66
-28
@@ -23,7 +23,13 @@ type Session struct {
|
|||||||
stopped bool
|
stopped bool
|
||||||
exited bool
|
exited bool
|
||||||
codeBase uint64 // base address of the JIT code in the debuggee
|
codeBase uint64 // base address of the JIT code in the debuggee
|
||||||
wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 BADVR0-15)
|
tmpDir string // scratch directory of the session, removed on Kill
|
||||||
|
wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 DBGWVR0-15)
|
||||||
|
// lastSignal holds the signal of the most recent stop when that stop
|
||||||
|
// was a genuine signal-delivery-stop the caller must see (a fault such
|
||||||
|
// as SIGSEGV, SIGBUS, SIGFPE or SIGILL); 0 for breakpoint traps,
|
||||||
|
// single-steps, SIGSTOP and suppressed runtime signals.
|
||||||
|
lastSignal syscall.Signal
|
||||||
}
|
}
|
||||||
|
|
||||||
// Launch starts the debuggee subprocess (gasm debug --target ...) and
|
// Launch starts the debuggee subprocess (gasm debug --target ...) and
|
||||||
@@ -78,7 +84,7 @@ func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec s
|
|||||||
return nil, nil, fmt.Errorf("debug: start debuggee: %w", err)
|
return nil, nil, fmt.Errorf("debug: start debuggee: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
s := &Session{pid: cmd.Process.Pid, cmd: cmd}
|
s := &Session{pid: cmd.Process.Pid, cmd: cmd, tmpDir: tmpDir}
|
||||||
|
|
||||||
readyFile := filepath.Join(tmpDir, "ready")
|
readyFile := filepath.Join(tmpDir, "ready")
|
||||||
for range 500 {
|
for range 500 {
|
||||||
@@ -125,29 +131,17 @@ func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec s
|
|||||||
return s, bufAddrs, nil
|
return s, bufAddrs, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// wait waits for the debuggee to stop and returns the wait status.
|
|
||||||
func (s *Session) wait() error {
|
|
||||||
var ws syscall.WaitStatus
|
|
||||||
_, err := syscall.Wait4(s.pid, &ws, 0, nil)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
if ws.Exited() {
|
|
||||||
s.exited = true
|
|
||||||
return fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
|
|
||||||
}
|
|
||||||
s.stopped = true
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// waitStopped consumes ptrace-stop events until one the debugger cares
|
// waitStopped consumes ptrace-stop events until one the debugger cares
|
||||||
// about arrives: SIGTRAP (a breakpoint or a completed single-step) or the
|
// about arrives: SIGTRAP (a breakpoint or a completed single-step), the
|
||||||
// debuggee's own SIGSTOP. A Go tracee's runtime raises SIGURG for
|
// debuggee's own SIGSTOP, or a genuine signal-delivery-stop. A Go tracee's
|
||||||
// asynchronous preemption, and every signal on a traced thread surfaces as
|
// runtime raises SIGURG for asynchronous preemption, and every signal on a
|
||||||
// a signal-delivery-stop, so those are suppressed and the tracee resumed
|
// traced thread surfaces as a signal-delivery-stop, so SIGURG is suppressed
|
||||||
// without them. Runtime noise is why a single wait can return in the
|
// and the tracee resumed without it. Every other signal (SIGSEGV, SIGBUS,
|
||||||
// middle of runtime code and a resume can then fail: the event stream must
|
// SIGFPE, SIGILL, ...) is returned to the caller: resuming with signal 0
|
||||||
// be drained by the tracer.
|
// would restart the faulting instruction and fault forever, so a faulting
|
||||||
|
// kernel must surface as a stop the caller reports. Runtime noise is also
|
||||||
|
// why a single wait can return in the middle of runtime code and a resume
|
||||||
|
// can then fail: the event stream must be drained by the tracer.
|
||||||
func (s *Session) waitStopped() (syscall.Signal, error) {
|
func (s *Session) waitStopped() (syscall.Signal, error) {
|
||||||
for {
|
for {
|
||||||
var ws syscall.WaitStatus
|
var ws syscall.WaitStatus
|
||||||
@@ -165,10 +159,12 @@ func (s *Session) waitStopped() (syscall.Signal, error) {
|
|||||||
switch sig := ws.StopSignal(); sig {
|
switch sig := ws.StopSignal(); sig {
|
||||||
case syscall.SIGTRAP, syscall.SIGSTOP:
|
case syscall.SIGTRAP, syscall.SIGSTOP:
|
||||||
s.stopped = true
|
s.stopped = true
|
||||||
|
s.lastSignal = 0
|
||||||
return sig, nil
|
return sig, nil
|
||||||
default:
|
case syscall.SIGURG:
|
||||||
// Runtime noise (SIGURG preemption and friends): resume the
|
// Go runtime asynchronous preemption: resume the tracee
|
||||||
// tracee without delivering the signal.
|
// without delivering the signal.
|
||||||
|
s.lastSignal = 0
|
||||||
if _, _, errno := syscall.Syscall6(
|
if _, _, errno := syscall.Syscall6(
|
||||||
syscall.SYS_PTRACE,
|
syscall.SYS_PTRACE,
|
||||||
uintptr(syscall.PTRACE_CONT),
|
uintptr(syscall.PTRACE_CONT),
|
||||||
@@ -177,10 +173,22 @@ func (s *Session) waitStopped() (syscall.Signal, error) {
|
|||||||
); errno != 0 {
|
); errno != 0 {
|
||||||
return 0, fmt.Errorf("debug: PTRACE_CONT: %w", errno)
|
return 0, fmt.Errorf("debug: PTRACE_CONT: %w", errno)
|
||||||
}
|
}
|
||||||
|
default:
|
||||||
|
// A genuine signal-delivery-stop. Report it; the caller
|
||||||
|
// decides how to proceed.
|
||||||
|
s.stopped = true
|
||||||
|
s.lastSignal = sig
|
||||||
|
return sig, nil
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// LastSignal returns the signal of the most recent stop when that stop was
|
||||||
|
// a genuine signal-delivery-stop (a fault such as SIGSEGV, SIGFPE, SIGILL
|
||||||
|
// or SIGBUS), and 0 for breakpoint traps, single-steps, SIGSTOP and
|
||||||
|
// suppressed runtime signals.
|
||||||
|
func (s *Session) LastSignal() syscall.Signal { return s.lastSignal }
|
||||||
|
|
||||||
// Peek reads a word (8 bytes) from the debuggee's memory at addr.
|
// Peek reads a word (8 bytes) from the debuggee's memory at addr.
|
||||||
func (s *Session) Peek(addr uint64) (uint64, error) {
|
func (s *Session) Peek(addr uint64) (uint64, error) {
|
||||||
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
|
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
|
||||||
@@ -294,7 +302,8 @@ func (s *Session) Pid() int { return s.pid }
|
|||||||
// CodeBase returns the base address of the JIT code in the debuggee.
|
// CodeBase returns the base address of the JIT code in the debuggee.
|
||||||
func (s *Session) CodeBase() uint64 { return s.codeBase }
|
func (s *Session) CodeBase() uint64 { return s.codeBase }
|
||||||
|
|
||||||
// Kill terminates the debuggee.
|
// Kill terminates the debuggee and removes the session's scratch
|
||||||
|
// directory, so a successful session leaves no gasm-debug-* debris behind.
|
||||||
func (s *Session) Kill() {
|
func (s *Session) Kill() {
|
||||||
if !s.exited {
|
if !s.exited {
|
||||||
syscall.Kill(s.pid, syscall.SIGKILL)
|
syscall.Kill(s.pid, syscall.SIGKILL)
|
||||||
@@ -304,6 +313,35 @@ func (s *Session) Kill() {
|
|||||||
if s.cmd != nil && s.cmd.Process != nil {
|
if s.cmd != nil && s.cmd.Process != nil {
|
||||||
s.cmd.Wait()
|
s.cmd.Wait()
|
||||||
}
|
}
|
||||||
|
if s.tmpDir != "" {
|
||||||
|
os.RemoveAll(s.tmpDir)
|
||||||
|
s.tmpDir = ""
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// execRange is one executable mapping of the debuggee.
|
||||||
|
type execRange struct {
|
||||||
|
lo, hi uint64
|
||||||
|
}
|
||||||
|
|
||||||
|
// execRanges parses the debuggee's executable mappings from /proc/pid/maps.
|
||||||
|
func execRanges(pid int) []execRange {
|
||||||
|
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
|
||||||
|
if err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
var out []execRange
|
||||||
|
for line := range strings.SplitSeq(string(data), "\n") {
|
||||||
|
fields := strings.Fields(line)
|
||||||
|
if len(fields) < 2 || !strings.Contains(fields[1], "x") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
var lo, hi uint64
|
||||||
|
if _, err := fmt.Sscanf(fields[0], "%x-%x", &lo, &hi); err == nil {
|
||||||
|
out = append(out, execRange{lo, hi})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
// findRWXMapping reads /proc/pid/maps and returns the base address of the
|
// findRWXMapping reads /proc/pid/maps and returns the base address of the
|
||||||
|
|||||||
+68
-11
@@ -6,6 +6,7 @@
|
|||||||
package debug
|
package debug
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"encoding/binary"
|
||||||
"fmt"
|
"fmt"
|
||||||
"syscall"
|
"syscall"
|
||||||
"unsafe"
|
"unsafe"
|
||||||
@@ -44,20 +45,24 @@ func (s *Session) SetRegs(regs *Regs) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// FPRegs holds the x87 FPU and SSE (XMM) register state from PTRACE_GETFPREGS.
|
// FPRegs holds the x87 FPU and SSE (XMM) register state from
|
||||||
|
// PTRACE_GETFPREGS. The layout is the kernel's struct user_fpregs_struct
|
||||||
|
// (sys/user.h), the FXSAVE image: 512 bytes with XMM0-15 at offset 160.
|
||||||
|
// The i387 fcs/ds segment fields do not exist in the 64-bit layout. The
|
||||||
|
// size matters: the copy fills all 512 bytes, so a short or misaligned
|
||||||
|
// struct makes PTRACE_GETFPREGS overflow the caller's memory.
|
||||||
type FPRegs struct {
|
type FPRegs struct {
|
||||||
FCW uint16
|
FCW uint16
|
||||||
FSW uint16
|
FSW uint16
|
||||||
FTW byte
|
FTW uint16
|
||||||
FOP uint16
|
FOP uint16
|
||||||
FIP uint64
|
FIP uint64
|
||||||
FCS uint16
|
|
||||||
FDP uint64
|
FDP uint64
|
||||||
FDS uint16
|
|
||||||
MXCSR uint32
|
MXCSR uint32
|
||||||
MXCSRMask uint32
|
MXCSRMask uint32
|
||||||
ST [8][16]byte // x87 stack (10 bytes per reg, padded to 16)
|
ST [8][16]byte // x87 stack (10 bytes per reg, padded to 16)
|
||||||
XMM [16][16]byte // XMM0-15
|
XMM [16][16]byte // XMM0-15, struct offset 160
|
||||||
|
Reserved [96]byte // FXSAVE padding, to the full 512 bytes
|
||||||
}
|
}
|
||||||
|
|
||||||
// GetFPRegs retrieves the FPU/SSE register state of the stopped debuggee.
|
// GetFPRegs retrieves the FPU/SSE register state of the stopped debuggee.
|
||||||
@@ -82,16 +87,68 @@ type VectorRegs struct {
|
|||||||
YMM [16][32]byte // YMM0-15 (full 256-bit values)
|
YMM [16][32]byte // YMM0-15 (full 256-bit values)
|
||||||
}
|
}
|
||||||
|
|
||||||
// GetVectorRegs retrieves the YMM registers via PTRACE_GETREGSET + XSAVE.
|
// NT_X86_XSTATE (0x202), the xsave extended-state regset
|
||||||
|
// (include/uapi/linux/elf.h).
|
||||||
|
const ntX86XState = 0x202
|
||||||
|
|
||||||
|
// Layout of the buffer PTRACE_GETREGSET returns for NT_X86_XSTATE: the
|
||||||
|
// 512-byte legacy fxsave image (x87 state in 0-159, XMM0-15 in 160-511),
|
||||||
|
// then the 64-byte xsave header whose first 8 bytes are xstate_bv, then one
|
||||||
|
// component per set feature bit, each 64-byte aligned. The YMM high halves
|
||||||
|
// are the first extended component, at offset 576; that offset is fixed by
|
||||||
|
// the ISA on AVX-capable x86-64. XFEATURE_MASK_YMM is bit 2 of xstate_bv
|
||||||
|
// (arch/x86/include/asm/fpu/types.h); the high halves are zero when the bit
|
||||||
|
// is clear.
|
||||||
|
const (
|
||||||
|
xsaveXMMOffset = 160
|
||||||
|
xsaveXMMSize = 256
|
||||||
|
xsaveHeaderOffset = 512
|
||||||
|
xsaveBVOffset = xsaveHeaderOffset
|
||||||
|
ymmOffset = xsaveHeaderOffset + 64 // 576
|
||||||
|
ymmSize = 256 // 16 registers, 16 bytes each
|
||||||
|
xfeatureMaskYMM = 1 << 2
|
||||||
|
xstateMaxBuffer = 4096 // CPUID(0xD).xsave_size is far below this
|
||||||
|
)
|
||||||
|
|
||||||
|
// GetVectorRegs retrieves the YMM registers via PTRACE_GETREGSET on
|
||||||
|
// NT_X86_XSTATE. The low (XMM) halves always come from the legacy image;
|
||||||
|
// the high halves are copied only when xstate_bv reports the YMM feature,
|
||||||
|
// and read as zero otherwise. When the regset request fails the FP image
|
||||||
|
// still provides correct XMM halves, so that is the fallback.
|
||||||
func (s *Session) GetVectorRegs() (VectorRegs, error) {
|
func (s *Session) GetVectorRegs() (VectorRegs, error) {
|
||||||
var v VectorRegs
|
var v VectorRegs
|
||||||
fp, err := s.GetFPRegs()
|
buf := make([]byte, xstateMaxBuffer)
|
||||||
if err != nil {
|
iovec := syscall.Iovec{
|
||||||
return v, err
|
Base: &buf[0],
|
||||||
|
Len: uint64(len(buf)),
|
||||||
}
|
}
|
||||||
|
_, _, errno := syscall.Syscall6(
|
||||||
|
syscall.SYS_PTRACE,
|
||||||
|
uintptr(syscall.PTRACE_GETREGSET),
|
||||||
|
uintptr(s.pid),
|
||||||
|
uintptr(ntX86XState),
|
||||||
|
uintptr(unsafe.Pointer(&iovec)),
|
||||||
|
0, 0,
|
||||||
|
)
|
||||||
|
if errno != 0 {
|
||||||
|
fp, err := s.GetFPRegs()
|
||||||
|
if err != nil {
|
||||||
|
return v, err
|
||||||
|
}
|
||||||
|
for i := range 16 {
|
||||||
|
copy(v.YMM[i][:16], fp.XMM[i][:])
|
||||||
|
}
|
||||||
|
return v, nil
|
||||||
|
}
|
||||||
|
n := int(iovec.Len)
|
||||||
for i := range 16 {
|
for i := range 16 {
|
||||||
for j := range 16 {
|
copy(v.YMM[i][:16], buf[xsaveXMMOffset+16*i:xsaveXMMOffset+16*i+16])
|
||||||
v.YMM[i][j] = fp.XMM[i][j]
|
}
|
||||||
|
if n >= ymmOffset+ymmSize {
|
||||||
|
if binary.LittleEndian.Uint64(buf[xsaveBVOffset:xsaveBVOffset+8])&xfeatureMaskYMM != 0 {
|
||||||
|
for i := range 16 {
|
||||||
|
copy(v.YMM[i][16:], buf[ymmOffset+16*i:ymmOffset+16*i+16])
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return v, nil
|
return v, nil
|
||||||
|
|||||||
@@ -91,5 +91,9 @@ func (r *Regs) RegValue(name string) (uint64, bool) {
|
|||||||
// breakpointInsn is the software breakpoint instruction.
|
// breakpointInsn is the software breakpoint instruction.
|
||||||
var breakpointInsn = []byte{0xCC} // INT3
|
var breakpointInsn = []byte{0xCC} // INT3
|
||||||
|
|
||||||
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
|
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
|
||||||
|
// a trap. x86-64 reports the #DB for INT3 with RIP on the byte after the
|
||||||
|
// INT3 (Intel SDM vol 3, "Debug Exceptions"), so the trap address is
|
||||||
|
// PC-1. The other supported architectures leave the PC on the trap
|
||||||
|
// instruction and use 0 there.
|
||||||
const breakpointPCAdjust = 1
|
const breakpointPCAdjust = 1
|
||||||
|
|||||||
@@ -130,5 +130,11 @@ func (r *Regs) RegValue(name string) (uint64, bool) {
|
|||||||
// breakpointInsn is the software breakpoint instruction (BRK #0).
|
// breakpointInsn is the software breakpoint instruction (BRK #0).
|
||||||
var breakpointInsn = []byte{0x00, 0x00, 0x20, 0xD4} // BRK #0
|
var breakpointInsn = []byte{0x00, 0x00, 0x20, 0xD4} // BRK #0
|
||||||
|
|
||||||
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
|
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
|
||||||
const breakpointPCAdjust = 4
|
// a trap: 0, because the arm64 kernel delivers the BRK SIGTRAP with the PC
|
||||||
|
// still on the BRK. do_el0_brk64 calls send_user_sigtrap, which uses
|
||||||
|
// instruction_pointer(regs) unmodified (arch/arm64/kernel/debug-monitors.c);
|
||||||
|
// only the kernel-internal skip paths advance the PC. GDB history agrees:
|
||||||
|
// decr_pc_after_break on aarch64 Linux is 0 (the +4 variant was a QEMU bug,
|
||||||
|
// sourceware PR 17280).
|
||||||
|
const breakpointPCAdjust = 0
|
||||||
|
|||||||
@@ -126,5 +126,9 @@ func (r *Regs) RegValue(name string) (uint64, bool) {
|
|||||||
// breakpointInsn is the software breakpoint instruction (BRK $0).
|
// breakpointInsn is the software breakpoint instruction (BRK $0).
|
||||||
var breakpointInsn = []byte{0x05, 0x00, 0x2a, 0x00} // break 0
|
var breakpointInsn = []byte{0x05, 0x00, 0x2a, 0x00} // break 0
|
||||||
|
|
||||||
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
|
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
|
||||||
const breakpointPCAdjust = 4
|
// a trap: 0, because the kernel delivers the break SIGTRAP with csr_era
|
||||||
|
// still on the break instruction. do_bp passes regs->csr_era straight to
|
||||||
|
// force_sig_fault(SIGTRAP, TRAP_BRKPT, ...) and never adjusts era on the
|
||||||
|
// signal path (arch/loongarch/kernel/traps.c).
|
||||||
|
const breakpointPCAdjust = 0
|
||||||
|
|||||||
@@ -126,5 +126,9 @@ func (r *Regs) RegValue(name string) (uint64, bool) {
|
|||||||
// breakpointInsn is the software breakpoint instruction (EBREAK).
|
// breakpointInsn is the software breakpoint instruction (EBREAK).
|
||||||
var breakpointInsn = []byte{0x73, 0x00, 0x10, 0x00} // ebreak
|
var breakpointInsn = []byte{0x73, 0x00, 0x10, 0x00} // ebreak
|
||||||
|
|
||||||
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
|
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
|
||||||
const breakpointPCAdjust = 4
|
// a trap: 0, because the kernel delivers the EBREAK SIGTRAP with sepc still
|
||||||
|
// on the ebreak. handle_break passes regs->epc straight to
|
||||||
|
// force_sig_fault(SIGTRAP, TRAP_BRKPT, ...) and only the kernel-internal
|
||||||
|
// WARN/CFI paths advance epc (arch/riscv/kernel/traps.c).
|
||||||
|
const breakpointPCAdjust = 0
|
||||||
|
|||||||
+76
-20
@@ -7,9 +7,10 @@ package debug
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"bufio"
|
"bufio"
|
||||||
|
"cmp"
|
||||||
"fmt"
|
"fmt"
|
||||||
"io"
|
"io"
|
||||||
"sort"
|
"slices"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
@@ -32,7 +33,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
|
|||||||
entryAddr := codeBase + uint64(funcOffset)
|
entryAddr := codeBase + uint64(funcOffset)
|
||||||
|
|
||||||
fmt.Printf("stopped at function entry: %#x (%d bytes)\n", entryAddr, funcSize)
|
fmt.Printf("stopped at function entry: %#x (%d bytes)\n", entryAddr, funcSize)
|
||||||
fmt.Println("commands: break <label|addr> | step [n] | continue | disas [n] | regs | where | x <addr> [len] | w <addr> <val...> | labels | quit")
|
fmt.Println("commands: break <label|addr|line> | step [n] | continue | disas [n] | regs | where | x <addr> [len] | w <addr> <val...> | labels | quit")
|
||||||
|
|
||||||
scanner := bufio.NewScanner(in)
|
scanner := bufio.NewScanner(in)
|
||||||
|
|
||||||
@@ -93,7 +94,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
|
|||||||
regs, _ := s.GetRegs()
|
regs, _ := s.GetRegs()
|
||||||
pc := regs.GetPC()
|
pc := regs.GetPC()
|
||||||
text, instLen, _ := s.Disassemble(pc)
|
text, instLen, _ := s.Disassemble(pc)
|
||||||
if strings.HasPrefix(strings.ToLower(text), "call") || strings.HasPrefix(strings.ToLower(text), "bl") {
|
if isCallInsn(text) {
|
||||||
afterAddr := pc + uint64(instLen)
|
afterAddr := pc + uint64(instLen)
|
||||||
_, err := bm.Set(afterAddr, "(next)")
|
_, err := bm.Set(afterAddr, "(next)")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -108,6 +109,21 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
|
|||||||
bm.Clear(afterAddr)
|
bm.Clear(afterAddr)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
if s.Exited() {
|
||||||
|
bm.Clear(afterAddr)
|
||||||
|
fmt.Println("debuggee exited")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if sig := s.LastSignal(); sig != 0 {
|
||||||
|
bm.Clear(afterAddr)
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Fetch the registers after the stop: the trap must be
|
||||||
|
// evaluated against the real PC, not the pre-Continue
|
||||||
|
// snapshot, and a stale SetRegs would clobber live state.
|
||||||
|
regs, _ = s.GetRegs()
|
||||||
bm.HandleTrap(®s)
|
bm.HandleTrap(®s)
|
||||||
bm.Clear(afterAddr)
|
bm.Clear(afterAddr)
|
||||||
} else {
|
} else {
|
||||||
@@ -143,9 +159,21 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
|
|||||||
bm.Clear(retAddr)
|
bm.Clear(retAddr)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
if !s.Exited() {
|
if s.Exited() {
|
||||||
bm.HandleTrap(®s)
|
bm.Clear(retAddr)
|
||||||
|
fmt.Println("debuggee exited")
|
||||||
|
continue
|
||||||
}
|
}
|
||||||
|
if sig := s.LastSignal(); sig != 0 {
|
||||||
|
bm.Clear(retAddr)
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Fetch the registers after the stop, as the continue case
|
||||||
|
// does: HandleTrap must see the PC the trap left behind.
|
||||||
|
regs, _ = s.GetRegs()
|
||||||
|
bm.HandleTrap(®s)
|
||||||
bm.Clear(retAddr)
|
bm.Clear(retAddr)
|
||||||
if s.Exited() {
|
if s.Exited() {
|
||||||
fmt.Println("debuggee exited")
|
fmt.Println("debuggee exited")
|
||||||
@@ -171,6 +199,15 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
|
|||||||
fmt.Println("debuggee exited")
|
fmt.Println("debuggee exited")
|
||||||
break
|
break
|
||||||
}
|
}
|
||||||
|
if sig := s.LastSignal(); sig != 0 {
|
||||||
|
// A genuine signal-delivery-stop (a fault): report it
|
||||||
|
// and return to the prompt. Continuing would restart
|
||||||
|
// the faulting instruction and fault forever.
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
fmt.Printf("stopped on signal %v at %#x (func+%#x)\n",
|
||||||
|
sig, regs.GetPC(), regs.GetPC()-codeBase-uint64(funcOffset))
|
||||||
|
break
|
||||||
|
}
|
||||||
reason, wpAddr := s.StopInfo()
|
reason, wpAddr := s.StopInfo()
|
||||||
if reason == StopWatchpoint {
|
if reason == StopWatchpoint {
|
||||||
fmt.Printf("watchpoint hit at %#x\n", wpAddr)
|
fmt.Printf("watchpoint hit at %#x\n", wpAddr)
|
||||||
@@ -196,7 +233,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
|
|||||||
|
|
||||||
case "break", "b":
|
case "break", "b":
|
||||||
if len(parts) < 2 {
|
if len(parts) < 2 {
|
||||||
fmt.Println("usage: break <label|addr|line> [if <reg> <op> <val>]")
|
fmt.Println("usage: break <label|addr|line> [if <reg> <op> <val|reg|*addr>]")
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
var addr uint64
|
var addr uint64
|
||||||
@@ -221,13 +258,27 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
|
|||||||
reg := strings.ToLower(parts[3])
|
reg := strings.ToLower(parts[3])
|
||||||
op := parts[4]
|
op := parts[4]
|
||||||
operand := parts[5]
|
operand := parts[5]
|
||||||
if val, err := strconv.ParseUint(operand, 0, 64); err == nil {
|
switch {
|
||||||
cond = &Condition{Reg: reg, Op: op, Value: val}
|
case strings.HasPrefix(operand, "*"):
|
||||||
} else {
|
// Memory operand: compare against the 8-byte word at
|
||||||
cond = &Condition{Reg: reg, Op: op, Reg2: strings.ToLower(operand)}
|
// the address, resolved in the debuggee when the
|
||||||
|
// breakpoint is evaluated.
|
||||||
|
addr, err := strconv.ParseUint(strings.TrimPrefix(operand, "*"), 0, 64)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Printf("invalid memory operand: %s\n", operand)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
cond = &Condition{Reg: reg, Op: op, MemAddr: addr}
|
||||||
|
default:
|
||||||
|
val, err := strconv.ParseUint(operand, 0, 64)
|
||||||
|
if err == nil {
|
||||||
|
cond = &Condition{Reg: reg, Op: op, Value: val}
|
||||||
|
} else {
|
||||||
|
cond = &Condition{Reg: reg, Op: op, Reg2: strings.ToLower(operand)}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
} else if len(parts) >= 4 && parts[2] == "if" {
|
} else if len(parts) >= 4 && parts[2] == "if" {
|
||||||
fmt.Println("usage: break <label|addr> if <reg> <op> <value|reg>")
|
fmt.Println("usage: break <label|addr|line> if <reg> <op> <value|reg|*addr>")
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
bp, err := bm.SetWithCond(addr, label, cond)
|
bp, err := bm.SetWithCond(addr, label, cond)
|
||||||
@@ -237,7 +288,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
|
|||||||
}
|
}
|
||||||
condStr := ""
|
condStr := ""
|
||||||
if cond != nil {
|
if cond != nil {
|
||||||
condStr = fmt.Sprintf(" if %s %s %#x", cond.Reg, cond.Op, cond.Value)
|
condStr = " if " + cond.String()
|
||||||
}
|
}
|
||||||
fmt.Printf("breakpoint set: %s at %#x (func+%#x)%s\n", bp.Label, bp.Addr, bp.Addr-codeBase-uint64(funcOffset), condStr)
|
fmt.Printf("breakpoint set: %s at %#x (func+%#x)%s\n", bp.Label, bp.Addr, bp.Addr-codeBase-uint64(funcOffset), condStr)
|
||||||
|
|
||||||
@@ -277,7 +328,11 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
|
|||||||
addr, _ = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
addr, _ = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||||
}
|
}
|
||||||
if len(parts) > 2 {
|
if len(parts) > 2 {
|
||||||
length, _ = strconv.Atoi(parts[2])
|
// A malformed or non-positive length would panic
|
||||||
|
// ReadMemory's make; fall back to the default instead.
|
||||||
|
if n, err := strconv.Atoi(parts[2]); err == nil && n > 0 {
|
||||||
|
length = n
|
||||||
|
}
|
||||||
}
|
}
|
||||||
mem, err := s.ReadMemory(addr, length)
|
mem, err := s.ReadMemory(addr, length)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -336,10 +391,8 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
|
|||||||
}
|
}
|
||||||
|
|
||||||
case "labels", "l":
|
case "labels", "l":
|
||||||
sorted := make([]Label, len(labels))
|
slices.SortFunc(labels, func(a, b Label) int { return cmp.Compare(a.Offset, b.Offset) })
|
||||||
copy(sorted, labels)
|
for _, l := range labels {
|
||||||
sort.Slice(sorted, func(i, j int) bool { return sorted[i].Offset < sorted[j].Offset })
|
|
||||||
for _, l := range sorted {
|
|
||||||
fmt.Printf(" func+%#04x %s\n", l.Offset, l.Name)
|
fmt.Printf(" func+%#04x %s\n", l.Offset, l.Name)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -369,7 +422,10 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
|
|||||||
fmt.Println()
|
fmt.Println()
|
||||||
|
|
||||||
case "help", "h", "?":
|
case "help", "h", "?":
|
||||||
fmt.Printf(` break <label|addr> [if <reg> <op> <val>] set a breakpoint
|
fmt.Printf(` break <label|addr|line> [if <reg> <op> <val|reg|*addr>]
|
||||||
|
set a breakpoint, optionally conditional on a
|
||||||
|
register compared to a constant, a register, or the
|
||||||
|
8-byte word at *addr
|
||||||
delete <label|addr> remove a breakpoint
|
delete <label|addr> remove a breakpoint
|
||||||
info break list all breakpoints
|
info break list all breakpoints
|
||||||
watch <addr> [r|w] [size] set a hardware watchpoint (write by default)
|
watch <addr> [r|w] [size] set a hardware watchpoint (write by default)
|
||||||
@@ -463,8 +519,8 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
|
|||||||
case "unwatch":
|
case "unwatch":
|
||||||
if len(parts) >= 2 {
|
if len(parts) >= 2 {
|
||||||
slot, err := strconv.Atoi(parts[1])
|
slot, err := strconv.Atoi(parts[1])
|
||||||
if err != nil || slot < 0 || slot > 3 {
|
if err != nil || slot < 0 || slot >= maxWatchpoints() {
|
||||||
fmt.Println("usage: unwatch [<slot>]")
|
fmt.Printf("usage: unwatch [<slot 0-%d>]\n", maxWatchpoints()-1)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
if err := s.ClearWatchpoint(slot); err != nil {
|
if err := s.ClearWatchpoint(slot); err != nil {
|
||||||
|
|||||||
+10
-2
@@ -6,6 +6,7 @@
|
|||||||
package debug
|
package debug
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"encoding/binary"
|
||||||
"syscall"
|
"syscall"
|
||||||
"unsafe"
|
"unsafe"
|
||||||
)
|
)
|
||||||
@@ -61,8 +62,15 @@ func (s *Session) StopInfo() (StopReason, uint64) {
|
|||||||
case trapBRKPT:
|
case trapBRKPT:
|
||||||
return StopBreakpoint, 0
|
return StopBreakpoint, 0
|
||||||
case trapHWBRKPT:
|
case trapHWBRKPT:
|
||||||
addr := *(*uint64)(unsafe.Add(unsafe.Pointer(&info), 16))
|
// si_addr sits at struct offset 16 (12 bytes of signo/errno/code
|
||||||
return StopWatchpoint, addr
|
// plus 4 bytes of union alignment). The siginfo buffer is only
|
||||||
|
// 4-byte aligned, so the address is read byte-wise to keep the
|
||||||
|
// load aligned on riscv64 and loong64. What si_addr names is
|
||||||
|
// architecture-specific (the data address on arm64, the
|
||||||
|
// instruction pointer on x86), so the per-architecture
|
||||||
|
// archWatchpointAddr resolves it to the watched address.
|
||||||
|
addr := binary.LittleEndian.Uint64(info._pad[4:12])
|
||||||
|
return StopWatchpoint, archWatchpointAddr(s, addr)
|
||||||
default:
|
default:
|
||||||
return StopSingleStep, 0
|
return StopSingleStep, 0
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -28,15 +28,7 @@ func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
|
|||||||
return fmt.Errorf("debug target: parse: %v", errs[0])
|
return fmt.Errorf("debug target: parse: %v", errs[0])
|
||||||
}
|
}
|
||||||
|
|
||||||
var img *asm.Image
|
img, err := asm.AssembleFileARM64(file)
|
||||||
switch "arm64" {
|
|
||||||
case "arm64":
|
|
||||||
img, err = asm.AssembleFileARM64(file)
|
|
||||||
case "riscv64":
|
|
||||||
img, err = asm.AssembleFileRISCV(file)
|
|
||||||
case "loong64":
|
|
||||||
img, err = asm.AssembleFileLOONG64(file)
|
|
||||||
}
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("debug target: assemble: %w", err)
|
return fmt.Errorf("debug target: assemble: %w", err)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -28,15 +28,7 @@ func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
|
|||||||
return fmt.Errorf("debug target: parse: %v", errs[0])
|
return fmt.Errorf("debug target: parse: %v", errs[0])
|
||||||
}
|
}
|
||||||
|
|
||||||
var img *asm.Image
|
img, err := asm.AssembleFileLOONG64(file)
|
||||||
switch "loong64" {
|
|
||||||
case "arm64":
|
|
||||||
img, err = asm.AssembleFileARM64(file)
|
|
||||||
case "riscv64":
|
|
||||||
img, err = asm.AssembleFileRISCV(file)
|
|
||||||
case "loong64":
|
|
||||||
img, err = asm.AssembleFileLOONG64(file)
|
|
||||||
}
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("debug target: assemble: %w", err)
|
return fmt.Errorf("debug target: assemble: %w", err)
|
||||||
}
|
}
|
||||||
|
|||||||
+8
-2
@@ -12,10 +12,11 @@ type tracer interface {
|
|||||||
Peek(addr uint64) (uint64, error)
|
Peek(addr uint64) (uint64, error)
|
||||||
Poke(addr uint64, val uint64) error
|
Poke(addr uint64, val uint64) error
|
||||||
SetRegs(regs *Regs) error
|
SetRegs(regs *Regs) error
|
||||||
|
Step() error
|
||||||
Pid() int
|
Pid() int
|
||||||
}
|
}
|
||||||
|
|
||||||
// mockTracer records Peek/Poke calls and provides fake register state.
|
// mockTracer records Peek/Poke/Step calls and provides fake register state.
|
||||||
type mockTracer struct {
|
type mockTracer struct {
|
||||||
mem map[uint64]byte
|
mem map[uint64]byte
|
||||||
peeks []uint64
|
peeks []uint64
|
||||||
@@ -23,7 +24,8 @@ type mockTracer struct {
|
|||||||
addr uint64
|
addr uint64
|
||||||
val uint64
|
val uint64
|
||||||
}
|
}
|
||||||
regs *Regs
|
steps int
|
||||||
|
regs *Regs
|
||||||
}
|
}
|
||||||
|
|
||||||
func newMockTracer() *mockTracer {
|
func newMockTracer() *mockTracer {
|
||||||
@@ -57,4 +59,8 @@ func (m *mockTracer) SetRegs(regs *Regs) error {
|
|||||||
m.regs = regs
|
m.regs = regs
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
func (m *mockTracer) Step() error {
|
||||||
|
m.steps++
|
||||||
|
return nil
|
||||||
|
}
|
||||||
func (m *mockTracer) Pid() int { return 42 }
|
func (m *mockTracer) Pid() int { return 42 }
|
||||||
|
|||||||
@@ -8,10 +8,48 @@ package debug
|
|||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
"syscall"
|
"syscall"
|
||||||
|
"unsafe"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Hardware watchpoint support via x86-64 debug registers (DR0-DR3, DR7).
|
// Hardware watchpoint support via x86-64 debug registers (DR0-DR3, DR7).
|
||||||
|
|
||||||
|
// The kernel translates PTRACE_POKEUSER/PEEKUSER offsets inside
|
||||||
|
// [offsetof(struct user, u_debugreg[0]), u_debugreg[7]] to DR0-DR7
|
||||||
|
// (arch/x86/kernel/ptrace.c, arch_ptrace). sys/user.h places u_debugreg at
|
||||||
|
// 0x350: DR0-DR3 are 0x350/0x358/0x360/0x368, DR6 (status) is 0x380 and
|
||||||
|
// DR7 (control) is 0x388. Offsets below 0x350 write user_regs_struct
|
||||||
|
// fields (r15 at 0x0, r10 at 0x38), not debug registers.
|
||||||
|
const (
|
||||||
|
drOffset = 0x350 // offsetof(struct user, u_debugreg[0]), DR0
|
||||||
|
dr6Off = 0x380 // offsetof(struct user, u_debugreg[6]), DR6
|
||||||
|
dr7Off = 0x388 // offsetof(struct user, u_debugreg[7]), DR7
|
||||||
|
)
|
||||||
|
|
||||||
|
// archWatchpointAddr resolves the address of the watchpoint that fired.
|
||||||
|
// x86 delivers si_addr = the instruction pointer of the trapping access
|
||||||
|
// (arch/x86/kernel/ptrace.c send_sigtrap passes regs->ip), so the watched
|
||||||
|
// data address is recovered from DR6's slot bits (B0-B3, positive polarity
|
||||||
|
// through PEEKUSER) and the matching DR0-DR3.
|
||||||
|
func archWatchpointAddr(s *Session, siAddr uint64) uint64 {
|
||||||
|
dr6, err := ptracePeekUser(s.pid, dr6Off)
|
||||||
|
if err != nil {
|
||||||
|
return siAddr
|
||||||
|
}
|
||||||
|
for slot := range 4 {
|
||||||
|
if dr6&(1<<slot) != 0 {
|
||||||
|
addr, err := ptracePeekUser(s.pid, drOffset+uintptr(slot*8))
|
||||||
|
if err == nil && addr != 0 {
|
||||||
|
return addr
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return siAddr
|
||||||
|
}
|
||||||
|
|
||||||
|
// maxWatchpoints reports the number of hardware watchpoint slots the
|
||||||
|
// architecture provides: four address registers, DR0-DR3.
|
||||||
|
func maxWatchpoints() int { return 4 }
|
||||||
|
|
||||||
// WatchpointType selects what triggers the watchpoint.
|
// WatchpointType selects what triggers the watchpoint.
|
||||||
type WatchpointType int
|
type WatchpointType int
|
||||||
|
|
||||||
@@ -62,23 +100,11 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
|
|||||||
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
|
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
|
||||||
}
|
}
|
||||||
|
|
||||||
var drAddr uintptr
|
if err := ptracePokeUser(s.pid, drOffset+uintptr(slot*8), addr); err != nil {
|
||||||
switch slot {
|
|
||||||
case 0:
|
|
||||||
drAddr = 0x0
|
|
||||||
case 1:
|
|
||||||
drAddr = 0x8
|
|
||||||
case 2:
|
|
||||||
drAddr = 0x10
|
|
||||||
case 3:
|
|
||||||
drAddr = 0x18
|
|
||||||
}
|
|
||||||
|
|
||||||
if err := ptracePokeUser(s.pid, drAddr, addr); err != nil {
|
|
||||||
return fmt.Errorf("debug: set DR%d: %w", slot, err)
|
return fmt.Errorf("debug: set DR%d: %w", slot, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
dr7, err := ptracePeekUser(s.pid, 0x38)
|
dr7, err := ptracePeekUser(s.pid, dr7Off)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("debug: read DR7: %w", err)
|
return fmt.Errorf("debug: read DR7: %w", err)
|
||||||
}
|
}
|
||||||
@@ -90,7 +116,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
|
|||||||
mask := ^((uint64(1) << (2 * slot)) | (uint64(3) << (16 + 4*slot)) | (uint64(3) << (18 + 4*slot)))
|
mask := ^((uint64(1) << (2 * slot)) | (uint64(3) << (16 + 4*slot)) | (uint64(3) << (18 + 4*slot)))
|
||||||
dr7 = (dr7 & mask) | enableBit | rwBits | lenField
|
dr7 = (dr7 & mask) | enableBit | rwBits | lenField
|
||||||
|
|
||||||
if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
|
if err := ptracePokeUser(s.pid, dr7Off, dr7); err != nil {
|
||||||
return fmt.Errorf("debug: set DR7: %w", err)
|
return fmt.Errorf("debug: set DR7: %w", err)
|
||||||
}
|
}
|
||||||
s.wpSlots[slot] = true
|
s.wpSlots[slot] = true
|
||||||
@@ -105,12 +131,12 @@ func (s *Session) ClearWatchpoint(slot int) error {
|
|||||||
if !s.wpSlots[slot] {
|
if !s.wpSlots[slot] {
|
||||||
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
|
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
|
||||||
}
|
}
|
||||||
dr7, err := ptracePeekUser(s.pid, 0x38)
|
dr7, err := ptracePeekUser(s.pid, dr7Off)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
dr7 &^= uint64(1) << (2 * slot)
|
dr7 &^= uint64(1) << (2 * slot)
|
||||||
if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
|
if err := ptracePokeUser(s.pid, dr7Off, dr7); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
s.wpSlots[slot] = false
|
s.wpSlots[slot] = false
|
||||||
@@ -119,7 +145,7 @@ func (s *Session) ClearWatchpoint(slot int) error {
|
|||||||
|
|
||||||
// ClearAllWatchpoints removes all hardware watchpoints.
|
// ClearAllWatchpoints removes all hardware watchpoints.
|
||||||
func (s *Session) ClearAllWatchpoints() error {
|
func (s *Session) ClearAllWatchpoints() error {
|
||||||
for slot := range 4 {
|
for slot := range maxWatchpoints() {
|
||||||
if s.wpSlots[slot] {
|
if s.wpSlots[slot] {
|
||||||
if err := s.ClearWatchpoint(slot); err != nil {
|
if err := s.ClearWatchpoint(slot); err != nil {
|
||||||
return err
|
return err
|
||||||
@@ -146,16 +172,21 @@ func ptracePokeUser(pid int, offset uintptr, val uint64) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func ptracePeekUser(pid int, offset uintptr) (uint64, error) {
|
func ptracePeekUser(pid int, offset uintptr) (uint64, error) {
|
||||||
|
// x86 PEEKUSR writes the word to the user-space pointer in data
|
||||||
|
// (arch/x86/kernel/ptrace.c uses put_user); passing 0 there fails with
|
||||||
|
// EFAULT, so the word is read through a real address.
|
||||||
const ptracePeekuser = 3
|
const ptracePeekuser = 3
|
||||||
val, _, errno := syscall.Syscall6(
|
var word uint64
|
||||||
|
_, _, errno := syscall.Syscall6(
|
||||||
syscall.SYS_PTRACE,
|
syscall.SYS_PTRACE,
|
||||||
uintptr(ptracePeekuser),
|
uintptr(ptracePeekuser),
|
||||||
uintptr(pid),
|
uintptr(pid),
|
||||||
offset,
|
offset,
|
||||||
0, 0, 0,
|
uintptr(unsafe.Pointer(&word)),
|
||||||
|
0, 0,
|
||||||
)
|
)
|
||||||
if errno != 0 {
|
if errno != 0 {
|
||||||
return 0, errno
|
return 0, errno
|
||||||
}
|
}
|
||||||
return uint64(val), nil
|
return word, nil
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -12,7 +12,7 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
// Hardware watchpoint support via arm64 debug registers (DBGWVR/DBGWCR).
|
// Hardware watchpoint support via arm64 debug registers (DBGWVR/DBGWCR).
|
||||||
// Accessed via PTRACE_SETREGSET with NT_ARM_HW_BREAK.
|
// Accessed via PTRACE_GETREGSET/SETREGSET with NT_ARM_HW_WATCH.
|
||||||
|
|
||||||
// WatchpointType selects what triggers the watchpoint.
|
// WatchpointType selects what triggers the watchpoint.
|
||||||
type WatchpointType int
|
type WatchpointType int
|
||||||
@@ -22,26 +22,35 @@ const (
|
|||||||
WatchRead WatchpointType = 3
|
WatchRead WatchpointType = 3
|
||||||
)
|
)
|
||||||
|
|
||||||
const maxWatchpoints = 16
|
// maxWatchpoints reports the number of hardware watchpoint slots the
|
||||||
|
// architecture provides: DBGWVR0-DBGWCR15.
|
||||||
|
func maxWatchpoints() int { return 16 }
|
||||||
|
|
||||||
// hwBreakState mirrors the kernel's struct user_hwdebug_state.
|
// hwWatchState mirrors the kernel's struct user_hwdebug_state.
|
||||||
type hwBreakState struct {
|
type hwWatchState struct {
|
||||||
DbgInfo uint32
|
DbgInfo uint32
|
||||||
_pad [4]byte
|
_pad [4]byte
|
||||||
DbgRegs [16]hwBreakReg
|
DbgRegs [16]hwWatchReg
|
||||||
}
|
}
|
||||||
|
|
||||||
type hwBreakReg struct {
|
type hwWatchReg struct {
|
||||||
Addr uint64
|
Addr uint64
|
||||||
Ctrl uint64
|
Ctrl uint64
|
||||||
}
|
}
|
||||||
|
|
||||||
const (
|
// ntArmHWWatch is NT_ARM_HW_WATCH (0x403), the watchpoint regset
|
||||||
ntArmHWBreak = 0x403 // NT_ARM_HW_BREAK
|
// (include/uapi/linux/elf.h; 0x402 is NT_ARM_HW_BREAK). Watchpoints and
|
||||||
)
|
// breakpoints live in different regsets with the same struct shape, so the
|
||||||
|
// constant is named for what it arms to keep a future edit from arming
|
||||||
|
// breakpoints instead.
|
||||||
|
const ntArmHWWatch = 0x403
|
||||||
|
|
||||||
|
// archWatchpointAddr resolves the address of the watchpoint that fired:
|
||||||
|
// the arm64 kernel already reports the watched data address as si_addr.
|
||||||
|
func archWatchpointAddr(s *Session, siAddr uint64) uint64 { return siAddr }
|
||||||
|
|
||||||
func (s *Session) FindFreeWatchpointSlot() int {
|
func (s *Session) FindFreeWatchpointSlot() int {
|
||||||
for i := range maxWatchpoints {
|
for i := range maxWatchpoints() {
|
||||||
if !s.wpSlots[i] {
|
if !s.wpSlots[i] {
|
||||||
return i
|
return i
|
||||||
}
|
}
|
||||||
@@ -50,7 +59,7 @@ func (s *Session) FindFreeWatchpointSlot() int {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
|
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
|
||||||
if slot < 0 || slot >= maxWatchpoints {
|
if slot < 0 || slot >= maxWatchpoints() {
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
return s.wpSlots[slot]
|
return s.wpSlots[slot]
|
||||||
@@ -58,27 +67,31 @@ func (s *Session) IsWatchpointSlotUsed(slot int) bool {
|
|||||||
|
|
||||||
// SetWatchpoint installs a hardware watchpoint on the given address.
|
// SetWatchpoint installs a hardware watchpoint on the given address.
|
||||||
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
|
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
|
||||||
if slot < 0 || slot >= maxWatchpoints {
|
if slot < 0 || slot >= maxWatchpoints() {
|
||||||
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
|
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints()-1)
|
||||||
}
|
}
|
||||||
if s.wpSlots[slot] {
|
if s.wpSlots[slot] {
|
||||||
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
|
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
|
||||||
}
|
}
|
||||||
|
|
||||||
state, err := s.getHWBreakState()
|
state, err := s.getHWWatchState()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("debug: read watchpoint state: %w", err)
|
return fmt.Errorf("debug: read watchpoint state: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
if uint32(slot) >= state.DbgInfo {
|
// MDSCR_EL1 packs (debug_arch << 8) | num_slots into dbg_info, so only
|
||||||
return fmt.Errorf("debug: slot %d exceeds available watchpoints (%d)", slot, state.DbgInfo)
|
// the low byte counts slots.
|
||||||
|
if uint32(slot) >= state.DbgInfo&0xff {
|
||||||
|
return fmt.Errorf("debug: slot %d exceeds available watchpoints (%d)", slot, state.DbgInfo&0xff)
|
||||||
}
|
}
|
||||||
|
|
||||||
state.DbgRegs[slot].Addr = addr
|
state.DbgRegs[slot].Addr = addr
|
||||||
|
// DBGWCR bits 3-4 select the access type: 01 load, 10 store, 11 either
|
||||||
|
// (ARM DDI 0487, DBGWCR<n>_EL1 watchpoint type field).
|
||||||
ctrl := uint64(1) // enable
|
ctrl := uint64(1) // enable
|
||||||
switch typ {
|
switch typ {
|
||||||
case WatchWrite:
|
case WatchWrite:
|
||||||
ctrl |= 1 << 3 // store only
|
ctrl |= 2 << 3 // store only
|
||||||
case WatchRead:
|
case WatchRead:
|
||||||
ctrl |= 3 << 3 // load+store
|
ctrl |= 3 << 3 // load+store
|
||||||
}
|
}
|
||||||
@@ -98,7 +111,7 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
|
|||||||
ctrl |= bas << 5
|
ctrl |= bas << 5
|
||||||
state.DbgRegs[slot].Ctrl = ctrl
|
state.DbgRegs[slot].Ctrl = ctrl
|
||||||
|
|
||||||
if err := s.setHWBreakState(state); err != nil {
|
if err := s.setHWWatchState(state); err != nil {
|
||||||
return fmt.Errorf("debug: set watchpoint: %w", err)
|
return fmt.Errorf("debug: set watchpoint: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -107,20 +120,20 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (s *Session) ClearWatchpoint(slot int) error {
|
func (s *Session) ClearWatchpoint(slot int) error {
|
||||||
if slot < 0 || slot >= maxWatchpoints {
|
if slot < 0 || slot >= maxWatchpoints() {
|
||||||
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
|
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints()-1)
|
||||||
}
|
}
|
||||||
if !s.wpSlots[slot] {
|
if !s.wpSlots[slot] {
|
||||||
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
|
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
|
||||||
}
|
}
|
||||||
|
|
||||||
state, err := s.getHWBreakState()
|
state, err := s.getHWWatchState()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
state.DbgRegs[slot].Addr = 0
|
state.DbgRegs[slot].Addr = 0
|
||||||
state.DbgRegs[slot].Ctrl = 0
|
state.DbgRegs[slot].Ctrl = 0
|
||||||
if err := s.setHWBreakState(state); err != nil {
|
if err := s.setHWWatchState(state); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
s.wpSlots[slot] = false
|
s.wpSlots[slot] = false
|
||||||
@@ -128,7 +141,7 @@ func (s *Session) ClearWatchpoint(slot int) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (s *Session) ClearAllWatchpoints() error {
|
func (s *Session) ClearAllWatchpoints() error {
|
||||||
for slot := 0; slot < maxWatchpoints; slot++ {
|
for slot := range maxWatchpoints() {
|
||||||
if s.wpSlots[slot] {
|
if s.wpSlots[slot] {
|
||||||
if err := s.ClearWatchpoint(slot); err != nil {
|
if err := s.ClearWatchpoint(slot); err != nil {
|
||||||
return err
|
return err
|
||||||
@@ -138,8 +151,8 @@ func (s *Session) ClearAllWatchpoints() error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (s *Session) getHWBreakState() (*hwBreakState, error) {
|
func (s *Session) getHWWatchState() (*hwWatchState, error) {
|
||||||
var state hwBreakState
|
var state hwWatchState
|
||||||
iovec := syscall.Iovec{
|
iovec := syscall.Iovec{
|
||||||
Base: (*byte)(unsafe.Pointer(&state)),
|
Base: (*byte)(unsafe.Pointer(&state)),
|
||||||
Len: uint64(unsafe.Sizeof(state)),
|
Len: uint64(unsafe.Sizeof(state)),
|
||||||
@@ -148,7 +161,7 @@ func (s *Session) getHWBreakState() (*hwBreakState, error) {
|
|||||||
syscall.SYS_PTRACE,
|
syscall.SYS_PTRACE,
|
||||||
uintptr(syscall.PTRACE_GETREGSET),
|
uintptr(syscall.PTRACE_GETREGSET),
|
||||||
uintptr(s.pid),
|
uintptr(s.pid),
|
||||||
uintptr(ntArmHWBreak),
|
uintptr(ntArmHWWatch),
|
||||||
uintptr(unsafe.Pointer(&iovec)),
|
uintptr(unsafe.Pointer(&iovec)),
|
||||||
0, 0,
|
0, 0,
|
||||||
)
|
)
|
||||||
@@ -158,7 +171,7 @@ func (s *Session) getHWBreakState() (*hwBreakState, error) {
|
|||||||
return &state, nil
|
return &state, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (s *Session) setHWBreakState(state *hwBreakState) error {
|
func (s *Session) setHWWatchState(state *hwWatchState) error {
|
||||||
iovec := syscall.Iovec{
|
iovec := syscall.Iovec{
|
||||||
Base: (*byte)(unsafe.Pointer(state)),
|
Base: (*byte)(unsafe.Pointer(state)),
|
||||||
Len: uint64(unsafe.Sizeof(*state)),
|
Len: uint64(unsafe.Sizeof(*state)),
|
||||||
@@ -167,7 +180,7 @@ func (s *Session) setHWBreakState(state *hwBreakState) error {
|
|||||||
syscall.SYS_PTRACE,
|
syscall.SYS_PTRACE,
|
||||||
uintptr(syscall.PTRACE_SETREGSET),
|
uintptr(syscall.PTRACE_SETREGSET),
|
||||||
uintptr(s.pid),
|
uintptr(s.pid),
|
||||||
uintptr(ntArmHWBreak),
|
uintptr(ntArmHWWatch),
|
||||||
uintptr(unsafe.Pointer(&iovec)),
|
uintptr(unsafe.Pointer(&iovec)),
|
||||||
0, 0,
|
0, 0,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -8,10 +8,47 @@ package debug
|
|||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
"syscall"
|
"syscall"
|
||||||
|
"unsafe"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Hardware watchpoint support for LoongArch via debug registers.
|
// Hardware watchpoint support via the NT_LOONGARCH_HW_WATCH regset.
|
||||||
// Uses PTRACE_POKEUSER/PEEKUSER to access HW watchpoint registers.
|
//
|
||||||
|
// The kernel's PTRACE_POKEUSER on loong64 accepts only the user_pt_regs
|
||||||
|
// indices 0-34 (GPRs, orig_a0, era, badv, per
|
||||||
|
// arch/loongarch/include/uapi/asm/ptrace.h), so there is no debug-register
|
||||||
|
// window to poke. The real interface is PTRACE_GETREGSET/SETREGSET on
|
||||||
|
// NT_LOONGARCH_HW_WATCH (0xa06, include/uapi/linux/elf.h) with struct
|
||||||
|
// user_watch_state_v2 (arch/loongarch/include/uapi/asm/ptrace.h): a dbg_info
|
||||||
|
// word followed by 14 slots of {addr u64, mask u64, ctrl u32, pad u32}.
|
||||||
|
// hw_break_get puts the slot count in the low byte of dbg_info
|
||||||
|
// (arch/loongarch/kernel/ptrace.c, ptrace_hbp_get_resource_info) and
|
||||||
|
// hw_break_set ignores dbg_info, reading addr, mask and ctrl per slot.
|
||||||
|
|
||||||
|
const ntLoongHWWatch = 0xa06
|
||||||
|
|
||||||
|
// loongWatchState mirrors the kernel's struct user_watch_state_v2.
|
||||||
|
type loongWatchState struct {
|
||||||
|
DbgInfo uint64
|
||||||
|
DbgRegs [14]loongWatchReg
|
||||||
|
}
|
||||||
|
|
||||||
|
type loongWatchReg struct {
|
||||||
|
Addr uint64
|
||||||
|
Mask uint64
|
||||||
|
Ctrl uint32
|
||||||
|
Pad uint32
|
||||||
|
}
|
||||||
|
|
||||||
|
// Control word bit layout (arch/loongarch/include/asm/hw_breakpoint.h):
|
||||||
|
// bits 1-4 privilege enables (CTRL_PLV3_ENABLE, 0x10, covers user mode),
|
||||||
|
// bits 8-9 access type (LOAD 1<<0, STORE 1<<1), bits 10-11 length
|
||||||
|
// (0=8 bytes, 1=4, 2=2, 3=1, inverted like the hardware FWP cfg).
|
||||||
|
const (
|
||||||
|
loongCtrlPLV3Enable = 0x10
|
||||||
|
loongTypeLoad = 1 << 8
|
||||||
|
loongTypeStore = 2 << 8
|
||||||
|
loongLenShift = 10
|
||||||
|
)
|
||||||
|
|
||||||
// WatchpointType selects what triggers the watchpoint.
|
// WatchpointType selects what triggers the watchpoint.
|
||||||
type WatchpointType int
|
type WatchpointType int
|
||||||
@@ -21,10 +58,16 @@ const (
|
|||||||
WatchRead WatchpointType = 3
|
WatchRead WatchpointType = 3
|
||||||
)
|
)
|
||||||
|
|
||||||
const maxWatchpoints = 4
|
// maxWatchpoints reports the slot capacity of the regset struct; the number
|
||||||
|
// the hardware actually provides is read from dbg_info at arm time.
|
||||||
|
func maxWatchpoints() int { return len(loongWatchState{}.DbgRegs) }
|
||||||
|
|
||||||
|
// archWatchpointAddr resolves the address of the watchpoint that fired:
|
||||||
|
// the loongarch kernel already reports the accessed address as si_addr.
|
||||||
|
func archWatchpointAddr(s *Session, siAddr uint64) uint64 { return siAddr }
|
||||||
|
|
||||||
func (s *Session) FindFreeWatchpointSlot() int {
|
func (s *Session) FindFreeWatchpointSlot() int {
|
||||||
for i := range maxWatchpoints {
|
for i := range maxWatchpoints() {
|
||||||
if !s.wpSlots[i] {
|
if !s.wpSlots[i] {
|
||||||
return i
|
return i
|
||||||
}
|
}
|
||||||
@@ -33,53 +76,56 @@ func (s *Session) FindFreeWatchpointSlot() int {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
|
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
|
||||||
if slot < 0 || slot >= maxWatchpoints {
|
if slot < 0 || slot >= maxWatchpoints() {
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
return s.wpSlots[slot]
|
return s.wpSlots[slot]
|
||||||
}
|
}
|
||||||
|
|
||||||
// SetWatchpoint installs a hardware watchpoint.
|
// SetWatchpoint installs a hardware watchpoint on the given address.
|
||||||
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
|
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
|
||||||
if slot < 0 || slot >= maxWatchpoints {
|
if slot < 0 || slot >= maxWatchpoints() {
|
||||||
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
|
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints()-1)
|
||||||
}
|
}
|
||||||
if s.wpSlots[slot] {
|
if s.wpSlots[slot] {
|
||||||
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
|
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
|
||||||
}
|
}
|
||||||
if size != 1 && size != 2 && size != 4 && size != 8 {
|
|
||||||
|
var ctrlType uint32
|
||||||
|
switch typ {
|
||||||
|
case WatchWrite:
|
||||||
|
ctrlType = loongTypeStore
|
||||||
|
case WatchRead:
|
||||||
|
ctrlType = loongTypeLoad | loongTypeStore
|
||||||
|
}
|
||||||
|
var lenBits uint32
|
||||||
|
switch size {
|
||||||
|
case 1:
|
||||||
|
lenBits = 3
|
||||||
|
case 2:
|
||||||
|
lenBits = 2
|
||||||
|
case 4:
|
||||||
|
lenBits = 1
|
||||||
|
case 8:
|
||||||
|
lenBits = 0
|
||||||
|
default:
|
||||||
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
|
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
|
||||||
}
|
}
|
||||||
|
|
||||||
// LoongArch debug registers: DBGWVR (watchpoint value) and DBGWCR (watchpoint control).
|
state, err := s.getLoongWatchState()
|
||||||
// Accessed via PTRACE_POKEUSER at architecture-specific offsets.
|
if err != nil {
|
||||||
if err := ptracePokeUser(s.pid, uintptr(0x1000+slot*8), addr); err != nil {
|
return fmt.Errorf("debug: read watchpoint state: %w", err)
|
||||||
return fmt.Errorf("debug: set watchpoint address: %w", err)
|
}
|
||||||
|
if uint64(slot) >= state.DbgInfo&0xff {
|
||||||
|
return fmt.Errorf("debug: slot %d exceeds available watchpoints (%d)", slot, state.DbgInfo&0xff)
|
||||||
}
|
}
|
||||||
|
|
||||||
// DBGWCR: enable + type + size.
|
state.DbgRegs[slot].Addr = addr
|
||||||
var wcr uint64 = 1 // enable
|
state.DbgRegs[slot].Mask = 0
|
||||||
switch typ {
|
state.DbgRegs[slot].Ctrl = loongCtrlPLV3Enable | ctrlType | lenBits<<loongLenShift
|
||||||
case WatchWrite:
|
|
||||||
wcr |= 1 << 3 // store
|
|
||||||
case WatchRead:
|
|
||||||
wcr |= 3 << 3 // load+store
|
|
||||||
}
|
|
||||||
var sizeBits uint64
|
|
||||||
switch size {
|
|
||||||
case 1:
|
|
||||||
sizeBits = 0
|
|
||||||
case 2:
|
|
||||||
sizeBits = 1
|
|
||||||
case 4:
|
|
||||||
sizeBits = 2
|
|
||||||
case 8:
|
|
||||||
sizeBits = 3
|
|
||||||
}
|
|
||||||
wcr |= sizeBits << 5
|
|
||||||
|
|
||||||
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), wcr); err != nil {
|
if err := s.setLoongWatchState(state); err != nil {
|
||||||
return fmt.Errorf("debug: set watchpoint control: %w", err)
|
return fmt.Errorf("debug: set watchpoint: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
s.wpSlots[slot] = true
|
s.wpSlots[slot] = true
|
||||||
@@ -87,14 +133,21 @@ func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (s *Session) ClearWatchpoint(slot int) error {
|
func (s *Session) ClearWatchpoint(slot int) error {
|
||||||
if slot < 0 || slot >= maxWatchpoints {
|
if slot < 0 || slot >= maxWatchpoints() {
|
||||||
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
|
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints()-1)
|
||||||
}
|
}
|
||||||
if !s.wpSlots[slot] {
|
if !s.wpSlots[slot] {
|
||||||
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
|
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
|
||||||
}
|
}
|
||||||
|
|
||||||
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), 0); err != nil {
|
state, err := s.getLoongWatchState()
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
state.DbgRegs[slot].Addr = 0
|
||||||
|
state.DbgRegs[slot].Mask = 0
|
||||||
|
state.DbgRegs[slot].Ctrl = 0
|
||||||
|
if err := s.setLoongWatchState(state); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
s.wpSlots[slot] = false
|
s.wpSlots[slot] = false
|
||||||
@@ -102,7 +155,7 @@ func (s *Session) ClearWatchpoint(slot int) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (s *Session) ClearAllWatchpoints() error {
|
func (s *Session) ClearAllWatchpoints() error {
|
||||||
for slot := 0; slot < maxWatchpoints; slot++ {
|
for slot := range maxWatchpoints() {
|
||||||
if s.wpSlots[slot] {
|
if s.wpSlots[slot] {
|
||||||
if err := s.ClearWatchpoint(slot); err != nil {
|
if err := s.ClearWatchpoint(slot); err != nil {
|
||||||
return err
|
return err
|
||||||
@@ -112,14 +165,37 @@ func (s *Session) ClearAllWatchpoints() error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func ptracePokeUser(pid int, offset uintptr, val uint64) error {
|
func (s *Session) getLoongWatchState() (*loongWatchState, error) {
|
||||||
const ptracePokeuser = 6
|
var state loongWatchState
|
||||||
|
iovec := syscall.Iovec{
|
||||||
|
Base: (*byte)(unsafe.Pointer(&state)),
|
||||||
|
Len: uint64(unsafe.Sizeof(state)),
|
||||||
|
}
|
||||||
_, _, errno := syscall.Syscall6(
|
_, _, errno := syscall.Syscall6(
|
||||||
syscall.SYS_PTRACE,
|
syscall.SYS_PTRACE,
|
||||||
uintptr(ptracePokeuser),
|
uintptr(syscall.PTRACE_GETREGSET),
|
||||||
uintptr(pid),
|
uintptr(s.pid),
|
||||||
offset,
|
uintptr(ntLoongHWWatch),
|
||||||
uintptr(val),
|
uintptr(unsafe.Pointer(&iovec)),
|
||||||
|
0, 0,
|
||||||
|
)
|
||||||
|
if errno != 0 {
|
||||||
|
return nil, errno
|
||||||
|
}
|
||||||
|
return &state, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *Session) setLoongWatchState(state *loongWatchState) error {
|
||||||
|
iovec := syscall.Iovec{
|
||||||
|
Base: (*byte)(unsafe.Pointer(state)),
|
||||||
|
Len: uint64(unsafe.Sizeof(*state)),
|
||||||
|
}
|
||||||
|
_, _, errno := syscall.Syscall6(
|
||||||
|
syscall.SYS_PTRACE,
|
||||||
|
uintptr(syscall.PTRACE_SETREGSET),
|
||||||
|
uintptr(s.pid),
|
||||||
|
uintptr(ntLoongHWWatch),
|
||||||
|
uintptr(unsafe.Pointer(&iovec)),
|
||||||
0, 0,
|
0, 0,
|
||||||
)
|
)
|
||||||
if errno != 0 {
|
if errno != 0 {
|
||||||
@@ -127,18 +203,3 @@ func ptracePokeUser(pid int, offset uintptr, val uint64) error {
|
|||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func ptracePeekUser(pid int, offset uintptr) (uint64, error) {
|
|
||||||
const ptracePeekuser = 3
|
|
||||||
val, _, errno := syscall.Syscall6(
|
|
||||||
syscall.SYS_PTRACE,
|
|
||||||
uintptr(ptracePeekuser),
|
|
||||||
uintptr(pid),
|
|
||||||
offset,
|
|
||||||
0, 0, 0,
|
|
||||||
)
|
|
||||||
if errno != 0 {
|
|
||||||
return 0, errno
|
|
||||||
}
|
|
||||||
return uint64(val), nil
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -7,11 +7,16 @@ package debug
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
"syscall"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// Hardware watchpoint support for RISC-V via Sdtrig trigger registers.
|
// Hardware watchpoints are not reachable through the riscv64 kernel ptrace
|
||||||
// Uses PTRACE_POKEUSER/PEEKUSER to access debug registers.
|
// interface. arch/riscv/kernel/ptrace.c forwards every POKEUSER/PEEKUSER to
|
||||||
|
// the generic ptrace_request, and the riscv user_regset view contains only
|
||||||
|
// the GPR, FP and vector regsets: there is no debug-register or trigger
|
||||||
|
// regset, and offsets outside the view fail with EIO. The Sdtrig CSRs
|
||||||
|
// (tselect/tdata1/tdata2) are not exposed to ptrace either. Until the
|
||||||
|
// kernel grows a trigger regset, SetWatchpoint reports the fact instead of
|
||||||
|
// poking a window that does not exist.
|
||||||
|
|
||||||
// WatchpointType selects what triggers the watchpoint.
|
// WatchpointType selects what triggers the watchpoint.
|
||||||
type WatchpointType int
|
type WatchpointType int
|
||||||
@@ -21,10 +26,18 @@ const (
|
|||||||
WatchRead WatchpointType = 3
|
WatchRead WatchpointType = 3
|
||||||
)
|
)
|
||||||
|
|
||||||
const maxWatchpoints = 4
|
// maxWatchpoints reports the number of hardware watchpoint slots the
|
||||||
|
// architecture provides. riscv64 exposes none via ptrace; the bound exists
|
||||||
|
// so the slot bookkeeping stays consistent.
|
||||||
|
func maxWatchpoints() int { return 4 }
|
||||||
|
|
||||||
|
// archWatchpointAddr resolves the address of the watchpoint that fired.
|
||||||
|
// Unreachable in practice (watchpoints cannot be armed), but si_addr names
|
||||||
|
// the accessed address where the kernel does report one.
|
||||||
|
func archWatchpointAddr(s *Session, siAddr uint64) uint64 { return siAddr }
|
||||||
|
|
||||||
func (s *Session) FindFreeWatchpointSlot() int {
|
func (s *Session) FindFreeWatchpointSlot() int {
|
||||||
for i := range maxWatchpoints {
|
for i := range maxWatchpoints() {
|
||||||
if !s.wpSlots[i] {
|
if !s.wpSlots[i] {
|
||||||
return i
|
return i
|
||||||
}
|
}
|
||||||
@@ -33,115 +46,26 @@ func (s *Session) FindFreeWatchpointSlot() int {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
|
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
|
||||||
if slot < 0 || slot >= maxWatchpoints {
|
if slot < 0 || slot >= maxWatchpoints() {
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
return s.wpSlots[slot]
|
return s.wpSlots[slot]
|
||||||
}
|
}
|
||||||
|
|
||||||
// SetWatchpoint installs a hardware watchpoint.
|
// SetWatchpoint always fails: the riscv64 kernel ptrace interface has no
|
||||||
|
// hardware-watchpoint access.
|
||||||
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
|
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
|
||||||
if slot < 0 || slot >= maxWatchpoints {
|
return fmt.Errorf("debug: hardware watchpoints are not supported by the riscv64 kernel ptrace interface")
|
||||||
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
|
|
||||||
}
|
|
||||||
if s.wpSlots[slot] {
|
|
||||||
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
|
|
||||||
}
|
|
||||||
if size != 1 && size != 2 && size != 4 && size != 8 {
|
|
||||||
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
|
|
||||||
}
|
|
||||||
|
|
||||||
// RISC-V trigger registers: tdata1 encodes type/control, tdata2 holds address.
|
|
||||||
// The exact encoding depends on the trigger implementation (Sdtrig).
|
|
||||||
// Use PTRACE_POKEUSER to write to the trigger CSRs via the kernel's
|
|
||||||
// debug register interface.
|
|
||||||
if err := ptracePokeUser(s.pid, uintptr(0x1000+slot*8), addr); err != nil {
|
|
||||||
return fmt.Errorf("debug: set watchpoint address: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// tdata1: set match control. Mode=2 (data match), select=0, action=1 (debug exception).
|
|
||||||
var tdata1 uint64 = 2 << 60 // type = match (2)
|
|
||||||
tdata1 |= 1 << 0 // action = enter debug mode
|
|
||||||
tdata1 |= 1 << 7 // store (write) trigger
|
|
||||||
if typ == WatchRead {
|
|
||||||
tdata1 |= 1 << 6 // load trigger
|
|
||||||
}
|
|
||||||
// Size encoding: 0=1byte, 1=2byte, 2=4byte, 3=8byte.
|
|
||||||
var sizeBits uint64
|
|
||||||
switch size {
|
|
||||||
case 1:
|
|
||||||
sizeBits = 0
|
|
||||||
case 2:
|
|
||||||
sizeBits = 1
|
|
||||||
case 4:
|
|
||||||
sizeBits = 2
|
|
||||||
case 8:
|
|
||||||
sizeBits = 3
|
|
||||||
}
|
|
||||||
tdata1 |= sizeBits << 16 // size field
|
|
||||||
|
|
||||||
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), tdata1); err != nil {
|
|
||||||
return fmt.Errorf("debug: set watchpoint control: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
s.wpSlots[slot] = true
|
|
||||||
return nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ClearWatchpoint always fails: no watchpoint can ever be armed.
|
||||||
func (s *Session) ClearWatchpoint(slot int) error {
|
func (s *Session) ClearWatchpoint(slot int) error {
|
||||||
if slot < 0 || slot >= maxWatchpoints {
|
if slot < 0 || slot >= maxWatchpoints() {
|
||||||
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints-1)
|
return fmt.Errorf("debug: watchpoint slot must be 0-%d", maxWatchpoints()-1)
|
||||||
}
|
}
|
||||||
if !s.wpSlots[slot] {
|
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
|
||||||
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Disable by clearing tdata1.
|
|
||||||
if err := ptracePokeUser(s.pid, uintptr(0x1001+slot*8), 0); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
s.wpSlots[slot] = false
|
|
||||||
return nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func (s *Session) ClearAllWatchpoints() error {
|
func (s *Session) ClearAllWatchpoints() error {
|
||||||
for slot := 0; slot < maxWatchpoints; slot++ {
|
|
||||||
if s.wpSlots[slot] {
|
|
||||||
if err := s.ClearWatchpoint(slot); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func ptracePokeUser(pid int, offset uintptr, val uint64) error {
|
|
||||||
const ptracePokeuser = 6
|
|
||||||
_, _, errno := syscall.Syscall6(
|
|
||||||
syscall.SYS_PTRACE,
|
|
||||||
uintptr(ptracePokeuser),
|
|
||||||
uintptr(pid),
|
|
||||||
offset,
|
|
||||||
uintptr(val),
|
|
||||||
0, 0,
|
|
||||||
)
|
|
||||||
if errno != 0 {
|
|
||||||
return errno
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func ptracePeekUser(pid int, offset uintptr) (uint64, error) {
|
|
||||||
const ptracePeekuser = 3
|
|
||||||
val, _, errno := syscall.Syscall6(
|
|
||||||
syscall.SYS_PTRACE,
|
|
||||||
uintptr(ptracePeekuser),
|
|
||||||
uintptr(pid),
|
|
||||||
offset,
|
|
||||||
0, 0, 0,
|
|
||||||
)
|
|
||||||
if errno != 0 {
|
|
||||||
return 0, errno
|
|
||||||
}
|
|
||||||
return uint64(val), nil
|
|
||||||
}
|
|
||||||
|
|||||||
+167
-52
@@ -4,26 +4,30 @@ How gasm-devkit is put together and why.
|
|||||||
|
|
||||||
Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
|
Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
|
||||||
|
|
||||||
## Design goals
|
## Overview
|
||||||
|
|
||||||
|
Three design goals shape everything below.
|
||||||
|
|
||||||
1. **A real AST, not a grammar hack.** The linter, analyser, assembler and
|
1. **A real AST, not a grammar hack.** The linter, analyser, assembler and
|
||||||
language server all need to *reason* about assembly, not just colour it.
|
language server all need to *reason* about assembly, not just colour it.
|
||||||
So the centre of the toolkit is a hand-written lexer and a parser that
|
So the centre of the toolkit is a hand-written lexer and a parser that
|
||||||
produce a typed AST with source positions on every node.
|
produce a typed AST with source positions on every node.
|
||||||
2. **Architecture as data, not code.** Per-architecture differences (amd64,
|
2. **Architecture as data, not code.** Per-architecture differences (amd64,
|
||||||
arm64, riscv64, loong64) live in register and instruction *tables* (`arch`),
|
arm64, riscv64, loong64) live in register and instruction *tables* (`arch`)
|
||||||
never in `if arch == …` branches scattered through the logic. The
|
and per-architecture encoders, rather than in `if arch == …` branches
|
||||||
instruction tables are generated from the Go toolchain's own assembler
|
threaded through the analysis; the arch tests that remain are dispatch and
|
||||||
source (`just gen`), so adding or refreshing an architecture is a data
|
policy points, such as which encoder a file name selects and which
|
||||||
operation, not a coding one.
|
registers the liveness pass audits. The instruction tables are generated
|
||||||
|
from the Go toolchain's own assembler source (`just gen`), so refreshing an
|
||||||
|
architecture is a data operation, not a coding one.
|
||||||
3. **Open integration surface.** Everything the toolkit can do is reachable
|
3. **Open integration surface.** Everything the toolkit can do is reachable
|
||||||
through two vendor-neutral interfaces: a CLI and an LSP server. No editor
|
through two vendor-neutral interfaces: a CLI and an LSP server. No editor
|
||||||
owns the toolkit; the toolkit is offered to editors on standard terms.
|
owns the toolkit; the toolkit is offered to editors on standard terms.
|
||||||
|
|
||||||
## Pipeline
|
The components, and how data moves between them:
|
||||||
|
|
||||||
```mermaid
|
```mermaid
|
||||||
graph TD
|
flowchart TD
|
||||||
SRC["source .s"] --> LEX["lexer<br/>token stream"]
|
SRC["source .s"] --> LEX["lexer<br/>token stream"]
|
||||||
LEX --> PAR["parser<br/>AST + diagnostics"]
|
LEX --> PAR["parser<br/>AST + diagnostics"]
|
||||||
LEX --> FMT["format<br/>re-space tokens"]
|
LEX --> FMT["format<br/>re-space tokens"]
|
||||||
@@ -33,18 +37,55 @@ graph TD
|
|||||||
ARCH["arch tables<br/>amd64 / arm64 / riscv64 / loong64"] --> LINT
|
ARCH["arch tables<br/>amd64 / arm64 / riscv64 / loong64"] --> LINT
|
||||||
ARCH --> LSP
|
ARCH --> LSP
|
||||||
LINT --> LSP
|
LINT --> LSP
|
||||||
|
PAR --> ASM["asm<br/>encoders, image, object emitters"]
|
||||||
|
ASM --> VER["verify<br/>JIT mapping, ABI checks, fuzzing"]
|
||||||
|
ASM --> DBG["debug<br/>ptrace session"]
|
||||||
|
VER --> DBG
|
||||||
|
DIS["disasm<br/>golang.org/x/arch"] --> DBG
|
||||||
FMT --> CLI["gasm CLI"]
|
FMT --> CLI["gasm CLI"]
|
||||||
LINT --> CLI
|
LINT --> CLI
|
||||||
PAR --> CLI
|
PAR --> CLI
|
||||||
LEX --> CLI
|
LEX --> CLI
|
||||||
|
ASM --> CLI
|
||||||
|
VER --> CLI
|
||||||
|
DBG --> CLI
|
||||||
|
DIS --> CLI
|
||||||
LSP --> EDITOR["any LSP editor"]
|
LSP --> EDITOR["any LSP editor"]
|
||||||
```
|
```
|
||||||
|
|
||||||
The lexer is the shared foundation: the parser builds the AST from it, the
|
The lexer is the shared foundation: the parser builds the AST from it, the
|
||||||
formatter re-spaces its tokens directly, and the language server uses it for
|
formatter re-spaces its tokens directly, and the language server uses it for
|
||||||
semantic highlighting.
|
semantic highlighting. The packages follow a dependency chain: static analysis
|
||||||
|
builds only on the AST, the standalone assembler emits object code, and both
|
||||||
|
the dynamic analysis and the debugger consume the execution substrate the
|
||||||
|
assembler provides.
|
||||||
|
|
||||||
## Components
|
## Packages
|
||||||
|
|
||||||
|
| Package | Responsibility |
|
||||||
|
|---|---|
|
||||||
|
| `token` | token kinds and positions |
|
||||||
|
| `lexer` | hand-written scanner; permissive, and it never panics |
|
||||||
|
| `ast` | the typed syntax tree: declarations, lines, operands |
|
||||||
|
| `parser` | line-oriented parser producing the AST and its diagnostics |
|
||||||
|
| `arch` | register and instruction tables for the four architectures |
|
||||||
|
| `lint` | static checks over the AST |
|
||||||
|
| `format` | canonical formatter over the token stream |
|
||||||
|
| `lsp` | the language server |
|
||||||
|
| `asm` | standalone assembler: encoders, image layout, object emitters |
|
||||||
|
| `disasm` | disassembly backend over golang.org/x/arch |
|
||||||
|
| `verify` | JIT execution, ABI checks, differential fuzzing |
|
||||||
|
| `debug` | interactive ptrace debugger |
|
||||||
|
| `cmd/gasm` | the CLI |
|
||||||
|
| `_gen` | rebuilds the `arch` tables from the Go toolchain source |
|
||||||
|
|
||||||
|
The boundaries matter as much as the responsibilities: `ast` records syntax
|
||||||
|
only, and whether a name is a register or a label is left to `arch`, so the
|
||||||
|
parser stays architecture-agnostic. `asm` produces the machine code, `verify`
|
||||||
|
and `debug` are the two packages that map it executable (read-execute in
|
||||||
|
`verify`, read-write-execute in the debuggee), and `cmd/gasm` is the CLI, with
|
||||||
|
the verify sweep orchestration and the audit, scaffold and unified-diff
|
||||||
|
helpers beside its flags and output.
|
||||||
|
|
||||||
### `token` and `lexer`
|
### `token` and `lexer`
|
||||||
|
|
||||||
@@ -84,14 +125,17 @@ Register files are generated programmatically (the regular `R8`-`R15`,
|
|||||||
`X0`-`X15`, `Y0`-`Y15`, `Z0`-`Z31`, `K0`-`K7` ranges) plus the irregularly
|
`X0`-`X15`, `Y0`-`Y15`, `Z0`-`Z31`, `K0`-`K7` ranges) plus the irregularly
|
||||||
named registers listed explicitly. Instruction names are **generated from the
|
named registers listed explicitly. Instruction names are **generated from the
|
||||||
Go toolchain's own assembler source** (`cmd/internal/obj/<arch>/anames.go`,
|
Go toolchain's own assembler source** (`cmd/internal/obj/<arch>/anames.go`,
|
||||||
plus the common opcodes and the per-architecture front-end aliases such as the
|
plus the common opcodes in `cmd/internal/obj/util.go`) by `just gen`, so the
|
||||||
arm64 `B`/`BL` branches and the `.P`/`.W` load-store addressing suffixes) by
|
tables always match what the real assembler accepts. The spellings the
|
||||||
`just gen`, so the tables always match what the real assembler accepts. Each
|
toolchain's tables do not carry are hand-maintained instead: the front-end
|
||||||
mnemonic maps to a summary and an optional operand-count range; counts are
|
alias lists in `arch/arm64.go`, `arch/amd64.go` and `arch/loong64.go` (the
|
||||||
recorded only where unambiguous (`-1` disables the operand-count lint for that
|
arm64 `B`/`BL` branches among them), and the arm64 `.P`/`.W` load-store suffix
|
||||||
instruction) so the linter stays silent rather than guess. For architectures
|
stripping in `arch/arch.go`. Each mnemonic maps to a summary and an optional
|
||||||
with highly variable operand forms (arm64, riscv64, loong64) only a few
|
operand-count range; counts are recorded only where unambiguous (`-1`
|
||||||
fixed-arity instructions (`RET`, `NOP`, `JMP`, `CALL`) carry counts at all.
|
disables the operand-count lint for that instruction) so the linter stays
|
||||||
|
silent rather than guess. For architectures with highly variable operand
|
||||||
|
forms (arm64, riscv64, loong64) `relaxCounts` clears those counts, leaving
|
||||||
|
`RET` and `NOP` with a range (`RET` alone on riscv64).
|
||||||
|
|
||||||
### `lint`
|
### `lint`
|
||||||
|
|
||||||
@@ -134,7 +178,8 @@ Two deeper analyses sit on top of the AST:
|
|||||||
- **`unreachable-code`.** Code after a `RET` and before the next label is
|
- **`unreachable-code`.** Code after a `RET` and before the next label is
|
||||||
dead. The check is suppressed for any function whose reachability cannot be
|
dead. The check is suppressed for any function whose reachability cannot be
|
||||||
decided statically: those using PC-relative jumps (`JMP 2(PC)`),
|
decided statically: those using PC-relative jumps (`JMP 2(PC)`),
|
||||||
register-indirect branches (`JALR`/`JR`/`JIRL`/`BR`/`BLR`), or living in a
|
register-indirect branches (`JALR`/`JR`/`JIRL`/`BR`/`BLR`, or a `JMP`/`CALL`
|
||||||
|
through a register or memory operand), or living in a
|
||||||
file with `#ifdef` conditionals. `UNDEF` is deliberately not a terminator:
|
file with `#ifdef` conditionals. `UNDEF` is deliberately not a terminator:
|
||||||
code after it is occasionally intentional metadata.
|
code after it is occasionally intentional metadata.
|
||||||
- **`register-clobber` (register liveness).** The linter builds the function's
|
- **`register-clobber` (register liveness).** The linter builds the function's
|
||||||
@@ -202,20 +247,20 @@ them from the standard LSP legend, so no editor-specific grammar is needed.
|
|||||||
|
|
||||||
### `asm`
|
### `asm`
|
||||||
|
|
||||||
The standalone assembler (Phase 2). Its core is an amd64 instruction encoder:
|
The standalone assembler. Its core is an amd64 instruction encoder:
|
||||||
a REX/ModR-M/SIB/displacement/immediate engine plus the scalar instruction set,
|
a REX/ModR-M/SIB/displacement/immediate engine plus the scalar instruction set,
|
||||||
with the Plan 9 operand order (source first) mapped onto the x86 encoding.
|
with the Plan 9 operand order (source first) mapped onto the x86 encoding.
|
||||||
Every encoding is validated by decoding it again with `golang.org/x/arch`, the
|
Every encoding is validated by decoding it again with `golang.org/x/arch`, the
|
||||||
one module dependency, used in tests only and never linked into the binary.
|
one module dependency, which also backs the `gasm dis` listings.
|
||||||
|
|
||||||
A **RISC-V encoder** (Phase 5, RV64IMAFDC + RVC compression) encodes the full
|
A **RISC-V encoder** (RV64IMAFDC + RVC compression) encodes the full
|
||||||
integer, atomic, float/double, FMA and CSR instruction sets with the MOV
|
integer, atomic, float/double, FMA and CSR instruction sets with the MOV
|
||||||
pseudo-instruction and SB/global symbol references (AUIPC pairs with
|
pseudo-instruction and SB/global symbol references (AUIPC pairs with
|
||||||
R_RISCV_PCREL_HI20/LO12 relocations). The encoder compresses eligible
|
R_RISCV_PCREL_HI20/LO12 relocations). The encoder compresses eligible
|
||||||
instructions to 16-bit RVC forms and is validated byte-for-byte against
|
instructions to 16-bit RVC forms and is validated byte-for-byte against
|
||||||
`GOARCH=riscv64 go tool asm`.
|
`GOARCH=riscv64 go tool asm`.
|
||||||
|
|
||||||
A **LoongArch encoder** (Phase 5, LoongArch64) encodes the integer and
|
A **LoongArch encoder** (LoongArch64) encodes the integer and
|
||||||
floating-point instruction sets with the dual-form arithmetic mnemonics (3R
|
floating-point instruction sets with the dual-form arithmetic mnemonics (3R
|
||||||
vs 2RI12), the 16/21-bit branch families, the MOV pseudo-instruction and its
|
vs 2RI12), the 16/21-bit branch families, the MOV pseudo-instruction and its
|
||||||
constant materialisation (the dcon classification driving lu12i.w/ori/lu32i.d/
|
constant materialisation (the dcon classification driving lu12i.w/ori/lu32i.d/
|
||||||
@@ -226,7 +271,7 @@ relocations). Like the RISC-V encoder it is validated byte-for-byte against
|
|||||||
`GOARCH=loong64 go tool asm`, and its GOOBJ output is proven end-to-end by
|
`GOARCH=loong64 go tool asm`, and its GOOBJ output is proven end-to-end by
|
||||||
substituting it into a cross-compiled `go build` and linking with `cmd/link`.
|
substituting it into a cross-compiled `go build` and linking with `cmd/link`.
|
||||||
|
|
||||||
An **AArch64 encoder** (Phase 5, arm64) encodes the integer instruction set
|
An **AArch64 encoder** (arm64) encodes the integer instruction set
|
||||||
with the data-processing (shifted register and immediate forms), load/store
|
with the data-processing (shifted register and immediate forms), load/store
|
||||||
(scaled unsigned immediate and unscaled9-bit immediate), conditional and
|
(scaled unsigned immediate and unscaled9-bit immediate), conditional and
|
||||||
unconditional branches, the MOV pseudo-instruction and its constant
|
unconditional branches, the MOV pseudo-instruction and its constant
|
||||||
@@ -262,11 +307,11 @@ registers are translated onto the hardware stack pointer: `x+N(FP)` becomes
|
|||||||
pointer is set up, with the matching Go prologue/epilogue generated, so the
|
pointer is set up, with the matching Go prologue/epilogue generated, so the
|
||||||
output is byte-identical to the Go assembler for these cases. SIMD is handled
|
output is byte-identical to the Go assembler for these cases. SIMD is handled
|
||||||
by a VEX (AVX/AVX2) encoder (the two- and three-byte VEX prefixes with XMM/YMM
|
by a VEX (AVX/AVX2) encoder (the two- and three-byte VEX prefixes with XMM/YMM
|
||||||
registers) across eight operand forms: the three-operand NDS form, the
|
registers) over nine operand forms plus a dedicated move encoder: the
|
||||||
two-operand reg/rm form, the immediate-shift form (plus the variable-count
|
three-operand NDS form, the two-operand reg/rm form, the immediate-shift form
|
||||||
shifts, which share the NDS shape with the count in an XMM register or
|
(plus the variable-count shifts, which share the NDS shape with the count in
|
||||||
memory), the immediate shuffle form (`VPSHUFD`, `VPERMQ`), the
|
an XMM register or memory), the immediate shuffle form (`VPSHUFD`, `VPERMQ`),
|
||||||
three-operand-plus-immediate form (`VSHUFPD`,
|
the three-operand-plus-immediate form (`VSHUFPD`,
|
||||||
`VPERM2I128`, `VINSERTI128`), the lane-extract form (`VEXTRACTI128`,
|
`VPERM2I128`, `VINSERTI128`), the lane-extract form (`VEXTRACTI128`,
|
||||||
`VEXTRACTF128`, where the YMM source occupies the reg field and the XMM or
|
`VEXTRACTF128`, where the YMM source occupies the reg field and the XMM or
|
||||||
memory destination r/m), the direction-sensitive moves (`VMOVDQU`, `VMOVUPD`,
|
memory destination r/m), the direction-sensitive moves (`VMOVDQU`, `VMOVUPD`,
|
||||||
@@ -311,10 +356,10 @@ b bit and the L'L rounding-control field (broadcast keeps the vector length
|
|||||||
and scales disp8 by the element size), and combine with the .Z zeroing
|
and scales disp8 by the element size), and combine with the .Z zeroing
|
||||||
suffix. Every encoding is validated two ways: by
|
suffix. Every encoding is validated two ways: by
|
||||||
round-trip decoding through `golang.org/x/arch`, and byte-for-byte against
|
round-trip decoding through `golang.org/x/arch`, and byte-for-byte against
|
||||||
the machine code the real Go assembler emits, a comparison that holds for
|
the machine code the real Go assembler emits; the parity suites carry that
|
||||||
whole functions: all 27 functions of both kernels assemble to exactly the Go
|
comparison over whole kernel files on all four architectures, with the
|
||||||
toolchain's bytes, the lone exception being the displacements of the
|
relocation fields masked because the Go linker fills those displacements at
|
||||||
static-constant loads, which the Go linker fills at link time.
|
link time.
|
||||||
|
|
||||||
File-level assembly (`AssembleFile`) goes beyond single functions: it
|
File-level assembly (`AssembleFile`) goes beyond single functions: it
|
||||||
materialises the file's static symbols (`GLOBL`/`DATA`) in a data section
|
materialises the file's static symbols (`GLOBL`/`DATA`) in a data section
|
||||||
@@ -361,7 +406,7 @@ compiled packages it references.
|
|||||||
|
|
||||||
### `verify`
|
### `verify`
|
||||||
|
|
||||||
The dynamic-analysis substrate (Phase 3). It JIT-loads assembled images into
|
The dynamic-analysis substrate. It JIT-loads assembled images into
|
||||||
executable memory and invokes them directly, enabling differential testing,
|
executable memory and invokes them directly, enabling differential testing,
|
||||||
runtime ABI checks and coverage profiling.
|
runtime ABI checks and coverage profiling.
|
||||||
|
|
||||||
@@ -381,10 +426,10 @@ every architecture too: `enterJITChecked` plants sentinels in the registers
|
|||||||
the Go ABI fixes across calls (amd64 `BP`/`R14`, arm64 `R29`/`R28`, riscv64
|
the Go ABI fixes across calls (amd64 `BP`/`R14`, arm64 `R29`/`R28`, riscv64
|
||||||
`X27`, loong64 `R22`; the latter two keep no hardware frame pointer) and the
|
`X27`, loong64 `R22`; the latter two keep no hardware frame pointer) and the
|
||||||
raw return trampoline `leaveJITCheckedRaw` verifies them, restoring the
|
raw return trampoline `leaveJITCheckedRaw` verifies them, restoring the
|
||||||
saved registers before Go code resumes. riscv64 is validated end to
|
saved registers before Go code resumes. All three non-amd64 trampolines
|
||||||
end under qemu-user emulation; arm64 shares the same stack convention and
|
are validated end to end under qemu-user emulation, the loong64 one
|
||||||
fix; loong64 stays ground-truth-only until hardware validation (see
|
through its raw-address leave handoff.
|
||||||
docs/DECISIONS.md). `gasm verify` runs the JIT checks when the host
|
`gasm verify` runs the JIT checks when the host
|
||||||
matches the kernel's architecture and the toolchain comparisons
|
matches the kernel's architecture and the toolchain comparisons
|
||||||
elsewhere.
|
elsewhere.
|
||||||
|
|
||||||
@@ -397,10 +442,13 @@ The `gasm verify` CLI subcommand exposes this: it loads a file, reports the
|
|||||||
available functions and (with `-smoke`) calls each NOSPLIT function with zeroed
|
available functions and (with `-smoke`) calls each NOSPLIT function with zeroed
|
||||||
arguments to confirm the trampoline round-trips. The `-smoke` and `-abi`
|
arguments to confirm the trampoline round-trips. The `-smoke` and `-abi`
|
||||||
sweeps run in parallel and each inside a child process, so a function that
|
sweeps run in parallel and each inside a child process, so a function that
|
||||||
faults is reported without ending the sweep. `gasm verify --fuzz` combines
|
faults is reported without ending the sweep; `-abi` is where the ABI check
|
||||||
ABI checks (sentinel registers, canary, stack bounds) with differential fuzz
|
lives, fuzzing each function with sentinel values in the registers the Go ABI
|
||||||
testing, comparing the JIT-assembled kernel against the portable Go reference
|
fixes across calls and a canary below `SP`, and reporting a violation on any
|
||||||
bit-for-bit while verifying the ABI contract on every iteration. When a fuzz
|
iteration. `gasm verify --fuzz` is the differential campaign instead: it
|
||||||
|
JIT-loads the kernel and the `go tool asm` build of the same kernel and
|
||||||
|
compares the output argument areas bit-for-bit, one child process per function
|
||||||
|
so a crash on a partial function is reported rather than fatal. When a fuzz
|
||||||
iteration crashes or mismatches, `FuzzResult.CrashInput` stores the exact input
|
iteration crashes or mismatches, `FuzzResult.CrashInput` stores the exact input
|
||||||
for reproducibility. `gasm verify --call <func> --buf name:size:pattern`
|
for reproducibility. `gasm verify --call <func> --buf name:size:pattern`
|
||||||
invokes a single function with user-supplied buffers (patterns: zero, ones,
|
invokes a single function with user-supplied buffers (patterns: zero, ones,
|
||||||
@@ -420,30 +468,97 @@ masked), reporting any encoding drift.
|
|||||||
The interactive debugger (all four architectures). It launches the target
|
The interactive debugger (all four architectures). It launches the target
|
||||||
function in a child process that maps the JIT code, calls
|
function in a child process that maps the JIT code, calls
|
||||||
`PTRACE_TRACEME`, and stops; the parent attaches via ptrace and controls
|
`PTRACE_TRACEME`, and stops; the parent attaches via ptrace and controls
|
||||||
execution. Breakpoints are patched as INT3 bytes through `/proc/pid/mem`
|
execution. Breakpoints are patched through `/proc/pid/mem`: the one-byte
|
||||||
(PTRACE_PEEKTEXT is unreliable with Go's multi-threaded runtime).
|
`INT3` on amd64, the four-byte break instruction on the other three (arm64
|
||||||
|
`BRK #0`, riscv64 `ebreak`, loong64 `break 0`).
|
||||||
The child pins its goroutine to the OS thread with `runtime.LockOSThread`
|
The child pins its goroutine to the OS thread with `runtime.LockOSThread`
|
||||||
so the traced thread is the one executing JIT code. The REPL provides
|
so the traced thread is the one executing JIT code. The REPL provides
|
||||||
single-step, register inspection (GPR + YMM/XMM via `PTRACE_GETFPREGS`),
|
single-step, register inspection (the GPRs on every architecture; on amd64 the
|
||||||
|
XMM set through `PTRACE_GETFPREGS` and the YMM set through `PTRACE_GETREGSET`
|
||||||
|
on `NT_X86_XSTATE`; on the other three the FP/SIMD regset through
|
||||||
|
`PTRACE_GETREGSET` on `NT_PRFPREG`),
|
||||||
label resolution, named buffer allocation with pattern filling
|
label resolution, named buffer allocation with pattern filling
|
||||||
(`--buf name:size:pattern`: zero, ones, seq, or hex), and breakpoint
|
(`--buf name:size:pattern`: zero, ones, seq, or hex), and breakpoint
|
||||||
management. Breakpoints accept conditions
|
management. Breakpoints accept conditions
|
||||||
(`break <label> if <reg> <op> <val>`, including register-against-register
|
(`break <label> if <reg> <op> <val>`, including register-against-register
|
||||||
comparisons), and hardware watchpoints work on all four architectures.
|
comparisons), and hardware watchpoints work on amd64 (the DR0-DR3 debug
|
||||||
|
registers), arm64 (`NT_ARM_HW_WATCH`) and loong64 (`NT_LOONGARCH_HW_WATCH`);
|
||||||
|
riscv64 reports that its kernel ptrace interface exposes no trigger regset.
|
||||||
|
The ptrace path is validated at run time on amd64, where the session tests are
|
||||||
|
built; arm64, riscv64 and loong64 compile and are covered by the
|
||||||
|
architecture-neutral units (label and line tables, the breakpoint manager).
|
||||||
For non-interactive use, `--script` runs REPL commands from a file (or
|
For non-interactive use, `--script` runs REPL commands from a file (or
|
||||||
stdin) and exits, `--timeout` kills the debuggee when a run hangs (the
|
stdin) and exits, `--timeout` kills the debuggee when a run hangs (the
|
||||||
watchdog is armed before the ptrace attach, so a sandboxed debuggee cannot
|
watchdog is armed before the ptrace attach, so a sandboxed debuggee cannot
|
||||||
block it), and `--cover` runs to completion with a breakpoint on every
|
block it), and `--cover` runs to completion with a breakpoint on every
|
||||||
label and reports which blocks executed.
|
instruction and reports which instructions executed and how often.
|
||||||
|
|
||||||
## Extension points
|
### Extending the toolkit
|
||||||
|
|
||||||
- **New architecture:** add an entry to the generator in `_gen`, run
|
- **New architecture:** add an entry to the generator in `_gen`, run
|
||||||
`just gen`, and add a `buildXXX()` register file plus a case in `ForArch`.
|
`just gen`, and add a `buildXXX()` register file plus a case in `ForArch`.
|
||||||
- **New lint rule:** add a function in `lint` and a rule-code constant.
|
- **New lint rule:** add a function in `lint` and a rule-code constant.
|
||||||
- **New LSP feature:** add a method case in `dispatch` and a handler.
|
- **New LSP feature:** add a method case in `dispatch` and a handler.
|
||||||
|
|
||||||
The phases follow a dependency chain. Phase 1 (static analysis) builds only on
|
## Data flow
|
||||||
the AST; Phase 2 (the standalone assembler) emits object code; Phases 3
|
|
||||||
(dynamic analysis) and 4 (the debugger) both consume the execution substrate
|
The main operation, assembling one file:
|
||||||
that the assembler provides.
|
|
||||||
|
```mermaid
|
||||||
|
sequenceDiagram
|
||||||
|
participant User
|
||||||
|
participant CLI as gasm CLI
|
||||||
|
participant Parser as parser
|
||||||
|
participant Asm as asm
|
||||||
|
participant Go as go toolchain
|
||||||
|
User->>CLI: gasm asm --format goobj -p pkg -o k.o k_amd64.s
|
||||||
|
CLI->>Parser: Parse(path, src)
|
||||||
|
Parser-->>CLI: AST, diagnostics
|
||||||
|
CLI->>Asm: AssembleFile(AST)
|
||||||
|
Asm->>Asm: encode operands, settle label offsets, lay out data
|
||||||
|
Asm-->>CLI: Image, code and data and relocations
|
||||||
|
CLI->>Asm: GOObject(pkg, path)
|
||||||
|
Asm->>Go: go list -json -export, externals only
|
||||||
|
Go-->>Asm: package and symbol indices
|
||||||
|
Asm-->>CLI: Go object bytes
|
||||||
|
CLI-->>User: wrote N bytes to k.o
|
||||||
|
```
|
||||||
|
|
||||||
|
Errors are produced where the parse or the encoding fails and become values at
|
||||||
|
the CLI boundary: the parser returns a diagnostic list and never aborts a file,
|
||||||
|
`AssembleFile` returns an error, and `cmd/gasm` prints what it has to stderr
|
||||||
|
and returns a non-zero exit code. The formatter and the linter take different
|
||||||
|
inputs from the assembler: `gasm fmt` re-spaces the token stream
|
||||||
|
(`format.Source` lexes the source text itself) and `gasm lint` walks the parsed
|
||||||
|
AST, so neither depends on an encoding.
|
||||||
|
|
||||||
|
## State and lifetime
|
||||||
|
|
||||||
|
- The analysis packages (`lexer`, `parser`, `format`, `lint`, `arch`) hold only
|
||||||
|
read-only lookup tables and no mutable state: every call allocates its own
|
||||||
|
tokens and AST, and any number of goroutines may read the `arch` tables.
|
||||||
|
- A `verify.Kernel` owns one executable mapping, which `Close` releases. The
|
||||||
|
JIT trampolines keep the Go stack pointer and the checked-call sentinels in
|
||||||
|
package globals, so a call is a process-wide, one-at-a-time operation. The
|
||||||
|
`gasm verify` sweeps therefore run each function in a child process, which
|
||||||
|
contains a crash and keeps the globals unshared.
|
||||||
|
- `lsp.Server` is long-lived: it runs a single read and dispatch loop over the
|
||||||
|
stream and touches its document store only from that loop, so one server
|
||||||
|
serves one connection.
|
||||||
|
- A `debug.Session` owns a traced child process and pins its goroutine to the
|
||||||
|
forking OS thread, because ptrace requests must stay on that thread.
|
||||||
|
|
||||||
|
## Dependencies
|
||||||
|
|
||||||
|
- **`golang.org/x/arch`** (v0.30.0) is the one module dependency: it is the
|
||||||
|
disassembler backend (`gasm dis` and the debugger's listings). The tests
|
||||||
|
additionally decode through it to validate the encodings.
|
||||||
|
- **The Go toolchain**, as an oracle and never as a library: `go tool asm`
|
||||||
|
supplies the object preamble and the ground truth for `gasm verify
|
||||||
|
--ground-truth`, `go list -json -export` locates the archives of the packages
|
||||||
|
a GOOBJ object references, and `_gen` parses
|
||||||
|
`$GOROOT/src/cmd/internal/obj/<arch>/anames.go` to rebuild the tables.
|
||||||
|
- **Linux process interfaces** for the dynamic work: `mmap` and `mprotect` for
|
||||||
|
the JIT mapping, ptrace with `/proc/pid/mem` for the debugger. That is why
|
||||||
|
`verify` runs a JIT check only when the host architecture matches the
|
||||||
|
kernel's, and why `debug` is Linux-only.
|
||||||
|
|||||||
+434
-162
@@ -1,215 +1,487 @@
|
|||||||
# CLI Reference
|
# Command line
|
||||||
|
|
||||||
Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
|
The reference below is taken from the program's own `--help`. If the two disagree, the
|
||||||
|
program is right and this file is a defect.
|
||||||
|
|
||||||
`gasm` is a single binary with subcommands. Run `gasm --help` for an
|
The same reference is installed as man pages: `just install-man` puts gasm(1) and a page
|
||||||
overview, or `gasm <command> -h` for a command's usage and flags.
|
for every command except `version` (which gasm(1) itself documents) into
|
||||||
|
~/.local/share/man (`MANDIR` overrides). A test in `cmd/gasm` keeps the two from
|
||||||
|
drifting: it compares each page's flag set and SYNOPSIS line with the binary's own `-h`
|
||||||
|
output, and gasm(1)'s COMMANDS list with the top-level help. The prose is not compared.
|
||||||
|
|
||||||
## Global Flags
|
## Synopsis
|
||||||
|
|
||||||
| Flag | Description |
|
```sh
|
||||||
|------|-------------|
|
gasm [global flags] <command> [command flags] [arguments]
|
||||||
| `-h`, `--help` | Show help |
|
```
|
||||||
| `-V`, `--version` | Print the version |
|
|
||||||
|
|
||||||
## `gasm tokens <file>`
|
## Commands
|
||||||
|
|
||||||
Print the lexical token stream of FILE: position, token kind, and text,
|
| Command | Purpose |
|
||||||
one token per line. FILE may be `-` to read standard input.
|
|---|---|
|
||||||
|
| `tokens` | print the lexical token stream |
|
||||||
|
| `parse` | parse a file and report syntax errors |
|
||||||
|
| `fmt` | canonicalise the formatting of `.s` files |
|
||||||
|
| `lint` | run the static checks |
|
||||||
|
| `asm` | assemble `.s` files to machine code |
|
||||||
|
| `dis` | disassemble machine code or an assembled file |
|
||||||
|
| `verify` | JIT-assemble and run the dynamic checks |
|
||||||
|
| `debug` | interactive source-level debugger |
|
||||||
|
| `diff` | compare the machine code of two `.s` files |
|
||||||
|
| `profile` | show the basic-block structure of the functions |
|
||||||
|
| `audit-instructions` | diff the encoder against the toolchain's name table |
|
||||||
|
| `scaffold` | generate a differential test skeleton for a kernel |
|
||||||
|
| `lsp` | run the language server over stdio |
|
||||||
|
| `version` | print the version |
|
||||||
|
|
||||||
## `gasm parse <file>`
|
## tokens
|
||||||
|
|
||||||
Parse FILE and report syntax errors on stderr. On success, prints how
|
```text
|
||||||
many declarations and TEXT functions the file contains.
|
Usage: gasm tokens <file>
|
||||||
|
```
|
||||||
|
|
||||||
## `gasm fmt [-w|-l|-d] [path...]`
|
Print the lexical token stream of FILE: position, token kind and text, one
|
||||||
|
token per line. FILE may be `-` to read standard input.
|
||||||
|
|
||||||
Canonicalise the formatting of Plan 9 assembly sources: indentation,
|
```sh
|
||||||
operand spacing, per-function mnemonic alignment, and blank-line layout.
|
gasm tokens hello_amd64.s
|
||||||
|
```
|
||||||
|
|
||||||
| Flag | Description |
|
```text
|
||||||
|------|-------------|
|
1:1 # "#"
|
||||||
| `-w` | Write result to the source file (default: print to stdout) |
|
1:2 IDENT "include"
|
||||||
| `-l` | List files whose formatting differs, one per line; write nothing |
|
1:10 STRING "\"textflag.h\""
|
||||||
| `-d` | Print a unified diff of the canonical formatting instead |
|
```
|
||||||
|
|
||||||
With no arguments, or with a directory argument, every `.s` file below
|
## parse
|
||||||
it is reformatted in place and the names of changed files are listed
|
|
||||||
(`go fmt` style). `.` and `_` directories are skipped.
|
|
||||||
|
|
||||||
## `gasm lint <file...>`
|
```text
|
||||||
|
Usage: gasm parse <file>
|
||||||
|
```
|
||||||
|
|
||||||
Run static checks and print diagnostics as
|
Parse FILE and report syntax errors on stderr. On success, print how many
|
||||||
`file:line:col: severity: message [code]`. Exit status is non-zero when
|
declarations and TEXT functions the file contains. FILE may be `-` to read
|
||||||
an error-severity diagnostic is found.
|
standard input.
|
||||||
|
|
||||||
| Flag | Description |
|
```sh
|
||||||
|------|-------------|
|
gasm parse hello_amd64.s
|
||||||
| `-disable` | Comma-separated rule codes to disable |
|
```
|
||||||
|
|
||||||
|
```text
|
||||||
|
hello_amd64.s: OK, 2 declarations, 1 functions
|
||||||
|
```
|
||||||
|
|
||||||
|
## fmt
|
||||||
|
|
||||||
|
```text
|
||||||
|
Usage: gasm fmt [-w|-l|-d] [path...]
|
||||||
|
```
|
||||||
|
|
||||||
|
| Flag | Default | Effect |
|
||||||
|
|---|---|---|
|
||||||
|
| `-w` | off | write the result back to the source file |
|
||||||
|
| `-l` | off | list the files whose formatting differs; write nothing |
|
||||||
|
| `-d` | off | print a unified diff of the canonical formatting instead |
|
||||||
|
|
||||||
|
`-l` and `-d` are mutually exclusive. With no arguments, or with a directory
|
||||||
|
argument, every `.s` file below it is reformatted in place and the names of the
|
||||||
|
changed files are listed, the way `go fmt` does; `.` and `_` directories are
|
||||||
|
skipped. Explicit file arguments print to stdout unless `-w` is given.
|
||||||
|
|
||||||
|
```sh
|
||||||
|
gasm fmt -l kernel_amd64.s
|
||||||
|
```
|
||||||
|
|
||||||
|
Empty output means every file is formatted, which is the shape a CI check
|
||||||
|
wants; `-d` shows what would change:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
gasm fmt -d ugly_amd64.s
|
||||||
|
```
|
||||||
|
|
||||||
|
```text
|
||||||
|
--- ugly_amd64.s
|
||||||
|
+++ ugly_amd64.s
|
||||||
|
@@ -2,8 +2,8 @@
|
||||||
|
|
||||||
|
// func add(a, b int) int
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
- MOVQ a+0(FP), AX
|
||||||
|
- ADDQ b+8(FP), AX
|
||||||
|
+ MOVQ a+0(FP), AX
|
||||||
|
+ ADDQ b+8(FP), AX
|
||||||
|
```
|
||||||
|
|
||||||
|
## lint
|
||||||
|
|
||||||
|
```text
|
||||||
|
Usage: gasm lint <file...>
|
||||||
|
```
|
||||||
|
|
||||||
|
| Flag | Default | Effect |
|
||||||
|
|---|---|---|
|
||||||
|
| `-disable` | empty | comma-separated rule codes to disable |
|
||||||
|
|
||||||
|
Diagnostics are printed as `file:line:col: severity: message [code]`. The exit
|
||||||
|
status is non-zero when an error-severity diagnostic is found; warnings (the
|
||||||
|
register-clobber audit, for example) do not affect it.
|
||||||
|
|
||||||
Rules: `unknown-instruction`, `operand-count`, `undefined-label`,
|
Rules: `unknown-instruction`, `operand-count`, `undefined-label`,
|
||||||
`duplicate-label`, `missing-ret`, `missing-textflag-include`,
|
`duplicate-label`, `missing-ret`, `missing-textflag-include`,
|
||||||
`abi-argsize`, `unreachable-code`, `register-clobber`,
|
`abi-argsize`, `unreachable-code`, `register-clobber`,
|
||||||
`funcdata-pcdata`, `unused-label`, `invalid-textflag`,
|
`funcdata-pcdata`, `unused-label`, `invalid-textflag`,
|
||||||
`stack-imbalance`, `register-width-mismatch`, `abi0-register-args`,
|
`stack-imbalance`, `register-width-mismatch`, `abi0-register-args`,
|
||||||
`nonportable-register-name` and `unencodable-instruction`.
|
`nonportable-register-name`, `unencodable-instruction` and
|
||||||
|
`reserved-register-write`.
|
||||||
|
|
||||||
## `gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>`
|
```sh
|
||||||
|
gasm lint kernel_amd64.s
|
||||||
|
```
|
||||||
|
|
||||||
Assemble FILE to machine code (amd64, arm64, riscv64, loong64).
|
## asm
|
||||||
|
|
||||||
| Flag | Description |
|
```text
|
||||||
|------|-------------|
|
Usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>
|
||||||
| `--format` | Output format: `raw` (default), `elf`, `goobj` |
|
```
|
||||||
| `-p` | Package path (required for `--format goobj`) |
|
|
||||||
| `-o` | Write output to file (default: hex dump to stdout) |
|
|
||||||
|
|
||||||
## `gasm dis [-a arch] <file>`
|
| Flag | Default | Effect |
|
||||||
|
|---|---|---|
|
||||||
|
| `-format` | `raw` | output format: `raw` (concatenated image), `elf` or `goobj` (Go object) |
|
||||||
|
| `-p` | empty | package path for `--format goobj`, qualifying the exported symbols |
|
||||||
|
| `-GOARCH` | empty | target architecture: `amd64`, `arm64`, `riscv64` or `loong64`; overrides the file-name suffix |
|
||||||
|
| `-o` | empty | write the output to this file instead of a hex dump on stdout |
|
||||||
|
|
||||||
Disassemble machine code to instruction text (via `golang.org/x/arch`).
|
Supported architectures: amd64 (VEX/AVX2 and EVEX/AVX-512 included), arm64,
|
||||||
|
riscv64 (RV64IMAFDC and RVC) and loong64, taken from the file's `_arch.s`
|
||||||
|
suffix or from `-GOARCH`, which is how files whose names carry no
|
||||||
|
recognisable suffix (most of GOROOT's, for example `cpu_x86.s`) are
|
||||||
|
assembled. `raw` concatenates the functions and the data section into one
|
||||||
|
self-consistent image; `elf` emits a relocatable object that links with the
|
||||||
|
system toolchain; `goobj` emits the Go toolchain's own object format, which
|
||||||
|
`cmd/link` consumes directly, and is the one format that needs the toolchain
|
||||||
|
installed: the object preamble is captured from `go tool asm` and the format
|
||||||
|
version from `go version`. `raw` and `elf` need no toolchain at all.
|
||||||
|
|
||||||
With a `.s` file, the file is assembled first and the listing follows the
|
```sh
|
||||||
real layout: one block per `TEXT` function, local labels printed at their
|
gasm asm hello_amd64.s
|
||||||
offsets. The architecture comes from the file name suffix, or from `-a`.
|
```
|
||||||
With any other file, or `-` for standard input, the bytes are
|
|
||||||
disassembled linearly and `-a` selects the architecture (amd64, arm64,
|
|
||||||
riscv64 or loong64).
|
|
||||||
|
|
||||||
| Flag | Description |
|
```text
|
||||||
|------|-------------|
|
add: 16 bytes
|
||||||
| `-a` | Architecture for raw input without a `_arch.s` name |
|
0000: 48 8b 44 24 08 48 03 44 24 10 48 89 44 24 18 c3
|
||||||
|
```
|
||||||
|
|
||||||
## `gasm verify [flags] <file.s>`
|
## dis
|
||||||
|
|
||||||
Assemble FILE, map it into executable memory, and run dynamic checks.
|
```text
|
||||||
|
Usage: gasm dis [-a arch] <file>
|
||||||
|
```
|
||||||
|
|
||||||
| Flag | Description |
|
| Flag | Default | Effect |
|
||||||
|------|-------------|
|
|---|---|---|
|
||||||
| `--ground-truth` | Compare machine code byte-for-byte against `go tool asm` |
|
| `-a` | empty | architecture for raw input without a `_arch.s` name |
|
||||||
| `--fuzz` | Differential fuzz: JIT both gasm and go-tool-asm, compare outputs |
|
|
||||||
| `-n` | Fuzz iterations per function (default: 1000) |
|
|
||||||
| `--abi` | Run ABI-checking calls (sentinel registers + red zone) |
|
|
||||||
| `--abi-n` | Number of ABI check iterations with varied inputs (default: 100) |
|
|
||||||
| `--profile` | List basic-block structure per function |
|
|
||||||
| `--smoke` | Call each NOSPLIT function with zeroed args |
|
|
||||||
| `--call <func>` | Invoke a single function with `--buf` instead of the sweeps |
|
|
||||||
| `--buf <spec>` | Buffer spec for `--call`: `name:size:pattern[,name:size:pattern]` |
|
|
||||||
| `--args <spec>` | Scalar args for `--call`: `name=value[,name=value]` (decimal or `0x` hex) |
|
|
||||||
| `--repeat <n>` | Number of times to repeat a `--call` invocation (default: 1) |
|
|
||||||
| `--save-corpus <dir>` | With `--fuzz`: write each failing input to DIR as replayable JSON |
|
|
||||||
| `--replay <dir>` | Re-run saved corpus entries (JSON in DIR), one child process per entry |
|
|
||||||
|
|
||||||
The `--fuzz` mode runs each function in a subprocess; a partial function
|
With a `.s` file the file is assembled first and the listing follows the real
|
||||||
(e.g. a decoder that faults on malformed input) is reported as
|
layout: one block per `TEXT` function, local labels printed at their offsets.
|
||||||
`CRASH` without killing the parent. Use `--call` with `--buf` to invoke
|
With any other file, or `-` for standard input, the bytes are disassembled
|
||||||
partial functions with valid data instead.
|
linearly and `-a` selects the architecture (amd64, arm64, riscv64 or loong64).
|
||||||
|
|
||||||
The `--call` mode parses the `// func` signature, allocates the requested
|
```sh
|
||||||
buffers (`zero`, `ones`, `seq`, or a hex blob), builds the ABI0 argument
|
gasm dis hello_amd64.s
|
||||||
block with buffer pointers/lengths/capacities at the matching parameter
|
```
|
||||||
offsets, and prints the arg block before and after the call, showing
|
|
||||||
return values and any output written to the buffers. Scalar parameters
|
|
||||||
are supplied with `--args` (decimal, or `0x` hex) at their ABI0 offsets.
|
|
||||||
|
|
||||||
The `--save-corpus` mode records the logical arguments (buffer contents and
|
```text
|
||||||
scalars, not raw pointers) of every failing fuzz input as JSON. `--replay`
|
add: 16 bytes
|
||||||
rebuilds a live argument block from each entry and calls it in its own child
|
0000: 48 8b 44 24 08 mov rax, qword ptr [rsp+0x8]
|
||||||
process, reporting `OK`, `CRASH (reproduced)` or `FAIL` per entry and
|
0005: 48 03 44 24 10 add rax, qword ptr [rsp+0x10]
|
||||||
exiting non-zero when any entry fails.
|
000a: 48 89 44 24 18 mov qword ptr [rsp+0x18], rax
|
||||||
|
000f: c3 ret
|
||||||
|
```
|
||||||
|
|
||||||
## `gasm debug [--func <name>] [--buf spec] [--script file] <file.s>`
|
## verify
|
||||||
|
|
||||||
Interactive debugger for JIT-assembled functions (amd64, arm64, riscv64,
|
```text
|
||||||
loong64). Requires a compiled binary on `$PATH` (not `go run`).
|
Usage: gasm verify [-smoke] [-abi] [-fuzz] [-ground-truth] [-profile] [-call] <file.s>
|
||||||
|
```
|
||||||
|
|
||||||
| Flag | Description |
|
| Flag | Default | Effect |
|
||||||
|------|-------------|
|
|---|---|---|
|
||||||
| `--func` | Function to debug (required) |
|
| `--ground-truth` | off | compare the machine code byte-for-byte against `go tool asm` |
|
||||||
| `--buf` | Buffer spec: `name:size:pattern[,name:size:pattern]` |
|
| `--fuzz` | off | differential fuzz against the `go tool asm` build |
|
||||||
| `--args <file>` | File containing the ABI0 argument block |
|
| `-n` | 1000 | fuzz iterations per function |
|
||||||
| `--script <file>` | Run REPL commands from a file (one per line) and exit; `-` reads stdin |
|
| `--abi` | off | ABI-checking calls: sentinel registers and a red-zone canary |
|
||||||
| `--timeout <dur>` | Kill the debuggee after this duration (e.g. `30s`); for headless `--script` runs |
|
| `--abi-n` | 100 | ABI check iterations with varied inputs |
|
||||||
| `--cover` | Run to completion with a breakpoint on every instruction; report which executed, how often, and which labels were reached |
|
| `--profile` | off | list the basic-block structure per function |
|
||||||
|
| `--smoke` | off | call each NOSPLIT function with zeroed arguments |
|
||||||
|
| `--call` | empty | invoke a single NOSPLIT function with `--buf` instead of the sweeps |
|
||||||
|
| `--buf` | empty | buffer spec for `--call`: `name:size:pattern[,name:size:pattern]` |
|
||||||
|
| `--args` | empty | scalar args for `--call`: `name=value[,name=value]` (decimal or `0x` hex) |
|
||||||
|
| `--repeat` | 1 | number of times to repeat a `--call` invocation |
|
||||||
|
| `--save-corpus` | empty | with `--fuzz`: write each failing input to this directory as replayable JSON |
|
||||||
|
| `--replay` | empty | re-run saved corpus entries, one child process per entry |
|
||||||
|
|
||||||
|
The JIT checks run when the host matches the file's architecture; the
|
||||||
|
toolchain comparison works everywhere. `--fuzz`, `--smoke` and `--abi` run each
|
||||||
|
function in its own child process, so a partial function that faults on random
|
||||||
|
input is reported as `CRASH` instead of ending the sweep; `--call` with `--buf`
|
||||||
|
invokes such a function with valid data. The function named by `--call` must be
|
||||||
|
NOSPLIT: a function with a stack frame is refused with a diagnostic and exits 1.
|
||||||
|
|
||||||
|
```sh
|
||||||
|
gasm verify --ground-truth hello_amd64.s
|
||||||
|
```
|
||||||
|
|
||||||
|
```text
|
||||||
|
hello_amd64.s: 1 functions JIT-loaded
|
||||||
|
add: MATCH (16 bytes)
|
||||||
|
ground truth: 1/1 functions byte-identical
|
||||||
|
add: 16 bytes, args=24, frame=0 NOSPLIT
|
||||||
|
```
|
||||||
|
|
||||||
|
```sh
|
||||||
|
gasm verify --call add --args a=2,b=3 hello_amd64.s
|
||||||
|
```
|
||||||
|
|
||||||
|
```text
|
||||||
|
add: 16 bytes, args=24
|
||||||
|
signature: func add(a int, b int) int
|
||||||
|
scalars:
|
||||||
|
a = 2
|
||||||
|
b = 3
|
||||||
|
args before: 02 00 00 00 00 00 00 00 03 00 00 00 00 00 00 00 00 00 00 00 00 00 00 00 (24 bytes)
|
||||||
|
args after: 02 00 00 00 00 00 00 00 03 00 00 00 00 00 00 00 05 00 00 00 00 00 00 00 (24 bytes)
|
||||||
|
call 1: OK
|
||||||
|
```
|
||||||
|
|
||||||
|
## debug
|
||||||
|
|
||||||
|
```text
|
||||||
|
Usage: gasm debug <file.s> --func <name>
|
||||||
|
```
|
||||||
|
|
||||||
|
| Flag | Default | Effect |
|
||||||
|
|---|---|---|
|
||||||
|
| `-func` | empty | the function to debug, required |
|
||||||
|
| `-buf` | empty | buffer spec: `name:size:pattern[,name:size:pattern]` (zero, ones, seq or hex) |
|
||||||
|
| `-args` | empty | file containing the ABI0 argument block |
|
||||||
|
| `-script` | empty | run REPL commands from a file, one per line, and exit; `-` reads stdin |
|
||||||
|
| `-timeout` | 0 | kill the debuggee after this duration, for headless `-script` runs; a timeout exits 3 |
|
||||||
|
| `-cover` | off | run to completion with a breakpoint on every instruction and report which executed |
|
||||||
|
|
||||||
|
The debugger re-executes the binary it is running as (`os.Executable()`) for the
|
||||||
|
traced child, so the child is the same `gasm`, whether it is installed on `$PATH`
|
||||||
|
or run with `go run ./cmd/gasm`; nothing has to be installed first. Requires
|
||||||
|
Linux (ptrace), and all four architectures are supported.
|
||||||
|
|
||||||
REPL commands:
|
REPL commands:
|
||||||
|
|
||||||
| Command | Description |
|
| Command | Effect |
|
||||||
|---------|-------------|
|
|---|---|
|
||||||
| `break <label\|addr> [if <reg> <op> <val>]` | Set a breakpoint, optionally conditional |
|
| `break <label\|addr\|line> [if <reg> <op> <val\|reg\|*addr>]`, `b` | set a breakpoint; the condition compares a register with a constant, another register or the 8-byte word at `*addr` |
|
||||||
| `delete <label\|addr>` | Remove a breakpoint |
|
| `delete <label\|addr>`, `d` | remove a breakpoint |
|
||||||
| `info break` | List all breakpoints |
|
| `info break`, `info breakpoints`, `info b` | list the breakpoints |
|
||||||
| `step [n]`, `s` | Single-step n instructions |
|
| `step [n]`, `s` | single-step n instructions |
|
||||||
| `next`, `n` | Step over CALL |
|
| `next`, `n` | step over a CALL |
|
||||||
| `finish`, `fin` | Run until the function returns |
|
| `finish`, `fin` | run until the function returns |
|
||||||
| `continue`, `c` | Run until breakpoint, watchpoint or exit |
|
| `continue`, `c` | run until a breakpoint, watchpoint or exit |
|
||||||
| `disas [n]`, `u` | Disassemble n instructions at PC |
|
| `disas [n]`, `u` | disassemble n instructions at the PC |
|
||||||
| `regs` | Print general-purpose + vector/FP registers |
|
| `regs` | print the general-purpose and vector/FP registers |
|
||||||
| `where` | Show source line and nearest label at PC |
|
| `where` | show the source line and the nearest label at the PC |
|
||||||
| `stack` | Show stack near RSP (return address + ABI0 args) |
|
| `stack` | show the stack near RSP, the return address and the ABI0 args |
|
||||||
| `bt`, `backtrace` | Backtrace (current frame + return address) |
|
| `bt`, `backtrace` | backtrace: the current frame and the return address |
|
||||||
| `x [addr] [len]` | Hex-dump memory |
|
| `x [addr] [len]` | hex-dump memory |
|
||||||
| `w <addr> <val...>` | Write bytes to memory |
|
| `w <addr> <val...>` | write bytes to memory |
|
||||||
| `set <reg> <value>` | Set a register |
|
| `set <reg> <value>` | set a register |
|
||||||
| `watch <addr> [r\|w] [size]` | Set a hardware watchpoint (write by default) |
|
| `watch <addr> [r\|w] [size]` | set a hardware watchpoint, write by default |
|
||||||
| `unwatch [<slot>]` | Clear one or all watchpoints |
|
| `unwatch [<slot>]` | clear one watchpoint or all of them |
|
||||||
| `labels`, `l` | List function labels and offsets |
|
| `labels`, `l` | list the function's labels and offsets |
|
||||||
| `help`, `h`, `?` | Show command help |
|
| `help`, `h`, `?` | show the command help |
|
||||||
| `quit`, `q` | Kill the debuggee and exit |
|
| `quit`, `q` | kill the debuggee and exit |
|
||||||
|
|
||||||
## `gasm diff [--map old=new,...] <file1.s> <file2.s>`
|
```sh
|
||||||
|
gasm debug --func add --cover hello_amd64.s
|
||||||
|
```
|
||||||
|
|
||||||
Compare the machine code produced by assembling two files. Shows which
|
## diff
|
||||||
functions differ and the first few differing bytes. Useful for verifying
|
|
||||||
that two implementations produce identical code, or for tracking encoding
|
|
||||||
changes between Go assembler versions.
|
|
||||||
|
|
||||||
| Flag | Description |
|
```text
|
||||||
|------|-------------|
|
Usage: gasm diff [-GOARCH arch] <file1.s> <file2.s>
|
||||||
| `--map` | Comma-separated `old=new` pairs to match functions with different names |
|
```
|
||||||
|
|
||||||
Without `--map`, functions are paired by exact name. With `--map`, a
|
| Flag | Default | Effect |
|
||||||
function named `old` in the first file is compared against the function
|
|---|---|---|
|
||||||
named `new` in the second file (e.g. `--map wideCopyAVX2=wideCopyAVX512`
|
| `-GOARCH` | empty | target architecture for both files, overriding the file-name suffixes |
|
||||||
pairs AVX2 and AVX-512 variants regardless of suffix).
|
| `-map` | empty | comma-separated `old=new` pairs to match functions with different names |
|
||||||
|
|
||||||
## `gasm profile <file.s>`
|
Functions are paired by exact name unless `--map` says otherwise, so
|
||||||
|
`--map wideCopyAVX2=wideCopyAVX512` pairs two variants regardless of suffix.
|
||||||
|
The exit status is non-zero when anything differs.
|
||||||
|
|
||||||
Show the basic-block structure of functions in an assembly file. Lists
|
```sh
|
||||||
each function's labels, their offsets, and the block boundaries. This is
|
gasm diff hello_amd64.s hello_amd64.s
|
||||||
the static structure; for runtime execution counts, use `gasm verify
|
```
|
||||||
--fuzz` which exercises the code paths.
|
|
||||||
|
|
||||||
## `gasm audit-instructions [amd64|arm64|riscv64|loong64]`
|
```text
|
||||||
|
add: identical (16 bytes)
|
||||||
|
all functions identical
|
||||||
|
```
|
||||||
|
|
||||||
Compare the gasm encoder for the given architecture (default amd64)
|
## profile
|
||||||
against the installed `go tool asm` and print the diff: superset
|
|
||||||
encodings (gasm-only spellings, shippable via `gasm asm --format goobj`),
|
|
||||||
known-but-unencodable names (the encoder backlog) and go-only names
|
|
||||||
(feature gaps). The Go side is probed black-box with a battery of operand
|
|
||||||
shapes per mnemonic, so the audit tracks whatever toolchain
|
|
||||||
`go env GOROOT` provides. On non-amd64 architectures the backlog is an
|
|
||||||
over-approximation: a name counts as encodable only when a probe shape
|
|
||||||
assembles cleanly, so a name whose real forms the battery misses lands
|
|
||||||
in the backlog.
|
|
||||||
|
|
||||||
## `gasm scaffold differential <file.s>`
|
```text
|
||||||
|
Usage: gasm profile <file.s>
|
||||||
|
```
|
||||||
|
|
||||||
Print a differential test skeleton for every `// func` signature in
|
Show the basic-block structure of each function: its labels, their offsets and
|
||||||
FILE. The generated test seeds random states, drives the kernel and a
|
the block boundaries. This is the static structure; for runtime execution
|
||||||
portable reference (`<name>Portable`), and compares outputs
|
counts use `gasm debug --cover`, and for input coverage `gasm verify --fuzz`.
|
||||||
byte-for-byte. Write the reference bodies, place the file in the
|
|
||||||
kernel's package, and run it in CI.
|
|
||||||
|
|
||||||
## `gasm lsp`
|
```sh
|
||||||
|
gasm profile hello_amd64.s
|
||||||
|
```
|
||||||
|
|
||||||
Run the language server over standard input/output (JSON-RPC 2.0 with
|
```text
|
||||||
Content-Length framing). Point an LSP-capable editor at the binary and
|
add: 16 bytes, args=24, frame=0 NOSPLIT
|
||||||
associate it with `.s` files. The target architecture is inferred from
|
basic blocks: 1
|
||||||
the file-name suffix (`_amd64.s`, `_arm64.s`, `_riscv64.s`,
|
```
|
||||||
`_loong64.s`).
|
|
||||||
|
|
||||||
Provides: completion, hover, document symbols, push and pull
|
## audit-instructions
|
||||||
diagnostics, semantic tokens, go-to-definition, find references, rename,
|
|
||||||
document formatting, inlay hints, code actions, signature help, document
|
```text
|
||||||
highlights, workspace symbol search, #include document links, and
|
Usage: gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64]
|
||||||
folding ranges for function bodies.
|
```
|
||||||
|
|
||||||
|
Compare the gasm encoder for the given architecture (default amd64) against the
|
||||||
|
installed `go tool asm` and print the diff: superset encodings (gasm-only
|
||||||
|
spellings, shippable via `gasm asm --format goobj`) and known-but-unencodable
|
||||||
|
names (the encoder backlog). The Go side is probed black-box one bare mnemonic
|
||||||
|
at a time, classified by the toolchain's diagnostic for an instruction it does
|
||||||
|
not know, so the audit tracks whatever toolchain `go env GOROOT` provides; the
|
||||||
|
gasm side answers from the encoder table on amd64 and from trial assembly over a
|
||||||
|
battery of operand shapes on the other architectures. On non-amd64
|
||||||
|
architectures the backlog is therefore an over-approximation: a name counts as
|
||||||
|
encodable only when a probe shape assembles cleanly, so a name whose real forms
|
||||||
|
the battery misses lands in the backlog. Names the toolchain knows and gasm does
|
||||||
|
not cannot be enumerated by probing at all, because Go's table is visible only
|
||||||
|
through names already in the gasm table; the report closes with a note saying
|
||||||
|
so rather than listing them.
|
||||||
|
|
||||||
|
```sh
|
||||||
|
gasm audit-instructions amd64
|
||||||
|
```
|
||||||
|
|
||||||
|
```text
|
||||||
|
gasm table (amd64, families excluded): 1542 mnemonics
|
||||||
|
gasm encodable: 587 go tool asm recognised: 1542
|
||||||
|
shared: 587
|
||||||
|
...
|
||||||
|
```
|
||||||
|
|
||||||
|
With `--corpus` the audit changes shape: it assembles every `.s` file under
|
||||||
|
DIR (default `GOROOT/src`) with the gasm encoder only, no toolchain probing.
|
||||||
|
A file whose name carries a recognisable `_arch` suffix is attempted for that
|
||||||
|
architecture; a file without one is attempted for all four, exactly as a
|
||||||
|
`GOARCH` build would compile it. The report gives the headline number (files
|
||||||
|
that assemble for every target architecture), the per-architecture pass rates
|
||||||
|
and the most common failure reasons with one representative file each, which
|
||||||
|
drive the encodability backlog by frequency rather than by table order. A run
|
||||||
|
over GOROOT takes under a second.
|
||||||
|
|
||||||
|
```sh
|
||||||
|
gasm audit-instructions --corpus
|
||||||
|
gasm audit-instructions --corpus "$(go env GOROOT)/src/crypto"
|
||||||
|
```
|
||||||
|
|
||||||
|
```text
|
||||||
|
corpus /usr/local/go/src: 627 files (365 generic, attempted for all architectures)
|
||||||
|
assemble for every target architecture: 127 (20.3%)
|
||||||
|
amd64: 82/464 attempted
|
||||||
|
165 unsupported operand form
|
||||||
|
e.g. /usr/local/go/src/cmd/asm/internal/asm/testdata/386enc.s
|
||||||
|
109 instruction not encodable
|
||||||
|
e.g. /usr/local/go/src/cmd/asm/internal/asm/testdata/386.s
|
||||||
|
...
|
||||||
|
```
|
||||||
|
|
||||||
|
## scaffold
|
||||||
|
|
||||||
|
```text
|
||||||
|
Usage: gasm scaffold differential <file.s>
|
||||||
|
```
|
||||||
|
|
||||||
|
Print a differential test skeleton for every `// func` signature in FILE. The
|
||||||
|
generated test seeds random states, drives the kernel and a portable reference
|
||||||
|
(`<name>Portable`), and compares the outputs byte-for-byte. Write the reference
|
||||||
|
bodies, place the file in the kernel's package, and run it in CI.
|
||||||
|
|
||||||
|
```sh
|
||||||
|
gasm scaffold differential kernel_amd64.s > kernel_differential_test.go
|
||||||
|
```
|
||||||
|
|
||||||
|
## lsp
|
||||||
|
|
||||||
|
```text
|
||||||
|
Usage: gasm lsp
|
||||||
|
```
|
||||||
|
|
||||||
|
Run the language server over standard input/output, JSON-RPC 2.0 with
|
||||||
|
`Content-Length` framing. Point an LSP-capable editor at the binary and
|
||||||
|
associate it with `.s` files; the target architecture is inferred from the
|
||||||
|
file-name suffix (`_amd64.s`, `_arm64.s`, `_riscv64.s`, `_loong64.s`).
|
||||||
|
|
||||||
|
Provides: completion, hover, document symbols, push and pull diagnostics,
|
||||||
|
semantic tokens, go-to-definition, find references, rename, document
|
||||||
|
formatting, inlay hints, code actions, signature help, document highlights,
|
||||||
|
workspace symbol search, #include document links, and folding ranges for
|
||||||
|
function bodies. Definition, references and rename work across every open
|
||||||
|
document.
|
||||||
|
|
||||||
|
## version
|
||||||
|
|
||||||
|
```text
|
||||||
|
Usage: gasm version
|
||||||
|
```
|
||||||
|
|
||||||
|
Print the version the toolchain recorded for the build, the same string as
|
||||||
|
`gasm --version`: the tag on a tagged checkout, a pseudo-version naming the
|
||||||
|
commit below one, with `+dirty` appended on a dirty tree and `(devel)` outside
|
||||||
|
version control.
|
||||||
|
|
||||||
|
## Global flags
|
||||||
|
|
||||||
|
| Flag | Default | Effect |
|
||||||
|
|---|---|---|
|
||||||
|
| `-h`, `--help` | off | print the usage |
|
||||||
|
| `-V`, `--version` | off | print the version |
|
||||||
|
|
||||||
|
## Exit codes
|
||||||
|
|
||||||
|
| Code | Meaning |
|
||||||
|
|---|---|
|
||||||
|
| `0` | success |
|
||||||
|
| `1` | a failure the program detected: a parse or assembly error, an error-severity lint diagnostic, a mismatch in `verify`, a file that cannot be read |
|
||||||
|
| `2` | the arguments were wrong: a missing or extra argument, an unknown command or format, an invalid `--map` pair |
|
||||||
|
| `3` | `debug --timeout` killed the debuggee |
|
||||||
|
|
||||||
|
## Examples
|
||||||
|
|
||||||
|
Assemble a kernel, check it, and run it:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
gasm lint kernel_amd64.s
|
||||||
|
gasm fmt -l kernel_amd64.s
|
||||||
|
gasm asm -o kernel.bin kernel_amd64.s
|
||||||
|
gasm verify --ground-truth kernel_amd64.s
|
||||||
|
```
|
||||||
|
|
||||||
|
Link the kernel into a Go program through the toolchain's own object format:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
gasm asm --format goobj -p example.com/kernel -o kernel.o kernel_amd64.s
|
||||||
|
```
|
||||||
|
|
||||||
|
Find which labels a failing kernel reaches, headlessly:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
gasm debug --func decodeBlockAVX2 --cover --timeout 30s kernel_amd64.s
|
||||||
|
```
|
||||||
|
|||||||
@@ -1,111 +0,0 @@
|
|||||||
# Deferred decisions
|
|
||||||
|
|
||||||
Design decisions deliberately postponed, with enough context to pick them up
|
|
||||||
again without re-deriving the analysis. Each entry records what is deferred,
|
|
||||||
why, the options on the table, and the trigger that should reopen it.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## GOOBJ external (cross-package) symbol references
|
|
||||||
|
|
||||||
**Status:** resolved (v0.29.0+, 2026-08-07).
|
|
||||||
|
|
||||||
**Approach taken.** Instead of parsing the compiler's iexport data (which
|
|
||||||
would have required either `golang.org/x/tools` or an in-house parser), the
|
|
||||||
resolver reads the **GOOBJ data directly** from the target package's `.a`
|
|
||||||
archive. The `.a` file contains a `_go_.o` member whose GOOBJ s is the
|
|
||||||
same one gasm writes; the parser reuses the same layout (`blkSymdef`,
|
|
||||||
`blkNonpkgdef`, the string table), so no new dependency was needed.
|
|
||||||
|
|
||||||
**How it works.**
|
|
||||||
|
|
||||||
1. `go list -json -export <pkg>` finds the target package's `.a` file.
|
|
||||||
2. `extractGOOBJ` reads the ar archive, finds the `_go_.o` member, skips
|
|
||||||
the `"go object …\n!\n"` preamble and parses the GOOBJ header.
|
|
||||||
3. `goobjFile.symbols()` walks `blkSymdef` and `blkNonpkgdef` in definition
|
|
||||||
order (the same order the linker uses) to build the symbol-to-index
|
|
||||||
mapping.
|
|
||||||
4. `resolveExternalSymbols` wires the resolved `{PkgIdx, SymIdx}` into the
|
|
||||||
GOOBJ emission.
|
|
||||||
|
|
||||||
The resolver is invoked automatically when `img.Externals` is non-empty; it
|
|
||||||
runs `go list` as a subprocess (consistent with `toolchainObjectPreamble`
|
|
||||||
which already calls `go tool asm`). All symbol data is cached per package
|
|
||||||
for the lifetime of the GOOBJ emission.
|
|
||||||
|
|
||||||
## 2026-08-30 non-amd64 JIT execution trampolines
|
|
||||||
|
|
||||||
**Status:** resolved for riscv64 (validated end to end under qemu-user)
|
|
||||||
and arm64 (fix in place, consistent with the observed frame convention);
|
|
||||||
open for loong64 until hardware validation.
|
|
||||||
|
|
||||||
**Root cause (found 2026-08-31).** The trampolines advanced SP past the
|
|
||||||
leave-address slot after loading it, while the assembled kernels read
|
|
||||||
their first argument at SP+8 per the frame convention (the amd64 path
|
|
||||||
already kept SP on that slot). Removing the advance fixed riscv64
|
|
||||||
immediately (plain and checked ABI tests pass under qemu-user); the
|
|
||||||
arm64 kernel's pre-fix trace showed exactly the same SP+8 reading. The
|
|
||||||
apparent arm64/loong64 "crashes in the JIT" turned out to be dominated
|
|
||||||
by an unrelated instability: the Go 1.26 and 1.27 runtimes crash under
|
|
||||||
qemu-user arm64 emulation (GC worker start, identical signature with the
|
|
||||||
JIT tests skipped, both qemu 7.2 and 10.2), and the Go loong64 runtime
|
|
||||||
does not start at all. `gasm verify` therefore keeps loong64 kernels on
|
|
||||||
the ground-truth path until hardware validation; the GOARCH-guarded
|
|
||||||
tests (`verify/jit_arch_test.go`, `verify/abi_arch_test.go`) are the
|
|
||||||
hardware validation entry point.
|
|
||||||
|
|
||||||
**State.** The per-architecture trampolines compile for all targets, the
|
|
||||||
kernels they execute are byte-for-byte correct against `go tool asm`, and
|
|
||||||
under `qemu-aarch64` the arm64 kernel demonstrably executes and stores its
|
|
||||||
result correctly. The failure is on the return path into Go code: arm64
|
|
||||||
and loong64 take a SIGSEGV after the kernel's RET (the Go-side unwind
|
|
||||||
through `leaveJIT` and its interposed ABIInternal wrapper is the suspect),
|
|
||||||
and riscv64 returns cleanly but with an untouched result area. amd64 is
|
|
||||||
unaffected (the checked trampoline saves and restores BP/R14 and the flow
|
|
||||||
is validated end to end).
|
|
||||||
|
|
||||||
**Evidence harness.** `verify/jit_arch_test.go` (plain call) and
|
|
||||||
`verify/abi_arch_test.go` (checked call) are GOARCH-guarded tests; build
|
|
||||||
the test binary per target (`GOARCH=arm64 go test -c -o v.test ./verify/`)
|
|
||||||
and run it under `qemu-aarch64-static` from the `verify/` directory. A
|
|
||||||
minimal reproducer pattern lives in the qemu exploration notes: verify
|
|
||||||
loads, the kernel executes, the fault follows the return.
|
|
||||||
|
|
||||||
**Fix direction.** Compare the amd64 checked trampoline (GLOBL/DATA raw
|
|
||||||
address, explicit SP/BP/R14 save-restore) against the arm64/riscv64/
|
|
||||||
loong64 `leaveJIT` unwind, in particular the interaction with the
|
|
||||||
ABIInternal wrapper that `reflect.ValueOf(leaveJIT).Pointer()` returns.
|
|
||||||
The plain-call path (no sentinels) fails the same way, so the checked
|
|
||||||
path is not the variable.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 2026-08-29 tooling round
|
|
||||||
|
|
||||||
- `lint abi0-register-args`: flags kernels whose `// func` parameters are
|
|
||||||
never read from the FP frame. Motivated by a real latent bug: kernels
|
|
||||||
reading arguments from registers pass every test while the autogenerated
|
|
||||||
`F.abi0` wrapper happens to leave the caller's register values intact, and
|
|
||||||
break on a toolchain upgrade.
|
|
||||||
- `lint nonportable-register-name`: the RAX/EAX register spellings are a gasm
|
|
||||||
extension; go tool asm rejects them, so files using them only link through
|
|
||||||
the gasm goobj path.
|
|
||||||
- `lint unencodable-instruction`: a mnemonic in the architecture table that
|
|
||||||
`asm.Encodable` rejects is flagged at edit time instead of failing at
|
|
||||||
assembly time.
|
|
||||||
- `audit-instructions`: black-box diff of the encoder against go tool asm.
|
|
||||||
As of this round the tables fully overlap on names; the audit exists to
|
|
||||||
catch drift in both directions (future supersets and future gaps).
|
|
||||||
- `scaffold differential`: generates the direct-call differential skeleton
|
|
||||||
(two independent seed sets, output and in-place buffer comparison) that a
|
|
||||||
pipeline-level fuzz can never replace.
|
|
||||||
- `verify --args`: scalar arguments for `-call`, closing the repro gap where
|
|
||||||
only buffers could be supplied.
|
|
||||||
- `debug --script/--timeout/--cover`: headless debugging with a watchdog
|
|
||||||
armed before the ptrace attach (untracing sandboxes hang the attach), and
|
|
||||||
label-level block coverage for the "did my test ever enter that branch"
|
|
||||||
question.
|
|
||||||
- Superset policy remains: gasm may accept spellings and encodings go tool
|
|
||||||
asm lacks, but such kernels ship only via `gasm asm --format goobj`; the
|
|
||||||
audit reports the superset surface. The register-alias superset is warned
|
|
||||||
about by lint because the default `go build` path cannot consume it.
|
|
||||||
+104
-63
@@ -4,79 +4,98 @@ Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrb
|
|||||||
|
|
||||||
## Prerequisites
|
## Prerequisites
|
||||||
|
|
||||||
- **Go** 1.27+ with `toolchain go1.27.0`
|
- **Go** 1.27.1, the exact version the `go` directive in `go.mod` declares
|
||||||
- **just**, the command runner; every task below is a just recipe
|
- **just**, the command runner; every task below is a just recipe
|
||||||
- No external dependencies beyond the Go toolchain
|
- **A C compiler** (`gcc`): `just race` runs the suite under the race detector,
|
||||||
|
which needs cgo
|
||||||
|
- **Perl**: the `test`, `fmt-check`, `install-man` and `uninstall-man` recipes
|
||||||
|
are Perl programs
|
||||||
|
- **`gzip`**: `install-man` compresses the man pages with it
|
||||||
|
- A Linux host on amd64, arm64, riscv64 or loong64: `gasm debug` needs ptrace
|
||||||
|
and the JIT checks of `gasm verify` need executable memory
|
||||||
|
- **`golang.org/x/arch`**, the one module dependency, which the Go toolchain
|
||||||
|
fetches; nothing else sits outside the standard library
|
||||||
|
|
||||||
## Quick Start
|
## Setup
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
|
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
|
||||||
cd gasm-devkit
|
cd gasm-devkit
|
||||||
just install # go mod download
|
just build # compile bin/gasm, zero errors and zero warnings
|
||||||
just build # go vet + gofmt, must pass with zero output
|
just gates # build, fmt-check, vet, test, race: the definition of done
|
||||||
just test # full suite, race detector, 80 % coverage gate
|
|
||||||
```
|
```
|
||||||
|
|
||||||
## Just Recipes
|
## Recipes
|
||||||
|
|
||||||
### `just install`
|
Every recipe in the `justfile`, and what it does.
|
||||||
|
|
||||||
`go mod download`. The only module dependency, `golang.org/x/arch`, is
|
| Recipe | What it does |
|
||||||
used in tests only.
|
|---|---|
|
||||||
|
| `default` (bare `just`) | prints the recipe list (`@just --list`) |
|
||||||
### `just build`
|
| `just build` | compiles `bin/gasm` with `CGO_ENABLED=0` and stripped symbols; zero errors and zero warnings |
|
||||||
|
| `just test` | the test gate: the suite with `-count=1`, the coverage profile and the 80 % floor, then the CLI and debugger tests outside the profile |
|
||||||
Runs `go vet ./...` and checks `gofmt -l .` produces no output. This is
|
| `just race` | the same suite under the race detector; the expensive one, so it runs once, inside `gates` |
|
||||||
the minimum bar before any commit.
|
| `just unit [packages] [run]` | fast, cached, scoped run for iterating: no race and no coverage, so an unchanged package reports instantly |
|
||||||
|
| `just fuzz <target> <pkg> [fuzztime]` | time-boxed fuzz of one target; the package is required, because `go test -fuzz` refuses more than one |
|
||||||
|
| `just bench [packages]` | benchmarks (`-benchmem -count=5`); on an idle machine only |
|
||||||
|
| `just fmt` | formats the tree in place with `gofmt` |
|
||||||
|
| `just fmt-check` | zero diff; prints nothing when everything is formatted, which is the shape the CI step wants |
|
||||||
|
| `just vet` | both static gates: `go vet` and `go fix -diff` |
|
||||||
|
| `just gates` | `build`, `fmt-check`, `vet`, `test` and `race`, in that order: the definition of done |
|
||||||
|
| `just clean` | removes the build artefacts, `bin/` and `coverage.out` |
|
||||||
|
| `just install` | builds, then copies the binary into `bindir` (`~/.local/bin`) |
|
||||||
|
| `just uninstall` | removes the installed binary from `bindir` |
|
||||||
|
| `just install-man` | installs the man pages under `docs/man` into `~/.local/share/man/man1` (`MANDIR` overrides), gzip-compressed; not a gate |
|
||||||
|
| `just uninstall-man` | removes the installed man pages |
|
||||||
|
| `just run` | runs the CLI with `go run -buildvcs=true`; the recipe takes no arguments, so flags go through the package instead |
|
||||||
|
| `just dev` | the same as `run`; the project has no watcher to add |
|
||||||
|
| `just gen` | regenerates the `arch` instruction tables from the Go toolchain source; not a gate |
|
||||||
|
|
||||||
### `just test`
|
### `just test`
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
go test -race -count=1 ./...
|
go test -count=1 -timeout 10m -coverprofile=coverage.out \
|
||||||
|
./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... \
|
||||||
|
./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
|
||||||
```
|
```
|
||||||
|
|
||||||
Plus a coverage run over the ten analysable packages (arch, asm, ast,
|
The suite runs over the logic packages (`-count=1`, so no cached pass
|
||||||
format, lexer, lint, lsp, parser, token, verify; `debug` and `cmd/gasm`
|
counts): arch, asm, ast, disasm, format, lexer, lint, lsp, parser,
|
||||||
need hardware or are CLI glue) and an `awk` gate that fails if total
|
token, verify. `debug` traces a live process and `cmd/gasm` is thin CLI
|
||||||
coverage is below 80 %.
|
glue, so both sit outside the profile sweep, and a thin `cmd/` in it
|
||||||
|
would drag the coverage total under the floor. Their tests still run, in
|
||||||
### `just fmt`
|
a second invocation without a profile:
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
gofmt -w .
|
go test -count=1 -timeout 10m ./cmd/... ./debug/...
|
||||||
```
|
```
|
||||||
|
|
||||||
Run after editing any Go source. The output must be idempotent.
|
That covers the CLI's exit codes and the guard that compares the manual
|
||||||
|
pages with the binary's own help, and the debugger's architecture-neutral
|
||||||
|
units. The floor fails if the total is below 80 %. CI runs the same two
|
||||||
|
commands with the same ten-minute bound, so
|
||||||
|
the number is the same everywhere.
|
||||||
|
|
||||||
### `just run -- <args>`
|
### `just run`
|
||||||
|
|
||||||
Runs the CLI via `go run` with the version string stamped:
|
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
just run -- lint kernel_amd64.s
|
just run
|
||||||
just run -- fmt -w kernel_amd64.s
|
go run -buildvcs=true ./cmd/gasm lint kernel_amd64.s
|
||||||
just run -- verify --ground-truth kernel_amd64.s
|
go run -buildvcs=true ./cmd/gasm verify --ground-truth kernel_amd64.s
|
||||||
```
|
```
|
||||||
|
|
||||||
### `just install-bin`
|
The flag on `go run` is there because it does not stamp the build otherwise,
|
||||||
|
which `--version` would then report as `(devel)`.
|
||||||
Installs the `gasm` binary into `$GOBIN` with the release version
|
|
||||||
embedded via `-ldflags "-X main.version=..."`.
|
|
||||||
|
|
||||||
### `just gen`
|
### `just gen`
|
||||||
|
|
||||||
Regenerates the architecture instruction tables in `arch/` by parsing
|
Regenerates the architecture instruction tables in `arch/` by parsing the Go
|
||||||
the Go toolchain's own assembler source
|
toolchain's own assembler source
|
||||||
(`$GOROOT/src/cmd/internal/obj/<arch>/anames.go`). Requires a Go
|
(`$GOROOT/src/cmd/internal/obj/<arch>/anames.go`). Requires a Go
|
||||||
installation. Output is committed, with no runtime dependency on the
|
installation. Output is committed, with no runtime dependency on the
|
||||||
toolchain.
|
toolchain.
|
||||||
|
|
||||||
### `just uninstall`
|
## Running a single test
|
||||||
|
|
||||||
Removes `coverage.out`, the `gasm` binary, and `*.test` artefacts.
|
|
||||||
|
|
||||||
## Running Individual Tests
|
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
go test -run TestVexGroundTruth ./asm/
|
go test -run TestVexGroundTruth ./asm/
|
||||||
@@ -85,32 +104,54 @@ go test -run TestGOObjectLinkAndRun ./asm/
|
|||||||
go test -run TestFuzzWideCopy ./verify/
|
go test -run TestFuzzWideCopy ./verify/
|
||||||
```
|
```
|
||||||
|
|
||||||
## Debugger Note
|
Add `-v` for the sub-test names, and `-race` when the change touches
|
||||||
|
concurrency. `-count=1` defeats the test cache when a result looks stale.
|
||||||
|
|
||||||
`gasm debug` spawns a child process from the binary on `$PATH`. It does
|
## Coverage
|
||||||
not work with `go run`; install first:
|
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
just install-bin
|
just test
|
||||||
gasm debug --func decodeBlockAVX2 path/to/kernel_amd64.s
|
go tool cover -func=coverage.out
|
||||||
```
|
```
|
||||||
|
|
||||||
## Project Layout
|
The `total:` line is the number that matters, and it stays at 80 percent or
|
||||||
|
more.
|
||||||
|
|
||||||
|
## Debugging the build
|
||||||
|
|
||||||
|
```sh
|
||||||
|
go build -gcflags='-m' ./... # inlining decisions
|
||||||
|
go build -gcflags='-S' ./... # what the compiler generated
|
||||||
|
go tool asm -S kernel_amd64.s # how the toolchain's assembler encodes a kernel
|
||||||
|
gasm dis kernel_amd64.s # what gasm makes of the same kernel
|
||||||
|
gasm tokens kernel_amd64.s # the token stream
|
||||||
|
gasm profile kernel_amd64.s # the basic blocks of each function
|
||||||
```
|
```
|
||||||
cmd/gasm/ CLI entry point (subcommands)
|
|
||||||
token/ Lexical token kinds and positions
|
`gasm verify --ground-truth` is the differential check that ties the two
|
||||||
lexer/ Hand-written scanner
|
together: it compares gasm's bytes with `go tool asm`'s, with the relocation
|
||||||
ast/ Abstract syntax tree
|
sites masked, so an encoding drift shows up as a byte difference rather than a
|
||||||
parser/ Line-oriented parser
|
crash later.
|
||||||
arch/ Register and instruction tables (generated)
|
|
||||||
lint/ Static analysis rules
|
## Continuous integration
|
||||||
format/ Canonical formatter
|
|
||||||
lsp/ Language Server Protocol server
|
Workflows live in `.gitea/workflows/` and run on the project's own runners:
|
||||||
asm/ Standalone assembler, encoder, object emitters
|
Test on a push or pull request to `development`, race dispatched by hand, and
|
||||||
verify/ JIT execution, differential testing, ABI checks
|
the release on a `v*` tag. They are written by hand rather than through
|
||||||
debug/ Interactive ptrace debugger (all four architectures)
|
`just`, but they enforce the same set of gates minus the race detector, which
|
||||||
_gen/ Instruction table generator
|
the shared runner cannot afford on a push; a green `just gates` locally is
|
||||||
testdata/ Test fixtures
|
therefore the fastest way to a green pipeline.
|
||||||
docs/ Architecture, development, CLI reference
|
|
||||||
```
|
## Releases
|
||||||
|
|
||||||
|
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`,
|
||||||
|
which triggers the release workflow: it builds the portable Linux targets,
|
||||||
|
takes the notes from the matching `CHANGELOG.md` section and uploads the
|
||||||
|
assets. `SECURITY.md` carries the supported-versions table, so that table
|
||||||
|
moves with the release; the pipeline refuses a tag the policy does not name.
|
||||||
|
|
||||||
|
The version is never injected. `gasm --version` prints what the
|
||||||
|
toolchain recorded in the build information: the tag on a tagged
|
||||||
|
checkout, a pseudo-version naming the commit below one, `+dirty` on a
|
||||||
|
dirty tree, and `(devel)` outside version control. There is no
|
||||||
|
`-ldflags "-X"` anywhere and no version constant in the source.
|
||||||
|
|||||||
@@ -0,0 +1,73 @@
|
|||||||
|
.TH GASM-ASM 1 "2026-09-19" "gasm" "User Commands"
|
||||||
|
.SH NAME
|
||||||
|
gasm-asm \- assemble Plan 9 assembly without the Go toolchain
|
||||||
|
.SH SYNOPSIS
|
||||||
|
.B gasm asm [\-\-format raw|elf|goobj] [\-p pkg] [\-GOARCH arch] [\-o out] <file>
|
||||||
|
.SH DESCRIPTION
|
||||||
|
Assemble FILE without the Go toolchain: every TEXT function is encoded
|
||||||
|
to machine code and printed as a hex dump. Supported architectures:
|
||||||
|
amd64 (including VEX/AVX2 and EVEX/AVX-512), arm64 (AArch64 integer,
|
||||||
|
FP, conditional select, CRC32 and MOV pseudo), riscv64 (RV64IMAFDC and
|
||||||
|
RVC) and loong64 (LoongArch base ISA).
|
||||||
|
.PP
|
||||||
|
With
|
||||||
|
.B \-o
|
||||||
|
the output is written to a file instead. The
|
||||||
|
.B \-\-format
|
||||||
|
flag selects what is written:
|
||||||
|
.B raw
|
||||||
|
(the default) concatenates the functions and the data section into one
|
||||||
|
self-consistent image;
|
||||||
|
.B elf
|
||||||
|
emits a relocatable object (.text/.data sections, a symbol table and
|
||||||
|
one relocation per static-symbol reference, in the architecture's own
|
||||||
|
form: R_X86_64_PC32 on amd64, R_AARCH64_*, R_RISCV_* or R_LARCH_* on the
|
||||||
|
others) that links with the
|
||||||
|
system toolchain;
|
||||||
|
.B goobj
|
||||||
|
emits the Go toolchain's own object format, which cmd/link consumes
|
||||||
|
directly (it requires
|
||||||
|
.BR \-p ,
|
||||||
|
the package path, and the installed Go toolchain: the object preamble is
|
||||||
|
captured from
|
||||||
|
.B go tool asm
|
||||||
|
and the format version from
|
||||||
|
.BR "go version" ).
|
||||||
|
.PP
|
||||||
|
.B raw
|
||||||
|
and
|
||||||
|
.B elf
|
||||||
|
need no toolchain at all.
|
||||||
|
.PP
|
||||||
|
Framed functions receive the stack-split guard and the trailing
|
||||||
|
morestack block, byte-identical to the toolchain's output, so split
|
||||||
|
functions link too.
|
||||||
|
.SH OPTIONS
|
||||||
|
.TP
|
||||||
|
.B \-\-format \fIraw|elf|goobj\fR
|
||||||
|
Output format; the default is raw.
|
||||||
|
.TP
|
||||||
|
.B \-p \fIpkg\fR
|
||||||
|
Package path for --format goobj, qualifying the exported symbols.
|
||||||
|
.TP
|
||||||
|
.B \-GOARCH \fIarch\fR
|
||||||
|
Target architecture: amd64, arm64, riscv64 or loong64; overrides the
|
||||||
|
file-name suffix, which is how the suffix-less majority of GOROOT's
|
||||||
|
files (cpu_x86.s, stub.s, ...) become assemblable.
|
||||||
|
.TP
|
||||||
|
.B \-o \fIfile\fR
|
||||||
|
Write the output to this file instead of a hex dump on stdout.
|
||||||
|
.SH EXIT STATUS
|
||||||
|
Exits 0 on success, 1 when parsing or assembly fails, and 2 on a usage
|
||||||
|
error.
|
||||||
|
.SH EXAMPLES
|
||||||
|
.nf
|
||||||
|
gasm asm \-o hello.bin hello_amd64.s raw image
|
||||||
|
gasm asm \-\-format elf \-o k.o k_amd64.s linkable ELF object
|
||||||
|
gasm asm \-\-format goobj \-p pkg/path \-o k.o k_amd64.s Go object for go build
|
||||||
|
gasm asm \-GOARCH amd64 cpu_x86.s arch override
|
||||||
|
.fi
|
||||||
|
.SH SEE ALSO
|
||||||
|
.BR gasm (1),
|
||||||
|
.BR gasm\-dis (1),
|
||||||
|
.BR gasm\-verify (1)
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
.TH GASM-AUDIT-INSTRUCTIONS 1 "2026-09-19" "gasm" "User Commands"
|
||||||
|
.SH NAME
|
||||||
|
gasm-audit-instructions \- diff the encoder against the Go toolchain, or measure a corpus
|
||||||
|
.SH SYNOPSIS
|
||||||
|
.B gasm audit\-instructions [\-\-corpus [\fIdir\fR]] [amd64|arm64|riscv64|loong64]
|
||||||
|
.SH DESCRIPTION
|
||||||
|
Compare the gasm encoder for the given architecture (default amd64)
|
||||||
|
against
|
||||||
|
.B go tool asm
|
||||||
|
and print the diff: superset encodings (gasm-only, shippable via
|
||||||
|
.BR "gasm asm \-\-format goobj" )
|
||||||
|
and known-but-unencodable names (the backlog). The Go side is probed
|
||||||
|
black-box one bare mnemonic at a time, so the audit tracks whatever
|
||||||
|
toolchain
|
||||||
|
.B go env GOROOT
|
||||||
|
provides; the gasm side answers from the encoder table on amd64 and from
|
||||||
|
trial assembly over a battery of operand shapes elsewhere. Names
|
||||||
|
.B go tool asm
|
||||||
|
knows and gasm does not cannot be enumerated by probing, because Go's
|
||||||
|
table is visible only through names already in the gasm table; the report
|
||||||
|
closes with a note saying so rather than listing them.
|
||||||
|
.PP
|
||||||
|
With
|
||||||
|
.BR \-\-corpus ,
|
||||||
|
the audit changes shape: it assembles every
|
||||||
|
.I .s
|
||||||
|
file under the given directory (default GOROOT/src) with the gasm
|
||||||
|
encoder only, no toolchain probing. A file whose name carries a
|
||||||
|
recognisable _arch suffix is attempted for that architecture; a file
|
||||||
|
without one is attempted for all four, exactly as a GOARCH build would
|
||||||
|
compile it. The report gives the headline number (files that assemble
|
||||||
|
for every target architecture), the per-architecture pass rates and the
|
||||||
|
most common failure reasons, which drive the encodability backlog by
|
||||||
|
frequency rather than by table order. A run over GOROOT takes under a
|
||||||
|
second.
|
||||||
|
.SH OPTIONS
|
||||||
|
.TP
|
||||||
|
.B \-\-corpus [\fIdir\fR]
|
||||||
|
Assemble a corpus of .s files and report pass rates and failure
|
||||||
|
reasons.
|
||||||
|
.SH EXIT STATUS
|
||||||
|
The mnemonic-diff mode reports through its output and exits 0; a failed
|
||||||
|
probe or an unknown architecture exits non-zero.
|
||||||
|
.SH EXAMPLES
|
||||||
|
.nf
|
||||||
|
gasm audit\-instructions amd64
|
||||||
|
gasm audit\-instructions \-\-corpus
|
||||||
|
gasm audit\-instructions \-\-corpus "$(go env GOROOT)/src/crypto"
|
||||||
|
.fi
|
||||||
|
.SH SEE ALSO
|
||||||
|
.BR gasm (1),
|
||||||
|
.BR gasm\-asm (1)
|
||||||
@@ -0,0 +1,117 @@
|
|||||||
|
.TH GASM-DEBUG 1 "2026-09-19" "gasm" "User Commands"
|
||||||
|
.SH NAME
|
||||||
|
gasm-debug \- interactive source-level debugger for JIT-assembled functions
|
||||||
|
.SH SYNOPSIS
|
||||||
|
.B gasm debug <file.s> \-\-func <name>
|
||||||
|
.SH DESCRIPTION
|
||||||
|
Interactive debugger for JIT-assembled functions. Launches the function
|
||||||
|
in a traced subprocess (ptrace), then provides a REPL for
|
||||||
|
single-stepping, breakpoints, register and memory inspection.
|
||||||
|
.PP
|
||||||
|
With
|
||||||
|
.B \-\-script
|
||||||
|
the REPL commands run from a file and the session ends: the headless
|
||||||
|
mode CI and scripts use.
|
||||||
|
.B \-\-cover
|
||||||
|
runs to completion with a breakpoint on every instruction and reports
|
||||||
|
which executed and how often, the label-level coverage view.
|
||||||
|
.SH REPL COMMANDS
|
||||||
|
.TP
|
||||||
|
.B break \fIlabel|addr|line\fR [\fBif \fIreg op val|reg|*addr\fR], b
|
||||||
|
Set a breakpoint at a label, an address or a source line number, optionally
|
||||||
|
conditional on a register comparison: against a constant, against another
|
||||||
|
register, or against the 8-byte word at
|
||||||
|
.BR *addr .
|
||||||
|
.TP
|
||||||
|
.B delete \fIlabel|addr\fR, d
|
||||||
|
Remove a breakpoint.
|
||||||
|
.TP
|
||||||
|
.B info break, info breakpoints, info b
|
||||||
|
List all breakpoints.
|
||||||
|
.TP
|
||||||
|
.BR step " [" n ], " s
|
||||||
|
Single-step n instructions; the default is 1.
|
||||||
|
.TP
|
||||||
|
.BR next ", " n
|
||||||
|
Step over a CALL.
|
||||||
|
.TP
|
||||||
|
.BR finish ", " fin
|
||||||
|
Run until the function returns.
|
||||||
|
.TP
|
||||||
|
.BR continue ", " c
|
||||||
|
Run until a breakpoint, watchpoint or exit.
|
||||||
|
.TP
|
||||||
|
.BR disas " [" n ], " u
|
||||||
|
Disassemble n instructions at PC.
|
||||||
|
.TP
|
||||||
|
.B regs
|
||||||
|
Print general-purpose and vector registers.
|
||||||
|
.TP
|
||||||
|
.B where
|
||||||
|
Show the source line and nearest label at PC.
|
||||||
|
.TP
|
||||||
|
.B stack
|
||||||
|
Show the stack near RSP (return address and ABI0 args).
|
||||||
|
.TP
|
||||||
|
.BR bt ", " backtrace
|
||||||
|
Backtrace: current frame plus return address.
|
||||||
|
.TP
|
||||||
|
.B x [\fIaddr\fR] [\fIlen\fR]
|
||||||
|
Hex-dump memory; the defaults are the current PC and 64 bytes.
|
||||||
|
.TP
|
||||||
|
.B w \fIaddr val...\fR
|
||||||
|
Write bytes to memory.
|
||||||
|
.TP
|
||||||
|
.B set \fIreg value\fR
|
||||||
|
Set a register.
|
||||||
|
.TP
|
||||||
|
.B watch \fIaddr\fR [\fBr|w\fR] [\fIsize\fR]
|
||||||
|
Set a hardware watchpoint; writes are watched by default.
|
||||||
|
.TP
|
||||||
|
.B unwatch [\fIslot\fR]
|
||||||
|
Clear one watchpoint, or all without an argument.
|
||||||
|
.TP
|
||||||
|
.BR labels ", " l
|
||||||
|
List function labels and offsets.
|
||||||
|
.TP
|
||||||
|
.BR help ", " h ", " ?
|
||||||
|
Show command help.
|
||||||
|
.TP
|
||||||
|
.BR quit ", " q
|
||||||
|
Kill the debuggee and exit.
|
||||||
|
.SH OPTIONS
|
||||||
|
.TP
|
||||||
|
.B \-args \fIfile\fR
|
||||||
|
File containing the ABI0 argument block.
|
||||||
|
.TP
|
||||||
|
.B \-buf \fIspec\fR
|
||||||
|
Buffer specification: name:size:pattern[,name:size:pattern...] where
|
||||||
|
pattern is zero, ones, seq, or hex.
|
||||||
|
.TP
|
||||||
|
.B \-cover
|
||||||
|
Run to completion with a breakpoint on every instruction and report
|
||||||
|
which executed and how often.
|
||||||
|
.TP
|
||||||
|
.B \-func \fIname\fR
|
||||||
|
Function to debug.
|
||||||
|
.TP
|
||||||
|
.B \-script \fIfile\fR
|
||||||
|
Run REPL commands from a file (one per line) and exit; - reads stdin.
|
||||||
|
.TP
|
||||||
|
.B \-timeout \fIduration\fR
|
||||||
|
Kill the debuggee after this duration (e.g. 30s); for headless --script
|
||||||
|
runs; a timeout exits 3.
|
||||||
|
.SH EXIT STATUS
|
||||||
|
Exits 0 when the scripted session completes, 1 when the debuggee crashes
|
||||||
|
or a check fails, and 3 when
|
||||||
|
.B \-\-timeout
|
||||||
|
kills the debuggee; the debugger is Linux-only.
|
||||||
|
.SH EXAMPLES
|
||||||
|
.nf
|
||||||
|
gasm debug \-\-func name k.s
|
||||||
|
gasm debug \-\-func name \-\-script cmds.txt \-\-timeout 30s k.s
|
||||||
|
gasm debug \-\-func name \-\-cover k.s
|
||||||
|
.fi
|
||||||
|
.SH SEE ALSO
|
||||||
|
.BR gasm (1),
|
||||||
|
.BR gasm\-verify (1)
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
.TH GASM-DIFF 1 "2026-09-19" "gasm" "User Commands"
|
||||||
|
.SH NAME
|
||||||
|
gasm-diff \- compare the machine code of two assembly files
|
||||||
|
.SH SYNOPSIS
|
||||||
|
.B gasm diff [\-GOARCH arch] <file1.s> <file2.s>
|
||||||
|
.SH DESCRIPTION
|
||||||
|
Compare the machine code produced by assembling two files. Shows which
|
||||||
|
functions differ and the byte-level differences. Useful for verifying
|
||||||
|
that two implementations produce identical code, or for tracking
|
||||||
|
encoding changes between Go assembler versions.
|
||||||
|
.PP
|
||||||
|
Functions are paired by exact name unless
|
||||||
|
.B \-\-map
|
||||||
|
says otherwise, so
|
||||||
|
.B \-\-map wideCopyAVX2=wideCopyAVX512
|
||||||
|
pairs two variants regardless of suffix.
|
||||||
|
.SH OPTIONS
|
||||||
|
.TP
|
||||||
|
.B \-GOARCH \fIarch\fR
|
||||||
|
Target architecture for both files: amd64, arm64, riscv64 or loong64;
|
||||||
|
overrides the file-name suffixes.
|
||||||
|
.TP
|
||||||
|
.B \-\-map \fIspec\fR
|
||||||
|
Comma-separated old=new pairs to match functions with different names.
|
||||||
|
.SH EXIT STATUS
|
||||||
|
Exits 0 when every paired function is identical and 1 when anything
|
||||||
|
differs; a usage error exits 2.
|
||||||
|
.SH EXAMPLES
|
||||||
|
.nf
|
||||||
|
gasm diff hello_amd64.s hello_amd64.s
|
||||||
|
gasm diff \-\-map wideCopyAVX2=wideCopyAVX512 avx2_amd64.s avx512_amd64.s
|
||||||
|
.fi
|
||||||
|
.SH SEE ALSO
|
||||||
|
.BR gasm (1),
|
||||||
|
.BR gasm\-asm (1)
|
||||||
@@ -0,0 +1,36 @@
|
|||||||
|
.TH GASM-DIS 1 "2026-09-19" "gasm" "User Commands"
|
||||||
|
.SH NAME
|
||||||
|
gasm-dis \- disassemble machine code to instruction text
|
||||||
|
.SH SYNOPSIS
|
||||||
|
.B gasm dis [\-a arch] <file>
|
||||||
|
.SH DESCRIPTION
|
||||||
|
Disassemble machine code to instruction text, decoded through
|
||||||
|
golang.org/x/arch.
|
||||||
|
.PP
|
||||||
|
With a
|
||||||
|
.I .s
|
||||||
|
file, the file is assembled first and the listing follows the real
|
||||||
|
layout: one block per TEXT function, local labels printed at their
|
||||||
|
offsets. The architecture comes from the file-name suffix, or from
|
||||||
|
.BR \-a .
|
||||||
|
.PP
|
||||||
|
With any other file, or
|
||||||
|
.B \-
|
||||||
|
for standard input, the bytes are disassembled linearly and
|
||||||
|
.B \-a
|
||||||
|
selects the architecture (amd64, arm64, riscv64 or loong64).
|
||||||
|
.SH OPTIONS
|
||||||
|
.TP
|
||||||
|
.B \-a \fIarch\fR
|
||||||
|
Architecture for raw input: amd64, arm64, riscv64 or loong64.
|
||||||
|
.SH EXIT STATUS
|
||||||
|
Exits 0 on success, 1 when assembly or decoding fails, and 2 on a usage
|
||||||
|
error.
|
||||||
|
.SH EXAMPLES
|
||||||
|
.nf
|
||||||
|
gasm dis k_amd64.s assemble, then list each function
|
||||||
|
gasm dis \-a amd64 \- < dump.bin disassemble raw bytes from stdin
|
||||||
|
.fi
|
||||||
|
.SH SEE ALSO
|
||||||
|
.BR gasm (1),
|
||||||
|
.BR gasm\-asm (1)
|
||||||
@@ -0,0 +1,59 @@
|
|||||||
|
.TH GASM-FMT 1 "2026-09-19" "gasm" "User Commands"
|
||||||
|
.SH NAME
|
||||||
|
gasm-fmt \- canonicalise the formatting of Plan 9 assembly sources
|
||||||
|
.SH SYNOPSIS
|
||||||
|
.B gasm fmt [\-w|\-l|\-d] [path...]
|
||||||
|
.SH DESCRIPTION
|
||||||
|
Canonicalise the formatting of Plan 9 assembly sources: indentation,
|
||||||
|
operand spacing, per-function mnemonic alignment and blank-line layout
|
||||||
|
(exactly one blank line before each label, TEXT and GLOBL block).
|
||||||
|
Formatting is idempotent and preserves every line, comments included.
|
||||||
|
.PP
|
||||||
|
With no paths, or a directory path, every
|
||||||
|
.I .s
|
||||||
|
file below it is reformatted in place and the changed files are listed,
|
||||||
|
the way
|
||||||
|
.B go fmt
|
||||||
|
does;
|
||||||
|
.B .
|
||||||
|
and
|
||||||
|
.B _
|
||||||
|
directories are skipped. Explicit file paths print to stdout unless
|
||||||
|
.B \-w
|
||||||
|
is given.
|
||||||
|
.PP
|
||||||
|
.B \-l
|
||||||
|
and
|
||||||
|
.B \-d
|
||||||
|
rewrite nothing:
|
||||||
|
.B \-l
|
||||||
|
prints the paths whose formatting differs from gasm's (empty output
|
||||||
|
means everything is formatted, which is what a CI check wants),
|
||||||
|
.B \-d
|
||||||
|
prints the diffs. They are mutually exclusive.
|
||||||
|
.SH OPTIONS
|
||||||
|
.TP
|
||||||
|
.B \-d
|
||||||
|
Print diffs instead of rewriting files.
|
||||||
|
.TP
|
||||||
|
.B \-l
|
||||||
|
List files whose formatting differs from gasm's.
|
||||||
|
.TP
|
||||||
|
.B \-w
|
||||||
|
Write the result to the source file.
|
||||||
|
.SH EXIT STATUS
|
||||||
|
Exits 0 on success, 1 when a path cannot be read or written, and 2 on a
|
||||||
|
usage error (combining
|
||||||
|
.B \-l
|
||||||
|
and
|
||||||
|
.BR \-d ,
|
||||||
|
or an unknown flag).
|
||||||
|
.SH EXAMPLES
|
||||||
|
.nf
|
||||||
|
gasm fmt reformat every .s below here
|
||||||
|
gasm fmt \-w kernel_amd64.s canonicalise one file in place
|
||||||
|
gasm fmt \-l *.s list files whose formatting differs
|
||||||
|
gasm fmt \-d kernel_amd64.s print a unified diff instead
|
||||||
|
.fi
|
||||||
|
.SH SEE ALSO
|
||||||
|
.BR gasm (1)
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user