Compare commits
141
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5a8e9acbf3 | ||
|
|
2747fce7d3 | ||
|
|
e02918c17b | ||
|
|
02a6359c1f | ||
|
|
b4c1e133c0 | ||
|
|
1107928870 | ||
|
|
bc4ac93fd9 | ||
|
|
c1bca7ce7e | ||
|
|
fefb76beb9 | ||
|
|
42bc1669d7 | ||
|
|
69dcbec8ef | ||
|
|
f405cea5bc | ||
|
|
f57377abb9 | ||
|
|
dd1782c538 | ||
|
|
17cc49fee4 | ||
|
|
bafb2fd130 | ||
|
|
2f679326c2 | ||
|
|
96f2dd65b4 | ||
|
|
ca887d3927 | ||
|
|
6570709226 | ||
|
|
9a5217d9c1 | ||
|
|
4d01bb3ecf | ||
|
|
332c63e440 | ||
|
|
b306c210c6 | ||
|
|
b9015e1c2e | ||
|
|
26c5008136 | ||
|
|
74d6b90d69 | ||
|
|
7b11c62f53 | ||
|
|
8eed54b3da | ||
|
|
4be16dcdf5 | ||
|
|
ded9cabdf4 | ||
|
|
cf6bc6987e | ||
|
|
ff7b1452b1 | ||
|
|
517c1cea25 | ||
|
|
a3e3010e0f | ||
|
|
057c4eb545 | ||
|
|
f720381d43 | ||
|
|
2c9042d62c | ||
|
|
82ef289d3a | ||
|
|
7246b0e002 | ||
|
|
8cfd40aac8 | ||
|
|
5382c9a8e4 | ||
|
|
53de91b2df | ||
|
|
8a36af7c7d | ||
|
|
e9789ce3f4 | ||
|
|
837231c068 | ||
|
|
95025be1bc | ||
|
|
03a964bb2d | ||
|
|
123a16e346 | ||
|
|
9701812bee | ||
|
|
29ac03468e | ||
|
|
bfb7701db1 | ||
|
|
e8b6ff5d7c | ||
|
|
1456907000 | ||
|
|
ec1c521187 | ||
|
|
a7744c24bd | ||
|
|
522e6f2ae8 | ||
|
|
81d4bd81e4 | ||
|
|
687678a2ea | ||
|
|
b0f9071bf5 | ||
|
|
81e2673923 | ||
|
|
75e9fd771b | ||
|
|
863926abd6 | ||
|
|
241e7256f6 | ||
|
|
6556b85abf | ||
|
|
289cabe993 | ||
|
|
d6cf7cfa44 | ||
|
|
4cc2f0eba5 | ||
|
|
97dfaa7526 | ||
|
|
66aa4dbc8b | ||
|
|
dce5d31462 | ||
|
|
9dc3987e02 | ||
|
|
9b238a525a | ||
|
|
ad82aac663 | ||
|
|
0629f5e2df | ||
|
|
ecb203dcf5 | ||
|
|
6c672567f3 | ||
|
|
cc6e416c59 | ||
|
|
c66a47973a | ||
|
|
9629897202 | ||
|
|
5399a8a724 | ||
|
|
de5d9f358e | ||
|
|
ca3fdce0e0 | ||
|
|
fc2d92eabd | ||
|
|
39d2e80145 | ||
|
|
9f4f949c1f | ||
|
|
f0d5238c47 | ||
|
|
2931bbd6b2 | ||
|
|
63562a503a | ||
|
|
e836d6150d | ||
|
|
8a51b060da | ||
|
|
d3d47db727 | ||
|
|
0758556b7d | ||
|
|
ddb8440340 | ||
|
|
f15ff66fb1 | ||
|
|
187e4856d3 | ||
|
|
d315a998ce | ||
|
|
a6f3828c02 | ||
|
|
e3b35bb817 | ||
|
|
eb0a89e58d | ||
|
|
dd32d9e66e | ||
|
|
3a73acb20a | ||
|
|
7604a9443f | ||
|
|
b3908fc43d | ||
|
|
eb8b0cd316 | ||
|
|
a8bfd54ed2 | ||
|
|
375182ef1f | ||
|
|
87b1081c53 | ||
|
|
f3c8510a58 | ||
|
|
ebdf14939f | ||
|
|
79a2c16bac | ||
|
|
401386956c | ||
|
|
4258131a3a | ||
|
|
94e09e8070 | ||
|
|
7aefe6a42d | ||
|
|
ac1c05c793 | ||
|
|
93c47a312a | ||
|
|
708d0a0a5e | ||
|
|
3c8f7cb411 | ||
|
|
7c5b7a1419 | ||
|
|
bc3f448738 | ||
|
|
f37f183577 | ||
|
|
1e77e58250 | ||
|
|
1d0969ed64 | ||
|
|
23c001be51 | ||
|
|
96e81cc98d | ||
|
|
c834d98210 | ||
|
|
03d6d4da54 | ||
|
|
0b42ce7952 | ||
|
|
288a64ccd2 | ||
|
|
5fddfa704b | ||
|
|
a2bb5eeb4e | ||
|
|
48449b7a7f | ||
|
|
3de043c494 | ||
|
|
0078f7be5c | ||
|
|
6a7317d141 | ||
|
|
d08523caa5 | ||
|
|
20e4b8d9c4 | ||
|
|
61f4247cef | ||
|
|
049872ddff | ||
|
|
3669f64ff6 |
@@ -0,0 +1,42 @@
|
|||||||
|
# FreeBSD compile gates. Dispatched by hand, never on a push.
|
||||||
|
#
|
||||||
|
# The debugger's ptrace surface and the JIT substrate are the two
|
||||||
|
# FreeBSD-portable layers the tree carries; the forge has no FreeBSD runner,
|
||||||
|
# so they can only be compile-gated, and three foreign-GOOS builds of the
|
||||||
|
# whole module are minutes of one-core work the push pipeline's budget cannot
|
||||||
|
# carry. The push pipeline stays fast and light; this workflow is the
|
||||||
|
# deliberate run, before a release or after touching the ported layers.
|
||||||
|
# Running the ptrace suite itself needs real FreeBSD hardware.
|
||||||
|
#
|
||||||
|
# A dispatched workflow takes no concurrency block: it is one deliberate run.
|
||||||
|
name: FreeBSD build
|
||||||
|
|
||||||
|
on:
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
env:
|
||||||
|
# One core: parallelism buys no speed here and costs memory the box does not have.
|
||||||
|
GOFLAGS: -p=1
|
||||||
|
GOMAXPROCS: "2"
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
build:
|
||||||
|
runs-on: fedora
|
||||||
|
timeout-minutes: 10
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
|
- uses: actions/setup-go@v6
|
||||||
|
with:
|
||||||
|
# The module is the source of truth for the version, so it cannot drift.
|
||||||
|
go-version-file: go.mod
|
||||||
|
cache: true
|
||||||
|
|
||||||
|
- name: FreeBSD build (amd64)
|
||||||
|
run: GOOS=freebsd GOARCH=amd64 go build ./...
|
||||||
|
|
||||||
|
- name: FreeBSD build (arm64)
|
||||||
|
run: GOOS=freebsd GOARCH=arm64 go build ./...
|
||||||
|
|
||||||
|
- name: FreeBSD build (riscv64)
|
||||||
|
run: GOOS=freebsd GOARCH=riscv64 go build ./...
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
# Race, Go. Dispatched by hand, and never a gate on a push or a tag: the release tag is
|
||||||
|
# cut only after `just gates` has already raced the tree, so this workflow is the
|
||||||
|
# explicit second opinion, not a step of the release.
|
||||||
|
#
|
||||||
|
# The race detector roughly doubles both time and memory, which the shared runner box
|
||||||
|
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once per
|
||||||
|
# task; here it is a decision rather than a routine.
|
||||||
|
#
|
||||||
|
# Every step is one command, so the step that fails is the gate that failed.
|
||||||
|
name: Race
|
||||||
|
|
||||||
|
on:
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
env:
|
||||||
|
# One core: parallelism buys no speed here and costs memory the box does not have.
|
||||||
|
GOFLAGS: -p=1
|
||||||
|
GOMAXPROCS: "2"
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
race:
|
||||||
|
runs-on: fedora
|
||||||
|
timeout-minutes: 20
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
|
- uses: actions/setup-go@v6
|
||||||
|
with:
|
||||||
|
go-version-file: go.mod
|
||||||
|
cache: true
|
||||||
|
|
||||||
|
- name: Install gcc
|
||||||
|
# The race detector needs cgo and the runner image carries no C compiler.
|
||||||
|
run: dnf install -y gcc
|
||||||
|
|
||||||
|
- name: Race
|
||||||
|
run: go test -race -count=1 -timeout 10m ./...
|
||||||
+319
-86
@@ -1,73 +1,222 @@
|
|||||||
# Release — gasm binaries. Runs on version tags (v0.28.0) pushed to main.
|
# Release, Go binaries. Runs on version tags (v1.2.3) pushed to main.
|
||||||
|
#
|
||||||
|
# The module sits at the repository root: the toolchain records a version only for a root
|
||||||
|
# module, measured on go1.27.1, so a build of a module in a subdirectory reports (devel)
|
||||||
|
# even at its own <module>/vX.Y.Z tag and this workflow's smoke test can never pass for
|
||||||
|
# it. A Go repository is one module at the root.
|
||||||
|
#
|
||||||
|
# The version contract these steps implement: nothing is injected. The toolchain records
|
||||||
|
# the tag into the binary's build information, so the build simply has to happen at the
|
||||||
|
# tag, which the trigger guarantees.
|
||||||
|
#
|
||||||
|
# The gates run in their own job, once, before the matrix, minus the race detector: race
|
||||||
|
# never runs on a push path or a tag, and the local gate raced this tree before the tag
|
||||||
|
# was cut. Putting the gates inside the matrix would run the whole suite once per target
|
||||||
|
# on the box that also hosts the forge. Each job validates the tag for itself rather than
|
||||||
|
# passing a value between jobs, so no workflow feature has to be trusted for the version
|
||||||
|
# to reach the file name.
|
||||||
name: Release
|
name: Release
|
||||||
|
|
||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
tags: ["v*"]
|
tags: ["v*"]
|
||||||
|
|
||||||
|
env:
|
||||||
|
# The box is shared with the forge, so parallelism is bounded on purpose. The gates job
|
||||||
|
# needs it most; the build jobs inherit it for their parallel compilation.
|
||||||
|
GOFLAGS: -p=1
|
||||||
|
GOMAXPROCS: "2"
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
|
gates:
|
||||||
|
runs-on: fedora
|
||||||
|
timeout-minutes: 10
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
|
- uses: actions/setup-go@v6
|
||||||
|
with:
|
||||||
|
go-version-file: go.mod
|
||||||
|
cache: true
|
||||||
|
|
||||||
|
- name: Install Perl
|
||||||
|
# Perl for the steps below. The install is a no-op where the package
|
||||||
|
# is already present.
|
||||||
|
run: dnf install -y perl
|
||||||
|
|
||||||
|
- name: Validate the tag
|
||||||
|
env:
|
||||||
|
VERSION: ${{ gitea.ref_name }}
|
||||||
|
run: |
|
||||||
|
perl -e '
|
||||||
|
my $v = $ENV{VERSION} // q{};
|
||||||
|
$v =~ m{^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$}
|
||||||
|
or die qq{ERROR: expected a semver tag like v1.2.3, got: $v\n};
|
||||||
|
print qq{tag $v\n};
|
||||||
|
'
|
||||||
|
|
||||||
|
- name: Security policy names this release
|
||||||
|
# The supported-versions table is the one part of SECURITY.md that
|
||||||
|
# carries a version, so it goes stale the moment a tag is cut. Fail
|
||||||
|
# here rather than publish a policy naming the previous release.
|
||||||
|
env:
|
||||||
|
VERSION: ${{ gitea.ref_name }}
|
||||||
|
run: |
|
||||||
|
perl -e '
|
||||||
|
my $v = $ENV{VERSION} // q{};
|
||||||
|
(my $nv = $v) =~ s/^v//;
|
||||||
|
open(my $f, q{<}, q{SECURITY.md}) or die qq{SECURITY.md: $!\n};
|
||||||
|
local $/;
|
||||||
|
my $t = <$f>;
|
||||||
|
close $f;
|
||||||
|
$t =~ m{^\|\s*\Q$nv\E\s*\|\s*yes\s*\|}m
|
||||||
|
or die qq{ERROR: SECURITY.md does not name $nv as supported; update the table before releasing.\n};
|
||||||
|
print qq{SECURITY.md names $nv\n};
|
||||||
|
'
|
||||||
|
|
||||||
|
- name: Build
|
||||||
|
run: go build ./...
|
||||||
|
|
||||||
|
- name: Format
|
||||||
|
run: |
|
||||||
|
perl -e '
|
||||||
|
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
|
||||||
|
my @bad = <$g>;
|
||||||
|
close($g);
|
||||||
|
print @bad;
|
||||||
|
exit(@bad ? 1 : 0);
|
||||||
|
'
|
||||||
|
|
||||||
|
- name: Vet
|
||||||
|
run: go vet ./...
|
||||||
|
|
||||||
|
- name: Modernise
|
||||||
|
run: go fix -diff ./...
|
||||||
|
|
||||||
|
- name: Tests
|
||||||
|
# The same command as in test.yml, so the floor is the same number everywhere.
|
||||||
|
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
|
||||||
|
|
||||||
|
- name: Tests outside the coverage set
|
||||||
|
# The same command as in test.yml: the CLI's exit codes and manual-page guard,
|
||||||
|
# and the debugger's architecture-neutral units, run outside the floor.
|
||||||
|
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
|
||||||
|
|
||||||
|
- name: Coverage floor
|
||||||
|
run: |
|
||||||
|
perl -e '
|
||||||
|
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
|
||||||
|
my $total;
|
||||||
|
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
|
||||||
|
close($c);
|
||||||
|
die qq{no total line in coverage.out\n} unless defined $total;
|
||||||
|
printf qq{Total coverage: %s%%\n}, $total;
|
||||||
|
exit($total < 80 ? 1 : 0);
|
||||||
|
'
|
||||||
|
|
||||||
build:
|
build:
|
||||||
runs-on: fedora
|
runs-on: fedora
|
||||||
|
timeout-minutes: 25
|
||||||
|
needs: gates
|
||||||
strategy:
|
strategy:
|
||||||
fail-fast: false
|
fail-fast: false
|
||||||
matrix:
|
matrix:
|
||||||
|
# Portable targets: amd64, arm64, loong64 and riscv64 on Linux, at the toolchain
|
||||||
|
# default level. No 32-bit, no wasm, no macOS, no Windows. FreeBSD stays out until
|
||||||
|
# verify/jit.go ports off syscall.Mprotect: the Go syscall package defines no
|
||||||
|
# Mprotect for freebsd, and verify/jit.go:50 calls it to drop the write bit from
|
||||||
|
# the JIT mapping, so every freebsd target fails to build with "undefined:
|
||||||
|
# syscall.Mprotect" (verified for amd64, arm64 and riscv64 on go1.27.1).
|
||||||
include:
|
include:
|
||||||
- goos: linux
|
- goos: linux
|
||||||
goarch: amd64
|
goarch: amd64
|
||||||
- goos: linux
|
- goos: linux
|
||||||
goarch: arm64
|
goarch: arm64
|
||||||
- goos: linux
|
|
||||||
goarch: riscv64
|
|
||||||
- goos: linux
|
- goos: linux
|
||||||
goarch: loong64
|
goarch: loong64
|
||||||
|
- goos: linux
|
||||||
|
goarch: riscv64
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v7
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
- uses: actions/setup-go@v6
|
- uses: actions/setup-go@v6
|
||||||
with:
|
with:
|
||||||
go-version: "1.27"
|
go-version-file: go.mod
|
||||||
|
cache: true
|
||||||
|
|
||||||
- name: Download dependencies
|
- name: Install Perl
|
||||||
run: go mod download
|
run: dnf install -y perl
|
||||||
|
|
||||||
- name: Validate tag and build
|
- name: Validate the tag
|
||||||
id: build
|
id: version
|
||||||
env:
|
env:
|
||||||
VERSION: ${{ gitea.ref_name }}
|
VERSION: ${{ gitea.ref_name }}
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
perl -e '
|
||||||
|
my $v = $ENV{VERSION} // q{};
|
||||||
|
$v =~ m{^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$}
|
||||||
|
or die qq{ERROR: expected a semver tag like v1.2.3, got: $v\n};
|
||||||
|
(my $nv = $v) =~ s{^v}{};
|
||||||
|
open(my $o, q{>>}, $ENV{GITEA_OUTPUT}) or die qq{GITEA_OUTPUT: $!};
|
||||||
|
print $o qq{version_no_v=$nv\n};
|
||||||
|
close($o);
|
||||||
|
print qq{version $nv\n};
|
||||||
|
'
|
||||||
|
|
||||||
if ! echo "$VERSION" | grep -qE '^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$'; then
|
- name: Build
|
||||||
echo "ERROR: expected a semver tag like v1.2.3, got: '$VERSION'"
|
env:
|
||||||
exit 1
|
VERSION_NO_V: ${{ steps.version.outputs.version_no_v }}
|
||||||
fi
|
GOOS: ${{ matrix.goos }}
|
||||||
|
GOARCH: ${{ matrix.goarch }}
|
||||||
VERSION_NO_V="${VERSION#v}"
|
CGO_ENABLED: "0"
|
||||||
echo "version_no_v=${VERSION_NO_V}" >> "$GITEA_OUTPUT"
|
run: |
|
||||||
|
# Nothing is injected. The toolchain records the tag into the binary's build
|
||||||
mkdir -p bin
|
# information, so the version is right because this build happens at the tag, and
|
||||||
GOOS=${{ matrix.goos }} GOARCH=${{ matrix.goarch }} CGO_ENABLED=0 \
|
# there is no path for anyone to get wrong. -s -w only strips symbols.
|
||||||
go build -ldflags "-s -w -X main.version=${VERSION_NO_V}" \
|
go build -ldflags "-s -w" -o "bin/gasm-${VERSION_NO_V}-${GOOS}-${GOARCH}" ./cmd/gasm
|
||||||
-o "bin/gasm-${VERSION_NO_V}-${{ matrix.goos }}-${{ matrix.goarch }}" \
|
|
||||||
./cmd/gasm
|
|
||||||
|
|
||||||
|
# Artifacts stay on v3: v4 and later detect Gitea as GHES and abort.
|
||||||
- name: Upload artifact
|
- name: Upload artifact
|
||||||
uses: actions/upload-artifact@v3
|
uses: actions/upload-artifact@v3
|
||||||
with:
|
with:
|
||||||
name: gasm-${{ matrix.goos }}-${{ matrix.goarch }}
|
name: gasm-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||||
path: bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
path: bin/gasm-${{ steps.version.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||||
if-no-files-found: error
|
if-no-files-found: error
|
||||||
|
|
||||||
- name: Smoke test
|
- name: Smoke test
|
||||||
|
# Only a binary matching the runner can be run here. The check is not that --version
|
||||||
|
# exits cleanly but that it reports the tag and nothing more: a build outside version
|
||||||
|
# control reports (devel), and a build whose tree was dirty reports +dirty, and both
|
||||||
|
# would otherwise be published.
|
||||||
if: matrix.goos == 'linux' && matrix.goarch == 'amd64'
|
if: matrix.goos == 'linux' && matrix.goarch == 'amd64'
|
||||||
|
env:
|
||||||
|
TAG: ${{ gitea.ref_name }}
|
||||||
|
BIN: bin/gasm-${{ steps.version.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||||
run: |
|
run: |
|
||||||
chmod +x bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
perl -e '
|
||||||
./bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }} --version
|
my $want = $ENV{TAG} // die qq{ERROR: no tag\n};
|
||||||
|
open(my $bin, q{-|}, $ENV{BIN}, q{--version}) or die qq{$ENV{BIN}: $!};
|
||||||
|
my $got = <$bin>;
|
||||||
|
close($bin);
|
||||||
|
$got = defined $got ? $got : q{};
|
||||||
|
chomp $got;
|
||||||
|
index($got, $want) >= 0
|
||||||
|
or die qq{ERROR: the binary printed "$got", which does not contain $want. Version control was disabled, so there is no recorded version.\n};
|
||||||
|
index($got, q{+dirty}) < 0
|
||||||
|
or die qq{ERROR: the binary printed "$got". The tree was dirty at build time, which means the checkout was not the tag, or the build artefacts are not ignored.\n};
|
||||||
|
print qq{$ENV{BIN} reports $got\n};
|
||||||
|
'
|
||||||
|
|
||||||
release:
|
release:
|
||||||
runs-on: fedora
|
runs-on: fedora
|
||||||
|
timeout-minutes: 15
|
||||||
needs: build
|
needs: build
|
||||||
permissions:
|
permissions:
|
||||||
|
# contents: read is required for the checkout: a job that declares any
|
||||||
|
# permissions gets a token scoped to exactly those, and releases: write
|
||||||
|
# alone leaves the fetch with no read access, which Gitea answers with
|
||||||
|
# a 404 "Repository not found". Verified on the instance 2026-09-16.
|
||||||
|
contents: read
|
||||||
releases: write
|
releases: write
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v7
|
- uses: actions/checkout@v7
|
||||||
@@ -77,81 +226,165 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
path: dist
|
path: dist
|
||||||
|
|
||||||
- name: Extract CHANGELOG section
|
- name: Install Perl
|
||||||
|
run: dnf install -y perl
|
||||||
|
|
||||||
|
- name: Extract the CHANGELOG section
|
||||||
env:
|
env:
|
||||||
VERSION: ${{ gitea.ref_name }}
|
VERSION: ${{ gitea.ref_name }}
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
# Each step derives what it needs from the tag, so no value has to travel between
|
||||||
VERSION_NO_V="${VERSION#v}"
|
# jobs.
|
||||||
|
perl -e '
|
||||||
|
my $v = $ENV{VERSION} // q{};
|
||||||
|
$v =~ s{^v}{};
|
||||||
|
open(my $vout, q{>}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
|
||||||
|
print $vout $v;
|
||||||
|
close($vout);
|
||||||
|
open(my $in, q{<}, q{CHANGELOG.md}) or die qq{CHANGELOG.md: $!};
|
||||||
|
my @lines = <$in>;
|
||||||
|
close($in);
|
||||||
|
my ($start, $end) = (-1, scalar @lines);
|
||||||
|
for my $i (0 .. $#lines) {
|
||||||
|
if ($start < 0) { $start = $i if $lines[$i] =~ m{^##\s+\[\Q$v\E\]} }
|
||||||
|
elsif ($lines[$i] =~ m{^##\s+\[}) { $end = $i; last }
|
||||||
|
}
|
||||||
|
$start >= 0 or die qq{ERROR: no CHANGELOG section for $v, expected a heading like: ## [$v] - YYYY-MM-DD\n};
|
||||||
|
my @body = grep { m{\S} } @lines[$start + 1 .. $end - 1];
|
||||||
|
@body or die qq{ERROR: the CHANGELOG section for $v is empty\n};
|
||||||
|
open(my $out, q{>}, q{release-body.md}) or die qq{release-body.md: $!};
|
||||||
|
print $out @body;
|
||||||
|
close($out);
|
||||||
|
printf qq{notes for %s: %d lines\n}, $v, scalar @body;
|
||||||
|
'
|
||||||
|
|
||||||
sed -n "/^## \[${VERSION_NO_V}\] /,/^## \[/p" CHANGELOG.md \
|
- name: Build the release request
|
||||||
| sed '$d' \
|
run: |
|
||||||
| tail -n +2 \
|
perl -e '
|
||||||
> release-body.md
|
open(my $vin, q{<}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
|
||||||
|
my $v = <$vin>;
|
||||||
|
close($vin);
|
||||||
|
chomp $v;
|
||||||
|
open(my $in, q{<:raw}, q{release-body.md}) or die qq{release-body.md: $!};
|
||||||
|
my $body = do { local $/; <$in> };
|
||||||
|
close($in);
|
||||||
|
# Byte-oriented escaping: JSON is UTF-8, so non-ASCII passes through and only the
|
||||||
|
# characters JSON forbids are rewritten.
|
||||||
|
$body =~ s/([\\"])/\\$1/g;
|
||||||
|
$body =~ s/\t/\\t/g;
|
||||||
|
$body =~ s/\r//g;
|
||||||
|
$body =~ s/\n/\\n/g;
|
||||||
|
$body =~ s/([\x00-\x08\x0b\x0c\x0e-\x1f])/sprintf(q{\u%04x}, ord($1))/ge;
|
||||||
|
my $json = sprintf(qq{{"tag_name":"v%s","name":"v%s","body":"%s","draft":false,"prerelease":false}}, $v, $v, $body);
|
||||||
|
open(my $out, q{>}, q{release.json}) or die qq{release.json: $!};
|
||||||
|
print $out $json;
|
||||||
|
close($out);
|
||||||
|
print qq{release.json written for v$v\n};
|
||||||
|
'
|
||||||
|
|
||||||
if [ ! -s release-body.md ]; then
|
- name: Create the release
|
||||||
echo "ERROR: no CHANGELOG section found for ${VERSION_NO_V}"
|
|
||||||
echo "Expected a heading like: ## [${VERSION_NO_V}] — YYYY-MM-DD"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
- name: Create release
|
|
||||||
env:
|
env:
|
||||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||||
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
||||||
GITEA_REPOSITORY: ${{ gitea.repository }}
|
GITEA_REPOSITORY: ${{ gitea.repository }}
|
||||||
GITEA_REF_NAME: ${{ gitea.ref_name }}
|
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
perl -e '
|
||||||
|
my @cmd = (q{curl}, q{-sS}, q{-o}, q{response.json}, q{-w}, q{%{http_code}},
|
||||||
BODY=$(sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' -e 's/\t/\\t/g' -e 's/\r//g' release-body.md | sed ':a;N;$!ba;s/\n/\\n/g')
|
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
|
||||||
BODY="\"${BODY}\""
|
q{-H}, q{Content-Type: application/json},
|
||||||
|
q{-X}, q{POST},
|
||||||
response=$(curl -sS -w '\n%{http_code}' \
|
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases},
|
||||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
q{--data-binary}, q{@release.json});
|
||||||
-H "Content-Type: application/json" \
|
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
|
||||||
-X POST \
|
my $code = <$curl>;
|
||||||
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases" \
|
my $ok = close($curl);
|
||||||
-d "{\"tag_name\":\"${GITEA_REF_NAME}\",\"name\":\"${GITEA_REF_NAME}\",\"body\":${BODY},\"draft\":false,\"prerelease\":false}")
|
my $exit = $? >> 8;
|
||||||
|
$code = defined $code ? $code : q{};
|
||||||
http_code=$(echo "$response" | tail -1)
|
$ok or die qq{ERROR: curl failed (exit $exit) calling $ENV{GITEA_SERVER_URL}\n};
|
||||||
payload=$(echo "$response" | sed '$d')
|
open(my $r, q{<:raw}, q{response.json}) or die qq{response.json: $!};
|
||||||
|
my $body = do { local $/; <$r> };
|
||||||
echo "HTTP ${http_code}"
|
close($r);
|
||||||
if [ "$http_code" != "201" ]; then
|
$code eq q{201} or die qq{ERROR: the release was not created, HTTP $code: $body\n};
|
||||||
echo "Failed to create release: ${payload}"
|
$body =~ m{"id"\s*:\s*([0-9]+)} or die qq{ERROR: no release id in the response: $body\n};
|
||||||
exit 1
|
open(my $o, q{>}, q{release-id.txt}) or die qq{release-id.txt: $!};
|
||||||
fi
|
print $o $1;
|
||||||
|
close($o);
|
||||||
RELEASE_ID=$(echo "$payload" | grep -oE '"id"[[:space:]]*:[[:space:]]*[0-9]+' | head -1 | grep -oE '[0-9]+')
|
print qq{release id $1\n};
|
||||||
echo "Created release ID=${RELEASE_ID}"
|
'
|
||||||
printf '%s' "${RELEASE_ID}" > release-id.txt
|
|
||||||
|
|
||||||
- name: Upload assets
|
- name: Upload assets
|
||||||
env:
|
env:
|
||||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||||
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
||||||
GITEA_REPOSITORY: ${{ gitea.repository }}
|
GITEA_REPOSITORY: ${{ gitea.repository }}
|
||||||
GITEA_REF_NAME: ${{ gitea.ref_name }}
|
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
perl -e '
|
||||||
RELEASE_ID=$(cat release-id.txt)
|
open(my $f, q{<}, q{release-id.txt}) or die qq{release-id.txt: $!};
|
||||||
|
my $id = <$f>;
|
||||||
for binary in dist/gasm-*/gasm-*; do
|
close($f);
|
||||||
[ -f "$binary" ] || continue
|
chomp $id;
|
||||||
fname=$(basename "$binary")
|
my @files = grep { -f $_ } glob(q{dist/*/*});
|
||||||
echo "Uploading ${fname}..."
|
@files or die qq{ERROR: no assets under dist/\n};
|
||||||
http_code=$(curl -sS -o /dev/null -w '%{http_code}' \
|
# A file that arrived empty from the artifact step would be uploaded as an
|
||||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
# empty attachment, every status would still be 201, and the run would go
|
||||||
-H "Content-Type: application/octet-stream" \
|
# green over a release nobody can install. Refuse it here, before the
|
||||||
-X POST \
|
# upload, and verify what was stored afterwards.
|
||||||
--data-binary "@${binary}" \
|
my %size;
|
||||||
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases/${RELEASE_ID}/assets?name=${fname}")
|
for my $path (@files) {
|
||||||
echo " HTTP ${http_code}"
|
my $n = -s $path // 0;
|
||||||
if [ "$http_code" != "201" ]; then
|
(my $name = $path) =~ s{.*/}{};
|
||||||
echo "Failed to upload ${fname}"
|
$n > 0 or die qq{ERROR: $path is empty, so there is nothing to upload\n};
|
||||||
exit 1
|
$size{$name} = $n;
|
||||||
fi
|
}
|
||||||
done
|
my $bad = 0;
|
||||||
|
for my $path (@files) {
|
||||||
echo "Release ${GITEA_REF_NAME} is live."
|
(my $name = $path) =~ s{.*/}{};
|
||||||
|
my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}},
|
||||||
|
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
|
||||||
|
q{-H}, q{Content-Type: application/octet-stream},
|
||||||
|
# The @ must not sit inside a qq{} string: there it starts an
|
||||||
|
# array interpolation and the upload body collapses to empty,
|
||||||
|
# which Gitea stores as a 201-created zero-byte attachment.
|
||||||
|
q{-X}, q{POST}, q{--data-binary}, q{@} . $path,
|
||||||
|
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name});
|
||||||
|
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
|
||||||
|
my $code = <$curl>;
|
||||||
|
my $ok = close($curl);
|
||||||
|
my $exit = $? >> 8;
|
||||||
|
$code = defined $code ? $code : q{};
|
||||||
|
unless ($ok) {
|
||||||
|
printf qq{%s: curl failed (exit %d)\n}, $name, $exit;
|
||||||
|
$bad = 1;
|
||||||
|
next;
|
||||||
|
}
|
||||||
|
printf qq{%s: HTTP %s\n}, $name, $code;
|
||||||
|
$bad = 1 if $code ne q{201};
|
||||||
|
}
|
||||||
|
# Read every asset back through the release download route and require the
|
||||||
|
# served length to be the file that was sent: stored but empty is a broken
|
||||||
|
# release however green the run looks.
|
||||||
|
open(my $v, q{<}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
|
||||||
|
my $v = <$v>;
|
||||||
|
close($v);
|
||||||
|
chomp $v;
|
||||||
|
for my $name (sort keys %size) {
|
||||||
|
my $url = qq{$ENV{GITEA_SERVER_URL}/$ENV{GITEA_REPOSITORY}/releases/download/v$v/$name};
|
||||||
|
my @head = (q{curl}, q{-sS}, q{-I}, q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}}, $url);
|
||||||
|
open(my $h, q{-|}, @head) or die qq{curl: $!};
|
||||||
|
my $len;
|
||||||
|
my $status;
|
||||||
|
while (my $l = <$h>) {
|
||||||
|
$status = $1 if $l =~ m{^HTTP/\S+\s+(\d+)};
|
||||||
|
$len = $1 if $l =~ m{^content-length:\s*(\d+)}i;
|
||||||
|
}
|
||||||
|
my $ok = close($h);
|
||||||
|
$len = defined $len ? $len : 0;
|
||||||
|
if (!$ok || $status != 200 || $len != $size{$name}) {
|
||||||
|
printf qq{ERROR: %s serves %s bytes, expected %d\n}, $name, $len, $size{$name};
|
||||||
|
$bad = 1;
|
||||||
|
next;
|
||||||
|
}
|
||||||
|
printf qq{%s: serves %d bytes\n}, $name, $len;
|
||||||
|
}
|
||||||
|
exit($bad ? 1 : 0);
|
||||||
|
'
|
||||||
|
|||||||
+98
-76
@@ -1,4 +1,22 @@
|
|||||||
# Test — gasm-devkit. Runs on push and pull request to development.
|
# Test, Go. Push and pull request to development. Never on main.
|
||||||
|
#
|
||||||
|
# The gates are the ones the justfile's `gates` recipe runs, minus race: the shared
|
||||||
|
# runner box cannot afford the race detector on every push, so it lives in race.yml.
|
||||||
|
# The box is one core and 2 GB beside Gitea, so parallelism is bounded on purpose and
|
||||||
|
# everything runs in one job. Extra jobs would duplicate the checkout, the Go setup and
|
||||||
|
# the dependency download three times without buying any parallelism.
|
||||||
|
#
|
||||||
|
# The budget is part of the contract: a push run is fast and light, about two minutes,
|
||||||
|
# and nothing that cannot run natively on the runner belongs here. The FreeBSD compile
|
||||||
|
# gates live in freebsd.yml behind workflow_dispatch for that reason; the GOOBJ link
|
||||||
|
# parity campaign is an opt-in local verification (just link-parity).
|
||||||
|
#
|
||||||
|
# Every step is one command, so the step that fails is the gate that failed, and no shell
|
||||||
|
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
|
||||||
|
# not Python: Perl behaves the same on both runner images, there is no bashism to trip over
|
||||||
|
# on ash, and it is one language instead of two. The Perl uses builtins only, because
|
||||||
|
# Fedora packages the Perl modules separately and nothing beyond `perl` itself may be
|
||||||
|
# assumed present.
|
||||||
name: Test
|
name: Test
|
||||||
|
|
||||||
on:
|
on:
|
||||||
@@ -7,90 +25,94 @@ on:
|
|||||||
pull_request:
|
pull_request:
|
||||||
branches: [development]
|
branches: [development]
|
||||||
|
|
||||||
|
env:
|
||||||
|
# One core: parallelism buys no speed here and costs memory the box does not have.
|
||||||
|
GOFLAGS: -p=1
|
||||||
|
GOMAXPROCS: "2"
|
||||||
|
|
||||||
|
# A superseded run of the same ref is cancelled instead of queueing behind one that
|
||||||
|
# no longer matters. Verified on Gitea 1.27.1 on 2026-09-17: a queued run whose ref
|
||||||
|
# moved on is cancelled before it ever reaches the runner, while a run already
|
||||||
|
# dispatched there runs to completion.
|
||||||
|
concurrency:
|
||||||
|
group: ${{ gitea.workflow }}-${{ gitea.ref }}
|
||||||
|
cancel-in-progress: true
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
vet:
|
|
||||||
runs-on: fedora
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v7
|
|
||||||
|
|
||||||
- uses: actions/setup-go@v6
|
|
||||||
with:
|
|
||||||
go-version: "1.27"
|
|
||||||
|
|
||||||
- name: Download dependencies
|
|
||||||
run: go mod download
|
|
||||||
|
|
||||||
- name: gofmt
|
|
||||||
run: |
|
|
||||||
set -euo pipefail
|
|
||||||
unformatted=$(gofmt -l .)
|
|
||||||
if [ -n "$unformatted" ]; then
|
|
||||||
echo "These files need gofmt:"
|
|
||||||
echo "$unformatted"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
- name: go vet
|
|
||||||
run: go vet ./...
|
|
||||||
|
|
||||||
test:
|
test:
|
||||||
runs-on: fedora
|
runs-on: fedora
|
||||||
needs: vet
|
timeout-minutes: 10
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v7
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
- uses: actions/setup-go@v6
|
- uses: actions/setup-go@v6
|
||||||
with:
|
with:
|
||||||
go-version: "1.27"
|
# The module is the source of truth for the version, so it cannot drift.
|
||||||
|
go-version-file: go.mod
|
||||||
|
cache: true
|
||||||
|
|
||||||
- name: Download dependencies
|
# No Install Perl step: the fedora image carries perl (verified by run
|
||||||
run: go mod download
|
# 76: the install degraded into a package upgrade costing ~50 s), and a
|
||||||
|
# dnf on the push path is network work the budget does not need.
|
||||||
- name: Install gcc
|
|
||||||
run: dnf install -y gcc
|
|
||||||
|
|
||||||
- name: go test -race
|
|
||||||
run: go test -race -count=1 ./...
|
|
||||||
|
|
||||||
- name: Coverage gate — 80 % minimum
|
|
||||||
run: |
|
|
||||||
set -euo pipefail
|
|
||||||
# Exclude packages inherently untestable without hardware:
|
|
||||||
# debug — interactive ptrace, requires a live process
|
|
||||||
# cmd/gasm — CLI glue, covered by integration tests
|
|
||||||
go test -coverprofile=coverage.out \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/arch \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/asm \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/ast \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/format \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/lexer \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/lint \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/lsp \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/parser \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/token \
|
|
||||||
sourcedock.dev/petrbalvin/gasm-devkit/verify
|
|
||||||
coverage=$(go tool cover -func=coverage.out | awk '/^total:/ { gsub("%", "", $3); print $3 }')
|
|
||||||
echo "Total coverage: ${coverage}%"
|
|
||||||
if awk -v c="$coverage" 'BEGIN { exit !(c+0 < 80) }'; then
|
|
||||||
echo "ERROR: coverage ${coverage}% is below the 80% threshold"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
build:
|
|
||||||
runs-on: fedora
|
|
||||||
needs: test
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v7
|
|
||||||
|
|
||||||
- uses: actions/setup-go@v6
|
|
||||||
with:
|
|
||||||
go-version: "1.27"
|
|
||||||
|
|
||||||
- name: Download dependencies
|
|
||||||
run: go mod download
|
|
||||||
|
|
||||||
|
# The steps follow the `gates` order of the justfile contract: build, format,
|
||||||
|
# vet, test. The vet gate is go vet and go fix -diff, two steps here.
|
||||||
- name: Build
|
- name: Build
|
||||||
run: go build -ldflags="-s -w" -o bin/gasm ./cmd/gasm
|
run: go build ./...
|
||||||
|
|
||||||
- name: Smoke test
|
- name: Format
|
||||||
run: ./bin/gasm --version
|
run: |
|
||||||
|
perl -e '
|
||||||
|
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
|
||||||
|
my @bad = <$g>;
|
||||||
|
close($g);
|
||||||
|
print @bad;
|
||||||
|
exit(@bad ? 1 : 0);
|
||||||
|
'
|
||||||
|
|
||||||
|
- name: Vet
|
||||||
|
run: go vet ./...
|
||||||
|
|
||||||
|
- name: Modernise
|
||||||
|
# Exits non-zero when it has something to rewrite, so it needs no output capture.
|
||||||
|
run: go fix -diff ./...
|
||||||
|
|
||||||
|
- name: Tests
|
||||||
|
# The suite must be fast: a push pipeline that cannot finish in a few minutes moves
|
||||||
|
# its heavy part behind a dispatch. The inner timeout matches the job's, so a
|
||||||
|
# hanging test reports its own goroutine dump rather than a silent job kill.
|
||||||
|
# The pattern is `packages` in the project's justfile: the logic packages, since a
|
||||||
|
# thin cmd/ would drag the total under the floor. release.yml runs the same
|
||||||
|
# command, so the floor is the same number everywhere. ./verify/... carries the
|
||||||
|
# live oracle-parity comparison against `go tool asm` (the TestGroundTruth
|
||||||
|
# suites); the runner's Go setup provides both the tool and GOROOT.
|
||||||
|
# -short skips the deliberate-run categories inside the suites (the live
|
||||||
|
# ptrace sessions above all): they need the machine to themselves and a
|
||||||
|
# starved single-core runner turns each into a timeout the budget cannot
|
||||||
|
# carry. The local `just test` gate runs everything, in full.
|
||||||
|
run: go test -short -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
|
||||||
|
|
||||||
|
- name: Tests outside the coverage set
|
||||||
|
# The CLI and the debugger sit outside `packages` because a thin main and a
|
||||||
|
# ptrace-bound package pull the total under the floor, but their tests guard
|
||||||
|
# shipped surfaces: the command exit codes, the manual pages against the
|
||||||
|
# binary's own help, and the debugger's architecture-neutral units. They run
|
||||||
|
# here so the floor stays a product measure and nothing is left untested.
|
||||||
|
# -short skips the debugger's live ptrace sessions, the deliberate-run
|
||||||
|
# category the runner cannot starve-proof. The live go-tool-asm oracle
|
||||||
|
# comparison (TestGroundTruth in ./verify/...) runs inside the coverage
|
||||||
|
# sweep above; it is not re-run as its own step, because every second on
|
||||||
|
# this box is budget.
|
||||||
|
run: go test -short -count=1 -timeout 10m ./cmd/... ./debug/...
|
||||||
|
|
||||||
|
- name: Coverage floor
|
||||||
|
run: |
|
||||||
|
perl -e '
|
||||||
|
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
|
||||||
|
my $total;
|
||||||
|
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
|
||||||
|
close($c);
|
||||||
|
die qq{no total line in coverage.out\n} unless defined $total;
|
||||||
|
printf qq{Total coverage: %s%%\n}, $total;
|
||||||
|
exit($total < 80 ? 1 : 0);
|
||||||
|
'
|
||||||
|
|||||||
+3
-15
@@ -1,25 +1,13 @@
|
|||||||
# Metadata (always first, per repo convention)
|
|
||||||
.idea/
|
.idea/
|
||||||
.zcode/
|
.zcode/
|
||||||
.qwen/
|
|
||||||
.mimocode/
|
|
||||||
|
|
||||||
# Binaries
|
# Build output
|
||||||
/gasm
|
|
||||||
/bin/
|
/bin/
|
||||||
*.exe
|
/gasm
|
||||||
|
|
||||||
# Test and coverage artefacts
|
|
||||||
coverage.out
|
coverage.out
|
||||||
*.test
|
*.test
|
||||||
|
|
||||||
# Crash dumps
|
# Crash dumps from the emulator runs
|
||||||
core
|
core
|
||||||
core.*
|
core.*
|
||||||
*.core
|
*.core
|
||||||
|
|
||||||
# Scratch / temporary work
|
|
||||||
_scratch/
|
|
||||||
|
|
||||||
# ZCode workspace
|
|
||||||
.zcode
|
|
||||||
|
|||||||
+744
-150
File diff suppressed because it is too large
Load Diff
+107
-80
@@ -1,107 +1,134 @@
|
|||||||
# Contributing to gasm-devkit
|
# Contributing
|
||||||
|
|
||||||
Thanks for contributing to gasm-devkit.
|
Contributions to **gasm-sdk** are governed by the Contributor terms
|
||||||
|
below; submitting one means you accept them.
|
||||||
|
|
||||||
|
## Contributor terms
|
||||||
|
|
||||||
|
1. This project belongs to its owner alone. The owner decides what is
|
||||||
|
accepted, in what form and when; the decision is final and needs no
|
||||||
|
justification.
|
||||||
|
2. By submitting a contribution you assign to Petr Balvín
|
||||||
|
<opensource@petrbalvin.org> all present and future copyright and
|
||||||
|
related rights in it, worldwide, for the full term of the rights,
|
||||||
|
with the right to relicense and sublicense without restriction,
|
||||||
|
including under proprietary terms.
|
||||||
|
3. Where that assignment is not effective, it counts as a perpetual,
|
||||||
|
irrevocable, royalty-free licence with the same scope.
|
||||||
|
4. To the fullest extent permitted by law, you waive any right of
|
||||||
|
attribution and integrity in the contribution. The project names no
|
||||||
|
contributors and keeps no credits list.
|
||||||
|
5. By submitting you represent that the work is yours and that you
|
||||||
|
hold the rights to assign it as above.
|
||||||
|
|
||||||
## Development setup
|
## Development setup
|
||||||
|
|
||||||
Requirements: Go 1.27 or later, the [just](https://github.com/casey/just)
|
Requirements: Go 1.27.1, the exact version the `go` directive in `go.mod`
|
||||||
command runner, and a Linux host on amd64, arm64, riscv64 or loong64.
|
declares, [just](https://github.com/casey/just) for the recipes, and a C
|
||||||
|
compiler (gcc), because `just gates` includes `just race` and the race
|
||||||
|
detector needs cgo.
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
|
git clone https://sourcedock.dev/petrbalvin/gasm-sdk.git
|
||||||
cd gasm-devkit
|
cd gasm-sdk
|
||||||
just install # download module dependencies
|
just build
|
||||||
just build # go vet + gofmt check
|
just gates
|
||||||
just test # full suite, race detector, 80 % coverage gate
|
|
||||||
```
|
```
|
||||||
|
|
||||||
## Workflow
|
## Workflow
|
||||||
|
|
||||||
1. Branch from `development`; never commit directly to `main` (`main` is
|
1. Branch from `development`. Never commit directly to `main`, which is release-only.
|
||||||
release-only: merge from `development`, then tag).
|
2. Commit in [Conventional Commits](https://www.conventionalcommits.org/) form:
|
||||||
2. Commit with [Conventional Commits](https://www.conventionalcommits.org/):
|
`type(scope): description`, subject line only, imperative mood, lowercase after the
|
||||||
`type(scope): description`: subject line only, imperative mood,
|
colon, no trailing full stop. Allowed types: `feat`, `fix`, `docs`, `style`,
|
||||||
lowercase after the colon, no trailing dot. Allowed types: `feat`,
|
`refactor`, `perf`, `test`, `chore`, `ci`, `build`, `revert`.
|
||||||
`fix`, `docs`, `style`, `refactor`, `perf`, `test`, `chore`, `ci`,
|
3. One logical change per commit. A refactor, a behaviour change and a formatting pass
|
||||||
`build`, `revert`. The only line after the subject is the trailer:
|
are three commits, never one.
|
||||||
`Assisted-by: <model-name>`. No `Co-Authored-By`, no `Signed-off-by`,
|
4. Record every user-visible change in `CHANGELOG.md` under `## [development]`.
|
||||||
no other trailers.
|
5. Add or update tests. Coverage stays at 80 percent or more; it is a hard gate.
|
||||||
3. Record every user-visible change in `CHANGELOG.md` under
|
6. Update the documentation when the public API, the configuration or the behaviour
|
||||||
`## [development]` (categories: Added, Changed, Fixed, Removed,
|
changes.
|
||||||
Security).
|
7. Open a pull request against `development`.
|
||||||
4. Add or update tests; coverage must stay **at or above 80 %** (hard
|
|
||||||
gate, enforced by CI).
|
|
||||||
5. Update the documentation when behaviour, flags or the public surface
|
|
||||||
change.
|
|
||||||
6. Open a pull request against `development`.
|
|
||||||
|
|
||||||
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`;
|
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The release
|
||||||
CI builds and publishes the binaries for all four architectures.
|
workflow builds the assets and publishes the release and its notes.
|
||||||
|
|
||||||
## Code style
|
## Code style
|
||||||
|
|
||||||
`gofmt` and `go vet` via `just fmt` / `just build`; both must pass with
|
`gofmt` and `go vet` run through `just fmt` and `just vet`, with zero diff and zero
|
||||||
zero output; `go fix -diff ./...` must report nothing on touched packages.
|
warnings tolerated. `just vet` is two gates, `go vet ./...` and `go fix -diff ./...`,
|
||||||
|
so the modernisation rewrites are enforced too. `just gates` is the definition of done in
|
||||||
|
one command, and the recipe file names what it contains. Errors are checked explicitly,
|
||||||
|
wrapped as `fmt.Errorf("context: %w", err)`, and nothing panics outside `main`. The
|
||||||
|
recipe file holds the commands, and the language and standard-library surface is the one
|
||||||
|
the `go` directive in `go.mod` pins.
|
||||||
|
|
||||||
- Standard library only in production code; `golang.org/x/arch` is used
|
- `golang.org/x/arch` is the one module dependency, and it is linked into the binary:
|
||||||
in tests only (round-trip decoding) and is never linked into the `gasm`
|
`gasm dis` and the debugger's listings decode through it. Everything else is the
|
||||||
binary.
|
standard library.
|
||||||
- No cgo, no C, no external toolchains at runtime.
|
- No cgo and no C. The standalone encoder paths (`gasm asm --format raw` and `--format
|
||||||
- Explicit `if err != nil`; errors wrapped with
|
elf`) need no Go installation; `gasm verify --ground-truth`, `gasm verify --fuzz`,
|
||||||
`fmt.Errorf("context: %w", err)`; no panics outside `main`.
|
`gasm audit-instructions` and `gasm asm --format goobj` resolve through the installed
|
||||||
- The parser, lexer and formatter are hand-written; the `arch` instruction
|
Go toolchain.
|
||||||
tables are generated only via `_gen/gen.go` (`just gen`), never edited.
|
- The parser, lexer and formatter are hand-written; the `arch` instruction tables are
|
||||||
|
generated only by `_gen/gen.go` (`just gen`) and never edited by hand.
|
||||||
|
- Assembly committed to the repository goes through `gasm fmt` and `gasm lint`, so a
|
||||||
|
`.s` file that `gasm fmt -l .` lists is unfinished.
|
||||||
|
|
||||||
## Running a single test
|
New source files open with the project's two-line licence header, whose SPDX
|
||||||
|
identifier matches `LICENSE`. Configuration files, workflows and dotfiles do not carry
|
||||||
|
it.
|
||||||
|
|
||||||
```sh
|
## AI contribution policy
|
||||||
go test -run TestVexGroundTruth ./asm/
|
|
||||||
go test -run TestGroundTruthBasic ./verify/
|
|
||||||
go test -run TestGOObjectLinkAndRun ./asm/
|
|
||||||
go test -run TestFuzzWideCopy ./verify/
|
|
||||||
```
|
|
||||||
|
|
||||||
The interactive debugger (`gasm debug`) requires a compiled binary on
|
AI tools are welcome as productivity aids and are a normal part of modern software
|
||||||
`$PATH`; `go run` does not work for the traced child process. Install
|
development. What matters is that the contribution stays understandable, reviewable and
|
||||||
first with `just install-bin`.
|
genuinely useful.
|
||||||
|
|
||||||
## CI (Gitea Actions)
|
- **Disclose the assistance.** If AI helped draft any part of a commit, issue, pull
|
||||||
|
request or review, say so.
|
||||||
|
- **Commit messages carry exactly one trailer**, as a git trailer on the line after a
|
||||||
|
blank line that closes the subject:
|
||||||
|
|
||||||
Workflows live in `.gitea/workflows/` and run on self-hosted runners:
|
```
|
||||||
|
Assisted-by: MODEL
|
||||||
|
```
|
||||||
|
|
||||||
|
Name the model that did the work, spelled the way its maker spells it, for example
|
||||||
|
`GLM 5.3`, `DeepSeek V4.1 Flash` or `Qwen 3.8 Flash`. No `Co-Authored-By`, no `Signed-off-by`,
|
||||||
|
no other trailers, and no prose: the trailer is the disclosure.
|
||||||
|
- **Issues and pull requests** attribute the assistance in a comment, for example
|
||||||
|
`_Assisted-by: GLM 5.3_`. It does not belong in the pull request description.
|
||||||
|
- **Take responsibility.** You are accountable for the accuracy, completeness and
|
||||||
|
intent of everything you submit, whether or not AI produced it.
|
||||||
|
- **Review before marking ready.** Read the diff carefully, run it locally, and add the
|
||||||
|
tests it needs. Do not mark a pull request ready until you can defend every change in
|
||||||
|
it.
|
||||||
|
- **Quality over quantity.** Contributions that look like un-reviewed output, or whose
|
||||||
|
author cannot engage substantively during review, may be closed.
|
||||||
|
- **Preferred models.** Prefer open-weight models with transparent training data and
|
||||||
|
minimal output filtering.
|
||||||
|
|
||||||
|
AI assists. It does not replace judgement.
|
||||||
|
|
||||||
|
## Continuous integration
|
||||||
|
|
||||||
|
Workflows live in `.gitea/workflows/` and run on the project's own runners:
|
||||||
|
|
||||||
| Workflow | Trigger | What it does |
|
| Workflow | Trigger | What it does |
|
||||||
|----------|---------|--------------|
|
|---|---|---|
|
||||||
| Test | push / PR to `development` | gofmt check, `go vet`, `go test -race`, 80 % coverage gate |
|
| Test | push or pull request to `development` | build, format check, vet, modernisation, the test suite with the coverage floor, the CLI and debugger tests outside the profile, then the oracle-parity rerun against `go tool asm` |
|
||||||
| Release | tag `v*` | cross-compiles binaries for linux/{amd64,arm64,riscv64,loong64} and publishes the Gitea release |
|
| Release | a `v*` tag | the same gates as Test minus the oracle-parity step, then the matrix build, the version smoke test and the release itself; the race detector runs locally in `just gates` before the tag is cut |
|
||||||
|
|
||||||
The Definition of Done (`just build` + `just test` + `just fmt`) must
|
The local equivalent is `just gates`, which is the same set plus the race detector. The
|
||||||
still pass locally before pushing.
|
race detector also has its own workflow, dispatched by hand; it never runs on a push or a
|
||||||
|
tag, where it would double the time and the memory a shared runner cannot spare.
|
||||||
## AI Contribution Policy
|
|
||||||
|
|
||||||
AI tools are welcome as productivity aids. What matters is that
|
|
||||||
contributions remain understandable, reviewable, and genuinely useful.
|
|
||||||
|
|
||||||
- **Disclose AI use.** If you used AI to draft or generate any part of a
|
|
||||||
commit, issue, pull request, or code review, say so clearly.
|
|
||||||
- **Commit messages:** end every commit with exactly one trailer:
|
|
||||||
`Assisted-by: <model-name>` (e.g. `Assisted-by: GLM 5.3`).
|
|
||||||
- **Pull requests and issues:** attribute AI assistance in one trailing
|
|
||||||
line, e.g. `_Assisted-by: GLM 5.3_`. Do not paste it into the PR
|
|
||||||
description as a section.
|
|
||||||
- **Take responsibility.** You remain accountable for the accuracy,
|
|
||||||
completeness, and intent of everything you submit.
|
|
||||||
- **Review before marking ready.** Read AI-generated diffs carefully, run
|
|
||||||
them locally, and add or update tests where appropriate.
|
|
||||||
- **Preferred models.** Prefer open-weight models with transparent
|
|
||||||
training data: **GLM**, **DeepSeek**, and **MiMo**.
|
|
||||||
|
|
||||||
## Reporting bugs
|
## Reporting bugs
|
||||||
|
|
||||||
Open an issue at
|
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-sdk/issues` with the
|
||||||
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit/issues)
|
version, the operating system and architecture, the exact command, the full output,
|
||||||
with the version (`gasm --version`), OS and architecture, the exact
|
and the expected against the actual behaviour.
|
||||||
command, the full output, and the expected versus actual behaviour.
|
|
||||||
|
|
||||||
**Security issues:** email **opensource@petrbalvin.org** instead of opening
|
**Security issues do not go in the issue tracker.** Report them as
|
||||||
a public issue.
|
[SECURITY.md](SECURITY.md) describes, to **opensource@petrbalvin.org**.
|
||||||
|
|||||||
@@ -1,13 +1,65 @@
|
|||||||
# gasm-devkit
|
# GAsm: Software Development Kit for Plan 9 Assembly
|
||||||
|
|
||||||
Developer tooling for **GAsm**, Go's built-in Plan 9 assembler.
|
> **Warning: this is an experiment.** gasm-sdk is under active
|
||||||
|
> development and is not stable. The version is 0.x.x: commands, flags,
|
||||||
|
> output formats and behaviour can change without warning at any time.
|
||||||
|
> A 1.0.0 release is light years away. Nothing in this document is a
|
||||||
|
> stability promise. For all of that, this is not a paper project: gasm
|
||||||
|
> is already in active use and is tested on real assembly work. Only
|
||||||
|
> amd64 is validated on real hardware; the other three architectures run
|
||||||
|
> under emulation ([Validation status](#validation-status)).
|
||||||
|
|
||||||
Go ships an assembler but no tooling for it: there is no syntax highlighting,
|
**GAsm** is Go's Plan 9 assembler, and Go ships it without tooling:
|
||||||
no autocomplete, no linter, no static analyser, no formatter, no standalone
|
there is no formatter, no linter and no debugger for `.s` files, and no
|
||||||
assembler and no debugger for `.s` files. Developers write assembly blind,
|
assembler that works without a Go installation. Developers write
|
||||||
validate it by benchmark, and debug it by print statement. gasm-devkit is the
|
assembly blind, validate it by benchmark, and debug it by print
|
||||||
missing toolkit: a single, self-contained binary, `gasm`, that brings proper
|
statement. gasm-sdk is the missing toolkit: a single, self-contained
|
||||||
developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
|
binary, `gasm`, that serves both purposes.
|
||||||
|
|
||||||
|
- **Help develop Plan 9 assembly.** Formatting, linting, disassembly,
|
||||||
|
dynamic verification, a source-level debugger and a language server,
|
||||||
|
for `.s` files in Go programs.
|
||||||
|
- **Use Plan 9 assembly outside the Go toolchain.** `gasm asm` encodes
|
||||||
|
on its own and writes raw images or linkable ELF objects with DWARF5
|
||||||
|
debug sections, with no Go installation in the loop; the Go
|
||||||
|
toolchain's own GOOBJ format, which `go build` consumes in place of
|
||||||
|
the toolchain's output, needs the installed toolchain.
|
||||||
|
|
||||||
|
## Why Plan 9 assembly
|
||||||
|
|
||||||
|
Plan 9 assembly is the quiet triumph of the field. One syntax across
|
||||||
|
every architecture Go builds for: the same source-first operand order,
|
||||||
|
the same four pseudo-registers, the same frame convention, whether the
|
||||||
|
target is x86, ARM, RISC-V or LoongArch. Learn it once and you can
|
||||||
|
read a kernel on any of them.
|
||||||
|
|
||||||
|
Compare the alternatives. Intel syntax and AT&T syntax disagree on the
|
||||||
|
one question every instruction answers, which operand is the source
|
||||||
|
and which is the destination, so half the world writes it one way,
|
||||||
|
half the other, and every assembly programmer carries both in their
|
||||||
|
head forever. GNU as settles the argument with directives that switch
|
||||||
|
dialects mid-file (`.intel_syntax noprefix`), a percent sign on every
|
||||||
|
register and a dollar on every immediate: punctuation that carries
|
||||||
|
nothing the operand order did not already say. And the x86 family
|
||||||
|
fragments again underneath: NASM is not MASM is not GAS, each with its
|
||||||
|
own directive zoo and macro language, so every project picks a dialect
|
||||||
|
and every reader learns a different one by accident.
|
||||||
|
|
||||||
|
Plan 9 assembly has none of it. Registers are bare names. Memory is
|
||||||
|
one notation, `offset(base)`, extended by an index and a scale when
|
||||||
|
the instruction needs it. Arguments arrive named and offset-checked:
|
||||||
|
`x+0(FP)` is the argument x, on every architecture, and `go vet`
|
||||||
|
polices the offsets against the Go prototype.
|
||||||
|
|
||||||
|
```text
|
||||||
|
AT&T (GNU as): movq %rax, -16(%rbp)
|
||||||
|
Plan 9 (Go): MOVQ AX, total-16(SP)
|
||||||
|
```
|
||||||
|
|
||||||
|
The same lines, but only one of them tells you what the number is for.
|
||||||
|
The syntax is uppercase, regular and boring, which is the highest
|
||||||
|
compliment a language for machine code can earn. gasm-sdk exists
|
||||||
|
to give that syntax the tooling it deserves.
|
||||||
|
|
||||||
## Features
|
## Features
|
||||||
|
|
||||||
@@ -19,15 +71,20 @@ developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
|
|||||||
operating recursively on directories the way `go fmt` does. `-l` lists
|
operating recursively on directories the way `go fmt` does. `-l` lists
|
||||||
files whose formatting differs and `-d` prints a unified diff.
|
files whose formatting differs and `-d` prints a unified diff.
|
||||||
- **Linter.** `gasm lint` runs 18 conservative static checks, among them
|
- **Linter.** `gasm lint` runs 18 conservative static checks, among them
|
||||||
`undefined-label`, `abi-argsize` (declared frame vs the `// func` signature),
|
`undefined-label`, `abi-argsize` (declared argument area vs the `// func`
|
||||||
`register-clobber` (Go ABI register liveness over the control-flow graph),
|
signature), `register-clobber` (Go ABI register liveness over the
|
||||||
`stack-imbalance`, `abi0-register-args` and `unencodable-instruction`.
|
control-flow graph), `stack-imbalance`, `abi0-register-args` and
|
||||||
|
`unencodable-instruction`.
|
||||||
- **Standalone assembler.** `gasm asm` encodes all four architectures without
|
- **Standalone assembler.** `gasm asm` encodes all four architectures without
|
||||||
the Go toolchain and writes raw images, linkable ELF objects (with DWARF5
|
the Go toolchain and writes raw images or linkable ELF objects (with DWARF5
|
||||||
debug sections) or the Go toolchain's own GOOBJ format, which `go build`
|
debug sections) with no Go installation needed, or the Go toolchain's own
|
||||||
|
GOOBJ format, which needs the installed toolchain and which `go build`
|
||||||
consumes in place of the toolchain's output. Framed functions get the
|
consumes in place of the toolchain's output. Framed functions get the
|
||||||
stack-split guard and the morestack block, byte-identical to the
|
stack-split guard and the morestack block, byte-identical to the
|
||||||
toolchain's, so split functions link too.
|
toolchain's, so split functions link too. The assembler preprocesses
|
||||||
|
like the toolchain (`#define`, `#include` with `-I`, `#ifdef`), generates
|
||||||
|
`go_asm.h` from the package's Go files, and carries `PCALIGN`, the
|
||||||
|
`LOCK`/`REP` prefixes and the literal-data pseudo-ops.
|
||||||
- **Disassembler.** `gasm dis` lists a `.s` file's functions at their real
|
- **Disassembler.** `gasm dis` lists a `.s` file's functions at their real
|
||||||
offsets after assembling, or disassembles raw bytes from a file or stdin.
|
offsets after assembling, or disassembles raw bytes from a file or stdin.
|
||||||
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
|
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
|
||||||
@@ -36,54 +93,162 @@ developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
|
|||||||
byte-for-byte ground-truth comparison of the machine code.
|
byte-for-byte ground-truth comparison of the machine code.
|
||||||
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
|
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
|
||||||
breakpoints (optionally conditional), hardware watchpoints, register and
|
breakpoints (optionally conditional), hardware watchpoints, register and
|
||||||
memory inspection, and headless script runs with label-level coverage.
|
memory inspection, and headless script runs that report instruction and
|
||||||
|
label coverage; it runs on Linux (all four architectures) and FreeBSD
|
||||||
|
(amd64, arm64, riscv64).
|
||||||
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
|
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
|
||||||
push and pull diagnostics, semantic-token highlighting, go-to-definition,
|
push and pull diagnostics, semantic-token highlighting, go-to-definition,
|
||||||
find references, rename, formatting, inlay hints, code actions, signature
|
find references, rename, formatting, inlay hints, code actions, signature
|
||||||
help, document highlights, workspace symbol search, #include document
|
help, document highlights, workspace symbol search, #include document
|
||||||
links and folding ranges over stdio; definition, references and rename
|
links and folding ranges over stdio; definition, references and rename
|
||||||
work across every open document.
|
work across every open document and the indexed workspace files beyond
|
||||||
|
them, and the quick fixes add a missing textflag.h include and set the
|
||||||
|
argument area from the // func signature.
|
||||||
- **Comparators and audits.** `gasm diff` compares the machine code of two
|
- **Comparators and audits.** `gasm diff` compares the machine code of two
|
||||||
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
|
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
|
||||||
`gasm audit-instructions` diffs the encoder against the installed toolchain,
|
`gasm audit-instructions` diffs the encoder against the installed toolchain,
|
||||||
and `gasm scaffold` generates a differential test skeleton for a kernel.
|
and `gasm scaffold` generates a differential test skeleton for a kernel.
|
||||||
- **Complete instruction coverage.** The instruction tables are generated
|
|
||||||
from the Go toolchain's own assembler source, so the toolkit recognises
|
|
||||||
every mnemonic the real assembler accepts; `just gen` refreshes them.
|
|
||||||
|
|
||||||
### Architecture support
|
### Architecture support
|
||||||
|
|
||||||
|
Four architectures, the four that matter in practice:
|
||||||
|
|
||||||
| Architecture | GOARCH | File suffix | Instructions recognised |
|
| Architecture | GOARCH | File suffix | Instructions recognised |
|
||||||
|--------------|-------------|--------------|---------------------------------------------|
|
|--------------|-------------|--------------|---------------------------------------------|
|
||||||
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
|
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
|
||||||
| ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes |
|
| ARM64 | `arm64` | `_arm64.s` | 645 + common opcodes |
|
||||||
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes |
|
| RISC-V | `riscv64` | `_riscv64.s` | 992 + common opcodes |
|
||||||
| LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes |
|
| LoongArch | `loong64` | `_loong64.s` | 808 + common opcodes |
|
||||||
|
|
||||||
"Common opcodes" are the instructions shared by every architecture (`RET`,
|
"Common opcodes" are the instructions shared by every architecture (`RET`,
|
||||||
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
|
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
|
||||||
carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
|
carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
|
||||||
...) that the assembler accepts as aliases. Regenerating the tables is one
|
...) that the assembler accepts as aliases. The tables are generated from
|
||||||
command (`just gen`) and requires only a Go installation; the committed output
|
the Go toolchain's own assembler source (`just gen` refreshes them), so
|
||||||
has no runtime dependency on the toolchain.
|
every mnemonic the real assembler accepts is recognised; what the encoder
|
||||||
|
can emit today is narrower, and a recognised but unencodable instruction is
|
||||||
|
reported as an explicit error, never as a wrong byte.
|
||||||
|
|
||||||
|
The same measurement runs over GOROOT's whole assembly corpus:
|
||||||
|
`gasm audit-instructions --corpus` reports every real-code GOROOT assembly
|
||||||
|
file (the tree without testdata) assembling for every target its build
|
||||||
|
admits: 250 of 250, 100 %. Over the whole tree including testdata the
|
||||||
|
measure is 271 of 322 attemptable (84.2 %); files named for other Go ports
|
||||||
|
are counted but never attempted, and `//go:build` constraints decide which
|
||||||
|
targets attempt a file at all, exactly as the build does. The number moves
|
||||||
|
with every release.
|
||||||
|
|
||||||
|
### Validation status
|
||||||
|
|
||||||
|
**Only amd64 is validated on real hardware.** The other three
|
||||||
|
architectures are validated under qemu-user emulation, because the
|
||||||
|
project owns no arm64, riscv64 or loong64 machine, and emulation is the
|
||||||
|
only substitute available for the hardware. The distinction matters and
|
||||||
|
is stated rather than implied: everything below is a claim about what has
|
||||||
|
actually been executed.
|
||||||
|
|
||||||
|
| Layer | amd64 | arm64, riscv64, loong64 |
|
||||||
|
|---|---|---|
|
||||||
|
| Encoding: byte-for-byte against `go tool asm` | native hardware | native hardware (the toolchain cross-assembles any GOARCH on any host) |
|
||||||
|
| Execution: JIT calls, ABI checks, differential fuzzing | native hardware | qemu-user emulation |
|
||||||
|
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more. FreeBSD (amd64, arm64, riscv64) is in the same position: the port compiles behind the cross-build gate and its integration test is ready, but no FreeBSD machine has executed it |
|
||||||
|
|
||||||
|
Consequences, stated plainly. An emulator is a model of a CPU, not the
|
||||||
|
CPU: instruction semantics are implemented in software and can differ
|
||||||
|
from silicon in ways a test suite does not reveal. A kernel that passes
|
||||||
|
under qemu-user is therefore not proven correct on real hardware, and a
|
||||||
|
discrepancy found on real hardware is a defect in gasm, reported like any
|
||||||
|
other. Encoding parity is the exception: the byte comparison against the
|
||||||
|
toolchain runs on the host for every architecture, so no emulator stands
|
||||||
|
between the claim and the evidence. The debugger is the weakest case: on
|
||||||
|
the three emulated architectures its per-architecture ptrace code has
|
||||||
|
been compiled and read, never executed. Its architecture-neutral units
|
||||||
|
run under `go test ./...`, which the race workflow and a manual run
|
||||||
|
perform; the default `just test` gate does not sweep `./debug/...`.
|
||||||
|
|
||||||
|
## The documentation goal
|
||||||
|
|
||||||
|
The toolkit is the primary goal. The secondary one is documentation: a
|
||||||
|
specification of the Plan 9 assembly language and of the GOOBJ object
|
||||||
|
format that is 100 % complete, detailed enough to implement against,
|
||||||
|
and written to a professional standard. These are the two subjects this
|
||||||
|
project works with every day, and they are the two for which no usable
|
||||||
|
documentation exists.
|
||||||
|
|
||||||
|
Go documents the language on a single page, "A Quick Guide to Go's
|
||||||
|
Assembler", which carries no section for loong64, one of the four
|
||||||
|
architectures gasm supports, and covers a fraction of what each
|
||||||
|
assembler accepts. What exists beyond it lives as comments inside the
|
||||||
|
toolchain's internal source: per-architecture reference manuals for
|
||||||
|
arm64, ppc64, riscv64 and loong64, written for the toolchain's own
|
||||||
|
maintainers rather than for an outside reader, and none at all for
|
||||||
|
amd64. GOOBJ fares worst of all. The format that `go build` consumes
|
||||||
|
has no specification anywhere: it is described by a comment in an
|
||||||
|
internal package, it is not a stable interface, and it can change with
|
||||||
|
any toolchain release.
|
||||||
|
|
||||||
|
The gap is therefore filled the only way it can be filled: by reverse
|
||||||
|
engineering the toolchain itself, the same work the encoders already
|
||||||
|
perform. Most of the documentation can come from nowhere else, and it
|
||||||
|
is written as that knowledge is produced during development. It is
|
||||||
|
verified the way the code is verified: an encoding documented here is
|
||||||
|
one that differential tests against `go tool asm` confirm
|
||||||
|
byte-for-byte, and a format field documented here is one the linker
|
||||||
|
demonstrably reads. The work has begun: [docs/GOOBJ.md](docs/GOOBJ.md)
|
||||||
|
specifies the object file format completely, and
|
||||||
|
[docs/asm/README.md](docs/asm/README.md) opens the language reference
|
||||||
|
with its common core. The per-architecture pages follow.
|
||||||
|
|
||||||
|
## Direction
|
||||||
|
|
||||||
|
The plan, in the order it is being worked:
|
||||||
|
|
||||||
|
- **Extended instruction support.** Two layers. First, encoding
|
||||||
|
coverage for every mnemonic the Go toolchain itself accepts, closed in
|
||||||
|
order of how often real code needs each instruction;
|
||||||
|
`gasm audit-instructions` measures the gap. Second, the larger work:
|
||||||
|
an extended instruction set the toolchain does not know at all. The
|
||||||
|
toolchain-derived tables stay generated and untouched; only the
|
||||||
|
extended instructions are hand-maintained, with their own spellings
|
||||||
|
and encoders, verified by execution (on real hardware for amd64, under
|
||||||
|
emulation for the rest, per the validation status above) because the
|
||||||
|
toolchain offers no ground truth to compare against. The gaps exist
|
||||||
|
on every architecture, amd64 included.
|
||||||
|
- **Full GOOBJ and ELF compilation.** The destination is a complete,
|
||||||
|
standalone compilation path: linkable ELF objects for consumers outside
|
||||||
|
Go, and GOOBJ objects that `go build` links directly. Through GOOBJ, a
|
||||||
|
Go program will be able to use machine instructions that the Go
|
||||||
|
toolchain itself does not support; through ELF, Plan 9 assembly becomes
|
||||||
|
usable outside Go entirely.
|
||||||
|
- **Platforms: Linux and FreeBSD.** Linux is supported today on all four
|
||||||
|
architectures and is where the binary builds. FreeBSD follows on amd64,
|
||||||
|
arm64 and riscv64: the JIT's executable-memory mapping and the ptrace
|
||||||
|
debugger layer are ported (the debugger's live validation awaits a
|
||||||
|
FreeBSD machine, as the validation status states). Other unix systems
|
||||||
|
may follow those two.
|
||||||
|
- **Four architectures, no more.** amd64, arm64, riscv64 and loong64.
|
||||||
|
No others are planned.
|
||||||
|
|
||||||
## Install
|
## Install
|
||||||
|
|
||||||
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
|
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
|
||||||
linux/loong64 are on the
|
linux/loong64 are on the
|
||||||
[releases page](https://sourcedock.dev/petrbalvin/gasm-devkit/releases).
|
[releases page](https://sourcedock.dev/petrbalvin/gasm-sdk/releases).
|
||||||
From source (Go 1.27 or later):
|
From source (Go 1.27.1):
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
go install sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm@latest
|
go install sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm@latest
|
||||||
```
|
```
|
||||||
|
|
||||||
Or from a repository checkout, with the development version stamped:
|
Or from a repository checkout:
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
just install-bin
|
just install
|
||||||
```
|
```
|
||||||
|
|
||||||
|
The installed binary reports the version the toolchain recorded: the tag
|
||||||
|
on a tagged checkout, a pseudo-version naming the commit below one.
|
||||||
|
|
||||||
## Quick start
|
## Quick start
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
@@ -119,7 +284,7 @@ gasm verify --ground-truth k.s # byte-for-byte vs go tool asm
|
|||||||
gasm verify --fuzz k.s # differential fuzz vs the go tool asm build
|
gasm verify --fuzz k.s # differential fuzz vs the go tool asm build
|
||||||
gasm debug --func name k.s # interactive debugger
|
gasm debug --func name k.s # interactive debugger
|
||||||
gasm debug --func name --script cmds.txt --timeout 30s k.s # headless run
|
gasm debug --func name --script cmds.txt --timeout 30s k.s # headless run
|
||||||
gasm debug --func name --cover k.s # which labels did execution reach?
|
gasm debug --func name --cover k.s # instruction and label coverage
|
||||||
gasm diff a.s b.s # compare machine code byte-for-byte
|
gasm diff a.s b.s # compare machine code byte-for-byte
|
||||||
gasm diff --map wideCopyAVX2=wideCopyAVX512 avx2.s avx512.s
|
gasm diff --map wideCopyAVX2=wideCopyAVX512 avx2.s avx512.s
|
||||||
gasm profile k.s # show basic-block structure
|
gasm profile k.s # show basic-block structure
|
||||||
@@ -142,9 +307,9 @@ infers the target architecture from the file-name suffix
|
|||||||
## Development
|
## Development
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
just install # download module dependencies
|
just build # compile, zero errors and zero warnings
|
||||||
just build # go vet + gofmt check, zero errors and zero warnings
|
just test # the suite, no cache, the 80 % coverage floor
|
||||||
just test # full suite, race detector, 80 % coverage gate
|
just gates # build, fmt-check, vet, test, race: the definition of done
|
||||||
just fmt # gofmt the tree
|
just fmt # gofmt the tree
|
||||||
just gen # regenerate the instruction tables from the Go toolchain
|
just gen # regenerate the instruction tables from the Go toolchain
|
||||||
```
|
```
|
||||||
@@ -155,14 +320,19 @@ recipe.
|
|||||||
|
|
||||||
## Documentation
|
## Documentation
|
||||||
|
|
||||||
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
|
|
||||||
- [docs/CLI.md](docs/CLI.md): full command reference
|
- [docs/CLI.md](docs/CLI.md): full command reference
|
||||||
|
- man pages: `just install-man` installs gasm(1) and one page per command
|
||||||
|
except `version`, which is documented inside gasm(1) instead, into
|
||||||
|
~/.local/share/man (MANDIR overrides); `just uninstall-man` removes
|
||||||
|
them
|
||||||
|
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
|
||||||
|
- [docs/GOOBJ.md](docs/GOOBJ.md): the GOOBJ object file format specification
|
||||||
|
- [docs/asm/](docs/asm/README.md): the Plan 9 assembly language reference
|
||||||
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
|
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
|
||||||
- [docs/DECISIONS.md](docs/DECISIONS.md): deferred design decisions
|
|
||||||
- [CHANGELOG.md](CHANGELOG.md): release history
|
- [CHANGELOG.md](CHANGELOG.md): release history
|
||||||
|
|
||||||
## Licence
|
## Licence
|
||||||
|
|
||||||
BSD-3-Clause — see [LICENSE](LICENSE).
|
BSD-3-Clause; see [LICENSE](LICENSE).
|
||||||
|
|
||||||
Copyright © 2026 [Petr Balvín](https://petrbalvin.org)
|
Copyright © 2026 [Petr Balvín](https://petrbalvin.org)
|
||||||
|
|||||||
+41
@@ -0,0 +1,41 @@
|
|||||||
|
# Security policy
|
||||||
|
|
||||||
|
## Supported versions
|
||||||
|
|
||||||
|
Security fixes go to the newest release and to the `development` branch. Older
|
||||||
|
releases do not receive them.
|
||||||
|
|
||||||
|
| Version | Supported |
|
||||||
|
|---|---|
|
||||||
|
| 0.35.0 | yes |
|
||||||
|
| older releases | no |
|
||||||
|
|
||||||
|
## Reporting a vulnerability
|
||||||
|
|
||||||
|
**Do not open a public issue for a security problem.** A public report tells everyone
|
||||||
|
about the flaw before there is a fix. Report it privately to
|
||||||
|
**opensource@petrbalvin.org**.
|
||||||
|
|
||||||
|
Include:
|
||||||
|
|
||||||
|
- the version or commit you tested, and the platform
|
||||||
|
- what the problem is, and what an attacker gains from it
|
||||||
|
- the smallest reproducer you have, ideally a test or a single command
|
||||||
|
- a suggested fix, if you have one
|
||||||
|
|
||||||
|
## What to expect
|
||||||
|
|
||||||
|
- A human reads the report, and you get an acknowledgement.
|
||||||
|
- You are kept informed while the fix is being made, and told when it ships.
|
||||||
|
- The fix is released before the details are published, and the timing is agreed with
|
||||||
|
you.
|
||||||
|
- The fix ships without naming you: the project keeps no credits list, so the release
|
||||||
|
notes, the changelog and the commits name no reporter.
|
||||||
|
|
||||||
|
## Out of scope
|
||||||
|
|
||||||
|
- Findings that require the attacker to already run code as the user, or to have local
|
||||||
|
access.
|
||||||
|
- Missing hardening with no demonstrated impact.
|
||||||
|
- Flaws in a third-party dependency: report them to that project, and to this one only
|
||||||
|
when this project's use of it makes them reachable.
|
||||||
+105
-7
@@ -5,9 +5,13 @@
|
|||||||
// toolchain's own assembler source. Go's Plan 9 assembler defines the exact,
|
// toolchain's own assembler source. Go's Plan 9 assembler defines the exact,
|
||||||
// complete set of mnemonics it accepts for each architecture in
|
// complete set of mnemonics it accepts for each architecture in
|
||||||
// $GOROOT/src/cmd/internal/obj/<arch>/anames.go; this tool extracts those
|
// $GOROOT/src/cmd/internal/obj/<arch>/anames.go; this tool extracts those
|
||||||
// names so gasm-devkit supports every instruction the real assembler does,
|
// names so gasm-sdk supports every instruction the real assembler does,
|
||||||
// with no hand-maintained (and therefore inevitably incomplete) lists.
|
// with no hand-maintained (and therefore inevitably incomplete) lists.
|
||||||
//
|
//
|
||||||
|
// The same data feeds the generated instruction appendices of the assembly
|
||||||
|
// language reference, docs/asm/INSTRUCTIONS-<ARCH>.md, so that the reference
|
||||||
|
// cannot drift from the tables it documents.
|
||||||
|
//
|
||||||
// Usage (via the justfile):
|
// Usage (via the justfile):
|
||||||
//
|
//
|
||||||
// just gen
|
// just gen
|
||||||
@@ -26,9 +30,12 @@ import (
|
|||||||
"path/filepath"
|
"path/filepath"
|
||||||
"sort"
|
"sort"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||||
)
|
)
|
||||||
|
|
||||||
// archDirs maps a gasm-devkit architecture name to its obj sub-directory.
|
// archDirs maps a gasm-sdk architecture name to its obj sub-directory.
|
||||||
var archDirs = []struct {
|
var archDirs = []struct {
|
||||||
arch string
|
arch string
|
||||||
sub string
|
sub string
|
||||||
@@ -39,11 +46,30 @@ var archDirs = []struct {
|
|||||||
{"loong64", "loong64"},
|
{"loong64", "loong64"},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// docPages maps an architecture to its generated appendix in the language
|
||||||
|
// reference. The amd64 page carries a per-mnemonic encodability column,
|
||||||
|
// decided by asm.Encodable, which mirrors the encoder's own dispatch; the
|
||||||
|
// other targets have no single cheap predicate, so their pages carry the
|
||||||
|
// inventory and point at the live measurement instead.
|
||||||
|
var docPages = []struct {
|
||||||
|
arch arch.Arch
|
||||||
|
title string
|
||||||
|
file string
|
||||||
|
anames string
|
||||||
|
encodable bool
|
||||||
|
}{
|
||||||
|
{arch.AMD64, "AMD64", "INSTRUCTIONS-AMD64.md", "cmd/internal/obj/x86/anames.go", true},
|
||||||
|
{arch.ARM64, "ARM64", "INSTRUCTIONS-ARM64.md", "cmd/internal/obj/arm64/anames.go", false},
|
||||||
|
{arch.RISCV, "RISC-V 64", "INSTRUCTIONS-RISCV64.md", "cmd/internal/obj/riscv/anames.go", false},
|
||||||
|
{arch.LOONG64, "LoongArch 64", "INSTRUCTIONS-LOONG64.md", "cmd/internal/obj/loong64/anames.go", false},
|
||||||
|
}
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
goroot := strings.TrimSpace(runGoEnvGOROOT())
|
goroot := strings.TrimSpace(runGoEnvGOROOT())
|
||||||
if goroot == "" {
|
if goroot == "" {
|
||||||
fatal("could not determine GOROOT")
|
fatal("could not determine GOROOT")
|
||||||
}
|
}
|
||||||
|
version := strings.TrimSpace(runGoEnv("GOVERSION"))
|
||||||
// The common opcodes shared by every architecture (RET, JMP, NOP, CALL,
|
// The common opcodes shared by every architecture (RET, JMP, NOP, CALL,
|
||||||
// TEXT, FUNCDATA, …) live in cmd/internal/obj/util.go.
|
// TEXT, FUNCDATA, …) live in cmd/internal/obj/util.go.
|
||||||
commonPath := filepath.Join(goroot, "src", "cmd", "internal", "obj", "util.go")
|
commonPath := filepath.Join(goroot, "src", "cmd", "internal", "obj", "util.go")
|
||||||
@@ -57,16 +83,24 @@ func main() {
|
|||||||
}
|
}
|
||||||
fmt.Printf("%-8s %4d instructions -> arch/common_gen.go\n", "common", len(common))
|
fmt.Printf("%-8s %4d instructions -> arch/common_gen.go\n", "common", len(common))
|
||||||
|
|
||||||
|
names := map[string][]string{}
|
||||||
for _, a := range archDirs {
|
for _, a := range archDirs {
|
||||||
path := filepath.Join(goroot, "src", "cmd", "internal", "obj", a.sub, "anames.go")
|
path := filepath.Join(goroot, "src", "cmd", "internal", "obj", a.sub, "anames.go")
|
||||||
names, err := extractInstrs(path)
|
names[a.arch], err = extractInstrs(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fatal("extract %s: %v", a.arch, err)
|
fatal("extract %s: %v", a.arch, err)
|
||||||
}
|
}
|
||||||
if err := writeGen(a.arch, a.sub, names); err != nil {
|
if err := writeGen(a.arch, a.sub, names[a.arch]); err != nil {
|
||||||
fatal("write %s: %v", a.arch, err)
|
fatal("write %s: %v", a.arch, err)
|
||||||
}
|
}
|
||||||
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names), a.arch)
|
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names[a.arch]), a.arch)
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, p := range docPages {
|
||||||
|
if err := writeDocPage(p.arch, p.title, p.file, p.anames, version, p.encodable); err != nil {
|
||||||
|
fatal("write %s: %v", p.file, err)
|
||||||
|
}
|
||||||
|
fmt.Printf("%-8s -> docs/asm/%s\n", p.arch, p.file)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -85,7 +119,7 @@ func filterCommon(names []string) []string {
|
|||||||
// writeCommon emits arch/common_gen.go.
|
// writeCommon emits arch/common_gen.go.
|
||||||
func writeCommon(names []string) error {
|
func writeCommon(names []string) error {
|
||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
|
||||||
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
|
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
|
||||||
b.WriteString("//\n")
|
b.WriteString("//\n")
|
||||||
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
||||||
@@ -156,7 +190,7 @@ func stringLit(elt ast.Expr) string {
|
|||||||
// writeGen emits arch/<arch>_gen.go.
|
// writeGen emits arch/<arch>_gen.go.
|
||||||
func writeGen(arch, sub string, names []string) error {
|
func writeGen(arch, sub string, names []string) error {
|
||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
|
||||||
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
|
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
|
||||||
b.WriteString("//\n")
|
b.WriteString("//\n")
|
||||||
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
||||||
@@ -172,6 +206,61 @@ func writeGen(arch, sub string, names []string) error {
|
|||||||
return os.WriteFile(filepath.Join("arch", arch+"_gen.go"), []byte(b.String()), 0o644)
|
return os.WriteFile(filepath.Join("arch", arch+"_gen.go"), []byte(b.String()), 0o644)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// writeDocPage emits docs/asm/<file>, the generated instruction appendix of
|
||||||
|
// the language reference for one architecture: every mnemonic the toolchain
|
||||||
|
// accepts, with the curated summary where the architecture table carries one
|
||||||
|
// and, on amd64, a per-mnemonic encodability column.
|
||||||
|
func writeDocPage(a arch.Arch, title, file, anames, version string, encodable bool) error {
|
||||||
|
table := arch.ForArch(a)
|
||||||
|
instrs := table.Instructions()
|
||||||
|
|
||||||
|
var b strings.Builder
|
||||||
|
b.WriteString("# " + title + ": instruction inventory\n\n")
|
||||||
|
b.WriteString("Generated by gasm-sdk's `_gen` from the Go toolchain's instruction table\n")
|
||||||
|
b.WriteString("(`" + anames + "`, " + version + "); DO NOT EDIT. This page lists every mnemonic\n")
|
||||||
|
b.WriteString("`go tool asm` accepts on this target, which is the upper bound of the\n")
|
||||||
|
b.WriteString("language on it: a name absent here is not an instruction of the target,\n")
|
||||||
|
b.WriteString("and a name present here may still be one gasm's encoder cannot emit yet.\n\n")
|
||||||
|
|
||||||
|
encodableCount := 0
|
||||||
|
if encodable {
|
||||||
|
b.WriteString("The `gasm encodes` column reports whether gasm's encoder can emit the\n")
|
||||||
|
b.WriteString("mnemonic today; the gap is the encoder backlog, measured live by\n")
|
||||||
|
b.WriteString("`gasm audit-instructions`.\n\n")
|
||||||
|
b.WriteString("| Mnemonic | gasm encodes | Notes |\n")
|
||||||
|
b.WriteString("|---|---|---|\n")
|
||||||
|
for _, in := range instrs {
|
||||||
|
ok := asm.Encodable(in.Name)
|
||||||
|
if ok {
|
||||||
|
encodableCount++
|
||||||
|
}
|
||||||
|
b.WriteString("| `" + in.Name + "` | " + yesNo(ok) + " | " + in.Summary + " |\n")
|
||||||
|
}
|
||||||
|
b.WriteString("\n")
|
||||||
|
fmt.Fprintf(&b, "Recognised: %d mnemonics. gasm encodes: %d.\n", len(instrs), encodableCount)
|
||||||
|
} else {
|
||||||
|
b.WriteString("The inventory carries no per-mnemonic encoder column: on this target\n")
|
||||||
|
b.WriteString("encodability is decided per operand shape, and the live measured\n")
|
||||||
|
b.WriteString("coverage is reported by `gasm audit-instructions`.\n\n")
|
||||||
|
b.WriteString("| Mnemonic | Notes |\n")
|
||||||
|
b.WriteString("|---|---|\n")
|
||||||
|
for _, in := range instrs {
|
||||||
|
b.WriteString("| `" + in.Name + "` | " + in.Summary + " |\n")
|
||||||
|
}
|
||||||
|
b.WriteString("\n")
|
||||||
|
fmt.Fprintf(&b, "Recognised: %d mnemonics.\n", len(instrs))
|
||||||
|
}
|
||||||
|
return os.WriteFile(filepath.Join("docs", "asm", file), []byte(b.String()), 0o644)
|
||||||
|
}
|
||||||
|
|
||||||
|
// yesNo renders a boolean as the word the appendix tables use.
|
||||||
|
func yesNo(v bool) string {
|
||||||
|
if v {
|
||||||
|
return "yes"
|
||||||
|
}
|
||||||
|
return "no"
|
||||||
|
}
|
||||||
|
|
||||||
func runGoEnvGOROOT() string {
|
func runGoEnvGOROOT() string {
|
||||||
out, err := exec.Command("go", "env", "GOROOT").Output()
|
out, err := exec.Command("go", "env", "GOROOT").Output()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -180,6 +269,15 @@ func runGoEnvGOROOT() string {
|
|||||||
return string(out)
|
return string(out)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// runGoEnv runs `go env` for a single variable.
|
||||||
|
func runGoEnv(name string) string {
|
||||||
|
out, err := exec.Command("go", "env", name).Output()
|
||||||
|
if err != nil {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
return string(out)
|
||||||
|
}
|
||||||
|
|
||||||
func fatal(format string, args ...any) {
|
func fatal(format string, args ...any) {
|
||||||
fmt.Fprintf(os.Stderr, "gen: "+format+"\n", args...)
|
fmt.Fprintf(os.Stderr, "gen: "+format+"\n", args...)
|
||||||
os.Exit(1)
|
os.Exit(1)
|
||||||
|
|||||||
@@ -70,6 +70,10 @@ func amd64Registers() []Register {
|
|||||||
for i := 0; i <= 7; i++ {
|
for i := 0; i <= 7; i++ {
|
||||||
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
|
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
|
||||||
}
|
}
|
||||||
|
// x87 stack registers (FMOVD and the other x87 moves).
|
||||||
|
for i := 0; i <= 7; i++ {
|
||||||
|
add(fmt.Sprintf("F%d", i), Float, "x87 stack register")
|
||||||
|
}
|
||||||
return regs
|
return regs
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
|
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
|
||||||
//
|
//
|
||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
|||||||
@@ -55,6 +55,7 @@ const (
|
|||||||
Mask // AVX-512 mask register (K)
|
Mask // AVX-512 mask register (K)
|
||||||
Float // arm64 floating-point register (F)
|
Float // arm64 floating-point register (F)
|
||||||
VecARM // arm64 SIMD/vector register (V)
|
VecARM // arm64 SIMD/vector register (V)
|
||||||
|
VecSIMD // architecture-neutral SIMD/vector register (LoongArch LSX/LASX)
|
||||||
Special // architecture-special register
|
Special // architecture-special register
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -73,6 +74,8 @@ func (c RegClass) String() string {
|
|||||||
return "float"
|
return "float"
|
||||||
case VecARM:
|
case VecARM:
|
||||||
return "vector (arm64)"
|
return "vector (arm64)"
|
||||||
|
case VecSIMD:
|
||||||
|
return "vector"
|
||||||
case Special:
|
case Special:
|
||||||
return "special"
|
return "special"
|
||||||
default:
|
default:
|
||||||
|
|||||||
+29
-1
@@ -33,6 +33,7 @@ func arm64Registers() []Register {
|
|||||||
for i := 0; i <= 30; i++ {
|
for i := 0; i <= 30; i++ {
|
||||||
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
|
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
|
||||||
}
|
}
|
||||||
|
add("R18_PLATFORM", GPR, "R18 under its toolchain-reserved Windows name (an alias of R18)")
|
||||||
add("ZR", Special, "zero register (reads as 0)")
|
add("ZR", Special, "zero register (reads as 0)")
|
||||||
add("SP", Special, "stack pointer")
|
add("SP", Special, "stack pointer")
|
||||||
add("LR", Special, "link register (alias of R30)")
|
add("LR", Special, "link register (alias of R30)")
|
||||||
@@ -145,11 +146,38 @@ func arm64Curated() []Instr {
|
|||||||
for _, op := range []string{
|
for _, op := range []string{
|
||||||
"LDAXR", "LDAXRB", "LDAXRH", "LDAXRW", "STXR", "STXRB", "STXRH", "STXRW",
|
"LDAXR", "LDAXRB", "LDAXRH", "LDAXRW", "STXR", "STXRB", "STXRH", "STXRW",
|
||||||
"LDAR", "LDARB", "LDARH", "LDARW", "STLR", "STLRB", "STLRH", "STLRW",
|
"LDAR", "LDARB", "LDARH", "LDARW", "STLR", "STLRB", "STLRH", "STLRW",
|
||||||
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL", "CASAL",
|
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL",
|
||||||
} {
|
} {
|
||||||
t = append(t, i(op, "Atomic memory operation"))
|
t = append(t, i(op, "Atomic memory operation"))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Register-pair loads and stores.
|
||||||
|
for _, op := range []string{"LDP", "STP", "LDPW", "STPW", "FLDPD", "FSTPD"} {
|
||||||
|
t = append(t, ic(op, "Register-pair load or store", 2, 2))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Cache maintenance and prefetch.
|
||||||
|
t = append(t, i("DC", "Data cache maintenance"))
|
||||||
|
t = append(t, i("PRFM", "Memory prefetch"))
|
||||||
|
for _, op := range []string{"LDADDAL", "LDCLRAL", "LDORAL", "SWPAL"} {
|
||||||
|
t = append(t, i(op, "Atomic memory operation with acquire and release semantics"))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Cryptographic extensions.
|
||||||
|
for _, op := range []string{"AESE", "AESD", "AESMC", "AESIMC"} {
|
||||||
|
t = append(t, i(op, "AES round"))
|
||||||
|
}
|
||||||
|
for _, op := range []string{
|
||||||
|
"SHA1C", "SHA1P", "SHA1M", "SHA1H", "SHA1SU0", "SHA1SU1",
|
||||||
|
"SHA256H", "SHA256H2", "SHA256SU0", "SHA256SU1",
|
||||||
|
"SHA512H", "SHA512H2", "SHA512SU0", "SHA512SU1",
|
||||||
|
} {
|
||||||
|
t = append(t, i(op, "SHA round"))
|
||||||
|
}
|
||||||
|
for _, op := range []string{"VEOR3", "VBCAX", "VXAR", "VRAX1"} {
|
||||||
|
t = append(t, i(op, "Three-way XOR / rotate crypto vector operation"))
|
||||||
|
}
|
||||||
|
|
||||||
// Floating-point scalar.
|
// Floating-point scalar.
|
||||||
for _, op := range []string{
|
for _, op := range []string{
|
||||||
"FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX",
|
"FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX",
|
||||||
|
|||||||
@@ -0,0 +1,624 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
// This file carries the extended-instruction layer: instructions the Go
|
||||||
|
// toolchain does not know at all, described as data and validated against
|
||||||
|
// golden vectors from the Arm Architecture Reference Manual rather than
|
||||||
|
// against the toolchain. It sits beside the generated tables, never inside
|
||||||
|
// them: arch/arm64_gen.go stays untouched, and Extensions returns the layer
|
||||||
|
// per architecture so a later amd64 table attaches through the same door.
|
||||||
|
//
|
||||||
|
// The first entry is the arm64 SVE and SVE2 integer add/subtract/multiply
|
||||||
|
// family (twenty-three forms over four word shapes). The encodings are
|
||||||
|
// transcribed from the manual and cross-checked against the GNU assembler's
|
||||||
|
// and LLVM's published encodings; the golden vectors in arm64_ext_test.go pin
|
||||||
|
// the bytes.
|
||||||
|
|
||||||
|
package arch
|
||||||
|
|
||||||
|
import "fmt"
|
||||||
|
|
||||||
|
// ExtOperandKind classifies one operand of an extended instruction.
|
||||||
|
type ExtOperandKind uint8
|
||||||
|
|
||||||
|
// Operand kinds.
|
||||||
|
const (
|
||||||
|
ExtZReg ExtOperandKind = iota // scalable vector register Z0-Z31
|
||||||
|
ExtPReg // predicate register P0-P15
|
||||||
|
ExtImm // immediate
|
||||||
|
)
|
||||||
|
|
||||||
|
// String returns a short label for the kind.
|
||||||
|
func (k ExtOperandKind) String() string {
|
||||||
|
switch k {
|
||||||
|
case ExtZReg:
|
||||||
|
return "scalable vector register"
|
||||||
|
case ExtPReg:
|
||||||
|
return "predicate register"
|
||||||
|
case ExtImm:
|
||||||
|
return "immediate"
|
||||||
|
default:
|
||||||
|
return "operand"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExtArrangement is the element-size suffix a scalable vector operand
|
||||||
|
// carries: .B, .H, .S, .D or .Q. ExtArrNone means the operand is written
|
||||||
|
// bare, which the SVE forms in this layer reject.
|
||||||
|
type ExtArrangement uint8
|
||||||
|
|
||||||
|
// Arrangements, widest last.
|
||||||
|
const (
|
||||||
|
ExtArrNone ExtArrangement = iota
|
||||||
|
ExtArrB // 8-bit elements
|
||||||
|
ExtArrH // 16-bit elements
|
||||||
|
ExtArrS // 32-bit elements
|
||||||
|
ExtArrD // 64-bit elements
|
||||||
|
ExtArrQ // 128-bit elements
|
||||||
|
)
|
||||||
|
|
||||||
|
// String returns the assembler suffix, with the leading dot.
|
||||||
|
func (a ExtArrangement) String() string {
|
||||||
|
switch a {
|
||||||
|
case ExtArrB:
|
||||||
|
return ".B"
|
||||||
|
case ExtArrH:
|
||||||
|
return ".H"
|
||||||
|
case ExtArrS:
|
||||||
|
return ".S"
|
||||||
|
case ExtArrD:
|
||||||
|
return ".D"
|
||||||
|
case ExtArrQ:
|
||||||
|
return ".Q"
|
||||||
|
default:
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Width returns the byte width of one element under the arrangement.
|
||||||
|
func (a ExtArrangement) Width() int {
|
||||||
|
switch a {
|
||||||
|
case ExtArrB:
|
||||||
|
return 1
|
||||||
|
case ExtArrH:
|
||||||
|
return 2
|
||||||
|
case ExtArrS:
|
||||||
|
return 4
|
||||||
|
case ExtArrD:
|
||||||
|
return 8
|
||||||
|
case ExtArrQ:
|
||||||
|
return 16
|
||||||
|
default:
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// sizeBits maps the arrangement onto the two-bit size field the integer SVE
|
||||||
|
// classes carry at bits 23..22: 00=B, 01=H, 10=S, 11=D. ok is false for the
|
||||||
|
// arrangements no such class accepts (.Q and the bare spelling).
|
||||||
|
func (a ExtArrangement) sizeBits() (uint32, bool) {
|
||||||
|
switch a {
|
||||||
|
case ExtArrB, ExtArrH, ExtArrS, ExtArrD:
|
||||||
|
return uint32(a) - 1, true
|
||||||
|
default:
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExtQualifier is the predicate qualifier spelled after the slash.
|
||||||
|
type ExtQualifier uint8
|
||||||
|
|
||||||
|
// Predicate qualifiers.
|
||||||
|
const (
|
||||||
|
ExtQualNone ExtQualifier = iota // bare Pn (non-predicating position)
|
||||||
|
ExtQualMerging // /M, inactive lanes keep the destination
|
||||||
|
ExtQualZeroing // /Z, inactive lanes become zero
|
||||||
|
)
|
||||||
|
|
||||||
|
// String returns the assembler spelling, with the leading slash.
|
||||||
|
func (q ExtQualifier) String() string {
|
||||||
|
switch q {
|
||||||
|
case ExtQualMerging:
|
||||||
|
return "/M"
|
||||||
|
case ExtQualZeroing:
|
||||||
|
return "/Z"
|
||||||
|
default:
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExtOperand is one operand of an extended instruction, already resolved to
|
||||||
|
// its pieces: a register with its arrangement and qualifier, or an immediate
|
||||||
|
// with its optional left shift. The assembler's future hook constructs these
|
||||||
|
// from the parsed statement; Encode consumes them.
|
||||||
|
type ExtOperand struct {
|
||||||
|
Kind ExtOperandKind
|
||||||
|
Reg int // register number (Z: 0..31, P: 0..15)
|
||||||
|
Arr ExtArrangement // element-size suffix; ExtArrNone when bare
|
||||||
|
Qual ExtQualifier // predicate qualifier; ExtQualNone elsewhere
|
||||||
|
Imm int64 // immediate value (ExtImm only)
|
||||||
|
// Shift carries the LSL amount an immediate form shifts the constant by
|
||||||
|
// before use (0 or 8 in the SVE add/subtract immediate class). HasShift
|
||||||
|
// separates a spelled shift (validated as written) from an unshifted
|
||||||
|
// operand (the encoder may derive the sh bit from the value).
|
||||||
|
Shift int
|
||||||
|
HasShift bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExtVector builds a scalable vector operand, ADD Z1.S style.
|
||||||
|
func ExtVector(reg int, arr ExtArrangement) ExtOperand {
|
||||||
|
return ExtOperand{Kind: ExtZReg, Reg: reg, Arr: arr}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExtPredicate builds a predicate operand with its qualifier, P0/M style.
|
||||||
|
func ExtPredicate(reg int, qual ExtQualifier) ExtOperand {
|
||||||
|
return ExtOperand{Kind: ExtPReg, Reg: reg, Qual: qual}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExtImmediate builds an unshifted immediate operand.
|
||||||
|
func ExtImmediate(v int64) ExtOperand {
|
||||||
|
return ExtOperand{Kind: ExtImm, Imm: v}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExtShiftedImmediate builds an immediate operand with a spelled LSL amount.
|
||||||
|
func ExtShiftedImmediate(v int64, shift int) ExtOperand {
|
||||||
|
return ExtOperand{Kind: ExtImm, Imm: v, Shift: shift, HasShift: true}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExtField is one named field of the 32-bit encoding word: a bit offset from
|
||||||
|
// the least significant end and the field's width.
|
||||||
|
type ExtField struct {
|
||||||
|
Off uint8
|
||||||
|
Width uint8
|
||||||
|
}
|
||||||
|
|
||||||
|
// extMask returns the field's bits as a mask.
|
||||||
|
func extMask(f ExtField) uint32 {
|
||||||
|
return ^uint32(0) >> (32 - f.Width)
|
||||||
|
}
|
||||||
|
|
||||||
|
// extSet ORs v into the field of word.
|
||||||
|
func extSet(word uint32, f ExtField, v uint32) uint32 {
|
||||||
|
return word | (v&extMask(f))<<f.Off
|
||||||
|
}
|
||||||
|
|
||||||
|
// The fields the SVE integer classes use. The 5-bit register fields are
|
||||||
|
// named after their role in the three-vector class; the predicated class
|
||||||
|
// reuses extFieldRn for its Zm operand and extFieldPg for the governing
|
||||||
|
// predicate, which that class narrows to three bits (P0-P7).
|
||||||
|
var (
|
||||||
|
extFieldRd = ExtField{0, 5} // destination (Zd or Zdn)
|
||||||
|
extFieldRn = ExtField{5, 5} // first source (Zn, or Zm in the predicated class)
|
||||||
|
extFieldRm = ExtField{16, 5} // second source (Zm in the three-vector class)
|
||||||
|
extFieldPg = ExtField{10, 3} // governing predicate P0-P7 (predicated class)
|
||||||
|
extFieldImm8 = ExtField{5, 8} // the immediate, bits 12..5
|
||||||
|
extFieldSh = ExtField{13, 1} // the shift flag: 1 means LSL #8
|
||||||
|
extSizeBHSD = ExtField{22, 2} // element-size field of every class here, bits 23..22
|
||||||
|
)
|
||||||
|
|
||||||
|
// ExtForm enumerates the operand shapes the extension layer defines, in Plan
|
||||||
|
// 9 order (sources first, destination last). A destructive SVE operand is
|
||||||
|
// written once, in destination position: the encoding carries no second copy.
|
||||||
|
type ExtForm uint8
|
||||||
|
|
||||||
|
// Operand shapes.
|
||||||
|
const (
|
||||||
|
// ExtFormVectors is the unpredicated three-vector form, the SVE integer
|
||||||
|
// add/subtract (unpredicated) class: ADD Z0.S, Z1.S, Z2.S computes
|
||||||
|
// Z0 = Z1 + Z2. Operands: Zn, Zm, Zd.
|
||||||
|
ExtFormVectors ExtForm = iota
|
||||||
|
// ExtFormPredicated is the governed destructive form, the SVE integer
|
||||||
|
// add/subtract vectors (predicated) class: ADD Z1.S, P0/M, Z0.S computes
|
||||||
|
// Z0 = Z0 + Z1 for the active lanes. Operands: Zm, Pg/M, Zdn. The
|
||||||
|
// governing predicate is a 3-bit field, so only P0-P7 encode here, and
|
||||||
|
// the class takes the merging qualifier alone: a zeroing form would need
|
||||||
|
// a MOVPRFX expansion, which one data word cannot carry.
|
||||||
|
ExtFormPredicated
|
||||||
|
// ExtFormImmediate is the add/subtract immediate form, the SVE integer
|
||||||
|
// add/subtract (immediate) class: ADD $255, Z0.S computes
|
||||||
|
// Z0 = Z0 + 255. Operands: imm{, LSL #8}, Zdn. The constant is an
|
||||||
|
// unsigned imm8, optionally shifted left by 8 bits; a bare multiple of
|
||||||
|
// 256 (up to 65280) derives the shift, the spelling the GNU assembler
|
||||||
|
// canonicalises too. .B takes no shift.
|
||||||
|
ExtFormImmediate
|
||||||
|
// ExtFormSignedImmediate is the signed immediate form of the SVE integer
|
||||||
|
// multiply (immediate) class: MUL $-128, Z0.B computes Z0 = Z0 * -128.
|
||||||
|
// Operands: simm8, Zdn. No shift exists in this class.
|
||||||
|
ExtFormSignedImmediate
|
||||||
|
)
|
||||||
|
|
||||||
|
// Arity returns the operand count the form takes.
|
||||||
|
func (f ExtForm) Arity() int {
|
||||||
|
switch f {
|
||||||
|
case ExtFormVectors, ExtFormPredicated:
|
||||||
|
return 3
|
||||||
|
case ExtFormImmediate, ExtFormSignedImmediate:
|
||||||
|
return 2
|
||||||
|
default:
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Kinds returns the operand kind each position of the form wants, in the
|
||||||
|
// order the operands arrive. The registry uses the list to pick the most
|
||||||
|
// specific rejection when every matching form refuses an operand list.
|
||||||
|
func (f ExtForm) Kinds() []ExtOperandKind {
|
||||||
|
switch f {
|
||||||
|
case ExtFormVectors:
|
||||||
|
return []ExtOperandKind{ExtZReg, ExtZReg, ExtZReg}
|
||||||
|
case ExtFormPredicated:
|
||||||
|
return []ExtOperandKind{ExtZReg, ExtPReg, ExtZReg}
|
||||||
|
case ExtFormImmediate, ExtFormSignedImmediate:
|
||||||
|
return []ExtOperandKind{ExtImm, ExtZReg}
|
||||||
|
default:
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// String returns a short label for the form, for diagnostics.
|
||||||
|
func (f ExtForm) String() string {
|
||||||
|
switch f {
|
||||||
|
case ExtFormVectors:
|
||||||
|
return "unpredicated vectors"
|
||||||
|
case ExtFormPredicated:
|
||||||
|
return "predicated (merging)"
|
||||||
|
case ExtFormImmediate:
|
||||||
|
return "unsigned immediate"
|
||||||
|
case ExtFormSignedImmediate:
|
||||||
|
return "signed immediate"
|
||||||
|
default:
|
||||||
|
return "unknown form"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExtFeature names the architecture feature an extended instruction belongs
|
||||||
|
// to. The field is metadata: the assembler offers every instruction it
|
||||||
|
// registers, and a feature check is the caller's decision, not the encoder's.
|
||||||
|
type ExtFeature string
|
||||||
|
|
||||||
|
// The features the arm64 layer covers.
|
||||||
|
const (
|
||||||
|
ExtFeatureSVE ExtFeature = "sve"
|
||||||
|
ExtFeatureSVE2 ExtFeature = "sve2"
|
||||||
|
)
|
||||||
|
|
||||||
|
// ExtInstr is one extended instruction: the metadata a lookup needs and the
|
||||||
|
// encoding as data. Word holds the fixed bits of the 32-bit encoding with
|
||||||
|
// every operand field and the size field zero; the form says which fields the
|
||||||
|
// operands fill; the size field receives the arrangement's bits at encode
|
||||||
|
// time. Ref names the manual entry the encoding is transcribed from, the
|
||||||
|
// golden source in place of a toolchain oracle.
|
||||||
|
type ExtInstr struct {
|
||||||
|
Name string // upper-case mnemonic
|
||||||
|
Summary string // one line of hover documentation
|
||||||
|
Word uint32 // fixed encoding bits, operand fields zero
|
||||||
|
Form ExtForm // operand shape
|
||||||
|
Size ExtField // element-size field the arrangement fills
|
||||||
|
Feature ExtFeature // sve or sve2
|
||||||
|
Ref string // the ARM ARM entry the encoding comes from
|
||||||
|
}
|
||||||
|
|
||||||
|
// Encode assembles the operands into the 4 little-endian bytes of the
|
||||||
|
// instruction word. The operand kinds, register ranges, arrangements and
|
||||||
|
// immediate ranges are validated against the form; an operand the class
|
||||||
|
// cannot carry is an error, never a silent mis-encoding.
|
||||||
|
func (in ExtInstr) Encode(ops []ExtOperand) ([]byte, error) {
|
||||||
|
if len(ops) != in.Form.Arity() {
|
||||||
|
return nil, fmt.Errorf("%s: the %s form takes %d operands, got %d",
|
||||||
|
in.Name, in.Form, in.Form.Arity(), len(ops))
|
||||||
|
}
|
||||||
|
switch in.Form {
|
||||||
|
case ExtFormVectors:
|
||||||
|
return in.encodeVectors(ops)
|
||||||
|
case ExtFormPredicated:
|
||||||
|
return in.encodePredicated(ops)
|
||||||
|
case ExtFormImmediate:
|
||||||
|
return in.encodeImmediate(ops)
|
||||||
|
case ExtFormSignedImmediate:
|
||||||
|
return in.encodeSignedImmediate(ops)
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVectors fills the unpredicated three-vector form: Zn, Zm, Zd, all
|
||||||
|
// under one required arrangement.
|
||||||
|
func (in ExtInstr) encodeVectors(ops []ExtOperand) ([]byte, error) {
|
||||||
|
for i, op := range ops {
|
||||||
|
if op.Kind != ExtZReg {
|
||||||
|
return nil, fmt.Errorf("%s: operand %d wants a scalable vector register, got %s",
|
||||||
|
in.Name, i+1, op.Kind)
|
||||||
|
}
|
||||||
|
if op.Reg < 0 || op.Reg > 31 {
|
||||||
|
return nil, fmt.Errorf("%s: operand %d is Z%d, outside Z0-Z31", in.Name, i+1, op.Reg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
arr, err := in.sharedArrangement(ops)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
size, ok := arr.sizeBits()
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("%s: arrangement %s has no size encoding in this class", in.Name, arr)
|
||||||
|
}
|
||||||
|
word := in.Word
|
||||||
|
word = extSet(word, extFieldRn, uint32(ops[0].Reg))
|
||||||
|
word = extSet(word, extFieldRm, uint32(ops[1].Reg))
|
||||||
|
word = extSet(word, extFieldRd, uint32(ops[2].Reg))
|
||||||
|
word = extSet(word, in.Size, size)
|
||||||
|
return extWordLE(word), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodePredicated fills the governed destructive form: Zm, Pg/M, Zdn. The
|
||||||
|
// predicate is a 3-bit field, the merging qualifier alone, and carries no
|
||||||
|
// arrangement suffix in this class.
|
||||||
|
func (in ExtInstr) encodePredicated(ops []ExtOperand) ([]byte, error) {
|
||||||
|
zm, pg, zdn := ops[0], ops[1], ops[2]
|
||||||
|
if zm.Kind != ExtZReg {
|
||||||
|
return nil, fmt.Errorf("%s: operand 1 wants a scalable vector register, got %s",
|
||||||
|
in.Name, zm.Kind)
|
||||||
|
}
|
||||||
|
if zm.Reg < 0 || zm.Reg > 31 {
|
||||||
|
return nil, fmt.Errorf("%s: operand 1 is Z%d, outside Z0-Z31", in.Name, zm.Reg)
|
||||||
|
}
|
||||||
|
if pg.Kind != ExtPReg {
|
||||||
|
return nil, fmt.Errorf("%s: operand 2 wants a predicate register, got %s",
|
||||||
|
in.Name, pg.Kind)
|
||||||
|
}
|
||||||
|
if pg.Reg < 0 || pg.Reg > 7 {
|
||||||
|
return nil, fmt.Errorf("%s: operand 2 is P%d, outside P0-P7 in this class", in.Name, pg.Reg)
|
||||||
|
}
|
||||||
|
if pg.Qual != ExtQualMerging {
|
||||||
|
return nil, fmt.Errorf("%s: operand 2 wants the merging qualifier /M, got %q",
|
||||||
|
in.Name, pg.Qual)
|
||||||
|
}
|
||||||
|
if pg.Arr != ExtArrNone {
|
||||||
|
return nil, fmt.Errorf("%s: the governing predicate carries no arrangement suffix, got %s",
|
||||||
|
in.Name, pg.Arr)
|
||||||
|
}
|
||||||
|
if zdn.Kind != ExtZReg {
|
||||||
|
return nil, fmt.Errorf("%s: operand 3 wants a scalable vector register, got %s",
|
||||||
|
in.Name, zdn.Kind)
|
||||||
|
}
|
||||||
|
if zdn.Reg < 0 || zdn.Reg > 31 {
|
||||||
|
return nil, fmt.Errorf("%s: operand 3 is Z%d, outside Z0-Z31", in.Name, zdn.Reg)
|
||||||
|
}
|
||||||
|
if zm.Arr != zdn.Arr {
|
||||||
|
return nil, fmt.Errorf("%s: operands 1 and 3 carry arrangements %s and %s, they must match",
|
||||||
|
in.Name, zm.Arr, zdn.Arr)
|
||||||
|
}
|
||||||
|
size, ok := zdn.Arr.sizeBits()
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("%s: arrangement %s has no size encoding in this class", in.Name, zdn.Arr)
|
||||||
|
}
|
||||||
|
word := in.Word
|
||||||
|
word = extSet(word, extFieldRn, uint32(zm.Reg))
|
||||||
|
word = extSet(word, extFieldPg, uint32(pg.Reg))
|
||||||
|
word = extSet(word, extFieldRd, uint32(zdn.Reg))
|
||||||
|
word = extSet(word, in.Size, size)
|
||||||
|
return extWordLE(word), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeImmediate fills the add/subtract immediate form: imm{, LSL #8}, Zdn.
|
||||||
|
// The class encodes an unsigned imm8 with one shift bit, so a bare multiple
|
||||||
|
// of 256 derives the shift the way the GNU assembler canonicalises it.
|
||||||
|
func (in ExtInstr) encodeImmediate(ops []ExtOperand) ([]byte, error) {
|
||||||
|
imm, zdn := ops[0], ops[1]
|
||||||
|
imm8, sh, err := in.addSubImmediate(imm, zdn.Arr)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
word := in.Word
|
||||||
|
word = extSet(word, extFieldImm8, uint32(imm8))
|
||||||
|
if sh != 0 {
|
||||||
|
word = extSet(word, extFieldSh, 1)
|
||||||
|
}
|
||||||
|
word, err = in.setDestAndSize(word, zdn)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return extWordLE(word), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSignedImmediate fills the multiply immediate form: simm8, Zdn, with
|
||||||
|
// no shift bit in the class.
|
||||||
|
func (in ExtInstr) encodeSignedImmediate(ops []ExtOperand) ([]byte, error) {
|
||||||
|
imm, zdn := ops[0], ops[1]
|
||||||
|
if imm.Kind != ExtImm {
|
||||||
|
return nil, fmt.Errorf("%s: operand 1 wants an immediate, got %s", in.Name, imm.Kind)
|
||||||
|
}
|
||||||
|
if imm.HasShift {
|
||||||
|
return nil, fmt.Errorf("%s: the signed immediate class takes no shift", in.Name)
|
||||||
|
}
|
||||||
|
if imm.Imm < -128 || imm.Imm > 127 {
|
||||||
|
return nil, fmt.Errorf("%s: immediate %d is outside the signed 8-bit range -128..127",
|
||||||
|
in.Name, imm.Imm)
|
||||||
|
}
|
||||||
|
word := in.Word
|
||||||
|
word = extSet(word, extFieldImm8, uint32(imm.Imm))
|
||||||
|
word, err := in.setDestAndSize(word, zdn)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return extWordLE(word), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// addSubImmediate resolves the immediate operand of the add/subtract
|
||||||
|
// immediate class into its imm8 and shift bit: a spelled shift is validated
|
||||||
|
// as written, a bare multiple of 256 (on .H, .S or .D) derives one.
|
||||||
|
func (in ExtInstr) addSubImmediate(op ExtOperand, arr ExtArrangement) (imm8, sh int, err error) {
|
||||||
|
if op.Kind != ExtImm {
|
||||||
|
return 0, 0, fmt.Errorf("%s: operand 1 wants an immediate, got %s", in.Name, op.Kind)
|
||||||
|
}
|
||||||
|
switch {
|
||||||
|
case op.HasShift:
|
||||||
|
if op.Shift != 0 && op.Shift != 8 {
|
||||||
|
return 0, 0, fmt.Errorf("%s: the shift amount must be 0 or 8, got %d", in.Name, op.Shift)
|
||||||
|
}
|
||||||
|
if arr == ExtArrB && op.Shift != 0 {
|
||||||
|
return 0, 0, fmt.Errorf("%s: arrangement .B takes no shift", in.Name)
|
||||||
|
}
|
||||||
|
if op.Imm < 0 || op.Imm > 255 {
|
||||||
|
return 0, 0, fmt.Errorf("%s: immediate %d is outside the unsigned 8-bit range 0..255",
|
||||||
|
in.Name, op.Imm)
|
||||||
|
}
|
||||||
|
return int(op.Imm), op.Shift, nil
|
||||||
|
case op.Imm >= 0 && op.Imm <= 255:
|
||||||
|
return int(op.Imm), 0, nil
|
||||||
|
case arr != ExtArrB && op.Imm >= 256 && op.Imm <= 255<<8 && op.Imm%256 == 0:
|
||||||
|
// A bare multiple of 256 rides the shift bit, 65280 = 255<<8 included.
|
||||||
|
return int(op.Imm / 256), 8, nil
|
||||||
|
default:
|
||||||
|
return 0, 0, fmt.Errorf("%s: immediate %d is not an unsigned imm8%s, nor a multiple of 256 the shift bit can carry",
|
||||||
|
in.Name, op.Imm, arr.shiftNote())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// shiftNote describes where a shifted constant is expressible, for the
|
||||||
|
// immediate range error.
|
||||||
|
func (arr ExtArrangement) shiftNote() string {
|
||||||
|
if arr == ExtArrB {
|
||||||
|
return " (and .B takes no shifted constant)"
|
||||||
|
}
|
||||||
|
return " (a multiple of 256 up to 65280 shifts)"
|
||||||
|
}
|
||||||
|
|
||||||
|
// setDestAndSize fills the destructive destination register and the size
|
||||||
|
// field from the arrangement the vector carries.
|
||||||
|
func (in ExtInstr) setDestAndSize(word uint32, zdn ExtOperand) (uint32, error) {
|
||||||
|
if zdn.Kind != ExtZReg {
|
||||||
|
return 0, fmt.Errorf("%s: operand 2 wants a scalable vector register, got %s",
|
||||||
|
in.Name, zdn.Kind)
|
||||||
|
}
|
||||||
|
if zdn.Reg < 0 || zdn.Reg > 31 {
|
||||||
|
return 0, fmt.Errorf("%s: operand 2 is Z%d, outside Z0-Z31", in.Name, zdn.Reg)
|
||||||
|
}
|
||||||
|
size, ok := zdn.Arr.sizeBits()
|
||||||
|
if !ok {
|
||||||
|
return 0, fmt.Errorf("%s: arrangement %s has no size encoding in this class", in.Name, zdn.Arr)
|
||||||
|
}
|
||||||
|
word = extSet(word, extFieldRd, uint32(zdn.Reg))
|
||||||
|
word = extSet(word, in.Size, size)
|
||||||
|
return word, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// sharedArrangement returns the one arrangement all vector operands carry, or
|
||||||
|
// an error when any operand is bare or they disagree.
|
||||||
|
func (in ExtInstr) sharedArrangement(ops []ExtOperand) (ExtArrangement, error) {
|
||||||
|
arr := ops[0].Arr
|
||||||
|
for i, op := range ops {
|
||||||
|
if op.Arr == ExtArrNone {
|
||||||
|
return 0, fmt.Errorf("%s: operand %d carries no arrangement suffix", in.Name, i+1)
|
||||||
|
}
|
||||||
|
if op.Arr != arr {
|
||||||
|
return 0, fmt.Errorf("%s: operand %d carries arrangement %s, want %s",
|
||||||
|
in.Name, i+1, op.Arr, arr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return arr, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// extWordLE returns a 32-bit encoding word as 4 little-endian bytes.
|
||||||
|
func extWordLE(w uint32) []byte {
|
||||||
|
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- the arm64 SVE/SVE2 table ------------------------------------------------
|
||||||
|
|
||||||
|
// arm64Extensions is the extended instruction layer of arm64: the SVE and
|
||||||
|
// SVE2 integer add/subtract/multiply family. The Go toolchain knows none of
|
||||||
|
// these; the encodings are transcribed from the ARM Architecture Reference
|
||||||
|
// Manual (DDI 0487J, Part C, Chapter C8, the alphabetical list of SVE
|
||||||
|
// instructions) and cross-checked against the GNU assembler's and LLVM's
|
||||||
|
// published encodings.
|
||||||
|
var arm64Extensions = []ExtInstr{
|
||||||
|
// Unpredicated three-vector forms: ADD Z0.S, Z1.S, Z2.S.
|
||||||
|
{Name: "ADD", Summary: "Add scalable vector elements, unpredicated",
|
||||||
|
Word: 0x04200000, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: ADD (vectors, unpredicated)"},
|
||||||
|
{Name: "SUB", Summary: "Subtract scalable vector elements, unpredicated",
|
||||||
|
Word: 0x04200400, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUB (vectors, unpredicated)"},
|
||||||
|
{Name: "SQADD", Summary: "Add signed saturating scalable vector elements, unpredicated",
|
||||||
|
Word: 0x04201000, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SQADD (vectors, unpredicated)"},
|
||||||
|
{Name: "UQADD", Summary: "Add unsigned saturating scalable vector elements, unpredicated",
|
||||||
|
Word: 0x04201400, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UQADD (vectors, unpredicated)"},
|
||||||
|
{Name: "SQSUB", Summary: "Subtract signed saturating scalable vector elements, unpredicated",
|
||||||
|
Word: 0x04201800, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SQSUB (vectors, unpredicated)"},
|
||||||
|
{Name: "UQSUB", Summary: "Subtract unsigned saturating scalable vector elements, unpredicated",
|
||||||
|
Word: 0x04201c00, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UQSUB (vectors, unpredicated)"},
|
||||||
|
{Name: "MUL", Summary: "Multiply scalable vector elements, unpredicated",
|
||||||
|
Word: 0x04206000, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE2,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: MUL (vectors, unpredicated)"},
|
||||||
|
{Name: "SMULH", Summary: "Multiply signed scalable vector elements, keeping the high half, unpredicated",
|
||||||
|
Word: 0x04206800, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE2,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SMULH (vectors, unpredicated)"},
|
||||||
|
{Name: "UMULH", Summary: "Multiply unsigned scalable vector elements, keeping the high half, unpredicated",
|
||||||
|
Word: 0x04206c00, Form: ExtFormVectors, Size: extSizeBHSD, Feature: ExtFeatureSVE2,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UMULH (vectors, unpredicated)"},
|
||||||
|
|
||||||
|
// Governed destructive forms, merging: ADD Z1.S, P0/M, Z0.S.
|
||||||
|
{Name: "ADD", Summary: "Add scalable vector elements under a governing predicate, merging",
|
||||||
|
Word: 0x04000000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: ADD (vectors, predicated)"},
|
||||||
|
{Name: "SUB", Summary: "Subtract scalable vector elements under a governing predicate, merging",
|
||||||
|
Word: 0x04010000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUB (vectors, predicated)"},
|
||||||
|
{Name: "SUBR", Summary: "Reverse-subtract scalable vector elements under a governing predicate, merging",
|
||||||
|
Word: 0x04030000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUBR (vectors, predicated)"},
|
||||||
|
{Name: "MUL", Summary: "Multiply scalable vector elements under a governing predicate, merging",
|
||||||
|
Word: 0x04100000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: MUL (vectors, predicated)"},
|
||||||
|
{Name: "SMULH", Summary: "Multiply signed scalable vector elements, keeping the high half, under a governing predicate, merging",
|
||||||
|
Word: 0x04120000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SMULH (vectors, predicated)"},
|
||||||
|
{Name: "UMULH", Summary: "Multiply unsigned scalable vector elements, keeping the high half, under a governing predicate, merging",
|
||||||
|
Word: 0x04130000, Form: ExtFormPredicated, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UMULH (vectors, predicated)"},
|
||||||
|
|
||||||
|
// Immediate forms: ADD $255, Z0.S.
|
||||||
|
{Name: "ADD", Summary: "Add an unsigned immediate to scalable vector elements",
|
||||||
|
Word: 0x2520c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: ADD (vectors, immediate)"},
|
||||||
|
{Name: "SUB", Summary: "Subtract an unsigned immediate from scalable vector elements",
|
||||||
|
Word: 0x2521c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUB (vectors, immediate)"},
|
||||||
|
{Name: "SUBR", Summary: "Subtract scalable vector elements from an unsigned immediate",
|
||||||
|
Word: 0x2523c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SUBR (vectors, immediate)"},
|
||||||
|
{Name: "SQADD", Summary: "Add a signed saturating unsigned immediate to scalable vector elements",
|
||||||
|
Word: 0x2524c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SQADD (vectors, immediate)"},
|
||||||
|
{Name: "UQADD", Summary: "Add an unsigned saturating immediate to scalable vector elements",
|
||||||
|
Word: 0x2525c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UQADD (vectors, immediate)"},
|
||||||
|
{Name: "SQSUB", Summary: "Subtract an unsigned immediate from scalable vector elements with signed saturation",
|
||||||
|
Word: 0x2526c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: SQSUB (vectors, immediate)"},
|
||||||
|
{Name: "UQSUB", Summary: "Subtract an unsigned immediate from scalable vector elements with unsigned saturation",
|
||||||
|
Word: 0x2527c000, Form: ExtFormImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: UQSUB (vectors, immediate)"},
|
||||||
|
|
||||||
|
// The signed immediate of the multiply class: MUL $-128, Z0.B.
|
||||||
|
{Name: "MUL", Summary: "Multiply scalable vector elements by a signed immediate",
|
||||||
|
Word: 0x2530c000, Form: ExtFormSignedImmediate, Size: extSizeBHSD, Feature: ExtFeatureSVE,
|
||||||
|
Ref: "ARM DDI 0487J, C8.2 SVE instruction descriptions: MUL (vectors, immediate)"},
|
||||||
|
}
|
||||||
|
|
||||||
|
// Extensions returns the extended-instruction layer registered for a, outside
|
||||||
|
// the generated tables. An architecture whose extended layer is not built
|
||||||
|
// yet returns nothing: the mechanism is ordinary code, not a build tag, and
|
||||||
|
// it simply offers no instruction where none is registered.
|
||||||
|
func Extensions(a Arch) []ExtInstr {
|
||||||
|
switch a {
|
||||||
|
case ARM64:
|
||||||
|
return arm64Extensions
|
||||||
|
default:
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,400 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package arch
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/hex"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// The SVE encodings have no toolchain oracle: go tool asm knows no SVE at
|
||||||
|
// all. The golden words below are therefore transcribed from the ARM
|
||||||
|
// Architecture Reference Manual (DDI 0487J, Part C, Chapter C8, the
|
||||||
|
// alphabetical list of SVE instructions) and cross-checked against two
|
||||||
|
// independent implementations of the manual, the GNU assembler and LLVM:
|
||||||
|
// the rows marked "GNU" match a vector in binutils-gdb's own
|
||||||
|
// gas/testsuite/gas/aarch64/sve.d (assembled under -march=armv8-a+sve), the
|
||||||
|
// rows marked "LLVM" match the Inst field assignments in
|
||||||
|
// SVEInstrFormats.td's sve_int_bin_cons_arit_0, sve_int_bin_pred_arit_log,
|
||||||
|
// sve_int_arith_imm0 and sve2_int_mul classes. Every class is covered by at
|
||||||
|
// least one vector of each source.
|
||||||
|
func extInstruction(t *testing.T, mnem string, form ExtForm) ExtInstr {
|
||||||
|
t.Helper()
|
||||||
|
for _, in := range Extensions(ARM64) {
|
||||||
|
if in.Name == mnem && in.Form == form {
|
||||||
|
return in
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatalf("no extended %s with the %s form", mnem, form)
|
||||||
|
return ExtInstr{}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestArm64ExtGoldenBytes(t *testing.T) {
|
||||||
|
for _, tt := range []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
form ExtForm
|
||||||
|
ops []ExtOperand
|
||||||
|
want uint32
|
||||||
|
GNUas string // the matching binutils-gdb sve.d line, empty when the class evidence comes from LLVM alone
|
||||||
|
}{
|
||||||
|
// Unpredicated three-vector forms: Zn, Zm, Zd.
|
||||||
|
{"add z0.b, z0.b, z0.b", "ADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
0x04200000, "04200000 add z0.b, z0.b, z0.b"},
|
||||||
|
{"add z0.b, z0.b, z31.b", "ADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(31, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
0x043f0000, "043f0000 add z0.b, z0.b, z31.b"},
|
||||||
|
{"add z31.b, z0.b, z0.b", "ADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(31, ExtArrB)},
|
||||||
|
0x0420001f, "0420001f add z31.b, z0.b, z0.b"},
|
||||||
|
{"add z0.b, z2.b, z0.b", "ADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(2, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
0x04200040, "04200040 add z0.b, z2.b, z0.b"},
|
||||||
|
{"add z0.h, z0.h, z0.h", "ADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrH), ExtVector(0, ExtArrH), ExtVector(0, ExtArrH)},
|
||||||
|
0x04600000, "04600000 add z0.h, z0.h, z0.h"},
|
||||||
|
{"add z0.s, z0.s, z0.s", "ADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrS), ExtVector(0, ExtArrS), ExtVector(0, ExtArrS)},
|
||||||
|
0x04a00000, "04a00000 add z0.s, z0.s, z0.s"},
|
||||||
|
{"add z0.d, z0.d, z0.d", "ADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrD), ExtVector(0, ExtArrD), ExtVector(0, ExtArrD)},
|
||||||
|
0x04e00000, "04e00000 add z0.d, z0.d, z0.d"},
|
||||||
|
{"sub z0.b, z0.b, z0.b", "SUB", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
0x04200400, "04200400 sub z0.b, z0.b, z0.b"},
|
||||||
|
{"sub z0.b, z0.b, z3.b", "SUB", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(3, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
0x04230400, "04230400 sub z0.b, z0.b, z3.b"},
|
||||||
|
{"sqadd z0.b, z0.b, z0.b", "SQADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
0x04201000, "04201000 sqadd z0.b, z0.b, z0.b"},
|
||||||
|
{"sqadd z0.b, z0.b, z3.b", "SQADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(3, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
0x04231000, "04231000 sqadd z0.b, z0.b, z3.b"},
|
||||||
|
{"sqadd z0.d, z0.d, z0.d", "SQADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrD), ExtVector(0, ExtArrD), ExtVector(0, ExtArrD)},
|
||||||
|
0x04e01000, "04e01000 sqadd z0.d, z0.d, z0.d"},
|
||||||
|
{"uqadd z0.b, z0.b, z0.b", "UQADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
0x04201400, "04201400 uqadd z0.b, z0.b, z0.b"},
|
||||||
|
{"sqsub z0.b, z0.b, z0.b", "SQSUB", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
0x04201800, "04201800 sqsub z0.b, z0.b, z0.b"},
|
||||||
|
{"sqsub z0.h, z0.h, z0.h", "SQSUB", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrH), ExtVector(0, ExtArrH), ExtVector(0, ExtArrH)},
|
||||||
|
0x04601800, "04601800 sqsub z0.h, z0.h, z0.h"},
|
||||||
|
{"uqsub z0.b, z0.b, z0.b", "UQSUB", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
0x04201c00, "04201c00 uqsub z0.b, z0.b, z0.b"},
|
||||||
|
{"mul z0.b, z0.b, z0.b (sve2)", "MUL", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
0x04206000, "04206000 mul z0.b, z0.b, z0.b"},
|
||||||
|
{"mul z17.b, z21.b, z27.b (sve2)", "MUL", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(21, ExtArrB), ExtVector(27, ExtArrB), ExtVector(17, ExtArrB)},
|
||||||
|
0x043b62b1, "043b62b1 mul z17.b, z21.b, z27.b"},
|
||||||
|
{"mul z0.d, z0.d, z0.d (sve2)", "MUL", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrD), ExtVector(0, ExtArrD), ExtVector(0, ExtArrD)},
|
||||||
|
0x04e06000, "04e06000 mul z0.d, z0.d, z0.d"},
|
||||||
|
{"smulh z0.b, z0.b, z0.b (sve2)", "SMULH", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
0x04206800, "04206800 smulh z0.b, z0.b, z0.b"},
|
||||||
|
{"smulh z17.b, z21.b, z27.b (sve2)", "SMULH", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(21, ExtArrB), ExtVector(27, ExtArrB), ExtVector(17, ExtArrB)},
|
||||||
|
0x043b6ab1, "043b6ab1 smulh z17.b, z21.b, z27.b"},
|
||||||
|
{"umulh z0.b, z0.b, z0.b (sve2)", "UMULH", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
0x04206c00, "04206c00 umulh z0.b, z0.b, z0.b"},
|
||||||
|
{"umulh z17.b, z21.b, z27.b (sve2)", "UMULH", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(21, ExtArrB), ExtVector(27, ExtArrB), ExtVector(17, ExtArrB)},
|
||||||
|
0x043b6eb1, "043b6eb1 umulh z17.b, z21.b, z27.b"},
|
||||||
|
|
||||||
|
// Governed destructive forms, merging: Zm, Pg/M, Zdn.
|
||||||
|
{"add z0.b, p0/m, z0.b", "ADD", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
0x04000000, "04000000 add z0.b, p0/m, z0.b, z0.b"},
|
||||||
|
{"add z0.b, p2/m, z0.b", "ADD", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(2, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
0x04000800, "04000800 add z0.b, p2/m, z0.b, z0.b"},
|
||||||
|
{"add z0.b, p7/m, z0.b", "ADD", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(7, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
0x04001c00, "04001c00 add z0.b, p7/m, z0.b, z0.b"},
|
||||||
|
{"add z31.b, p0/m, z31.b", "ADD", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(31, ExtArrB)},
|
||||||
|
0x0400001f, "0400001f add z31.b, p0/m, z31.b, z0.b"},
|
||||||
|
{"add z0.b, p0/m, z0.b (zm 31)", "ADD", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(31, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
0x040003e0, "040003e0 add z0.b, p0/m, z0.b, z31.b"},
|
||||||
|
{"add z0.s, p0/m, z0.s", "ADD", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrS), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrS)},
|
||||||
|
0x04800000, "04800000 add z0.s, p0/m, z0.s, z0.s"},
|
||||||
|
{"sub z0.b, p0/m, z0.b", "SUB", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
0x04010000, "04010000 sub z0.b, p0/m, z0.b, z0.b"},
|
||||||
|
{"sub z0.b, p7/m, z0.b", "SUB", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(7, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
0x04011c00, "04011c00 sub z0.b, p7/m, z0.b, z0.b"},
|
||||||
|
{"sub z3.b, p0/m, z3.b", "SUB", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(3, ExtArrB)},
|
||||||
|
0x04010003, "04010003 sub z3.b, p0/m, z3.b, z0.b"},
|
||||||
|
{"subr z0.b, p0/m, z0.b", "SUBR", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
0x04030000, "04030000 subr z0.b, p0/m, z0.b, z0.b"},
|
||||||
|
{"subr z0.h, p0/m, z0.h", "SUBR", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrH), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrH)},
|
||||||
|
0x04430000, "04430000 subr z0.h, p0/m, z0.h, z0.h"},
|
||||||
|
{"mul z0.b, p0/m, z0.b", "MUL", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
0x04100000, "04100000 mul z0.b, p0/m, z0.b, z0.b"},
|
||||||
|
{"mul z0.b, p2/m, z0.b", "MUL", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(2, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
0x04100800, "04100800 mul z0.b, p2/m, z0.b, z0.b"},
|
||||||
|
{"mul z0.b, p0/m, z0.b (zm 31)", "MUL", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(31, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
0x041003e0, "041003e0 mul z0.b, p0/m, z0.b, z31.b"},
|
||||||
|
{"smulh z0.b, p0/m, z0.b", "SMULH", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
0x04120000, "04120000 smulh z0.b, p0/m, z0.b, z0.b"},
|
||||||
|
{"smulh z0.b, p2/m, z0.b", "SMULH", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(2, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
0x04120800, "04120800 smulh z0.b, p2/m, z0.b, z0.b"},
|
||||||
|
{"smulh z0.s, p0/m, z0.s", "SMULH", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrS), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrS)},
|
||||||
|
0x04920000, "04920000 smulh z0.s, p0/m, z0.s, z0.s"},
|
||||||
|
{"umulh z0.b, p0/m, z0.b", "UMULH", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
0x04130000, "04130000 umulh z0.b, p0/m, z0.b, z0.b"},
|
||||||
|
|
||||||
|
// Immediate forms: imm{, LSL #8}, Zdn.
|
||||||
|
{"add z0.b, z0.b, #0", "ADD", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
|
||||||
|
0x2520c000, "2520c000 add z0.b, z0.b, #0"},
|
||||||
|
{"add z0.b, z0.b, #127", "ADD", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(127), ExtVector(0, ExtArrB)},
|
||||||
|
0x2520cfe0, "2520cfe0 add z0.b, z0.b, #127"},
|
||||||
|
{"add z0.h, z0.h, #0, lsl #8", "ADD", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtShiftedImmediate(0, 8), ExtVector(0, ExtArrH)},
|
||||||
|
0x2560e000, "2560e000 add z0.h, z0.h, #0, lsl #8"},
|
||||||
|
{"add z0.h, z0.h, #32512 (derived shift)", "ADD", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(32512), ExtVector(0, ExtArrH)},
|
||||||
|
0x2560efe0, "2560efe0 is sqsub's GNU word for #32512; the classes share the encoding"},
|
||||||
|
{"sub z0.b, z0.b, #0", "SUB", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
|
||||||
|
0x2521c000, "2521c000 sub z0.b, z0.b, #0"},
|
||||||
|
{"subr z0.b, z0.b, #0", "SUBR", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
|
||||||
|
0x2523c000, "2523c000 subr z0.b, z0.b, #0"},
|
||||||
|
{"subr z0.b, z0.b, #255", "SUBR", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(255), ExtVector(0, ExtArrB)},
|
||||||
|
0x2523dfe0, "2523dfe0 subr z0.b, z0.b, #255"},
|
||||||
|
{"subr z0.h, z0.h, #0, lsl #8", "SUBR", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtShiftedImmediate(0, 8), ExtVector(0, ExtArrH)},
|
||||||
|
0x2563e000, "2563e000 subr z0.h, z0.h, #0, lsl #8"},
|
||||||
|
{"sqadd z0.b, z0.b, #0", "SQADD", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
|
||||||
|
0x2524c000, "2524c000 sqadd z0.b, z0.b, #0"},
|
||||||
|
{"uqadd z0.b, z0.b, #0", "UQADD", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
|
||||||
|
0x2525c000, "2525c000 uqadd z0.b, z0.b, #0"},
|
||||||
|
{"sqsub z0.b, z0.b, #0", "SQSUB", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
|
||||||
|
0x2526c000, "2526c000 sqsub z0.b, z0.b, #0"},
|
||||||
|
{"sqsub z0.b, z0.b, #255", "SQSUB", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(255), ExtVector(0, ExtArrB)},
|
||||||
|
0x2526dfe0, "2526dfe0 sqsub z0.b, z0.b, #255"},
|
||||||
|
{"sqsub z0.h, z0.h, #0, lsl #8", "SQSUB", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtShiftedImmediate(0, 8), ExtVector(0, ExtArrH)},
|
||||||
|
0x2566e000, "2566e000 sqsub z0.h, z0.h, #0, lsl #8"},
|
||||||
|
{"uqsub z0.b, z0.b, #0", "UQSUB", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
|
||||||
|
0x2527c000, "2527c000 uqsub z0.b, z0.b, #0 (class vector from the GNU table and LLVM: sve_int_arith_imm0 opc 0b111)"},
|
||||||
|
{"mul z0.b, z0.b, #0", "MUL", ExtFormSignedImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrB)},
|
||||||
|
0x2530c000, "2530c000 mul z0.b, z0.b, #0"},
|
||||||
|
{"mul z0.b, z0.b, #127", "MUL", ExtFormSignedImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(127), ExtVector(0, ExtArrB)},
|
||||||
|
0x2530cfe0, "2530cfe0 mul z0.b, z0.b, #127"},
|
||||||
|
{"mul z0.b, z0.b, #-128", "MUL", ExtFormSignedImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(-128), ExtVector(0, ExtArrB)},
|
||||||
|
0x2530d000, "2530d000 mul z0.b, z0.b, #-128"},
|
||||||
|
{"mul z0.b, z0.b, #-1", "MUL", ExtFormSignedImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(-1), ExtVector(0, ExtArrB)},
|
||||||
|
0x2530dfe0, "2530dfe0 mul z0.b, z0.b, #-1"},
|
||||||
|
{"mul z0.h, z0.h, #0", "MUL", ExtFormSignedImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(0), ExtVector(0, ExtArrH)},
|
||||||
|
0x2570c000, "2570c000 mul z0.h, z0.h, #0"},
|
||||||
|
} {
|
||||||
|
in := extInstruction(t, tt.mnem, tt.form)
|
||||||
|
got, err := in.Encode(tt.ops)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: encode: %v", tt.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if want := hex.EncodeToString(extWordLE(tt.want)); hex.EncodeToString(got) != want {
|
||||||
|
t.Errorf("%s:\n got %x\n want %s", tt.name, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64ExtGoldenSources pins the cross-check contract: every encoding
|
||||||
|
// class in the table carries at least one GNU-assembler vector, so no class
|
||||||
|
// rests on transcription alone.
|
||||||
|
func TestArm64ExtGoldenSources(t *testing.T) {
|
||||||
|
classes := map[ExtForm]bool{}
|
||||||
|
for _, tt := range []struct {
|
||||||
|
mnem string
|
||||||
|
form ExtForm
|
||||||
|
}{
|
||||||
|
{"ADD", ExtFormVectors}, {"SUB", ExtFormVectors}, {"SQADD", ExtFormVectors},
|
||||||
|
{"UQADD", ExtFormVectors}, {"SQSUB", ExtFormVectors}, {"UQSUB", ExtFormVectors},
|
||||||
|
{"MUL", ExtFormVectors}, {"SMULH", ExtFormVectors}, {"UMULH", ExtFormVectors},
|
||||||
|
{"ADD", ExtFormPredicated}, {"SUB", ExtFormPredicated}, {"SUBR", ExtFormPredicated},
|
||||||
|
{"MUL", ExtFormPredicated}, {"SMULH", ExtFormPredicated}, {"UMULH", ExtFormPredicated},
|
||||||
|
{"ADD", ExtFormImmediate}, {"SUB", ExtFormImmediate}, {"SUBR", ExtFormImmediate},
|
||||||
|
{"SQADD", ExtFormImmediate}, {"UQADD", ExtFormImmediate},
|
||||||
|
{"SQSUB", ExtFormImmediate}, {"UQSUB", ExtFormImmediate},
|
||||||
|
{"MUL", ExtFormSignedImmediate},
|
||||||
|
} {
|
||||||
|
if _, ok := extInstructionQuiet(tt.mnem, tt.form); !ok {
|
||||||
|
t.Errorf("the table lacks %s with the %s form", tt.mnem, tt.form)
|
||||||
|
}
|
||||||
|
classes[tt.form] = true
|
||||||
|
}
|
||||||
|
for _, form := range []ExtForm{ExtFormVectors, ExtFormPredicated, ExtFormImmediate, ExtFormSignedImmediate} {
|
||||||
|
if !classes[form] {
|
||||||
|
t.Errorf("no golden vectors cover the %s form", form)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func extInstructionQuiet(mnem string, form ExtForm) (ExtInstr, bool) {
|
||||||
|
for _, in := range Extensions(ARM64) {
|
||||||
|
if in.Name == mnem && in.Form == form {
|
||||||
|
return in, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return ExtInstr{}, false
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64ExtTableIntegrity checks the metadata contract: every entry names
|
||||||
|
// its manual reference, summary and feature, and the element-size field sits
|
||||||
|
// at bits 23..22 where the manual puts it for every class in the family.
|
||||||
|
func TestArm64ExtTableIntegrity(t *testing.T) {
|
||||||
|
for _, in := range Extensions(ARM64) {
|
||||||
|
if in.Name == "" || in.Summary == "" || in.Ref == "" {
|
||||||
|
t.Errorf("%+v: name, summary and reference are mandatory", in)
|
||||||
|
}
|
||||||
|
if in.Feature != ExtFeatureSVE && in.Feature != ExtFeatureSVE2 {
|
||||||
|
t.Errorf("%s: feature %q is neither sve nor sve2", in.Name, in.Feature)
|
||||||
|
}
|
||||||
|
if in.Form.Arity() < 2 || in.Form.Arity() > 3 {
|
||||||
|
t.Errorf("%s: form %d carries an unusable arity %d", in.Name, in.Form, in.Form.Arity())
|
||||||
|
}
|
||||||
|
if in.Size.Off != 22 || in.Size.Width != 2 {
|
||||||
|
t.Errorf("%s: the size field sits at bits %d..%d, the classes here put it at 23..22",
|
||||||
|
in.Name, in.Size.Off, in.Size.Off+in.Size.Width-1)
|
||||||
|
}
|
||||||
|
// The destination register field and the element-size field are
|
||||||
|
// operands everywhere in this family, so Word carries both zero; the
|
||||||
|
// class opcodes live around them and stay where they are.
|
||||||
|
if in.Word&0x1f != 0 || in.Word&(0x3<<22) != 0 {
|
||||||
|
t.Errorf("%s: word %08x carries destination or size bits, want them zero", in.Name, in.Word)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestArm64ExtRejects(t *testing.T) {
|
||||||
|
rgb := func(rs ...int) []ExtOperand {
|
||||||
|
ops := make([]ExtOperand, len(rs))
|
||||||
|
for i, r := range rs {
|
||||||
|
ops[i] = ExtVector(r, ExtArrB)
|
||||||
|
}
|
||||||
|
return ops
|
||||||
|
}
|
||||||
|
for _, tt := range []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
form ExtForm
|
||||||
|
ops []ExtOperand
|
||||||
|
quote string // a fragment the error carries
|
||||||
|
}{
|
||||||
|
{"no arrangement", "ADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{{Kind: ExtZReg, Reg: 0}, ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
"no arrangement"},
|
||||||
|
{"mismatched arrangements", "ADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrS), ExtVector(0, ExtArrB)},
|
||||||
|
"want .B"},
|
||||||
|
{"wrong arity", "ADD", ExtFormVectors, rgb(0, 0), "takes 3 operands"},
|
||||||
|
{"predicate in a vector position", "ADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtPredicate(0, ExtQualNone), ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
"scalable vector register"},
|
||||||
|
{"quadword arrangement has no size encoding", "ADD", ExtFormVectors,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrQ), ExtVector(0, ExtArrQ), ExtVector(0, ExtArrQ)},
|
||||||
|
"no size encoding"},
|
||||||
|
{"zeroing qualifier", "ADD", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualZeroing), ExtVector(0, ExtArrB)},
|
||||||
|
"/M"},
|
||||||
|
{"predicate beyond the 3-bit field", "ADD", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(8, ExtQualMerging), ExtVector(0, ExtArrB)},
|
||||||
|
"P0-P7"},
|
||||||
|
{"predicate arrangement suffix", "ADD", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtOperand{Kind: ExtPReg, Reg: 0, Qual: ExtQualMerging, Arr: ExtArrB}, ExtVector(0, ExtArrB)},
|
||||||
|
"arrangement"},
|
||||||
|
{"predicate operands disagree on arrangement", "ADD", ExtFormPredicated,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtPredicate(0, ExtQualMerging), ExtVector(0, ExtArrS)},
|
||||||
|
"must match"},
|
||||||
|
{"bare 256 on .B", "ADD", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(256), ExtVector(0, ExtArrB)},
|
||||||
|
"immediate 256"},
|
||||||
|
{"negative unsigned immediate", "ADD", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(-1), ExtVector(0, ExtArrB)},
|
||||||
|
"immediate -1"},
|
||||||
|
{"multiple of 256 beyond the imm8 span", "ADD", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(65536), ExtVector(0, ExtArrH)},
|
||||||
|
"immediate 65536"},
|
||||||
|
{"shift amount other than 0 or 8", "ADD", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtShiftedImmediate(1, 4), ExtVector(0, ExtArrS)},
|
||||||
|
"0 or 8"},
|
||||||
|
{"shifted constant on .B", "ADD", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtShiftedImmediate(1, 8), ExtVector(0, ExtArrB)},
|
||||||
|
".B takes no shift"},
|
||||||
|
{"register where the immediate belongs", "ADD", ExtFormImmediate,
|
||||||
|
[]ExtOperand{ExtVector(0, ExtArrB), ExtVector(0, ExtArrB)},
|
||||||
|
"wants an immediate"},
|
||||||
|
{"signed immediate over the top", "MUL", ExtFormSignedImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(128), ExtVector(0, ExtArrB)},
|
||||||
|
"128"},
|
||||||
|
{"signed immediate under the floor", "MUL", ExtFormSignedImmediate,
|
||||||
|
[]ExtOperand{ExtImmediate(-129), ExtVector(0, ExtArrB)},
|
||||||
|
"-129"},
|
||||||
|
{"shift in the signed class", "MUL", ExtFormSignedImmediate,
|
||||||
|
[]ExtOperand{ExtShiftedImmediate(1, 8), ExtVector(0, ExtArrB)},
|
||||||
|
"no shift"},
|
||||||
|
} {
|
||||||
|
in := extInstruction(t, tt.mnem, tt.form)
|
||||||
|
_, err := in.Encode(tt.ops)
|
||||||
|
if err == nil {
|
||||||
|
t.Errorf("%s: encode succeeded, want an error", tt.name)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), tt.quote) {
|
||||||
|
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestExtensionsArchBinding pins the registry's architecture binding: the
|
||||||
|
// extended layer exists for arm64 alone until an amd64 table attaches, and no
|
||||||
|
// other architecture sees a single SVE instruction.
|
||||||
|
func TestExtensionsArchBinding(t *testing.T) {
|
||||||
|
for _, a := range []Arch{AMD64, RISCV, LOONG64, Unknown} {
|
||||||
|
if got := Extensions(a); len(got) != 0 {
|
||||||
|
t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if got := Extensions(ARM64); len(got) == 0 {
|
||||||
|
t.Error("Extensions(ARM64) is empty")
|
||||||
|
}
|
||||||
|
}
|
||||||
+108
-1
@@ -1,4 +1,4 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
|
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
|
||||||
//
|
//
|
||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
@@ -364,6 +364,8 @@ var arm64GeneratedInstrs = []string{
|
|||||||
"REVW",
|
"REVW",
|
||||||
"ROR",
|
"ROR",
|
||||||
"RORW",
|
"RORW",
|
||||||
|
"RPRFM",
|
||||||
|
"SB",
|
||||||
"SBC",
|
"SBC",
|
||||||
"SBCS",
|
"SBCS",
|
||||||
"SBCSW",
|
"SBCSW",
|
||||||
@@ -477,23 +479,68 @@ var arm64GeneratedInstrs = []string{
|
|||||||
"UXTH",
|
"UXTH",
|
||||||
"UXTHW",
|
"UXTHW",
|
||||||
"UXTW",
|
"UXTW",
|
||||||
|
"VABS",
|
||||||
"VADD",
|
"VADD",
|
||||||
"VADDP",
|
"VADDP",
|
||||||
"VADDV",
|
"VADDV",
|
||||||
"VAND",
|
"VAND",
|
||||||
"VBCAX",
|
"VBCAX",
|
||||||
|
"VBIC",
|
||||||
"VBIF",
|
"VBIF",
|
||||||
"VBIT",
|
"VBIT",
|
||||||
"VBSL",
|
"VBSL",
|
||||||
|
"VCLS",
|
||||||
|
"VCLZ",
|
||||||
"VCMEQ",
|
"VCMEQ",
|
||||||
|
"VCMGE",
|
||||||
|
"VCMGT",
|
||||||
|
"VCMHI",
|
||||||
|
"VCMHS",
|
||||||
|
"VCMLE",
|
||||||
|
"VCMLT",
|
||||||
"VCMTST",
|
"VCMTST",
|
||||||
"VCNT",
|
"VCNT",
|
||||||
"VDUP",
|
"VDUP",
|
||||||
"VEOR",
|
"VEOR",
|
||||||
"VEOR3",
|
"VEOR3",
|
||||||
"VEXT",
|
"VEXT",
|
||||||
|
"VFABS",
|
||||||
|
"VFADD",
|
||||||
|
"VFADDP",
|
||||||
|
"VFCMEQ",
|
||||||
|
"VFCMGE",
|
||||||
|
"VFCMGT",
|
||||||
|
"VFCMLE",
|
||||||
|
"VFCMLT",
|
||||||
|
"VFCVTL",
|
||||||
|
"VFCVTL2",
|
||||||
|
"VFCVTN",
|
||||||
|
"VFCVTN2",
|
||||||
|
"VFCVTZS",
|
||||||
|
"VFCVTZU",
|
||||||
|
"VFDIV",
|
||||||
|
"VFMAX",
|
||||||
|
"VFMAXNM",
|
||||||
|
"VFMAXNMP",
|
||||||
|
"VFMAXNMV",
|
||||||
|
"VFMAXP",
|
||||||
|
"VFMAXV",
|
||||||
|
"VFMIN",
|
||||||
|
"VFMINNM",
|
||||||
|
"VFMINNMP",
|
||||||
|
"VFMINNMV",
|
||||||
|
"VFMINP",
|
||||||
|
"VFMINV",
|
||||||
"VFMLA",
|
"VFMLA",
|
||||||
"VFMLS",
|
"VFMLS",
|
||||||
|
"VFMUL",
|
||||||
|
"VFNEG",
|
||||||
|
"VFRINTM",
|
||||||
|
"VFRINTN",
|
||||||
|
"VFRINTP",
|
||||||
|
"VFRINTZ",
|
||||||
|
"VFSQRT",
|
||||||
|
"VFSUB",
|
||||||
"VLD1",
|
"VLD1",
|
||||||
"VLD1R",
|
"VLD1R",
|
||||||
"VLD2",
|
"VLD2",
|
||||||
@@ -502,11 +549,17 @@ var arm64GeneratedInstrs = []string{
|
|||||||
"VLD3R",
|
"VLD3R",
|
||||||
"VLD4",
|
"VLD4",
|
||||||
"VLD4R",
|
"VLD4R",
|
||||||
|
"VMLA",
|
||||||
|
"VMLS",
|
||||||
"VMOV",
|
"VMOV",
|
||||||
"VMOVD",
|
"VMOVD",
|
||||||
"VMOVI",
|
"VMOVI",
|
||||||
"VMOVQ",
|
"VMOVQ",
|
||||||
"VMOVS",
|
"VMOVS",
|
||||||
|
"VMUL",
|
||||||
|
"VNEG",
|
||||||
|
"VNOT",
|
||||||
|
"VORN",
|
||||||
"VORR",
|
"VORR",
|
||||||
"VPMULL",
|
"VPMULL",
|
||||||
"VPMULL2",
|
"VPMULL2",
|
||||||
@@ -515,14 +568,47 @@ var arm64GeneratedInstrs = []string{
|
|||||||
"VREV16",
|
"VREV16",
|
||||||
"VREV32",
|
"VREV32",
|
||||||
"VREV64",
|
"VREV64",
|
||||||
|
"VSCVTF",
|
||||||
|
"VSHADD",
|
||||||
"VSHL",
|
"VSHL",
|
||||||
|
"VSHRN",
|
||||||
|
"VSHRN2",
|
||||||
"VSLI",
|
"VSLI",
|
||||||
|
"VSMAX",
|
||||||
|
"VSMAXP",
|
||||||
|
"VSMAXV",
|
||||||
|
"VSMIN",
|
||||||
|
"VSMINP",
|
||||||
|
"VSMINV",
|
||||||
|
"VSMLAL",
|
||||||
|
"VSMLAL2",
|
||||||
|
"VSMLSL",
|
||||||
|
"VSMLSL2",
|
||||||
|
"VSMULL",
|
||||||
|
"VSMULL2",
|
||||||
|
"VSQABS",
|
||||||
|
"VSQADD",
|
||||||
|
"VSQNEG",
|
||||||
|
"VSQSHL",
|
||||||
|
"VSQSUB",
|
||||||
|
"VSQXTN",
|
||||||
|
"VSQXTN2",
|
||||||
|
"VSQXTUN",
|
||||||
|
"VSQXTUN2",
|
||||||
|
"VSRHADD",
|
||||||
"VSRI",
|
"VSRI",
|
||||||
|
"VSRSHR",
|
||||||
|
"VSSHL",
|
||||||
|
"VSSHLL",
|
||||||
|
"VSSHLL2",
|
||||||
|
"VSSHR",
|
||||||
"VST1",
|
"VST1",
|
||||||
"VST2",
|
"VST2",
|
||||||
"VST3",
|
"VST3",
|
||||||
"VST4",
|
"VST4",
|
||||||
"VSUB",
|
"VSUB",
|
||||||
|
"VSXTL",
|
||||||
|
"VSXTL2",
|
||||||
"VTBL",
|
"VTBL",
|
||||||
"VTBX",
|
"VTBX",
|
||||||
"VTRN1",
|
"VTRN1",
|
||||||
@@ -530,8 +616,27 @@ var arm64GeneratedInstrs = []string{
|
|||||||
"VUADDLV",
|
"VUADDLV",
|
||||||
"VUADDW",
|
"VUADDW",
|
||||||
"VUADDW2",
|
"VUADDW2",
|
||||||
|
"VUCVTF",
|
||||||
|
"VUHADD",
|
||||||
"VUMAX",
|
"VUMAX",
|
||||||
|
"VUMAXP",
|
||||||
|
"VUMAXV",
|
||||||
"VUMIN",
|
"VUMIN",
|
||||||
|
"VUMINP",
|
||||||
|
"VUMINV",
|
||||||
|
"VUMLAL",
|
||||||
|
"VUMLAL2",
|
||||||
|
"VUMLSL",
|
||||||
|
"VUMLSL2",
|
||||||
|
"VUMULL",
|
||||||
|
"VUMULL2",
|
||||||
|
"VUQADD",
|
||||||
|
"VUQSHL",
|
||||||
|
"VUQSUB",
|
||||||
|
"VUQXTN",
|
||||||
|
"VUQXTN2",
|
||||||
|
"VURHADD",
|
||||||
|
"VUSHL",
|
||||||
"VUSHLL",
|
"VUSHLL",
|
||||||
"VUSHLL2",
|
"VUSHLL2",
|
||||||
"VUSHR",
|
"VUSHR",
|
||||||
@@ -541,6 +646,8 @@ var arm64GeneratedInstrs = []string{
|
|||||||
"VUZP1",
|
"VUZP1",
|
||||||
"VUZP2",
|
"VUZP2",
|
||||||
"VXAR",
|
"VXAR",
|
||||||
|
"VXTN",
|
||||||
|
"VXTN2",
|
||||||
"VZIP1",
|
"VZIP1",
|
||||||
"VZIP2",
|
"VZIP2",
|
||||||
"WFE",
|
"WFE",
|
||||||
|
|||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/util.go from the Go toolchain.
|
// Source: cmd/internal/obj/util.go from the Go toolchain.
|
||||||
//
|
//
|
||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
|||||||
+2
-2
@@ -31,10 +31,10 @@ func loong64Registers() []Register {
|
|||||||
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
|
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
|
||||||
}
|
}
|
||||||
for i := 0; i <= 31; i++ {
|
for i := 0; i <= 31; i++ {
|
||||||
add(fmt.Sprintf("V%d", i), VecARM, "LSX 128-bit vector register")
|
add(fmt.Sprintf("V%d", i), VecSIMD, "LSX 128-bit vector register")
|
||||||
}
|
}
|
||||||
for i := 0; i <= 31; i++ {
|
for i := 0; i <= 31; i++ {
|
||||||
add(fmt.Sprintf("X%d", i), VecARM, "LASX 256-bit vector register")
|
add(fmt.Sprintf("X%d", i), VecSIMD, "LASX 256-bit vector register")
|
||||||
}
|
}
|
||||||
return regs
|
return regs
|
||||||
}
|
}
|
||||||
|
|||||||
+10
-1
@@ -1,4 +1,4 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
|
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
|
||||||
//
|
//
|
||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
@@ -152,6 +152,8 @@ var loong64GeneratedInstrs = []string{
|
|||||||
"FNMADDF",
|
"FNMADDF",
|
||||||
"FNMSUBD",
|
"FNMSUBD",
|
||||||
"FNMSUBF",
|
"FNMSUBF",
|
||||||
|
"FRINTD",
|
||||||
|
"FRINTF",
|
||||||
"FSCALEBD",
|
"FSCALEBD",
|
||||||
"FSCALEBF",
|
"FSCALEBF",
|
||||||
"FSEL",
|
"FSEL",
|
||||||
@@ -177,7 +179,10 @@ var loong64GeneratedInstrs = []string{
|
|||||||
"FTINTWF",
|
"FTINTWF",
|
||||||
"JIRL",
|
"JIRL",
|
||||||
"LL",
|
"LL",
|
||||||
|
"LLACQV",
|
||||||
|
"LLACQW",
|
||||||
"LLV",
|
"LLV",
|
||||||
|
"LLW",
|
||||||
"LU12IW",
|
"LU12IW",
|
||||||
"LU32ID",
|
"LU32ID",
|
||||||
"LU52ID",
|
"LU52ID",
|
||||||
@@ -248,7 +253,11 @@ var loong64GeneratedInstrs = []string{
|
|||||||
"ROTR",
|
"ROTR",
|
||||||
"ROTRV",
|
"ROTRV",
|
||||||
"SC",
|
"SC",
|
||||||
|
"SCQ",
|
||||||
|
"SCRELV",
|
||||||
|
"SCRELW",
|
||||||
"SCV",
|
"SCV",
|
||||||
|
"SCW",
|
||||||
"SGT",
|
"SGT",
|
||||||
"SGTU",
|
"SGTU",
|
||||||
"SLL",
|
"SLL",
|
||||||
|
|||||||
+32
-1
@@ -1,4 +1,4 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
|
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
|
||||||
//
|
//
|
||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
@@ -81,6 +81,9 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"CLD",
|
"CLD",
|
||||||
"CLDSP",
|
"CLDSP",
|
||||||
"CLI",
|
"CLI",
|
||||||
|
"CLMUL",
|
||||||
|
"CLMULH",
|
||||||
|
"CLMULR",
|
||||||
"CLUI",
|
"CLUI",
|
||||||
"CLW",
|
"CLW",
|
||||||
"CLWSP",
|
"CLWSP",
|
||||||
@@ -95,13 +98,20 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"CSDSP",
|
"CSDSP",
|
||||||
"CSLLI",
|
"CSLLI",
|
||||||
"CSRAI",
|
"CSRAI",
|
||||||
|
"CSRC",
|
||||||
|
"CSRCI",
|
||||||
"CSRLI",
|
"CSRLI",
|
||||||
|
"CSRR",
|
||||||
"CSRRC",
|
"CSRRC",
|
||||||
"CSRRCI",
|
"CSRRCI",
|
||||||
"CSRRS",
|
"CSRRS",
|
||||||
"CSRRSI",
|
"CSRRSI",
|
||||||
"CSRRW",
|
"CSRRW",
|
||||||
"CSRRWI",
|
"CSRRWI",
|
||||||
|
"CSRS",
|
||||||
|
"CSRSI",
|
||||||
|
"CSRW",
|
||||||
|
"CSRWI",
|
||||||
"CSUB",
|
"CSUB",
|
||||||
"CSUBW",
|
"CSUBW",
|
||||||
"CSW",
|
"CSW",
|
||||||
@@ -259,6 +269,7 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"ORCB",
|
"ORCB",
|
||||||
"ORI",
|
"ORI",
|
||||||
"ORN",
|
"ORN",
|
||||||
|
"PAUSE",
|
||||||
"RDCYCLE",
|
"RDCYCLE",
|
||||||
"RDINSTRET",
|
"RDINSTRET",
|
||||||
"RDTIME",
|
"RDTIME",
|
||||||
@@ -322,6 +333,8 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"VADDVI",
|
"VADDVI",
|
||||||
"VADDVV",
|
"VADDVV",
|
||||||
"VADDVX",
|
"VADDVX",
|
||||||
|
"VANDNVV",
|
||||||
|
"VANDNVX",
|
||||||
"VANDVI",
|
"VANDVI",
|
||||||
"VANDVV",
|
"VANDVV",
|
||||||
"VANDVX",
|
"VANDVX",
|
||||||
@@ -329,8 +342,17 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"VASUBUVX",
|
"VASUBUVX",
|
||||||
"VASUBVV",
|
"VASUBVV",
|
||||||
"VASUBVX",
|
"VASUBVX",
|
||||||
|
"VBREV8V",
|
||||||
|
"VBREVV",
|
||||||
|
"VCLMULHVV",
|
||||||
|
"VCLMULHVX",
|
||||||
|
"VCLMULVV",
|
||||||
|
"VCLMULVX",
|
||||||
|
"VCLZV",
|
||||||
"VCOMPRESSVM",
|
"VCOMPRESSVM",
|
||||||
"VCPOPM",
|
"VCPOPM",
|
||||||
|
"VCPOPV",
|
||||||
|
"VCTZV",
|
||||||
"VDIVUVV",
|
"VDIVUVV",
|
||||||
"VDIVUVX",
|
"VDIVUVX",
|
||||||
"VDIVVV",
|
"VDIVVV",
|
||||||
@@ -743,10 +765,16 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"VREMUVX",
|
"VREMUVX",
|
||||||
"VREMVV",
|
"VREMVV",
|
||||||
"VREMVX",
|
"VREMVX",
|
||||||
|
"VREV8V",
|
||||||
"VRGATHEREI16VV",
|
"VRGATHEREI16VV",
|
||||||
"VRGATHERVI",
|
"VRGATHERVI",
|
||||||
"VRGATHERVV",
|
"VRGATHERVV",
|
||||||
"VRGATHERVX",
|
"VRGATHERVX",
|
||||||
|
"VROLVV",
|
||||||
|
"VROLVX",
|
||||||
|
"VRORVI",
|
||||||
|
"VRORVV",
|
||||||
|
"VRORVX",
|
||||||
"VRSUBVI",
|
"VRSUBVI",
|
||||||
"VRSUBVX",
|
"VRSUBVX",
|
||||||
"VS1RV",
|
"VS1RV",
|
||||||
@@ -950,6 +978,9 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"VWMULVX",
|
"VWMULVX",
|
||||||
"VWREDSUMUVS",
|
"VWREDSUMUVS",
|
||||||
"VWREDSUMVS",
|
"VWREDSUMVS",
|
||||||
|
"VWSLLVI",
|
||||||
|
"VWSLLVV",
|
||||||
|
"VWSLLVX",
|
||||||
"VWSUBUVV",
|
"VWSUBUVV",
|
||||||
"VWSUBUVX",
|
"VWSUBUVX",
|
||||||
"VWSUBUWV",
|
"VWSUBUWV",
|
||||||
|
|||||||
+158
-39
@@ -4,13 +4,14 @@
|
|||||||
package asm
|
package asm
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"encoding/binary"
|
||||||
"os"
|
"os"
|
||||||
"os/exec"
|
"os/exec"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestGOObjectAARCH64Structure checks the basic structure of the emitted
|
// TestGOObjectAARCH64Structure checks the basic structure of the emitted
|
||||||
@@ -61,6 +62,62 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestGOObjectAARCH64PairReloc pins the ADRP-pair relocation shape against
|
||||||
|
// the toolchain's own object for the same source: exactly one R_ADDRARM64
|
||||||
|
// of Siz 8 at the ADRP word (cmd/internal/obj/arm64/asm7.go adds a single
|
||||||
|
// Siz-8 relocation per pair and the linker patches both instructions from
|
||||||
|
// it). gasm's assembler records the ADRP+ADD form as two word relocs; the
|
||||||
|
// emitter must coalesce them, not emit two Siz-4 records.
|
||||||
|
func TestGOObjectAARCH64PairReloc(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("gv_arm64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·getv(SB), NOSPLIT, $0-8
|
||||||
|
MOVD $v<>(SB), R4
|
||||||
|
MOVD R4, ret+0(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
GLOBL v<>(SB), RODATA, $8
|
||||||
|
DATA v<>+0(SB)/8, $7
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.GOObjectAARCH64("main", "gv_arm64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObjectAARCH64: %v", err)
|
||||||
|
}
|
||||||
|
v := openGoobj(t, obj)
|
||||||
|
relocs := v.blk(blkReloc)
|
||||||
|
le := binary.LittleEndian
|
||||||
|
// Two DWARF relocs on the lines/DIE symbols, then the code's one pair
|
||||||
|
// relocation.
|
||||||
|
if len(relocs) != 3*23 {
|
||||||
|
t.Fatalf("relocs = %d bytes, want three entries", len(relocs))
|
||||||
|
}
|
||||||
|
cr := relocs[2*23:]
|
||||||
|
if off := int32(le.Uint32(cr[0:])); off != 0 {
|
||||||
|
t.Errorf("pair reloc off = %d, want 0 (the ADRP word)", off)
|
||||||
|
}
|
||||||
|
if siz := cr[4]; siz != 8 {
|
||||||
|
t.Errorf("pair reloc siz = %d, want 8", siz)
|
||||||
|
}
|
||||||
|
if typ := le.Uint16(cr[5:]); typ != relocArm64Addr {
|
||||||
|
t.Errorf("pair reloc type = %d, want %d (R_ADDRARM64)", typ, relocArm64Addr)
|
||||||
|
}
|
||||||
|
if pkg := le.Uint32(cr[15:]); pkg != pkgIdxSelf {
|
||||||
|
t.Errorf("pair reloc PkgIdx = %#x, want pkgIdxSelf", pkg)
|
||||||
|
}
|
||||||
|
// The GLOBL is the first package definition.
|
||||||
|
if sym := le.Uint32(cr[19:]); sym != 0 {
|
||||||
|
t.Errorf("pair reloc SymIdx = %d, want 0 (the GLOBL definition)", sym)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestGOObjectAARCH64Link does an end-to-end link test: it cross-compiles a
|
// TestGOObjectAARCH64Link does an end-to-end link test: it cross-compiles a
|
||||||
// Go program for arm64, substitutes the gasm-produced object into the package
|
// Go program for arm64, substitutes the gasm-produced object into the package
|
||||||
// archive, re-links with cmd/link, and verifies the symbol appears in the
|
// archive, re-links with cmd/link, and verifies the symbol appears in the
|
||||||
@@ -79,6 +136,14 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
ADD R5, R4, R4
|
ADD R5, R4, R4
|
||||||
MOVD R4, ret+16(FP)
|
MOVD R4, ret+16(FP)
|
||||||
RET
|
RET
|
||||||
|
|
||||||
|
TEXT ·getv(SB), NOSPLIT, $0-8
|
||||||
|
MOVD $v<>(SB), R4
|
||||||
|
MOVD R4, ret+0(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
GLOBL v<>(SB), RODATA, $8
|
||||||
|
DATA v<>+0(SB)/8, $7
|
||||||
`
|
`
|
||||||
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
|
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
@@ -86,11 +151,15 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
mainSrc := `package main
|
mainSrc := `package main
|
||||||
|
|
||||||
func add(a, b int64) int64
|
func add(a, b int64) int64
|
||||||
|
func getv() *int64
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
if add(20, 22) != 42 {
|
if add(20, 22) != 42 {
|
||||||
panic("bad add")
|
panic("bad add")
|
||||||
}
|
}
|
||||||
|
if getv() == nil {
|
||||||
|
panic("bad getv")
|
||||||
|
}
|
||||||
}
|
}
|
||||||
`
|
`
|
||||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||||
@@ -109,26 +178,10 @@ func main() {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||||
}
|
}
|
||||||
var work, linkLine, asmObj string
|
st := parseBuildLog(t, buildLog, "main_arm64.s")
|
||||||
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
defer os.RemoveAll(st.work)
|
||||||
switch {
|
|
||||||
case strings.HasPrefix(line, "WORK="):
|
|
||||||
work = strings.TrimPrefix(line, "WORK=")
|
|
||||||
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
|
|
||||||
asmObj = fieldAfter(line, "-o")
|
|
||||||
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
|
||||||
linkLine = line
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if work == "" || asmObj == "" {
|
|
||||||
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
|
|
||||||
}
|
|
||||||
defer os.RemoveAll(work)
|
|
||||||
|
|
||||||
// Expand $WORK in the object path.
|
// Assemble the same source with gasm and substitute the object.
|
||||||
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
|
|
||||||
|
|
||||||
// Read the toolchain-produced object and assemble the same source with gasm.
|
|
||||||
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
|
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
@@ -141,30 +194,16 @@ func main() {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("AssembleFileARM64: %v", err)
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
}
|
}
|
||||||
gasmObj, err := img.GOObjectAARCH64("a64link", "main_arm64.s")
|
// The package path is "main", the prefix the Go code's references carry.
|
||||||
|
gasmObj, err := img.GOObjectAARCH64("main", "main_arm64.s")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("GOObjectAARCH64: %v", err)
|
t.Fatalf("GOObjectAARCH64: %v", err)
|
||||||
}
|
}
|
||||||
|
substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"),
|
||||||
// Replace the toolchain-produced object with gasm's.
|
gasmObj, "GOARCH=arm64")
|
||||||
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
|
|
||||||
t.Fatalf("write gasm object: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Re-link.
|
|
||||||
if linkLine == "" {
|
|
||||||
t.Skip("could not find link command in build log")
|
|
||||||
}
|
|
||||||
// Expand $WORK in the link command.
|
|
||||||
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
|
||||||
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
|
|
||||||
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
|
|
||||||
if out, err := linkCmd.CombinedOutput(); err != nil {
|
|
||||||
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Verify the binary exists and contains the symbol.
|
// Verify the binary exists and contains the symbol.
|
||||||
binPath := filepath.Join(dir, "prog")
|
binPath := filepath.Join(dir, "prog2")
|
||||||
if _, err := os.Stat(binPath); err != nil {
|
if _, err := os.Stat(binPath); err != nil {
|
||||||
t.Fatalf("binary not found: %v", err)
|
t.Fatalf("binary not found: %v", err)
|
||||||
}
|
}
|
||||||
@@ -176,3 +215,83 @@ func main() {
|
|||||||
t.Error("binary does not contain expected symbol")
|
t.Error("binary does not contain expected symbol")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestGOObjectAARCH64DataSymbolLink does for symbol-valued DATA fields what
|
||||||
|
// the rt0 files do ("DATA _rt0…lib+0(SB)/8, $_rt0…lib(SB)"): the gasm object
|
||||||
|
// carries an R_ADDR against the file's own TEXT symbol, the toolchain links
|
||||||
|
// it, and the binary is checked for the symbol (no arm64 host to run it).
|
||||||
|
func TestGOObjectAARCH64DataSymbolLink(t *testing.T) {
|
||||||
|
goBin, err := exec.LookPath("go")
|
||||||
|
if err != nil {
|
||||||
|
t.Skip("no Go toolchain available")
|
||||||
|
}
|
||||||
|
dir := t.TempDir()
|
||||||
|
asmSrc := `#include "textflag.h"
|
||||||
|
GLOBL entry(SB), NOPTR, $8
|
||||||
|
DATA entry+0(SB)/8, $·keepme(SB)
|
||||||
|
|
||||||
|
TEXT ·keepme(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·entryptr(SB), NOSPLIT, $0-8
|
||||||
|
MOVD entry+0(SB), R4
|
||||||
|
MOVD R4, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
mainSrc := `package main
|
||||||
|
|
||||||
|
func keepme()
|
||||||
|
func entryptr() uintptr
|
||||||
|
|
||||||
|
func main() {
|
||||||
|
if entryptr() == 0 {
|
||||||
|
panic("the entry word is empty")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module a64dlink\n\ngo 1.21\n"), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||||
|
build.Dir = dir
|
||||||
|
build.Env = append(os.Environ(), "GOARCH=arm64")
|
||||||
|
buildLog, err := build.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||||
|
}
|
||||||
|
st := parseBuildLog(t, buildLog, "main_arm64.s")
|
||||||
|
defer os.RemoveAll(st.work)
|
||||||
|
|
||||||
|
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
f, errs := parser.Parse("main_arm64.s", string(src))
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
gasmObj, err := img.GOObjectAARCH64("main", "main_arm64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObjectAARCH64: %v", err)
|
||||||
|
}
|
||||||
|
substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"),
|
||||||
|
gasmObj, "GOARCH=arm64")
|
||||||
|
binData, err := os.ReadFile(filepath.Join(dir, "prog2"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if !strings.Contains(string(binData), "keepme") {
|
||||||
|
t.Error("binary does not contain the keepme symbol")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+3492
-188
File diff suppressed because it is too large
Load Diff
+975
-106
File diff suppressed because it is too large
Load Diff
+1177
-7
File diff suppressed because it is too large
Load Diff
+86
-21
@@ -53,7 +53,7 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
)
|
)
|
||||||
|
|
||||||
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
|
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
|
||||||
@@ -187,10 +187,32 @@ func arm64Prologue(fi arm64FrameInfo) []byte {
|
|||||||
return a64WordsLE(ws...)
|
return a64WordsLE(ws...)
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value
|
// arm64SplitImm12 reports whether the toolchain decomposes ADD/SUB $imm into
|
||||||
// fits the imm12 field (plain, or shifted left by 12 when it is a multiple
|
// two imm12 instructions instead of materialising it into REGTMP
|
||||||
// of 4096); otherwise the toolchain materialises it into REGTMP (R27) and
|
// (asm7.go case 48, the C_ADDCON2 class): the value must fit 24 bits
|
||||||
// subtracts the register in the extended-register form.
|
// unsigned and be neither encodable as one imm12 (checked by the callers
|
||||||
|
// first), nor loadable into a register in a single MOVZ/MOVN word, nor a
|
||||||
|
// logical immediate, because conclass tests all three before C_ADDCON2.
|
||||||
|
func arm64SplitImm12(imm uint32) bool {
|
||||||
|
if imm > 0xFFFFFF {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
if _, _, _, ok := arm64Bitmask(uint64(imm), 1); ok {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
return arm64Movcon(int64(imm)) < 0 && arm64Movcon(^int64(imm)) < 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// arm64SubImmWords emits SUB $imm, SP, Rd with the toolchain's ladder for an
|
||||||
|
// ADD/SUB constant (asm7.go conclass and cases 2, 48, 62 and 13): the
|
||||||
|
// immediate form when the value fits imm12 (plain, or shifted left by 12
|
||||||
|
// when it is a multiple of 4096); a value with a single 16-bit chunk, a
|
||||||
|
// logical immediate, or one wider than 24 bits is materialised into REGTMP
|
||||||
|
// (R27) and subtracted in the extended-register form; everything else up to
|
||||||
|
// 0xFFFFFF is split into two imm12 instructions:
|
||||||
|
//
|
||||||
|
// SUB $(imm&0xfff), SP, Rd
|
||||||
|
// SUB $((imm&0xfff000)>>12)<<12, Rd, Rd
|
||||||
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
||||||
if imm <= 0xFFF {
|
if imm <= 0xFFF {
|
||||||
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
|
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
|
||||||
@@ -198,15 +220,21 @@ func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
|||||||
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
||||||
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
|
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
|
||||||
}
|
}
|
||||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
if !arm64SplitImm12(imm) {
|
||||||
if err != nil {
|
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||||
mov = nil
|
if err != nil {
|
||||||
|
mov = nil
|
||||||
|
}
|
||||||
|
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
|
||||||
|
}
|
||||||
|
return []uint32{
|
||||||
|
a64AddSub(1, 1, 0, 0, imm&0xFFF, 31, rd),
|
||||||
|
a64AddSub(1, 1, 0, 1, (imm&0xFFF000)>>12, rd, rd),
|
||||||
}
|
}
|
||||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12
|
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12,
|
||||||
// and REGTMP fallback ladder.
|
// split and REGTMP ladder as arm64SubImmWords.
|
||||||
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
||||||
if imm <= 0xFFF {
|
if imm <= 0xFFF {
|
||||||
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
|
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
|
||||||
@@ -214,11 +242,35 @@ func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
|||||||
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
||||||
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
|
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
|
||||||
}
|
}
|
||||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
if !arm64SplitImm12(imm) {
|
||||||
|
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||||
|
if err != nil {
|
||||||
|
mov = nil
|
||||||
|
}
|
||||||
|
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
|
||||||
|
}
|
||||||
|
return []uint32{
|
||||||
|
a64AddSub(1, 0, 0, 0, imm&0xFFF, 31, rd),
|
||||||
|
a64AddSub(1, 0, 0, 1, (imm&0xFFF000)>>12, rd, rd),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// arm64RetAddWords emits the frame deallocation of a non-leaf RET with a
|
||||||
|
// large frame. The toolchain adds the frame back with a single instruction:
|
||||||
|
// a plain imm12 ADD when autosize fits 12 bits, otherwise the value is
|
||||||
|
// materialised into REGTMP and added as a register, so the epilogue never
|
||||||
|
// leaves a partially deallocated frame (obj7.go ARET, issue 73259). The
|
||||||
|
// shifted-imm12 and split-imm12 forms are therefore never used here, unlike
|
||||||
|
// the leaf epilogue's plain ADD instructions.
|
||||||
|
func arm64RetAddWords(autosize uint32) []uint32 {
|
||||||
|
if autosize < 1<<12 {
|
||||||
|
return []uint32{a64AddSub(1, 0, 0, 0, autosize, 31, 31)}
|
||||||
|
}
|
||||||
|
mov, err := encodeARM64LoadImm(27, int64(autosize), "MOVD")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
mov = nil
|
mov = nil
|
||||||
}
|
}
|
||||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
|
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, 31))
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
|
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
|
||||||
@@ -237,11 +289,11 @@ func arm64Return(fi arm64FrameInfo) []byte {
|
|||||||
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
|
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
|
||||||
)
|
)
|
||||||
} else {
|
} else {
|
||||||
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
|
// Large frame: LDP -8(SP), (FP, LR), then deallocate.
|
||||||
ws = append(ws,
|
ws = append(ws,
|
||||||
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
||||||
)
|
)
|
||||||
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
|
ws = append(ws, arm64RetAddWords(uint32(fi.autosize))...)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// RET: BR LR (0xd65f03c0)
|
// RET: BR LR (0xd65f03c0)
|
||||||
@@ -258,22 +310,32 @@ func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
|
|||||||
if fi.autosize <= 0xf0 {
|
if fi.autosize <= 0xf0 {
|
||||||
return 4 // MOVD.W instruction decrements SP
|
return 4 // MOVD.W instruction decrements SP
|
||||||
}
|
}
|
||||||
return 8 // SUB + STP + MOVD (3 instructions, SP updated at the MOVD)
|
// Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose
|
||||||
|
// position depends on how many words the SUB itself took (immediate,
|
||||||
|
// shifted immediate, the two-word imm12 split, or a materialised REGTMP
|
||||||
|
// sequence).
|
||||||
|
return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1)
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
||||||
// (but not including) the final RET instruction.
|
// (but not including) the final RET instruction. The lengths are read from
|
||||||
|
// the same word-emitting helpers the epilogue uses rather than assumed: the
|
||||||
|
// leaf path shares the prologue's immediate ladder, and a materialised
|
||||||
|
// autosize costs its MOV words plus the ADD itself.
|
||||||
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
|
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
|
||||||
if fi.autosize == 0 {
|
if fi.autosize == 0 {
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
if fi.leaf {
|
if fi.leaf {
|
||||||
return 8 // ADD + ADD
|
return 4 * (len(arm64AddImmWords(uint32(fi.autosize-8), 29)) +
|
||||||
|
len(arm64AddImmWords(uint32(fi.autosize), 31)))
|
||||||
}
|
}
|
||||||
if fi.autosize <= 0xf0 {
|
if fi.autosize <= 0xf0 {
|
||||||
return 8 // LDR + LDR.P
|
return 8 // LDR + LDR.P
|
||||||
}
|
}
|
||||||
return 8 // LDP + ADD
|
// LDP + the deallocation emitted by arm64RetAddWords, so the length
|
||||||
|
// tracks whatever the MOVD ladder needs.
|
||||||
|
return 4 + 4*len(arm64RetAddWords(uint32(fi.autosize)))
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64ResolvePseudo translates a pseudo-register memory reference into a
|
// arm64ResolvePseudo translates a pseudo-register memory reference into a
|
||||||
@@ -382,9 +444,12 @@ func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte {
|
|||||||
ws = append(ws, wordsOf(mov)...)
|
ws = append(ws, wordsOf(mov)...)
|
||||||
ml := len(mov) / 4
|
ml := len(mov) / 4
|
||||||
ws = append(ws, arm64DPExtWords(arm64OpSubs, 27, 31, 17)) // SUBS R17, RSP, R27
|
ws = append(ws, arm64DPExtWords(arm64OpSubs, 27, 31, 17)) // SUBS R17, RSP, R27
|
||||||
ws = append(ws, br(8+ml, a64CondLO))
|
// The branches sit at fixed byte offsets in the guard prefix: after
|
||||||
|
// the LDR (4), the ml MOV words (4*ml) and the SUBS (4) for B.LO,
|
||||||
|
// then a further B.LO word and the CMP for B.LS.
|
||||||
|
ws = append(ws, br(8+4*ml, a64CondLO))
|
||||||
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
|
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
|
||||||
ws = append(ws, br(8+ml+8, a64CondLS))
|
ws = append(ws, br(16+4*ml, a64CondLS))
|
||||||
}
|
}
|
||||||
return a64WordsLE(ws...)
|
return a64WordsLE(ws...)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ import (
|
|||||||
"encoding/binary"
|
"encoding/binary"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// parseArm64File is a helper assembling one arm64 source file.
|
// parseArm64File is a helper assembling one arm64 source file.
|
||||||
|
|||||||
@@ -0,0 +1,588 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
// arm64 system registers and system-instruction aliases.
|
||||||
|
//
|
||||||
|
// The tables are transcribed from the data the Go toolchain itself carries
|
||||||
|
// (cmd/internal/obj/arm64/sysRegEnc.go and the sysInstFields map of asm7.go),
|
||||||
|
// which the ARM ARM defines: every system register is the packed field set
|
||||||
|
// op0<<19 | op1<<16 | CRn<<12 | CRm<<8 | op2<<5, and the read/write flags are
|
||||||
|
// the toolchain's own access classification. The encoding tables live here so
|
||||||
|
// the encoder stays testable against the GOROOT testdata word for word.
|
||||||
|
|
||||||
|
// a64SysReg is one system register: the packed encoding fields and the
|
||||||
|
// directions the register supports.
|
||||||
|
type a64SysReg struct {
|
||||||
|
v uint32
|
||||||
|
read bool
|
||||||
|
write bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64SysRegs maps the system register names the toolchain knows to their
|
||||||
|
// encodings. MRS reads 0xd5300000 | v | Rd and MSR writes
|
||||||
|
// 0xd5100000 | v | Rt.
|
||||||
|
var a64SysRegs = map[string]a64SysReg{
|
||||||
|
"ACTLR_EL1": a64SysReg{0x181020, true, true},
|
||||||
|
"AFSR0_EL1": a64SysReg{0x185100, true, true},
|
||||||
|
"AFSR1_EL1": a64SysReg{0x185120, true, true},
|
||||||
|
"AIDR_EL1": a64SysReg{0x1900e0, true, false},
|
||||||
|
"AMAIR_EL1": a64SysReg{0x18a300, true, true},
|
||||||
|
"AMCFGR_EL0": a64SysReg{0x1bd220, true, false},
|
||||||
|
"AMCGCR_EL0": a64SysReg{0x1bd240, true, false},
|
||||||
|
"AMCNTENCLR0_EL0": a64SysReg{0x1bd280, true, true},
|
||||||
|
"AMCNTENCLR1_EL0": a64SysReg{0x1bd300, true, true},
|
||||||
|
"AMCNTENSET0_EL0": a64SysReg{0x1bd2a0, true, true},
|
||||||
|
"AMCNTENSET1_EL0": a64SysReg{0x1bd320, true, true},
|
||||||
|
"AMCR_EL0": a64SysReg{0x1bd200, true, true},
|
||||||
|
"AMEVCNTR00_EL0": a64SysReg{0x1bd400, true, true},
|
||||||
|
"AMEVCNTR01_EL0": a64SysReg{0x1bd420, true, true},
|
||||||
|
"AMEVCNTR02_EL0": a64SysReg{0x1bd440, true, true},
|
||||||
|
"AMEVCNTR03_EL0": a64SysReg{0x1bd460, true, true},
|
||||||
|
"AMEVCNTR04_EL0": a64SysReg{0x1bd480, true, true},
|
||||||
|
"AMEVCNTR05_EL0": a64SysReg{0x1bd4a0, true, true},
|
||||||
|
"AMEVCNTR06_EL0": a64SysReg{0x1bd4c0, true, true},
|
||||||
|
"AMEVCNTR07_EL0": a64SysReg{0x1bd4e0, true, true},
|
||||||
|
"AMEVCNTR08_EL0": a64SysReg{0x1bd500, true, true},
|
||||||
|
"AMEVCNTR09_EL0": a64SysReg{0x1bd520, true, true},
|
||||||
|
"AMEVCNTR010_EL0": a64SysReg{0x1bd540, true, true},
|
||||||
|
"AMEVCNTR011_EL0": a64SysReg{0x1bd560, true, true},
|
||||||
|
"AMEVCNTR012_EL0": a64SysReg{0x1bd580, true, true},
|
||||||
|
"AMEVCNTR013_EL0": a64SysReg{0x1bd5a0, true, true},
|
||||||
|
"AMEVCNTR014_EL0": a64SysReg{0x1bd5c0, true, true},
|
||||||
|
"AMEVCNTR015_EL0": a64SysReg{0x1bd5e0, true, true},
|
||||||
|
"AMEVCNTR10_EL0": a64SysReg{0x1bdc00, true, true},
|
||||||
|
"AMEVCNTR11_EL0": a64SysReg{0x1bdc20, true, true},
|
||||||
|
"AMEVCNTR12_EL0": a64SysReg{0x1bdc40, true, true},
|
||||||
|
"AMEVCNTR13_EL0": a64SysReg{0x1bdc60, true, true},
|
||||||
|
"AMEVCNTR14_EL0": a64SysReg{0x1bdc80, true, true},
|
||||||
|
"AMEVCNTR15_EL0": a64SysReg{0x1bdca0, true, true},
|
||||||
|
"AMEVCNTR16_EL0": a64SysReg{0x1bdcc0, true, true},
|
||||||
|
"AMEVCNTR17_EL0": a64SysReg{0x1bdce0, true, true},
|
||||||
|
"AMEVCNTR18_EL0": a64SysReg{0x1bdd00, true, true},
|
||||||
|
"AMEVCNTR19_EL0": a64SysReg{0x1bdd20, true, true},
|
||||||
|
"AMEVCNTR110_EL0": a64SysReg{0x1bdd40, true, true},
|
||||||
|
"AMEVCNTR111_EL0": a64SysReg{0x1bdd60, true, true},
|
||||||
|
"AMEVCNTR112_EL0": a64SysReg{0x1bdd80, true, true},
|
||||||
|
"AMEVCNTR113_EL0": a64SysReg{0x1bdda0, true, true},
|
||||||
|
"AMEVCNTR114_EL0": a64SysReg{0x1bddc0, true, true},
|
||||||
|
"AMEVCNTR115_EL0": a64SysReg{0x1bdde0, true, true},
|
||||||
|
"AMEVTYPER00_EL0": a64SysReg{0x1bd600, true, false},
|
||||||
|
"AMEVTYPER01_EL0": a64SysReg{0x1bd620, true, false},
|
||||||
|
"AMEVTYPER02_EL0": a64SysReg{0x1bd640, true, false},
|
||||||
|
"AMEVTYPER03_EL0": a64SysReg{0x1bd660, true, false},
|
||||||
|
"AMEVTYPER04_EL0": a64SysReg{0x1bd680, true, false},
|
||||||
|
"AMEVTYPER05_EL0": a64SysReg{0x1bd6a0, true, false},
|
||||||
|
"AMEVTYPER06_EL0": a64SysReg{0x1bd6c0, true, false},
|
||||||
|
"AMEVTYPER07_EL0": a64SysReg{0x1bd6e0, true, false},
|
||||||
|
"AMEVTYPER08_EL0": a64SysReg{0x1bd700, true, false},
|
||||||
|
"AMEVTYPER09_EL0": a64SysReg{0x1bd720, true, false},
|
||||||
|
"AMEVTYPER010_EL0": a64SysReg{0x1bd740, true, false},
|
||||||
|
"AMEVTYPER011_EL0": a64SysReg{0x1bd760, true, false},
|
||||||
|
"AMEVTYPER012_EL0": a64SysReg{0x1bd780, true, false},
|
||||||
|
"AMEVTYPER013_EL0": a64SysReg{0x1bd7a0, true, false},
|
||||||
|
"AMEVTYPER014_EL0": a64SysReg{0x1bd7c0, true, false},
|
||||||
|
"AMEVTYPER015_EL0": a64SysReg{0x1bd7e0, true, false},
|
||||||
|
"AMEVTYPER10_EL0": a64SysReg{0x1bde00, true, true},
|
||||||
|
"AMEVTYPER11_EL0": a64SysReg{0x1bde20, true, true},
|
||||||
|
"AMEVTYPER12_EL0": a64SysReg{0x1bde40, true, true},
|
||||||
|
"AMEVTYPER13_EL0": a64SysReg{0x1bde60, true, true},
|
||||||
|
"AMEVTYPER14_EL0": a64SysReg{0x1bde80, true, true},
|
||||||
|
"AMEVTYPER15_EL0": a64SysReg{0x1bdea0, true, true},
|
||||||
|
"AMEVTYPER16_EL0": a64SysReg{0x1bdec0, true, true},
|
||||||
|
"AMEVTYPER17_EL0": a64SysReg{0x1bdee0, true, true},
|
||||||
|
"AMEVTYPER18_EL0": a64SysReg{0x1bdf00, true, true},
|
||||||
|
"AMEVTYPER19_EL0": a64SysReg{0x1bdf20, true, true},
|
||||||
|
"AMEVTYPER110_EL0": a64SysReg{0x1bdf40, true, true},
|
||||||
|
"AMEVTYPER111_EL0": a64SysReg{0x1bdf60, true, true},
|
||||||
|
"AMEVTYPER112_EL0": a64SysReg{0x1bdf80, true, true},
|
||||||
|
"AMEVTYPER113_EL0": a64SysReg{0x1bdfa0, true, true},
|
||||||
|
"AMEVTYPER114_EL0": a64SysReg{0x1bdfc0, true, true},
|
||||||
|
"AMEVTYPER115_EL0": a64SysReg{0x1bdfe0, true, true},
|
||||||
|
"AMUSERENR_EL0": a64SysReg{0x1bd260, true, true},
|
||||||
|
"APDAKeyHi_EL1": a64SysReg{0x182220, true, true},
|
||||||
|
"APDAKeyLo_EL1": a64SysReg{0x182200, true, true},
|
||||||
|
"APDBKeyHi_EL1": a64SysReg{0x182260, true, true},
|
||||||
|
"APDBKeyLo_EL1": a64SysReg{0x182240, true, true},
|
||||||
|
"APGAKeyHi_EL1": a64SysReg{0x182320, true, true},
|
||||||
|
"APGAKeyLo_EL1": a64SysReg{0x182300, true, true},
|
||||||
|
"APIAKeyHi_EL1": a64SysReg{0x182120, true, true},
|
||||||
|
"APIAKeyLo_EL1": a64SysReg{0x182100, true, true},
|
||||||
|
"APIBKeyHi_EL1": a64SysReg{0x182160, true, true},
|
||||||
|
"APIBKeyLo_EL1": a64SysReg{0x182140, true, true},
|
||||||
|
"CCSIDR2_EL1": a64SysReg{0x190040, true, false},
|
||||||
|
"CCSIDR_EL1": a64SysReg{0x190000, true, false},
|
||||||
|
"CLIDR_EL1": a64SysReg{0x190020, true, false},
|
||||||
|
"CNTFRQ_EL0": a64SysReg{0x1be000, true, true},
|
||||||
|
"CNTKCTL_EL1": a64SysReg{0x18e100, true, true},
|
||||||
|
"CNTP_CTL_EL0": a64SysReg{0x1be220, true, true},
|
||||||
|
"CNTP_CVAL_EL0": a64SysReg{0x1be240, true, true},
|
||||||
|
"CNTP_TVAL_EL0": a64SysReg{0x1be200, true, true},
|
||||||
|
"CNTPCT_EL0": a64SysReg{0x1be020, true, false},
|
||||||
|
"CNTPS_CTL_EL1": a64SysReg{0x1fe220, true, true},
|
||||||
|
"CNTPS_CVAL_EL1": a64SysReg{0x1fe240, true, true},
|
||||||
|
"CNTPS_TVAL_EL1": a64SysReg{0x1fe200, true, true},
|
||||||
|
"CNTV_CTL_EL0": a64SysReg{0x1be320, true, true},
|
||||||
|
"CNTV_CVAL_EL0": a64SysReg{0x1be340, true, true},
|
||||||
|
"CNTV_TVAL_EL0": a64SysReg{0x1be300, true, true},
|
||||||
|
"CNTVCT_EL0": a64SysReg{0x1be040, true, false},
|
||||||
|
"CONTEXTIDR_EL1": a64SysReg{0x18d020, true, true},
|
||||||
|
"CPACR_EL1": a64SysReg{0x181040, true, true},
|
||||||
|
"CSSELR_EL1": a64SysReg{0x1a0000, true, true},
|
||||||
|
"CTR_EL0": a64SysReg{0x1b0020, true, false},
|
||||||
|
"CurrentEL": a64SysReg{0x184240, true, false},
|
||||||
|
"DAIF": a64SysReg{0x1b4220, true, true},
|
||||||
|
"DBGAUTHSTATUS_EL1": a64SysReg{0x107ec0, true, false},
|
||||||
|
"DBGBCR0_EL1": a64SysReg{0x1000a0, true, true},
|
||||||
|
"DBGBCR1_EL1": a64SysReg{0x1001a0, true, true},
|
||||||
|
"DBGBCR2_EL1": a64SysReg{0x1002a0, true, true},
|
||||||
|
"DBGBCR3_EL1": a64SysReg{0x1003a0, true, true},
|
||||||
|
"DBGBCR4_EL1": a64SysReg{0x1004a0, true, true},
|
||||||
|
"DBGBCR5_EL1": a64SysReg{0x1005a0, true, true},
|
||||||
|
"DBGBCR6_EL1": a64SysReg{0x1006a0, true, true},
|
||||||
|
"DBGBCR7_EL1": a64SysReg{0x1007a0, true, true},
|
||||||
|
"DBGBCR8_EL1": a64SysReg{0x1008a0, true, true},
|
||||||
|
"DBGBCR9_EL1": a64SysReg{0x1009a0, true, true},
|
||||||
|
"DBGBCR10_EL1": a64SysReg{0x100aa0, true, true},
|
||||||
|
"DBGBCR11_EL1": a64SysReg{0x100ba0, true, true},
|
||||||
|
"DBGBCR12_EL1": a64SysReg{0x100ca0, true, true},
|
||||||
|
"DBGBCR13_EL1": a64SysReg{0x100da0, true, true},
|
||||||
|
"DBGBCR14_EL1": a64SysReg{0x100ea0, true, true},
|
||||||
|
"DBGBCR15_EL1": a64SysReg{0x100fa0, true, true},
|
||||||
|
"DBGBVR0_EL1": a64SysReg{0x100080, true, true},
|
||||||
|
"DBGBVR1_EL1": a64SysReg{0x100180, true, true},
|
||||||
|
"DBGBVR2_EL1": a64SysReg{0x100280, true, true},
|
||||||
|
"DBGBVR3_EL1": a64SysReg{0x100380, true, true},
|
||||||
|
"DBGBVR4_EL1": a64SysReg{0x100480, true, true},
|
||||||
|
"DBGBVR5_EL1": a64SysReg{0x100580, true, true},
|
||||||
|
"DBGBVR6_EL1": a64SysReg{0x100680, true, true},
|
||||||
|
"DBGBVR7_EL1": a64SysReg{0x100780, true, true},
|
||||||
|
"DBGBVR8_EL1": a64SysReg{0x100880, true, true},
|
||||||
|
"DBGBVR9_EL1": a64SysReg{0x100980, true, true},
|
||||||
|
"DBGBVR10_EL1": a64SysReg{0x100a80, true, true},
|
||||||
|
"DBGBVR11_EL1": a64SysReg{0x100b80, true, true},
|
||||||
|
"DBGBVR12_EL1": a64SysReg{0x100c80, true, true},
|
||||||
|
"DBGBVR13_EL1": a64SysReg{0x100d80, true, true},
|
||||||
|
"DBGBVR14_EL1": a64SysReg{0x100e80, true, true},
|
||||||
|
"DBGBVR15_EL1": a64SysReg{0x100f80, true, true},
|
||||||
|
"DBGCLAIMCLR_EL1": a64SysReg{0x1079c0, true, true},
|
||||||
|
"DBGCLAIMSET_EL1": a64SysReg{0x1078c0, true, true},
|
||||||
|
"DBGDTR_EL0": a64SysReg{0x130400, true, true},
|
||||||
|
"DBGDTRRX_EL0": a64SysReg{0x130500, true, false},
|
||||||
|
"DBGDTRTX_EL0": a64SysReg{0x130500, false, true},
|
||||||
|
"DBGPRCR_EL1": a64SysReg{0x101480, true, true},
|
||||||
|
"DBGWCR0_EL1": a64SysReg{0x1000e0, true, true},
|
||||||
|
"DBGWCR1_EL1": a64SysReg{0x1001e0, true, true},
|
||||||
|
"DBGWCR2_EL1": a64SysReg{0x1002e0, true, true},
|
||||||
|
"DBGWCR3_EL1": a64SysReg{0x1003e0, true, true},
|
||||||
|
"DBGWCR4_EL1": a64SysReg{0x1004e0, true, true},
|
||||||
|
"DBGWCR5_EL1": a64SysReg{0x1005e0, true, true},
|
||||||
|
"DBGWCR6_EL1": a64SysReg{0x1006e0, true, true},
|
||||||
|
"DBGWCR7_EL1": a64SysReg{0x1007e0, true, true},
|
||||||
|
"DBGWCR8_EL1": a64SysReg{0x1008e0, true, true},
|
||||||
|
"DBGWCR9_EL1": a64SysReg{0x1009e0, true, true},
|
||||||
|
"DBGWCR10_EL1": a64SysReg{0x100ae0, true, true},
|
||||||
|
"DBGWCR11_EL1": a64SysReg{0x100be0, true, true},
|
||||||
|
"DBGWCR12_EL1": a64SysReg{0x100ce0, true, true},
|
||||||
|
"DBGWCR13_EL1": a64SysReg{0x100de0, true, true},
|
||||||
|
"DBGWCR14_EL1": a64SysReg{0x100ee0, true, true},
|
||||||
|
"DBGWCR15_EL1": a64SysReg{0x100fe0, true, true},
|
||||||
|
"DBGWVR0_EL1": a64SysReg{0x1000c0, true, true},
|
||||||
|
"DBGWVR1_EL1": a64SysReg{0x1001c0, true, true},
|
||||||
|
"DBGWVR2_EL1": a64SysReg{0x1002c0, true, true},
|
||||||
|
"DBGWVR3_EL1": a64SysReg{0x1003c0, true, true},
|
||||||
|
"DBGWVR4_EL1": a64SysReg{0x1004c0, true, true},
|
||||||
|
"DBGWVR5_EL1": a64SysReg{0x1005c0, true, true},
|
||||||
|
"DBGWVR6_EL1": a64SysReg{0x1006c0, true, true},
|
||||||
|
"DBGWVR7_EL1": a64SysReg{0x1007c0, true, true},
|
||||||
|
"DBGWVR8_EL1": a64SysReg{0x1008c0, true, true},
|
||||||
|
"DBGWVR9_EL1": a64SysReg{0x1009c0, true, true},
|
||||||
|
"DBGWVR10_EL1": a64SysReg{0x100ac0, true, true},
|
||||||
|
"DBGWVR11_EL1": a64SysReg{0x100bc0, true, true},
|
||||||
|
"DBGWVR12_EL1": a64SysReg{0x100cc0, true, true},
|
||||||
|
"DBGWVR13_EL1": a64SysReg{0x100dc0, true, true},
|
||||||
|
"DBGWVR14_EL1": a64SysReg{0x100ec0, true, true},
|
||||||
|
"DBGWVR15_EL1": a64SysReg{0x100fc0, true, true},
|
||||||
|
"DCZID_EL0": a64SysReg{0x1b00e0, true, false},
|
||||||
|
"DISR_EL1": a64SysReg{0x18c120, true, true},
|
||||||
|
"DIT": a64SysReg{0x1b42a0, true, true},
|
||||||
|
"DLR_EL0": a64SysReg{0x1b4520, true, true},
|
||||||
|
"DSPSR_EL0": a64SysReg{0x1b4500, true, true},
|
||||||
|
"ELR_EL1": a64SysReg{0x184020, true, true},
|
||||||
|
"ERRIDR_EL1": a64SysReg{0x185300, true, false},
|
||||||
|
"ERRSELR_EL1": a64SysReg{0x185320, true, true},
|
||||||
|
"ERXADDR_EL1": a64SysReg{0x185460, true, true},
|
||||||
|
"ERXCTLR_EL1": a64SysReg{0x185420, true, true},
|
||||||
|
"ERXFR_EL1": a64SysReg{0x185400, true, false},
|
||||||
|
"ERXMISC0_EL1": a64SysReg{0x185500, true, true},
|
||||||
|
"ERXMISC1_EL1": a64SysReg{0x185520, true, true},
|
||||||
|
"ERXMISC2_EL1": a64SysReg{0x185540, true, true},
|
||||||
|
"ERXMISC3_EL1": a64SysReg{0x185560, true, true},
|
||||||
|
"ERXPFGCDN_EL1": a64SysReg{0x1854c0, true, true},
|
||||||
|
"ERXPFGCTL_EL1": a64SysReg{0x1854a0, true, true},
|
||||||
|
"ERXPFGF_EL1": a64SysReg{0x185480, true, false},
|
||||||
|
"ERXSTATUS_EL1": a64SysReg{0x185440, true, true},
|
||||||
|
"ESR_EL1": a64SysReg{0x185200, true, true},
|
||||||
|
"FAR_EL1": a64SysReg{0x186000, true, true},
|
||||||
|
"FPCR": a64SysReg{0x1b4400, true, true},
|
||||||
|
"FPSR": a64SysReg{0x1b4420, true, true},
|
||||||
|
"GCR_EL1": a64SysReg{0x1810c0, true, true},
|
||||||
|
"GMID_EL1": a64SysReg{0x31400, true, false},
|
||||||
|
"ICC_AP0R0_EL1": a64SysReg{0x18c880, true, true},
|
||||||
|
"ICC_AP0R1_EL1": a64SysReg{0x18c8a0, true, true},
|
||||||
|
"ICC_AP0R2_EL1": a64SysReg{0x18c8c0, true, true},
|
||||||
|
"ICC_AP0R3_EL1": a64SysReg{0x18c8e0, true, true},
|
||||||
|
"ICC_AP1R0_EL1": a64SysReg{0x18c900, true, true},
|
||||||
|
"ICC_AP1R1_EL1": a64SysReg{0x18c920, true, true},
|
||||||
|
"ICC_AP1R2_EL1": a64SysReg{0x18c940, true, true},
|
||||||
|
"ICC_AP1R3_EL1": a64SysReg{0x18c960, true, true},
|
||||||
|
"ICC_ASGI1R_EL1": a64SysReg{0x18cbc0, false, true},
|
||||||
|
"ICC_BPR0_EL1": a64SysReg{0x18c860, true, true},
|
||||||
|
"ICC_BPR1_EL1": a64SysReg{0x18cc60, true, true},
|
||||||
|
"ICC_CTLR_EL1": a64SysReg{0x18cc80, true, true},
|
||||||
|
"ICC_DIR_EL1": a64SysReg{0x18cb20, false, true},
|
||||||
|
"ICC_EOIR0_EL1": a64SysReg{0x18c820, false, true},
|
||||||
|
"ICC_EOIR1_EL1": a64SysReg{0x18cc20, false, true},
|
||||||
|
"ICC_HPPIR0_EL1": a64SysReg{0x18c840, true, false},
|
||||||
|
"ICC_HPPIR1_EL1": a64SysReg{0x18cc40, true, false},
|
||||||
|
"ICC_IAR0_EL1": a64SysReg{0x18c800, true, false},
|
||||||
|
"ICC_IAR1_EL1": a64SysReg{0x18cc00, true, false},
|
||||||
|
"ICC_IGRPEN0_EL1": a64SysReg{0x18ccc0, true, true},
|
||||||
|
"ICC_IGRPEN1_EL1": a64SysReg{0x18cce0, true, true},
|
||||||
|
"ICC_PMR_EL1": a64SysReg{0x184600, true, true},
|
||||||
|
"ICC_RPR_EL1": a64SysReg{0x18cb60, true, false},
|
||||||
|
"ICC_SGI0R_EL1": a64SysReg{0x18cbe0, false, true},
|
||||||
|
"ICC_SGI1R_EL1": a64SysReg{0x18cba0, false, true},
|
||||||
|
"ICC_SRE_EL1": a64SysReg{0x18cca0, true, true},
|
||||||
|
"ICV_AP0R0_EL1": a64SysReg{0x18c880, true, true},
|
||||||
|
"ICV_AP0R1_EL1": a64SysReg{0x18c8a0, true, true},
|
||||||
|
"ICV_AP0R2_EL1": a64SysReg{0x18c8c0, true, true},
|
||||||
|
"ICV_AP0R3_EL1": a64SysReg{0x18c8e0, true, true},
|
||||||
|
"ICV_AP1R0_EL1": a64SysReg{0x18c900, true, true},
|
||||||
|
"ICV_AP1R1_EL1": a64SysReg{0x18c920, true, true},
|
||||||
|
"ICV_AP1R2_EL1": a64SysReg{0x18c940, true, true},
|
||||||
|
"ICV_AP1R3_EL1": a64SysReg{0x18c960, true, true},
|
||||||
|
"ICV_BPR0_EL1": a64SysReg{0x18c860, true, true},
|
||||||
|
"ICV_BPR1_EL1": a64SysReg{0x18cc60, true, true},
|
||||||
|
"ICV_CTLR_EL1": a64SysReg{0x18cc80, true, true},
|
||||||
|
"ICV_DIR_EL1": a64SysReg{0x18cb20, false, true},
|
||||||
|
"ICV_EOIR0_EL1": a64SysReg{0x18c820, false, true},
|
||||||
|
"ICV_EOIR1_EL1": a64SysReg{0x18cc20, false, true},
|
||||||
|
"ICV_HPPIR0_EL1": a64SysReg{0x18c840, true, false},
|
||||||
|
"ICV_HPPIR1_EL1": a64SysReg{0x18cc40, true, false},
|
||||||
|
"ICV_IAR0_EL1": a64SysReg{0x18c800, true, false},
|
||||||
|
"ICV_IAR1_EL1": a64SysReg{0x18cc00, true, false},
|
||||||
|
"ICV_IGRPEN0_EL1": a64SysReg{0x18ccc0, true, true},
|
||||||
|
"ICV_IGRPEN1_EL1": a64SysReg{0x18cce0, true, true},
|
||||||
|
"ICV_PMR_EL1": a64SysReg{0x184600, true, true},
|
||||||
|
"ICV_RPR_EL1": a64SysReg{0x18cb60, true, false},
|
||||||
|
"ID_AA64AFR0_EL1": a64SysReg{0x180580, true, false},
|
||||||
|
"ID_AA64AFR1_EL1": a64SysReg{0x1805a0, true, false},
|
||||||
|
"ID_AA64DFR0_EL1": a64SysReg{0x180500, true, false},
|
||||||
|
"ID_AA64DFR1_EL1": a64SysReg{0x180520, true, false},
|
||||||
|
"ID_AA64ISAR0_EL1": a64SysReg{0x180600, true, false},
|
||||||
|
"ID_AA64ISAR1_EL1": a64SysReg{0x180620, true, false},
|
||||||
|
"ID_AA64MMFR0_EL1": a64SysReg{0x180700, true, false},
|
||||||
|
"ID_AA64MMFR1_EL1": a64SysReg{0x180720, true, false},
|
||||||
|
"ID_AA64MMFR2_EL1": a64SysReg{0x180740, true, false},
|
||||||
|
"ID_AA64PFR0_EL1": a64SysReg{0x180400, true, false},
|
||||||
|
"ID_AA64PFR1_EL1": a64SysReg{0x180420, true, false},
|
||||||
|
"ID_AA64ZFR0_EL1": a64SysReg{0x180480, true, false},
|
||||||
|
"ID_AFR0_EL1": a64SysReg{0x180160, true, false},
|
||||||
|
"ID_DFR0_EL1": a64SysReg{0x180140, true, false},
|
||||||
|
"ID_ISAR0_EL1": a64SysReg{0x180200, true, false},
|
||||||
|
"ID_ISAR1_EL1": a64SysReg{0x180220, true, false},
|
||||||
|
"ID_ISAR2_EL1": a64SysReg{0x180240, true, false},
|
||||||
|
"ID_ISAR3_EL1": a64SysReg{0x180260, true, false},
|
||||||
|
"ID_ISAR4_EL1": a64SysReg{0x180280, true, false},
|
||||||
|
"ID_ISAR5_EL1": a64SysReg{0x1802a0, true, false},
|
||||||
|
"ID_ISAR6_EL1": a64SysReg{0x1802e0, true, false},
|
||||||
|
"ID_MMFR0_EL1": a64SysReg{0x180180, true, false},
|
||||||
|
"ID_MMFR1_EL1": a64SysReg{0x1801a0, true, false},
|
||||||
|
"ID_MMFR2_EL1": a64SysReg{0x1801c0, true, false},
|
||||||
|
"ID_MMFR3_EL1": a64SysReg{0x1801e0, true, false},
|
||||||
|
"ID_MMFR4_EL1": a64SysReg{0x1802c0, true, false},
|
||||||
|
"ID_PFR0_EL1": a64SysReg{0x180100, true, false},
|
||||||
|
"ID_PFR1_EL1": a64SysReg{0x180120, true, false},
|
||||||
|
"ID_PFR2_EL1": a64SysReg{0x180380, true, false},
|
||||||
|
"ISR_EL1": a64SysReg{0x18c100, true, false},
|
||||||
|
"LORC_EL1": a64SysReg{0x18a460, true, true},
|
||||||
|
"LOREA_EL1": a64SysReg{0x18a420, true, true},
|
||||||
|
"LORID_EL1": a64SysReg{0x18a4e0, true, false},
|
||||||
|
"LORN_EL1": a64SysReg{0x18a440, true, true},
|
||||||
|
"LORSA_EL1": a64SysReg{0x18a400, true, true},
|
||||||
|
"MAIR_EL1": a64SysReg{0x18a200, true, true},
|
||||||
|
"MDCCINT_EL1": a64SysReg{0x100200, true, true},
|
||||||
|
"MDCCSR_EL0": a64SysReg{0x130100, true, false},
|
||||||
|
"MDRAR_EL1": a64SysReg{0x101000, true, false},
|
||||||
|
"MDSCR_EL1": a64SysReg{0x100240, true, true},
|
||||||
|
"MIDR_EL1": a64SysReg{0x180000, true, false},
|
||||||
|
"MPAM0_EL1": a64SysReg{0x18a520, true, true},
|
||||||
|
"MPAM1_EL1": a64SysReg{0x18a500, true, true},
|
||||||
|
"MPAMIDR_EL1": a64SysReg{0x18a480, true, false},
|
||||||
|
"MPIDR_EL1": a64SysReg{0x1800a0, true, false},
|
||||||
|
"MVFR0_EL1": a64SysReg{0x180300, true, false},
|
||||||
|
"MVFR1_EL1": a64SysReg{0x180320, true, false},
|
||||||
|
"MVFR2_EL1": a64SysReg{0x180340, true, false},
|
||||||
|
"NZCV": a64SysReg{0x1b4200, true, true},
|
||||||
|
"OSDLR_EL1": a64SysReg{0x101380, true, true},
|
||||||
|
"OSDTRRX_EL1": a64SysReg{0x100040, true, true},
|
||||||
|
"OSDTRTX_EL1": a64SysReg{0x100340, true, true},
|
||||||
|
"OSECCR_EL1": a64SysReg{0x100640, true, true},
|
||||||
|
"OSLAR_EL1": a64SysReg{0x101080, false, true},
|
||||||
|
"OSLSR_EL1": a64SysReg{0x101180, true, false},
|
||||||
|
"PAN": a64SysReg{0x184260, true, true},
|
||||||
|
"PAR_EL1": a64SysReg{0x187400, true, true},
|
||||||
|
"PMBIDR_EL1": a64SysReg{0x189ae0, true, false},
|
||||||
|
"PMBLIMITR_EL1": a64SysReg{0x189a00, true, true},
|
||||||
|
"PMBPTR_EL1": a64SysReg{0x189a20, true, true},
|
||||||
|
"PMBSR_EL1": a64SysReg{0x189a60, true, true},
|
||||||
|
"PMCCFILTR_EL0": a64SysReg{0x1befe0, true, true},
|
||||||
|
"PMCCNTR_EL0": a64SysReg{0x1b9d00, true, true},
|
||||||
|
"PMCEID0_EL0": a64SysReg{0x1b9cc0, true, false},
|
||||||
|
"PMCEID1_EL0": a64SysReg{0x1b9ce0, true, false},
|
||||||
|
"PMCNTENCLR_EL0": a64SysReg{0x1b9c40, true, true},
|
||||||
|
"PMCNTENSET_EL0": a64SysReg{0x1b9c20, true, true},
|
||||||
|
"PMCR_EL0": a64SysReg{0x1b9c00, true, true},
|
||||||
|
"PMEVCNTR0_EL0": a64SysReg{0x1be800, true, true},
|
||||||
|
"PMEVCNTR1_EL0": a64SysReg{0x1be820, true, true},
|
||||||
|
"PMEVCNTR2_EL0": a64SysReg{0x1be840, true, true},
|
||||||
|
"PMEVCNTR3_EL0": a64SysReg{0x1be860, true, true},
|
||||||
|
"PMEVCNTR4_EL0": a64SysReg{0x1be880, true, true},
|
||||||
|
"PMEVCNTR5_EL0": a64SysReg{0x1be8a0, true, true},
|
||||||
|
"PMEVCNTR6_EL0": a64SysReg{0x1be8c0, true, true},
|
||||||
|
"PMEVCNTR7_EL0": a64SysReg{0x1be8e0, true, true},
|
||||||
|
"PMEVCNTR8_EL0": a64SysReg{0x1be900, true, true},
|
||||||
|
"PMEVCNTR9_EL0": a64SysReg{0x1be920, true, true},
|
||||||
|
"PMEVCNTR10_EL0": a64SysReg{0x1be940, true, true},
|
||||||
|
"PMEVCNTR11_EL0": a64SysReg{0x1be960, true, true},
|
||||||
|
"PMEVCNTR12_EL0": a64SysReg{0x1be980, true, true},
|
||||||
|
"PMEVCNTR13_EL0": a64SysReg{0x1be9a0, true, true},
|
||||||
|
"PMEVCNTR14_EL0": a64SysReg{0x1be9c0, true, true},
|
||||||
|
"PMEVCNTR15_EL0": a64SysReg{0x1be9e0, true, true},
|
||||||
|
"PMEVCNTR16_EL0": a64SysReg{0x1bea00, true, true},
|
||||||
|
"PMEVCNTR17_EL0": a64SysReg{0x1bea20, true, true},
|
||||||
|
"PMEVCNTR18_EL0": a64SysReg{0x1bea40, true, true},
|
||||||
|
"PMEVCNTR19_EL0": a64SysReg{0x1bea60, true, true},
|
||||||
|
"PMEVCNTR20_EL0": a64SysReg{0x1bea80, true, true},
|
||||||
|
"PMEVCNTR21_EL0": a64SysReg{0x1beaa0, true, true},
|
||||||
|
"PMEVCNTR22_EL0": a64SysReg{0x1beac0, true, true},
|
||||||
|
"PMEVCNTR23_EL0": a64SysReg{0x1beae0, true, true},
|
||||||
|
"PMEVCNTR24_EL0": a64SysReg{0x1beb00, true, true},
|
||||||
|
"PMEVCNTR25_EL0": a64SysReg{0x1beb20, true, true},
|
||||||
|
"PMEVCNTR26_EL0": a64SysReg{0x1beb40, true, true},
|
||||||
|
"PMEVCNTR27_EL0": a64SysReg{0x1beb60, true, true},
|
||||||
|
"PMEVCNTR28_EL0": a64SysReg{0x1beb80, true, true},
|
||||||
|
"PMEVCNTR29_EL0": a64SysReg{0x1beba0, true, true},
|
||||||
|
"PMEVCNTR30_EL0": a64SysReg{0x1bebc0, true, true},
|
||||||
|
"PMEVTYPER0_EL0": a64SysReg{0x1bec00, true, true},
|
||||||
|
"PMEVTYPER1_EL0": a64SysReg{0x1bec20, true, true},
|
||||||
|
"PMEVTYPER2_EL0": a64SysReg{0x1bec40, true, true},
|
||||||
|
"PMEVTYPER3_EL0": a64SysReg{0x1bec60, true, true},
|
||||||
|
"PMEVTYPER4_EL0": a64SysReg{0x1bec80, true, true},
|
||||||
|
"PMEVTYPER5_EL0": a64SysReg{0x1beca0, true, true},
|
||||||
|
"PMEVTYPER6_EL0": a64SysReg{0x1becc0, true, true},
|
||||||
|
"PMEVTYPER7_EL0": a64SysReg{0x1bece0, true, true},
|
||||||
|
"PMEVTYPER8_EL0": a64SysReg{0x1bed00, true, true},
|
||||||
|
"PMEVTYPER9_EL0": a64SysReg{0x1bed20, true, true},
|
||||||
|
"PMEVTYPER10_EL0": a64SysReg{0x1bed40, true, true},
|
||||||
|
"PMEVTYPER11_EL0": a64SysReg{0x1bed60, true, true},
|
||||||
|
"PMEVTYPER12_EL0": a64SysReg{0x1bed80, true, true},
|
||||||
|
"PMEVTYPER13_EL0": a64SysReg{0x1beda0, true, true},
|
||||||
|
"PMEVTYPER14_EL0": a64SysReg{0x1bedc0, true, true},
|
||||||
|
"PMEVTYPER15_EL0": a64SysReg{0x1bede0, true, true},
|
||||||
|
"PMEVTYPER16_EL0": a64SysReg{0x1bee00, true, true},
|
||||||
|
"PMEVTYPER17_EL0": a64SysReg{0x1bee20, true, true},
|
||||||
|
"PMEVTYPER18_EL0": a64SysReg{0x1bee40, true, true},
|
||||||
|
"PMEVTYPER19_EL0": a64SysReg{0x1bee60, true, true},
|
||||||
|
"PMEVTYPER20_EL0": a64SysReg{0x1bee80, true, true},
|
||||||
|
"PMEVTYPER21_EL0": a64SysReg{0x1beea0, true, true},
|
||||||
|
"PMEVTYPER22_EL0": a64SysReg{0x1beec0, true, true},
|
||||||
|
"PMEVTYPER23_EL0": a64SysReg{0x1beee0, true, true},
|
||||||
|
"PMEVTYPER24_EL0": a64SysReg{0x1bef00, true, true},
|
||||||
|
"PMEVTYPER25_EL0": a64SysReg{0x1bef20, true, true},
|
||||||
|
"PMEVTYPER26_EL0": a64SysReg{0x1bef40, true, true},
|
||||||
|
"PMEVTYPER27_EL0": a64SysReg{0x1bef60, true, true},
|
||||||
|
"PMEVTYPER28_EL0": a64SysReg{0x1bef80, true, true},
|
||||||
|
"PMEVTYPER29_EL0": a64SysReg{0x1befa0, true, true},
|
||||||
|
"PMEVTYPER30_EL0": a64SysReg{0x1befc0, true, true},
|
||||||
|
"PMINTENCLR_EL1": a64SysReg{0x189e40, true, true},
|
||||||
|
"PMINTENSET_EL1": a64SysReg{0x189e20, true, true},
|
||||||
|
"PMMIR_EL1": a64SysReg{0x189ec0, true, false},
|
||||||
|
"PMOVSCLR_EL0": a64SysReg{0x1b9c60, true, true},
|
||||||
|
"PMOVSSET_EL0": a64SysReg{0x1b9e60, true, true},
|
||||||
|
"PMSCR_EL1": a64SysReg{0x189900, true, true},
|
||||||
|
"PMSELR_EL0": a64SysReg{0x1b9ca0, true, true},
|
||||||
|
"PMSEVFR_EL1": a64SysReg{0x1899a0, true, true},
|
||||||
|
"PMSFCR_EL1": a64SysReg{0x189980, true, true},
|
||||||
|
"PMSICR_EL1": a64SysReg{0x189940, true, true},
|
||||||
|
"PMSIDR_EL1": a64SysReg{0x1899e0, true, false},
|
||||||
|
"PMSIRR_EL1": a64SysReg{0x189960, true, true},
|
||||||
|
"PMSLATFR_EL1": a64SysReg{0x1899c0, true, true},
|
||||||
|
"PMSWINC_EL0": a64SysReg{0x1b9c80, false, true},
|
||||||
|
"PMUSERENR_EL0": a64SysReg{0x1b9e00, true, true},
|
||||||
|
"PMXEVCNTR_EL0": a64SysReg{0x1b9d40, true, true},
|
||||||
|
"PMXEVTYPER_EL0": a64SysReg{0x1b9d20, true, true},
|
||||||
|
"REVIDR_EL1": a64SysReg{0x1800c0, true, false},
|
||||||
|
"RGSR_EL1": a64SysReg{0x1810a0, true, true},
|
||||||
|
"RMR_EL1": a64SysReg{0x18c040, true, true},
|
||||||
|
"RNDR": a64SysReg{0x1b2400, true, false},
|
||||||
|
"RNDRRS": a64SysReg{0x1b2420, true, false},
|
||||||
|
"RVBAR_EL1": a64SysReg{0x18c020, true, false},
|
||||||
|
"SCTLR_EL1": a64SysReg{0x181000, true, true},
|
||||||
|
"SCXTNUM_EL0": a64SysReg{0x1bd0e0, true, true},
|
||||||
|
"SCXTNUM_EL1": a64SysReg{0x18d0e0, true, true},
|
||||||
|
"SP_EL0": a64SysReg{0x184100, true, true},
|
||||||
|
"SP_EL1": a64SysReg{0x1c4100, true, true},
|
||||||
|
"SPSel": a64SysReg{0x184200, true, true},
|
||||||
|
"SPSR_abt": a64SysReg{0x1c4320, true, true},
|
||||||
|
"SPSR_EL1": a64SysReg{0x184000, true, true},
|
||||||
|
"SPSR_fiq": a64SysReg{0x1c4360, true, true},
|
||||||
|
"SPSR_irq": a64SysReg{0x1c4300, true, true},
|
||||||
|
"SPSR_und": a64SysReg{0x1c4340, true, true},
|
||||||
|
"SSBS": a64SysReg{0x1b42c0, true, true},
|
||||||
|
"TCO": a64SysReg{0x1b42e0, true, true},
|
||||||
|
"TCR_EL1": a64SysReg{0x182040, true, true},
|
||||||
|
"TFSR_EL1": a64SysReg{0x185600, true, true},
|
||||||
|
"TFSRE0_EL1": a64SysReg{0x185620, true, true},
|
||||||
|
"TPIDR_EL0": a64SysReg{0x1bd040, true, true},
|
||||||
|
"TPIDR_EL1": a64SysReg{0x18d080, true, true},
|
||||||
|
"TPIDRRO_EL0": a64SysReg{0x1bd060, true, true},
|
||||||
|
"TRFCR_EL1": a64SysReg{0x181220, true, true},
|
||||||
|
"TTBR0_EL1": a64SysReg{0x182000, true, true},
|
||||||
|
"TTBR1_EL1": a64SysReg{0x182020, true, true},
|
||||||
|
"UAO": a64SysReg{0x184280, true, true},
|
||||||
|
"VBAR_EL1": a64SysReg{0x18c000, true, true},
|
||||||
|
"ZCR_EL1": a64SysReg{0x181200, true, true},
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64SysInst is one TLBI alias: the fields the SYS encoding carries beside
|
||||||
|
// the fixed op0 = 01 and CRn = 8.
|
||||||
|
type a64SysInst struct {
|
||||||
|
op1, cm, op2 uint32
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64TLBIOps maps the TLBI operation names to their fields; the register
|
||||||
|
// operand is optional and defaults to ZR.
|
||||||
|
var a64TLBIOps = map[string]a64SysInst{
|
||||||
|
"ALLE1": {0x4, 0x7, 0x4},
|
||||||
|
"ALLE1IS": {0x4, 0x3, 0x4},
|
||||||
|
"ALLE1OS": {0x4, 0x1, 0x4},
|
||||||
|
"ALLE2": {0x4, 0x7, 0x0},
|
||||||
|
"ALLE2IS": {0x4, 0x3, 0x0},
|
||||||
|
"ALLE2OS": {0x4, 0x1, 0x0},
|
||||||
|
"ALLE3": {0x6, 0x7, 0x0},
|
||||||
|
"ALLE3IS": {0x6, 0x3, 0x0},
|
||||||
|
"ALLE3OS": {0x6, 0x1, 0x0},
|
||||||
|
"ASIDE1": {0x0, 0x7, 0x2},
|
||||||
|
"ASIDE1IS": {0x0, 0x3, 0x2},
|
||||||
|
"ASIDE1OS": {0x0, 0x1, 0x2},
|
||||||
|
"IPAS2E1": {0x4, 0x4, 0x1},
|
||||||
|
"IPAS2E1IS": {0x4, 0x0, 0x1},
|
||||||
|
"IPAS2E1OS": {0x4, 0x4, 0x0},
|
||||||
|
"IPAS2LE1": {0x4, 0x4, 0x5},
|
||||||
|
"IPAS2LE1IS": {0x4, 0x0, 0x5},
|
||||||
|
"IPAS2LE1OS": {0x4, 0x4, 0x4},
|
||||||
|
"RIPAS2E1": {0x4, 0x4, 0x2},
|
||||||
|
"RIPAS2E1IS": {0x4, 0x0, 0x2},
|
||||||
|
"RIPAS2E1OS": {0x4, 0x4, 0x3},
|
||||||
|
"RIPAS2LE1": {0x4, 0x4, 0x6},
|
||||||
|
"RIPAS2LE1IS": {0x4, 0x0, 0x6},
|
||||||
|
"RIPAS2LE1OS": {0x4, 0x4, 0x7},
|
||||||
|
"RVAAE1": {0x0, 0x6, 0x3},
|
||||||
|
"RVAAE1IS": {0x0, 0x2, 0x3},
|
||||||
|
"RVAAE1OS": {0x0, 0x5, 0x3},
|
||||||
|
"RVAALE1": {0x0, 0x6, 0x7},
|
||||||
|
"RVAALE1IS": {0x0, 0x2, 0x7},
|
||||||
|
"RVAALE1OS": {0x0, 0x5, 0x7},
|
||||||
|
"RVAE1": {0x0, 0x6, 0x1},
|
||||||
|
"RVAE1IS": {0x0, 0x2, 0x1},
|
||||||
|
"RVAE1OS": {0x0, 0x5, 0x1},
|
||||||
|
"RVAE2": {0x4, 0x6, 0x1},
|
||||||
|
"RVAE2IS": {0x4, 0x2, 0x1},
|
||||||
|
"RVAE2OS": {0x4, 0x5, 0x1},
|
||||||
|
"RVAE3": {0x6, 0x6, 0x1},
|
||||||
|
"RVAE3IS": {0x6, 0x2, 0x1},
|
||||||
|
"RVAE3OS": {0x6, 0x5, 0x1},
|
||||||
|
"RVALE1": {0x0, 0x6, 0x5},
|
||||||
|
"RVALE1IS": {0x0, 0x2, 0x5},
|
||||||
|
"RVALE1OS": {0x0, 0x5, 0x5},
|
||||||
|
"RVALE2": {0x4, 0x6, 0x5},
|
||||||
|
"RVALE2IS": {0x4, 0x2, 0x5},
|
||||||
|
"RVALE2OS": {0x4, 0x5, 0x5},
|
||||||
|
"RVALE3": {0x6, 0x6, 0x5},
|
||||||
|
"RVALE3IS": {0x6, 0x2, 0x5},
|
||||||
|
"RVALE3OS": {0x6, 0x5, 0x5},
|
||||||
|
"VAAE1": {0x0, 0x7, 0x3},
|
||||||
|
"VAAE1IS": {0x0, 0x3, 0x3},
|
||||||
|
"VAAE1OS": {0x0, 0x1, 0x3},
|
||||||
|
"VAALE1": {0x0, 0x7, 0x7},
|
||||||
|
"VAALE1IS": {0x0, 0x3, 0x7},
|
||||||
|
"VAALE1OS": {0x0, 0x1, 0x7},
|
||||||
|
"VAE1": {0x0, 0x7, 0x1},
|
||||||
|
"VAE1IS": {0x0, 0x3, 0x1},
|
||||||
|
"VAE1OS": {0x0, 0x1, 0x1},
|
||||||
|
"VAE2": {0x4, 0x7, 0x1},
|
||||||
|
"VAE2IS": {0x4, 0x3, 0x1},
|
||||||
|
"VAE2OS": {0x4, 0x1, 0x1},
|
||||||
|
"VAE3": {0x6, 0x7, 0x1},
|
||||||
|
"VAE3IS": {0x6, 0x3, 0x1},
|
||||||
|
"VAE3OS": {0x6, 0x1, 0x1},
|
||||||
|
"VALE1": {0x0, 0x7, 0x5},
|
||||||
|
"VALE1IS": {0x0, 0x3, 0x5},
|
||||||
|
"VALE1OS": {0x0, 0x1, 0x5},
|
||||||
|
"VALE2": {0x4, 0x7, 0x5},
|
||||||
|
"VALE2IS": {0x4, 0x3, 0x5},
|
||||||
|
"VALE2OS": {0x4, 0x1, 0x5},
|
||||||
|
"VALE3": {0x6, 0x7, 0x5},
|
||||||
|
"VALE3IS": {0x6, 0x3, 0x5},
|
||||||
|
"VALE3OS": {0x6, 0x1, 0x5},
|
||||||
|
"VMALLE1": {0x0, 0x7, 0x0},
|
||||||
|
"VMALLE1IS": {0x0, 0x3, 0x0},
|
||||||
|
"VMALLE1OS": {0x0, 0x1, 0x0},
|
||||||
|
"VMALLS12E1": {0x4, 0x7, 0x6},
|
||||||
|
"VMALLS12E1IS": {0x4, 0x3, 0x6},
|
||||||
|
"VMALLS12E1OS": {0x4, 0x1, 0x6},
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64DCOps2 maps the DC operation names to their fields; the register
|
||||||
|
// operand is mandatory.
|
||||||
|
var a64DCOps2 = map[string]a64SysInst{
|
||||||
|
"CGDSW": {0x0, 0xa, 0x6},
|
||||||
|
"CGDVAC": {0x3, 0xa, 0x5},
|
||||||
|
"CGDVADP": {0x3, 0xd, 0x5},
|
||||||
|
"CGDVAP": {0x3, 0xc, 0x5},
|
||||||
|
"CGSW": {0x0, 0xa, 0x4},
|
||||||
|
"CGVAC": {0x3, 0xa, 0x3},
|
||||||
|
"CGVADP": {0x3, 0xd, 0x3},
|
||||||
|
"CGVAP": {0x3, 0xc, 0x3},
|
||||||
|
"CIGDSW": {0x0, 0xe, 0x6},
|
||||||
|
"CIGDVAC": {0x3, 0xe, 0x5},
|
||||||
|
"CIGSW": {0x0, 0xe, 0x4},
|
||||||
|
"CIGVAC": {0x3, 0xe, 0x3},
|
||||||
|
"CISW": {0x0, 0xe, 0x2},
|
||||||
|
"CIVAC": {0x3, 0xe, 0x1},
|
||||||
|
"CSW": {0x0, 0xa, 0x2},
|
||||||
|
"CVAC": {0x3, 0xa, 0x1},
|
||||||
|
"CVADP": {0x3, 0xd, 0x1},
|
||||||
|
"CVAP": {0x3, 0xc, 0x1},
|
||||||
|
"CVAU": {0x3, 0xb, 0x1},
|
||||||
|
"GVA": {0x3, 0x4, 0x3},
|
||||||
|
"GZVA": {0x3, 0x4, 0x4},
|
||||||
|
"IGDSW": {0x0, 0x6, 0x6},
|
||||||
|
"IGDVAC": {0x0, 0x6, 0x5},
|
||||||
|
"IGSW": {0x0, 0x6, 0x4},
|
||||||
|
"IGVAC": {0x0, 0x6, 0x3},
|
||||||
|
"ISW": {0x0, 0x6, 0x2},
|
||||||
|
"IVAC": {0x0, 0x6, 0x1},
|
||||||
|
"ZVA": {0x3, 0x4, 0x1},
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64RPRFOps maps the range-prefetch operation names to their 6-bit values.
|
||||||
|
var a64RPRFOps = map[string]uint32{
|
||||||
|
"PLDKEEP": 0,
|
||||||
|
"PLDSTRM": 4,
|
||||||
|
"PSTKEEP": 1,
|
||||||
|
"PSTSTRM": 5,
|
||||||
|
}
|
||||||
@@ -0,0 +1,164 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"slices"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestARM64SysRegsDifferential proves the whole system-register table against
|
||||||
|
// the toolchain at once: one TEXT whose body reads every register the table
|
||||||
|
// carries (and writes every writable one), assembled by gasm and by
|
||||||
|
// go tool asm, must agree byte for byte. A single wrong op0/op1/CRn/CRm/op2
|
||||||
|
// packing names its register through the first differing word.
|
||||||
|
func TestARM64SysRegsDifferential(t *testing.T) {
|
||||||
|
names := make([]string, 0, len(a64SysRegs))
|
||||||
|
for name := range a64SysRegs {
|
||||||
|
names = append(names, name)
|
||||||
|
}
|
||||||
|
slices.Sort(names)
|
||||||
|
|
||||||
|
var body strings.Builder
|
||||||
|
for i, name := range names {
|
||||||
|
// R18 is the arm64 platform register and R29-R31 carry dedicated
|
||||||
|
// meanings; a plain read/write destination keeps to R0-R17.
|
||||||
|
reg := fmt.Sprintf("R%d", i%18)
|
||||||
|
if a64SysRegs[name].read {
|
||||||
|
body.WriteString(fmt.Sprintf("\tMRS %s, %s\n", name, reg))
|
||||||
|
}
|
||||||
|
if a64SysRegs[name].write {
|
||||||
|
body.WriteString(fmt.Sprintf("\tMSR %s, %s\n", reg, name))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
src := "#include \"textflag.h\"\n\nTEXT ·sysregs(SB), NOSPLIT, $0\n" + body.String() + "\tRET\n"
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
path := filepath.Join(dir, "sysregs_arm64.s")
|
||||||
|
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
assertARM64Differential(t, path, src, "sysregs")
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestARM64FamiliesDifferential pins the non-sysreg families the arm64
|
||||||
|
// campaign added: the LSE compare-and-swap pairs, the VMOVI immediate, the
|
||||||
|
// SIMD narrow/long shift pairs, the VLD2/VLD3/VLD4 and VST2/VST3/VST4
|
||||||
|
// structure accesses with their post-index and replicate forms, LDPSW, the
|
||||||
|
// pointer-authentication hint and the DC maintenance operation. Every
|
||||||
|
// spelling is the toolchain's own, taken from its arm64 testdata, and the
|
||||||
|
// bytes must agree word for word.
|
||||||
|
func TestARM64FamiliesDifferential(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·families(SB), NOSPLIT, $0
|
||||||
|
CASPD (R2, R3), (R2), (R8, R9)
|
||||||
|
CASPW (R6, R7), (R8), (R4, R5)
|
||||||
|
VMOVI $82, V0.B16
|
||||||
|
VMOVI $146, V22.B16
|
||||||
|
VSSHLL $0, V1.B8, V2.H8
|
||||||
|
VSSHLL $7, V1.B8, V2.H8
|
||||||
|
VSSHLL2 $0, V1.B16, V2.H8
|
||||||
|
VSHRN $7, V1.H8, V0.B8
|
||||||
|
VSHRN2 $31, V1.D2, V0.S4
|
||||||
|
VLD2 (R29), [V23.H8, V24.H8]
|
||||||
|
VLD2.P 16(R0), [V18.B8, V19.B8]
|
||||||
|
VLD2.P (R1)(R2), [V15.S2, V16.S2]
|
||||||
|
VLD3 (R27), [V11.S4, V12.S4, V13.S4]
|
||||||
|
VLD3.P 48(RSP), [V11.S4, V12.S4, V13.S4]
|
||||||
|
VLD4 (R15), [V10.H4, V11.H4, V12.H4, V13.H4]
|
||||||
|
VLD4.P 32(R24), [V31.B8, V0.B8, V1.B8, V2.B8]
|
||||||
|
VLD1R (R1), [V9.B8]
|
||||||
|
VLD1R.P (R0), [V0.B16]
|
||||||
|
VLD1R.P 2(R1), [V2.H4]
|
||||||
|
VLD2R (R15), [V15.H4, V16.H4]
|
||||||
|
VLD2R.P 16(R0), [V0.D2, V1.D2]
|
||||||
|
VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]
|
||||||
|
VLD4R.P 16(RSP), [V31.S4, V0.S4, V1.S4, V2.S4]
|
||||||
|
VST2 [V22.H8, V23.H8], (R23)
|
||||||
|
VST2.P [V14.H4, V15.H4], 16(R17)
|
||||||
|
VST2.P [V14.H4, V15.H4], (R3)(R17)
|
||||||
|
VST3 [V1.D2, V2.D2, V3.D2], (R11)
|
||||||
|
VST3.P [V18.S4, V19.S4, V20.S4], 48(R25)
|
||||||
|
VST4 [V22.D2, V23.D2, V24.D2, V25.D2], (R3)
|
||||||
|
VST4.P [V14.D2, V15.D2, V16.D2, V17.D2], 64(R15)
|
||||||
|
LDPSW (R0), (R1, R2)
|
||||||
|
LDPSW 4(R0), (R1, R2)
|
||||||
|
LDPSW -4(R0), (R1, R2)
|
||||||
|
PACIASP
|
||||||
|
DC IVAC, R1
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
dir := t.TempDir()
|
||||||
|
path := filepath.Join(dir, "families_arm64.s")
|
||||||
|
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
assertARM64Differential(t, path, src, "families")
|
||||||
|
}
|
||||||
|
|
||||||
|
// assertARM64Differential assembles the same source with gasm and with the
|
||||||
|
// toolchain for arm64 and requires the named function's code bytes to agree.
|
||||||
|
// The live oracle is a deliberate-run comparison, so -short skips it (the
|
||||||
|
// push pipeline's mode); the golden bytes of the individual encoders are
|
||||||
|
// pinned separately in every mode.
|
||||||
|
func assertARM64Differential(t *testing.T, path, src, fn string) {
|
||||||
|
t.Helper()
|
||||||
|
oracle := oracleFuncCode(t, toolAsmObject(t, path, "arm64"))
|
||||||
|
|
||||||
|
// The oracle keys its functions by the qualified object name
|
||||||
|
// (pkg.name); match on the local part.
|
||||||
|
want := map[string][]byte{}
|
||||||
|
for name, code := range oracle {
|
||||||
|
if _, after, ok := strings.Cut(name, "."); ok {
|
||||||
|
want[after] = code
|
||||||
|
} else {
|
||||||
|
want[name] = code
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if want[fn] == nil {
|
||||||
|
t.Fatalf("the oracle object carries no function %q (has %v)", fn, keysOf(want))
|
||||||
|
}
|
||||||
|
|
||||||
|
f, perrs := parser.Parse(path, src)
|
||||||
|
if len(perrs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", perrs[0])
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
got := trimTrailingZeroWords(img.Code)
|
||||||
|
wantB := trimTrailingZeroWords(want[fn])
|
||||||
|
if len(got) != len(wantB) {
|
||||||
|
t.Fatalf("gasm %d bytes, oracle %d bytes", len(got), len(wantB))
|
||||||
|
}
|
||||||
|
for i := range wantB {
|
||||||
|
if got[i] != wantB[i] {
|
||||||
|
t.Fatalf("word %d differs: gasm %08x, oracle %08x", i/4,
|
||||||
|
binary.LittleEndian.Uint32(got[i:i+4]), binary.LittleEndian.Uint32(wantB[i:i+4]))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// trimTrailingZeroWords drops whole zero words off the end of a code span:
|
||||||
|
// an object pads a function to its alignment, and the raw image does not.
|
||||||
|
// A difference in the middle survives the trim untouched.
|
||||||
|
func trimTrailingZeroWords(b []byte) []byte {
|
||||||
|
for len(b) >= 4 {
|
||||||
|
last := b[len(b)-4:]
|
||||||
|
if last[0]|last[1]|last[2]|last[3] != 0 {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
b = b[:len(b)-4]
|
||||||
|
}
|
||||||
|
return b
|
||||||
|
}
|
||||||
+865
-74
File diff suppressed because it is too large
Load Diff
+274
-4
@@ -10,8 +10,8 @@ import (
|
|||||||
|
|
||||||
"golang.org/x/arch/x86/x86asm"
|
"golang.org/x/arch/x86/x86asm"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// firstText parses src and returns its first TEXT function.
|
// firstText parses src and returns its first TEXT function.
|
||||||
@@ -160,6 +160,57 @@ TEXT ·loadarg(SB), NOSPLIT, $0-24
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAssembleFramelessCall verifies the forced base-pointer frame a $0-frame
|
||||||
|
// function containing a CALL receives: the PUSHQ BP prologue with no stack
|
||||||
|
// adjustment and the x+N(FP) → (N+16)(SP) translation, against the bytes the
|
||||||
|
// Go assembler produces. The push is the frame, so the offset must not count
|
||||||
|
// it twice.
|
||||||
|
func TestAssembleFramelessCall(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("frameless_call_amd64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·withcall(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ x+0(FP), AX
|
||||||
|
CALL ·other(SB)
|
||||||
|
MOVQ AX, ret+8(FP)
|
||||||
|
RET
|
||||||
|
TEXT ·other(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
code := append([]byte(nil), img.Code[img.Funcs[0].Offset:img.Funcs[0].Offset+img.Funcs[0].Size]...)
|
||||||
|
for _, r := range img.Funcs[0].Relocs {
|
||||||
|
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||||
|
code[j] = 0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// From `go tool objdump` of the Go-assembled function:
|
||||||
|
// PUSHQ BP 55
|
||||||
|
// MOVQ SP, BP 4889e5
|
||||||
|
// MOVQ 0x10(SP), AX 488b442410
|
||||||
|
// CALL other e800000000
|
||||||
|
// MOVQ AX, 0x18(SP) 4889442418
|
||||||
|
// POPQ BP 5d
|
||||||
|
// RET c3
|
||||||
|
want := []byte{
|
||||||
|
0x55,
|
||||||
|
0x48, 0x89, 0xe5,
|
||||||
|
0x48, 0x8b, 0x44, 0x24, 0x10,
|
||||||
|
0xe8, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x48, 0x89, 0x44, 0x24, 0x18,
|
||||||
|
0x5d,
|
||||||
|
0xc3,
|
||||||
|
}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("frameless CALL FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style
|
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style
|
||||||
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
|
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
|
||||||
// the bytes the Go assembler produces.
|
// the bytes the Go assembler produces.
|
||||||
@@ -202,8 +253,8 @@ TEXT ·withframe(SB), NOSPLIT, $16-16
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestAssembleVexKernel assembles the horizontal-sum reduction the go-flac
|
// TestAssembleVexKernel assembles the horizontal-sum reduction the go-flac
|
||||||
// kernels end with — exercising the VEX moves, shuffle and extract forms
|
// kernels end with; exercising the VEX moves, shuffle and extract forms
|
||||||
// through the full parser → encoder path — and checks the output is
|
// through the full parser → encoder path; and checks the output is
|
||||||
// byte-identical to the Go assembler's.
|
// byte-identical to the Go assembler's.
|
||||||
func TestAssembleVexKernel(t *testing.T) {
|
func TestAssembleVexKernel(t *testing.T) {
|
||||||
fn := firstText(t, `
|
fn := firstText(t, `
|
||||||
@@ -319,6 +370,58 @@ end:
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAssembleNumericPCJumps pins the numeric ±N(PC) branch operands: N
|
||||||
|
// counts instruction statements, skipping labels, in both directions (the
|
||||||
|
// runtime's exit loops write JMP -3(PC)), N = 0 parks on the jump itself.
|
||||||
|
func TestAssembleNumericPCJumps(t *testing.T) {
|
||||||
|
fn := firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·exit(SB), NOSPLIT, $0
|
||||||
|
MOVB $1, AL
|
||||||
|
lab:
|
||||||
|
MOVB $2, AL
|
||||||
|
MOVB $3, AL
|
||||||
|
JMP -3(PC)
|
||||||
|
MOVB $4, AL
|
||||||
|
park:
|
||||||
|
JMP 0(PC)
|
||||||
|
MOVB $5, AL
|
||||||
|
JMP 2(PC)
|
||||||
|
MOVB $6, AL
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, err := Assemble(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Assemble: %v", err)
|
||||||
|
}
|
||||||
|
// From the Go-assembled function:
|
||||||
|
// MOVB $1, AL b001
|
||||||
|
// MOVB $2, AL b002
|
||||||
|
// MOVB $3, AL b003
|
||||||
|
// JMP -3(PC) ebf8 (three instructions back, past lab:)
|
||||||
|
// MOVB $4, AL b004
|
||||||
|
// JMP 0(PC) ebfe (the park loop)
|
||||||
|
// MOVB $5, AL b005
|
||||||
|
// JMP 2(PC) eb02 (over MOVB $6 to the RET)
|
||||||
|
// MOVB $6, AL b006
|
||||||
|
// RET c3
|
||||||
|
want := []byte{
|
||||||
|
0xb0, 0x01,
|
||||||
|
0xb0, 0x02,
|
||||||
|
0xb0, 0x03,
|
||||||
|
0xeb, 0xf8,
|
||||||
|
0xb0, 0x04,
|
||||||
|
0xeb, 0xfe,
|
||||||
|
0xb0, 0x05,
|
||||||
|
0xeb, 0x02,
|
||||||
|
0xb0, 0x06,
|
||||||
|
0xc3,
|
||||||
|
}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("numeric-PC mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestAssemblePrefetch(t *testing.T) {
|
func TestAssemblePrefetch(t *testing.T) {
|
||||||
fn := firstText(t, `
|
fn := firstText(t, `
|
||||||
#include "textflag.h"
|
#include "textflag.h"
|
||||||
@@ -353,6 +456,19 @@ TEXT ·pf(SB), NOSPLIT, $0
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAssembleBareJump checks that a zero-operand jump (which parses, because
|
||||||
|
// the parser does not arity-check mnemonics) is rejected with an error rather
|
||||||
|
// than panicking in the layout loop, which indexes Operands[0] before the
|
||||||
|
// emission pass gets a chance to diagnose the arity.
|
||||||
|
func TestAssembleBareJump(t *testing.T) {
|
||||||
|
for _, mnem := range []string{"JE", "JMP", "JLT", "CALL"} {
|
||||||
|
fn := firstText(t, "TEXT ·bare(SB), $16-0\n\t"+mnem+"\n")
|
||||||
|
if _, _, err := Assemble(fn); err == nil {
|
||||||
|
t.Errorf("%s with no operand: expected an error, got none", mnem)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
|
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
|
||||||
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
|
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
|
||||||
// larger. The intermediate 129..255 range used to encode an ADD with a
|
// larger. The intermediate 129..255 range used to encode an ADD with a
|
||||||
@@ -375,3 +491,157 @@ func TestSubSPEncodings(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAssemblePseudoStatements runs LOCK/REP, BYTE/WORD and END through the
|
||||||
|
// full statement pipeline, pinned against go tool asm (Go 1.27, amd64). It
|
||||||
|
// asserts the three behaviours the toolchain shows: each prefix statement is
|
||||||
|
// a standalone byte with a PC of its own (so a label placed on the LOCK
|
||||||
|
// points at the F0), the data pseudo-ops write their literal bytes inline,
|
||||||
|
// and END terminates nothing (the statements after it still belong to the
|
||||||
|
// function and carry no trace of it).
|
||||||
|
func TestAssemblePseudoStatements(t *testing.T) {
|
||||||
|
fn := firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·pseudo(SB), NOSPLIT, $0-0
|
||||||
|
pfx:
|
||||||
|
LOCK
|
||||||
|
CMPXCHGQ AX, (BX)
|
||||||
|
REP
|
||||||
|
MOVSQ
|
||||||
|
BYTE $0x0f
|
||||||
|
BYTE $0x1f
|
||||||
|
WORD $0x1234
|
||||||
|
END
|
||||||
|
BYTE $0x02
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, labels, err := Assemble(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Assemble: %v", err)
|
||||||
|
}
|
||||||
|
// go tool asm: f0 480fb103 f3 48a5 0f 1f 3412 02 c3
|
||||||
|
want := []byte{
|
||||||
|
0xf0,
|
||||||
|
0x48, 0x0f, 0xb1, 0x03,
|
||||||
|
0xf3, 0x48, 0xa5,
|
||||||
|
0x0f, 0x1f, 0x34, 0x12,
|
||||||
|
0x02, 0xc3,
|
||||||
|
}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("pseudo statements:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
// The label sits on the LOCK byte, exactly where the toolchain's PC
|
||||||
|
// listing puts it.
|
||||||
|
if off := labels["pfx"]; off != 0 {
|
||||||
|
t.Errorf("label pfx = %d, want 0 (the LOCK's own byte)", off)
|
||||||
|
}
|
||||||
|
// The trailing BYTE lands where the layout says: after the 8 bytes of
|
||||||
|
// LOCK, CMPXCHGQ, REP and MOVSQ plus the 4 data bytes, END contributing
|
||||||
|
// none.
|
||||||
|
if code[12] != 0x02 {
|
||||||
|
t.Errorf("byte at 12 = %02x, want 02 (the BYTE after END)", code[12])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAssembleAdjspBalance pins the toolchain's push/pop balance rule over
|
||||||
|
// ADJSP: the straight-line sum of the adjustments must be zero at each
|
||||||
|
// RET, branches in between counting for nothing (verified against go tool
|
||||||
|
// asm: ADJSP $16 before a RET is reported as "unbalanced PUSH/POP", a
|
||||||
|
// $16/$-16 pair with a JMP in between assembles).
|
||||||
|
func TestAssembleAdjspBalance(t *testing.T) {
|
||||||
|
// Balanced pair with a branch in between, bytes pinned from go tool asm.
|
||||||
|
fn := firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·adjsp(SB), NOSPLIT, $0-0
|
||||||
|
ADJSP $16
|
||||||
|
JMP body
|
||||||
|
body:
|
||||||
|
ADJSP $-16
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, err := Assemble(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Assemble: %v", err)
|
||||||
|
}
|
||||||
|
want := []byte{0x48, 0x83, 0xEC, 0x10, 0xEB, 0x00, 0x48, 0x83, 0xC4, 0x10, 0xC3}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("adjsp pair:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Unbalanced at the RET: the toolchain diagnoses, so must we.
|
||||||
|
_, _, err = Assemble(firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·unbalanced(SB), NOSPLIT, $0-0
|
||||||
|
ADJSP $16
|
||||||
|
RET
|
||||||
|
`))
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
|
||||||
|
t.Errorf("unbalanced ADJSP: err = %v, want unbalanced PUSH/POP", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The check runs per RET: a closed pair before the first RET does not
|
||||||
|
// excuse an open adjustment before the second.
|
||||||
|
_, _, err = Assemble(firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·tworet(SB), NOSPLIT, $0-0
|
||||||
|
ADJSP $8
|
||||||
|
ADJSP $-8
|
||||||
|
RET
|
||||||
|
mid:
|
||||||
|
ADJSP $8
|
||||||
|
RET
|
||||||
|
`))
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
|
||||||
|
t.Errorf("second RET with open ADJSP: err = %v, want unbalanced PUSH/POP", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A framed function: the assembler's own prologue and epilogue
|
||||||
|
// contribute matching deltas, so the pair in the body still balances,
|
||||||
|
// and the bytes match go tool asm end to end.
|
||||||
|
fn = firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·framed(SB), $16-8
|
||||||
|
ADJSP $8
|
||||||
|
ADJSP $-8
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, err = Assemble(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Assemble framed: %v", err)
|
||||||
|
}
|
||||||
|
want = []byte{
|
||||||
|
0x55, 0x48, 0x89, 0xE5, 0x48, 0x83, 0xEC, 0x10, // prologue
|
||||||
|
0x48, 0x83, 0xEC, 0x08, // ADJSP $8
|
||||||
|
0x48, 0x83, 0xC4, 0x08, // ADJSP $-8
|
||||||
|
0x48, 0x83, 0xC4, 0x10, 0x5D, // epilogue
|
||||||
|
0xC3,
|
||||||
|
}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAssembleRegRange pins the bracketed register range at the statement
|
||||||
|
// level: exactly four consecutive same-width vector registers assemble, the
|
||||||
|
// toolchain's rejected shapes all report an error.
|
||||||
|
func TestAssembleRegRange(t *testing.T) {
|
||||||
|
asm := func(t *testing.T, op string) ([]byte, error) {
|
||||||
|
t.Helper()
|
||||||
|
f, errs := parser.Parse("f_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tV4FMADDPS 17(SP), "+op+", K2, Z0\n\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse %s: %v", op, errs)
|
||||||
|
}
|
||||||
|
code, _, err := Assemble(f.Decls[0].(*ast.Text))
|
||||||
|
return code, err
|
||||||
|
}
|
||||||
|
for _, op := range []string{"[Z0-Z3]", "[Z4-Z7]", "[Z28-Z31]"} {
|
||||||
|
if _, err := asm(t, op); err != nil {
|
||||||
|
t.Errorf("%s: %v", op, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, op := range []string{"[Z0-Z4]", "[Z0-Z2]", "[Z0-Z0]", "[Z4-Z0]", "[Z1-Z0]", "[AX-Z3]", "[Z0-AX]"} {
|
||||||
|
if _, err := asm(t, op); err == nil {
|
||||||
|
t.Errorf("%s: assembled, want an error", op)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+114
-17
@@ -42,8 +42,13 @@ const (
|
|||||||
sttSection = 3
|
sttSection = 3
|
||||||
stInfoShift = 4
|
stInfoShift = 4
|
||||||
|
|
||||||
rX8664PC32 = 2
|
rX8664PC32 = 2
|
||||||
rX8664TPOFF32 = 20
|
// R_X86_64_32 (debug/elf): the absolute 32-bit address of a symbol, the
|
||||||
|
// R_ADDR shape a 4-byte DATA field carries.
|
||||||
|
rX8664Abs32 = 10
|
||||||
|
// R_X86_64_TPOFF32 (debug/elf): the local-exec TLS offset the stack
|
||||||
|
// guard loads from FS. 20 is R_X86_64_TLSLD, a different relocation.
|
||||||
|
rX8664TPOFF32 = 23
|
||||||
)
|
)
|
||||||
|
|
||||||
// elfSym is one symbol-table entry in construction.
|
// elfSym is one symbol-table entry in construction.
|
||||||
@@ -156,6 +161,50 @@ func (img *Image) ELFObject() ([]byte, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||||
|
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||||
|
// DATA line's width at the field's data-section offset, S + A with no
|
||||||
|
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||||
|
// cannot hold an address, so they are refused rather than truncated.
|
||||||
|
var dataRelas []elfRela
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
for _, r := range d.Relocs {
|
||||||
|
idx, ok := symIdx[r.Name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||||
|
}
|
||||||
|
var typ uint32
|
||||||
|
switch r.Siz {
|
||||||
|
case 8:
|
||||||
|
typ = rX8664Abs64
|
||||||
|
case 4:
|
||||||
|
typ = rX8664Abs32
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||||
|
}
|
||||||
|
dataRelas = append(dataRelas, elfRela{
|
||||||
|
off: uint64(d.Offset + r.Off),
|
||||||
|
sym: idx,
|
||||||
|
typ: typ,
|
||||||
|
addend: r.Addend,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Section presence: .rela.text only when there are code relocations,
|
||||||
|
// .rela.data only when a DATA line holds a symbol value.
|
||||||
|
hasRela := len(relas) > 0
|
||||||
|
hasDataRela := len(dataRelas) > 0
|
||||||
|
nSections := 6 // NULL, .text, .data, .symtab, .strtab, .shstrtab
|
||||||
|
if hasRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
secSymtab, secStrtab := 3, 4
|
||||||
|
secShstr := nSections - 1
|
||||||
|
|
||||||
// Serialise the string tables.
|
// Serialise the string tables.
|
||||||
stNames := newElfStrtab()
|
stNames := newElfStrtab()
|
||||||
for _, s := range syms {
|
for _, s := range syms {
|
||||||
@@ -165,19 +214,13 @@ func (img *Image) ELFObject() ([]byte, error) {
|
|||||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
stSections.add(".rela.data")
|
||||||
|
}
|
||||||
for _, n := range dwarfSectionNames {
|
for _, n := range dwarfSectionNames {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Section presence: .rela.text only when there are relocations.
|
|
||||||
hasRela := len(relas) > 0
|
|
||||||
nSections := 6 // NULL, .text, .data, .symtab, .strtab, .shstrtab
|
|
||||||
if hasRela {
|
|
||||||
nSections = 7
|
|
||||||
}
|
|
||||||
secSymtab, secStrtab := 3, 4
|
|
||||||
secShstr := nSections - 1
|
|
||||||
|
|
||||||
// Lay the file out: header, section data, section headers.
|
// Lay the file out: header, section data, section headers.
|
||||||
var out []byte
|
var out []byte
|
||||||
out = append(out, make([]byte, 64)...) // ELF header, filled last
|
out = append(out, make([]byte, 64)...) // ELF header, filled last
|
||||||
@@ -212,14 +255,25 @@ func (img *Image) ELFObject() ([]byte, error) {
|
|||||||
strtabOff := len(out)
|
strtabOff := len(out)
|
||||||
out = append(out, stNames.bytes()...)
|
out = append(out, stNames.bytes()...)
|
||||||
|
|
||||||
var relaOff int
|
var relaOff, relaDataOff int
|
||||||
if hasRela {
|
if hasRela {
|
||||||
align(8)
|
align(8)
|
||||||
relaOff = len(out)
|
relaOff = len(out)
|
||||||
for _, r := range relas {
|
for _, r := range relas {
|
||||||
var b [24]byte
|
var b [24]byte
|
||||||
le.PutUint64(b[0:], r.off)
|
le.PutUint64(b[0:], r.off)
|
||||||
le.PutUint64(b[8:], uint64(r.sym)<<32|rX8664PC32)
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||||
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
align(8)
|
||||||
|
relaDataOff = len(out)
|
||||||
|
for _, r := range dataRelas {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint64(b[0:], r.off)
|
||||||
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||||
le.PutUint64(b[16:], uint64(r.addend))
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
out = append(out, b[:]...)
|
out = append(out, b[:]...)
|
||||||
}
|
}
|
||||||
@@ -228,15 +282,32 @@ func (img *Image) ELFObject() ([]byte, error) {
|
|||||||
shstrOff := len(out)
|
shstrOff := len(out)
|
||||||
out = append(out, stSections.bytes()...)
|
out = append(out, stSections.bytes()...)
|
||||||
|
|
||||||
// DWARF debug sections (no relocations, the linker resolves DWARF fixups).
|
// DWARF debug sections; the address placeholders they leave are carried
|
||||||
|
// as .rela.debug_info/.rela.debug_line entries the system linker applies.
|
||||||
dwAlign := func(n int) {
|
dwAlign := func(n int) {
|
||||||
for len(out)%n != 0 {
|
for len(out)%n != 0 {
|
||||||
out = append(out, 0)
|
out = append(out, 0)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
|
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiAMD64)
|
||||||
|
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
nSections += 4 // .debug_abbrev, .debug_info, .debug_line, .debug_line_str
|
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
|
||||||
|
// .debug_line_str and .debug_frame (the CIE is unconditional, so
|
||||||
|
// the frame section is always present), plus the relocation
|
||||||
|
// sections below when they carry entries.
|
||||||
|
dwarfStart = nSections
|
||||||
|
nSections += 5
|
||||||
|
appendDWARFRelas(&out, dw, rX8664Abs64, dwAlign)
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
align(8)
|
align(8)
|
||||||
@@ -265,16 +336,42 @@ func (img *Image) ELFObject() ([]byte, error) {
|
|||||||
if hasRela {
|
if hasRela {
|
||||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||||
|
}
|
||||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
|
|
||||||
// DWARF section headers.
|
// DWARF section headers; their indices follow the write order.
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
|
// secIdx is a running section index: each putSh below emits the
|
||||||
|
// next header, and the sh_info of a .rela section names the index
|
||||||
|
// of the section it relocates.
|
||||||
|
secIdx := dwarfStart
|
||||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||||
|
secInfoIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||||
|
secLineIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
if dw.frameSize > 0 {
|
if dw.frameSize > 0 {
|
||||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||||
|
secFrameIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+161
-74
@@ -12,43 +12,74 @@ import (
|
|||||||
// self-contained sections because the system linker only performs fixup
|
// self-contained sections because the system linker only performs fixup
|
||||||
// relocations, not assembly.
|
// relocations, not assembly.
|
||||||
|
|
||||||
|
// DWARF5 attribute, form and line-table constants (the values the
|
||||||
|
// toolchain uses, cmd/internal/dwarf/dwarf_defs.go; the DIE streams below
|
||||||
|
// are written against these forms).
|
||||||
|
const (
|
||||||
|
dwAtName = 0x03 // DW_AT_name
|
||||||
|
dwAtStmtList = 0x10 // DW_AT_stmt_list
|
||||||
|
dwAtLowPC = 0x11 // DW_AT_low_pc
|
||||||
|
dwAtHighPC = 0x12 // DW_AT_high_pc
|
||||||
|
dwAtDeclFile = 0x3a // DW_AT_decl_file
|
||||||
|
dwAtDeclLine = 0x3b // DW_AT_decl_line
|
||||||
|
dwAtExternal = 0x3f // DW_AT_external
|
||||||
|
dwAtFrameBase = 0x40 // DW_AT_frame_base
|
||||||
|
dwTagSubprog = 0x2e // DW_TAG_subprogram
|
||||||
|
dwTagCompUnit = 0x11 // DW_TAG_compile_unit
|
||||||
|
dwFormAddr = 0x01 // DW_FORM_addr
|
||||||
|
dwFormData8 = 0x07 // DW_FORM_data8
|
||||||
|
dwFormString = 0x08 // DW_FORM_string
|
||||||
|
dwFormData1 = 0x0b // DW_FORM_data1
|
||||||
|
dwFormUdata = 0x0f // DW_FORM_udata
|
||||||
|
dwFormSecOff = 0x17 // DW_FORM_sec_offset
|
||||||
|
dwFormExprloc = 0x18 // DW_FORM_exprloc
|
||||||
|
dwFormLineStrp = 0x1f // DW_FORM_line_strp
|
||||||
|
dwLnctPath = 0x01 // DW_LNCT_path
|
||||||
|
dwLnctDirIndex = 0x02 // DW_LNCT_directory_index
|
||||||
|
)
|
||||||
|
|
||||||
// dwarfAbbrevTable returns the .debug_abbrev content: a single compilation
|
// dwarfAbbrevTable returns the .debug_abbrev content: a single compilation
|
||||||
// unit with DW_TAG_compile_unit and DW_TAG_subprogram entries.
|
// unit with DW_TAG_compile_unit and DW_TAG_subprogram entries. The
|
||||||
|
// attribute/form pairs must match the DIE streams dwarfBuildInfoSection
|
||||||
|
// writes byte for byte, in the same order, or every consumer's parse of
|
||||||
|
// .debug_info desynchronises.
|
||||||
func dwarfAbbrevTable() []byte {
|
func dwarfAbbrevTable() []byte {
|
||||||
var b []byte
|
var b []byte
|
||||||
// Abbrev 1: DW_TAG_compile_unit
|
// Abbrev 1: DW_TAG_compile_unit.
|
||||||
b = append(b, 1) // abbreviation code
|
b = append(b, 1) // abbreviation code
|
||||||
b = append(b, 0x11) // DW_TAG_compile_unit
|
b = appendUleb(b, dwTagCompUnit) // DW_TAG_compile_unit
|
||||||
b = append(b, 1) // DW_CHILDREN_yes
|
b = append(b, 1) // DW_CHILDREN_yes
|
||||||
b = appendUleb(b, 0x1b) // DW_AT_low_pc
|
b = appendUleb(b, dwAtLowPC) // DW_AT_low_pc
|
||||||
b = appendUleb(b, 0x01) // DW_FORM_addr
|
b = appendUleb(b, dwFormAddr) // DW_FORM_addr
|
||||||
b = appendUleb(b, 0x29) // DW_AT_high_pc
|
b = appendUleb(b, dwAtHighPC) // DW_AT_high_pc
|
||||||
b = appendUleb(b, 0x07) // DW_FORM_data8
|
b = appendUleb(b, dwFormData8) // DW_FORM_data8
|
||||||
b = appendUleb(b, 0x10) // DW_AT_stmt_list
|
b = appendUleb(b, dwAtStmtList) // DW_AT_stmt_list
|
||||||
b = appendUleb(b, 0x25) // DW_FORM_sec_offset
|
b = appendUleb(b, dwFormSecOff) // DW_FORM_sec_offset (4 bytes here)
|
||||||
b = appendUleb(b, 0x01) // DW_AT_name
|
b = appendUleb(b, dwAtName) // DW_AT_name
|
||||||
b = appendUleb(b, 0x08) // DW_FORM_string
|
b = appendUleb(b, dwFormString) // DW_FORM_string
|
||||||
b = appendUleb(b, 0) // end of attributes
|
b = appendUleb(b, 0) // end of attributes: attr 0
|
||||||
|
b = appendUleb(b, 0) // ... paired with form 0
|
||||||
|
|
||||||
// Abbrev 2: DW_TAG_subprogram
|
// Abbrev 2: DW_TAG_subprogram.
|
||||||
b = append(b, 2) // abbreviation code
|
b = append(b, 2) // abbreviation code
|
||||||
b = append(b, 0x2e) // DW_TAG_subprogram
|
b = appendUleb(b, dwTagSubprog) // DW_TAG_subprogram
|
||||||
b = append(b, 0) // DW_CHILDREN_no
|
b = append(b, 0) // DW_CHILDREN_no
|
||||||
b = appendUleb(b, 0x03) // DW_AT_name
|
b = appendUleb(b, dwAtName) // DW_AT_name
|
||||||
b = appendUleb(b, 0x08) // DW_FORM_string
|
b = appendUleb(b, dwFormString) // DW_FORM_string
|
||||||
b = appendUleb(b, 0x11) // DW_AT_low_pc
|
b = appendUleb(b, dwAtLowPC) // DW_AT_low_pc
|
||||||
b = appendUleb(b, 0x01) // DW_FORM_addr
|
b = appendUleb(b, dwFormAddr) // DW_FORM_addr
|
||||||
b = appendUleb(b, 0x29) // DW_AT_high_pc
|
b = appendUleb(b, dwAtHighPC) // DW_AT_high_pc
|
||||||
b = appendUleb(b, 0x07) // DW_FORM_data8
|
b = appendUleb(b, dwFormData8) // DW_FORM_data8
|
||||||
b = appendUleb(b, 0x3f) // DW_AT_frame_base
|
b = appendUleb(b, dwAtFrameBase) // DW_AT_frame_base
|
||||||
b = appendUleb(b, 0x18) // DW_FORM_exprloc
|
b = appendUleb(b, dwFormExprloc) // DW_FORM_exprloc
|
||||||
b = appendUleb(b, 0x3b) // DW_AT_decl_file
|
b = appendUleb(b, dwAtDeclFile) // DW_AT_decl_file
|
||||||
b = appendUleb(b, 0x0b) // DW_FORM_data1
|
b = appendUleb(b, dwFormData1) // DW_FORM_data1
|
||||||
b = appendUleb(b, 0x37) // DW_AT_decl_line
|
b = appendUleb(b, dwAtDeclLine) // DW_AT_decl_line
|
||||||
b = appendUleb(b, 0x0b) // DW_FORM_data1
|
b = appendUleb(b, dwFormData1) // DW_FORM_data1
|
||||||
b = appendUleb(b, 0x63) // DW_AT_external
|
b = appendUleb(b, dwAtExternal) // DW_AT_external
|
||||||
b = appendUleb(b, 0x0b) // DW_FORM_flag
|
b = appendUleb(b, 0x0c) // DW_FORM_flag (one byte, 0 or 1)
|
||||||
b = appendUleb(b, 0) // end of attributes
|
b = appendUleb(b, 0) // end of attributes: attr 0
|
||||||
|
b = appendUleb(b, 0) // ... paired with form 0
|
||||||
|
|
||||||
// End of table.
|
// End of table.
|
||||||
b = append(b, 0)
|
b = append(b, 0)
|
||||||
@@ -68,6 +99,9 @@ type dwarfSections struct {
|
|||||||
infoRelocs []dwarfReloc
|
infoRelocs []dwarfReloc
|
||||||
// Relocations for .debug_line: (offset, symbol name, addend).
|
// Relocations for .debug_line: (offset, symbol name, addend).
|
||||||
lineRelocs []dwarfReloc
|
lineRelocs []dwarfReloc
|
||||||
|
// Relocations for .debug_frame: (offset, symbol name, addend), one per
|
||||||
|
// FDE initial_location.
|
||||||
|
frameRelocs []dwarfReloc
|
||||||
}
|
}
|
||||||
|
|
||||||
type dwarfReloc struct {
|
type dwarfReloc struct {
|
||||||
@@ -76,8 +110,9 @@ type dwarfReloc struct {
|
|||||||
addend int64
|
addend int64
|
||||||
}
|
}
|
||||||
|
|
||||||
// emitDWARF generates complete DWARF5 sections for the image.
|
// emitDWARF generates complete DWARF5 sections for the image. cfi carries
|
||||||
func emitDWARF(img *Image, srcFile string) *dwarfSections {
|
// the architecture's .debug_frame register conventions.
|
||||||
|
func emitDWARF(img *Image, srcFile string, cfi cfiArch) *dwarfSections {
|
||||||
ds := &dwarfSections{}
|
ds := &dwarfSections{}
|
||||||
ds.debugAbbrev = dwarfAbbrevTable()
|
ds.debugAbbrev = dwarfAbbrevTable()
|
||||||
|
|
||||||
@@ -86,19 +121,21 @@ func emitDWARF(img *Image, srcFile string) *dwarfSections {
|
|||||||
lineStr.add(srcFile)
|
lineStr.add(srcFile)
|
||||||
ds.debugLineStr = lineStr.bytes()
|
ds.debugLineStr = lineStr.bytes()
|
||||||
|
|
||||||
// Build .debug_line.
|
// Build .debug_line; the file table references the source name through
|
||||||
ds.debugLine = dwarfBuildLineSection(img, ds)
|
// its offset in .debug_line_str.
|
||||||
|
ds.debugLine = dwarfBuildLineSection(img, uint32(lineStr.at(srcFile)), ds)
|
||||||
|
|
||||||
// Build .debug_info.
|
// Build .debug_info.
|
||||||
ds.debugInfo = dwarfBuildInfoSection(img, srcFile, ds)
|
ds.debugInfo = dwarfBuildInfoSection(img, srcFile, ds)
|
||||||
|
|
||||||
// Build .debug_frame.
|
// Build .debug_frame.
|
||||||
ds.debugFrame = dwarfBuildFrameSection(img)
|
ds.debugFrame = dwarfBuildFrameSection(img, cfi, ds)
|
||||||
return ds
|
return ds
|
||||||
}
|
}
|
||||||
|
|
||||||
// dwarfBuildLineSection builds a complete .debug_line section.
|
// dwarfBuildLineSection builds a complete .debug_line section. srcStrOff is
|
||||||
func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
|
// the source file name's offset in .debug_line_str.
|
||||||
|
func dwarfBuildLineSection(img *Image, srcStrOff uint32, ds *dwarfSections) []byte {
|
||||||
var b []byte
|
var b []byte
|
||||||
le := binary.LittleEndian
|
le := binary.LittleEndian
|
||||||
|
|
||||||
@@ -120,15 +157,25 @@ func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
|
|||||||
// Standard opcode lengths (opcode 1..opcode_base-1).
|
// Standard opcode lengths (opcode 1..opcode_base-1).
|
||||||
b = append(b, 0, 1, 1, 1, 1, 0, 0, 0, 1, 0)
|
b = append(b, 0, 1, 1, 1, 1, 0, 0, 0, 1, 0)
|
||||||
|
|
||||||
// Directory table (DWARF5 format).
|
// Directory table (DWARF5 §6.2.4): entry format descriptors followed by
|
||||||
b = append(b, 0) // one directory entry (index 0 = empty)
|
// the entries. One directory, the compilation directory, whose path is
|
||||||
// File table.
|
// the empty string at .debug_line_str offset 0.
|
||||||
b = appendUleb(b, 1) // file count
|
b = append(b, 1) // directory_entry_format_count
|
||||||
// File 1: name index into .debug_line_str, dir index, time, size.
|
b = appendUleb(b, dwLnctPath) // DW_LNCT_path
|
||||||
b = appendUleb(b, 0) // name (index 0 in line_str)
|
b = appendUleb(b, dwFormLineStrp) // DW_FORM_line_strp
|
||||||
b = appendUleb(b, 0) // directory index
|
b = appendUleb(b, 1) // directories_count
|
||||||
b = appendUleb(b, 0) // last modification time
|
b = le.AppendUint32(b, 0) // .debug_line_str offset of ""
|
||||||
b = appendUleb(b, 0) // file size
|
|
||||||
|
// File table (DWARF5 §6.2.5). v5 indexes files from 0, so the source
|
||||||
|
// file is entry 0, matching the DW_AT_decl_file value 0 the DIEs carry.
|
||||||
|
b = append(b, 2) // file_name_entry_format_count
|
||||||
|
b = appendUleb(b, dwLnctPath) // DW_LNCT_path
|
||||||
|
b = appendUleb(b, dwFormLineStrp) // DW_FORM_line_strp
|
||||||
|
b = appendUleb(b, dwLnctDirIndex) // DW_LNCT_directory_index
|
||||||
|
b = appendUleb(b, dwFormUdata) // DW_FORM_udata
|
||||||
|
b = appendUleb(b, 1) // file_names_count
|
||||||
|
b = le.AppendUint32(b, srcStrOff) // .debug_line_str offset of the source name
|
||||||
|
b = appendUleb(b, 0) // directory index 0 (the compilation directory)
|
||||||
|
|
||||||
headerEnd := len(b)
|
headerEnd := len(b)
|
||||||
|
|
||||||
@@ -176,8 +223,11 @@ func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
|
|||||||
|
|
||||||
// Patch unit_length.
|
// Patch unit_length.
|
||||||
le.PutUint32(b[headerStart:], uint32(len(b)-headerStart-4))
|
le.PutUint32(b[headerStart:], uint32(len(b)-headerStart-4))
|
||||||
// Patch header_length.
|
// Patch header_length. In the v5 header it follows the one-byte
|
||||||
le.PutUint32(b[headerStart+6:], uint32(headerEnd-headerStart-10))
|
// address_size and segment_selector_size (offset 8, not the DWARF2-4
|
||||||
|
// offset 6), and counts from just past itself to the first program
|
||||||
|
// byte.
|
||||||
|
le.PutUint32(b[headerStart+8:], uint32(headerEnd-headerStart-12))
|
||||||
return b
|
return b
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -195,14 +245,16 @@ func dwarfBuildInfoSection(img *Image, srcFile string, ds *dwarfSections) []byte
|
|||||||
|
|
||||||
// DW_TAG_compile_unit (abbrev 1).
|
// DW_TAG_compile_unit (abbrev 1).
|
||||||
b = append(b, 1) // abbreviation code
|
b = append(b, 1) // abbreviation code
|
||||||
// DW_AT_low_pc: address of .text start.
|
// DW_AT_low_pc: address of .text start. A data-only image has no
|
||||||
infoRelocBase := len(b)
|
// functions to relocate against; its CU covers no code, so the base
|
||||||
|
// stays zero (the DWARF "no base address" value) with no relocation.
|
||||||
b = le.AppendUint64(b, 0) // placeholder
|
b = le.AppendUint64(b, 0) // placeholder
|
||||||
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
|
if len(img.Funcs) > 0 {
|
||||||
off: uint64(infoRelocBase),
|
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
|
||||||
name: img.Funcs[0].Name,
|
off: uint64(len(b) - 8),
|
||||||
addend: 0,
|
name: img.Funcs[0].Name,
|
||||||
})
|
})
|
||||||
|
}
|
||||||
// DW_AT_high_pc: size of .text.
|
// DW_AT_high_pc: size of .text.
|
||||||
b = le.AppendUint64(b, uint64(len(img.Code)))
|
b = le.AppendUint64(b, uint64(len(img.Code)))
|
||||||
// DW_AT_stmt_list: offset into .debug_line (0).
|
// DW_AT_stmt_list: offset into .debug_line (0).
|
||||||
@@ -229,8 +281,9 @@ func dwarfBuildInfoSection(img *Image, srcFile string, ds *dwarfSections) []byte
|
|||||||
b = le.AppendUint64(b, uint64(fn.Size))
|
b = le.AppendUint64(b, uint64(fn.Size))
|
||||||
// DW_AT_frame_base: DW_OP_call_frame_cfa.
|
// DW_AT_frame_base: DW_OP_call_frame_cfa.
|
||||||
b = append(b, 1, 0x9c)
|
b = append(b, 1, 0x9c)
|
||||||
// DW_AT_decl_file: file index 1.
|
// DW_AT_decl_file: the single file-table entry, index 0 (v5 indexes
|
||||||
b = append(b, 1)
|
// files from 0).
|
||||||
|
b = append(b, 0)
|
||||||
// DW_AT_decl_line.
|
// DW_AT_decl_line.
|
||||||
b = append(b, uint8(fn.Line))
|
b = append(b, uint8(fn.Line))
|
||||||
// DW_AT_external.
|
// DW_AT_external.
|
||||||
@@ -253,32 +306,61 @@ func appendUleb(b []byte, v uint64) []byte {
|
|||||||
return binary.AppendUvarint(b, v)
|
return binary.AppendUvarint(b, v)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// appendSleb appends v in signed LEB128, the encoding DWARF specifies:
|
||||||
|
// two's-complement sign extension, which is NOT Go's zigzag varint
|
||||||
|
// (binary.AppendVarint(-8) encodes 15, where DWARF wants 0x78).
|
||||||
func appendSleb(b []byte, v int64) []byte {
|
func appendSleb(b []byte, v int64) []byte {
|
||||||
return binary.AppendVarint(b, v)
|
for {
|
||||||
|
c := byte(v & 0x7f)
|
||||||
|
v >>= 7
|
||||||
|
if (v == 0 && c&0x40 == 0) || (v == -1 && c&0x40 != 0) {
|
||||||
|
return append(b, c)
|
||||||
|
}
|
||||||
|
b = append(b, c|0x80)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// cfiArch carries the .debug_frame CIE parameters that differ per
|
||||||
|
// architecture: the DWARF register numbers of the stack pointer the initial
|
||||||
|
// CFA rule names and of the return address. The values are the ones the Go
|
||||||
|
// linker writes into its own CIE (cmd/link/internal/ld/dwarf.go uses
|
||||||
|
// Dwarfregsp and Dwarfreglr; the per-architecture constants live in
|
||||||
|
// cmd/link/internal/<arch>/l.go).
|
||||||
|
type cfiArch struct {
|
||||||
|
name string
|
||||||
|
cfaReg byte // the stack-pointer register the initial CFA rule names
|
||||||
|
raReg byte // the return-address register
|
||||||
|
}
|
||||||
|
|
||||||
|
var (
|
||||||
|
cfiAMD64 = cfiArch{"amd64", 7, 16} // RSP, RIP
|
||||||
|
cfiARM64 = cfiArch{"arm64", 31, 30} // SP (X31), LR (X30)
|
||||||
|
cfiRISCV64 = cfiArch{"riscv64", 2, 1} // X2 (sp), X1 (ra)
|
||||||
|
cfiLOONG64 = cfiArch{"loong64", 3, 1} // $r3 (sp), $r1 (ra)
|
||||||
|
)
|
||||||
|
|
||||||
// dwarfBuildFrameSection builds a .debug_frame section with CFI for stack
|
// dwarfBuildFrameSection builds a .debug_frame section with CFI for stack
|
||||||
// unwinding. It emits one CIE and one FDE per function, encoding the
|
// unwinding. It emits one CIE and one FDE per function, encoding the
|
||||||
// CFA (Canonical Frame Address) rule changes at each stack-adjustment
|
// CFA (Canonical Frame Address) rule changes at each stack-adjustment
|
||||||
// boundary recorded in FuncLayout.Spadj.
|
// boundary recorded in FuncLayout.Spadj.
|
||||||
func dwarfBuildFrameSection(img *Image) []byte {
|
func dwarfBuildFrameSection(img *Image, cfi cfiArch, ds *dwarfSections) []byte {
|
||||||
var b []byte
|
var b []byte
|
||||||
le := binary.LittleEndian
|
le := binary.LittleEndian
|
||||||
|
|
||||||
// CIE (Common Information Entry).
|
// CIE (Common Information Entry).
|
||||||
cieStart := len(b)
|
cieStart := len(b)
|
||||||
b = append(b, 0, 0, 0, 0) // length (placeholder)
|
b = append(b, 0, 0, 0, 0) // length (placeholder)
|
||||||
b = le.AppendUint32(b, 0xFFFFFFFF) // CIE marker
|
b = le.AppendUint32(b, 0xFFFFFFFF) // CIE marker
|
||||||
b = append(b, 3) // version (DWARF3, widely supported)
|
b = append(b, 3) // version (DWARF3, widely supported)
|
||||||
b = append(b, 0) // augmentation (empty)
|
b = append(b, 0) // augmentation (empty)
|
||||||
b = appendUleb(b, 1) // code alignment
|
b = appendUleb(b, 1) // code alignment
|
||||||
b = appendSleb(b, -8) // data alignment (-8 for 64-bit)
|
b = appendSleb(b, -8) // data alignment (-8 for 64-bit)
|
||||||
b = appendUleb(b, 16) // return address register (LR on arm64, RIP on amd64)
|
b = appendUleb(b, uint64(cfi.raReg)) // return address register
|
||||||
// Initial CFA rule: DW_CFA_def_cfa (SP, 0)
|
// Initial CFA rule: DW_CFA_def_cfa (SP, 0)
|
||||||
b = append(b, 0x0c) // DW_CFA_def_cfa
|
b = append(b, 0x0c) // DW_CFA_def_cfa
|
||||||
b = appendUleb(b, 31) // register: SP (RSP=7 on amd64, SP=31 on arm64)
|
b = appendUleb(b, uint64(cfi.cfaReg)) // the architecture's stack pointer
|
||||||
b = appendUleb(b, 0) // offset: 0
|
b = appendUleb(b, 0) // offset: 0
|
||||||
b = append(b, 0) // DW_CFA_nop (padding)
|
b = append(b, 0) // DW_CFA_nop (padding)
|
||||||
// Patch CIE length.
|
// Patch CIE length.
|
||||||
le.PutUint32(b[cieStart:], uint32(len(b)-cieStart-4))
|
le.PutUint32(b[cieStart:], uint32(len(b)-cieStart-4))
|
||||||
|
|
||||||
@@ -287,7 +369,12 @@ func dwarfBuildFrameSection(img *Image) []byte {
|
|||||||
fdeStart := len(b)
|
fdeStart := len(b)
|
||||||
b = append(b, 0, 0, 0, 0) // length (placeholder)
|
b = append(b, 0, 0, 0, 0) // length (placeholder)
|
||||||
b = le.AppendUint32(b, uint32(cieStart)) // CIE pointer (offset from start)
|
b = le.AppendUint32(b, uint32(cieStart)) // CIE pointer (offset from start)
|
||||||
// Initial location: function offset in .text (relocated by linker).
|
// Initial location: function offset in .text, referenced through
|
||||||
|
// the function's symbol so the linker relocates it.
|
||||||
|
ds.frameRelocs = append(ds.frameRelocs, dwarfReloc{
|
||||||
|
off: uint64(fdeStart + 8),
|
||||||
|
name: fn.Name,
|
||||||
|
})
|
||||||
b = le.AppendUint64(b, uint64(fn.Offset))
|
b = le.AppendUint64(b, uint64(fn.Offset))
|
||||||
// Address range: function size.
|
// Address range: function size.
|
||||||
b = le.AppendUint64(b, uint64(fn.Size))
|
b = le.AppendUint64(b, uint64(fn.Size))
|
||||||
|
|||||||
+84
-24
@@ -3,6 +3,17 @@
|
|||||||
|
|
||||||
package asm
|
package asm
|
||||||
|
|
||||||
|
import "encoding/binary"
|
||||||
|
|
||||||
|
// Absolute 64-bit relocation types for the DWARF address fixups, one per
|
||||||
|
// supported architecture (the numbers debug/elf carries).
|
||||||
|
const (
|
||||||
|
rX8664Abs64 = 1 // R_X86_64_64
|
||||||
|
rAARCH64Abs64 = 257 // R_AARCH64_ABS64
|
||||||
|
rRISCVAbs64 = 2 // R_RISCV_64
|
||||||
|
rLarchAbs64 = 2 // R_LARCH_64
|
||||||
|
)
|
||||||
|
|
||||||
// dwarfELFSections holds the laid-out DWARF sections ready for inclusion
|
// dwarfELFSections holds the laid-out DWARF sections ready for inclusion
|
||||||
// in an ELF file.
|
// in an ELF file.
|
||||||
type dwarfELFSections struct {
|
type dwarfELFSections struct {
|
||||||
@@ -11,15 +22,23 @@ type dwarfELFSections struct {
|
|||||||
lineOff, lineSize int
|
lineOff, lineSize int
|
||||||
lineStrOff, lineStrSize int
|
lineStrOff, lineStrSize int
|
||||||
frameOff, frameSize int
|
frameOff, frameSize int
|
||||||
// Relocations for .debug_info address references.
|
// .rela.debug_info and .rela.debug_line contents: file offsets and
|
||||||
|
// entry counts (zero count: the section is absent).
|
||||||
|
infoRelaOff, infoRelaCount int
|
||||||
|
lineRelaOff, lineRelaCount int
|
||||||
|
frameRelaOff, frameRelaCount int
|
||||||
|
// Relocations for .debug_info address references, offsets relative to
|
||||||
|
// the section start (what an r_offset in .rela.debug_info means).
|
||||||
infoRelocs []elfDwarfReloc
|
infoRelocs []elfDwarfReloc
|
||||||
// Relocations for .debug_line address references.
|
// Relocations for .debug_line address references, section-relative.
|
||||||
lineRelocs []elfDwarfReloc
|
lineRelocs []elfDwarfReloc
|
||||||
|
// Relocations for .debug_frame FDE initial locations, section-relative.
|
||||||
|
frameRelocs []elfDwarfReloc
|
||||||
}
|
}
|
||||||
|
|
||||||
type elfDwarfReloc struct {
|
type elfDwarfReloc struct {
|
||||||
off uint64
|
off uint64 // offset within the target section
|
||||||
sym int // symbol index in .symtab
|
sym int // symbol index in .symtab
|
||||||
addend int64
|
addend int64
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -29,8 +48,9 @@ type elfDwarfReloc struct {
|
|||||||
//
|
//
|
||||||
// symIdx maps function names to their .symtab indices (needed for relocations
|
// symIdx maps function names to their .symtab indices (needed for relocations
|
||||||
// against .text symbols). The map uses objectName format (pkg.name); the
|
// against .text symbols). The map uses objectName format (pkg.name); the
|
||||||
// DWARF code uses bare function names, so we build a reverse lookup.
|
// DWARF code uses bare function names, so we build a reverse lookup. cfi
|
||||||
func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[string]int, align func(int)) *dwarfELFSections {
|
// carries the architecture's .debug_frame register conventions.
|
||||||
|
func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[string]int, align func(int), cfi cfiArch) *dwarfELFSections {
|
||||||
// Build a lookup from bare function name to symbol index.
|
// Build a lookup from bare function name to symbol index.
|
||||||
nameToIdx := make(map[string]int, len(symIdx))
|
nameToIdx := make(map[string]int, len(symIdx))
|
||||||
for name, idx := range symIdx {
|
for name, idx := range symIdx {
|
||||||
@@ -45,7 +65,7 @@ func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[str
|
|||||||
}
|
}
|
||||||
nameToIdx[name] = idx
|
nameToIdx[name] = idx
|
||||||
}
|
}
|
||||||
ds := emitDWARF(img, srcFile)
|
ds := emitDWARF(img, srcFile, cfi)
|
||||||
if ds == nil || len(ds.debugAbbrev) == 0 {
|
if ds == nil || len(ds.debugAbbrev) == 0 {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
@@ -68,15 +88,11 @@ func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[str
|
|||||||
align(1)
|
align(1)
|
||||||
result.lineOff = len(*out)
|
result.lineOff = len(*out)
|
||||||
result.lineSize = len(ds.debugLine)
|
result.lineSize = len(ds.debugLine)
|
||||||
lineBase := len(*out)
|
|
||||||
*out = append(*out, ds.debugLine...)
|
*out = append(*out, ds.debugLine...)
|
||||||
|
|
||||||
// Patch .debug_line relocations: replace placeholder addresses with
|
|
||||||
// actual .text offsets via symbol lookup.
|
|
||||||
for _, dr := range ds.lineRelocs {
|
for _, dr := range ds.lineRelocs {
|
||||||
if idx, ok := nameToIdx[dr.name]; ok {
|
if idx, ok := nameToIdx[dr.name]; ok {
|
||||||
result.lineRelocs = append(result.lineRelocs, elfDwarfReloc{
|
result.lineRelocs = append(result.lineRelocs, elfDwarfReloc{
|
||||||
off: uint64(lineBase) + dr.off,
|
off: dr.off,
|
||||||
sym: idx,
|
sym: idx,
|
||||||
addend: dr.addend,
|
addend: dr.addend,
|
||||||
})
|
})
|
||||||
@@ -87,33 +103,77 @@ func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[str
|
|||||||
align(1)
|
align(1)
|
||||||
result.infoOff = len(*out)
|
result.infoOff = len(*out)
|
||||||
result.infoSize = len(ds.debugInfo)
|
result.infoSize = len(ds.debugInfo)
|
||||||
infoBase := len(*out)
|
|
||||||
*out = append(*out, ds.debugInfo...)
|
*out = append(*out, ds.debugInfo...)
|
||||||
|
|
||||||
// .debug_frame
|
|
||||||
if len(ds.debugFrame) > 0 {
|
|
||||||
align(1)
|
|
||||||
result.frameOff = len(*out)
|
|
||||||
result.frameSize = len(ds.debugFrame)
|
|
||||||
*out = append(*out, ds.debugFrame...)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Patch .debug_info relocations.
|
|
||||||
for _, dr := range ds.infoRelocs {
|
for _, dr := range ds.infoRelocs {
|
||||||
if idx, ok := nameToIdx[dr.name]; ok {
|
if idx, ok := nameToIdx[dr.name]; ok {
|
||||||
result.infoRelocs = append(result.infoRelocs, elfDwarfReloc{
|
result.infoRelocs = append(result.infoRelocs, elfDwarfReloc{
|
||||||
off: uint64(infoBase) + dr.off,
|
off: dr.off,
|
||||||
sym: idx,
|
sym: idx,
|
||||||
addend: dr.addend,
|
addend: dr.addend,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// .debug_frame: the section header declares alignment 8, so the data is
|
||||||
|
// padded to 8, matching it.
|
||||||
|
if len(ds.debugFrame) > 0 {
|
||||||
|
align(8)
|
||||||
|
result.frameOff = len(*out)
|
||||||
|
result.frameSize = len(ds.debugFrame)
|
||||||
|
*out = append(*out, ds.debugFrame...)
|
||||||
|
for _, dr := range ds.frameRelocs {
|
||||||
|
if idx, ok := nameToIdx[dr.name]; ok {
|
||||||
|
result.frameRelocs = append(result.frameRelocs, elfDwarfReloc{
|
||||||
|
off: dr.off,
|
||||||
|
sym: idx,
|
||||||
|
addend: dr.addend,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
return result
|
return result
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// appendDWARFRelas writes the .rela.debug_info and .rela.debug_line section
|
||||||
|
// bodies from the relocations appendDWARFSections recorded, with the
|
||||||
|
// architecture's absolute 64-bit relocation type, and records their file
|
||||||
|
// offsets and entry counts on dw. Called after the DWARF sections
|
||||||
|
// themselves so the r_offsets (section-relative) need no adjustment.
|
||||||
|
func appendDWARFRelas(out *[]byte, dw *dwarfELFSections, abs64 uint32, align func(int)) {
|
||||||
|
le := binary.LittleEndian
|
||||||
|
write := func(relas []elfDwarfReloc) (off, count int) {
|
||||||
|
if len(relas) == 0 {
|
||||||
|
return 0, 0
|
||||||
|
}
|
||||||
|
align(8)
|
||||||
|
off = len(*out)
|
||||||
|
for _, r := range relas {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint64(b[0:], r.off)
|
||||||
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(abs64))
|
||||||
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
|
*out = append(*out, b[:]...)
|
||||||
|
}
|
||||||
|
return off, len(relas)
|
||||||
|
}
|
||||||
|
dw.infoRelaOff, dw.infoRelaCount = write(dw.infoRelocs)
|
||||||
|
dw.lineRelaOff, dw.lineRelaCount = write(dw.lineRelocs)
|
||||||
|
dw.frameRelaOff, dw.frameRelaCount = write(dw.frameRelocs)
|
||||||
|
}
|
||||||
|
|
||||||
|
// dwarfSourceName returns the source name the DWARF sections record: the
|
||||||
|
// image's source path when the assembler captured one, "gasm.s" otherwise.
|
||||||
|
func dwarfSourceName(img *Image) string {
|
||||||
|
if img.SourcePath != "" {
|
||||||
|
return img.SourcePath
|
||||||
|
}
|
||||||
|
return "gasm.s"
|
||||||
|
}
|
||||||
|
|
||||||
// dwarfSectionNames returns the DWARF section names for the string table.
|
// dwarfSectionNames returns the DWARF section names for the string table.
|
||||||
var dwarfSectionNames = []string{
|
var dwarfSectionNames = []string{
|
||||||
".debug_abbrev", ".debug_info", ".debug_line", ".debug_line_str",
|
".debug_abbrev", ".debug_info", ".debug_line", ".debug_line_str",
|
||||||
".debug_frame", ".rela.debug_info", ".rela.debug_line",
|
".debug_frame", ".rela.debug_info", ".rela.debug_line",
|
||||||
|
".rela.debug_frame",
|
||||||
}
|
}
|
||||||
|
|||||||
+304
-13
@@ -4,11 +4,313 @@
|
|||||||
package asm
|
package asm
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/binary"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// ulebIter reads ULEB128 values, the .debug_abbrev and line-header
|
||||||
|
// encoding.
|
||||||
|
type ulebIter struct {
|
||||||
|
b []byte
|
||||||
|
i int
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *ulebIter) uleb(t *testing.T) uint64 {
|
||||||
|
t.Helper()
|
||||||
|
v, n := binary.Uvarint(r.b[r.i:])
|
||||||
|
if n <= 0 {
|
||||||
|
t.Fatalf("bad ULEB at %d", r.i)
|
||||||
|
}
|
||||||
|
r.i += n
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *ulebIter) byteAt(t *testing.T) byte {
|
||||||
|
t.Helper()
|
||||||
|
if r.i >= len(r.b) {
|
||||||
|
t.Fatalf("read past end at %d", r.i)
|
||||||
|
}
|
||||||
|
c := r.b[r.i]
|
||||||
|
r.i++
|
||||||
|
return c
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *ulebIter) uint32At(t *testing.T) uint32 {
|
||||||
|
t.Helper()
|
||||||
|
v := binary.LittleEndian.Uint32(r.b[r.i:])
|
||||||
|
r.i += 4
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
|
||||||
|
// sleb reads a signed LEB128, the DWARF encoding (sign-extended two's
|
||||||
|
// complement, not Go's zigzag varint).
|
||||||
|
func (r *ulebIter) sleb(t *testing.T) int64 {
|
||||||
|
t.Helper()
|
||||||
|
var v int64
|
||||||
|
var shift uint
|
||||||
|
for {
|
||||||
|
c := r.byteAt(t)
|
||||||
|
v |= int64(c&0x7f) << shift
|
||||||
|
shift += 7
|
||||||
|
if c&0x80 == 0 {
|
||||||
|
if c&0x40 != 0 {
|
||||||
|
v |= -1 << shift
|
||||||
|
}
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// dwarfAttr is one attribute/form pair of an abbreviation.
|
||||||
|
type dwarfAttr struct{ attr, form uint64 }
|
||||||
|
|
||||||
|
// dwarfAbbrev is one parsed abbreviation declaration.
|
||||||
|
type dwarfAbbrev struct {
|
||||||
|
code uint64
|
||||||
|
tag uint64
|
||||||
|
children bool
|
||||||
|
attrs []dwarfAttr
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseAbbrevs walks a .debug_abbrev table: abbreviation code, tag,
|
||||||
|
// children flag, then attr/form ULEB pairs terminated by a double zero.
|
||||||
|
func parseAbbrevs(t *testing.T, b []byte) map[uint64]dwarfAbbrev {
|
||||||
|
t.Helper()
|
||||||
|
out := map[uint64]dwarfAbbrev{}
|
||||||
|
r := &ulebIter{b: b}
|
||||||
|
for {
|
||||||
|
code := r.uleb(t)
|
||||||
|
if code == 0 {
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
ab := dwarfAbbrev{code: code, tag: r.uleb(t)}
|
||||||
|
ab.children = r.byteAt(t) == 1
|
||||||
|
for {
|
||||||
|
attr := r.uleb(t)
|
||||||
|
form := r.uleb(t)
|
||||||
|
if attr == 0 && form == 0 {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
if attr == 0 || form == 0 {
|
||||||
|
t.Fatalf("abbrev %d: half-terminated attr/form pair (%d, %d)", code, attr, form)
|
||||||
|
}
|
||||||
|
ab.attrs = append(ab.attrs, dwarfAttr{attr, form})
|
||||||
|
}
|
||||||
|
out[code] = ab
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func eqAttrs(t *testing.T, ab dwarfAbbrev, want []dwarfAttr) {
|
||||||
|
t.Helper()
|
||||||
|
if len(ab.attrs) != len(want) {
|
||||||
|
t.Fatalf("abbrev %d attrs = %v, want %v", ab.code, ab.attrs, want)
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
if ab.attrs[i] != w {
|
||||||
|
t.Fatalf("abbrev %d attr %d = (%#x, %#x), want (%#x, %#x)", ab.code, i, ab.attrs[i].attr, ab.attrs[i].form, w.attr, w.form)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDwarfAbbrevTable walks the abbreviation table as a consumer does and
|
||||||
|
// checks the attribute/form sets against the constants the toolchain uses
|
||||||
|
// (cmd/internal/dwarf/dwarf_defs.go). A wrong constant here renames an
|
||||||
|
// attribute (0x1b is comp_dir, not low_pc; 0x29 and 0x37 are bounds and
|
||||||
|
// count) and a wrong form desynchronises the DIE parse: 0x25 is strx1, one
|
||||||
|
// byte, where the writer emits four for a section offset.
|
||||||
|
func TestDwarfAbbrevTable(t *testing.T) {
|
||||||
|
abbrev := dwarfAbbrevTable()
|
||||||
|
if len(abbrev) == 0 {
|
||||||
|
t.Fatal("empty abbrev table")
|
||||||
|
}
|
||||||
|
// Must end with a zero byte (end of table).
|
||||||
|
if abbrev[len(abbrev)-1] != 0 {
|
||||||
|
t.Fatalf("abbrev table last byte = %d, want 0", abbrev[len(abbrev)-1])
|
||||||
|
}
|
||||||
|
abs := parseAbbrevs(t, abbrev)
|
||||||
|
if len(abs) != 2 {
|
||||||
|
t.Fatalf("abbreviations = %d, want 2", len(abs))
|
||||||
|
}
|
||||||
|
cu, ok := abs[1]
|
||||||
|
if !ok {
|
||||||
|
t.Fatal("missing abbreviation 1 (compile unit)")
|
||||||
|
}
|
||||||
|
if cu.tag != dwTagCompUnit || !cu.children {
|
||||||
|
t.Errorf("abbrev 1: tag %#x children %v, want compile unit with children", cu.tag, cu.children)
|
||||||
|
}
|
||||||
|
eqAttrs(t, cu, []dwarfAttr{
|
||||||
|
{dwAtLowPC, dwFormAddr},
|
||||||
|
{dwAtHighPC, dwFormData8},
|
||||||
|
{dwAtStmtList, dwFormSecOff},
|
||||||
|
{dwAtName, dwFormString},
|
||||||
|
})
|
||||||
|
sp, ok := abs[2]
|
||||||
|
if !ok {
|
||||||
|
t.Fatal("missing abbreviation 2 (subprogram)")
|
||||||
|
}
|
||||||
|
if sp.tag != dwTagSubprog || sp.children {
|
||||||
|
t.Errorf("abbrev 2: tag %#x children %v, want subprogram without children", sp.tag, sp.children)
|
||||||
|
}
|
||||||
|
eqAttrs(t, sp, []dwarfAttr{
|
||||||
|
{dwAtName, dwFormString},
|
||||||
|
{dwAtLowPC, dwFormAddr},
|
||||||
|
{dwAtHighPC, dwFormData8},
|
||||||
|
{dwAtFrameBase, dwFormExprloc},
|
||||||
|
{dwAtDeclFile, dwFormData1},
|
||||||
|
{dwAtDeclLine, dwFormData1},
|
||||||
|
{dwAtExternal, 0x0c}, // DW_FORM_flag
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDwarfLineHeaderV5 parses the .debug_line header under DWARF5 rules:
|
||||||
|
// the directory and file tables are format-descriptor lists, not the
|
||||||
|
// DWARF2-4 shape of null-terminated strings, and the file entry references
|
||||||
|
// the source name through .debug_line_str.
|
||||||
|
func TestDwarfLineHeaderV5(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ a+0(FP), AX
|
||||||
|
MOVQ b+8(FP), BX
|
||||||
|
ADDQ BX, AX
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("test_amd64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
|
||||||
|
r := &ulebIter{b: ds.debugLine}
|
||||||
|
r.uint32At(t) // unit_length
|
||||||
|
if v := binary.LittleEndian.Uint16(ds.debugLine[4:]); v != 5 {
|
||||||
|
t.Fatalf("version = %d, want 5", v)
|
||||||
|
}
|
||||||
|
r.i = 6
|
||||||
|
r.byteAt(t) // address_size
|
||||||
|
r.byteAt(t) // segment_selector_size
|
||||||
|
r.uint32At(t) // header_length
|
||||||
|
r.byteAt(t) // minimum_instruction_length
|
||||||
|
r.byteAt(t) // maximum_ops_per_instruction
|
||||||
|
r.byteAt(t) // default_is_stmt
|
||||||
|
r.byteAt(t) // line_base
|
||||||
|
r.byteAt(t) // line_range
|
||||||
|
opcodeBase := r.byteAt(t)
|
||||||
|
for range int(opcodeBase) - 1 {
|
||||||
|
r.byteAt(t) // standard opcode lengths
|
||||||
|
}
|
||||||
|
|
||||||
|
// Directory table (DWARF5 §6.2.4).
|
||||||
|
if n := r.byteAt(t); n != 1 {
|
||||||
|
t.Fatalf("directory_entry_format_count = %d, want 1", n)
|
||||||
|
}
|
||||||
|
if lnct := r.uleb(t); lnct != dwLnctPath {
|
||||||
|
t.Errorf("directory content type = %#x, want DW_LNCT_path", lnct)
|
||||||
|
}
|
||||||
|
if form := r.uleb(t); form != dwFormLineStrp {
|
||||||
|
t.Errorf("directory form = %#x, want DW_FORM_line_strp", form)
|
||||||
|
}
|
||||||
|
if n := r.uleb(t); n != 1 {
|
||||||
|
t.Fatalf("directories_count = %d, want 1", n)
|
||||||
|
}
|
||||||
|
if off := r.uint32At(t); off != 0 {
|
||||||
|
t.Errorf("compilation directory line_strp = %d, want 0 (the empty string)", off)
|
||||||
|
}
|
||||||
|
|
||||||
|
// File table (DWARF5 §6.2.5).
|
||||||
|
if n := r.byteAt(t); n != 2 {
|
||||||
|
t.Fatalf("file_name_entry_format_count = %d, want 2", n)
|
||||||
|
}
|
||||||
|
if lnct := r.uleb(t); lnct != dwLnctPath {
|
||||||
|
t.Errorf("file content type = %#x, want DW_LNCT_path", lnct)
|
||||||
|
}
|
||||||
|
if form := r.uleb(t); form != dwFormLineStrp {
|
||||||
|
t.Errorf("file path form = %#x, want DW_FORM_line_strp", form)
|
||||||
|
}
|
||||||
|
if lnct := r.uleb(t); lnct != dwLnctDirIndex {
|
||||||
|
t.Errorf("file content type = %#x, want DW_LNCT_directory_index", lnct)
|
||||||
|
}
|
||||||
|
if form := r.uleb(t); form != dwFormUdata {
|
||||||
|
t.Errorf("file dir-index form = %#x, want DW_FORM_udata", form)
|
||||||
|
}
|
||||||
|
if n := r.uleb(t); n != 1 {
|
||||||
|
t.Fatalf("file_names_count = %d, want 1", n)
|
||||||
|
}
|
||||||
|
strOff := r.uint32At(t)
|
||||||
|
if dirIdx := r.uleb(t); dirIdx != 0 {
|
||||||
|
t.Errorf("file directory index = %d, want 0", dirIdx)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The file entry's line_strp must resolve to the source name.
|
||||||
|
end := int(strOff) + len("test_amd64.s")
|
||||||
|
if int(strOff) >= len(ds.debugLineStr) || !bytes.Equal(ds.debugLineStr[strOff:end], []byte("test_amd64.s")) {
|
||||||
|
t.Errorf("file entry line_strp %d does not name the source: %q", strOff, ds.debugLineStr)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The fixed header fields: address_size 8 and a header_length that
|
||||||
|
// points just past the file table (the patch site is offset 8 in the
|
||||||
|
// v5 header, and the field counts from its own end).
|
||||||
|
if ds.debugLine[6] != 8 || ds.debugLine[7] != 0 {
|
||||||
|
t.Errorf("address_size/segment_selector = %d/%d, want 8/0", ds.debugLine[6], ds.debugLine[7])
|
||||||
|
}
|
||||||
|
if hl := binary.LittleEndian.Uint32(ds.debugLine[8:]); hl != uint32(r.i-12) {
|
||||||
|
t.Errorf("header_length = %d, want %d (the byte after the file table is %d)", hl, r.i-12, r.i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDwarfFrameCIEArch checks the shared CIE carries each architecture's
|
||||||
|
// stack-pointer and return-address registers: the values the Go linker
|
||||||
|
// writes (cmd/link/internal/<arch>/l.go dwarfRegSP/dwarfRegLR).
|
||||||
|
func TestDwarfFrameCIEArch(t *testing.T) {
|
||||||
|
for _, tc := range []struct {
|
||||||
|
name string
|
||||||
|
cfi cfiArch
|
||||||
|
}{
|
||||||
|
{"amd64", cfiAMD64},
|
||||||
|
{"arm64", cfiARM64},
|
||||||
|
{"riscv64", cfiRISCV64},
|
||||||
|
{"loong64", cfiLOONG64},
|
||||||
|
} {
|
||||||
|
frame := dwarfBuildFrameSection(&Image{}, tc.cfi, &dwarfSections{})
|
||||||
|
r := &ulebIter{b: frame}
|
||||||
|
r.uint32At(t) // length
|
||||||
|
if cid := r.uint32At(t); cid != 0xFFFFFFFF {
|
||||||
|
t.Errorf("%s: CIE id = %#x, want 0xffffffff", tc.name, cid)
|
||||||
|
}
|
||||||
|
if v := r.byteAt(t); v != 3 {
|
||||||
|
t.Errorf("%s: CIE version = %d, want 3", tc.name, v)
|
||||||
|
}
|
||||||
|
if aug := r.byteAt(t); aug != 0 {
|
||||||
|
t.Errorf("%s: CIE augmentation = %d, want 0", tc.name, aug)
|
||||||
|
}
|
||||||
|
if ca := r.uleb(t); ca != 1 {
|
||||||
|
t.Errorf("%s: code alignment = %d, want 1", tc.name, ca)
|
||||||
|
}
|
||||||
|
if da := r.sleb(t); da != -8 {
|
||||||
|
t.Errorf("%s: data alignment = %d, want -8 (signed LEB128, not zigzag)", tc.name, da)
|
||||||
|
}
|
||||||
|
if ra := r.uleb(t); ra != uint64(tc.cfi.raReg) {
|
||||||
|
t.Errorf("%s: return-address register = %d, want %d", tc.name, ra, tc.cfi.raReg)
|
||||||
|
}
|
||||||
|
if op := r.byteAt(t); op != 0x0c {
|
||||||
|
t.Errorf("%s: expected DW_CFA_def_cfa, got opcode %#x", tc.name, op)
|
||||||
|
}
|
||||||
|
if cfa := r.uleb(t); cfa != uint64(tc.cfi.cfaReg) {
|
||||||
|
t.Errorf("%s: CFA register = %d, want %d", tc.name, cfa, tc.cfi.cfaReg)
|
||||||
|
}
|
||||||
|
if off := r.uleb(t); off != 0 {
|
||||||
|
t.Errorf("%s: CFA offset = %d, want 0", tc.name, off)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestEmitDWARF(t *testing.T) {
|
func TestEmitDWARF(t *testing.T) {
|
||||||
src := `#include "textflag.h"
|
src := `#include "textflag.h"
|
||||||
TEXT ·add(SB), NOSPLIT, $0-24
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
@@ -27,7 +329,7 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
t.Fatalf("assemble: %v", err)
|
t.Fatalf("assemble: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
ds := emitDWARF(img, "test_amd64.s")
|
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
|
||||||
|
|
||||||
// .debug_abbrev must not be empty and must start with abbrev code 1.
|
// .debug_abbrev must not be empty and must start with abbrev code 1.
|
||||||
if len(ds.debugAbbrev) == 0 {
|
if len(ds.debugAbbrev) == 0 {
|
||||||
@@ -68,14 +370,3 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
t.Fatal("no .debug_info relocations")
|
t.Fatal("no .debug_info relocations")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestDwarfAbbrevTable(t *testing.T) {
|
|
||||||
abbrev := dwarfAbbrevTable()
|
|
||||||
if len(abbrev) == 0 {
|
|
||||||
t.Fatal("empty abbrev table")
|
|
||||||
}
|
|
||||||
// Must end with a zero byte (end of table).
|
|
||||||
if abbrev[len(abbrev)-1] != 0 {
|
|
||||||
t.Fatalf("abbrev table last byte = %d, want 0", abbrev[len(abbrev)-1])
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
+552
-2
@@ -12,7 +12,8 @@ import (
|
|||||||
"path/filepath"
|
"path/filepath"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// The object-file tests share one source: two exported functions, one
|
// The object-file tests share one source: two exported functions, one
|
||||||
@@ -52,7 +53,7 @@ func elfTestImage(t *testing.T) *Image {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestAssembleFileExternals checks that a reference to a symbol no GLOBL
|
// TestAssembleFileExternals checks that a reference to a symbol no GLOBL
|
||||||
// defines is recorded as an external relocation instead of failing — the
|
// defines is recorded as an external relocation instead of failing; the
|
||||||
// raw image leaves the displacement zero, the object emitters carry it.
|
// raw image leaves the displacement zero, the object emitters carry it.
|
||||||
func TestAssembleFileExternals(t *testing.T) {
|
func TestAssembleFileExternals(t *testing.T) {
|
||||||
img := elfTestImage(t)
|
img := elfTestImage(t)
|
||||||
@@ -211,6 +212,75 @@ func TestELFObject(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestELFObjectTLSGuardReloc checks that a non-NOSPLIT function's stack
|
||||||
|
// guard carries an R_X86_64_TPOFF32 relocation against the null symbol in
|
||||||
|
// .rela.text. The serialisation must honour the record's type field: a
|
||||||
|
// hardcoded R_X86_64_PC32 mislinks the TLS load as an ordinary
|
||||||
|
// PC-relative reference.
|
||||||
|
func TestELFObjectTLSGuardReloc(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("g_amd64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·grow(SB), $0
|
||||||
|
CALL ·other(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·other(SB), NOSPLIT, $0
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
var haveTLS bool
|
||||||
|
for _, fn := range img.Funcs {
|
||||||
|
for _, r := range fn.Relocs {
|
||||||
|
if r.Kind == RelTLSLE {
|
||||||
|
haveTLS = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !haveTLS {
|
||||||
|
t.Fatal("test source produced no RelTLSLE relocation")
|
||||||
|
}
|
||||||
|
obj, err := img.ELFObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFObject: %v", err)
|
||||||
|
}
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
relaSec := ef.Section(".rela.text")
|
||||||
|
if relaSec == nil {
|
||||||
|
t.Fatal("missing .rela.text")
|
||||||
|
}
|
||||||
|
raw, err := relaSec.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
found := false
|
||||||
|
for i := 0; i+24 <= len(raw); i += 24 {
|
||||||
|
e := raw[i:]
|
||||||
|
info := binary.LittleEndian.Uint64(e[8:])
|
||||||
|
typ := info & 0xffffffff
|
||||||
|
sym := int(info >> 32)
|
||||||
|
if typ == uint64(elf.R_X86_64_TPOFF32) {
|
||||||
|
found = true
|
||||||
|
if sym != 0 {
|
||||||
|
t.Errorf("TPOFF32 relocation against symbol %d, want 0 (the null symbol)", sym)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !found {
|
||||||
|
t.Errorf("no R_X86_64_TPOFF32 relocation in .rela.text (%d bytes)", len(raw))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestELFObjectNoRelocations checks a file with no static-symbol references
|
// TestELFObjectNoRelocations checks a file with no static-symbol references
|
||||||
// emits a valid object without a .rela.text section.
|
// emits a valid object without a .rela.text section.
|
||||||
func TestELFObjectNoRelocations(t *testing.T) {
|
func TestELFObjectNoRelocations(t *testing.T) {
|
||||||
@@ -253,6 +323,238 @@ TEXT ·nop(SB), NOSPLIT, $0
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// elfSectionHeaderCount returns the e_shnum the ELF header declares.
|
||||||
|
func elfSectionHeaderCount(t *testing.T, obj []byte) int {
|
||||||
|
t.Helper()
|
||||||
|
return int(binary.LittleEndian.Uint16(obj[60:]))
|
||||||
|
}
|
||||||
|
|
||||||
|
// checkELFSectionAccounting verifies the number of section headers the
|
||||||
|
// writer physically laid out equals e_shnum: every DWARF section written
|
||||||
|
// after .shstrtab must be counted, or the last ones (always .debug_frame)
|
||||||
|
// are invisible to every consumer, debug/elf included.
|
||||||
|
func checkELFSectionAccounting(t *testing.T, obj []byte) {
|
||||||
|
t.Helper()
|
||||||
|
shoff := int(binary.LittleEndian.Uint64(obj[40:]))
|
||||||
|
shentsize := int(binary.LittleEndian.Uint16(obj[58:]))
|
||||||
|
shnum := elfSectionHeaderCount(t, obj)
|
||||||
|
if shentsize != 64 {
|
||||||
|
t.Fatalf("e_shentsize = %d, want 64", shentsize)
|
||||||
|
}
|
||||||
|
if (len(obj)-shoff)%shentsize != 0 {
|
||||||
|
t.Fatalf("section header table is not a whole number of entries: shoff=%d len=%d", shoff, len(obj))
|
||||||
|
}
|
||||||
|
if present := (len(obj) - shoff) / shentsize; present != shnum {
|
||||||
|
t.Errorf("e_shnum = %d but %d section headers are laid out", shnum, present)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestELFDWARFSectionAccounting runs the header accounting check over all
|
||||||
|
// four architecture emitters, and additionally checks the .debug_frame
|
||||||
|
// section is visible (its data aligned as its header declares).
|
||||||
|
func TestELFDWARFSectionAccounting(t *testing.T) {
|
||||||
|
parse := func(name, src string) *ast.File {
|
||||||
|
f, errs := parser.Parse(name, src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse %s: %v", name, errs)
|
||||||
|
}
|
||||||
|
return f
|
||||||
|
}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
img *Image
|
||||||
|
emit func(*Image) ([]byte, error)
|
||||||
|
}{
|
||||||
|
{"amd64", elfTestImage(t), (*Image).ELFObject},
|
||||||
|
{"arm64", mustImage(t, func() (*Image, error) {
|
||||||
|
return AssembleFileARM64(parse("k_arm64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOVD a+0(FP), R4
|
||||||
|
MOVD b+8(FP), R5
|
||||||
|
ADD R5, R4, R4
|
||||||
|
MOVD R4, ret+16(FP)
|
||||||
|
RET
|
||||||
|
`))
|
||||||
|
}), (*Image).ELFAARCH64Object},
|
||||||
|
{"riscv64", mustImage(t, func() (*Image, error) {
|
||||||
|
return AssembleFileRISCV(parse("k_riscv64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·sb(SB), NOSPLIT, $0-0
|
||||||
|
MOV $answer<>(SB), X10
|
||||||
|
RET
|
||||||
|
|
||||||
|
GLOBL answer<>(SB), RODATA, $8
|
||||||
|
DATA answer<>+0(SB)/8, $42
|
||||||
|
`))
|
||||||
|
}), (*Image).ELFRISCVObject},
|
||||||
|
{"loong64", mustImage(t, func() (*Image, error) {
|
||||||
|
return AssembleFileLOONG64(parse("k_loong64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOVV a+0(FP), R4
|
||||||
|
MOVV b+8(FP), R5
|
||||||
|
ADDV R5, R4, R4
|
||||||
|
MOVV R4, ret+16(FP)
|
||||||
|
RET
|
||||||
|
`))
|
||||||
|
}), (*Image).ELFLOONG64Object},
|
||||||
|
}
|
||||||
|
for _, tc := range cases {
|
||||||
|
obj, err := tc.emit(tc.img)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("%s: emit: %v", tc.name, err)
|
||||||
|
}
|
||||||
|
checkELFSectionAccounting(t, obj)
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("%s: parse emitted object: %v", tc.name, err)
|
||||||
|
}
|
||||||
|
frame := ef.Section(".debug_frame")
|
||||||
|
if frame == nil {
|
||||||
|
t.Errorf("%s: .debug_frame invisible to debug/elf (e_shnum too small?)", tc.name)
|
||||||
|
ef.Close()
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if frame.Offset%8 != 0 || frame.Addralign != 8 {
|
||||||
|
t.Errorf("%s: .debug_frame offset %d align %d, want offset%%8==0 align 8", tc.name, frame.Offset, frame.Addralign)
|
||||||
|
}
|
||||||
|
ef.Close()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func mustImage(t *testing.T, f func() (*Image, error)) *Image {
|
||||||
|
t.Helper()
|
||||||
|
img, err := f()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return img
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestELFDWARFRelocations checks the .rela.debug_info and .rela.debug_line
|
||||||
|
// sections exist and carry absolute 64-bit relocations against the
|
||||||
|
// function symbols, with r_offsets inside their target sections.
|
||||||
|
func TestELFDWARFRelocations(t *testing.T) {
|
||||||
|
img := elfTestImage(t)
|
||||||
|
obj, err := img.ELFObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFObject: %v", err)
|
||||||
|
}
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
// The DWARF must record the assembled file's path (threaded through
|
||||||
|
// Image.SourcePath), not a placeholder name.
|
||||||
|
info, err := ef.Section(".debug_info").Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if img.SourcePath != "t_amd64.s" || !bytes.Contains(info, []byte(img.SourcePath)) {
|
||||||
|
t.Errorf("DWARF compilation unit does not name the source %q", img.SourcePath)
|
||||||
|
}
|
||||||
|
for _, tc := range []struct {
|
||||||
|
rela string
|
||||||
|
target string
|
||||||
|
want uint32
|
||||||
|
}{
|
||||||
|
{".rela.debug_info", ".debug_info", rX8664Abs64},
|
||||||
|
{".rela.debug_line", ".debug_line", rX8664Abs64},
|
||||||
|
{".rela.debug_frame", ".debug_frame", rX8664Abs64},
|
||||||
|
} {
|
||||||
|
rs := ef.Section(tc.rela)
|
||||||
|
if rs == nil {
|
||||||
|
t.Fatalf("missing %s", tc.rela)
|
||||||
|
}
|
||||||
|
if rs.Type != elf.SHT_RELA {
|
||||||
|
t.Errorf("%s: type %v, want SHT_RELA", tc.rela, rs.Type)
|
||||||
|
}
|
||||||
|
target := ef.Section(tc.target)
|
||||||
|
if target == nil {
|
||||||
|
t.Fatalf("missing %s", tc.target)
|
||||||
|
}
|
||||||
|
if rs.Link == 0 || ef.Sections[rs.Info] != target {
|
||||||
|
t.Errorf("%s: link %d info %d, want the symtab and %s", tc.rela, rs.Link, rs.Info, tc.target)
|
||||||
|
}
|
||||||
|
b, err := rs.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
// .debug_line has one address per function; .debug_info adds the
|
||||||
|
// compile unit's own low_pc.
|
||||||
|
want := len(img.Funcs)
|
||||||
|
if tc.target == ".debug_info" {
|
||||||
|
want++
|
||||||
|
}
|
||||||
|
if len(b)/24 != want {
|
||||||
|
t.Errorf("%s: %d entries, want %d", tc.rela, len(b)/24, want)
|
||||||
|
}
|
||||||
|
for i := 0; i+24 <= len(b); i += 24 {
|
||||||
|
r_offset := binary.LittleEndian.Uint64(b[i:])
|
||||||
|
info := binary.LittleEndian.Uint64(b[i+8:])
|
||||||
|
typ := uint32(info)
|
||||||
|
sym := int(info >> 32)
|
||||||
|
if typ != tc.want {
|
||||||
|
t.Errorf("%s entry %d: type %d, want R_X86_64_64 (%d)", tc.rela, i/24, typ, tc.want)
|
||||||
|
}
|
||||||
|
if r_offset >= uint64(target.Size) {
|
||||||
|
t.Errorf("%s entry %d: r_offset %d outside %s (%d bytes)", tc.rela, i/24, r_offset, tc.target, target.Size)
|
||||||
|
}
|
||||||
|
if sym == 0 {
|
||||||
|
t.Errorf("%s entry %d: against the null symbol", tc.rela, i/24)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestELFDataOnly checks a source with GLOBL data and no TEXT emits a valid
|
||||||
|
// ELF object: the DWARF compilation unit of a code-less image has no
|
||||||
|
// function to relocate against and must not reach for one.
|
||||||
|
func TestELFDataOnly(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("d0_amd64.s", `
|
||||||
|
GLOBL table<>(SB), RODATA, $8
|
||||||
|
DATA table<>+0(SB)/8, $12345
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFObject: %v", err)
|
||||||
|
}
|
||||||
|
checkELFSectionAccounting(t, obj)
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
found := false
|
||||||
|
for _, s := range syms {
|
||||||
|
if s.Name == "table" && s.Size == 8 {
|
||||||
|
found = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !found {
|
||||||
|
t.Errorf("data symbol table missing: %v", syms)
|
||||||
|
}
|
||||||
|
if ef.Section(".rela.debug_info") != nil || ef.Section(".rela.debug_line") != nil {
|
||||||
|
t.Error("data-only image must not emit DWARF address relocations")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestELFLinkAndRun is the end-to-end check: assemble the test functions,
|
// TestELFLinkAndRun is the end-to-end check: assemble the test functions,
|
||||||
// link the emitted object with a C driver that defines the external symbol,
|
// link the emitted object with a C driver that defines the external symbol,
|
||||||
// and run the result. Skipped when no C compiler is available.
|
// and run the result. Skipped when no C compiler is available.
|
||||||
@@ -307,4 +609,252 @@ int main(void) {
|
|||||||
if got := string(run); got != "42 42 7\n" {
|
if got := string(run); got != "42 42 7\n" {
|
||||||
t.Errorf("output %q, want \"42 42 7\\n\"", got)
|
t.Errorf("output %q, want \"42 42 7\\n\"", got)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The DWARF addresses must have resolved at link time: the .debug_info
|
||||||
|
// placeholders were carried by .rela.debug_info, so every subprogram's
|
||||||
|
// low_pc must now equal its linked symbol address.
|
||||||
|
bin, err := os.ReadFile(appPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
lef, err := elf.NewFile(bytes.NewReader(bin))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse linked binary: %v", err)
|
||||||
|
}
|
||||||
|
defer lef.Close()
|
||||||
|
syms, err := lef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
addrByName := map[string]uint64{}
|
||||||
|
for _, s := range syms {
|
||||||
|
if elf.ST_TYPE(s.Info) == elf.STT_FUNC && s.Value != 0 {
|
||||||
|
addrByName[s.Name] = s.Value
|
||||||
|
}
|
||||||
|
}
|
||||||
|
lowPCs := dwarfSubprogramLowPCs(t, lef)
|
||||||
|
if len(lowPCs) == 0 {
|
||||||
|
t.Fatal("no subprogram DW_AT_low_pc parsed from the linked binary")
|
||||||
|
}
|
||||||
|
for name, pc := range lowPCs {
|
||||||
|
addr, ok := addrByName[name]
|
||||||
|
if !ok {
|
||||||
|
t.Errorf("subprogram %q not in the linked symbol table", name)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if pc != addr {
|
||||||
|
t.Errorf("subprogram %q: DW_AT_low_pc = %#x, linked address %#x (DWARF relocation unresolved)", name, pc, addr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// dwarfSubprogramLowPCs walks the linked binary's .debug_info with its own
|
||||||
|
// .debug_abbrev and returns each DW_TAG_subprogram's DW_AT_low_pc by name.
|
||||||
|
func dwarfSubprogramLowPCs(t *testing.T, ef *elf.File) map[string]uint64 {
|
||||||
|
t.Helper()
|
||||||
|
abbrevSec := ef.Section(".debug_abbrev")
|
||||||
|
infoSec := ef.Section(".debug_info")
|
||||||
|
if abbrevSec == nil || infoSec == nil {
|
||||||
|
t.Fatal("linked binary lacks .debug_abbrev or .debug_info")
|
||||||
|
}
|
||||||
|
abbrev, err := abbrevSec.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
info, err := infoSec.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
abs := parseAbbrevs(t, abbrev)
|
||||||
|
le := binary.LittleEndian
|
||||||
|
out := map[string]uint64{}
|
||||||
|
r := &ulebIter{b: info}
|
||||||
|
r.uint32At(t) // unit_length
|
||||||
|
if v := le.Uint16(info[4:]); v != 5 {
|
||||||
|
t.Fatalf(".debug_info version %d, want 5", v)
|
||||||
|
}
|
||||||
|
r.i = 6
|
||||||
|
r.byteAt(t) // unit_type
|
||||||
|
r.byteAt(t) // address_size
|
||||||
|
r.uint32At(t) // debug_abbrev_offset
|
||||||
|
var name string
|
||||||
|
var lowPC uint64
|
||||||
|
for r.i < len(r.b) {
|
||||||
|
code := r.uleb(t)
|
||||||
|
if code == 0 {
|
||||||
|
continue // end of the CU's children
|
||||||
|
}
|
||||||
|
ab, ok := abs[code]
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("unknown abbreviation code %d", code)
|
||||||
|
}
|
||||||
|
name, lowPC = "", 0
|
||||||
|
for _, a := range ab.attrs {
|
||||||
|
switch a.attr {
|
||||||
|
case dwAtName:
|
||||||
|
readFormKeep(t, r, a.form, &name, nil)
|
||||||
|
case dwAtLowPC:
|
||||||
|
readFormKeep(t, r, a.form, nil, &lowPC)
|
||||||
|
default:
|
||||||
|
readFormSkip(t, r, a.form)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if ab.tag == dwTagSubprog && name != "" {
|
||||||
|
out[name] = lowPC
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// readFormKeep reads one DIE attribute value, keeping a string or an
|
||||||
|
// address into the pointer it was given (nil keeps nothing).
|
||||||
|
func readFormKeep(t *testing.T, r *ulebIter, form uint64, name *string, addr *uint64) {
|
||||||
|
t.Helper()
|
||||||
|
switch form {
|
||||||
|
case dwFormString:
|
||||||
|
end := r.i
|
||||||
|
for end < len(r.b) && r.b[end] != 0 {
|
||||||
|
end++
|
||||||
|
}
|
||||||
|
if name != nil {
|
||||||
|
*name = string(r.b[r.i:end])
|
||||||
|
}
|
||||||
|
r.i = end + 1
|
||||||
|
case dwFormAddr:
|
||||||
|
if addr != nil {
|
||||||
|
*addr = binary.LittleEndian.Uint64(r.b[r.i:])
|
||||||
|
}
|
||||||
|
r.i += 8
|
||||||
|
default:
|
||||||
|
readFormSkip(t, r, form)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func readFormSkip(t *testing.T, r *ulebIter, form uint64) {
|
||||||
|
t.Helper()
|
||||||
|
switch form {
|
||||||
|
case dwFormString:
|
||||||
|
for r.i < len(r.b) && r.b[r.i] != 0 {
|
||||||
|
r.i++
|
||||||
|
}
|
||||||
|
r.i++
|
||||||
|
case dwFormAddr, dwFormData8:
|
||||||
|
r.i += 8
|
||||||
|
case dwFormSecOff:
|
||||||
|
r.i += 4
|
||||||
|
case dwFormExprloc:
|
||||||
|
r.i += int(r.uleb(t))
|
||||||
|
case dwFormData1, 0x0c:
|
||||||
|
r.i++
|
||||||
|
default:
|
||||||
|
t.Fatalf("unsupported form %#x", form)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestELFObjectDataRelocation checks that a symbol-valued DATA field ("DATA
|
||||||
|
// s+0(SB)/8, $other(SB)") reaches the ELF object as a .rela.data entry: an
|
||||||
|
// absolute 64-bit relocation at the field's offset within .data, against
|
||||||
|
// the named symbol, external targets included.
|
||||||
|
func TestELFObjectDataRelocation(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("t_amd64.s", `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-8
|
||||||
|
RET
|
||||||
|
GLOBL holder(SB), NOPTR, $24
|
||||||
|
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||||
|
DATA holder+8(SB)/8, $holder(SB)
|
||||||
|
DATA holder+16(SB)/8, $extvar(SB)
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFObject: %v", err)
|
||||||
|
}
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
relaData := ef.Section(".rela.data")
|
||||||
|
if relaData == nil {
|
||||||
|
t.Fatal("missing .rela.data section")
|
||||||
|
}
|
||||||
|
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||||
|
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||||
|
}
|
||||||
|
if ef.Sections[relaData.Info].Name != ".data" {
|
||||||
|
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||||
|
}
|
||||||
|
relas, err := relaData.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var got []struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}
|
||||||
|
for i := 0; i+24 <= len(relas); i += 24 {
|
||||||
|
got = append(got, struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}{
|
||||||
|
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||||
|
// r_info packs the type in the low dword and the symbol index
|
||||||
|
// in the high dword.
|
||||||
|
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||||
|
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||||
|
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
name := func(idx uint32) string {
|
||||||
|
if idx >= 1 && int(idx) <= len(syms) {
|
||||||
|
return syms[idx-1].Name
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
// The offsets are data-section-relative: the field's DATA offset plus
|
||||||
|
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||||
|
base := uint64(0)
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
if d.Name == "holder" {
|
||||||
|
base = uint64(d.Offset)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
want := []struct {
|
||||||
|
off uint64
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
target string
|
||||||
|
}{
|
||||||
|
{off: base + 0, typ: uint32(elf.R_X86_64_64), addend: 5, target: "Keep"},
|
||||||
|
{off: base + 8, typ: uint32(elf.R_X86_64_64), addend: 0, target: "holder"},
|
||||||
|
{off: base + 16, typ: uint32(elf.R_X86_64_64), addend: 0, target: "extvar"},
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
g := got[i]
|
||||||
|
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||||
|
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||||
|
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||||
|
}
|
||||||
|
if n := name(g.sym); n != w.target {
|
||||||
|
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+139
-30
@@ -18,12 +18,16 @@ const (
|
|||||||
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset)
|
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset)
|
||||||
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
|
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
|
||||||
rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset)
|
rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset)
|
||||||
|
// R_AARCH64_ABS32 (debug/elf 258): the absolute 32-bit address of a
|
||||||
|
// symbol, the R_ADDR shape a 4-byte DATA field carries. ABS64 (257)
|
||||||
|
// lives with the DWARF fixup constants as rAARCH64Abs64.
|
||||||
|
rArm64Abs32 = 258
|
||||||
)
|
)
|
||||||
|
|
||||||
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
|
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
|
||||||
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
|
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
|
||||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
|
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
|
||||||
// optional .rela.text.
|
// optional .rela.text and an optional .rela.data.
|
||||||
func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||||
le := binary.LittleEndian
|
le := binary.LittleEndian
|
||||||
|
|
||||||
@@ -81,10 +85,15 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Build relocations. Each SB reference is an ADRP pair:
|
// Build relocations. Each SB reference is an ADRP pair:
|
||||||
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21
|
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21 at the ADRP
|
||||||
// ADD → R_AARCH64_ADD_ABS_LO12_NC
|
// ADD → R_AARCH64_ADD_ABS_LO12_NC at the ADD word
|
||||||
// LDR/STR X → R_AARCH64_LDST64_ABS_LO12_NC
|
// LDR/STR X → R_AARCH64_LDST64_ABS_LO12_NC at the LDR/STR word
|
||||||
// BL → R_AARCH64_CALL26
|
// BL → R_AARCH64_CALL26
|
||||||
|
// cmd/link's own conversion emits the HI21 at sectoff and the LO12 at
|
||||||
|
// sectoff+4 (cmd/link/internal/arm64/asm.go), so the ADD or load word
|
||||||
|
// carries the page-offset relocation, never a second HI21. The
|
||||||
|
// assembler records two RelArm64Addr relocs per ADRP+ADD pair (one per
|
||||||
|
// word), so the second of the pair is consumed here.
|
||||||
// Addends stay raw: ADR_PREL_PG_HI21 and the ABS_LO12_NC forms resolve
|
// Addends stay raw: ADR_PREL_PG_HI21 and the ABS_LO12_NC forms resolve
|
||||||
// against S+A, and CALL26 branches take the branch instruction's own
|
// against S+A, and CALL26 branches take the branch instruction's own
|
||||||
// place as the PC-relative base, so subtracting the field width (the
|
// place as the PC-relative base, so subtracting the field width (the
|
||||||
@@ -97,31 +106,81 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
}
|
}
|
||||||
var relas []elfRela
|
var relas []elfRela
|
||||||
for _, fn := range img.Funcs {
|
for _, fn := range img.Funcs {
|
||||||
for _, r := range fn.Relocs {
|
for i := 0; i < len(fn.Relocs); i++ {
|
||||||
|
r := fn.Relocs[i]
|
||||||
idx, ok := symIdx[r.Name]
|
idx, ok := symIdx[r.Name]
|
||||||
if !ok {
|
if !ok {
|
||||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||||
}
|
}
|
||||||
var typ uint32
|
switch r.Kind {
|
||||||
switch {
|
case RelArm64Branch:
|
||||||
case r.Kind == RelArm64Branch:
|
relas = append(relas, elfRela{
|
||||||
typ = rArm64Call26
|
off: uint64(fn.Offset + r.Off), typ: rArm64Call26, sym: idx, addend: r.Addend,
|
||||||
case r.Kind == RelArm64LDST64 && r.Off%4 == 4:
|
})
|
||||||
typ = rArm64Ldst64Lo12NC
|
case RelArm64Addr:
|
||||||
case r.Kind == RelArm64Addr && r.Off%4 == 4:
|
// ADRP+ADD: the pair's second reloc (at Off+4) is the
|
||||||
typ = rArm64AddAbsLo12NC
|
// assembler's twin of the same pair; skip it.
|
||||||
|
relas = append(relas,
|
||||||
|
elfRela{off: uint64(fn.Offset + r.Off), typ: rArm64PrelPgHi21, sym: idx, addend: r.Addend},
|
||||||
|
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rArm64AddAbsLo12NC, sym: idx, addend: r.Addend},
|
||||||
|
)
|
||||||
|
i++
|
||||||
|
case RelArm64LDST64:
|
||||||
|
// ADRP+LDR/STR: one assembler reloc covers the pair.
|
||||||
|
relas = append(relas,
|
||||||
|
elfRela{off: uint64(fn.Offset + r.Off), typ: rArm64PrelPgHi21, sym: idx, addend: r.Addend},
|
||||||
|
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rArm64Ldst64Lo12NC, sym: idx, addend: r.Addend},
|
||||||
|
)
|
||||||
default:
|
default:
|
||||||
typ = rArm64PrelPgHi21
|
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
|
||||||
}
|
}
|
||||||
relas = append(relas, elfRela{
|
}
|
||||||
off: uint64(fn.Offset + r.Off),
|
}
|
||||||
typ: typ,
|
|
||||||
|
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||||
|
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||||
|
// DATA line's width at the field's data-section offset, S + A with no
|
||||||
|
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||||
|
// cannot hold an address, so they are refused rather than truncated.
|
||||||
|
var dataRelas []elfRela
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
for _, r := range d.Relocs {
|
||||||
|
idx, ok := symIdx[r.Name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||||
|
}
|
||||||
|
var typ uint32
|
||||||
|
switch r.Siz {
|
||||||
|
case 8:
|
||||||
|
typ = rAARCH64Abs64
|
||||||
|
case 4:
|
||||||
|
typ = rArm64Abs32
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||||
|
}
|
||||||
|
dataRelas = append(dataRelas, elfRela{
|
||||||
|
off: uint64(d.Offset + r.Off),
|
||||||
sym: idx,
|
sym: idx,
|
||||||
|
typ: typ,
|
||||||
addend: r.Addend,
|
addend: r.Addend,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Section presence: .rela.text only when there are code relocations,
|
||||||
|
// .rela.data only when a DATA line holds a symbol value.
|
||||||
|
hasRela := len(relas) > 0
|
||||||
|
hasDataRela := len(dataRelas) > 0
|
||||||
|
nSections := 6
|
||||||
|
if hasRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
secSymtab, secStrtab := 3, 4
|
||||||
|
secShstr := nSections - 1
|
||||||
|
|
||||||
// String tables.
|
// String tables.
|
||||||
stNames := newElfStrtab()
|
stNames := newElfStrtab()
|
||||||
for _, s := range syms {
|
for _, s := range syms {
|
||||||
@@ -131,18 +190,13 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
stSections.add(".rela.data")
|
||||||
|
}
|
||||||
for _, n := range dwarfSectionNames {
|
for _, n := range dwarfSectionNames {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
|
||||||
hasRela := len(relas) > 0
|
|
||||||
nSections := 6
|
|
||||||
if hasRela {
|
|
||||||
nSections = 7
|
|
||||||
}
|
|
||||||
secSymtab, secStrtab := 3, 4
|
|
||||||
secShstr := nSections - 1
|
|
||||||
|
|
||||||
// Layout.
|
// Layout.
|
||||||
var out []byte
|
var out []byte
|
||||||
out = append(out, make([]byte, 64)...)
|
out = append(out, make([]byte, 64)...)
|
||||||
@@ -177,7 +231,7 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
strtabOff := len(out)
|
strtabOff := len(out)
|
||||||
out = append(out, stNames.bytes()...)
|
out = append(out, stNames.bytes()...)
|
||||||
|
|
||||||
var relaOff int
|
var relaOff, relaDataOff int
|
||||||
if hasRela {
|
if hasRela {
|
||||||
align(8)
|
align(8)
|
||||||
relaOff = len(out)
|
relaOff = len(out)
|
||||||
@@ -189,19 +243,47 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
out = append(out, b[:]...)
|
out = append(out, b[:]...)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
align(8)
|
||||||
|
relaDataOff = len(out)
|
||||||
|
for _, r := range dataRelas {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint64(b[0:], r.off)
|
||||||
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||||
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
shstrOff := len(out)
|
shstrOff := len(out)
|
||||||
out = append(out, stSections.bytes()...)
|
out = append(out, stSections.bytes()...)
|
||||||
|
|
||||||
// DWARF debug sections.
|
// DWARF debug sections; the address placeholders they leave are carried
|
||||||
|
// as .rela.debug_info/.rela.debug_line entries the system linker applies.
|
||||||
dwAlign := func(n int) {
|
dwAlign := func(n int) {
|
||||||
for len(out)%n != 0 {
|
for len(out)%n != 0 {
|
||||||
out = append(out, 0)
|
out = append(out, 0)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
|
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiARM64)
|
||||||
|
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
nSections += 4
|
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
|
||||||
|
// .debug_line_str and .debug_frame (the CIE is unconditional, so
|
||||||
|
// the frame section is always present), plus the relocation
|
||||||
|
// sections below when they carry entries.
|
||||||
|
dwarfStart = nSections
|
||||||
|
nSections += 5
|
||||||
|
appendDWARFRelas(&out, dw, rAARCH64Abs64, dwAlign)
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
align(8)
|
align(8)
|
||||||
@@ -229,14 +311,41 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
if hasRela {
|
if hasRela {
|
||||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||||
|
}
|
||||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
|
// DWARF section headers; their indices follow the write order.
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
|
// secIdx is a running section index: each putSh below emits the
|
||||||
|
// next header, and the sh_info of a .rela section names the index
|
||||||
|
// of the section it relocates.
|
||||||
|
secIdx := dwarfStart
|
||||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||||
|
secInfoIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||||
|
secLineIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
if dw.frameSize > 0 {
|
if dw.frameSize > 0 {
|
||||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||||
|
secFrameIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+174
-2
@@ -6,9 +6,10 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
"debug/elf"
|
"debug/elf"
|
||||||
|
"encoding/binary"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
|
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
|
||||||
@@ -28,6 +29,7 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
|
|
||||||
TEXT ·getanswer(SB), NOSPLIT, $0-8
|
TEXT ·getanswer(SB), NOSPLIT, $0-8
|
||||||
MOVD answer<>(SB), R4
|
MOVD answer<>(SB), R4
|
||||||
|
MOVD $answer<>(SB), R5
|
||||||
MOVD R4, ret+0(FP)
|
MOVD R4, ret+0(FP)
|
||||||
RET
|
RET
|
||||||
|
|
||||||
@@ -102,8 +104,63 @@ DATA answer<>+0(SB)/8, $42
|
|||||||
// Check that .rela.text exists (getanswer has SB reference).
|
// Check that .rela.text exists (getanswer has SB reference).
|
||||||
relaText := ef.Section(".rela.text")
|
relaText := ef.Section(".rela.text")
|
||||||
if relaText == nil {
|
if relaText == nil {
|
||||||
t.Error("missing .rela.text section")
|
t.Fatal("missing .rela.text section")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The SB references of getanswer form two ADRP pairs: the load
|
||||||
|
// (MOVD answer<>(SB), R4) is ADRP+LDR carrying HI21 at the ADRP and
|
||||||
|
// LDST64_ABS_LO12_NC at the LDR word, and the address-of
|
||||||
|
// (MOVD $answer<>(SB), R5) is ADRP+ADD carrying HI21 and
|
||||||
|
// ADD_ABS_LO12_NC. cmd/link's own conversion emits exactly this
|
||||||
|
// sectoff / sectoff+4 pairing; a second HI21 at the ADD or LDR word
|
||||||
|
// corrupts the pair.
|
||||||
|
raw, err := relaText.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if len(raw)%24 != 0 || len(raw)/24 != 4 {
|
||||||
|
t.Fatalf(".rela.text has %d bytes, want four 24-byte entries", len(raw))
|
||||||
|
}
|
||||||
|
wantRela := []struct {
|
||||||
|
typ elf.R_AARCH64
|
||||||
|
off uint64 // relative to the getanswer function start
|
||||||
|
}{
|
||||||
|
{elf.R_AARCH64_ADR_PREL_PG_HI21, 0},
|
||||||
|
{elf.R_AARCH64_LDST64_ABS_LO12_NC, 4},
|
||||||
|
{elf.R_AARCH64_ADR_PREL_PG_HI21, 8},
|
||||||
|
{elf.R_AARCH64_ADD_ABS_LO12_NC, 12},
|
||||||
|
}
|
||||||
|
getanswer := byNameElf(t, ef, "getanswer")
|
||||||
|
for i, w := range wantRela {
|
||||||
|
e := raw[i*24 : (i+1)*24]
|
||||||
|
off := binary.LittleEndian.Uint64(e[0:])
|
||||||
|
info := binary.LittleEndian.Uint64(e[8:])
|
||||||
|
typ := elf.R_AARCH64(info & 0xffffffff)
|
||||||
|
sym := int(info >> 32)
|
||||||
|
if typ != w.typ || off != getanswer.Value+w.off {
|
||||||
|
t.Errorf("reloc %d: type %v off %d, want %v at %d", i, typ, off, w.typ, getanswer.Value+w.off)
|
||||||
|
}
|
||||||
|
if sym != 3 { // NULL, .text, .data, then the first local: answer
|
||||||
|
t.Errorf("reloc %d: symbol index %d, want 3 (answer)", i, sym)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// byNameElf returns the symbol table entry for name from the raw .symtab,
|
||||||
|
// which carries every entry including the null and section symbols in order.
|
||||||
|
func byNameElf(t *testing.T, ef *elf.File, name string) elf.Symbol {
|
||||||
|
t.Helper()
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("symbols: %v", err)
|
||||||
|
}
|
||||||
|
for _, s := range syms {
|
||||||
|
if s.Name == name {
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatalf("symbol %q not found", name)
|
||||||
|
return elf.Symbol{}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestELFAARCH64ObjectNoRelocations checks the ELF output when there are no
|
// TestELFAARCH64ObjectNoRelocations checks the ELF output when there are no
|
||||||
@@ -140,3 +197,118 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
t.Error("unexpected .rela.text section when there are no relocations")
|
t.Error("unexpected .rela.text section when there are no relocations")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestELFAARCH64ObjectDataRelocation checks that a symbol-valued DATA field
|
||||||
|
// ("DATA s+0(SB)/8, $other(SB)") reaches the AArch64 ELF object as a
|
||||||
|
// .rela.data entry: an R_AARCH64_ABS64 (ABS32 for a width-4 field) at the
|
||||||
|
// field's offset within .data, against the named symbol, external targets
|
||||||
|
// included.
|
||||||
|
func TestELFAARCH64ObjectDataRelocation(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("t_arm64.s", `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
GLOBL holder(SB), NOPTR, $32
|
||||||
|
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||||
|
DATA holder+8(SB)/8, $holder(SB)
|
||||||
|
DATA holder+16(SB)/8, $extvar(SB)
|
||||||
|
DATA holder+24(SB)/4, $Keep(SB)
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFAARCH64Object()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFAARCH64Object: %v", err)
|
||||||
|
}
|
||||||
|
checkELFSectionAccounting(t, obj)
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
relaData := ef.Section(".rela.data")
|
||||||
|
if relaData == nil {
|
||||||
|
t.Fatal("missing .rela.data section")
|
||||||
|
}
|
||||||
|
if relaData.Type != elf.SHT_RELA {
|
||||||
|
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
|
||||||
|
}
|
||||||
|
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||||
|
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||||
|
}
|
||||||
|
if ef.Sections[relaData.Info].Name != ".data" {
|
||||||
|
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||||
|
}
|
||||||
|
relas, err := relaData.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var got []struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}
|
||||||
|
for i := 0; i+24 <= len(relas); i += 24 {
|
||||||
|
got = append(got, struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}{
|
||||||
|
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||||
|
// r_info packs the type in the low dword and the symbol index
|
||||||
|
// in the high dword.
|
||||||
|
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||||
|
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||||
|
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
name := func(idx uint32) string {
|
||||||
|
if idx >= 1 && int(idx) <= len(syms) {
|
||||||
|
return syms[idx-1].Name
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
// The offsets are data-section-relative: the field's DATA offset plus
|
||||||
|
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||||
|
base := uint64(0)
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
if d.Name == "holder" {
|
||||||
|
base = uint64(d.Offset)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
want := []struct {
|
||||||
|
off uint64
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
target string
|
||||||
|
}{
|
||||||
|
{off: base + 0, typ: uint32(elf.R_AARCH64_ABS64), addend: 5, target: "Keep"},
|
||||||
|
{off: base + 8, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "holder"},
|
||||||
|
{off: base + 16, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "extvar"},
|
||||||
|
{off: base + 24, typ: uint32(elf.R_AARCH64_ABS32), addend: 0, target: "Keep"},
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
g := got[i]
|
||||||
|
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||||
|
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||||
|
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||||
|
}
|
||||||
|
if n := name(g.sym); n != w.target {
|
||||||
|
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+117
-14
@@ -13,16 +13,26 @@ import (
|
|||||||
const (
|
const (
|
||||||
emLOONGARCH = 258 // EM_LOONGARCH
|
emLOONGARCH = 258 // EM_LOONGARCH
|
||||||
|
|
||||||
|
// EF_LOONGARCH_ABI_DOUBLE_FLOAT | EF_LOONGARCH_OBJABI_V1: the flags the
|
||||||
|
// Go toolchain writes (cmd/link/internal/ld/elf.go: Flags = 0x43 for
|
||||||
|
// Loong64). System linkers refuse to merge ET_REL objects whose float
|
||||||
|
// ABI differs, so 0 (soft-float) would make the object unlinkable.
|
||||||
|
efLarchAbiDoubleObjV1 = 0x43
|
||||||
|
|
||||||
// LoongArch relocation types (the ELF psABI).
|
// LoongArch relocation types (the ELF psABI).
|
||||||
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
|
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
|
||||||
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
|
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
|
||||||
rLarchB26 = 66 // R_LARCH_B26 (b/bl, matches the Go linker's mapping)
|
rLarchB26 = 66 // R_LARCH_B26 (b/bl, matches the Go linker's mapping)
|
||||||
|
// R_LARCH_32 (debug/elf 1): the absolute 32-bit address of a symbol,
|
||||||
|
// the R_ADDR shape a 4-byte DATA field carries. R_LARCH_64 (2) lives
|
||||||
|
// with the DWARF fixup constants as rLarchAbs64.
|
||||||
|
rLarchAbs32 = 1
|
||||||
)
|
)
|
||||||
|
|
||||||
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
|
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
|
||||||
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
|
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
|
||||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
|
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
|
||||||
// optional .rela.text.
|
// optional .rela.text and an optional .rela.data.
|
||||||
func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||||
le := binary.LittleEndian
|
le := binary.LittleEndian
|
||||||
|
|
||||||
@@ -111,6 +121,50 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||||
|
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||||
|
// DATA line's width at the field's data-section offset, S + A with no
|
||||||
|
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||||
|
// cannot hold an address, so they are refused rather than truncated.
|
||||||
|
var dataRelas []elfRela
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
for _, r := range d.Relocs {
|
||||||
|
idx, ok := symIdx[r.Name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||||
|
}
|
||||||
|
var typ uint32
|
||||||
|
switch r.Siz {
|
||||||
|
case 8:
|
||||||
|
typ = rLarchAbs64
|
||||||
|
case 4:
|
||||||
|
typ = rLarchAbs32
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||||
|
}
|
||||||
|
dataRelas = append(dataRelas, elfRela{
|
||||||
|
off: uint64(d.Offset + r.Off),
|
||||||
|
sym: idx,
|
||||||
|
typ: typ,
|
||||||
|
addend: r.Addend,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Section presence: .rela.text only when there are code relocations,
|
||||||
|
// .rela.data only when a DATA line holds a symbol value.
|
||||||
|
hasRela := len(relas) > 0
|
||||||
|
hasDataRela := len(dataRelas) > 0
|
||||||
|
nSections := 6
|
||||||
|
if hasRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
secSymtab, secStrtab := 3, 4
|
||||||
|
secShstr := nSections - 1
|
||||||
|
|
||||||
// String tables.
|
// String tables.
|
||||||
stNames := newElfStrtab()
|
stNames := newElfStrtab()
|
||||||
for _, s := range syms {
|
for _, s := range syms {
|
||||||
@@ -120,18 +174,13 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
stSections.add(".rela.data")
|
||||||
|
}
|
||||||
for _, n := range dwarfSectionNames {
|
for _, n := range dwarfSectionNames {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
|
||||||
hasRela := len(relas) > 0
|
|
||||||
nSections := 6
|
|
||||||
if hasRela {
|
|
||||||
nSections = 7
|
|
||||||
}
|
|
||||||
secSymtab, secStrtab := 3, 4
|
|
||||||
secShstr := nSections - 1
|
|
||||||
|
|
||||||
// Layout.
|
// Layout.
|
||||||
var out []byte
|
var out []byte
|
||||||
out = append(out, make([]byte, 64)...)
|
out = append(out, make([]byte, 64)...)
|
||||||
@@ -166,7 +215,7 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
strtabOff := len(out)
|
strtabOff := len(out)
|
||||||
out = append(out, stNames.bytes()...)
|
out = append(out, stNames.bytes()...)
|
||||||
|
|
||||||
var relaOff int
|
var relaOff, relaDataOff int
|
||||||
if hasRela {
|
if hasRela {
|
||||||
align(8)
|
align(8)
|
||||||
relaOff = len(out)
|
relaOff = len(out)
|
||||||
@@ -178,6 +227,17 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
out = append(out, b[:]...)
|
out = append(out, b[:]...)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
align(8)
|
||||||
|
relaDataOff = len(out)
|
||||||
|
for _, r := range dataRelas {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint64(b[0:], r.off)
|
||||||
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||||
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
shstrOff := len(out)
|
shstrOff := len(out)
|
||||||
out = append(out, stSections.bytes()...)
|
out = append(out, stSections.bytes()...)
|
||||||
@@ -187,9 +247,25 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
out = append(out, 0)
|
out = append(out, 0)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
|
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiLOONG64)
|
||||||
|
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
nSections += 4
|
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
|
||||||
|
// .debug_line_str and .debug_frame (the CIE is unconditional, so
|
||||||
|
// the frame section is always present), plus the relocation
|
||||||
|
// sections below when they carry entries.
|
||||||
|
dwarfStart = nSections
|
||||||
|
nSections += 5
|
||||||
|
appendDWARFRelas(&out, dw, rLarchAbs64, dwAlign)
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
align(8)
|
align(8)
|
||||||
@@ -217,14 +293,41 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
if hasRela {
|
if hasRela {
|
||||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||||
|
}
|
||||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
|
// DWARF section headers; their indices follow the write order.
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
|
// secIdx is a running section index: each putSh below emits the
|
||||||
|
// next header, and the sh_info of a .rela section names the index
|
||||||
|
// of the section it relocates.
|
||||||
|
secIdx := dwarfStart
|
||||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||||
|
secInfoIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||||
|
secLineIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
if dw.frameSize > 0 {
|
if dw.frameSize > 0 {
|
||||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||||
|
secFrameIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -237,7 +340,7 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
le.PutUint64(hdr[24:], 0)
|
le.PutUint64(hdr[24:], 0)
|
||||||
le.PutUint64(hdr[32:], 0)
|
le.PutUint64(hdr[32:], 0)
|
||||||
le.PutUint64(hdr[40:], uint64(shoff))
|
le.PutUint64(hdr[40:], uint64(shoff))
|
||||||
le.PutUint32(hdr[48:], 0)
|
le.PutUint32(hdr[48:], efLarchAbiDoubleObjV1)
|
||||||
le.PutUint16(hdr[52:], 64)
|
le.PutUint16(hdr[52:], 64)
|
||||||
le.PutUint16(hdr[54:], 0)
|
le.PutUint16(hdr[54:], 0)
|
||||||
le.PutUint16(hdr[56:], 0)
|
le.PutUint16(hdr[56:], 0)
|
||||||
|
|||||||
+121
-1
@@ -9,7 +9,7 @@ import (
|
|||||||
"encoding/binary"
|
"encoding/binary"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
|
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
|
||||||
@@ -55,6 +55,11 @@ DATA answer<>+0(SB)/8, $42
|
|||||||
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_LOONGARCH {
|
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_LOONGARCH {
|
||||||
t.Errorf("type/machine = %v/%v, want ET_REL/EM_LOONGARCH", ef.Type, ef.Machine)
|
t.Errorf("type/machine = %v/%v, want ET_REL/EM_LOONGARCH", ef.Type, ef.Machine)
|
||||||
}
|
}
|
||||||
|
// The double-float ABI plus OBJABI_V1 flags the Go toolchain writes;
|
||||||
|
// system linkers refuse ABI-mismatched merges.
|
||||||
|
if flags := binary.LittleEndian.Uint32(obj[48:]); flags != efLarchAbiDoubleObjV1 {
|
||||||
|
t.Errorf("e_flags = %#x, want %#x (double-float, OBJABI_V1)", flags, efLarchAbiDoubleObjV1)
|
||||||
|
}
|
||||||
|
|
||||||
text := ef.Section(".text")
|
text := ef.Section(".text")
|
||||||
data := ef.Section(".data")
|
data := ef.Section(".data")
|
||||||
@@ -240,3 +245,118 @@ func TestELFLOONG64BranchRelocation(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestELFLOONG64ObjectDataRelocation checks that a symbol-valued DATA field
|
||||||
|
// ("DATA s+0(SB)/8, $other(SB)") reaches the LoongArch ELF object as a
|
||||||
|
// .rela.data entry: an R_LARCH_64 (R_LARCH_32 for a width-4 field) at the
|
||||||
|
// field's offset within .data, against the named symbol, external targets
|
||||||
|
// included.
|
||||||
|
func TestELFLOONG64ObjectDataRelocation(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("t_loong64.s", `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
GLOBL holder(SB), NOPTR, $32
|
||||||
|
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||||
|
DATA holder+8(SB)/8, $holder(SB)
|
||||||
|
DATA holder+16(SB)/8, $extvar(SB)
|
||||||
|
DATA holder+24(SB)/4, $Keep(SB)
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileLOONG64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFLOONG64Object()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFLOONG64Object: %v", err)
|
||||||
|
}
|
||||||
|
checkELFSectionAccounting(t, obj)
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
relaData := ef.Section(".rela.data")
|
||||||
|
if relaData == nil {
|
||||||
|
t.Fatal("missing .rela.data section")
|
||||||
|
}
|
||||||
|
if relaData.Type != elf.SHT_RELA {
|
||||||
|
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
|
||||||
|
}
|
||||||
|
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||||
|
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||||
|
}
|
||||||
|
if ef.Sections[relaData.Info].Name != ".data" {
|
||||||
|
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||||
|
}
|
||||||
|
relas, err := relaData.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var got []struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}
|
||||||
|
for i := 0; i+24 <= len(relas); i += 24 {
|
||||||
|
got = append(got, struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}{
|
||||||
|
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||||
|
// r_info packs the type in the low dword and the symbol index
|
||||||
|
// in the high dword.
|
||||||
|
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||||
|
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||||
|
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
name := func(idx uint32) string {
|
||||||
|
if idx >= 1 && int(idx) <= len(syms) {
|
||||||
|
return syms[idx-1].Name
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
// The offsets are data-section-relative: the field's DATA offset plus
|
||||||
|
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||||
|
base := uint64(0)
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
if d.Name == "holder" {
|
||||||
|
base = uint64(d.Offset)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
want := []struct {
|
||||||
|
off uint64
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
target string
|
||||||
|
}{
|
||||||
|
{off: base + 0, typ: uint32(elf.R_LARCH_64), addend: 5, target: "Keep"},
|
||||||
|
{off: base + 8, typ: uint32(elf.R_LARCH_64), addend: 0, target: "holder"},
|
||||||
|
{off: base + 16, typ: uint32(elf.R_LARCH_64), addend: 0, target: "extvar"},
|
||||||
|
{off: base + 24, typ: uint32(elf.R_LARCH_32), addend: 0, target: "Keep"},
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
g := got[i]
|
||||||
|
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||||
|
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||||
|
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||||
|
}
|
||||||
|
if n := name(g.sym); n != w.target {
|
||||||
|
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+128
-21
@@ -13,17 +13,27 @@ import (
|
|||||||
const (
|
const (
|
||||||
emRISCV = 243 // EM_RISCV
|
emRISCV = 243 // EM_RISCV
|
||||||
|
|
||||||
|
// EF_RISCV_FLOAT_ABI_DOUBLE: the double-precision float ABI the Go
|
||||||
|
// toolchain targets (cmd/link/internal/ld/elf.go writes Flags = 0x4 for
|
||||||
|
// RISCV64). System linkers refuse to merge ET_REL objects whose float
|
||||||
|
// ABI differs, so 0 (soft-float) would make the object unlinkable.
|
||||||
|
efRISCVFloatAbiDouble = 0x4
|
||||||
|
|
||||||
// RISC-V relocation types.
|
// RISC-V relocation types.
|
||||||
rRISCV32 = 1
|
|
||||||
rRISCVJAL = 17 // R_RISCV_JAL
|
rRISCVJAL = 17 // R_RISCV_JAL
|
||||||
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
|
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
|
||||||
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
|
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
|
||||||
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
|
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
|
||||||
|
// R_RISCV_32 (debug/elf 1): the absolute 32-bit address of a symbol,
|
||||||
|
// the R_ADDR shape a 4-byte DATA field carries. R_RISCV_64 (2) lives
|
||||||
|
// with the DWARF fixup constants as rRISCVAbs64.
|
||||||
|
rRISVCAbs32 = 1
|
||||||
)
|
)
|
||||||
|
|
||||||
// ELFRISCVObject returns the image as an ELF64 relocatable object file for
|
// ELFRISCVObject returns the image as an ELF64 relocatable object file for
|
||||||
// RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64
|
// RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64
|
||||||
// ELF emission: .text, .data, .symtab, .strtab and optional .rela.text.
|
// ELF emission: .text, .data, .symtab, .strtab, an optional .rela.text and
|
||||||
|
// an optional .rela.data.
|
||||||
func (img *Image) ELFRISCVObject() ([]byte, error) {
|
func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||||
le := binary.LittleEndian
|
le := binary.LittleEndian
|
||||||
|
|
||||||
@@ -83,9 +93,14 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
// Build relocations. Each SB reference is an AUIPC + second-instruction
|
// Build relocations. Each SB reference is an AUIPC + second-instruction
|
||||||
// pair carrying a single relocation kind; the ELF writer expands it into
|
// pair carrying a single relocation kind; the ELF writer expands it into
|
||||||
// the R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I/S pair the psABI expects.
|
// the R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I/S pair the psABI expects.
|
||||||
// The HI20 carries the symbol addend; the LO12 addend is zero, matching
|
// The HI20 carries the symbol and its addend. The LO12's symbol must
|
||||||
// cmd/link's own ELF conversion (the LO12 resolves against the HI20's
|
// denote the AUIPC site the HI20 relocates (psABI §8.4.9: the pair is
|
||||||
// AUIPC location).
|
// resolved against the label of the AUIPC, not the target symbol;
|
||||||
|
// cmd/link generates one local text symbol per AUIPC for exactly this,
|
||||||
|
// cmd/link/internal/riscv64/asm.go). The .text section symbol with the
|
||||||
|
// AUIPC's section-relative offset as addend gives S + A = the AUIPC
|
||||||
|
// address, which is that label.
|
||||||
|
const secSymText = 1 // syms[1], the .text section symbol
|
||||||
type elfRela struct {
|
type elfRela struct {
|
||||||
off uint64
|
off uint64
|
||||||
typ uint32
|
typ uint32
|
||||||
@@ -99,27 +114,70 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||||
}
|
}
|
||||||
|
auipc := int64(fn.Offset + r.Off)
|
||||||
switch r.Kind {
|
switch r.Kind {
|
||||||
case RelRISCVPCRELIType:
|
case RelRISCVPCRELIType:
|
||||||
relas = append(relas,
|
relas = append(relas,
|
||||||
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
||||||
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: idx, addend: 0},
|
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: secSymText, addend: auipc},
|
||||||
)
|
)
|
||||||
case RelRISCVPCRELSType:
|
case RelRISCVPCRELSType:
|
||||||
relas = append(relas,
|
relas = append(relas,
|
||||||
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
||||||
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: idx, addend: 0},
|
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: secSymText, addend: auipc},
|
||||||
)
|
)
|
||||||
case RelRISCVJal:
|
case RelRISCVJal:
|
||||||
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVJAL, sym: idx, addend: r.Addend})
|
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVJAL, sym: idx, addend: r.Addend})
|
||||||
case RelPCRelAbs:
|
|
||||||
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCV32, sym: idx, addend: r.Addend})
|
|
||||||
default:
|
default:
|
||||||
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
|
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||||
|
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||||
|
// DATA line's width at the field's data-section offset, S + A with no
|
||||||
|
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||||
|
// cannot hold an address, so they are refused rather than truncated.
|
||||||
|
var dataRelas []elfRela
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
for _, r := range d.Relocs {
|
||||||
|
idx, ok := symIdx[r.Name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||||
|
}
|
||||||
|
var typ uint32
|
||||||
|
switch r.Siz {
|
||||||
|
case 8:
|
||||||
|
typ = rRISCVAbs64
|
||||||
|
case 4:
|
||||||
|
typ = rRISVCAbs32
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||||
|
}
|
||||||
|
dataRelas = append(dataRelas, elfRela{
|
||||||
|
off: uint64(d.Offset + r.Off),
|
||||||
|
sym: idx,
|
||||||
|
typ: typ,
|
||||||
|
addend: r.Addend,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Section presence: .rela.text only when there are code relocations,
|
||||||
|
// .rela.data only when a DATA line holds a symbol value.
|
||||||
|
hasRela := len(relas) > 0
|
||||||
|
hasDataRela := len(dataRelas) > 0
|
||||||
|
nSections := 6
|
||||||
|
if hasRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
secSymtab, secStrtab := 3, 4
|
||||||
|
secShstr := nSections - 1
|
||||||
|
|
||||||
// String tables.
|
// String tables.
|
||||||
stNames := newElfStrtab()
|
stNames := newElfStrtab()
|
||||||
for _, s := range syms {
|
for _, s := range syms {
|
||||||
@@ -129,18 +187,13 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
stSections.add(".rela.data")
|
||||||
|
}
|
||||||
for _, n := range dwarfSectionNames {
|
for _, n := range dwarfSectionNames {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
|
||||||
hasRela := len(relas) > 0
|
|
||||||
nSections := 6
|
|
||||||
if hasRela {
|
|
||||||
nSections = 7
|
|
||||||
}
|
|
||||||
secSymtab, secStrtab := 3, 4
|
|
||||||
secShstr := nSections - 1
|
|
||||||
|
|
||||||
// Layout.
|
// Layout.
|
||||||
var out []byte
|
var out []byte
|
||||||
out = append(out, make([]byte, 64)...)
|
out = append(out, make([]byte, 64)...)
|
||||||
@@ -175,7 +228,7 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
strtabOff := len(out)
|
strtabOff := len(out)
|
||||||
out = append(out, stNames.bytes()...)
|
out = append(out, stNames.bytes()...)
|
||||||
|
|
||||||
var relaOff int
|
var relaOff, relaDataOff int
|
||||||
if hasRela {
|
if hasRela {
|
||||||
align(8)
|
align(8)
|
||||||
relaOff = len(out)
|
relaOff = len(out)
|
||||||
@@ -187,6 +240,17 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
out = append(out, b[:]...)
|
out = append(out, b[:]...)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
align(8)
|
||||||
|
relaDataOff = len(out)
|
||||||
|
for _, r := range dataRelas {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint64(b[0:], r.off)
|
||||||
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||||
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
shstrOff := len(out)
|
shstrOff := len(out)
|
||||||
out = append(out, stSections.bytes()...)
|
out = append(out, stSections.bytes()...)
|
||||||
@@ -196,9 +260,25 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
out = append(out, 0)
|
out = append(out, 0)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
|
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiRISCV64)
|
||||||
|
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
nSections += 4
|
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
|
||||||
|
// .debug_line_str and .debug_frame (the CIE is unconditional, so
|
||||||
|
// the frame section is always present), plus the relocation
|
||||||
|
// sections below when they carry entries.
|
||||||
|
dwarfStart = nSections
|
||||||
|
nSections += 5
|
||||||
|
appendDWARFRelas(&out, dw, rRISCVAbs64, dwAlign)
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
align(8)
|
align(8)
|
||||||
@@ -226,14 +306,41 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
if hasRela {
|
if hasRela {
|
||||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||||
|
}
|
||||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
|
// DWARF section headers; their indices follow the write order.
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
|
// secIdx is a running section index: each putSh below emits the
|
||||||
|
// next header, and the sh_info of a .rela section names the index
|
||||||
|
// of the section it relocates.
|
||||||
|
secIdx := dwarfStart
|
||||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||||
|
secInfoIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.infoRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||||
|
secLineIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.lineRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
|
||||||
|
secIdx++
|
||||||
|
}
|
||||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||||
|
secIdx++
|
||||||
if dw.frameSize > 0 {
|
if dw.frameSize > 0 {
|
||||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||||
|
secFrameIdx := secIdx
|
||||||
|
secIdx++
|
||||||
|
if dw.frameRelaCount > 0 {
|
||||||
|
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -246,7 +353,7 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
le.PutUint64(hdr[24:], 0)
|
le.PutUint64(hdr[24:], 0)
|
||||||
le.PutUint64(hdr[32:], 0)
|
le.PutUint64(hdr[32:], 0)
|
||||||
le.PutUint64(hdr[40:], uint64(shoff))
|
le.PutUint64(hdr[40:], uint64(shoff))
|
||||||
le.PutUint32(hdr[48:], 0)
|
le.PutUint32(hdr[48:], efRISCVFloatAbiDouble)
|
||||||
le.PutUint16(hdr[52:], 64)
|
le.PutUint16(hdr[52:], 64)
|
||||||
le.PutUint16(hdr[54:], 0)
|
le.PutUint16(hdr[54:], 0)
|
||||||
le.PutUint16(hdr[56:], 0)
|
le.PutUint16(hdr[56:], 0)
|
||||||
|
|||||||
@@ -0,0 +1,128 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"debug/elf"
|
||||||
|
"encoding/binary"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestELFRISCVObjectDataRelocation checks that a symbol-valued DATA field
|
||||||
|
// ("DATA s+0(SB)/8, $other(SB)") reaches the RISC-V ELF object as a
|
||||||
|
// .rela.data entry: an R_RISCV_64 (R_RISCV_32 for a width-4 field) at the
|
||||||
|
// field's offset within .data, against the named symbol, external targets
|
||||||
|
// included.
|
||||||
|
func TestELFRISCVObjectDataRelocation(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("t_riscv64.s", `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
GLOBL holder(SB), NOPTR, $32
|
||||||
|
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||||
|
DATA holder+8(SB)/8, $holder(SB)
|
||||||
|
DATA holder+16(SB)/8, $extvar(SB)
|
||||||
|
DATA holder+24(SB)/4, $Keep(SB)
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFRISCVObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFRISCVObject: %v", err)
|
||||||
|
}
|
||||||
|
checkELFSectionAccounting(t, obj)
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
relaData := ef.Section(".rela.data")
|
||||||
|
if relaData == nil {
|
||||||
|
t.Fatal("missing .rela.data section")
|
||||||
|
}
|
||||||
|
if relaData.Type != elf.SHT_RELA {
|
||||||
|
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
|
||||||
|
}
|
||||||
|
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||||
|
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||||
|
}
|
||||||
|
if ef.Sections[relaData.Info].Name != ".data" {
|
||||||
|
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||||
|
}
|
||||||
|
relas, err := relaData.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var got []struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}
|
||||||
|
for i := 0; i+24 <= len(relas); i += 24 {
|
||||||
|
got = append(got, struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}{
|
||||||
|
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||||
|
// r_info packs the type in the low dword and the symbol index
|
||||||
|
// in the high dword.
|
||||||
|
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||||
|
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||||
|
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
name := func(idx uint32) string {
|
||||||
|
if idx >= 1 && int(idx) <= len(syms) {
|
||||||
|
return syms[idx-1].Name
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
// The offsets are data-section-relative: the field's DATA offset plus
|
||||||
|
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||||
|
base := uint64(0)
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
if d.Name == "holder" {
|
||||||
|
base = uint64(d.Offset)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
want := []struct {
|
||||||
|
off uint64
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
target string
|
||||||
|
}{
|
||||||
|
{off: base + 0, typ: uint32(elf.R_RISCV_64), addend: 5, target: "Keep"},
|
||||||
|
{off: base + 8, typ: uint32(elf.R_RISCV_64), addend: 0, target: "holder"},
|
||||||
|
{off: base + 16, typ: uint32(elf.R_RISCV_64), addend: 0, target: "extvar"},
|
||||||
|
{off: base + 24, typ: uint32(elf.R_RISCV_32), addend: 0, target: "Keep"},
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
g := got[i]
|
||||||
|
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||||
|
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||||
|
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||||
|
}
|
||||||
|
if n := name(g.sym); n != w.target {
|
||||||
|
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
+63
-11
@@ -18,7 +18,27 @@ func Encodable(mnemonic string) bool {
|
|||||||
|
|
||||||
// Fixed-name instructions (no size suffix).
|
// Fixed-name instructions (no size suffix).
|
||||||
switch upper {
|
switch upper {
|
||||||
case "RET", "NOP", "CALL", "JMP":
|
case "RET", "NOP", "CALL", "JMP",
|
||||||
|
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
|
||||||
|
// The SSE compare family sharing CMPSD's predicate-last shape, the
|
||||||
|
// far return with its stack pop, the loop family, the bank-crossing
|
||||||
|
// MMX moves and the one-operand system controls.
|
||||||
|
"CMPSS", "CMPPS", "CMPPD", "RETFL",
|
||||||
|
"LOOP", "LOOPE", "LOOPNE",
|
||||||
|
"MOVDQ2Q", "MOVQ2DQ",
|
||||||
|
"ENDBR64", "CLWB", "TPAUSE", "UMONITOR", "UMWAIT", "RDPID", "CLDEMOTE",
|
||||||
|
// The literal-data pseudo-ops, the accepted-and-ignored END and
|
||||||
|
// bookkeeping statements, and the SP adjust.
|
||||||
|
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sysUnaryTable[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sseStoreOnly[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := noOperandTable[upper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
if _, ok := condCode(upper); ok {
|
if _, ok := condCode(upper); ok {
|
||||||
@@ -31,15 +51,20 @@ func Encodable(mnemonic string) bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
||||||
base == "KMOVW" || base == "KMOVQ" {
|
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
// CMOV carries size then condition (CMOVLGT); SET carries the condition
|
// CMOV carries size then condition (CMOVLGT); SET carries the condition
|
||||||
// alone (SETNE).
|
// alone (SETNE). The size letter is checked exactly as encodeCmov does,
|
||||||
|
// so a spelling like CMOVBGT is not reported encodable when Encode
|
||||||
|
// would reject it.
|
||||||
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
|
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
|
||||||
if _, ok := jccMap[rest[1:]]; ok {
|
switch rest[0] {
|
||||||
return true
|
case 'W', 'L', 'Q':
|
||||||
|
if _, ok := jccMap[rest[1:]]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
|
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
|
||||||
@@ -48,13 +73,28 @@ func Encodable(mnemonic string) bool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Legacy SSE shuffles and packed binaries dispatch on the full name.
|
// Legacy SSE shuffles and packed binaries dispatch on the full name; so
|
||||||
|
// do the imm8-controlled instructions, the lane extracts and inserts and
|
||||||
|
// the packed integer shifts (their trailing width letters belong to the
|
||||||
|
// mnemonic).
|
||||||
if _, ok := sseShufTable[upper]; ok {
|
if _, ok := sseShufTable[upper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
if _, ok := sseBinTable[upper]; ok {
|
if _, ok := sseBinTable[upper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
if _, ok := sseImm3Table[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sseExtractTable[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sseInsertTable[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sseShiftImm[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
// The size-suffix split: retry the tables and the scalar switch on the
|
// The size-suffix split: retry the tables and the scalar switch on the
|
||||||
// base.
|
// base.
|
||||||
@@ -69,20 +109,32 @@ func Encodable(mnemonic string) bool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
switch base2 {
|
switch base2 {
|
||||||
case "MOV",
|
case "MOV", "MOVD",
|
||||||
"ADD", "SUB", "AND", "OR", "XOR", "CMP",
|
"ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB",
|
||||||
"TEST",
|
"TEST",
|
||||||
"LEA",
|
"LEA",
|
||||||
"INC", "DEC", "NEG", "NOT",
|
"INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV",
|
||||||
"SHL", "SHR", "SAR",
|
"SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR",
|
||||||
|
"BT", "BTS", "BTR", "BTC",
|
||||||
|
"XCHG", "CMPXCHG", "XADD", "CRC32", "ADCX", "ADOX",
|
||||||
|
"MOVS", "STOS",
|
||||||
"IMUL", "IMUL3",
|
"IMUL", "IMUL3",
|
||||||
"PUSH", "POP",
|
"PUSH", "POP",
|
||||||
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
|
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
|
||||||
"BSWAP",
|
"BSWAP",
|
||||||
"PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2",
|
"PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2",
|
||||||
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
|
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
|
||||||
|
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX",
|
||||||
"CVTSL2SD", "CVTSQ2SD",
|
"CVTSL2SD", "CVTSQ2SD",
|
||||||
"MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
"CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S",
|
||||||
|
"FMOVD",
|
||||||
|
"MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
// Full-name dispatches the size split would eat (a trailing width
|
||||||
|
// letter that is part of the mnemonic).
|
||||||
|
switch upper {
|
||||||
|
case "PMOVMSKB":
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
return false
|
return false
|
||||||
|
|||||||
+489
-16
@@ -5,6 +5,8 @@ package asm
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"math"
|
||||||
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -21,6 +23,39 @@ func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
|
|||||||
type enc struct {
|
type enc struct {
|
||||||
out []byte
|
out []byte
|
||||||
patches []encPatch // disp32 fields awaiting static-symbol resolution
|
patches []encPatch // disp32 fields awaiting static-symbol resolution
|
||||||
|
|
||||||
|
// FloatPool collects the pooled constants the floating-point
|
||||||
|
// immediates reference, in first-use order.
|
||||||
|
floatPool []floatPoolEntry
|
||||||
|
floatPoolSeen map[string]bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// floatPoolEntry is one pooled floating-point constant: the symbol name
|
||||||
|
// the emitted RIP-relative load refers to and its IEEE-754 bytes.
|
||||||
|
type floatPoolEntry struct {
|
||||||
|
name string
|
||||||
|
data []byte
|
||||||
|
}
|
||||||
|
|
||||||
|
// addFloatPool records a pooled constant, deduplicated by symbol name.
|
||||||
|
func (e *enc) addFloatPool(name string, bits uint64, width int) {
|
||||||
|
if e.floatPoolSeen == nil {
|
||||||
|
e.floatPoolSeen = map[string]bool{}
|
||||||
|
}
|
||||||
|
if e.floatPoolSeen[name] {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
e.floatPoolSeen[name] = true
|
||||||
|
data := make([]byte, width)
|
||||||
|
for i := range width {
|
||||||
|
data[i] = byte(bits >> (8 * i))
|
||||||
|
}
|
||||||
|
e.floatPool = append(e.floatPool, floatPoolEntry{name: name, data: data})
|
||||||
|
}
|
||||||
|
|
||||||
|
// floatPoolList returns the pooled constants in first-use order.
|
||||||
|
func (e *enc) floatPoolList() []floatPoolEntry {
|
||||||
|
return e.floatPool
|
||||||
}
|
}
|
||||||
|
|
||||||
// encPatch marks a 4-byte displacement field in enc.out that must receive the
|
// encPatch marks a 4-byte displacement field in enc.out that must receive the
|
||||||
@@ -29,6 +64,7 @@ type encPatch struct {
|
|||||||
off int
|
off int
|
||||||
name string
|
name string
|
||||||
addend int64
|
addend int64
|
||||||
|
tls bool // a TLS slot offset: the patch is R_TLSLE with no symbol
|
||||||
}
|
}
|
||||||
|
|
||||||
func (e *enc) encode(mnem string, ops []Operand) error {
|
func (e *enc) encode(mnem string, ops []Operand) error {
|
||||||
@@ -37,17 +73,149 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
// Fixed-name instructions (no size suffix).
|
// Fixed-name instructions (no size suffix).
|
||||||
switch {
|
switch {
|
||||||
case upper == "RET":
|
case upper == "RET":
|
||||||
|
// RET sym(SB), the absolute return: the toolchain encodes it as a
|
||||||
|
// tail jump, E9 rel32 with a call relocation against the symbol.
|
||||||
|
if len(ops) == 1 {
|
||||||
|
if m, ok := ops[0].(sbMem); ok {
|
||||||
|
return e.emit(&instr{opcode: []byte{0xE9}, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}})
|
||||||
|
}
|
||||||
|
return fmt.Errorf("RET: unsupported operand")
|
||||||
|
}
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("RET expects no operands, got %d", len(ops))
|
||||||
|
}
|
||||||
return e.encodeRet()
|
return e.encodeRet()
|
||||||
case upper == "NOP":
|
case upper == "NOP":
|
||||||
return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1})
|
// The toolchain consumes every NOP statement as a pseudo and emits
|
||||||
case upper == "CALL":
|
// nothing for it, operands included (a bare NOP, NOP AX and
|
||||||
return e.encodeJmpRel(ops, []byte{0xE8})
|
// NOP sym(SB) all vanish from the object).
|
||||||
case upper == "JMP":
|
return nil
|
||||||
return e.encodeJmpRel(ops, []byte{0xE9})
|
case upper == "CALL" || upper == "JMP":
|
||||||
|
// Through a register or memory: FF /2 (CALL) or FF /4 (JMP).
|
||||||
|
// Anything else is a rel32 against a label resolved by the assembler.
|
||||||
|
if len(ops) == 1 {
|
||||||
|
switch ops[0].(type) {
|
||||||
|
case Reg, Mem:
|
||||||
|
return e.encodeIndirectBranch(upper, ops)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
opcode := []byte{0xE8}
|
||||||
|
if upper == "JMP" {
|
||||||
|
opcode = []byte{0xE9}
|
||||||
|
}
|
||||||
|
return e.encodeJmpRel(ops, opcode)
|
||||||
}
|
}
|
||||||
if cc, ok := condCode(upper); ok {
|
if cc, ok := condCode(upper); ok {
|
||||||
return e.encodeJcc(cc, ops)
|
return e.encodeJcc(cc, ops)
|
||||||
}
|
}
|
||||||
|
// No-operand system and string-control instructions (CPUID, RDTSC,
|
||||||
|
// SYSCALL, the fences, UNDEF, …).
|
||||||
|
if op, ok := noOperandTable[upper]; ok {
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("%s takes no operands, got %d", upper, len(ops))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
|
||||||
|
}
|
||||||
|
// One-operand system instructions whose reg field is a fixed digit:
|
||||||
|
// the cache and wait controls under 0F AE/0F 1C and the RDPID read.
|
||||||
|
if m, ok := sysUnaryTable[upper]; ok {
|
||||||
|
return e.encodeSysUnary(upper, m, ops)
|
||||||
|
}
|
||||||
|
// The store-only SSE moves (the non-temporal store).
|
||||||
|
if m, ok := sseStoreOnly[upper]; ok {
|
||||||
|
return e.encodeSSEStoreOnly(upper, m, ops)
|
||||||
|
}
|
||||||
|
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
|
||||||
|
// are rejected by go tool asm in 64-bit mode, so they stay unsupported.
|
||||||
|
switch upper {
|
||||||
|
case "POPFQ":
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("POPFQ takes no operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: []byte{0x9D}, modrm: -1, sib: -1})
|
||||||
|
case "PUSHFQ":
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("PUSHFQ takes no operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
|
||||||
|
case "INT":
|
||||||
|
return e.encodeInt(ops)
|
||||||
|
// The LOOP family outside the assembler's label settlement: the operand
|
||||||
|
// is the already-computed rel8 (E0-E2).
|
||||||
|
case "LOOP", "LOOPE", "LOOPNE":
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("%s expects 1 operand, got %d", upper, len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok || !fits8(int64(imm)) {
|
||||||
|
return fmt.Errorf("%s: relative offset must be a signed byte", upper)
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: []byte{loopOpcode(upper)}, modrm: -1, sib: -1, imm: []byte{byte(int8(imm))}})
|
||||||
|
case "LDMXCSR":
|
||||||
|
return e.encodeMxcsr(2, ops)
|
||||||
|
case "STMXCSR":
|
||||||
|
return e.encodeMxcsr(3, ops)
|
||||||
|
// CMPSD is the scalar double compare, whose predicate immediate comes
|
||||||
|
// LAST in Plan 9 order (src, dst, $imm); the family shares the shape.
|
||||||
|
case "CMPSD":
|
||||||
|
return e.encodeSSECmp("CMPSD", 0xF2, ops)
|
||||||
|
case "CMPSS":
|
||||||
|
return e.encodeSSECmp("CMPSS", 0xF3, ops)
|
||||||
|
case "CMPPS":
|
||||||
|
return e.encodeSSECmp("CMPPS", 0x00, ops)
|
||||||
|
case "CMPPD":
|
||||||
|
return e.encodeSSECmp("CMPPD", 0x66, ops)
|
||||||
|
// RETFL pops the immediate's worth of bytes after the far return
|
||||||
|
// (LRET iw: CA imm16), the toolchain's RETF spelling with a stack
|
||||||
|
// adjustment.
|
||||||
|
case "RETFL":
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("RETFL expects 1 operand, got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("RETFL expects an immediate")
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: []byte{0xCA}, modrm: -1, sib: -1, imm: le16(int64(imm))})
|
||||||
|
// MOVDQ2Q/MOVQ2DQ cross the MMX and XMM banks (F2 0F D6), the register
|
||||||
|
// in the reg field, the other bank's in r/m.
|
||||||
|
case "MOVDQ2Q", "MOVQ2DQ":
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||||
|
}
|
||||||
|
srcReg, ok1 := ops[0].(Reg)
|
||||||
|
dstReg, ok2 := ops[1].(Reg)
|
||||||
|
if !ok1 || !ok2 {
|
||||||
|
return fmt.Errorf("%s takes register operands alone", upper)
|
||||||
|
}
|
||||||
|
if upper == "MOVDQ2Q" && (!srcReg.isVec() || !dstReg.mmx) ||
|
||||||
|
upper == "MOVQ2DQ" && (!srcReg.mmx || !dstReg.isVec()) {
|
||||||
|
return fmt.Errorf("%s crosses the XMM and MMX banks in that order", upper)
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dstReg, srcReg, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
|
||||||
|
case "SHA256RNDS2":
|
||||||
|
return e.encodeSha256rnds2(ops)
|
||||||
|
// BYTE, WORD, LONG and QUAD write the immediate into the text stream
|
||||||
|
// itself: 1, 2, 4 or 8 literal bytes, little-endian. END is accepted
|
||||||
|
// and ignored. ADJSP adjusts SP by the immediate, sign-chosen between
|
||||||
|
// the SUBQ and ADDQ forms.
|
||||||
|
case "BYTE", "WORD", "LONG", "QUAD":
|
||||||
|
return e.encodeData(upper, ops)
|
||||||
|
case "END":
|
||||||
|
return e.encodeEnd(ops)
|
||||||
|
case "ADJSP":
|
||||||
|
return e.encodeAdjsp(ops)
|
||||||
|
// The runtime's bookkeeping statements carry no text bytes: go tool asm
|
||||||
|
// records FUNCDATA and PCDATA in the program list only, so the encoded
|
||||||
|
// body shows nothing, on every architecture.
|
||||||
|
case "FUNCDATA", "PCDATA":
|
||||||
|
return e.encodeFuncdata(upper, ops)
|
||||||
|
}
|
||||||
|
|
||||||
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
||||||
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
|
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
|
||||||
@@ -57,7 +225,9 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
|
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
||||||
|
isEvexPrefGather(base) ||
|
||||||
|
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
|
||||||
return e.encodeVec(base, ops, sfx)
|
return e.encodeVec(base, ops, sfx)
|
||||||
}
|
}
|
||||||
if sfx.any() {
|
if sfx.any() {
|
||||||
@@ -84,43 +254,100 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
}
|
}
|
||||||
// Legacy SSE packed binaries dispatch on the full name: the packed
|
// Legacy SSE packed binaries dispatch on the full name: the packed
|
||||||
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
|
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
|
||||||
// which the size split must not eat.
|
// which the size split must not eat. A floating-point immediate
|
||||||
|
// rewrites into a pooled-constant read on the scalar members.
|
||||||
if m, ok := sseBinTable[upper]; ok {
|
if m, ok := sseBinTable[upper]; ok {
|
||||||
|
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||||
|
return e.encodeSSEFloatBin(upper, m, f, ops)
|
||||||
|
}
|
||||||
return e.encodeSSEBin(m, ops)
|
return e.encodeSSEBin(m, ops)
|
||||||
}
|
}
|
||||||
if m, ok := sseBinTable[base]; ok {
|
if m, ok := sseBinTable[base]; ok {
|
||||||
|
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||||
|
return e.encodeSSEFloatBin(upper, m, f, ops)
|
||||||
|
}
|
||||||
return e.encodeSSEBin(m, ops)
|
return e.encodeSSEBin(m, ops)
|
||||||
}
|
}
|
||||||
|
// The imm8-controlled legacy instructions, the lane extracts and inserts
|
||||||
|
// and the packed integer shifts all dispatch on the full name: a trailing
|
||||||
|
// width letter here belongs to the mnemonic, not to the size split.
|
||||||
|
if m, ok := sseImm3Table[upper]; ok {
|
||||||
|
return e.encodeSSEImm3(m, ops)
|
||||||
|
}
|
||||||
|
if m, ok := sseExtractTable[upper]; ok {
|
||||||
|
return e.encodeSSEExtract(m, ops)
|
||||||
|
}
|
||||||
|
if m, ok := sseInsertTable[upper]; ok {
|
||||||
|
return e.encodeSSEInsert(m, ops)
|
||||||
|
}
|
||||||
|
if _, ok := sseShiftImm[upper]; ok {
|
||||||
|
return e.encodeSSEShift(upper, ops)
|
||||||
|
}
|
||||||
|
// PMOVMSKB ends in a width letter the size split would eat, so it
|
||||||
|
// dispatches on the full name like the packed binaries above.
|
||||||
|
if upper == "PMOVMSKB" {
|
||||||
|
return e.encodePmovmskb(upper, ops)
|
||||||
|
}
|
||||||
switch base {
|
switch base {
|
||||||
case "MOV":
|
case "MOV":
|
||||||
return e.encodeMov(ops, size)
|
return e.encodeMov(ops, size)
|
||||||
case "ADD", "SUB", "AND", "OR", "XOR", "CMP":
|
// MOVD is the Go assembler's alias of MOVQ: the same byte forms, 64-bit
|
||||||
|
// REX.W and all.
|
||||||
|
case "MOVD":
|
||||||
|
return e.encodeMov(ops, 8)
|
||||||
|
case "ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB":
|
||||||
return e.encodeALU(aluOp[base], ops, size)
|
return e.encodeALU(aluOp[base], ops, size)
|
||||||
case "TEST":
|
case "TEST":
|
||||||
return e.encodeTest(ops, size)
|
return e.encodeTest(ops, size)
|
||||||
case "LEA":
|
case "LEA":
|
||||||
return e.encodeLea(ops, size)
|
return e.encodeLea(ops, size)
|
||||||
case "INC", "DEC", "NEG", "NOT":
|
case "INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV":
|
||||||
return e.encodeUnary(unaryOp[base], ops, size)
|
return e.encodeUnary(unaryOp[base], ops, size)
|
||||||
case "SHL", "SHR", "SAR":
|
case "SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR":
|
||||||
return e.encodeShift(shiftOp[base], ops, size)
|
return e.encodeShift(base, ops, size)
|
||||||
|
case "BT", "BTS", "BTR", "BTC":
|
||||||
|
return e.encodeBitTest(base, ops, size)
|
||||||
|
case "XCHG":
|
||||||
|
return e.encodeExchange(ops, size)
|
||||||
|
case "CMPXCHG":
|
||||||
|
return e.encodeRegRegOp(0xB0, 0xB1, base, ops, size)
|
||||||
|
case "XADD":
|
||||||
|
return e.encodeRegRegOp(0xC0, 0xC1, base, ops, size)
|
||||||
|
case "CRC32":
|
||||||
|
return e.encodeCrc32(ops, size)
|
||||||
|
case "ADCX":
|
||||||
|
return e.encodeCarryExt(0x66, ops, size)
|
||||||
|
case "ADOX":
|
||||||
|
return e.encodeCarryExt(0xF3, ops, size)
|
||||||
|
case "MOVS", "STOS":
|
||||||
|
return e.encodeStringOp(base, ops, size)
|
||||||
case "IMUL", "IMUL3":
|
case "IMUL", "IMUL3":
|
||||||
return e.encodeImul(ops, size)
|
return e.encodeImul(ops, size)
|
||||||
case "PUSH":
|
case "PUSH":
|
||||||
return e.encodePushPop(ops, true)
|
return e.encodePushPop(ops, size, true)
|
||||||
case "POP":
|
case "POP":
|
||||||
return e.encodePushPop(ops, false)
|
return e.encodePushPop(ops, size, false)
|
||||||
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
|
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
|
||||||
return e.encodeCount(base, ops, size)
|
return e.encodeCount(base, ops, size)
|
||||||
case "BSWAP":
|
case "BSWAP":
|
||||||
return e.encodeBswap(ops, size)
|
return e.encodeBswap(ops, size)
|
||||||
case "PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2":
|
case "PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2":
|
||||||
return e.encodePrefetch(base, ops)
|
return e.encodePrefetch(base, ops)
|
||||||
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX":
|
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
|
||||||
|
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX":
|
||||||
return e.encodeMovExtend(base, ops)
|
return e.encodeMovExtend(base, ops)
|
||||||
case "CVTSL2SD", "CVTSQ2SD":
|
case "CVTSL2SD", "CVTSQ2SD":
|
||||||
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
|
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
|
||||||
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
case "CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S":
|
||||||
|
return e.encodeCvtInt(base, ops, size)
|
||||||
|
case "FMOVD":
|
||||||
|
return e.encodeFmov(ops)
|
||||||
|
case "MOVSD", "MOVSS":
|
||||||
|
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||||
|
return e.encodeSSEFloatMove(upper, f, ops)
|
||||||
|
}
|
||||||
|
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||||
|
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD":
|
||||||
return e.encodeSSEMove(sseMoveTable[base], ops)
|
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("unsupported instruction %q", mnem)
|
return fmt.Errorf("unsupported instruction %q", mnem)
|
||||||
@@ -150,6 +377,214 @@ var prefetchVariant = map[string]int{
|
|||||||
"PREFETCHT2": 3,
|
"PREFETCHT2": 3,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// dataWidth is the literal byte count of each data-emission pseudo-op.
|
||||||
|
var dataWidth = map[string]int{
|
||||||
|
"BYTE": 1,
|
||||||
|
"WORD": 2,
|
||||||
|
"LONG": 4,
|
||||||
|
"QUAD": 8,
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeData emits the literal-data pseudo-ops: BYTE, WORD, LONG and QUAD
|
||||||
|
// write the immediate into the text stream as 1, 2, 4 or 8 bytes,
|
||||||
|
// little-endian, with no opcode lookup. The value is truncated to the
|
||||||
|
// width rather than range-checked, exactly as go tool asm behaves (BYTE
|
||||||
|
// $0x1FF emits FF, WORD $0x12345 emits 45 23, both without an error), and
|
||||||
|
// exactly one immediate is accepted: the toolchain rejects a list such as
|
||||||
|
// BYTE $1, $2, $3.
|
||||||
|
func (e *enc) encodeData(mnem string, ops []Operand) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("%s expects 1 immediate operand, got %d", mnem, len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s requires an integer immediate", mnem)
|
||||||
|
}
|
||||||
|
width := dataWidth[mnem]
|
||||||
|
out := make([]byte, width)
|
||||||
|
u := uint64(imm)
|
||||||
|
for i := range width {
|
||||||
|
out[i] = byte(u >> (8 * i))
|
||||||
|
}
|
||||||
|
e.out = append(e.out, out...)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeFuncdata accepts-and-ignores the runtime bookkeeping statements:
|
||||||
|
// FUNCDATA $n, sym(SB) and PCDATA $n, $m. go tool asm emits no text bytes
|
||||||
|
// for either (the entries live in the object's ancillary tables, not the
|
||||||
|
// function body), and the operand shapes it takes are exactly these: an
|
||||||
|
// integer count first, then a symbol reference for FUNCDATA and an integer
|
||||||
|
// value for PCDATA. The other architectures accept-and-ignore the same
|
||||||
|
// statements; amd64 now matches.
|
||||||
|
func (e *enc) encodeFuncdata(upper string, ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||||
|
}
|
||||||
|
if _, ok := ops[0].(Imm); !ok {
|
||||||
|
return fmt.Errorf("%s: first operand must be an integer immediate", upper)
|
||||||
|
}
|
||||||
|
switch upper {
|
||||||
|
case "FUNCDATA":
|
||||||
|
if _, ok := ops[1].(sbMem); !ok {
|
||||||
|
return fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
|
||||||
|
}
|
||||||
|
case "PCDATA":
|
||||||
|
if _, ok := ops[1].(Imm); !ok {
|
||||||
|
return fmt.Errorf("PCDATA: second operand must be an integer immediate")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeEnd accepts-and-ignores END. go tool asm drops the statement
|
||||||
|
// entirely: the AEND Prog is skipped when the program list is flushed, so
|
||||||
|
// the statements after an END still belong to the same function and the
|
||||||
|
// encoded body carries no trace of it, whatever operands follow the name
|
||||||
|
// (the toolchain takes END $0 and END AX alike). Zero bytes, no effect.
|
||||||
|
func (e *enc) encodeEnd(ops []Operand) error {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeAdjsp emits ADJSP $imm: a positive value is SUBQ $imm, SP, a
|
||||||
|
// negative one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude
|
||||||
|
// picks (the same selection subSP and addSP make for the frame). go tool
|
||||||
|
// asm refuses ADJSP $0 outright, so a zero value is an error here too; the
|
||||||
|
// statement's effect on the SP balance is checked by the function-level
|
||||||
|
// assembly (checkAdjspBalance), as the toolchain's push/pop walk does.
|
||||||
|
func (e *enc) encodeAdjsp(ops []Operand) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("ADJSP expects 1 immediate operand, got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("ADJSP requires an integer immediate")
|
||||||
|
}
|
||||||
|
switch v := int(imm); {
|
||||||
|
case v > 0:
|
||||||
|
e.out = append(e.out, subSP(v)...)
|
||||||
|
case v < 0:
|
||||||
|
e.out = append(e.out, addSP(-v)...)
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("ADJSP $0 has no encoding")
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- floating-point immediates ----------------------------------------------
|
||||||
|
|
||||||
|
// sseFloatImm lists the mnemonics whose first operand may be a floating-point
|
||||||
|
// immediate, the set go tool asm rewrites into a pooled-constant read: the
|
||||||
|
// scalar moves, the four scalar arithmetic pairs and the scalar compares.
|
||||||
|
// The packed members and the uniform forms (MAXSD, MINSD, SQRTSD, CMPSD)
|
||||||
|
// reject the immediate in the toolchain and are absent here on purpose.
|
||||||
|
var sseFloatImm = map[string]bool{
|
||||||
|
"MOVSD": true, "MOVSS": true,
|
||||||
|
"ADDSD": true, "ADDSS": true,
|
||||||
|
"SUBSD": true, "SUBSS": true,
|
||||||
|
"MULSD": true, "MULSS": true,
|
||||||
|
"DIVSD": true, "DIVSS": true,
|
||||||
|
"COMISD": true, "COMISS": true,
|
||||||
|
"UCOMISD": true, "UCOMISS": true,
|
||||||
|
}
|
||||||
|
|
||||||
|
// floatImmOperand reports whether the operand list opens with a
|
||||||
|
// floating-point immediate in the two-operand spelling (imm, dst).
|
||||||
|
func floatImmOperand(ops []Operand) (FloatImm, bool) {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return FloatImm{}, false
|
||||||
|
}
|
||||||
|
f, ok := ops[0].(FloatImm)
|
||||||
|
return f, ok
|
||||||
|
}
|
||||||
|
|
||||||
|
// floatPoolValue evaluates a floating-point immediate at the width its
|
||||||
|
// mnemonic encodes and names the pool constant the toolchain synthesises:
|
||||||
|
// $f64.<16 hex> for the doubles, $f32.<8 hex> for the singles (the float32
|
||||||
|
// rounding of the parsed value). The name carries the IEEE-754 bits; the
|
||||||
|
// section holds them little-endian.
|
||||||
|
func floatPoolValue(mnem string, f FloatImm) (bits uint64, name string, err error) {
|
||||||
|
v, err := strconv.ParseFloat(f.Text, 64)
|
||||||
|
if err != nil {
|
||||||
|
return 0, "", fmt.Errorf("invalid floating-point immediate %q", f.Text)
|
||||||
|
}
|
||||||
|
if f.Neg {
|
||||||
|
v = -v
|
||||||
|
}
|
||||||
|
if strings.HasSuffix(mnem, "D") {
|
||||||
|
bits = math.Float64bits(v)
|
||||||
|
return bits, fmt.Sprintf("$f64.%016x", bits), nil
|
||||||
|
}
|
||||||
|
bits = uint64(math.Float32bits(float32(v)))
|
||||||
|
return bits, fmt.Sprintf("$f32.%08x", bits), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSSEFloatMove encodes MOVSD/MOVSS with a floating-point immediate
|
||||||
|
// source. A positive zero needs no memory read: the toolchain emits
|
||||||
|
// XORPS dst, dst. Anything else loads the pooled constant RIP-relative
|
||||||
|
// ($f64.<hex>(SB) / $f32.<hex>(SB)), the displacement a patch site the
|
||||||
|
// file-level layout or the linker resolves.
|
||||||
|
func (e *enc) encodeSSEFloatMove(mnem string, f FloatImm, ops []Operand) error {
|
||||||
|
if !sseFloatImm[mnem] {
|
||||||
|
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
|
||||||
|
}
|
||||||
|
dst, ok := ops[1].(Reg)
|
||||||
|
if !ok || !dst.isVec() {
|
||||||
|
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||||
|
}
|
||||||
|
bits, name, err := floatPoolValue(mnem, f)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
e.addFloatPool(name, bits, mwidth(mnem))
|
||||||
|
if bits == 0 {
|
||||||
|
i := &instr{opcode: []byte{0x0F, 0x57}, modrm: -1, sib: -1} // XORPS
|
||||||
|
if err := setRM(i, dst, dst, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
m := sseMoveTable[mnem]
|
||||||
|
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.load}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSSEFloatBin encodes the scalar arithmetic and compare mnemonics with
|
||||||
|
// a floating-point immediate source: the constant is read from the pool into
|
||||||
|
// the instruction's r/m side (reg = destination), the rewrite go tool asm
|
||||||
|
// performs at the source level.
|
||||||
|
func (e *enc) encodeSSEFloatBin(mnem string, m sseBin, f FloatImm, ops []Operand) error {
|
||||||
|
if !sseFloatImm[mnem] {
|
||||||
|
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
|
||||||
|
}
|
||||||
|
dst, ok := ops[1].(Reg)
|
||||||
|
if !ok || !dst.isVec() {
|
||||||
|
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||||
|
}
|
||||||
|
bits, name, err := floatPoolValue(mnem, f)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
e.addFloatPool(name, bits, mwidth(mnem))
|
||||||
|
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// mwidth returns the operand width a scalar SSE mnemonic encodes: the double
|
||||||
|
// spellings end in D, the single spellings in S.
|
||||||
|
func mwidth(mnem string) int {
|
||||||
|
if strings.HasSuffix(mnem, "D") {
|
||||||
|
return 8
|
||||||
|
}
|
||||||
|
return 4
|
||||||
|
}
|
||||||
|
|
||||||
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
|
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
|
||||||
func splitSize(upper string) (base string, size int) {
|
func splitSize(upper string) (base string, size int) {
|
||||||
if upper == "" {
|
if upper == "" {
|
||||||
@@ -179,7 +614,7 @@ func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
|
|||||||
if ss, ok := scatterTable[upper]; ok {
|
if ss, ok := scatterTable[upper]; ok {
|
||||||
return e.encodeScatter(upper, ss, ops, sfx)
|
return e.encodeScatter(upper, ss, ops, sfx)
|
||||||
}
|
}
|
||||||
if upper == "KMOVW" || upper == "KMOVQ" {
|
if upper == "KMOVW" || upper == "KMOVQ" || upper == "KMOVB" || upper == "KMOVD" {
|
||||||
if sfx.any() {
|
if sfx.any() {
|
||||||
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
||||||
}
|
}
|
||||||
@@ -216,6 +651,7 @@ type instr struct {
|
|||||||
disp []byte
|
disp []byte
|
||||||
imm []byte
|
imm []byte
|
||||||
sb *sbRef // static-symbol displacement in disp, awaiting resolution
|
sb *sbRef // static-symbol displacement in disp, awaiting resolution
|
||||||
|
tls bool // the displacement is a TLS slot offset, patched R_TLSLE
|
||||||
}
|
}
|
||||||
|
|
||||||
// sbRef records that an instruction's displacement refers to a static symbol
|
// sbRef records that an instruction's displacement refers to a static symbol
|
||||||
@@ -258,6 +694,9 @@ func (e *enc) emit(i *instr) error {
|
|||||||
if i.sb != nil {
|
if i.sb != nil {
|
||||||
e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend})
|
e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend})
|
||||||
}
|
}
|
||||||
|
if i.tls {
|
||||||
|
e.patches = append(e.patches, encPatch{off: len(e.out), tls: true})
|
||||||
|
}
|
||||||
e.out = append(e.out, i.disp...)
|
e.out = append(e.out, i.disp...)
|
||||||
e.out = append(e.out, i.imm...)
|
e.out = append(e.out, i.imm...)
|
||||||
return nil
|
return nil
|
||||||
@@ -312,12 +751,30 @@ func setRMReg(i *instr, regField int, rexR, regForced bool, rm Operand, opSize i
|
|||||||
i.disp = le32(0)
|
i.disp = le32(0)
|
||||||
i.sb = &sbRef{name: r.name, addend: r.addend}
|
i.sb = &sbRef{name: r.name, addend: r.addend}
|
||||||
return nil
|
return nil
|
||||||
|
case TLSMem:
|
||||||
|
// off(TLS): the segment-prefixed absolute access, mod=00 with the
|
||||||
|
// SIB escape's disp32 absolute form. The displacement is the TLS
|
||||||
|
// slot offset, patched by the linker's TLS relocation.
|
||||||
|
i.prefix = r.Seg
|
||||||
|
i.modrm = 0x04 | regField<<3
|
||||||
|
i.sib = 0x25
|
||||||
|
i.disp = le32(r.Disp)
|
||||||
|
i.tls = true
|
||||||
|
return nil
|
||||||
|
case SegAbs:
|
||||||
|
// 0x30(GS): the segment override with the SIB escape's disp32
|
||||||
|
// absolute form, no relocation.
|
||||||
|
setSegAbs(i, regField, r)
|
||||||
|
return nil
|
||||||
default:
|
default:
|
||||||
return fmt.Errorf("invalid r/m operand %T", rm)
|
return fmt.Errorf("invalid r/m operand %T", rm)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func setMem(i *instr, regField int, m Mem) error {
|
func setMem(i *instr, regField int, m Mem) error {
|
||||||
|
if m.Seg != 0 {
|
||||||
|
i.prefix = m.Seg
|
||||||
|
}
|
||||||
modrm, sib, disp, xBit, bBit, err := memComponents(regField, m)
|
modrm, sib, disp, xBit, bBit, err := memComponents(regField, m)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
@@ -330,11 +787,27 @@ func setMem(i *instr, regField int, m Mem) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// setSegAbs assembles a segment-absolute operand, 0x30(GS): the segment
|
||||||
|
// override with the mod=00 SIB escape's disp32 absolute form and no
|
||||||
|
// relocation.
|
||||||
|
func setSegAbs(i *instr, regField int, m SegAbs) {
|
||||||
|
i.prefix = m.Seg
|
||||||
|
i.modrm = 0x04 | regField<<3
|
||||||
|
i.sib = 0x25
|
||||||
|
i.disp = le32(m.Disp)
|
||||||
|
}
|
||||||
|
|
||||||
// memComponents computes the ModR/M byte (with the given reg field), the SIB
|
// memComponents computes the ModR/M byte (with the given reg field), the SIB
|
||||||
// byte (-1 if none), the displacement bytes, and the high index/base bits, for
|
// byte (-1 if none), the displacement bytes, and the high index/base bits, for
|
||||||
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
|
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
|
||||||
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
|
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
|
||||||
sib = -1
|
sib = -1
|
||||||
|
// A displacement wider than int32 fits no encoding form; truncating it
|
||||||
|
// would address a different location, and go tool asm reports "offset
|
||||||
|
// too large" for the same operand.
|
||||||
|
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
|
||||||
|
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
|
||||||
|
}
|
||||||
// RIP-relative: neither base nor index.
|
// RIP-relative: neither base nor index.
|
||||||
if !m.HasBase && !m.HasIndex {
|
if !m.HasBase && !m.HasIndex {
|
||||||
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
|
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
|
||||||
|
|||||||
+933
-3
File diff suppressed because it is too large
Load Diff
+709
-57
File diff suppressed because it is too large
Load Diff
+254
-13
@@ -16,7 +16,7 @@ import (
|
|||||||
// kernels use: NDS arithmetic, immediate and variable shifts, shuffles with
|
// kernels use: NDS arithmetic, immediate and variable shifts, shuffles with
|
||||||
// an immediate, lane extracts, narrowing stores, broadcasts from a GPR or
|
// an immediate, lane extracts, narrowing stores, broadcasts from a GPR or
|
||||||
// memory, mask destinations, mask moves, disp8×N compression and the 5-bit
|
// memory, mask destinations, mask moves, disp8×N compression and the 5-bit
|
||||||
// register fields (X/Y 16–31, Z 0–31).
|
// register fields (X/Y 16-31, Z 0-31).
|
||||||
func TestEvexGroundTruth(t *testing.T) {
|
func TestEvexGroundTruth(t *testing.T) {
|
||||||
cases := []struct {
|
cases := []struct {
|
||||||
name string
|
name string
|
||||||
@@ -38,6 +38,15 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
|
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
|
||||||
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
|
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
|
||||||
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
|
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
|
||||||
|
// The qword OR spelling always encodes through EVEX.
|
||||||
|
{"VPORQ Y0,Y1,Y2", "VPORQ", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "62f1f528ebd0"},
|
||||||
|
{"VPORQ X0,X1,X2", "VPORQ", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f1f508ebd0"},
|
||||||
|
// Byte permute and population count.
|
||||||
|
{"VPERMI2B X0,X1,X2", "VPERMI2B", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f2750875d0"},
|
||||||
|
{"VPOPCNTB X0,X1", "VPOPCNTB", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0854c8"},
|
||||||
|
{"VPOPCNTD X0,X1", "VPOPCNTD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0855c8"},
|
||||||
|
{"VPOPCNTD Y0,Y1", "VPOPCNTD", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "62f27d2855c8"},
|
||||||
|
{"VPOPCNTQ X0,X1", "VPOPCNTQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f2fd0855c8"},
|
||||||
// Align (NDS + imm8).
|
// Align (NDS + imm8).
|
||||||
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
|
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
|
||||||
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
|
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
|
||||||
@@ -52,13 +61,23 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
|
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
|
||||||
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
|
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
|
||||||
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
|
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
|
||||||
|
{"KMOVB K1,K2", "KMOVB", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f990d1"},
|
||||||
|
{"KMOVB AX,K1", "KMOVB", []Operand{AX, vreg(t, "K1")}, "c5f992c8"},
|
||||||
|
{"KMOVB K1,AX", "KMOVB", []Operand{vreg(t, "K1"), AX}, "c5f993c1"},
|
||||||
|
{"KMOVB K1,(AX)", "KMOVB", []Operand{vreg(t, "K1"), Ptr(AX, 0, 1)}, "c5f99108"},
|
||||||
|
{"KMOVD K1,K2", "KMOVD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f990d1"},
|
||||||
|
{"KMOVD AX,K1", "KMOVD", []Operand{AX, vreg(t, "K1")}, "c5fb92c8"},
|
||||||
|
{"KMOVD K1,AX", "KMOVD", []Operand{vreg(t, "K1"), AX}, "c5fb93c1"},
|
||||||
|
{"KMOVD K1,(AX)", "KMOVD", []Operand{vreg(t, "K1"), Ptr(AX, 0, 4)}, "c4e1f99108"},
|
||||||
|
{"KMOVB (AX),K1", "KMOVB", []Operand{Ptr(AX, 0, 1), vreg(t, "K1")}, "c5f99008"},
|
||||||
|
{"KMOVQ (AX),K1", "KMOVQ", []Operand{Ptr(AX, 0, 8), vreg(t, "K1")}, "c4e1f89008"},
|
||||||
// Moves, incl. disp8×N (64 for a 512-bit operand).
|
// Moves, incl. disp8×N (64 for a 512-bit operand).
|
||||||
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
|
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
|
||||||
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
|
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
|
||||||
{"VMOVDQU32 16(SI)(R15*4),Z4", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 16, 64), vreg(t, "Z4")}, "62b17e486fa4be10000000"},
|
{"VMOVDQU32 16(SI)(R15*4),Z4", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 16, 64), vreg(t, "Z4")}, "62b17e486fa4be10000000"},
|
||||||
{"VMOVDQU32 Z0,4(SI)(AX*1)", "VMOVDQU32", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f17e487f840604000000"},
|
{"VMOVDQU32 Z0,4(SI)(AX*1)", "VMOVDQU32", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f17e487f840604000000"},
|
||||||
{"VMOVDQU32 Z3,(DI)(R15*4)", "VMOVDQU32", []Operand{vreg(t, "Z3"), Idx(DI, vreg(t, "R15"), 4, 0, 64)}, "62b17e487f1cbf"},
|
{"VMOVDQU32 Z3,(DI)(R15*4)", "VMOVDQU32", []Operand{vreg(t, "Z3"), Idx(DI, vreg(t, "R15"), 4, 0, 64)}, "62b17e487f1cbf"},
|
||||||
// VMOVDQU64 — the W1 qword variant.
|
// VMOVDQU64; the W1 qword variant.
|
||||||
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
|
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
|
||||||
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
|
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
|
||||||
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
|
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
|
||||||
@@ -77,7 +96,7 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"VPSHUFB Z1,Z2,Z3", "VPSHUFB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4800d9"},
|
{"VPSHUFB Z1,Z2,Z3", "VPSHUFB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4800d9"},
|
||||||
{"VMOVDQU8 Z1,Z2", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17f487fca"},
|
{"VMOVDQU8 Z1,Z2", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17f487fca"},
|
||||||
{"VMOVDQU16 Z1,Z2", "VMOVDQU16", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff487fca"},
|
{"VMOVDQU16 Z1,Z2", "VMOVDQU16", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff487fca"},
|
||||||
// Indices 16–31: rm[4] rides in X̄ for register operands.
|
// Indices 16-31: rm[4] rides in X̄ for register operands.
|
||||||
{"VPSHUFD $1,X16,X17", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}, "62a17d0870c801"},
|
{"VPSHUFD $1,X16,X17", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}, "62a17d0870c801"},
|
||||||
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
|
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
|
||||||
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
|
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
|
||||||
@@ -96,7 +115,7 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"VPBROADCASTD 4(SI),Z10", "VPBROADCASTD", []Operand{Ptr(SI, 4, 4), vreg(t, "Z10")}, "62727d48585601"},
|
{"VPBROADCASTD 4(SI),Z10", "VPBROADCASTD", []Operand{Ptr(SI, 4, 4), vreg(t, "Z10")}, "62727d48585601"},
|
||||||
{"VPBROADCASTQ R8,X31", "VPBROADCASTQ", []Operand{vreg(t, "R8"), vreg(t, "X31")}, "6242fd087cf8"},
|
{"VPBROADCASTQ R8,X31", "VPBROADCASTQ", []Operand{vreg(t, "R8"), vreg(t, "X31")}, "6242fd087cf8"},
|
||||||
{"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"},
|
{"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"},
|
||||||
// Register indices 16–31 exist only in EVEX encodings.
|
// Register indices 16-31 exist only in EVEX encodings.
|
||||||
{"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"},
|
{"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"},
|
||||||
// Packed double arithmetic / unpack (EVEX forms carry W=1).
|
// Packed double arithmetic / unpack (EVEX forms carry W=1).
|
||||||
{"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"},
|
{"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"},
|
||||||
@@ -107,7 +126,7 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"},
|
{"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"},
|
||||||
{"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"},
|
{"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"},
|
||||||
{"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"},
|
{"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"},
|
||||||
// VMOVDDUP — duplicate the low double; disp8×N = 64 at 512 bits, and
|
// VMOVDDUP; duplicate the low double; disp8×N = 64 at 512 bits, and
|
||||||
// X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄).
|
// X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄).
|
||||||
{"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"},
|
{"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"},
|
||||||
{"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"},
|
{"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"},
|
||||||
@@ -150,7 +169,7 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestEvexMasking checks the AVX-512 mask operand (K1–K7, placed freely among
|
// TestEvexMasking checks the AVX-512 mask operand (K1-K7, placed freely among
|
||||||
// the operands) and the .Z zeroing suffix, byte for byte against the Go
|
// the operands) and the .Z zeroing suffix, byte for byte against the Go
|
||||||
// assembler.
|
// assembler.
|
||||||
func TestEvexMasking(t *testing.T) {
|
func TestEvexMasking(t *testing.T) {
|
||||||
@@ -241,11 +260,11 @@ func TestEvexMasking(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set — ternary
|
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set; ternary
|
||||||
// logic, lane shuffles/inserts/extracts, compares with a K destination,
|
// logic, lane shuffles/inserts/extracts, compares with a K destination,
|
||||||
// permutes, the wider integer families, expand/compress, broadcasts,
|
// permutes, the wider integer families, expand/compress, broadcasts,
|
||||||
// rotates and word shifts, the opmask instructions, the EVEX suffixes
|
// rotates and word shifts, the opmask instructions, the EVEX suffixes
|
||||||
// (rounding/SAE/broadcast) and the aligned/scalar moves — byte for byte
|
// (rounding/SAE/broadcast) and the aligned/scalar moves; byte for byte
|
||||||
// against the Go assembler.
|
// against the Go assembler.
|
||||||
func TestEvexExtendedGroundTruth(t *testing.T) {
|
func TestEvexExtendedGroundTruth(t *testing.T) {
|
||||||
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
|
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
|
||||||
@@ -275,7 +294,7 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
|
|||||||
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
|
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
|
||||||
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
|
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
|
||||||
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
|
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
|
||||||
// Packed single arithmetic (same opcodes, no mandatory prefix) —
|
// Packed single arithmetic (same opcodes, no mandatory prefix);
|
||||||
// ZMM, YMM and XMM widths, rounding and broadcast.
|
// ZMM, YMM and XMM widths, rounding and broadcast.
|
||||||
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
|
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
|
||||||
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
|
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
|
||||||
@@ -399,8 +418,8 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
|
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
|
||||||
// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef,
|
// tail of the EVEX set; reciprocals, rsqrt, getexp/getmant, scalef,
|
||||||
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions —
|
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions;
|
||||||
// plus gather/scatter with VSIB addressing, byte for byte against the Go
|
// plus gather/scatter with VSIB addressing, byte for byte against the Go
|
||||||
// assembler.
|
// assembler.
|
||||||
func TestEvexHelperGroundTruth(t *testing.T) {
|
func TestEvexHelperGroundTruth(t *testing.T) {
|
||||||
@@ -502,9 +521,9 @@ func TestEvexHelperGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestEvexGprGroundTruth covers the scalar conversions between vector and
|
// TestEvexGprGroundTruth covers the scalar conversions between vector and
|
||||||
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
|
// general-purpose registers; the signed and truncated VCVT{,T}S{D,S}2SI
|
||||||
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
|
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
|
||||||
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
|
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv; byte
|
||||||
// for byte against the Go assembler, including memory sources and extended
|
// for byte against the Go assembler, including memory sources and extended
|
||||||
// GPRs.
|
// GPRs.
|
||||||
func TestEvexGprGroundTruth(t *testing.T) {
|
func TestEvexGprGroundTruth(t *testing.T) {
|
||||||
@@ -675,6 +694,15 @@ func TestEvexErrors(t *testing.T) {
|
|||||||
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
|
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
|
||||||
// VEX-only mnemonics reject registers only EVEX can encode.
|
// VEX-only mnemonics reject registers only EVEX can encode.
|
||||||
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
|
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
|
||||||
|
// The scalar EVEX move matches its VEX twin and the Go assembler:
|
||||||
|
// XMM↔memory only, never reg-reg and never a wider register (the
|
||||||
|
// toolchain rejects every one of these shapes).
|
||||||
|
{"VMOVSS X1,X2", "VMOVSS", []Operand{vreg(t, "X1"), vreg(t, "X2")}},
|
||||||
|
{"VMOVSS X16,X2", "VMOVSS", []Operand{vreg(t, "X16"), vreg(t, "X2")}},
|
||||||
|
{"VMOVSS Y1,(AX)", "VMOVSS", []Operand{vreg(t, "Y1"), Ptr(AX, 0, 4)}},
|
||||||
|
{"VMOVSS Z1,Z2", "VMOVSS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}},
|
||||||
|
{"VMOVSS Z1,(AX)", "VMOVSS", []Operand{vreg(t, "Z1"), Ptr(AX, 0, 4)}},
|
||||||
|
{"VMOVSS (AX),Z2", "VMOVSS", []Operand{Ptr(AX, 0, 4), vreg(t, "Z2")}},
|
||||||
}
|
}
|
||||||
for _, c := range cases {
|
for _, c := range cases {
|
||||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||||
@@ -693,3 +721,216 @@ func hexCompact(b []byte) string {
|
|||||||
}
|
}
|
||||||
return string(out)
|
return string(out)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAvx512CorpusFamilies pins representative encodings of the AVX-512
|
||||||
|
// families the toolchain's avx512enc corpus exercises: the bytes are the
|
||||||
|
// go tool asm output for exactly these operands, and the same families are
|
||||||
|
// covered end to end by the avx512_amd64.s differential kernel.
|
||||||
|
func TestAvx512CorpusFamilies(t *testing.T) {
|
||||||
|
vsib := func(base, idx string, scale int) Operand {
|
||||||
|
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
|
||||||
|
}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
// AES rounds (EVEX NDS, VEX twin routed by operand width).
|
||||||
|
{"VAESDEC Z", "VAESDEC", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d48ded9"},
|
||||||
|
// Integer VNNI and the bit algorithm group.
|
||||||
|
{"VPDPBUSD", "VPDPBUSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f26d4a50d9"},
|
||||||
|
{"VPOPCNTW", "VPOPCNTW", []Operand{vreg(t, "Z1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f2fd4b54d1"},
|
||||||
|
{"VPCONFLICTD", "VPCONFLICTD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d49c4d1"},
|
||||||
|
{"VPLZCNTQ masked", "VPLZCNTQ", []Operand{vreg(t, "Z7"), vreg(t, "K1"), vreg(t, "Z8")}, "6272fd4944c7"},
|
||||||
|
{"VPERMT2B", "VPERMT2B", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f26d497dd9"},
|
||||||
|
{"VPMULTISHIFTQB", "VPMULTISHIFTQB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f2ed4b83e1"},
|
||||||
|
{"VDBPSADBW", "VDBPSADBW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z3")}, "62f36d4b42d903"},
|
||||||
|
{"VPSHUFBITQMB", "VPSHUFBITQMB", []Operand{vreg(t, "Z9"), vreg(t, "Z10"), vreg(t, "K3")}, "62d22d488fd9"},
|
||||||
|
{"VPTESTNMQ", "VPTESTNMQ", []Operand{vreg(t, "Z13"), vreg(t, "Z14"), vreg(t, "K5")}, "62d28e4827ed"},
|
||||||
|
// Permutations: immediate and register counts.
|
||||||
|
{"VALIGNQ", "VALIGNQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f3ed4903d903"},
|
||||||
|
{"VPERMQ imm", "VPERMQ", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f3fd4a00d101"},
|
||||||
|
{"VPERMQ reg", "VPERMQ", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K2"), vreg(t, "Z5")}, "62f2dd4a36eb"},
|
||||||
|
{"VPERMPD reg", "VPERMPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4816d9"},
|
||||||
|
{"VPERMILPS imm", "VPERMILPS", []Operand{Imm(5), vreg(t, "Z9"), vreg(t, "K2"), vreg(t, "Z10")}, "62537d4a04d105"},
|
||||||
|
{"VPERMILPS reg", "VPERMILPS", []Operand{vreg(t, "Z11"), vreg(t, "Z12"), vreg(t, "K2"), vreg(t, "Z13")}, "62521d4a0ceb"},
|
||||||
|
// Shifts: immediate, register-count and memory-count forms; the
|
||||||
|
// count source carries its own XMM tuple width.
|
||||||
|
{"VPSLLW imm mask", "VPSLLW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f16d4a71f103"},
|
||||||
|
{"VPSLLD reg count", "VPSLLD", []Operand{vreg(t, "X1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49f2d9"},
|
||||||
|
{"VPSLLDQ", "VPSLLDQ", []Operand{Imm(9), vreg(t, "Z7"), vreg(t, "Z8")}, "62f13d4873ff09"},
|
||||||
|
{"VPSRLDQ mem", "VPSRLDQ", []Operand{Imm(11), Ptr(SI, 16, 16), vreg(t, "Z4")}, "62f15d48739e100000000b"},
|
||||||
|
{"VPSRLVW", "VPSRLVW", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K1"), vreg(t, "Z5")}, "62f2dd4910eb"},
|
||||||
|
// Conversions and shuffles with the F2 prefix and no prefix.
|
||||||
|
{"VCVTUDQ2PS", "VCVTUDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f17f497ad1"},
|
||||||
|
{"VSHUFPS", "VSHUFPS", []Operand{Imm(2), vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f15449c6f402"},
|
||||||
|
// Gather and scatter prefetch hints (memory-only, /digit in reg).
|
||||||
|
{"VGATHERPF0DPD", "VGATHERPF0DPD", []Operand{vreg(t, "K5"), vsib("R10", "Y29", 8)}, "6292fd45c60cea"},
|
||||||
|
{"VSCATTERPF1DPS", "VSCATTERPF1DPS", []Operand{vreg(t, "K2"), vsib("R10", "Z28", 4)}, "62927d42c634a2"},
|
||||||
|
// Opmask broadcasts and the K logic.
|
||||||
|
{"VPBROADCASTMB2Q", "VPBROADCASTMB2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe482ad1"},
|
||||||
|
{"VPBROADCASTMW2D", "VPBROADCASTMW2D", []Operand{vreg(t, "K3"), vreg(t, "Z4")}, "62f27e483ae3"},
|
||||||
|
{"KUNPCKWD", "KUNPCKWD", []Operand{vreg(t, "K6"), vreg(t, "K4"), vreg(t, "K1")}, "c5dc4bce"},
|
||||||
|
{"KADDB", "KADDB", []Operand{vreg(t, "K2"), vreg(t, "K3"), vreg(t, "K5")}, "c5e54aea"},
|
||||||
|
// Lane extracts to general registers (EVEX and VEX routes).
|
||||||
|
{"VPEXTRB", "VPEXTRB", []Operand{Imm(3), vreg(t, "X26"), AX}, "62637d0814d003"},
|
||||||
|
{"VPEXTRD", "VPEXTRD", []Operand{Imm(1), vreg(t, "X26"), vreg(t, "R9")}, "62437d0816d101"},
|
||||||
|
{"VPEXTRD vex", "VPEXTRD", []Operand{Imm(1), vreg(t, "X2"), DI}, "c4e37916d701"},
|
||||||
|
{"VPINSRQ", "VPINSRQ", []Operand{Imm(1), DI, vreg(t, "X3"), vreg(t, "X4")}, "c4e3e122e701"},
|
||||||
|
// Moves: masked unaligned, masked scalar register form, half moves
|
||||||
|
// and non-temporal stores.
|
||||||
|
{"VMOVUPS mask", "VMOVUPS", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f17c4a11cb"},
|
||||||
|
{"VMOVSD 3op", "VMOVSD", []Operand{vreg(t, "X14"), vreg(t, "X5"), vreg(t, "K3"), vreg(t, "X22")}, "6231d70b11f6"},
|
||||||
|
{"VMOVSS 3op", "VMOVSS", []Operand{vreg(t, "X18"), vreg(t, "X3"), vreg(t, "K2"), vreg(t, "X25")}, "6281660a11d1"},
|
||||||
|
{"VMOVHPS insert", "VMOVHPS", []Operand{Ptr(SI, 0, 8), vreg(t, "X18"), vreg(t, "X19")}, "62e16c00161e"},
|
||||||
|
{"VMOVHPS store", "VMOVHPS", []Operand{vreg(t, "X20"), Ptr(SI, 8, 8)}, "62e17c08176601"},
|
||||||
|
{"VMOVLHPS", "VMOVLHPS", []Operand{vreg(t, "X16"), vreg(t, "X5"), vreg(t, "X17")}, "62a1540816c8"},
|
||||||
|
{"VMOVNTDQ", "VMOVNTDQ", []Operand{vreg(t, "Z7"), Ptr(SI, 0, 64)}, "62f17d48e73e"},
|
||||||
|
{"VMOVNTDQA", "VMOVNTDQA", []Operand{Ptr(SI, 64, 64), vreg(t, "Z8")}, "62727d482a4601"},
|
||||||
|
{"VMOVNTPS", "VMOVNTPS", []Operand{vreg(t, "Z9"), Ptr(SI, 0, 64)}, "62717c482b0e"},
|
||||||
|
// Scalar compares with and without the 66 prefix.
|
||||||
|
{"VCOMISD", "VCOMISD", []Operand{vreg(t, "X5"), vreg(t, "X6")}, "c5f92ff5"},
|
||||||
|
{"VUCOMISS", "VUCOMISS", []Operand{vreg(t, "X7"), vreg(t, "X8")}, "c5782ec7"},
|
||||||
|
// Floating point helpers.
|
||||||
|
{"VSQRTSD", "VSQRTSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f1ef0951d9"},
|
||||||
|
{"VEXP2PD", "VEXP2PD", []Operand{vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f2fd49c8f5"},
|
||||||
|
{"VRCP28SD", "VRCP28SD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "K1"), vreg(t, "X10")}, "6252bd09cbd1"},
|
||||||
|
{"VBROADCASTF32X2", "VBROADCASTF32X2", []Operand{vreg(t, "X1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d4919d1"},
|
||||||
|
{"VPCOMPRESSB", "VPCOMPRESSB", []Operand{vreg(t, "Z1"), vreg(t, "K1"), Ptr(SI, 0, 64)}, "62f27d49630e"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvexQuadRegisterGroundTruth pins the quad-register instructions (the
|
||||||
|
// 4FMAPS and 4VNNIW families) byte for byte against go tool asm: the memory
|
||||||
|
// source keeps r/m, the bracketed list's LOW register travels the inverted
|
||||||
|
// 5-bit V'VVVV field, the destination sits in reg, the opmask rides aaa and
|
||||||
|
// the vector length follows the destination (L'L=512 for the ZMM forms,
|
||||||
|
// 128 for the scalar ones) while the disp8×N multiplier stays 16 for every
|
||||||
|
// member. The x86 decoder has no view of these forms, so no decode check
|
||||||
|
// runs.
|
||||||
|
func TestEvexQuadRegisterGroundTruth(t *testing.T) {
|
||||||
|
sp := vreg(t, "RSP")
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"V4FMADDPS 17(SP) [Z0-Z3] K2 Z0", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||||
|
"62f27f4a9a842411000000"},
|
||||||
|
{"V4FMADDPS [Z10-Z13]", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z10"), vreg(t, "Z13")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||||
|
"62f22f4a9a842411000000"},
|
||||||
|
{"V4FMADDPS [Z20-Z23]", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z20"), vreg(t, "Z23")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||||
|
"62f25f429a842411000000"},
|
||||||
|
{"V4FMADDPS Z8 dst", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z8")},
|
||||||
|
"62727f4a9a842411000000"},
|
||||||
|
{"V4FMADDPS disp8x16", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 64, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||||
|
"62f27f4a9a442404"},
|
||||||
|
{"V4FMADDPS unmasked", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
|
||||||
|
"62f27f489a842411000000"},
|
||||||
|
{"V4FMADDSS 7(AX) [X0-X3] K5 X22", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||||
|
"62e27f0d9bb007000000"},
|
||||||
|
{"V4FMADDSS (DI)", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(DI, 0, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||||
|
"62e27f0d9b37"},
|
||||||
|
{"V4FMADDSS [X10-X13]", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X10"), vreg(t, "X13")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||||
|
"62e22f0d9bb007000000"},
|
||||||
|
{"V4FMADDSS [X20-X23]", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||||
|
"62e25f059bb007000000"},
|
||||||
|
{"V4FMADDSS X30 dst", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X30")},
|
||||||
|
"62627f0d9bb007000000"},
|
||||||
|
{"V4FMADDSS X3 dst", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X3")},
|
||||||
|
"62f27f0d9b9807000000"},
|
||||||
|
{"V4FMADDSS disp8x16", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 16, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X30")},
|
||||||
|
"62625f059b7001"},
|
||||||
|
{"V4FNMADDPS", "V4FNMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||||
|
"62f27f4aaa842411000000"},
|
||||||
|
{"V4FNMADDSS", "V4FNMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||||
|
"62e27f0dabb007000000"},
|
||||||
|
{"VP4DPWSSD", "VP4DPWSSD",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||||
|
"62f27f4a52842411000000"},
|
||||||
|
{"VP4DPWSSDS unmasked", "VP4DPWSSDS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
|
||||||
|
"62f27f4853842411000000"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvexQuadRegisterErrors pins the operand shapes the toolchain rejects:
|
||||||
|
// the register class the list and the destination take is fixed per
|
||||||
|
// instruction, the source is memory only, the opmask slot is positional and
|
||||||
|
// the list's low register owns V'VVVV.
|
||||||
|
func TestEvexQuadRegisterErrors(t *testing.T) {
|
||||||
|
sp := vreg(t, "RSP")
|
||||||
|
list := func(lo, hi string) RegList {
|
||||||
|
return RegList{vreg(t, lo), vreg(t, hi)}
|
||||||
|
}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
}{
|
||||||
|
{"X list on the PS form", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("X0", "X3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||||
|
{"Z list on the SS form", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 0, 8), list("Z0", "Z3"), vreg(t, "K5"), vreg(t, "X22")}},
|
||||||
|
{"Y destination", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Y0")}},
|
||||||
|
{"register source", "V4FMADDPS",
|
||||||
|
[]Operand{vreg(t, "Z1"), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||||
|
{"non-mask third operand", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z4"), vreg(t, "Z0")}},
|
||||||
|
{"k0 mask", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K0"), vreg(t, "Z0")}},
|
||||||
|
{"K after the destination", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0"), vreg(t, "K2")}},
|
||||||
|
{"zeroing without a mask", "V4FMADDPS.Z",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0")}},
|
||||||
|
{"SAE suffix", "V4FMADDPS.SAE",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||||
|
{"high index source", "VP4DPWSSD",
|
||||||
|
[]Operand{Idx(DI, vreg(t, "X16"), 1, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||||
|
{"short operand list", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3")}},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", c.name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,159 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
// The extended-instruction registry: the lookup over and above the generated
|
||||||
|
// architecture tables. The generated tables (arch/*_gen.go) list the
|
||||||
|
// mnemonics the Go toolchain knows; the extension layer carries the
|
||||||
|
// instructions it does not, and this file indexes them per architecture so
|
||||||
|
// the assembler and the linter can consult the layer without touching the
|
||||||
|
// generated lists or the main encoders. A later hook wires
|
||||||
|
// ExtensionEncodable into the Encodable mirror and EncodeExtension into the
|
||||||
|
// per-architecture assembly paths; nothing existing changes until then.
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"slices"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
|
)
|
||||||
|
|
||||||
|
// extensionIndex is the per-architecture index of the extension layer, keyed
|
||||||
|
// by upper-case mnemonic. One mnemonic registers several forms (the SVE ADD
|
||||||
|
// carries unpredicated, predicated and immediate shapes), so the value is the
|
||||||
|
// full candidate list in table order.
|
||||||
|
type extensionIndex struct {
|
||||||
|
byName map[string][]arch.ExtInstr
|
||||||
|
}
|
||||||
|
|
||||||
|
// extensionIndexes builds one index per known architecture. Architectures
|
||||||
|
// whose extension layer is not built yet get an empty index, which keeps the
|
||||||
|
// queries answering false rather than failing on a missing entry.
|
||||||
|
var extensionIndexes = buildExtensionIndexes()
|
||||||
|
|
||||||
|
func buildExtensionIndexes() map[arch.Arch]*extensionIndex {
|
||||||
|
m := make(map[arch.Arch]*extensionIndex)
|
||||||
|
for _, a := range []arch.Arch{arch.AMD64, arch.ARM64, arch.RISCV, arch.LOONG64} {
|
||||||
|
idx := &extensionIndex{byName: make(map[string][]arch.ExtInstr)}
|
||||||
|
for _, in := range arch.Extensions(a) {
|
||||||
|
key := strings.ToUpper(in.Name)
|
||||||
|
idx.byName[key] = append(idx.byName[key], in)
|
||||||
|
}
|
||||||
|
m[a] = idx
|
||||||
|
}
|
||||||
|
return m
|
||||||
|
}
|
||||||
|
|
||||||
|
// LookupExtension returns the extended instructions registered for the
|
||||||
|
// mnemonic on a, outside the generated architecture table. It reports false
|
||||||
|
// when a carries no extended layer or the mnemonic is not in it; a mnemonic
|
||||||
|
// the base table knows is not thereby covered, the layers stay independent.
|
||||||
|
func LookupExtension(a arch.Arch, mnemonic string) ([]arch.ExtInstr, bool) {
|
||||||
|
idx, ok := extensionIndexes[a]
|
||||||
|
if !ok || idx == nil {
|
||||||
|
return nil, false
|
||||||
|
}
|
||||||
|
cands, ok := idx.byName[strings.ToUpper(mnemonic)]
|
||||||
|
return cands, ok && len(cands) > 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExtensionNames returns the mnemonics the extension layer of a registers,
|
||||||
|
// in table order, without duplicates.
|
||||||
|
func ExtensionNames(a arch.Arch) []string {
|
||||||
|
var names []string
|
||||||
|
seen := make(map[string]bool)
|
||||||
|
for _, in := range arch.Extensions(a) {
|
||||||
|
key := strings.ToUpper(in.Name)
|
||||||
|
if !seen[key] {
|
||||||
|
seen[key] = true
|
||||||
|
names = append(names, in.Name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return names
|
||||||
|
}
|
||||||
|
|
||||||
|
// EncodeExtension encodes one extended instruction on a: it resolves the
|
||||||
|
// mnemonic through the extension registry, picks the registered form whose
|
||||||
|
// arity matches the operands and encodes against it. The first form that
|
||||||
|
// encodes wins. When every matching form rejects the operands, the error
|
||||||
|
// comes from the form whose operand kinds the list points at (the one with
|
||||||
|
// the most matching positions), so a mis-spelled predicate qualifier is
|
||||||
|
// diagnosed as one, not as the unpredicated form's register complaint.
|
||||||
|
func EncodeExtension(a arch.Arch, mnemonic string, ops ...arch.ExtOperand) ([]byte, error) {
|
||||||
|
cands, ok := LookupExtension(a, mnemonic)
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("%s registers no extended instruction %q", a, mnemonic)
|
||||||
|
}
|
||||||
|
var bestErr error
|
||||||
|
var bestScore int
|
||||||
|
var tried int
|
||||||
|
for _, in := range cands {
|
||||||
|
if in.Form.Arity() != len(ops) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
tried++
|
||||||
|
b, err := in.Encode(ops)
|
||||||
|
if err == nil {
|
||||||
|
return b, nil
|
||||||
|
}
|
||||||
|
if score := kindScore(in.Form, ops); bestErr == nil || score > bestScore {
|
||||||
|
bestErr, bestScore = err, score
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if tried == 0 {
|
||||||
|
return nil, fmt.Errorf("%s: extended %q takes %s, got %d operands",
|
||||||
|
a, mnemonic, extensionAritySummary(cands), len(ops))
|
||||||
|
}
|
||||||
|
return nil, bestErr
|
||||||
|
}
|
||||||
|
|
||||||
|
// kindScore counts the positions whose operand kind matches what the form
|
||||||
|
// wants, the tie-break that picks the most specific rejection.
|
||||||
|
func kindScore(form arch.ExtForm, ops []arch.ExtOperand) int {
|
||||||
|
kinds := form.Kinds()
|
||||||
|
score := 0
|
||||||
|
for i, op := range ops {
|
||||||
|
if i < len(kinds) && op.Kind == kinds[i] {
|
||||||
|
score++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return score
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExtensionEncodable reports whether the extension layer of a encodes the
|
||||||
|
// mnemonic with these operands. It mirrors asm.Encodable for the extension
|
||||||
|
// layer: the predicate the linter consults once the hook wires it in.
|
||||||
|
func ExtensionEncodable(a arch.Arch, mnemonic string, ops ...arch.ExtOperand) bool {
|
||||||
|
_, err := EncodeExtension(a, mnemonic, ops...)
|
||||||
|
return err == nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// extensionAritySummary describes the operand counts the candidate forms
|
||||||
|
// take, "2 or 3" style, for the arity error.
|
||||||
|
func extensionAritySummary(cands []arch.ExtInstr) string {
|
||||||
|
counts := make([]int, 0, len(cands))
|
||||||
|
seen := make(map[int]bool)
|
||||||
|
for _, in := range cands {
|
||||||
|
n := in.Form.Arity()
|
||||||
|
if !seen[n] {
|
||||||
|
seen[n] = true
|
||||||
|
counts = append(counts, n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
slices.Sort(counts)
|
||||||
|
var b strings.Builder
|
||||||
|
for i, n := range counts {
|
||||||
|
if i > 0 {
|
||||||
|
if i == len(counts)-1 {
|
||||||
|
b.WriteString(" or ")
|
||||||
|
} else {
|
||||||
|
b.WriteString(", ")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
fmt.Fprintf(&b, "%d", n)
|
||||||
|
}
|
||||||
|
b.WriteString(" operands")
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
@@ -0,0 +1,197 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/hex"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestExtensionRegistryARM64 checks the mnemonic lookup over and above the
|
||||||
|
// generated arm64 table: one mnemonic, several forms, case-insensitive, and
|
||||||
|
// nothing offered for a spelling the layer does not carry.
|
||||||
|
func TestExtensionRegistryARM64(t *testing.T) {
|
||||||
|
add, ok := LookupExtension(arch.ARM64, "ADD")
|
||||||
|
if !ok {
|
||||||
|
t.Fatal("LookupExtension(ARM64, ADD) found nothing")
|
||||||
|
}
|
||||||
|
var forms []arch.ExtForm
|
||||||
|
for _, in := range add {
|
||||||
|
if in.Name != "ADD" {
|
||||||
|
t.Errorf("candidate %q leaked into the ADD lookup", in.Name)
|
||||||
|
}
|
||||||
|
forms = append(forms, in.Form)
|
||||||
|
}
|
||||||
|
if len(forms) != 3 ||
|
||||||
|
forms[0] != arch.ExtFormVectors ||
|
||||||
|
forms[1] != arch.ExtFormPredicated ||
|
||||||
|
forms[2] != arch.ExtFormImmediate {
|
||||||
|
t.Errorf("ADD registers forms %v, want unpredicated, predicated and immediate", forms)
|
||||||
|
}
|
||||||
|
if _, ok := LookupExtension(arch.ARM64, "add"); !ok {
|
||||||
|
t.Error("the lookup is case-sensitive")
|
||||||
|
}
|
||||||
|
if _, ok := LookupExtension(arch.ARM64, "NOSUCHINSTR"); ok {
|
||||||
|
t.Error("a non-extended mnemonic resolved")
|
||||||
|
}
|
||||||
|
sqadd, ok := LookupExtension(arch.ARM64, "SQADD")
|
||||||
|
if !ok || len(sqadd) != 2 {
|
||||||
|
t.Errorf("SQADD registers %d forms, want the unpredicated and immediate pair", len(sqadd))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestExtensionAboveGeneratedTable pins the layering: SQADD is nowhere in the
|
||||||
|
// generated arm64 table (the toolchain knows only the NEON spelling VSQADD)
|
||||||
|
// yet the extension layer carries it, while ADD sits in both layers
|
||||||
|
// independently.
|
||||||
|
func TestExtensionAboveGeneratedTable(t *testing.T) {
|
||||||
|
if _, found := arch.ForArch(arch.ARM64).Lookup("SQADD"); found {
|
||||||
|
t.Error("SQADD is in the generated table, the layering assumption broke")
|
||||||
|
}
|
||||||
|
if _, ok := LookupExtension(arch.ARM64, "SQADD"); !ok {
|
||||||
|
t.Error("SQADD is missing from the extension layer")
|
||||||
|
}
|
||||||
|
if _, found := arch.ForArch(arch.ARM64).Lookup("ADD"); !found {
|
||||||
|
t.Error("ADD vanished from the generated table")
|
||||||
|
}
|
||||||
|
if add, ok := LookupExtension(arch.ARM64, "ADD"); !ok || len(add) != 3 {
|
||||||
|
t.Errorf("ADD carries %d extension forms, want 3", len(add))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEncodeExtensionGolden encodes through the registry and pins the same
|
||||||
|
// golden words the arch table tests pin, proving the registry resolves to the
|
||||||
|
// right encoding.
|
||||||
|
func TestEncodeExtensionGolden(t *testing.T) {
|
||||||
|
for _, tt := range []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []arch.ExtOperand
|
||||||
|
want uint32
|
||||||
|
}{
|
||||||
|
{"unpredicated add", "ADD",
|
||||||
|
[]arch.ExtOperand{
|
||||||
|
arch.ExtVector(2, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB),
|
||||||
|
},
|
||||||
|
0x04200040},
|
||||||
|
{"predicated mul", "MUL",
|
||||||
|
[]arch.ExtOperand{
|
||||||
|
arch.ExtVector(0, arch.ExtArrB), arch.ExtPredicate(2, arch.ExtQualMerging), arch.ExtVector(0, arch.ExtArrB),
|
||||||
|
},
|
||||||
|
0x04100800},
|
||||||
|
{"immediate add with derived shift", "ADD",
|
||||||
|
[]arch.ExtOperand{arch.ExtImmediate(32512), arch.ExtVector(0, arch.ExtArrH)},
|
||||||
|
0x2560efe0},
|
||||||
|
{"signed immediate mul", "MUL",
|
||||||
|
[]arch.ExtOperand{arch.ExtImmediate(-1), arch.ExtVector(0, arch.ExtArrB)},
|
||||||
|
0x2530dfe0},
|
||||||
|
} {
|
||||||
|
got, err := EncodeExtension(arch.ARM64, tt.mnem, tt.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: encode: %v", tt.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if want := hex.EncodeToString([]byte{
|
||||||
|
byte(tt.want), byte(tt.want >> 8), byte(tt.want >> 16), byte(tt.want >> 24),
|
||||||
|
}); hex.EncodeToString(got) != want {
|
||||||
|
t.Errorf("%s:\n got %x\n want %s", tt.name, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEncodeExtensionErrors checks the registry's diagnostics: a wrong arity
|
||||||
|
// names every form's count, an operand the first candidate rejects surfaces
|
||||||
|
// its own message once a later form takes over.
|
||||||
|
func TestEncodeExtensionErrors(t *testing.T) {
|
||||||
|
if _, err := EncodeExtension(arch.ARM64, "ADD", arch.ExtVector(0, arch.ExtArrB)); err == nil {
|
||||||
|
t.Error("one operand encoded, want an arity error")
|
||||||
|
} else if !strings.Contains(err.Error(), "2 or 3 operands") {
|
||||||
|
t.Errorf("arity error %q does not name the counts", err)
|
||||||
|
}
|
||||||
|
// The predicated candidate must answer for its own operands: the /Z
|
||||||
|
// qualifier is rejected with the merging message, not the unpredicated
|
||||||
|
// form's register-kind complaint.
|
||||||
|
_, err := EncodeExtension(arch.ARM64, "ADD",
|
||||||
|
arch.ExtVector(0, arch.ExtArrB), arch.ExtPredicate(0, arch.ExtQualZeroing), arch.ExtVector(0, arch.ExtArrB))
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("/Z encoded, want an error")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), "/M") {
|
||||||
|
t.Errorf("error %q does not name the merging qualifier", err)
|
||||||
|
}
|
||||||
|
if _, err := EncodeExtension(arch.ARM64, "NOSUCHINSTR", arch.ExtVector(0, arch.ExtArrB)); err == nil ||
|
||||||
|
!strings.Contains(err.Error(), "registers no extended instruction") {
|
||||||
|
t.Errorf("unknown mnemonic error = %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestExtensionEncodable checks the predicate the later Encodable hook will
|
||||||
|
// call: true exactly when the registry encodes the operand list.
|
||||||
|
func TestExtensionEncodable(t *testing.T) {
|
||||||
|
if !ExtensionEncodable(arch.ARM64, "ADD",
|
||||||
|
arch.ExtVector(0, arch.ExtArrS), arch.ExtVector(1, arch.ExtArrS), arch.ExtVector(2, arch.ExtArrS)) {
|
||||||
|
t.Error("an encodable unpredicated add reported false")
|
||||||
|
}
|
||||||
|
if !ExtensionEncodable(arch.ARM64, "ADD",
|
||||||
|
arch.ExtVector(1, arch.ExtArrS), arch.ExtPredicate(0, arch.ExtQualMerging), arch.ExtVector(0, arch.ExtArrS)) {
|
||||||
|
t.Error("an encodable predicated add reported false")
|
||||||
|
}
|
||||||
|
if ExtensionEncodable(arch.ARM64, "ADD",
|
||||||
|
arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrS), arch.ExtVector(0, arch.ExtArrB)) {
|
||||||
|
t.Error("mismatched arrangements reported encodable")
|
||||||
|
}
|
||||||
|
if ExtensionEncodable(arch.ARM64, "ADD", arch.ExtVector(0, arch.ExtArrB)) {
|
||||||
|
t.Error("a one-operand add reported encodable")
|
||||||
|
}
|
||||||
|
if ExtensionEncodable(arch.ARM64, "NOSUCHINSTR") {
|
||||||
|
t.Error("an unregistered mnemonic reported encodable")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestExtensionArchIsolation is the architecture-binding negative case: the
|
||||||
|
// extension layer is registered for arm64 alone, and no other architecture
|
||||||
|
// answers its queries, not even for a mnemonic the amd64 base table carries.
|
||||||
|
func TestExtensionArchIsolation(t *testing.T) {
|
||||||
|
ops := []arch.ExtOperand{
|
||||||
|
arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB),
|
||||||
|
}
|
||||||
|
for _, a := range []arch.Arch{arch.AMD64, arch.RISCV, arch.LOONG64, arch.Unknown} {
|
||||||
|
if cands, ok := LookupExtension(a, "ADD"); ok || cands != nil {
|
||||||
|
t.Errorf("LookupExtension(%s, ADD) offered %d candidates", a, len(cands))
|
||||||
|
}
|
||||||
|
if cands, ok := LookupExtension(a, "MUL"); ok || cands != nil {
|
||||||
|
t.Errorf("LookupExtension(%s, MUL) offered %d candidates", a, len(cands))
|
||||||
|
}
|
||||||
|
if got, err := EncodeExtension(a, "ADD", ops...); err == nil {
|
||||||
|
t.Errorf("EncodeExtension(%s, ADD) encoded %x, want a refusal", a, got)
|
||||||
|
} else if !strings.Contains(err.Error(), string(a)) {
|
||||||
|
t.Errorf("EncodeExtension(%s) error %q does not name the architecture", a, err)
|
||||||
|
}
|
||||||
|
if ExtensionEncodable(a, "ADD", ops...) {
|
||||||
|
t.Errorf("ExtensionEncodable(%s, ADD) reported true", a)
|
||||||
|
}
|
||||||
|
if names := ExtensionNames(a); len(names) != 0 {
|
||||||
|
t.Errorf("ExtensionNames(%s) = %v, want none", a, names)
|
||||||
|
}
|
||||||
|
if got := arch.Extensions(a); len(got) != 0 {
|
||||||
|
t.Errorf("arch.Extensions(%s) carries %d instructions", a, len(got))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestExtensionNamesARM64 checks the completion-facing name list: every
|
||||||
|
// distinct mnemonic of the family, first-occurrence order, no duplicates.
|
||||||
|
func TestExtensionNamesARM64(t *testing.T) {
|
||||||
|
want := []string{"ADD", "SUB", "SQADD", "UQADD", "SQSUB", "UQSUB", "MUL", "SMULH", "UMULH", "SUBR"}
|
||||||
|
got := ExtensionNames(arch.ARM64)
|
||||||
|
if strings.Join(got, ",") != strings.Join(want, ",") {
|
||||||
|
t.Errorf("ExtensionNames(ARM64) = %v, want %v", got, want)
|
||||||
|
}
|
||||||
|
if n := len(arch.Extensions(arch.ARM64)); n != 23 {
|
||||||
|
t.Errorf("the family registers %d instructions, want 23", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
+68
-4
@@ -63,16 +63,18 @@ const (
|
|||||||
const (
|
const (
|
||||||
kindSTEXT = 1
|
kindSTEXT = 1
|
||||||
kindSRODATA = 3
|
kindSRODATA = 3
|
||||||
|
kindSNOPTRDATA = 5
|
||||||
kindSDATA = 7
|
kindSDATA = 7
|
||||||
kindSDWARFFCN = 14
|
kindSDWARFFCN = 14
|
||||||
kindSDWARFLINES = 20
|
kindSDWARFLINES = 20
|
||||||
)
|
)
|
||||||
|
|
||||||
// Symbol flags (cmd/internal/goobj).
|
// Symbol flags (cmd/internal/goobj). The linkname flag is set only for
|
||||||
|
// //go:linkname symbols (and main.main); ordinary assembly symbols carry
|
||||||
|
// none, matching cmd/asm's output.
|
||||||
const (
|
const (
|
||||||
symFlagDupok = 0x01
|
symFlagDupok = 0x01
|
||||||
symFlagNoSplit = 0x10
|
symFlagNoSplit = 0x10
|
||||||
symFlag2Link = 0x10 // asm objects flag every named symbol as linkname
|
|
||||||
symABIStatic = 0xffff
|
symABIStatic = 0xffff
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -287,7 +289,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
nps = append(nps, npSym{
|
nps = append(nps, npSym{
|
||||||
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
|
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, size: uint32(fn.Size)},
|
||||||
data: code,
|
data: code,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
@@ -304,9 +306,16 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
|||||||
if !d.Static {
|
if !d.Static {
|
||||||
name = pkgPath + "." + name
|
name = pkgPath + "." + name
|
||||||
}
|
}
|
||||||
|
// RODATA implies no pointers, so it wins over NOPTR: the kind is
|
||||||
|
// SRODATA either way, exactly as the toolchain chooses it. Plain
|
||||||
|
// NOPTR data is SNOPTRDATA, which the linker keeps out of the GC's
|
||||||
|
// type scan; a plain SDATA symbol would demand Go type information
|
||||||
|
// no assembly file can supply, and the link would fail.
|
||||||
typ := uint8(kindSDATA)
|
typ := uint8(kindSDATA)
|
||||||
if d.Rodata {
|
if d.Rodata {
|
||||||
typ = kindSRODATA
|
typ = kindSRODATA
|
||||||
|
} else if d.Noptr {
|
||||||
|
typ = kindSNOPTRDATA
|
||||||
}
|
}
|
||||||
flag := uint8(0)
|
flag := uint8(0)
|
||||||
if d.Dupok {
|
if d.Dupok {
|
||||||
@@ -317,7 +326,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
|||||||
abi = symABIStatic
|
abi = symABIStatic
|
||||||
}
|
}
|
||||||
defIdx[d.Name] = len(defs)
|
defIdx[d.Name] = len(defs)
|
||||||
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)})
|
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, size: uint32(d.Size)})
|
||||||
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
|
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
|
||||||
}
|
}
|
||||||
fnFiIdx := make([]int, len(img.Funcs))
|
fnFiIdx := make([]int, len(img.Funcs))
|
||||||
@@ -462,6 +471,61 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
|||||||
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// The data symbols' own relocations: the symbol-valued DATA fields
|
||||||
|
// ("DATA s+0(SB)/8, $other(SB)"). The toolchain patches each field
|
||||||
|
// with the target's absolute address through an R_ADDR of the DATA
|
||||||
|
// line's width, on every architecture (the code relocations are
|
||||||
|
// per-architecture PC-relative shapes; a data pointer word is not), so
|
||||||
|
// this mapping bypasses relocField. The definitions were appended in
|
||||||
|
// DataSyms order, so data symbol i is definition index i.
|
||||||
|
for i, d := range img.DataSyms {
|
||||||
|
for _, r := range d.Relocs {
|
||||||
|
if r.Kind != RelAddr {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: data symbol %q carries a non-data relocation", d.Name)
|
||||||
|
}
|
||||||
|
var rec [23]byte
|
||||||
|
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
|
||||||
|
rec[4] = r.Siz
|
||||||
|
binary.LittleEndian.PutUint16(rec[5:], relocAddr)
|
||||||
|
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
|
||||||
|
switch {
|
||||||
|
case r.External && r.Name == goobjBuiltinMorestack:
|
||||||
|
binary.LittleEndian.PutUint32(rec[15:], pkgIdxBuiltin)
|
||||||
|
binary.LittleEndian.PutUint32(rec[19:], goobjBuiltinMorestackNoctxt)
|
||||||
|
case r.External:
|
||||||
|
pkg, name := splitQualified(r.Name)
|
||||||
|
if pkg == "" {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: external symbol %q has no package prefix", r.Name)
|
||||||
|
}
|
||||||
|
pIdx, ok := extPkgIdx[pkg]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: package %q not resolved", pkg)
|
||||||
|
}
|
||||||
|
sIdx, ok := extSymIdx[pkg+"·"+name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: symbol %s·%s not resolved", pkg, name)
|
||||||
|
}
|
||||||
|
binary.LittleEndian.PutUint32(rec[15:], uint32(pIdx))
|
||||||
|
binary.LittleEndian.PutUint32(rec[19:], uint32(sIdx))
|
||||||
|
default:
|
||||||
|
if di, ok := defIdx[r.Name]; ok {
|
||||||
|
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
|
||||||
|
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
|
||||||
|
break
|
||||||
|
}
|
||||||
|
// A DATA field may hold the address of a TEXT function of
|
||||||
|
// the same file (the rt0 lib entry spelling), which is a
|
||||||
|
// non-package definition.
|
||||||
|
ni, isText := textNpIdx[r.Name]
|
||||||
|
if !isText {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
|
||||||
|
}
|
||||||
|
binary.LittleEndian.PutUint32(rec[15:], pkgIdxNone)
|
||||||
|
binary.LittleEndian.PutUint32(rec[19:], uint32(ni))
|
||||||
|
}
|
||||||
|
symRelocs[i] = append(symRelocs[i], rec[:]...)
|
||||||
|
}
|
||||||
|
}
|
||||||
// The DWARF symbols' own relocations (the function address references).
|
// The DWARF symbols' own relocations (the function address references).
|
||||||
for _, ds := range dwarfRelocs {
|
for _, ds := range dwarfRelocs {
|
||||||
for _, r := range ds.relocs {
|
for _, r := range ds.relocs {
|
||||||
|
|||||||
+72
-32
@@ -40,8 +40,11 @@ func exportPath(importPath string) (string, error) {
|
|||||||
//
|
//
|
||||||
// refs maps package import paths to the symbol names referenced from that
|
// refs maps package import paths to the symbol names referenced from that
|
||||||
// package. The returned pkgIdx maps each import path to its position in
|
// package. The returned pkgIdx maps each import path to its position in
|
||||||
// the blkPkgIdx table (0-based), and symIdx gives each symbol's index within
|
// the blkPkgIdx table, which reserves index 0 for the dummy invalid
|
||||||
// its package.
|
// package (cmd/internal/obj/sym.go: "0 is invalid index"; the loader's
|
||||||
|
// reader loop starts at 1), so package i sits at block index i+1 and its
|
||||||
|
// relocations carry i+1. symIdx gives each symbol's index within its
|
||||||
|
// package.
|
||||||
func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symIdx map[string]int, err error) {
|
func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symIdx map[string]int, err error) {
|
||||||
pkgIdx = make(map[string]int, len(refs))
|
pkgIdx = make(map[string]int, len(refs))
|
||||||
symIdx = make(map[string]int)
|
symIdx = make(map[string]int)
|
||||||
@@ -50,7 +53,9 @@ func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symI
|
|||||||
packages := sortedPkgRefs(refs)
|
packages := sortedPkgRefs(refs)
|
||||||
|
|
||||||
for i, pkg := range packages {
|
for i, pkg := range packages {
|
||||||
pkgIdx[pkg.path] = i
|
// Block index 0 is the dummy invalid package; the first real
|
||||||
|
// package starts at 1.
|
||||||
|
pkgIdx[pkg.path] = i + 1
|
||||||
exp, err := exportPath(pkg.path)
|
exp, err := exportPath(pkg.path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, nil, err
|
return nil, nil, err
|
||||||
@@ -145,40 +150,65 @@ func parseArDecimal(b []byte) int {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// goobjFile is a parsed GOOBJ file: the string table and the symbol-definition
|
// goobjFile is a parsed GOOBJ file: the string table and the symbol-definition
|
||||||
// block.
|
// blocks. The hashed blocks are kept raw: their symbols carry no names, only
|
||||||
|
// the loader needs their counts.
|
||||||
type goobjFile struct {
|
type goobjFile struct {
|
||||||
strTab []byte // string table, at headerSize + n
|
strTab []byte // string table, at headerSize + n
|
||||||
symdef []byte // blkSymdef raw block
|
symdef []byte // blkSymdef raw block
|
||||||
npdef []byte // blkNonpkgdef raw block
|
hashed64 []byte // blkHashed64def raw block
|
||||||
|
hashed []byte // blkHasheddef raw block
|
||||||
|
npdef []byte // blkNonpkgdef raw block
|
||||||
}
|
}
|
||||||
|
|
||||||
// symbols returns all symbol names in definition order by scanning the
|
// loaderIndexBase returns the index the first nonpkgdef symbol occupies in the
|
||||||
// symdef and nonpkgdef blocks and resolving each name through the string
|
// loader's per-object symbol array. cmd/link lays the definition blocks out as
|
||||||
// table. Package definitions (blkSymdef) use fully-qualified names like
|
// symdef, hashed64def, hasheddef, nonpkgdef, nonpkgref (loader.go: preloadSyms
|
||||||
// "runtime.morestack"; non-package definitions (blkNonpkgdef) use bare
|
// fills r.syms in exactly that order, and resolve() indexes PkgIdxNone and
|
||||||
// names like "morestack". This combined list matches the index the
|
// cross-package SymIdx into it), so a symbol found in blkNonpkgdef carries the
|
||||||
// linker expects for cross-package references.
|
// three leading blocks' symbol counts as its base.
|
||||||
|
func (f *goobjFile) loaderIndexBase() int {
|
||||||
|
return len(f.symdef)/recSymSize + len(f.hashed64)/recSymSize + len(f.hashed)/recSymSize
|
||||||
|
}
|
||||||
|
|
||||||
|
// symbols returns the names of the symdef and nonpkgdef blocks in
|
||||||
|
// definition order. Package definitions (blkSymdef) use fully-qualified
|
||||||
|
// names like "runtime.morestack"; non-package definitions (blkNonpkgdef)
|
||||||
|
// use bare names like "morestack". For lookups by index prefer
|
||||||
|
// findSymbol: it adds the hashed blocks' count the loader's array
|
||||||
|
// interleaves between the two.
|
||||||
func (f *goobjFile) symbols() []string {
|
func (f *goobjFile) symbols() []string {
|
||||||
return append(f.defNames(), f.npdefNames()...)
|
return append(f.defNames(), f.npdefNames()...)
|
||||||
}
|
}
|
||||||
|
|
||||||
// findSymbol returns the index of a symbol within the combined symbol list,
|
// findSymbol returns the index of a symbol within the loader's per-object
|
||||||
// or -1 if not found. It first tries the fully-qualified name (pkg.name),
|
// symbol array, or -1 if not found. It first tries the fully-qualified
|
||||||
// then the bare name.
|
// name (pkg.name), then the bare name (assembly objects store dotless
|
||||||
|
// names, e.g. runtime's "gogo", for symbols other packages reach through
|
||||||
|
// a linkname).
|
||||||
func (f *goobjFile) findSymbol(pkg, name string) int {
|
func (f *goobjFile) findSymbol(pkg, name string) int {
|
||||||
|
base := f.loaderIndexBase()
|
||||||
qualified := pkg + "." + name
|
qualified := pkg + "." + name
|
||||||
syms := f.symbols()
|
for i, s := range f.defNames() {
|
||||||
for i, s := range syms {
|
|
||||||
if s == qualified {
|
if s == qualified {
|
||||||
return i
|
return i
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Try bare name (for non-package definitions).
|
for i, s := range f.npdefNames() {
|
||||||
for i, s := range syms {
|
if s == qualified {
|
||||||
|
return base + i
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Try bare name (for dotless assembly definitions).
|
||||||
|
for i, s := range f.defNames() {
|
||||||
if s == name {
|
if s == name {
|
||||||
return i
|
return i
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
for i, s := range f.npdefNames() {
|
||||||
|
if s == name {
|
||||||
|
return base + i
|
||||||
|
}
|
||||||
|
}
|
||||||
return -1
|
return -1
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -192,12 +222,16 @@ func (f *goobjFile) npdefNames() []string {
|
|||||||
return f.readSymNames(f.npdef)
|
return f.readSymNames(f.npdef)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// recSymSize is the size of one Sym record in the definition blocks
|
||||||
|
// (goobj.SymSize: stringRefSize + 2 + 1 + 1 + 1 + 4 + 4).
|
||||||
|
const recSymSize = 21
|
||||||
|
|
||||||
// readSymNames reads symbol names from a symdef/nonpkgdef block. Each record
|
// readSymNames reads symbol names from a symdef/nonpkgdef block. Each record
|
||||||
// is 21 bytes: nameLen (u32), nameOff (u32), abi (u16), typ, flag, flag2,
|
// is 21 bytes: nameLen (u32), nameOff (u32), abi (u16), typ, flag, flag2,
|
||||||
// size (u32), align (u32). nameOff is an absolute offset into the string
|
// size (u32), align (u32). nameOff is an absolute offset into the string
|
||||||
// table.
|
// table.
|
||||||
func (f *goobjFile) readSymNames(block []byte) []string {
|
func (f *goobjFile) readSymNames(block []byte) []string {
|
||||||
const recSize = 21
|
const recSize = recSymSize
|
||||||
if len(block) < recSize {
|
if len(block) < recSize {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
@@ -247,16 +281,18 @@ func parseGOOBJ(data []byte) (*goobjFile, error) {
|
|||||||
// [16:20] flags
|
// [16:20] flags
|
||||||
// [20:96] 19 × uint32 offsets
|
// [20:96] 19 × uint32 offsets
|
||||||
var offs [blkEnd + 1]uint32
|
var offs [blkEnd + 1]uint32
|
||||||
for i := 0; i <= blkEnd; i++ {
|
for i := range blkEnd + 1 {
|
||||||
offs[i] = binary.LittleEndian.Uint32(payload[20+4*i:])
|
offs[i] = binary.LittleEndian.Uint32(payload[20+4*i:])
|
||||||
}
|
}
|
||||||
// The string table lives at headerSize.
|
// The string table lives at headerSize.
|
||||||
strTabStart := uint32(goobjHeaderSize)
|
strTabStart := uint32(goobjHeaderSize)
|
||||||
|
|
||||||
f := &goobjFile{
|
f := &goobjFile{
|
||||||
strTab: payload[strTabStart:offs[0]],
|
strTab: payload[strTabStart:offs[0]],
|
||||||
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
|
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
|
||||||
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
|
hashed64: blockSlice(payload, offs, blkHashed64def, blkHashed64def+1),
|
||||||
|
hashed: blockSlice(payload, offs, blkHasheddef, blkHasheddef+1),
|
||||||
|
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
|
||||||
}
|
}
|
||||||
return f, nil
|
return f, nil
|
||||||
}
|
}
|
||||||
@@ -306,22 +342,26 @@ func resolveExternalSymbols(externals []string) (pkgTable []string, pkgIdxMap ma
|
|||||||
return nil, nil, nil, err
|
return nil, nil, nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
// Build the package table in pkgIdx order.
|
// Build the package table in pkgIdx order. The indices are 1-based
|
||||||
|
// (0 is the dummy invalid package, written by the emitter itself), so
|
||||||
|
// the table without the dummy is indexed one below.
|
||||||
pkgTable = make([]string, len(pkgIdx1))
|
pkgTable = make([]string, len(pkgIdx1))
|
||||||
for pkg, idx := range pkgIdx1 {
|
for pkg, idx := range pkgIdx1 {
|
||||||
pkgTable[idx] = pkg
|
pkgTable[idx-1] = pkg
|
||||||
}
|
}
|
||||||
|
|
||||||
return pkgTable, pkgIdx1, symIdx1, nil
|
return pkgTable, pkgIdx1, symIdx1, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// splitQualified splits a qualified Go symbol name (pkgpath·name) into its
|
// splitQualified splits a qualified Go symbol name (pkgpath·name) into its
|
||||||
// package path and local name. The separator is the middle dot (U+00B7).
|
// package path and local name. The separator is the middle dot (U+00B7),
|
||||||
// If no separator is found, the symbol is assumed to be in the current
|
// whose UTF-8 encoding is two bytes, so the search must be string-based:
|
||||||
// package (empty pkg).
|
// IndexByte would match only the second byte and leave the lead byte on
|
||||||
|
// the package path. If no separator is found, the symbol is assumed to be
|
||||||
|
// in the current package (empty pkg).
|
||||||
func splitQualified(full string) (pkg, name string) {
|
func splitQualified(full string) (pkg, name string) {
|
||||||
if idx := strings.IndexByte(full, '\u00b7'); idx >= 0 {
|
if before, after, ok := strings.Cut(full, "\u00b7"); ok {
|
||||||
return full[:idx], full[idx+len("\u00b7"):]
|
return before, after
|
||||||
}
|
}
|
||||||
if before, after, ok := strings.Cut(full, "."); ok {
|
if before, after, ok := strings.Cut(full, "."); ok {
|
||||||
return before, after
|
return before, after
|
||||||
|
|||||||
@@ -45,6 +45,9 @@ func TestReadRuntimeSymbols(t *testing.T) {
|
|||||||
// TestResolveExternalSymbols verifies end-to-end resolution of external
|
// TestResolveExternalSymbols verifies end-to-end resolution of external
|
||||||
// symbol references.
|
// symbol references.
|
||||||
func TestResolveExternalSymbols(t *testing.T) {
|
func TestResolveExternalSymbols(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("resolves through a live go list -export: skipped in -short mode")
|
||||||
|
}
|
||||||
if _, err := exec.LookPath("go"); err != nil {
|
if _, err := exec.LookPath("go"); err != nil {
|
||||||
t.Skip("go toolchain not available")
|
t.Skip("go toolchain not available")
|
||||||
}
|
}
|
||||||
@@ -56,8 +59,12 @@ func TestResolveExternalSymbols(t *testing.T) {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("resolveExternalGOOBJ: %v", err)
|
t.Fatalf("resolveExternalGOOBJ: %v", err)
|
||||||
}
|
}
|
||||||
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 0 {
|
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 1 {
|
||||||
t.Errorf("pkgIdx = %v, want runtime→0", pkgIdx)
|
// Index 0 is the dummy invalid package in the blkPkgIdx table;
|
||||||
|
// the loader's reader loop starts at 1 (cmd/link/internal/
|
||||||
|
// loader/loader.go: "PkgIdx 0 is a dummy invalid package"), so
|
||||||
|
// the first real package must carry index 1.
|
||||||
|
t.Errorf("pkgIdx = %v, want runtime→1", pkgIdx)
|
||||||
}
|
}
|
||||||
if _, ok := symIdx["runtime·g0"]; !ok {
|
if _, ok := symIdx["runtime·g0"]; !ok {
|
||||||
t.Errorf("symIdx missing runtime·g0, got %v", symIdx)
|
t.Errorf("symIdx missing runtime·g0, got %v", symIdx)
|
||||||
|
|||||||
+335
-6
@@ -12,7 +12,7 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// goobjView is a minimal parsed view of a GOOBJ payload, enough to check
|
// goobjView is a minimal parsed view of a GOOBJ payload, enough to check
|
||||||
@@ -117,7 +117,9 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
|||||||
if len(defs) != 7 {
|
if len(defs) != 7 {
|
||||||
t.Fatalf("symdefs = %d, want 7", len(defs))
|
t.Fatalf("symdefs = %d, want 7", len(defs))
|
||||||
}
|
}
|
||||||
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
|
// The linkname flag stays clear: the toolchain sets it only for
|
||||||
|
// //go:linkname symbols, and an ordinary static GLOBL is not one.
|
||||||
|
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != 0 {
|
||||||
t.Errorf("mask symbol = %+v", defs[0])
|
t.Errorf("mask symbol = %+v", defs[0])
|
||||||
}
|
}
|
||||||
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
|
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
|
||||||
@@ -158,8 +160,8 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
|||||||
t.Errorf("funcinfo bytes %x", fi)
|
t.Errorf("funcinfo bytes %x", fi)
|
||||||
}
|
}
|
||||||
|
|
||||||
// The pc-value tables of addq (non-package indices 0–3, so global
|
// The pc-value tables of addq (non-package indices 0-3, so global
|
||||||
// indices 7–10): pcsp a flat zero over the whole function, pcinline a
|
// indices 7-10): pcsp a flat zero over the whole function, pcinline a
|
||||||
// flat -1, both with the pc delta in MinLC (1) units.
|
// flat -1, both with the pc delta in MinLC (1) units.
|
||||||
pcsp := data[le.Uint32(didx[4*7:]):]
|
pcsp := data[le.Uint32(didx[4*7:]):]
|
||||||
if got := pcsp[:3]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
|
if got := pcsp[:3]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
|
||||||
@@ -294,7 +296,7 @@ TEXT ·framed(SB), NOSPLIT, $8-0
|
|||||||
}
|
}
|
||||||
for i := range wantPCs {
|
for i := range wantPCs {
|
||||||
if pcs[i] != wantPCs[i] || vals[i] != wantVals[i] {
|
if pcs[i] != wantPCs[i] || vals[i] != wantVals[i] {
|
||||||
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d) — all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
|
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d); all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// The last two steps unwind the epilogue to zero.
|
// The last two steps unwind the epilogue to zero.
|
||||||
@@ -333,7 +335,7 @@ TEXT ·useext(SB), NOSPLIT, $0-8
|
|||||||
|
|
||||||
// TestGOObjectLinkAndRun is the end-to-end check: assemble the test
|
// TestGOObjectLinkAndRun is the end-to-end check: assemble the test
|
||||||
// functions to a GOOBJ, swap it into a go build in place of the toolchain's
|
// functions to a GOOBJ, swap it into a go build in place of the toolchain's
|
||||||
// assembly object, link, and run — the output must match the baseline
|
// assembly object, link, and run; the output must match the baseline
|
||||||
// binary the Go assembler produced. Skipped when no Go toolchain is
|
// binary the Go assembler produced. Skipped when no Go toolchain is
|
||||||
// available.
|
// available.
|
||||||
func TestGOObjectLinkAndRun(t *testing.T) {
|
func TestGOObjectLinkAndRun(t *testing.T) {
|
||||||
@@ -511,3 +513,330 @@ func fieldAfter(line, flag string) string {
|
|||||||
}
|
}
|
||||||
return ""
|
return ""
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// buildLogSteps is what a substitution test needs from a `go build -x -work`
|
||||||
|
// log: the work directory, the assembler's object, the package archive and
|
||||||
|
// the link command line.
|
||||||
|
type buildLogSteps struct {
|
||||||
|
work string
|
||||||
|
asmObj string // $WORK expanded
|
||||||
|
pkgArch string // $WORK expanded
|
||||||
|
linkLine string // still carries $WORK placeholders
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseBuildLog extracts the build steps from a `go build -x -work` log.
|
||||||
|
// asmFile names the assembly file whose object the test substitutes. A
|
||||||
|
// missing step is a failure, not a skip: the toolchain changed shape and the
|
||||||
|
// substitution would silently test nothing.
|
||||||
|
func parseBuildLog(t *testing.T, log []byte, asmFile string) buildLogSteps {
|
||||||
|
t.Helper()
|
||||||
|
var st buildLogSteps
|
||||||
|
for line := range strings.SplitSeq(string(log), "\n") {
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(line, "WORK="):
|
||||||
|
st.work = strings.TrimPrefix(line, "WORK=")
|
||||||
|
case strings.Contains(line, "/asm ") && strings.Contains(line, asmFile) && !strings.Contains(line, "-gensymabis"):
|
||||||
|
st.asmObj = fieldAfter(line, "-o")
|
||||||
|
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
|
||||||
|
rest := strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
|
||||||
|
st.pkgArch = strings.Fields(strings.SplitN(rest, "#", 2)[0])[0]
|
||||||
|
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||||
|
st.linkLine = line
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if st.work == "" || st.asmObj == "" || st.pkgArch == "" || st.linkLine == "" {
|
||||||
|
t.Fatalf("could not locate the build steps (work=%q asmObj=%q pkgArch=%q link=%q):\n%s",
|
||||||
|
st.work, st.asmObj, st.pkgArch, st.linkLine, log)
|
||||||
|
}
|
||||||
|
st.asmObj = strings.ReplaceAll(st.asmObj, "$WORK", st.work)
|
||||||
|
st.pkgArch = strings.ReplaceAll(st.pkgArch, "$WORK", st.work)
|
||||||
|
return st
|
||||||
|
}
|
||||||
|
|
||||||
|
// substituteAndRelink swaps the gasm object into the package archive the
|
||||||
|
// baseline build produced and re-runs the captured link line against the
|
||||||
|
// rebuilt archive, writing the binary to outBin (the -x log's link step
|
||||||
|
// always targets the action graph's internal a.out, which the helper
|
||||||
|
// redirects; the copy to the -o target is a separate build action the helper
|
||||||
|
// does not need). The archive handed to the linker is proven to carry the
|
||||||
|
// gasm object byte for byte, so a build-layout change that skipped the
|
||||||
|
// substitution fails here instead of passing vacuously.
|
||||||
|
func substituteAndRelink(t *testing.T, goBin, dir string, st buildLogSteps, outBin string, gasmObj []byte, extraEnv ...string) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
// The deliberate-run boundary: this path drives a real `go build` and
|
||||||
|
// cmd/link per invocation, minutes-scale work on the small single-core
|
||||||
|
// CI runner. Under -short (the push pipeline's mode) it skips; the
|
||||||
|
// local test gate and the dispatched workflows run it in full.
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("end-to-end go build and link: skipped in -short mode")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Extract the archive, overwrite the assembler's member with the gasm
|
||||||
|
// object and repack (go tool pack has no replace-in-place).
|
||||||
|
membersDir := filepath.Join(dir, "members")
|
||||||
|
if err := os.MkdirAll(membersDir, 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
extract := exec.Command(goBin, "tool", "pack", "x", st.pkgArch)
|
||||||
|
extract.Dir = membersDir
|
||||||
|
if out, err := extract.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("pack x: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
member := filepath.Join(membersDir, filepath.Base(st.asmObj))
|
||||||
|
if _, err := os.Stat(member); err != nil {
|
||||||
|
t.Fatalf("the assembler's archive member was not extracted: %v", err)
|
||||||
|
}
|
||||||
|
if err := os.Chmod(member, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(member, gasmObj, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
listCmd := exec.Command(goBin, "tool", "pack", "t", st.pkgArch)
|
||||||
|
listOut, err := listCmd.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("pack t: %v\n%s", err, listOut)
|
||||||
|
}
|
||||||
|
newArch := filepath.Join(dir, "pkg.a")
|
||||||
|
args := []string{"tool", "pack", "c", newArch}
|
||||||
|
seen := map[string]bool{}
|
||||||
|
for m := range strings.FieldsSeq(string(listOut)) {
|
||||||
|
if seen[m] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
seen[m] = true
|
||||||
|
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
args = append(args, m)
|
||||||
|
}
|
||||||
|
pack := exec.Command(goBin, args...)
|
||||||
|
pack.Dir = membersDir
|
||||||
|
if out, err := pack.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("pack c: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Prove the substitution: the archive the linker is about to consume
|
||||||
|
// holds the gasm object, byte for byte.
|
||||||
|
checkDir := filepath.Join(dir, "check")
|
||||||
|
if err := os.MkdirAll(checkDir, 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
check := exec.Command(goBin, "tool", "pack", "x", newArch)
|
||||||
|
check.Dir = checkDir
|
||||||
|
if out, err := check.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("pack x (verification): %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
got, err := os.ReadFile(filepath.Join(checkDir, filepath.Base(st.asmObj)))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read the substituted member back: %v", err)
|
||||||
|
}
|
||||||
|
if !bytes.Equal(got, gasmObj) {
|
||||||
|
t.Fatal("the repacked archive does not carry the gasm object")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Re-link. The line carries a GOROOT assignment and $WORK placeholders;
|
||||||
|
// GOEXPERIMENT must match the toolchain's own, because the linker
|
||||||
|
// compares the object header against its configuration.
|
||||||
|
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
|
||||||
|
linkLine := strings.ReplaceAll(st.linkLine, "$WORK", st.work)
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, filepath.Join(st.work, "b001", "_pkg_.a"), newArch)
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, filepath.Join(st.work, "b001", "exe", "a.out"), outBin)
|
||||||
|
env := append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)))
|
||||||
|
env = append(env, extraEnv...)
|
||||||
|
link := exec.Command("sh", "-c", linkLine)
|
||||||
|
link.Dir = dir
|
||||||
|
link.Env = env
|
||||||
|
if out, err := link.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("link with the gasm object: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGOObjectExternalPackageLink is the cross-package end-to-end check: a
|
||||||
|
// GOOBJ whose code references a real external package symbol (runtime's
|
||||||
|
// morestack, a plain reference rather than the builtin noctxt form) must
|
||||||
|
// carry a package index that points past the blkPkgIdx table's dummy entry
|
||||||
|
// 0, and the object must link against the real runtime. Pre-fix, the
|
||||||
|
// relocations carried block index 0, which the loader never fills, so the
|
||||||
|
// reference resolved against whatever object was loaded first and the link
|
||||||
|
// failed. The binary is not run: morestack returns to the call site's
|
||||||
|
// stack check, which a hand-written caller has none of.
|
||||||
|
func TestGOObjectExternalPackageLink(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("end-to-end go build and link: skipped in -short mode")
|
||||||
|
}
|
||||||
|
goBin, err := exec.LookPath("go")
|
||||||
|
if err != nil {
|
||||||
|
t.Skip("no Go toolchain available")
|
||||||
|
}
|
||||||
|
dir := t.TempDir()
|
||||||
|
|
||||||
|
const asmSrc = `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·fn(SB), NOSPLIT, $0-0
|
||||||
|
CALL ·helper(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·helper(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
const mainSrc = `package main
|
||||||
|
|
||||||
|
func fn()
|
||||||
|
func helper()
|
||||||
|
|
||||||
|
func main() {
|
||||||
|
fn()
|
||||||
|
helper()
|
||||||
|
}
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module extlink\n\ngo 1.27\n"), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Capture the build the toolchain performs and re-run only its link
|
||||||
|
// step with our object swapped into the package archive, mirroring
|
||||||
|
// TestGOObjectLinkAndRun.
|
||||||
|
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||||
|
build.Dir = dir
|
||||||
|
buildLog, err := build.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||||
|
}
|
||||||
|
var work, linkLine, asmObj, pkgArch string
|
||||||
|
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(line, "WORK="):
|
||||||
|
work = strings.TrimPrefix(line, "WORK=")
|
||||||
|
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
|
||||||
|
asmObj = fieldAfter(line, "-o")
|
||||||
|
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
|
||||||
|
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
|
||||||
|
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
|
||||||
|
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||||
|
linkLine = line
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if work == "" || asmObj == "" || pkgArch == "" || linkLine == "" {
|
||||||
|
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
|
||||||
|
}
|
||||||
|
defer os.RemoveAll(work)
|
||||||
|
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
|
||||||
|
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
|
||||||
|
|
||||||
|
// Assemble the source with gasm, then retarget fn's internal call at
|
||||||
|
// a real external package symbol: the reloc's qualified name drives
|
||||||
|
// the export-data resolution the way a source-level runtime·sym(SB)
|
||||||
|
// reference would.
|
||||||
|
f, errs := parser.Parse("main_amd64.s", asmSrc)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
fn := &img.Funcs[0]
|
||||||
|
for i := range fn.Relocs {
|
||||||
|
fn.Relocs[i].Name = "runtime\u00b7morestack"
|
||||||
|
fn.Relocs[i].External = true
|
||||||
|
}
|
||||||
|
img.Externals = []string{"runtime\u00b7morestack"}
|
||||||
|
obj, err := img.GOObject("main", "main_amd64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObject: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Structural check: the blkPkgIdx block reserves entry 0 for the
|
||||||
|
// dummy invalid package and places runtime at entry 1, and fn's call
|
||||||
|
// relocation carries PkgIdx 1.
|
||||||
|
v := openGoobj(t, obj)
|
||||||
|
pkgBlk := v.blk(blkPkgIdx)
|
||||||
|
if len(pkgBlk) != 2*8 {
|
||||||
|
t.Fatalf("blkPkgIdx = %d bytes, want two entries", len(pkgBlk))
|
||||||
|
}
|
||||||
|
le := binary.LittleEndian
|
||||||
|
strEntry := func(i int) string {
|
||||||
|
e := pkgBlk[i*8 : (i+1)*8]
|
||||||
|
return v.str(le.Uint32(e[4:]), le.Uint32(e[0:]))
|
||||||
|
}
|
||||||
|
if s := strEntry(0); s != "" {
|
||||||
|
t.Errorf("blkPkgIdx[0] = %q, want the dummy empty package", s)
|
||||||
|
}
|
||||||
|
if s := strEntry(1); s != "runtime" {
|
||||||
|
t.Errorf("blkPkgIdx[1] = %q, want runtime", s)
|
||||||
|
}
|
||||||
|
relocs := v.blk(blkReloc)
|
||||||
|
// fn is the last non-package symbol (two functions, four pc tables
|
||||||
|
// each); its one reloc is the final record.
|
||||||
|
fnRec := relocs[len(relocs)-23:]
|
||||||
|
if pIdx := le.Uint32(fnRec[15:]); pIdx != 1 {
|
||||||
|
t.Errorf("external reloc PkgIdx = %d, want 1 (runtime)", pIdx)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Swap the object into the package archive and link with cmd/link;
|
||||||
|
// the link line consumes the archive, not the loose object file.
|
||||||
|
membersDir := filepath.Join(dir, "members")
|
||||||
|
if err := os.MkdirAll(membersDir, 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
|
||||||
|
extract.Dir = membersDir
|
||||||
|
if out, err := extract.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("pack x: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
member := filepath.Join(membersDir, filepath.Base(asmObj))
|
||||||
|
if err := os.Chmod(member, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(member, obj, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
|
||||||
|
listOut, err := listCmd.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("pack t: %v\n%s", err, listOut)
|
||||||
|
}
|
||||||
|
newArch := filepath.Join(dir, "pkg.a")
|
||||||
|
args := []string{"tool", "pack", "c", newArch}
|
||||||
|
seen := map[string]bool{}
|
||||||
|
for m := range strings.FieldsSeq(string(listOut)) {
|
||||||
|
if seen[m] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
seen[m] = true
|
||||||
|
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
args = append(args, filepath.Join(membersDir, m))
|
||||||
|
}
|
||||||
|
pack := exec.Command(goBin, args...)
|
||||||
|
pack.Dir = membersDir
|
||||||
|
if out, err := pack.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("pack c: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, pkgArch, newArch)
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "prog2"))
|
||||||
|
linkCmd := exec.Command("sh", "-c", "cd "+dir+" && "+linkLine)
|
||||||
|
if out, err := linkCmd.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The call must have resolved to the real runtime symbol.
|
||||||
|
dump, err := exec.Command(goBin, "tool", "objdump", "-s", "main.fn", filepath.Join(dir, "prog2")).CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("objdump main.fn: %v\n%s", err, dump)
|
||||||
|
}
|
||||||
|
if !bytes.Contains(dump, []byte("runtime.morestack")) {
|
||||||
|
t.Errorf("main.fn does not call runtime.morestack:\n%s", dump)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+24
-3
@@ -18,19 +18,40 @@ import (
|
|||||||
// reloc/aux/data index arrays, with the arm64 preamble, the MinLC of 4
|
// reloc/aux/data index arrays, with the arm64 preamble, the MinLC of 4
|
||||||
// for the pc-value deltas, and the arm64 relocation types for the ADRP
|
// for the pc-value deltas, and the arm64 relocation types for the ADRP
|
||||||
// pairs and BL calls.
|
// pairs and BL calls.
|
||||||
|
//
|
||||||
|
// The toolchain records one relocation per ADRP pair: a single R_ADDRARM64
|
||||||
|
// or R_ARM64_PCREL_LDST64 of Siz 8 at the ADRP word, from which the linker
|
||||||
|
// patches both instructions of the pair (cmd/internal/obj/arm64/asm7.go,
|
||||||
|
// the ADRP cases: one AddRel with Off at the pair's pc and Siz 8). gasm's
|
||||||
|
// assembler records the ADRP+ADD form as two word relocs, so the second
|
||||||
|
// word's twin is dropped here before emission.
|
||||||
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
|
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
|
||||||
pre, err := toolchainObjectPreambleAARCH64()
|
pre, err := toolchainObjectPreambleAARCH64()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
|
coalesced := *img
|
||||||
|
coalesced.Funcs = append([]FuncLayout(nil), img.Funcs...)
|
||||||
|
for i := range coalesced.Funcs {
|
||||||
|
rs := coalesced.Funcs[i].Relocs
|
||||||
|
var keep []Reloc
|
||||||
|
for j := 0; j < len(rs); j++ {
|
||||||
|
keep = append(keep, rs[j])
|
||||||
|
if rs[j].Kind == RelArm64Addr && j+1 < len(rs) &&
|
||||||
|
rs[j+1].Kind == RelArm64Addr && rs[j+1].Off == rs[j].Off+4 {
|
||||||
|
j++ // the ADD word's twin: the Siz-8 pair reloc covers it
|
||||||
|
}
|
||||||
|
}
|
||||||
|
coalesced.Funcs[i].Relocs = keep
|
||||||
|
}
|
||||||
|
return coalesced.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
|
||||||
switch r.Kind {
|
switch r.Kind {
|
||||||
case RelArm64Branch:
|
case RelArm64Branch:
|
||||||
return relocArm64Branch, 4
|
return relocArm64Branch, 4
|
||||||
case RelArm64LDST64:
|
case RelArm64LDST64:
|
||||||
return relocArm64LDST64, 4
|
return relocArm64LDST64, 8
|
||||||
default:
|
default:
|
||||||
return relocArm64Addr, 4
|
return relocArm64Addr, 8
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|||||||
+52
-6
@@ -6,10 +6,11 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
"encoding/hex"
|
"encoding/hex"
|
||||||
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64,
|
// The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64,
|
||||||
@@ -27,6 +28,12 @@ func TestStackGuardBytes(t *testing.T) {
|
|||||||
"644c8b3425000000004c8da42478ffffff4d3b66107614554889e54881ec000100004881c4000100005dc3e800000000ebce"},
|
"644c8b3425000000004c8da42478ffffff4d3b66107614554889e54881ec000100004881c4000100005dc3e800000000ebce"},
|
||||||
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
|
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
|
||||||
"644c8b3425000000004989e44981ec881f0000721a4d3b66107614554889e54881ec002000004881c4002000005dc3e800000000ebca"},
|
"644c8b3425000000004989e44981ec881f0000721a4d3b66107614554889e54881ec002000004881c4002000005dc3e800000000ebca"},
|
||||||
|
// Class 2 with a body long enough that the underflow JB relaxes to
|
||||||
|
// rel32: its displacement must span the real 6-byte JB, else the
|
||||||
|
// branch lands 4 bytes past the morestack block, inside the CALL
|
||||||
|
// displacement field.
|
||||||
|
{"leafbiglong", "TEXT \u00b7leafbiglong(SB), $8192-0\n" + strings.Repeat("\tMOVQ AX, BX\n", 40) + "\tRET\n",
|
||||||
|
"644c8b3425000000004989e44981ec881f00000f82960000004d3b66100f868c000000554889e54881ec00200000" + strings.Repeat("4889c3", 40) + "4881c4002000005dc3e800000000e947ffffff"},
|
||||||
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
|
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
|
||||||
"644c8b342500000000493b66107613554889e54883ec10e8000000004883c4105dc3e800000000ebd7"},
|
"644c8b342500000000493b66107613554889e54883ec10e8000000004883c4105dc3e800000000ebd7"},
|
||||||
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
|
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
|
||||||
@@ -145,6 +152,45 @@ func TestStackGuardBytesARM64(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestStackGuardBranchTargetsARM64 checks the class-2 guard's branch
|
||||||
|
// positions for a frame whose guard constant needs two MOV words: the
|
||||||
|
// displacements must be computed from byte offsets (8+4*ml and 16+4*ml), so
|
||||||
|
// both branches land on the morestack block rather than inside the body.
|
||||||
|
// The frame size makes the toolchain switch its own prologue decomposition,
|
||||||
|
// so the assertion is on the branch targets, not pinned bytes.
|
||||||
|
func TestStackGuardBranchTargetsARM64(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("g_arm64.s", "TEXT \u00b7f(SB), $65664-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
fn := img.Funcs[0]
|
||||||
|
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||||
|
if len(code)%4 != 0 {
|
||||||
|
t.Fatalf("function size %d is not a word multiple", len(code))
|
||||||
|
}
|
||||||
|
// autosize = 65680, so the guard materialises 65552 = MOVZ+MOVK: ml = 2
|
||||||
|
// and the branches sit at bytes 16 and 24 of the guard prefix.
|
||||||
|
const morestackBlock = 12 // MOVD R30, R3; BL; B back
|
||||||
|
blockStart := len(code) - morestackBlock
|
||||||
|
check := func(name string, off int) {
|
||||||
|
t.Helper()
|
||||||
|
w := leWord(code[off:])
|
||||||
|
imm19 := int32(w>>5) & 0x7FFFF
|
||||||
|
if imm19&(1<<18) != 0 {
|
||||||
|
imm19 -= 1 << 19
|
||||||
|
}
|
||||||
|
if target := off + int(imm19)*4; target != blockStart {
|
||||||
|
t.Errorf("%s at byte %d targets byte %d, want the morestack block at %d", name, off, target, blockStart)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
check("B.LO", 16)
|
||||||
|
check("B.LS", 24)
|
||||||
|
}
|
||||||
|
|
||||||
// The riscv64 stack-split guard, pinned from `go tool asm` (Go 1.27,
|
// The riscv64 stack-split guard, pinned from `go tool asm` (Go 1.27,
|
||||||
// riscv64): the morestack call sits between the guard and the body, and the
|
// riscv64): the morestack call sits between the guard and the body, and the
|
||||||
// guard branches forward over it. Relocation fields are masked.
|
// guard branches forward over it. Relocation fields are masked.
|
||||||
@@ -261,12 +307,12 @@ func TestStackGuardBytesLOONG64(t *testing.T) {
|
|||||||
func TestStackGuardGOObjInternalCall(t *testing.T) {
|
func TestStackGuardGOObjInternalCall(t *testing.T) {
|
||||||
for _, tt := range []struct {
|
for _, tt := range []struct {
|
||||||
src string
|
src string
|
||||||
assemble func(*ast.File) (*Image, error)
|
assemble func(*ast.File, ...AssembleOption) (*Image, error)
|
||||||
}{
|
}{
|
||||||
{"g_amd64.s", AssembleFile},
|
{"g_amd64.s", AssembleFile},
|
||||||
{"g_arm64.s", AssembleFileARM64},
|
{"g_arm64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileARM64(f) }},
|
||||||
{"g_riscv64.s", AssembleFileRISCV},
|
{"g_riscv64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileRISCV(f) }},
|
||||||
{"g_loong64.s", AssembleFileLOONG64},
|
{"g_loong64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileLOONG64(f) }},
|
||||||
} {
|
} {
|
||||||
f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
|
f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
if len(errs) > 0 {
|
if len(errs) > 0 {
|
||||||
|
|||||||
+1112
-64
File diff suppressed because it is too large
Load Diff
@@ -16,16 +16,16 @@ import (
|
|||||||
|
|
||||||
"golang.org/x/arch/x86/x86asm"
|
"golang.org/x/arch/x86/x86asm"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel —
|
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel;
|
||||||
// all functions plus the file-local mask24 constant — and checks that every
|
// all functions plus the file-local mask24 constant; and checks that every
|
||||||
// static-symbol load resolves to the right bytes in the image.
|
// static-symbol load resolves to the right bytes in the image.
|
||||||
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
||||||
path := "../../go-libraries/go-flac/avx2_amd64.s"
|
path := "../../go-libraries/go-flac/avx2_amd64.s"
|
||||||
if _, err := os.Stat(path); err != nil {
|
if _, err := os.Stat(path); err != nil {
|
||||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
t.Skip("go-libraries repository not present next to gasm-sdk")
|
||||||
}
|
}
|
||||||
src, err := os.ReadFile(path)
|
src, err := os.ReadFile(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -81,12 +81,12 @@ func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
|
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
|
||||||
// kernel — all functions plus the file-global idx16 constant — and checks
|
// kernel, all functions plus the file-global idx16 constant, and checks
|
||||||
// that the static-symbol load resolves to the right bytes in the image.
|
// that the static-symbol load resolves to the right bytes in the image.
|
||||||
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
|
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
|
||||||
path := "../../go-libraries/go-flac/avx512_amd64.s"
|
path := "../../go-libraries/go-flac/avx512_amd64.s"
|
||||||
if _, err := os.Stat(path); err != nil {
|
if _, err := os.Stat(path); err != nil {
|
||||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
t.Skip("go-libraries repository not present next to gasm-sdk")
|
||||||
}
|
}
|
||||||
src, err := os.ReadFile(path)
|
src, err := os.ReadFile(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
@@ -0,0 +1,228 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/binary"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"runtime"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// The differential kernels for the DATA-path and front-end gaps are kept in
|
||||||
|
// testdata/verify beside the campaign's other kernels; the verify package's
|
||||||
|
// suites are not open to the asm package, so this test is their runner: each
|
||||||
|
// kernel assembles through gasm and through go tool asm, and the functions'
|
||||||
|
// bytes must agree with the relocation sites masked on both sides.
|
||||||
|
|
||||||
|
// toolAsmObject assembles path with the installed toolchain's assembler for
|
||||||
|
// goarch ("" = the host) and returns the object bytes. Every live-oracle
|
||||||
|
// comparison funnels through here, so this is also where the deliberate-run
|
||||||
|
// boundary sits: under -short (the push pipeline's mode) the comparisons
|
||||||
|
// skip, because each spawns a go tool asm subprocess and the small single-
|
||||||
|
// core runner pays seconds per spawn. The encodings stay pinned by the
|
||||||
|
// golden-byte tests in every mode; the live oracle runs in the local test
|
||||||
|
// gate and the dispatched workflows.
|
||||||
|
func toolAsmObject(t *testing.T, path, goarch string) []byte {
|
||||||
|
t.Helper()
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("live go tool asm oracle: skipped in -short mode")
|
||||||
|
}
|
||||||
|
goBin, err := exec.LookPath("go")
|
||||||
|
if err != nil {
|
||||||
|
t.Skip("no Go toolchain available")
|
||||||
|
}
|
||||||
|
out, err := exec.Command(goBin, "env", "GOROOT").Output()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("go env GOROOT: %v", err)
|
||||||
|
}
|
||||||
|
includeDir := filepath.Join(strings.TrimSpace(string(out)), "pkg", "include")
|
||||||
|
|
||||||
|
pkg := strings.TrimSuffix(filepath.Base(path), ".s")
|
||||||
|
pkg = strings.TrimSuffix(pkg, "_amd64")
|
||||||
|
pkg = strings.TrimSuffix(pkg, "_arm64")
|
||||||
|
|
||||||
|
objPath := filepath.Join(t.TempDir(), "oracle.o")
|
||||||
|
cmd := exec.Command(goBin, "tool", "asm", "-I", includeDir, "-p", pkg, "-o", objPath, path)
|
||||||
|
if goarch != "" {
|
||||||
|
environ := os.Environ()
|
||||||
|
env := make([]string, 0, len(environ)+1)
|
||||||
|
for _, e := range environ {
|
||||||
|
if !strings.HasPrefix(e, "GOARCH=") {
|
||||||
|
env = append(env, e)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
cmd.Env = append(env, "GOARCH="+goarch)
|
||||||
|
}
|
||||||
|
if out, err := cmd.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("go tool asm %s: %v\n%s", filepath.Base(path), err, out)
|
||||||
|
}
|
||||||
|
obj, err := os.ReadFile(objPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return obj
|
||||||
|
}
|
||||||
|
|
||||||
|
// oracleFuncCode extracts the non-package TEXT functions' code bytes from a
|
||||||
|
// toolchain object, keyed by the name the object records (pkg.name). Each
|
||||||
|
// function's span is its own symbol size: a toolchain object that follows
|
||||||
|
// the text with data symbols (the synthesised float-constant pool) would
|
||||||
|
// otherwise fold them into the last function's bytes.
|
||||||
|
func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
|
||||||
|
t.Helper()
|
||||||
|
v := openGoobj(t, obj)
|
||||||
|
le := binary.LittleEndian
|
||||||
|
const symSize = 21
|
||||||
|
nps := v.syms(blkNonpkgdef)
|
||||||
|
data := v.blk(blkData)
|
||||||
|
didx := v.blk(blkDataIdx)
|
||||||
|
preceding := 0
|
||||||
|
for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef} {
|
||||||
|
preceding += len(v.blk(bi)) / symSize
|
||||||
|
}
|
||||||
|
out := make(map[string][]byte, len(nps))
|
||||||
|
for i, s := range nps {
|
||||||
|
if s.typ != kindSTEXT {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
start := le.Uint32(didx[4*(preceding+i):])
|
||||||
|
out[s.name] = data[start : start+s.size]
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// maskCode zeroes every relocation field, the way the toolchain's object
|
||||||
|
// leaves them for the linker.
|
||||||
|
func maskCode(code []byte, relocs []Reloc) []byte {
|
||||||
|
for _, r := range relocs {
|
||||||
|
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||||
|
code[j] = 0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return code
|
||||||
|
}
|
||||||
|
|
||||||
|
// code assembles src for amd64 and returns the image's code bytes.
|
||||||
|
func code(path, src string) []byte {
|
||||||
|
f, errs := parser.Parse(path, src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return img.Code
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDifferentialKernels pins the new kernels against the oracle.
|
||||||
|
func TestDifferentialKernels(t *testing.T) {
|
||||||
|
if runtime.GOARCH != "amd64" {
|
||||||
|
t.Skip("the amd64 kernels assume an amd64 host assembler default")
|
||||||
|
}
|
||||||
|
for _, k := range []struct {
|
||||||
|
path string
|
||||||
|
goarch string
|
||||||
|
arm64 bool
|
||||||
|
}{
|
||||||
|
{filepath.Join("..", "testdata", "verify", "datarel_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "divslash_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "semicolons_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "forms_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true},
|
||||||
|
} {
|
||||||
|
t.Run(filepath.Base(k.path), func(t *testing.T) {
|
||||||
|
src, err := os.ReadFile(k.path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read: %v", err)
|
||||||
|
}
|
||||||
|
f, errs := parser.Parse(k.path, string(src))
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
var img *Image
|
||||||
|
if k.arm64 {
|
||||||
|
img, err = AssembleFileARM64(f)
|
||||||
|
} else {
|
||||||
|
img, err = AssembleFile(f)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
gt := oracleFuncCode(t, toolAsmObject(t, k.path, k.goarch))
|
||||||
|
// The oracle keys its functions by the qualified object name
|
||||||
|
// (pkg.name); match on the local part.
|
||||||
|
byLocal := make(map[string][]byte, len(gt))
|
||||||
|
for name, code := range gt {
|
||||||
|
if _, after, ok := strings.Cut(name, "."); ok {
|
||||||
|
name = after
|
||||||
|
}
|
||||||
|
byLocal[name] = code
|
||||||
|
}
|
||||||
|
|
||||||
|
matched := 0
|
||||||
|
for _, fn := range img.Funcs {
|
||||||
|
gasmCode := maskCode(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
|
||||||
|
goCode, ok := byLocal[fn.Name]
|
||||||
|
if !ok {
|
||||||
|
t.Errorf("%s: not in ground truth (%d functions: %v)", fn.Name, len(gt), keysOf(byLocal))
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
goCode = maskCode(append([]byte(nil), goCode...), fn.Relocs)
|
||||||
|
cmpLen := min(len(goCode), len(gasmCode))
|
||||||
|
if !bytes.Equal(gasmCode[:cmpLen], goCode[:cmpLen]) {
|
||||||
|
t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\ngasm %x\ngo %x", fn.Name, len(gasmCode), len(goCode), gasmCode, goCode)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for _, b := range goCode[len(gasmCode):] {
|
||||||
|
if b != 0 {
|
||||||
|
t.Errorf("%s: non-zero trailing bytes in go tool asm output", fn.Name)
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
matched++
|
||||||
|
t.Logf("%s: MATCH (%d bytes)", fn.Name, len(gasmCode))
|
||||||
|
}
|
||||||
|
if matched == 0 {
|
||||||
|
t.Fatal("no functions matched")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func keysOf(m map[string][]byte) []string {
|
||||||
|
out := make([]string, 0, len(m))
|
||||||
|
for k := range m {
|
||||||
|
out = append(out, k)
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSemicolonSpellingParity pins that the ';' statement separator changes
|
||||||
|
// nothing about the encoding: the one-line spelling assembles to exactly the
|
||||||
|
// bytes of the same statements written one per line.
|
||||||
|
func TestSemicolonSpellingParity(t *testing.T) {
|
||||||
|
for _, tt := range []struct{ one, two string }{
|
||||||
|
{"\tROLQ $3, DI; ROLQ $13, DI\n", "\tROLQ $3, DI\n\tROLQ $13, DI\n"},
|
||||||
|
{"\tREP; MOVSQ\n", "\tREP\n\tMOVSQ\n"},
|
||||||
|
{"\tXORQ AX, AX; XORQ CX, CX\n", "\tXORQ AX, AX\n\tXORQ CX, CX\n"},
|
||||||
|
} {
|
||||||
|
one := code("t.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n"+tt.one+"\tRET\n")
|
||||||
|
two := code("t.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n"+tt.two+"\tRET\n")
|
||||||
|
if !bytes.Equal(one, two) {
|
||||||
|
t.Errorf("semicolon spelling %q: %x, want the two-line bytes %x", tt.one, one, two)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -12,7 +12,7 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
|
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
|
||||||
@@ -94,9 +94,9 @@ DATA ·table<>+0(SB)/8, $0x1122334455667788
|
|||||||
}
|
}
|
||||||
|
|
||||||
// The debug_line program: LNE_set_address (the R_ADDR relocation
|
// The debug_line program: LNE_set_address (the R_ADDR relocation
|
||||||
// carries the function address), then one row per line change — the
|
// carries the function address), then one row per line change; the
|
||||||
// TEXT is on line 4 (a leading blank line precedes the include), the
|
// TEXT is on line 4 (a leading blank line precedes the include), the
|
||||||
// instructions on lines 5–9 — an advance to the 20-byte end and an
|
// instructions on lines 5-9; an advance to the 20-byte end and an
|
||||||
// end-of-sequence.
|
// end-of-sequence.
|
||||||
linesOff := le.Uint32(dataIdx[4*2:])
|
linesOff := le.Uint32(dataIdx[4*2:])
|
||||||
lines := dataBlk[linesOff : linesOff+21]
|
lines := dataBlk[linesOff : linesOff+21]
|
||||||
|
|||||||
+266
-77
@@ -5,10 +5,11 @@ package asm
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"math"
|
||||||
"sort"
|
"sort"
|
||||||
"strconv"
|
"strconv"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Image is an assembled file: the function bodies laid out in source order,
|
// Image is an assembled file: the function bodies laid out in source order,
|
||||||
@@ -25,6 +26,10 @@ type Image struct {
|
|||||||
Symbols map[string]int // static symbol → byte offset within the image
|
Symbols map[string]int // static symbol → byte offset within the image
|
||||||
DataSyms []DataSymbol // GLOBL symbols, in layout order
|
DataSyms []DataSymbol // GLOBL symbols, in layout order
|
||||||
Externals []string // referenced but undefined symbols, sorted
|
Externals []string // referenced but undefined symbols, sorted
|
||||||
|
// SourcePath is the assembled file's path, recorded in the DWARF
|
||||||
|
// sections in place of a placeholder name. Empty when the image was
|
||||||
|
// not built from a named file.
|
||||||
|
SourcePath string
|
||||||
}
|
}
|
||||||
|
|
||||||
// FuncLayout describes one assembled function within an Image.
|
// FuncLayout describes one assembled function within an Image.
|
||||||
@@ -81,12 +86,9 @@ func (fl *FuncLayout) LineAt(offset int) int {
|
|||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
// RelocKind Reloc is one static-symbol reference within a function body: the disp32
|
// RelocKind discriminates the relocation a static-symbol reference needs;
|
||||||
// field at Off (function-relative) must reach the symbol plus Addend,
|
// the encoders record one per SB reference, and the object-file emitters map
|
||||||
// measured from After, the address just past the instruction. An External
|
// it to their format's relocation type.
|
||||||
// relocation names a symbol no GLOBL in the file defines; the object-file
|
|
||||||
// emitters carry it into the output's relocation table.
|
|
||||||
// RelocKind discriminates the type of relocation needed.
|
|
||||||
type RelocKind int
|
type RelocKind int
|
||||||
|
|
||||||
const (
|
const (
|
||||||
@@ -96,22 +98,33 @@ const (
|
|||||||
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
|
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
|
||||||
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
|
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
|
||||||
RelRISCVJal // R_RISCV_JAL (J-type call)
|
RelRISCVJal // R_RISCV_JAL (J-type call)
|
||||||
RelPCRelAbs // 32-bit absolute (R_RISCV_32)
|
|
||||||
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
|
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
|
||||||
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
|
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
|
||||||
RelArm64Addr // R_ADDRARM64 (ADRP + ADD pair)
|
RelArm64Addr // R_ADDRARM64 (ADRP + ADD pair)
|
||||||
RelArm64Branch // R_CALLARM64 (BL instruction)
|
RelArm64Branch // R_CALLARM64 (BL instruction)
|
||||||
RelArm64LDST64 // R_ARM64_PCREL_LDST64 (ADRP + 64-bit LDR/STR pair)
|
RelArm64LDST64 // R_ARM64_PCREL_LDST64 (ADRP + 64-bit LDR/STR pair)
|
||||||
RelLoong64Branch // R_CALLLOONG64 (BL instruction)
|
RelLoong64Branch // R_CALLLOONG64 (BL instruction)
|
||||||
|
RelAddr // R_ADDR: the absolute address of a symbol held in a DATA field
|
||||||
)
|
)
|
||||||
|
|
||||||
type Reloc struct {
|
type Reloc struct {
|
||||||
|
// Off is the function-relative offset of the field the linker patches
|
||||||
|
// and After the address just past the instruction, the base the
|
||||||
|
// assembler measures PC-relative displacements from. Name plus
|
||||||
|
// Addend select the target: the symbol plus the byte offset. An
|
||||||
|
// External relocation names a symbol no GLOBL in the file defines;
|
||||||
|
// the object-file emitters carry it into the output's relocation
|
||||||
|
// table. Siz is the width of the patched field and is set only for
|
||||||
|
// data-field relocations (RelAddr, Off relative to the data symbol),
|
||||||
|
// whose width is the DATA line's; code relocations take their width
|
||||||
|
// from the architecture's instruction encoding.
|
||||||
Off int
|
Off int
|
||||||
After int
|
After int
|
||||||
Name string
|
Name string
|
||||||
Addend int64
|
Addend int64
|
||||||
External bool
|
External bool
|
||||||
Kind RelocKind
|
Kind RelocKind
|
||||||
|
Siz uint8
|
||||||
}
|
}
|
||||||
|
|
||||||
// DataSymbol describes one GLOBL symbol laid out in the data section.
|
// DataSymbol describes one GLOBL symbol laid out in the data section.
|
||||||
@@ -121,8 +134,14 @@ type DataSymbol struct {
|
|||||||
Offset int // byte offset within Data
|
Offset int // byte offset within Data
|
||||||
Size int
|
Size int
|
||||||
Static bool // the <> marker: file-local, not exported
|
Static bool // the <> marker: file-local, not exported
|
||||||
Rodata bool // the RODATA flag: read-only data
|
Rodata bool // the RODATA flag: read-only data (implies no pointers)
|
||||||
|
Noptr bool // the NOPTR flag: data with no pointers, kept out of GC scanning
|
||||||
Dupok bool // the DUPOK flag: duplicate-OK
|
Dupok bool // the DUPOK flag: duplicate-OK
|
||||||
|
// Relocs carries the symbol-valued DATA initialisers ("DATA s+0(SB)/8,
|
||||||
|
// $other(SB)"): fields of this symbol's data that hold another symbol's
|
||||||
|
// address, resolved by the linker. Off is relative to the symbol's
|
||||||
|
// data start.
|
||||||
|
Relocs []Reloc
|
||||||
}
|
}
|
||||||
|
|
||||||
// Bytes returns the whole image: code, then data.
|
// Bytes returns the whole image: code, then data.
|
||||||
@@ -132,6 +151,18 @@ func (img *Image) Bytes() []byte {
|
|||||||
return append(out, img.Data...)
|
return append(out, img.Data...)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// AssembleOption adjusts the file-level assembly context.
|
||||||
|
type AssembleOption func(*linkInfo)
|
||||||
|
|
||||||
|
// WithGOOS selects the target operating system for the forms that depend on
|
||||||
|
// it, the TLS access shape above all: linux and freebsd take the
|
||||||
|
// one-instruction form, windows and plan9 keep the two-instruction load.
|
||||||
|
func WithGOOS(goos string) AssembleOption {
|
||||||
|
return func(l *linkInfo) {
|
||||||
|
l.goos = goos
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// AssembleFile assembles every TEXT function of a parsed file and lays out
|
// AssembleFile assembles every TEXT function of a parsed file and lays out
|
||||||
// its static symbols (GLOBL/DATA) in a data section behind the code. Each
|
// its static symbols (GLOBL/DATA) in a data section behind the code. Each
|
||||||
// reference to a file-local static symbol becomes a RIP-relative load whose
|
// reference to a file-local static symbol becomes a RIP-relative load whose
|
||||||
@@ -139,7 +170,7 @@ func (img *Image) Bytes() []byte {
|
|||||||
// GLOBL defines is recorded as an external relocation (Externals) with its
|
// GLOBL defines is recorded as an external relocation (Externals) with its
|
||||||
// displacement left zero, the object-file emitters resolve it at link
|
// displacement left zero, the object-file emitters resolve it at link
|
||||||
// time, while the raw image (Bytes) cannot represent it.
|
// time, while the raw image (Bytes) cannot represent it.
|
||||||
func AssembleFile(f *ast.File) (*Image, error) {
|
func AssembleFile(f *ast.File, opts ...AssembleOption) (*Image, error) {
|
||||||
dataSyms, err := collectData(f)
|
dataSyms, err := collectData(f)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
@@ -148,9 +179,20 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
|||||||
for _, d := range dataSyms {
|
for _, d := range dataSyms {
|
||||||
known[d.name] = true
|
known[d.name] = true
|
||||||
}
|
}
|
||||||
|
// TEXT symbols are file-level definitions too: a symbol immediate
|
||||||
|
// ($fn(SB)) may name one, exactly as a data reference names a GLOBL.
|
||||||
|
for _, d := range f.Decls {
|
||||||
|
if t, ok := d.(*ast.Text); ok {
|
||||||
|
known[t.Name.Name] = true
|
||||||
|
}
|
||||||
|
}
|
||||||
link := &linkInfo{symbols: known, allowExternal: true}
|
link := &linkInfo{symbols: known, allowExternal: true}
|
||||||
|
for _, o := range opts {
|
||||||
|
o(link)
|
||||||
|
}
|
||||||
|
poolSeen := map[string]bool{}
|
||||||
|
|
||||||
img := &Image{Symbols: map[string]int{}}
|
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
||||||
textOff := map[string]int{}
|
textOff := map[string]int{}
|
||||||
type asmFunc struct {
|
type asmFunc struct {
|
||||||
name string
|
name string
|
||||||
@@ -162,7 +204,26 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
|||||||
if !ok {
|
if !ok {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
code, patches, labels, steps, lines, err := assemble(t, link)
|
code, patches, labels, steps, lines, pool, err := assemble(t, link)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||||
|
}
|
||||||
|
// The pooled floating-point constants join the declared data as
|
||||||
|
// read-only symbols, deduplicated across the file (the toolchain
|
||||||
|
// synthesises the same symbols into its rodata).
|
||||||
|
for _, entry := range pool {
|
||||||
|
if poolSeen[entry.name] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
poolSeen[entry.name] = true
|
||||||
|
dataSyms = append(dataSyms, dataSym{
|
||||||
|
name: entry.name,
|
||||||
|
buf: entry.data,
|
||||||
|
size: len(entry.data),
|
||||||
|
rodata: true,
|
||||||
|
dupok: true,
|
||||||
|
})
|
||||||
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||||
}
|
}
|
||||||
@@ -209,6 +270,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
|||||||
Size: len(d.buf),
|
Size: len(d.buf),
|
||||||
Static: d.static,
|
Static: d.static,
|
||||||
Rodata: d.rodata,
|
Rodata: d.rodata,
|
||||||
|
Noptr: d.noptr,
|
||||||
Dupok: d.dupok,
|
Dupok: d.dupok,
|
||||||
})
|
})
|
||||||
img.Data = append(img.Data, d.buf...)
|
img.Data = append(img.Data, d.buf...)
|
||||||
@@ -250,6 +312,22 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
|||||||
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
|
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// The data symbols' symbol-valued DATA fields resolve the same way the
|
||||||
|
// code references do: a name the file defines (GLOBL or TEXT) stays an
|
||||||
|
// internal reference the emitters resolve, anything else is external.
|
||||||
|
// img.DataSyms was laid out in dataSyms order, so the indexes line up.
|
||||||
|
for i := range img.DataSyms {
|
||||||
|
for _, r := range dataSyms[i].relocs {
|
||||||
|
reloc := r
|
||||||
|
if _, ok := img.Symbols[reloc.Name]; !ok {
|
||||||
|
if _, ok := textOff[reloc.Name]; !ok {
|
||||||
|
reloc.External = true
|
||||||
|
externals[reloc.Name] = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
img.DataSyms[i].Relocs = append(img.DataSyms[i].Relocs, reloc)
|
||||||
|
}
|
||||||
|
}
|
||||||
for name := range externals {
|
for name := range externals {
|
||||||
img.Externals = append(img.Externals, name)
|
img.Externals = append(img.Externals, name)
|
||||||
}
|
}
|
||||||
@@ -266,17 +344,34 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
// The pooled $i64 constants the wide MOV immediate loads refer to join
|
||||||
|
// the declared data as read-only symbols, deduplicated across the file
|
||||||
|
// (the toolchain synthesises the same symbols into its rodata).
|
||||||
|
litSeen := map[string]bool{}
|
||||||
|
|
||||||
img := &Image{Symbols: map[string]int{}}
|
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
||||||
for _, d := range f.Decls {
|
for _, d := range f.Decls {
|
||||||
t, ok := d.(*ast.Text)
|
t, ok := d.(*ast.Text)
|
||||||
if !ok {
|
if !ok {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
code, labels, relocs, lines, spadj, err := assembleRISCV(t)
|
code, labels, relocs, lines, spadj, lits, err := assembleRISCV(t)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||||
}
|
}
|
||||||
|
for _, lit := range lits {
|
||||||
|
if litSeen[lit.Name] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
litSeen[lit.Name] = true
|
||||||
|
dataSyms = append(dataSyms, dataSym{
|
||||||
|
name: lit.Name,
|
||||||
|
buf: lit.Data,
|
||||||
|
size: len(lit.Data),
|
||||||
|
rodata: true,
|
||||||
|
dupok: true,
|
||||||
|
})
|
||||||
|
}
|
||||||
fl := FuncLayout{
|
fl := FuncLayout{
|
||||||
Name: t.Name.Name,
|
Name: t.Name.Name,
|
||||||
Pkg: t.Name.Pkg,
|
Pkg: t.Name.Pkg,
|
||||||
@@ -320,6 +415,7 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
|
|||||||
Size: d.size,
|
Size: d.size,
|
||||||
Static: d.static,
|
Static: d.static,
|
||||||
Rodata: d.rodata,
|
Rodata: d.rodata,
|
||||||
|
Noptr: d.noptr,
|
||||||
Dupok: d.dupok,
|
Dupok: d.dupok,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
@@ -339,7 +435,7 @@ func AssembleFileLOONG64(f *ast.File) (*Image, error) {
|
|||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
img := &Image{Symbols: map[string]int{}}
|
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
||||||
for _, d := range f.Decls {
|
for _, d := range f.Decls {
|
||||||
t, ok := d.(*ast.Text)
|
t, ok := d.(*ast.Text)
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -392,6 +488,7 @@ func AssembleFileLOONG64(f *ast.File) (*Image, error) {
|
|||||||
Size: d.size,
|
Size: d.size,
|
||||||
Static: d.static,
|
Static: d.static,
|
||||||
Rodata: d.rodata,
|
Rodata: d.rodata,
|
||||||
|
Noptr: d.noptr,
|
||||||
Dupok: d.dupok,
|
Dupok: d.dupok,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
@@ -422,6 +519,23 @@ func markExternals(img *Image, dataSyms []dataSym) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// The declared data symbols carry the file's own relocations (the
|
||||||
|
// symbol-valued DATA fields); the layouts appended img.DataSyms in
|
||||||
|
// dataSyms order, so the indexes line up. The trailing entries (the
|
||||||
|
// pooled arm64 literals) have no source relocations.
|
||||||
|
for i := range img.DataSyms {
|
||||||
|
if i >= len(dataSyms) {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
for _, r := range dataSyms[i].relocs {
|
||||||
|
reloc := r
|
||||||
|
if !known[reloc.Name] {
|
||||||
|
reloc.External = true
|
||||||
|
externals[reloc.Name] = true
|
||||||
|
}
|
||||||
|
img.DataSyms[i].Relocs = append(img.DataSyms[i].Relocs, reloc)
|
||||||
|
}
|
||||||
|
}
|
||||||
for name := range externals {
|
for name := range externals {
|
||||||
img.Externals = append(img.Externals, name)
|
img.Externals = append(img.Externals, name)
|
||||||
}
|
}
|
||||||
@@ -436,87 +550,162 @@ type dataSym struct {
|
|||||||
size int
|
size int
|
||||||
static bool
|
static bool
|
||||||
rodata bool
|
rodata bool
|
||||||
|
noptr bool
|
||||||
dupok bool
|
dupok bool
|
||||||
|
// relocs are the symbol-valued DATA fields, in declaration order; Off
|
||||||
|
// is relative to the symbol's data start.
|
||||||
|
relocs []Reloc
|
||||||
}
|
}
|
||||||
|
|
||||||
// collectData gathers the file's static symbols (GLOBL) and their initial
|
// collectData gathers the file's static symbols (GLOBL) and their initial
|
||||||
// contents (DATA) into byte buffers, in declaration order.
|
// contents (DATA) into byte buffers. Two passes: the Plan 9 convention puts
|
||||||
|
// every DATA line before its symbol's GLOBL, so the symbols are registered
|
||||||
|
// before the initialisers are applied.
|
||||||
func collectData(f *ast.File) ([]dataSym, error) {
|
func collectData(f *ast.File) ([]dataSym, error) {
|
||||||
index := map[string]int{}
|
index := map[string]int{}
|
||||||
var syms []dataSym
|
var syms []dataSym
|
||||||
for _, d := range f.Decls {
|
for _, d := range f.Decls {
|
||||||
switch dd := d.(type) {
|
gd, ok := d.(*ast.Globl)
|
||||||
case *ast.Globl:
|
if !ok {
|
||||||
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
continue
|
||||||
continue
|
}
|
||||||
}
|
if gd.Name == nil || gd.Name.Pseudo != "SB" {
|
||||||
name := dd.Name.Name
|
continue
|
||||||
if _, dup := index[name]; dup {
|
}
|
||||||
return nil, fmt.Errorf("duplicate GLOBL %q", name)
|
name := gd.Name.Name
|
||||||
}
|
if _, dup := index[name]; dup {
|
||||||
size := 0
|
return nil, fmt.Errorf("duplicate GLOBL %q", name)
|
||||||
if dd.Size != nil && dd.Size.Imm.HasVal {
|
}
|
||||||
size = int(dd.Size.Imm.Val)
|
size := 0
|
||||||
}
|
if gd.Size != nil && gd.Size.Imm.HasVal {
|
||||||
index[name] = len(syms)
|
size = int(gd.Size.Imm.Val)
|
||||||
ds := dataSym{
|
}
|
||||||
name: name,
|
index[name] = len(syms)
|
||||||
pkg: dd.Name.Pkg,
|
ds := dataSym{
|
||||||
buf: make([]byte, size),
|
name: name,
|
||||||
size: size,
|
pkg: gd.Name.Pkg,
|
||||||
static: dd.Name.Static,
|
buf: make([]byte, size),
|
||||||
}
|
size: size,
|
||||||
for _, f := range dd.Flags {
|
static: gd.Name.Static,
|
||||||
switch f {
|
}
|
||||||
case "RODATA":
|
for _, f := range gd.Flags {
|
||||||
ds.rodata = true
|
switch f {
|
||||||
case "DUPOK":
|
case "RODATA":
|
||||||
ds.dupok = true
|
ds.rodata = true
|
||||||
default:
|
case "NOPTR":
|
||||||
// Legacy numeric flag constants (runtime/textflag.h):
|
ds.noptr = true
|
||||||
// DUPOK is 2, RODATA is 8; combinations arrive as one
|
case "DUPOK":
|
||||||
// number (e.g. 10 = RODATA|DUPOK).
|
ds.dupok = true
|
||||||
if n, err := strconv.Atoi(f); err == nil {
|
default:
|
||||||
if n&2 != 0 {
|
// Legacy numeric flag constants (runtime/textflag.h):
|
||||||
ds.dupok = true
|
// DUPOK is 2, RODATA is 8, NOPTR is 16; combinations arrive
|
||||||
}
|
// as one number (e.g. 10 = RODATA|DUPOK).
|
||||||
if n&8 != 0 {
|
if n, err := strconv.Atoi(f); err == nil {
|
||||||
ds.rodata = true
|
if n&2 != 0 {
|
||||||
}
|
ds.dupok = true
|
||||||
|
}
|
||||||
|
if n&8 != 0 {
|
||||||
|
ds.rodata = true
|
||||||
|
}
|
||||||
|
if n&16 != 0 {
|
||||||
|
ds.noptr = true
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
syms = append(syms, ds)
|
}
|
||||||
|
syms = append(syms, ds)
|
||||||
case *ast.Data:
|
}
|
||||||
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
for _, d := range f.Decls {
|
||||||
continue
|
dd, ok := d.(*ast.Data)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
i, ok := index[dd.Name.Name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
|
||||||
|
}
|
||||||
|
if dd.Value == nil {
|
||||||
|
return nil, fmt.Errorf("DATA %q: missing value", dd.Name.Name)
|
||||||
|
}
|
||||||
|
w := dd.Width
|
||||||
|
off := dd.Name.Offset
|
||||||
|
buf := syms[i].buf
|
||||||
|
if off < 0 || off+int64(w) > int64(len(buf)) {
|
||||||
|
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
|
||||||
|
}
|
||||||
|
// A symbol value ("DATA s+0(SB)/8, $other(SB)", the rt0 spelling)
|
||||||
|
// leaves the field zero and records a relocation against the named
|
||||||
|
// symbol: the linker patches the absolute address at this data
|
||||||
|
// offset. The toolchain emits the same shape, an R_ADDR of the
|
||||||
|
// DATA width with the value's offset as the addend, on every
|
||||||
|
// architecture.
|
||||||
|
if sym := dd.Value.Imm.Sym; !dd.Value.Imm.HasVal && sym != nil {
|
||||||
|
syms[i].relocs = append(syms[i].relocs, Reloc{
|
||||||
|
Off: int(off),
|
||||||
|
Name: sym.Name,
|
||||||
|
Addend: sym.Offset,
|
||||||
|
Kind: RelAddr,
|
||||||
|
Siz: uint8(w),
|
||||||
|
})
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// A string or rune value ("DATA s+0(SB)/20, $"text"") writes its
|
||||||
|
// bytes into the field and leaves the rest zero, the toolchain's
|
||||||
|
// WriteString: the declared width must hold every byte, and any
|
||||||
|
// width is legal.
|
||||||
|
if s := dd.Value.Imm.Str; s != "" && !dd.Value.Imm.HasVal {
|
||||||
|
text, err := strconv.Unquote(s)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("DATA %q: invalid string value %s", dd.Name.Name, s)
|
||||||
}
|
}
|
||||||
i, ok := index[dd.Name.Name]
|
if len(text) > w {
|
||||||
if !ok {
|
return nil, fmt.Errorf("DATA %q: string of %d bytes does not fit width %d", dd.Name.Name, len(text), w)
|
||||||
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
|
|
||||||
}
|
}
|
||||||
if dd.Value == nil || !dd.Value.Imm.HasVal {
|
copy(buf[off:], text)
|
||||||
return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name)
|
continue
|
||||||
|
}
|
||||||
|
// A floating-point value stores its IEEE-754 bits: /4 the float32
|
||||||
|
// rounding of the parsed double, /8 the full 64 bits, the
|
||||||
|
// toolchain's WriteFloat32 and WriteFloat64.
|
||||||
|
if f := dd.Value.Imm.Float; f != "" && !dd.Value.Imm.HasVal {
|
||||||
|
num, err := strconv.ParseFloat(f, 64)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("DATA %q: invalid floating-point value %q", dd.Name.Name, f)
|
||||||
}
|
}
|
||||||
w := dd.Width
|
|
||||||
switch w {
|
|
||||||
case 1, 2, 4, 8:
|
|
||||||
default:
|
|
||||||
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
|
|
||||||
}
|
|
||||||
off := dd.Name.Offset
|
|
||||||
buf := syms[i].buf
|
|
||||||
if off < 0 || off+int64(w) > int64(len(buf)) {
|
|
||||||
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
|
|
||||||
}
|
|
||||||
v := dd.Value.Imm.Val
|
|
||||||
if dd.Value.Imm.Neg {
|
if dd.Value.Imm.Neg {
|
||||||
v = -v
|
num = -num
|
||||||
|
}
|
||||||
|
var v uint64
|
||||||
|
switch w {
|
||||||
|
case 4:
|
||||||
|
v = uint64(math.Float32bits(float32(num)))
|
||||||
|
case 8:
|
||||||
|
v = math.Float64bits(num)
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("DATA %q: invalid width %d for a float (want 4 or 8)", dd.Name.Name, w)
|
||||||
}
|
}
|
||||||
for j := range w {
|
for j := range w {
|
||||||
buf[off+int64(j)] = byte(v >> (8 * j))
|
buf[off+int64(j)] = byte(v >> (8 * j))
|
||||||
}
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !dd.Value.Imm.HasVal {
|
||||||
|
return nil, fmt.Errorf("DATA %q: value must be an integer immediate or a symbol address", dd.Name.Name)
|
||||||
|
}
|
||||||
|
switch w {
|
||||||
|
case 1, 2, 4, 8:
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
|
||||||
|
}
|
||||||
|
v := dd.Value.Imm.Val
|
||||||
|
if dd.Value.Imm.Neg {
|
||||||
|
v = -v
|
||||||
|
}
|
||||||
|
for j := range w {
|
||||||
|
buf[off+int64(j)] = byte(v >> (8 * j))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return syms, nil
|
return syms, nil
|
||||||
|
|||||||
+313
-14
@@ -4,14 +4,19 @@
|
|||||||
package asm
|
package asm
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestAssembleFileStaticData checks the whole-image layout — code, padding
|
// TestAssembleFileStaticData checks the whole-image layout; code, padding
|
||||||
// and the data section — and that the RIP-relative displacements of static
|
// and the data section; and that the RIP-relative displacements of static
|
||||||
// symbol loads resolve to the right bytes.
|
// symbol loads resolve to the right bytes.
|
||||||
func TestAssembleFileStaticData(t *testing.T) {
|
func TestAssembleFileStaticData(t *testing.T) {
|
||||||
f, errs := parser.Parse("d_amd64.s", `
|
f, errs := parser.Parse("d_amd64.s", `
|
||||||
@@ -55,23 +60,15 @@ DATA small<>+0(SB)/4, $0x1234
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestAssembleFileErrors checks the static-symbol error paths.
|
// TestAssembleFileErrors checks the static-symbol error paths. A reference
|
||||||
|
// to a static symbol no GLOBL defines defers to the linker exactly as the
|
||||||
|
// toolchain does (an external relocation), so it is not an error here.
|
||||||
func TestAssembleFileErrors(t *testing.T) {
|
func TestAssembleFileErrors(t *testing.T) {
|
||||||
cases := []struct {
|
cases := []struct {
|
||||||
name string
|
name string
|
||||||
src string
|
src string
|
||||||
want string // substring of the error
|
want string // substring of the error
|
||||||
}{
|
}{
|
||||||
{
|
|
||||||
"undefined symbol",
|
|
||||||
`
|
|
||||||
#include "textflag.h"
|
|
||||||
TEXT ·f(SB), NOSPLIT, $0
|
|
||||||
VMOVDQU nope<>(SB), X0
|
|
||||||
RET
|
|
||||||
`,
|
|
||||||
"undefined symbol",
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
"DATA without GLOBL",
|
"DATA without GLOBL",
|
||||||
`
|
`
|
||||||
@@ -166,3 +163,305 @@ func TestCollectDataNumericFlags(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestCollectDataSymbolValue covers the symbol-valued DATA field ("DATA
|
||||||
|
// s+0(SB)/8, $other(SB)", the rt0 spelling): the field stays zero in the
|
||||||
|
// image and the relocation is recorded against the named symbol, whatever
|
||||||
|
// the file defines (a TEXT function, a GLOBL) or leaves external.
|
||||||
|
func TestCollectDataSymbolValue(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-8
|
||||||
|
MOVQ target+0(FP), AX
|
||||||
|
RET
|
||||||
|
GLOBL holder(SB), NOPTR, $32
|
||||||
|
DATA holder+0(SB)/8, $·Keep(SB)
|
||||||
|
DATA holder+8(SB)/8, $·Keep+5(SB)
|
||||||
|
DATA holder+16(SB)/8, $holder(SB)
|
||||||
|
GLOBL spare(SB), NOPTR, $8
|
||||||
|
DATA spare+0(SB)/8, $extvar(SB)
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("f_amd64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
byName := map[string]DataSymbol{}
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
byName[d.Name] = d
|
||||||
|
}
|
||||||
|
want := []struct {
|
||||||
|
sym string
|
||||||
|
off int
|
||||||
|
name string
|
||||||
|
addend int64
|
||||||
|
ext bool
|
||||||
|
}{
|
||||||
|
{"holder", 0, "Keep", 0, false},
|
||||||
|
{"holder", 8, "Keep", 5, false},
|
||||||
|
{"holder", 16, "holder", 0, false},
|
||||||
|
{"spare", 0, "extvar", 0, true},
|
||||||
|
}
|
||||||
|
var flat []struct {
|
||||||
|
sym string
|
||||||
|
r Reloc
|
||||||
|
}
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
for _, r := range d.Relocs {
|
||||||
|
flat = append(flat, struct {
|
||||||
|
sym string
|
||||||
|
r Reloc
|
||||||
|
}{d.Name, r})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(flat) != len(want) {
|
||||||
|
t.Fatalf("data relocations = %d, want %d", len(flat), len(want))
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
g := flat[i]
|
||||||
|
r := g.r
|
||||||
|
if g.sym != w.sym {
|
||||||
|
t.Errorf("relocation %d sits on %q, want %q", i, g.sym, w.sym)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if r.Off != w.off || r.Name != w.name || r.Addend != w.addend || r.External != w.ext {
|
||||||
|
t.Errorf("relocation %d = {+%d %q addend %d ext %v}, want {+%d %q addend %d ext %v}",
|
||||||
|
i, r.Off, r.Name, r.Addend, r.External, w.off, w.name, w.addend, w.ext)
|
||||||
|
}
|
||||||
|
if r.Kind != RelAddr {
|
||||||
|
t.Errorf("relocation %d kind = %v, want RelAddr", i, r.Kind)
|
||||||
|
}
|
||||||
|
if r.Siz != 8 {
|
||||||
|
t.Errorf("relocation %d siz = %d, want 8", i, r.Siz)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The fields themselves stay zero: only the linker fills them.
|
||||||
|
for _, b := range img.Data {
|
||||||
|
if b != 0 {
|
||||||
|
t.Fatal("data section is not all zero before relocation")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(img.Externals) != 1 || img.Externals[0] != "extvar" {
|
||||||
|
t.Errorf("Externals = %v, want [extvar]", img.Externals)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGOObjectDataSymbolReloc pins the GOOBJ record a symbol-valued DATA
|
||||||
|
// field produces, against the shape the toolchain emits for the same
|
||||||
|
// source: an R_ADDR of the DATA width at the field offset, pkgIdxNone plus
|
||||||
|
// the non-package definition index when the target is the file's own TEXT
|
||||||
|
// function (the rt0 lib entry spelling).
|
||||||
|
func TestGOObjectDataSymbolReloc(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("f_amd64.s", `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-8
|
||||||
|
RET
|
||||||
|
GLOBL holder(SB), NOPTR, $16
|
||||||
|
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.GOObject("main", "f_amd64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObject: %v", err)
|
||||||
|
}
|
||||||
|
v := openGoobj(t, obj)
|
||||||
|
// Walk every relocation record; the data record is the one of Siz 8
|
||||||
|
// and type R_ADDR.
|
||||||
|
var off, add int64
|
||||||
|
var pkg, sym uint32
|
||||||
|
found := false
|
||||||
|
for data := v.blk(blkReloc); len(data) >= 23; data = data[23:] {
|
||||||
|
if data[4] != 8 || binary.LittleEndian.Uint16(data[5:]) != relocAddr {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
found = true
|
||||||
|
off = int64(int32(binary.LittleEndian.Uint32(data[0:])))
|
||||||
|
add = int64(binary.LittleEndian.Uint64(data[7:]))
|
||||||
|
pkg = binary.LittleEndian.Uint32(data[15:])
|
||||||
|
sym = binary.LittleEndian.Uint32(data[19:])
|
||||||
|
break
|
||||||
|
}
|
||||||
|
if !found {
|
||||||
|
t.Fatal("no data relocation record in the object")
|
||||||
|
}
|
||||||
|
if off != 0 || add != 5 {
|
||||||
|
t.Errorf("data reloc = {off %d addend %d}, want {off 0 addend 5}", off, add)
|
||||||
|
}
|
||||||
|
if pkg != pkgIdxNone {
|
||||||
|
t.Errorf("data reloc pkg = %#x, want pkgIdxNone (the TEXT function)", pkg)
|
||||||
|
}
|
||||||
|
// The function's non-package definition index: the four pc tables
|
||||||
|
// precede it, so index 4.
|
||||||
|
if sym != 4 {
|
||||||
|
t.Errorf("data reloc sym = %d, want 4", sym)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGOObjectDataSymbolLink is the end-to-end proof for symbol-valued DATA
|
||||||
|
// fields: the gasm object is substituted for the toolchain's and re-linked,
|
||||||
|
// then executed, and the linked data word must hold the real address of the
|
||||||
|
// function the DATA line named (runtime.FuncForPC identifies it).
|
||||||
|
func TestGOObjectDataSymbolLink(t *testing.T) {
|
||||||
|
goBin, err := exec.LookPath("go")
|
||||||
|
if err != nil {
|
||||||
|
t.Skip("no Go toolchain available")
|
||||||
|
}
|
||||||
|
dir := t.TempDir()
|
||||||
|
asmSrc := `#include "textflag.h"
|
||||||
|
GLOBL entry(SB), NOPTR, $8
|
||||||
|
DATA entry+0(SB)/8, $·keepme(SB)
|
||||||
|
|
||||||
|
TEXT ·keepme(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·entryptr(SB), NOSPLIT, $0-8
|
||||||
|
MOVQ entry+0(SB), AX
|
||||||
|
MOVQ AX, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
mainSrc := `package main
|
||||||
|
|
||||||
|
import "runtime"
|
||||||
|
|
||||||
|
func keepme()
|
||||||
|
func entryptr() uintptr
|
||||||
|
|
||||||
|
func main() {
|
||||||
|
pc := entryptr()
|
||||||
|
fn := runtime.FuncForPC(pc)
|
||||||
|
if fn == nil {
|
||||||
|
panic("the entry word does not point at a function")
|
||||||
|
}
|
||||||
|
if fn.Name() != "main.keepme" {
|
||||||
|
panic("the entry word points at " + fn.Name())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module dlink\n\ngo 1.21\n"), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Capture the build: the package archive's asm object and the link line.
|
||||||
|
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||||
|
build.Dir = dir
|
||||||
|
buildLog, err := build.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||||
|
}
|
||||||
|
st := parseBuildLog(t, buildLog, "main_amd64.s")
|
||||||
|
defer os.RemoveAll(st.work)
|
||||||
|
|
||||||
|
// Assemble the same source with gasm and substitute the object.
|
||||||
|
src, err := os.ReadFile(filepath.Join(dir, "main_amd64.s"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
f, errs := parser.Parse("main_amd64.s", string(src))
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
// The package path is "main": the linker resolves the Go code's
|
||||||
|
// references against main.<name>, so the object must define the symbols
|
||||||
|
// under that prefix whatever the module is called.
|
||||||
|
gasmObj, err := img.GOObject("main", "main_amd64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObject: %v", err)
|
||||||
|
}
|
||||||
|
substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"), gasmObj)
|
||||||
|
|
||||||
|
// The linked program must run and find the right function behind the
|
||||||
|
// data word.
|
||||||
|
out, err := exec.Command(filepath.Join(dir, "prog2")).CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("linked program failed: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCollectDataFloatAndStringValues covers the non-integer DATA values the
|
||||||
|
// runtime's math and asm files use: floating-point initialisers store their
|
||||||
|
// IEEE-754 bits (/4 the float32 rounding, /8 the full double) and string
|
||||||
|
// initialisers write their bytes zero-padded within the declared width.
|
||||||
|
func TestCollectDataFloatAndStringValues(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-8
|
||||||
|
RET
|
||||||
|
GLOBL vals<>(SB), RODATA, $44
|
||||||
|
DATA vals<>+0(SB)/8, $0.5
|
||||||
|
DATA vals<>+8(SB)/8, $-1.0
|
||||||
|
DATA vals<>+16(SB)/4, $1.5
|
||||||
|
DATA vals<>+20(SB)/16, $"call frame too "
|
||||||
|
DATA vals<>+36(SB)/4, $"hi"
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("fvals_amd64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
byName := map[string]DataSymbol{}
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
byName[d.Name] = d
|
||||||
|
}
|
||||||
|
d := byName["vals"]
|
||||||
|
if d.Size != 44 {
|
||||||
|
t.Fatalf("vals size = %d, want 44", d.Size)
|
||||||
|
}
|
||||||
|
buf := img.Data[d.Offset : d.Offset+44]
|
||||||
|
// 0.5 = 0x3FE0000000000000, -1.0 = 0xBFF0000000000000 (float64);
|
||||||
|
// 1.5 = 0x3FC00000 (float32).
|
||||||
|
for _, c := range []struct {
|
||||||
|
off int
|
||||||
|
want []byte
|
||||||
|
}{
|
||||||
|
{0, []byte{0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xE0, 0x3F}},
|
||||||
|
{8, []byte{0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xF0, 0xBF}},
|
||||||
|
{16, []byte{0x00, 0x00, 0xC0, 0x3F}},
|
||||||
|
{20, []byte("call frame too ")},
|
||||||
|
{36, []byte{'h', 'i', 0x00, 0x00}},
|
||||||
|
} {
|
||||||
|
if string(buf[c.off:c.off+len(c.want)]) != string(c.want) {
|
||||||
|
t.Errorf("vals+%d: got % x, want % x", c.off, buf[c.off:c.off+len(c.want)], c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCollectDataValueErrors pins the value-kind width rules: a float needs
|
||||||
|
// width 4 or 8, a string must fit its declared width, and a bad float
|
||||||
|
// literal is diagnosed rather than stored.
|
||||||
|
func TestCollectDataValueErrors(t *testing.T) {
|
||||||
|
cases := []string{
|
||||||
|
`GLOBL v<>(SB), RODATA, $4
|
||||||
|
DATA v<>+0(SB)/1, $0.5`,
|
||||||
|
`GLOBL v<>(SB), RODATA, $2
|
||||||
|
DATA v<>+0(SB)/2, $"toolarge"`,
|
||||||
|
}
|
||||||
|
for i, src := range cases {
|
||||||
|
full := "#include \"textflag.h\"\nTEXT ·Keep(SB), NOSPLIT, $0-8\n\tRET\n" + src
|
||||||
|
f, errs := parser.Parse(fmt.Sprintf("verr%d_amd64.s", i), full)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("case %d parse: %v", i, errs)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFile(f); err == nil {
|
||||||
|
t.Errorf("case %d: expected an error, got none", i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+1212
-63
File diff suppressed because it is too large
Load Diff
+639
-8
@@ -30,7 +30,10 @@ package asm
|
|||||||
// of the immediate and register fields), mirroring the toolchain's OP_*
|
// of the immediate and register fields), mirroring the toolchain's OP_*
|
||||||
// helpers, so each l64* function only ORs its fields in.
|
// helpers, so each l64* function only ORs its fields in.
|
||||||
|
|
||||||
import "maps"
|
import (
|
||||||
|
"maps"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
// loong64RegNum returns the 5-bit register number for a LoongArch register
|
// loong64RegNum returns the 5-bit register number for a LoongArch register
|
||||||
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
|
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
|
||||||
@@ -103,7 +106,12 @@ func loong64RegNum(name string) int {
|
|||||||
case "R31", "S8":
|
case "R31", "S8":
|
||||||
return 31
|
return 31
|
||||||
}
|
}
|
||||||
// F0-F31, FCC0-FCC7, FCSR0-FCSR31.
|
// F0-F31, FCC0-FCC7, FCSR0-FCSR31. The LSX/LASX vector banks (V0-V31,
|
||||||
|
// X0-X31) are deliberately NOT accepted here: they are a separate
|
||||||
|
// register class, and the toolchain rejects V/X names wherever an
|
||||||
|
// integer or FP register is expected (GOARCH=loong64 go tool asm reports
|
||||||
|
// "unrecognized instruction" for `BEQZ X0`). Vector operands are
|
||||||
|
// resolved only through loong64VecRegNum.
|
||||||
if len(name) >= 4 && name[:4] == "FCSR" {
|
if len(name) >= 4 && name[:4] == "FCSR" {
|
||||||
return loong64RegSpecial(name[4:], 31)
|
return loong64RegSpecial(name[4:], 31)
|
||||||
}
|
}
|
||||||
@@ -148,6 +156,19 @@ func loong64RegSpecial(digits string, max int) int {
|
|||||||
return -1
|
return -1
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// loong64VecRegNum resolves an LSX/LASX vector register name (V0-V31 or
|
||||||
|
// X0-X31) to its 5-bit number, or -1. The vector banks are a register class
|
||||||
|
// of their own: the toolchain accepts them only in the vector operands of the
|
||||||
|
// LSX/LASX instructions (GOARCH=loong64 go tool asm assembles `VADDV V0, V1,
|
||||||
|
// V2` and `XVADDV X0, X1, X2`, and rejects `VADDV R4, R5, R6`), so the V/X
|
||||||
|
// spellings never reach the integer/FP resolver.
|
||||||
|
func loong64VecRegNum(name string) int {
|
||||||
|
if len(name) < 2 || (name[0] != 'V' && name[0] != 'X') {
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
return loong64RegSpecial(name[1:], 31)
|
||||||
|
}
|
||||||
|
|
||||||
// ---- format helpers ----
|
// ---- format helpers ----
|
||||||
|
|
||||||
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
|
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
|
||||||
@@ -199,7 +220,9 @@ func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
|
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
|
||||||
// The msb/lsb fields are 6 bits wide (0-63) and are validated by the caller.
|
// The msb/lsb fields are 6 bits wide and are inserted unmasked: the caller
|
||||||
|
// must have validated them (0..31 for the .w forms, 0..63 for the .d forms,
|
||||||
|
// lsb <= msb), the same rule the toolchain enforces as "illegal bit number".
|
||||||
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
|
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
|
||||||
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
|
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
|
||||||
}
|
}
|
||||||
@@ -245,7 +268,7 @@ const (
|
|||||||
l64Firr14 // 2RI14 (ldptr/stptr)
|
l64Firr14 // 2RI14 (ldptr/stptr)
|
||||||
l64Firr16 // 2RI16 (addu16i.d)
|
l64Firr16 // 2RI16 (addu16i.d)
|
||||||
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
|
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
|
||||||
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub)
|
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub, fsel)
|
||||||
l64Firir // bstrins/bstrpick
|
l64Firir // bstrins/bstrpick
|
||||||
l64Firrr // alsl
|
l64Firrr // alsl
|
||||||
l64Fi15 // syscall/break/dbar
|
l64Fi15 // syscall/break/dbar
|
||||||
@@ -253,6 +276,11 @@ const (
|
|||||||
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
|
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
|
||||||
l64Fshift // 2RI12 with a 5/6-bit shift immediate
|
l64Fshift // 2RI12 with a 5/6-bit shift immediate
|
||||||
l64Fpreld // preld (2RI12 + 5-bit hint)
|
l64Fpreld // preld (2RI12 + 5-bit hint)
|
||||||
|
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
|
||||||
|
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
|
||||||
|
l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd
|
||||||
|
l64Fllsc // acquire/release LL/SC (2R against a zero-offset memory operand)
|
||||||
|
l64Fscq // sc.q: op | middle<<10 | base<<5 | first against a zero-offset memory operand
|
||||||
)
|
)
|
||||||
|
|
||||||
// l64Enc is one instruction's encoding: its bit layout (format) and the
|
// l64Enc is one instruction's encoding: its bit layout (format) and the
|
||||||
@@ -275,17 +303,90 @@ type l64DualEnc struct {
|
|||||||
var l64DualTable = map[string]l64DualEnc{}
|
var l64DualTable = map[string]l64DualEnc{}
|
||||||
|
|
||||||
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
|
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
|
||||||
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered
|
// to their encoding.
|
||||||
// yet; the base integer, memory and floating-point ISA is complete.
|
|
||||||
var l64InstrTable = map[string]l64Enc{}
|
var l64InstrTable = map[string]l64Enc{}
|
||||||
|
|
||||||
|
// l64Vec3Enc pairs a vector opcode with its register bank: false = LSX
|
||||||
|
// (V0-V31), true = LASX (X0-X31). The toolchain accepts one bank per
|
||||||
|
// spelling: GOARCH=loong64 go tool asm assembles `VADDV V1, V2, V3` and
|
||||||
|
// `XVADDV X1, X2, X3`, and rejects the crossed spellings.
|
||||||
|
type l64Vec3Enc struct {
|
||||||
|
op uint32
|
||||||
|
lasx bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64VecImmEnc carries the immediate-form encoding of a vector mnemonic:
|
||||||
|
// the opcode, the bank, the accepted immediate range, the bias the toolchain
|
||||||
|
// adds (vsrai.b encodes imm+8) and the mask of the encoded field (vseqi.b
|
||||||
|
// keeps a 5-bit two's-complement value, vseqi.d a 7-bit one).
|
||||||
|
type l64VecImmEnc struct {
|
||||||
|
op uint32
|
||||||
|
lasx bool
|
||||||
|
min, max int
|
||||||
|
bias int
|
||||||
|
mask int
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64VecBank marks the LSX/LASX mnemonics and records which register bank
|
||||||
|
// each accepts; presence in the map routes the mnemonic through the vector
|
||||||
|
// dispatcher rather than the integer/FP formats.
|
||||||
|
var l64VecBank = map[string]bool{}
|
||||||
|
|
||||||
|
// l64VecImmInfo mirrors l64VecImmTable for the dispatcher.
|
||||||
|
var l64VecImmInfo = map[string]l64VecImmEnc{}
|
||||||
|
|
||||||
|
// l64Vec2R marks the two-operand vector mnemonics (INSTR vj, vd, such as
|
||||||
|
// vpcnt.v).
|
||||||
|
var l64Vec2R = map[string]bool{}
|
||||||
|
|
||||||
|
// l64Vec4R marks the four-operand vector mnemonics (INSTR va, vk, vj, vd,
|
||||||
|
// such as vshuf.b).
|
||||||
|
var l64Vec4R = map[string]bool{}
|
||||||
|
|
||||||
|
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants, each pre-shifted to
|
||||||
|
// its exact bit range, read off `go tool objdump` of GOARCH=loong64
|
||||||
|
// `go tool asm` kernels and the toolchain's specialLsxMovInst table.
|
||||||
|
type l64VmovqEnc struct {
|
||||||
|
ld, st, ldx, stx uint32 // plain and indexed load/store
|
||||||
|
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
|
||||||
|
pickS, pickU uint32 // vpickve2gr.{,u} element extract
|
||||||
|
ins uint32 // vinsgr2vr element insert
|
||||||
|
dup uint32 // vreplgr2vr duplicate (width in [11:10])
|
||||||
|
move uint32 // vori.b/xvori.b $0 register move
|
||||||
|
rveiB, rveiH, rveiW, rveiD uint32 // vreplvei: broadcast one element (LSX)
|
||||||
|
rve0B, rve0H, rve0W uint32 // xvreplve0 broadcast of element zero (LASX)
|
||||||
|
rve0D, rve0Q uint32 // xvreplve0.{d,q}, ditto
|
||||||
|
xinsW, xinsD uint32 // xvinsve0: insert element zero (LASX)
|
||||||
|
xpickW, xpickD uint32 // xvpickve: extract element (LASX)
|
||||||
|
}
|
||||||
|
|
||||||
|
var l64VmovqTable = map[bool]l64VmovqEnc{
|
||||||
|
false: { // VMOVQ, the LSX (V) bank
|
||||||
|
ld: 0x5800 << 15, st: 0x5880 << 15, ldx: 0x7080 << 15, stx: 0x7088 << 15,
|
||||||
|
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
|
||||||
|
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
|
||||||
|
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
|
||||||
|
rveiB: 0x01CBDE << 14, rveiH: 0x0397BE << 13, rveiW: 0x072F7E << 12, rveiD: 0x0E5EFE << 11,
|
||||||
|
},
|
||||||
|
true: { // XVMOVQ, the LASX (X) bank
|
||||||
|
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
|
||||||
|
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
|
||||||
|
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
|
||||||
|
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
|
||||||
|
rve0B: 0x1DC1C0 << 10, rve0H: 0x1DC1E0 << 10, rve0W: 0x1DC1F0 << 10,
|
||||||
|
rve0D: 0x1DC1F8 << 10, rve0Q: 0x1DC1FC << 10,
|
||||||
|
xinsW: 0x03B7FE << 13, xinsD: 0x076FFE << 12,
|
||||||
|
xpickW: 0x03B81E << 13, xpickD: 0x07703E << 12,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
func init() {
|
func init() {
|
||||||
// 3R, integer.
|
// 3R, integer.
|
||||||
rrr := map[string]uint32{
|
rrr := map[string]uint32{
|
||||||
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
|
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
|
||||||
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
|
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
|
||||||
"SGT": 0x24 << 15, "SGTU": 0x25 << 15,
|
"SGT": 0x24 << 15, "SGTU": 0x25 << 15,
|
||||||
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15,
|
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15,
|
||||||
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
|
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
|
||||||
"ORN": 0x2c << 15, "ANDN": 0x2d << 15,
|
"ORN": 0x2c << 15, "ANDN": 0x2d << 15,
|
||||||
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
|
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
|
||||||
@@ -358,6 +459,18 @@ func init() {
|
|||||||
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
|
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
|
||||||
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
|
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
|
||||||
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
|
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
|
||||||
|
// LSX: convert a 64-bit integer lane to a double float. The operand
|
||||||
|
// bank is the FP registers (the toolchain spells it `FFINTDV F0, F1`),
|
||||||
|
// so the entry stays on the 2R integer/FP format.
|
||||||
|
"FFINTDV": 0x474a << 10,
|
||||||
|
// The rest of the scalar conversions (all F-bank, 2R).
|
||||||
|
"FFINTFW": 0x4744 << 10, // ffint.s.w
|
||||||
|
"FFINTFV": 0x4746 << 10, // ffint.s.l
|
||||||
|
"FFINTDW": 0x4748 << 10, // ffint.d.w
|
||||||
|
"FTINTWF": 0x46c1 << 10, // ftint.w.s
|
||||||
|
"FTINTWD": 0x46c2 << 10, // ftint.w.d
|
||||||
|
"FTINTVF": 0x46c9 << 10, // ftint.l.s
|
||||||
|
"FTINTVD": 0x46ca << 10, // ftint.l.d
|
||||||
}
|
}
|
||||||
for m, op := range rr {
|
for m, op := range rr {
|
||||||
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
|
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
|
||||||
@@ -367,6 +480,20 @@ func init() {
|
|||||||
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
|
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
|
||||||
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
|
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
|
||||||
|
|
||||||
|
// Acquire/release LL/SC (2R against a zero-offset memory operand):
|
||||||
|
// LLACQV (Rj), Rd loads, SCRELV Rd, (Rj) stores, both encoding
|
||||||
|
// op | rj<<5 | rd. Opcodes from cmd/internal/obj/loong64/instOp.go
|
||||||
|
// (ll.acq.{w,d}, sc.rel.{w,d}).
|
||||||
|
l64InstrTable["LLACQW"] = l64Enc{format: l64Fllsc, op: 0x0E15E0 << 10}
|
||||||
|
l64InstrTable["SCRELW"] = l64Enc{format: l64Fllsc, op: 0x0E15E1 << 10}
|
||||||
|
l64InstrTable["LLACQV"] = l64Enc{format: l64Fllsc, op: 0x0E15E2 << 10}
|
||||||
|
l64InstrTable["SCRELV"] = l64Enc{format: l64Fllsc, op: 0x0E15E3 << 10}
|
||||||
|
|
||||||
|
// SCQ (sc.q first, middle, (base)) keeps its own operand order: the
|
||||||
|
// encoding is op | middle<<10 | base<<5 | first, the memory operand's
|
||||||
|
// base in the rj field, not the toolchain's generic 3R layout.
|
||||||
|
l64InstrTable["SCQ"] = l64Enc{format: l64Fscq, op: 0x070AE << 15}
|
||||||
|
|
||||||
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
|
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
|
||||||
// selected by the operand kind; the shift mnemonics pair the 3R form
|
// selected by the operand kind; the shift mnemonics pair the 3R form
|
||||||
// with a 5/6-bit shift immediate.
|
// with a 5/6-bit shift immediate.
|
||||||
@@ -414,12 +541,14 @@ func init() {
|
|||||||
// LUI is the Plan 9 spelling of lu12i.w.
|
// LUI is the Plan 9 spelling of lu12i.w.
|
||||||
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
||||||
|
|
||||||
// 4R, fused multiply-add.
|
// 4R, fused multiply-add, and FSEL (fsel.d: the first operand is a FCC
|
||||||
|
// condition flag, the layout matches the 4R shape).
|
||||||
rrrr := map[string]uint32{
|
rrrr := map[string]uint32{
|
||||||
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
|
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
|
||||||
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
|
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
|
||||||
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
|
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
|
||||||
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
|
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
|
||||||
|
"FSEL": 0x340 << 18,
|
||||||
}
|
}
|
||||||
for m, op := range rrrr {
|
for m, op := range rrrr {
|
||||||
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
|
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
|
||||||
@@ -453,6 +582,10 @@ func init() {
|
|||||||
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
|
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
|
||||||
|
|
||||||
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
|
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
|
||||||
|
// The toolchain's form is three operands, `AMADDW rk, (rj), rd`
|
||||||
|
// (cmd/asm/internal/asm/testdata/loong64enc1.s and
|
||||||
|
// internal/runtime/atomic/atomic_loong64.s); the two-register spelling
|
||||||
|
// is rejected by the oracle.
|
||||||
am := map[string]uint32{
|
am := map[string]uint32{
|
||||||
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
|
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
|
||||||
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
|
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
|
||||||
@@ -470,10 +603,508 @@ func init() {
|
|||||||
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
|
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
|
||||||
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
|
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
|
||||||
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
|
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
|
||||||
|
// The _dbar (acquire/release) add, and, or variants: opcodes read off
|
||||||
|
// `go tool objdump` of `AMADDDBW R14, (R13), R12` and friends.
|
||||||
|
"AMADDDBW": 0x070D4 << 15, "AMADDDBV": 0x070D5 << 15,
|
||||||
|
"AMANDDBW": 0x070D6 << 15, "AMANDDBV": 0x070D7 << 15,
|
||||||
|
"AMORDBW": 0x070D8 << 15, "AMORDBV": 0x070D9 << 15,
|
||||||
|
// The remaining _dbar exchange variants (loong64enc1.s).
|
||||||
|
"AMXORDBW": 0x070DA << 15, "AMXORDBV": 0x070DB << 15,
|
||||||
|
"AMMAXDBW": 0x070DC << 15, "AMMAXDBV": 0x070DD << 15,
|
||||||
|
"AMMINDBW": 0x070DE << 15, "AMMINDBV": 0x070DF << 15,
|
||||||
|
"AMMAXDBWU": 0x070E0 << 15, "AMMAXDBVU": 0x070E1 << 15,
|
||||||
|
"AMMINDBWU": 0x070E2 << 15, "AMMINDBVU": 0x070E3 << 15,
|
||||||
}
|
}
|
||||||
for m, op := range am {
|
for m, op := range am {
|
||||||
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
|
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- LSX/LASX (V*/XV*) ----
|
||||||
|
// Every opcode below was read off `go tool objdump` of a GOARCH=loong64
|
||||||
|
// `go tool asm` kernel (the toolchain's own loong64enc1.s cross-checks
|
||||||
|
// most of them), not assumed from the LoongArch manual.
|
||||||
|
|
||||||
|
// Three vector registers: INSTR vk, vj, vd (or INSTR vk, vd with
|
||||||
|
// vj = vd). l64Vec3Enc.lasx selects the register bank the toolchain
|
||||||
|
// accepts: LSX spellings take V0-V31, LASX spellings X0-X31.
|
||||||
|
vec3 := map[string]l64Vec3Enc{
|
||||||
|
"VADDW": {0xE016 << 15, false}, "VADDV": {0xE017 << 15, false},
|
||||||
|
"VANDV": {0xE24C << 15, false}, "VXORV": {0xE24E << 15, false},
|
||||||
|
"VSEQB": {0xE000 << 15, false}, "VSEQV": {0xE003 << 15, false},
|
||||||
|
"VSRAB": {0xE1D8 << 15, false}, "VROTRW": {0xE1DE << 15, false},
|
||||||
|
"XVADDV": {0xE817 << 15, true},
|
||||||
|
"XVANDV": {0xEA4C << 15, true}, "XVXORV": {0xEA4E << 15, true},
|
||||||
|
"XVSEQB": {0xE800 << 15, true}, "XVSEQV": {0xE803 << 15, true},
|
||||||
|
}
|
||||||
|
|
||||||
|
// The integer and FP add/subtract families: [X]VADD and [X]VSUB by lane
|
||||||
|
// width, plus the [X]VSADD/[X]VSSUB saturating pairs.
|
||||||
|
// Opcodes transcribed from the toolchain's loong64enc1.s.
|
||||||
|
addsub := map[string]l64Vec3Enc{
|
||||||
|
"VADDB": {0xE014 << 15, false}, "VADDH": {0xE015 << 15, false},
|
||||||
|
"VADDD": {0xE262 << 15, false}, "VADDF": {0xE261 << 15, false},
|
||||||
|
"VADDQ": {0xE25A << 15, false},
|
||||||
|
"VSUBB": {0xE018 << 15, false}, "VSUBH": {0xE019 << 15, false},
|
||||||
|
"VSUBW": {0xE01A << 15, false}, "VSUBV": {0xE01B << 15, false},
|
||||||
|
"VSUBQ": {0xE25B << 15, false},
|
||||||
|
"VSUBF": {0xE265 << 15, false}, "VSUBD": {0xE266 << 15, false},
|
||||||
|
"VSADDB": {0xE08C << 15, false}, "VSADDH": {0xE08D << 15, false},
|
||||||
|
"VSADDW": {0xE08E << 15, false}, "VSADDV": {0xE08F << 15, false},
|
||||||
|
"VSADDBU": {0xE094 << 15, false}, "VSADDHU": {0xE095 << 15, false},
|
||||||
|
"VSADDWU": {0xE096 << 15, false}, "VSADDVU": {0xE097 << 15, false},
|
||||||
|
"VSSUBB": {0xE090 << 15, false}, "VSSUBH": {0xE091 << 15, false},
|
||||||
|
"VSSUBW": {0xE092 << 15, false}, "VSSUBV": {0xE093 << 15, false},
|
||||||
|
"VSSUBBU": {0xE098 << 15, false}, "VSSUBHU": {0xE099 << 15, false},
|
||||||
|
"VSSUBWU": {0xE09A << 15, false}, "VSSUBVU": {0xE09B << 15, false},
|
||||||
|
"XVADDB": {0xE814 << 15, true}, "XVADDH": {0xE815 << 15, true},
|
||||||
|
"XVADDW": {0xE816 << 15, true},
|
||||||
|
"XVADDD": {0xEA62 << 15, true}, "XVADDF": {0xEA61 << 15, true},
|
||||||
|
"XVADDQ": {0xEA5A << 15, true},
|
||||||
|
"XVSUBB": {0xE818 << 15, true}, "XVSUBH": {0xE819 << 15, true},
|
||||||
|
"XVSUBW": {0xE81A << 15, true}, "XVSUBV": {0xE81B << 15, true},
|
||||||
|
"XVSUBQ": {0xEA5B << 15, true},
|
||||||
|
"XVSUBF": {0xEA65 << 15, true}, "XVSUBD": {0xEA66 << 15, true},
|
||||||
|
"XVSADDB": {0xE88C << 15, true}, "XVSADDH": {0xE88D << 15, true},
|
||||||
|
"XVSADDW": {0xE88E << 15, true}, "XVSADDV": {0xE88F << 15, true},
|
||||||
|
"XVSADDBU": {0xE894 << 15, true}, "XVSADDHU": {0xE895 << 15, true},
|
||||||
|
"XVSADDWU": {0xE896 << 15, true}, "XVSADDVU": {0xE897 << 15, true},
|
||||||
|
"XVSSUBB": {0xE890 << 15, true}, "XVSSUBH": {0xE891 << 15, true},
|
||||||
|
"XVSSUBW": {0xE892 << 15, true}, "XVSSUBV": {0xE893 << 15, true},
|
||||||
|
"XVSSUBBU": {0xE898 << 15, true}, "XVSSUBHU": {0xE899 << 15, true},
|
||||||
|
"XVSSUBWU": {0xE89A << 15, true}, "XVSSUBVU": {0xE89B << 15, true},
|
||||||
|
}
|
||||||
|
|
||||||
|
// The multiply families: plain and high-half [X]VMUL/[X]VMUH, the
|
||||||
|
// widening [X]VMULW{EV,OD} ladder and its accumulating [X]VMADDW twins,
|
||||||
|
// plus the [X]VMADD/[X]VMSUB fused multiply-add and the [X]VDIV/[X]VMOD
|
||||||
|
// divide and modulo pairs.
|
||||||
|
muldiv := map[string]l64Vec3Enc{
|
||||||
|
"VMULB": {0xE108 << 15, false}, "VMULH": {0xE109 << 15, false},
|
||||||
|
"VMULW": {0xE10A << 15, false}, "VMULV": {0xE10B << 15, false},
|
||||||
|
"VMUHB": {0xE10C << 15, false}, "VMUHH": {0xE10D << 15, false},
|
||||||
|
"VMUHW": {0xE10E << 15, false}, "VMUHV": {0xE10F << 15, false},
|
||||||
|
"VMUHBU": {0xE110 << 15, false}, "VMUHHU": {0xE111 << 15, false},
|
||||||
|
"VMUHWU": {0xE112 << 15, false}, "VMUHVU": {0xE113 << 15, false},
|
||||||
|
"VMULWEVHB": {0xE120 << 15, false}, "VMULWEVWH": {0xE121 << 15, false},
|
||||||
|
"VMULWEVVW": {0xE122 << 15, false}, "VMULWEVQV": {0xE123 << 15, false},
|
||||||
|
"VMULWODHB": {0xE124 << 15, false}, "VMULWODWH": {0xE125 << 15, false},
|
||||||
|
"VMULWODVW": {0xE126 << 15, false}, "VMULWODQV": {0xE127 << 15, false},
|
||||||
|
"VMULWEVHBU": {0xE130 << 15, false}, "VMULWEVWHU": {0xE131 << 15, false},
|
||||||
|
"VMULWEVVWU": {0xE132 << 15, false}, "VMULWEVQVU": {0xE133 << 15, false},
|
||||||
|
"VMULWODHBU": {0xE134 << 15, false}, "VMULWODWHU": {0xE135 << 15, false},
|
||||||
|
"VMULWODVWU": {0xE136 << 15, false}, "VMULWODQVU": {0xE137 << 15, false},
|
||||||
|
"VMULWEVHBUB": {0xE140 << 15, false}, "VMULWEVWHUH": {0xE141 << 15, false},
|
||||||
|
"VMULWEVVWUW": {0xE142 << 15, false}, "VMULWEVQVUV": {0xE143 << 15, false},
|
||||||
|
"VMULWODHBUB": {0xE144 << 15, false}, "VMULWODWHUH": {0xE145 << 15, false},
|
||||||
|
"VMULWODVWUW": {0xE146 << 15, false}, "VMULWODQVUV": {0xE147 << 15, false},
|
||||||
|
"VMADDB": {0xE150 << 15, false}, "VMADDH": {0xE151 << 15, false},
|
||||||
|
"VMADDW": {0xE152 << 15, false}, "VMADDV": {0xE153 << 15, false},
|
||||||
|
"VMSUBB": {0xE154 << 15, false}, "VMSUBH": {0xE155 << 15, false},
|
||||||
|
"VMSUBW": {0xE156 << 15, false}, "VMSUBV": {0xE157 << 15, false},
|
||||||
|
"VMADDWEVHB": {0xE158 << 15, false}, "VMADDWEVWH": {0xE159 << 15, false},
|
||||||
|
"VMADDWEVVW": {0xE15A << 15, false}, "VMADDWEVQV": {0xE15B << 15, false},
|
||||||
|
"VMADDWODHB": {0xE15C << 15, false}, "VMADDWODWH": {0xE15D << 15, false},
|
||||||
|
"VMADDWODVW": {0xE15E << 15, false}, "VMADDWODQV": {0xE15F << 15, false},
|
||||||
|
"VMADDWEVHBU": {0xE168 << 15, false}, "VMADDWEVWHU": {0xE169 << 15, false},
|
||||||
|
"VMADDWEVVWU": {0xE16A << 15, false}, "VMADDWEVQVU": {0xE16B << 15, false},
|
||||||
|
"VMADDWODHBU": {0xE16C << 15, false}, "VMADDWODWHU": {0xE16D << 15, false},
|
||||||
|
"VMADDWODVWU": {0xE16E << 15, false}, "VMADDWODQVU": {0xE16F << 15, false},
|
||||||
|
"VMADDWEVHBUB": {0xE178 << 15, false}, "VMADDWEVWHUH": {0xE179 << 15, false},
|
||||||
|
"VMADDWEVVWUW": {0xE17A << 15, false}, "VMADDWEVQVUV": {0xE17B << 15, false},
|
||||||
|
"VMADDWODHBUB": {0xE17C << 15, false}, "VMADDWODWHUH": {0xE17D << 15, false},
|
||||||
|
"VMADDWODVWUW": {0xE17E << 15, false}, "VMADDWODQVUV": {0xE17F << 15, false},
|
||||||
|
"VDIVB": {0xE1C0 << 15, false}, "VDIVH": {0xE1C1 << 15, false},
|
||||||
|
"VDIVW": {0xE1C2 << 15, false}, "VDIVV": {0xE1C3 << 15, false},
|
||||||
|
"VMODB": {0xE1C4 << 15, false}, "VMODH": {0xE1C5 << 15, false},
|
||||||
|
"VMODW": {0xE1C6 << 15, false}, "VMODV": {0xE1C7 << 15, false},
|
||||||
|
"VDIVBU": {0xE1C8 << 15, false}, "VDIVHU": {0xE1C9 << 15, false},
|
||||||
|
"VDIVWU": {0xE1CA << 15, false}, "VDIVVU": {0xE1CB << 15, false},
|
||||||
|
"VMODBU": {0xE1CC << 15, false}, "VMODHU": {0xE1CD << 15, false},
|
||||||
|
"VMODWU": {0xE1CE << 15, false}, "VMODVU": {0xE1CF << 15, false},
|
||||||
|
"VMULF": {0xE271 << 15, false}, "VMULD": {0xE272 << 15, false},
|
||||||
|
"VDIVF": {0xE275 << 15, false}, "VDIVD": {0xE276 << 15, false},
|
||||||
|
"XVMULB": {0xE908 << 15, true}, "XVMULH": {0xE909 << 15, true},
|
||||||
|
"XVMULW": {0xE90A << 15, true}, "XVMULV": {0xE90B << 15, true},
|
||||||
|
"XVMUHB": {0xE90C << 15, true}, "XVMUHH": {0xE90D << 15, true},
|
||||||
|
"XVMUHW": {0xE90E << 15, true}, "XVMUHV": {0xE90F << 15, true},
|
||||||
|
"XVMUHBU": {0xE910 << 15, true}, "XVMUHHU": {0xE911 << 15, true},
|
||||||
|
"XVMUHWU": {0xE912 << 15, true}, "XVMUHVU": {0xE913 << 15, true},
|
||||||
|
"XVMULWEVHB": {0xE920 << 15, true}, "XVMULWEVWH": {0xE921 << 15, true},
|
||||||
|
"XVMULWEVVW": {0xE922 << 15, true}, "XVMULWEVQV": {0xE923 << 15, true},
|
||||||
|
"XVMULWODHB": {0xE924 << 15, true}, "XVMULWODWH": {0xE925 << 15, true},
|
||||||
|
"XVMULWODVW": {0xE926 << 15, true}, "XVMULWODQV": {0xE927 << 15, true},
|
||||||
|
"XVMULWEVHBU": {0xE930 << 15, true}, "XVMULWEVWHU": {0xE931 << 15, true},
|
||||||
|
"XVMULWEVVWU": {0xE932 << 15, true}, "XVMULWEVQVU": {0xE933 << 15, true},
|
||||||
|
"XVMULWODHBU": {0xE934 << 15, true}, "XVMULWODWHU": {0xE935 << 15, true},
|
||||||
|
"XVMULWODVWU": {0xE936 << 15, true}, "XVMULWODQVU": {0xE937 << 15, true},
|
||||||
|
"XVMULWEVHBUB": {0xE940 << 15, true}, "XVMULWEVWHUH": {0xE941 << 15, true},
|
||||||
|
"XVMULWEVVWUW": {0xE942 << 15, true}, "XVMULWEVQVUV": {0xE943 << 15, true},
|
||||||
|
"XVMULWODHBUB": {0xE944 << 15, true}, "XVMULWODWHUH": {0xE945 << 15, true},
|
||||||
|
"XVMULWODVWUW": {0xE946 << 15, true}, "XVMULWODQVUV": {0xE947 << 15, true},
|
||||||
|
"XVMADDB": {0xE950 << 15, true}, "XVMADDH": {0xE951 << 15, true},
|
||||||
|
"XVMADDW": {0xE952 << 15, true}, "XVMADDV": {0xE953 << 15, true},
|
||||||
|
"XVMSUBB": {0xE954 << 15, true}, "XVMSUBH": {0xE955 << 15, true},
|
||||||
|
"XVMSUBW": {0xE956 << 15, true}, "XVMSUBV": {0xE957 << 15, true},
|
||||||
|
"XVMADDWEVHB": {0xE958 << 15, true}, "XVMADDWEVWH": {0xE959 << 15, true},
|
||||||
|
"XVMADDWEVVW": {0xE95A << 15, true}, "XVMADDWEVQV": {0xE95B << 15, true},
|
||||||
|
"XVMADDWODHB": {0xE95C << 15, true}, "XVMADDWODWH": {0xE95D << 15, true},
|
||||||
|
"XVMADDWODVW": {0xE95E << 15, true}, "XVMADDWODQV": {0xE95F << 15, true},
|
||||||
|
"XVMADDWEVHBU": {0xE968 << 15, true}, "XVMADDWEVWHU": {0xE969 << 15, true},
|
||||||
|
"XVMADDWEVVWU": {0xE96A << 15, true}, "XVMADDWEVQVU": {0xE96B << 15, true},
|
||||||
|
"XVMADDWODHBU": {0xE96C << 15, true}, "XVMADDWODWHU": {0xE96D << 15, true},
|
||||||
|
"XVMADDWODVWU": {0xE96E << 15, true}, "XVMADDWODQVU": {0xE96F << 15, true},
|
||||||
|
"XVMADDWEVHBUB": {0xE978 << 15, true}, "XVMADDWEVWHUH": {0xE979 << 15, true},
|
||||||
|
"XVMADDWEVVWUW": {0xE97A << 15, true}, "XVMADDWEVQVUV": {0xE97B << 15, true},
|
||||||
|
"XVMADDWODHBUB": {0xE97C << 15, true}, "XVMADDWODWHUH": {0xE97D << 15, true},
|
||||||
|
"XVMADDWODVWUW": {0xE97E << 15, true}, "XVMADDWODQVUV": {0xE97F << 15, true},
|
||||||
|
"XVDIVB": {0xE9C0 << 15, true}, "XVDIVH": {0xE9C1 << 15, true},
|
||||||
|
"XVDIVW": {0xE9C2 << 15, true}, "XVDIVV": {0xE9C3 << 15, true},
|
||||||
|
"XVMODB": {0xE9C4 << 15, true}, "XVMODH": {0xE9C5 << 15, true},
|
||||||
|
"XVMODW": {0xE9C6 << 15, true}, "XVMODV": {0xE9C7 << 15, true},
|
||||||
|
"XVDIVBU": {0xE9C8 << 15, true}, "XVDIVHU": {0xE9C9 << 15, true},
|
||||||
|
"XVDIVWU": {0xE9CA << 15, true}, "XVDIVVU": {0xE9CB << 15, true},
|
||||||
|
"XVMODBU": {0xE9CC << 15, true}, "XVMODHU": {0xE9CD << 15, true},
|
||||||
|
"XVMODWU": {0xE9CE << 15, true}, "XVMODVU": {0xE9CF << 15, true},
|
||||||
|
"XVMULF": {0xEA71 << 15, true}, "XVMULD": {0xEA72 << 15, true},
|
||||||
|
"XVDIVF": {0xEA75 << 15, true}, "XVDIVD": {0xEA76 << 15, true},
|
||||||
|
}
|
||||||
|
|
||||||
|
// The lane-wise shifts and rotates (three-register forms; the immediate
|
||||||
|
// forms live in l64VecImmInfo), the interleave families, the bit
|
||||||
|
// clear/set/rev register forms, the remaining logic and compare
|
||||||
|
// spellings, the widening add/subtract ladder and the vector FP
|
||||||
|
// arithmetic.
|
||||||
|
vecmisc := map[string]l64Vec3Enc{
|
||||||
|
"VSLLB": {0xE1D0 << 15, false}, "VSLLH": {0xE1D1 << 15, false},
|
||||||
|
"VSLLW": {0xE1D2 << 15, false}, "VSLLV": {0xE1D3 << 15, false},
|
||||||
|
"VSRLB": {0xE1D4 << 15, false}, "VSRLH": {0xE1D5 << 15, false},
|
||||||
|
"VSRLW": {0xE1D6 << 15, false}, "VSRLV": {0xE1D7 << 15, false},
|
||||||
|
"VSRAH": {0xE1D9 << 15, false}, "VSRAW": {0xE1DA << 15, false},
|
||||||
|
"VSRAV": {0xE1DB << 15, false},
|
||||||
|
"VROTRB": {0xE1DC << 15, false}, "VROTRH": {0xE1DD << 15, false},
|
||||||
|
"VROTRV": {0xE1DF << 15, false},
|
||||||
|
"VILVLB": {0xE234 << 15, false}, "VILVLH": {0xE235 << 15, false},
|
||||||
|
"VILVLW": {0xE236 << 15, false}, "VILVLV": {0xE237 << 15, false},
|
||||||
|
"VILVHB": {0xE238 << 15, false}, "VILVHH": {0xE239 << 15, false},
|
||||||
|
"VILVHW": {0xE23A << 15, false}, "VILVHV": {0xE23B << 15, false},
|
||||||
|
"VBITCLRB": {0xE218 << 15, false}, "VBITCLRH": {0xE219 << 15, false},
|
||||||
|
"VBITCLRW": {0xE21A << 15, false}, "VBITCLRV": {0xE21B << 15, false},
|
||||||
|
"VBITSETB": {0xE21C << 15, false}, "VBITSETH": {0xE21D << 15, false},
|
||||||
|
"VBITSETW": {0xE21E << 15, false}, "VBITSETV": {0xE21F << 15, false},
|
||||||
|
"VBITREVB": {0xE220 << 15, false}, "VBITREVH": {0xE221 << 15, false},
|
||||||
|
"VBITREVW": {0xE222 << 15, false}, "VBITREVV": {0xE223 << 15, false},
|
||||||
|
"VORV": {0xE24D << 15, false}, "VNORV": {0xE24F << 15, false},
|
||||||
|
"VANDNV": {0xE250 << 15, false}, "VORNV": {0xE251 << 15, false},
|
||||||
|
"VSEQH": {0xE001 << 15, false}, "VSEQW": {0xE002 << 15, false},
|
||||||
|
"VSLTB": {0xE00C << 15, false}, "VSLTH": {0xE00D << 15, false},
|
||||||
|
"VSLTW": {0xE00E << 15, false}, "VSLTV": {0xE00F << 15, false},
|
||||||
|
"VSLTBU": {0xE010 << 15, false}, "VSLTHU": {0xE011 << 15, false},
|
||||||
|
"VSLTWU": {0xE012 << 15, false}, "VSLTVU": {0xE013 << 15, false},
|
||||||
|
"VADDWEVHB": {0xE03C << 15, false}, "VADDWEVWH": {0xE03D << 15, false},
|
||||||
|
"VADDWEVVW": {0xE03E << 15, false}, "VADDWEVQV": {0xE03F << 15, false},
|
||||||
|
"VSUBWEVHB": {0xE040 << 15, false}, "VSUBWEVWH": {0xE041 << 15, false},
|
||||||
|
"VSUBWEVVW": {0xE042 << 15, false}, "VSUBWEVQV": {0xE043 << 15, false},
|
||||||
|
"VADDWODHB": {0xE044 << 15, false}, "VADDWODWH": {0xE045 << 15, false},
|
||||||
|
"VADDWODVW": {0xE046 << 15, false}, "VADDWODQV": {0xE047 << 15, false},
|
||||||
|
"VSUBWODHB": {0xE048 << 15, false}, "VSUBWODWH": {0xE049 << 15, false},
|
||||||
|
"VSUBWODVW": {0xE04A << 15, false}, "VSUBWODQV": {0xE04B << 15, false},
|
||||||
|
"VSUBWEVHBU": {0xE060 << 15, false}, "VSUBWEVWHU": {0xE061 << 15, false},
|
||||||
|
"VSUBWEVVWU": {0xE062 << 15, false}, "VSUBWEVQVU": {0xE063 << 15, false},
|
||||||
|
"VADDWEVHBU": {0xE05C << 15, false}, "VADDWEVWHU": {0xE05D << 15, false},
|
||||||
|
"VADDWEVVWU": {0xE05E << 15, false}, "VADDWEVQVU": {0xE05F << 15, false},
|
||||||
|
"VADDWODHBU": {0xE064 << 15, false}, "VADDWODWHU": {0xE065 << 15, false},
|
||||||
|
"VADDWODVWU": {0xE066 << 15, false}, "VADDWODQVU": {0xE067 << 15, false},
|
||||||
|
"VSUBWODHBU": {0xE068 << 15, false}, "VSUBWODWHU": {0xE069 << 15, false},
|
||||||
|
"VSUBWODVWU": {0xE06A << 15, false}, "VSUBWODQVU": {0xE06B << 15, false},
|
||||||
|
"VSHUFH": {0xE2F5 << 15, false}, "VSHUFW": {0xE2F6 << 15, false},
|
||||||
|
"VSHUFV": {0xE2F7 << 15, false},
|
||||||
|
"XVSLLB": {0xE9D0 << 15, true}, "XVSLLH": {0xE9D1 << 15, true},
|
||||||
|
"XVSLLW": {0xE9D2 << 15, true}, "XVSLLV": {0xE9D3 << 15, true},
|
||||||
|
"XVSRLB": {0xE9D4 << 15, true}, "XVSRLH": {0xE9D5 << 15, true},
|
||||||
|
"XVSRLW": {0xE9D6 << 15, true}, "XVSRLV": {0xE9D7 << 15, true},
|
||||||
|
"XVSRAB": {0xE9D8 << 15, true}, "XVSRAH": {0xE9D9 << 15, true},
|
||||||
|
"XVSRAW": {0xE9DA << 15, true}, "XVSRAV": {0xE9DB << 15, true},
|
||||||
|
"XVROTRB": {0xE9DC << 15, true}, "XVROTRH": {0xE9DD << 15, true},
|
||||||
|
"XVROTRW": {0xE9DE << 15, true}, "XVROTRV": {0xE9DF << 15, true},
|
||||||
|
"XVILVLB": {0xEA34 << 15, true}, "XVILVLH": {0xEA35 << 15, true},
|
||||||
|
"XVILVLW": {0xEA36 << 15, true}, "XVILVLV": {0xEA37 << 15, true},
|
||||||
|
"XVILVHB": {0xEA38 << 15, true}, "XVILVHH": {0xEA39 << 15, true},
|
||||||
|
"XVILVHW": {0xEA3A << 15, true}, "XVILVHV": {0xEA3B << 15, true},
|
||||||
|
"XVBITCLRB": {0xEA18 << 15, true}, "XVBITCLRH": {0xEA19 << 15, true},
|
||||||
|
"XVBITCLRW": {0xEA1A << 15, true}, "XVBITCLRV": {0xEA1B << 15, true},
|
||||||
|
"XVBITSETB": {0xEA1C << 15, true}, "XVBITSETH": {0xEA1D << 15, true},
|
||||||
|
"XVBITSETW": {0xEA1E << 15, true}, "XVBITSETV": {0xEA1F << 15, true},
|
||||||
|
"XVBITREVB": {0xEA20 << 15, true}, "XVBITREVH": {0xEA21 << 15, true},
|
||||||
|
"XVBITREVW": {0xEA22 << 15, true}, "XVBITREVV": {0xEA23 << 15, true},
|
||||||
|
"XVORV": {0xEA4D << 15, true}, "XVNORV": {0xEA4F << 15, true},
|
||||||
|
"XVANDNV": {0xEA50 << 15, true}, "XVORNV": {0xEA51 << 15, true},
|
||||||
|
"XVSEQH": {0xE801 << 15, true}, "XVSEQW": {0xE802 << 15, true},
|
||||||
|
"XVSLTB": {0xE80C << 15, true}, "XVSLTH": {0xE80D << 15, true},
|
||||||
|
"XVSLTW": {0xE80E << 15, true}, "XVSLTV": {0xE80F << 15, true},
|
||||||
|
"XVSLTBU": {0xE810 << 15, true}, "XVSLTHU": {0xE811 << 15, true},
|
||||||
|
"XVSLTWU": {0xE812 << 15, true}, "XVSLTVU": {0xE813 << 15, true},
|
||||||
|
"XVADDWEVHB": {0xE83C << 15, true}, "XVADDWEVWH": {0xE83D << 15, true},
|
||||||
|
"XVADDWEVVW": {0xE83E << 15, true}, "XVADDWEVQV": {0xE83F << 15, true},
|
||||||
|
"XVSUBWEVHB": {0xE840 << 15, true}, "XVSUBWEVWH": {0xE841 << 15, true},
|
||||||
|
"XVSUBWEVVW": {0xE842 << 15, true}, "XVSUBWEVQV": {0xE843 << 15, true},
|
||||||
|
"XVADDWODHB": {0xE844 << 15, true}, "XVADDWODWH": {0xE845 << 15, true},
|
||||||
|
"XVADDWODVW": {0xE846 << 15, true}, "XVADDWODQV": {0xE847 << 15, true},
|
||||||
|
"XVSUBWODHB": {0xE848 << 15, true}, "XVSUBWODWH": {0xE849 << 15, true},
|
||||||
|
"XVSUBWODVW": {0xE84A << 15, true}, "XVSUBWODQV": {0xE84B << 15, true},
|
||||||
|
"XVADDWEVHBU": {0xE85C << 15, true}, "XVADDWEVWHU": {0xE85D << 15, true},
|
||||||
|
"XVADDWEVVWU": {0xE85E << 15, true}, "XVADDWEVQVU": {0xE85F << 15, true},
|
||||||
|
"XVSUBWEVHBU": {0xE860 << 15, true}, "XVSUBWEVWHU": {0xE861 << 15, true},
|
||||||
|
"XVSUBWEVVWU": {0xE862 << 15, true}, "XVSUBWEVQVU": {0xE863 << 15, true},
|
||||||
|
"XVADDWODHBU": {0xE864 << 15, true}, "XVADDWODWHU": {0xE865 << 15, true},
|
||||||
|
"XVADDWODVWU": {0xE866 << 15, true}, "XVADDWODQVU": {0xE867 << 15, true},
|
||||||
|
"XVSUBWODHBU": {0xE868 << 15, true}, "XVSUBWODWHU": {0xE869 << 15, true},
|
||||||
|
"XVSUBWODVWU": {0xE86A << 15, true}, "XVSUBWODQVU": {0xE86B << 15, true},
|
||||||
|
"XVSHUFH": {0xEAF5 << 15, true}, "XVSHUFW": {0xEAF6 << 15, true},
|
||||||
|
"XVSHUFV": {0xEAF7 << 15, true},
|
||||||
|
}
|
||||||
|
for _, tab := range []map[string]l64Vec3Enc{addsub, muldiv, vecmisc} {
|
||||||
|
for m, e := range tab {
|
||||||
|
if _, dup := vec3[m]; dup {
|
||||||
|
panic("loong64: duplicate vector mnemonic " + m)
|
||||||
|
}
|
||||||
|
vec3[m] = e
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for m, e := range vec3 {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Fvvv, op: e.op}
|
||||||
|
l64VecBank[m] = e.lasx
|
||||||
|
}
|
||||||
|
|
||||||
|
// Immediate forms: INSTR $imm, vj, vd (or INSTR $imm, vd). The immediate
|
||||||
|
// range, bias and field mask are the ones the toolchain encodes: vandi.b
|
||||||
|
// stores the raw 8-bit constant, vsrari.b stores imm+8 (lane-width
|
||||||
|
// bias), the si5 compares store 5-bit two's-complement values and vseqi.d
|
||||||
|
// a 7-bit field the toolchain range-checks down to si5.
|
||||||
|
// The mnemonics that also have a register form (the shifts, the bit
|
||||||
|
// clear/set/rev families, VSEQ and the logic immediates) keep their
|
||||||
|
// three-register entry in l64InstrTable; the dispatcher picks the
|
||||||
|
// immediate opcode from l64VecImmInfo by operand kind, so the immediate
|
||||||
|
// entries must not overwrite the table.
|
||||||
|
vecImm := map[string]l64VecImmEnc{
|
||||||
|
"VANDB": {0xE7A0 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVANDB": {0xEFA0 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VORB": {0xE7A8 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVORB": {0xEFA8 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VXORB": {0xE7B0 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVXORB": {0xEFB0 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSEQB": {0xED00 << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
// vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a
|
||||||
|
// 7-bit field, but the toolchain range-checks it down to si5 as well
|
||||||
|
// (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64).
|
||||||
|
"VSEQH": {0xE501 << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSEQH": {0xED01 << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
"VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
"VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F},
|
||||||
|
"XVSEQV": {0xED03 << 15, true, -16, 15, 0, 0x7F},
|
||||||
|
// vslti compares against a signed (or, in the U spellings, unsigned)
|
||||||
|
// si5/ui5 constant.
|
||||||
|
"VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSLTB": {0xED0C << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
"VSLTH": {0xE50D << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSLTH": {0xED0D << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
"VSLTW": {0xE50E << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSLTW": {0xED0E << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
"VSLTV": {0xE50F << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSLTV": {0xED0F << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
"VSLTBU": {0xE510 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSLTBU": {0xED10 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSLTHU": {0xE511 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSLTHU": {0xED11 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSLTWU": {0xE512 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSLTWU": {0xED12 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSLTVU": {0xE513 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSLTVU": {0xED13 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
// vaddi/vsubi take ui5 constants for every width on this toolchain
|
||||||
|
// (VADDVU $32 is rejected by the oracle although the field is ui8).
|
||||||
|
"VADDBU": {0xE514 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVADDBU": {0xED14 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VADDHU": {0xE515 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVADDHU": {0xED15 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VADDWU": {0xE516 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVADDWU": {0xED16 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VADDVU": {0xE517 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVADDVU": {0xED17 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSUBBU": {0xE518 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSUBBU": {0xED18 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSUBHU": {0xE519 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSUBHU": {0xED19 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSUBWU": {0xE51A << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSUBWU": {0xED1A << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSUBVU": {0xE51B << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSUBVU": {0xED1B << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
// The shift/rotate immediates ride in a width-sized field whose upper
|
||||||
|
// bits carry the lane-width code: vslli.b stores ui3 at [12:0] with
|
||||||
|
// bits [14:13] inside the opcode, vslli.h ui4 under a 4 bit mask, and
|
||||||
|
// the .w/.d spellings a raw ui5/ui6.
|
||||||
|
"VSLLB": {0x732C2000, false, 0, 7, 0, 0x7},
|
||||||
|
"XVSLLB": {0x772C2000, true, 0, 7, 0, 0x7},
|
||||||
|
"VSLLH": {0x732C4000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVSLLH": {0x772C4000, true, 0, 15, 0, 0xF},
|
||||||
|
"VSLLW": {0xE659 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSLLW": {0xEE59 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSLLV": {0xE65A << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVSLLV": {0xEE5A << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
"VSRLB": {0x73302000, false, 0, 7, 0, 0x7},
|
||||||
|
"XVSRLB": {0x77302000, true, 0, 7, 0, 0x7},
|
||||||
|
"VSRLH": {0x73304000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVSRLH": {0x77304000, true, 0, 15, 0, 0xF},
|
||||||
|
"VSRLW": {0xE661 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSRLW": {0xEE61 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSRLV": {0xE662 << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVSRLV": {0xEE62 << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
// vsrari/vrotri bias the field so the lane-width code rides above the
|
||||||
|
// shift amount (.b adds 8, .h 16, .w 32; .d is a raw ui6).
|
||||||
|
"VSRAB": {0xE668 << 15, false, 0, 7, 8, 0x1F},
|
||||||
|
"XVSRAB": {0xEE68 << 15, true, 0, 7, 8, 0x1F},
|
||||||
|
"VSRAH": {0x73344000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVSRAH": {0x77344000, true, 0, 15, 0, 0xF},
|
||||||
|
"VSRAW": {0xE669 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSRAW": {0xEE69 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSRAV": {0xE66A << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVSRAV": {0xEE6A << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
"VROTRB": {0x72A02000, false, 0, 7, 0, 0x7},
|
||||||
|
"XVROTRB": {0x76A02000, true, 0, 7, 0, 0x7},
|
||||||
|
"VROTRH": {0x72A04000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVROTRH": {0x76A04000, true, 0, 15, 0, 0xF},
|
||||||
|
"VROTRW": {0xE541 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVROTRW": {0xED41 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VROTRV": {0xE542 << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVROTRV": {0xED42 << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
// vbitclri/vbitseti/vbitrevi follow the same width-coded layout.
|
||||||
|
"VBITCLRB": {0x73102000, false, 0, 7, 0, 0x7},
|
||||||
|
"XVBITCLRB": {0x77102000, true, 0, 7, 0, 0x7},
|
||||||
|
"VBITCLRH": {0x73104000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVBITCLRH": {0x77104000, true, 0, 15, 0, 0xF},
|
||||||
|
"VBITCLRW": {0xE621 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVBITCLRW": {0xEE21 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VBITCLRV": {0xE622 << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVBITCLRV": {0xEE22 << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
"VBITSETB": {0x73142000, false, 0, 7, 0, 0x7},
|
||||||
|
"XVBITSETB": {0x77142000, true, 0, 7, 0, 0x7},
|
||||||
|
"VBITSETH": {0x73144000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVBITSETH": {0x77144000, true, 0, 15, 0, 0xF},
|
||||||
|
"VBITSETW": {0xE629 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVBITSETW": {0xEE29 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VBITSETV": {0xE62A << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVBITSETV": {0xEE2A << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
"VBITREVB": {0x73182000, false, 0, 7, 0, 0x7},
|
||||||
|
"XVBITREVB": {0x77182000, true, 0, 7, 0, 0x7},
|
||||||
|
"VBITREVH": {0x73184000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVBITREVH": {0x77184000, true, 0, 15, 0, 0xF},
|
||||||
|
"VBITREVW": {0xE631 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVBITREVW": {0xEE31 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VBITREVV": {0xE632 << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVBITREVV": {0xEE32 << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
// The 4-bit-select shuffles and the byte-extract/insert permutations
|
||||||
|
// take ui8 (the .d shuffle ui4 range-checked to 0..15 by the
|
||||||
|
// toolchain) packing both position nibbles.
|
||||||
|
"VSHUF4IB": {0xE720 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVSHUF4IB": {0xEF20 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VSHUF4IH": {0xE728 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVSHUF4IH": {0xEF28 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VSHUF4IW": {0xE730 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVSHUF4IW": {0xEF30 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VSHUF4IV": {0xE738 << 15, false, 0, 15, 0, 0xFF},
|
||||||
|
"XVSHUF4IV": {0xEF38 << 15, true, 0, 15, 0, 0xFF},
|
||||||
|
"VPERMIW": {0xE7C8 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVPERMIW": {0xEFC8 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"XVPERMIV": {0xEFD0 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"XVPERMIQ": {0xEFD8 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VEXTRINSB": {0xE718 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVEXTRINSB": {0xEF18 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VEXTRINSH": {0xE710 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVEXTRINSH": {0xEF10 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VEXTRINSW": {0xE708 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVEXTRINSW": {0xEF08 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VEXTRINSV": {0xE700 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVEXTRINSV": {0xEF00 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
}
|
||||||
|
for m, e := range vecImm {
|
||||||
|
l64VecImmInfo[m] = e
|
||||||
|
l64VecBank[m] = e.lasx
|
||||||
|
}
|
||||||
|
|
||||||
|
// Vector-to-condition flag: INSTR vj, FCCn (vsetnez.v, vsetanyeqz.*,
|
||||||
|
// vsetallnez.*): the sub-op rides in the rk field.
|
||||||
|
vecCf := map[string]uint32{
|
||||||
|
"VSETNEV": 0xE539<<15 | 7<<10, "XVSETNEV": 0xED39<<15 | 7<<10,
|
||||||
|
"VSETANYEQB": 0xE539<<15 | 8<<10, "XVSETANYEQB": 0xED39<<15 | 8<<10,
|
||||||
|
"VSETANYEQV": 0xE539<<15 | 11<<10, "XVSETANYEQV": 0xED39<<15 | 11<<10,
|
||||||
|
"VSETALLNEV": 0xE539<<15 | 15<<10, "XVSETALLNEV": 0xED39<<15 | 15<<10,
|
||||||
|
"VSETEQV": 0xE539<<15 | 6<<10, "XVSETEQV": 0xED39<<15 | 6<<10,
|
||||||
|
"VSETANYEQH": 0xE539<<15 | 9<<10, "XVSETANYEQH": 0xED39<<15 | 9<<10,
|
||||||
|
"VSETANYEQW": 0xE539<<15 | 10<<10, "XVSETANYEQW": 0xED39<<15 | 10<<10,
|
||||||
|
"VSETALLNEB": 0xE539<<15 | 12<<10, "XVSETALLNEB": 0xED39<<15 | 12<<10,
|
||||||
|
"VSETALLNEH": 0xE539<<15 | 13<<10, "XVSETALLNEH": 0xED39<<15 | 13<<10,
|
||||||
|
"VSETALLNEW": 0xE539<<15 | 14<<10, "XVSETALLNEW": 0xED39<<15 | 14<<10,
|
||||||
|
}
|
||||||
|
for m, op := range vecCf {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Fvcf, op: op}
|
||||||
|
l64VecBank[m] = strings.HasPrefix(m, "XV")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Lane popcount and the two-operand vector FP/unary spellings: INSTR vj,
|
||||||
|
// vd (the 2R layout with the opcode extending over the unused vk field;
|
||||||
|
// the low byte of each constant is the instruction's own sub-op).
|
||||||
|
vec2r := map[string]l64Vec3Enc{
|
||||||
|
"VPCNTV": {0x1CA70B << 10, false}, "XVPCNTV": {0x1DA70B << 10, true},
|
||||||
|
}
|
||||||
|
// The rest of the lane popcounts, the vector negations and the vector FP
|
||||||
|
// unary conversions (loong64enc1.s).
|
||||||
|
vec2rMore := map[string]l64Vec3Enc{
|
||||||
|
"VPCNTB": {0x1CA708 << 10, false}, "VPCNTH": {0x1CA709 << 10, false},
|
||||||
|
"VPCNTW": {0x1CA70A << 10, false},
|
||||||
|
"VNEGB": {0x1CA70C << 10, false}, "VNEGH": {0x1CA70D << 10, false},
|
||||||
|
"VNEGW": {0x1CA70E << 10, false}, "VNEGV": {0x1CA70F << 10, false},
|
||||||
|
"VFCLASSF": {0x1CA735 << 10, false}, "VFCLASSD": {0x1CA736 << 10, false},
|
||||||
|
"VFSQRTF": {0x1CA739 << 10, false}, "VFSQRTD": {0x1CA73A << 10, false},
|
||||||
|
"VFRECIPF": {0x1CA73D << 10, false}, "VFRECIPD": {0x1CA73E << 10, false},
|
||||||
|
"VFRSQRTF": {0x1CA741 << 10, false}, "VFRSQRTD": {0x1CA742 << 10, false},
|
||||||
|
"VFRINTF": {0x1CA74D << 10, false}, "VFRINTD": {0x1CA74E << 10, false},
|
||||||
|
"VFRINTRMF": {0x1CA751 << 10, false}, "VFRINTRMD": {0x1CA752 << 10, false},
|
||||||
|
"VFRINTRPF": {0x1CA755 << 10, false}, "VFRINTRPD": {0x1CA756 << 10, false},
|
||||||
|
"VFRINTRZF": {0x1CA759 << 10, false}, "VFRINTRZD": {0x1CA75A << 10, false},
|
||||||
|
"VFRINTRNEF": {0x1CA75D << 10, false}, "VFRINTRNED": {0x1CA75E << 10, false},
|
||||||
|
"XVPCNTB": {0x1DA708 << 10, true}, "XVPCNTH": {0x1DA709 << 10, true},
|
||||||
|
"XVPCNTW": {0x1DA70A << 10, true},
|
||||||
|
"XVNEGB": {0x1DA70C << 10, true}, "XVNEGH": {0x1DA70D << 10, true},
|
||||||
|
"XVNEGW": {0x1DA70E << 10, true}, "XVNEGV": {0x1DA70F << 10, true},
|
||||||
|
"XVFCLASSF": {0x1DA735 << 10, true}, "XVFCLASSD": {0x1DA736 << 10, true},
|
||||||
|
"XVFSQRTF": {0x1DA739 << 10, true}, "XVFSQRTD": {0x1DA73A << 10, true},
|
||||||
|
"XVFRECIPF": {0x1DA73D << 10, true}, "XVFRECIPD": {0x1DA73E << 10, true},
|
||||||
|
"XVFRSQRTF": {0x1DA741 << 10, true}, "XVFRSQRTD": {0x1DA742 << 10, true},
|
||||||
|
"XVFRINTF": {0x1DA74D << 10, true}, "XVFRINTD": {0x1DA74E << 10, true},
|
||||||
|
"XVFRINTRMF": {0x1DA751 << 10, true}, "XVFRINTRMD": {0x1DA752 << 10, true},
|
||||||
|
"XVFRINTRPF": {0x1DA755 << 10, true}, "XVFRINTRPD": {0x1DA756 << 10, true},
|
||||||
|
"XVFRINTRZF": {0x1DA759 << 10, true}, "XVFRINTRZD": {0x1DA75A << 10, true},
|
||||||
|
"XVFRINTRNEF": {0x1DA75D << 10, true}, "XVFRINTRNED": {0x1DA75E << 10, true},
|
||||||
|
}
|
||||||
|
maps.Copy(vec2r, vec2rMore)
|
||||||
|
for m, e := range vec2r {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Frr, op: e.op}
|
||||||
|
l64VecBank[m] = e.lasx
|
||||||
|
l64Vec2R[m] = true
|
||||||
|
}
|
||||||
|
|
||||||
|
// The four-register byte shuffle: INSTR va, vk, vj, vd (the operand the
|
||||||
|
// table reads in each field position, va at bits [19:15]).
|
||||||
|
vec4r := map[string]l64Vec3Enc{
|
||||||
|
"VSHUFB": {0x0D50 << 16, false}, "XVSHUFB": {0x0D60 << 16, true},
|
||||||
|
}
|
||||||
|
for m, e := range vec4r {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Fvvvv, op: e.op}
|
||||||
|
l64VecBank[m] = e.lasx
|
||||||
|
l64Vec4R[m] = true
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
|
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
|
||||||
|
|||||||
+769
-2
@@ -8,8 +8,8 @@ import (
|
|||||||
"encoding/binary"
|
"encoding/binary"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// firstTextLOONG64 parses assembly source and returns the first TEXT body.
|
// firstTextLOONG64 parses assembly source and returns the first TEXT body.
|
||||||
@@ -263,6 +263,20 @@ func TestLOONG64_regNames(t *testing.T) {
|
|||||||
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
|
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// The X/V spellings name the LSX/LASX vector banks, a register class of
|
||||||
|
// their own: the oracle (GOARCH=loong64 go tool asm) rejects `BEQZ X0`
|
||||||
|
// with "unrecognized instruction" while assembling `VADDV V0, V1, V2`
|
||||||
|
// and `XVADDV X0, X1, X2`, so loong64RegNum stays strict and the vector
|
||||||
|
// operands resolve through loong64VecRegNum only.
|
||||||
|
vecCases := map[string]int{
|
||||||
|
"V0": 0, "V31": 31, "X0": 0, "X31": 31,
|
||||||
|
"R4": -1, "F0": -1, "FCC0": -1, "V32": -1, "X32": -1, "V": -1, "X": -1,
|
||||||
|
}
|
||||||
|
for name, want := range vecCases {
|
||||||
|
if got := loong64VecRegNum(name); got != want {
|
||||||
|
t.Errorf("loong64VecRegNum(%q) = %d, want %d", name, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
|
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
|
||||||
@@ -291,3 +305,756 @@ done:
|
|||||||
t.Errorf("code = % x\nwant % x", code, want)
|
t.Errorf("code = % x\nwant % x", code, want)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestLOONG64IndirectBranch pins the indirect branch encodings: JMP (Rj) and
|
||||||
|
// JAL (Rj) lower to jirl, and the raw JIRL spelling encodes the written
|
||||||
|
// offset (the Go loong64 assembler deletes raw JIRL instructions entirely,
|
||||||
|
// so this form is a gasm-only superset with faithful semantics).
|
||||||
|
func TestLOONG64IndirectBranch(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·f(SB), NOSPLIT, $0-0
|
||||||
|
JMP (R4)
|
||||||
|
JIRL R0, R4, 8
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x4C000080, // jirl r0, r4, 0
|
||||||
|
0x4C002080, // jirl r0, r4, 8
|
||||||
|
0x4C000020, // jirl r0, r1, 0 (RET)
|
||||||
|
)
|
||||||
|
|
||||||
|
// JAL (R5) links, so the toolchain gives the function its autosize-8
|
||||||
|
// prologue and epilogue around the call and the closing RET.
|
||||||
|
fn = firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·f(SB), NOSPLIT, $0-0
|
||||||
|
JAL (R5)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code = assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x29FFE061, // st.d r1, -8(r3) (prologue saves RA below the new SP)
|
||||||
|
0x02FFE063, // addi.d r3, r3, -8 (prologue opens the frame)
|
||||||
|
0x29C00061, // st.d r1, 0(r3) (prologue saves RA at SP)
|
||||||
|
0x4C0000A1, // jirl r1, r5, 0
|
||||||
|
0x28C00061, // ld.d r1, 0(r3) (epilogue restores RA)
|
||||||
|
0x02C02063, // addi.d r3, r3, 8
|
||||||
|
0x4C000020, // jirl r0, r1, 0 (RET)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_vector pins the LSX/LASX slice against words read off
|
||||||
|
// GOARCH=loong64 go tool asm (cross-checked against the toolchain's own
|
||||||
|
// loong64enc1.s): the three-register forms, the immediate forms with their
|
||||||
|
// biases, the vector-to-condition forms, lane popcount, the FP conversion,
|
||||||
|
// FSEL and the VMOVQ move family.
|
||||||
|
func TestLOONG64_vector(t *testing.T) {
|
||||||
|
t.Run("three-register and immediate forms", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VADDV V1, V2, V3
|
||||||
|
VADDW V1, V2, V3
|
||||||
|
VADDV V2, V1
|
||||||
|
VANDV V1, V2
|
||||||
|
VXORV V1, V2, V3
|
||||||
|
VSEQB V1, V2, V3
|
||||||
|
VSEQV V1, V2, V3
|
||||||
|
VSRAB V1, V2, V3
|
||||||
|
VROTRW V1, V2, V3
|
||||||
|
VANDB $0, V2, V3
|
||||||
|
VANDB $255, V2
|
||||||
|
VSEQB $3, V2, V3
|
||||||
|
VSEQV $15, V2, V3
|
||||||
|
VSEQV $-15, V2, V3
|
||||||
|
VSRAB $7, V1, V2
|
||||||
|
VROTRW $16, V1, V2
|
||||||
|
VPCNTV V1, V2
|
||||||
|
XVADDV X1, X2, X3
|
||||||
|
XVXORV X1, X2, X3
|
||||||
|
XVSEQB X1, X2, X3
|
||||||
|
XVPCNTV X1, X2
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x700B8443, // vadd.v v3, v2, v1
|
||||||
|
0x700B0443, // vadd.w
|
||||||
|
0x700B8821, // vadd.v v1, v1, v2 (two-operand form)
|
||||||
|
0x71260442, // vand.v v2, v2, v1
|
||||||
|
0x71270443, // vxor.v
|
||||||
|
0x70000443, // vseq.b
|
||||||
|
0x70018443, // vseq.d
|
||||||
|
0x70EC0443, // vsra.b
|
||||||
|
0x70EF0443, // vrotr.w
|
||||||
|
0x73D00043, // vandi.b v3, v2, 0
|
||||||
|
0x73D3FC42, // vandi.b v2, v2, 255 (two-operand form)
|
||||||
|
0x72800C43, // vseqi.b v3, v2, 3
|
||||||
|
0x7281BC43, // vseqi.d v3, v2, 15
|
||||||
|
0x7281C443, // vseqi.d v3, v2, -15 (7-bit two's complement)
|
||||||
|
0x73343C22, // vsrai.b v2, v1, 7 (encoded as 7+8)
|
||||||
|
0x72A0C022, // vrotri.w v2, v1, 16
|
||||||
|
0x729C2C22, // vpcnt.d v2, v1
|
||||||
|
0x740B8443, // xvadd.d x3, x2, x1
|
||||||
|
0x75270443, // xvxor.d
|
||||||
|
0x74000443, // xvseq.b
|
||||||
|
0x769C2C22, // xvpcnt.d x2, x1
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("vector-to-condition", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VSETNEV V1, FCC0
|
||||||
|
VSETANYEQB V1, FCC0
|
||||||
|
VSETANYEQV V2, FCC0
|
||||||
|
VSETALLNEV V0, FCC0
|
||||||
|
XVSETNEV X1, FCC0
|
||||||
|
XVSETALLNEV X1, FCC0
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x729C9C20, // vsetnez.d fcc0, v1
|
||||||
|
0x729CA020, // vsetanyeqz.b
|
||||||
|
0x729CAC40, // vsetanyeqz.d
|
||||||
|
0x729CBC00, // vsetallnez.d
|
||||||
|
0x769C9C20, // xvsetnez.d
|
||||||
|
0x769CBC20, // xvsetallnez.d
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("FP convert and FSEL", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
FFINTDV F0, F1
|
||||||
|
FSEL FCC0, F3, F4, F3
|
||||||
|
FSEL FCC1, F1, F2
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x011D2801, // ffint.d.v f1, f0
|
||||||
|
0x0D000C83, // fsel f3, f4, f3, fcc0
|
||||||
|
0x0D008442, // fsel f2, f2, f1, fcc1
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("VMOVQ move family", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VMOVQ V1, V9
|
||||||
|
VMOVQ (R4), V2
|
||||||
|
VMOVQ 16(R4), V2
|
||||||
|
VMOVQ V0, (R4)
|
||||||
|
VMOVQ V0, 32(R4)
|
||||||
|
VMOVQ (R4)(R7), V3
|
||||||
|
VMOVQ V3, (R4)(R7)
|
||||||
|
VMOVQ R6, V0.B16
|
||||||
|
VMOVQ R6, V12.W4
|
||||||
|
VMOVQ (R4), V4.W4
|
||||||
|
XVMOVQ X3, X7
|
||||||
|
XVMOVQ (R4), X2
|
||||||
|
XVMOVQ X0, (R4)
|
||||||
|
XVMOVQ (R4)(R7), X4
|
||||||
|
XVMOVQ X0, (R4)(R7)
|
||||||
|
XVMOVQ R6, X0.B32
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x732D0029, // vori.b v9, v1, 0 (register move)
|
||||||
|
0x2C000082, // vld v2, r4, 0
|
||||||
|
0x2C004082, // vld v2, r4, 16
|
||||||
|
0x2C400080, // vst v0, r4, 0
|
||||||
|
0x2C408080, // vst v0, r4, 32
|
||||||
|
0x38401C83, // vldx v3, r4, r7
|
||||||
|
0x38441C83, // vstx v3, r4, r7
|
||||||
|
0x729F00C0, // vreplgr2vr.b v0, r6
|
||||||
|
0x729F08CC, // vreplgr2vr.w v12, r6
|
||||||
|
0x30200084, // vldrepl.w v4, r4, 0
|
||||||
|
0x772D0067, // xvori.b x7, x3, 0
|
||||||
|
0x2C800082, // xvld x2, r4, 0
|
||||||
|
0x2CC00080, // xvst x0, r4, 0
|
||||||
|
0x38481C84, // xvldx x4, r4, r7
|
||||||
|
0x384C1C80, // xvstx x0, r4, r7
|
||||||
|
0x769F00C0, // xvreplgr2vr.b x0, r6
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("element extract and insert", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VMOVQ V0.V[0], R10
|
||||||
|
VMOVQ V6.V[1], R8
|
||||||
|
VMOVQ R9, V1.V[0]
|
||||||
|
XVMOVQ X0.V[0], R10
|
||||||
|
XVMOVQ X5.W[7], R7
|
||||||
|
XVMOVQ R4, X7.V[3]
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x72EFF00A, // vpickve2gr.d r10, v0, 0
|
||||||
|
0x72EFF4C8, // vpickve2gr.d r8, v6, 1
|
||||||
|
0x72EBF121, // vinsgr2vr.d v1, r9, 0
|
||||||
|
0x76EFE00A, // xvpickve2gr.d r10, x0, 0
|
||||||
|
0x76EFDCA7, // xvpickve2gr.w r7, x5, 7
|
||||||
|
0x76EBEC87, // xvinsgr2vr.d x7, r4, 3
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
// The integer and FP add/subtract families with their saturating pairs
|
||||||
|
// and immediate spellings (loong64enc1.s words).
|
||||||
|
t.Run("add and subtract families", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VADDB V1, V2, V3
|
||||||
|
VADDF V1, V2, V3
|
||||||
|
VADDD V1, V2, V3
|
||||||
|
VSUBD V1, V2, V3
|
||||||
|
VSADDV V1, V2, V3
|
||||||
|
VSSUBVU V1, V2, V3
|
||||||
|
VADDBU $1, V2, V1
|
||||||
|
VADDBU $1, V2
|
||||||
|
VSUBVU $31, V2
|
||||||
|
XVSADDV X3, X2, X1
|
||||||
|
XVSUBD X1, X2, X3
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x700A0443, // vadd.b
|
||||||
|
0x71308443, // vadd.f
|
||||||
|
0x71310443, // vadd.d
|
||||||
|
0x71330443, // vsub.d
|
||||||
|
0x70478443, // vsadd.v
|
||||||
|
0x704D8443, // vssub.u.d
|
||||||
|
0x728A0441, // vaddi.bu v1, v2, 1
|
||||||
|
0x728A0442, // vaddi.bu v2, v2, 1 (two-operand form)
|
||||||
|
0x728DFC42, // vsubi.du v2, v2, 31 (two-operand form)
|
||||||
|
0x74478C41, // xvsadd.d x1, x2, x3
|
||||||
|
0x75330443, // xvsub.d x3, x2, x1
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
// The multiply, divide and accumulate families.
|
||||||
|
t.Run("multiply and divide families", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VMULV V1, V2, V3
|
||||||
|
VMUHHU V1, V2, V3
|
||||||
|
VDIVBU V1, V2, V3
|
||||||
|
VMODV V1, V2, V3
|
||||||
|
VMADDB V1, V2, V3
|
||||||
|
VMSUBV V1, V2, V3
|
||||||
|
VMULWEVHB V1, V2, V3
|
||||||
|
VMULWODQV V1, V2, V3
|
||||||
|
VMADDWEVHBUB V1, V2, V3
|
||||||
|
XVDIVD X1, X2, X3
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x70858443, // vmul.v
|
||||||
|
0x70888443, // vmuh.u.d
|
||||||
|
0x70E40443, // vdiv.u.b
|
||||||
|
0x70E38443, // vmod.d
|
||||||
|
0x70A80443, // vmadd.b
|
||||||
|
0x70AB8443, // vmsub.d
|
||||||
|
0x70900443, // vmulwev.h.b
|
||||||
|
0x70938443, // vmulwod.q.d
|
||||||
|
0x70BC0443, // vmaddwev.h.bu.b
|
||||||
|
0x753B0443, // xvdiv.d
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
// The shift, bit and interleave families in register and immediate
|
||||||
|
// spellings, with the width-coded shift immediates.
|
||||||
|
t.Run("shift, bit and interleave families", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VSLLV V1, V2, V3
|
||||||
|
VROTRB V1, V2, V3
|
||||||
|
VBITCLRV V1, V2, V3
|
||||||
|
VBITSETW V1, V2, V3
|
||||||
|
VBITREVV V1, V2, V3
|
||||||
|
VILVLB V1, V2, V3
|
||||||
|
VILVHV V1, V2, V3
|
||||||
|
VSLLB $7, V1, V2
|
||||||
|
VSLLB $5, V1
|
||||||
|
VSRLH $15, V1, V2
|
||||||
|
VSRAW $31, V1, V2
|
||||||
|
VSRAV $63, V1, V2
|
||||||
|
VROTRV $63, V1, V2
|
||||||
|
VBITCLRB $7, V2, V3
|
||||||
|
VBITREVV $63, V2, V3
|
||||||
|
VSEQH $-16, V2, V3
|
||||||
|
VSLTB $1, V2, V3
|
||||||
|
VSLTHU $31, V2, V3
|
||||||
|
XVILVLV X3, X2, X1
|
||||||
|
XVSLLB $7, X2, X1
|
||||||
|
XVSRAV $63, X2, X1
|
||||||
|
XVBITREVV $63, X2, X1
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x70E98443, // vsll.d
|
||||||
|
0x70EE0443, // vrotr.b
|
||||||
|
0x710D8443, // vbitclr.d
|
||||||
|
0x710F0443, // vbitset.w
|
||||||
|
0x71118443, // vbitrev.d
|
||||||
|
0x711A0443, // vilvl.b
|
||||||
|
0x711D8443, // vilvh.d
|
||||||
|
0x732C3C22, // vslli.b v2, v1, 7
|
||||||
|
0x732C3421, // vslli.b v1, v1, 5 (two-operand form)
|
||||||
|
0x73307C22, // vsrli.h v2, v1, 15
|
||||||
|
0x7334FC22, // vsrai.w v2, v1, 31
|
||||||
|
0x7335FC22, // vsrai.d v2, v1, 63
|
||||||
|
0x72A1FC22, // vrotri.d v2, v1, 63
|
||||||
|
0x73103C43, // vbitclri.b v3, v2, 7
|
||||||
|
0x7319FC43, // vbitrevi.d v3, v2, 63
|
||||||
|
0x7280C043, // vseqi.h v3, v2, -16
|
||||||
|
0x72860443, // vslti.b v3, v2, 1
|
||||||
|
0x7288FC43, // vslti.hu v3, v2, 31
|
||||||
|
0x751B8C41, // xvilvl.d x1, x2, x3
|
||||||
|
0x772C3C41, // xvslli.b x1, x2, 7
|
||||||
|
0x7735FC41, // xvsrai.d x1, x2, 63
|
||||||
|
0x7719FC41, // xvbitrevi.d x1, x2, 63
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
// The shuffle, select and permutation families, including the
|
||||||
|
// four-register byte shuffle.
|
||||||
|
t.Run("shuffle and permutation families", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VSHUFH V1, V2, V3
|
||||||
|
VSHUFW V1, V2, V3
|
||||||
|
VSHUFV V1, V2, V3
|
||||||
|
VSHUFB V1, V2, V3, V4
|
||||||
|
XVSHUFB X1, X2, X3, X4
|
||||||
|
VSHUF4IB $255, V2, V1
|
||||||
|
VSHUF4IV $15, V2, V1
|
||||||
|
XVSHUF4IV $15, X1, X2
|
||||||
|
VEXTRINSB $0x18, V1, V2
|
||||||
|
XVEXTRINSV $0x81, X1, X2
|
||||||
|
VPERMIW $0x1B, V1, V2
|
||||||
|
XVPERMIQ $0x4B, X1, X2
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x717A8443, // vshuf.h
|
||||||
|
0x717B0443, // vshuf.w
|
||||||
|
0x717B8443, // vshuf.d
|
||||||
|
0x0D508864, // vshuf.b v4, v3, v2, v1
|
||||||
|
0x0D608864, // xvshuf.b
|
||||||
|
0x7393FC41, // vshuf4i.b v1, v2, 255
|
||||||
|
0x739C3C41, // vshuf4i.d v1, v2, 15
|
||||||
|
0x779C3C22, // xvshuf4i.d x2, x1, 15
|
||||||
|
0x738C6022, // vextrins.b v2, v1, 0x18
|
||||||
|
0x77820422, // xvextrins.d x2, x1, 0x81
|
||||||
|
0x73E46C22, // vpermi.w v2, v1, 0x1b
|
||||||
|
0x77ED2C22, // xvpermi.q x2, x1, 0x4b
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
// The vector FP families, the unary spellings, the compare-to-flag
|
||||||
|
// additions and the scalar int/float conversions.
|
||||||
|
t.Run("FP and conversion families", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VADDF V1, V2, V3
|
||||||
|
VMULF V1, V2, V3
|
||||||
|
VFCLASSD V1, V2
|
||||||
|
VFSQRTF V1, V2
|
||||||
|
VFRECIPD V1, V2
|
||||||
|
VFRSQRTF V1, V2
|
||||||
|
VFRINTF V1, V2
|
||||||
|
VFRINTRNED V1, V2
|
||||||
|
VNEGB V1, V2
|
||||||
|
VPCNTB V1, V2
|
||||||
|
XVNEGV X2, X1
|
||||||
|
XVPCNTW X3, X2
|
||||||
|
XVFRINTRNEF X1, X2
|
||||||
|
VSETEQV V1, FCC0
|
||||||
|
VSETANYEQH V1, FCC0
|
||||||
|
VSETALLNEB V1, FCC0
|
||||||
|
XVSETALLNEW X1, FCC0
|
||||||
|
FFINTFW F0, F1
|
||||||
|
FTINTVD F0, F1
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x71308443, // vfadd.s
|
||||||
|
0x71388443, // vfmul.s
|
||||||
|
0x729CD822, // vfclass.d
|
||||||
|
0x729CE422, // vfsqrt.s
|
||||||
|
0x729CF822, // vfrecip.d
|
||||||
|
0x729D0422, // vfrsqrt.s
|
||||||
|
0x729D3422, // vfrint.s
|
||||||
|
0x729D7822, // vfrintne.s
|
||||||
|
0x729C3022, // vneg.b
|
||||||
|
0x729C2022, // vpcnt.b
|
||||||
|
0x769C3C41, // xvneg.d x1, x2
|
||||||
|
0x769C2862, // xvpcnt.w x2, x3
|
||||||
|
0x769D7422, // xvfrintne.s x2, x1
|
||||||
|
0x729C9820, // vseteqz.d fcc0, v1
|
||||||
|
0x729CA420, // vsetanyeqz.h
|
||||||
|
0x729CB020, // vsetallnez.b
|
||||||
|
0x769CB820, // xvsetallnez.w
|
||||||
|
0x011D1001, // ffint.s.w f1, f0
|
||||||
|
0x011B2801, // ftint.l.d f1, f0
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_vectorErrors pins the register-class and range diagnostics of
|
||||||
|
// the vector slice; each shape is rejected by the oracle as well
|
||||||
|
// (GOARCH=loong64 go tool asm).
|
||||||
|
func TestLOONG64_vectorErrors(t *testing.T) {
|
||||||
|
cases := []string{
|
||||||
|
// Integer registers in vector positions.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VADDV R4, R5, R6
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// Crossed banks: LSX spellings take V, LASX spellings X.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VADDV X1, X2, X3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
XVADDV V1, V2, V3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// The LASX bank has no .b/.h element forms.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
XVMOVQ R4, X2.B[0]
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// Immediate ranges.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VANDB $256, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSEQB $16, V2, V3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VROTRW $32, V1, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VADDVU $32, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSEQV $32, V2, V3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSHUF4IV $16, V2, V1
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VEXTRINSB $256, V1, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSLTV $-17, V2, V3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// VSHUFB wants four vector registers.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSHUFB V1, V2, V3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// The FCC forms still refuse vector registers.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSETEQV V1, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// VSET* wants an FCC flag, not a vector register.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSETNEV V1, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
}
|
||||||
|
for i, src := range cases {
|
||||||
|
fn := firstTextLOONG64(t, src)
|
||||||
|
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||||
|
t.Errorf("case %d: expected an error, got none", i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_dbarAtomics pins the _dbar (acquire/release) AMO variants.
|
||||||
|
// The oracle words come from GOARCH=loong64 go tool objdump of kernels
|
||||||
|
// assembled with go tool asm, and match the toolchain's loong64enc1.s.
|
||||||
|
func TestLOONG64_dbarAtomics(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·atoms(SB), NOSPLIT, $0
|
||||||
|
AMADDDBW R14, (R13), R12
|
||||||
|
AMADDDBV R14, (R13), R12
|
||||||
|
AMANDDBW R5, (R4), R6
|
||||||
|
AMANDDBV R5, (R4), R6
|
||||||
|
AMORDBW R5, (R4), R0
|
||||||
|
AMORDBV R5, (R4), R6
|
||||||
|
AMSWAPDBW R5, (R4), R6
|
||||||
|
AMCASDBV R6, (R4), R5
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x386A39AC, // amadd_db.w r12, r13, r14
|
||||||
|
0x386AB9AC, // amadd_db.d
|
||||||
|
0x386B1486, // amand_db.w r6, r4, r5
|
||||||
|
0x386B9486, // amand_db.d
|
||||||
|
0x386C1480, // amor_db.w r0, r4, r5
|
||||||
|
0x386C9486, // amor_db.d
|
||||||
|
0x38691486, // amswap_db.w
|
||||||
|
0x385B9885, // amcas_db.w
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_llacqScrel pins the acquire/release LL/SC pair. The oracle
|
||||||
|
// words come from GOARCH=loong64 go tool objdump and the toolchain's own
|
||||||
|
// loong64enc1.s golden bytes.
|
||||||
|
func TestLOONG64_llacqScrel(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·llsc(SB), NOSPLIT, $0
|
||||||
|
LLACQW (R5), R4
|
||||||
|
LLACQV (R5), R4
|
||||||
|
SCRELW R4, (R6)
|
||||||
|
SCRELV R4, (R6)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x385780A4, // ll.acq.w r4, r5
|
||||||
|
0x385788A4, // ll.acq.d r4, r5
|
||||||
|
0x385784C4, // sc.rel.w r4, r6
|
||||||
|
0x38578CC4, // sc.rel.d r4, r6
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
// The toolchain accepts the zero-offset memory form alone.
|
||||||
|
for i, src := range []string{
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
LLACQW 4(R5), R4
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
SCRELV R4, 8(R6)
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
} {
|
||||||
|
fn := firstTextLOONG64(t, src)
|
||||||
|
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||||
|
t.Errorf("case %d: expected an error, got none", i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_vmovqSuffixed pins the element-broadcast and element-move
|
||||||
|
// VMOVQ/XVMOVQ forms, with the oracle words lifted verbatim from the
|
||||||
|
// toolchain's loong64enc1.s.
|
||||||
|
func TestLOONG64_vmovqSuffixed(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·vmovq(SB), NOSPLIT, $0
|
||||||
|
VMOVQ V1.B[3], V9.B16
|
||||||
|
VMOVQ V2.H[2], V8.H8
|
||||||
|
VMOVQ V3.W[1], V7.W4
|
||||||
|
VMOVQ V4.V[0], V6.V2
|
||||||
|
XVMOVQ X0, X31.B32
|
||||||
|
XVMOVQ X1, X30.H16
|
||||||
|
XVMOVQ X2, X29.W8
|
||||||
|
XVMOVQ X3, X28.V4
|
||||||
|
XVMOVQ X3, X27.Q2
|
||||||
|
XVMOVQ X0, X31.W[7]
|
||||||
|
XVMOVQ X1, X29.W[0]
|
||||||
|
XVMOVQ X3, X28.V[3]
|
||||||
|
XVMOVQ X4, X27.V[0]
|
||||||
|
XVMOVQ X31.W[7], X0
|
||||||
|
XVMOVQ X29.W[0], X1
|
||||||
|
XVMOVQ X28.V[3], X8
|
||||||
|
XVMOVQ X27.V[0], X9
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x72F78C29, // vreplvei.b v9, v1, 3
|
||||||
|
0x72F7C848, // vreplvei.h v8, v2, 2
|
||||||
|
0x72F7E467, // vreplvei.w v7, v3, 1
|
||||||
|
0x72F7F086, // vreplvei.d v6, v4, 0
|
||||||
|
0x7707001F, // xvreplve0.b x31, x0
|
||||||
|
0x7707803E, // xvreplve0.h x30, x1
|
||||||
|
0x7707C05D, // xvreplve0.w x29, x2
|
||||||
|
0x7707E07C, // xvreplve0.d x28, x3
|
||||||
|
0x7707F07B, // xvreplve0.q x27, x3
|
||||||
|
0x76FFDC1F, // xvinsve0.w x31, x0, 7
|
||||||
|
0x76FFC03D, // xvinsve0.w x29, x1, 0
|
||||||
|
0x76FFEC7C, // xvinsve0.d x28, x3, 3
|
||||||
|
0x76FFE09B, // xvinsve0.d x27, x4, 0
|
||||||
|
0x7703DFE0, // xvpickve.w x0, x31, 7
|
||||||
|
0x7703C3A1, // xvpickve.w x1, x29, 0
|
||||||
|
0x7703EF88, // xvpickve.d x8, x28, 3
|
||||||
|
0x7703E369, // xvpickve.d x9, x27, 0
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
// The rejected shapes: a width mismatch between the element and the
|
||||||
|
// arrangement, an element index past the lane count, a wrong-bank
|
||||||
|
// vreplvei and an arrangement the LASX bank does not spell.
|
||||||
|
for i, src := range []string{
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VMOVQ V1.H[3], V9.B16
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VMOVQ V1.B[16], V9.B16
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
XVMOVQ X1.B[3], X9.B32
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
XVMOVQ X0, X31.B16
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
XVMOVQ X0, X31.W[8]
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
} {
|
||||||
|
fn := firstTextLOONG64(t, src)
|
||||||
|
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||||
|
t.Errorf("case %d: expected an error, got none", i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_parityFixes pins the operand forms whose encodings were found
|
||||||
|
// diverging from the toolchain by the loong64enc1.s differential: the
|
||||||
|
// $off(reg) address immediate (addi.d), the SCQ operand order, the scaled
|
||||||
|
// vldrepl offsets (with their field masks), the XVSEQB/XVSEQV immediate
|
||||||
|
// opcodes and the zero-register base the toolchain gives FP-relative
|
||||||
|
// VMOVQ/XVMOVQ memory operands. Golden words from loong64enc1.s.
|
||||||
|
func TestLOONG64_parityFixes(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·parity(SB), NOSPLIT, $0-32
|
||||||
|
MOVW $4(R4), R5
|
||||||
|
MOVV $4(R4), R5
|
||||||
|
MOVW $65536(R4), R5
|
||||||
|
MOVW $-4096(R4), R5
|
||||||
|
SCQ R4, R5, (R6)
|
||||||
|
VMOVQ 2(R4), V1.H8
|
||||||
|
VMOVQ -6(R4), V1.H8
|
||||||
|
VMOVQ -12(R4), V2.W4
|
||||||
|
VMOVQ -16(R4), V3.V2
|
||||||
|
XVMOVQ -10(R4), X1.H16
|
||||||
|
XVSEQB $0, X2, X4
|
||||||
|
XVSEQH $3, X2, X4
|
||||||
|
XVSEQW $12, X2, X4
|
||||||
|
XVSEQV $15, X2, X4
|
||||||
|
XVSEQV $-15, X2, X4
|
||||||
|
VMOVQ V2, y+16(FP)
|
||||||
|
VMOVQ y+16(FP), V2
|
||||||
|
VMOVQ V2, x+2030(FP)
|
||||||
|
XVMOVQ X6, y+16(FP)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x02C01085, // addi.d $4, r4, r5
|
||||||
|
0x02C01085, // addi.d $4, r4, r5 (MOVW keeps the 64-bit addi.d)
|
||||||
|
0x1400021E, // lu12i.w $16, r30
|
||||||
|
0x038003DE, // ori $0, r30, r30
|
||||||
|
0x0010F885, // add.d r5, r4, r30
|
||||||
|
0x15FFFFFE, // lu12i.w $-1, r30
|
||||||
|
0x038003DE, // ori $0, r30, r30
|
||||||
|
0x0010F885, // add.d r5, r4, r30
|
||||||
|
0x385714C4, // sc.q r4, r5, (r6): middle<<10 | base<<5 | first
|
||||||
|
0x30400481, // vldrepl.h v1, 2(r4)
|
||||||
|
0x305FF481, // vldrepl.h v1, -6(r4)
|
||||||
|
0x302FF482, // vldrepl.w v2, -12(r4)
|
||||||
|
0x3017F883, // vldrepl.d v3, -16(r4)
|
||||||
|
0x325FEC81, // xvldrepl.h x1, -10(r4)
|
||||||
|
0x76800044, // xvseqi.b x4, x2, 0
|
||||||
|
0x76808C44, // xvseqi.h x4, x2, 3
|
||||||
|
0x76813044, // xvseqi.w x4, x2, 12
|
||||||
|
0x7681BC44, // xvseqi.d x4, x2, 15
|
||||||
|
0x7681C444, // xvseqi.d x4, x2, -15
|
||||||
|
0x2C406002, // vst v2, 24(r0): FP-relative keeps the zero base
|
||||||
|
0x2C006002, // vld v2, 24(r0)
|
||||||
|
0x2C5FD802, // vst v2, 2038(r0)
|
||||||
|
0x2CC06006, // xvst x6, 24(r0)
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
// Misaligned vldrepl offsets are rejected, as the toolchain does.
|
||||||
|
for i, src := range []string{
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VMOVQ 3(R4), V1.H8
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
MOVW $4(R4), F1
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
} {
|
||||||
|
fn := firstTextLOONG64(t, src)
|
||||||
|
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||||
|
t.Errorf("case %d: expected an error, got none", i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_bytePseudo pins the BYTE literal-data pseudo-op, which the
|
||||||
|
// loong64 toolchain does not spell but the arm64 and riscv64 encoders of
|
||||||
|
// this package already accept for byte-exact data layout (a superset
|
||||||
|
// spelling, shippable via the goobj path).
|
||||||
|
func TestLOONG64_bytePseudo(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·bytes(SB), NOSPLIT, $0
|
||||||
|
BYTE $2
|
||||||
|
BYTE $1; BYTE $0
|
||||||
|
BYTE $255
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
// Four literal bytes, then RET (jirl r0, r1, 0); the trailing bytes pad
|
||||||
|
// the final word the way any sub-word tail does.
|
||||||
|
want := []byte{2, 1, 0, 0xFF, 0x20, 0x00, 0x00, 0x4C}
|
||||||
|
if !bytes.Equal(code[:len(want)], want) {
|
||||||
|
t.Errorf("bytes = % x, want % x", code, want)
|
||||||
|
}
|
||||||
|
for _, src := range []string{
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
BYTE $256
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
BYTE $-1
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
} {
|
||||||
|
fn := firstTextLOONG64(t, src)
|
||||||
|
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||||
|
t.Errorf("%q: expected an error, got none", src)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Loong64 frame mapping, matching the Go toolchain's loong64 backend.
|
// Loong64 frame mapping, matching the Go toolchain's loong64 backend.
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ import (
|
|||||||
"bytes"
|
"bytes"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestLOONG64_sys exercises the no-operand system instructions and the
|
// TestLOONG64_sys exercises the no-operand system instructions and the
|
||||||
@@ -232,7 +232,10 @@ DATA ·table+0(SB)/8, $42
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
|
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
|
||||||
// invalid register operands and operand-count mismatches.
|
// invalid register operands and operand-count mismatches. The X0 and
|
||||||
|
// AMADDW cases follow the oracle: GOARCH=loong64 go tool asm rejects
|
||||||
|
// `BEQZ X0` (the X bank is not an integer register) and the two-register
|
||||||
|
// `AMADDW R4, R5` (the AM* family is strictly `val, (addr), result`).
|
||||||
func TestLOONG64_errors(t *testing.T) {
|
func TestLOONG64_errors(t *testing.T) {
|
||||||
cases := []string{
|
cases := []string{
|
||||||
`TEXT ·e(SB), NOSPLIT, $0
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
@@ -280,7 +283,7 @@ done:
|
|||||||
// TestLOONG64_pcsp checks the stack-adjustment table of a framed function:
|
// TestLOONG64_pcsp checks the stack-adjustment table of a framed function:
|
||||||
// the prologue raises the SP delta by autosize (in effect from the third
|
// the prologue raises the SP delta by autosize (in effect from the third
|
||||||
// instruction) and the RET's epilogue restores it to zero, with the pc deltas
|
// instruction) and the RET's epilogue restores it to zero, with the pc deltas
|
||||||
// in MinLC (4) units — byte-identical to `go tool asm`.
|
// in MinLC (4) units; byte-identical to `go tool asm`.
|
||||||
func TestLOONG64_pcsp(t *testing.T) {
|
func TestLOONG64_pcsp(t *testing.T) {
|
||||||
cases := []struct {
|
cases := []struct {
|
||||||
name string
|
name string
|
||||||
@@ -347,21 +350,94 @@ TEXT ·sb(SB), NOSPLIT, $0
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestLOONG64_movImmToFp checks the immediate-to-FP move forms.
|
// TestLOONG64_movImmToFp checks the immediate-to-FP move: MOVW $c, Fd is the
|
||||||
|
// only spelling the toolchain accepts, expanding to ori (or addi.w for the
|
||||||
|
// negative span) into R30 plus movgr2fr.w. The pinned words are the
|
||||||
|
// toolchain's own bytes; the other widths and out-of-range constants are
|
||||||
|
// illegal combinations there and are diagnosed here.
|
||||||
func TestLOONG64_movImmToFp(t *testing.T) {
|
func TestLOONG64_movImmToFp(t *testing.T) {
|
||||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
TEXT ·fpmov(SB), NOSPLIT, $0
|
TEXT ·fpmov(SB), NOSPLIT, $0
|
||||||
MOVV $0x1, F0
|
MOVW $0x1, F0
|
||||||
MOVW $0x2, F4
|
MOVW $0x2, F4
|
||||||
|
MOVW $-1, F4
|
||||||
RET
|
RET
|
||||||
`)
|
`)
|
||||||
code := assembleLOONG64Helper(t, fn)
|
code := assembleLOONG64Helper(t, fn)
|
||||||
want := []byte{
|
want := []byte{
|
||||||
0x00, 0x04, 0x80, 0x03, // ori f0, r0, 1
|
0x1e, 0x04, 0x80, 0x03, // ori r30, r0, 1
|
||||||
0x04, 0x08, 0x80, 0x03, // ori f4, r0, 2
|
0xc0, 0xa7, 0x14, 0x01, // movgr2fr.w f0, r30
|
||||||
|
0x1e, 0x08, 0x80, 0x03, // ori r30, r0, 2
|
||||||
|
0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30
|
||||||
|
0x1e, 0xfc, 0xbf, 0x02, // addi.w r30, r0, -1
|
||||||
|
0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30
|
||||||
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
|
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
|
||||||
}
|
}
|
||||||
if !bytes.Equal(code, want) {
|
if !bytes.Equal(code, want) {
|
||||||
t.Errorf("code = % x\nwant % x", code, want)
|
t.Errorf("code = % x\nwant % x", code, want)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_movImmToFpErrors checks the immediate-to-FP diagnostics: the
|
||||||
|
// widths the toolchain rejects as illegal combinations, and constants beyond
|
||||||
|
// the 12-bit ori/addi.w span (the toolchain never materialises a wider
|
||||||
|
// constant on this path).
|
||||||
|
func TestLOONG64_movImmToFpErrors(t *testing.T) {
|
||||||
|
cases := []string{
|
||||||
|
"MOVV $1, F0",
|
||||||
|
"MOVF $2, F4",
|
||||||
|
"MOVD $2, F4",
|
||||||
|
"MOVW $100000, F1",
|
||||||
|
"MOVW $-2049, F1",
|
||||||
|
"MOVW $4096, F1",
|
||||||
|
}
|
||||||
|
for _, src := range cases {
|
||||||
|
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
|
||||||
|
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", src)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_branch16Unsigned pins the unsigned two-operand branches: with
|
||||||
|
// one register BLTU/BGEU keep the register-register form against R0 (never
|
||||||
|
// taken), the toolchain's encoding, where a beqz would test the wrong
|
||||||
|
// condition; the three-operand forms are unchanged.
|
||||||
|
func TestLOONG64_branch16Unsigned(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·u(SB), NOSPLIT, $0
|
||||||
|
BLTU R4, done
|
||||||
|
BGEU R5, done
|
||||||
|
BLTU R6, R7, done
|
||||||
|
BGEU R8, R9, done
|
||||||
|
done:
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x68001080, // bltu r4, r0, +4
|
||||||
|
0x6C000CA0, // bgeu r5, r0, +3
|
||||||
|
0x680008C7, // bltu r6, r7, +2
|
||||||
|
0x6C000509, // bgeu r8, r9, +1
|
||||||
|
0x4C000020, // jirl r0, r1, 0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_bitFieldRange checks the BSTRINS/BSTRPICK bit-number
|
||||||
|
// validation, mirroring the toolchain's "illegal bit number" rule: 0..31 for
|
||||||
|
// the .w forms, 0..63 for the .d forms, and lsb <= msb.
|
||||||
|
func TestLOONG64_bitFieldRange(t *testing.T) {
|
||||||
|
cases := []string{
|
||||||
|
"BSTRINSW $32, R4, $0, R5",
|
||||||
|
"BSTRPICKW $31, R4, $32, R5",
|
||||||
|
"BSTRINSV $64, R4, $0, R5",
|
||||||
|
"BSTRPICKV $3, R4, $4, R5",
|
||||||
|
"BSTRINSW $-1, R4, $0, R5",
|
||||||
|
}
|
||||||
|
for _, src := range cases {
|
||||||
|
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
|
||||||
|
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", src)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestLOONG64RelocOffsetsIncludePrologue pins the function-relative
|
// TestLOONG64RelocOffsetsIncludePrologue pins the function-relative
|
||||||
|
|||||||
@@ -14,6 +14,51 @@ type Imm int64
|
|||||||
|
|
||||||
func (Imm) isOperand() {}
|
func (Imm) isOperand() {}
|
||||||
|
|
||||||
|
// RegList is a bracketed register range, [Z0-Z3]: the four-register source
|
||||||
|
// of the 4FMAPS and 4VNNIW families. The EVEX emit path carries the list's
|
||||||
|
// low register through the inverted 5-bit V'VVVV field; the three higher
|
||||||
|
// registers are implied by the instruction, so only the pair travels here.
|
||||||
|
type RegList struct {
|
||||||
|
Lo Reg
|
||||||
|
Hi Reg // implied by the encoding; Lo.idx+3 by construction
|
||||||
|
}
|
||||||
|
|
||||||
|
func (RegList) isOperand() {}
|
||||||
|
|
||||||
|
// FloatImm is a floating-point immediate ($-1.0). The SSE mnemonics whose
|
||||||
|
// encoding takes an XMM/memory source at that position rewrite it as a read
|
||||||
|
// from a read-only pool constant ($f64.<hex> or $f32.<hex>), the toolchain's
|
||||||
|
// own behaviour; every other instruction rejects it.
|
||||||
|
type FloatImm struct {
|
||||||
|
Text string // the numeric text as written, sign excluded
|
||||||
|
Neg bool // a leading minus
|
||||||
|
}
|
||||||
|
|
||||||
|
func (FloatImm) isOperand() {}
|
||||||
|
|
||||||
|
// TLSMem is a thread-local access, the source form off(base)(TLS*1) with the
|
||||||
|
// base dropped: the toolchain's one-instruction TLS rewrite assembles it as
|
||||||
|
// the segment-prefixed absolute whose disp32 carries an R_TLS_LE patch site
|
||||||
|
// (the linker fills the TLS slot offset).
|
||||||
|
type TLSMem struct {
|
||||||
|
Disp int64
|
||||||
|
Size int
|
||||||
|
Seg byte // the segment override: FS (0x64) or GS (0x65) on windows
|
||||||
|
}
|
||||||
|
|
||||||
|
func (TLSMem) isOperand() {}
|
||||||
|
|
||||||
|
// SegAbs is a segment-absolute access, 0x30(GS): the segment override
|
||||||
|
// prefixes a disp32 absolute reference with no relocation. The base
|
||||||
|
// register spellings GS and FS produce it.
|
||||||
|
type SegAbs struct {
|
||||||
|
Disp int64
|
||||||
|
Size int
|
||||||
|
Seg byte // 0x64 FS, 0x65 GS
|
||||||
|
}
|
||||||
|
|
||||||
|
func (SegAbs) isOperand() {}
|
||||||
|
|
||||||
// Mem is a memory operand of the form disp(base)(index*scale).
|
// Mem is a memory operand of the form disp(base)(index*scale).
|
||||||
type Mem struct {
|
type Mem struct {
|
||||||
Base Reg
|
Base Reg
|
||||||
@@ -23,6 +68,7 @@ type Mem struct {
|
|||||||
Size int // operand width in bytes
|
Size int // operand width in bytes
|
||||||
HasBase bool
|
HasBase bool
|
||||||
HasIndex bool
|
HasIndex bool
|
||||||
|
Seg byte // segment override prefix (0x64 FS, 0x65 GS); 0 = none
|
||||||
}
|
}
|
||||||
|
|
||||||
func (Mem) isOperand() {}
|
func (Mem) isOperand() {}
|
||||||
@@ -48,3 +94,16 @@ type sbMem struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (sbMem) isOperand() {}
|
func (sbMem) isOperand() {}
|
||||||
|
|
||||||
|
// isX86Mem reports whether the operand is an amd64 memory reference: a base
|
||||||
|
// or indexed Mem, or an SB-relative sbMem. Encoders that gate on "memory in
|
||||||
|
// this position" must accept both; the r/m emitters distinguish the two
|
||||||
|
// themselves.
|
||||||
|
func isX86Mem(o Operand) bool {
|
||||||
|
switch o.(type) {
|
||||||
|
case Mem, sbMem:
|
||||||
|
return true
|
||||||
|
default:
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+35
-1
@@ -17,12 +17,28 @@ import "strings"
|
|||||||
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
|
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
|
||||||
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
|
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
|
||||||
// those indices but require one. The mask flag marks the AVX-512 opmask
|
// those indices but require one. The mask flag marks the AVX-512 opmask
|
||||||
// registers K0-K7.
|
// registers K0-K7, the fp flag the x87 stack registers F0-F7, the mmx flag the
|
||||||
|
// MMX registers M0-M7, the seg field a bare segment register (FS, GS) and the
|
||||||
|
// ctl field the control and debug registers, whose number rides an
|
||||||
|
// instruction's reg field rather than r/m.
|
||||||
type Reg struct {
|
type Reg struct {
|
||||||
idx int
|
idx int
|
||||||
size int // informational width implied by the name; the mnemonic decides
|
size int // informational width implied by the name; the mnemonic decides
|
||||||
high bool // AH/CH/DH/BH
|
high bool // AH/CH/DH/BH
|
||||||
mask bool // K0-K7 opmask register
|
mask bool // K0-K7 opmask register
|
||||||
|
fp bool // F0-F7 x87 stack register
|
||||||
|
mmx bool // M0-M7 MMX register
|
||||||
|
seg int // segment register number plus one (ES=1..GS=6); 0 = not one
|
||||||
|
ctl byte // 0 none, 1 CRn control register, 2 DRn debug register
|
||||||
|
}
|
||||||
|
|
||||||
|
// segNumber returns the segment register number (ES=0..GS=5) when r names a
|
||||||
|
// bare segment register.
|
||||||
|
func (r Reg) segNumber() (int, bool) {
|
||||||
|
if r.seg == 0 {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
return r.seg - 1, true
|
||||||
}
|
}
|
||||||
|
|
||||||
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
|
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
|
||||||
@@ -144,6 +160,24 @@ func buildRegByName() map[string]Reg {
|
|||||||
for i := 0; i <= 7; i++ {
|
for i := 0; i <= 7; i++ {
|
||||||
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
|
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
|
||||||
}
|
}
|
||||||
|
// x87 stack: F0..F7.
|
||||||
|
for i := 0; i <= 7; i++ {
|
||||||
|
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
|
||||||
|
}
|
||||||
|
// MMX: M0..M7.
|
||||||
|
for i := 0; i <= 7; i++ {
|
||||||
|
m["M"+itoa(i)] = Reg{idx: i, size: 8, mmx: true}
|
||||||
|
}
|
||||||
|
// Bare segment registers: ES, CS, SS, DS, FS, GS (the memory-base and
|
||||||
|
// index spellings of FS and GS are handled before register lookup).
|
||||||
|
for i, n := range []string{"ES", "CS", "SS", "DS", "FS", "GS"} {
|
||||||
|
m[n] = Reg{idx: i, size: 2, seg: i + 1}
|
||||||
|
}
|
||||||
|
// Control and debug registers: CR0..CR15, DR0..DR15.
|
||||||
|
for i := 0; i <= 15; i++ {
|
||||||
|
m["CR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 1}
|
||||||
|
m["DR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 2}
|
||||||
|
}
|
||||||
return m
|
return m
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+1692
-73
File diff suppressed because it is too large
Load Diff
+160
-36
@@ -63,9 +63,9 @@ func riscvRegNum(name string) int {
|
|||||||
return 24
|
return 24
|
||||||
case "X25", "S9":
|
case "X25", "S9":
|
||||||
return 25
|
return 25
|
||||||
case "X26", "S10":
|
case "X26", "S10", "CTXT":
|
||||||
return 26
|
return 26
|
||||||
case "X27", "S11":
|
case "X27", "S11", "g":
|
||||||
return 27
|
return 27
|
||||||
case "X28", "T3":
|
case "X28", "T3":
|
||||||
return 28
|
return 28
|
||||||
@@ -141,10 +141,36 @@ func riscvRegNum(name string) int {
|
|||||||
case "F31", "FT11":
|
case "F31", "FT11":
|
||||||
return 31
|
return 31
|
||||||
default:
|
default:
|
||||||
|
// Vector registers V0-V31 (the "V" extension). They share the
|
||||||
|
// register numbering with the integer file: a bare number 0-31.
|
||||||
|
if len(name) >= 2 && name[0] == 'V' {
|
||||||
|
if n, ok := parseRegDigits(name[1:], 31); ok {
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
}
|
||||||
return -1
|
return -1
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// parseRegDigits parses a decimal register suffix and reports whether it is
|
||||||
|
// within [0, max].
|
||||||
|
func parseRegDigits(digits string, max int) (int, bool) {
|
||||||
|
if digits == "" {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
n := 0
|
||||||
|
for i := 0; i < len(digits); i++ {
|
||||||
|
if digits[i] < '0' || digits[i] > '9' {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
n = n*10 + int(digits[i]-'0')
|
||||||
|
if n > max {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n, true
|
||||||
|
}
|
||||||
|
|
||||||
// RISC-V instruction encoding parameters.
|
// RISC-V instruction encoding parameters.
|
||||||
type riscvEnc struct {
|
type riscvEnc struct {
|
||||||
opcode uint32 // bits [6:0]
|
opcode uint32 // bits [6:0]
|
||||||
@@ -193,6 +219,9 @@ var riscvInstrTable = map[string]riscvEnc{
|
|||||||
"DIVUW": {0x3B, 0x5, 0x01},
|
"DIVUW": {0x3B, 0x5, 0x01},
|
||||||
"REMW": {0x3B, 0x6, 0x01},
|
"REMW": {0x3B, 0x6, 0x01},
|
||||||
"REMUW": {0x3B, 0x7, 0x01},
|
"REMUW": {0x3B, 0x7, 0x01},
|
||||||
|
// Zicond conditional zeroing.
|
||||||
|
"CZEROEQZ": {0x33, 0x5, 0x07},
|
||||||
|
"CZERONEZ": {0x33, 0x7, 0x07},
|
||||||
// RV64I, I-type arithmetic.
|
// RV64I, I-type arithmetic.
|
||||||
"ADDI": {0x13, 0x0, 0x00},
|
"ADDI": {0x13, 0x0, 0x00},
|
||||||
"ADDIW": {0x1B, 0x0, 0x00},
|
"ADDIW": {0x1B, 0x0, 0x00},
|
||||||
@@ -221,36 +250,47 @@ var riscvInstrTable = map[string]riscvEnc{
|
|||||||
"BGE": {0x63, 0x5, 0x00},
|
"BGE": {0x63, 0x5, 0x00},
|
||||||
"BLTU": {0x63, 0x6, 0x00},
|
"BLTU": {0x63, 0x6, 0x00},
|
||||||
"BGEU": {0x63, 0x7, 0x00},
|
"BGEU": {0x63, 0x7, 0x00},
|
||||||
|
// The swapped-spelling comparison forms: encoded as BLT/BGE/BLTU/BGEU
|
||||||
|
// with the register operands swapped.
|
||||||
|
"BGT": {0x63, 0x4, 0x00},
|
||||||
|
"BLE": {0x63, 0x5, 0x00},
|
||||||
|
"BGTU": {0x63, 0x6, 0x00},
|
||||||
|
"BLEU": {0x63, 0x7, 0x00},
|
||||||
// U-type.
|
// U-type.
|
||||||
"LUI": {0x37, 0x0, 0x00},
|
"LUI": {0x37, 0x0, 0x00},
|
||||||
"AUIPC": {0x17, 0x0, 0x00},
|
"AUIPC": {0x17, 0x0, 0x00},
|
||||||
// System.
|
// System.
|
||||||
"ECALL": {0x73, 0x0, 0x00},
|
"ECALL": {0x73, 0x0, 0x00},
|
||||||
"EBREAK": {0x73, 0x0, 0x00},
|
"EBREAK": {0x73, 0x0, 0x00},
|
||||||
"FENCE": {0x0F, 0x0, 0x00},
|
"FENCE": {0x0F, 0x0, 0x00},
|
||||||
|
"FENCE.TSO": {0x0F, 0x0, 0x00},
|
||||||
|
"PAUSE": {0x0F, 0x0, 0x00},
|
||||||
// JALR, indirect jump/call (I-type).
|
// JALR, indirect jump/call (I-type).
|
||||||
"JALR": {0x67, 0x0, 0x00},
|
"JALR": {0x67, 0x0, 0x00},
|
||||||
|
|
||||||
// RV64A, atomics (AMO opcode 0x2F).
|
// RV64A, atomics (AMO opcode 0x2F).
|
||||||
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
|
// funct3: 0x2 = word, 0x3 = doubleword. The stored funct7 is the full
|
||||||
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
|
// 7-bit field: funct5 in the upper five bits and the aq/rl ordering bits in
|
||||||
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
|
// the lower two, exactly as the toolchain writes them: every AMO sets both
|
||||||
"AMOADDW": {0x2F, 0x2, 0x00 << 2},
|
// aq and rl (funct7 |= 3).
|
||||||
"AMOADDD": {0x2F, 0x3, 0x00 << 2},
|
"AMOSWAPW": {0x2F, 0x2, 0x01<<2 | 0x3},
|
||||||
"AMOANDW": {0x2F, 0x2, 0x0C << 2},
|
"AMOSWAPD": {0x2F, 0x3, 0x01<<2 | 0x3},
|
||||||
"AMOANDD": {0x2F, 0x3, 0x0C << 2},
|
"AMOADDW": {0x2F, 0x2, 0x00<<2 | 0x3},
|
||||||
"AMOORW": {0x2F, 0x2, 0x06 << 2},
|
"AMOADDD": {0x2F, 0x3, 0x00<<2 | 0x3},
|
||||||
"AMOORD": {0x2F, 0x3, 0x06 << 2},
|
"AMOANDW": {0x2F, 0x2, 0x0C<<2 | 0x3},
|
||||||
"AMOXORW": {0x2F, 0x2, 0x04 << 2},
|
"AMOANDD": {0x2F, 0x3, 0x0C<<2 | 0x3},
|
||||||
"AMOXORD": {0x2F, 0x3, 0x04 << 2},
|
"AMOORW": {0x2F, 0x2, 0x08<<2 | 0x3},
|
||||||
"AMOMAXW": {0x2F, 0x2, 0x14 << 2},
|
"AMOORD": {0x2F, 0x3, 0x08<<2 | 0x3},
|
||||||
"AMOMAXD": {0x2F, 0x3, 0x14 << 2},
|
"AMOXORW": {0x2F, 0x2, 0x04<<2 | 0x3},
|
||||||
"AMOMINW": {0x2F, 0x2, 0x10 << 2},
|
"AMOXORD": {0x2F, 0x3, 0x04<<2 | 0x3},
|
||||||
"AMOMIND": {0x2F, 0x3, 0x10 << 2},
|
"AMOMAXW": {0x2F, 0x2, 0x14<<2 | 0x3},
|
||||||
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2},
|
"AMOMAXD": {0x2F, 0x3, 0x14<<2 | 0x3},
|
||||||
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2},
|
"AMOMINW": {0x2F, 0x2, 0x10<<2 | 0x3},
|
||||||
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
|
"AMOMIND": {0x2F, 0x3, 0x10<<2 | 0x3},
|
||||||
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
|
"AMOMAXUW": {0x2F, 0x2, 0x1C<<2 | 0x3},
|
||||||
|
"AMOMAXUD": {0x2F, 0x3, 0x1C<<2 | 0x3},
|
||||||
|
"AMOMINUW": {0x2F, 0x2, 0x18<<2 | 0x3},
|
||||||
|
"AMOMINUD": {0x2F, 0x3, 0x18<<2 | 0x3},
|
||||||
|
|
||||||
// RV64F/D, floating-point arithmetic.
|
// RV64F/D, floating-point arithmetic.
|
||||||
"FADDS": {0x53, 0x0, 0x00},
|
"FADDS": {0x53, 0x0, 0x00},
|
||||||
@@ -273,12 +313,23 @@ var riscvInstrTable = map[string]riscvEnc{
|
|||||||
"FMAXS": {0x53, 0x1, 0x14},
|
"FMAXS": {0x53, 0x1, 0x14},
|
||||||
"FMIND": {0x53, 0x0, 0x15},
|
"FMIND": {0x53, 0x0, 0x15},
|
||||||
"FMAXD": {0x53, 0x1, 0x15},
|
"FMAXD": {0x53, 0x1, 0x15},
|
||||||
|
// FP sign injection (double): rs2 carries the sign source.
|
||||||
|
"FSGNJD": {0x53, 0x0, 0x11},
|
||||||
|
"FSGNJS": {0x53, 0x0, 0x10},
|
||||||
|
"FSGNJX": {0x53, 0x0, 0x14},
|
||||||
|
"FSGNJXD": {0x53, 0x0, 0x15},
|
||||||
|
"FSGNJXS": {0x53, 0x0, 0x14},
|
||||||
|
"FSGNJND": {0x53, 0x1, 0x11},
|
||||||
|
"FSGNJNS": {0x53, 0x1, 0x10},
|
||||||
|
"FSGNJNX": {0x53, 0x1, 0x14},
|
||||||
|
|
||||||
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
|
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
|
||||||
"LRW": {0x2F, 0x2, 0x02 << 2},
|
// The toolchain gives LR acquire ordering (aq = 1) and SC release
|
||||||
"LRD": {0x2F, 0x3, 0x02 << 2},
|
// ordering (rl = 1).
|
||||||
"SCW": {0x2F, 0x2, 0x03 << 2},
|
"LRW": {0x2F, 0x2, 0x02<<2 | 0x2},
|
||||||
"SCD": {0x2F, 0x3, 0x03 << 2},
|
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
|
||||||
|
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
|
||||||
|
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
|
||||||
|
|
||||||
// FP compare, result in integer register (funct7 0x50/0x51).
|
// FP compare, result in integer register (funct7 0x50/0x51).
|
||||||
"FEQS": {0x53, 0x2, 0x50},
|
"FEQS": {0x53, 0x2, 0x50},
|
||||||
@@ -296,11 +347,11 @@ func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// riscvAMOType encodes an atomic (AMO) instruction.
|
// riscvAMOType encodes an atomic (AMO) instruction.
|
||||||
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode.
|
// Layout: funct7 | rs2 | rs1 | funct3 | rd | opcode, where funct7 carries the
|
||||||
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2).
|
// funct5 in its upper five bits and the aq/rl ordering bits in the lower two
|
||||||
|
// (the table stores the full field, so the word needs no reassembly).
|
||||||
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||||
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value
|
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||||
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
|
||||||
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -328,6 +379,8 @@ var riscvCvtTable = map[string]riscvCvtEnc{
|
|||||||
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
|
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
|
||||||
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
|
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
|
||||||
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
|
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
|
||||||
|
"FCLASSS": {0x70, 0x0, 0x53}, // classify float32 → GPR mask
|
||||||
|
"FCLASSD": {0x70, 0x0, 0x53}, // classify float64 → GPR mask
|
||||||
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
|
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
|
||||||
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
|
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
|
||||||
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
|
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
|
||||||
@@ -340,6 +393,10 @@ var riscvCvtTable = map[string]riscvCvtEnc{
|
|||||||
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
|
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
|
||||||
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
|
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
|
||||||
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
|
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
|
||||||
|
// The toolchain's W/D suffix spellings of the same moves.
|
||||||
|
"FMVXS": {0x70, 0x0, 0x53},
|
||||||
|
"FMVFS": {0x78, 0x0, 0x53},
|
||||||
|
"FMVSX": {0x79, 0x0, 0x53},
|
||||||
}
|
}
|
||||||
|
|
||||||
// riscvCvtType encodes an FP conversion instruction.
|
// riscvCvtType encodes an FP conversion instruction.
|
||||||
@@ -371,7 +428,7 @@ var riscvFmaTable = map[string]riscvFmaEnc{
|
|||||||
// riscvFmaType encodes an R4-type fused multiply-add instruction.
|
// riscvFmaType encodes an R4-type fused multiply-add instruction.
|
||||||
func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 {
|
func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 {
|
||||||
return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) |
|
return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) |
|
||||||
(uint32(rs1) << 15) | (0x0 << 12) /* rm=dynamic */ | (uint32(rd) << 7) | enc.opcode
|
(uint32(rs1) << 15) | (0x0 << 12) /* rm=RNE */ | (uint32(rd) << 7) | enc.opcode
|
||||||
}
|
}
|
||||||
|
|
||||||
// CSR (Control and Status Register) instructions.
|
// CSR (Control and Status Register) instructions.
|
||||||
@@ -439,6 +496,71 @@ func riscvJType(rd int, offset int32) uint32 {
|
|||||||
0x6F // JAL opcode
|
0x6F // JAL opcode
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- RVV ("V" extension) encoding helpers ----
|
||||||
|
|
||||||
|
// The OP-V major opcode and its funct3 subclasses.
|
||||||
|
const (
|
||||||
|
riscvOpV = 0x57 // the vector operation opcode (also OPcfg for vset*)
|
||||||
|
// funct3 values: 0 OPIVV, 1 OPFVV, 2 OPMVV, 3 OPIVI, 4 OPIVX,
|
||||||
|
// 5 OPFVF, 6 OPMVX, 7 vsetvli.
|
||||||
|
riscvVf3VV = 0x0 // vector-vector
|
||||||
|
riscvVf3MV = 0x2 // vector mask
|
||||||
|
riscvVf3VI = 0x3 // vector-immediate
|
||||||
|
riscvVf3VX = 0x4 // vector-scalar
|
||||||
|
riscvVf3Cfg = 0x7 // vsetvli
|
||||||
|
)
|
||||||
|
|
||||||
|
// riscvVType composes the vsetvli/vsetivli vtype immediate: the register
|
||||||
|
// group multiplier in [2:0], the selected element width in [5:3] and the
|
||||||
|
// tail-agnostic and mask-agnostic policies in bits 6 and 7.
|
||||||
|
func riscvVType(vsew, vlmul, vta, vma int) int {
|
||||||
|
return vlmul | vsew<<3 | vta<<6 | vma<<7
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvVSetEnc encodes VSETVLI and VSETIVLI: imm[31:20] = vtype, rs1 = the
|
||||||
|
// avl register or 5-bit uimm, rd = the destination. Both carry funct3 7; a
|
||||||
|
// vsetivli is distinguished by bits [31:30] set in the immediate (the 0xC00
|
||||||
|
// the toolchain writes above its 10-bit vtype).
|
||||||
|
func riscvVSetEnc(vsetivli bool, avl, vtype, rd int) uint32 {
|
||||||
|
imm := vtype & 0x3FF
|
||||||
|
if vsetivli {
|
||||||
|
imm |= 0xC00
|
||||||
|
}
|
||||||
|
return uint32(imm)<<20 | uint32(avl&0x1F)<<15 | uint32(riscvVf3Cfg)<<12 |
|
||||||
|
uint32(rd)<<7 | riscvOpV
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvVLSType encodes a vector load or store: the full 32-bit word with the
|
||||||
|
// segment count in bits [31:29], the addressing mode in bits [28:26], the
|
||||||
|
// unmasked bit at 25 and the width in funct3. width follows the load
|
||||||
|
// convention (0 = 8-bit, 5 = 16-bit, 6 = 32-bit, 7 = 64-bit).
|
||||||
|
func riscvVLSType(op uint32, nf, mop, width int, rs2 int32, rs1, rd int) uint32 {
|
||||||
|
return uint32(nf&0x7)<<29 | uint32(mop&0x7)<<26 | 1<<25 |
|
||||||
|
uint32(rs2)<<20 | uint32(rs1)<<15 | uint32(width&0x7)<<12 |
|
||||||
|
uint32(rd)<<7 | op
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvVVInstr encodes an OP-V instruction with the six-bit operation code in
|
||||||
|
// funct7's upper bits, bit 25 as the unmasked flag and the three registers in
|
||||||
|
// the standard positions. vs1 may name an integer register for the *VX forms
|
||||||
|
// (the scalar sits in the rs1 field) or an immediate for the *VI forms.
|
||||||
|
func riscvVVInstr(funct6, funct3 int, vs1 int32, vs2, vd int) uint32 {
|
||||||
|
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs1)<<15 |
|
||||||
|
uint32(funct3)<<12 | uint32(vs2)<<20 | uint32(vd)<<7 | riscvOpV
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvVUnaryInstr encodes a one-vector-operand OP-V instruction whose fixed
|
||||||
|
// fields live where the second source register would be: rs1Field and vs2 are
|
||||||
|
// written verbatim (the oracle writes fixed non-zero constants there for some
|
||||||
|
// instructions, such as 0x11 in the rs1 field of vmfirst.m and vid.v).
|
||||||
|
func riscvVUnaryInstr(funct6, funct3 int, rs1Field int32, vs2, vd int) uint32 {
|
||||||
|
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs2&0x1F)<<20 |
|
||||||
|
uint32(rs1Field&0x1F)<<15 | uint32(funct3&0x7)<<12 | uint32(vd&0x1F)<<7 | riscvOpV
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvSegNF maps a segment count to the 3-bit nf field (count - 1).
|
||||||
|
func riscvSegNF(n int) int32 { return int32(n - 1) }
|
||||||
|
|
||||||
// ---- RVC (compressed) encoding helpers ----
|
// ---- RVC (compressed) encoding helpers ----
|
||||||
|
|
||||||
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
|
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
|
||||||
@@ -520,11 +642,13 @@ func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 {
|
|||||||
|
|
||||||
// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
|
// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
|
||||||
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
|
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
|
||||||
// offset; the immediate bits are extracted per the RISC-V CS format.
|
// offset; the immediate bits are extracted per the RISC-V CS format, with the
|
||||||
|
// same five-bit patterns as the load side ({5,4,3,7,6} and {5,4,3,2,6},
|
||||||
|
// matching the toolchain's encodeCS).
|
||||||
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
|
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
|
||||||
pattern := []int{5, 3, 7, 6}
|
pattern := []int{5, 4, 3, 7, 6}
|
||||||
if funct3 == 0x6 {
|
if funct3 == 0x6 {
|
||||||
pattern = []int{5, 3, 2, 6}
|
pattern = []int{5, 4, 3, 2, 6}
|
||||||
}
|
}
|
||||||
packed := encodeRVCPattern(imm, pattern)
|
packed := encodeRVCPattern(imm, pattern)
|
||||||
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
|
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
|
||||||
|
|||||||
+591
-12
@@ -5,10 +5,13 @@ package asm
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
|
"encoding/binary"
|
||||||
|
"encoding/hex"
|
||||||
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// firstTextRISCV parses assembly source and returns the first TEXT function body.
|
// firstTextRISCV parses assembly source and returns the first TEXT function body.
|
||||||
@@ -30,7 +33,7 @@ func firstTextRISCV(t *testing.T, src string) *ast.Text {
|
|||||||
// assembleRISCVHelper assembles one TEXT function and returns its code bytes.
|
// assembleRISCVHelper assembles one TEXT function and returns its code bytes.
|
||||||
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
|
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
code, _, _, _, _, err := assembleRISCV(fn)
|
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("assemble: %v", err)
|
t.Fatalf("assemble: %v", err)
|
||||||
}
|
}
|
||||||
@@ -289,7 +292,7 @@ TEXT ·cmp(SB), NOSPLIT, $0
|
|||||||
}
|
}
|
||||||
|
|
||||||
func TestRISCV_forwardBranch(t *testing.T) {
|
func TestRISCV_forwardBranch(t *testing.T) {
|
||||||
// Forward label reference — must not fail.
|
// Forward label reference; must not fail.
|
||||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
TEXT ·fwd(SB), NOSPLIT, $0
|
TEXT ·fwd(SB), NOSPLIT, $0
|
||||||
ADDI $1, X10, X10
|
ADDI $1, X10, X10
|
||||||
@@ -691,6 +694,81 @@ DATA answer<>+0(SB)/8, $42
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestRISCV_RVC_StorePatterns pins the register-relative compressed store
|
||||||
|
// encodings for offsets with immediate bits 4 and 5 set, byte-identical to
|
||||||
|
// the toolchain's encodeCS (patterns {5,4,3,7,6} and {5,4,3,2,6}).
|
||||||
|
// Regression: the store-side patterns dropped imm[4], so every such store
|
||||||
|
// silently encoded the wrong address while the loads stayed correct.
|
||||||
|
func TestRISCV_RVC_StorePatterns(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·csstores(SB), NOSPLIT, $0
|
||||||
|
SD X9, 24(X8)
|
||||||
|
SW X10, 16(X11)
|
||||||
|
FSD F8, 40(X12)
|
||||||
|
LD 24(X8), X9
|
||||||
|
LW 16(X11), X10
|
||||||
|
FLD 40(X12), F8
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
want := []byte{
|
||||||
|
0x04, 0xec, // c.sd x9, 24(x8)
|
||||||
|
0x88, 0xc9, // c.sw x10, 16(x11)
|
||||||
|
0x00, 0xb6, // c.fsd f8, 40(x12)
|
||||||
|
0x04, 0x6c, // c.ld x9, 24(x8)
|
||||||
|
0x88, 0x49, // c.lw x10, 16(x11)
|
||||||
|
0x00, 0x36, // c.fld f8, 40(x12)
|
||||||
|
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
|
||||||
|
}
|
||||||
|
if !bytes.Equal(code, want) {
|
||||||
|
t.Errorf("code = % x\nwant % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_FENCE pins the FENCE encoding: the toolchain expands the bare
|
||||||
|
// mnemonic to fence iorw, iorw (0x0FF0000F), not fence 0,0.
|
||||||
|
func TestRISCV_FENCE(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·fence(SB), NOSPLIT, $0
|
||||||
|
FENCE
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
want := []byte{
|
||||||
|
0x0f, 0x00, 0xf0, 0x0f, // fence iorw, iorw
|
||||||
|
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
|
||||||
|
}
|
||||||
|
if !bytes.Equal(code, want) {
|
||||||
|
t.Errorf("code = % x\nwant % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_RVC_WidthSpellings pins the compression of the GOROOT width
|
||||||
|
// spellings: MOVW and MOVD lower to their base load/store and compress
|
||||||
|
// exactly like LW/SW/FLD/FSD would (the toolchain compresses these shapes;
|
||||||
|
// before the normalisation they stayed 4 bytes).
|
||||||
|
func TestRISCV_RVC_WidthSpellings(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·widths(SB), NOSPLIT, $0-16
|
||||||
|
MOVW w+0(FP), X9
|
||||||
|
MOVW X9, v+4(FP)
|
||||||
|
MOVD d+0(FP), F8
|
||||||
|
MOVD F8, r+8(FP)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
want := []byte{
|
||||||
|
0xa2, 0x44, // c.lwsp x9, 8
|
||||||
|
0x26, 0xc6, // c.swsp x9, 12
|
||||||
|
0x22, 0x24, // c.fldsp f8, 8
|
||||||
|
0x22, 0xa8, // c.fsdsp f8, 16
|
||||||
|
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
|
||||||
|
}
|
||||||
|
if !bytes.Equal(code, want) {
|
||||||
|
t.Errorf("code = % x\nwant % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestRISCV_system_instrs(t *testing.T) {
|
func TestRISCV_system_instrs(t *testing.T) {
|
||||||
// Test FENCE, ECALL, EBREAK encoding.
|
// Test FENCE, ECALL, EBREAK encoding.
|
||||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
@@ -707,16 +785,140 @@ TEXT ·sys(SB), NOSPLIT, $0
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRISCV_MOV_sym_FP_error(t *testing.T) {
|
func TestRISCV_MOV_sym_FP(t *testing.T) {
|
||||||
// MOV $sym(FP), rd should return an error (unsupported).
|
// MOV $sym(FP), rd lowers to the frame-adjusted ADDI against SP: the
|
||||||
|
// toolchain's argframe spelling. A zero frame leaves the offset at the
|
||||||
|
// 8-byte link slot, compressed to C.ADDI4SPN.
|
||||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
TEXT ·badfp(SB), NOSPLIT, $0
|
TEXT ·argfp(SB), NOSPLIT, $0
|
||||||
MOV $arg(FP), X10
|
MOV $arg(FP), X10
|
||||||
RET
|
RET
|
||||||
`)
|
`)
|
||||||
_, _, _, _, _, err := assembleRISCV(fn)
|
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
if err == nil {
|
if err != nil {
|
||||||
t.Error("expected error for MOV $arg(FP), got nil")
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
// prologue (0: leaf, zero frame) + C.ADDI4SPN (2) + RET (4) = 6
|
||||||
|
want := []byte{0x28, 0x00, 0x67, 0x80, 0x00, 0x00}
|
||||||
|
if string(code) != string(want) {
|
||||||
|
t.Errorf("got % x, want % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_Bookkeeping(t *testing.T) {
|
||||||
|
// FUNCDATA and PCDATA contribute no bytes; UNDEF is the toolchain's
|
||||||
|
// ebreak, compressed to C.EBREAK under RVC.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·book(SB), NOSPLIT, $0-8
|
||||||
|
FUNCDATA $0, marks<>(SB)
|
||||||
|
PCDATA $1, $1
|
||||||
|
UNDEF
|
||||||
|
MOV $1, X10
|
||||||
|
MOV X10, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
// C.EBREAK (2) + C.LI X10, 1 (2) + C.SWSP (2) + RET (4) = 10: the
|
||||||
|
// FUNCDATA and PCDATA statements contribute nothing.
|
||||||
|
want := []byte{0x02, 0x90, 0x05, 0x45, 0x2a, 0xe4, 0x67, 0x80, 0x00, 0x00}
|
||||||
|
if string(code) != string(want) {
|
||||||
|
t.Errorf("got % x, want % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_JMPPCRel(t *testing.T) {
|
||||||
|
// JMP N(PC): the displacement tracks the instruction N source slots
|
||||||
|
// away in the final layout (0 the jump itself, negative backwards).
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·slots(SB), NOSPLIT, $0-0
|
||||||
|
JMP 2(PC)
|
||||||
|
MOV $1, X11
|
||||||
|
MOV $2, X12
|
||||||
|
MOV X12, X11
|
||||||
|
JMP -3(PC)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
// JMP 2(PC) lands on the C.MV six bytes ahead; JMP -3(PC) lands back on
|
||||||
|
// the first C.LI, six bytes behind.
|
||||||
|
want := []byte{
|
||||||
|
0x6f, 0x00, 0x60, 0x00, // JAL X0, 6
|
||||||
|
0x85, 0x45, // C.LI X11, 1
|
||||||
|
0x09, 0x46, // C.LI X12, 2
|
||||||
|
0xb2, 0x85, // C.MV X11, X12
|
||||||
|
0x6f, 0xf0, 0xbf, 0xff, // JAL X0, -6
|
||||||
|
0x67, 0x80, 0x00, 0x00, // RET
|
||||||
|
}
|
||||||
|
if string(code) != string(want) {
|
||||||
|
t.Errorf("got % x, want % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_MOVWideImm(t *testing.T) {
|
||||||
|
// Shift-sequence constants compress like the toolchain's expansion.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·wide(SB), NOSPLIT, $0-0
|
||||||
|
MOV $0x8000000000000000, X5
|
||||||
|
MOV $0x100000000, X5
|
||||||
|
MOV $0x000fffffffffffda, X5
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
// C.LI -1, C.SLLI 63; C.LI 1, C.SLLI 32; C.LI -19, C.SLLI 13, SRLI 12.
|
||||||
|
want := []byte{
|
||||||
|
0xfd, 0x52, 0xfe, 0x12,
|
||||||
|
0x85, 0x42, 0x82, 0x12,
|
||||||
|
0xb5, 0x52, 0xb6, 0x02, 0x93, 0xd2, 0xc2, 0x00,
|
||||||
|
0x67, 0x80, 0x00, 0x00,
|
||||||
|
}
|
||||||
|
if string(code) != string(want) {
|
||||||
|
t.Errorf("got % x, want % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_MOVImmPool(t *testing.T) {
|
||||||
|
// A constant outside the shift shapes loads from the pooled $i64 data
|
||||||
|
// symbol via AUIPC+LD, named like the toolchain's pool.
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·pool(SB), NOSPLIT, $0-8
|
||||||
|
MOV $0x0101010101010101, X16
|
||||||
|
MOV X16, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("pool_riscv64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
// AUIPC X16, 0 + LD X16, 0(X16): the relocation pair carries the symbol.
|
||||||
|
wantCode := []byte{0x17, 0x08, 0x00, 0x00, 0x03, 0x38, 0x08, 0x00}
|
||||||
|
if string(img.Code[0:8]) != string(wantCode) {
|
||||||
|
t.Errorf("pool load: got % x", img.Code[0:8])
|
||||||
|
}
|
||||||
|
var lit *DataSymbol
|
||||||
|
for i := range img.DataSyms {
|
||||||
|
if img.DataSyms[i].Name == "$i64.0101010101010101" {
|
||||||
|
lit = &img.DataSyms[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if lit == nil {
|
||||||
|
t.Fatalf("pool symbol missing: %v", img.DataSyms)
|
||||||
|
}
|
||||||
|
wantData := []byte{0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01}
|
||||||
|
if string(img.Data[lit.Offset:lit.Offset+8]) != string(wantData) {
|
||||||
|
t.Errorf("pool bytes: got % x", img.Data[lit.Offset:lit.Offset+8])
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -727,7 +929,7 @@ TEXT ·calltest(SB), NOSPLIT, $0
|
|||||||
CALL ext(SB)
|
CALL ext(SB)
|
||||||
RET
|
RET
|
||||||
`)
|
`)
|
||||||
code, _, relocs, _, _, err := assembleRISCV(fn)
|
code, _, relocs, _, _, _, err := assembleRISCV(fn)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("assemble: %v", err)
|
t.Fatalf("assemble: %v", err)
|
||||||
}
|
}
|
||||||
@@ -756,8 +958,385 @@ TEXT ·calllocal(SB), NOSPLIT, $0
|
|||||||
sub:
|
sub:
|
||||||
RET
|
RET
|
||||||
`)
|
`)
|
||||||
_, _, _, _, _, err := assembleRISCV(fn)
|
_, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
if err == nil {
|
if err == nil {
|
||||||
t.Error("expected error for CALL to local label, got nil")
|
t.Error("expected error for CALL to local label, got nil")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestRISCVIndirectBranch pins the indirect branch encodings: JMP (X5) is the
|
||||||
|
// toolchain's JALR X0, 0(X5), and the trampoline form JALR rd, offset(rs1)
|
||||||
|
// takes its destination from the first operand (regression: the base
|
||||||
|
// register was once read as the destination, silently jumping to X0).
|
||||||
|
func TestRISCVIndirectBranch(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·f(SB), NOSPLIT, $0-0
|
||||||
|
JMP (X5)
|
||||||
|
JALR X0, 0(X6)
|
||||||
|
JALR X28, 0(X9)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x00028067, // jalr x0, 5(x0), 0
|
||||||
|
0x00030067, // jalr x0, 6(x0), 0
|
||||||
|
0x00048e67, // jalr x28, 9(x0), 0
|
||||||
|
0x00008067, // jalr x0, 1(x0), 0 (RET)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeOneInstrRISCV encodes a single parsed instruction against a synthetic
|
||||||
|
// offsets map, the smallest honest harness for the branch-range diagnostics:
|
||||||
|
// the spans are far larger than any source a test would want to spell out.
|
||||||
|
func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]int) ([]byte, error) {
|
||||||
|
t.Helper()
|
||||||
|
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
|
||||||
|
instr := fn.Body[0].(*ast.Instr)
|
||||||
|
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil, nil, nil)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span
|
||||||
|
// [-4096, 4094] and the J-type span [-1048576, 1048574] are diagnosed instead
|
||||||
|
// of wrapping silently to a wrong target.
|
||||||
|
func TestRISCVBranchJumpRange(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
src string
|
||||||
|
off int // the target's function-relative offset (pc 0)
|
||||||
|
ok bool
|
||||||
|
}{
|
||||||
|
{"branch max", "BEQ X10, X11, tgt\nRET\n", 4094, true},
|
||||||
|
{"branch past max", "BEQ X10, X11, tgt\nRET\n", 4096, false},
|
||||||
|
{"branch back max", "BEQ X10, X11, tgt\nRET\n", -4096, true},
|
||||||
|
{"branch back past max", "BEQ X10, X11, tgt\nRET\n", -4098, false},
|
||||||
|
{"branchz past max", "BEQZ X10, tgt\nRET\n", 4096, false},
|
||||||
|
{"jump max", "JMP tgt\nRET\n", 1048574, true},
|
||||||
|
{"jump past max", "JMP tgt\nRET\n", 1048576, false},
|
||||||
|
{"jump back max", "JMP tgt\nRET\n", -1048576, true},
|
||||||
|
{"jump back past max", "JMP tgt\nRET\n", -1048578, false},
|
||||||
|
{"jal past max", "JAL tgt\nRET\n", 1048576, false},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
_, err := encodeOneInstrRISCV(t, "TEXT ·f(SB), NOSPLIT, $0\n\t"+c.src, 0, map[string]int{"tgt": c.off})
|
||||||
|
if c.ok && err != nil {
|
||||||
|
t.Fatalf("unexpected error: %v", err)
|
||||||
|
}
|
||||||
|
if !c.ok && err == nil {
|
||||||
|
t.Fatal("expected an out-of-range diagnostic, got none")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCVBranchFarBody drives the relaxation pass through the full
|
||||||
|
// assembler: a forward branch over a body larger than the B-type span is
|
||||||
|
// rewritten as an inverted branch over an inserted JMP, the same layout the
|
||||||
|
// toolchain produces, instead of wrapping to a wrong target.
|
||||||
|
func TestRISCVBranchFarBody(t *testing.T) {
|
||||||
|
var sb strings.Builder
|
||||||
|
sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n")
|
||||||
|
for range 1100 {
|
||||||
|
sb.WriteString("\tADD X10, X11, X12\n")
|
||||||
|
}
|
||||||
|
sb.WriteString("done:\n\tRET\n")
|
||||||
|
fn := firstTextRISCV(t, sb.String())
|
||||||
|
out, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("unexpected error: %v", err)
|
||||||
|
}
|
||||||
|
// The relaxed branch at offset 0 targets the inserted JMP at 4 (bne
|
||||||
|
// x10, x11, +4); the JMP at 4 carries the far forward displacement.
|
||||||
|
wantBranch := wordLE(riscvBType(riscvEnc{0x63, 0x1, 0x00}, 10, 11, 4))
|
||||||
|
if !bytes.Equal(out[0:4], wantBranch) {
|
||||||
|
t.Errorf("relaxed branch = %x, want %x", out[0:4], wantBranch)
|
||||||
|
}
|
||||||
|
// done sits after 1100 ADDs: 4 + 4400, i.e. offset 4404 from the JMP at 4.
|
||||||
|
wantJmp := wordLE(riscvJType(0, 4404))
|
||||||
|
if !bytes.Equal(out[4:8], wantJmp) {
|
||||||
|
t.Errorf("inserted JMP = %x, want %x", out[4:8], wantJmp)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_CSRRange checks the CSR address range: the 12-bit field is
|
||||||
|
// diagnosed rather than masked, so CSRRW $4096 does not silently address
|
||||||
|
// CSR 0.
|
||||||
|
func TestRISCV_CSRRange(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·csrhi(SB), NOSPLIT, $0
|
||||||
|
CSRRW $4096, X10, X11
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if _, _, _, _, _, _, err := assembleRISCV(fn); err == nil {
|
||||||
|
t.Error("expected an out-of-range error for CSR $4096, got none")
|
||||||
|
}
|
||||||
|
fn = firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·csrmax(SB), NOSPLIT, $0
|
||||||
|
CSRRW $4095, X10, X11
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
|
||||||
|
t.Errorf("CSR $4095 must assemble: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_Imm64Rejected checks that immediates outside the signed 32-bit
|
||||||
|
// span are diagnosed instead of silently truncated to their low 32 bits for
|
||||||
|
// the I-type arithmetic; the MOV forms materialise the wide constant instead
|
||||||
|
// (shift sequence or pooled load), like the toolchain.
|
||||||
|
func TestRISCV_Imm64Rejected(t *testing.T) {
|
||||||
|
cases := []string{
|
||||||
|
"ADDI $0x100000000, X10, X11",
|
||||||
|
"ANDI $-0x800000001, X10, X11",
|
||||||
|
"SUB $0x100000000, X10, X11",
|
||||||
|
}
|
||||||
|
for _, src := range cases {
|
||||||
|
fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·wide(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
|
||||||
|
if _, _, _, _, _, _, err := assembleRISCV(fn); err == nil {
|
||||||
|
t.Errorf("%s: expected an out-of-range error, got none", src)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The full signed 32-bit span still assembles, including the SUB form
|
||||||
|
// whose negated immediate only just fits.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·edge(SB), NOSPLIT, $0
|
||||||
|
MOV $2147483647, X10
|
||||||
|
MOV $-2147483648, X11
|
||||||
|
SUB $0x80000000, X12, X13
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
|
||||||
|
t.Errorf("int32-span immediates must assemble: %v", err)
|
||||||
|
}
|
||||||
|
// Beyond the span the MOV forms materialise the constant like the
|
||||||
|
// toolchain instead of diagnosing it.
|
||||||
|
fn = firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·pool(SB), NOSPLIT, $0
|
||||||
|
MOV $0x123456789, X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
|
||||||
|
t.Errorf("MOV with a 64-bit immediate must assemble: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvWants decodes code as little-endian words and pins each one; the
|
||||||
|
// expected values below were read off GOARCH=riscv64 go tool objdump of
|
||||||
|
// kernels assembled with go tool asm (the toolchain's riscv64.s testdata
|
||||||
|
// cross-checks the same words).
|
||||||
|
func riscvWants(t *testing.T, code []byte, want ...uint32) {
|
||||||
|
t.Helper()
|
||||||
|
got := make([]uint32, 0, len(code)/4)
|
||||||
|
for i := 0; i+4 <= len(code); i += 4 {
|
||||||
|
got = append(got, binary.LittleEndian.Uint32(code[i:]))
|
||||||
|
}
|
||||||
|
if len(got) < len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
|
||||||
|
}
|
||||||
|
// The RET (JALR) ends the sequence; only the pinned prefix is compared.
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvWantsHex pins the exact hex encoding of a function's instruction
|
||||||
|
// bytes, including any 2-byte compressed instructions in the stream; the
|
||||||
|
// expected strings were read off GOARCH=riscv64 go tool objdump of kernels
|
||||||
|
// assembled with go tool asm (the toolchain's riscv64.s testdata
|
||||||
|
// cross-checks the same words).
|
||||||
|
func riscvWantsHex(t *testing.T, code []byte, wantHex string) {
|
||||||
|
t.Helper()
|
||||||
|
got := hex.EncodeToString(code)
|
||||||
|
if got != wantHex {
|
||||||
|
t.Errorf("code = %s, want %s", got, wantHex)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_extendedPseudos pins the toolchain-synthesised instructions:
|
||||||
|
// ANDN/ORN (XORI + AND/OR through the destination or TMP), the five-word
|
||||||
|
// MIN/MAX expansion, the four-word rotate, ROR's compressed reverse shift
|
||||||
|
// (C.SLLI when rd == rs1, both non-zero, 1 <= sll <= 63), the identical-
|
||||||
|
// input MIN/MAX fold to C.MV, FABSD (FSGNJX.D), SEQZ and RDTIME (csrrs with
|
||||||
|
// the time CSR).
|
||||||
|
func TestRISCV_extendedPseudos(t *testing.T) {
|
||||||
|
t.Run("logic and minmax", func(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·l(SB), NOSPLIT, $0
|
||||||
|
ANDN X19, X20, X21
|
||||||
|
ANDN X19, X20
|
||||||
|
ORN X20, X19
|
||||||
|
MAX X26, X28, X29
|
||||||
|
MIN X29, X30, X5
|
||||||
|
MAX X5, X5
|
||||||
|
MAX X5, X5, X6
|
||||||
|
SEQZ X5, X6
|
||||||
|
NEG X5, X6
|
||||||
|
NOT X5
|
||||||
|
RDTIME X5
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// Words 0-10 up to the folded C.MV pair (halfwords 96 82 and 16 83),
|
||||||
|
// then SEQZ, NEG, NOT and RDTIME.
|
||||||
|
riscvWantsHex(t, code,
|
||||||
|
"93caf9ffb37a5a01"+"93cff9ff337afa01"+"934ffaffb3e9f901"+
|
||||||
|
"b32fae01b30ff041b34eae01b3fedf01b34ede01"+
|
||||||
|
"b3afee01b30ff041b342df01b3f25f00b3425f00"+
|
||||||
|
"9682"+"1683"+
|
||||||
|
"13b31200"+"33035040"+"93c2f2ff"+"f32210c0"+"67800000")
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("rotate", func(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·r(SB), NOSPLIT, $0
|
||||||
|
ROR X10, X11, X12
|
||||||
|
ROR X10, X11
|
||||||
|
ROR $63, X11
|
||||||
|
RORIW $31, X13, X14
|
||||||
|
RORIW $1, X14, X15
|
||||||
|
RORIW $3, X14
|
||||||
|
RORW X15, X16, X17
|
||||||
|
RORW $31, X13
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// The third ROR carries the compressed C.SLLI (05 86) in mid-stream.
|
||||||
|
riscvWantsHex(t, code,
|
||||||
|
"b30fa040b39ff50133d6a50033e6cf00"+
|
||||||
|
"b30fa040b39ff501b3d5a500b3e5bf00"+
|
||||||
|
"93dff5038605b3e5bf00"+
|
||||||
|
"9bdff6011b97160033e7ef00"+
|
||||||
|
"9b5f17009b17f701b3e7ff00"+
|
||||||
|
"9b5f37001b17d70133e7ef00"+
|
||||||
|
"b30ff040bb1ff801bb58f800b3e81f01"+
|
||||||
|
"9bdff6019b961600b3e6df00"+"67800000")
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("fp and branches", func(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·f(SB), NOSPLIT, $0
|
||||||
|
FABSD F1, F2
|
||||||
|
FSGNJD F1, F0, F2
|
||||||
|
FMADDD F1, F2, F3, F4
|
||||||
|
FMSUBD F1, F2, F3, F4
|
||||||
|
FNMSUBD F1, F2, F3, F4
|
||||||
|
BGT X5, X6, tgt
|
||||||
|
BLE X5, X6, tgt
|
||||||
|
BGTU X5, X6, tgt
|
||||||
|
BLEU X5, X6, tgt
|
||||||
|
tgt:
|
||||||
|
RDTIME X5
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
riscvWantsHex(t, code,
|
||||||
|
"53a11022"+"53011022"+"4382201a4782201a4b82201a"+
|
||||||
|
"63485300635653006364530063725300"+ // blt/bge/bltu/bgeu x6, x5
|
||||||
|
"f32210c0"+"67800000")
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_amoWords pins the full AMO family: every AMO carries aq and rl
|
||||||
|
// (funct7 |= 3), LR is acquire (funct7 |= 2) and SC release (funct7 |= 1),
|
||||||
|
// exactly as GOARCH=riscv64 go tool asm encodes them.
|
||||||
|
func TestRISCV_amoWords(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·amo(SB), NOSPLIT, $0
|
||||||
|
AMOSWAPW X5, (X6), X7
|
||||||
|
AMOSWAPD X5, (X6), X7
|
||||||
|
AMOADDW X5, (X6), X7
|
||||||
|
AMOADDD X5, (X6), X7
|
||||||
|
AMOANDW X5, (X6), X7
|
||||||
|
AMOANDD X5, (X6), X7
|
||||||
|
AMOORW X5, (X6), X7
|
||||||
|
AMOORD X5, (X6), X7
|
||||||
|
AMOXORW X5, (X6), X7
|
||||||
|
AMOXORD X5, (X6), X7
|
||||||
|
AMOMAXW X5, (X6), X7
|
||||||
|
AMOMAXD X5, (X6), X7
|
||||||
|
AMOMAXUW X5, (X6), X7
|
||||||
|
AMOMAXUD X5, (X6), X7
|
||||||
|
AMOMINUW X5, (X6), X7
|
||||||
|
AMOMINUD X5, (X6), X7
|
||||||
|
LRW (X5), X6
|
||||||
|
LRD (X5), X6
|
||||||
|
SCW X5, (X6), X7
|
||||||
|
SCD X5, (X6), X7
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
riscvWants(t, code,
|
||||||
|
0x0E5323AF, // amoswap.w
|
||||||
|
0x0E5333AF, // amoswap.d
|
||||||
|
0x065323AF, // amoaddd.w
|
||||||
|
0x065333AF, // amoadd.d
|
||||||
|
0x665323AF, // amoand.w
|
||||||
|
0x665333AF, // amoand.d
|
||||||
|
0x465323AF, // amoor.w
|
||||||
|
0x465333AF, // amoor.d
|
||||||
|
0x265323AF, // amoxor.w
|
||||||
|
0x265333AF, // amoxor.d
|
||||||
|
0xA65323AF, // amomax.w
|
||||||
|
0xA65333AF, // amomax.d
|
||||||
|
0xE65323AF, // amomaxu.w
|
||||||
|
0xE65333AF, // amomaxu.d
|
||||||
|
0xC65323AF, // amominu.w
|
||||||
|
0xC65333AF, // amominu.d
|
||||||
|
0x1402A32F, // lr.w (aq)
|
||||||
|
0x1402B32F, // lr.d
|
||||||
|
0x1A5323AF, // sc.w (rl)
|
||||||
|
0x1A5333AF, // sc.d
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_vectorWords pins the RVV slice and the VSET* encodings. The
|
||||||
|
// toolchain canonicalises an immediate avl to vsetivli even under the
|
||||||
|
// VSETVLI spelling (`VSETVLI $15` and `VSETIVLI $15` come out byte-
|
||||||
|
// identical), which is what the 0xC00 bit of the first word carries.
|
||||||
|
func TestRISCV_vectorWords(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VSETVLI X5, E8, M8, TA, MA, X6
|
||||||
|
VSETIVLI $4, E32, M1, TA, MA, X0
|
||||||
|
VSETVLI $15, E32, M1, TA, MA, X12
|
||||||
|
VADDVV V1, V2, V3
|
||||||
|
VADDVX X12, V12, V12
|
||||||
|
VXORVV V8, V16, V24
|
||||||
|
VMSEQVX X12, V8, V0
|
||||||
|
VMSNEVV V8, V16, V0
|
||||||
|
VSLLVI $8, V28, V30
|
||||||
|
VSRLVI $25, V29, V29
|
||||||
|
VFIRSTM V0, X6
|
||||||
|
VIDV V12
|
||||||
|
VMV4RV V8, V24
|
||||||
|
VLE8V (X10), V8
|
||||||
|
VSE8V V24, (X10)
|
||||||
|
VSE32V V9, (X11)
|
||||||
|
VLSSEG4E32V (X14), X0, V0
|
||||||
|
VLSSEG8E32V (X10), X0, V4
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
riscvWants(t, code,
|
||||||
|
0x0C32F357, // vsetvli x6, x5, vtype 0xc3 (E8, M8, TA, MA)
|
||||||
|
0xCD027057, // vsetivli x0, 4
|
||||||
|
0xCD07F657, // vsetivli x12, 15: VSETVLI $15 canonicalises to the same word
|
||||||
|
0x022081D7, // vadd.vv v3, v2, v1
|
||||||
|
0x02C64657, // vadd.vx v12, v12, x12
|
||||||
|
0x2F040C57, // vxor.vv v24, v16, v8
|
||||||
|
0x62864057, // vmseq.vx v0, v8, x12
|
||||||
|
0x67040057, // vmsne.vv v0, v16, v8
|
||||||
|
0x97C43F57, // vsll.vi v30, v28, 8
|
||||||
|
0xA3DCBED7, // vsrl.vi v29, v29, 25
|
||||||
|
0x4208A357, // vmfirst.m x6, v0
|
||||||
|
0x5208A657, // vid.v v12
|
||||||
|
0x9E81BC57, // vmv4r.v v24, v8
|
||||||
|
0x02050407, // vle8.v v8, (x10)
|
||||||
|
0x02050C27, // vse8.v v24, (x10)
|
||||||
|
0x0205E4A7, // vse32.v v9, (x11)
|
||||||
|
0x6A076007, // vlsseg4e32.v v0, (x14), x0
|
||||||
|
0xEA056207, // vlsseg8e32.v v4, (x10), x0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|||||||
+45
-35
@@ -4,9 +4,10 @@
|
|||||||
package asm
|
package asm
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"fmt"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
)
|
)
|
||||||
|
|
||||||
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
|
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
|
||||||
@@ -93,12 +94,19 @@ func riscvIsLeaf(t *ast.Text) bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
case "JALR":
|
case "JALR":
|
||||||
// JALR rs1, rd, a call when rd is X1; JALR offset(rs1) always
|
// JALR rd, offset(rs1) links when the destination register (the
|
||||||
// links to X1.
|
// first operand) is X1; JALR rs1, rd links when the second
|
||||||
|
// register is X1; JALR offset(rs1) always links to X1.
|
||||||
if len(in.Operands) == 1 {
|
if len(in.Operands) == 1 {
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
if len(in.Operands) >= 2 && regFromOperand(in.Operands[1]) == 1 {
|
if isMemOperand(in.Operands[1]) {
|
||||||
|
if regFromOperand(in.Operands[0]) == 1 {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if regFromOperand(in.Operands[1]) == 1 {
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -234,33 +242,36 @@ func riscvFitsCAddi(imm int32) bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// riscvPrologueSpadjPC returns the function-relative byte offset where the
|
// riscvPrologueSpadjPC returns the function-relative byte offset where the
|
||||||
// prologue has finished decrementing SP (the delta becomes autosize).
|
// prologue has finished decrementing SP (the delta becomes autosize). It is
|
||||||
|
// computed from the same expansion functions the prologue emits, so the
|
||||||
|
// large-frame X31 materialisations are counted: C.LUI + C.ADD before the SD,
|
||||||
|
// C.LUI + ADDIW + C.ADD for the SP adjust.
|
||||||
func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
|
func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
|
||||||
if fi.autosize == 0 {
|
if fi.autosize == 0 {
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
adj := int32(-fi.autosize)
|
||||||
return 4 + riscvSPAdjustLen(int32(-fi.autosize))
|
if fits12(adj) {
|
||||||
|
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
||||||
|
return 4 + len(riscvSPAdjust(adj))
|
||||||
|
}
|
||||||
|
return len(riscvAddressInX31(adj)) + 4 + len(riscvAddToSP(adj))
|
||||||
}
|
}
|
||||||
|
|
||||||
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
||||||
// (but not including) the final JALR, the point where SP is restored.
|
// (but not including) the final JALR, the point where SP is restored. The
|
||||||
|
// small frame closes with C.LDSP + ADDI/C.ADDI; the large frame materialises
|
||||||
|
// the adjustment through X31 (C.LUI + ADDIW + C.ADD).
|
||||||
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
|
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
|
||||||
if fi.autosize == 0 {
|
if fi.autosize == 0 {
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
adj := int32(fi.autosize)
|
||||||
return 2 + riscvSPAdjustLen(int32(fi.autosize))
|
if fits12(adj) {
|
||||||
}
|
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
||||||
|
return 2 + len(riscvSPAdjust(adj))
|
||||||
func riscvSPAdjustLen(imm int32) int {
|
|
||||||
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
|
|
||||||
return 2
|
|
||||||
}
|
}
|
||||||
if riscvFitsCAddi(imm) {
|
return 2 + len(riscvAddToSP(adj))
|
||||||
return 2
|
|
||||||
}
|
|
||||||
return 4
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// riscvResolvePseudo translates a pseudo-register memory reference into a
|
// riscvResolvePseudo translates a pseudo-register memory reference into a
|
||||||
@@ -286,19 +297,21 @@ func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32
|
|||||||
// including the inline morestack call (zero when the function needs no
|
// including the inline morestack call (zero when the function needs no
|
||||||
// guard). Unlike amd64 and arm64, the toolchain places the morestack call
|
// guard). Unlike amd64 and arm64, the toolchain places the morestack call
|
||||||
// between the guard and the body: the guard branches forward over it.
|
// between the guard and the body: the guard branches forward over it.
|
||||||
func riscvGuardLen(fi riscvFrameInfo) int {
|
func riscvGuardLen(fi riscvFrameInfo) (int, error) {
|
||||||
_, reloc := riscvGuard(fi)
|
g, _, err := riscvGuard(fi)
|
||||||
_ = reloc
|
if err != nil {
|
||||||
return len(riscvGuardBytes(fi))
|
return 0, err
|
||||||
|
}
|
||||||
|
return len(g), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// riscvGuard emits the stack-split guard prefix with the inline morestack
|
// riscvGuard emits the stack-split guard prefix with the inline morestack
|
||||||
// call: the branch skips forward over JAL X5 and JAL X0 straight into the
|
// call: the branch skips forward over JAL X5 and JAL X0 straight into the
|
||||||
// body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are
|
// body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are
|
||||||
// relative to the guard itself, which sits at function offset 0.
|
// relative to the guard itself, which sits at function offset 0.
|
||||||
func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
|
func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc, error) {
|
||||||
if !fi.needSplit {
|
if !fi.needSplit {
|
||||||
return nil, Reloc{}
|
return nil, Reloc{}, nil
|
||||||
}
|
}
|
||||||
// MOV 16(g), X6 (g.stackguard0), g = X27.
|
// MOV 16(g), X6 (g.stackguard0), g = X27.
|
||||||
out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16))
|
out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16))
|
||||||
@@ -310,14 +323,14 @@ func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
|
|||||||
var reloc Reloc
|
var reloc Reloc
|
||||||
switch fi.splitClass {
|
switch fi.splitClass {
|
||||||
case 0:
|
case 0:
|
||||||
// BLTU X6, SP, done (+8: over the CALL and the JMP back)
|
// BLTU X6, SP, done (+12: over the CALL and the JMP back)
|
||||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...)
|
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...)
|
||||||
call := len(out)
|
call := len(out)
|
||||||
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
|
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
|
||||||
out = append(out, wordLE(riscvJType(5, 0))...)
|
out = append(out, wordLE(riscvJType(5, 0))...)
|
||||||
out = append(out, jalBack()...)
|
out = append(out, jalBack()...)
|
||||||
case 1:
|
case 1:
|
||||||
// ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+8)
|
// ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+12)
|
||||||
off := int32(fi.autosize - stackSmall)
|
off := int32(fi.autosize - stackSmall)
|
||||||
out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...)
|
out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...)
|
||||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
|
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
|
||||||
@@ -335,7 +348,10 @@ func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
|
|||||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...)
|
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...)
|
||||||
addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off)
|
addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
addi = nil
|
// The ADDI expansion failed: the SP adjustment this class
|
||||||
|
// depends on is not emittable, and silently dropping it would
|
||||||
|
// corrupt every stack reference in the body.
|
||||||
|
return nil, Reloc{}, fmt.Errorf("stack-split guard: %w", err)
|
||||||
}
|
}
|
||||||
out = append(out, addi...)
|
out = append(out, addi...)
|
||||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
|
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
|
||||||
@@ -344,11 +360,5 @@ func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
|
|||||||
out = append(out, wordLE(riscvJType(5, 0))...)
|
out = append(out, wordLE(riscvJType(5, 0))...)
|
||||||
out = append(out, jalBack()...)
|
out = append(out, jalBack()...)
|
||||||
}
|
}
|
||||||
return out, reloc
|
return out, reloc, nil
|
||||||
}
|
|
||||||
|
|
||||||
// riscvGuardBytes emits the guard prefix bytes alone (sizing helper).
|
|
||||||
func riscvGuardBytes(fi riscvFrameInfo) []byte {
|
|
||||||
g, _ := riscvGuard(fi)
|
|
||||||
return g
|
|
||||||
}
|
}
|
||||||
|
|||||||
+42
-1
@@ -6,7 +6,7 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestRISCVFrameSpadjAndLines checks that a framed function records its
|
// TestRISCVFrameSpadjAndLines checks that a framed function records its
|
||||||
@@ -65,6 +65,47 @@ TEXT ·framed(SB), NOSPLIT, $16-16
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestRISCVFrameSpadjLargeFrame checks the stack-adjustment boundaries of a
|
||||||
|
// frame past the imm12 range: the prologue materialises the LR-store address
|
||||||
|
// and the SP adjustment through X31 (C.LUI + C.ADD + SD, then C.LUI + ADDIW +
|
||||||
|
// C.ADD), so the SP boundary lands at PC 16, and the RET closes with
|
||||||
|
// C.LDSP plus the same X31 adjustment, 10 bytes. Regression: both helpers
|
||||||
|
// assumed the small-frame prologue and reported 8 and 6.
|
||||||
|
func TestRISCVFrameSpadjLargeFrame(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("bigframe_riscv64.s", `#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·big(SB), NOSPLIT, $9000-8
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
fn := img.Funcs[0]
|
||||||
|
|
||||||
|
// autosize = 9008. Prologue: C.LUI X31 + C.ADD X31,SP (4) + SD (4) +
|
||||||
|
// C.LUI X31 + ADDIW X31 + C.ADD SP,X31 (8) = 16 bytes to the SP boundary;
|
||||||
|
// C.SDSP X1 (2) follows, so the body starts at 18.
|
||||||
|
wantSpadj := []SpadjStep{{PC: 16, Value: 9008}, {PC: 36, Value: 0}}
|
||||||
|
if len(fn.Spadj) != len(wantSpadj) {
|
||||||
|
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
|
||||||
|
}
|
||||||
|
for i := range wantSpadj {
|
||||||
|
if fn.Spadj[i] != wantSpadj[i] {
|
||||||
|
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The FP load materialises its 9016-byte offset through X31 as well
|
||||||
|
// (8 bytes), then RET's epilogue (C.LDSP + X31 adjust = 10) plus JALR.
|
||||||
|
if fn.Size != 18+8+14 {
|
||||||
|
t.Errorf("size = %d, want %d", fn.Size, 18+8+14)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestRISCVRegAliases checks the Go ABI register aliases that the toolchain
|
// TestRISCVRegAliases checks the Go ABI register aliases that the toolchain
|
||||||
// defines: LR is the link register (X1) and TMP is the assembler scratch
|
// defines: LR is the link register (X1) and TMP is the assembler scratch
|
||||||
// register (X31/T6).
|
// register (X31/T6).
|
||||||
|
|||||||
+84
-1
@@ -13,7 +13,7 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
|
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
|
||||||
@@ -74,6 +74,89 @@ DATA callee<>+0(SB)/8, $42
|
|||||||
t.Error("ELF object missing R_RISCV_JAL relocation")
|
t.Error("ELF object missing R_RISCV_JAL relocation")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestELFRISCVPCRELLO12Anchor checks the psABI's LO12 pairing rule: the
|
||||||
|
// R_RISCV_PCREL_LO12_I/S relocation must reference a symbol whose value is
|
||||||
|
// the AUIPC site of its HI20 partner (psABI §8.4.9; cmd/link generates one
|
||||||
|
// local text symbol per AUIPC for exactly this). The emitter pairs each
|
||||||
|
// HI20 (against the target symbol) with a LO12 against the .text section
|
||||||
|
// symbol whose addend is the AUIPC's section-relative offset, so S + A is
|
||||||
|
// the AUIPC address.
|
||||||
|
func TestELFRISCVPCRELLO12Anchor(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("k_riscv64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·sb(SB), NOSPLIT, $0-0
|
||||||
|
MOV $answer<>(SB), X10
|
||||||
|
MOV answer<>(SB), X11
|
||||||
|
MOV X12, answer<>(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
GLOBL answer<>(SB), RODATA, $8
|
||||||
|
DATA answer<>+0(SB)/8, $42
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFRISCVObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFRISCVObject: %v", err)
|
||||||
|
}
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse ELF: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
if flags := binary.LittleEndian.Uint32(obj[48:]); flags != efRISCVFloatAbiDouble {
|
||||||
|
t.Errorf("e_flags = %#x, want %#x (EF_RISCV_FLOAT_ABI_DOUBLE)", flags, efRISCVFloatAbiDouble)
|
||||||
|
}
|
||||||
|
rela := ef.Section(".rela.text")
|
||||||
|
if rela == nil {
|
||||||
|
t.Fatal("missing .rela.text")
|
||||||
|
}
|
||||||
|
b, err := rela.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if len(b) != 6*24 {
|
||||||
|
t.Fatalf(".rela.text holds %d entries, want six (three HI20/LO12 pairs)", len(b)/24)
|
||||||
|
}
|
||||||
|
le := binary.LittleEndian
|
||||||
|
wantLo := []uint32{rRISCVPCRELLO12I, rRISCVPCRELLO12I, rRISCVPCRELLO12S}
|
||||||
|
for p := range 3 {
|
||||||
|
auipc := 8 * p
|
||||||
|
hi := b[p*2*24:]
|
||||||
|
lo := b[(p*2+1)*24:]
|
||||||
|
if off := le.Uint64(hi[0:]); off != uint64(auipc) {
|
||||||
|
t.Errorf("pair %d: HI20 r_offset = %d, want %d (the AUIPC)", p, off, auipc)
|
||||||
|
}
|
||||||
|
if typ := uint32(le.Uint64(hi[8:])); typ != rRISCVPCRELHI20 {
|
||||||
|
t.Errorf("pair %d: HI20 type = %d, want %d", p, typ, rRISCVPCRELHI20)
|
||||||
|
}
|
||||||
|
if sym := int(le.Uint64(hi[8:]) >> 32); sym == 0 || sym == 1 {
|
||||||
|
t.Errorf("pair %d: HI20 against symbol %d, want the target", p, sym)
|
||||||
|
}
|
||||||
|
if off := le.Uint64(lo[0:]); off != uint64(auipc+4) {
|
||||||
|
t.Errorf("pair %d: LO12 r_offset = %d, want %d", p, off, auipc+4)
|
||||||
|
}
|
||||||
|
if typ := uint32(le.Uint64(lo[8:])); typ != wantLo[p] {
|
||||||
|
t.Errorf("pair %d: LO12 type = %d, want %d", p, typ, wantLo[p])
|
||||||
|
}
|
||||||
|
// The LO12 must denote the AUIPC site: the .text section symbol
|
||||||
|
// (index 1) plus the AUIPC's section-relative offset as addend.
|
||||||
|
if sym := int(le.Uint64(lo[8:]) >> 32); sym != 1 {
|
||||||
|
t.Errorf("pair %d: LO12 against symbol %d, want 1 (the .text section symbol)", p, sym)
|
||||||
|
}
|
||||||
|
if add := int64(le.Uint64(lo[16:])); add != int64(auipc) {
|
||||||
|
t.Errorf("pair %d: LO12 addend = %d, want %d (S + A = the AUIPC address)", p, add, auipc)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestGOObjectRISCVStructure(t *testing.T) {
|
func TestGOObjectRISCVStructure(t *testing.T) {
|
||||||
f, errs := parser.Parse("k_riscv64.s", `
|
f, errs := parser.Parse("k_riscv64.s", `
|
||||||
#include "textflag.h"
|
#include "textflag.h"
|
||||||
|
|||||||
+485
-9
@@ -41,7 +41,7 @@ const (
|
|||||||
vexExtract
|
vexExtract
|
||||||
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
||||||
// in ModRM.reg and the destination in r/m, the layout of the EVEX
|
// in ModRM.reg and the destination in r/m, the layout of the EVEX
|
||||||
// narrowing stores (VPMOVDW, VPMOVQD).
|
// narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ.
|
||||||
vexRMRev
|
vexRMRev
|
||||||
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
||||||
// vector length follows the source: the packed-double → dword
|
// vector length follows the source: the packed-double → dword
|
||||||
@@ -52,6 +52,34 @@ const (
|
|||||||
vexRMSrcLen
|
vexRMSrcLen
|
||||||
// vexZero is the no-operand form (VZEROUPPER).
|
// vexZero is the no-operand form (VZEROUPPER).
|
||||||
vexZero
|
vexZero
|
||||||
|
// vexZeroAll is the no-operand form that zeroes the full upper state
|
||||||
|
// (VZEROALL, the L = 1 twin of VZEROUPPER).
|
||||||
|
vexZeroAll
|
||||||
|
// vexNDS3GPR is the three-operand NDS form over general-purpose
|
||||||
|
// registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0.
|
||||||
|
vexNDS3GPR
|
||||||
|
// vexImmRMGPR is the immediate form over general-purpose registers
|
||||||
|
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
|
||||||
|
vexImmRMGPR
|
||||||
|
// vexRMOpGPR is the two-operand /digit form over general-purpose
|
||||||
|
// registers (BLSI, BLSMSK, BLSR): ModRM.reg = /digit, ModRM.rm = src
|
||||||
|
// (op0), VEX.vvvv = dst (op1), L = 0.
|
||||||
|
vexRMOpGPR
|
||||||
|
// vexCountGPR is the three-operand count form over general-purpose
|
||||||
|
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): the first operand rides
|
||||||
|
// VEX.vvvv and the second is r/m, the opposite pairing of the ANDN
|
||||||
|
// family, with reg = dst (op2), L = 0.
|
||||||
|
vexCountGPR
|
||||||
|
// vexExtractGPR is the lane-extract-to-GPR form `OP $imm, xsrc, GPR/mem
|
||||||
|
// dst`: ModRM.reg = xsrc (op1), ModRM.rm = destination (op2), imm8 =
|
||||||
|
// op0, the VPEXTRB/W/D/Q layout. EVEX only; the destination never
|
||||||
|
// carries a vector length, so the register the L'L field follows is the
|
||||||
|
// XMM source.
|
||||||
|
vexExtractGPR
|
||||||
|
// vexBlend4 is the four-operand variable blend `OP mask, src2, src1,
|
||||||
|
// dst` (VPBLENDVB): ModRM.reg = dst (op3), VEX.vvvv = src1 (op2),
|
||||||
|
// ModRM.rm = src2 (op1) and the mask register in the /is4 byte (op0).
|
||||||
|
vexBlend4
|
||||||
)
|
)
|
||||||
|
|
||||||
// vexSpec describes one VEX instruction's encoding parameters.
|
// vexSpec describes one VEX instruction's encoding parameters.
|
||||||
@@ -125,6 +153,12 @@ var vexTable = map[string]vexSpec{
|
|||||||
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
||||||
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
|
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
|
||||||
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
||||||
|
// Scalar fused multiply-add (NDS form). The Go assembler carries the
|
||||||
|
// same 66 prefix as the packed forms on every FMA row, and W1 on the
|
||||||
|
// double-precision spellings, so SD shares PD's prefix/W pair and the
|
||||||
|
// scalar width rides on the W bit.
|
||||||
|
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3},
|
||||||
|
|
||||||
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
|
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
|
||||||
// no vvvv).
|
// no vvvv).
|
||||||
@@ -172,8 +206,28 @@ var vexTable = map[string]vexSpec{
|
|||||||
|
|
||||||
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
|
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
|
||||||
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
|
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
|
||||||
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8).
|
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8), and its
|
||||||
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
|
// double twin under op 01; the in-lane permutes under 04/05.
|
||||||
|
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
|
||||||
|
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM},
|
||||||
|
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM},
|
||||||
|
"VPERMILPD": {3, 0x05, 0, 1, -1, vexImmRM},
|
||||||
|
// VEX.66.0F3A.W0, the immediate-controlled AVX tail: the rounding
|
||||||
|
// pair, the AES key assistant and the string compares.
|
||||||
|
"VROUNDPD": {3, 0x09, 0, 1, -1, vexImmRM},
|
||||||
|
"VROUNDPS": {3, 0x08, 0, 1, -1, vexImmRM},
|
||||||
|
"VAESKEYGENASSIST": {3, 0xDF, 0, 1, -1, vexImmRM},
|
||||||
|
"VPCMPESTRI": {3, 0x61, 0, 1, -1, vexImmRM},
|
||||||
|
"VPCMPESTRM": {3, 0x60, 0, 1, -1, vexImmRM},
|
||||||
|
"VPCMPISTRI": {3, 0x63, 0, 1, -1, vexImmRM},
|
||||||
|
"VPCMPISTRM": {3, 0x62, 0, 1, -1, vexImmRM},
|
||||||
|
// VEX.128.66.0F3A.W0, the scalar lane extract to a GPR or memory
|
||||||
|
// (reg = the XMM source, r/m = the destination).
|
||||||
|
"VEXTRACTPS": {3, 0x17, 0, 1, -1, vexExtractGPR},
|
||||||
|
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR},
|
||||||
|
// VEX.128.66.0F3A.W0, the four-operand variable blend with its mask
|
||||||
|
// register in the /is4 byte.
|
||||||
|
"VPBLENDVB": {3, 0x4C, 0, 1, -1, vexBlend4},
|
||||||
|
|
||||||
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
|
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
|
||||||
// imm8).
|
// imm8).
|
||||||
@@ -192,6 +246,57 @@ var vexTable = map[string]vexSpec{
|
|||||||
|
|
||||||
// VEX.128.0F.W0, no operands.
|
// VEX.128.0F.W0, no operands.
|
||||||
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
||||||
|
// VEX.256.0F.W0, zero all vector registers (the L = 1 twin).
|
||||||
|
"VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll},
|
||||||
|
// VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare.
|
||||||
|
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm},
|
||||||
|
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm},
|
||||||
|
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3},
|
||||||
|
// VEX.128/256.0F.WIG, packed single XOR (NDS form).
|
||||||
|
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3},
|
||||||
|
// VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control.
|
||||||
|
"VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm},
|
||||||
|
"VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm},
|
||||||
|
// VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling
|
||||||
|
// falls through to the EVEX table.
|
||||||
|
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm},
|
||||||
|
// VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst).
|
||||||
|
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm},
|
||||||
|
// VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8).
|
||||||
|
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm},
|
||||||
|
// BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR).
|
||||||
|
"ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR},
|
||||||
|
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
|
||||||
|
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
|
||||||
|
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
|
||||||
|
// VEX.NDS.LZ.0F38, the BMI2 three-operand bit ops: BEXTR and BZHI
|
||||||
|
// share the F7/F5 opcodes across W, the variable shifts carry their
|
||||||
|
// direction in the prefix (SHLX 66, SHRX F2, SARX F3) and PDEP/PEXT
|
||||||
|
// in F2/F3.
|
||||||
|
"BEXTRL": {2, 0xF7, 0, 0, -1, vexCountGPR},
|
||||||
|
"BEXTRQ": {2, 0xF7, 1, 0, -1, vexCountGPR},
|
||||||
|
"BZHIL": {2, 0xF5, 0, 0, -1, vexCountGPR},
|
||||||
|
"BZHIQ": {2, 0xF5, 1, 0, -1, vexCountGPR},
|
||||||
|
"SARXL": {2, 0xF7, 0, 2, -1, vexCountGPR},
|
||||||
|
"SARXQ": {2, 0xF7, 1, 2, -1, vexCountGPR},
|
||||||
|
"SHLXL": {2, 0xF7, 0, 1, -1, vexCountGPR},
|
||||||
|
"SHLXQ": {2, 0xF7, 1, 1, -1, vexCountGPR},
|
||||||
|
"SHRXL": {2, 0xF7, 0, 3, -1, vexCountGPR},
|
||||||
|
"SHRXQ": {2, 0xF7, 1, 3, -1, vexCountGPR},
|
||||||
|
"PDEPL": {2, 0xF5, 0, 3, -1, vexNDS3GPR},
|
||||||
|
"PDEPQ": {2, 0xF5, 1, 3, -1, vexNDS3GPR},
|
||||||
|
"PEXTL": {2, 0xF5, 0, 2, -1, vexNDS3GPR},
|
||||||
|
"PEXTQ": {2, 0xF5, 1, 2, -1, vexNDS3GPR},
|
||||||
|
// VEX.LZ.0F38.W, the BMI1 unary bit ops (src, dst: ModRM.reg = /digit,
|
||||||
|
// rm = src, vvvv = dst).
|
||||||
|
"BLSIL": {2, 0xF3, 0, 0, 3, vexRMOpGPR},
|
||||||
|
"BLSIQ": {2, 0xF3, 1, 0, 3, vexRMOpGPR},
|
||||||
|
"BLSMSKL": {2, 0xF3, 0, 0, 2, vexRMOpGPR},
|
||||||
|
"BLSMSKQ": {2, 0xF3, 1, 0, 2, vexRMOpGPR},
|
||||||
|
"BLSRL": {2, 0xF3, 0, 0, 1, vexRMOpGPR},
|
||||||
|
"BLSRQ": {2, 0xF3, 1, 0, 1, vexRMOpGPR},
|
||||||
|
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
|
||||||
|
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
|
||||||
|
|
||||||
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
||||||
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
||||||
@@ -200,6 +305,14 @@ var vexTable = map[string]vexSpec{
|
|||||||
// rm=scalar memory; SD is 256-bit only).
|
// rm=scalar memory; SD is 256-bit only).
|
||||||
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
||||||
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
||||||
|
// VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a
|
||||||
|
// YMM (the encoder rejects an XMM destination, as go tool asm does).
|
||||||
|
"VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM},
|
||||||
|
// VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg,
|
||||||
|
// memory destination in rm).
|
||||||
|
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev},
|
||||||
|
// VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv).
|
||||||
|
"VPTEST": {2, 0x17, 0, 1, -1, vexRM},
|
||||||
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
|
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
|
||||||
// source).
|
// source).
|
||||||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
||||||
@@ -242,6 +355,128 @@ var vexTable = map[string]vexSpec{
|
|||||||
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
||||||
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||||
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||||
|
|
||||||
|
// --- the VEX forms the avx512enc corpus exercises alongside the EVEX
|
||||||
|
// spellings, read off the toolchain opcode tables ---
|
||||||
|
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3},
|
||||||
|
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3},
|
||||||
|
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3},
|
||||||
|
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3},
|
||||||
|
"VANDNPD": {1, 0x55, 0, 1, -1, vexNDS3},
|
||||||
|
"VANDPD": {1, 0x54, 0, 1, -1, vexNDS3},
|
||||||
|
"VCOMISD": {1, 0x2F, 0, 1, -1, vexRM},
|
||||||
|
"VCVTSD2SS": {1, 0x5A, 0, 3, -1, vexNDS3},
|
||||||
|
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3},
|
||||||
|
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3},
|
||||||
|
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm},
|
||||||
|
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3},
|
||||||
|
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM},
|
||||||
|
"VMOVNTPD": {1, 0x2B, 0, 1, -1, vexRMRev},
|
||||||
|
"VORPD": {1, 0x56, 0, 1, -1, vexNDS3},
|
||||||
|
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3},
|
||||||
|
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3},
|
||||||
|
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3},
|
||||||
|
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3},
|
||||||
|
"VPCMPEQQ": {2, 0x29, 0, 1, -1, vexNDS3},
|
||||||
|
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3},
|
||||||
|
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3},
|
||||||
|
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3},
|
||||||
|
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3},
|
||||||
|
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3},
|
||||||
|
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtract},
|
||||||
|
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtract},
|
||||||
|
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtract},
|
||||||
|
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm},
|
||||||
|
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm},
|
||||||
|
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3},
|
||||||
|
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3},
|
||||||
|
"VPMULUDQ": {1, 0xF4, 0, 1, -1, vexNDS3},
|
||||||
|
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3},
|
||||||
|
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3},
|
||||||
|
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3},
|
||||||
|
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3},
|
||||||
|
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3},
|
||||||
|
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3},
|
||||||
|
"VPUNPCKHQDQ": {1, 0x6D, 0, 1, -1, vexNDS3},
|
||||||
|
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3},
|
||||||
|
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3},
|
||||||
|
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3},
|
||||||
|
"VSQRTPD": {1, 0x51, 0, 1, -1, vexRM},
|
||||||
|
"VSQRTSD": {1, 0x51, 0, 3, -1, vexNDS3},
|
||||||
|
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3},
|
||||||
|
"VUCOMISD": {1, 0x2E, 0, 1, -1, vexRM},
|
||||||
|
|
||||||
|
// VEX.0F.WIG, the plain-prefix single/double arithmetic and unpack
|
||||||
|
// spellings (no 66 prefix; WIG, so W = 0).
|
||||||
|
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3},
|
||||||
|
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3},
|
||||||
|
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3},
|
||||||
|
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3},
|
||||||
|
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3},
|
||||||
|
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM},
|
||||||
|
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev},
|
||||||
|
// VEX.128.66.0F, the scalar and packed compare forms.
|
||||||
|
"VCOMISS": {1, 0x2F, 0, 1, -1, vexRM},
|
||||||
|
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM},
|
||||||
|
// VEX.128.0F.F3/F2.W0, the high/low word shuffles ($imm, src, dst).
|
||||||
|
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM},
|
||||||
|
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM},
|
||||||
}
|
}
|
||||||
|
|
||||||
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
|
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
|
||||||
@@ -290,6 +525,8 @@ type vexMoveSpec struct {
|
|||||||
var vexMoveTable = map[string]vexMoveSpec{
|
var vexMoveTable = map[string]vexMoveSpec{
|
||||||
// VEX.128/256.F3.0F.WIG, unaligned integer move.
|
// VEX.128/256.F3.0F.WIG, unaligned integer move.
|
||||||
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
||||||
|
// VEX.128/256.66.0F.WIG, aligned integer move.
|
||||||
|
"VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
||||||
// VEX.128/256.66.0F.WIG, unaligned packed double move.
|
// VEX.128/256.66.0F.WIG, unaligned packed double move.
|
||||||
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
|
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
|
||||||
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
|
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
|
||||||
@@ -311,8 +548,16 @@ func isVex(mnemUpper string) bool {
|
|||||||
if _, ok := vexTable[mnemUpper]; ok {
|
if _, ok := vexTable[mnemUpper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
_, ok := vexMoveTable[mnemUpper]
|
if _, ok := vexMoveTable[mnemUpper]; ok {
|
||||||
return ok
|
return true
|
||||||
|
}
|
||||||
|
// The dual-shape moves (VMOVHPD/VMOVLPD) pick their VEX form by operand
|
||||||
|
// count in encodeVex.
|
||||||
|
switch mnemUpper {
|
||||||
|
case "VMOVHPD", "VMOVLPD":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
|
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
|
||||||
@@ -324,6 +569,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
|||||||
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination
|
||||||
|
// only; an XMM destination is rejected exactly as go tool asm does.
|
||||||
|
if mnemUpper == "VBROADCASTI128" {
|
||||||
|
dstReg, ok := ops[len(ops)-1].(Reg)
|
||||||
|
if len(ops) != 2 || !ok || dstReg.size != 32 {
|
||||||
|
return fmt.Errorf("VBROADCASTI128 requires a YMM destination")
|
||||||
|
}
|
||||||
|
}
|
||||||
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
||||||
return e.encodeVexMove(mnemUpper, ms, ops)
|
return e.encodeVexMove(mnemUpper, ms, ops)
|
||||||
}
|
}
|
||||||
@@ -338,6 +591,22 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
|||||||
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
|
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// The high/low double moves split by operand count: three operands
|
||||||
|
// load-and-insert (mem, src, dst, an NDS form), two store (xmm, m64,
|
||||||
|
// the reversed store layout).
|
||||||
|
if mnemUpper == "VMOVHPD" || mnemUpper == "VMOVLPD" {
|
||||||
|
loadOp, storeOp := byte(0x16), byte(0x17)
|
||||||
|
if mnemUpper == "VMOVLPD" {
|
||||||
|
loadOp, storeOp = 0x12, 0x13
|
||||||
|
}
|
||||||
|
switch len(ops) {
|
||||||
|
case 3:
|
||||||
|
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: loadOp, w: 0, pp: 1, opdigit: -1, form: vexNDS3}, ops)
|
||||||
|
case 2:
|
||||||
|
return e.encodeVexRMRev(vexSpec{mapSel: 1, opcode: storeOp, w: 0, pp: 1, opdigit: -1, form: vexRMRev}, ops)
|
||||||
|
}
|
||||||
|
return fmt.Errorf("%s expects 2 or 3 operands, got %d", mnemUpper, len(ops))
|
||||||
|
}
|
||||||
spec := vexTable[mnemUpper]
|
spec := vexTable[mnemUpper]
|
||||||
switch spec.form {
|
switch spec.form {
|
||||||
case vexNDS3:
|
case vexNDS3:
|
||||||
@@ -352,10 +621,26 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
|||||||
return e.encodeVexNDS3Imm(spec, ops)
|
return e.encodeVexNDS3Imm(spec, ops)
|
||||||
case vexExtract:
|
case vexExtract:
|
||||||
return e.encodeVexExtract(spec, ops)
|
return e.encodeVexExtract(spec, ops)
|
||||||
|
case vexExtractGPR:
|
||||||
|
return e.encodeVexExtractGPR(spec, ops)
|
||||||
|
case vexBlend4:
|
||||||
|
return e.encodeVexBlend4(spec, ops)
|
||||||
case vexRMSrcLen:
|
case vexRMSrcLen:
|
||||||
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
||||||
case vexZero:
|
case vexZero:
|
||||||
return e.encodeVexZero(mnemUpper, spec, ops)
|
return e.encodeVexZero(mnemUpper, spec, ops)
|
||||||
|
case vexZeroAll:
|
||||||
|
return e.encodeVexZeroAll(mnemUpper, spec, ops)
|
||||||
|
case vexNDS3GPR:
|
||||||
|
return e.encodeVexNDS3GPR(spec, ops)
|
||||||
|
case vexImmRMGPR:
|
||||||
|
return e.encodeVexImmRMGPR(spec, ops)
|
||||||
|
case vexRMOpGPR:
|
||||||
|
return e.encodeVexRMOpGPR(spec, ops)
|
||||||
|
case vexCountGPR:
|
||||||
|
return e.encodeVexCountGPR(spec, ops)
|
||||||
|
case vexRMRev:
|
||||||
|
return e.encodeVexRMRev(spec, ops)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
|
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
|
||||||
}
|
}
|
||||||
@@ -457,9 +742,10 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return fmt.Errorf("shift count must be an immediate")
|
return fmt.Errorf("shift count must be an immediate")
|
||||||
}
|
}
|
||||||
srcReg, ok := src.(Reg)
|
// The count source is a vector register or memory; the VEX length
|
||||||
if !ok || !srcReg.isVec() {
|
// follows the destination register either way.
|
||||||
return fmt.Errorf("shift source must be a vector register")
|
if !vecOrMem(src) {
|
||||||
|
return fmt.Errorf("shift source must be a vector register or memory")
|
||||||
}
|
}
|
||||||
dstReg, ok := dst.(Reg)
|
dstReg, ok := dst.(Reg)
|
||||||
if !ok || !dstReg.isVec() {
|
if !ok || !dstReg.isVec() {
|
||||||
@@ -467,7 +753,7 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
vvvvBar := 15 - (dstReg.idx & 15)
|
vvvvBar := 15 - (dstReg.idx & 15)
|
||||||
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, srcReg); err != nil {
|
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, src); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
immByte, err := imm8(int64(immVal))
|
immByte, err := imm8(int64(immVal))
|
||||||
@@ -607,6 +893,196 @@ func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1
|
||||||
|
// twin of VZEROUPPER.
|
||||||
|
func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
|
||||||
|
}
|
||||||
|
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1.
|
||||||
|
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose
|
||||||
|
// registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1,
|
||||||
|
// rm = src2 and L = 0.
|
||||||
|
func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
src2, src1, dst := ops[0], ops[1], ops[2]
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||||
|
}
|
||||||
|
vvvvReg, ok := src1.(Reg)
|
||||||
|
if !ok || vvvvReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
|
||||||
|
}
|
||||||
|
rBit := 0
|
||||||
|
if dstReg.idx >= 8 {
|
||||||
|
rBit = 1
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexImmRMGPR encodes the immediate form over general-purpose
|
||||||
|
// registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0.
|
||||||
|
func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, src, dst := ops[0], ops[1], ops[2]
|
||||||
|
immVal, ok := imm.(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("shift control must be an immediate")
|
||||||
|
}
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(immVal))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if err := e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
e.out = append(e.out, immByte)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexRMOpGPR encodes the two-operand /digit form over general-purpose
|
||||||
|
// registers (BLSI, BLSMSK, BLSR): OP src, dst with ModRM.reg = /digit,
|
||||||
|
// ModRM.rm = src and VEX.vvvv = dst.
|
||||||
|
func (e *enc) encodeVexRMOpGPR(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("instruction expects 2 operands (src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
src, dst := ops[0], ops[1]
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, 0, spec.opdigit, 0, 15-(dstReg.idx&15), src)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexCountGPR encodes the three-operand count form over general-purpose
|
||||||
|
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): OP src, count, dst with
|
||||||
|
// VEX.vvvv = src (op0), ModRM.rm = count (op1), ModRM.reg = dst (op2).
|
||||||
|
func (e *enc) encodeVexCountGPR(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("VEX count instruction expects 3 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
src, count, dst := ops[0], ops[1], ops[2]
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||||
|
}
|
||||||
|
countReg, ok := count.(Reg)
|
||||||
|
if !ok || countReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX count operand must be a general-purpose register")
|
||||||
|
}
|
||||||
|
srcReg, ok := src.(Reg)
|
||||||
|
if !ok || srcReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX count source must be a general-purpose register")
|
||||||
|
}
|
||||||
|
rBit := 0
|
||||||
|
if dstReg.idx >= 8 {
|
||||||
|
rBit = 1
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(srcReg.idx&15), count)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
|
||||||
|
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
|
||||||
|
// a store with no register-destination form).
|
||||||
|
func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("store expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
srcReg, ok := ops[0].(Reg)
|
||||||
|
if !ok || !srcReg.isVec() {
|
||||||
|
return fmt.Errorf("store source must be a vector register")
|
||||||
|
}
|
||||||
|
if !memOperand(ops[1]) {
|
||||||
|
return fmt.Errorf("store destination must be memory")
|
||||||
|
}
|
||||||
|
rBit := 0
|
||||||
|
if srcReg.idx >= 8 {
|
||||||
|
rBit = 1
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexExtractGPR encodes the lane extract to a general-purpose register
|
||||||
|
// or memory (VEXTRACTPS): OP $imm, xsrc, gpr/mem with the XMM source in
|
||||||
|
// ModRM.reg and the destination in r/m, L = 0.
|
||||||
|
func (e *enc) encodeVexExtractGPR(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, src, dst := ops[0], ops[1], ops[2]
|
||||||
|
immVal, ok := imm.(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("extract lane must be an immediate")
|
||||||
|
}
|
||||||
|
srcReg, ok := src.(Reg)
|
||||||
|
if !ok || !srcReg.isVec() || srcReg.size != 16 {
|
||||||
|
return fmt.Errorf("extract source must be an XMM register")
|
||||||
|
}
|
||||||
|
if _, isReg := dst.(Reg); !isReg && !memOperand(dst) {
|
||||||
|
return fmt.Errorf("extract destination must be a register or memory")
|
||||||
|
}
|
||||||
|
rBit := 0
|
||||||
|
if srcReg.idx >= 8 {
|
||||||
|
rBit = 1
|
||||||
|
}
|
||||||
|
if err := e.emitVexFields(spec, 0, srcReg.idx&7, rBit, 15, dst); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(immVal))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
e.out = append(e.out, immByte)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexBlend4 encodes the four-operand variable blend (VPBLENDVB):
|
||||||
|
// OP mask, src2, src1, dst with ModRM.reg = dst, VEX.vvvv = src1, r/m =
|
||||||
|
// src2 and the mask XMM register in the trailing /is4 byte.
|
||||||
|
func (e *enc) encodeVexBlend4(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 4 {
|
||||||
|
return fmt.Errorf("blend expects 4 operands (mask, src2, src1, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
mask, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
|
||||||
|
maskReg, ok := mask.(Reg)
|
||||||
|
if !ok || !maskReg.isVec() || maskReg.size != 16 {
|
||||||
|
return fmt.Errorf("blend mask must be an XMM register")
|
||||||
|
}
|
||||||
|
vvvvReg, ok := src1.(Reg)
|
||||||
|
if !ok || !vvvvReg.isVec() {
|
||||||
|
return fmt.Errorf("blend second source must be a vector register")
|
||||||
|
}
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("blend destination must be a vector register")
|
||||||
|
}
|
||||||
|
rBit := 0
|
||||||
|
if dstReg.idx >= 8 {
|
||||||
|
rBit = 1
|
||||||
|
}
|
||||||
|
if err := e.emitVexFields(spec, dstReg.vecLenBit(), dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
// The /is4 byte names the mask register: bits [3:0] its low nibble,
|
||||||
|
// bit 7 the fourth register bit (X8-X15).
|
||||||
|
e.out = append(e.out, byte(maskReg.idx&7)|byte((maskReg.idx&8)<<4))
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
|
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
|
||||||
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
|
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
|
||||||
// move uses the store-form layout (reg = source, rm = destination), matching
|
// move uses the store-form layout (reg = source, rm = destination), matching
|
||||||
|
|||||||
+121
-5
@@ -19,6 +19,65 @@ func vreg(t *testing.T, name string) Reg {
|
|||||||
return r
|
return r
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// x86asmUnrecognised lists the VEX mnemonics whose machine code the
|
||||||
|
// golang.org/x/arch decoder cannot resolve; their bytes are verified against
|
||||||
|
// go tool asm in the ground-truth tests instead.
|
||||||
|
var x86asmUnrecognised = map[string]bool{
|
||||||
|
"ANDNL": true,
|
||||||
|
"ANDNQ": true,
|
||||||
|
"MULXL": true,
|
||||||
|
"MULXQ": true,
|
||||||
|
"RORXL": true,
|
||||||
|
"RORXQ": true,
|
||||||
|
"VFMADD213SD": true,
|
||||||
|
"VFNMADD231SD": true,
|
||||||
|
// The scalar FMA spellings the decoder's tables lack entirely.
|
||||||
|
"VFMADD132SD": true,
|
||||||
|
"VFMADD132SS": true,
|
||||||
|
"VFMADD213SS": true,
|
||||||
|
"VFMADD231SD": true,
|
||||||
|
"VFMADD231SS": true,
|
||||||
|
"VFMSUB132SD": true,
|
||||||
|
"VFMSUB132SS": true,
|
||||||
|
"VFMSUB213SD": true,
|
||||||
|
"VFMSUB213SS": true,
|
||||||
|
"VFMSUB231SD": true,
|
||||||
|
"VFMSUB231SS": true,
|
||||||
|
"VFNMADD132SD": true,
|
||||||
|
"VFNMADD132SS": true,
|
||||||
|
"VFNMADD213SD": true,
|
||||||
|
"VFNMADD213SS": true,
|
||||||
|
"VFNMADD231SS": true,
|
||||||
|
"VFNMSUB132SD": true,
|
||||||
|
"VFNMSUB132SS": true,
|
||||||
|
"VFNMSUB213SD": true,
|
||||||
|
"VFNMSUB213SS": true,
|
||||||
|
"VFNMSUB231SD": true,
|
||||||
|
"VFNMSUB231SS": true,
|
||||||
|
// The BMI1 unary bit ops the decoder's AVX tables lack.
|
||||||
|
"BLSIL": true,
|
||||||
|
"BLSIQ": true,
|
||||||
|
"BLSMSKL": true,
|
||||||
|
"BLSMSKQ": true,
|
||||||
|
"BLSRL": true,
|
||||||
|
"BLSRQ": true,
|
||||||
|
// The BMI2 bit ops whose W1/LZ rows the decoder misses.
|
||||||
|
"BEXTRL": true,
|
||||||
|
"BEXTRQ": true,
|
||||||
|
"BZHIL": true,
|
||||||
|
"BZHIQ": true,
|
||||||
|
"PDEPL": true,
|
||||||
|
"PDEPQ": true,
|
||||||
|
"PEXTL": true,
|
||||||
|
"PEXTQ": true,
|
||||||
|
"SARXL": true,
|
||||||
|
"SARXQ": true,
|
||||||
|
"SHLXL": true,
|
||||||
|
"SHLXQ": true,
|
||||||
|
"SHRXL": true,
|
||||||
|
"SHRXQ": true,
|
||||||
|
}
|
||||||
|
|
||||||
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
|
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
|
||||||
// instruction and verifies it round-trips through the x86 decoder to the same
|
// instruction and verifies it round-trips through the x86 decoder to the same
|
||||||
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
|
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
|
||||||
@@ -37,8 +96,15 @@ func TestVexNDS3(t *testing.T) {
|
|||||||
t.Errorf("%s: Encode: %v", mnem, err)
|
t.Errorf("%s: Encode: %v", mnem, err)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
// The x86 decoder's table lacks a handful of rows the Go assembler
|
||||||
|
// emits (the scalar 213/231 FMA spellings among them); those are
|
||||||
|
// pinned byte for byte against go tool asm in TestVexGroundTruth
|
||||||
|
// instead of round-tripped here.
|
||||||
inst, err := x86asm.Decode(code, 64)
|
inst, err := x86asm.Decode(code, 64)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
if strings.Contains(err.Error(), "unrecognized instruction") && x86asmUnrecognised[mnem] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
@@ -173,7 +239,7 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""},
|
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""},
|
||||||
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""},
|
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""},
|
||||||
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""},
|
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""},
|
||||||
// Floating point (packed and scalar) and FMA — same NDS form, the pp
|
// Floating point (packed and scalar) and FMA; same NDS form, the pp
|
||||||
// bits and map select the operation.
|
// bits and map select the operation.
|
||||||
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""},
|
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""},
|
||||||
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""},
|
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""},
|
||||||
@@ -184,6 +250,50 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
|
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
|
||||||
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
|
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
|
||||||
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
|
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
|
||||||
|
{"VFMADD213SD X0,X1,X2", "VFMADD213SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1a9d0", ""},
|
||||||
|
{"VFNMADD231SD X0,X1,X2", "VFNMADD231SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1bdd0", ""},
|
||||||
|
// Packed single XOR and byte compare (NDS form).
|
||||||
|
{"VXORPS Y0,Y1,Y2", "VXORPS", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f457d0", ""},
|
||||||
|
{"VPCMPEQB Y0,Y1,Y2", "VPCMPEQB", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f574d0", ""},
|
||||||
|
// Octa byte shifts (vvvv carries the destination).
|
||||||
|
{"VPSLLDQ $2,X0,X1", "VPSLLDQ", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "c5f173f802", ""},
|
||||||
|
{"VPSRLDQ $2,Y0,Y1", "VPSRLDQ", []Operand{Imm(2), vreg(t, "Y0"), vreg(t, "Y1")}, "c5f573d802", ""},
|
||||||
|
// Two-source shuffle, blend and carry-less multiply (NDS + imm8).
|
||||||
|
{"VPERM2F128 $3,Y0,Y1,Y2", "VPERM2F128", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37506d003", ""},
|
||||||
|
{"VPBLENDD $3,X0,X1,X2", "VPBLENDD", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37102d003", ""},
|
||||||
|
{"VPBLENDD $3,Y0,Y1,Y2", "VPBLENDD", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37502d003", ""},
|
||||||
|
{"VPCLMULQDQ $0,X0,X1,X2", "VPCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37144d000", ""},
|
||||||
|
{"VGF2P8AFFINEQB $0,X0,X1,X2", "VGF2P8AFFINEQB", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e3f1ced000", ""},
|
||||||
|
// Two-operand test and the non-temporal and broadcast stores.
|
||||||
|
{"VPTEST X0,X1", "VPTEST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c4e27917c8", ""},
|
||||||
|
{"VPTEST Y0,Y1", "VPTEST", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c4e27d17c8", ""},
|
||||||
|
{"VMOVNTDQ Y0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "Y0"), Ptr(AX, 0, 32)}, "c5fde700", ""},
|
||||||
|
{"VMOVNTDQ X0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "c5f9e700", ""},
|
||||||
|
{"VBROADCASTI128 (AX),Y1", "VBROADCASTI128", []Operand{Ptr(AX, 0, 16), vreg(t, "Y1")}, "c4e27d5a08", ""},
|
||||||
|
// Aligned integer move and the full zeroing form.
|
||||||
|
{"VMOVDQA X0,X1", "VMOVDQA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c5f97fc1", ""},
|
||||||
|
{"VMOVDQA (AX),X1", "VMOVDQA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "c5f96f08", ""},
|
||||||
|
{"VMOVDQA Y0,Y1", "VMOVDQA", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c5fd7fc1", ""},
|
||||||
|
{"VZEROALL", "VZEROALL", []Operand{}, "c5fc77", ""},
|
||||||
|
// BMI1/BMI2 general-register VEX forms.
|
||||||
|
{"ANDNL AX,BX,CX", "ANDNL", []Operand{AX, BX, CX}, "c4e260f2c8", ""},
|
||||||
|
{"ANDNQ AX,BX,CX", "ANDNQ", []Operand{AX, BX, CX}, "c4e2e0f2c8", ""},
|
||||||
|
{"MULXL AX,BX,CX", "MULXL", []Operand{AX, BX, CX}, "c4e263f6c8", ""},
|
||||||
|
{"MULXQ AX,BX,CX", "MULXQ", []Operand{AX, BX, CX}, "c4e2e3f6c8", ""},
|
||||||
|
{"RORXL $3,AX,CX", "RORXL", []Operand{Imm(3), AX, CX}, "c4e37bf0c803", ""},
|
||||||
|
{"RORXQ $3,AX,CX", "RORXQ", []Operand{Imm(3), AX, CX}, "c4e3fbf0c803", ""},
|
||||||
|
// BMI2 variable shifts and bit ops (three general registers).
|
||||||
|
{"SHLXL AX,CX,R15", "SHLXL", []Operand{AX, CX, vreg(t, "R15")}, "c46279f7f9", ""},
|
||||||
|
{"SHRXQ R8,DX,AX", "SHRXQ", []Operand{vreg(t, "R8"), DX, AX}, "c4e2bbf7c2", ""},
|
||||||
|
{"SARXQ AX,DX,R9", "SARXQ", []Operand{AX, DX, vreg(t, "R9")}, "c462faf7ca", ""},
|
||||||
|
{"BEXTRL AX,CX,R15", "BEXTRL", []Operand{AX, CX, vreg(t, "R15")}, "c46278f7f9", ""},
|
||||||
|
{"BZHIQ AX,CX,R15", "BZHIQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f8f5f9", ""},
|
||||||
|
{"PDEPQ AX,CX,R15", "PDEPQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f3f5f8", ""},
|
||||||
|
{"PEXTQ AX,CX,R15", "PEXTQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f2f5f8", ""},
|
||||||
|
// BMI1 unary bit ops (src, dst: /digit in ModRM.reg, dst in vvvv).
|
||||||
|
{"BLSIL AX,CX", "BLSIL", []Operand{AX, CX}, "c4e270f3d8", ""},
|
||||||
|
{"BLSRQ AX,CX", "BLSRQ", []Operand{AX, CX}, "c4e2f0f3c8", ""},
|
||||||
|
{"BLSMSKQ AX,CX", "BLSMSKQ", []Operand{AX, CX}, "c4e2f0f3d0", ""},
|
||||||
// Two-operand reg/rm form (v̄vvv must be 1111).
|
// Two-operand reg/rm form (v̄vvv must be 1111).
|
||||||
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
|
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
|
||||||
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
|
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
|
||||||
@@ -217,7 +327,7 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""},
|
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""},
|
||||||
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""},
|
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""},
|
||||||
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""},
|
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""},
|
||||||
// Moves — each direction picks its own opcode and VEX.W.
|
// Moves; each direction picks its own opcode and VEX.W.
|
||||||
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""},
|
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""},
|
||||||
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""},
|
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""},
|
||||||
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""},
|
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""},
|
||||||
@@ -234,7 +344,7 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""},
|
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""},
|
||||||
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""},
|
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""},
|
||||||
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""},
|
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""},
|
||||||
// Packed double arithmetic and unpack — the NDS form, the opcode
|
// Packed double arithmetic and unpack; the NDS form, the opcode
|
||||||
// selects the operation.
|
// selects the operation.
|
||||||
{"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""},
|
{"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""},
|
||||||
{"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""},
|
{"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""},
|
||||||
@@ -255,12 +365,12 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
{"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""},
|
{"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""},
|
||||||
{"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""},
|
{"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""},
|
||||||
{"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""},
|
{"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""},
|
||||||
// VMOVDDUP — duplicate the low double (reg=dst, rm=src, F2 pp).
|
// VMOVDDUP; duplicate the low double (reg=dst, rm=src, F2 pp).
|
||||||
{"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""},
|
{"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""},
|
||||||
{"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""},
|
{"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""},
|
||||||
{"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""},
|
{"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""},
|
||||||
// Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3
|
// Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3
|
||||||
// prefix — see the table comment), DQ→PD.
|
// prefix; see the table comment), DQ→PD.
|
||||||
{"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""},
|
{"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""},
|
||||||
{"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""},
|
{"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""},
|
||||||
{"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""},
|
{"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""},
|
||||||
@@ -287,6 +397,12 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
inst, err := x86asm.Decode(code, 64)
|
inst, err := x86asm.Decode(code, 64)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
// The decoder's AVX/BMI table lacks a few rows the Go
|
||||||
|
// assembler emits (the GPR VEX forms and the scalar FMA
|
||||||
|
// spellings); their bytes are the ground truth here.
|
||||||
|
if x86asmUnrecognised[c.mnem] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
|
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|||||||
+19
-8
@@ -8,7 +8,7 @@
|
|||||||
// to the arch package; the AST records syntax only.
|
// to the arch package; the AST records syntax only.
|
||||||
package ast
|
package ast
|
||||||
|
|
||||||
import "sourcedock.dev/petrbalvin/gasm-devkit/token"
|
import "sourcedock.dev/petrbalvin/gasm-sdk/token"
|
||||||
|
|
||||||
// File is the parsed representation of one .s source file.
|
// File is the parsed representation of one .s source file.
|
||||||
type File struct {
|
type File struct {
|
||||||
@@ -114,6 +114,7 @@ type Symbol struct {
|
|||||||
Pkg string // package prefix before the middle dot ("" = current package)
|
Pkg string // package prefix before the middle dot ("" = current package)
|
||||||
Name string // identifier without the middle dot or <>
|
Name string // identifier without the middle dot or <>
|
||||||
Static bool // the <> marker is present
|
Static bool // the <> marker is present
|
||||||
|
ABI string // the <NAME> ABI marker, e.g. ABIInternal ("" when absent)
|
||||||
Pseudo string // FP, SP, SB or PC ("" for a bare name)
|
Pseudo string // FP, SP, SB or PC ("" for a bare name)
|
||||||
Offset int64
|
Offset int64
|
||||||
HasOff bool
|
HasOff bool
|
||||||
@@ -150,11 +151,21 @@ type Immediate struct {
|
|||||||
// Address is a non-immediate operand: a register, a memory reference, a symbol
|
// Address is a non-immediate operand: a register, a memory reference, a symbol
|
||||||
// reference or a label. Fields are populated best-effort from the syntax.
|
// reference or a label. Fields are populated best-effort from the syntax.
|
||||||
type Address struct {
|
type Address struct {
|
||||||
Sym *Symbol // name reference (bare ident, or name+off(pseudo))
|
Sym *Symbol // name reference (bare ident, or name+off(pseudo))
|
||||||
Base string // base register, from (base)
|
Base string // base register, from (base)
|
||||||
Index string // index register, from (index*scale)
|
Index string // index register, from (index*scale)
|
||||||
Scale int // index scale; 0 when absent
|
Scale int // index scale; 0 when absent
|
||||||
Offset int64 // leading displacement, from off(base)
|
Offset int64 // leading displacement, from off(base)
|
||||||
HasOff bool // a leading displacement is present
|
HasOff bool // a leading displacement is present
|
||||||
Shift string // verbatim arm64 shift suffix, e.g. "<<2"
|
Shift string // verbatim arm64 shift suffix, e.g. "<< 2"
|
||||||
|
Range *RegRange // bracketed register range; nil for every other form
|
||||||
|
}
|
||||||
|
|
||||||
|
// RegRange is a bracketed register range, [Z0-Z3]: the amd64 spelling of
|
||||||
|
// the four-register source of the 4FMAPS/4VNNIW families. Lo and Hi carry
|
||||||
|
// the verbatim register spellings; the range is inclusive at both ends.
|
||||||
|
type RegRange struct {
|
||||||
|
Lo string
|
||||||
|
Hi string
|
||||||
|
Pos token.Position
|
||||||
}
|
}
|
||||||
|
|||||||
+1
-1
@@ -6,7 +6,7 @@ package ast
|
|||||||
import (
|
import (
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
"sourcedock.dev/petrbalvin/gasm-sdk/token"
|
||||||
)
|
)
|
||||||
|
|
||||||
func pos(line, col int) token.Position { return token.Position{Line: line, Column: col} }
|
func pos(line, col int) token.Position { return token.Position{Line: line, Column: col} }
|
||||||
|
|||||||
@@ -0,0 +1,367 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"go/ast"
|
||||||
|
"go/build"
|
||||||
|
"go/constant"
|
||||||
|
"go/parser"
|
||||||
|
"go/token"
|
||||||
|
"go/types"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"regexp"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
|
)
|
||||||
|
|
||||||
|
// go_asm.h is the header the Go compiler writes for every package that
|
||||||
|
// carries assembly (the compiler's -asmhdr output): "#define const_NAME
|
||||||
|
// value" for each package constant, and for each named struct type
|
||||||
|
// "#define TYPE__size size" plus one "#define TYPE_field offset" per field.
|
||||||
|
// GOROOT assembly includes it, and a standalone assembler has no compiler
|
||||||
|
// to have produced it, so gasm generates the equivalent itself: the package
|
||||||
|
// the .s file lives in is parsed and type-checked here, with the target
|
||||||
|
// architecture's own sizes, and the same defines are written out. The
|
||||||
|
// type-checking GOOS is selected by the caller: a GOOS-specific file
|
||||||
|
// (sys_darwin_arm64.s) needs its platform's defines, which a header from
|
||||||
|
// the ambient GOOS silently omits.
|
||||||
|
//
|
||||||
|
// The emitter mirrors cmd/compile's dumpasmhdr exactly: constants come out
|
||||||
|
// as "const_NAME", struct entries as "NAME__size" followed by the fields in
|
||||||
|
// declaration order, blank names are skipped, and float and complex
|
||||||
|
// constants are omitted (the assembler carries integers, bools and strings
|
||||||
|
// only). Aliases to structs are emitted, generic types are not: they have
|
||||||
|
// no fixed size. A define the assembly references but this header does not
|
||||||
|
// carry surfaces later as the assembler's own "undefined" diagnostic naming
|
||||||
|
// the define, which is the honest failure.
|
||||||
|
|
||||||
|
// goAsmInclude matches the #include "go_asm.h" directive, tolerant of
|
||||||
|
// whitespace, so the wiring knows which files need a generated header
|
||||||
|
// before the preprocessor runs and would report the header as missing.
|
||||||
|
var goAsmInclude = regexp.MustCompile(`(?m)^\s*#\s*include\s+"go_asm\.h"`)
|
||||||
|
|
||||||
|
// needsGoAsmHeader reports whether src includes go_asm.h.
|
||||||
|
func needsGoAsmHeader(src string) bool {
|
||||||
|
return goAsmInclude.MatchString(src)
|
||||||
|
}
|
||||||
|
|
||||||
|
// goAsmHeaderResolved reports whether the include of go_asm.h from a file in
|
||||||
|
// asmDir already resolves: to a header in the package directory itself, or
|
||||||
|
// in one of the -I directories, the way the preprocessor searches. Only an
|
||||||
|
// unresolved include is generated for; a header someone placed by hand is
|
||||||
|
// the tool the author chose, and it also wins the preprocessor's own search
|
||||||
|
// order, so generating a second copy would be dead weight at best.
|
||||||
|
func goAsmHeaderResolved(asmDir string, dirs []string) bool {
|
||||||
|
candidates := []string{filepath.Join(asmDir, "go_asm.h")}
|
||||||
|
for _, d := range dirs {
|
||||||
|
candidates = append(candidates, filepath.Join(d, "go_asm.h"))
|
||||||
|
}
|
||||||
|
for _, candidate := range candidates {
|
||||||
|
if st, err := os.Stat(candidate); err == nil && !st.IsDir() {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// generateGoAsmHeader type-checks the Go package in pkgDir for goos and
|
||||||
|
// goarch, writes its go_asm.h equivalent into dir, and returns dir. An
|
||||||
|
// empty goos means the ambient one. The caller owns the directory and its
|
||||||
|
// removal.
|
||||||
|
func generateGoAsmHeader(pkgDir, goos, goarch, dir string) (string, error) {
|
||||||
|
if goos == "" {
|
||||||
|
goos = build.Default.GOOS
|
||||||
|
}
|
||||||
|
imp := newSourceImporter(goos, goarch)
|
||||||
|
if imp.sizes == nil {
|
||||||
|
return "", fmt.Errorf("go_asm.h: unknown GOARCH %q", goarch)
|
||||||
|
}
|
||||||
|
bp, err := imp.ctxt.ImportDir(pkgDir, 0)
|
||||||
|
if err != nil {
|
||||||
|
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %w", goarch, pkgDir, err)
|
||||||
|
}
|
||||||
|
files, errs := imp.parse(bp)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %s", goarch, pkgDir, errorList(errs))
|
||||||
|
}
|
||||||
|
_, info, errs := imp.checkPackage(bp, files)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: package does not type-check: %s", goarch, pkgDir, errorList(errs))
|
||||||
|
}
|
||||||
|
|
||||||
|
var b strings.Builder
|
||||||
|
fmt.Fprintf(&b, "// generated by gasm from package %s (GOOS %s, GOARCH %s)\n\n", bp.Name, goos, goarch)
|
||||||
|
// Files in the build's own order and declarations in source order: the
|
||||||
|
// same walk the compiler's reader makes, so the header reads the same
|
||||||
|
// way the toolchain's does. Order carries no meaning to the assembler
|
||||||
|
// (defines form a table), only to a human diffing against one.
|
||||||
|
for _, f := range files {
|
||||||
|
for _, decl := range f.Decls {
|
||||||
|
gd, ok := decl.(*ast.GenDecl)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for _, spec := range gd.Specs {
|
||||||
|
switch gd.Tok {
|
||||||
|
case token.CONST:
|
||||||
|
vs, ok := spec.(*ast.ValueSpec)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for _, name := range vs.Names {
|
||||||
|
emitConst(&b, info.Defs[name], name.Name)
|
||||||
|
}
|
||||||
|
case token.TYPE:
|
||||||
|
ts, ok := spec.(*ast.TypeSpec)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
emitStruct(&b, imp.sizes, info.Defs[ts.Name], ts.Name.Name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||||
|
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %w", goarch, pkgDir, err)
|
||||||
|
}
|
||||||
|
out := filepath.Join(dir, "go_asm.h")
|
||||||
|
if err := os.WriteFile(out, []byte(b.String()), 0o644); err != nil {
|
||||||
|
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %w", goarch, pkgDir, err)
|
||||||
|
}
|
||||||
|
return dir, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// emitConst writes one const define, skipping what the toolchain skips:
|
||||||
|
// blank names, and float and complex values the assembler has no syntax for.
|
||||||
|
func emitConst(b *strings.Builder, obj types.Object, name string) {
|
||||||
|
c, ok := obj.(*types.Const)
|
||||||
|
if !ok || name == "_" {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
switch c.Val().Kind() {
|
||||||
|
case constant.Float, constant.Complex, constant.Unknown:
|
||||||
|
return
|
||||||
|
}
|
||||||
|
fmt.Fprintf(b, "#define const_%s %s\n", name, c.Val().ExactString())
|
||||||
|
}
|
||||||
|
|
||||||
|
// emitStruct writes one named struct type's size and field offsets,
|
||||||
|
// skipping what the toolchain skips: blank names, non-struct types, and
|
||||||
|
// generic types, whose size depends on their instantiation.
|
||||||
|
func emitStruct(b *strings.Builder, sizes types.Sizes, obj types.Object, name string) {
|
||||||
|
tn, ok := obj.(*types.TypeName)
|
||||||
|
if !ok || name == "_" {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
t := types.Unalias(tn.Type())
|
||||||
|
// Generic types are spelled *types.Named with a type-parameter list;
|
||||||
|
// a plain struct type or an instantiated one carries none.
|
||||||
|
if named, ok := t.(*types.Named); ok && named.TypeParams().Len() > 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
st, ok := t.Underlying().(*types.Struct)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
fmt.Fprintf(b, "#define %s__size %d\n", name, sizes.Sizeof(t))
|
||||||
|
fields := make([]*types.Var, st.NumFields())
|
||||||
|
for i := range st.NumFields() {
|
||||||
|
fields[i] = st.Field(i)
|
||||||
|
}
|
||||||
|
for i, off := range sizes.Offsetsof(fields) {
|
||||||
|
fld := fields[i]
|
||||||
|
if fld.Name() == "_" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
fmt.Fprintf(b, "#define %s_%s %d\n", name, fld.Name(), off)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// errorList renders at most three errors, enough to say what is wrong
|
||||||
|
// without burying the diagnostic the caller actually reads.
|
||||||
|
func errorList(errs []error) string {
|
||||||
|
if len(errs) > 3 {
|
||||||
|
errs = errs[:3]
|
||||||
|
}
|
||||||
|
msgs := make([]string, len(errs))
|
||||||
|
for i, err := range errs {
|
||||||
|
msgs[i] = err.Error()
|
||||||
|
}
|
||||||
|
return strings.Join(msgs, "; ")
|
||||||
|
}
|
||||||
|
|
||||||
|
// sourceImporter type-checks imported packages from source with the target
|
||||||
|
// architecture's sizes. go/importer's "source" importer pins the host
|
||||||
|
// GOARCH, which would lay out imported types (internal/cpu, internal/abi)
|
||||||
|
// for the wrong target on a cross-architecture header, so the recursion is
|
||||||
|
// carried here with one build context and one sizes instance per
|
||||||
|
// architecture.
|
||||||
|
type sourceImporter struct {
|
||||||
|
fset *token.FileSet
|
||||||
|
ctxt *build.Context
|
||||||
|
sizes types.Sizes
|
||||||
|
pkgs map[string]*types.Package
|
||||||
|
}
|
||||||
|
|
||||||
|
// newSourceImporter returns the importer for one target GOOS and GOARCH.
|
||||||
|
// Cgo is disabled so the file set is deterministic and independent of the
|
||||||
|
// host's C toolchain: cgo-tagged files drop out of the build exactly as
|
||||||
|
// they do from a CGO_ENABLED=0 build, whose assembly is what gasm targets.
|
||||||
|
func newSourceImporter(goos, goarch string) *sourceImporter {
|
||||||
|
ctxt := new(build.Context)
|
||||||
|
*ctxt = build.Default
|
||||||
|
ctxt.GOOS = goos
|
||||||
|
ctxt.GOARCH = goarch
|
||||||
|
ctxt.CgoEnabled = false
|
||||||
|
return &sourceImporter{
|
||||||
|
fset: token.NewFileSet(),
|
||||||
|
ctxt: ctxt,
|
||||||
|
sizes: types.SizesFor("gc", goarch),
|
||||||
|
pkgs: map[string]*types.Package{},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Import type-checks one imported package and memoises it. "unsafe" must
|
||||||
|
// resolve to go/types' own package, never to the source in GOROOT/src/unsafe:
|
||||||
|
// the source declares Sizeof and Offsetof as ordinary functions over
|
||||||
|
// ArbitraryType, and checking against that signature rejects half the
|
||||||
|
// unsafe arithmetic the gc compiler accepts, which is exactly the divergence
|
||||||
|
// srcimporter guards against the same way.
|
||||||
|
func (im *sourceImporter) Import(path string) (*types.Package, error) {
|
||||||
|
if path == "unsafe" {
|
||||||
|
return types.Unsafe, nil
|
||||||
|
}
|
||||||
|
if p, ok := im.pkgs[path]; ok {
|
||||||
|
return p, nil
|
||||||
|
}
|
||||||
|
bp, err := im.ctxt.Import(path, "", 0)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
files, errs := im.parse(bp)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
return nil, errors.New(errorList(errs))
|
||||||
|
}
|
||||||
|
pkg, _, _ := im.checkPackage(bp, files)
|
||||||
|
im.pkgs[path] = pkg
|
||||||
|
return pkg, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// parse reads the build package's Go files. Import-level failures (no Go
|
||||||
|
// files for the target, unreadable files) come back as errors, and the
|
||||||
|
// type-check decides the rest.
|
||||||
|
func (im *sourceImporter) parse(bp *build.Package) ([]*ast.File, []error) {
|
||||||
|
if len(bp.GoFiles) == 0 {
|
||||||
|
return nil, []error{fmt.Errorf("no Go source files for GOOS=%s GOARCH=%s", im.ctxt.GOOS, im.ctxt.GOARCH)}
|
||||||
|
}
|
||||||
|
var (
|
||||||
|
files []*ast.File
|
||||||
|
errs []error
|
||||||
|
)
|
||||||
|
for _, name := range bp.GoFiles {
|
||||||
|
f, err := parser.ParseFile(im.fset, filepath.Join(bp.Dir, name), nil, parser.SkipObjectResolution)
|
||||||
|
if err != nil {
|
||||||
|
errs = append(errs, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
files = append(files, f)
|
||||||
|
}
|
||||||
|
return files, errs
|
||||||
|
}
|
||||||
|
|
||||||
|
// checkPackage type-checks one package's files with the importer's sizes,
|
||||||
|
// recording every error: a header from a package that does not type-check
|
||||||
|
// could silently mis-state an offset, so the caller refuses the header
|
||||||
|
// rather than trusting it. The returned Defs map backs the root package's
|
||||||
|
// emission walk; imports only need the checked package itself.
|
||||||
|
func (im *sourceImporter) checkPackage(bp *build.Package, files []*ast.File) (*types.Package, *types.Info, []error) {
|
||||||
|
var errs []error
|
||||||
|
conf := &types.Config{
|
||||||
|
Importer: im,
|
||||||
|
Sizes: im.sizes,
|
||||||
|
Error: func(err error) { errs = append(errs, err) },
|
||||||
|
}
|
||||||
|
info := &types.Info{Defs: map[*ast.Ident]types.Object{}}
|
||||||
|
pkg, _ := conf.Check(bp.ImportPath, im.fset, files, info)
|
||||||
|
return pkg, info, errs
|
||||||
|
}
|
||||||
|
|
||||||
|
// asmhdrCache generates one go_asm.h per package directory and target
|
||||||
|
// architecture under one temp root, for callers that assemble many files
|
||||||
|
// (the corpus audit). Failures are cached too: a package that does not
|
||||||
|
// type-check must not be re-checked once per file.
|
||||||
|
type asmhdrCache struct {
|
||||||
|
root string
|
||||||
|
dirs map[string]string // "pkgDir\x00goos\x00goarch" -> directory holding go_asm.h
|
||||||
|
errs map[string]error
|
||||||
|
}
|
||||||
|
|
||||||
|
func newAsmhdrCache() (*asmhdrCache, error) {
|
||||||
|
root, err := os.MkdirTemp("", "gasm-asmhdr")
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return &asmhdrCache{root: root, dirs: map[string]string{}, errs: map[string]error{}}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// dirFor returns the directory holding the generated go_asm.h for pkgDir
|
||||||
|
// under goos and goarch, generating it on first use. An empty goos means
|
||||||
|
// the ambient one, resolved here so that one package cannot generate twice
|
||||||
|
// under an explicit and an implicit spelling of the same GOOS.
|
||||||
|
func (c *asmhdrCache) dirFor(pkgDir, goos, goarch string) (string, error) {
|
||||||
|
if goos == "" {
|
||||||
|
goos = build.Default.GOOS
|
||||||
|
}
|
||||||
|
key := pkgDir + "\x00" + goos + "\x00" + goarch
|
||||||
|
if dir, ok := c.dirs[key]; ok {
|
||||||
|
return dir, nil
|
||||||
|
}
|
||||||
|
if err, ok := c.errs[key]; ok {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
dir := filepath.Join(c.root, fmt.Sprintf("h%d_%s_%s", len(c.dirs), goos, goarch))
|
||||||
|
if _, err := generateGoAsmHeader(pkgDir, goos, goarch, dir); err != nil {
|
||||||
|
c.errs[key] = err
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
c.dirs[key] = dir
|
||||||
|
return dir, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// close removes the temp root.
|
||||||
|
func (c *asmhdrCache) close() { os.RemoveAll(c.root) }
|
||||||
|
|
||||||
|
// ensureGoAsmHeader prepares the include directory a file that includes
|
||||||
|
// go_asm.h needs: the generated header for the package in path's directory,
|
||||||
|
// for the file's target GOOS and architecture. It reports a usage error
|
||||||
|
// when the architecture cannot be determined, and passes through the
|
||||||
|
// generator's diagnostics, which name the package.
|
||||||
|
func ensureGoAsmHeader(path string, target arch.Arch, goos string, cache *asmhdrCache) (string, func(), error) {
|
||||||
|
if path == "-" {
|
||||||
|
return "", nil, errors.New("cannot generate go_asm.h for standard input (no package directory)")
|
||||||
|
}
|
||||||
|
if target == arch.Unknown {
|
||||||
|
return "", nil, errors.New("a file that includes go_asm.h needs a target architecture: name the file _<arch>.s or pass -GOARCH")
|
||||||
|
}
|
||||||
|
if cache != nil {
|
||||||
|
dir, err := cache.dirFor(filepath.Dir(path), goos, goarchName(target))
|
||||||
|
return dir, func() {}, err
|
||||||
|
}
|
||||||
|
root, err := os.MkdirTemp("", "gasm-asmhdr")
|
||||||
|
if err != nil {
|
||||||
|
return "", nil, err
|
||||||
|
}
|
||||||
|
dir, err := generateGoAsmHeader(filepath.Dir(path), goos, goarchName(target), root)
|
||||||
|
if err != nil {
|
||||||
|
os.RemoveAll(root)
|
||||||
|
return "", nil, err
|
||||||
|
}
|
||||||
|
return dir, func() { os.RemoveAll(root) }, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,476 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// writePkg lays out a minimal Go package in a temp directory.
|
||||||
|
func writePkg(t *testing.T, files map[string]string) string {
|
||||||
|
t.Helper()
|
||||||
|
dir := t.TempDir()
|
||||||
|
for name, src := range files {
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return dir
|
||||||
|
}
|
||||||
|
|
||||||
|
// generateFor generates the header for dir and returns its text. An empty
|
||||||
|
// goos means the ambient one.
|
||||||
|
func generateFor(t *testing.T, dir, goos, goarch string) string {
|
||||||
|
t.Helper()
|
||||||
|
hdrDir, err := generateGoAsmHeader(dir, goos, goarch, t.TempDir())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("generateGoAsmHeader(%q, %s, %s): %v", dir, goos, goarch, err)
|
||||||
|
}
|
||||||
|
b, err := os.ReadFile(filepath.Join(hdrDir, "go_asm.h"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return string(b)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestGenerateGoAsmHeaderShape(t *testing.T) {
|
||||||
|
dir := writePkg(t, map[string]string{"sample.go": `package sample
|
||||||
|
|
||||||
|
const bufSize = 1024
|
||||||
|
|
||||||
|
const (
|
||||||
|
a = iota * 8
|
||||||
|
b
|
||||||
|
c
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
strConst = "hello"
|
||||||
|
boolConst = true
|
||||||
|
floatConst = 1.5
|
||||||
|
_ = "the blank identifier is skipped"
|
||||||
|
)
|
||||||
|
|
||||||
|
const shift = 1 << 20
|
||||||
|
|
||||||
|
type reader struct {
|
||||||
|
r int64
|
||||||
|
w int64
|
||||||
|
_ [4]byte
|
||||||
|
name string
|
||||||
|
}
|
||||||
|
|
||||||
|
type scalar int
|
||||||
|
|
||||||
|
type aliased struct {
|
||||||
|
k uint32
|
||||||
|
v uint32
|
||||||
|
}
|
||||||
|
|
||||||
|
type alias = aliased
|
||||||
|
`})
|
||||||
|
hdr := generateFor(t, dir, "", "amd64")
|
||||||
|
want := []string{
|
||||||
|
"#define const_bufSize 1024",
|
||||||
|
// iota resolves through go/types, one define per name.
|
||||||
|
"#define const_a 0",
|
||||||
|
"#define const_b 8",
|
||||||
|
"#define const_c 16",
|
||||||
|
`#define const_strConst "hello"`,
|
||||||
|
"#define const_boolConst true",
|
||||||
|
// Floats are the toolchain's own skip, as are blank names.
|
||||||
|
"#define const_shift 1048576",
|
||||||
|
// The blank field still occupies its bytes: the pad after w runs to
|
||||||
|
// the string's 8-byte alignment.
|
||||||
|
"#define reader__size 40",
|
||||||
|
"#define reader_r 0",
|
||||||
|
"#define reader_w 8",
|
||||||
|
"#define reader_name 24",
|
||||||
|
// Non-struct named types carry no defines; aliases to structs do.
|
||||||
|
"#define aliased__size 8",
|
||||||
|
"#define aliased_k 0",
|
||||||
|
"#define aliased_v 4",
|
||||||
|
"#define alias__size 8",
|
||||||
|
"#define alias_k 0",
|
||||||
|
"#define alias_v 4",
|
||||||
|
}
|
||||||
|
for _, w := range want {
|
||||||
|
if !strings.Contains(hdr, w+"\n") {
|
||||||
|
t.Errorf("header misses %q\ngot:\n%s", w, hdr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, banned := range []string{"#define const_floatConst", "#define _ ", "#define scalar"} {
|
||||||
|
if strings.Contains(hdr, banned) {
|
||||||
|
t.Errorf("header must not carry %s\ngot:\n%s", banned, hdr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestGenerateGoAsmHeaderPerArch(t *testing.T) {
|
||||||
|
dir := writePkg(t, map[string]string{
|
||||||
|
"common.go": `package perarch
|
||||||
|
|
||||||
|
type layout struct {
|
||||||
|
a int32
|
||||||
|
p uintptr
|
||||||
|
}
|
||||||
|
`,
|
||||||
|
// The build-tagged file set is part of the contract: a per-arch
|
||||||
|
// package is exactly how internal/cpu declares its layouts.
|
||||||
|
"const_amd64.go": `//go:build amd64
|
||||||
|
|
||||||
|
package perarch
|
||||||
|
|
||||||
|
const flavour = 1
|
||||||
|
`,
|
||||||
|
"const_arm64.go": `//go:build arm64
|
||||||
|
|
||||||
|
package perarch
|
||||||
|
|
||||||
|
const flavour = 2
|
||||||
|
`,
|
||||||
|
})
|
||||||
|
amd64 := generateFor(t, dir, "", "amd64")
|
||||||
|
arm64 := generateFor(t, dir, "", "arm64")
|
||||||
|
if !strings.Contains(amd64, "#define const_flavour 1\n") {
|
||||||
|
t.Errorf("amd64 header misses const_flavour 1:\n%s", amd64)
|
||||||
|
}
|
||||||
|
if !strings.Contains(arm64, "#define const_flavour 2\n") {
|
||||||
|
t.Errorf("arm64 header misses const_flavour 2:\n%s", arm64)
|
||||||
|
}
|
||||||
|
if strings.Contains(arm64, "#define const_flavour 1\n") {
|
||||||
|
t.Errorf("arm64 header must not carry the amd64 file's value")
|
||||||
|
}
|
||||||
|
// SizesFor makes the layout the target's: uintptr is 4 bytes wide on
|
||||||
|
// 386 and 8 on amd64, which must move p and grow the struct.
|
||||||
|
if !strings.Contains(amd64, "#define layout__size 16\n") || !strings.Contains(amd64, "#define layout_p 8\n") {
|
||||||
|
t.Errorf("amd64 layout wrong:\n%s", amd64)
|
||||||
|
}
|
||||||
|
w386 := generateFor(t, dir, "", "386")
|
||||||
|
if !strings.Contains(w386, "#define layout__size 8\n") || !strings.Contains(w386, "#define layout_p 4\n") {
|
||||||
|
t.Errorf("386 layout wrong:\n%s", w386)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGenerateGoAsmHeaderGOOS pins the GOOS half of the target: only the
|
||||||
|
// platform's own files type-check into the header, which is why
|
||||||
|
// sys_darwin_arm64.s cannot assemble against a linux-generated one.
|
||||||
|
func TestGenerateGoAsmHeaderGOOS(t *testing.T) {
|
||||||
|
dir := writePkg(t, map[string]string{
|
||||||
|
"common.go": `package goosaware
|
||||||
|
|
||||||
|
type shared struct {
|
||||||
|
a int32
|
||||||
|
}
|
||||||
|
`,
|
||||||
|
"plat_darwin.go": `//go:build darwin
|
||||||
|
|
||||||
|
package goosaware
|
||||||
|
|
||||||
|
type platform struct {
|
||||||
|
trampoline_numer int64
|
||||||
|
}
|
||||||
|
`,
|
||||||
|
"plat_windows.go": `//go:build windows
|
||||||
|
|
||||||
|
package goosaware
|
||||||
|
|
||||||
|
type platform struct {
|
||||||
|
callbackArgs__size int32
|
||||||
|
}
|
||||||
|
`,
|
||||||
|
})
|
||||||
|
darwin := generateFor(t, dir, "darwin", "arm64")
|
||||||
|
if !strings.Contains(darwin, "#define platform__size 8\n") || !strings.Contains(darwin, "#define platform_trampoline_numer 0\n") {
|
||||||
|
t.Errorf("darwin header misses the darwin layout:\n%s", darwin)
|
||||||
|
}
|
||||||
|
if strings.Contains(darwin, "callbackArgs") {
|
||||||
|
t.Errorf("darwin header must not carry the windows layout:\n%s", darwin)
|
||||||
|
}
|
||||||
|
windows := generateFor(t, dir, "windows", "arm64")
|
||||||
|
if !strings.Contains(windows, "#define platform_callbackArgs__size 0\n") {
|
||||||
|
t.Errorf("windows header misses the windows layout:\n%s", windows)
|
||||||
|
}
|
||||||
|
if strings.Contains(windows, "trampoline_numer") {
|
||||||
|
t.Errorf("windows header must not carry the darwin layout:\n%s", windows)
|
||||||
|
}
|
||||||
|
// The ambient GOOS is neither of the two, so only shared's defines are
|
||||||
|
// emitted; the shared type keeps its layout there.
|
||||||
|
ambient := generateFor(t, dir, "", "arm64")
|
||||||
|
if !strings.Contains(ambient, "#define shared__size 4\n") {
|
||||||
|
t.Errorf("ambient header misses the shared layout:\n%s", ambient)
|
||||||
|
}
|
||||||
|
if strings.Contains(ambient, "#define platform_") {
|
||||||
|
t.Errorf("ambient header must not carry either platform layout:\n%s", ambient)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestGoosFromFilename(t *testing.T) {
|
||||||
|
for path, want := range map[string]string{
|
||||||
|
"/x/sys_darwin_arm64.s": "darwin",
|
||||||
|
"/x/sys_windows_arm64.s": "windows",
|
||||||
|
"/x/asm_linux_amd64.s": "linux",
|
||||||
|
"/x/rt0_darwin_arm64.s": "darwin",
|
||||||
|
"/x/vgetrandom_zos_s390x.s": "zos",
|
||||||
|
"/x/rt0_js_wasm.s": "js",
|
||||||
|
"/x/memmove_amd64.s": "",
|
||||||
|
"/x/vlop_arm.s": "",
|
||||||
|
"/x/stubs.s": "",
|
||||||
|
} {
|
||||||
|
if got := goosFromFilename(path); got != want {
|
||||||
|
t.Errorf("goosFromFilename(%q) = %q, want %q", path, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestGenerateGoAsmHeaderErrors(t *testing.T) {
|
||||||
|
t.Run("type error", func(t *testing.T) {
|
||||||
|
dir := writePkg(t, map[string]string{"bad.go": `package bad
|
||||||
|
|
||||||
|
const x = undefinedIdent
|
||||||
|
`})
|
||||||
|
_, err := generateGoAsmHeader(dir, "", "amd64", t.TempDir())
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("generation must fail for a package that does not type-check")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), dir) {
|
||||||
|
t.Errorf("error must name the package directory: %v", err)
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), "type-check") {
|
||||||
|
t.Errorf("error must say the package does not type-check: %v", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
t.Run("no go files", func(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
_, err := generateGoAsmHeader(dir, "", "amd64", t.TempDir())
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("generation must fail without Go files")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), dir) {
|
||||||
|
t.Errorf("error must name the package directory: %v", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestNeedsGoAsmHeader(t *testing.T) {
|
||||||
|
yes := "#include \"go_asm.h\"\n#include \"textflag.h\"\n"
|
||||||
|
no := "#include \"textflag.h\"\n#include \"funcdata.h\"\n"
|
||||||
|
if !needsGoAsmHeader(yes) {
|
||||||
|
t.Error("needsGoAsmHeader(missing on a go_asm.h include)")
|
||||||
|
}
|
||||||
|
if needsGoAsmHeader(no) {
|
||||||
|
t.Error("needsGoAsmHeader claims other headers need generation")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestGoAsmHeaderResolved(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
if goAsmHeaderResolved(dir, nil) {
|
||||||
|
t.Error("resolved with no header anywhere")
|
||||||
|
}
|
||||||
|
other := t.TempDir()
|
||||||
|
if goAsmHeaderResolved(dir, []string{other}) {
|
||||||
|
t.Error("resolved with an empty -I directory")
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "go_asm.h"), nil, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if !goAsmHeaderResolved(dir, nil) {
|
||||||
|
t.Error("not resolved with the header in the package directory")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestOtherGOOSFile(t *testing.T) {
|
||||||
|
for path, want := range map[string]bool{
|
||||||
|
"/x/sys_windows_amd64.s": true,
|
||||||
|
"/x/rt0_js_wasm.s": true,
|
||||||
|
"/x/sys_darwin_arm64.s": true,
|
||||||
|
"/x/sys_linux_amd64.s": false,
|
||||||
|
"/x/time_linux_amd64.s": false,
|
||||||
|
"/x/memmove_amd64.s": false,
|
||||||
|
"/x/generic.s": false,
|
||||||
|
} {
|
||||||
|
if got := otherGOOSFile(path); got != want {
|
||||||
|
t.Errorf("otherGOOSFile(%q) = %v, want %v", path, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRunCorpusAuditGoAsm covers the audit wiring end to end: a package
|
||||||
|
// beside its kernel, the kernel living off the generated defines, and the
|
||||||
|
// histogram recording a generation failure as its own reason.
|
||||||
|
func TestRunCorpusAuditGoAsm(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
write := func(name, src string) {
|
||||||
|
t.Helper()
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
write("pkg.go", `package corpus
|
||||||
|
|
||||||
|
const pageSize = 4096
|
||||||
|
|
||||||
|
type header struct {
|
||||||
|
magic uint64
|
||||||
|
flags uint64
|
||||||
|
}
|
||||||
|
`)
|
||||||
|
write("kern_amd64.s", "#include \"go_asm.h\"\nTEXT \xc2\xb7f(SB), NOSPLIT, $0-16\n\tMOVQ\t$const_pageSize, AX\n\tMOVQ\t$header__size, BX\n\tRET\n")
|
||||||
|
// The defines live in the file's own package; a kernel in a directory
|
||||||
|
// without Go files has no package to generate from.
|
||||||
|
if err := os.MkdirAll(filepath.Join(dir, "sub"), 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
write(filepath.Join("sub", "lonely_arm64.s"), "#include \"go_asm.h\"\nTEXT \xc2\xb7g(SB), NOSPLIT, $0-0\n\tRET\n")
|
||||||
|
|
||||||
|
stats, err := runCorpusAudit(dir, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("runCorpusAudit: %v", err)
|
||||||
|
}
|
||||||
|
get := func(name string) *corpusTally {
|
||||||
|
for i, tg := range stats.targets {
|
||||||
|
if tg.name == name {
|
||||||
|
return stats.tallies[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatalf("no tally for %s", name)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if a := get("amd64"); a.attempted != 1 || a.assembled != 1 {
|
||||||
|
t.Errorf("amd64 = %d/%d, want 1/1", a.assembled, a.attempted)
|
||||||
|
}
|
||||||
|
// lonely_arm64.s is an arm64 file whose package cannot be generated.
|
||||||
|
if a := get("arm64"); a.attempted != 1 || a.assembled != 0 {
|
||||||
|
t.Errorf("arm64 = %d/%d, want 0/1", a.assembled, a.attempted)
|
||||||
|
}
|
||||||
|
if r := get("arm64").reasons["go_asm.h generation failed"]; r != 1 {
|
||||||
|
t.Errorf("arm64 go_asm.h failure count = %d, want 1", r)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRunCorpusAuditGOOS covers the filename-derived GOOS end to end: a
|
||||||
|
// kernel whose name names darwin must have its header type-checked with
|
||||||
|
// GOOS=darwin, so the darwin-only constant it offsets with is defined. The
|
||||||
|
// operand mirrors sys_darwin_arm64.s's trampoline, where a missing define
|
||||||
|
// leaves an unexpanded symbol in the offset and fails.
|
||||||
|
func TestRunCorpusAuditGOOS(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
write := func(name, src string) {
|
||||||
|
t.Helper()
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
write("pkg.go", "package corpus\n")
|
||||||
|
write("plat_darwin.go", "//go:build darwin\n\npackage corpus\n\nconst trampolineNumer = 8\n")
|
||||||
|
write("kern_darwin_arm64.s", "#include \"go_asm.h\"\n"+
|
||||||
|
"GLOBL timebase<>(SB), NOPTR, $16\n"+
|
||||||
|
"TEXT \xc2\xb7g(SB), NOSPLIT, $0-0\n"+
|
||||||
|
"\tMOVD\ttimebase<>+const_trampolineNumer(SB), R0\n"+
|
||||||
|
"\tRET\n")
|
||||||
|
|
||||||
|
stats, err := runCorpusAudit(dir, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("runCorpusAudit: %v", err)
|
||||||
|
}
|
||||||
|
var arm *corpusTally
|
||||||
|
for i, tg := range stats.targets {
|
||||||
|
if tg.name == "arm64" {
|
||||||
|
arm = stats.tallies[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if arm == nil {
|
||||||
|
t.Fatal("no arm64 tally")
|
||||||
|
}
|
||||||
|
if arm.attempted != 1 || arm.assembled != 1 {
|
||||||
|
t.Errorf("arm64 = %d/%d, want 1/1; reasons: %v", arm.assembled, arm.attempted, arm.reasons)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRunCorpusAuditBuildConstraint covers the //go:build classification end
|
||||||
|
// to end: a generic-named file whose constraint admits one target is
|
||||||
|
// attempted there alone (cpu_x86.s on amd64), and a file whose constraint
|
||||||
|
// admits none of the four targets is never attempted (the msan and
|
||||||
|
// goexperiment trees).
|
||||||
|
func TestRunCorpusAuditBuildConstraint(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
write := func(name, src string) {
|
||||||
|
t.Helper()
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
write("x86.s", "//go:build 386 || amd64\n\nTEXT \xc2\xb7f(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
|
write("racey.s", "//go:build race\n\nTEXT \xc2\xb7r(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
|
write("plain.s", "TEXT \xc2\xb7p(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
|
|
||||||
|
stats, err := runCorpusAudit(dir, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("runCorpusAudit: %v", err)
|
||||||
|
}
|
||||||
|
tally := func(name string) *corpusTally {
|
||||||
|
for i, tg := range stats.targets {
|
||||||
|
if tg.name == name {
|
||||||
|
return stats.tallies[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatalf("no tally for %s", name)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if stats.narrowed != 1 || stats.excluded != 1 || stats.generic != 1 {
|
||||||
|
t.Errorf("buckets = narrowed %d, excluded %d, generic %d; want 1, 1, 1", stats.narrowed, stats.excluded, stats.generic)
|
||||||
|
}
|
||||||
|
if a := tally("amd64"); a.attempted != 2 || a.assembled != 2 {
|
||||||
|
t.Errorf("amd64 = %d/%d, want 2/2 (x86.s and plain.s)", a.assembled, a.attempted)
|
||||||
|
}
|
||||||
|
for _, name := range []string{"arm64", "riscv64", "loong64"} {
|
||||||
|
if a := tally(name); a.attempted != 1 || a.assembled != 1 {
|
||||||
|
t.Errorf("%s = %d/%d, want 1/1 (plain.s only)", name, a.assembled, a.attempted)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if stats.full != 2 {
|
||||||
|
t.Errorf("full = %d, want 2 (x86.s over its one target, plain.s over all four)", stats.full)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGenerateGoAsmHeaderRuntime pins the generator against the real thing:
|
||||||
|
// the runtime package of the ambient toolchain, whose header the toolchain's
|
||||||
|
// own -asmhdr output was sampled from. Skipped in short mode: it type-checks
|
||||||
|
// the whole package. The GOROOT comes from the go command itself, so the
|
||||||
|
// test follows whatever toolchain the host provides.
|
||||||
|
func TestGenerateGoAsmHeaderRuntime(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("type-checks the whole runtime package")
|
||||||
|
}
|
||||||
|
out, err := exec.Command("go", "env", "GOROOT").Output()
|
||||||
|
if err != nil {
|
||||||
|
t.Skipf("no Go toolchain: %v", err)
|
||||||
|
}
|
||||||
|
runtimeDir := filepath.Join(strings.TrimSpace(string(out)), "src", "runtime")
|
||||||
|
dir, err := generateGoAsmHeader(runtimeDir, "", "amd64", t.TempDir())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("generateGoAsmHeader(runtime): %v", err)
|
||||||
|
}
|
||||||
|
b, err := os.ReadFile(dir + "/go_asm.h")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
hdr := string(b)
|
||||||
|
for _, want := range []string{
|
||||||
|
"#define const_hashSize 8\n",
|
||||||
|
"#define const_avxSupported 1\n",
|
||||||
|
"#define const_pageSize 8192\n",
|
||||||
|
"#define g_stackguard0 16\n",
|
||||||
|
"#define m__size ",
|
||||||
|
} {
|
||||||
|
if !strings.Contains(hdr, want) {
|
||||||
|
t.Errorf("runtime header misses %q", want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
+616
-11
@@ -5,6 +5,8 @@ package main
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"go/build/constraint"
|
||||||
|
"maps"
|
||||||
"os"
|
"os"
|
||||||
"os/exec"
|
"os/exec"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
@@ -14,9 +16,9 @@ import (
|
|||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// cmdAuditInstructions cross-checks a gasm encoder against the Go toolchain's
|
// cmdAuditInstructions cross-checks a gasm encoder against the Go toolchain's
|
||||||
@@ -37,21 +39,41 @@ import (
|
|||||||
// construction and are excluded from the diff; the other architectures list
|
// construction and are excluded from the diff; the other architectures list
|
||||||
// their conditional branches outright.
|
// their conditional branches outright.
|
||||||
func cmdAuditInstructions(args []string) error {
|
func cmdAuditInstructions(args []string) error {
|
||||||
fs := newCommand("audit-instructions", "gasm audit-instructions [amd64|arm64|riscv64|loong64]", `
|
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [--list] [-I dir] [amd64|arm64|riscv64|loong64]", `
|
||||||
Compare the gasm encoder for the given architecture (default amd64) against
|
Compare the gasm encoder for the given architecture (default amd64) against
|
||||||
go tool asm and print the diff: superset encodings (gasm-only, shippable via
|
go tool asm and print the diff: superset encodings (gasm-only, shippable via
|
||||||
gasm asm --format goobj), known-but-unencodable names (the backlog) and go-
|
gasm asm --format goobj) and known-but-unencodable names (the backlog). The
|
||||||
only names (feature gaps). The Go side is probed black-box with a battery
|
Go side is probed black-box one bare mnemonic at a time, so the audit tracks
|
||||||
of bare mnemonics, so the audit tracks whatever toolchain `+"`go env GOROOT`"+`
|
whatever toolchain `+"`go env GOROOT`"+` provides; the gasm side answers from
|
||||||
provides.
|
the encoder table on amd64 and from trial assembly over a battery of operand
|
||||||
|
shapes elsewhere. Names go tool asm knows and gasm does not cannot be
|
||||||
|
enumerated by probing, because Go's table is visible only through names
|
||||||
|
already in the gasm table; the report closes with a note saying so.
|
||||||
|
|
||||||
|
With --corpus the audit changes shape: it assembles every .s file under the
|
||||||
|
given directory (default GOROOT/src) with the gasm encoder only, no
|
||||||
|
toolchain probing. A file whose name carries a recognisable _arch suffix is
|
||||||
|
attempted for that architecture; a file without one is attempted for all
|
||||||
|
four, exactly as a GOARCH build would compile it. The report gives the
|
||||||
|
per-architecture pass rates and the most common failure reasons, which drive
|
||||||
|
the encodability backlog by frequency rather than by table order. With
|
||||||
|
-list the report also prints every failing file with its reason, per
|
||||||
|
architecture.
|
||||||
`)
|
`)
|
||||||
|
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
|
||||||
|
list := fs.Bool("list", false, "with --corpus, list every failing file with its reason, per architecture")
|
||||||
|
var dirs includeDirs
|
||||||
|
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
|
||||||
if err := fs.Parse(args); err != nil {
|
if err := fs.Parse(args); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
if *corpus {
|
||||||
|
return cmdAuditCorpus(fs.Args(), dirs, *list)
|
||||||
|
}
|
||||||
archName := "amd64"
|
archName := "amd64"
|
||||||
switch n := len(fs.Args()); {
|
switch n := len(fs.Args()); {
|
||||||
case n > 1:
|
case n > 1:
|
||||||
return fmt.Errorf("audit-instructions takes at most one architecture argument")
|
return &usageError{fmt.Errorf("audit-instructions takes at most one architecture argument")}
|
||||||
case n == 1:
|
case n == 1:
|
||||||
archName = strings.ToLower(fs.Arg(0))
|
archName = strings.ToLower(fs.Arg(0))
|
||||||
}
|
}
|
||||||
@@ -97,7 +119,7 @@ provides.
|
|||||||
|
|
||||||
w := os.Stdout
|
w := os.Stdout
|
||||||
fmt.Fprintf(w, "gasm table (%s, families excluded): %d mnemonics\n", archName, len(names))
|
fmt.Fprintf(w, "gasm table (%s, families excluded): %d mnemonics\n", archName, len(names))
|
||||||
fmt.Fprintf(w, "gasm encodable: %d go tool asm recognized: %d\n", len(shared)+len(superset), countTrue(goKnown))
|
fmt.Fprintf(w, "gasm encodable: %d go tool asm recognised: %d\n", len(shared)+len(superset), countTrue(goKnown))
|
||||||
fmt.Fprintf(w, "shared: %d\n", len(shared))
|
fmt.Fprintf(w, "shared: %d\n", len(shared))
|
||||||
fmt.Fprintf(w, "\nSuperset encodings (gasm-only; ship via gasm asm --format goobj):\n")
|
fmt.Fprintf(w, "\nSuperset encodings (gasm-only; ship via gasm asm --format goobj):\n")
|
||||||
for _, n := range superset {
|
for _, n := range superset {
|
||||||
@@ -124,7 +146,7 @@ func auditArch(name string) (arch.Arch, error) {
|
|||||||
case "loong64", "loong":
|
case "loong64", "loong":
|
||||||
return arch.LOONG64, nil
|
return arch.LOONG64, nil
|
||||||
}
|
}
|
||||||
return arch.Unknown, fmt.Errorf("unknown architecture %q: want amd64, arm64, riscv64 or loong64", name)
|
return arch.Unknown, &usageError{fmt.Errorf("unknown architecture %q: want amd64, arm64, riscv64 or loong64", name)}
|
||||||
}
|
}
|
||||||
|
|
||||||
// goarchName maps an arch identifier onto its GOARCH spelling.
|
// goarchName maps an arch identifier onto its GOARCH spelling.
|
||||||
@@ -205,6 +227,13 @@ func probeGoAsm(goarch string, names []string) (map[string]bool, error) {
|
|||||||
cmd := exec.Command(asmBin, "-p", "probe", "-o", filepath.Join(dir, "probe.o"), probePath)
|
cmd := exec.Command(asmBin, "-p", "probe", "-o", filepath.Join(dir, "probe.o"), probePath)
|
||||||
cmd.Env = append(os.Environ(), "GOARCH="+goarch, "GOOS="+runtime.GOOS)
|
cmd.Env = append(os.Environ(), "GOARCH="+goarch, "GOOS="+runtime.GOOS)
|
||||||
out, _ := cmd.CombinedOutput()
|
out, _ := cmd.CombinedOutput()
|
||||||
|
// The expected failure mode is a non-zero exit with compiler diagnostics
|
||||||
|
// on stdout; empty output means the probe broke at the exec level (a
|
||||||
|
// killed child, a tool that would not start), and seeding every name as
|
||||||
|
// recognized on that silence would fake a clean audit.
|
||||||
|
if len(out) == 0 {
|
||||||
|
return nil, fmt.Errorf("go tool asm probe for GOARCH=%s produced no output", goarch)
|
||||||
|
}
|
||||||
|
|
||||||
result := map[string]bool{}
|
result := map[string]bool{}
|
||||||
for _, name := range names {
|
for _, name := range names {
|
||||||
@@ -244,18 +273,76 @@ func probeShapes(a arch.Arch) []string {
|
|||||||
// and takes R register spellings.
|
// and takes R register spellings.
|
||||||
"EQ, R0, R1, R2", "EQ, R0, R1", "EQ, R0",
|
"EQ, R0, R1, R2", "EQ, R0, R1", "EQ, R0",
|
||||||
"GE, F0, F1, F2", "NE, F0, F1, $0",
|
"GE, F0, F1, F2", "NE, F0, F1, $0",
|
||||||
|
// Pairs, acquire/release and exclusive atomics, LSE-AL forms.
|
||||||
|
"(R0), R1", "R0, (R1)", "R1, (R2), R3", "(R2, R3), 8(R1)",
|
||||||
|
"8(R1), (R2, R3)", "R1, R2, (R3)", "(R0)",
|
||||||
|
// System operations and their register/operand names.
|
||||||
|
"$4, R1, p2", "$35943", "$1", "$1, SPSel", "SPSel, R0",
|
||||||
|
"IVAC, R0", "(R0), PLDL1KEEP", "R1, R2, R3, R4",
|
||||||
|
// SIMD element, structure and literal-pool forms.
|
||||||
|
"(R0), [V1.B16]", "[V1.B16], (R0)", "V13.S[0], R1",
|
||||||
|
"R1, V2.B[3]", "$4, V1.B16, V2.B16", "V1.B16, (R0)",
|
||||||
|
"(R0), V1.B16", "",
|
||||||
|
// The spellings GOROOT's own kernels use, from the
|
||||||
|
// differential kernels this table was proven against.
|
||||||
|
"R0, p2", "R0, R1", "F0, F1, F2, F3", "$4, V1.B16, V2.B16, V3.B16, V4.B16",
|
||||||
|
"(R0), [V0.B8, V1.B8, V2.B8, V3.B8]", "$1, $2, V1",
|
||||||
|
"R0, R1, p2", "p2, R1", "$1234, R1", "DCZID_EL0, R1",
|
||||||
|
"$0", "R1, $4, EQ", "$33, R1, $25, R2", "$4, R1, p2",
|
||||||
|
"$4, V1.B8, V2.B8, V3.B8", "$63, V1.D2, V2.D2, V3.D2",
|
||||||
|
"V1.B16, [V2.B16], V3.B16", "V1.B8, [V2.B16, V3.B16], V4.B8",
|
||||||
|
"$4, V1.B16, V2.B16, V3.B16", "$15, V1", "V1, V2, p2",
|
||||||
|
"R0, R1, $1, $4, p2",
|
||||||
|
// The landing-pad kind, the compiler's PCDATA
|
||||||
|
// bookkeeping and the four-operand bitfield
|
||||||
|
// insert/extract family, as the toolchain's own
|
||||||
|
// testdata spells them.
|
||||||
|
"C", "$1, $0", "$0, R1, $1, R2",
|
||||||
}
|
}
|
||||||
case arch.RISCV:
|
case arch.RISCV:
|
||||||
return []string{
|
return []string{
|
||||||
"X5, X6, X7", "X5, X6", "X5", "$1, X5", "X5, (X6)", "$1, X5, X6",
|
"X5, X6, X7", "X5, X6", "X5", "$1, X5", "X5, (X6)", "$1, X5, X6",
|
||||||
"(X5), X6", "F0, F1, F2", "F0, F1", "p2", "X1, p2", "X0, p2",
|
"(X5), X6", "F0, F1, F2", "F0, F1", "p2", "X1, p2", "X0, p2",
|
||||||
"X5, X6, p2", "p2(SB)",
|
"X5, X6, p2", "p2(SB)",
|
||||||
|
// AMO atomics: destination, base, source.
|
||||||
|
"R5, (R4), R6", "X5, (X4), X6",
|
||||||
|
// Segment stores take the first vector register aligned
|
||||||
|
// to the segment count, as the toolchain requires.
|
||||||
|
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
|
||||||
|
// The FP multiply-add family takes four registers.
|
||||||
|
"F0, F1, F2, F3",
|
||||||
|
// The RVV slice: register, vector-register and vtype forms.
|
||||||
|
"V1, V2, V3", "V1, X5, V2", "V1", "V1, (X5)", "(X5), V1",
|
||||||
|
"$15, V1", "$15", "V1, V2", "V1, X5",
|
||||||
|
"X5, X6, p2", "R5, R6, p2",
|
||||||
|
"X5, E8, M8, TA, MA, X6", "$4, E32, M1, TA, MA, X1",
|
||||||
|
"(X5), X6, V1, V2",
|
||||||
|
// The CSR immediate forms the toolchain's testdata spells:
|
||||||
|
// immediate, CSR name, destination.
|
||||||
|
"$2, TIME, X5",
|
||||||
|
"",
|
||||||
}
|
}
|
||||||
case arch.LOONG64:
|
case arch.LOONG64:
|
||||||
return []string{
|
return []string{
|
||||||
"R4, R5, R6", "R4, R5", "R4", "$1, R4", "R4, (R5)", "(R4), R5",
|
"R4, R5, R6", "R4, R5", "R4", "$1, R4", "R4, (R5)", "(R4), R5",
|
||||||
"F0, F1, F2", "F0, F1", "p2", "R1, p2", "R4, p2",
|
"F0, F1, F2", "F0, F1", "p2", "R1, p2", "R4, p2",
|
||||||
"$1, R4, R5, R6", "$65536, R4", "R4, R5, p2", "p2(SB)",
|
"$1, R4, R5, R6", "$65536, R4", "R4, R5, p2", "p2(SB)",
|
||||||
|
// AMO atomics: destination, base, source.
|
||||||
|
"R5, (R4), R6", "X5, (X4), X6",
|
||||||
|
// Segment stores take the first vector register aligned
|
||||||
|
// to the segment count, as the toolchain requires.
|
||||||
|
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
|
||||||
|
// The LSX and LASX banks share the 5-bit numbering with F.
|
||||||
|
"V1, V2, V3", "X1, X2, X3", "V1, V2", "X1, X2", "V1", "X1",
|
||||||
|
// The vector compare-to-flag forms land in an FCC register.
|
||||||
|
"V1, FCC0", "X1, FCC0",
|
||||||
|
// The compiler's bookkeeping pair and the raw spellings the
|
||||||
|
// toolchain's own testdata carries: JIRL rd, rj, offset (the
|
||||||
|
// form RET lowers to), the prefetch with a 32-bit address and
|
||||||
|
// hint, and the byte-shuffle quads.
|
||||||
|
"$1, $0", "R1, R5, 0", "0(R7), $5, $0", "(R7), $5, $0",
|
||||||
|
"V1, V2, V3, V4", "X1, X2, X3, X4",
|
||||||
|
"",
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
@@ -305,3 +392,521 @@ func gasmAssembles(a arch.Arch, name, shape string) bool {
|
|||||||
func sanitize(name string) string {
|
func sanitize(name string) string {
|
||||||
return strings.NewReplacer(".", "_", "$", "_").Replace(name)
|
return strings.NewReplacer(".", "_", "$", "_").Replace(name)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- corpus audit -----------------------------------------------------------
|
||||||
|
|
||||||
|
// corpusTarget is one architecture row of the corpus report.
|
||||||
|
type corpusTarget struct {
|
||||||
|
a arch.Arch
|
||||||
|
name string
|
||||||
|
}
|
||||||
|
|
||||||
|
// corpusTally accumulates one architecture's attempts over the corpus.
|
||||||
|
type corpusTally struct {
|
||||||
|
attempted int
|
||||||
|
assembled int
|
||||||
|
reasons map[string]int // failure reason → count
|
||||||
|
example map[string]string // failure reason → one representative file
|
||||||
|
fails []corpusFailure // every failure, in file order, for --list
|
||||||
|
}
|
||||||
|
|
||||||
|
// corpusFailure is one failed attempt, recorded for the --list report.
|
||||||
|
type corpusFailure struct {
|
||||||
|
path string
|
||||||
|
reason string
|
||||||
|
detail string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (t *corpusTally) fail(path string, err error) {
|
||||||
|
reason := corpusReason(err)
|
||||||
|
t.reasons[reason]++
|
||||||
|
if t.example[reason] == "" {
|
||||||
|
t.example[reason] = path
|
||||||
|
}
|
||||||
|
t.fails = append(t.fails, corpusFailure{path: path, reason: reason, detail: firstLine(err.Error())})
|
||||||
|
}
|
||||||
|
|
||||||
|
// cmdAuditCorpus implements audit-instructions --corpus. The include
|
||||||
|
// directories carry #include resolution over a corpus whose files refer to
|
||||||
|
// headers such as GOROOT/pkg/include, the same -I a toolchain comparison
|
||||||
|
// needs.
|
||||||
|
func cmdAuditCorpus(args []string, dirs includeDirs, list bool) error {
|
||||||
|
if len(args) > 1 {
|
||||||
|
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
|
||||||
|
}
|
||||||
|
root := ""
|
||||||
|
if len(args) == 1 {
|
||||||
|
root = args[0]
|
||||||
|
} else {
|
||||||
|
out, err := exec.Command("go", "env", "GOROOT").Output()
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("locate GOROOT: %w", err)
|
||||||
|
}
|
||||||
|
root = filepath.Join(strings.TrimSpace(string(out)), "src")
|
||||||
|
}
|
||||||
|
// The toolchain's shipped headers (funcdata.h and friends) define the
|
||||||
|
// macros GOROOT files include; a corpus audit measures those files, so
|
||||||
|
// the header directory joins the search path automatically. go_asm.h
|
||||||
|
// is compiler-generated per package, so it is not resolved from here:
|
||||||
|
// files that include it get one generated per target architecture,
|
||||||
|
// which runCorpusAudit arranges.
|
||||||
|
if out, err := exec.Command("go", "env", "GOROOT").Output(); err == nil {
|
||||||
|
pkgInclude := filepath.Join(strings.TrimSpace(string(out)), "pkg", "include")
|
||||||
|
if fi, err := os.Stat(pkgInclude); err == nil && fi.IsDir() {
|
||||||
|
seen := false
|
||||||
|
for _, d := range dirs {
|
||||||
|
if d == pkgInclude {
|
||||||
|
seen = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !seen {
|
||||||
|
dirs = append(dirs, pkgInclude)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
stats, err := runCorpusAudit(root, dirs)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
printCorpusStats(stats, list)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// corpusStats is the outcome of one corpus audit run.
|
||||||
|
type corpusStats struct {
|
||||||
|
root string
|
||||||
|
files int
|
||||||
|
generic int // files attempted for all four architectures
|
||||||
|
narrowed int // files whose //go:build admits a proper subset of the four
|
||||||
|
excluded int // files whose //go:build admits none of the four: never compiled
|
||||||
|
otherPort int // files named for another Go port: never attempted
|
||||||
|
full int // files that assembled for every applicable target architecture
|
||||||
|
targets []corpusTarget
|
||||||
|
tallies []*corpusTally
|
||||||
|
}
|
||||||
|
|
||||||
|
// runCorpusAudit assembles every .s file under root and returns the stats.
|
||||||
|
// goPortSuffixes lists every architecture the Go project ports to. A file
|
||||||
|
// named for one of them belongs to that port's build, not to the generic
|
||||||
|
// set, even when gasm does not support the architecture.
|
||||||
|
var goPortSuffixes = []string{
|
||||||
|
"386", "amd64", "arm", "arm64", "loong64", "mips", "mips64",
|
||||||
|
"mips64le", "mipsle", "mips64x", "mipsx", "ppc64", "ppc64le",
|
||||||
|
"ppc64x", "riscv", "riscv64", "s390x", "wasm",
|
||||||
|
}
|
||||||
|
|
||||||
|
// otherPortFile reports whether the file belongs to a build no supported
|
||||||
|
// target ever compiles: either its name carries a Go-architecture suffix
|
||||||
|
// gasm does not support, or, for a file with no architecture suffix at all,
|
||||||
|
// it names another GOOS, which go/build drops from the file set
|
||||||
|
// (rt0_js_wasm.s is a javascript build, not a generic one).
|
||||||
|
func otherPortFile(path string) bool {
|
||||||
|
if otherGOOSFile(path) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
base := path
|
||||||
|
if i := strings.LastIndexByte(base, '/'); i >= 0 {
|
||||||
|
base = base[i+1:]
|
||||||
|
}
|
||||||
|
for _, sfx := range goPortSuffixes {
|
||||||
|
if strings.HasSuffix(base, "_"+sfx+".s") {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// goOSNames are the GOOS values go/build recognises in file names.
|
||||||
|
var goOSNames = map[string]bool{
|
||||||
|
"aix": true, "android": true, "darwin": true, "dragonfly": true,
|
||||||
|
"freebsd": true, "hurd": true, "illumos": true, "ios": true,
|
||||||
|
"js": true, "linux": true, "nacl": true, "netbsd": true,
|
||||||
|
"openbsd": true, "plan9": true, "solaris": true, "wasip1": true,
|
||||||
|
"windows": true, "zos": true,
|
||||||
|
}
|
||||||
|
|
||||||
|
// resolveGOOS validates a -GOOS flag value, mirroring the architecture
|
||||||
|
// check's surface: a usage error naming what the tool accepts.
|
||||||
|
func resolveGOOS(name string) (string, error) {
|
||||||
|
lower := strings.ToLower(name)
|
||||||
|
if goOSNames[lower] {
|
||||||
|
return lower, nil
|
||||||
|
}
|
||||||
|
return "", &usageError{fmt.Errorf("unknown GOOS %q: want one of %s", name, strings.Join(slices.Sorted(maps.Keys(goOSNames)), ", "))}
|
||||||
|
}
|
||||||
|
|
||||||
|
// goosFromFilename returns the GOOS the file's name carries, by go/build's
|
||||||
|
// goodOSArchFile rule: the GOOS segment sits last, or last before the
|
||||||
|
// architecture segment (sys_darwin_arm64.s, vlop_arm.s carries none). An
|
||||||
|
// empty result means the name names no GOOS and the ambient one applies.
|
||||||
|
func goosFromFilename(path string) string {
|
||||||
|
base := path
|
||||||
|
if i := strings.LastIndexByte(base, '/'); i >= 0 {
|
||||||
|
base = base[i+1:]
|
||||||
|
}
|
||||||
|
base = strings.TrimSuffix(base, ".s")
|
||||||
|
// go/build ignores everything before the first underscore, so a GOOS
|
||||||
|
// segment is only ever looked for from there on.
|
||||||
|
i := strings.IndexByte(base, '_')
|
||||||
|
if i < 0 {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
segs := strings.Split(base[i:], "_")
|
||||||
|
if n := len(segs); n >= 2 && goOSNames[segs[n-2]] && slices.Contains(goPortSuffixes, segs[n-1]) {
|
||||||
|
return segs[n-2]
|
||||||
|
}
|
||||||
|
if goOSNames[segs[len(segs)-1]] {
|
||||||
|
return segs[len(segs)-1]
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// otherGOOSFile reports whether the file's name names a GOOS other than the
|
||||||
|
// host's, by go/build's file-name rules.
|
||||||
|
func otherGOOSFile(path string) bool {
|
||||||
|
base := path
|
||||||
|
if i := strings.LastIndexByte(base, '/'); i >= 0 {
|
||||||
|
base = base[i+1:]
|
||||||
|
}
|
||||||
|
for seg := range strings.SplitSeq(strings.TrimSuffix(base, ".s"), "_") {
|
||||||
|
if goOSNames[seg] && seg != runtime.GOOS {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// buildConstraint returns the file's leading //go:build expression, or nil
|
||||||
|
// when the file carries none. The constraint governs the same header block
|
||||||
|
// go/build reads: blank lines and comments may precede it, and the first
|
||||||
|
// line that is neither ends the block. A constraint that does not parse
|
||||||
|
// narrows nothing, so the file stays in the attempted set: the audit must
|
||||||
|
// never exclude a file the toolchain would compile.
|
||||||
|
func buildConstraint(src string) constraint.Expr {
|
||||||
|
for line := range strings.SplitSeq(src, "\n") {
|
||||||
|
t := strings.TrimSpace(line)
|
||||||
|
switch {
|
||||||
|
case t == "":
|
||||||
|
continue
|
||||||
|
case strings.HasPrefix(t, "//"):
|
||||||
|
if constraint.IsGoBuild(t) {
|
||||||
|
e, err := constraint.Parse(t)
|
||||||
|
if err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return e
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
default:
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// unixOS is go/build's unixOS set: the GOOSes the unix build tag admits.
|
||||||
|
var unixOS = map[string]bool{
|
||||||
|
"aix": true, "android": true, "darwin": true, "dragonfly": true,
|
||||||
|
"freebsd": true, "hurd": true, "illumos": true, "ios": true,
|
||||||
|
"linux": true, "netbsd": true, "openbsd": true, "solaris": true,
|
||||||
|
}
|
||||||
|
|
||||||
|
// constraintTags answers the build tags a plain `go build` sets for a
|
||||||
|
// target: the GOOS and GOARCH, gc, and unix on the unix-like GOOSes. No
|
||||||
|
// experiment, sanitiser or cgo tag is ever true: the audit models the
|
||||||
|
// default build, and no GOROOT assembly file's constraint hinges on cgo.
|
||||||
|
func constraintTags(goarch, goos string) func(string) bool {
|
||||||
|
return func(tag string) bool {
|
||||||
|
switch tag {
|
||||||
|
case goarch, goos, "gc":
|
||||||
|
return true
|
||||||
|
case "unix":
|
||||||
|
return unixOS[goos]
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
|
||||||
|
files, err := asmFiles(root)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
targets := []corpusTarget{
|
||||||
|
{arch.AMD64, "amd64"},
|
||||||
|
{arch.ARM64, "arm64"},
|
||||||
|
{arch.RISCV, "riscv64"},
|
||||||
|
{arch.LOONG64, "loong64"},
|
||||||
|
}
|
||||||
|
tallies := make([]*corpusTally, len(targets))
|
||||||
|
for i := range tallies {
|
||||||
|
tallies[i] = &corpusTally{reasons: map[string]int{}, example: map[string]string{}}
|
||||||
|
}
|
||||||
|
// full is the north-star number: a file counts when every architecture
|
||||||
|
// its build admits assembles it.
|
||||||
|
full, generic, otherPort, narrowedCount, excluded := 0, 0, 0, 0, 0
|
||||||
|
|
||||||
|
// Header generation is created on first use, so a corpus with no
|
||||||
|
// go_asm.h includes never pays for a temp directory.
|
||||||
|
var hdr *asmhdrCache
|
||||||
|
defer func() {
|
||||||
|
if hdr != nil {
|
||||||
|
hdr.close()
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
|
||||||
|
for _, path := range files {
|
||||||
|
src, err := readSource(path)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
// The GOOS the header generation type-checks under follows the
|
||||||
|
// file's name when the name carries one; the ambient GOOS is the
|
||||||
|
// honest guess otherwise (a build tag naming another GOOS is
|
||||||
|
// invisible to a file-name rule).
|
||||||
|
goos := goosFromFilename(path)
|
||||||
|
|
||||||
|
// The GOOS the header generation type-checks under follows the
|
||||||
|
// file's name when the name carries one; the ambient GOOS is the
|
||||||
|
// honest guess otherwise.
|
||||||
|
namedArch := arch.FromFilename(path)
|
||||||
|
var wanted []int // indexes into targets
|
||||||
|
other := false
|
||||||
|
switch {
|
||||||
|
case namedArch != arch.Unknown:
|
||||||
|
for i, tg := range targets {
|
||||||
|
if tg.a == namedArch {
|
||||||
|
wanted = append(wanted, i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
case otherPortFile(path):
|
||||||
|
// A file named for a Go port gasm does not support (arm,
|
||||||
|
// 386, s390x, ...) or for another GOOS is compiled by no
|
||||||
|
// supported-arch build, so it is neither generic nor a
|
||||||
|
// per-arch attempt: counting it as generic would make the
|
||||||
|
// headline unreachably low for reasons no supported target
|
||||||
|
// can fix.
|
||||||
|
other = true
|
||||||
|
otherPort++
|
||||||
|
default:
|
||||||
|
for i := range targets {
|
||||||
|
wanted = append(wanted, i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A //go:build constraint narrows the set of targets the file is
|
||||||
|
// assembled for, the way the go command compiles the file only for
|
||||||
|
// the targets the expression admits: cpu_x86.s belongs to the x86
|
||||||
|
// build alone, and a file whose constraint admits none of the four
|
||||||
|
// targets (the goexperiment.runtimesecret and msan trees) is
|
||||||
|
// compiled by no supported build. The tags mirror what a plain
|
||||||
|
// `go build` sets: the GOOS and GOARCH, gc, and unix on the
|
||||||
|
// unix-like GOOSes; no experiment, sanitiser or cgo tag is ever
|
||||||
|
// true. The GOOS is the file's own when the name carries one,
|
||||||
|
// else the ambient one.
|
||||||
|
goosForEval := goos
|
||||||
|
if goosForEval == "" {
|
||||||
|
goosForEval = runtime.GOOS
|
||||||
|
}
|
||||||
|
narrowed := false
|
||||||
|
if len(wanted) > 0 {
|
||||||
|
if ce := buildConstraint(src); ce != nil {
|
||||||
|
kept := make([]int, 0, len(wanted))
|
||||||
|
for _, i := range wanted {
|
||||||
|
tg := targets[i]
|
||||||
|
if ce.Eval(constraintTags(goarchName(tg.a), goosForEval)) {
|
||||||
|
kept = append(kept, i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(kept) < len(wanted) {
|
||||||
|
narrowed = true
|
||||||
|
}
|
||||||
|
wanted = kept
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
switch {
|
||||||
|
case other:
|
||||||
|
// already tallied above
|
||||||
|
case len(wanted) == 0:
|
||||||
|
excluded++
|
||||||
|
case namedArch != arch.Unknown:
|
||||||
|
// a per-arch attempt over the constraint's subset
|
||||||
|
case narrowed:
|
||||||
|
narrowedCount++
|
||||||
|
default:
|
||||||
|
generic++
|
||||||
|
}
|
||||||
|
|
||||||
|
// A file that includes go_asm.h parses against a per-target header:
|
||||||
|
// the defines differ per architecture (internal/cpu's layout, for
|
||||||
|
// one) and per GOOS (sys_darwin_arm64.s's trampoline constants,
|
||||||
|
// for another), so the parse cannot be shared the way a
|
||||||
|
// header-free file's can. A generation failure is a failure for
|
||||||
|
// every target, named for the package rather than a bare "include
|
||||||
|
// not found". A header already resolvable in the package
|
||||||
|
// directory or the -I list is left alone.
|
||||||
|
if len(wanted) > 0 && needsGoAsmHeader(src) && !goAsmHeaderResolved(filepath.Dir(path), dirs) {
|
||||||
|
if hdr == nil {
|
||||||
|
if hdr, err = newAsmhdrCache(); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
pkgDir := filepath.Dir(path)
|
||||||
|
ok := true
|
||||||
|
for _, i := range wanted {
|
||||||
|
tg, t := targets[i], tallies[i]
|
||||||
|
t.attempted++
|
||||||
|
hdrDir, err := hdr.dirFor(pkgDir, goos, goarchName(tg.a))
|
||||||
|
if err != nil {
|
||||||
|
ok = false
|
||||||
|
t.fail(path, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
f, errs := parser.ParseWithOptions(path, src, parser.Options{
|
||||||
|
Expand: true,
|
||||||
|
IncludeDirs: append(slices.Clone(dirs), hdrDir),
|
||||||
|
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
|
||||||
|
})
|
||||||
|
if len(errs) > 0 {
|
||||||
|
ok = false
|
||||||
|
t.fail(path, errs[0])
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if _, err := assembleFile(tg.a, f, goos); err != nil {
|
||||||
|
ok = false
|
||||||
|
t.fail(path, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
t.assembled++
|
||||||
|
}
|
||||||
|
if ok && len(wanted) > 0 {
|
||||||
|
full++
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
ok := true
|
||||||
|
for _, i := range wanted {
|
||||||
|
tg, t := targets[i], tallies[i]
|
||||||
|
t.attempted++
|
||||||
|
// The parse carries the target's platform predefines, so it
|
||||||
|
// cannot be shared across targets the way a header-free file's
|
||||||
|
// could: a #ifdef GOARCH_arm block must be live on arm64 and
|
||||||
|
// dead everywhere else.
|
||||||
|
f, errs := parser.ParseWithOptions(path, src, parser.Options{
|
||||||
|
Expand: true,
|
||||||
|
IncludeDirs: dirs,
|
||||||
|
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
|
||||||
|
})
|
||||||
|
var err error
|
||||||
|
if len(errs) > 0 {
|
||||||
|
err = errs[0] // a parse failure is a failure for every target
|
||||||
|
} else {
|
||||||
|
_, err = assembleFile(tg.a, f, goos)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
ok = false
|
||||||
|
t.fail(path, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
t.assembled++
|
||||||
|
}
|
||||||
|
if ok && len(wanted) > 0 {
|
||||||
|
full++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return &corpusStats{
|
||||||
|
root: root,
|
||||||
|
files: len(files),
|
||||||
|
generic: generic,
|
||||||
|
narrowed: narrowedCount,
|
||||||
|
excluded: excluded,
|
||||||
|
otherPort: otherPort,
|
||||||
|
full: full,
|
||||||
|
targets: targets,
|
||||||
|
tallies: tallies,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// printCorpusStats renders the corpus audit report.
|
||||||
|
func printCorpusStats(s *corpusStats, list bool) {
|
||||||
|
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d narrowed by //go:build; %d excluded by //go:build; %d named for other Go ports, never attempted)\n",
|
||||||
|
s.root, s.files, s.generic, s.narrowed, s.excluded, s.otherPort)
|
||||||
|
// The rate is over the files a supported build would attempt: the
|
||||||
|
// other ports' files and the ones no supported target compiles sit in
|
||||||
|
// the count for completeness but can never assemble, so counting them
|
||||||
|
// in the denominator would report the gap of platforms gasm
|
||||||
|
// deliberately does not target.
|
||||||
|
attemptable := max(s.files-s.otherPort-s.excluded, 1)
|
||||||
|
fmt.Printf(" assemble for every applicable target: %d of %d attemptable (%.1f%%)\n", s.full, attemptable, 100*float64(s.full)/float64(attemptable))
|
||||||
|
for i, tg := range s.targets {
|
||||||
|
t := s.tallies[i]
|
||||||
|
fmt.Printf(" %s: %d/%d attempted\n", tg.name, t.assembled, t.attempted)
|
||||||
|
for _, r := range topReasons(t) {
|
||||||
|
fmt.Printf(" %4d %s\n", t.reasons[r], r)
|
||||||
|
fmt.Printf(" e.g. %s\n", t.example[r])
|
||||||
|
}
|
||||||
|
if !list {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for _, f := range t.fails {
|
||||||
|
fmt.Printf(" FAIL %s\n", f.path)
|
||||||
|
fmt.Printf(" %s: %s\n", f.reason, f.detail)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// corpusReason buckets an assembly or parse failure for the histogram.
|
||||||
|
func corpusReason(err error) string {
|
||||||
|
msg := err.Error()
|
||||||
|
switch {
|
||||||
|
case strings.Contains(msg, "go_asm.h for GOARCH"):
|
||||||
|
return "go_asm.h generation failed"
|
||||||
|
case strings.Contains(msg, "unsupported"), strings.Contains(msg, "cannot encode"):
|
||||||
|
return "instruction not encodable"
|
||||||
|
case strings.Contains(msg, "undefined label"):
|
||||||
|
return "undefined label"
|
||||||
|
case strings.Contains(msg, "undefined symbol"), strings.Contains(msg, "external symbol"), strings.Contains(msg, "file-level assembly"):
|
||||||
|
return "undefined symbol or external"
|
||||||
|
case strings.Contains(msg, "operand"), strings.Contains(msg, "operand form"):
|
||||||
|
return "unsupported operand form"
|
||||||
|
default:
|
||||||
|
return "other: " + firstLine(msg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// topReasons returns at most five reasons, most frequent first.
|
||||||
|
func topReasons(t *corpusTally) []string {
|
||||||
|
type kv struct {
|
||||||
|
k string
|
||||||
|
n int
|
||||||
|
}
|
||||||
|
var kvs []kv
|
||||||
|
for k, n := range t.reasons {
|
||||||
|
kvs = append(kvs, kv{k, n})
|
||||||
|
}
|
||||||
|
slices.SortFunc(kvs, func(a, b kv) int { return b.n - a.n })
|
||||||
|
if len(kvs) > 5 {
|
||||||
|
kvs = kvs[:5]
|
||||||
|
}
|
||||||
|
out := make([]string, len(kvs))
|
||||||
|
for i, kv := range kvs {
|
||||||
|
out[i] = kv.k
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// firstLine returns the first line of an error message, truncated.
|
||||||
|
func firstLine(msg string) string {
|
||||||
|
if i := strings.IndexByte(msg, '\n'); i >= 0 {
|
||||||
|
msg = msg[:i]
|
||||||
|
}
|
||||||
|
if len(msg) > 80 {
|
||||||
|
msg = msg[:80]
|
||||||
|
}
|
||||||
|
return msg
|
||||||
|
}
|
||||||
|
|||||||
+61
-1
@@ -4,9 +4,10 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"runtime"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
)
|
)
|
||||||
|
|
||||||
func TestDerivedFamily(t *testing.T) {
|
func TestDerivedFamily(t *testing.T) {
|
||||||
@@ -64,3 +65,62 @@ func TestGasmEncodable(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestBuildConstraint pins the //go:build reader: the constraint governs the
|
||||||
|
// leading comment block, the first non-comment line ends it (a tag below a
|
||||||
|
// #include governs nothing, exactly as go/build drops it), and a file
|
||||||
|
// without one admits every target.
|
||||||
|
func TestBuildConstraint(t *testing.T) {
|
||||||
|
admits := func(src, goarch, goos string) bool {
|
||||||
|
t.Helper()
|
||||||
|
e := buildConstraint(src)
|
||||||
|
if e == nil {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return e.Eval(constraintTags(goarch, goos))
|
||||||
|
}
|
||||||
|
const ret = "TEXT \xc2\xb7f(SB), NOSPLIT, $0\n\tRET\n"
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
src string
|
||||||
|
amd64, arm64 bool
|
||||||
|
}{
|
||||||
|
{"no constraint", ret, true, true},
|
||||||
|
{"x86 only", "//go:build 386 || amd64\n\n" + ret, true, false},
|
||||||
|
{"arm64 and linux", "//go:build arm64 && linux\n\n" + ret, false, true},
|
||||||
|
{"msan never", "//go:build msan\n\n" + ret, false, false},
|
||||||
|
{"experiment never", "//go:build goexperiment.runtimesecret\n\n" + ret, false, false},
|
||||||
|
{"below an include governs nothing", "#include \"textflag.h\"\n//go:build amd64\n" + ret, true, true},
|
||||||
|
{"unparsable narrows nothing", "//go:build (amd64\n" + ret, true, true},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
if got := admits(c.src, "amd64", runtime.GOOS); got != c.amd64 {
|
||||||
|
t.Errorf("amd64 admission = %v, want %v", got, c.amd64)
|
||||||
|
}
|
||||||
|
if got := admits(c.src, "arm64", runtime.GOOS); got != c.arm64 {
|
||||||
|
t.Errorf("arm64 admission = %v, want %v", got, c.arm64)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestConstraintTags pins the tag set a plain `go build` sets: the GOOS and
|
||||||
|
// GOARCH, gc, unix on the unix-like GOOSes; nothing else is ever true.
|
||||||
|
func TestConstraintTags(t *testing.T) {
|
||||||
|
ok := constraintTags("amd64", "linux")
|
||||||
|
for _, tag := range []string{"amd64", "linux", "gc", "unix"} {
|
||||||
|
if !ok(tag) {
|
||||||
|
t.Errorf("tag %q = false, want true", tag)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, tag := range []string{"arm64", "freebsd", "darwin", "cgo", "race", "msan", "goexperiment.runtimesecret"} {
|
||||||
|
if ok(tag) {
|
||||||
|
t.Errorf("tag %q = true, want false", tag)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
fb := constraintTags("arm64", "freebsd")
|
||||||
|
if !fb("unix") {
|
||||||
|
t.Error("unix on freebsd = false, want true")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
// SPDX-License-Identifier: BSD-3-Clause
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
//go:build !linux
|
//go:build !(linux || (freebsd && (amd64 || arm64 || riscv64)))
|
||||||
|
|
||||||
package main
|
package main
|
||||||
|
|
||||||
@@ -11,6 +11,6 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
func cmdDebug(args []string) int {
|
func cmdDebug(args []string) int {
|
||||||
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux (ptrace)")
|
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux or FreeBSD (ptrace)")
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
// SPDX-License-Identifier: BSD-3-Clause
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
//go:build linux
|
//go:build linux || (freebsd && (amd64 || arm64 || riscv64))
|
||||||
|
|
||||||
package main
|
package main
|
||||||
|
|
||||||
@@ -13,8 +13,8 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/debug"
|
"sourcedock.dev/petrbalvin/gasm-sdk/debug"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
|
||||||
)
|
)
|
||||||
|
|
||||||
func cmdDebug(args []string) int {
|
func cmdDebug(args []string) int {
|
||||||
@@ -24,9 +24,10 @@ function in a traced subprocess (ptrace), then provides a REPL for
|
|||||||
single-stepping, breakpoints, register and memory inspection.
|
single-stepping, breakpoints, register and memory inspection.
|
||||||
|
|
||||||
REPL commands:
|
REPL commands:
|
||||||
break <label|addr> [if <reg> <op> <val>]
|
break <label|addr|line> [if <reg> <op> <val|reg|*addr>]
|
||||||
set a breakpoint, optionally conditional on a
|
set a breakpoint, optionally conditional on a
|
||||||
register comparison (reg-reg or reg-immediate)
|
comparison of one register against a constant,
|
||||||
|
another register, or the 8-byte word at *addr
|
||||||
delete <label|addr> remove a breakpoint
|
delete <label|addr> remove a breakpoint
|
||||||
info break list all breakpoints
|
info break list all breakpoints
|
||||||
step [n], s single-step n instructions (default 1)
|
step [n], s single-step n instructions (default 1)
|
||||||
@@ -53,7 +54,7 @@ REPL commands:
|
|||||||
bufSpec := fs.String("buf", "", "buffer specification: name:size:pattern[,name:size:pattern...] where pattern is zero, ones, seq, or hex")
|
bufSpec := fs.String("buf", "", "buffer specification: name:size:pattern[,name:size:pattern...] where pattern is zero, ones, seq, or hex")
|
||||||
script := fs.String("script", "", "run REPL commands from a file (one per line) and exit; '-' reads stdin")
|
script := fs.String("script", "", "run REPL commands from a file (one per line) and exit; '-' reads stdin")
|
||||||
cover := fs.Bool("cover", false, "run to completion with a breakpoint on every instruction and report which executed and how often")
|
cover := fs.Bool("cover", false, "run to completion with a breakpoint on every instruction and report which executed and how often")
|
||||||
timeout := fs.Duration("timeout", 0, "kill the debuggee after this duration (e.g. 30s); for headless --script runs")
|
timeout := fs.Duration("timeout", 0, "kill the debuggee after this duration (e.g. 30s); for headless --script runs; a timeout exits 3")
|
||||||
fs.Parse(args)
|
fs.Parse(args)
|
||||||
|
|
||||||
// --- Debuggee mode (internal, spawned by the debugger) ---
|
// --- Debuggee mode (internal, spawned by the debugger) ---
|
||||||
@@ -231,10 +232,19 @@ REPL commands:
|
|||||||
if sess.Exited() {
|
if sess.Exited() {
|
||||||
break
|
break
|
||||||
}
|
}
|
||||||
|
// A genuine signal-delivery-stop (a fault in the kernel): the
|
||||||
|
// run cannot make progress, because resuming would restart the
|
||||||
|
// faulting instruction and fault forever. Report and stop.
|
||||||
|
if sig := sess.LastSignal(); sig != 0 {
|
||||||
|
fmt.Printf("gasm debug: cover: stopped on signal %v\n", sig)
|
||||||
|
break
|
||||||
|
}
|
||||||
|
reason, _ := sess.StopInfo()
|
||||||
regs, rerr := sess.GetRegs()
|
regs, rerr := sess.GetRegs()
|
||||||
if rerr != nil {
|
if rerr != nil {
|
||||||
break
|
break
|
||||||
}
|
}
|
||||||
|
trapPC := regs.GetPC()
|
||||||
// HandleTrap restores the original byte, rewinds PC and counts
|
// HandleTrap restores the original byte, rewinds PC and counts
|
||||||
// the hit on the breakpoint itself. Single-step over the
|
// the hit on the breakpoint itself. Single-step over the
|
||||||
// restored instruction so the reinsertion at the top of the
|
// restored instruction so the reinsertion at the top of the
|
||||||
@@ -243,6 +253,12 @@ REPL commands:
|
|||||||
if err := sess.Step(); err != nil {
|
if err := sess.Step(); err != nil {
|
||||||
break
|
break
|
||||||
}
|
}
|
||||||
|
} else if debug.TrapStray(sess, reason, trapPC) {
|
||||||
|
// A breakpoint-class trap that matches none of ours and left
|
||||||
|
// the PC in place: resuming would re-execute the trapping
|
||||||
|
// instruction forever, so the coverage run stops here.
|
||||||
|
fmt.Printf("gasm debug: cover: SIGTRAP at %#x matches no breakpoint; the PC did not advance\n", trapPC)
|
||||||
|
break
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
hits := map[uint64]int{}
|
hits := map[uint64]int{}
|
||||||
+4
-4
@@ -9,9 +9,9 @@ import (
|
|||||||
"sort"
|
"sort"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// cmdDis disassembles machine code: either a raw binary (standard input with
|
// cmdDis disassembles machine code: either a raw binary (standard input with
|
||||||
@@ -84,7 +84,7 @@ func disSource(path string, target arch.Arch) int {
|
|||||||
if len(errs) > 0 {
|
if len(errs) > 0 {
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
img, err := assembleFile(target, f)
|
img, err := assembleFile(target, f, "")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
|
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
|
|||||||
+266
-103
@@ -10,6 +10,7 @@ package main
|
|||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
"flag"
|
"flag"
|
||||||
"fmt"
|
"fmt"
|
||||||
"io"
|
"io"
|
||||||
@@ -18,6 +19,7 @@ import (
|
|||||||
"os/exec"
|
"os/exec"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"runtime"
|
"runtime"
|
||||||
|
"runtime/debug"
|
||||||
"slices"
|
"slices"
|
||||||
"sort"
|
"sort"
|
||||||
"strconv"
|
"strconv"
|
||||||
@@ -25,20 +27,43 @@ import (
|
|||||||
"sync"
|
"sync"
|
||||||
"syscall"
|
"syscall"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/format"
|
"sourcedock.dev/petrbalvin/gasm-sdk/format"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
"sourcedock.dev/petrbalvin/gasm-sdk/lexer"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
|
"sourcedock.dev/petrbalvin/gasm-sdk/lint"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/lsp"
|
"sourcedock.dev/petrbalvin/gasm-sdk/lsp"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
|
||||||
)
|
)
|
||||||
|
|
||||||
// version is the release version, stamped at build time via
|
// version reports the release the toolchain recorded for this build: the
|
||||||
// -ldflags "-X main.version=…" (defaulting to the current release).
|
// tag on a tag, a pseudo-version below one, and (devel) outside version
|
||||||
var version = "0.33.0"
|
// control. Nothing is injected; the recorded value cannot go stale.
|
||||||
|
func version() string {
|
||||||
|
bi, ok := debug.ReadBuildInfo()
|
||||||
|
if !ok || bi.Main.Version == "" {
|
||||||
|
return "(devel)"
|
||||||
|
}
|
||||||
|
return bi.Main.Version
|
||||||
|
}
|
||||||
|
|
||||||
|
// usageError marks an error the caller's arguments caused, which exits 2
|
||||||
|
// instead of the 1 a runtime failure gets.
|
||||||
|
type usageError struct{ err error }
|
||||||
|
|
||||||
|
func (e *usageError) Error() string { return e.err.Error() }
|
||||||
|
func (e *usageError) Unwrap() error { return e.err }
|
||||||
|
|
||||||
|
// exitCodeFor maps an error onto the process exit status: 2 for a usage
|
||||||
|
// error, 1 for anything else.
|
||||||
|
func exitCodeFor(err error) int {
|
||||||
|
if _, ok := errors.AsType[*usageError](err); ok {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
if len(os.Args) < 2 {
|
if len(os.Args) < 2 {
|
||||||
@@ -69,12 +94,12 @@ func main() {
|
|||||||
case "audit-instructions":
|
case "audit-instructions":
|
||||||
if err := cmdAuditInstructions(os.Args[2:]); err != nil {
|
if err := cmdAuditInstructions(os.Args[2:]); err != nil {
|
||||||
fmt.Fprintln(os.Stderr, err)
|
fmt.Fprintln(os.Stderr, err)
|
||||||
os.Exit(1)
|
os.Exit(exitCodeFor(err))
|
||||||
}
|
}
|
||||||
case "scaffold":
|
case "scaffold":
|
||||||
if err := cmdScaffold(os.Args[2:]); err != nil {
|
if err := cmdScaffold(os.Args[2:]); err != nil {
|
||||||
fmt.Fprintln(os.Stderr, err)
|
fmt.Fprintln(os.Stderr, err)
|
||||||
os.Exit(1)
|
os.Exit(exitCodeFor(err))
|
||||||
}
|
}
|
||||||
case "lsp":
|
case "lsp":
|
||||||
os.Exit(cmdLSP(os.Args[2:]))
|
os.Exit(cmdLSP(os.Args[2:]))
|
||||||
@@ -88,22 +113,22 @@ func main() {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// cmdVersion prints the release version.
|
// cmdVersion prints the recorded version.
|
||||||
func cmdVersion() int {
|
func cmdVersion() int {
|
||||||
fmt.Printf("gasm %s\n", version)
|
fmt.Printf("gasm %s\n", version())
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
// ANSI color helpers for terminal output.
|
// ANSI colour helpers for terminal output.
|
||||||
const (
|
const (
|
||||||
colorReset = "\033[0m"
|
colourReset = "\033[0m"
|
||||||
colorBold = "\033[1m"
|
colourBold = "\033[1m"
|
||||||
colorCyan = "\033[36m"
|
colourCyan = "\033[36m"
|
||||||
colorYellow = "\033[33m"
|
colourYellow = "\033[33m"
|
||||||
colorGray = "\033[90m"
|
colourGrey = "\033[90m"
|
||||||
)
|
)
|
||||||
|
|
||||||
// isTTY reports whether the writer is a terminal (for color output).
|
// isTTY reports whether the writer is a terminal (for colour output).
|
||||||
func isTTY(w io.Writer) bool {
|
func isTTY(w io.Writer) bool {
|
||||||
if f, ok := w.(*os.File); ok {
|
if f, ok := w.(*os.File); ok {
|
||||||
stat, _ := f.Stat()
|
stat, _ := f.Stat()
|
||||||
@@ -114,12 +139,12 @@ func isTTY(w io.Writer) bool {
|
|||||||
|
|
||||||
func usage(w io.Writer) {
|
func usage(w io.Writer) {
|
||||||
useColor := isTTY(w)
|
useColor := isTTY(w)
|
||||||
bold, cyan, yellow, gray, reset := "", "", "", "", ""
|
bold, cyan, yellow, grey, reset := "", "", "", "", ""
|
||||||
if useColor {
|
if useColor {
|
||||||
bold, cyan, yellow, gray, reset = colorBold, colorCyan, colorYellow, colorGray, colorReset
|
bold, cyan, yellow, grey, reset = colourBold, colourCyan, colourYellow, colourGrey, colourReset
|
||||||
}
|
}
|
||||||
|
|
||||||
fmt.Fprintf(w, "%sgasm %s%s: developer tooling for Go's Plan 9 assembler (GAsm)%s\n\n", bold, version, reset, reset)
|
fmt.Fprintf(w, "%sgasm %s%s: developer tooling for Go's Plan 9 assembler (GAsm)%s\n\n", bold, version(), reset, reset)
|
||||||
fmt.Fprintf(w, "gasm bundles a lexer, parser, formatter, linter, standalone assembler and\n")
|
fmt.Fprintf(w, "gasm bundles a lexer, parser, formatter, linter, standalone assembler and\n")
|
||||||
fmt.Fprintf(w, "language server for Plan 9 assembly into one self-contained binary.\n\n")
|
fmt.Fprintf(w, "language server for Plan 9 assembly into one self-contained binary.\n\n")
|
||||||
|
|
||||||
@@ -145,12 +170,12 @@ func usage(w io.Writer) {
|
|||||||
{"version", "print the version (same as --version)"},
|
{"version", "print the version (same as --version)"},
|
||||||
}
|
}
|
||||||
for _, c := range commands {
|
for _, c := range commands {
|
||||||
fmt.Fprintf(w, " %s%-10s%s %s%s%s\n", cyan, c.name, reset, gray, c.desc, reset)
|
fmt.Fprintf(w, " %s%-10s%s %s%s%s\n", cyan, c.name, reset, grey, c.desc, reset)
|
||||||
}
|
}
|
||||||
|
|
||||||
fmt.Fprintf(w, "\n%sFlags:%s\n", yellow, reset)
|
fmt.Fprintf(w, "\n%sFlags:%s\n", yellow, reset)
|
||||||
fmt.Fprintf(w, " %s-h, --help%s %sshow this help%s\n", cyan, reset, gray, reset)
|
fmt.Fprintf(w, " %s-h, --help%s %sshow this help%s\n", cyan, reset, grey, reset)
|
||||||
fmt.Fprintf(w, " %s-V, --version%s %sprint the version%s\n", cyan, reset, gray, reset)
|
fmt.Fprintf(w, " %s-V, --version%s %sprint the version%s\n", cyan, reset, grey, reset)
|
||||||
|
|
||||||
fmt.Fprintf(w, "\nRun \"gasm <command> -h\" for a command's usage and flags.\n\n")
|
fmt.Fprintf(w, "\nRun \"gasm <command> -h\" for a command's usage and flags.\n\n")
|
||||||
|
|
||||||
@@ -164,7 +189,7 @@ func usage(w io.Writer) {
|
|||||||
}
|
}
|
||||||
for _, e := range examples {
|
for _, e := range examples {
|
||||||
if e.desc != "" {
|
if e.desc != "" {
|
||||||
fmt.Fprintf(w, " %s%s%s %s%s%s\n", cyan, e.cmd, reset, gray, e.desc, reset)
|
fmt.Fprintf(w, " %s%s%s %s%s%s\n", cyan, e.cmd, reset, grey, e.desc, reset)
|
||||||
} else {
|
} else {
|
||||||
fmt.Fprintf(w, " %s%s%s\n", cyan, e.cmd, reset)
|
fmt.Fprintf(w, " %s%s%s\n", cyan, e.cmd, reset)
|
||||||
}
|
}
|
||||||
@@ -215,6 +240,16 @@ func readSource(path string) (string, error) {
|
|||||||
return string(b), err
|
return string(b), err
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// includeDirs collects repeatable -I flags: the directories searched for
|
||||||
|
// #include files during macro expansion and include splicing.
|
||||||
|
type includeDirs []string
|
||||||
|
|
||||||
|
func (d *includeDirs) String() string { return strings.Join(*d, ",") }
|
||||||
|
func (d *includeDirs) Set(v string) error {
|
||||||
|
*d = append(*d, v)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
func cmdTokens(args []string) int {
|
func cmdTokens(args []string) int {
|
||||||
fs := newCommand("tokens", "gasm tokens <file>", `
|
fs := newCommand("tokens", "gasm tokens <file>", `
|
||||||
Print the lexical token stream of FILE: position, token kind and text, one
|
Print the lexical token stream of FILE: position, token kind and text, one
|
||||||
@@ -442,7 +477,7 @@ hover, document symbols, diagnostics and semantic-token highlighting.
|
|||||||
`)
|
`)
|
||||||
fs.Parse(args)
|
fs.Parse(args)
|
||||||
srv := lsp.New(os.Stdin, os.Stdout)
|
srv := lsp.New(os.Stdin, os.Stdout)
|
||||||
srv.SetVersion(version)
|
srv.SetVersion(version())
|
||||||
if err := srv.Run(); err != nil {
|
if err := srv.Run(); err != nil {
|
||||||
fmt.Fprintln(os.Stderr, "gasm lsp:", err)
|
fmt.Fprintln(os.Stderr, "gasm lsp:", err)
|
||||||
return 1
|
return 1
|
||||||
@@ -451,7 +486,7 @@ hover, document symbols, diagnostics and semantic-token highlighting.
|
|||||||
}
|
}
|
||||||
|
|
||||||
func cmdAsm(args []string) int {
|
func cmdAsm(args []string) int {
|
||||||
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>", `
|
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-GOOS os] [-o out] <file>", `
|
||||||
Assemble FILE without the Go toolchain: every TEXT function is encoded to
|
Assemble FILE without the Go toolchain: every TEXT function is encoded to
|
||||||
machine code and printed as a hex dump. Supported architectures: amd64
|
machine code and printed as a hex dump. Supported architectures: amd64
|
||||||
(including VEX/AVX2 and EVEX/AVX-512), arm64 (AArch64 integer, FP,
|
(including VEX/AVX2 and EVEX/AVX-512), arm64 (AArch64 integer, FP,
|
||||||
@@ -461,27 +496,85 @@ func cmdAsm(args []string) int {
|
|||||||
With -o the output is written to a file instead. The --format flag selects
|
With -o the output is written to a file instead. The --format flag selects
|
||||||
what is written: raw (the default) concatenates the functions and the data
|
what is written: raw (the default) concatenates the functions and the data
|
||||||
section into one self-consistent image; elf emits a relocatable object
|
section into one self-consistent image; elf emits a relocatable object
|
||||||
(.text/.data sections, a symbol table and one PC32 relocation per
|
(.text/.data sections, a symbol table and one relocation per static-symbol
|
||||||
static-symbol reference) that links with the system toolchain; goobj emits
|
reference, in the architecture's own form: R_X86_64_PC32 on amd64,
|
||||||
the Go toolchain's own object format, which cmd/link consumes directly (it
|
R_AARCH64_*, R_RISCV_* or R_LARCH_* on the others) that links with the
|
||||||
requires -p, the package path, and the installed Go toolchain).
|
system toolchain; goobj emits the Go toolchain's own object format, which
|
||||||
|
cmd/link consumes directly (it requires -p, the package path, and the
|
||||||
|
installed Go toolchain: the object preamble is captured from go tool asm
|
||||||
|
and the format version from go version).
|
||||||
|
|
||||||
|
A file that includes go_asm.h gets that header generated automatically from
|
||||||
|
the package it lives in (the .go files beside it, type-checked for the
|
||||||
|
target architecture, the toolchain's own defines), so GOROOT assembly
|
||||||
|
assembles without a compiler. -GOOS selects the type-checking GOOS for
|
||||||
|
that header: a GOOS-specific file (sys_darwin_arm64.s) needs its platform's
|
||||||
|
defines, which a header from the ambient GOOS silently omits. A package
|
||||||
|
that has no Go files for the target or does not type-check is a hard error
|
||||||
|
naming the package.
|
||||||
`)
|
`)
|
||||||
out := fs.String("o", "", "write the output to this file")
|
out := fs.String("o", "", "write the output to this file")
|
||||||
format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
|
format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
|
||||||
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
|
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
|
||||||
|
archName := fs.String("GOARCH", "", "target architecture: amd64, arm64, riscv64 or loong64 (overrides the file-name suffix)")
|
||||||
|
goosName := fs.String("GOOS", "", "operating system for go_asm.h generation: a GOOS go/build recognises (default: the host's)")
|
||||||
|
var dirs includeDirs
|
||||||
|
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
|
||||||
fs.Parse(args)
|
fs.Parse(args)
|
||||||
if fs.NArg() != 1 {
|
if fs.NArg() != 1 {
|
||||||
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>")
|
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-GOOS os] [-o out] <file>")
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
// The format is validated before anything else, so a bogus value exits 2
|
||||||
|
// with or without -o instead of silently dumping the hex of a raw image.
|
||||||
|
switch *format {
|
||||||
|
case "raw", "elf", "goobj":
|
||||||
|
default:
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm asm: unknown format %q (want raw, elf or goobj)\n", *format)
|
||||||
return 2
|
return 2
|
||||||
}
|
}
|
||||||
path := fs.Arg(0)
|
path := fs.Arg(0)
|
||||||
targetArch := arch.FromFilename(path)
|
targetArch := arch.FromFilename(path)
|
||||||
|
if *archName != "" {
|
||||||
|
a, err := auditArch(*archName)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm asm: %v\n", err)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
targetArch = a
|
||||||
|
}
|
||||||
|
goos := ""
|
||||||
|
if *goosName != "" {
|
||||||
|
g, err := resolveGOOS(*goosName)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm asm: %v\n", err)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
goos = g
|
||||||
|
}
|
||||||
src, err := readSource(path)
|
src, err := readSource(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
f, errs := parser.Parse(path, src)
|
// A file that includes go_asm.h cannot assemble without the package's
|
||||||
|
// defines, and without a compiler nothing else has generated them, so
|
||||||
|
// gasm produces the equivalent itself: automatic, because the compiler
|
||||||
|
// behaves the same way and a flag would only ever be forgotten. A
|
||||||
|
// generation failure is fatal and names the package: assembling against
|
||||||
|
// a missing header would fail later with a bare "undefined" instead.
|
||||||
|
// A go_asm.h that already resolves (placed by hand, or passed with -I)
|
||||||
|
// is left alone.
|
||||||
|
if needsGoAsmHeader(src) && !goAsmHeaderResolved(filepath.Dir(path), dirs) {
|
||||||
|
hdrDir, cleanup, err := ensureGoAsmHeader(path, targetArch, goos, nil)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintln(os.Stderr, "gasm asm:", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
defer cleanup()
|
||||||
|
dirs = append(dirs, hdrDir)
|
||||||
|
}
|
||||||
|
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs, Predefines: platformPredefinesFor(string(targetArch), goos)})
|
||||||
for _, e := range errs {
|
for _, e := range errs {
|
||||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
||||||
}
|
}
|
||||||
@@ -489,47 +582,53 @@ requires -p, the package path, and the installed Go toolchain).
|
|||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
|
||||||
img, err := assembleFile(targetArch, f)
|
img, err := assembleFile(targetArch, f, goos)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
|
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
if len(img.Funcs) == 0 {
|
if len(img.Funcs) == 0 && len(img.Data) == 0 {
|
||||||
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions found")
|
// A file with neither code nor data assembles to nothing, which is
|
||||||
|
// almost always a wrong architecture rather than an intent.
|
||||||
|
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions or GLOBL data found")
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
for _, fn := range img.Funcs {
|
// Without -o the hex dump on stdout is the output; with -o the file is,
|
||||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
// and the dump is skipped, as the -o help text promises.
|
||||||
fmt.Printf("%s: %d bytes\n", fn.Name, fn.Size)
|
if *out == "" {
|
||||||
for i := 0; i < len(code); i += 16 {
|
for _, fn := range img.Funcs {
|
||||||
end := min(i+16, len(code))
|
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||||
fmt.Printf(" %04x:", i)
|
fmt.Printf("%s: %d bytes\n", fn.Name, fn.Size)
|
||||||
for _, b := range code[i:end] {
|
for i := 0; i < len(code); i += 16 {
|
||||||
fmt.Printf(" %02x", b)
|
end := min(i+16, len(code))
|
||||||
|
fmt.Printf(" %04x:", i)
|
||||||
|
for _, b := range code[i:end] {
|
||||||
|
fmt.Printf(" %02x", b)
|
||||||
|
}
|
||||||
|
fmt.Println()
|
||||||
}
|
}
|
||||||
fmt.Println()
|
|
||||||
}
|
}
|
||||||
}
|
if len(img.Data) > 0 {
|
||||||
if len(img.Data) > 0 {
|
fmt.Printf("data: %d bytes at 0x%x\n", len(img.Data), len(img.Code))
|
||||||
fmt.Printf("data: %d bytes at 0x%x\n", len(img.Data), len(img.Code))
|
for _, d := range f.Decls {
|
||||||
for _, d := range f.Decls {
|
g, ok := d.(*ast.Globl)
|
||||||
g, ok := d.(*ast.Globl)
|
if !ok || g.Name == nil || g.Name.Pseudo != "SB" {
|
||||||
if !ok || g.Name == nil || g.Name.Pseudo != "SB" {
|
continue
|
||||||
continue
|
}
|
||||||
|
size := 0
|
||||||
|
if g.Size != nil && g.Size.Imm.HasVal {
|
||||||
|
size = int(g.Size.Imm.Val)
|
||||||
|
}
|
||||||
|
fmt.Printf(" %s: %d bytes at 0x%x\n", g.Name.Name, size, img.Symbols[g.Name.Name])
|
||||||
}
|
}
|
||||||
size := 0
|
for i := 0; i < len(img.Data); i += 16 {
|
||||||
if g.Size != nil && g.Size.Imm.HasVal {
|
end := min(i+16, len(img.Data))
|
||||||
size = int(g.Size.Imm.Val)
|
fmt.Printf(" %04x:", len(img.Code)+i)
|
||||||
|
for _, b := range img.Data[i:end] {
|
||||||
|
fmt.Printf(" %02x", b)
|
||||||
|
}
|
||||||
|
fmt.Println()
|
||||||
}
|
}
|
||||||
fmt.Printf(" %s: %d bytes at 0x%x\n", g.Name.Name, size, img.Symbols[g.Name.Name])
|
|
||||||
}
|
|
||||||
for i := 0; i < len(img.Data); i += 16 {
|
|
||||||
end := min(i+16, len(img.Data))
|
|
||||||
fmt.Printf(" %04x:", len(img.Code)+i)
|
|
||||||
for _, b := range img.Data[i:end] {
|
|
||||||
fmt.Printf(" %02x", b)
|
|
||||||
}
|
|
||||||
fmt.Println()
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if *out != "" {
|
if *out != "" {
|
||||||
@@ -567,9 +666,6 @@ requires -p, the package path, and the installed Go toolchain).
|
|||||||
obj, err = img.GOObject(*pkg, path)
|
obj, err = img.GOObject(*pkg, path)
|
||||||
}
|
}
|
||||||
kind = "Go object"
|
kind = "Go object"
|
||||||
default:
|
|
||||||
fmt.Fprintf(os.Stderr, "gasm asm: unknown format %q (want raw, elf or goobj)\n", *format)
|
|
||||||
return 2
|
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintln(os.Stderr, "gasm asm:", err)
|
fmt.Fprintln(os.Stderr, "gasm asm:", err)
|
||||||
@@ -586,7 +682,7 @@ requires -p, the package path, and the installed Go toolchain).
|
|||||||
|
|
||||||
// cmdDiff compares the machine code of two assembly files.
|
// cmdDiff compares the machine code of two assembly files.
|
||||||
func cmdDiff(args []string) int {
|
func cmdDiff(args []string) int {
|
||||||
set := newCommand("diff", "gasm diff <file1.s> <file2.s>", `
|
set := newCommand("diff", "gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>", `
|
||||||
Compare the machine code produced by assembling two files.
|
Compare the machine code produced by assembling two files.
|
||||||
Shows which functions differ and the byte-level differences.
|
Shows which functions differ and the byte-level differences.
|
||||||
Useful for verifying that two implementations produce identical code,
|
Useful for verifying that two implementations produce identical code,
|
||||||
@@ -596,12 +692,24 @@ Use --map to compare functions whose names differ between the files,
|
|||||||
e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
|
e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
|
||||||
`)
|
`)
|
||||||
mapSpec := set.String("map", "", "comma-separated old=new pairs to match functions with different names")
|
mapSpec := set.String("map", "", "comma-separated old=new pairs to match functions with different names")
|
||||||
|
archName := set.String("GOARCH", "", "target architecture for both files: amd64, arm64, riscv64 or loong64")
|
||||||
|
var dirs includeDirs
|
||||||
|
set.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
|
||||||
set.Parse(args)
|
set.Parse(args)
|
||||||
if set.NArg() != 2 {
|
if set.NArg() != 2 {
|
||||||
fmt.Fprintln(os.Stderr, "usage: gasm diff <file1.s> <file2.s>")
|
fmt.Fprintln(os.Stderr, "usage: gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>")
|
||||||
return 2
|
return 2
|
||||||
}
|
}
|
||||||
path1, path2 := set.Arg(0), set.Arg(1)
|
path1, path2 := set.Arg(0), set.Arg(1)
|
||||||
|
forced := arch.Unknown
|
||||||
|
if *archName != "" {
|
||||||
|
a, err := auditArch(*archName)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm diff: %v\n", err)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
forced = a
|
||||||
|
}
|
||||||
|
|
||||||
// Parse the name mapping (file1 name → file2 name).
|
// Parse the name mapping (file1 name → file2 name).
|
||||||
nameMap := make(map[string]string)
|
nameMap := make(map[string]string)
|
||||||
@@ -617,12 +725,12 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Assemble both files.
|
// Assemble both files.
|
||||||
img1, err := assemblePath(path1)
|
img1, err := assemblePath(path1, forced, dirs)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err)
|
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
img2, err := assemblePath(path2)
|
img2, err := assemblePath(path2, forced, dirs)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err)
|
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err)
|
||||||
return 1
|
return 1
|
||||||
@@ -681,11 +789,34 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
|
|||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// platformPredefines mirrors the go command's assembler invocation, which
|
||||||
|
// defines GOOS_<goos> and GOARCH_<arch> as -D macros: GOROOT headers
|
||||||
|
// (go_tls.h, asm_riscv64.h) select their platform blocks with #ifdef on
|
||||||
|
// exactly those names, so an assembler without them cannot see the platform
|
||||||
|
// definitions at all.
|
||||||
|
func platformPredefines(goarch, goos string) map[string]string {
|
||||||
|
return map[string]string{
|
||||||
|
"GOARCH_" + goarch: "1",
|
||||||
|
"GOOS_" + goos: "1",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// platformPredefinesFor resolves the ambient GOOS the way a build would: a
|
||||||
|
// file whose name carries one (sys_darwin_arm64.s) is compiled for that GOOS
|
||||||
|
// and nothing else.
|
||||||
|
func platformPredefinesFor(goarch string, fileGoos string) map[string]string {
|
||||||
|
goos := fileGoos
|
||||||
|
if goos == "" {
|
||||||
|
goos = runtime.GOOS
|
||||||
|
}
|
||||||
|
return platformPredefines(goarch, goos)
|
||||||
|
}
|
||||||
|
|
||||||
// assembleFile assembles a parsed file for the given architecture and returns the image.
|
// assembleFile assembles a parsed file for the given architecture and returns the image.
|
||||||
func assembleFile(targetArch arch.Arch, f *ast.File) (*asm.Image, error) {
|
func assembleFile(targetArch arch.Arch, f *ast.File, goos string) (*asm.Image, error) {
|
||||||
switch targetArch {
|
switch targetArch {
|
||||||
case arch.AMD64:
|
case arch.AMD64:
|
||||||
return asm.AssembleFile(f)
|
return asm.AssembleFile(f, asm.WithGOOS(goos))
|
||||||
case arch.RISCV:
|
case arch.RISCV:
|
||||||
return asm.AssembleFileRISCV(f)
|
return asm.AssembleFileRISCV(f)
|
||||||
case arch.ARM64:
|
case arch.ARM64:
|
||||||
@@ -697,20 +828,26 @@ func assembleFile(targetArch arch.Arch, f *ast.File) (*asm.Image, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// assemblePath reads, parses and assembles a file (used by cmdDiff).
|
// assemblePath reads, preprocesses, parses and assembles a file (used by
|
||||||
func assemblePath(path string) (*asm.Image, error) {
|
// cmdDiff). A non-Unknown forced architecture overrides the file-name
|
||||||
|
// suffix.
|
||||||
|
func assemblePath(path string, forced arch.Arch, dirs includeDirs) (*asm.Image, error) {
|
||||||
src, err := readSource(path)
|
src, err := readSource(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
f, errs := parser.Parse(path, src)
|
target := forced
|
||||||
|
if target == arch.Unknown {
|
||||||
|
target = arch.FromFilename(path)
|
||||||
|
}
|
||||||
|
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs, Predefines: platformPredefinesFor(string(target), "")})
|
||||||
for _, e := range errs {
|
for _, e := range errs {
|
||||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
||||||
}
|
}
|
||||||
if len(errs) > 0 {
|
if len(errs) > 0 {
|
||||||
return nil, fmt.Errorf("parse errors")
|
return nil, fmt.Errorf("parse errors")
|
||||||
}
|
}
|
||||||
return assembleFile(arch.FromFilename(path), f)
|
return assembleFile(target, f, "")
|
||||||
}
|
}
|
||||||
|
|
||||||
// printByteDiff shows the first few byte differences between two code blocks.
|
// printByteDiff shows the first few byte differences between two code blocks.
|
||||||
@@ -733,8 +870,8 @@ func cmdProfile(args []string) int {
|
|||||||
flagSet := newCommand("profile", "gasm profile <file.s>", `
|
flagSet := newCommand("profile", "gasm profile <file.s>", `
|
||||||
Show the basic-block structure of functions in an assembly file.
|
Show the basic-block structure of functions in an assembly file.
|
||||||
Lists each function's labels, their offsets, and the block boundaries.
|
Lists each function's labels, their offsets, and the block boundaries.
|
||||||
This is the static structure; for runtime execution counts, use
|
This is the static structure; for runtime execution counts use
|
||||||
gasm verify --fuzz which exercises the code paths.
|
gasm debug --cover, and for input coverage gasm verify --fuzz.
|
||||||
`)
|
`)
|
||||||
flagSet.Parse(args)
|
flagSet.Parse(args)
|
||||||
if flagSet.NArg() != 1 {
|
if flagSet.NArg() != 1 {
|
||||||
@@ -806,7 +943,7 @@ func cmdVerifyNonJIT(path string, targetArch arch.Arch, groundTruth, profile boo
|
|||||||
if len(errs) > 0 {
|
if len(errs) > 0 {
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
img, err := assembleFile(targetArch, f)
|
img, err := assembleFile(targetArch, f, "")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
@@ -878,11 +1015,40 @@ func compareGroundTruth(img *asm.Image, gt map[string][]byte) (matched, total, d
|
|||||||
goCmp[j] = 0
|
goCmp[j] = 0
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if bytes.Equal(gasmCmp, goCmp) {
|
// The toolchain pads text symbols to 16-byte boundaries with
|
||||||
|
// zeros, so a function whose size is not a multiple of 16
|
||||||
|
// carries trailing zeros in the ground truth that are not part
|
||||||
|
// of the encoding. Compare up to the shorter side and require
|
||||||
|
// the remainder of whichever is longer to be zero, so padding
|
||||||
|
// never masks a real difference.
|
||||||
|
cmpLen := min(len(gasmCmp), len(goCmp))
|
||||||
|
equal := bytes.Equal(gasmCmp[:cmpLen], goCmp[:cmpLen])
|
||||||
|
if equal {
|
||||||
|
for _, b := range gasmCmp[cmpLen:] {
|
||||||
|
if b != 0 {
|
||||||
|
equal = false
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if equal {
|
||||||
|
for _, b := range goCmp[cmpLen:] {
|
||||||
|
if b != 0 {
|
||||||
|
equal = false
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if equal {
|
||||||
matched++
|
matched++
|
||||||
if len(fn.Relocs) > 0 {
|
switch {
|
||||||
|
case len(fn.Relocs) > 0 && len(goCmp) > cmpLen:
|
||||||
|
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked, %d padding)\n", fn.Name, fn.Size, len(fn.Relocs), len(goCmp)-cmpLen)
|
||||||
|
case len(fn.Relocs) > 0:
|
||||||
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
|
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
|
||||||
} else {
|
case len(goCmp) > cmpLen:
|
||||||
|
fmt.Printf(" %s: MATCH (%d bytes, %d padding)\n", fn.Name, fn.Size, len(goCmp)-cmpLen)
|
||||||
|
default:
|
||||||
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
|
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
@@ -930,8 +1096,8 @@ that tolerate nil pointers and zero lengths in their arguments.
|
|||||||
With -abi, each function is called with sentinel values in the registers
|
With -abi, each function is called with sentinel values in the registers
|
||||||
the Go ABI fixes across calls (the frame pointer and the goroutine
|
the Go ABI fixes across calls (the frame pointer and the goroutine
|
||||||
pointer) plus a canary below SP; violations are reported. JIT-based
|
pointer) plus a canary below SP; violations are reported. JIT-based
|
||||||
checks run when the host matches the file's architecture (all but
|
checks run when the host matches the file's architecture, on all four
|
||||||
loong64, which is ground-truth only for now).
|
architectures.
|
||||||
|
|
||||||
With -fuzz, each function with a // func signature is differentially fuzzed
|
With -fuzz, each function with a // func signature is differentially fuzzed
|
||||||
against the go-tool-asm version in a subprocess (so a crash on a partial
|
against the go-tool-asm version in a subprocess (so a crash on a partial
|
||||||
@@ -944,7 +1110,8 @@ With -profile, the static basic-block structure is listed for each function.
|
|||||||
|
|
||||||
With -call, a single function is invoked with user-supplied buffers (-buf)
|
With -call, a single function is invoked with user-supplied buffers (-buf)
|
||||||
instead of the smoke/abi/fuzz sweeps. Useful for partial functions (e.g.
|
instead of the smoke/abi/fuzz sweeps. Useful for partial functions (e.g.
|
||||||
decoders) that crash on random input but should succeed on valid data.
|
decoders) that crash on random input but should succeed on valid data. The
|
||||||
|
function named must be NOSPLIT: a function with a stack frame is refused.
|
||||||
|
|
||||||
With -save-corpus (and -fuzz), every input that crashes or mismatches is
|
With -save-corpus (and -fuzz), every input that crashes or mismatches is
|
||||||
written to the directory as replayable JSON. -replay re-runs saved
|
written to the directory as replayable JSON. -replay re-runs saved
|
||||||
@@ -973,20 +1140,16 @@ each entry reproduces.
|
|||||||
path := set.Arg(0)
|
path := set.Arg(0)
|
||||||
targetArch := arch.FromFilename(path)
|
targetArch := arch.FromFilename(path)
|
||||||
// JIT execution runs when the host CPU matches the kernel's
|
// JIT execution runs when the host CPU matches the kernel's
|
||||||
// architecture, except loong64: its trampoline is implemented but not
|
// architecture; every trampoline is validated end to end under
|
||||||
// yet validated against real hardware (the Go runtime cannot start
|
// qemu-user emulation (the loong64 one included, via the raw-address
|
||||||
// under the available loong64 emulators), so those kernels take the
|
// leave handoff).
|
||||||
// toolchain-comparison path.
|
if targetArch != hostArch() {
|
||||||
if targetArch != hostArch() || targetArch == arch.LOONG64 {
|
// No JIT on this host: ground truth and profile remain available for
|
||||||
// No JIT on this host: ground truth and profile remain available.
|
// every architecture, because cmdVerifyNonJIT assembles and compares
|
||||||
// (loong64 is ground-truth-only everywhere for now: its trampoline
|
// against the toolchain without executing anything.
|
||||||
// is implemented but not yet validated against real hardware.)
|
|
||||||
switch targetArch {
|
switch targetArch {
|
||||||
case arch.RISCV, arch.LOONG64, arch.ARM64:
|
case arch.AMD64, arch.RISCV, arch.LOONG64, arch.ARM64:
|
||||||
return cmdVerifyNonJIT(path, targetArch, *groundTruth, *profile)
|
return cmdVerifyNonJIT(path, targetArch, *groundTruth, *profile)
|
||||||
case arch.AMD64:
|
|
||||||
fmt.Fprintln(os.Stderr, "gasm verify: JIT-based checks need an amd64 host; use --ground-truth here")
|
|
||||||
return 1
|
|
||||||
default:
|
default:
|
||||||
fmt.Fprintln(os.Stderr, "gasm verify: unsupported architecture")
|
fmt.Fprintln(os.Stderr, "gasm verify: unsupported architecture")
|
||||||
return 1
|
return 1
|
||||||
|
|||||||
+171
-3
@@ -13,6 +13,9 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"syscall"
|
"syscall"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||||
)
|
)
|
||||||
|
|
||||||
const clean = "#include \"textflag.h\"\n" +
|
const clean = "#include \"textflag.h\"\n" +
|
||||||
@@ -219,8 +222,9 @@ func TestCmdVersion(t *testing.T) {
|
|||||||
if code != 0 {
|
if code != 0 {
|
||||||
t.Fatalf("code = %d", code)
|
t.Fatalf("code = %d", code)
|
||||||
}
|
}
|
||||||
if !strings.Contains(out, version) {
|
got := version()
|
||||||
t.Errorf("version output %q does not mention %q", out, version)
|
if !strings.Contains(out, got) {
|
||||||
|
t.Errorf("version output %q does not mention %q", out, got)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -241,6 +245,96 @@ func TestCmdArgErrors(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestUsageExitCodes pins the exit-code contract for the commands whose main
|
||||||
|
// dispatches on a returned error: a wrong argument set exits 2, the same as
|
||||||
|
// the commands that count their arguments themselves, while a runtime
|
||||||
|
// failure (an unreadable file) keeps exit 1.
|
||||||
|
func TestUsageExitCodes(t *testing.T) {
|
||||||
|
for name, err := range map[string]error{
|
||||||
|
"audit-instructions extra argument": cmdAuditInstructions([]string{"amd64", "extra"}),
|
||||||
|
"audit-instructions unknown arch": cmdAuditInstructions([]string{"mips"}),
|
||||||
|
"audit-instructions corpus extra": cmdAuditInstructions([]string{"--corpus", "a", "b"}),
|
||||||
|
"scaffold no arguments": cmdScaffold(nil),
|
||||||
|
"scaffold extra arguments": cmdScaffold([]string{"differential", "a.s", "b.s"}),
|
||||||
|
} {
|
||||||
|
if err == nil {
|
||||||
|
t.Errorf("%s: expected an error", name)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if code := exitCodeFor(err); code != 2 {
|
||||||
|
t.Errorf("%s: exit code = %d, want 2 (err: %v)", name, code, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if err := cmdScaffold([]string{"differential", "/nonexistent/file.s"}); err == nil {
|
||||||
|
t.Error("scaffold on a missing file should fail")
|
||||||
|
} else if code := exitCodeFor(err); code != 1 {
|
||||||
|
t.Errorf("scaffold on a missing file: exit code = %d, want 1", code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCmdAsmFormatValidation checks that an unknown --format exits 2 with
|
||||||
|
// and without -o, instead of assembling and silently dumping a raw image.
|
||||||
|
func TestCmdAsmFormatValidation(t *testing.T) {
|
||||||
|
path := writeTemp(t, "f_amd64.s", clean)
|
||||||
|
out := filepath.Join(t.TempDir(), "f.bin")
|
||||||
|
if _, _, code := capture(func() int { return cmdAsm([]string{"--format", "bogus", path}) }); code != 2 {
|
||||||
|
t.Errorf("asm --format bogus without -o: code = %d, want 2", code)
|
||||||
|
}
|
||||||
|
if _, _, code := capture(func() int { return cmdAsm([]string{"--format", "bogus", "-o", out, path}) }); code != 2 {
|
||||||
|
t.Errorf("asm --format bogus with -o: code = %d, want 2", code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCmdAsmOutputFile pins the documented -o behaviour: the output goes to
|
||||||
|
// the file and stdout carries no hex dump; without -o the dump is the output.
|
||||||
|
func TestCmdAsmOutputFile(t *testing.T) {
|
||||||
|
path := writeTemp(t, "f_amd64.s", clean)
|
||||||
|
out := filepath.Join(t.TempDir(), "f.bin")
|
||||||
|
stdout, _, code := capture(func() int { return cmdAsm([]string{"-o", out, path}) })
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("code = %d", code)
|
||||||
|
}
|
||||||
|
if strings.Contains(stdout, "0000:") {
|
||||||
|
t.Errorf("stdout carries a hex dump despite -o:\n%s", stdout)
|
||||||
|
}
|
||||||
|
if !strings.Contains(stdout, "wrote ") {
|
||||||
|
t.Errorf("stdout misses the wrote line:\n%s", stdout)
|
||||||
|
}
|
||||||
|
b, err := os.ReadFile(out)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if len(b) == 0 {
|
||||||
|
t.Error("the output file is empty")
|
||||||
|
}
|
||||||
|
|
||||||
|
stdout, _, code = capture(func() int { return cmdAsm([]string{path}) })
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("without -o: code = %d", code)
|
||||||
|
}
|
||||||
|
if !strings.Contains(stdout, "0000:") {
|
||||||
|
t.Errorf("without -o the hex dump is missing:\n%s", stdout)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestVerifyNonJITAMD64GroundTruth drives the cross-architecture
|
||||||
|
// ground-truth path for an amd64 kernel: the path a host of any other
|
||||||
|
// architecture takes, which must compare against the toolchain rather than
|
||||||
|
// refuse to run.
|
||||||
|
func TestVerifyNonJITAMD64GroundTruth(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("runs go tool asm")
|
||||||
|
}
|
||||||
|
path := writeTemp(t, "f_amd64.s", clean)
|
||||||
|
out, _, code := capture(func() int { return cmdVerifyNonJIT(path, arch.AMD64, true, false) })
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("code = %d (%s)", code, out)
|
||||||
|
}
|
||||||
|
if !strings.Contains(out, "1/1 matched") {
|
||||||
|
t.Errorf("output misses the matched report:\n%s", out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestVerifySmokeCrashIsolation checks that a function faulting on its
|
// TestVerifySmokeCrashIsolation checks that a function faulting on its
|
||||||
// zeroed smoke arguments is reported as CRASH by a child process instead of
|
// zeroed smoke arguments is reported as CRASH by a child process instead of
|
||||||
// killing `gasm verify` itself.
|
// killing `gasm verify` itself.
|
||||||
@@ -274,7 +368,7 @@ func TestVerifySmokeCrashIsolation(t *testing.T) {
|
|||||||
}
|
}
|
||||||
if exitErr, ok := err.(*exec.ExitError); ok {
|
if exitErr, ok := err.(*exec.ExitError); ok {
|
||||||
if ws, ok := exitErr.Sys().(syscall.WaitStatus); ok && ws.Signaled() {
|
if ws, ok := exitErr.Sys().(syscall.WaitStatus); ok && ws.Signaled() {
|
||||||
t.Fatalf("verify died from %v — the crash was not isolated:\n%s", ws.Signal(), out)
|
t.Fatalf("verify died from %v; the crash was not isolated:\n%s", ws.Signal(), out)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if !strings.Contains(string(out), "CRASH") {
|
if !strings.Contains(string(out), "CRASH") {
|
||||||
@@ -292,3 +386,77 @@ func TestSweepCheckLines(t *testing.T) {
|
|||||||
t.Errorf("sweepCheckLines = %q, want %q", got, want)
|
t.Errorf("sweepCheckLines = %q, want %q", got, want)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestRunCorpusAudit drives the corpus audit over a small fixture tree: one
|
||||||
|
// suffixed amd64 file, one suffixed arm64 file whose body is not arm64, one
|
||||||
|
// generic file, and one file that does not parse.
|
||||||
|
func TestRunCorpusAudit(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
write := func(name, src string) {
|
||||||
|
t.Helper()
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
write("good_amd64.s", "#include \"textflag.h\"\nTEXT ·add(SB), NOSPLIT, $0-0\n\tMOVQ AX, BX\n\tRET\n")
|
||||||
|
write("bad_arm64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0-0\n\tMOVQ AX, BX\n\tRET\n")
|
||||||
|
write("generic.s", "#include \"textflag.h\"\nTEXT ·g(SB), NOSPLIT, $0-0\n\tRET\n")
|
||||||
|
write("broken.s", "#include \"textflag.h\"\nTEXT ·b(SB), NOSPLIT, $0-0\n\tJMP nowhere\n\tRET\n")
|
||||||
|
|
||||||
|
stats, err := runCorpusAudit(dir, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("runCorpusAudit: %v", err)
|
||||||
|
}
|
||||||
|
if stats.files != 4 {
|
||||||
|
t.Errorf("files = %d, want 4", stats.files)
|
||||||
|
}
|
||||||
|
if stats.generic != 2 {
|
||||||
|
t.Errorf("generic = %d, want 2 (generic.s and broken.s)", stats.generic)
|
||||||
|
}
|
||||||
|
// good_amd64 and generic.s assemble everywhere they are attempted.
|
||||||
|
if stats.full != 2 {
|
||||||
|
t.Errorf("full = %d, want 2", stats.full)
|
||||||
|
}
|
||||||
|
get := func(name string) *corpusTally {
|
||||||
|
for i, tg := range stats.targets {
|
||||||
|
if tg.name == name {
|
||||||
|
return stats.tallies[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatalf("no tally for %s", name)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
// amd64: good_amd64 + generic.s + broken.s; the broken file fails to parse.
|
||||||
|
if a := get("amd64"); a.attempted != 3 || a.assembled != 2 {
|
||||||
|
t.Errorf("amd64 = %d/%d, want 2/3", a.assembled, a.attempted)
|
||||||
|
}
|
||||||
|
// arm64: bad_arm64 (MOVQ is not arm64) + generic.s + broken.s.
|
||||||
|
if a := get("arm64"); a.attempted != 3 || a.assembled != 1 {
|
||||||
|
t.Errorf("arm64 = %d/%d, want 1/3", a.assembled, a.attempted)
|
||||||
|
}
|
||||||
|
if r := get("amd64").reasons["instruction not encodable"]; r != 0 {
|
||||||
|
t.Errorf("amd64 unexpected unencodable reason: %d", r)
|
||||||
|
}
|
||||||
|
if r := get("arm64").reasons["instruction not encodable"]; r != 1 {
|
||||||
|
t.Errorf("arm64 unencodable reasons = %d, want 1", r)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCompareGroundTruthPadding pins the padding-aware ground-truth
|
||||||
|
// comparison: the toolchain pads text symbols to 16-byte boundaries, so
|
||||||
|
// trailing zeros in the reference must not read as a mismatch, while any
|
||||||
|
// non-zero tail still must.
|
||||||
|
func TestCompareGroundTruthPadding(t *testing.T) {
|
||||||
|
code := []byte{0x48, 0x8b, 0x07, 0xc3} // 4 bytes, not a multiple of 16
|
||||||
|
img := &asm.Image{Code: code, Funcs: []asm.FuncLayout{{Name: "f", Offset: 0, Size: len(code)}}}
|
||||||
|
padded := append(append([]byte(nil), code...), 0, 0, 0)
|
||||||
|
matched, total, diffs := compareGroundTruth(img, map[string][]byte{"f": padded})
|
||||||
|
if matched != 1 || total != 1 || diffs != 0 {
|
||||||
|
t.Fatalf("zero padding should match: matched=%d total=%d diffs=%d", matched, total, diffs)
|
||||||
|
}
|
||||||
|
dirty := append(append([]byte(nil), code...), 0, 0x90, 0)
|
||||||
|
matched, _, diffs = compareGroundTruth(img, map[string][]byte{"f": dirty})
|
||||||
|
if matched != 0 || diffs != 1 {
|
||||||
|
t.Fatalf("non-zero padding must mismatch: matched=%d diffs=%d", matched, diffs)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,241 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"regexp"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestManPagesTrackTheCLI builds the binary once, then compares every
|
||||||
|
// command's live `-h` output with its docs/man/gasm-<command>.1 page: the
|
||||||
|
// flag sets must agree both ways, and the page's SYNOPSIS line must carry
|
||||||
|
// the command's usage line. A flag or a usage change that skips the man
|
||||||
|
// page fails here, so the pages cannot drift from the binary.
|
||||||
|
func TestManPagesTrackTheCLI(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("builds the gasm binary")
|
||||||
|
}
|
||||||
|
bin := filepath.Join(t.TempDir(), "gasm")
|
||||||
|
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("build gasm: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, cmd := range []string{
|
||||||
|
"tokens", "parse", "fmt", "lint", "asm", "dis", "verify",
|
||||||
|
"debug", "diff", "profile", "audit-instructions", "scaffold", "lsp",
|
||||||
|
} {
|
||||||
|
t.Run(cmd, func(t *testing.T) {
|
||||||
|
raw, err := os.ReadFile(filepath.Join("..", "..", "docs", "man", "gasm-"+cmd+".1"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read man page: %v", err)
|
||||||
|
}
|
||||||
|
page := string(raw)
|
||||||
|
|
||||||
|
out, _ := exec.Command(bin, cmd, "-h").CombinedOutput()
|
||||||
|
help := string(out)
|
||||||
|
|
||||||
|
binFlags := helpFlags(help)
|
||||||
|
pageFlags := roffFlags(page)
|
||||||
|
for f := range binFlags {
|
||||||
|
if !pageFlags[f] {
|
||||||
|
t.Errorf("flag -%s is in the binary's help but missing from the man page", f)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for f := range pageFlags {
|
||||||
|
if !binFlags[f] {
|
||||||
|
t.Errorf("flag -%s is in the man page but the binary does not accept it", f)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
want := helpUsage(help)
|
||||||
|
got := roffSynopsis(page)
|
||||||
|
if want != "" && got != want {
|
||||||
|
t.Errorf("SYNOPSIS drift:\n page: %s\nbinary: %s", got, want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestManCommandsTrackHelp compares the gasm(1) COMMANDS list with the
|
||||||
|
// top-level help output, so a subcommand added to the binary cannot miss
|
||||||
|
// its man entry and a stale entry cannot outlive its command.
|
||||||
|
func TestManCommandsTrackHelp(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("builds the gasm binary")
|
||||||
|
}
|
||||||
|
bin := filepath.Join(t.TempDir(), "gasm")
|
||||||
|
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("build gasm: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
raw, err := os.ReadFile(filepath.Join("..", "..", "docs", "man", "gasm.1"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read man page: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
helpOut, err := exec.Command(bin, "--help").Output()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("gasm --help: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
binCmds := helpCommands(string(helpOut))
|
||||||
|
pageCmds := roffCommands(string(raw))
|
||||||
|
for c := range binCmds {
|
||||||
|
if !pageCmds[c] {
|
||||||
|
t.Errorf("command %q is in the binary's help but missing from gasm(1) COMMANDS", c)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for c := range pageCmds {
|
||||||
|
if !binCmds[c] {
|
||||||
|
t.Errorf("command %q is in gasm(1) COMMANDS but the binary does not list it", c)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// helpFlags extracts the flag names from a `gasm <cmd> -h` output.
|
||||||
|
func helpFlags(help string) map[string]bool {
|
||||||
|
m := map[string]bool{}
|
||||||
|
inFlags := false
|
||||||
|
for line := range strings.SplitSeq(help, "\n") {
|
||||||
|
if strings.TrimRight(line, " \t") == "Flags:" {
|
||||||
|
inFlags = true
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !inFlags {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !strings.HasPrefix(line, " -") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
token := strings.FieldsFunc(strings.TrimLeft(line, " "), func(r rune) bool {
|
||||||
|
return r == ' ' || r == '\t'
|
||||||
|
})
|
||||||
|
if len(token) == 0 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
m[strings.TrimLeft(token[0], "-")] = true
|
||||||
|
}
|
||||||
|
return m
|
||||||
|
}
|
||||||
|
|
||||||
|
var roffEscape = regexp.MustCompile(`\\f[BIRP]`)
|
||||||
|
|
||||||
|
// roffFlags extracts the flag names from a man page's OPTIONS section.
|
||||||
|
func roffFlags(page string) map[string]bool {
|
||||||
|
m := map[string]bool{}
|
||||||
|
inOptions := false
|
||||||
|
for line := range strings.SplitSeq(page, "\n") {
|
||||||
|
if strings.HasPrefix(line, ".SH ") {
|
||||||
|
inOptions = strings.HasPrefix(line, ".SH OPTIONS")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !inOptions {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Flag entries are written as either `.B \-flag` or `\fB\-flag`.
|
||||||
|
var body string
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(line, `.B \-`):
|
||||||
|
body = line[3:]
|
||||||
|
case strings.HasPrefix(line, `\fB\-`):
|
||||||
|
body = line[1:]
|
||||||
|
default:
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
name := roffEscape.ReplaceAllString(body, "")
|
||||||
|
name = strings.ReplaceAll(name, `\-`, "-")
|
||||||
|
name = strings.TrimSpace(name)
|
||||||
|
if i := strings.IndexAny(name, " \t"); i >= 0 {
|
||||||
|
name = name[:i]
|
||||||
|
}
|
||||||
|
m[strings.TrimLeft(name, "-")] = true
|
||||||
|
}
|
||||||
|
return m
|
||||||
|
}
|
||||||
|
|
||||||
|
// helpCommands extracts the command names from the top-level help output's
|
||||||
|
// Commands section.
|
||||||
|
func helpCommands(help string) map[string]bool {
|
||||||
|
m := map[string]bool{}
|
||||||
|
inCmds := false
|
||||||
|
for line := range strings.SplitSeq(help, "\n") {
|
||||||
|
if strings.TrimSpace(line) == "Commands:" {
|
||||||
|
inCmds = true
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !inCmds {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
t := strings.TrimSpace(line)
|
||||||
|
if t == "" {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
name, _, _ := strings.Cut(t, " ")
|
||||||
|
m[name] = true
|
||||||
|
}
|
||||||
|
return m
|
||||||
|
}
|
||||||
|
|
||||||
|
// roffCommands extracts the command names from gasm(1)'s COMMANDS section,
|
||||||
|
// where each entry is written as `.B gasm\-<name>(1)` or `.B gasm <name>`.
|
||||||
|
func roffCommands(page string) map[string]bool {
|
||||||
|
m := map[string]bool{}
|
||||||
|
inCmds := false
|
||||||
|
for line := range strings.SplitSeq(page, "\n") {
|
||||||
|
if strings.HasPrefix(line, ".SH ") {
|
||||||
|
inCmds = strings.HasPrefix(line, ".SH COMMANDS")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !inCmds || !strings.HasPrefix(line, ".B gasm") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
entry := strings.ReplaceAll(strings.TrimPrefix(line, ".B "), `\-`, "-")
|
||||||
|
entry = strings.TrimSuffix(entry, "(1)")
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(entry, "gasm-"):
|
||||||
|
m[strings.TrimPrefix(entry, "gasm-")] = true
|
||||||
|
case strings.HasPrefix(entry, "gasm "):
|
||||||
|
m[strings.TrimPrefix(entry, "gasm ")] = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return m
|
||||||
|
}
|
||||||
|
|
||||||
|
// helpUsage returns the command's usage line without the "Usage: " prefix.
|
||||||
|
func helpUsage(help string) string {
|
||||||
|
for line := range strings.SplitSeq(help, "\n") {
|
||||||
|
if strings.HasPrefix(line, "Usage: ") {
|
||||||
|
return normaliseUsage(line[len("Usage: "):])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// roffSynopsis returns the page's SYNOPSIS usage line, unescaped.
|
||||||
|
func roffSynopsis(page string) string {
|
||||||
|
inSyn := false
|
||||||
|
for line := range strings.SplitSeq(page, "\n") {
|
||||||
|
if strings.HasPrefix(line, ".SH ") {
|
||||||
|
inSyn = strings.HasPrefix(line, ".SH SYNOPSIS")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !inSyn || !strings.HasPrefix(line, ".B ") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
return normaliseUsage(strings.ReplaceAll(line[3:], `\-`, "-"))
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// normaliseUsage flattens whitespace and drops the roff font escapes so that
|
||||||
|
// the binary's usage line and the page's SYNOPSIS line compare equal.
|
||||||
|
func normaliseUsage(s string) string {
|
||||||
|
s = roffEscape.ReplaceAllString(s, "")
|
||||||
|
return strings.Join(strings.Fields(s), " ")
|
||||||
|
}
|
||||||
@@ -0,0 +1,109 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// writeTree writes a directory of files and returns its root.
|
||||||
|
func writeTree(t *testing.T, files map[string]string) string {
|
||||||
|
t.Helper()
|
||||||
|
dir := t.TempDir()
|
||||||
|
for name, content := range files {
|
||||||
|
path := filepath.Join(dir, name)
|
||||||
|
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return dir
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAsmMacroAndIncludeEndToEnd drives `gasm asm` over a source with an
|
||||||
|
// in-file parameterised macro and an include resolved through -I, and checks
|
||||||
|
// the assembled bytes came from the expansion (the loop body counts six
|
||||||
|
// increments, two per expanded iteration).
|
||||||
|
func TestAsmMacroAndIncludeEndToEnd(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("runs the assembler end to end")
|
||||||
|
}
|
||||||
|
dir := writeTree(t, map[string]string{
|
||||||
|
"inc/consts.h": "#define NITER 3\n",
|
||||||
|
"main_amd64.s": "#include \"textflag.h\"\n" +
|
||||||
|
"#include \"consts.h\"\n" +
|
||||||
|
"#define STEP(r) ADDQ $1, r; ADDQ $1, r\n" +
|
||||||
|
"TEXT ·f(SB), NOSPLIT, $0-8\n" +
|
||||||
|
"\tXORQ AX, AX\n" +
|
||||||
|
"\tMOVQ $NITER, CX\n" +
|
||||||
|
"loop:\n" +
|
||||||
|
"\tSTEP(AX)\n" +
|
||||||
|
"\tDECQ CX\n" +
|
||||||
|
"\tJNZ loop\n" +
|
||||||
|
"\tMOVQ AX, ret+0(FP)\n" +
|
||||||
|
"\tRET\n",
|
||||||
|
})
|
||||||
|
stdout, stderr, code := capture(func() int {
|
||||||
|
return cmdAsm([]string{"-I", filepath.Join(dir, "inc"), "-GOARCH", "amd64", filepath.Join(dir, "main_amd64.s")})
|
||||||
|
})
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
|
||||||
|
}
|
||||||
|
// The macro expanded to two ADDQ $1 encodings in the static body; the
|
||||||
|
// iteration count lives in the runtime loop.
|
||||||
|
if n := strings.Count(stdout, "83 c0 01"); n != 2 {
|
||||||
|
t.Errorf("found %d ADDQ $1 encodings in the image, want 2:\n%s", n, stdout)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAsmIncludeResolutionOrder pins the -I search order end to end: the
|
||||||
|
// including file's directory wins over the -I directories.
|
||||||
|
func TestAsmIncludeResolutionOrder(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("runs the assembler end to end")
|
||||||
|
}
|
||||||
|
dir := writeTree(t, map[string]string{
|
||||||
|
"src/main_amd64.s": "#include \"textflag.h\"\n" +
|
||||||
|
"#include \"vals.h\"\n" +
|
||||||
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||||
|
"\tMOVQ $VAL, AX\n" +
|
||||||
|
"\tRET\n",
|
||||||
|
"src/vals.h": "#define VAL 1\n",
|
||||||
|
"late/vals.h": "#define VAL 2\n",
|
||||||
|
"early/vals.h": "#define VAL 3\n",
|
||||||
|
})
|
||||||
|
stdout, stderr, code := capture(func() int {
|
||||||
|
return cmdAsm([]string{"-I", filepath.Join(dir, "early"), "-I", filepath.Join(dir, "late"),
|
||||||
|
"-GOARCH", "amd64", filepath.Join(dir, "src", "main_amd64.s")})
|
||||||
|
})
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
|
||||||
|
}
|
||||||
|
// VAL came from src/vals.h, not from either -I directory: the image
|
||||||
|
// loads the immediate 1.
|
||||||
|
if !strings.Contains(stdout, "b8 01 00 00 00") {
|
||||||
|
t.Errorf("expected the source-directory VAL (immediate 1) in:\n%s", stdout)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAsmMissingIncludeIsAnError pins the diagnostic for an include that
|
||||||
|
// resolves nowhere on the assembly path.
|
||||||
|
func TestAsmMissingIncludeIsAnError(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("runs the assembler end to end")
|
||||||
|
}
|
||||||
|
path := writeTemp(t, "main_amd64.s", "#include \"textflag.h\"\n#include \"nothere.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
|
_, stderr, code := capture(func() int { return cmdAsm([]string{"-GOARCH", "amd64", path}) })
|
||||||
|
if code == 0 {
|
||||||
|
t.Fatal("gasm asm accepted a file whose include resolves nowhere")
|
||||||
|
}
|
||||||
|
if !strings.Contains(stderr, `#include "nothere.h"`) {
|
||||||
|
t.Errorf("stderr does not name the failing include: %s", stderr)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -11,8 +11,8 @@ import (
|
|||||||
"os"
|
"os"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
gasmast "sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
gasmast "sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
gasmparser "sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
gasmparser "sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// cmdScaffold generates a differential test skeleton for every kernel in a
|
// cmdScaffold generates a differential test skeleton for every kernel in a
|
||||||
@@ -44,7 +44,7 @@ bodies, place the file in the kernel's package, and run it in CI.
|
|||||||
rest = rest[1:]
|
rest = rest[1:]
|
||||||
}
|
}
|
||||||
if len(rest) != 1 {
|
if len(rest) != 1 {
|
||||||
return fmt.Errorf("usage: gasm scaffold differential <file.s>")
|
return &usageError{fmt.Errorf("usage: gasm scaffold differential <file.s>")}
|
||||||
}
|
}
|
||||||
path := rest[0]
|
path := rest[0]
|
||||||
src, err := os.ReadFile(path)
|
src, err := os.ReadFile(path)
|
||||||
|
|||||||
+130
-51
@@ -1,19 +1,23 @@
|
|||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
// SPDX-License-Identifier: BSD-3-Clause
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
//go:build linux
|
//go:build linux || (freebsd && (amd64 || arm64 || riscv64))
|
||||||
|
|
||||||
package debug
|
package debug
|
||||||
|
|
||||||
import "strings"
|
import "strings"
|
||||||
|
|
||||||
import "fmt"
|
import (
|
||||||
|
"cmp"
|
||||||
|
"fmt"
|
||||||
|
"slices"
|
||||||
|
)
|
||||||
|
|
||||||
// Breakpoint is one INT3 breakpoint in the debuggee.
|
// Breakpoint is one software breakpoint in the debuggee.
|
||||||
type Breakpoint struct {
|
type Breakpoint struct {
|
||||||
Addr uint64 // absolute address in the debuggee
|
Addr uint64 // absolute address in the debuggee
|
||||||
Label string // source label ("" for raw addresses)
|
Label string // source label ("" for raw addresses)
|
||||||
Orig byte // original byte at Addr (restored on removal)
|
Orig []byte // original bytes at Addr (restored on removal)
|
||||||
Enabled bool
|
Enabled bool
|
||||||
Cond *Condition // optional condition (nil = unconditional)
|
Cond *Condition // optional condition (nil = unconditional)
|
||||||
hits int
|
hits int
|
||||||
@@ -32,8 +36,12 @@ type Condition struct {
|
|||||||
MemAddr uint64 // memory address (for register-memory comparison, prefixed with *)
|
MemAddr uint64 // memory address (for register-memory comparison, prefixed with *)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Eval checks the condition against the current registers.
|
// Eval checks the condition against the current registers. For the
|
||||||
func (c *Condition) Eval(regs *Regs) bool {
|
// register-memory form, mem reads an 8-byte little-endian word from the
|
||||||
|
// debuggee; it may be nil when no reader is available. Anything that cannot
|
||||||
|
// be decided (unknown register or operator, unreadable memory) does not
|
||||||
|
// block the breakpoint.
|
||||||
|
func (c *Condition) Eval(regs *Regs, mem func(addr uint64) (uint64, bool)) bool {
|
||||||
actual, ok := regs.RegValue(c.Reg)
|
actual, ok := regs.RegValue(c.Reg)
|
||||||
if !ok {
|
if !ok {
|
||||||
return true // unknown register, don't block
|
return true // unknown register, don't block
|
||||||
@@ -48,9 +56,16 @@ func (c *Condition) Eval(regs *Regs) bool {
|
|||||||
}
|
}
|
||||||
expected = v
|
expected = v
|
||||||
case c.MemAddr != 0:
|
case c.MemAddr != 0:
|
||||||
// Register-memory comparison, requires a Session, not available here.
|
// Register-memory comparison, resolved in the debuggee at
|
||||||
// Fall back to treating as constant (the caller should resolve).
|
// evaluation time.
|
||||||
expected = c.Value
|
if mem == nil {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
v, ok := mem(c.MemAddr)
|
||||||
|
if !ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
expected = v
|
||||||
default:
|
default:
|
||||||
expected = c.Value
|
expected = c.Value
|
||||||
}
|
}
|
||||||
@@ -72,6 +87,18 @@ func (c *Condition) Eval(regs *Regs) bool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// String renders the condition for display.
|
||||||
|
func (c *Condition) String() string {
|
||||||
|
switch {
|
||||||
|
case c.Reg2 != "":
|
||||||
|
return fmt.Sprintf("%s %s %s", c.Reg, c.Op, c.Reg2)
|
||||||
|
case c.MemAddr != 0:
|
||||||
|
return fmt.Sprintf("%s %s *%#x", c.Reg, c.Op, c.MemAddr)
|
||||||
|
default:
|
||||||
|
return fmt.Sprintf("%s %s %#x", c.Reg, c.Op, c.Value)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Breakpoints manages the software breakpoints of one Session.
|
// Breakpoints manages the software breakpoints of one Session.
|
||||||
type Breakpoints struct {
|
type Breakpoints struct {
|
||||||
t tracer
|
t tracer
|
||||||
@@ -83,6 +110,18 @@ func NewBreakpoints(t tracer) *Breakpoints {
|
|||||||
return &Breakpoints{t: t, bps: make(map[uint64]*Breakpoint)}
|
return &Breakpoints{t: t, bps: make(map[uint64]*Breakpoint)}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// breakpointMask is the byte mask of the breakpoint instruction inside a
|
||||||
|
// peeked word: the low len(breakpointInsn) bytes, because every supported
|
||||||
|
// architecture is little-endian and patches the instruction at the lowest
|
||||||
|
// address of the word.
|
||||||
|
func breakpointMask() uint64 {
|
||||||
|
var mask uint64
|
||||||
|
for range breakpointInsn {
|
||||||
|
mask = (mask << 8) | 0xFF
|
||||||
|
}
|
||||||
|
return mask
|
||||||
|
}
|
||||||
|
|
||||||
// Set installs a breakpoint at addr (replaces any existing one).
|
// Set installs a breakpoint at addr (replaces any existing one).
|
||||||
func (bm *Breakpoints) Set(addr uint64, label string) (*Breakpoint, error) {
|
func (bm *Breakpoints) Set(addr uint64, label string) (*Breakpoint, error) {
|
||||||
return bm.SetWithCond(addr, label, nil)
|
return bm.SetWithCond(addr, label, nil)
|
||||||
@@ -100,13 +139,12 @@ func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
orig := byte(word)
|
orig := make([]byte, len(breakpointInsn))
|
||||||
// Patch with the breakpoint instruction, preserving the rest of the word.
|
for i := range orig {
|
||||||
mask := uint64(0)
|
orig[i] = byte(word >> (8 * i))
|
||||||
for range breakpointInsn {
|
|
||||||
mask = (mask << 8) | 0xFF
|
|
||||||
}
|
}
|
||||||
patched := (word &^ mask) | breakpointWord(breakpointInsn)
|
// Patch with the breakpoint instruction, preserving the rest of the word.
|
||||||
|
patched := (word &^ breakpointMask()) | breakpointWord(breakpointInsn)
|
||||||
if err := bm.t.Poke(addr, patched); err != nil {
|
if err := bm.t.Poke(addr, patched); err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
@@ -115,15 +153,16 @@ func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (
|
|||||||
return bp, nil
|
return bp, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// Info returns a formatted list of all breakpoints.
|
// Info returns a formatted list of all breakpoints, ordered by address so
|
||||||
|
// the numbering is stable across calls (map iteration order is not).
|
||||||
func (bm *Breakpoints) Info() string {
|
func (bm *Breakpoints) Info() string {
|
||||||
if len(bm.bps) == 0 {
|
if len(bm.bps) == 0 {
|
||||||
return "no breakpoints set\n"
|
return "no breakpoints set\n"
|
||||||
}
|
}
|
||||||
var result strings.Builder
|
var result strings.Builder
|
||||||
i := 0
|
bps := bm.All()
|
||||||
for _, bp := range bm.bps {
|
slices.SortFunc(bps, func(a, b *Breakpoint) int { return cmp.Compare(a.Addr, b.Addr) })
|
||||||
i++
|
for i, bp := range bps {
|
||||||
status := "enabled"
|
status := "enabled"
|
||||||
if !bp.Enabled {
|
if !bp.Enabled {
|
||||||
status = "disabled"
|
status = "disabled"
|
||||||
@@ -134,26 +173,40 @@ func (bm *Breakpoints) Info() string {
|
|||||||
}
|
}
|
||||||
cond := ""
|
cond := ""
|
||||||
if bp.Cond != nil {
|
if bp.Cond != nil {
|
||||||
cond = fmt.Sprintf(" if %s %s %#x", bp.Cond.Reg, bp.Cond.Op, bp.Cond.Value)
|
cond = " if " + bp.Cond.String()
|
||||||
}
|
}
|
||||||
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond))
|
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i+1, label, bp.Addr, status, bp.hits, cond))
|
||||||
}
|
}
|
||||||
return result.String()
|
return result.String()
|
||||||
}
|
}
|
||||||
|
|
||||||
// Clear removes the breakpoint at addr, restoring the original byte.
|
// restore writes the saved original bytes back over the breakpoint
|
||||||
|
// instruction, preserving the rest of the peeked word. It reports whether
|
||||||
|
// both the peek and the poke succeeded.
|
||||||
|
func (bm *Breakpoints) restore(addr uint64, bp *Breakpoint) bool {
|
||||||
|
word, err := bm.t.Peek(addr)
|
||||||
|
if err != nil {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
orig := uint64(0)
|
||||||
|
for i, b := range bp.Orig {
|
||||||
|
orig |= uint64(b) << (8 * i)
|
||||||
|
}
|
||||||
|
return bm.t.Poke(addr, (word&^breakpointMask())|orig) == nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Clear removes the breakpoint at addr, restoring the original bytes.
|
||||||
func (bm *Breakpoints) Clear(addr uint64) error {
|
func (bm *Breakpoints) Clear(addr uint64) error {
|
||||||
bp, ok := bm.bps[addr]
|
bp, ok := bm.bps[addr]
|
||||||
if !ok {
|
if !ok {
|
||||||
return fmt.Errorf("debug: no breakpoint at %#x", addr)
|
return fmt.Errorf("debug: no breakpoint at %#x", addr)
|
||||||
}
|
}
|
||||||
word, err := bm.t.Peek(addr)
|
if !bm.restore(addr, bp) {
|
||||||
if err != nil {
|
word, err := bm.t.Peek(addr)
|
||||||
return err
|
if err != nil {
|
||||||
}
|
return err
|
||||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
}
|
||||||
if err := bm.t.Poke(addr, restored); err != nil {
|
return fmt.Errorf("debug: restore breakpoint at %#x failed, word is %#x", addr, word)
|
||||||
return err
|
|
||||||
}
|
}
|
||||||
delete(bm.bps, addr)
|
delete(bm.bps, addr)
|
||||||
return nil
|
return nil
|
||||||
@@ -185,43 +238,73 @@ func (bm *Breakpoints) All() []*Breakpoint {
|
|||||||
|
|
||||||
// HandleTrap is called after the debuggee stops on SIGTRAP. It checks
|
// HandleTrap is called after the debuggee stops on SIGTRAP. It checks
|
||||||
// whether the trap was caused by one of our breakpoints (PC-adjust matches
|
// whether the trap was caused by one of our breakpoints (PC-adjust matches
|
||||||
// a breakpoint address), restores the original byte, rewinds PC, and
|
// a breakpoint address), restores the original bytes, rewinds PC, and
|
||||||
// returns the breakpoint that was hit (or nil if it was a single-step).
|
// returns the breakpoint that was hit (or nil if it was a single-step).
|
||||||
// Hits returns how many times the breakpoint has been hit.
|
// Hits returns how many times the breakpoint has been hit.
|
||||||
func (bp *Breakpoint) Hits() int { return bp.hits }
|
func (bp *Breakpoint) Hits() int { return bp.hits }
|
||||||
|
|
||||||
|
// TrapStray reports whether a stop is a breakpoint-class trap that matches
|
||||||
|
// no breakpoint of ours and cannot be resumed: the PC still stands on the
|
||||||
|
// trapping instruction (the kernel's own BRK, EBREAK or break, on an
|
||||||
|
// architecture that reports the trap in place), so the next resume would
|
||||||
|
// re-execute it and trap forever. trapPC is the PC the stop reported,
|
||||||
|
// before any HandleTrap rewinding; reason is the stop's StopInfo class.
|
||||||
|
// The debuggee's SIGSTOP barriers also stop without PC movement, and they
|
||||||
|
// never carry the breakpoint class, so they are unaffected.
|
||||||
|
func TrapStray(s *Session, reason StopReason, trapPC uint64) bool {
|
||||||
|
if reason != StopBreakpoint {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
after, err := s.GetRegs()
|
||||||
|
if err != nil {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
return after.GetPC() <= trapPC-uint64(breakpointPCAdjust)
|
||||||
|
}
|
||||||
|
|
||||||
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
|
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
|
||||||
// After a breakpoint trap, PC points past the breakpoint instruction.
|
// On amd64 the kernel reports the trap with RIP past the INT3; on the
|
||||||
|
// other supported architectures the PC still stands on the trap
|
||||||
|
// instruction, which breakpointPCAdjust encodes per architecture.
|
||||||
trapAddr := regs.GetPC() - uint64(breakpointPCAdjust)
|
trapAddr := regs.GetPC() - uint64(breakpointPCAdjust)
|
||||||
bp, ok := bm.bps[trapAddr]
|
bp, ok := bm.bps[trapAddr]
|
||||||
if !ok || !bp.Enabled {
|
if !ok || !bp.Enabled {
|
||||||
return nil // single-step trap or unknown
|
return nil // single-step trap or unknown
|
||||||
}
|
}
|
||||||
// Check the condition (if any).
|
// Check the condition (if any).
|
||||||
if bp.Cond != nil && !bp.Cond.Eval(regs) {
|
if bp.Cond != nil && !bp.Cond.Eval(regs, bm.peekValue) {
|
||||||
// Condition not met, restore the byte but do NOT rewind RIP.
|
// Condition not met: step the original instruction and re-arm the
|
||||||
// The process continues from the next instruction (past the INT3).
|
// breakpoint, leaving the debuggee stopped just past it, ready to
|
||||||
word, err := bm.t.Peek(trapAddr)
|
// resume silently. The PC must be rewound first: on architectures
|
||||||
if err == nil {
|
// that report the trap past the instruction (amd64) it would
|
||||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
// otherwise sit on the second byte of the replaced instruction.
|
||||||
bm.t.Poke(trapAddr, restored)
|
if !bm.restore(trapAddr, bp) {
|
||||||
|
return nil
|
||||||
}
|
}
|
||||||
// RIP is already past the INT3 (trapAddr + 1). Don't rewind.
|
regs.SetPC(trapAddr)
|
||||||
|
if err := bm.t.SetRegs(regs); err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if err := bm.t.Step(); err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
bm.Reinsert(trapAddr)
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
bp.hits++
|
bp.hits++
|
||||||
// Restore the original byte.
|
// Restore the original bytes and rewind PC to re-execute them.
|
||||||
word, err := bm.t.Peek(trapAddr)
|
bm.restore(trapAddr, bp)
|
||||||
if err == nil {
|
|
||||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
|
||||||
bm.t.Poke(trapAddr, restored)
|
|
||||||
}
|
|
||||||
// Rewind PC to re-execute the original instruction.
|
|
||||||
regs.SetPC(trapAddr)
|
regs.SetPC(trapAddr)
|
||||||
bm.t.SetRegs(regs)
|
bm.t.SetRegs(regs)
|
||||||
return bp
|
return bp
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// peekValue adapts tracer.Peek to the Condition value reader.
|
||||||
|
func (bm *Breakpoints) peekValue(addr uint64) (uint64, bool) {
|
||||||
|
v, err := bm.t.Peek(addr)
|
||||||
|
return v, err == nil
|
||||||
|
}
|
||||||
|
|
||||||
// Reinsert re-inserts the breakpoint at addr after a single-step past it.
|
// Reinsert re-inserts the breakpoint at addr after a single-step past it.
|
||||||
// Called after Step() when we want the breakpoint to fire again on the
|
// Called after Step() when we want the breakpoint to fire again on the
|
||||||
// next Continue().
|
// next Continue().
|
||||||
@@ -234,11 +317,7 @@ func (bm *Breakpoints) Reinsert(addr uint64) error {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
mask := uint64(0)
|
patched := (word &^ breakpointMask()) | breakpointWord(breakpointInsn)
|
||||||
for range breakpointInsn {
|
|
||||||
mask = (mask << 8) | 0xFF
|
|
||||||
}
|
|
||||||
patched := (word &^ mask) | breakpointWord(breakpointInsn)
|
|
||||||
return bm.t.Poke(addr, patched)
|
return bm.t.Poke(addr, patched)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,309 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build linux
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
// Architecture-neutral tests: label and line tables, and the breakpoint
|
||||||
|
// manager against the mock tracer. These do not launch a debuggee, so they
|
||||||
|
// build on every supported linux architecture.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"slices"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestLineAt(t *testing.T) {
|
||||||
|
lines := []SourceLine{
|
||||||
|
{Offset: 0, Line: 5},
|
||||||
|
{Offset: 5, Line: 6},
|
||||||
|
{Offset: 10, Line: 7},
|
||||||
|
{Offset: 15, Line: 8},
|
||||||
|
}
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
offset int
|
||||||
|
want int
|
||||||
|
}{
|
||||||
|
{0, 5},
|
||||||
|
{1, 5},
|
||||||
|
{4, 5},
|
||||||
|
{5, 6},
|
||||||
|
{7, 6},
|
||||||
|
{10, 7},
|
||||||
|
{12, 7},
|
||||||
|
{15, 8},
|
||||||
|
{20, 8},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
got := lineAt(lines, tt.offset)
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Empty table.
|
||||||
|
if lineAt(nil, 5) != 0 {
|
||||||
|
t.Error("lineAt(nil, 5) should return 0")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestOffsetForLine(t *testing.T) {
|
||||||
|
lines := []SourceLine{
|
||||||
|
{Offset: 0, Line: 5},
|
||||||
|
{Offset: 5, Line: 6},
|
||||||
|
{Offset: 10, Line: 7},
|
||||||
|
}
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
line int
|
||||||
|
want int
|
||||||
|
}{
|
||||||
|
{5, 0},
|
||||||
|
{6, 5},
|
||||||
|
{7, 10},
|
||||||
|
{99, -1}, // not found
|
||||||
|
{0, -1}, // not found
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
got := offsetForLine(lines, tt.line)
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestNearestLabel(t *testing.T) {
|
||||||
|
labels := []Label{
|
||||||
|
{Name: "start", Offset: 0},
|
||||||
|
{Name: "loop", Offset: 10},
|
||||||
|
{Name: "done", Offset: 20},
|
||||||
|
}
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
offset int
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{0, "start"},
|
||||||
|
{5, "start"},
|
||||||
|
{10, "loop"},
|
||||||
|
{15, "loop"},
|
||||||
|
{20, "done"},
|
||||||
|
{25, "done"},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
got := nearestLabel(labels, tt.offset)
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBreakpointsSetAndClear(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
|
||||||
|
// Set a breakpoint at address 0x1000.
|
||||||
|
bp, err := bm.Set(0x1000, "test")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Set: %v", err)
|
||||||
|
}
|
||||||
|
if !bp.Enabled {
|
||||||
|
t.Error("breakpoint not enabled")
|
||||||
|
}
|
||||||
|
if bp.Label != "test" {
|
||||||
|
t.Errorf("label = %q, want test", bp.Label)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Verify Peek was called.
|
||||||
|
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
|
||||||
|
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Verify Poke wrote the breakpoint instruction's bytes.
|
||||||
|
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
|
||||||
|
t.Errorf("pokes = %v", tr.pokes)
|
||||||
|
}
|
||||||
|
if got := tr.pokes[0].val & breakpointMask(); got != breakpointWord(breakpointInsn) {
|
||||||
|
t.Errorf("patched bytes %#x, want %#x", got, breakpointWord(breakpointInsn))
|
||||||
|
}
|
||||||
|
|
||||||
|
// At should find it.
|
||||||
|
if bm.At(0x1000) == nil {
|
||||||
|
t.Error("At(0x1000) returned nil")
|
||||||
|
}
|
||||||
|
|
||||||
|
// All should return it.
|
||||||
|
all := bm.All()
|
||||||
|
if len(all) != 1 {
|
||||||
|
t.Errorf("All() = %d breakpoints, want 1", len(all))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Clear it.
|
||||||
|
if err := bm.Clear(0x1000); err != nil {
|
||||||
|
t.Fatalf("Clear: %v", err)
|
||||||
|
}
|
||||||
|
if bm.At(0x1000) != nil {
|
||||||
|
t.Error("At(0x1000) after Clear should be nil")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestBreakpointRestoreWidth proves the restore path writes back every
|
||||||
|
// byte of the breakpoint instruction's width, not just the first byte: on
|
||||||
|
// arm64, riscv64 and loong64 the instruction is four bytes, and restoring
|
||||||
|
// one byte would leave three bytes of the trap instruction in place.
|
||||||
|
func TestBreakpointRestoreWidth(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
tr.mem[0x3000] = 0x11
|
||||||
|
tr.mem[0x3001] = 0x22
|
||||||
|
tr.mem[0x3002] = 0x33
|
||||||
|
tr.mem[0x3003] = 0x44
|
||||||
|
|
||||||
|
if _, err := bm.Set(0x3000, "width"); err != nil {
|
||||||
|
t.Fatalf("Set: %v", err)
|
||||||
|
}
|
||||||
|
for i, b := range breakpointInsn {
|
||||||
|
if tr.mem[0x3000+uint64(i)] != b {
|
||||||
|
t.Fatalf("byte %d after Set = %#x, want the breakpoint byte %#x", i, tr.mem[0x3000+uint64(i)], b)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(bm.At(0x3000).Orig) != len(breakpointInsn) {
|
||||||
|
t.Fatalf("Orig holds %d bytes, want %d", len(bm.At(0x3000).Orig), len(breakpointInsn))
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := bm.Clear(0x3000); err != nil {
|
||||||
|
t.Fatalf("Clear: %v", err)
|
||||||
|
}
|
||||||
|
want := []byte{0x11, 0x22, 0x33, 0x44}
|
||||||
|
for i, b := range want {
|
||||||
|
if tr.mem[0x3000+uint64(i)] != b {
|
||||||
|
t.Errorf("byte %d after Clear = %#x, want %#x (restore must cover the full instruction width)", i, tr.mem[0x3000+uint64(i)], b)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBreakpointsSetWithCond(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
|
||||||
|
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
|
||||||
|
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("SetWithCond: %v", err)
|
||||||
|
}
|
||||||
|
if bp.Cond == nil || bp.Cond.Value != 42 {
|
||||||
|
t.Error("condition not set")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Re-setting the same address should update the condition.
|
||||||
|
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
|
||||||
|
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("SetWithCond (update): %v", err)
|
||||||
|
}
|
||||||
|
if bp2.Cond.Value != 100 {
|
||||||
|
t.Error("condition not updated")
|
||||||
|
}
|
||||||
|
// Should have only 1 Peek (first Set), second is update (no Peek needed).
|
||||||
|
if len(tr.peeks) != 1 {
|
||||||
|
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBreakpointsClearAll(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
|
||||||
|
bm.Set(0x1000, "a")
|
||||||
|
bm.Set(0x2000, "b")
|
||||||
|
bm.Set(0x3000, "c")
|
||||||
|
|
||||||
|
if len(bm.All()) != 3 {
|
||||||
|
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
|
||||||
|
}
|
||||||
|
|
||||||
|
bm.ClearAll()
|
||||||
|
if len(bm.All()) != 0 {
|
||||||
|
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBreakpointInfo(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
bm.Set(0x4000, "info_test")
|
||||||
|
|
||||||
|
info := bm.Info()
|
||||||
|
if info == "" {
|
||||||
|
t.Error("Info returned empty string")
|
||||||
|
}
|
||||||
|
if !strings.Contains(info, "info_test") {
|
||||||
|
t.Errorf("Info %q does not contain label", info)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestConditionString covers the display of all three condition forms.
|
||||||
|
func TestConditionString(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
cond Condition
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{Condition{Reg: "rax", Op: "==", Value: 42}, "rax == 0x2a"},
|
||||||
|
{Condition{Reg: "rax", Op: "!=", Reg2: "rbx"}, "rax != rbx"},
|
||||||
|
{Condition{Reg: "rax", Op: "<", MemAddr: 0x5000}, "rax < *0x5000"},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
if got := tt.cond.String(); got != tt.want {
|
||||||
|
t.Errorf("Condition.String() = %q, want %q", got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestBreakpointsInfoOrdered proves the listing is ordered by address: the
|
||||||
|
// numbers it prints are map keys rendered in iteration order otherwise, so
|
||||||
|
// the same set of breakpoints would renumber itself between calls.
|
||||||
|
func TestBreakpointsInfoOrdered(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
addrs := []uint64{0x9000, 0x1000, 0x7000, 0x3000, 0x8000, 0x2000,
|
||||||
|
0x6000, 0x4000, 0x5000, 0xa000}
|
||||||
|
for i, a := range addrs {
|
||||||
|
if _, err := bm.Set(a, fmt.Sprintf("bp%d", i)); err != nil {
|
||||||
|
t.Fatalf("Set(%#x): %v", a, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
sorted := append([]uint64(nil), addrs...)
|
||||||
|
slices.Sort(sorted)
|
||||||
|
info := bm.Info()
|
||||||
|
for i, a := range sorted {
|
||||||
|
want := fmt.Sprintf(" %d: bp%d at %#x", i+1, indexOf(addrs, a), a)
|
||||||
|
if !strings.Contains(info, want) {
|
||||||
|
t.Errorf("Info() missing %q; listing:\n%s", want, info)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The numbers themselves must ascend: "1:" before "2" ... "10".
|
||||||
|
pos := 0
|
||||||
|
for i := range len(addrs) {
|
||||||
|
next := strings.Index(info[pos:], fmt.Sprintf(" %d: ", i+1))
|
||||||
|
if next < 0 {
|
||||||
|
t.Fatalf("Info() has no entry %d; listing:\n%s", i+1, info)
|
||||||
|
}
|
||||||
|
pos += next
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func indexOf(addrs []uint64, a uint64) int {
|
||||||
|
for i, v := range addrs {
|
||||||
|
if v == a {
|
||||||
|
return i
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return -1
|
||||||
|
}
|
||||||
@@ -0,0 +1,325 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build linux && amd64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"runtime"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Regression tests for the debugger audit: memory access at mapping
|
||||||
|
// boundaries, watchpoint slot attribution, launch failure latency, stray
|
||||||
|
// trap instructions and the REPL's argument validation. All drive a real
|
||||||
|
// ptrace session, so they run on amd64 hosts only.
|
||||||
|
|
||||||
|
// memMap is one line of /proc/pid/maps.
|
||||||
|
type memMap struct {
|
||||||
|
lo, hi uint64
|
||||||
|
perms string
|
||||||
|
name string
|
||||||
|
}
|
||||||
|
|
||||||
|
// readMaps parses the debuggee's memory map.
|
||||||
|
func readMaps(t *testing.T, pid int) []memMap {
|
||||||
|
t.Helper()
|
||||||
|
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read maps: %v", err)
|
||||||
|
}
|
||||||
|
var out []memMap
|
||||||
|
for line := range strings.SplitSeq(string(data), "\n") {
|
||||||
|
fields := strings.Fields(line)
|
||||||
|
if len(fields) < 2 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
var lo, hi uint64
|
||||||
|
if _, err := fmt.Sscanf(fields[0], "%x-%x", &lo, &hi); err != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
m := memMap{lo: lo, hi: hi, perms: fields[1]}
|
||||||
|
if len(fields) >= 6 {
|
||||||
|
m.name = fields[5]
|
||||||
|
}
|
||||||
|
out = append(out, m)
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// boundaryByte returns the last byte of a writable, ordinary mapping that is
|
||||||
|
// followed by an unmapped gap: an access there is inside the mapping, while
|
||||||
|
// the 8-byte word starting at it crosses into unmapped memory.
|
||||||
|
func boundaryByte(t *testing.T, pid int) uint64 {
|
||||||
|
t.Helper()
|
||||||
|
maps := readMaps(t, pid)
|
||||||
|
for i, m := range maps {
|
||||||
|
if !strings.Contains(m.perms, "rw") ||
|
||||||
|
strings.Contains(m.name, "vvar") || strings.Contains(m.name, "vdso") ||
|
||||||
|
strings.Contains(m.name, "vsyscall") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
gap := uint64(1) << 62
|
||||||
|
if i+1 < len(maps) {
|
||||||
|
gap = maps[i+1].lo - m.hi
|
||||||
|
}
|
||||||
|
if gap >= 4096 {
|
||||||
|
return m.hi - 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Skip("no writable mapping followed by a hole; cannot construct the boundary")
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestReadMemoryPageBoundary proves ReadMemory never reads past the requested
|
||||||
|
// range: one byte at the end of a mapping followed by a hole must be
|
||||||
|
// readable, which the old word-at-a-time tail read failed because its final
|
||||||
|
// 8-byte Peek crossed into the unmapped page.
|
||||||
|
func TestReadMemoryPageBoundary(t *testing.T) {
|
||||||
|
sess, _, _ := launchKernel(t, buildGasm(t), boundaryKernel(t), "boundary", nil)
|
||||||
|
addr := boundaryByte(t, sess.Pid())
|
||||||
|
mem, err := sess.ReadMemory(addr, 1)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ReadMemory(%#x, 1): %v (the read must not cross into the unmapped page)", addr, err)
|
||||||
|
}
|
||||||
|
if len(mem) != 1 {
|
||||||
|
t.Fatalf("ReadMemory returned %d bytes, want 1", len(mem))
|
||||||
|
}
|
||||||
|
// A request whose own range crosses into the hole must still fail.
|
||||||
|
if _, err := sess.ReadMemory(addr, 8); err == nil {
|
||||||
|
t.Fatal("ReadMemory past the mapping end should fail")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDisassemblePageBoundary proves the disassembler shrinks its read
|
||||||
|
// window at a mapping end instead of failing: the instruction stream cannot
|
||||||
|
// be decoded at all when the fixed 15-byte read crosses into the hole.
|
||||||
|
func TestDisassemblePageBoundary(t *testing.T) {
|
||||||
|
sess, _, _ := launchKernel(t, buildGasm(t), boundaryKernel(t), "boundary", nil)
|
||||||
|
addr := boundaryByte(t, sess.Pid())
|
||||||
|
if _, _, err := sess.Disassemble(addr); err != nil {
|
||||||
|
t.Fatalf("Disassemble(%#x): %v (the read window must shrink at the mapping end)", addr, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestWriteMemoryPageBoundary proves WriteMemory writes exactly the bytes it
|
||||||
|
// is given: one byte at the end of a mapping followed by a hole must be
|
||||||
|
// writable, which the old read-modify-write of the final partial word failed
|
||||||
|
// because its Peek crossed into the unmapped page.
|
||||||
|
func TestWriteMemoryPageBoundary(t *testing.T) {
|
||||||
|
sess, _, _ := launchKernel(t, buildGasm(t), boundaryKernel(t), "boundary", nil)
|
||||||
|
addr := boundaryByte(t, sess.Pid())
|
||||||
|
orig, err := sess.ReadMemory(addr, 1)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ReadMemory(%#x, 1): %v", addr, err)
|
||||||
|
}
|
||||||
|
if err := sess.WriteMemory(addr, []byte{orig[0]}); err != nil {
|
||||||
|
t.Fatalf("WriteMemory(%#x, 1): %v (the write must not read past the range)", addr, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestWatchpointSlotAttribution proves a hit is attributed to the slot that
|
||||||
|
// fired, not to an earlier one whose DR6 status bit is still set: the B0-B3
|
||||||
|
// bits are sticky, so they must be acknowledged when read.
|
||||||
|
func TestWatchpointSlotAttribution(t *testing.T) {
|
||||||
|
bin := buildGasm(t)
|
||||||
|
const kernel = `#include "textflag.h"
|
||||||
|
|
||||||
|
// func wptwo(x, y int64) (a, b int64)
|
||||||
|
TEXT ·wptwo(SB), NOSPLIT, $0-32
|
||||||
|
MOVQ $0x1111, AX
|
||||||
|
MOVQ AX, a+16(FP)
|
||||||
|
MOVQ $0x2222, BX
|
||||||
|
MOVQ BX, b+24(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
path := writeKernel(t, kernel)
|
||||||
|
sess, bm, fl := launchKernel(t, bin, path, "wptwo", nil)
|
||||||
|
|
||||||
|
entry := sess.CodeBase() + uint64(fl.Offset)
|
||||||
|
if _, err := bm.Set(entry, "entry"); err != nil {
|
||||||
|
t.Fatalf("Set: %v", err)
|
||||||
|
}
|
||||||
|
runToEntry(t, sess, bm, entry)
|
||||||
|
regs, err := sess.GetRegs()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetRegs: %v", err)
|
||||||
|
}
|
||||||
|
// FP sits one word above the entry stack pointer (the return address
|
||||||
|
// occupies [RSP]), so a+16(FP) = RSP+24 and b+24(FP) = RSP+32.
|
||||||
|
watchA := regs.RSP + 24
|
||||||
|
watchB := regs.RSP + 32
|
||||||
|
|
||||||
|
if err := sess.SetWatchpoint(0, watchA, WatchWrite, 8); err != nil {
|
||||||
|
t.Fatalf("SetWatchpoint(0): %v", err)
|
||||||
|
}
|
||||||
|
if err := sess.SetWatchpoint(1, watchB, WatchWrite, 8); err != nil {
|
||||||
|
t.Fatalf("SetWatchpoint(1): %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
for i, want := range []uint64{watchA, watchB} {
|
||||||
|
if err := sess.Continue(); err != nil {
|
||||||
|
t.Fatalf("Continue (hit %d): %v", i+1, err)
|
||||||
|
}
|
||||||
|
reason, addr := sess.StopInfo()
|
||||||
|
if reason != StopWatchpoint {
|
||||||
|
t.Fatalf("hit %d: stop reason = %v, want StopWatchpoint", i+1, reason)
|
||||||
|
}
|
||||||
|
if addr != want {
|
||||||
|
t.Fatalf("hit %d reported %#x, want %#x (the sticky DR6 bit misattributes the slot)", i+1, addr, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Clearing a watchpoint must zero its address register: a stale
|
||||||
|
// address in a disabled slot turns any sticky status bit into a
|
||||||
|
// misattributed report later.
|
||||||
|
if err := sess.ClearWatchpoint(0); err != nil {
|
||||||
|
t.Fatalf("ClearWatchpoint(0): %v", err)
|
||||||
|
}
|
||||||
|
dr0, err := ptracePeekUser(sess.Pid(), drOffset)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read DR0: %v", err)
|
||||||
|
}
|
||||||
|
if dr0 != 0 {
|
||||||
|
t.Fatalf("DR0 = %#x after ClearWatchpoint, want 0 (the address register must be cleared)", dr0)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLaunchFailsFastOnDeadDebuggee proves a debuggee that dies before
|
||||||
|
// signalling readiness surfaces promptly: the ready poll used to run its
|
||||||
|
// full 2.5 seconds before the wait discovered the exit.
|
||||||
|
func TestLaunchFailsFastOnDeadDebuggee(t *testing.T) {
|
||||||
|
runtime.LockOSThread()
|
||||||
|
defer runtime.UnlockOSThread()
|
||||||
|
bin := buildGasm(t)
|
||||||
|
path := boundaryKernel(t)
|
||||||
|
start := time.Now()
|
||||||
|
sess, err := Launch(bin, path, "nosuchfunction", nil)
|
||||||
|
elapsed := time.Since(start)
|
||||||
|
if err == nil {
|
||||||
|
sess.Kill()
|
||||||
|
t.Fatal("Launch with an unknown function should fail")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), "before signalling readiness") &&
|
||||||
|
!strings.Contains(err.Error(), "debuggee exited") {
|
||||||
|
t.Errorf("error does not name the dead debuggee: %v", err)
|
||||||
|
}
|
||||||
|
if elapsed >= 1500*time.Millisecond {
|
||||||
|
t.Fatalf("Launch took %v to report the dead debuggee; the readiness poll must detect the exit, not time out", elapsed)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestStrayTrapRunsThrough proves the continue loop survives a trap
|
||||||
|
// instruction planted in the kernel itself (BYTE $0xCC, the same byte the
|
||||||
|
// debugger patches in): on architectures that report the trap in place the
|
||||||
|
// loop must surface the stop, and on amd64 it runs through to the exit. A
|
||||||
|
// regression here hangs, so a watchdog fails the run.
|
||||||
|
func TestStrayTrapRunsThrough(t *testing.T) {
|
||||||
|
bin := buildGasm(t)
|
||||||
|
const kernel = `#include "textflag.h"
|
||||||
|
|
||||||
|
// func stray() int64
|
||||||
|
TEXT ·stray(SB), NOSPLIT, $0-8
|
||||||
|
MOVQ $7, AX
|
||||||
|
BYTE $0xCC
|
||||||
|
MOVQ AX, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
path := writeKernel(t, kernel)
|
||||||
|
sess, bm, _ := launchKernel(t, bin, path, "stray", nil)
|
||||||
|
|
||||||
|
timer := time.AfterFunc(time.Minute, func() {
|
||||||
|
panic("watchdog: the continue loop hung on the stray trap instruction")
|
||||||
|
})
|
||||||
|
defer timer.Stop()
|
||||||
|
|
||||||
|
out := captureStdout(t, func() {
|
||||||
|
REPL(sess, bm, sess.CodeBase(), 0, 0, 0, nil, nil,
|
||||||
|
strings.NewReader("continue\nquit\n"))
|
||||||
|
})
|
||||||
|
if !strings.Contains(out, "debuggee exited") {
|
||||||
|
t.Errorf("the stray trap wedged the continue loop; output:\n%s", out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestStepIntoFaultReportsSignal proves the step command reports a genuine
|
||||||
|
// signal-delivery-stop instead of silently printing the faulting
|
||||||
|
// instruction as if the step had succeeded.
|
||||||
|
func TestStepIntoFaultReportsSignal(t *testing.T) {
|
||||||
|
bin := buildGasm(t)
|
||||||
|
const kernel = `#include "textflag.h"
|
||||||
|
|
||||||
|
// func crash() int64
|
||||||
|
TEXT ·crash(SB), NOSPLIT, $0-8
|
||||||
|
XORQ AX, AX
|
||||||
|
MOVQ (AX), AX
|
||||||
|
MOVQ AX, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
path := writeKernel(t, kernel)
|
||||||
|
sess, bm, fl := launchKernel(t, bin, path, "crash", nil)
|
||||||
|
entry := sess.CodeBase() + uint64(fl.Offset)
|
||||||
|
if _, err := bm.Set(entry, "entry"); err != nil {
|
||||||
|
t.Fatalf("Set: %v", err)
|
||||||
|
}
|
||||||
|
runToEntry(t, sess, bm, entry)
|
||||||
|
|
||||||
|
out := captureStdout(t, func() {
|
||||||
|
REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, nil, nil,
|
||||||
|
strings.NewReader("step 2\nquit\n"))
|
||||||
|
})
|
||||||
|
if !strings.Contains(out, "stopped on signal") {
|
||||||
|
t.Errorf("stepping into the fault did not report the signal; output:\n%s", out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestREPLRejectsBadArguments proves the command loop reports malformed
|
||||||
|
// input instead of silently defaulting: an unknown label for x would read
|
||||||
|
// address 0, and a malformed count would silently step one instruction.
|
||||||
|
func TestREPLRejectsBadArguments(t *testing.T) {
|
||||||
|
bin := buildGasm(t)
|
||||||
|
path := boundaryKernel(t)
|
||||||
|
sess, bm, fl := launchKernel(t, bin, path, "boundary", nil)
|
||||||
|
entry := sess.CodeBase() + uint64(fl.Offset)
|
||||||
|
if _, err := bm.Set(entry, "entry"); err != nil {
|
||||||
|
t.Fatalf("Set: %v", err)
|
||||||
|
}
|
||||||
|
runToEntry(t, sess, bm, entry)
|
||||||
|
|
||||||
|
out := captureStdout(t, func() {
|
||||||
|
REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, nil, nil,
|
||||||
|
strings.NewReader("x nosuchlabel\nstep abc\ndisas abc\nwatch 0x1000 q 8\nquit\n"))
|
||||||
|
})
|
||||||
|
for _, want := range []string{
|
||||||
|
"unknown address: nosuchlabel",
|
||||||
|
"invalid count: abc",
|
||||||
|
"unknown watchpoint type: q",
|
||||||
|
} {
|
||||||
|
if !strings.Contains(out, want) {
|
||||||
|
t.Errorf("output missing %q:\n%s", want, out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if got := strings.Count(out, "invalid count: abc"); got != 2 {
|
||||||
|
t.Errorf("invalid count reported %d times, want 2 (step and disas):\n%s", got, out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// boundaryKernel is a minimal kernel for the boundary tests, which only need
|
||||||
|
// a live, stopped debuggee.
|
||||||
|
func boundaryKernel(t *testing.T) string {
|
||||||
|
t.Helper()
|
||||||
|
const kernel = `#include "textflag.h"
|
||||||
|
|
||||||
|
// func boundary() int64
|
||||||
|
TEXT ·boundary(SB), NOSPLIT, $0-8
|
||||||
|
MOVQ $1, AX
|
||||||
|
MOVQ AX, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
return writeKernel(t, kernel)
|
||||||
|
}
|
||||||
+38
-188
@@ -6,7 +6,6 @@
|
|||||||
package debug
|
package debug
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"strings"
|
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -48,7 +47,7 @@ func TestConditionEval(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
for _, tt := range tests {
|
for _, tt := range tests {
|
||||||
got := tt.cond.Eval(regs)
|
got := tt.cond.Eval(regs, nil)
|
||||||
if got != tt.want {
|
if got != tt.want {
|
||||||
t.Errorf("Condition{%q %q %d}.Eval() = %v, want %v",
|
t.Errorf("Condition{%q %q %d}.Eval() = %v, want %v",
|
||||||
tt.cond.Reg, tt.cond.Op, tt.cond.Value, got, tt.want)
|
tt.cond.Reg, tt.cond.Op, tt.cond.Value, got, tt.want)
|
||||||
@@ -56,65 +55,33 @@ func TestConditionEval(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestLineAt(t *testing.T) {
|
// TestConditionEvalMem covers the register-memory form: the value is read
|
||||||
lines := []SourceLine{
|
// through the supplied reader, and a missing or failing reader must not
|
||||||
{Offset: 0, Line: 5},
|
// block the breakpoint.
|
||||||
{Offset: 5, Line: 6},
|
func TestConditionEvalMem(t *testing.T) {
|
||||||
{Offset: 10, Line: 7},
|
regs := &Regs{RAX: 7}
|
||||||
{Offset: 15, Line: 8},
|
mem := func(addr uint64) (uint64, bool) {
|
||||||
}
|
if addr == 0x5000 {
|
||||||
|
return 7, true
|
||||||
tests := []struct {
|
|
||||||
offset int
|
|
||||||
want int
|
|
||||||
}{
|
|
||||||
{0, 5},
|
|
||||||
{1, 5},
|
|
||||||
{4, 5},
|
|
||||||
{5, 6},
|
|
||||||
{7, 6},
|
|
||||||
{10, 7},
|
|
||||||
{12, 7},
|
|
||||||
{15, 8},
|
|
||||||
{20, 8},
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, tt := range tests {
|
|
||||||
got := lineAt(lines, tt.offset)
|
|
||||||
if got != tt.want {
|
|
||||||
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
|
|
||||||
}
|
}
|
||||||
|
return 0, false
|
||||||
}
|
}
|
||||||
|
|
||||||
// Empty table.
|
eq := Condition{Reg: "rax", Op: "==", MemAddr: 0x5000}
|
||||||
if lineAt(nil, 5) != 0 {
|
if !eq.Eval(regs, mem) {
|
||||||
t.Error("lineAt(nil, 5) should return 0")
|
t.Error("register-memory comparison with matching word should hold")
|
||||||
}
|
}
|
||||||
}
|
ne := Condition{Reg: "rax", Op: "!=", MemAddr: 0x5000}
|
||||||
|
if ne.Eval(regs, mem) {
|
||||||
func TestOffsetForLine(t *testing.T) {
|
t.Error("register-memory comparison with mismatching word should not hold")
|
||||||
lines := []SourceLine{
|
|
||||||
{Offset: 0, Line: 5},
|
|
||||||
{Offset: 5, Line: 6},
|
|
||||||
{Offset: 10, Line: 7},
|
|
||||||
}
|
}
|
||||||
|
bad := Condition{Reg: "rax", Op: "==", MemAddr: 0x6000}
|
||||||
tests := []struct {
|
if !bad.Eval(regs, mem) {
|
||||||
line int
|
t.Error("unreadable memory must not block the breakpoint")
|
||||||
want int
|
|
||||||
}{
|
|
||||||
{5, 0},
|
|
||||||
{6, 5},
|
|
||||||
{7, 10},
|
|
||||||
{99, -1}, // not found
|
|
||||||
{0, -1}, // not found
|
|
||||||
}
|
}
|
||||||
|
noReader := Condition{Reg: "rax", Op: "==", MemAddr: 0x5000}
|
||||||
for _, tt := range tests {
|
if !noReader.Eval(regs, nil) {
|
||||||
got := offsetForLine(lines, tt.line)
|
t.Error("missing memory reader must not block the breakpoint")
|
||||||
if got != tt.want {
|
|
||||||
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -141,139 +108,6 @@ func TestDecodeRflags(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestNearestLabel(t *testing.T) {
|
|
||||||
labels := []Label{
|
|
||||||
{Name: "start", Offset: 0},
|
|
||||||
{Name: "loop", Offset: 10},
|
|
||||||
{Name: "done", Offset: 20},
|
|
||||||
}
|
|
||||||
|
|
||||||
tests := []struct {
|
|
||||||
offset int
|
|
||||||
want string
|
|
||||||
}{
|
|
||||||
{0, "start"},
|
|
||||||
{5, "start"},
|
|
||||||
{10, "loop"},
|
|
||||||
{15, "loop"},
|
|
||||||
{20, "done"},
|
|
||||||
{25, "done"},
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, tt := range tests {
|
|
||||||
got := nearestLabel(labels, tt.offset)
|
|
||||||
if got != tt.want {
|
|
||||||
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestBreakpointsSetAndClear(t *testing.T) {
|
|
||||||
tr := newMockTracer()
|
|
||||||
bm := NewBreakpoints(tr)
|
|
||||||
|
|
||||||
// Set a breakpoint at address 0x1000.
|
|
||||||
bp, err := bm.Set(0x1000, "test")
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("Set: %v", err)
|
|
||||||
}
|
|
||||||
if !bp.Enabled {
|
|
||||||
t.Error("breakpoint not enabled")
|
|
||||||
}
|
|
||||||
if bp.Label != "test" {
|
|
||||||
t.Errorf("label = %q, want test", bp.Label)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Verify Peek was called.
|
|
||||||
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
|
|
||||||
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Verify Poke wrote INT3.
|
|
||||||
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
|
|
||||||
t.Errorf("pokes = %v", tr.pokes)
|
|
||||||
}
|
|
||||||
|
|
||||||
// At should find it.
|
|
||||||
if bm.At(0x1000) == nil {
|
|
||||||
t.Error("At(0x1000) returned nil")
|
|
||||||
}
|
|
||||||
|
|
||||||
// All should return it.
|
|
||||||
all := bm.All()
|
|
||||||
if len(all) != 1 {
|
|
||||||
t.Errorf("All() = %d breakpoints, want 1", len(all))
|
|
||||||
}
|
|
||||||
|
|
||||||
// Clear it.
|
|
||||||
if err := bm.Clear(0x1000); err != nil {
|
|
||||||
t.Fatalf("Clear: %v", err)
|
|
||||||
}
|
|
||||||
if bm.At(0x1000) != nil {
|
|
||||||
t.Error("At(0x1000) after Clear should be nil")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestBreakpointsSetWithCond(t *testing.T) {
|
|
||||||
tr := newMockTracer()
|
|
||||||
bm := NewBreakpoints(tr)
|
|
||||||
|
|
||||||
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
|
|
||||||
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("SetWithCond: %v", err)
|
|
||||||
}
|
|
||||||
if bp.Cond == nil || bp.Cond.Value != 42 {
|
|
||||||
t.Error("condition not set")
|
|
||||||
}
|
|
||||||
|
|
||||||
// Re-setting the same address should update the condition.
|
|
||||||
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
|
|
||||||
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("SetWithCond (update): %v", err)
|
|
||||||
}
|
|
||||||
if bp2.Cond.Value != 100 {
|
|
||||||
t.Error("condition not updated")
|
|
||||||
}
|
|
||||||
// Should have only 1 Peek (first Set), second is update (no Peek needed).
|
|
||||||
if len(tr.peeks) != 1 {
|
|
||||||
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestBreakpointsClearAll(t *testing.T) {
|
|
||||||
tr := newMockTracer()
|
|
||||||
bm := NewBreakpoints(tr)
|
|
||||||
|
|
||||||
bm.Set(0x1000, "a")
|
|
||||||
bm.Set(0x2000, "b")
|
|
||||||
bm.Set(0x3000, "c")
|
|
||||||
|
|
||||||
if len(bm.All()) != 3 {
|
|
||||||
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
|
|
||||||
}
|
|
||||||
|
|
||||||
bm.ClearAll()
|
|
||||||
if len(bm.All()) != 0 {
|
|
||||||
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestBreakpointInfo(t *testing.T) {
|
|
||||||
tr := newMockTracer()
|
|
||||||
bm := NewBreakpoints(tr)
|
|
||||||
bm.Set(0x4000, "info_test")
|
|
||||||
|
|
||||||
info := bm.Info()
|
|
||||||
if info == "" {
|
|
||||||
t.Error("Info returned empty string")
|
|
||||||
}
|
|
||||||
if !strings.Contains(info, "info_test") {
|
|
||||||
t.Errorf("Info %q does not contain label", info)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestWatchpointSlotTracking(t *testing.T) {
|
func TestWatchpointSlotTracking(t *testing.T) {
|
||||||
s := &Session{} // per-session slots start free
|
s := &Session{} // per-session slots start free
|
||||||
|
|
||||||
@@ -323,3 +157,19 @@ func TestWatchpointSlotTracking(t *testing.T) {
|
|||||||
t.Errorf("FindFreeWatchpointSlot() with all slots used = %d, want -1", got)
|
t.Errorf("FindFreeWatchpointSlot() with all slots used = %d, want -1", got)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestUnwatchSlotBound checks the bound the REPL parses against: it must
|
||||||
|
// cover the architecture's whole slot range, not a hardcoded 0-3.
|
||||||
|
func TestUnwatchSlotBound(t *testing.T) {
|
||||||
|
max := maxWatchpoints()
|
||||||
|
if max < 4 {
|
||||||
|
t.Fatalf("maxWatchpoints() = %d, want at least 4", max)
|
||||||
|
}
|
||||||
|
s := &Session{}
|
||||||
|
if s.IsWatchpointSlotUsed(max - 1) {
|
||||||
|
t.Errorf("slot %d should be free initially", max-1)
|
||||||
|
}
|
||||||
|
if s.IsWatchpointSlotUsed(max) {
|
||||||
|
t.Errorf("slot %d must be out of range", max)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,65 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && amd64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||||
|
// memory and returns its text representation and length in bytes. An amd64
|
||||||
|
// instruction is up to 15 bytes long, but the read must not reach past the
|
||||||
|
// end of the mapping: when the full 15-byte window crosses into unmapped
|
||||||
|
// memory the window shrinks, because an instruction at the mapping's end is
|
||||||
|
// by construction no longer than the readable bytes that hold it.
|
||||||
|
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||||
|
var lastErr error
|
||||||
|
for _, n := range []int{15, 8, 4, 2, 1} {
|
||||||
|
mem, err := s.ReadMemory(addr, n)
|
||||||
|
if err != nil {
|
||||||
|
lastErr = err
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
ins, derr := disasm.Decode(arch.AMD64, mem, addr)
|
||||||
|
if derr != nil {
|
||||||
|
return "", 0, derr
|
||||||
|
}
|
||||||
|
return ins.Text, ins.Len, nil
|
||||||
|
}
|
||||||
|
return "", 0, lastErr
|
||||||
|
}
|
||||||
|
|
||||||
|
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||||
|
// them as a formatted string with addresses and byte offsets.
|
||||||
|
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||||
|
var result strings.Builder
|
||||||
|
pc := addr
|
||||||
|
for range n {
|
||||||
|
text, length, err := s.Disassemble(pc)
|
||||||
|
if err != nil {
|
||||||
|
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
|
||||||
|
break
|
||||||
|
}
|
||||||
|
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
|
||||||
|
if length == 0 {
|
||||||
|
length = 1
|
||||||
|
}
|
||||||
|
pc += uint64(length)
|
||||||
|
}
|
||||||
|
return result.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// isCallInsn reports whether disassembled text (x86asm.IntelSyntax) is a
|
||||||
|
// call. The first token must match exactly: a prefix test would also catch
|
||||||
|
// unrelated mnemonics.
|
||||||
|
func isCallInsn(text string) bool {
|
||||||
|
m, _, _ := strings.Cut(text, " ")
|
||||||
|
return strings.ToLower(m) == "call"
|
||||||
|
}
|
||||||
@@ -0,0 +1,60 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && arm64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||||
|
// memory and returns its text representation and length in bytes.
|
||||||
|
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||||
|
mem, err := s.ReadMemory(addr, 4)
|
||||||
|
if err != nil {
|
||||||
|
return "", 0, err
|
||||||
|
}
|
||||||
|
ins, err := disasm.Decode(arch.ARM64, mem, addr)
|
||||||
|
if err != nil {
|
||||||
|
return "", 0, err
|
||||||
|
}
|
||||||
|
return ins.Text, ins.Len, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||||
|
// them as a formatted string with addresses and byte offsets.
|
||||||
|
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||||
|
var result strings.Builder
|
||||||
|
pc := addr
|
||||||
|
for range n {
|
||||||
|
text, length, err := s.Disassemble(pc)
|
||||||
|
if err != nil {
|
||||||
|
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
|
||||||
|
break
|
||||||
|
}
|
||||||
|
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
|
||||||
|
if length == 0 {
|
||||||
|
length = 1
|
||||||
|
}
|
||||||
|
pc += uint64(length)
|
||||||
|
}
|
||||||
|
return result.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// isCallInsn reports whether disassembled text (arm64asm.GoSyntax) is a
|
||||||
|
// call. GoSyntax renders bl as CALL; the native mnemonic is accepted too.
|
||||||
|
// The first token must match exactly so branches never match.
|
||||||
|
func isCallInsn(text string) bool {
|
||||||
|
m, _, _ := strings.Cut(text, " ")
|
||||||
|
switch strings.ToLower(m) {
|
||||||
|
case "call", "bl":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
@@ -0,0 +1,70 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && riscv64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||||
|
// memory and returns its text representation and length in bytes. The read
|
||||||
|
// shrinks from 4 to 2 bytes when the full word crosses into unmapped memory:
|
||||||
|
// a compressed instruction at the mapping's end still fits the shorter
|
||||||
|
// window, and an instruction can never extend past the mapping that holds it.
|
||||||
|
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||||
|
var lastErr error
|
||||||
|
for _, n := range []int{4, 2} {
|
||||||
|
mem, err := s.ReadMemory(addr, n)
|
||||||
|
if err != nil {
|
||||||
|
lastErr = err
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
ins, derr := disasm.Decode(arch.RISCV, mem, addr)
|
||||||
|
if derr != nil {
|
||||||
|
return "", 0, derr
|
||||||
|
}
|
||||||
|
return ins.Text, ins.Len, nil
|
||||||
|
}
|
||||||
|
return "", 0, lastErr
|
||||||
|
}
|
||||||
|
|
||||||
|
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||||
|
// them as a formatted string with addresses and byte offsets.
|
||||||
|
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||||
|
var result strings.Builder
|
||||||
|
pc := addr
|
||||||
|
for range n {
|
||||||
|
text, length, err := s.Disassemble(pc)
|
||||||
|
if err != nil {
|
||||||
|
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
|
||||||
|
break
|
||||||
|
}
|
||||||
|
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
|
||||||
|
if length == 0 {
|
||||||
|
length = 1
|
||||||
|
}
|
||||||
|
pc += uint64(length)
|
||||||
|
}
|
||||||
|
return result.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// isCallInsn reports whether disassembled text (riscv64asm.GoSyntax) is a
|
||||||
|
// call. GoSyntax renders jal and jalr calls as CALL; the native mnemonics
|
||||||
|
// are accepted too. The first token must match exactly: a prefix test on
|
||||||
|
// "bl" would catch branches on other architectures, and jalr as ret prints
|
||||||
|
// RET, which must not be stepped over.
|
||||||
|
func isCallInsn(text string) bool {
|
||||||
|
m, _, _ := strings.Cut(text, " ")
|
||||||
|
switch strings.ToLower(m) {
|
||||||
|
case "call", "jal", "jalr":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
+28
-11
@@ -9,22 +9,31 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||||
// memory and returns its text representation and length in bytes.
|
// memory and returns its text representation and length in bytes. An amd64
|
||||||
|
// instruction is up to 15 bytes long, but the read must not reach past the
|
||||||
|
// end of the mapping: when the full 15-byte window crosses into unmapped
|
||||||
|
// memory the window shrinks, because an instruction at the mapping's end is
|
||||||
|
// by construction no longer than the readable bytes that hold it.
|
||||||
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||||
mem, err := s.ReadMemory(addr, 15)
|
var lastErr error
|
||||||
if err != nil {
|
for _, n := range []int{15, 8, 4, 2, 1} {
|
||||||
return "", 0, err
|
mem, err := s.ReadMemory(addr, n)
|
||||||
|
if err != nil {
|
||||||
|
lastErr = err
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
ins, derr := disasm.Decode(arch.AMD64, mem, addr)
|
||||||
|
if derr != nil {
|
||||||
|
return "", 0, derr
|
||||||
|
}
|
||||||
|
return ins.Text, ins.Len, nil
|
||||||
}
|
}
|
||||||
ins, err := disasm.Decode(arch.AMD64, mem, addr)
|
return "", 0, lastErr
|
||||||
if err != nil {
|
|
||||||
return "", 0, err
|
|
||||||
}
|
|
||||||
return ins.Text, ins.Len, nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// DisassembleN decodes up to n instructions starting at addr and returns
|
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||||
@@ -46,3 +55,11 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
|
|||||||
}
|
}
|
||||||
return result.String()
|
return result.String()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// isCallInsn reports whether disassembled text (x86asm.IntelSyntax) is a
|
||||||
|
// call. The first token must match exactly: a prefix test would also catch
|
||||||
|
// unrelated mnemonics.
|
||||||
|
func isCallInsn(text string) bool {
|
||||||
|
m, _, _ := strings.Cut(text, " ")
|
||||||
|
return strings.ToLower(m) == "call"
|
||||||
|
}
|
||||||
|
|||||||
@@ -7,9 +7,10 @@ package debug
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||||
@@ -44,3 +45,15 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
|
|||||||
}
|
}
|
||||||
return result
|
return result
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// isCallInsn reports whether disassembled text (arm64asm.GoSyntax) is a
|
||||||
|
// call. GoSyntax renders bl as CALL; the native mnemonic is accepted too.
|
||||||
|
// The first token must match exactly so branches never match.
|
||||||
|
func isCallInsn(text string) bool {
|
||||||
|
m, _, _ := strings.Cut(text, " ")
|
||||||
|
switch strings.ToLower(m) {
|
||||||
|
case "call", "bl":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user