Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
22226d59a5 | ||
|
|
a0ae0e4e37 | ||
|
|
94c4756d47 | ||
|
|
a5a59d6503 | ||
|
|
9cb1666b35 | ||
|
|
96cc70731f | ||
|
|
b4da13d0f6 | ||
|
|
41b387e54d | ||
|
|
4171e412b5 | ||
|
|
57c0ca8b09 | ||
|
|
75bd83fd52 | ||
|
|
62f6fb4faf | ||
|
|
8f84dac10b | ||
|
|
6d7f10f13e | ||
|
|
56f8babbce | ||
|
|
6c1c8d9d96 | ||
|
|
93794c02f7 | ||
|
|
56ad158772 | ||
|
|
15e8b88d32 | ||
|
|
9beff4ae85 | ||
|
|
eacf33d0f7 | ||
|
|
9a34733615 | ||
|
|
970df7c32a | ||
|
|
49572efe16 | ||
|
|
568c553986 | ||
|
|
7a30e902fc | ||
|
|
cb398d9498 | ||
|
|
e44162a749 | ||
|
|
6fb9629ab6 | ||
|
|
8bda4066e3 | ||
|
|
685b150ecf | ||
|
|
c92e6bed3a | ||
|
|
d75e6bcae6 | ||
|
|
6699ebd34f | ||
|
|
ba4d961b20 | ||
|
|
78b12dd427 | ||
|
|
19a26e049b | ||
|
|
163480e283 | ||
|
|
0d62818db7 | ||
|
|
be2ceaafb9 | ||
|
|
941f7fa990 | ||
|
|
eb3c79c3c8 | ||
|
|
eade875b53 | ||
|
|
b08005753e | ||
|
|
5e06d6a6aa | ||
|
|
e4c9d78968 | ||
|
|
f5fcaf9fa6 | ||
|
|
94b98468bb | ||
|
|
746cef8100 | ||
|
|
f153be8158 | ||
|
|
1bf95a169e | ||
|
|
f9eb4021d6 | ||
|
|
c4930438fd | ||
|
|
f372e2db75 | ||
|
|
ac02c83a86 | ||
|
|
ce5ec24fa8 | ||
|
|
181d8e508c | ||
|
|
1160c96427 | ||
|
|
de9e211ff1 | ||
|
|
3acbdd6533 | ||
|
|
ba502c9b79 | ||
|
|
874e054ecb | ||
|
|
f1960febdc | ||
|
|
ae550cc05a | ||
|
|
cc50035375 | ||
|
|
8054fff9ac | ||
|
|
6a79c35bf7 | ||
|
|
6f4f2096e9 | ||
|
|
56630f8624 | ||
|
|
459f4a2b6e | ||
|
|
5d66343488 | ||
|
|
48334c4d5a | ||
|
|
7629963cab | ||
|
|
97951cbeb6 | ||
|
|
6e73f59e78 | ||
|
|
4221ec5741 | ||
|
|
01dcc3b86e | ||
|
|
b08885bd31 | ||
|
|
c05c53452f | ||
|
|
681a449c01 | ||
|
|
31a2cee382 | ||
|
|
3bc7c18bc3 | ||
|
|
f0512a4e1c | ||
|
|
373c09f725 | ||
|
|
b78b6c5004 | ||
|
|
9b5c878f9e | ||
|
|
31ee8e7941 | ||
|
|
2d1176e045 | ||
|
|
2a27a3a52b | ||
|
|
cde7d0f96a | ||
|
|
d7ee1b78d4 | ||
|
|
30c53565a7 | ||
|
|
ebd8ab8a3c | ||
|
|
176d856f67 | ||
|
|
eace06bbd6 | ||
|
|
f97bea61c5 | ||
|
|
19b37569c0 | ||
|
|
23b3d3e152 | ||
|
|
b0c62be8ce | ||
|
|
ece0d3f127 | ||
|
|
ad6e3360df | ||
|
|
49566de7fb | ||
|
|
6228d77566 | ||
|
|
c0e280ee3c | ||
|
|
2d45dbf7ff | ||
|
|
8ddac0135e | ||
|
|
f58a4fe51d | ||
|
|
5cb7e3e231 | ||
|
|
36bbc0c13b | ||
|
|
32afa3449f | ||
|
|
2ab6b9eb84 | ||
|
|
f41a86b660 | ||
|
|
7721353d44 | ||
|
|
243b087116 | ||
|
|
eee7a6d4a4 | ||
|
|
801fb963c9 | ||
|
|
a2acc9b5a3 | ||
|
|
f860bf8ce6 | ||
|
|
e7df5e5225 | ||
|
|
d114b3412c | ||
|
|
c77d68018c | ||
|
|
89d633f4bb | ||
|
|
f20e0bf1e7 | ||
|
|
1a45b66139 | ||
|
|
382efe538a | ||
|
|
f8d28a42ba | ||
|
|
51a2854d7f | ||
|
|
c234c3dd5b | ||
|
|
9262990ce5 | ||
|
|
d9f6167a4d | ||
|
|
f5088c52fc | ||
|
|
f52e23f1bc | ||
|
|
c9775c2b95 | ||
|
|
5af12e15ac | ||
|
|
db8e3fc160 | ||
|
|
1312122a99 | ||
|
|
11f962fbcc | ||
|
|
ee68859beb | ||
|
|
0920edb092 | ||
|
|
900c9772b1 | ||
|
|
b914c0e390 | ||
|
|
0f3146ff2c | ||
|
|
9370f9c3ee | ||
|
|
e98680597d | ||
|
|
1a01870695 | ||
|
|
458cfb626e | ||
|
|
56ecc39539 |
@@ -0,0 +1,157 @@
|
||||
# Release — gasm binaries. Runs on version tags (v0.28.0) pushed to main.
|
||||
name: Release
|
||||
|
||||
on:
|
||||
push:
|
||||
tags: ["v*"]
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: fedora
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- goos: linux
|
||||
goarch: amd64
|
||||
- goos: linux
|
||||
goarch: arm64
|
||||
- goos: linux
|
||||
goarch: riscv64
|
||||
- goos: linux
|
||||
goarch: loong64
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.27"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
|
||||
- name: Validate tag and build
|
||||
id: build
|
||||
env:
|
||||
VERSION: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
if ! echo "$VERSION" | grep -qE '^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$'; then
|
||||
echo "ERROR: expected a semver tag like v1.2.3, got: '$VERSION'"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
VERSION_NO_V="${VERSION#v}"
|
||||
echo "version_no_v=${VERSION_NO_V}" >> "$GITEA_OUTPUT"
|
||||
|
||||
mkdir -p bin
|
||||
GOOS=${{ matrix.goos }} GOARCH=${{ matrix.goarch }} CGO_ENABLED=0 \
|
||||
go build -ldflags "-s -w -X main.version=${VERSION_NO_V}" \
|
||||
-o "bin/gasm-${VERSION_NO_V}-${{ matrix.goos }}-${{ matrix.goarch }}" \
|
||||
./cmd/gasm
|
||||
|
||||
- name: Upload artifact
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: gasm-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||
path: bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||
if-no-files-found: error
|
||||
|
||||
- name: Smoke test
|
||||
if: matrix.goos == 'linux' && matrix.goarch == 'amd64'
|
||||
run: |
|
||||
chmod +x bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||
./bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }} --version
|
||||
|
||||
release:
|
||||
runs-on: fedora
|
||||
needs: build
|
||||
permissions:
|
||||
releases: write
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
path: dist
|
||||
|
||||
- name: Extract CHANGELOG section
|
||||
env:
|
||||
VERSION: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
VERSION_NO_V="${VERSION#v}"
|
||||
|
||||
sed -n "/^## \[${VERSION_NO_V}\] /,/^## \[/p" CHANGELOG.md \
|
||||
| sed '$d' \
|
||||
| tail -n +2 \
|
||||
> release-body.md
|
||||
|
||||
if [ ! -s release-body.md ]; then
|
||||
echo "ERROR: no CHANGELOG section found for ${VERSION_NO_V}"
|
||||
echo "Expected a heading like: ## [${VERSION_NO_V}] — YYYY-MM-DD"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Create release
|
||||
env:
|
||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
||||
GITEA_REPOSITORY: ${{ gitea.repository }}
|
||||
GITEA_REF_NAME: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
BODY=$(sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' -e 's/\t/\\t/g' -e 's/\r//g' release-body.md | sed ':a;N;$!ba;s/\n/\\n/g')
|
||||
BODY="\"${BODY}\""
|
||||
|
||||
response=$(curl -sS -w '\n%{http_code}' \
|
||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
||||
-H "Content-Type: application/json" \
|
||||
-X POST \
|
||||
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases" \
|
||||
-d "{\"tag_name\":\"${GITEA_REF_NAME}\",\"name\":\"${GITEA_REF_NAME}\",\"body\":${BODY},\"draft\":false,\"prerelease\":false}")
|
||||
|
||||
http_code=$(echo "$response" | tail -1)
|
||||
payload=$(echo "$response" | sed '$d')
|
||||
|
||||
echo "HTTP ${http_code}"
|
||||
if [ "$http_code" != "201" ]; then
|
||||
echo "Failed to create release: ${payload}"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
RELEASE_ID=$(echo "$payload" | grep -oE '"id"[[:space:]]*:[[:space:]]*[0-9]+' | head -1 | grep -oE '[0-9]+')
|
||||
echo "Created release ID=${RELEASE_ID}"
|
||||
printf '%s' "${RELEASE_ID}" > release-id.txt
|
||||
|
||||
- name: Upload assets
|
||||
env:
|
||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
||||
GITEA_REPOSITORY: ${{ gitea.repository }}
|
||||
GITEA_REF_NAME: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
RELEASE_ID=$(cat release-id.txt)
|
||||
|
||||
for binary in dist/gasm-*/gasm-*; do
|
||||
[ -f "$binary" ] || continue
|
||||
fname=$(basename "$binary")
|
||||
echo "Uploading ${fname}..."
|
||||
http_code=$(curl -sS -o /dev/null -w '%{http_code}' \
|
||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
||||
-H "Content-Type: application/octet-stream" \
|
||||
-X POST \
|
||||
--data-binary "@${binary}" \
|
||||
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases/${RELEASE_ID}/assets?name=${fname}")
|
||||
echo " HTTP ${http_code}"
|
||||
if [ "$http_code" != "201" ]; then
|
||||
echo "Failed to upload ${fname}"
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
|
||||
echo "Release ${GITEA_REF_NAME} is live."
|
||||
@@ -0,0 +1,96 @@
|
||||
# Test — gasm-devkit. Runs on push and pull request to development.
|
||||
name: Test
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [development]
|
||||
pull_request:
|
||||
branches: [development]
|
||||
|
||||
jobs:
|
||||
vet:
|
||||
runs-on: fedora
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.27"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
|
||||
- name: gofmt
|
||||
run: |
|
||||
set -euo pipefail
|
||||
unformatted=$(gofmt -l .)
|
||||
if [ -n "$unformatted" ]; then
|
||||
echo "These files need gofmt:"
|
||||
echo "$unformatted"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: go vet
|
||||
run: go vet ./...
|
||||
|
||||
test:
|
||||
runs-on: fedora
|
||||
needs: vet
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.27"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
|
||||
- name: Install gcc
|
||||
run: dnf install -y gcc
|
||||
|
||||
- name: go test -race
|
||||
run: go test -race -count=1 ./...
|
||||
|
||||
- name: Coverage gate — 80 % minimum
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# Exclude packages inherently untestable without hardware:
|
||||
# debug — interactive ptrace, requires a live process
|
||||
# cmd/gasm — CLI glue, covered by integration tests
|
||||
go test -coverprofile=coverage.out \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/arch \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/asm \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/ast \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/format \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/lexer \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/lint \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/lsp \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/parser \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/token \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/verify
|
||||
coverage=$(go tool cover -func=coverage.out | awk '/^total:/ { gsub("%", "", $3); print $3 }')
|
||||
echo "Total coverage: ${coverage}%"
|
||||
if awk -v c="$coverage" 'BEGIN { exit !(c+0 < 80) }'; then
|
||||
echo "ERROR: coverage ${coverage}% is below the 80% threshold"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
build:
|
||||
runs-on: fedora
|
||||
needs: test
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.27"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
|
||||
- name: Build
|
||||
run: go build -ldflags="-s -w" -o bin/gasm ./cmd/gasm
|
||||
|
||||
- name: Smoke test
|
||||
run: ./bin/gasm --version
|
||||
+15
-3
@@ -1,3 +1,8 @@
|
||||
# Metadata (always first, per repo convention)
|
||||
.idea/
|
||||
.zcode/
|
||||
.mimocode/
|
||||
|
||||
# Binaries
|
||||
/gasm
|
||||
/bin/
|
||||
@@ -7,6 +12,13 @@
|
||||
coverage.out
|
||||
*.test
|
||||
|
||||
# Editor detritus
|
||||
*.swp
|
||||
.DS_Store
|
||||
# Crash dumps
|
||||
core
|
||||
core.*
|
||||
*.core
|
||||
|
||||
# Scratch / temporary work
|
||||
_scratch/
|
||||
|
||||
# ZCode workspace
|
||||
.zcode
|
||||
|
||||
+1066
File diff suppressed because it is too large
Load Diff
+107
@@ -0,0 +1,107 @@
|
||||
# Contributing to gasm-devkit
|
||||
|
||||
Thanks for contributing to gasm-devkit.
|
||||
|
||||
## Development setup
|
||||
|
||||
Requirements: Go 1.27 or later, the [just](https://github.com/casey/just)
|
||||
command runner, and a Linux host on amd64, arm64, riscv64 or loong64.
|
||||
|
||||
```sh
|
||||
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
|
||||
cd gasm-devkit
|
||||
just install # download module dependencies
|
||||
just build # go vet + gofmt check
|
||||
just test # full suite, race detector, 80 % coverage gate
|
||||
```
|
||||
|
||||
## Workflow
|
||||
|
||||
1. Branch from `development`; never commit directly to `main` (`main` is
|
||||
release-only: merge from `development`, then tag).
|
||||
2. Commit with [Conventional Commits](https://www.conventionalcommits.org/):
|
||||
`type(scope): description`: subject line only, imperative mood,
|
||||
lowercase after the colon, no trailing dot. Allowed types: `feat`,
|
||||
`fix`, `docs`, `style`, `refactor`, `perf`, `test`, `chore`, `ci`,
|
||||
`build`, `revert`. The only line after the subject is the trailer:
|
||||
`Assisted-by: <model-name>`. No `Co-Authored-By`, no `Signed-off-by`,
|
||||
no other trailers.
|
||||
3. Record every user-visible change in `CHANGELOG.md` under
|
||||
`## [development]` (categories: Added, Changed, Fixed, Removed,
|
||||
Security).
|
||||
4. Add or update tests; coverage must stay **at or above 80 %** (hard
|
||||
gate, enforced by CI).
|
||||
5. Update the documentation when behaviour, flags or the public surface
|
||||
change.
|
||||
6. Open a pull request against `development`.
|
||||
|
||||
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`;
|
||||
CI builds and publishes the binaries for all four architectures.
|
||||
|
||||
## Code style
|
||||
|
||||
`gofmt` and `go vet` via `just fmt` / `just build`; both must pass with
|
||||
zero output; `go fix -diff ./...` must report nothing on touched packages.
|
||||
|
||||
- Standard library only in production code; `golang.org/x/arch` is used
|
||||
in tests only (round-trip decoding) and is never linked into the `gasm`
|
||||
binary.
|
||||
- No cgo, no C, no external toolchains at runtime.
|
||||
- Explicit `if err != nil`; errors wrapped with
|
||||
`fmt.Errorf("context: %w", err)`; no panics outside `main`.
|
||||
- The parser, lexer and formatter are hand-written; the `arch` instruction
|
||||
tables are generated only via `_gen/gen.go` (`just gen`), never edited.
|
||||
|
||||
## Running a single test
|
||||
|
||||
```sh
|
||||
go test -run TestVexGroundTruth ./asm/
|
||||
go test -run TestGroundTruthBasic ./verify/
|
||||
go test -run TestGOObjectLinkAndRun ./asm/
|
||||
go test -run TestFuzzWideCopy ./verify/
|
||||
```
|
||||
|
||||
The interactive debugger (`gasm debug`) requires a compiled binary on
|
||||
`$PATH`; `go run` does not work for the traced child process. Install
|
||||
first with `just install-bin`.
|
||||
|
||||
## CI (Gitea Actions)
|
||||
|
||||
Workflows live in `.gitea/workflows/` and run on self-hosted runners:
|
||||
|
||||
| Workflow | Trigger | What it does |
|
||||
|----------|---------|--------------|
|
||||
| Test | push / PR to `development` | gofmt check, `go vet`, `go test -race`, 80 % coverage gate |
|
||||
| Release | tag `v*` | cross-compiles binaries for linux/{amd64,arm64,riscv64,loong64} and publishes the Gitea release |
|
||||
|
||||
The Definition of Done (`just build` + `just test` + `just fmt`) must
|
||||
still pass locally before pushing.
|
||||
|
||||
## AI Contribution Policy
|
||||
|
||||
AI tools are welcome as productivity aids. What matters is that
|
||||
contributions remain understandable, reviewable, and genuinely useful.
|
||||
|
||||
- **Disclose AI use.** If you used AI to draft or generate any part of a
|
||||
commit, issue, pull request, or code review, say so clearly.
|
||||
- **Commit messages:** end every commit with exactly one trailer:
|
||||
`Assisted-by: <model-name>` (e.g. `Assisted-by: GLM 5.3`).
|
||||
- **Pull requests and issues:** attribute AI assistance in one trailing
|
||||
line, e.g. `_Assisted-by: GLM 5.3_`. Do not paste it into the PR
|
||||
description as a section.
|
||||
- **Take responsibility.** You remain accountable for the accuracy,
|
||||
completeness, and intent of everything you submit.
|
||||
- **Review before marking ready.** Read AI-generated diffs carefully, run
|
||||
them locally, and add or update tests where appropriate.
|
||||
- **Preferred models.** Prefer open-weight models with transparent
|
||||
training data: **GLM**, **DeepSeek**, and **MiMo**.
|
||||
|
||||
## Reporting bugs
|
||||
|
||||
Open an issue at
|
||||
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit/issues)
|
||||
with the version (`gasm --version`), OS and architecture, the exact
|
||||
command, the full output, and the expected versus actual behaviour.
|
||||
|
||||
**Security issues:** email **opensource@petrbalvin.org** instead of opening
|
||||
a public issue.
|
||||
@@ -0,0 +1,158 @@
|
||||
# gasm-devkit
|
||||
|
||||
Developer tooling for **GAsm**, Go's built-in Plan 9 assembler.
|
||||
|
||||
Go ships an assembler but no tooling for it: there is no syntax highlighting,
|
||||
no autocomplete, no linter, no static analyser, no formatter, no standalone
|
||||
assembler and no debugger for `.s` files. Developers write assembly blind,
|
||||
validate it by benchmark, and debug it by print statement. gasm-devkit is the
|
||||
missing toolkit: a single, self-contained binary, `gasm`, that brings proper
|
||||
developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
|
||||
|
||||
## Features
|
||||
|
||||
- **Front end.** A hand-written lexer and an error-tolerant parser produce a
|
||||
typed AST with source positions; `gasm tokens` and `gasm parse` expose them
|
||||
directly.
|
||||
- **Formatter.** `gasm fmt` canonicalises indentation, operand spacing,
|
||||
per-function mnemonic alignment and blank-line layout: `gofmt` for assembly,
|
||||
operating recursively on directories the way `go fmt` does.
|
||||
- **Linter.** `gasm lint` runs 18 conservative static checks, among them
|
||||
`undefined-label`, `abi-argsize` (declared frame vs the `// func` signature),
|
||||
`register-clobber` (Go ABI register liveness over the control-flow graph),
|
||||
`stack-imbalance`, `abi0-register-args` and `unencodable-instruction`.
|
||||
- **Standalone assembler.** `gasm asm` encodes all four architectures without
|
||||
the Go toolchain and writes raw images, linkable ELF objects (with DWARF5
|
||||
debug sections) or the Go toolchain's own GOOBJ format, which `go build`
|
||||
consumes in place of the toolchain's output.
|
||||
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
|
||||
executable memory: smoke calls, ABI checks (sentinel registers, red-zone
|
||||
canary), differential fuzzing against the `go tool asm` build, and
|
||||
byte-for-byte ground-truth comparison of the machine code.
|
||||
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
|
||||
breakpoints (optionally conditional), hardware watchpoints, register and
|
||||
memory inspection, and headless script runs with label-level coverage.
|
||||
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
|
||||
push and pull diagnostics, semantic-token highlighting, go-to-definition,
|
||||
find references, rename, formatting, inlay hints, code actions, signature
|
||||
help, document highlights, workspace symbol search, #include document
|
||||
links and folding ranges over stdio.
|
||||
- **Comparators and audits.** `gasm diff` compares the machine code of two
|
||||
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
|
||||
`gasm audit-instructions` diffs the encoder against the installed toolchain,
|
||||
and `gasm scaffold` generates a differential test skeleton for a kernel.
|
||||
- **Complete instruction coverage.** The instruction tables are generated
|
||||
from the Go toolchain's own assembler source, so the toolkit recognises
|
||||
every mnemonic the real assembler accepts; `just gen` refreshes them.
|
||||
|
||||
### Architecture support
|
||||
|
||||
| Architecture | GOARCH | File suffix | Instructions recognised |
|
||||
|--------------|-------------|--------------|---------------------------------------------|
|
||||
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
|
||||
| ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes |
|
||||
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes |
|
||||
| LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes |
|
||||
|
||||
"Common opcodes" are the instructions shared by every architecture (`RET`,
|
||||
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
|
||||
carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
|
||||
...) that the assembler accepts as aliases. Regenerating the tables is one
|
||||
command (`just gen`) and requires only a Go installation; the committed output
|
||||
has no runtime dependency on the toolchain.
|
||||
|
||||
## Install
|
||||
|
||||
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
|
||||
linux/loong64 are on the
|
||||
[releases page](https://sourcedock.dev/petrbalvin/gasm-devkit/releases).
|
||||
From source (Go 1.27 or later):
|
||||
|
||||
```sh
|
||||
go install sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm@latest
|
||||
```
|
||||
|
||||
Or from a repository checkout, with the development version stamped:
|
||||
|
||||
```sh
|
||||
just install-bin
|
||||
```
|
||||
|
||||
## Quick start
|
||||
|
||||
```sh
|
||||
cat > hello_amd64.s <<'EOF'
|
||||
#include "textflag.h"
|
||||
|
||||
// func add(a, b int) int
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVQ a+0(FP), AX
|
||||
ADDQ b+8(FP), AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
EOF
|
||||
|
||||
gasm lint hello_amd64.s # static checks
|
||||
gasm asm -o hello.bin hello_amd64.s # assemble to a raw image
|
||||
gasm verify --call add --args a=2,b=3 hello_amd64.s # JIT-call it with arguments
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
```sh
|
||||
gasm fmt # reformat every .s below here, like go fmt
|
||||
gasm fmt -w kernel_amd64.s # canonicalise one file in place
|
||||
gasm lint *.s # static checks
|
||||
gasm asm --format elf -o k.o k.s # assemble to a linkable ELF object
|
||||
gasm asm --format goobj -p pkg/path -o k.o k.s # Go object, consumed by go build
|
||||
gasm verify --ground-truth k.s # byte-for-byte vs go tool asm
|
||||
gasm verify --fuzz k.s # differential fuzz vs the go tool asm build
|
||||
gasm debug --func name k.s # interactive debugger
|
||||
gasm debug --func name --script cmds.txt --timeout 30s k.s # headless run
|
||||
gasm debug --func name --cover k.s # which labels did execution reach?
|
||||
gasm diff a.s b.s # compare machine code byte-for-byte
|
||||
gasm diff --map wideCopyAVX2=wideCopyAVX512 avx2.s avx512.s
|
||||
gasm profile k.s # show basic-block structure
|
||||
gasm audit-instructions # encoder vs go tool asm name diff
|
||||
gasm scaffold differential k.s # generate a differential test skeleton
|
||||
```
|
||||
|
||||
Run `gasm --help` for the command overview and `gasm <command> -h` for a
|
||||
command's flags. [docs/CLI.md](docs/CLI.md) is the full reference.
|
||||
|
||||
### Editor integration
|
||||
|
||||
`gasm lsp` speaks the Language Server Protocol over standard input/output, so
|
||||
any LSP-capable editor can use it: point your editor's LSP client at the
|
||||
binary and associate it with `.s` files. Syntax highlighting is delivered as
|
||||
LSP semantic tokens, so no editor-specific grammar is required. The server
|
||||
infers the target architecture from the file-name suffix
|
||||
(`_amd64.s` / `_arm64.s` / `_riscv64.s` / `_loong64.s`).
|
||||
|
||||
## Development
|
||||
|
||||
```sh
|
||||
just install # download module dependencies
|
||||
just build # go vet + gofmt check, zero errors and zero warnings
|
||||
just test # full suite, race detector, 80 % coverage gate
|
||||
just fmt # gofmt the tree
|
||||
just gen # regenerate the instruction tables from the Go toolchain
|
||||
```
|
||||
|
||||
See [CONTRIBUTING.md](CONTRIBUTING.md) for the development workflow and
|
||||
[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for setup details and every
|
||||
recipe.
|
||||
|
||||
## Documentation
|
||||
|
||||
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
|
||||
- [docs/CLI.md](docs/CLI.md): full command reference
|
||||
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
|
||||
- [docs/DECISIONS.md](docs/DECISIONS.md): deferred design decisions
|
||||
- [CHANGELOG.md](CHANGELOG.md): release history
|
||||
|
||||
## Licence
|
||||
|
||||
BSD-3-Clause — see [LICENSE](LICENSE).
|
||||
|
||||
Copyright © 2026 [Petr Balvín](https://petrbalvin.org)
|
||||
+8
-2
@@ -86,7 +86,10 @@ func filterCommon(names []string) []string {
|
||||
func writeCommon(names []string) error {
|
||||
var b strings.Builder
|
||||
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
||||
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n\n")
|
||||
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
|
||||
b.WriteString("//\n")
|
||||
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
||||
b.WriteString("// SPDX-License-Identifier: BSD-3-Clause\n\n")
|
||||
b.WriteString("package arch\n\n")
|
||||
b.WriteString("// commonGeneratedInstrs is the set of opcodes shared by every architecture\n")
|
||||
b.WriteString("// (RET, JMP, NOP, CALL, TEXT, FUNCDATA, PCDATA, …).\n")
|
||||
@@ -154,7 +157,10 @@ func stringLit(elt ast.Expr) string {
|
||||
func writeGen(arch, sub string, names []string) error {
|
||||
var b strings.Builder
|
||||
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
||||
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n\n")
|
||||
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
|
||||
b.WriteString("//\n")
|
||||
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
||||
b.WriteString("// SPDX-License-Identifier: BSD-3-Clause\n\n")
|
||||
b.WriteString("package arch\n\n")
|
||||
b.WriteString("// " + arch + "GeneratedInstrs is the complete set of " + arch +
|
||||
" mnemonics accepted by\n// Go's Plan 9 assembler.\n")
|
||||
|
||||
@@ -270,6 +270,8 @@ func amd64Curated() []Instr {
|
||||
"VMINPD", "VMINPS", "VMINSD", "VMINSS", "VMAXPD", "VMAXPS", "VMAXSD", "VMAXSS",
|
||||
"VXORPD", "VXORPS", "VANDPD", "VANDPS", "VANDNPD", "VANDNPS", "VORPD", "VORPS",
|
||||
"VUNPCKHPD", "VUNPCKLPD", "VUNPCKHPS", "VUNPCKLPS",
|
||||
"PSHUFD", "PSHUFHW", "PSHUFLW", "SHUFPS", "SHUFPD",
|
||||
"UNPCKLPS", "UNPCKHPS", "UNPCKLPD", "UNPCKHPD",
|
||||
"VSQRTPD", "VSQRTPS", "VSQRTSD", "VSQRTSS", "VRSQRTPS", "VRCPPS",
|
||||
"VCMPPD", "VCMPPS", "VCMPSD", "VCMPSS",
|
||||
} {
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
|
||||
//
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package arch
|
||||
|
||||
|
||||
@@ -156,6 +156,16 @@ func (t *Table) Lookup(mnemonic string) (Instr, bool) {
|
||||
}
|
||||
}
|
||||
}
|
||||
// amd64 EVEX instructions take a .Z zeroing suffix (masking is written as
|
||||
// an explicit K operand rather than a suffix); strip it so the base
|
||||
// instruction is still recognised.
|
||||
if t.Arch == AMD64 {
|
||||
if base, ok := strings.CutSuffix(key, ".Z"); ok {
|
||||
if in, found := t.instrs[base]; found {
|
||||
return in, true
|
||||
}
|
||||
}
|
||||
}
|
||||
return Instr{}, false
|
||||
}
|
||||
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
|
||||
//
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package arch
|
||||
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/util.go from the Go toolchain.
|
||||
//
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package arch
|
||||
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
|
||||
//
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package arch
|
||||
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
|
||||
//
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package arch
|
||||
|
||||
|
||||
@@ -0,0 +1,178 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestGOObjectAARCH64Structure checks the basic structure of the emitted
|
||||
// AArch64 GOOBJ: the preamble, the magic, the block offsets and the
|
||||
// non-package symbol definitions.
|
||||
func TestGOObjectAARCH64Structure(t *testing.T) {
|
||||
f, errs := parser.Parse("k_arm64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVD a+0(FP), R4
|
||||
MOVD b+8(FP), R5
|
||||
ADD R5, R4, R4
|
||||
MOVD R4, ret+16(FP)
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
obj, err := img.GOObjectAARCH64("testpkg", "k_arm64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObjectAARCH64: %v", err)
|
||||
}
|
||||
|
||||
// Check preamble.
|
||||
idx := strings.Index(string(obj), "\n!\n")
|
||||
if idx < 0 {
|
||||
t.Fatal("missing preamble separator")
|
||||
}
|
||||
preamble := string(obj[:idx])
|
||||
if !strings.HasPrefix(preamble, "go object") {
|
||||
t.Errorf("preamble = %q, want 'go object ...'", preamble)
|
||||
}
|
||||
|
||||
// Check GOOBJ magic.
|
||||
magicIdx := idx + 3
|
||||
if magicIdx+8 > len(obj) || string(obj[magicIdx:magicIdx+8]) != "\x00go120ld" {
|
||||
t.Error("missing GOOBJ magic")
|
||||
}
|
||||
|
||||
// The object should contain the function's code.
|
||||
if len(img.Code) == 0 {
|
||||
t.Error("no code generated")
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectAARCH64Link does an end-to-end link test: it cross-compiles a
|
||||
// Go program for arm64, substitutes the gasm-produced object into the package
|
||||
// archive, re-links with cmd/link, and verifies the symbol appears in the
|
||||
// resulting binary. The binary is not executed (no arm64 host or qemu).
|
||||
// Skipped when no Go toolchain is available.
|
||||
func TestGOObjectAARCH64Link(t *testing.T) {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
t.Skip("no Go toolchain available")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
asmSrc := `#include "textflag.h"
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVD a+0(FP), R4
|
||||
MOVD b+8(FP), R5
|
||||
ADD R5, R4, R4
|
||||
MOVD R4, ret+16(FP)
|
||||
RET
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
mainSrc := `package main
|
||||
|
||||
func add(a, b int64) int64
|
||||
|
||||
func main() {
|
||||
if add(20, 22) != 42 {
|
||||
panic("bad add")
|
||||
}
|
||||
}
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module a64link\n\ngo 1.21\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Capture the cross build (GOARCH=arm64): the package archive and the
|
||||
// link line.
|
||||
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||
build.Dir = dir
|
||||
build.Env = append(os.Environ(), "GOARCH=arm64")
|
||||
buildLog, err := build.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||
}
|
||||
var work, linkLine, asmObj string
|
||||
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
|
||||
asmObj = fieldAfter(line, "-o")
|
||||
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||
linkLine = line
|
||||
}
|
||||
}
|
||||
if work == "" || asmObj == "" {
|
||||
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
|
||||
}
|
||||
defer os.RemoveAll(work)
|
||||
|
||||
// Expand $WORK in the object path.
|
||||
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
|
||||
|
||||
// Read the toolchain-produced object and assemble the same source with gasm.
|
||||
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f, errs := parser.Parse("main_arm64.s", string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
gasmObj, err := img.GOObjectAARCH64("a64link", "main_arm64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObjectAARCH64: %v", err)
|
||||
}
|
||||
|
||||
// Replace the toolchain-produced object with gasm's.
|
||||
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
|
||||
t.Fatalf("write gasm object: %v", err)
|
||||
}
|
||||
|
||||
// Re-link.
|
||||
if linkLine == "" {
|
||||
t.Skip("could not find link command in build log")
|
||||
}
|
||||
// Expand $WORK in the link command.
|
||||
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
|
||||
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
|
||||
if out, err := linkCmd.CombinedOutput(); err != nil {
|
||||
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
|
||||
}
|
||||
|
||||
// Verify the binary exists and contains the symbol.
|
||||
binPath := filepath.Join(dir, "prog")
|
||||
if _, err := os.Stat(binPath); err != nil {
|
||||
t.Fatalf("binary not found: %v", err)
|
||||
}
|
||||
binData, err := os.ReadFile(binPath)
|
||||
if err != nil {
|
||||
t.Fatalf("read binary: %v", err)
|
||||
}
|
||||
if !strings.Contains(string(binData), "add") && !strings.Contains(string(binData), "a64link") {
|
||||
t.Error("binary does not contain expected symbol")
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,687 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
// arm64 (AArch64) instruction encoding.
|
||||
//
|
||||
// The encoder is data-driven: each mnemonic maps to an instruction format and
|
||||
// an opcode constant, and the format selects the bit layout. The opcode
|
||||
// constants and formats are transcribed from the Go toolchain's own arm64
|
||||
// backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm`
|
||||
// exactly — the ground-truth oracle for the verify suite.
|
||||
//
|
||||
// All AArch64 instructions are 32 bits, little-endian. The formats used here
|
||||
// (per the ARM Architecture Reference Manual):
|
||||
//
|
||||
// DP-shifted-reg sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
|
||||
// DP-immediate sf<<31 | op<<30 | S<<29 | 0x11<<24 | imm12<<10 | Rn<<5 | Rd
|
||||
// Logical-imm sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd
|
||||
// Move-wide sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd
|
||||
// Load/store size<<30 | 0x7<<27 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt
|
||||
// LDST-unscaled size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<5 | Rt (actually imm9<<12 | Rn<<5 | Rt)
|
||||
// LDST-pair opc<<30 | 0x5<<27 | V<<26 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt
|
||||
// Branch-imm 0<<31 | 0x5<<26 | imm26 (B)
|
||||
// Branch-imm 1<<31 | 0x5<<26 | imm26 (BL)
|
||||
// Branch-cond 0x2A<<25 | imm19<<5 | cond (B.cond)
|
||||
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
|
||||
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
|
||||
|
||||
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
|
||||
// R0–R30 (integer), F0–F31 (floating point), and the ABI aliases the
|
||||
// runtime's assembly uses. Returns -1 for an unrecognised name.
|
||||
func arm64RegNum(name string) int {
|
||||
switch name {
|
||||
case "R0":
|
||||
return 0
|
||||
case "R1":
|
||||
return 1
|
||||
case "R2":
|
||||
return 2
|
||||
case "R3":
|
||||
return 3
|
||||
case "R4":
|
||||
return 4
|
||||
case "R5":
|
||||
return 5
|
||||
case "R6":
|
||||
return 6
|
||||
case "R7":
|
||||
return 7
|
||||
case "R8":
|
||||
return 8
|
||||
case "R9":
|
||||
return 9
|
||||
case "R10":
|
||||
return 10
|
||||
case "R11":
|
||||
return 11
|
||||
case "R12":
|
||||
return 12
|
||||
case "R13":
|
||||
return 13
|
||||
case "R14":
|
||||
return 14
|
||||
case "R15":
|
||||
return 15
|
||||
case "R16":
|
||||
return 16
|
||||
case "R17":
|
||||
return 17
|
||||
case "R18":
|
||||
return 18
|
||||
case "R19":
|
||||
return 19
|
||||
case "R20":
|
||||
return 20
|
||||
case "R21":
|
||||
return 21
|
||||
case "R22":
|
||||
return 22
|
||||
case "R23":
|
||||
return 23
|
||||
case "R24":
|
||||
return 24
|
||||
case "R25":
|
||||
return 25
|
||||
case "R26", "REGCTXT", "CTXT":
|
||||
return 26
|
||||
case "R27", "REGTMP", "TMP":
|
||||
return 27
|
||||
case "R28", "REGG", "g":
|
||||
return 28
|
||||
case "R29", "FP":
|
||||
return 29
|
||||
case "R30", "LR", "LINK":
|
||||
return 30
|
||||
case "R31", "ZR":
|
||||
return 31
|
||||
case "SP":
|
||||
return 31 // SP and ZR share encoding 31; context determines meaning
|
||||
}
|
||||
// F0–F31.
|
||||
if len(name) >= 1 && name[0] == 'F' {
|
||||
n := 0
|
||||
for i := 1; i < len(name); i++ {
|
||||
if name[i] < '0' || name[i] > '9' {
|
||||
return -1
|
||||
}
|
||||
n = n*10 + int(name[i]-'0')
|
||||
}
|
||||
if n <= 31 {
|
||||
return n
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// ---- format helpers ----
|
||||
|
||||
// a64wordLE encodes a uint32 as 4 little-endian bytes.
|
||||
func a64wordLE(w uint32) []byte {
|
||||
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
|
||||
}
|
||||
|
||||
// a64WordsLE concatenates one or more instruction words as little-endian bytes.
|
||||
func a64WordsLE(ws ...uint32) []byte {
|
||||
var out []byte
|
||||
for _, w := range ws {
|
||||
out = append(out, a64wordLE(w)...)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// ---- data-processing (immediate) ----
|
||||
|
||||
// a64AddSub encodes an ADD/SUB (immediate) instruction:
|
||||
// sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | Rn<<5 | Rd.
|
||||
func a64AddSub(sf, op, S, sh, imm12, rn, rd uint32) uint32 {
|
||||
return sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// ---- move wide ----
|
||||
|
||||
// a64MoveWide encodes a MOVZ/MOVK/MOVN instruction:
|
||||
// sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd.
|
||||
func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 {
|
||||
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
|
||||
}
|
||||
|
||||
// ---- load/store (unsigned immediate, scaled) ----
|
||||
|
||||
// a64LSU encodes a load/store register (unsigned immediate, scaled):
|
||||
// size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt.
|
||||
// (0x39<<24 encodes bits 29:24 = 111001, the scaled unsigned offset form.)
|
||||
func a64LSU(size, V, opc, imm12, rn, rt uint32) uint32 {
|
||||
return size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | rn<<5 | rt
|
||||
}
|
||||
|
||||
// ---- load/store (unscaled immediate) ----
|
||||
|
||||
// a64LSUnscaled encodes a load/store register (unscaled immediate, 9-bit signed):
|
||||
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<12 | Rn<<5 | Rt.
|
||||
// Note: the 0<<24 distinguishes unscaled from the pre/post-index forms.
|
||||
func a64LSUnscaled(size, V, opc int, imm9 int32, rn, rt int) uint32 {
|
||||
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | uint32(opc)<<22 |
|
||||
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
||||
}
|
||||
|
||||
// ---- load/store pair ----
|
||||
|
||||
// a64LSP encodes a load/store pair instruction (signed offset):
|
||||
// opc<<30 | 0x5<<27 | V<<26 | 2<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
|
||||
// opc: 0=32-bit, 1=reserved, 2=64-bit. V: 0=integer, 1=FP/SIMD.
|
||||
// L: 0=store, 1=load. imm7 is the signed scaled offset (÷8 for 64-bit pairs).
|
||||
func a64LSP(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
|
||||
return opc<<30 | 5<<27 | V<<26 | 2<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
|
||||
}
|
||||
|
||||
// ---- branches ----
|
||||
|
||||
// a64Branch encodes an unconditional branch (B/BL):
|
||||
// op<<31 | 0x5<<26 | imm26.
|
||||
func a64Branch(op uint32, imm26 int32) uint32 {
|
||||
return op<<31 | 5<<26 | (uint32(imm26) & 0x03FFFFFF)
|
||||
}
|
||||
|
||||
// a64BranchCond encodes a conditional branch (B.cond):
|
||||
// 0x2A<<25 | imm19<<5 | cond.
|
||||
func a64BranchCond(imm19 int32, cond uint32) uint32 {
|
||||
return 0x2A<<25 | (uint32(imm19)&0x7FFFF)<<5 | cond&0xF
|
||||
}
|
||||
|
||||
// a64UncondBranch encodes an unconditional branch register (BR/BLR/RET):
|
||||
// 0x6B<<25 | opc<<21 | 0x1F<<16 | Rn<<5 | Rd.
|
||||
// opc: 0=BR, 1=BLR, 2=RET. For RET, Rn defaults to LR(30).
|
||||
func a64UncondBranch(opc, rn, rd uint32) uint32 {
|
||||
return 0x6B<<25 | opc<<21 | 0x1F<<16 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// ---- ADR/ADRP ----
|
||||
|
||||
// a64ADR encodes an ADR instruction (p=0) or ADRP instruction (p=1):
|
||||
// p<<31 | immlo<<29 | 0x10<<24 | immhi<<5 | Rd.
|
||||
func a64ADR(p uint32, immhi int32, immlo uint32, rd uint32) uint32 {
|
||||
return p<<31 | immlo<<29 | 0x10<<24 | (uint32(immhi)&0x7FFFF)<<5 | rd
|
||||
}
|
||||
|
||||
// ---- system ----
|
||||
|
||||
// a64NOP encodes a NOP: 0xd503201f.
|
||||
const a64NOP uint32 = 0xd503201f
|
||||
|
||||
// a64BRK encodes a BRK instruction: 0xd4200000 | imm16<<5.
|
||||
func a64BRK(imm16 uint32) uint32 {
|
||||
return 0xd4200000 | imm16<<5
|
||||
}
|
||||
|
||||
// ---- condition codes ----
|
||||
|
||||
const (
|
||||
a64CondEQ = 0x0
|
||||
a64CondNE = 0x1
|
||||
a64CondCS = 0x2
|
||||
a64CondHS = 0x2
|
||||
a64CondCC = 0x3
|
||||
a64CondLO = 0x3
|
||||
a64CondMI = 0x4
|
||||
a64CondPL = 0x5
|
||||
a64CondVS = 0x6
|
||||
a64CondVC = 0x7
|
||||
a64CondHI = 0x8
|
||||
a64CondLS = 0x9
|
||||
a64CondGE = 0xa
|
||||
a64CondLT = 0xb
|
||||
a64CondGT = 0xc
|
||||
a64CondLE = 0xd
|
||||
)
|
||||
|
||||
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
|
||||
var arm64CondMap = map[string]uint32{
|
||||
"EQ": a64CondEQ,
|
||||
"NE": a64CondNE,
|
||||
"CS": a64CondCS,
|
||||
"HS": a64CondHS,
|
||||
"CC": a64CondCC,
|
||||
"LO": a64CondLO,
|
||||
"MI": a64CondMI,
|
||||
"PL": a64CondPL,
|
||||
"VS": a64CondVS,
|
||||
"VC": a64CondVC,
|
||||
"HI": a64CondHI,
|
||||
"LS": a64CondLS,
|
||||
"GE": a64CondGE,
|
||||
"LT": a64CondLT,
|
||||
"GT": a64CondGT,
|
||||
"LE": a64CondLE,
|
||||
}
|
||||
|
||||
// ---- instruction format tags ----
|
||||
|
||||
type a64Format uint8
|
||||
|
||||
const (
|
||||
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
|
||||
a64FDPIR // data-processing (immediate): ADD/SUB $imm
|
||||
a64FLogImm // logical (immediate): AND/ORR/EOR $imm
|
||||
a64FMovWide // move wide: MOVZ, MOVN, MOVK
|
||||
a64FLSU // load/store (unsigned immediate, scaled)
|
||||
a64FLSUnscaled // load/store (unscaled immediate)
|
||||
a64FLSPair // load/store pair
|
||||
a64FBranch // unconditional branch (B/BL)
|
||||
a64FBranchCond // conditional branch (B.cond)
|
||||
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
|
||||
a64FADR // ADR/ADRP
|
||||
a64FEXTR // EXTR
|
||||
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
||||
a64FSystem // system: NOP, BRK, etc.
|
||||
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
||||
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
|
||||
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
|
||||
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
|
||||
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
|
||||
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
|
||||
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
||||
a64FFMovGR // FMOV between GP and FP registers
|
||||
a64FCRC32 // CRC32
|
||||
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
||||
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR
|
||||
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
||||
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
|
||||
)
|
||||
|
||||
// a64Enc is one instruction's encoding: its bit layout (format) and the
|
||||
// opcode constant, positioned at its exact bit range.
|
||||
type a64Enc struct {
|
||||
format a64Format
|
||||
op uint32 // the pre-positioned opcode bits
|
||||
}
|
||||
|
||||
// a64InstrTable maps AArch64 mnemonics (as the Go assembler spells them) to
|
||||
// their encoding. The base integer, memory, floating-point and SIMD
|
||||
// instruction sets are covered.
|
||||
var a64InstrTable = map[string]a64Enc{}
|
||||
|
||||
func init() {
|
||||
// ---- data-processing (shifted register) ----
|
||||
// Format: sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
|
||||
dpsr := map[string]uint32{
|
||||
// Add/Sub
|
||||
"ADD": 1<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=1, op=0, S=0 (64-bit default)
|
||||
"ADDW": 0<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=0
|
||||
"ADDS": 1<<31 | 0<<30 | 1<<29 | 0x0b<<24,
|
||||
"ADDSW": 0<<31 | 0<<30 | 1<<29 | 0x0b<<24,
|
||||
"SUB": 1<<31 | 1<<30 | 0<<29 | 0x0b<<24,
|
||||
"SUBW": 0<<31 | 1<<30 | 0<<29 | 0x0b<<24,
|
||||
"SUBS": 1<<31 | 1<<30 | 1<<29 | 0x0b<<24,
|
||||
"SUBSW": 0<<31 | 1<<30 | 1<<29 | 0x0b<<24,
|
||||
// Logical (shifted register)
|
||||
"AND": 1<<31 | 0<<29 | 0x0a<<24,
|
||||
"ANDW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
"BIC": 1<<31 | 0<<29 | 0x0a<<24 | 1<<21,
|
||||
"BICW": 0<<31 | 0<<29 | 0x0a<<24 | 1<<21,
|
||||
"ORR": 1<<31 | 1<<29 | 0x0a<<24,
|
||||
"ORRW": 0<<31 | 1<<29 | 0x0a<<24,
|
||||
"ORN": 1<<31 | 1<<29 | 0x0a<<24 | 1<<21,
|
||||
"ORNW": 0<<31 | 1<<29 | 0x0a<<24 | 1<<21,
|
||||
"EOR": 1<<31 | 2<<29 | 0x0a<<24,
|
||||
"EORW": 0<<31 | 2<<29 | 0x0a<<24,
|
||||
"EON": 1<<31 | 2<<29 | 0x0a<<24 | 1<<21,
|
||||
"EONW": 0<<31 | 2<<29 | 0x0a<<24 | 1<<21,
|
||||
"ANDS": 1<<31 | 3<<29 | 0x0a<<24,
|
||||
"ANDSW": 0<<31 | 3<<29 | 0x0a<<24,
|
||||
"BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21,
|
||||
"BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21,
|
||||
// Shift
|
||||
"LSL": 1<<31 | 0<<29 | 0x0a<<24, // alias of UBFM
|
||||
"LSLW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
"LSR": 1<<31 | 0<<29 | 0x0a<<24,
|
||||
"LSRW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
"ASR": 1<<31 | 0<<29 | 0x0a<<24,
|
||||
"ASRW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
"ROR": 1<<31 | 0<<29 | 0x0a<<24,
|
||||
"RORW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
// Multiply
|
||||
"MADD": 1<<31 | 0<<29 | 0x1b<<24 | 0<<21,
|
||||
"MADDW": 0<<31 | 0<<29 | 0x1b<<24 | 0<<21,
|
||||
"MSUB": 1<<31 | 0<<29 | 0x1b<<24 | 1<<21,
|
||||
"MSUBW": 0<<31 | 0<<29 | 0x1b<<24 | 1<<21,
|
||||
// Divide
|
||||
"SDIV": 1<<31 | 0<<29 | 0x0d<<24,
|
||||
"SDIVW": 0<<31 | 0<<29 | 0x0d<<24,
|
||||
"UDIV": 1<<31 | 0<<29 | 0x0d<<24 | 1<<10,
|
||||
"UDIVW": 0<<31 | 0<<29 | 0x0d<<24 | 1<<10,
|
||||
// CRC
|
||||
"CRC32B": 0<<31 | 0<<29 | 0x1b<<24 | 4<<10,
|
||||
"CRC32H": 0<<31 | 0<<29 | 0x1b<<24 | 5<<10,
|
||||
"CRC32W": 0<<31 | 0<<29 | 0x1b<<24 | 6<<10,
|
||||
"CRC32X": 1<<31 | 0<<29 | 0x1b<<24 | 7<<10,
|
||||
// Conditional select
|
||||
"CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10,
|
||||
"CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10,
|
||||
"CSINC": 1<<31 | 0<<29 | 0x1d<<24 | 1<<10,
|
||||
"CSINCW": 0<<31 | 0<<29 | 0x1d<<24 | 1<<10,
|
||||
"CSINV": 1<<31 | 0<<29 | 0x1d<<24 | 2<<10,
|
||||
"CSINVW": 0<<31 | 0<<29 | 0x1d<<24 | 2<<10,
|
||||
"CSNEG": 1<<31 | 0<<29 | 0x1d<<24 | 3<<10,
|
||||
"CSNEGW": 0<<31 | 0<<29 | 0x1d<<24 | 3<<10,
|
||||
}
|
||||
for m, op := range dpsr {
|
||||
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
|
||||
}
|
||||
|
||||
// Aliases that map to the same encoding as their target.
|
||||
a64InstrTable["CMP"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
|
||||
a64InstrTable["CMPW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBSW"]}
|
||||
a64InstrTable["CMN"] = a64Enc{format: a64FDPSR, op: dpsr["ADDS"]}
|
||||
a64InstrTable["CMNW"] = a64Enc{format: a64FDPSR, op: dpsr["ADDSW"]}
|
||||
a64InstrTable["TST"] = a64Enc{format: a64FDPSR, op: dpsr["ANDS"]}
|
||||
a64InstrTable["TSTW"] = a64Enc{format: a64FDPSR, op: dpsr["ANDSW"]}
|
||||
a64InstrTable["NEG"] = a64Enc{format: a64FDPSR, op: dpsr["SUB"]}
|
||||
a64InstrTable["NEGW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBW"]}
|
||||
a64InstrTable["NEGS"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
|
||||
a64InstrTable["MVN"] = a64Enc{format: a64FDPSR, op: dpsr["ORN"]}
|
||||
a64InstrTable["MVNW"] = a64Enc{format: a64FDPSR, op: dpsr["ORNW"]}
|
||||
a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]}
|
||||
a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]}
|
||||
|
||||
// ---- data-processing (immediate) ----
|
||||
// ADD/SUB $imm, Rn, Rd
|
||||
a64InstrTable["ADDImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 0<<29 | 0x11<<24}
|
||||
a64InstrTable["ADDWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 0<<30 | 0<<29 | 0x11<<24}
|
||||
a64InstrTable["SUBImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 0<<29 | 0x11<<24}
|
||||
a64InstrTable["SUBWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 1<<30 | 0<<29 | 0x11<<24}
|
||||
a64InstrTable["ADDSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 1<<29 | 0x11<<24}
|
||||
a64InstrTable["SUBSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 1<<29 | 0x11<<24}
|
||||
|
||||
// ---- move wide ----
|
||||
// MOVZ/MOVN/MOVK
|
||||
a64InstrTable["MOVZ"] = a64Enc{format: a64FMovWide, op: 1<<31 | 2<<29 | 0x25<<23}
|
||||
a64InstrTable["MOVZW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 2<<29 | 0x25<<23}
|
||||
a64InstrTable["MOVN"] = a64Enc{format: a64FMovWide, op: 1<<31 | 0<<29 | 0x25<<23}
|
||||
a64InstrTable["MOVNW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 0<<29 | 0x25<<23}
|
||||
a64InstrTable["MOVK"] = a64Enc{format: a64FMovWide, op: 1<<31 | 3<<29 | 0x25<<23}
|
||||
a64InstrTable["MOVKW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 3<<29 | 0x25<<23}
|
||||
|
||||
// ---- ADR/ADRP ----
|
||||
a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0}
|
||||
a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1}
|
||||
|
||||
// ---- load/store (unsigned immediate) ----
|
||||
a64InstrTable["MOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<22} // LDR 64-bit
|
||||
a64InstrTable["MOVWU"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<22} // LDR 32-bit unsigned
|
||||
a64InstrTable["MOVHU"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 1<<22} // LDRH unsigned
|
||||
a64InstrTable["MOVBU"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 1<<22} // LDRB unsigned
|
||||
a64InstrTable["MOVW"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 2<<22} // LDRSW (signed 32→64)
|
||||
a64InstrTable["MOVH"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 2<<22} // LDRSH (signed half)
|
||||
a64InstrTable["MOVB"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 2<<22} // LDRSB (signed byte)
|
||||
a64InstrTable["FMOVS"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 32-bit FP
|
||||
a64InstrTable["FMOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 64-bit FP
|
||||
|
||||
// Store opcodes (load ^ (1<<22)):
|
||||
// STR 64-bit: size=3, V=0, opc=00 → 3<<30 | 7<<27 | 0<<22
|
||||
// STR 32-bit: size=2, V=0, opc=00 → 2<<30 | 7<<27 | 0<<22
|
||||
// STRH: size=1, V=0, opc=00 → 1<<30 | 7<<27 | 0<<22
|
||||
// STRB: size=0, V=0, opc=00 → 0<<30 | 7<<27 | 0<<22
|
||||
|
||||
// ---- branches ----
|
||||
a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26}
|
||||
a64InstrTable["BL"] = a64Enc{format: a64FBranch, op: 1<<31 | 5<<26}
|
||||
|
||||
// Conditional branches.
|
||||
condBranches := map[string]uint32{
|
||||
"BEQ": 0x0, "BNE": 0x1, "BCS": 0x2, "BHS": 0x2,
|
||||
"BCC": 0x3, "BLO": 0x3, "BMI": 0x4, "BPL": 0x5,
|
||||
"BVS": 0x6, "BVC": 0x7, "BHI": 0x8, "BLS": 0x9,
|
||||
"BGE": 0xa, "BLT": 0xb, "BGT": 0xc, "BLE": 0xd,
|
||||
}
|
||||
for name, cond := range condBranches {
|
||||
a64InstrTable[name] = a64Enc{format: a64FBranchCond, op: 0x2A<<25 | cond}
|
||||
}
|
||||
|
||||
// Unconditional branch register (BR/BLR/RET).
|
||||
a64InstrTable["BR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 0<<21}
|
||||
a64InstrTable["BLR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 1<<21}
|
||||
a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21}
|
||||
|
||||
// ---- system ----
|
||||
a64InstrTable["NOP"] = a64Enc{format: a64FSystem, op: a64NOP}
|
||||
a64InstrTable["NOOP"] = a64Enc{format: a64FSystem, op: a64NOP}
|
||||
a64InstrTable["BRK"] = a64Enc{format: a64FSystem, op: 0xd4200000}
|
||||
a64InstrTable["UNDEF"] = a64Enc{format: a64FSystem, op: a64BRK(0)}
|
||||
|
||||
// ---- EXTR ----
|
||||
a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22}
|
||||
a64InstrTable["EXTRW"] = a64Enc{format: a64FEXTR, op: 0<<31 | 0x27<<23 | 0<<22}
|
||||
|
||||
// ---- bitfield ----
|
||||
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
|
||||
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
|
||||
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["UBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
|
||||
a64InstrTable["BFI"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
|
||||
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
|
||||
|
||||
// ---- FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL ----
|
||||
fp3 := map[string]uint32{
|
||||
"FADDS": 0x1e202800, "FADDD": 0x1e602800,
|
||||
"FSUBS": 0x1e203800, "FSUBD": 0x1e603800,
|
||||
"FMULS": 0x1e200800, "FMULD": 0x1e600800,
|
||||
"FDIVS": 0x1e201800, "FDIVD": 0x1e601800,
|
||||
"FMAXS": 0x1e204800, "FMAXD": 0x1e604800,
|
||||
"FMINS": 0x1e205800, "FMIND": 0x1e605800,
|
||||
"FMAXNMS": 0x1e206800, "FMAXNMD": 0x1e606800,
|
||||
"FMINNMS": 0x1e207800, "FMINNMD": 0x1e607800,
|
||||
"FNMULS": 0x1e208800, "FNMULD": 0x1e608800,
|
||||
}
|
||||
for m, op := range fp3 {
|
||||
a64InstrTable[m] = a64Enc{format: a64FFP3, op: op}
|
||||
}
|
||||
|
||||
// ---- FP unary (Rn, Rd): FMOV reg-reg, FABS, FNEG, FSQRT, FCVT, FRINT* ----
|
||||
fp1 := map[string]uint32{
|
||||
"FMOVS": 0x1e204000, "FMOVD": 0x1e604000,
|
||||
"FABSS": 0x1e20c000, "FABSD": 0x1e60c000,
|
||||
"FNEGS": 0x1e214000, "FNEGD": 0x1e614000,
|
||||
"FSQRTS": 0x1e21c000, "FSQRTD": 0x1e61c000,
|
||||
"FCVTSD": 0x1e22c000, "FCVTDS": 0x1e624000,
|
||||
"FRINTNS": 0x1e244000, "FRINTND": 0x1e644000,
|
||||
"FRINTPS": 0x1e24c000, "FRINTPD": 0x1e64c000,
|
||||
"FRINTMS": 0x1e254000, "FRINTMD": 0x1e654000,
|
||||
"FRINTZS": 0x1e25c000, "FRINTZD": 0x1e65c000,
|
||||
"FRINTAS": 0x1e264000, "FRINTAD": 0x1e664000,
|
||||
"FRINTXS": 0x1e274000, "FRINTXD": 0x1e674000,
|
||||
"FRINTIS": 0x1e27c000, "FRINTID": 0x1e67c000,
|
||||
}
|
||||
for m, op := range fp1 {
|
||||
a64InstrTable[m] = a64Enc{format: a64FFPUnary, op: op}
|
||||
}
|
||||
|
||||
// ---- FP 4-operand FMA (Ra, Rm, Rn, Rd) ----
|
||||
fp4 := map[string]uint32{
|
||||
"FMADDS": 0x1f000000, "FMADDD": 0x1f400000,
|
||||
"FMSUBS": 0x1f008000, "FMSUBD": 0x1f408000,
|
||||
"FNMADDS": 0x1f200000, "FNMADDD": 0x1f600000,
|
||||
"FNMSUBS": 0x1f208000, "FNMSUBD": 0x1f608000,
|
||||
}
|
||||
for m, op := range fp4 {
|
||||
a64InstrTable[m] = a64Enc{format: a64FFP4, op: op}
|
||||
}
|
||||
|
||||
// ---- FP compare (Rm, Rn or #0, Rn) ----
|
||||
fpcmp := map[string]uint32{
|
||||
"FCMPS": 0x1e202000, "FCMPD": 0x1e602000,
|
||||
"FCMPES": 0x1e202010, "FCMPED": 0x1e602010,
|
||||
}
|
||||
for m, op := range fpcmp {
|
||||
a64InstrTable[m] = a64Enc{format: a64FFPCmp, op: op}
|
||||
}
|
||||
|
||||
// ---- FP conditional compare (Rm, Rn, #nzcv, cond) ----
|
||||
fpccmp := map[string]uint32{
|
||||
"FCCMPS": 0x1e200400, "FCCMPD": 0x1e600400,
|
||||
"FCCMPES": 0x1e200410, "FCCMPED": 0x1e600410,
|
||||
}
|
||||
for m, op := range fpccmp {
|
||||
a64InstrTable[m] = a64Enc{format: a64FFPCCmp, op: op}
|
||||
}
|
||||
|
||||
// ---- FP conditional select (Rm, Rn, Rd, cond) ----
|
||||
a64InstrTable["FCSELS"] = a64Enc{format: a64FFPSel, op: 0x1e200c00}
|
||||
a64InstrTable["FCSELD"] = a64Enc{format: a64FFPSel, op: 0x1e600c00}
|
||||
|
||||
// ---- FP ↔ integer conversion ----
|
||||
fpcvt := map[string]uint32{
|
||||
"FCVTZSD": 0x9e780000, "FCVTZSDW": 0x1e780000,
|
||||
"FCVTZSS": 0x9e380000, "FCVTZSSW": 0x1e380000,
|
||||
"FCVTZUD": 0x9e790000, "FCVTZUDW": 0x1e790000,
|
||||
"FCVTZUS": 0x9e390000, "FCVTZUSW": 0x1e390000,
|
||||
"SCVTFD": 0x9e620000, "SCVTFS": 0x9e220000,
|
||||
"SCVTFWD": 0x1e620000, "SCVTFWS": 0x1e220000,
|
||||
"UCVTFD": 0x9e630000, "UCVTFS": 0x9e230000,
|
||||
"UCVTFWD": 0x1e630000, "UCVTFWS": 0x1e230000,
|
||||
}
|
||||
for m, op := range fpcvt {
|
||||
a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op}
|
||||
}
|
||||
|
||||
// ---- FMOV between GP and FP registers ----
|
||||
a64InstrTable["FMOVGR"] = a64Enc{format: a64FFMovGR, op: 0x1e260000} // placeholder, actual encoding depends on direction
|
||||
|
||||
// ---- conditional select: CSEL, CSINC, CSINV, CSNEG ----
|
||||
csel := map[string]uint32{
|
||||
"CSEL": 0x9a800000, "CSELW": 0x1a800000,
|
||||
"CSINC": 0x9a800400, "CSINCW": 0x1a800400,
|
||||
"CSINV": 0xda800000, "CSINVW": 0x5a800000,
|
||||
"CSNEG": 0xda800400, "CSNEGW": 0x5a800400,
|
||||
}
|
||||
for m, op := range csel {
|
||||
a64InstrTable[m] = a64Enc{format: a64FCSEL, op: op}
|
||||
}
|
||||
// Aliases
|
||||
a64InstrTable["CSET"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
|
||||
a64InstrTable["CSETW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
|
||||
a64InstrTable["CSETM"] = a64Enc{format: a64FCSEL, op: 0xda800000}
|
||||
a64InstrTable["CSETMW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
|
||||
a64InstrTable["CINC"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
|
||||
a64InstrTable["CINCW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
|
||||
a64InstrTable["CINV"] = a64Enc{format: a64FCSEL, op: 0xda800000}
|
||||
a64InstrTable["CINVW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
|
||||
a64InstrTable["CNEG"] = a64Enc{format: a64FCSEL, op: 0xda800400}
|
||||
a64InstrTable["CNEGW"] = a64Enc{format: a64FCSEL, op: 0x5a800400}
|
||||
|
||||
// ---- CRC32 ----
|
||||
crc32 := map[string]uint32{
|
||||
"CRC32B": 0x1ac04000, "CRC32H": 0x1ac04400,
|
||||
"CRC32W": 0x1ac04800, "CRC32X": 0x9ac04c00,
|
||||
"CRC32CB": 0x1ac05000, "CRC32CH": 0x1ac05400,
|
||||
"CRC32CW": 0x1ac05800, "CRC32CX": 0x9ac05c00,
|
||||
}
|
||||
for m, op := range crc32 {
|
||||
a64InstrTable[m] = a64Enc{format: a64FCRC32, op: op}
|
||||
}
|
||||
|
||||
// ---- exclusive load/store ----
|
||||
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
|
||||
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
|
||||
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
|
||||
a64InstrTable["LDXRW"] = a64Enc{format: a64FExcl, op: 0x885f7c00}
|
||||
a64InstrTable["LDAXR"] = a64Enc{format: a64FExcl, op: 0xc85ffc00}
|
||||
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
|
||||
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
|
||||
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
|
||||
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
|
||||
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
|
||||
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
|
||||
a64InstrTable["STXRW"] = a64Enc{format: a64FExcl, op: 0x88007c00}
|
||||
a64InstrTable["STLXR"] = a64Enc{format: a64FExcl, op: 0xc800fc00}
|
||||
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
|
||||
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
|
||||
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
|
||||
|
||||
// ---- LSE atomics ----
|
||||
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
|
||||
a64InstrTable["LDADDW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x00<<10}
|
||||
a64InstrTable["LDADDB"] = a64Enc{format: a64FLSE, op: 0<<30 | 0x1c1<<21 | 0x00<<10}
|
||||
a64InstrTable["LDADDH"] = a64Enc{format: a64FLSE, op: 1<<30 | 0x1c1<<21 | 0x00<<10}
|
||||
a64InstrTable["CASD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x45<<21 | 0x1f<<10}
|
||||
a64InstrTable["CASW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x45<<21 | 0x1f<<10}
|
||||
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
|
||||
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
|
||||
|
||||
// ---- SIMD basics ----
|
||||
a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400}
|
||||
a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400}
|
||||
a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00}
|
||||
}
|
||||
|
||||
// ---- load/store helper tables ----
|
||||
|
||||
// a64LSType describes the load/store parameters for a MOV width mnemonic.
|
||||
type a64LSType struct {
|
||||
size int // 0=byte, 1=half, 2=word, 3=dword
|
||||
V int // 0=integer, 1=FP
|
||||
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
|
||||
}
|
||||
|
||||
// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters.
|
||||
// For loads, opc selects signed vs unsigned; for stores, we flip the opc.
|
||||
var a64LoadTable = map[string]a64LSType{
|
||||
"MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset)
|
||||
"MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned)
|
||||
"MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit)
|
||||
"MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned)
|
||||
"MOVH": {1, 0, 2}, // LDRSH (16-bit signed)
|
||||
"MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned)
|
||||
"MOVB": {0, 0, 2}, // LDRSB (8-bit signed)
|
||||
"FMOVS": {2, 1, 1}, // LDR S (32-bit FP)
|
||||
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
|
||||
}
|
||||
|
||||
// a64StoreOpc returns the store opc for a given load type.
|
||||
// For integer: store opc = 00 (the load opc bits cleared).
|
||||
// For FP: store opc = 00 (same pattern).
|
||||
func a64StoreOpc(t a64LSType) int {
|
||||
if t.V == 1 {
|
||||
return 0 // FP store
|
||||
}
|
||||
return 0 // integer store
|
||||
}
|
||||
|
||||
// arm64RegClass discriminates integer (R), floating-point (F) registers for
|
||||
// the MOV pseudo-instruction.
|
||||
type arm64RegClass int
|
||||
|
||||
const (
|
||||
arm64ClsNone arm64RegClass = iota
|
||||
arm64ClsGR
|
||||
arm64ClsFP
|
||||
)
|
||||
|
||||
// arm64RegClassOf reports the register class of a register operand name.
|
||||
func arm64RegClassOf(name string) arm64RegClass {
|
||||
switch {
|
||||
case name == "":
|
||||
return arm64ClsNone
|
||||
case len(name) >= 1 && name[0] == 'F':
|
||||
return arm64ClsFP
|
||||
default:
|
||||
return arm64ClsGR
|
||||
}
|
||||
}
|
||||
|
||||
// arm64Movcon returns the shift (in units of 16 bits) at which a non-zero
|
||||
// 16-bit chunk of v sits, or -1 if v cannot be represented as a single
|
||||
// MOVZ/MOVN immediate. This is the Go toolchain's movcon function.
|
||||
func arm64Movcon(v int64) int {
|
||||
for s := 0; s < 64; s += 16 {
|
||||
if (uint64(v) &^ (uint64(0xFFFF) << uint(s))) == 0 {
|
||||
return s
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
@@ -0,0 +1,574 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
func TestArm64LDRSTREncoding(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
got uint32
|
||||
want uint32
|
||||
}{
|
||||
{"LDR X4, [SP, #56]", a64LSU(3, 0, 1, 7, 31, 4), 0xf9401fe4},
|
||||
{"STR X4, [SP, #64]", a64LSU(3, 0, 0, 8, 31, 4), 0xf90023e4},
|
||||
{"STR X5, [SP, #32]", a64LSU(3, 0, 0, 4, 31, 5), 0xf90013e5},
|
||||
{"LDR X6, [SP, #32]", a64LSU(3, 0, 1, 4, 31, 6), 0xf94013e6},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
if tt.got != tt.want {
|
||||
t.Errorf("%s: got %08x, want %08x", tt.name, tt.got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64PrologueEncoding(t *testing.T) {
|
||||
fi := arm64FrameInfo{autosize: 48, frame: 32, leaf: false}
|
||||
pro := arm64Prologue(fi)
|
||||
if len(pro) != 12 {
|
||||
t.Fatalf("prologue length: got %d, want 12", len(pro))
|
||||
}
|
||||
expected := []uint32{0xf81d0ffe, 0xf81f83fd, 0xd10023fd}
|
||||
for i, w := range leWords(pro) {
|
||||
if w != expected[i] {
|
||||
t.Errorf("prologue word %d: got %08x, want %08x", i, w, expected[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64EpilogueSmallEncoding(t *testing.T) {
|
||||
fi := arm64FrameInfo{autosize: 48, frame: 32, leaf: false}
|
||||
ret := arm64Return(fi)
|
||||
if len(ret) != 12 {
|
||||
t.Fatalf("epilogue length: got %d, want 12", len(ret))
|
||||
}
|
||||
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #48; RET
|
||||
expected := []uint32{0xf85f83fd, 0xf84307fe, 0xd65f03c0}
|
||||
for i, w := range leWords(ret) {
|
||||
if w != expected[i] {
|
||||
t.Errorf("epilogue word %d: got %08x, want %08x", i, w, expected[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64LargeFrameEncoding(t *testing.T) {
|
||||
fi := arm64FrameInfo{autosize: 272, frame: 256, leaf: false}
|
||||
pro := arm64Prologue(fi)
|
||||
if len(pro) != 16 {
|
||||
t.Fatalf("prologue length: got %d, want 16", len(pro))
|
||||
}
|
||||
expected := []uint32{0xd10443f4, 0xa93ffa9d, 0x9100029f, 0xd10023fd}
|
||||
for i, w := range leWords(pro) {
|
||||
if w != expected[i] {
|
||||
t.Errorf("prologue word %d: got %08x, want %08x", i, w, expected[i])
|
||||
}
|
||||
}
|
||||
|
||||
epi := arm64Return(fi)
|
||||
if len(epi) != 12 {
|
||||
t.Fatalf("epilogue length: got %d, want 12", len(epi))
|
||||
}
|
||||
eexpected := []uint32{0xa97ffbfd, 0x910443ff, 0xd65f03c0}
|
||||
for i, w := range leWords(epi) {
|
||||
if w != eexpected[i] {
|
||||
t.Errorf("epilogue word %d: got %08x, want %08x", i, w, eexpected[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64NoFrame(t *testing.T) {
|
||||
fi := arm64FrameInfo{autosize: 0, frame: 0, leaf: true}
|
||||
pro := arm64Prologue(fi)
|
||||
if len(pro) != 0 {
|
||||
t.Errorf("no-frame prologue: got %d bytes, want 0", len(pro))
|
||||
}
|
||||
ret := arm64Return(fi)
|
||||
if len(ret) != 4 {
|
||||
t.Fatalf("no-frame return: got %d bytes, want 4", len(ret))
|
||||
}
|
||||
if leWord(ret) != 0xd65f03c0 {
|
||||
t.Errorf("no-frame RET: got %08x, want d65f03c0", leWord(ret))
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64RegNum(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
want int
|
||||
}{
|
||||
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
|
||||
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
|
||||
{"F0", 0}, {"F4", 4}, {"F31", 31},
|
||||
{"INVALID", -1}, {"X0", -1}, {"", -1},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
got := arm64RegNum(tt.name)
|
||||
if got != tt.want {
|
||||
t.Errorf("arm64RegNum(%q) = %d, want %d", tt.name, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64ComputeFrame(t *testing.T) {
|
||||
src := "TEXT ·f(SB), NOSPLIT, $32-0\n\tADD\tR4, R5\n\tRET\n"
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
fi := arm64ComputeFrame(f.Decls[0].(*ast.Text))
|
||||
if fi.frame != 32 {
|
||||
t.Errorf("frame: got %d, want 32", fi.frame)
|
||||
}
|
||||
if fi.autosize != 48 { // 32+8=40, aligned to48
|
||||
t.Errorf("autosize: got %d, want 48", fi.autosize)
|
||||
}
|
||||
// ADD + RET with no CALL/BL → leaf
|
||||
if !fi.leaf {
|
||||
t.Error("expected leaf")
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64IsLeaf(t *testing.T) {
|
||||
src := "TEXT ·f(SB), NOSPLIT, $0-0\n\tADD\tR4, R5\n\tRET\n"
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
if !arm64IsLeaf(f.Decls[0].(*ast.Text)) {
|
||||
t.Error("expected leaf")
|
||||
}
|
||||
|
||||
src2 := "TEXT ·f(SB), NOSPLIT, $0-0\n\tBL\tother(SB)\n\tRET\n"
|
||||
f2, errs := parser.Parse("test_arm64.s", src2)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
if arm64IsLeaf(f2.Decls[0].(*ast.Text)) {
|
||||
t.Error("expected non-leaf")
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64Bitmask(t *testing.T) {
|
||||
tests := []struct {
|
||||
v uint64
|
||||
sf int
|
||||
N, immr, imms uint32
|
||||
ok bool
|
||||
}{
|
||||
{1, 1, 1, 0, 0, true}, // single bit at pos 0
|
||||
{2, 1, 1, 63, 0, true}, // single bit at pos 1 (immr = esize-1)
|
||||
{0, 1, 0, 0, 0, false}, // zero is not a bitmask
|
||||
{0xFFFFFFFFFFFFFFFF, 1, 0, 0, 0, false}, // all ones is not a bitmask
|
||||
{0x5555555555555555, 1, 0, 0, 0x3E, true}, // alternating bits (esize=2, ones=1)
|
||||
{0xFFFFFFFF00000000, 1, 1, 32, 31, true}, // upper 32 bits set (esize=64, ones=32)
|
||||
}
|
||||
for _, tt := range tests {
|
||||
N, immr, imms, ok := arm64Bitmask(tt.v, tt.sf)
|
||||
if ok != tt.ok {
|
||||
t.Errorf("arm64Bitmask(%#x, %d): ok=%v, want %v", tt.v, tt.sf, ok, tt.ok)
|
||||
continue
|
||||
}
|
||||
if ok && (N != tt.N || immr != tt.immr || imms != tt.imms) {
|
||||
t.Errorf("arm64Bitmask(%#x, %d): N=%d immr=%d imms=%d, want N=%d immr=%d imms=%d",
|
||||
tt.v, tt.sf, N, immr, imms, tt.N, tt.immr, tt.imms)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64AssembleFile(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
|
||||
TEXT ·simple(SB), NOSPLIT, $0-0
|
||||
MOV R4, R5
|
||||
ADD R4, R5, R6
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
if len(img.Funcs) != 1 {
|
||||
t.Fatalf("got %d funcs, want 1", len(img.Funcs))
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
if fn.Name != "simple" {
|
||||
t.Errorf("func name: got %q, want %q", fn.Name, "simple")
|
||||
}
|
||||
//3 instructions ×4 bytes =12
|
||||
if fn.Size != 12 {
|
||||
t.Errorf("func size: got %d, want 12", fn.Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64AssembleFileWithFrame(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
|
||||
TEXT ·framed(SB), NOSPLIT, $16-8
|
||||
MOVD arg+0(FP), R4
|
||||
ADD $1, R4, R4
|
||||
MOVD R4, ret+0(FP)
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
if len(img.Funcs) != 1 {
|
||||
t.Fatalf("got %d funcs, want 1", len(img.Funcs))
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
if fn.Frame != 16 {
|
||||
t.Errorf("frame: got %d, want 16", fn.Frame)
|
||||
}
|
||||
// Prologue (3×4=12) + body (3×4=12) + RET epilogue (3×4=12) = 36
|
||||
if fn.Size != 36 {
|
||||
t.Errorf("func size: got %d, want 36", fn.Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64AssembleFileWithBranches(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
|
||||
TEXT ·branch(SB), NOSPLIT, $0-0
|
||||
BEQ done
|
||||
BNE skip
|
||||
skip:
|
||||
ADD R4, R5
|
||||
done:
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
if fn.Size != 16 {
|
||||
t.Errorf("func size: got %d, want 16", fn.Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64AssembleFileWithJumpChain(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
|
||||
TEXT ·chain(SB), NOSPLIT, $0-0
|
||||
BNE skip
|
||||
ADD R4, R5
|
||||
RET
|
||||
skip:
|
||||
B target
|
||||
target:
|
||||
ADD R6, R7
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
// BNE should be redirected past skip→target to target directly.
|
||||
if img.Funcs[0].Size != 24 {
|
||||
t.Errorf("func size: got %d, want 24", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64AssembleErrors(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
src string
|
||||
}{
|
||||
{"unsupported", "TEXT ·f(SB), NOSPLIT, $0-0\n\tINVALID\tR4, R5\n\tRET\n"},
|
||||
{"undefined label", "TEXT ·f(SB), NOSPLIT, $0-0\n\tB\tnosuch\n\tRET\n"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
f, errs := parser.Parse("test_arm64.s", tt.src)
|
||||
if len(errs) > 0 {
|
||||
return // parse error, that's fine
|
||||
}
|
||||
_, err := AssembleFileARM64(f)
|
||||
if err == nil {
|
||||
t.Error("expected error, got nil")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64Movcon(t *testing.T) {
|
||||
tests := []struct {
|
||||
v int64
|
||||
want int
|
||||
}{
|
||||
{0, 0}, // 0 fits at shift 0
|
||||
{1, 0}, // single bit at shift 0
|
||||
{0x10000, 16}, // single bit at shift 16
|
||||
{0x100000000, 32}, // single bit at shift 32
|
||||
{0xFF, 0}, // 0xFF fits at shift 0
|
||||
{0x12345, -1}, // multiple chunks, not movcon
|
||||
}
|
||||
for _, tt := range tests {
|
||||
got := arm64Movcon(tt.v)
|
||||
if got != tt.want {
|
||||
t.Errorf("arm64Movcon(%#x) = %d, want %d", tt.v, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64RegClassOf(t *testing.T) {
|
||||
if arm64RegClassOf("R4") != arm64ClsGR {
|
||||
t.Error("R4 should be GR")
|
||||
}
|
||||
if arm64RegClassOf("F4") != arm64ClsFP {
|
||||
t.Error("F4 should be FP")
|
||||
}
|
||||
if arm64RegClassOf("") != arm64ClsNone {
|
||||
t.Error("empty should be None")
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64ResolvePseudo(t *testing.T) {
|
||||
fi := arm64FrameInfo{autosize: 48, frame: 32}
|
||||
// FP: offset = sym.Offset + autosize +8
|
||||
base, off := arm64ResolvePseudo(&ast.Symbol{Pseudo: "FP", Offset: 0}, fi)
|
||||
if base != 31 || off != 56 {
|
||||
t.Errorf("FP: base=%d off=%d, want 31, 56", base, off)
|
||||
}
|
||||
// SP: offset = sym.Offset + frame +8
|
||||
base, off = arm64ResolvePseudo(&ast.Symbol{Pseudo: "SP", Offset: -8}, fi)
|
||||
if base != 31 || off != 32 {
|
||||
t.Errorf("SP: base=%d off=%d, want 31, 32", base, off)
|
||||
}
|
||||
// SB: unresolved
|
||||
base, _ = arm64ResolvePseudo(&ast.Symbol{Pseudo: "SB"}, fi)
|
||||
if base != -1 {
|
||||
t.Errorf("SB: base=%d, want -1", base)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64FPSel tests FP conditional select encoding.
|
||||
func TestArm64FPSel(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
FCSELD GE, F10, F11, F12
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
// FCSELD should be 4 bytes + RET 4 bytes = 8
|
||||
if img.Funcs[0].Size != 8 {
|
||||
t.Errorf("size: got %d, want 8", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64FPCvt tests FP conversion encoding.
|
||||
func TestArm64FPCvt(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
FCVTZSD F4, R0
|
||||
SCVTFD R4, F8
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
if img.Funcs[0].Size != 12 {
|
||||
t.Errorf("size: got %d, want 12", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64CSEL tests conditional select encoding.
|
||||
func TestArm64CSEL(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
CSEL EQ, R0, R1, R2
|
||||
CSET NE, R3
|
||||
CINC GE, R4, R5
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
if img.Funcs[0].Size != 16 {
|
||||
t.Errorf("size: got %d, want 16", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64CRC32 tests CRC32 encoding.
|
||||
func TestArm64CRC32(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
CRC32B R0, R2
|
||||
CRC32W R6, R8
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
if img.Funcs[0].Size != 12 {
|
||||
t.Errorf("size: got %d, want 12", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64Bitfield tests bitfield/shift encoding.
|
||||
func TestArm64Bitfield(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
ASR $4, R0, R1
|
||||
LSL $12, R4, R5
|
||||
EXTR $8, R0, R1, R2
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
if img.Funcs[0].Size != 16 {
|
||||
t.Errorf("size: got %d, want 16", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64SIMD tests SIMD encoding (via the instruction table).
|
||||
func TestArm64SIMD(t *testing.T) {
|
||||
// Verify SIMD instructions are in the table.
|
||||
for _, mnem := range []string{"VADD", "VSUB", "VMUL"} {
|
||||
if _, ok := a64InstrTable[mnem]; !ok {
|
||||
t.Errorf("%s not in instruction table", mnem)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64LoadImm64 tests 64-bit immediate loading.
|
||||
func TestArm64LoadImm64(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
MOVD $0x123456789ABCDEF0, R0
|
||||
MOVD $0, R1
|
||||
MOVD $1, R2
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
// $0x123456789ABCDEF0 needs 4 MOVZ/MOVK instructions (16 bytes)
|
||||
// $0 is 1 instruction (4 bytes)
|
||||
// $1 is 1 bitmask instruction (4 bytes)
|
||||
// RET is 1 instruction (4 bytes)
|
||||
if img.Funcs[0].Size != 28 {
|
||||
t.Errorf("size: got %d, want 28", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64BranchCond tests conditional branch encoding.
|
||||
func TestArm64BranchCond(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
BEQ done
|
||||
BNE done
|
||||
BGE done
|
||||
BLT done
|
||||
ADD R4, R5
|
||||
done:
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
// 4 branches + 1 ADD + 1 RET = 24 bytes
|
||||
if img.Funcs[0].Size != 24 {
|
||||
t.Errorf("size: got %d, want 24", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64Errors tests error paths.
|
||||
func TestArm64Errors(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
src string
|
||||
}{
|
||||
{"bad mnemonic", "TEXT ·f(SB), NOSPLIT, $0-0\n\tINVALID\tR4\n\tRET\n"},
|
||||
{"bad label", "TEXT ·f(SB), NOSPLIT, $0-0\n\tB\tnosuch\n\tRET\n"},
|
||||
{"bad register", "TEXT ·f(SB), NOSPLIT, $0-0\n\tADD\tR99, R0\n\tRET\n"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
f, errs := parser.Parse("test_arm64.s", tt.src)
|
||||
if len(errs) > 0 {
|
||||
return
|
||||
}
|
||||
_, err := AssembleFileARM64(f)
|
||||
if err == nil {
|
||||
t.Error("expected error, got nil")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// leWord reads a little-endian uint32 from b.
|
||||
func leWord(b []byte) uint32 {
|
||||
return uint32(b[0]) | uint32(b[1])<<8 | uint32(b[2])<<16 | uint32(b[3])<<24
|
||||
}
|
||||
|
||||
// leWords reads all little-endian uint32s from b.
|
||||
func leWords(b []byte) []uint32 {
|
||||
n := len(b) / 4
|
||||
w := make([]uint32, n)
|
||||
for i := range w {
|
||||
w[i] = leWord(b[i*4:])
|
||||
}
|
||||
return w
|
||||
}
|
||||
@@ -0,0 +1,237 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
// arm64 frame mapping, matching the Go toolchain's arm64 backend.
|
||||
//
|
||||
// Go's arm64 functions use R29 as the frame pointer (FP) and R30 as the link
|
||||
// register (LR). R31 is the stack pointer (SP). FP and SP in the source
|
||||
// are synthetic pseudo-registers resolved against the hardware SP and the
|
||||
// frame size.
|
||||
//
|
||||
// The autosize is the real stack adjustment: the declared local frame plus
|
||||
// 8 bytes for the saved link register, rounded up to a 16-byte multiple.
|
||||
// The toolchain adds an "extrasize" to align: if autosize%16 == 8, add 8;
|
||||
// if autosize%16 == 0, add 16.
|
||||
//
|
||||
// Prologue (autosize > 0, small frame ≤ 0xf0):
|
||||
//
|
||||
// MOVD.W LR, -autosize(SP) // pre-index: SP -= autosize, store LR at SP
|
||||
// MOVD FP, -8(SP) // store FP at SP-8
|
||||
// SUB $8, SP, FP // FP = SP - 8
|
||||
//
|
||||
// Prologue (autosize > 0, large frame > 0xf0):
|
||||
//
|
||||
// SUB $autosize, SP, R20 // R20 = SP - autosize
|
||||
// STP (FP, LR), -8(R20) // store FP,LR at R20-8
|
||||
// MOVD R20, SP // SP = R20
|
||||
// SUB $8, SP, FP // FP = SP - 8
|
||||
//
|
||||
// Epilogue (non-leaf, small frame):
|
||||
//
|
||||
// ADD $autosize-8, SP, FP // restore FP
|
||||
// ADD $autosize, SP, SP // deallocate frame
|
||||
// MOVD -8(SP), FP // (actually the reverse of prologue)
|
||||
// Actually:
|
||||
// MOVD -8(SP), FP // load FP from SP-8
|
||||
// MOVD.P autosize(SP), LR // post-index: load LR, SP += autosize
|
||||
//
|
||||
// Epilogue (non-leaf, large frame):
|
||||
// ADD $autosize-8, SP, FP
|
||||
// ADD $autosize, SP, SP
|
||||
// Actually:
|
||||
// LDP -8(SP), (FP, LR) // load FP,LR
|
||||
// ADD $autosize, SP, SP // deallocate frame
|
||||
//
|
||||
// Epilogue (leaf with frame):
|
||||
// ADD $autosize-8, SP, FP
|
||||
// ADD $autosize, SP, SP
|
||||
//
|
||||
// RET always emits as BR LR (0xd65f03c0).
|
||||
|
||||
import (
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
)
|
||||
|
||||
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
|
||||
type arm64FrameInfo struct {
|
||||
autosize int // the real SP adjustment (locals + saved LR + alignment)
|
||||
frame int // the declared $framesize
|
||||
args int // the declared -argsize
|
||||
noSplit bool // the NOSPLIT flag
|
||||
leaf bool // no call instructions in the body
|
||||
}
|
||||
|
||||
// arm64ComputeFrame derives the frame layout for a TEXT function.
|
||||
func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
|
||||
fi := arm64FrameInfo{
|
||||
frame: frameSize(t),
|
||||
args: argsSize(t),
|
||||
}
|
||||
for _, f := range t.Flags {
|
||||
if f == "NOSPLIT" {
|
||||
fi.noSplit = true
|
||||
}
|
||||
}
|
||||
fi.leaf = arm64IsLeaf(t)
|
||||
|
||||
if fi.frame != 0 || !fi.leaf {
|
||||
fi.autosize = fi.frame + 8 // space for the saved LR
|
||||
if fi.autosize%16 != 0 {
|
||||
// The toolchain aligns to 16: if autosize%16 == 8, add 8;
|
||||
// otherwise add whatever is needed.
|
||||
fi.autosize += 16 - (fi.autosize % 16)
|
||||
}
|
||||
}
|
||||
return fi
|
||||
}
|
||||
|
||||
// arm64IsLeaf reports whether a function contains no call instructions
|
||||
// (BL/CALL), matching the toolchain's LEAF mark.
|
||||
func arm64IsLeaf(t *ast.Text) bool {
|
||||
for _, stmt := range t.Body {
|
||||
in, ok := stmt.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
switch strings.ToUpper(in.Mnemonic.Text) {
|
||||
case "BL", "CALL":
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// arm64Prologue returns the prologue bytes for an arm64 function.
|
||||
func arm64Prologue(fi arm64FrameInfo) []byte {
|
||||
if fi.autosize == 0 {
|
||||
return nil
|
||||
}
|
||||
if fi.autosize <= 0xf0 {
|
||||
// Small frame: MOVD.W LR, -autosize(SP); MOVD FP, -8(SP); SUB $8, SP, FP
|
||||
return a64WordsLE(
|
||||
arm64PreStoreImm(3, 0, int32(-fi.autosize), 31, 30), // STR.W LR, -autosize(SP) (pre-index store)
|
||||
arm64UnscaledStore(3, 0, -8, 31, 29), // STUR FP, [SP, #-8]
|
||||
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
|
||||
)
|
||||
}
|
||||
// Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP
|
||||
return a64WordsLE(
|
||||
a64AddSub(1, 1, 0, 0, uint32(fi.autosize), 31, 20), // SUB $autosize, SP, R20
|
||||
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
|
||||
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
|
||||
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
|
||||
)
|
||||
}
|
||||
|
||||
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
|
||||
// deallocate the frame when present) followed by RET (BR LR).
|
||||
func arm64Return(fi arm64FrameInfo) []byte {
|
||||
var ws []uint32
|
||||
if fi.autosize != 0 {
|
||||
if fi.leaf {
|
||||
// Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP
|
||||
ws = append(ws,
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize-8), 31, 29), // ADD $autosize-8, SP, FP
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
|
||||
)
|
||||
} else if fi.autosize <= 0xf0 {
|
||||
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize
|
||||
ws = append(ws,
|
||||
arm64UnscaledLoad(3, 0, -8, 31, 29), // LDR FP, [SP, #-8]
|
||||
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
|
||||
)
|
||||
} else {
|
||||
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
|
||||
ws = append(ws,
|
||||
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
|
||||
)
|
||||
}
|
||||
}
|
||||
// RET: BR LR (0xd65f03c0)
|
||||
ws = append(ws, a64UncondBranch(2, 30, 0)) // opc=2(RET), Rn=LR(30), Rd=0
|
||||
return a64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// arm64PrologueSpadjPC returns the function-relative byte offset where the
|
||||
// prologue has finished decrementing SP (the delta becomes autosize).
|
||||
func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
|
||||
if fi.autosize == 0 {
|
||||
return 0
|
||||
}
|
||||
if fi.autosize <= 0xf0 {
|
||||
return 4 // MOVD.W instruction decrements SP
|
||||
}
|
||||
return 8 // SUB + STP + MOVD (3 instructions, SP updated at the MOVD)
|
||||
}
|
||||
|
||||
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
||||
// (but not including) the final RET instruction.
|
||||
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
|
||||
if fi.autosize == 0 {
|
||||
return 0
|
||||
}
|
||||
if fi.leaf {
|
||||
return 8 // ADD + ADD
|
||||
}
|
||||
if fi.autosize <= 0xf0 {
|
||||
return 8 // LDR + LDR.P
|
||||
}
|
||||
return 8 // LDP + ADD
|
||||
}
|
||||
|
||||
// arm64ResolvePseudo translates a pseudo-register memory reference into a
|
||||
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
|
||||
// x+N(SP) → (N + frame + 8)(SP). Returns base = -1 for an unresolvable
|
||||
// reference (SB: static data, handled by the relocation path).
|
||||
//
|
||||
// The Go toolchain resolves all pseudo-register references against the
|
||||
// hardware stack pointer (R31/SP): FP references add autosize+8 (the
|
||||
// distance from SP after the prologue to the caller's argument area),
|
||||
// SP references add frame+8 (the distance to the local area).
|
||||
func arm64ResolvePseudo(sym *ast.Symbol, fi arm64FrameInfo) (base int, off int32) {
|
||||
if sym == nil {
|
||||
return -1, 0
|
||||
}
|
||||
switch sym.Pseudo {
|
||||
case "FP":
|
||||
return 31, int32(sym.Offset) + int32(fi.autosize) + 8
|
||||
case "SP":
|
||||
return 31, int32(sym.Offset) + int32(fi.frame) + 8
|
||||
case "SB":
|
||||
return -1, int32(sym.Offset)
|
||||
}
|
||||
return -1, 0
|
||||
}
|
||||
|
||||
// arm64PreStoreImm encodes a pre-index store (STR with writeback):
|
||||
// size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
|
||||
func arm64PreStoreImm(size, V int, imm9 int32, rn, rt int) uint32 {
|
||||
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 |
|
||||
3<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
||||
}
|
||||
|
||||
// arm64UnscaledStore encodes an unscaled store (STUR):
|
||||
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt.
|
||||
func arm64UnscaledStore(size, V int, imm9 int32, rn, rt int) uint32 {
|
||||
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 |
|
||||
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
||||
}
|
||||
|
||||
// arm64UnscaledLoad encodes an unscaled load (LDUR):
|
||||
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt.
|
||||
func arm64UnscaledLoad(size, V int, imm9 int32, rn, rt int) uint32 {
|
||||
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
|
||||
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
||||
}
|
||||
|
||||
// arm64PostLoad encodes a post-index load (LDR with post-increment):
|
||||
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
|
||||
func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 {
|
||||
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
|
||||
1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
||||
}
|
||||
+113
-19
@@ -23,6 +23,44 @@ import (
|
||||
// operands require relocations and are not yet supported; the SIMD (VEX/AVX2)
|
||||
// integer and shuffle/extract/permute/move set is in.
|
||||
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
|
||||
code, _, labels, _, _, err := assemble(t, nil)
|
||||
return code, labels, err
|
||||
}
|
||||
|
||||
// linkInfo carries file-level symbol context into a single-function assembly:
|
||||
// the set of static symbols a GLOBL in the same file defines. A nil link
|
||||
// rejects SB operands outright (single-function assembly cannot resolve
|
||||
// them). When allowExternal is set, a reference to a symbol no GLOBL in the
|
||||
// file defines is recorded as an external relocation instead of failing —
|
||||
// the object-file emitters resolve it at link time.
|
||||
type linkInfo struct {
|
||||
symbols map[string]bool
|
||||
allowExternal bool
|
||||
}
|
||||
|
||||
// sbPatch is a function-relative static-symbol relocation: the disp32 field
|
||||
// at off must become the symbol's address minus after, where after is the
|
||||
// function-relative address just past the instruction.
|
||||
type sbPatch struct {
|
||||
off int
|
||||
after int
|
||||
name string
|
||||
addend int64
|
||||
}
|
||||
|
||||
// spadjStep is one stack-adjustment boundary within a function: Value is the
|
||||
// SP delta from the entry state (just below the return address) in effect
|
||||
// from PC (function-relative) until the next step. The steps feed the
|
||||
// pcsp table of the object-file emitters.
|
||||
type spadjStep struct {
|
||||
pc int
|
||||
value int
|
||||
}
|
||||
|
||||
// assemble encodes a TEXT body, returning the machine code, the static-symbol
|
||||
// patch sites (for the file-level layout to resolve), the label table and the
|
||||
// stack-adjustment boundaries.
|
||||
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
|
||||
fi := computeFrame(t)
|
||||
chain := jumpChain(t)
|
||||
resolve := func(name string) string {
|
||||
@@ -44,9 +82,9 @@ func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
|
||||
case *ast.Label:
|
||||
offsets[s.Name.Text] = pos
|
||||
case *ast.Instr:
|
||||
sz, err := instrSize(s, fi, long[i])
|
||||
sz, err := instrSize(s, fi, long[i], link)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
}
|
||||
sizes[i] = sz
|
||||
pcs[i] = pos
|
||||
@@ -85,23 +123,45 @@ func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
|
||||
|
||||
// Pass 2: emit.
|
||||
out := append([]byte(nil), fi.prologue...)
|
||||
var patches []sbPatch
|
||||
var steps []spadjStep
|
||||
var lines []LineEntry
|
||||
if fi.useFP {
|
||||
// PUSHQ BP saves the return-address-relative base (+8); the MOVQ
|
||||
// changes nothing; SUBQ $size, SP completes the frame.
|
||||
steps = append(steps,
|
||||
spadjStep{1, 8},
|
||||
spadjStep{len(fi.prologue), 8 + fi.size},
|
||||
)
|
||||
}
|
||||
pos := len(fi.prologue)
|
||||
for i, stmt := range t.Body {
|
||||
s, ok := stmt.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, err := encodeInstr(s, pos, offsets, fi, long[i], resolve)
|
||||
if strings.ToUpper(s.Mnemonic.Text) == "RET" && fi.useFP {
|
||||
// The RET's epilogue prefix unwinds: ADDQ $size, SP restores
|
||||
// the saved-BP-only stack, POPQ BP the entry state.
|
||||
epi := len(fi.epilogue)
|
||||
steps = append(steps,
|
||||
spadjStep{pos + epi - 1, 8},
|
||||
spadjStep{pos + epi, 0},
|
||||
)
|
||||
}
|
||||
code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
}
|
||||
if len(code) != sizes[i] {
|
||||
return nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
||||
}
|
||||
patches = append(patches, ps...)
|
||||
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
|
||||
out = append(out, code...)
|
||||
pos += len(code)
|
||||
}
|
||||
return out, offsets, nil
|
||||
return out, patches, offsets, steps, lines, nil
|
||||
}
|
||||
|
||||
// jumpChain precomputes jump-to-jump folding: a label whose first instruction
|
||||
@@ -201,6 +261,11 @@ func subSP(size int) []byte { // SUBQ $size, SP
|
||||
if size >= -128 && size <= 127 {
|
||||
return []byte{0x48, 0x83, 0xEC, byte(int8(size))}
|
||||
}
|
||||
// 128..255 do not fit SUB's unsigned imm8, but the Go assembler
|
||||
// switches to ADDQ $-size, SP whose sign-extended imm8 does.
|
||||
if size >= -255 && size <= 255 {
|
||||
return []byte{0x48, 0x83, 0xC4, byte(int8(-size))}
|
||||
}
|
||||
return append([]byte{0x48, 0x81, 0xEC}, le32(int64(size))...)
|
||||
}
|
||||
|
||||
@@ -214,12 +279,12 @@ func addSP(size int) []byte { // ADDQ $size, SP
|
||||
// instrSize returns the encoded length of an instruction (layout pass).
|
||||
// encodeInstr already includes the epilogue for a RET in a frame-pointer
|
||||
// function; jumps use their short or long form (never an epilogue).
|
||||
func instrSize(s *ast.Instr, fi frameInfo, long bool) (int, error) {
|
||||
func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, error) {
|
||||
mnem := strings.ToUpper(s.Mnemonic.Text)
|
||||
if isJumpMnemonic(mnem) {
|
||||
return jumpSize(mnem, long), nil
|
||||
}
|
||||
code, err := encodeInstr(s, 0, nil, fi, false, nil)
|
||||
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
@@ -254,7 +319,7 @@ func jumpSize(mnem string, long bool) int {
|
||||
// (relative to pc, the instruction's own offset). A RET in a frame-pointer
|
||||
// function is prefixed with the epilogue. resolve, when non-nil, redirects a
|
||||
// jump label through the jump-to-jump chain before the offset lookup.
|
||||
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string) ([]byte, error) {
|
||||
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, error) {
|
||||
mnem := strings.ToUpper(s.Mnemonic.Text)
|
||||
|
||||
var prefix []byte
|
||||
@@ -263,32 +328,48 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
||||
}
|
||||
|
||||
var code []byte
|
||||
var ps []sbPatch
|
||||
var err error
|
||||
if isJumpMnemonic(mnem) {
|
||||
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
|
||||
} else {
|
||||
code, err = encodeNormal(s, fi)
|
||||
code, ps, err = encodeNormal(s, fi, link)
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
return nil, nil, err
|
||||
}
|
||||
return append(prefix, code...), nil
|
||||
// Anchor the patch fields at function-relative positions: off indexes the
|
||||
// disp32 field, after is the address just past the instruction.
|
||||
body := pc + len(prefix)
|
||||
for i := range ps {
|
||||
ps[i].off += body
|
||||
ps[i].after = body + len(code)
|
||||
}
|
||||
return append(prefix, code...), ps, nil
|
||||
}
|
||||
|
||||
func encodeNormal(s *ast.Instr, fi frameInfo) ([]byte, error) {
|
||||
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) {
|
||||
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
|
||||
if size == 0 {
|
||||
size = 8
|
||||
}
|
||||
ops := make([]Operand, len(s.Operands))
|
||||
for i, op := range s.Operands {
|
||||
o, err := operandFromAST(op, size, fi)
|
||||
o, err := operandFromAST(op, size, fi, link)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
return nil, nil, err
|
||||
}
|
||||
ops[i] = o
|
||||
}
|
||||
return Encode(s.Mnemonic.Text, ops...)
|
||||
e := &enc{}
|
||||
if err := e.encode(s.Mnemonic.Text, ops); err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
ps := make([]sbPatch, len(e.patches))
|
||||
for i, p := range e.patches {
|
||||
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
|
||||
}
|
||||
return e.out, ps, nil
|
||||
}
|
||||
|
||||
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
|
||||
@@ -345,7 +426,7 @@ var spReg = Reg{idx: 4, size: 8}
|
||||
|
||||
// operandFromAST converts a parsed operand into an encoder Operand, applying
|
||||
// the frame translation to FP/SP pseudo-register operands.
|
||||
func operandFromAST(op *ast.Operand, size int, fi frameInfo) (Operand, error) {
|
||||
func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) {
|
||||
switch op.Kind {
|
||||
case ast.OpImmediate:
|
||||
if op.Imm.HasVal {
|
||||
@@ -371,9 +452,22 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo) (Operand, error) {
|
||||
off := fi.spAdjust + a.Sym.Offset
|
||||
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
|
||||
}
|
||||
// SB (global symbol) needs a relocation — not yet supported.
|
||||
// SB (global symbol): a symbol defined in the same file (GLOBL) is
|
||||
// encoded RIP-relative and resolved by the file-level layout;
|
||||
// anything not defined here needs object-file emission.
|
||||
if a.Sym != nil && a.Sym.Pseudo == "SB" {
|
||||
return nil, fmt.Errorf("SB (global symbol) operands need relocation support (pending)")
|
||||
if link == nil || link.symbols == nil {
|
||||
return nil, fmt.Errorf("symbol %q needs file-level assembly (AssembleFile)", a.Sym.Name)
|
||||
}
|
||||
if !link.symbols[a.Sym.Name] {
|
||||
if a.Sym.Static {
|
||||
return nil, fmt.Errorf("undefined symbol %q", a.Sym.Name)
|
||||
}
|
||||
if !link.allowExternal {
|
||||
return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name)
|
||||
}
|
||||
}
|
||||
return sbMem{size: size, name: a.Sym.Name, addend: a.Sym.Offset}, nil
|
||||
}
|
||||
|
||||
// Memory with a real base register: (base), off(base), (base)(index*scale).
|
||||
|
||||
+35
-1
@@ -62,7 +62,7 @@ TEXT ·f(SB), NOSPLIT, $0
|
||||
XORQ AX, AX
|
||||
loop:
|
||||
ADDQ $1, AX
|
||||
CMPQ $10, AX
|
||||
CMPQ AX, $10
|
||||
JLT loop
|
||||
RET
|
||||
`)
|
||||
@@ -317,3 +317,37 @@ end:
|
||||
t.Errorf("jump-folding mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssemblePrefetch(t *testing.T) {
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·pf(SB), NOSPLIT, $0
|
||||
PREFETCHNTA (AX)
|
||||
PREFETCHT0 (BX)
|
||||
PREFETCHT1 8(CX)
|
||||
PREFETCHT2 -1(AX)(R12*1)
|
||||
RET
|
||||
`)
|
||||
code, _, err := Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble: %v", err)
|
||||
}
|
||||
got := strings.Join(disasm(t, code), "\n")
|
||||
want := strings.Join([]string{
|
||||
"prefetchnta zmmword ptr [rax]",
|
||||
"prefetcht0 zmmword ptr [rbx]",
|
||||
"prefetcht1 zmmword ptr [rcx+0x8]",
|
||||
"prefetcht2 zmmword ptr [rax+r12-0x1]",
|
||||
"ret",
|
||||
}, "\n")
|
||||
if got != want {
|
||||
t.Errorf("prefetch disassembly mismatch:\n got:\n%s\n want:\n%s", got, want)
|
||||
}
|
||||
// Byte-level expectations: 0F 18 with the variant in the reg field.
|
||||
if hex := hexBytes(code[:3]); hex != "0f 18 00" {
|
||||
t.Errorf("PREFETCHNTA bytes: got %s, want 0f 18 00", hex)
|
||||
}
|
||||
if hex := hexBytes(code[3:6]); hex != "0f 18 0b" {
|
||||
t.Errorf("PREFETCHT0 bytes: got %s, want 0f 18 0b", hex)
|
||||
}
|
||||
}
|
||||
|
||||
+321
@@ -0,0 +1,321 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
)
|
||||
|
||||
// This file emits ELF64 relocatable objects (ET_REL) from an assembled
|
||||
// Image: a .text section holding the function bodies, a .data section
|
||||
// holding the GLOBL initialisers, a symbol table with one symbol per TEXT
|
||||
// and GLOBL (file-local <> symbols are STB_LOCAL, the rest STB_GLOBAL), and
|
||||
// a .rela.text relocation table — one R_X86_64_PC32 entry per static-symbol
|
||||
// reference, internal references resolving against the local data symbols
|
||||
// and external ones against undefined globals. The output links with the
|
||||
// system toolchain (cc/ld) the way a hand-assembled .o would.
|
||||
|
||||
const (
|
||||
elfClass64 = 2
|
||||
elfDataLSB = 1
|
||||
elfVersion = 1
|
||||
|
||||
etREL = 1 // relocatable object
|
||||
emX8664 = 62
|
||||
|
||||
shtNull = 0
|
||||
shtProgbits = 1
|
||||
shtSymtab = 2
|
||||
shtStrtab = 3
|
||||
shtRela = 4
|
||||
|
||||
shfWrite = 1
|
||||
shfAlloc = 2
|
||||
shfExecInstr = 4
|
||||
|
||||
stbGlobal = 1
|
||||
|
||||
sttObject = 1
|
||||
sttFunc = 2
|
||||
sttSection = 3
|
||||
stInfoShift = 4
|
||||
|
||||
rX8664PC32 = 2
|
||||
)
|
||||
|
||||
// elfSym is one symbol-table entry in construction.
|
||||
type elfSym struct {
|
||||
name string
|
||||
info byte
|
||||
shndx uint16
|
||||
value uint64
|
||||
size uint64
|
||||
}
|
||||
|
||||
// ELFObject returns the image as an ELF64 relocatable object file, ready for
|
||||
// the system linker. Symbol names are the TEXT and GLOBL identifiers as
|
||||
// written (the middle dot stripped); a package prefix, when present, is
|
||||
// joined with a dot. Every static-symbol reference becomes an
|
||||
// R_X86_64_PC32 relocation, so the code is position-independent and links
|
||||
// at any address.
|
||||
func (img *Image) ELFObject() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
// Section indices: 0 NULL, 1 .text, 2 .data; the tables follow.
|
||||
const (
|
||||
secText = 1
|
||||
secData = 2
|
||||
)
|
||||
|
||||
// Build the symbol table: the null entry and the two section symbols
|
||||
// come first, then the local symbols (static TEXT and GLOBL), then the
|
||||
// globals (exported TEXT and GLOBL, and the undefined externals) — ELF
|
||||
// requires every local to precede every global, and sh_info records the
|
||||
// boundary. symIdx maps a symbol name to its index for the relocations.
|
||||
var locals, globals []elfSym
|
||||
for _, fn := range img.Funcs {
|
||||
s := elfSym{
|
||||
name: objectName(fn.Pkg, fn.Name),
|
||||
info: sttFunc,
|
||||
shndx: secText,
|
||||
value: uint64(fn.Offset),
|
||||
size: uint64(fn.Size),
|
||||
}
|
||||
if fn.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, d := range img.DataSyms {
|
||||
s := elfSym{
|
||||
name: objectName(d.Pkg, d.Name),
|
||||
info: sttObject,
|
||||
shndx: secData,
|
||||
value: uint64(d.Offset),
|
||||
size: uint64(d.Size),
|
||||
}
|
||||
if d.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, name := range img.Externals {
|
||||
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
|
||||
}
|
||||
syms := []elfSym{
|
||||
{}, // the mandatory null entry
|
||||
{name: ".text", info: sttSection, shndx: secText},
|
||||
{name: ".data", info: sttSection, shndx: secData},
|
||||
}
|
||||
syms = append(syms, locals...)
|
||||
shInfo := len(syms) // first global symbol
|
||||
syms = append(syms, globals...)
|
||||
symIdx := map[string]int{}
|
||||
for i, s := range syms {
|
||||
symIdx[s.name] = i
|
||||
}
|
||||
|
||||
// Build the relocations.
|
||||
type elfRela struct {
|
||||
off uint64
|
||||
sym int
|
||||
addend int64
|
||||
}
|
||||
var relas []elfRela
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
relas = append(relas, elfRela{
|
||||
off: uint64(fn.Offset + r.Off),
|
||||
sym: idx,
|
||||
// R_X86_64_PC32 computes S + A − P with P the patch site; the
|
||||
// assembler measures the symbol from the instruction end,
|
||||
// After − Off bytes past the field, so the addend carries
|
||||
// that distance with a negative sign.
|
||||
addend: r.Addend - int64(r.After-r.Off),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Serialise the string tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
stNames.add(s.name)
|
||||
}
|
||||
stSections := newElfStrtab()
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
for _, n := range dwarfSectionNames {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
// Section presence: .rela.text only when there are relocations.
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6 // NULL, .text, .data, .symtab, .strtab, .shstrtab
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Lay the file out: header, section data, section headers.
|
||||
var out []byte
|
||||
out = append(out, make([]byte, 64)...) // ELF header, filled last
|
||||
|
||||
align := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
|
||||
align(16)
|
||||
textOff := len(out)
|
||||
out = append(out, img.Code...)
|
||||
|
||||
align(16)
|
||||
dataOff := len(out)
|
||||
out = append(out, img.Data...)
|
||||
|
||||
align(8)
|
||||
symtabOff := len(out)
|
||||
for _, s := range syms {
|
||||
var b [24]byte
|
||||
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
|
||||
b[4] = s.info
|
||||
b[5] = 0 // st_other
|
||||
le.PutUint16(b[6:], s.shndx)
|
||||
le.PutUint64(b[8:], s.value)
|
||||
le.PutUint64(b[16:], s.size)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
for _, r := range relas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|rX8664PC32)
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
|
||||
// DWARF debug sections (no relocations — the linker resolves DWARF fixups).
|
||||
dwAlign := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
|
||||
if dw != nil {
|
||||
nSections += 4 // .debug_abbrev, .debug_info, .debug_line, .debug_line_str
|
||||
}
|
||||
|
||||
align(8)
|
||||
shoff := len(out)
|
||||
|
||||
// Section headers.
|
||||
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
|
||||
var b [64]byte
|
||||
le.PutUint32(b[0:], uint32(stSections.at(name)))
|
||||
le.PutUint32(b[4:], uint32(typ))
|
||||
le.PutUint64(b[8:], flags)
|
||||
le.PutUint64(b[16:], 0) // sh_addr
|
||||
le.PutUint64(b[24:], uint64(off))
|
||||
le.PutUint64(b[32:], uint64(size))
|
||||
le.PutUint32(b[40:], uint32(link))
|
||||
le.PutUint32(b[44:], uint32(info))
|
||||
le.PutUint64(b[48:], alignV)
|
||||
le.PutUint64(b[56:], entsize)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
|
||||
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
|
||||
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
|
||||
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
|
||||
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
|
||||
// DWARF section headers.
|
||||
if dw != nil {
|
||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||
if dw.frameSize > 0 {
|
||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||
}
|
||||
}
|
||||
|
||||
// The ELF header.
|
||||
hdr := out[:64]
|
||||
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
|
||||
le.PutUint16(hdr[16:], etREL)
|
||||
le.PutUint16(hdr[18:], emX8664)
|
||||
le.PutUint32(hdr[20:], elfVersion)
|
||||
le.PutUint64(hdr[24:], 0) // e_entry
|
||||
le.PutUint64(hdr[32:], 0) // e_phoff
|
||||
le.PutUint64(hdr[40:], uint64(shoff)) // e_shoff
|
||||
le.PutUint32(hdr[48:], 0) // e_flags
|
||||
le.PutUint16(hdr[52:], 64) // e_ehsize
|
||||
le.PutUint16(hdr[54:], 0) // e_phentsize
|
||||
le.PutUint16(hdr[56:], 0) // e_phnum
|
||||
le.PutUint16(hdr[58:], 64) // e_shentsize
|
||||
le.PutUint16(hdr[60:], uint16(nSections))
|
||||
le.PutUint16(hdr[62:], uint16(secShstr))
|
||||
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// objectName renders a symbol's object-file name: the identifier as written,
|
||||
// with an explicit package prefix joined by a dot.
|
||||
func objectName(pkg, name string) string {
|
||||
if pkg == "" {
|
||||
return name
|
||||
}
|
||||
return pkg + "." + name
|
||||
}
|
||||
|
||||
// elfStrtab is an ELF string table under construction.
|
||||
type elfStrtab struct {
|
||||
buf []byte
|
||||
off map[string]int
|
||||
}
|
||||
|
||||
func newElfStrtab() *elfStrtab {
|
||||
return &elfStrtab{buf: []byte{0}, off: map[string]int{"": 0}}
|
||||
}
|
||||
|
||||
func (s *elfStrtab) add(name string) {
|
||||
if _, ok := s.off[name]; ok {
|
||||
return
|
||||
}
|
||||
s.off[name] = len(s.buf)
|
||||
s.buf = append(s.buf, name...)
|
||||
s.buf = append(s.buf, 0)
|
||||
}
|
||||
|
||||
func (s *elfStrtab) at(name string) int { return s.off[name] }
|
||||
|
||||
func (s *elfStrtab) bytes() []byte { return s.buf }
|
||||
@@ -0,0 +1,319 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
)
|
||||
|
||||
// DWARF5 section generation for ELF output. Unlike the GOOBJ path (where
|
||||
// the linker assembles the final DWARF), the ELF path must emit complete,
|
||||
// self-contained sections because the system linker only performs fixup
|
||||
// relocations, not assembly.
|
||||
|
||||
// dwarfAbbrevTable returns the .debug_abbrev content: a single compilation
|
||||
// unit with DW_TAG_compile_unit and DW_TAG_subprogram entries.
|
||||
func dwarfAbbrevTable() []byte {
|
||||
var b []byte
|
||||
// Abbrev 1: DW_TAG_compile_unit
|
||||
b = append(b, 1) // abbreviation code
|
||||
b = append(b, 0x11) // DW_TAG_compile_unit
|
||||
b = append(b, 1) // DW_CHILDREN_yes
|
||||
b = appendUleb(b, 0x1b) // DW_AT_low_pc
|
||||
b = appendUleb(b, 0x01) // DW_FORM_addr
|
||||
b = appendUleb(b, 0x29) // DW_AT_high_pc
|
||||
b = appendUleb(b, 0x07) // DW_FORM_data8
|
||||
b = appendUleb(b, 0x10) // DW_AT_stmt_list
|
||||
b = appendUleb(b, 0x25) // DW_FORM_sec_offset
|
||||
b = appendUleb(b, 0x01) // DW_AT_name
|
||||
b = appendUleb(b, 0x08) // DW_FORM_string
|
||||
b = appendUleb(b, 0) // end of attributes
|
||||
|
||||
// Abbrev 2: DW_TAG_subprogram
|
||||
b = append(b, 2) // abbreviation code
|
||||
b = append(b, 0x2e) // DW_TAG_subprogram
|
||||
b = append(b, 0) // DW_CHILDREN_no
|
||||
b = appendUleb(b, 0x03) // DW_AT_name
|
||||
b = appendUleb(b, 0x08) // DW_FORM_string
|
||||
b = appendUleb(b, 0x11) // DW_AT_low_pc
|
||||
b = appendUleb(b, 0x01) // DW_FORM_addr
|
||||
b = appendUleb(b, 0x29) // DW_AT_high_pc
|
||||
b = appendUleb(b, 0x07) // DW_FORM_data8
|
||||
b = appendUleb(b, 0x3f) // DW_AT_frame_base
|
||||
b = appendUleb(b, 0x18) // DW_FORM_exprloc
|
||||
b = appendUleb(b, 0x3b) // DW_AT_decl_file
|
||||
b = appendUleb(b, 0x0b) // DW_FORM_data1
|
||||
b = appendUleb(b, 0x37) // DW_AT_decl_line
|
||||
b = appendUleb(b, 0x0b) // DW_FORM_data1
|
||||
b = appendUleb(b, 0x63) // DW_AT_external
|
||||
b = appendUleb(b, 0x0b) // DW_FORM_flag
|
||||
b = appendUleb(b, 0) // end of attributes
|
||||
|
||||
// End of table.
|
||||
b = append(b, 0)
|
||||
return b
|
||||
}
|
||||
|
||||
// dwarfSections holds the generated DWARF section payloads and their
|
||||
// relocations (byte offsets within .debug_info and .debug_line that need
|
||||
// fixup against .text symbols).
|
||||
type dwarfSections struct {
|
||||
debugAbbrev []byte
|
||||
debugInfo []byte
|
||||
debugLine []byte
|
||||
debugLineStr []byte
|
||||
debugFrame []byte
|
||||
// Relocations for .debug_info: (offset, symbol name, addend).
|
||||
infoRelocs []dwarfReloc
|
||||
// Relocations for .debug_line: (offset, symbol name, addend).
|
||||
lineRelocs []dwarfReloc
|
||||
}
|
||||
|
||||
type dwarfReloc struct {
|
||||
off uint64
|
||||
name string
|
||||
addend int64
|
||||
}
|
||||
|
||||
// emitDWARF generates complete DWARF5 sections for the image.
|
||||
func emitDWARF(img *Image, srcFile string) *dwarfSections {
|
||||
ds := &dwarfSections{}
|
||||
ds.debugAbbrev = dwarfAbbrevTable()
|
||||
|
||||
// Build the string table for .debug_line_str.
|
||||
lineStr := newElfStrtab()
|
||||
lineStr.add(srcFile)
|
||||
ds.debugLineStr = lineStr.bytes()
|
||||
|
||||
// Build .debug_line.
|
||||
ds.debugLine = dwarfBuildLineSection(img, ds)
|
||||
|
||||
// Build .debug_info.
|
||||
ds.debugInfo = dwarfBuildInfoSection(img, srcFile, ds)
|
||||
|
||||
// Build .debug_frame.
|
||||
ds.debugFrame = dwarfBuildFrameSection(img)
|
||||
return ds
|
||||
}
|
||||
|
||||
// dwarfBuildLineSection builds a complete .debug_line section.
|
||||
func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
|
||||
var b []byte
|
||||
le := binary.LittleEndian
|
||||
|
||||
// We'll build the header first, then the programs, then patch the length.
|
||||
headerStart := len(b)
|
||||
b = append(b, 0, 0, 0, 0) // unit_length (placeholder)
|
||||
b = le.AppendUint16(b, 5) // version (DWARF5)
|
||||
b = append(b, 8) // address_size
|
||||
b = append(b, 0) // segment_selector_size
|
||||
b = append(b, 0, 0, 0, 0) // header_length (placeholder)
|
||||
|
||||
// Line program parameters.
|
||||
b = append(b, 1) // minimum_instruction_length
|
||||
b = append(b, 1) // maximum_ops_per_instruction
|
||||
b = append(b, 1) // default_is_stmt
|
||||
b = append(b, byte(dwLineBase&0xFF)) // line_base (-4 as unsigned)
|
||||
b = append(b, uint8(dwLineRange)) // line_range
|
||||
b = append(b, uint8(dwOpcodeBase)) // opcode_base
|
||||
// Standard opcode lengths (opcode 1..opcode_base-1).
|
||||
b = append(b, 0, 1, 1, 1, 1, 0, 0, 0, 1, 0)
|
||||
|
||||
// Directory table (DWARF5 format).
|
||||
b = append(b, 0) // one directory entry (index 0 = empty)
|
||||
// File table.
|
||||
b = appendUleb(b, 1) // file count
|
||||
// File 1: name index into .debug_line_str, dir index, time, size.
|
||||
b = appendUleb(b, 0) // name (index 0 in line_str)
|
||||
b = appendUleb(b, 0) // directory index
|
||||
b = appendUleb(b, 0) // last modification time
|
||||
b = appendUleb(b, 0) // file size
|
||||
|
||||
headerEnd := len(b)
|
||||
|
||||
// Per-function line programs.
|
||||
for _, fn := range img.Funcs {
|
||||
// LNE_set_address with the function's offset in .text.
|
||||
b = append(b, 0, 9, 2) // extended opcode, length 9, DW_LNE_set_address
|
||||
addrOff := len(b)
|
||||
b = le.AppendUint64(b, 0) // placeholder for address
|
||||
ds.lineRelocs = append(ds.lineRelocs, dwarfReloc{
|
||||
off: uint64(addrOff),
|
||||
name: fn.Name,
|
||||
addend: 0,
|
||||
})
|
||||
|
||||
// Build the line entries.
|
||||
pts := make([]LineEntry, 0, len(fn.Lines)+1)
|
||||
if len(fn.Lines) == 0 || fn.Lines[0].Offset > 0 {
|
||||
pts = append(pts, LineEntry{Offset: 0, Line: fn.Line})
|
||||
}
|
||||
pts = append(pts, fn.Lines...)
|
||||
|
||||
line := int64(1)
|
||||
pc := uint64(0)
|
||||
for _, p := range pts {
|
||||
if p.Line == 0 || uint64(p.Offset) < pc {
|
||||
continue
|
||||
}
|
||||
if int64(p.Line) == line {
|
||||
continue
|
||||
}
|
||||
deltaPC := uint64(p.Offset) - pc
|
||||
deltaLC := int64(p.Line) - line
|
||||
b = dwPutPCLCDelta(b, deltaPC, deltaLC)
|
||||
line, pc = int64(p.Line), uint64(p.Offset)
|
||||
}
|
||||
|
||||
// Advance to end of function.
|
||||
if end := uint64(fn.Size) - pc; end > 0 {
|
||||
b = append(b, 2) // DW_LNS_advance_pc
|
||||
b = appendUleb(b, end)
|
||||
}
|
||||
b = append(b, 0, 1, 1) // LNE_end_sequence
|
||||
}
|
||||
|
||||
// Patch unit_length.
|
||||
le.PutUint32(b[headerStart:], uint32(len(b)-headerStart-4))
|
||||
// Patch header_length.
|
||||
le.PutUint32(b[headerStart+6:], uint32(headerEnd-headerStart-10))
|
||||
return b
|
||||
}
|
||||
|
||||
// dwarfBuildInfoSection builds a complete .debug_info section.
|
||||
func dwarfBuildInfoSection(img *Image, srcFile string, ds *dwarfSections) []byte {
|
||||
var b []byte
|
||||
le := binary.LittleEndian
|
||||
|
||||
cuStart := len(b)
|
||||
b = append(b, 0, 0, 0, 0) // unit_length (placeholder)
|
||||
b = le.AppendUint16(b, 5) // version (DWARF5)
|
||||
b = append(b, 0x01) // unit_type (DW_UT_compile)
|
||||
b = append(b, 8) // address_size
|
||||
b = le.AppendUint32(b, 0) // debug_abbrev_offset (0 since single CU)
|
||||
|
||||
// DW_TAG_compile_unit (abbrev 1).
|
||||
b = append(b, 1) // abbreviation code
|
||||
// DW_AT_low_pc: address of .text start.
|
||||
infoRelocBase := len(b)
|
||||
b = le.AppendUint64(b, 0) // placeholder
|
||||
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
|
||||
off: uint64(infoRelocBase),
|
||||
name: img.Funcs[0].Name,
|
||||
addend: 0,
|
||||
})
|
||||
// DW_AT_high_pc: size of .text.
|
||||
b = le.AppendUint64(b, uint64(len(img.Code)))
|
||||
// DW_AT_stmt_list: offset into .debug_line (0).
|
||||
b = le.AppendUint32(b, 0)
|
||||
// DW_AT_name: source file name.
|
||||
b = append(b, srcFile...)
|
||||
b = append(b, 0)
|
||||
|
||||
// DW_TAG_subprogram entries (abbrev 2).
|
||||
for _, fn := range img.Funcs {
|
||||
b = append(b, 2) // abbreviation code
|
||||
// DW_AT_name.
|
||||
b = append(b, fn.Name...)
|
||||
b = append(b, 0)
|
||||
// DW_AT_low_pc.
|
||||
addrOff := len(b)
|
||||
b = le.AppendUint64(b, 0) // placeholder
|
||||
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
|
||||
off: uint64(addrOff),
|
||||
name: fn.Name,
|
||||
addend: 0,
|
||||
})
|
||||
// DW_AT_high_pc: function size.
|
||||
b = le.AppendUint64(b, uint64(fn.Size))
|
||||
// DW_AT_frame_base: DW_OP_call_frame_cfa.
|
||||
b = append(b, 1, 0x9c)
|
||||
// DW_AT_decl_file: file index 1.
|
||||
b = append(b, 1)
|
||||
// DW_AT_decl_line.
|
||||
b = append(b, uint8(fn.Line))
|
||||
// DW_AT_external.
|
||||
if fn.Static {
|
||||
b = append(b, 0)
|
||||
} else {
|
||||
b = append(b, 1)
|
||||
}
|
||||
}
|
||||
|
||||
// End of compile unit children.
|
||||
b = append(b, 0)
|
||||
|
||||
// Patch unit_length.
|
||||
le.PutUint32(b[cuStart:], uint32(len(b)-cuStart-4))
|
||||
return b
|
||||
}
|
||||
|
||||
func appendUleb(b []byte, v uint64) []byte {
|
||||
return binary.AppendUvarint(b, v)
|
||||
}
|
||||
|
||||
func appendSleb(b []byte, v int64) []byte {
|
||||
return binary.AppendVarint(b, v)
|
||||
}
|
||||
|
||||
// dwarfBuildFrameSection builds a .debug_frame section with CFI for stack
|
||||
// unwinding. It emits one CIE and one FDE per function, encoding the
|
||||
// CFA (Canonical Frame Address) rule changes at each stack-adjustment
|
||||
// boundary recorded in FuncLayout.Spadj.
|
||||
func dwarfBuildFrameSection(img *Image) []byte {
|
||||
var b []byte
|
||||
le := binary.LittleEndian
|
||||
|
||||
// CIE (Common Information Entry).
|
||||
cieStart := len(b)
|
||||
b = append(b, 0, 0, 0, 0) // length (placeholder)
|
||||
b = le.AppendUint32(b, 0xFFFFFFFF) // CIE marker
|
||||
b = append(b, 3) // version (DWARF3, widely supported)
|
||||
b = append(b, 0) // augmentation (empty)
|
||||
b = appendUleb(b, 1) // code alignment
|
||||
b = appendSleb(b, -8) // data alignment (-8 for 64-bit)
|
||||
b = appendUleb(b, 16) // return address register (LR on arm64, RIP on amd64)
|
||||
// Initial CFA rule: DW_CFA_def_cfa (SP, 0)
|
||||
b = append(b, 0x0c) // DW_CFA_def_cfa
|
||||
b = appendUleb(b, 31) // register: SP (RSP=7 on amd64, SP=31 on arm64)
|
||||
b = appendUleb(b, 0) // offset: 0
|
||||
b = append(b, 0) // DW_CFA_nop (padding)
|
||||
// Patch CIE length.
|
||||
le.PutUint32(b[cieStart:], uint32(len(b)-cieStart-4))
|
||||
|
||||
// FDEs (Frame Description Entries) — one per function.
|
||||
for _, fn := range img.Funcs {
|
||||
fdeStart := len(b)
|
||||
b = append(b, 0, 0, 0, 0) // length (placeholder)
|
||||
b = le.AppendUint32(b, uint32(cieStart)) // CIE pointer (offset from start)
|
||||
// Initial location: function offset in .text (relocated by linker).
|
||||
b = le.AppendUint64(b, uint64(fn.Offset))
|
||||
// Address range: function size.
|
||||
b = le.AppendUint64(b, uint64(fn.Size))
|
||||
|
||||
// Emit CFA rule changes at each Spadj boundary.
|
||||
for _, step := range fn.Spadj {
|
||||
if step.Value == 0 {
|
||||
continue
|
||||
}
|
||||
// DW_CFA_def_cfa_offset: set CFA = SP + |delta|.
|
||||
// The delta is negative (stack grows down), so CFA offset = -delta.
|
||||
offset := -step.Value
|
||||
if offset > 0 {
|
||||
b = append(b, 0x0e) // DW_CFA_def_cfa_offset
|
||||
b = appendUleb(b, uint64(offset))
|
||||
}
|
||||
}
|
||||
|
||||
// Pad to alignment.
|
||||
for len(b)%4 != 0 {
|
||||
b = append(b, 0) // DW_CFA_nop
|
||||
}
|
||||
|
||||
// Patch FDE length.
|
||||
le.PutUint32(b[fdeStart:], uint32(len(b)-fdeStart-4))
|
||||
}
|
||||
|
||||
return b
|
||||
}
|
||||
@@ -0,0 +1,119 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
// dwarfELFSections holds the laid-out DWARF sections ready for inclusion
|
||||
// in an ELF file.
|
||||
type dwarfELFSections struct {
|
||||
abbrevOff, abbrevSize int
|
||||
infoOff, infoSize int
|
||||
lineOff, lineSize int
|
||||
lineStrOff, lineStrSize int
|
||||
frameOff, frameSize int
|
||||
// Relocations for .debug_info address references.
|
||||
infoRelocs []elfDwarfReloc
|
||||
// Relocations for .debug_line address references.
|
||||
lineRelocs []elfDwarfReloc
|
||||
}
|
||||
|
||||
type elfDwarfReloc struct {
|
||||
off uint64
|
||||
sym int // symbol index in .symtab
|
||||
addend int64
|
||||
}
|
||||
|
||||
// appendDWARFSections generates and appends DWARF5 debug sections to the ELF
|
||||
// output. It returns the section offsets/sizes and relocations for the caller
|
||||
// to emit section headers and relocation records.
|
||||
//
|
||||
// symIdx maps function names to their .symtab indices (needed for relocations
|
||||
// against .text symbols). The map uses objectName format (pkg.name); the
|
||||
// DWARF code uses bare function names, so we build a reverse lookup.
|
||||
func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[string]int, align func(int)) *dwarfELFSections {
|
||||
// Build a lookup from bare function name to symbol index.
|
||||
nameToIdx := make(map[string]int, len(symIdx))
|
||||
for name, idx := range symIdx {
|
||||
// Strip package prefix: "pkg.name" → "name".
|
||||
if i := len(name) - 1; i >= 0 {
|
||||
for j := len(name) - 1; j >= 0; j-- {
|
||||
if name[j] == '.' {
|
||||
nameToIdx[name[j+1:]] = idx
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
nameToIdx[name] = idx
|
||||
}
|
||||
ds := emitDWARF(img, srcFile)
|
||||
if ds == nil || len(ds.debugAbbrev) == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
result := &dwarfELFSections{}
|
||||
|
||||
// .debug_abbrev
|
||||
align(1)
|
||||
result.abbrevOff = len(*out)
|
||||
result.abbrevSize = len(ds.debugAbbrev)
|
||||
*out = append(*out, ds.debugAbbrev...)
|
||||
|
||||
// .debug_line_str
|
||||
align(1)
|
||||
result.lineStrOff = len(*out)
|
||||
result.lineStrSize = len(ds.debugLineStr)
|
||||
*out = append(*out, ds.debugLineStr...)
|
||||
|
||||
// .debug_line
|
||||
align(1)
|
||||
result.lineOff = len(*out)
|
||||
result.lineSize = len(ds.debugLine)
|
||||
lineBase := len(*out)
|
||||
*out = append(*out, ds.debugLine...)
|
||||
|
||||
// Patch .debug_line relocations: replace placeholder addresses with
|
||||
// actual .text offsets via symbol lookup.
|
||||
for _, dr := range ds.lineRelocs {
|
||||
if idx, ok := nameToIdx[dr.name]; ok {
|
||||
result.lineRelocs = append(result.lineRelocs, elfDwarfReloc{
|
||||
off: uint64(lineBase) + dr.off,
|
||||
sym: idx,
|
||||
addend: dr.addend,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// .debug_info
|
||||
align(1)
|
||||
result.infoOff = len(*out)
|
||||
result.infoSize = len(ds.debugInfo)
|
||||
infoBase := len(*out)
|
||||
*out = append(*out, ds.debugInfo...)
|
||||
|
||||
// .debug_frame
|
||||
if len(ds.debugFrame) > 0 {
|
||||
align(1)
|
||||
result.frameOff = len(*out)
|
||||
result.frameSize = len(ds.debugFrame)
|
||||
*out = append(*out, ds.debugFrame...)
|
||||
}
|
||||
|
||||
// Patch .debug_info relocations.
|
||||
for _, dr := range ds.infoRelocs {
|
||||
if idx, ok := nameToIdx[dr.name]; ok {
|
||||
result.infoRelocs = append(result.infoRelocs, elfDwarfReloc{
|
||||
off: uint64(infoBase) + dr.off,
|
||||
sym: idx,
|
||||
addend: dr.addend,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
return result
|
||||
}
|
||||
|
||||
// dwarfSectionNames returns the DWARF section names for the string table.
|
||||
var dwarfSectionNames = []string{
|
||||
".debug_abbrev", ".debug_info", ".debug_line", ".debug_line_str",
|
||||
".debug_frame", ".rela.debug_info", ".rela.debug_line",
|
||||
}
|
||||
@@ -0,0 +1,81 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
func TestEmitDWARF(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVQ a+0(FP), AX
|
||||
MOVQ b+8(FP), BX
|
||||
ADDQ BX, AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_amd64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
|
||||
ds := emitDWARF(img, "test_amd64.s")
|
||||
|
||||
// .debug_abbrev must not be empty and must start with abbrev code 1.
|
||||
if len(ds.debugAbbrev) == 0 {
|
||||
t.Fatal("empty .debug_abbrev")
|
||||
}
|
||||
if ds.debugAbbrev[0] != 1 {
|
||||
t.Fatalf(".debug_abbrev first byte = %d, want 1", ds.debugAbbrev[0])
|
||||
}
|
||||
|
||||
// .debug_info must have a compile unit header (DWARF5 version 5).
|
||||
if len(ds.debugInfo) < 12 {
|
||||
t.Fatalf(".debug_info too short: %d bytes", len(ds.debugInfo))
|
||||
}
|
||||
// Version field at offset 4 (after unit_length).
|
||||
if ds.debugInfo[4] != 5 || ds.debugInfo[5] != 0 {
|
||||
t.Fatalf(".debug_info version = %d, want 5", uint16(ds.debugInfo[4])|uint16(ds.debugInfo[5])<<8)
|
||||
}
|
||||
|
||||
// .debug_line must have a header.
|
||||
if len(ds.debugLine) < 20 {
|
||||
t.Fatalf(".debug_line too short: %d bytes", len(ds.debugLine))
|
||||
}
|
||||
// Version at offset 4.
|
||||
if ds.debugLine[4] != 5 || ds.debugLine[5] != 0 {
|
||||
t.Fatalf(".debug_line version = %d, want 5", uint16(ds.debugLine[4])|uint16(ds.debugLine[5])<<8)
|
||||
}
|
||||
|
||||
// .debug_line_str must contain the source file name.
|
||||
if len(ds.debugLineStr) == 0 {
|
||||
t.Fatal("empty .debug_line_str")
|
||||
}
|
||||
|
||||
// Relocations must reference the function.
|
||||
if len(ds.lineRelocs) == 0 {
|
||||
t.Fatal("no .debug_line relocations")
|
||||
}
|
||||
if len(ds.infoRelocs) == 0 {
|
||||
t.Fatal("no .debug_info relocations")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDwarfAbbrevTable(t *testing.T) {
|
||||
abbrev := dwarfAbbrevTable()
|
||||
if len(abbrev) == 0 {
|
||||
t.Fatal("empty abbrev table")
|
||||
}
|
||||
// Must end with a zero byte (end of table).
|
||||
if abbrev[len(abbrev)-1] != 0 {
|
||||
t.Fatalf("abbrev table last byte = %d, want 0", abbrev[len(abbrev)-1])
|
||||
}
|
||||
}
|
||||
+310
@@ -0,0 +1,310 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"debug/elf"
|
||||
"encoding/binary"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// The object-file tests share one source: two exported functions, one
|
||||
// file-local constant reached through a relocation, and one external symbol
|
||||
// the linker must resolve. The functions take their arguments in the System
|
||||
// V registers (not the Go stack ABI) so a C driver can call them directly.
|
||||
const elfTestSrc = `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·addq(SB), NOSPLIT, $0
|
||||
LEAQ (DI)(SI*1), AX
|
||||
RET
|
||||
|
||||
TEXT ·getanswer(SB), NOSPLIT, $0
|
||||
MOVQ answer<>(SB), AX
|
||||
RET
|
||||
|
||||
TEXT ·useextern(SB), NOSPLIT, $0
|
||||
MOVQ extvar(SB), AX
|
||||
RET
|
||||
|
||||
GLOBL answer<>(SB), RODATA, $8
|
||||
DATA answer<>+0(SB)/8, $42
|
||||
`
|
||||
|
||||
func elfTestImage(t *testing.T) *Image {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse("t_amd64.s", elfTestSrc)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
return img
|
||||
}
|
||||
|
||||
// TestAssembleFileExternals checks that a reference to a symbol no GLOBL
|
||||
// defines is recorded as an external relocation instead of failing — the
|
||||
// raw image leaves the displacement zero, the object emitters carry it.
|
||||
func TestAssembleFileExternals(t *testing.T) {
|
||||
img := elfTestImage(t)
|
||||
if len(img.Externals) != 1 || img.Externals[0] != "extvar" {
|
||||
t.Fatalf("Externals = %v, want [extvar]", img.Externals)
|
||||
}
|
||||
var ext, local int
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
if r.External {
|
||||
ext++
|
||||
if r.Name != "extvar" {
|
||||
t.Errorf("external reloc names %q, want extvar", r.Name)
|
||||
}
|
||||
} else {
|
||||
local++
|
||||
if r.Name != "answer" {
|
||||
t.Errorf("local reloc names %q, want answer", r.Name)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if ext != 1 || local != 1 {
|
||||
t.Errorf("relocs = %d external, %d local; want 1 and 1", ext, local)
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFObject checks the structure of the emitted ELF64 relocatable
|
||||
// object: sections, the symbol table (bindings, types, values, sizes) and
|
||||
// the .rela.text relocations, parsed back with debug/elf.
|
||||
func TestELFObject(t *testing.T) {
|
||||
img := elfTestImage(t)
|
||||
obj, err := img.ELFObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFObject: %v", err)
|
||||
}
|
||||
f, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
if f.Type != elf.ET_REL || f.Machine != elf.EM_X86_64 {
|
||||
t.Errorf("type/machine = %v/%v, want ET_REL/EM_X86_64", f.Type, f.Machine)
|
||||
}
|
||||
|
||||
text := f.Section(".text")
|
||||
data := f.Section(".data")
|
||||
if text == nil || data == nil {
|
||||
t.Fatal("missing .text or .data section")
|
||||
}
|
||||
if text.Flags&elf.SHF_EXECINSTR == 0 || text.Flags&elf.SHF_ALLOC == 0 {
|
||||
t.Errorf(".text flags = %v", text.Flags)
|
||||
}
|
||||
if data.Flags&elf.SHF_WRITE == 0 {
|
||||
t.Errorf(".data flags = %v", data.Flags)
|
||||
}
|
||||
textData, err := text.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !bytes.Equal(textData, img.Code) {
|
||||
t.Errorf(".text contents differ from the image code")
|
||||
}
|
||||
|
||||
syms, err := f.Symbols()
|
||||
if err != nil {
|
||||
t.Fatalf("symbols: %v", err)
|
||||
}
|
||||
byName := map[string]elf.Symbol{}
|
||||
for _, s := range syms {
|
||||
byName[s.Name] = s
|
||||
}
|
||||
wantSym := func(name string, bind elf.SymBind, typ elf.SymType, section elf.SectionIndex, size uint64) {
|
||||
t.Helper()
|
||||
s, ok := byName[name]
|
||||
if !ok {
|
||||
t.Errorf("symbol %q not found", name)
|
||||
return
|
||||
}
|
||||
if elf.ST_BIND(s.Info) != bind || elf.ST_TYPE(s.Info) != typ {
|
||||
t.Errorf("%s: bind/type = %v/%v, want %v/%v", name, elf.ST_BIND(s.Info), elf.ST_TYPE(s.Info), bind, typ)
|
||||
}
|
||||
if s.Section != section {
|
||||
t.Errorf("%s: section = %v, want %v", name, s.Section, section)
|
||||
}
|
||||
if s.Size != size {
|
||||
t.Errorf("%s: size = %d, want %d", name, s.Size, size)
|
||||
}
|
||||
}
|
||||
// The emitted layout is fixed: 0 NULL, 1 .text, 2 .data.
|
||||
if f.Sections[1].Name != ".text" || f.Sections[2].Name != ".data" {
|
||||
t.Fatalf("section layout = %s, %s; want .text, .data", f.Sections[1].Name, f.Sections[2].Name)
|
||||
}
|
||||
textIdx := elf.SectionIndex(1)
|
||||
dataIdx := elf.SectionIndex(2)
|
||||
wantSym("addq", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 5)
|
||||
wantSym("getanswer", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 8)
|
||||
wantSym("useextern", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 8)
|
||||
wantSym("answer", elf.STB_LOCAL, elf.STT_OBJECT, dataIdx, 8)
|
||||
wantSym("extvar", elf.STB_GLOBAL, elf.STT_NOTYPE, elf.SHN_UNDEF, 0)
|
||||
|
||||
// Relocations: one for the file-local constant (resolving against the
|
||||
// local data symbol) and one for the external (against the undefined
|
||||
// global), both R_X86_64_PC32 with the −4 addend the PC-relative form
|
||||
// needs. debug/elf does not surface rela entries, so read the section
|
||||
// directly.
|
||||
relaSec := f.Section(".rela.text")
|
||||
if relaSec == nil {
|
||||
t.Fatal("missing .rela.text")
|
||||
}
|
||||
raw, err := relaSec.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(raw)%24 != 0 || len(raw)/24 != 2 {
|
||||
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
|
||||
}
|
||||
// Symbol names straight from the raw tables: r_info carries an index
|
||||
// into .symtab including the null entry, which debug/elf's Symbols()
|
||||
// slice may not mirror.
|
||||
symtabRaw, err := f.Section(".symtab").Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
strtabRaw, err := f.Section(".strtab").Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
symName := func(idx int) string {
|
||||
stName := binary.LittleEndian.Uint32(symtabRaw[idx*24:])
|
||||
end := bytes.IndexByte(strtabRaw[stName:], 0)
|
||||
return string(strtabRaw[stName : int(stName)+end])
|
||||
}
|
||||
for i := range 2 {
|
||||
e := raw[i*24 : (i+1)*24]
|
||||
off := binary.LittleEndian.Uint64(e[0:])
|
||||
info := binary.LittleEndian.Uint64(e[8:])
|
||||
addend := int64(binary.LittleEndian.Uint64(e[16:]))
|
||||
typ := info & 0xffffffff
|
||||
sym := int(info >> 32)
|
||||
if typ != uint64(elf.R_X86_64_PC32) {
|
||||
t.Errorf("reloc %d: type %d, want R_X86_64_PC32", i, typ)
|
||||
}
|
||||
if addend != -4 {
|
||||
t.Errorf("reloc %d: addend %d, want -4", i, addend)
|
||||
}
|
||||
if name := symName(sym); name != "answer" && name != "extvar" {
|
||||
t.Errorf("reloc %d: symbol %q, want answer or extvar", i, name)
|
||||
}
|
||||
// The relocation offset lands on the disp32 field: the four bytes
|
||||
// before a RET-terminated eight-byte MOVQ.
|
||||
if off+4 > uint64(len(textData)) {
|
||||
t.Errorf("reloc %d: offset %d outside .text", i, off)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFObjectNoRelocations checks a file with no static-symbol references
|
||||
// emits a valid object without a .rela.text section.
|
||||
func TestELFObjectNoRelocations(t *testing.T) {
|
||||
f, errs := parser.Parse("n_amd64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·nop(SB), NOSPLIT, $0
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
obj, err := img.ELFObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFObject: %v", err)
|
||||
}
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
if ef.Section(".rela.text") != nil {
|
||||
t.Error("unexpected .rela.text section")
|
||||
}
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
found := false
|
||||
for _, s := range syms {
|
||||
if s.Name == "nop" && elf.ST_TYPE(s.Info) == elf.STT_FUNC {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Error("function symbol nop not found")
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFLinkAndRun is the end-to-end check: assemble the test functions,
|
||||
// link the emitted object with a C driver that defines the external symbol,
|
||||
// and run the result. Skipped when no C compiler is available.
|
||||
func TestELFLinkAndRun(t *testing.T) {
|
||||
cc, err := exec.LookPath("cc")
|
||||
if err != nil {
|
||||
t.Skip("no C compiler available")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
|
||||
img := elfTestImage(t)
|
||||
obj, err := img.ELFObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFObject: %v", err)
|
||||
}
|
||||
objPath := filepath.Join(dir, "t.o")
|
||||
if err := os.WriteFile(objPath, obj, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
const driver = `
|
||||
#include <stdio.h>
|
||||
|
||||
long addq(long a, long b);
|
||||
long getanswer(void);
|
||||
long useextern(void);
|
||||
|
||||
long extvar = 7;
|
||||
|
||||
int main(void) {
|
||||
printf("%ld %ld %ld\n", addq(41, 1), getanswer(), useextern());
|
||||
return 0;
|
||||
}
|
||||
`
|
||||
driverPath := filepath.Join(dir, "driver.c")
|
||||
if err := os.WriteFile(driverPath, []byte(driver), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// -no-pie: the encoder emits R_X86_64_PC32 for external references,
|
||||
// which a position-independent executable would reject (it wants
|
||||
// PLT32/GOT relocations, a future increment).
|
||||
appPath := filepath.Join(dir, "app")
|
||||
out, err := exec.Command(cc, "-no-pie", "-o", appPath, driverPath, objPath).CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("link failed: %v\n%s", err, out)
|
||||
}
|
||||
run, err := exec.Command(appPath).CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("run failed: %v\n%s", err, run)
|
||||
}
|
||||
if got := string(run); got != "42 42 7\n" {
|
||||
t.Errorf("output %q, want \"42 42 7\\n\"", got)
|
||||
}
|
||||
}
|
||||
+252
@@ -0,0 +1,252 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
)
|
||||
|
||||
// AArch64 ELF64 relocatable object emission.
|
||||
|
||||
const (
|
||||
emAARCH64 = 183 // EM_AARCH64
|
||||
|
||||
// AArch64 relocation types (the ELF psABI).
|
||||
rArm64PrelPgHi21 = 275 // R_AARCH64_ADR_PREL_PG_HI21 (ADRP page)
|
||||
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD/STR/LDR page offset)
|
||||
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
|
||||
)
|
||||
|
||||
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
|
||||
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
|
||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
|
||||
// optional .rela.text.
|
||||
func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
const (
|
||||
secText = 1
|
||||
secData = 2
|
||||
)
|
||||
|
||||
// Build symbol table.
|
||||
var locals, globals []elfSym
|
||||
for _, fn := range img.Funcs {
|
||||
s := elfSym{
|
||||
name: objectName(fn.Pkg, fn.Name),
|
||||
info: sttFunc,
|
||||
shndx: secText,
|
||||
value: uint64(fn.Offset),
|
||||
size: uint64(fn.Size),
|
||||
}
|
||||
if fn.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, d := range img.DataSyms {
|
||||
s := elfSym{
|
||||
name: objectName(d.Pkg, d.Name),
|
||||
info: sttObject,
|
||||
shndx: secData,
|
||||
value: uint64(d.Offset),
|
||||
size: uint64(d.Size),
|
||||
}
|
||||
if d.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, name := range img.Externals {
|
||||
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
|
||||
}
|
||||
syms := []elfSym{
|
||||
{},
|
||||
{name: ".text", info: sttSection, shndx: secText},
|
||||
{name: ".data", info: sttSection, shndx: secData},
|
||||
}
|
||||
syms = append(syms, locals...)
|
||||
shInfo := len(syms)
|
||||
syms = append(syms, globals...)
|
||||
symIdx := map[string]int{}
|
||||
for i, s := range syms {
|
||||
symIdx[s.name] = i
|
||||
}
|
||||
|
||||
// Build relocations. Each SB reference is an ADRP pair:
|
||||
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21
|
||||
// ADD/LDR/STR → R_AARCH64_ADD_ABS_LO12_NC
|
||||
type elfRela struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
sym int
|
||||
addend int64
|
||||
}
|
||||
var relas []elfRela
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
var typ uint32
|
||||
switch {
|
||||
case r.Kind == RelArm64Branch:
|
||||
typ = rArm64Call26
|
||||
case r.Kind == RelArm64Addr && r.Off%4 == 4:
|
||||
typ = rArm64AddAbsLo12NC
|
||||
default:
|
||||
typ = rArm64PrelPgHi21
|
||||
}
|
||||
relas = append(relas, elfRela{
|
||||
off: uint64(fn.Offset + r.Off),
|
||||
typ: typ,
|
||||
sym: idx,
|
||||
addend: r.Addend - int64(r.After-r.Off),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// String tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
stNames.add(s.name)
|
||||
}
|
||||
stSections := newElfStrtab()
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
for _, n := range dwarfSectionNames {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Layout.
|
||||
var out []byte
|
||||
out = append(out, make([]byte, 64)...)
|
||||
|
||||
align := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
|
||||
align(16)
|
||||
textOff := len(out)
|
||||
out = append(out, img.Code...)
|
||||
|
||||
align(16)
|
||||
dataOff := len(out)
|
||||
out = append(out, img.Data...)
|
||||
|
||||
align(8)
|
||||
symtabOff := len(out)
|
||||
for _, s := range syms {
|
||||
var b [24]byte
|
||||
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
|
||||
b[4] = s.info
|
||||
b[5] = 0
|
||||
le.PutUint16(b[6:], s.shndx)
|
||||
le.PutUint64(b[8:], s.value)
|
||||
le.PutUint64(b[16:], s.size)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
for _, r := range relas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
|
||||
// DWARF debug sections.
|
||||
dwAlign := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
|
||||
if dw != nil {
|
||||
nSections += 4
|
||||
}
|
||||
|
||||
align(8)
|
||||
shoff := len(out)
|
||||
|
||||
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
|
||||
var b [64]byte
|
||||
le.PutUint32(b[0:], uint32(stSections.at(name)))
|
||||
le.PutUint32(b[4:], uint32(typ))
|
||||
le.PutUint64(b[8:], flags)
|
||||
le.PutUint64(b[16:], 0)
|
||||
le.PutUint64(b[24:], uint64(off))
|
||||
le.PutUint64(b[32:], uint64(size))
|
||||
le.PutUint32(b[40:], uint32(link))
|
||||
le.PutUint32(b[44:], uint32(info))
|
||||
le.PutUint64(b[48:], alignV)
|
||||
le.PutUint64(b[56:], entsize)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
|
||||
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
|
||||
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
|
||||
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
|
||||
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
if dw != nil {
|
||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||
if dw.frameSize > 0 {
|
||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||
}
|
||||
}
|
||||
|
||||
// ELF header.
|
||||
hdr := out[:64]
|
||||
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
|
||||
le.PutUint16(hdr[16:], etREL)
|
||||
le.PutUint16(hdr[18:], emAARCH64)
|
||||
le.PutUint32(hdr[20:], elfVersion)
|
||||
le.PutUint64(hdr[24:], 0)
|
||||
le.PutUint64(hdr[32:], 0)
|
||||
le.PutUint64(hdr[40:], uint64(shoff))
|
||||
le.PutUint32(hdr[48:], 0)
|
||||
le.PutUint16(hdr[52:], 64)
|
||||
le.PutUint16(hdr[54:], 0)
|
||||
le.PutUint16(hdr[56:], 0)
|
||||
le.PutUint16(hdr[58:], 64)
|
||||
le.PutUint16(hdr[60:], uint16(nSections))
|
||||
le.PutUint16(hdr[62:], uint16(secShstr))
|
||||
|
||||
return out, nil
|
||||
}
|
||||
@@ -0,0 +1,142 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"debug/elf"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
|
||||
// relocatable object: sections, the symbol table (bindings, types, values,
|
||||
// sizes) and the .rela.text relocation pair for the static-symbol load,
|
||||
// parsed back with debug/elf.
|
||||
func TestELFAARCH64Object(t *testing.T) {
|
||||
f, errs := parser.Parse("k_arm64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVD a+0(FP), R4
|
||||
MOVD b+8(FP), R5
|
||||
ADD R5, R4, R4
|
||||
MOVD R4, ret+16(FP)
|
||||
RET
|
||||
|
||||
TEXT ·getanswer(SB), NOSPLIT, $0-8
|
||||
MOVD answer<>(SB), R4
|
||||
MOVD R4, ret+0(FP)
|
||||
RET
|
||||
|
||||
GLOBL answer<>(SB), RODATA, $8
|
||||
DATA answer<>+0(SB)/8, $42
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
obj, err := img.ELFAARCH64Object()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFAARCH64Object: %v", err)
|
||||
}
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
|
||||
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_AARCH64 {
|
||||
t.Errorf("type/machine = %v/%v, want ET_REL/EM_AARCH64", ef.Type, ef.Machine)
|
||||
}
|
||||
|
||||
text := ef.Section(".text")
|
||||
data := ef.Section(".data")
|
||||
if text == nil || data == nil {
|
||||
t.Fatal("missing .text or .data section")
|
||||
}
|
||||
if text.Size == 0 {
|
||||
t.Error(".text section is empty")
|
||||
}
|
||||
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatalf("symbols: %v", err)
|
||||
}
|
||||
|
||||
foundAdd, foundGetanswer, foundAnswer := false, false, false
|
||||
for _, s := range syms {
|
||||
switch s.Name {
|
||||
case "add":
|
||||
foundAdd = true
|
||||
if elf.SymType(s.Info&0xf) != elf.STT_FUNC || elf.SymBind(s.Info>>4) != elf.STB_GLOBAL {
|
||||
t.Errorf("add: info=0x%02x, want STT_FUNC|STB_GLOBAL", s.Info)
|
||||
}
|
||||
case "getanswer":
|
||||
foundGetanswer = true
|
||||
if elf.SymType(s.Info&0xf) != elf.STT_FUNC || elf.SymBind(s.Info>>4) != elf.STB_GLOBAL {
|
||||
t.Errorf("getanswer: info=0x%02x, want STT_FUNC|STB_GLOBAL", s.Info)
|
||||
}
|
||||
case "answer":
|
||||
foundAnswer = true
|
||||
if elf.SymType(s.Info&0xf) != elf.STT_OBJECT || elf.SymBind(s.Info>>4) != elf.STB_LOCAL {
|
||||
t.Errorf("answer: info=0x%02x, want STT_OBJECT|STB_LOCAL", s.Info)
|
||||
}
|
||||
}
|
||||
}
|
||||
if !foundAdd {
|
||||
t.Error("symbol 'add' not found")
|
||||
}
|
||||
if !foundGetanswer {
|
||||
t.Error("symbol 'getanswer' not found")
|
||||
}
|
||||
if !foundAnswer {
|
||||
t.Error("symbol 'answer' not found")
|
||||
}
|
||||
|
||||
// Check that .rela.text exists (getanswer has SB reference).
|
||||
relaText := ef.Section(".rela.text")
|
||||
if relaText == nil {
|
||||
t.Error("missing .rela.text section")
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFAARCH64ObjectNoRelocations checks the ELF output when there are no
|
||||
// static-symbol references (no .rela.text section).
|
||||
func TestELFAARCH64ObjectNoRelocations(t *testing.T) {
|
||||
f, errs := parser.Parse("k_arm64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVD a+0(FP), R4
|
||||
MOVD b+8(FP), R5
|
||||
ADD R5, R4, R4
|
||||
MOVD R4, ret+16(FP)
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
obj, err := img.ELFAARCH64Object()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFAARCH64Object: %v", err)
|
||||
}
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
|
||||
if ef.Section(".rela.text") != nil {
|
||||
t.Error("unexpected .rela.text section when there are no relocations")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,245 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
)
|
||||
|
||||
// LoongArch ELF64 relocatable object emission.
|
||||
|
||||
const (
|
||||
emLOONGARCH = 258 // EM_LOONGARCH
|
||||
|
||||
// LoongArch relocation types (the ELF psABI).
|
||||
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
|
||||
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
|
||||
)
|
||||
|
||||
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
|
||||
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
|
||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
|
||||
// optional .rela.text.
|
||||
func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
const (
|
||||
secText = 1
|
||||
secData = 2
|
||||
)
|
||||
|
||||
// Build symbol table.
|
||||
var locals, globals []elfSym
|
||||
for _, fn := range img.Funcs {
|
||||
s := elfSym{
|
||||
name: objectName(fn.Pkg, fn.Name),
|
||||
info: sttFunc,
|
||||
shndx: secText,
|
||||
value: uint64(fn.Offset),
|
||||
size: uint64(fn.Size),
|
||||
}
|
||||
if fn.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, d := range img.DataSyms {
|
||||
s := elfSym{
|
||||
name: objectName(d.Pkg, d.Name),
|
||||
info: sttObject,
|
||||
shndx: secData,
|
||||
value: uint64(d.Offset),
|
||||
size: uint64(d.Size),
|
||||
}
|
||||
if d.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, name := range img.Externals {
|
||||
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
|
||||
}
|
||||
syms := []elfSym{
|
||||
{},
|
||||
{name: ".text", info: sttSection, shndx: secText},
|
||||
{name: ".data", info: sttSection, shndx: secData},
|
||||
}
|
||||
syms = append(syms, locals...)
|
||||
shInfo := len(syms)
|
||||
syms = append(syms, globals...)
|
||||
symIdx := map[string]int{}
|
||||
for i, s := range syms {
|
||||
symIdx[s.name] = i
|
||||
}
|
||||
|
||||
// Build relocations. Each SB reference is a pcalau12i pair:
|
||||
// pcalau12i rd, 0 → R_LARCH_PCALA_HI20
|
||||
// addi.d/ld/st → R_LARCH_PCALA_LO12
|
||||
type elfRela struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
sym int
|
||||
addend int64
|
||||
}
|
||||
var relas []elfRela
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
typ := uint32(rLarchPCALAHI20)
|
||||
if r.Kind == RelLoong64AddrLo {
|
||||
typ = rLarchPCALALO12
|
||||
}
|
||||
relas = append(relas, elfRela{
|
||||
off: uint64(fn.Offset + r.Off),
|
||||
typ: typ,
|
||||
sym: idx,
|
||||
addend: r.Addend - int64(r.After-r.Off),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// String tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
stNames.add(s.name)
|
||||
}
|
||||
stSections := newElfStrtab()
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
for _, n := range dwarfSectionNames {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Layout.
|
||||
var out []byte
|
||||
out = append(out, make([]byte, 64)...)
|
||||
|
||||
align := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
|
||||
align(16)
|
||||
textOff := len(out)
|
||||
out = append(out, img.Code...)
|
||||
|
||||
align(16)
|
||||
dataOff := len(out)
|
||||
out = append(out, img.Data...)
|
||||
|
||||
align(8)
|
||||
symtabOff := len(out)
|
||||
for _, s := range syms {
|
||||
var b [24]byte
|
||||
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
|
||||
b[4] = s.info
|
||||
b[5] = 0
|
||||
le.PutUint16(b[6:], s.shndx)
|
||||
le.PutUint64(b[8:], s.value)
|
||||
le.PutUint64(b[16:], s.size)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
for _, r := range relas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
|
||||
dwAlign := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
|
||||
if dw != nil {
|
||||
nSections += 4
|
||||
}
|
||||
|
||||
align(8)
|
||||
shoff := len(out)
|
||||
|
||||
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
|
||||
var b [64]byte
|
||||
le.PutUint32(b[0:], uint32(stSections.at(name)))
|
||||
le.PutUint32(b[4:], uint32(typ))
|
||||
le.PutUint64(b[8:], flags)
|
||||
le.PutUint64(b[16:], 0)
|
||||
le.PutUint64(b[24:], uint64(off))
|
||||
le.PutUint64(b[32:], uint64(size))
|
||||
le.PutUint32(b[40:], uint32(link))
|
||||
le.PutUint32(b[44:], uint32(info))
|
||||
le.PutUint64(b[48:], alignV)
|
||||
le.PutUint64(b[56:], entsize)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
|
||||
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
|
||||
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
|
||||
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
|
||||
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
if dw != nil {
|
||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||
if dw.frameSize > 0 {
|
||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||
}
|
||||
}
|
||||
|
||||
// ELF header.
|
||||
hdr := out[:64]
|
||||
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
|
||||
le.PutUint16(hdr[16:], etREL)
|
||||
le.PutUint16(hdr[18:], emLOONGARCH)
|
||||
le.PutUint32(hdr[20:], elfVersion)
|
||||
le.PutUint64(hdr[24:], 0)
|
||||
le.PutUint64(hdr[32:], 0)
|
||||
le.PutUint64(hdr[40:], uint64(shoff))
|
||||
le.PutUint32(hdr[48:], 0)
|
||||
le.PutUint16(hdr[52:], 64)
|
||||
le.PutUint16(hdr[54:], 0)
|
||||
le.PutUint16(hdr[56:], 0)
|
||||
le.PutUint16(hdr[58:], 64)
|
||||
le.PutUint16(hdr[60:], uint16(nSections))
|
||||
le.PutUint16(hdr[62:], uint16(secShstr))
|
||||
|
||||
return out, nil
|
||||
}
|
||||
@@ -0,0 +1,200 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"debug/elf"
|
||||
"encoding/binary"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
|
||||
// relocatable object: sections, the symbol table (bindings, types, values,
|
||||
// sizes) and the .rela.text relocation pair for the static-symbol load,
|
||||
// parsed back with debug/elf.
|
||||
func TestELFLOONG64Object(t *testing.T) {
|
||||
f, errs := parser.Parse("k_loong64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVV a+0(FP), R4
|
||||
MOVV b+8(FP), R5
|
||||
ADDV R5, R4, R4
|
||||
MOVV R4, ret+16(FP)
|
||||
RET
|
||||
|
||||
TEXT ·getanswer(SB), NOSPLIT, $0-8
|
||||
MOVV answer<>(SB), R4
|
||||
MOVV R4, ret+0(FP)
|
||||
RET
|
||||
|
||||
GLOBL answer<>(SB), RODATA, $8
|
||||
DATA answer<>+0(SB)/8, $42
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileLOONG64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||
}
|
||||
obj, err := img.ELFLOONG64Object()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFLOONG64Object: %v", err)
|
||||
}
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
|
||||
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_LOONGARCH {
|
||||
t.Errorf("type/machine = %v/%v, want ET_REL/EM_LOONGARCH", ef.Type, ef.Machine)
|
||||
}
|
||||
|
||||
text := ef.Section(".text")
|
||||
data := ef.Section(".data")
|
||||
if text == nil || data == nil {
|
||||
t.Fatal("missing .text or .data section")
|
||||
}
|
||||
if text.Flags&elf.SHF_EXECINSTR == 0 || text.Flags&elf.SHF_ALLOC == 0 {
|
||||
t.Errorf(".text flags = %v", text.Flags)
|
||||
}
|
||||
if data.Flags&elf.SHF_WRITE == 0 {
|
||||
t.Errorf(".data flags = %v", data.Flags)
|
||||
}
|
||||
textData, err := text.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !bytes.Equal(textData, img.Code) {
|
||||
t.Errorf(".text contents differ from the image code")
|
||||
}
|
||||
dataData, err := data.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatalf("symbols: %v", err)
|
||||
}
|
||||
byName := map[string]elf.Symbol{}
|
||||
for _, s := range syms {
|
||||
byName[s.Name] = s
|
||||
}
|
||||
wantSym := func(name string, bind elf.SymBind, typ elf.SymType, section elf.SectionIndex, size uint64) {
|
||||
t.Helper()
|
||||
s, ok := byName[name]
|
||||
if !ok {
|
||||
t.Errorf("symbol %q not found", name)
|
||||
return
|
||||
}
|
||||
if elf.ST_BIND(s.Info) != bind || elf.ST_TYPE(s.Info) != typ {
|
||||
t.Errorf("%s: bind/type = %v/%v, want %v/%v", name, elf.ST_BIND(s.Info), elf.ST_TYPE(s.Info), bind, typ)
|
||||
}
|
||||
if s.Section != section {
|
||||
t.Errorf("%s: section = %v, want %v", name, s.Section, section)
|
||||
}
|
||||
if s.Size != size {
|
||||
t.Errorf("%s: size = %d, want %d", name, s.Size, size)
|
||||
}
|
||||
}
|
||||
if ef.Sections[1].Name != ".text" || ef.Sections[2].Name != ".data" {
|
||||
t.Fatalf("section layout = %s, %s; want .text, .data", ef.Sections[1].Name, ef.Sections[2].Name)
|
||||
}
|
||||
textIdx := elf.SectionIndex(1)
|
||||
dataIdx := elf.SectionIndex(2)
|
||||
wantSym("add", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 20)
|
||||
wantSym("getanswer", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 16)
|
||||
wantSym("answer", elf.STB_LOCAL, elf.STT_OBJECT, dataIdx, 8)
|
||||
|
||||
// The data section carries 16-byte alignment padding; the answer
|
||||
// symbol sits at its padded offset.
|
||||
ans := byName["answer"]
|
||||
if ans.Value+8 > uint64(len(dataData)) {
|
||||
t.Fatalf("answer value %d outside .data (%d bytes)", ans.Value, len(dataData))
|
||||
}
|
||||
if got := dataData[ans.Value : ans.Value+8]; !bytes.Equal(got, []byte{42, 0, 0, 0, 0, 0, 0, 0}) {
|
||||
t.Errorf("answer data = % x, want $42", got)
|
||||
}
|
||||
|
||||
// Relocations: the static-symbol load is a pcalau12i+ld.d pair, so one
|
||||
// R_LARCH_PCALA_HI20 and one R_LARCH_PCALA_LO12, both against the local
|
||||
// data symbol. debug/elf does not surface rela entries, so read the
|
||||
// section directly.
|
||||
relaSec := ef.Section(".rela.text")
|
||||
if relaSec == nil {
|
||||
t.Fatal("missing .rela.text")
|
||||
}
|
||||
raw, err := relaSec.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(raw)%24 != 0 || len(raw)/24 != 2 {
|
||||
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
|
||||
}
|
||||
le := binary.LittleEndian
|
||||
for i := range 2 {
|
||||
e := raw[i*24 : (i+1)*24]
|
||||
off := le.Uint64(e[0:])
|
||||
info := le.Uint64(e[8:])
|
||||
typ := info & 0xffffffff
|
||||
sym := int(info >> 32)
|
||||
if i == 0 && (typ != uint64(elf.R_LARCH_PCALA_HI20) || off != 20) {
|
||||
t.Errorf("reloc %d: type %d off %d, want R_LARCH_PCALA_HI20 at 20", i, typ, off)
|
||||
}
|
||||
if i == 1 && (typ != uint64(elf.R_LARCH_PCALA_LO12) || off != 24) {
|
||||
t.Errorf("reloc %d: type %d off %d, want R_LARCH_PCALA_LO12 at 24", i, typ, off)
|
||||
}
|
||||
if sym != 3 { // NULL, .text, .data, then the first local: answer
|
||||
t.Errorf("reloc %d: symbol index %d, want 3 (answer)", i, sym)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFLOONG64ObjectNoRelocations checks a file with no static-symbol
|
||||
// references emits a valid object without a .rela.text section.
|
||||
func TestELFLOONG64ObjectNoRelocations(t *testing.T) {
|
||||
f, errs := parser.Parse("n_loong64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·nop(SB), NOSPLIT, $0
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileLOONG64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||
}
|
||||
obj, err := img.ELFLOONG64Object()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFLOONG64Object: %v", err)
|
||||
}
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
if ef.Section(".rela.text") != nil {
|
||||
t.Error("unexpected .rela.text section")
|
||||
}
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
found := false
|
||||
for _, s := range syms {
|
||||
if s.Name == "nop" && elf.ST_TYPE(s.Info) == elf.STT_FUNC {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Error("function symbol nop not found")
|
||||
}
|
||||
}
|
||||
+258
@@ -0,0 +1,258 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
)
|
||||
|
||||
// RISC-V ELF64 relocatable object emission.
|
||||
|
||||
const (
|
||||
emRISCV = 243 // EM_RISCV
|
||||
|
||||
// RISC-V relocation types.
|
||||
rRISCV32 = 1
|
||||
rRISCVJAL = 17 // R_RISCV_JAL
|
||||
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
|
||||
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
|
||||
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
|
||||
)
|
||||
|
||||
// ELFRISCVObject returns the image as an ELF64 relocatable object file for
|
||||
// RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64
|
||||
// ELF emission: .text, .data, .symtab, .strtab and optional .rela.text.
|
||||
func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
const (
|
||||
secText = 1
|
||||
secData = 2
|
||||
)
|
||||
|
||||
// Build symbol table.
|
||||
var locals, globals []elfSym
|
||||
for _, fn := range img.Funcs {
|
||||
s := elfSym{
|
||||
name: objectName(fn.Pkg, fn.Name),
|
||||
info: sttFunc,
|
||||
shndx: secText,
|
||||
value: uint64(fn.Offset),
|
||||
size: uint64(fn.Size),
|
||||
}
|
||||
if fn.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, d := range img.DataSyms {
|
||||
s := elfSym{
|
||||
name: objectName(d.Pkg, d.Name),
|
||||
info: sttObject,
|
||||
shndx: secData,
|
||||
value: uint64(d.Offset),
|
||||
size: uint64(d.Size),
|
||||
}
|
||||
if d.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, name := range img.Externals {
|
||||
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
|
||||
}
|
||||
syms := []elfSym{
|
||||
{},
|
||||
{name: ".text", info: sttSection, shndx: secText},
|
||||
{name: ".data", info: sttSection, shndx: secData},
|
||||
}
|
||||
syms = append(syms, locals...)
|
||||
shInfo := len(syms)
|
||||
syms = append(syms, globals...)
|
||||
symIdx := map[string]int{}
|
||||
for i, s := range syms {
|
||||
symIdx[s.name] = i
|
||||
}
|
||||
|
||||
// Build relocations. Each SB reference is an AUIPC + second-instruction
|
||||
// pair carrying a single relocation kind; the ELF writer expands it into
|
||||
// the R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I/S pair the psABI expects.
|
||||
// The HI20 carries the symbol addend; the LO12 addend is zero, matching
|
||||
// cmd/link's own ELF conversion (the LO12 resolves against the HI20's
|
||||
// AUIPC location).
|
||||
type elfRela struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
sym int
|
||||
addend int64
|
||||
}
|
||||
var relas []elfRela
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
switch r.Kind {
|
||||
case RelRISCVPCRELIType:
|
||||
relas = append(relas,
|
||||
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
||||
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: idx, addend: 0},
|
||||
)
|
||||
case RelRISCVPCRELSType:
|
||||
relas = append(relas,
|
||||
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
||||
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: idx, addend: 0},
|
||||
)
|
||||
case RelRISCVJal:
|
||||
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVJAL, sym: idx, addend: r.Addend})
|
||||
case RelPCRelAbs:
|
||||
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCV32, sym: idx, addend: r.Addend})
|
||||
default:
|
||||
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// String tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
stNames.add(s.name)
|
||||
}
|
||||
stSections := newElfStrtab()
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
for _, n := range dwarfSectionNames {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Layout.
|
||||
var out []byte
|
||||
out = append(out, make([]byte, 64)...)
|
||||
|
||||
align := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
|
||||
align(16)
|
||||
textOff := len(out)
|
||||
out = append(out, img.Code...)
|
||||
|
||||
align(16)
|
||||
dataOff := len(out)
|
||||
out = append(out, img.Data...)
|
||||
|
||||
align(8)
|
||||
symtabOff := len(out)
|
||||
for _, s := range syms {
|
||||
var b [24]byte
|
||||
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
|
||||
b[4] = s.info
|
||||
b[5] = 0
|
||||
le.PutUint16(b[6:], s.shndx)
|
||||
le.PutUint64(b[8:], s.value)
|
||||
le.PutUint64(b[16:], s.size)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
for _, r := range relas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
|
||||
dwAlign := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
|
||||
if dw != nil {
|
||||
nSections += 4
|
||||
}
|
||||
|
||||
align(8)
|
||||
shoff := len(out)
|
||||
|
||||
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
|
||||
var b [64]byte
|
||||
le.PutUint32(b[0:], uint32(stSections.at(name)))
|
||||
le.PutUint32(b[4:], uint32(typ))
|
||||
le.PutUint64(b[8:], flags)
|
||||
le.PutUint64(b[16:], 0)
|
||||
le.PutUint64(b[24:], uint64(off))
|
||||
le.PutUint64(b[32:], uint64(size))
|
||||
le.PutUint32(b[40:], uint32(link))
|
||||
le.PutUint32(b[44:], uint32(info))
|
||||
le.PutUint64(b[48:], alignV)
|
||||
le.PutUint64(b[56:], entsize)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
|
||||
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
|
||||
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
|
||||
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
|
||||
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
if dw != nil {
|
||||
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
|
||||
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
|
||||
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
|
||||
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
|
||||
if dw.frameSize > 0 {
|
||||
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
|
||||
}
|
||||
}
|
||||
|
||||
// ELF header.
|
||||
hdr := out[:64]
|
||||
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
|
||||
le.PutUint16(hdr[16:], etREL)
|
||||
le.PutUint16(hdr[18:], emRISCV)
|
||||
le.PutUint32(hdr[20:], elfVersion)
|
||||
le.PutUint64(hdr[24:], 0)
|
||||
le.PutUint64(hdr[32:], 0)
|
||||
le.PutUint64(hdr[40:], uint64(shoff))
|
||||
le.PutUint32(hdr[48:], 0)
|
||||
le.PutUint16(hdr[52:], 64)
|
||||
le.PutUint16(hdr[54:], 0)
|
||||
le.PutUint16(hdr[56:], 0)
|
||||
le.PutUint16(hdr[58:], 64)
|
||||
le.PutUint16(hdr[60:], uint16(nSections))
|
||||
le.PutUint16(hdr[62:], uint16(secShstr))
|
||||
|
||||
return out, nil
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import "strings"
|
||||
|
||||
// Encodable reports whether the amd64 encoder knows how to encode the
|
||||
// mnemonic. It mirrors the dispatch in (*enc).encode: the fixed-name
|
||||
// instructions, conditional jumps, the CMOV/SET condition families, the
|
||||
// VEX/EVEX/opmask/gather/scatter vector paths, the legacy SSE tables and the
|
||||
// explicit scalar cases. A mnemonic that parses (is in the architecture
|
||||
// table) but is not encodable would otherwise surface only at assembly time,
|
||||
// deep inside a build; the linter uses this predicate to flag it at edit
|
||||
// time.
|
||||
func Encodable(mnemonic string) bool {
|
||||
upper := strings.ToUpper(mnemonic)
|
||||
|
||||
// Fixed-name instructions (no size suffix).
|
||||
switch upper {
|
||||
case "RET", "NOP", "CALL", "JMP":
|
||||
return true
|
||||
}
|
||||
if _, ok := condCode(upper); ok {
|
||||
return true
|
||||
}
|
||||
|
||||
// VEX/EVEX and friends: the trailing B/W/L/Q/D is part of the mnemonic.
|
||||
base, _, err := parseEvexSuffix(upper)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
||||
base == "KMOVW" || base == "KMOVQ" {
|
||||
return true
|
||||
}
|
||||
|
||||
// CMOV carries size then condition (CMOVLGT); SET carries the condition
|
||||
// alone (SETNE).
|
||||
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
|
||||
if _, ok := jccMap[rest[1:]]; ok {
|
||||
return true
|
||||
}
|
||||
}
|
||||
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
|
||||
if _, ok := jccMap[rest]; ok {
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// Legacy SSE shuffles and packed binaries dispatch on the full name.
|
||||
if _, ok := sseShufTable[upper]; ok {
|
||||
return true
|
||||
}
|
||||
if _, ok := sseBinTable[upper]; ok {
|
||||
return true
|
||||
}
|
||||
|
||||
// The size-suffix split: retry the tables and the scalar switch on the
|
||||
// base.
|
||||
base2, size := splitSize(upper)
|
||||
if size == 0 {
|
||||
size = 8
|
||||
}
|
||||
_ = size
|
||||
if base2 != upper {
|
||||
if _, ok := sseBinTable[base2]; ok {
|
||||
return true
|
||||
}
|
||||
}
|
||||
switch base2 {
|
||||
case "MOV",
|
||||
"ADD", "SUB", "AND", "OR", "XOR", "CMP",
|
||||
"TEST",
|
||||
"LEA",
|
||||
"INC", "DEC", "NEG", "NOT",
|
||||
"SHL", "SHR", "SAR",
|
||||
"IMUL", "IMUL3",
|
||||
"PUSH", "POP",
|
||||
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
|
||||
"BSWAP",
|
||||
"PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2",
|
||||
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
|
||||
"CVTSL2SD", "CVTSQ2SD",
|
||||
"MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
+127
-7
@@ -19,7 +19,16 @@ func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
|
||||
}
|
||||
|
||||
type enc struct {
|
||||
out []byte
|
||||
out []byte
|
||||
patches []encPatch // disp32 fields awaiting static-symbol resolution
|
||||
}
|
||||
|
||||
// encPatch marks a 4-byte displacement field in enc.out that must receive the
|
||||
// RIP-relative offset of a static symbol once the file layout is settled.
|
||||
type encPatch struct {
|
||||
off int
|
||||
name string
|
||||
addend int64
|
||||
}
|
||||
|
||||
func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
@@ -40,10 +49,19 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
return e.encodeJcc(cc, ops)
|
||||
}
|
||||
|
||||
// VEX (AVX/AVX2) instructions: the trailing B/W/L/Q/D is part of the
|
||||
// mnemonic, not a size suffix, so dispatch before splitSize.
|
||||
if isVex(upper) {
|
||||
return e.encodeVex(upper, ops)
|
||||
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
||||
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
|
||||
// before splitSize. EVEX suffixes (.Z, .SAE, rounding, .BCST) split
|
||||
// off the mnemonic too.
|
||||
base, sfx, err := parseEvexSuffix(upper)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
|
||||
return e.encodeVec(base, ops, sfx)
|
||||
}
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s: the suffix requires an EVEX instruction", mnem)
|
||||
}
|
||||
|
||||
// CMOVcc and SETcc carry the condition in the mnemonic (CMOVLGT, SETNE).
|
||||
@@ -58,6 +76,21 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
if size == 0 {
|
||||
size = 8 // default operand size in 64-bit mode (e.g. PUSHQ)
|
||||
}
|
||||
// Legacy SSE imm8 shuffles whose names end in W/H (PSHUFLW,
|
||||
// PSHUFHW) must dispatch BEFORE the size-suffix split, and the
|
||||
// others ride along.
|
||||
if m, ok := sseShufTable[upper]; ok {
|
||||
return e.encodeSSEShuf(m, ops)
|
||||
}
|
||||
// Legacy SSE packed binaries dispatch on the full name: the packed
|
||||
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
|
||||
// which the size split must not eat.
|
||||
if m, ok := sseBinTable[upper]; ok {
|
||||
return e.encodeSSEBin(m, ops)
|
||||
}
|
||||
if m, ok := sseBinTable[base]; ok {
|
||||
return e.encodeSSEBin(m, ops)
|
||||
}
|
||||
switch base {
|
||||
case "MOV":
|
||||
return e.encodeMov(ops, size)
|
||||
@@ -77,16 +110,46 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
return e.encodePushPop(ops, true)
|
||||
case "POP":
|
||||
return e.encodePushPop(ops, false)
|
||||
case "LZCNT", "TZCNT":
|
||||
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
|
||||
return e.encodeCount(base, ops, size)
|
||||
case "BSWAP":
|
||||
return e.encodeBswap(ops, size)
|
||||
case "PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2":
|
||||
return e.encodePrefetch(base, ops)
|
||||
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX":
|
||||
return e.encodeMovExtend(base, ops)
|
||||
case "CVTSL2SD", "CVTSQ2SD":
|
||||
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
|
||||
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||
}
|
||||
return fmt.Errorf("unsupported instruction %q", mnem)
|
||||
}
|
||||
|
||||
// encodePrefetch emits the 0F 18 /r prefetch hints: the reg field selects
|
||||
// the locality (NTA=0, T0=1, T1=2, T2=3) and the single operand is memory.
|
||||
func (e *enc) encodePrefetch(base string, ops []Operand) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("%s expects one memory operand", base)
|
||||
}
|
||||
m, ok := ops[0].(Mem)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s requires a memory operand", base)
|
||||
}
|
||||
i := newInstr(0, []byte{0x0F, 0x18})
|
||||
if err := setMem(i, prefetchVariant[base], m); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
var prefetchVariant = map[string]int{
|
||||
"PREFETCHNTA": 0,
|
||||
"PREFETCHT0": 1,
|
||||
"PREFETCHT1": 2,
|
||||
"PREFETCHT2": 3,
|
||||
}
|
||||
|
||||
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
|
||||
func splitSize(upper string) (base string, size int) {
|
||||
if upper == "" {
|
||||
@@ -105,6 +168,38 @@ func splitSize(upper string) (base string, size int) {
|
||||
return upper, 0
|
||||
}
|
||||
|
||||
// encodeVec dispatches a VEX/EVEX mnemonic to the right encoding: KMOVW has
|
||||
// its own direction-dependent opcodes; KTESTW is always VEX; everything else
|
||||
// takes EVEX when an operand demands it (a ZMM or K register, or an
|
||||
// EVEX-only mnemonic) and VEX otherwise.
|
||||
func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
|
||||
if gs, ok := gatherTable[upper]; ok {
|
||||
return e.encodeGather(upper, gs, ops, sfx)
|
||||
}
|
||||
if ss, ok := scatterTable[upper]; ok {
|
||||
return e.encodeScatter(upper, ss, ops, sfx)
|
||||
}
|
||||
if upper == "KMOVW" || upper == "KMOVQ" {
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
||||
}
|
||||
return e.encodeKmov(upper, ops)
|
||||
}
|
||||
if isKOp(upper) {
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
||||
}
|
||||
return e.encodeKOp(upper, ops)
|
||||
}
|
||||
if upper == "KTESTW" || (!evexRequired(upper, ops) && !sfx.evexOnly()) {
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s: the .Z suffix requires an EVEX instruction", upper)
|
||||
}
|
||||
return e.encodeVex(upper, ops)
|
||||
}
|
||||
return e.encodeEvex(upper, ops, sfx)
|
||||
}
|
||||
|
||||
// --- instruction components -------------------------------------------------
|
||||
|
||||
type instr struct {
|
||||
@@ -120,6 +215,14 @@ type instr struct {
|
||||
sib int // -1 if absent
|
||||
disp []byte
|
||||
imm []byte
|
||||
sb *sbRef // static-symbol displacement in disp, awaiting resolution
|
||||
}
|
||||
|
||||
// sbRef records that an instruction's displacement refers to a static symbol
|
||||
// rather than holding a literal value.
|
||||
type sbRef struct {
|
||||
name string
|
||||
addend int64
|
||||
}
|
||||
|
||||
func (e *enc) emit(i *instr) error {
|
||||
@@ -152,6 +255,9 @@ func (e *enc) emit(i *instr) error {
|
||||
if i.sib >= 0 {
|
||||
e.out = append(e.out, byte(i.sib))
|
||||
}
|
||||
if i.sb != nil {
|
||||
e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend})
|
||||
}
|
||||
e.out = append(e.out, i.disp...)
|
||||
e.out = append(e.out, i.imm...)
|
||||
return nil
|
||||
@@ -199,6 +305,13 @@ func setRMReg(i *instr, regField int, rexR, regForced bool, rm Operand, opSize i
|
||||
return nil
|
||||
case Mem:
|
||||
return setMem(i, regField, r)
|
||||
case sbMem:
|
||||
// RIP-relative reference; the displacement is patched once the static
|
||||
// symbol's address is known.
|
||||
i.modrm = regField<<3 | 0x05 // mod=00, rm=101 → (RIP)+disp32
|
||||
i.disp = le32(0)
|
||||
i.sb = &sbRef{name: r.name, addend: r.addend}
|
||||
return nil
|
||||
default:
|
||||
return fmt.Errorf("invalid r/m operand %T", rm)
|
||||
}
|
||||
@@ -227,6 +340,13 @@ func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit
|
||||
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
|
||||
}
|
||||
|
||||
// The SIB scale field only encodes 1/2/4/8; the Go assembler rejects
|
||||
// anything else ("bad scale: 16"), so a silent fallback to scale 1 here
|
||||
// would mis-assemble the operand instead of reporting it.
|
||||
if m.HasIndex && m.Scale != 1 && m.Scale != 2 && m.Scale != 4 && m.Scale != 8 {
|
||||
return 0, -1, nil, 0, 0, fmt.Errorf("bad scale: %d", m.Scale)
|
||||
}
|
||||
|
||||
needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4)
|
||||
|
||||
var mod int
|
||||
@@ -299,7 +419,7 @@ func le16(v int64) []byte {
|
||||
func le64(v int64) []byte {
|
||||
u := uint64(v)
|
||||
b := make([]byte, 8)
|
||||
for i := 0; i < 8; i++ {
|
||||
for i := range 8 {
|
||||
b[i] = byte(u >> (8 * i))
|
||||
}
|
||||
return b
|
||||
|
||||
+200
-3
@@ -4,6 +4,7 @@
|
||||
package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
@@ -57,8 +58,11 @@ func TestMov(t *testing.T) {
|
||||
checkSyntax(t, "mov qword ptr [rbx], rax", "MOVQ", AX, Ptr(BX, 0, 8))
|
||||
checkSyntax(t, "mov rbx, qword ptr [rax+0x10]", "MOVQ", Ptr(AX, 0x10, 8), BX)
|
||||
checkSyntax(t, "mov rbx, qword ptr [rsi+4*rbx]", "MOVQ", Idx(SI, BX, 4, 0, 8), BX)
|
||||
checkSyntax(t, "mov rax, 0x5", "MOVQ", Imm(5), AX)
|
||||
checkSyntax(t, "mov r8, 0x5", "MOVQ", Imm(5), Reg{idx: 8, size: 8})
|
||||
// A small positive immediate compresses to the 32-bit zero-extending
|
||||
// form (matching go tool asm), so the disassembler renders the 32-bit
|
||||
// register name even for MOVQ.
|
||||
checkSyntax(t, "mov eax, 0x5", "MOVQ", Imm(5), AX)
|
||||
checkSyntax(t, "mov r8d, 0x5", "MOVQ", Imm(5), Reg{idx: 8, size: 8})
|
||||
checkSyntax(t, "mov qword ptr [rax], 0x5", "MOVQ", Imm(5), Ptr(AX, 0, 8))
|
||||
checkSyntax(t, "mov r12, r13", "MOVQ", Reg{idx: 13, size: 8}, Reg{idx: 12, size: 8})
|
||||
}
|
||||
@@ -74,13 +78,63 @@ func TestALU(t *testing.T) {
|
||||
checkSyntax(t, "cmp rsi, r10", "CMPQ", SI, Reg{idx: 10, size: 8})
|
||||
checkSyntax(t, "add rbx, qword ptr [rax]", "ADDQ", Ptr(AX, 0, 8), BX)
|
||||
checkSyntax(t, "add qword ptr [rax], rbx", "ADDQ", BX, Ptr(AX, 0, 8))
|
||||
checkSyntax(t, "cmp rbx, -0x20", "CMPQ", Imm(-32), BX)
|
||||
// The Go assembler rejects the immediate-first CMP spelling outright,
|
||||
// so Encode errors instead of silently emitting the swapped form.
|
||||
if _, err := Encode("CMPQ", Imm(-32), BX); err == nil {
|
||||
t.Errorf("Encode(CMPQ imm-first) should error, got success")
|
||||
}
|
||||
// The Go assembler's own spelling: immediate second.
|
||||
checkSyntax(t, "cmp ecx, 0x1f", "CMPL", CX, Imm(31))
|
||||
checkSyntax(t, "cmp ecx, -0x80000000", "CMPL", CX, Imm(-2147483648))
|
||||
checkSyntax(t, "cmp r9, -0x80000000", "CMPQ", Reg{idx: 9, size: 8}, Imm(-2147483648))
|
||||
}
|
||||
|
||||
// TestScalarXmmRegMoves pins the Go-assembler byte forms of scalar
|
||||
// MOVQ/MOVL between GPRs and XMM registers (66 REX.W 0F 6E/0F 7E) and the
|
||||
// memory forms (F3 0F 7E load, 66 0F D6 store), all byte-for-byte.
|
||||
func TestScalarXmmRegMoves(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"MOVQ AX,X1", "MOVQ", []Operand{AX, vreg(t, "X1")}, "66480f6ec8"},
|
||||
{"MOVQ DX,X2", "MOVQ", []Operand{DX, vreg(t, "X2")}, "66480f6ed2"},
|
||||
{"MOVQ X1,AX", "MOVQ", []Operand{vreg(t, "X1"), AX}, "66480f7ec8"},
|
||||
{"MOVQ X0,DX", "MOVQ", []Operand{vreg(t, "X0"), DX}, "66480f7ec2"},
|
||||
{"MOVL AX,X1", "MOVL", []Operand{AX, vreg(t, "X1")}, "660f6ec8"},
|
||||
{"MOVL X1,AX", "MOVL", []Operand{vreg(t, "X1"), AX}, "660f7ec8"},
|
||||
{"MOVQ (SI),X1", "MOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X1")}, "f30f7e0e"},
|
||||
{"MOVQ X3,(DI)", "MOVQ", []Operand{vreg(t, "X3"), Ptr(DI, 0, 8)}, "660fd61f"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestBadScale pins the go-tool-asm parity of rejecting SIB scales the
|
||||
// hardware cannot encode.
|
||||
func TestBadScale(t *testing.T) {
|
||||
for _, sc := range []int{3, 5, 16, 32} {
|
||||
if _, err := Encode("LEAQ", Idx(SI, BX, sc, 0, 8), AX); err == nil {
|
||||
t.Errorf("LEAQ scale %d: expected error, got success", sc)
|
||||
}
|
||||
}
|
||||
for _, sc := range []int{1, 2, 4, 8} {
|
||||
if _, err := Encode("LEAQ", Idx(SI, BX, sc, 0, 8), AX); err != nil {
|
||||
t.Errorf("LEAQ scale %d: %v", sc, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestLea(t *testing.T) {
|
||||
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
|
||||
checkSyntax(t, "lea rax, ptr [rbx+0x8]", "LEAQ", Ptr(BX, 0x8, 8), AX)
|
||||
@@ -127,6 +181,52 @@ func TestControl(t *testing.T) {
|
||||
checkOp(t, x86asm.JBE, "JLS", Imm(0))
|
||||
}
|
||||
|
||||
// TestSSEMoveGroundTruth checks the legacy (non-VEX) SSE moves byte for byte
|
||||
// against the Go assembler. wantOp is the decoder's name, which differs from
|
||||
// the Plan 9 spelling for the octa moves (MOVOU = MOVDQU, MOVO = MOVDQA).
|
||||
func TestSSEMoveGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
wantOp string
|
||||
}{
|
||||
{"MOVOU (SI),X1", "MOVOU", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "f30f6f0e", "MOVDQU"},
|
||||
{"MOVOU X3,(DI)", "MOVOU", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "f30f7f1f", "MOVDQU"},
|
||||
{"MOVOU X1,X2", "MOVOU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f30f6fd1", "MOVDQU"},
|
||||
{"MOVOU (SI)(BX*4),X9", "MOVOU", []Operand{Idx(SI, BX, 4, 0, 16), vreg(t, "X9")}, "f3440f6f0c9e", "MOVDQU"},
|
||||
{"MOVO (SI),X1", "MOVO", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "660f6f0e", "MOVDQA"},
|
||||
{"MOVO X3,(DI)", "MOVO", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "660f7f1f", "MOVDQA"},
|
||||
{"MOVUPS (SI),X1", "MOVUPS", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "0f100e", "MOVUPS"},
|
||||
{"MOVAPS X3,(DI)", "MOVAPS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "0f291f", "MOVAPS"},
|
||||
{"MOVUPD (SI),X1", "MOVUPD", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "660f100e", "MOVUPD"},
|
||||
{"MOVAPD X3,(DI)", "MOVAPD", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "660f291f", "MOVAPD"},
|
||||
{"MOVSD (SI),X1", "MOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X1")}, "f20f100e", "MOVSD_XMM"},
|
||||
{"MOVSD X1,X2", "MOVSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f20f10d1", "MOVSD_XMM"},
|
||||
{"MOVSS X3,(DI)", "MOVSS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 4)}, "f30f111f", "MOVSS"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
if inst.Op.String() != c.wantOp {
|
||||
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestGoFlacScalarTail encodes the scalar tail of an analyze kernel to confirm
|
||||
// the encoder handles a realistic instruction sequence.
|
||||
func TestGoFlacScalarTail(t *testing.T) {
|
||||
@@ -157,6 +257,21 @@ func TestScalarGroundTruth(t *testing.T) {
|
||||
{"LZCNTQ R8,R9", "LZCNTQ", []Operand{r8, r9}, "f34d0fbdc8", "LZCNT"},
|
||||
{"LZCNTW AX,CX", "LZCNTW", []Operand{AX, CX}, "66f30fbdc8", "LZCNT"},
|
||||
{"TZCNTL AX,CX", "TZCNTL", []Operand{AX, CX}, "f30fbcc8", "TZCNT"},
|
||||
// Bit scan: BSF/BSR are the unprefixed forms of TZCNT/LZCNT's map.
|
||||
{"BSFL AX,CX", "BSFL", []Operand{AX, CX}, "0fbcc8", "BSF"},
|
||||
{"BSFQ R8,R9", "BSFQ", []Operand{r8, r9}, "4d0fbcc8", "BSF"},
|
||||
{"BSFW AX,CX", "BSFW", []Operand{AX, CX}, "660fbcc8", "BSF"},
|
||||
{"BSRL AX,CX", "BSRL", []Operand{AX, CX}, "0fbdc8", "BSR"},
|
||||
{"BSRQ AX,CX", "BSRQ", []Operand{AX, CX}, "480fbdc8", "BSR"},
|
||||
{"POPCNTL AX,CX", "POPCNTL", []Operand{AX, CX}, "f30fb8c8", "POPCNT"},
|
||||
{"POPCNTQ R8,R9", "POPCNTQ", []Operand{r8, r9}, "f34d0fb8c8", "POPCNT"},
|
||||
// A 64-bit immediate that fits a signed int32 is compressed exactly
|
||||
// as the Go assembler does: positive via B8+rd without REX.W
|
||||
// (zero-extended), negative via REX.W C7 /0 (sign-extended).
|
||||
{"MOVQ $4,BX", "MOVQ", []Operand{Imm(4), BX}, "bb04000000", "MOV"},
|
||||
{"MOVQ $4,R8", "MOVQ", []Operand{Imm(4), r8}, "41b804000000", "MOV"},
|
||||
{"MOVQ $-1,BX", "MOVQ", []Operand{Imm(-1), BX}, "48c7c3ffffffff", "MOV"},
|
||||
{"MOVQ big,BX", "MOVQ", []Operand{Imm(0x1122334455667788), BX}, "48bb8877665544332211", "MOV"},
|
||||
{"CMOVLGT CX,AX", "CMOVLGT", []Operand{CX, AX}, "0f4fc1", "CMOVG"},
|
||||
{"CMOVLEQ CX,AX", "CMOVLEQ", []Operand{CX, AX}, "0f44c1", "CMOVE"},
|
||||
{"CMOVQGT R9,R8", "CMOVQGT", []Operand{r9, r8}, "4d0f4fc1", "CMOVG"},
|
||||
@@ -248,3 +363,85 @@ func TestScalarErrors(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
|
||||
// byte for byte (no prefix / 66 / F2 / F3 variants).
|
||||
func TestSSEBinGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"MULPS X0,X1", "MULPS", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f59c8"},
|
||||
{"MULPS (DI),X1", "MULPS", []Operand{Ptr(DI, 0, 16), vreg(t, "X1")}, "0f590f"},
|
||||
{"ADDPD X1,X2", "ADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "660f58d1"},
|
||||
{"XORPS X0,X0", "XORPS", []Operand{vreg(t, "X0"), vreg(t, "X0")}, "0f57c0"},
|
||||
{"UNPCKLPS X0,X0", "UNPCKLPS", []Operand{vreg(t, "X0"), vreg(t, "X0")}, "0f14c0"},
|
||||
{"MULSD X1,X2", "MULSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f20f59d1"},
|
||||
{"ADDSS (DI),X0", "ADDSS", []Operand{Ptr(DI, 0, 4), vreg(t, "X0")}, "f30f5807"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSSEShuffleGroundTruth checks the imm8 shuffle family: immediate
|
||||
// first in Plan 9 order, encoded last on the wire.
|
||||
func TestSSEShuffleGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"SHUFPS $0,X0,X0", "SHUFPS", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X0")}, "0fc6c000"},
|
||||
{"SHUFPS $27,X1,X2", "SHUFPS", []Operand{Imm(27), vreg(t, "X1"), vreg(t, "X2")}, "0fc6d11b"},
|
||||
{"PSHUFD $0,X0,X0", "PSHUFD", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X0")}, "660f70c000"},
|
||||
{"PSHUFLW $3,(DI),X1", "PSHUFLW", []Operand{Imm(3), Ptr(DI, 0, 8), vreg(t, "X1")}, "f20f700f03"},
|
||||
{"PSHUFHW $2,X1,X2", "PSHUFHW", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2")}, "f30f70d102"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestMOVQXMMGroundTruth pins the SSE2 packed-quadword move encodings:
|
||||
// loads and register moves on F3 0F 7E, stores on 66 0F D6 — the forms
|
||||
// the GPR-move fallback silently corrupted.
|
||||
func TestMOVQXMMGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"MOVQ (DI),X0", "MOVQ", []Operand{Ptr(DI, 0, 8), vreg(t, "X0")}, "f30f7e07"},
|
||||
{"MOVQ X1,X2", "MOVQ", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f30f7ed1"},
|
||||
{"MOVQ X0,(DI)", "MOVQ", []Operand{vreg(t, "X0"), Ptr(DI, 0, 8)}, "660fd607"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+1619
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,695 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
)
|
||||
|
||||
// TestEvexGroundTruth checks the EVEX (AVX-512) encodings byte for byte
|
||||
// against machine code extracted from the Go toolchain's assembly of the
|
||||
// same instructions, covering every operand shape the go-flac AVX-512
|
||||
// kernels use: NDS arithmetic, immediate and variable shifts, shuffles with
|
||||
// an immediate, lane extracts, narrowing stores, broadcasts from a GPR or
|
||||
// memory, mask destinations, mask moves, disp8×N compression and the 5-bit
|
||||
// register fields (X/Y 16–31, Z 0–31).
|
||||
func TestEvexGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// NDS integer arithmetic / logic.
|
||||
{"VPXORD Z12,Z12,Z12", "VPXORD", []Operand{vreg(t, "Z12"), vreg(t, "Z12"), vreg(t, "Z12")}, "62511d48efe4"},
|
||||
{"VPXORQ Z8,Z9,Z10", "VPXORQ", []Operand{vreg(t, "Z8"), vreg(t, "Z9"), vreg(t, "Z10")}, "6251b548efd0"},
|
||||
{"VPADDD Z1,Z0,Z0", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z0"), vreg(t, "Z0")}, "62f17d48fec1"},
|
||||
{"VPSUBQ Z8,Z11,Z11", "VPSUBQ", []Operand{vreg(t, "Z8"), vreg(t, "Z11"), vreg(t, "Z11")}, "6251a548fbd8"},
|
||||
{"VPUNPCKLDQ Z5,Z3,Z6", "VPUNPCKLDQ", []Operand{vreg(t, "Z5"), vreg(t, "Z3"), vreg(t, "Z6")}, "62f1654862f5"},
|
||||
{"VPUNPCKHDQ Z5,Z3,Z7", "VPUNPCKHDQ", []Operand{vreg(t, "Z5"), vreg(t, "Z3"), vreg(t, "Z7")}, "62f165486afd"},
|
||||
{"VPMULLQ Z9,Z10,Z10", "VPMULLQ", []Operand{vreg(t, "Z9"), vreg(t, "Z10"), vreg(t, "Z10")}, "6252ad4840d1"},
|
||||
{"VPMULLD Z13,Z11,Z2", "VPMULLD", []Operand{vreg(t, "Z13"), vreg(t, "Z11"), vreg(t, "Z2")}, "62d2254840d5"},
|
||||
{"VPERMD Z0,Z15,Z8", "VPERMD", []Operand{vreg(t, "Z0"), vreg(t, "Z15"), vreg(t, "Z8")}, "6272054836c0"},
|
||||
// Packed-double arithmetic (EVEX forms carry W=1).
|
||||
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
|
||||
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
|
||||
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
|
||||
// Align (NDS + imm8).
|
||||
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
|
||||
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
|
||||
// Shifts: immediate (/digit) and variable (XMM count).
|
||||
{"VPSRAD $31,Z3,Z5", "VPSRAD", []Operand{Imm(31), vreg(t, "Z3"), vreg(t, "Z5")}, "62f1554872e31f"},
|
||||
{"VPSLLD $1,Z3,Z4", "VPSLLD", []Operand{Imm(1), vreg(t, "Z3"), vreg(t, "Z4")}, "62f15d4872f301"},
|
||||
{"VPSRAQ X31,Z8,Z8", "VPSRAQ", []Operand{vreg(t, "X31"), vreg(t, "Z8"), vreg(t, "Z8")}, "6211bd48e2c7"},
|
||||
// Mask destinations (the K register occupies the reg field).
|
||||
{"VPCMPEQD Z0,Z3,K1", "VPCMPEQD", []Operand{vreg(t, "Z0"), vreg(t, "Z3"), vreg(t, "K1")}, "62f1654876c8"},
|
||||
{"VPCMPEQD Y30,Y11,K1", "VPCMPEQD", []Operand{vreg(t, "Y30"), vreg(t, "Y11"), vreg(t, "K1")}, "6291252876ce"},
|
||||
// Mask moves and test (VEX-encoded).
|
||||
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
|
||||
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
|
||||
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
|
||||
// Moves, incl. disp8×N (64 for a 512-bit operand).
|
||||
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
|
||||
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
|
||||
{"VMOVDQU32 16(SI)(R15*4),Z4", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 16, 64), vreg(t, "Z4")}, "62b17e486fa4be10000000"},
|
||||
{"VMOVDQU32 Z0,4(SI)(AX*1)", "VMOVDQU32", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f17e487f840604000000"},
|
||||
{"VMOVDQU32 Z3,(DI)(R15*4)", "VMOVDQU32", []Operand{vreg(t, "Z3"), Idx(DI, vreg(t, "R15"), 4, 0, 64)}, "62b17e487f1cbf"},
|
||||
// VMOVDQU64 — the W1 qword variant.
|
||||
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
|
||||
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
|
||||
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
|
||||
// The wider AVX-512 F/BW integer set.
|
||||
{"VPADDB Z1,Z2,Z3", "VPADDB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48fcd9"},
|
||||
{"VPSUBW Z1,Z2,Z3", "VPSUBW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48f9d9"},
|
||||
{"VPANDQ Z1,Z2,Z3", "VPANDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed48dbd9"},
|
||||
{"VPANDND Z1,Z2,Z3", "VPANDND", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48dfd9"},
|
||||
{"VPMULLW Z1,Z2,Z3", "VPMULLW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48d5d9"},
|
||||
{"VPMINUB Z1,Z2,Z3", "VPMINUB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48dad9"},
|
||||
{"VPMAXUQ Z1,Z2,Z3", "VPMAXUQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed483fd9"},
|
||||
{"VPAVGW Z1,Z2,Z3", "VPAVGW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48e3d9"},
|
||||
{"VPSLLVQ Z3,Z1,Z2", "VPSLLVQ", []Operand{vreg(t, "Z3"), vreg(t, "Z1"), vreg(t, "Z2")}, "62f2f54847d3"},
|
||||
{"VPSRAVQ Z3,Z1,Z2", "VPSRAVQ", []Operand{vreg(t, "Z3"), vreg(t, "Z1"), vreg(t, "Z2")}, "62f2f54846d3"},
|
||||
{"VPSHUFD $0x1B,Z1,Z2", "VPSHUFD", []Operand{Imm(0x1B), vreg(t, "Z1"), vreg(t, "Z2")}, "62f17d4870d11b"},
|
||||
{"VPSHUFB Z1,Z2,Z3", "VPSHUFB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4800d9"},
|
||||
{"VMOVDQU8 Z1,Z2", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17f487fca"},
|
||||
{"VMOVDQU16 Z1,Z2", "VMOVDQU16", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff487fca"},
|
||||
// Indices 16–31: rm[4] rides in X̄ for register operands.
|
||||
{"VPSHUFD $1,X16,X17", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}, "62a17d0870c801"},
|
||||
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
|
||||
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
|
||||
// Conversions and narrowing stores (reg = wide source).
|
||||
{"VCVTQQ2PD Z12,Z12", "VCVTQQ2PD", []Operand{vreg(t, "Z12"), vreg(t, "Z12")}, "6251fe48e6e4"},
|
||||
{"VCVTQQ2PD X13,X13", "VCVTQQ2PD", []Operand{vreg(t, "X13"), vreg(t, "X13")}, "6251fe08e6ed"},
|
||||
{"VPMOVSXDQ 32(SI),Z12", "VPMOVSXDQ", []Operand{Ptr(SI, 32, 32), vreg(t, "Z12")}, "62727d48256601"},
|
||||
{"VPMOVDW Z0,Y0", "VPMOVDW", []Operand{vreg(t, "Z0"), vreg(t, "Y0")}, "62f27e4833c0"},
|
||||
{"VPMOVQD Z11,Y11", "VPMOVQD", []Operand{vreg(t, "Z11"), vreg(t, "Y11")}, "62527e4835db"},
|
||||
// Lane extracts.
|
||||
{"VEXTRACTI64X4 $1,Z8,Y9", "VEXTRACTI64X4", []Operand{Imm(1), vreg(t, "Z8"), vreg(t, "Y9")}, "6253fd483bc101"},
|
||||
{"VEXTRACTF64X4 $1,Z10,Y11", "VEXTRACTF64X4", []Operand{Imm(1), vreg(t, "Z10"), vreg(t, "Y11")}, "6253fd481bd301"},
|
||||
// Broadcasts: GPR source (0x7C) vs memory source (0x58/0x59, disp8×4/8).
|
||||
{"VPBROADCASTD AX,Z15", "VPBROADCASTD", []Operand{AX, vreg(t, "Z15")}, "62727d487cf8"},
|
||||
{"VPBROADCASTD (SI),Z8", "VPBROADCASTD", []Operand{Ptr(SI, 0, 4), vreg(t, "Z8")}, "62727d485806"},
|
||||
{"VPBROADCASTD 4(SI),Z10", "VPBROADCASTD", []Operand{Ptr(SI, 4, 4), vreg(t, "Z10")}, "62727d48585601"},
|
||||
{"VPBROADCASTQ R8,X31", "VPBROADCASTQ", []Operand{vreg(t, "R8"), vreg(t, "X31")}, "6242fd087cf8"},
|
||||
{"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"},
|
||||
// Register indices 16–31 exist only in EVEX encodings.
|
||||
{"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"},
|
||||
// Packed double arithmetic / unpack (EVEX forms carry W=1).
|
||||
{"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"},
|
||||
{"VDIVPD Z4,Z5,Z6", "VDIVPD", []Operand{vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "Z6")}, "62f1d5485ef4"},
|
||||
{"VMINPD Z7,Z8,Z9", "VMINPD", []Operand{vreg(t, "Z7"), vreg(t, "Z8"), vreg(t, "Z9")}, "6271bd485dcf"},
|
||||
{"VMAXPD Z10,Z11,Z12", "VMAXPD", []Operand{vreg(t, "Z10"), vreg(t, "Z11"), vreg(t, "Z12")}, "6251a5485fe2"},
|
||||
{"VUNPCKLPD Z1,Z2,Z3", "VUNPCKLPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4814d9"},
|
||||
{"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"},
|
||||
{"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"},
|
||||
{"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"},
|
||||
// VMOVDDUP — duplicate the low double; disp8×N = 64 at 512 bits, and
|
||||
// X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄).
|
||||
{"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"},
|
||||
{"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"},
|
||||
{"VMOVDDUP X16,X17", "VMOVDDUP", []Operand{vreg(t, "X16"), vreg(t, "X17")}, "62a1ff0812c8"},
|
||||
// Conversions: DQ→PS, PS→PD (pp = 00, the Go assembler's choice),
|
||||
// DQ→PD (the destination sets the length).
|
||||
{"VCVTDQ2PS Z1,Z2", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"},
|
||||
{"VCVTPS2PD Y1,Z2", "VCVTPS2PD", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17c485ad1"},
|
||||
{"VCVTPS2PD 32(AX),Z2", "VCVTPS2PD", []Operand{Ptr(AX, 32, 32), vreg(t, "Z2")}, "62f17c485a5001"},
|
||||
{"VCVTDQ2PD Y1,Z2", "VCVTDQ2PD", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17e48e6d1"},
|
||||
// PD→DQ conversions: the source is the wide operand and fixes the
|
||||
// length (ZMM source → L'L = 10 even with an XMM destination; a
|
||||
// memory source takes the length the mnemonic's spelling implies).
|
||||
{"VCVTPD2DQ Z1,Y2", "VCVTPD2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff48e6d1"},
|
||||
{"VCVTPD2DQ 64(AX),Y2", "VCVTPD2DQ", []Operand{Ptr(AX, 64, 64), vreg(t, "Y2")}, "62f1ff48e65001"},
|
||||
{"VCVTTPD2DQ Z3,Y4", "VCVTTPD2DQ", []Operand{vreg(t, "Z3"), vreg(t, "Y4")}, "62f1fd48e6e3"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
want := strings.ReplaceAll(c.want, " ", "")
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
if inst.Len != len(code) {
|
||||
t.Errorf("%s: Decode consumed %d of %d bytes", c.name, inst.Len, len(code))
|
||||
}
|
||||
if inst.Op.String() != c.mnem {
|
||||
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexMasking checks the AVX-512 mask operand (K1–K7, placed freely among
|
||||
// the operands) and the .Z zeroing suffix, byte for byte against the Go
|
||||
// assembler.
|
||||
func TestEvexMasking(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// Masked arithmetic: K anywhere among the operands; .Z sets the z bit.
|
||||
{"VPADDD.Z merging+zeroing", "VPADDD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f16dcafed9"},
|
||||
{"VPADDD merging", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49fed9"},
|
||||
{"VADDPD.Z", "VADDPD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1edca58d9"},
|
||||
{"VPMINSD.Z", "VPMINSD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K5"), vreg(t, "Z3")}, "62f26dcd39d9"},
|
||||
{"VPMINSQ.Z", "VPMINSQ.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K5"), vreg(t, "Z3")}, "62f2edcd39d9"},
|
||||
// Masked immediate shift (K before the destination).
|
||||
{"VPSRAD.Z", "VPSRAD.Z", []Operand{Imm(1), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f165c972e201"},
|
||||
{"VPSLLD merge", "VPSLLD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1654a72f104"},
|
||||
// Masked align.
|
||||
{"VALIGND", "VALIGND", []Operand{Imm(12), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f36d4b03e10c"},
|
||||
// Masked conversion and extract.
|
||||
{"VCVTQQ2PD.Z", "VCVTQQ2PD.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1fecae6d9"},
|
||||
{"VEXTRACTI64X4", "VEXTRACTI64X4", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f3fd4a3bcb01"},
|
||||
// Masked moves: K sits between the register and memory operands.
|
||||
{"VMOVDQU8 store", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "K3"), Ptr(SI, 0, 64)}, "62f17f4b7f0e"},
|
||||
{"VMOVDQU32 load", "VMOVDQU32", []Operand{Ptr(SI, 0, 64), vreg(t, "K4"), vreg(t, "Z1")}, "62f17e4c6f0e"},
|
||||
{"VMOVDQU32 store", "VMOVDQU32", []Operand{vreg(t, "Z1"), vreg(t, "K4"), Ptr(DI, 0, 64)}, "62f17e4c7f0f"},
|
||||
// Masked comparison with a K destination: dst K1, mask K2.
|
||||
{"VPCMPEQD k-dst+mask", "VPCMPEQD", []Operand{vreg(t, "Z0"), vreg(t, "Z3"), vreg(t, "K2"), vreg(t, "K1")}, "62f1654a76c8"},
|
||||
// Masked floating point: packed double, the scalar SD/SS forms (which
|
||||
// exist under EVEX only for masked and zeroing use) and conversions.
|
||||
{"VSUBPD.Z", "VSUBPD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f1edcb5ce1"},
|
||||
{"VADDSD merge", "VADDSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K3"), vreg(t, "X4")}, "62f1ef0b58e1"},
|
||||
{"VSUBSD.Z", "VSUBSD.Z", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K5"), vreg(t, "X3")}, "62f1ef8d5cd9"},
|
||||
{"VADDSS merge", "VADDSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f16e0958d9"},
|
||||
{"VCVTPD2DQ merge", "VCVTPD2DQ", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f1ff4ae6d9"},
|
||||
{"VCVTTPD2DQ.Z", "VCVTTPD2DQ.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f1fdcae6d9"},
|
||||
{"VCVTDQ2PS.Z", "VCVTDQ2PS.Z", []Operand{vreg(t, "Z1"), vreg(t, "K4"), vreg(t, "Z2")}, "62f17ccc5bd1"},
|
||||
{"VCVTDQ2PD merge", "VCVTDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "K2"), vreg(t, "X3")}, "62f17e0ae6d9"},
|
||||
{"VCVTDQ2PD.Z", "VCVTDQ2PD.Z", []Operand{vreg(t, "Y1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f17ecae6d1"},
|
||||
{"VCVTPS2PD.Z", "VCVTPS2PD.Z", []Operand{vreg(t, "Y1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f17ccb5ad1"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
want := c.mnem
|
||||
if i := len(want) - 2; i > 0 && want[i:] == ".Z" {
|
||||
want = want[:i]
|
||||
}
|
||||
if inst.Op.String() != want {
|
||||
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
|
||||
}
|
||||
}
|
||||
|
||||
// Error cases.
|
||||
bad := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"zeroing without mask", "VPADDD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"K0 mask", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K0"), vreg(t, "Z3")}},
|
||||
{"two masks", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "K2"), vreg(t, "Z3")}},
|
||||
{".Z on VEX-only", "VPSHUFD.Z", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}},
|
||||
{"broadcast unsupported", "VPXORD.BCST", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"rounding unsupported", "VPXORD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"bcst with rounding", "VADDPD.BCST.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"Z not last", "VADDPD.Z.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"duplicate suffix", "VADDPD.Z.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"KMOVW.Z", "KMOVW.Z", []Operand{vreg(t, "K1"), vreg(t, "K2")}},
|
||||
}
|
||||
for _, c := range bad {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", c.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set — ternary
|
||||
// logic, lane shuffles/inserts/extracts, compares with a K destination,
|
||||
// permutes, the wider integer families, expand/compress, broadcasts,
|
||||
// rotates and word shifts, the opmask instructions, the EVEX suffixes
|
||||
// (rounding/SAE/broadcast) and the aligned/scalar moves — byte for byte
|
||||
// against the Go assembler.
|
||||
func TestEvexExtendedGroundTruth(t *testing.T) {
|
||||
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// Ternary logic and lane shuffles (NDS + imm8).
|
||||
{"VPTERNLOGD", "VPTERNLOGD", []Operand{Imm(0xE8), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4825d9e8"},
|
||||
{"VPTERNLOGQ", "VPTERNLOGQ", []Operand{Imm(0x96), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4825d996"},
|
||||
{"VSHUFI32X4", "VSHUFI32X4", []Operand{Imm(0x4E), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "62f36d2843d94e"},
|
||||
{"VSHUFF64X2", "VSHUFF64X2", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4823d901"},
|
||||
{"VPALIGNR", "VPALIGNR", []Operand{Imm(7), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d480fd907"},
|
||||
// Permutes.
|
||||
{"VPERMB", "VPERMB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d488dd9"},
|
||||
{"VPERMW", "VPERMW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed488dd9"},
|
||||
{"VPERMI2D", "VPERMI2D", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4876d9"},
|
||||
{"VPERMT2PD", "VPERMT2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed487fd9"},
|
||||
// Compare with a K destination (and an immediate predicate).
|
||||
{"VCMPPD", "VCMPPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3")}, "62f1ed48c2d904"},
|
||||
{"VCMPPS", "VCMPPS", []Operand{Imm(0), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "K4")}, "62f16c28c2e100"},
|
||||
{"VCMPSD", "VCMPSD", []Operand{Imm(17), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K5")}, "62f1ef08c2e911"},
|
||||
// Rounding / SAE / broadcast suffixes.
|
||||
{"VADDPD.RN_SAE", "VADDPD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed1858d9"},
|
||||
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
|
||||
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
|
||||
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
|
||||
// Packed single arithmetic (same opcodes, no mandatory prefix) —
|
||||
// ZMM, YMM and XMM widths, rounding and broadcast.
|
||||
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
|
||||
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
|
||||
{"VMAXPS", "VMAXPS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e85fd9"},
|
||||
{"VDIVPS.RD_SAE", "VDIVPS.RD_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c385ed9"},
|
||||
{"VADDPS.BCST", "VADDPS.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f174585810"},
|
||||
// Compress / expand.
|
||||
{"VCOMPRESSPD", "VCOMPRESSPD", []Operand{vreg(t, "Z1"), mem64(DI)}, "62f2fd488a0f"},
|
||||
{"VEXPANDPS", "VEXPANDPS", []Operand{mem64(SI), vreg(t, "Y2")}, "62f27d288816"},
|
||||
{"VPCOMPRESSD.Z", "VPCOMPRESSD.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), mem64(DI)}, "62f27dca8b0f"},
|
||||
// Broadcasts.
|
||||
{"VPBROADCASTB gpr", "VPBROADCASTB", []Operand{BX, vreg(t, "Z1")}, "62f27d487acb"},
|
||||
{"VPBROADCASTW mem", "VPBROADCASTW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d487910"},
|
||||
{"VBROADCASTSS", "VBROADCASTSS", []Operand{mem64(AX), vreg(t, "Y3")}, "c4e27d1818"},
|
||||
{"VBROADCASTSD", "VBROADCASTSD", []Operand{mem64(AX), vreg(t, "Z4")}, "62f2fd481920"},
|
||||
// Wider integer families.
|
||||
{"VPMADDWD", "VPMADDWD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48f5d9"},
|
||||
{"VPMADDUBSW", "VPMADDUBSW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4804d9"},
|
||||
{"VPMULHUW", "VPMULHUW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48e4d9"},
|
||||
{"VPSLLVW", "VPSLLVW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4812d9"},
|
||||
{"VPACKSSWB", "VPACKSSWB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d4863d9"},
|
||||
{"VPACKUSDW", "VPACKUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482bd9"},
|
||||
// Absolute values and replicating moves.
|
||||
{"VPABSD", "VPABSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d481ed1"},
|
||||
{"VPABSQ mem", "VPABSQ", []Operand{mem64(AX), vreg(t, "Z2")}, "62f2fd481f10"},
|
||||
{"VMOVSLDUP", "VMOVSLDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa12d1"},
|
||||
{"VMOVSHDUP", "VMOVSHDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e4816d1"},
|
||||
// Rotates and word/qword shifts.
|
||||
{"VPROLD", "VPROLD", []Operand{Imm(5), vreg(t, "Z1"), vreg(t, "Z2")}, "62f16d4872c905"},
|
||||
{"VPRORQ", "VPRORQ", []Operand{Imm(63), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4872c13f"},
|
||||
{"VPSLLW", "VPSLLW", []Operand{Imm(9), vreg(t, "X1"), vreg(t, "X2")}, "c5e971f109"},
|
||||
{"VPSRLQ", "VPSRLQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4873d103"},
|
||||
// Opmask instructions (VEX-encoded, the width in the L/W/pp bits).
|
||||
{"KANDW", "KANDW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec41d9"},
|
||||
{"KORD", "KORD", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d545f4"},
|
||||
{"KXNORQ", "KXNORQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec46d9"},
|
||||
{"KNOTB", "KNOTB", []Operand{vreg(t, "K4"), vreg(t, "K5")}, "c5f944ec"},
|
||||
{"KUNPCKBW", "KUNPCKBW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ed4bd9"},
|
||||
{"KSHIFTLW", "KSHIFTLW", []Operand{Imm(2), vreg(t, "K1"), vreg(t, "K2")}, "c4e3f932d102"},
|
||||
{"KADDQ", "KADDQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec4ad9"},
|
||||
{"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"},
|
||||
{"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"},
|
||||
{"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"},
|
||||
// Completed opmask families (ANDN, NOT, OR/XOR word+qword, TEST,
|
||||
// word-width shifts; byte-exact against go tool asm).
|
||||
{"KANDNW", "KANDNW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec42d9"},
|
||||
{"KANDNB", "KANDNB", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c5d542f4"},
|
||||
{"KANDND", "KANDND", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ed42d9"},
|
||||
{"KANDNQ", "KANDNQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d442f4"},
|
||||
{"KANDD", "KANDD", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ed41d9"},
|
||||
{"KADDD", "KADDD", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d54af4"},
|
||||
{"KNOTW", "KNOTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f844d1"},
|
||||
{"KNOTD", "KNOTD", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c4e1f944e3"},
|
||||
{"KNOTQ", "KNOTQ", []Operand{vreg(t, "K5"), vreg(t, "K6")}, "c4e1f844f5"},
|
||||
{"KORW", "KORW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec45d9"},
|
||||
{"KORQ", "KORQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d445f4"},
|
||||
{"KXNORB", "KXNORB", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ed46d9"},
|
||||
{"KXORW", "KXORW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec47d9"},
|
||||
{"KXORQ", "KXORQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d447f4"},
|
||||
{"KORTESTW", "KORTESTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f898d1"},
|
||||
{"KORTESTB", "KORTESTB", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c5f998e3"},
|
||||
{"KORTESTQ", "KORTESTQ", []Operand{vreg(t, "K5"), vreg(t, "K6")}, "c4e1f898f5"},
|
||||
{"KTESTW", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f899d1"},
|
||||
{"KTESTD", "KTESTD", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c4e1f999e3"},
|
||||
{"KSHIFTLB", "KSHIFTLB", []Operand{Imm(1), vreg(t, "K1"), vreg(t, "K2")}, "c4e37932d101"},
|
||||
{"KSHIFTLD", "KSHIFTLD", []Operand{Imm(2), vreg(t, "K3"), vreg(t, "K4")}, "c4e37933e302"},
|
||||
{"KSHIFTLQ", "KSHIFTLQ", []Operand{Imm(3), vreg(t, "K5"), vreg(t, "K6")}, "c4e3f933f503"},
|
||||
{"KSHIFTRB", "KSHIFTRB", []Operand{Imm(4), vreg(t, "K1"), vreg(t, "K2")}, "c4e37930d104"},
|
||||
{"KSHIFTRW", "KSHIFTRW", []Operand{Imm(5), vreg(t, "K3"), vreg(t, "K4")}, "c4e3f930e305"},
|
||||
{"KSHIFTRQ", "KSHIFTRQ", []Operand{Imm(6), vreg(t, "K5"), vreg(t, "K6")}, "c4e3f931f506"},
|
||||
// Integer compares with an opmask destination (0F3A map, the
|
||||
// go-bzip2 partition kernel's classify instructions).
|
||||
{"VPCMPUB", "VPCMPUB", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X0"), vreg(t, "K1")}, "62f37d083ec901"},
|
||||
{"VPCMPB", "VPCMPB", []Operand{Imm(2), vreg(t, "Y2"), vreg(t, "Y3"), vreg(t, "K2")}, "62f365283fd202"},
|
||||
{"VPCMPUW", "VPCMPUW", []Operand{Imm(5), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3")}, "62f3ed483ed905"},
|
||||
{"VPCMPW", "VPCMPW", []Operand{Imm(6), vreg(t, "X3"), vreg(t, "X4"), vreg(t, "K4")}, "62f3dd083fe306"},
|
||||
{"VPCMPD", "VPCMPD", []Operand{Imm(0), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "K1")}, "62f36d281fc900"},
|
||||
{"VPCMPUD", "VPCMPUD", []Operand{Imm(1), vreg(t, "Z2"), vreg(t, "Z3"), vreg(t, "K2")}, "62f365481ed201"},
|
||||
{"VPCMPQ", "VPCMPQ", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K3")}, "62f3ed081fd902"},
|
||||
{"VPCMPUQ", "VPCMPUQ", []Operand{Imm(3), vreg(t, "Y3"), vreg(t, "Y4"), vreg(t, "K4")}, "62f3dd281ee303"},
|
||||
// Lane extract / insert.
|
||||
{"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"},
|
||||
{"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"},
|
||||
{"VINSERTF32X8", "VINSERTF32X8", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d481ad901"},
|
||||
{"VINSERTI64X4", "VINSERTI64X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed483ad901"},
|
||||
// Aligned moves and the scalar single move.
|
||||
{"VMOVAPS", "VMOVAPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4829ca"},
|
||||
{"VMOVDQA64 mem", "VMOVDQA64", []Operand{mem64(AX), vreg(t, "Z2")}, "62f1fd486f10"},
|
||||
{"VMOVSS mem", "VMOVSS", []Operand{mem64(AX), vreg(t, "X2")}, "c5fa1010"},
|
||||
// Conversions and extending/narrowing moves.
|
||||
{"VCVTPS2DQ", "VCVTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17d485bd1"},
|
||||
{"VCVTTPS2DQ", "VCVTTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e485bd1"},
|
||||
{"VPMOVZXBW", "VPMOVZXBW", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d30d1"},
|
||||
{"VPMOVSXBW mem", "VPMOVSXBW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d482010"},
|
||||
{"VPMOVWB", "VPMOVWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4830ca"},
|
||||
{"VPMOVQB", "VPMOVQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4832ca"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
want := c.mnem
|
||||
if i := strings.IndexByte(want, '.'); i > 0 {
|
||||
want = want[:i]
|
||||
}
|
||||
if inst.Op.String() != want {
|
||||
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
|
||||
// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef,
|
||||
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions —
|
||||
// plus gather/scatter with VSIB addressing, byte for byte against the Go
|
||||
// assembler.
|
||||
func TestEvexHelperGroundTruth(t *testing.T) {
|
||||
vsib := func(base, idx string, scale int) Operand {
|
||||
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
|
||||
}
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// Reciprocals and rsqrt (packed RM, scalar NDS).
|
||||
{"VRCP14PD", "VRCP14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484cd1"},
|
||||
{"VRCP14PS", "VRCP14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484cd1"},
|
||||
{"VRCP14SD", "VRCP14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084dd9"},
|
||||
{"VRCP14SS", "VRCP14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084dd9"},
|
||||
{"VRSQRT14PD", "VRSQRT14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484ed1"},
|
||||
{"VRSQRT14PS", "VRSQRT14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484ed1"},
|
||||
{"VRSQRT14SD", "VRSQRT14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084fd9"},
|
||||
{"VRSQRT14SS", "VRSQRT14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084fd9"},
|
||||
// Getexp (packed RM, scalar NDS).
|
||||
{"VGETEXPPD", "VGETEXPPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd4842d1"},
|
||||
{"VGETEXPPS", "VGETEXPPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d4842d1"},
|
||||
{"VGETEXPSD", "VGETEXPSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed0843d9"},
|
||||
{"VGETEXPSS", "VGETEXPSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d0843d9"},
|
||||
// Scalef (NDS).
|
||||
{"VSCALEFPD", "VSCALEFPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed482cd9"},
|
||||
{"VSCALEFPS", "VSCALEFPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482cd9"},
|
||||
{"VSCALEFSD", "VSCALEFSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed082dd9"},
|
||||
{"VSCALEFSS", "VSCALEFSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d082dd9"},
|
||||
// Rndscale / getmant / reduce (packed $imm,src,dst; scalar NDS+imm).
|
||||
{"VRNDSCALEPD", "VRNDSCALEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4809d104"},
|
||||
{"VRNDSCALEPS", "VRNDSCALEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4808d104"},
|
||||
{"VRNDSCALESD", "VRNDSCALESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed080bd904"},
|
||||
{"VRNDSCALESS", "VRNDSCALESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d080ad904"},
|
||||
{"VGETMANTPD", "VGETMANTPD", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4826d103"},
|
||||
{"VGETMANTPS", "VGETMANTPS", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4826d103"},
|
||||
{"VGETMANTSD", "VGETMANTSD", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0827d903"},
|
||||
{"VGETMANTSS", "VGETMANTSS", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0827d903"},
|
||||
{"VREDUCEPD", "VREDUCEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4856d104"},
|
||||
{"VREDUCEPS", "VREDUCEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4856d104"},
|
||||
{"VREDUCESD", "VREDUCESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0857d904"},
|
||||
{"VREDUCESS", "VREDUCESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0857d904"},
|
||||
// Fixupimm / range (NDS + imm8).
|
||||
{"VFIXUPIMMPD", "VFIXUPIMMPD", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4854d902"},
|
||||
{"VFIXUPIMMPS", "VFIXUPIMMPS", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4854d902"},
|
||||
{"VFIXUPIMMSD", "VFIXUPIMMSD", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0855d902"},
|
||||
{"VFIXUPIMMSS", "VFIXUPIMMSS", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0855d902"},
|
||||
{"VRANGEPD", "VRANGEPD", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4850d901"},
|
||||
{"VRANGEPS", "VRANGEPS", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4850d901"},
|
||||
{"VRANGESD", "VRANGESD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0851d901"},
|
||||
{"VRANGESS", "VRANGESS", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0851d901"},
|
||||
// FP class test ($imm, src, kdst; packed forms carry the length in
|
||||
// the X/Y/Z mnemonic suffix the decoder drops).
|
||||
{"VFPCLASSPDZ", "VFPCLASSPDZ", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "K2")}, "62f3fd4866d104"},
|
||||
{"VFPCLASSPSY", "VFPCLASSPSY", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "K2")}, "62f37d2866d104"},
|
||||
{"VFPCLASSSD", "VFPCLASSSD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f3fd0867d104"},
|
||||
{"VFPCLASSSS", "VFPCLASSSS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f37d0867d104"},
|
||||
// Gather: VEX spelling (mask register, VSIB, destination) and EVEX
|
||||
// spelling (VSIB, K mask, destination; L'L follows the VSIB index).
|
||||
{"VGATHERDPS vex", "VGATHERDPS", []Operand{vreg(t, "X2"), vsib("SI", "X1", 4), vreg(t, "X3")}, "c4e269921c8e"},
|
||||
{"VPGATHERDD vex", "VPGATHERDD", []Operand{vreg(t, "Y2"), vsib("SI", "Y1", 4), vreg(t, "Y3")}, "c4e26d901c8e"},
|
||||
{"VGATHERDPS evex", "VGATHERDPS", []Operand{vsib("SI", "X1", 4), vreg(t, "K2"), vreg(t, "X3")}, "62f27d0a921c8e"},
|
||||
{"VPGATHERQD evex", "VPGATHERQD", []Operand{vsib("SI", "Z1", 8), vreg(t, "K2"), vreg(t, "Y3")}, "62f27d4a911cce"},
|
||||
// Scatter (EVEX only: source, K mask, VSIB).
|
||||
{"VSCATTERDPS", "VSCATTERDPS", []Operand{vreg(t, "X3"), vreg(t, "K1"), vsib("SI", "X1", 4)}, "62f27d09a21c8e"},
|
||||
{"VSCATTERQPD", "VSCATTERQPD", []Operand{vreg(t, "Z3"), vreg(t, "K1"), vsib("SI", "Z1", 8)}, "62f2fd49a31cce"},
|
||||
// The remaining conversions.
|
||||
{"VCVTDQ2PS", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"},
|
||||
{"VCVTQQ2PS", "VCVTQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc485bd1"},
|
||||
{"VCVTPD2QQ", "VCVTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487bd1"},
|
||||
{"VCVTPS2QQ", "VCVTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487bd1"},
|
||||
{"VCVTUDQ2PD", "VCVTUDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "62f17e287ad1"},
|
||||
{"VCVTPH2PS", "VCVTPH2PS", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f27d4813d1"},
|
||||
{"VCVTPS2PH", "VCVTPS2PH", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "X2")}, "c4e37d1dca04"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
want := c.mnem
|
||||
got := inst.Op.String()
|
||||
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s", c.name, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexGprGroundTruth covers the scalar conversions between vector and
|
||||
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
|
||||
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
|
||||
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
|
||||
// for byte against the Go assembler, including memory sources and extended
|
||||
// GPRs.
|
||||
func TestEvexGprGroundTruth(t *testing.T) {
|
||||
mem := func(b Reg) Operand { return Ptr(b, 0, 8) }
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"VCVTSD2SI", "VCVTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2dc1"},
|
||||
{"VCVTSD2SIQ", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2dc1"},
|
||||
{"VCVTSS2SI", "VCVTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2dc1"},
|
||||
{"VCVTSS2SIQ", "VCVTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2dc1"},
|
||||
{"VCVTTSD2SI", "VCVTTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2cc1"},
|
||||
{"VCVTTSD2SIQ", "VCVTTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2cc1"},
|
||||
{"VCVTTSS2SI", "VCVTTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2cc1"},
|
||||
{"VCVTTSS2SIQ", "VCVTTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2cc1"},
|
||||
{"VCVTSD2USIL", "VCVTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0879c1"},
|
||||
{"VCVTSD2USIQ", "VCVTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0879c1"},
|
||||
{"VCVTSS2USIL", "VCVTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0879c1"},
|
||||
{"VCVTSS2USIQ", "VCVTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0879c1"},
|
||||
{"VCVTTSD2USIL", "VCVTTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0878c1"},
|
||||
{"VCVTTSD2USIQ", "VCVTTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0878c1"},
|
||||
{"VCVTTSS2USIL", "VCVTTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0878c1"},
|
||||
{"VCVTTSS2USIQ", "VCVTTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0878c1"},
|
||||
{"VCVTSI2SDL", "VCVTSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f32ad0"},
|
||||
{"VCVTSI2SDQ", "VCVTSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32ad0"},
|
||||
{"VCVTSI2SSL", "VCVTSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f22ad0"},
|
||||
{"VCVTSI2SSQ", "VCVTSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f22ad0"},
|
||||
{"VCVTUSI2SDL", "VCVTUSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f177087bd0"},
|
||||
{"VCVTUSI2SDQ", "VCVTUSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f7087bd0"},
|
||||
{"VCVTUSI2SSL", "VCVTUSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f176087bd0"},
|
||||
{"VCVTUSI2SSQ", "VCVTUSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f6087bd0"},
|
||||
{"VCVTSD2SI mem", "VCVTSD2SI", []Operand{mem(AX), BX}, "c5fb2d18"},
|
||||
{"VCVTSI2SDQ mem", "VCVTSI2SDQ", []Operand{mem(BX), vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32a13"},
|
||||
{"VCVTSD2SIQ hi gpr", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), vreg(t, "R9")}, "c461fb2dc9"},
|
||||
{"VCVTSI2SDQ hi gpr", "VCVTSI2SDQ", []Operand{vreg(t, "R10"), vreg(t, "X1"), vreg(t, "X2")}, "c4c1f32ad2"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
// The decoder does not distinguish the Plan 9 SIQ spelling (the
|
||||
// 64-bit GPR destination) from the base name; the W bit carries it.
|
||||
want := c.mnem
|
||||
got := inst.Op.String()
|
||||
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s", c.name, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexConversionGroundTruth covers the unsigned and truncating VCVT*
|
||||
// conversions, the remaining sign/zero-extending moves, the signed/unsigned
|
||||
// narrowing stores and the mask/vector conversions, byte for byte against
|
||||
// the Go assembler.
|
||||
func TestEvexConversionGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// Unsigned and truncating conversions.
|
||||
{"VCVTPD2PS", "VCVTPD2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fd485ad1"},
|
||||
{"VCVTPD2PSX", "VCVTPD2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f95ad1"},
|
||||
{"VCVTPD2PSY", "VCVTPD2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5fd5ad1"},
|
||||
{"VCVTPD2UDQ", "VCVTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4879d1"},
|
||||
{"VCVTPD2UDQX", "VCVTPD2UDQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc0879d1"},
|
||||
{"VCVTTPD2UDQ", "VCVTTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4878d1"},
|
||||
{"VCVTTPD2UDQY", "VCVTTPD2UDQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc2878d1"},
|
||||
{"VCVTTPD2UQQ", "VCVTTPD2UQQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd4878d1"},
|
||||
{"VCVTPS2UDQ", "VCVTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4879d1"},
|
||||
{"VCVTTPS2UDQ", "VCVTTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4878d1"},
|
||||
{"VCVTPS2UQQ", "VCVTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4879d1"},
|
||||
{"VCVTTPS2UQQ", "VCVTTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4878d1"},
|
||||
{"VCVTTPD2QQ", "VCVTTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487ad1"},
|
||||
{"VCVTTPS2QQ", "VCVTTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487ad1"},
|
||||
{"VCVTUQQ2PD", "VCVTUQQ2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487ad1"},
|
||||
{"VCVTUQQ2PS", "VCVTUQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff487ad1"},
|
||||
{"VCVTUQQ2PSX", "VCVTUQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1ff087ad1"},
|
||||
{"VCVTQQ2PSX", "VCVTQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc085bd1"},
|
||||
{"VCVTQQ2PSY", "VCVTQQ2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc285bd1"},
|
||||
// The remaining sign/zero-extending moves.
|
||||
{"VPMOVSXBD", "VPMOVSXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d21d1"},
|
||||
{"VPMOVSXBQ evex", "VPMOVSXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4822d1"},
|
||||
{"VPMOVSXWQ", "VPMOVSXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d24d1"},
|
||||
{"VPMOVSXWD", "VPMOVSXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d23d1"},
|
||||
{"VPMOVZXBD", "VPMOVZXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d31d1"},
|
||||
{"VPMOVZXBQ evex", "VPMOVZXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4832d1"},
|
||||
{"VPMOVZXWD", "VPMOVZXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d33d1"},
|
||||
{"VPMOVZXWQ", "VPMOVZXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d34d1"},
|
||||
// Signed narrowing stores.
|
||||
{"VPMOVSDB", "VPMOVSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4821ca"},
|
||||
{"VPMOVSDW", "VPMOVSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4823ca"},
|
||||
{"VPMOVSQB", "VPMOVSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4822ca"},
|
||||
{"VPMOVSQD", "VPMOVSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4825ca"},
|
||||
{"VPMOVSQW", "VPMOVSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4824ca"},
|
||||
{"VPMOVSWB", "VPMOVSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4820ca"},
|
||||
// Unsigned narrowing stores.
|
||||
{"VPMOVUSDB", "VPMOVUSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4811ca"},
|
||||
{"VPMOVUSDW", "VPMOVUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4813ca"},
|
||||
{"VPMOVUSQB", "VPMOVUSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4812ca"},
|
||||
{"VPMOVUSQD", "VPMOVUSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4815ca"},
|
||||
{"VPMOVUSQW", "VPMOVUSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4814ca"},
|
||||
{"VPMOVUSWB", "VPMOVUSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4810ca"},
|
||||
{"VPMOVDB", "VPMOVDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4831ca"},
|
||||
{"VPMOVQW", "VPMOVQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4834ca"},
|
||||
// Mask/vector conversions (the K register is an operand, not a
|
||||
// mask).
|
||||
{"VPMOVM2B", "VPMOVM2B", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0828d1"},
|
||||
{"VPMOVM2W", "VPMOVM2W", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f2fe0828d1"},
|
||||
{"VPMOVM2D", "VPMOVM2D", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0838d1"},
|
||||
{"VPMOVM2Q", "VPMOVM2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe4838d1"},
|
||||
{"VPMOVB2M", "VPMOVB2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f27e0829d1"},
|
||||
{"VPMOVW2M", "VPMOVW2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f2fe0829d1"},
|
||||
{"VPMOVD2M", "VPMOVD2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f27e4839d1"},
|
||||
{"VPMOVQ2M", "VPMOVQ2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f2fe4839d1"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
want := c.mnem
|
||||
got := inst.Op.String()
|
||||
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s", c.name, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexErrors checks the EVEX-specific error paths.
|
||||
func TestEvexErrors(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"NDS arity", "VPXORD", []Operand{vreg(t, "Z0"), vreg(t, "Z1")}},
|
||||
{"KMOVW arity", "KMOVW", []Operand{vreg(t, "K1")}},
|
||||
{"KMOVW no K", "KMOVW", []Operand{AX, CX}},
|
||||
{"VMOVUPD Z gpr", "VMOVUPD", []Operand{AX, vreg(t, "Z1")}},
|
||||
{"broadcast src", "VPBROADCASTD", []Operand{Imm(1), vreg(t, "Z1")}},
|
||||
{"VPMOVDW src", "VPMOVDW", []Operand{AX, vreg(t, "Y0")}},
|
||||
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
|
||||
// VEX-only mnemonics reject registers only EVEX can encode.
|
||||
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", c.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// hexCompact renders bytes as a lowercase hex string without separators.
|
||||
func hexCompact(b []byte) string {
|
||||
const hexdig = "0123456789abcdef"
|
||||
out := make([]byte, len(b)*2)
|
||||
for i, c := range b {
|
||||
out[i*2] = hexdig[c>>4]
|
||||
out[i*2+1] = hexdig[c&0xf]
|
||||
}
|
||||
return string(out)
|
||||
}
|
||||
+623
@@ -0,0 +1,623 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
)
|
||||
|
||||
// This file emits GOOBJ — the Go toolchain's object format, which cmd/link
|
||||
// consumes directly — so gasm-assembled functions drop into a go build
|
||||
// without the Go assembler. The layout follows cmd/internal/goobj: a
|
||||
// toolchain preamble ("go object ...\n!\n"), the go120ld header with its
|
||||
// block offsets, a string table, symbol definitions, the relocation /
|
||||
// aux / data index arrays, and the three blocks themselves.
|
||||
//
|
||||
// The object carries what the linker requires of an assembly object: the
|
||||
// functions (non-package symbols, as cmd/asm emits them), the GLOBL data,
|
||||
// one FuncInfo per function, the per-function DWARF symbols (the
|
||||
// .debug_line program and the subprogram DIE, which the linker's DWARF
|
||||
// pass reads verbatim), and the pc-value tables (pcsp, pcfile, pcline,
|
||||
// pcinline). The implicit funcdata symbols are omitted; the linker fills
|
||||
// their defaults.
|
||||
//
|
||||
// emitGOObject is architecture-agnostic; the per-architecture GOObject*
|
||||
// methods supply the toolchain preamble, the MinLC (pc-value delta unit)
|
||||
// and the relocation-type mapping for code relocations.
|
||||
|
||||
// GOOBJ block indices (cmd/internal/goobj). These MUST match the real
|
||||
// archive layout: the emitter writes the header offsets per index and the
|
||||
// reader (groundtruth, goobj_resolve) parses real Go archives with them.
|
||||
// blkAutolib is unused by the emitter but still defines index 0.
|
||||
const (
|
||||
blkAutolib = iota
|
||||
blkPkgIdx
|
||||
blkFile
|
||||
blkSymdef
|
||||
blkHashed64def
|
||||
blkHasheddef
|
||||
blkNonpkgdef
|
||||
blkNonpkgref
|
||||
blkRefFlags
|
||||
blkHash64
|
||||
blkHash
|
||||
blkRelocIdx
|
||||
blkAuxIdx
|
||||
blkDataIdx
|
||||
blkReloc
|
||||
blkAux
|
||||
blkData
|
||||
blkRefName
|
||||
blkEnd
|
||||
)
|
||||
|
||||
// Symbol kinds used by assembly objects (cmd/internal/objabi).
|
||||
const (
|
||||
kindSTEXT = 1
|
||||
kindSRODATA = 3
|
||||
kindSDATA = 7
|
||||
kindSDWARFFCN = 14
|
||||
kindSDWARFLINES = 20
|
||||
)
|
||||
|
||||
// Symbol flags (cmd/internal/goobj).
|
||||
const (
|
||||
symFlagDupok = 0x01
|
||||
symFlagNoSplit = 0x10
|
||||
symFlag2Link = 0x10 // asm objects flag every named symbol as linkname
|
||||
symABIStatic = 0xffff
|
||||
)
|
||||
|
||||
// Aux entry types (cmd/internal/goobj).
|
||||
const (
|
||||
auxFuncInfo = 1
|
||||
auxDwarfInfo = 3
|
||||
auxDwarfLines = 6
|
||||
auxPcsp = 7
|
||||
auxPcfile = 8
|
||||
auxPcline = 9
|
||||
auxPcinline = 10
|
||||
)
|
||||
|
||||
// FuncInfo flags (internal/abi).
|
||||
const (
|
||||
funcFlagSPWrite = 2
|
||||
funcFlagAsm = 4
|
||||
)
|
||||
|
||||
// Relocation types (cmd/internal/objabi).
|
||||
// R_PCREL and R_ADDR are stable across Go versions.
|
||||
const (
|
||||
relocPCRel = 14 // R_PCREL
|
||||
relocAddr = 1 // R_ADDR
|
||||
)
|
||||
|
||||
// relocDWTXTADDRU4 returns the R_DWTXTADDR_U4 relocation type for the
|
||||
// installed Go toolchain. The value shifted between Go 1.26 (103) and
|
||||
// Go 1.27 (106) because new LoongArch relocations were inserted before it.
|
||||
func relocDWTXTADDRU4() uint16 {
|
||||
if isGo127OrLater() {
|
||||
return 106
|
||||
}
|
||||
return 103
|
||||
}
|
||||
|
||||
var (
|
||||
goVersionOnce sync.Once
|
||||
goVersionGT26 bool
|
||||
)
|
||||
|
||||
// isGo127OrLater reports whether the installed Go toolchain is 1.27 or later.
|
||||
func isGo127OrLater() bool {
|
||||
goVersionOnce.Do(func() {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
out, err := exec.Command(goBin, "version").Output()
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
// "go version go1.27rc1 linux/amd64"
|
||||
s := string(out)
|
||||
for _, prefix := range []string{"go version go1.27", "go version go1.28", "go version go1.29", "go version go2."} {
|
||||
if strings.Contains(s, prefix) {
|
||||
goVersionGT26 = true
|
||||
return
|
||||
}
|
||||
}
|
||||
})
|
||||
return goVersionGT26
|
||||
}
|
||||
|
||||
// Special package indices for symbol references.
|
||||
const (
|
||||
pkgIdxNone = 0x7fffffff
|
||||
pkgIdxSelf = 0x7ffffffb
|
||||
)
|
||||
|
||||
const goobjMagic = "\x00go120ld"
|
||||
|
||||
// goSym is one symbol definition under construction.
|
||||
type goSym struct {
|
||||
name string
|
||||
abi uint16
|
||||
typ uint8
|
||||
flag uint8
|
||||
flag2 uint8
|
||||
size uint32
|
||||
align uint32
|
||||
}
|
||||
|
||||
func (s goSym) append(b []byte, strOff map[string]uint32) []byte {
|
||||
b = binary.LittleEndian.AppendUint32(b, uint32(len(s.name)))
|
||||
b = binary.LittleEndian.AppendUint32(b, strOff[s.name])
|
||||
b = binary.LittleEndian.AppendUint16(b, s.abi)
|
||||
b = append(b, s.typ, s.flag, s.flag2)
|
||||
b = binary.LittleEndian.AppendUint32(b, s.size)
|
||||
return binary.LittleEndian.AppendUint32(b, s.align)
|
||||
}
|
||||
|
||||
// dwarfRelocSet attaches emitter-generated relocations (the DWARF
|
||||
// lines/info symbols' address references) to a definition index.
|
||||
type dwarfRelocSet struct {
|
||||
si int
|
||||
relocs []goobjReloc
|
||||
}
|
||||
|
||||
// GOObject returns the image as a GOOBJ object file for the given package
|
||||
// path (the linker qualifies the exported symbols with it, the way cmd/asm
|
||||
// does with its -p flag). srcPath names the source file recorded in the
|
||||
// object's file table and line tables. The toolchain's object preamble is
|
||||
// captured from the installed go tool asm, so the output links with the
|
||||
// toolchain it was produced on — exactly like a real assembly object.
|
||||
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
|
||||
pre, err := toolchainObjectPreamble()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// amd64: MinLC 1, R_PCREL for the code relocations.
|
||||
return img.emitGOObject(pkgPath, srcPath, pre, 1, func(Reloc) (uint16, uint8) { return relocPCRel, 4 })
|
||||
}
|
||||
|
||||
// emitGOObject assembles the GOOBJ payload for any architecture. pre is
|
||||
// the toolchain's object preamble; minLC is the architecture's minimum
|
||||
// instruction length, the unit of the pc-value table deltas; relocField
|
||||
// maps a code relocation to its objabi relocation type and the width of
|
||||
// the instruction field the linker writes.
|
||||
func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, relocField func(Reloc) (uint16, uint8)) ([]byte, error) {
|
||||
if pkgPath == "" {
|
||||
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
|
||||
}
|
||||
|
||||
// The non-package definitions first — the DWARF symbols reference the
|
||||
// functions by these indices: per function the four pc-value tables
|
||||
// and the function itself, as cmd/asm lays them out.
|
||||
type npSym struct {
|
||||
sym goSym
|
||||
data []byte
|
||||
}
|
||||
var nps []npSym
|
||||
type pcRefs struct{ sp, file, line, inl int }
|
||||
pcIdx := make([]pcRefs, len(img.Funcs))
|
||||
fnNpIdx := make([]int, len(img.Funcs))
|
||||
for i, fn := range img.Funcs {
|
||||
tables := []struct {
|
||||
data []byte
|
||||
dst *int
|
||||
}{
|
||||
{pcspTable(fn, minLC), &pcIdx[i].sp},
|
||||
{pcValueFlat(0, fn.Size, minLC), &pcIdx[i].file},
|
||||
{pcValueFlat(int32(fn.Line), fn.Size, minLC), &pcIdx[i].line},
|
||||
{pcValueFlat(-1, fn.Size, minLC), &pcIdx[i].inl},
|
||||
}
|
||||
for _, t := range tables {
|
||||
*t.dst = len(nps)
|
||||
nps = append(nps, npSym{
|
||||
sym: goSym{typ: kindSRODATA, size: uint32(len(t.data)), align: 1},
|
||||
data: t.data,
|
||||
})
|
||||
}
|
||||
name := fn.Name
|
||||
abi := uint16(0)
|
||||
if fn.Static {
|
||||
abi = symABIStatic
|
||||
} else {
|
||||
name = pkgPath + "." + name
|
||||
}
|
||||
flag := uint8(0)
|
||||
if fn.NoSplit {
|
||||
flag |= symFlagNoSplit
|
||||
}
|
||||
fnNpIdx[i] = len(nps)
|
||||
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
|
||||
for _, r := range fn.Relocs {
|
||||
// Only the amd64 encoder resolves file-local static symbols
|
||||
// into a disp32 field at assemble time; GOOBJ must leave that
|
||||
// field zero for the linker to fill. The RISC-V and LoongArch
|
||||
// encoders emit zero immediates with a relocation instead, and
|
||||
// their relocations cover whole AUIPC/pcalau12i pairs, so
|
||||
// zeroing r.Off would erase the opcode/register bits the linker
|
||||
// preserves when it patches only the immediate.
|
||||
if r.Kind != RelPCRel32 {
|
||||
continue
|
||||
}
|
||||
if r.Off >= 0 && r.Off+4 <= len(code) {
|
||||
code[r.Off], code[r.Off+1], code[r.Off+2], code[r.Off+3] = 0, 0, 0, 0
|
||||
}
|
||||
}
|
||||
nps = append(nps, npSym{
|
||||
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
|
||||
data: code,
|
||||
})
|
||||
}
|
||||
|
||||
// The package definitions: the GLOBL symbols, then, per function, the
|
||||
// FuncInfo and the two DWARF symbols (the .debug_line program and the
|
||||
// subprogram DIE). defIdx maps a GLOBL's bare name to its definition
|
||||
// index for the code relocations.
|
||||
var defs []goSym
|
||||
var defData [][]byte
|
||||
defIdx := map[string]int{}
|
||||
for _, d := range img.DataSyms {
|
||||
name := d.Name
|
||||
if !d.Static {
|
||||
name = pkgPath + "." + name
|
||||
}
|
||||
typ := uint8(kindSDATA)
|
||||
if d.Rodata {
|
||||
typ = kindSRODATA
|
||||
}
|
||||
flag := uint8(0)
|
||||
if d.Dupok {
|
||||
flag = symFlagDupok
|
||||
}
|
||||
abi := uint16(0)
|
||||
if d.Static {
|
||||
abi = symABIStatic
|
||||
}
|
||||
defIdx[d.Name] = len(defs)
|
||||
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)})
|
||||
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
|
||||
}
|
||||
fnFiIdx := make([]int, len(img.Funcs))
|
||||
fnLinesIdx := make([]int, len(img.Funcs))
|
||||
fnDIEIdx := make([]int, len(img.Funcs))
|
||||
var dwarfRelocs []dwarfRelocSet
|
||||
for i, fn := range img.Funcs {
|
||||
data := marshalFuncInfo(fn)
|
||||
fnFiIdx[i] = len(defs)
|
||||
defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))})
|
||||
defData = append(defData, data)
|
||||
|
||||
name := fn.Name
|
||||
if !fn.Static {
|
||||
name = pkgPath + "." + name
|
||||
}
|
||||
|
||||
// The DWARF symbols: the .debug_line state-machine program and the
|
||||
// subprogram DIE, both referencing the function by its non-package
|
||||
// index (package definitions, like cmd/asm's).
|
||||
lines, lrel := goobjDwarfLines(fn, fnNpIdx[i])
|
||||
fnLinesIdx[i] = len(defs)
|
||||
defs = append(defs, goSym{typ: kindSDWARFLINES, size: uint32(len(lines))})
|
||||
defData = append(defData, lines)
|
||||
die, drel := goobjDwarfInfo(fn, name, fnNpIdx[i])
|
||||
fnDIEIdx[i] = len(defs)
|
||||
defs = append(defs, goSym{typ: kindSDWARFFCN, size: uint32(len(die))})
|
||||
defData = append(defData, die)
|
||||
dwarfRelocs = append(dwarfRelocs,
|
||||
dwarfRelocSet{si: fnLinesIdx[i], relocs: lrel},
|
||||
dwarfRelocSet{si: fnDIEIdx[i], relocs: drel},
|
||||
)
|
||||
}
|
||||
|
||||
// Resolve external symbol references (cross-package). Build the
|
||||
// package index table and determine each external symbol's SymIdx
|
||||
// by reading the target package's export data.
|
||||
var extPkgTable []string
|
||||
var extPkgIdx map[string]int
|
||||
var extSymIdx map[string]int
|
||||
if len(img.Externals) > 0 {
|
||||
var err error
|
||||
extPkgTable, extPkgIdx, extSymIdx, err = resolveExternalSymbols(img.Externals)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("GOOBJ emission: resolving external symbols: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
// Relocations, per defined symbol in definition order (package defs,
|
||||
// then non-package defs).
|
||||
nsyms := len(defs) + len(nps)
|
||||
symRelocs := make([][]byte, nsyms) // flat 23-byte records
|
||||
for i, fn := range img.Funcs {
|
||||
si := len(defs) + fnNpIdx[i]
|
||||
for _, r := range fn.Relocs {
|
||||
typ, size := relocField(r)
|
||||
if r.External {
|
||||
// Split package-qualified name: "runtime·morestack" → runtime, morestack.
|
||||
pkg, name := splitQualified(r.Name)
|
||||
if pkg == "" {
|
||||
return nil, fmt.Errorf("GOOBJ emission: external symbol %q has no package prefix", r.Name)
|
||||
}
|
||||
pIdx, ok := extPkgIdx[pkg]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("GOOBJ emission: package %q not resolved", pkg)
|
||||
}
|
||||
sIdx, ok := extSymIdx[pkg+"·"+name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("GOOBJ emission: symbol %s·%s not resolved", pkg, name)
|
||||
}
|
||||
var rec [23]byte
|
||||
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
|
||||
rec[4] = size // field width
|
||||
binary.LittleEndian.PutUint16(rec[5:], typ)
|
||||
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
|
||||
binary.LittleEndian.PutUint32(rec[15:], uint32(pIdx))
|
||||
binary.LittleEndian.PutUint32(rec[19:], uint32(sIdx))
|
||||
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
||||
continue
|
||||
}
|
||||
di, ok := defIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
|
||||
}
|
||||
var rec [23]byte
|
||||
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
|
||||
rec[4] = size // field width
|
||||
binary.LittleEndian.PutUint16(rec[5:], typ)
|
||||
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
|
||||
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
|
||||
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
|
||||
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
||||
}
|
||||
}
|
||||
// The DWARF symbols' own relocations (the function address references).
|
||||
for _, ds := range dwarfRelocs {
|
||||
for _, r := range ds.relocs {
|
||||
var rec [23]byte
|
||||
binary.LittleEndian.PutUint32(rec[0:], uint32(r.off))
|
||||
rec[4] = r.siz
|
||||
binary.LittleEndian.PutUint16(rec[5:], r.typ)
|
||||
binary.LittleEndian.PutUint64(rec[7:], uint64(r.add))
|
||||
binary.LittleEndian.PutUint32(rec[15:], r.pkg)
|
||||
binary.LittleEndian.PutUint32(rec[19:], r.sym)
|
||||
symRelocs[ds.si] = append(symRelocs[ds.si], rec[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
// Aux entries per function: FuncInfo, the DWARF symbols, then the four
|
||||
// pc tables. References into the non-package table use pkgIdxNone.
|
||||
symAux := make([][]byte, nsyms)
|
||||
for i := range img.Funcs {
|
||||
si := len(defs) + fnNpIdx[i]
|
||||
aux := func(typ uint8, pkg, idx uint32) {
|
||||
var rec [9]byte
|
||||
rec[0] = typ
|
||||
binary.LittleEndian.PutUint32(rec[1:], pkg)
|
||||
binary.LittleEndian.PutUint32(rec[5:], idx)
|
||||
symAux[si] = append(symAux[si], rec[:]...)
|
||||
}
|
||||
aux(auxFuncInfo, pkgIdxSelf, uint32(fnFiIdx[i]))
|
||||
aux(auxDwarfInfo, pkgIdxSelf, uint32(fnDIEIdx[i]))
|
||||
aux(auxDwarfLines, pkgIdxSelf, uint32(fnLinesIdx[i]))
|
||||
// The pc-table references are 0-based within the non-package
|
||||
// definitions; the loader adds the package-definition count itself.
|
||||
aux(auxPcsp, pkgIdxNone, uint32(pcIdx[i].sp))
|
||||
aux(auxPcfile, pkgIdxNone, uint32(pcIdx[i].file))
|
||||
aux(auxPcline, pkgIdxNone, uint32(pcIdx[i].line))
|
||||
aux(auxPcinline, pkgIdxNone, uint32(pcIdx[i].inl))
|
||||
}
|
||||
|
||||
// The string table. Absolute offsets: it starts right after the
|
||||
// 96-byte header (magic, fingerprint, flags, the 19 block offsets).
|
||||
const headerSize = 8 + 8 + 4 + 4*(blkEnd+1)
|
||||
var strTab []byte
|
||||
strOff := map[string]uint32{}
|
||||
addStr := func(s string) {
|
||||
if _, ok := strOff[s]; ok {
|
||||
return
|
||||
}
|
||||
strOff[s] = uint32(headerSize + len(strTab))
|
||||
strTab = append(strTab, s...)
|
||||
}
|
||||
addStr("")
|
||||
addStr(srcPath)
|
||||
for _, s := range defs {
|
||||
addStr(s.name)
|
||||
}
|
||||
for _, s := range nps {
|
||||
addStr(s.sym.name)
|
||||
}
|
||||
stringRef := func(b []byte, s string) []byte {
|
||||
b = binary.LittleEndian.AppendUint32(b, uint32(len(s)))
|
||||
return binary.LittleEndian.AppendUint32(b, strOff[s])
|
||||
}
|
||||
|
||||
// Serialise the block bodies.
|
||||
var symdefBlk, npdefBlk []byte
|
||||
for _, s := range defs {
|
||||
symdefBlk = s.append(symdefBlk, strOff)
|
||||
}
|
||||
for _, s := range nps {
|
||||
npdefBlk = s.sym.append(npdefBlk, strOff)
|
||||
}
|
||||
|
||||
// Package index table: index 0 is the dummy invalid package.
|
||||
// External packages follow, in pkgIdx order.
|
||||
for _, pkg := range extPkgTable {
|
||||
addStr(pkg)
|
||||
}
|
||||
pkgIdxBlk := stringRef(nil, "") // index 0: dummy
|
||||
for _, pkg := range extPkgTable {
|
||||
pkgIdxBlk = stringRef(pkgIdxBlk, pkg)
|
||||
}
|
||||
fileBlk := stringRef(nil, srcPath)
|
||||
|
||||
var relocBlk, auxBlk, dataBlk []byte
|
||||
relocIdxBlk := make([]byte, 0, 4*(nsyms+1))
|
||||
auxIdxBlk := make([]byte, 0, 4*(nsyms+1))
|
||||
dataIdxBlk := make([]byte, 0, 4*(nsyms+1))
|
||||
var nr, na, nd uint32
|
||||
for si := range nsyms {
|
||||
relocIdxBlk = binary.LittleEndian.AppendUint32(relocIdxBlk, nr)
|
||||
auxIdxBlk = binary.LittleEndian.AppendUint32(auxIdxBlk, na)
|
||||
dataIdxBlk = binary.LittleEndian.AppendUint32(dataIdxBlk, nd)
|
||||
relocBlk = append(relocBlk, symRelocs[si]...)
|
||||
auxBlk = append(auxBlk, symAux[si]...)
|
||||
var d []byte
|
||||
if si < len(defData) {
|
||||
d = defData[si]
|
||||
} else {
|
||||
d = nps[si-len(defData)].data
|
||||
}
|
||||
dataBlk = append(dataBlk, d...)
|
||||
nr += uint32(len(symRelocs[si])) / 23
|
||||
na += uint32(len(symAux[si])) / 9
|
||||
nd += uint32(len(d))
|
||||
}
|
||||
relocIdxBlk = binary.LittleEndian.AppendUint32(relocIdxBlk, nr)
|
||||
auxIdxBlk = binary.LittleEndian.AppendUint32(auxIdxBlk, na)
|
||||
dataIdxBlk = binary.LittleEndian.AppendUint32(dataIdxBlk, nd)
|
||||
|
||||
blocks := [blkEnd][]byte{
|
||||
blkPkgIdx: pkgIdxBlk,
|
||||
blkFile: fileBlk,
|
||||
blkSymdef: symdefBlk,
|
||||
blkNonpkgdef: npdefBlk,
|
||||
blkRelocIdx: relocIdxBlk,
|
||||
blkAuxIdx: auxIdxBlk,
|
||||
blkDataIdx: dataIdxBlk,
|
||||
blkReloc: relocBlk,
|
||||
blkAux: auxBlk,
|
||||
blkData: dataBlk,
|
||||
}
|
||||
|
||||
// Assemble the payload: header (offsets filled once known), string
|
||||
// table, blocks in order.
|
||||
payload := make([]byte, headerSize)
|
||||
copy(payload, goobjMagic)
|
||||
// The fingerprint stays zero, as cmd/asm leaves it.
|
||||
binary.LittleEndian.PutUint32(payload[16:], 4) // ObjFlagFromAssembly
|
||||
off := uint32(headerSize + len(strTab))
|
||||
for i := range blkEnd {
|
||||
binary.LittleEndian.PutUint32(payload[20+4*i:], off)
|
||||
off += uint32(len(blocks[i]))
|
||||
}
|
||||
binary.LittleEndian.PutUint32(payload[20+4*blkEnd:], off)
|
||||
payload = append(payload, strTab...)
|
||||
for _, blk := range blocks {
|
||||
payload = append(payload, blk...)
|
||||
}
|
||||
|
||||
out := make([]byte, 0, len(pre)+len(payload))
|
||||
out = append(out, pre...)
|
||||
return append(out, payload...), nil
|
||||
}
|
||||
|
||||
// marshalFuncInfo serialises a function's goobj.FuncInfo: sizes, flags,
|
||||
// start line, the one-element file table and an empty inline tree.
|
||||
func marshalFuncInfo(fn FuncLayout) []byte {
|
||||
flag := uint8(funcFlagAsm)
|
||||
if fn.SPWrite {
|
||||
flag |= funcFlagSPWrite
|
||||
}
|
||||
b := make([]byte, 0, 28)
|
||||
b = binary.LittleEndian.AppendUint32(b, uint32(fn.Args))
|
||||
b = binary.LittleEndian.AppendUint32(b, uint32(fn.Frame))
|
||||
b = append(b, 0, flag, 0, 0) // FuncID normal, flags, padding
|
||||
b = binary.LittleEndian.AppendUint32(b, uint32(int32(fn.Line)))
|
||||
b = binary.LittleEndian.AppendUint32(b, 1) // one file
|
||||
b = binary.LittleEndian.AppendUint32(b, 0) // file index 0
|
||||
b = binary.LittleEndian.AppendUint32(b, 0) // no inline tree
|
||||
return b
|
||||
}
|
||||
|
||||
// pcValueFlat encodes a pc-value table holding v over the whole function.
|
||||
// The pc deltas are in MinLC units (the runtime scales them by the
|
||||
// architecture's minimum instruction length).
|
||||
func pcValueFlat(v int32, size, minLC int) []byte {
|
||||
// The table is delta-encoded from an implicit value of -1: a varint
|
||||
// value delta, an unsigned pc delta to the end, and a zero terminator.
|
||||
out := binary.AppendVarint(nil, int64(v)+1)
|
||||
out = binary.AppendUvarint(out, uint64(size/minLC))
|
||||
return append(out, 0)
|
||||
}
|
||||
|
||||
// pcspTable encodes the stack-adjustment table: the SP delta in effect at
|
||||
// every pc, from the function's prologue and epilogue boundaries.
|
||||
func pcspTable(fn FuncLayout, minLC int) []byte {
|
||||
if len(fn.Spadj) == 0 {
|
||||
return pcValueFlat(0, fn.Size, minLC)
|
||||
}
|
||||
pts := make([]SpadjStep, 0, len(fn.Spadj)+1)
|
||||
pts = append(pts, SpadjStep{PC: 0, Value: 0})
|
||||
pts = append(pts, fn.Spadj...)
|
||||
out := binary.AppendVarint(nil, int64(pts[0].Value)+1)
|
||||
cur, old := pts[0].PC, pts[0].Value
|
||||
for _, p := range pts[1:] {
|
||||
out = binary.AppendUvarint(out, uint64((p.PC-cur)/minLC))
|
||||
out = binary.AppendVarint(out, int64(p.Value-old))
|
||||
cur, old = p.PC, p.Value
|
||||
}
|
||||
out = binary.AppendUvarint(out, uint64((fn.Size-cur)/minLC))
|
||||
return append(out, 0)
|
||||
}
|
||||
|
||||
// toolchainObjectPreamble returns the "go object ...\n!\n" header the
|
||||
// installed go tool asm writes, captured by assembling a one-instruction
|
||||
// probe. The linker compares this string verbatim against its own, so it
|
||||
// must come from the toolchain itself, not be reconstructed.
|
||||
var (
|
||||
preambleOnce sync.Once
|
||||
preamble []byte
|
||||
preambleErr error
|
||||
)
|
||||
|
||||
func toolchainObjectPreamble() ([]byte, error) {
|
||||
preambleOnce.Do(func() {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
preambleErr = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
|
||||
return
|
||||
}
|
||||
dir, err := os.MkdirTemp("", "gasm-preamble")
|
||||
if err != nil {
|
||||
preambleErr = err
|
||||
return
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
src := filepath.Join(dir, "probe_amd64.s")
|
||||
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
|
||||
preambleErr = err
|
||||
return
|
||||
}
|
||||
obj := filepath.Join(dir, "probe.o")
|
||||
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
|
||||
cmd.Env = append(os.Environ(), "GOARCH=amd64")
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
preambleErr = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
|
||||
return
|
||||
}
|
||||
data, err := os.ReadFile(obj)
|
||||
if err != nil {
|
||||
preambleErr = err
|
||||
return
|
||||
}
|
||||
i := bytes.Index(data, []byte("\n!\n"))
|
||||
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
|
||||
preambleErr = fmt.Errorf("unrecognised assembler object layout")
|
||||
return
|
||||
}
|
||||
preamble = data[:i+3]
|
||||
})
|
||||
return preamble, preambleErr
|
||||
}
|
||||
@@ -0,0 +1,185 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
)
|
||||
|
||||
// This file generates the per-function DWARF symbols the linker's DWARF
|
||||
// pass requires of an assembly object, byte-identical to what cmd/asm
|
||||
// emits: the .debug_line state-machine program (SDWARFLINES) and the
|
||||
// subprogram DIE (SDWARFFCN). The linker copies the DIE and line-program
|
||||
// bytes verbatim into .debug_info and .debug_line, fixing up their
|
||||
// relocations, so the formats here must match cmd/internal/dwarf's
|
||||
// DW_ABRV_FUNCTION and generateDebugLinesSymbol exactly.
|
||||
//
|
||||
// DWARF5 is assumed throughout (the toolchain's default on Linux and the
|
||||
// other non-Darwin targets gasm supports).
|
||||
|
||||
// Line-program parameters (cmd/internal/obj/dwarf.go).
|
||||
const (
|
||||
dwLineBase = -4
|
||||
dwLineRange = 10
|
||||
dwOpcodeBase = 11
|
||||
dwPCRange = (255 - dwOpcodeBase) / dwLineRange
|
||||
)
|
||||
|
||||
// goobjReloc is one relocation attached to an emitter-generated symbol
|
||||
// (the DWARF lines/info symbols), in goobj's on-disk encoding fields.
|
||||
type goobjReloc struct {
|
||||
off int32
|
||||
siz uint8
|
||||
typ uint16
|
||||
add int64
|
||||
pkg uint32
|
||||
sym uint32
|
||||
}
|
||||
|
||||
// goobjDwarfLines builds the function's .debug_line state-machine program:
|
||||
// an LNE_set_address extended opcode establishing the function's start
|
||||
// address (carrying the R_ADDR relocation), one row per source line
|
||||
// change across the function's instructions, an advance to the end of the
|
||||
// function and an end-of-sequence opcode. The linker appends these bytes
|
||||
// after the unit's line header, so they must start with the address and
|
||||
// leave the state machine terminated.
|
||||
func goobjDwarfLines(fn FuncLayout, fnNpIdx int) ([]byte, []goobjReloc) {
|
||||
// Rows: the prologue, if any, then the body instructions (fn.Lines
|
||||
// covers the body only). The first body offset > 0 means a prologue
|
||||
// precedes it; the toolchain reports the prologue on the TEXT line.
|
||||
pts := make([]LineEntry, 0, len(fn.Lines)+1)
|
||||
if len(fn.Lines) == 0 || fn.Lines[0].Offset > 0 {
|
||||
pts = append(pts, LineEntry{Offset: 0, Line: fn.Line})
|
||||
}
|
||||
pts = append(pts, fn.Lines...)
|
||||
|
||||
out := []byte{0, 9, 2, 0, 0, 0, 0, 0, 0, 0, 0} // LNE_set_address, address zeroed
|
||||
relocs := []goobjReloc{{
|
||||
off: 3, siz: 8, typ: relocAddr,
|
||||
pkg: pkgIdxNone, sym: uint32(fnNpIdx),
|
||||
}}
|
||||
|
||||
// The state machine starts at line 1, pc 0 (function-relative); the
|
||||
// implicit initial pc is the function entry, so the first pc delta is
|
||||
// against 0.
|
||||
line := int64(1)
|
||||
pc := uint64(0)
|
||||
for _, p := range pts {
|
||||
if p.Line == 0 || uint64(p.Offset) < pc {
|
||||
continue
|
||||
}
|
||||
// Rows mark source-line changes only; the pc delta is measured from
|
||||
// the previous row, not the previous instruction.
|
||||
if int64(p.Line) == line {
|
||||
continue
|
||||
}
|
||||
deltaPC := uint64(p.Offset) - pc
|
||||
deltaLC := int64(p.Line) - line
|
||||
out = dwPutPCLCDelta(out, deltaPC, deltaLC)
|
||||
line, pc = int64(p.Line), uint64(p.Offset)
|
||||
}
|
||||
|
||||
// Cover the rest of the function and close the sequence.
|
||||
if end := uint64(fn.Size) - pc; end > 0 {
|
||||
out = append(out, 2) // DW_LNS_advance_pc
|
||||
out = binary.AppendUvarint(out, end)
|
||||
}
|
||||
out = append(out, 0, 1, 1) // LNE_end_sequence
|
||||
return out, relocs
|
||||
}
|
||||
|
||||
// dwPutPCLCDelta encodes one (pcDelta, lineDelta) step as the shortest
|
||||
// special opcode plus any standard-opcode remainder, exactly like
|
||||
// cmd/internal/obj's putpclcdelta.
|
||||
func dwPutPCLCDelta(b []byte, deltaPC uint64, deltaLC int64) []byte {
|
||||
opcode := dwSelectOpcode(deltaPC, deltaLC)
|
||||
deltaPC -= uint64((opcode - dwOpcodeBase) / dwLineRange)
|
||||
deltaLC -= (opcode-dwOpcodeBase)%dwLineRange + dwLineBase
|
||||
|
||||
// The remainder: standard opcodes first, then the special opcode
|
||||
// (which emits the row).
|
||||
if deltaPC != 0 {
|
||||
switch {
|
||||
case deltaPC <= uint64(dwPCRange):
|
||||
opcode -= dwLineRange * int64(uint64(dwPCRange)-deltaPC)
|
||||
b = append(b, 8) // DW_LNS_const_add_pc
|
||||
case (1<<14) <= deltaPC && deltaPC < (1<<16):
|
||||
b = append(b, 9) // DW_LNS_fixed_advance_pc
|
||||
b = binary.LittleEndian.AppendUint16(b, uint16(deltaPC))
|
||||
default:
|
||||
b = append(b, 2) // DW_LNS_advance_pc
|
||||
b = binary.AppendUvarint(b, deltaPC)
|
||||
}
|
||||
}
|
||||
if deltaLC != 0 {
|
||||
b = append(b, 3) // DW_LNS_advance_line
|
||||
b = binary.AppendVarint(b, deltaLC)
|
||||
}
|
||||
return append(b, byte(opcode))
|
||||
}
|
||||
|
||||
// dwSelectOpcode picks the special opcode for (deltaPC, deltaLC) per
|
||||
// cmd/internal/obj's putpclcdelta selection logic.
|
||||
func dwSelectOpcode(deltaPC uint64, deltaLC int64) int64 {
|
||||
switch {
|
||||
case deltaLC < dwLineBase:
|
||||
if deltaPC >= uint64(dwPCRange) {
|
||||
return dwOpcodeBase + dwLineRange*dwPCRange
|
||||
}
|
||||
return dwOpcodeBase + dwLineRange*int64(deltaPC)
|
||||
case deltaLC < dwLineBase+dwLineRange:
|
||||
if deltaPC >= uint64(dwPCRange) {
|
||||
op := int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*dwPCRange
|
||||
if op > 255 {
|
||||
op -= dwLineRange
|
||||
}
|
||||
return op
|
||||
}
|
||||
return int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*int64(deltaPC)
|
||||
default:
|
||||
if deltaPC <= uint64(dwPCRange) {
|
||||
op := min(int64(dwOpcodeBase)+(dwLineRange-1)+dwLineRange*int64(deltaPC), 255)
|
||||
return op
|
||||
}
|
||||
switch deltaPC - uint64(dwPCRange) {
|
||||
case uint64(dwPCRange), (1 << 7) - 1, (1 << 16) - 1, (1 << 21) - 1,
|
||||
(1 << 28) - 1, (1 << 35) - 1, (1 << 42) - 1, (1 << 49) - 1,
|
||||
(1 << 56) - 1, (1 << 63) - 1:
|
||||
return 255
|
||||
default:
|
||||
// 250: the toolchain's "249" comment is stale.
|
||||
return dwOpcodeBase + dwLineRange*dwPCRange - 1
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// goobjDwarfInfo builds the function's DWARF5 subprogram DIE (abbrev
|
||||
// DW_ABRV_FUNCTION): name, low_pc as a .debug_addr index (the
|
||||
// R_DWTXTADDR_U4 relocation), high_pc as the size, the call-frame-CFA
|
||||
// frame base, the decl file/line and the external flag. name is the
|
||||
// symbol's object name (package-qualified unless static).
|
||||
func goobjDwarfInfo(fn FuncLayout, name string, fnNpIdx int) ([]byte, []goobjReloc) {
|
||||
out := []byte{3} // DW_ABRV_FUNCTION
|
||||
out = append(out, name...)
|
||||
out = append(out, 0)
|
||||
|
||||
addrx := len(out)
|
||||
out = append(out, 0, 0, 0, 0) // DW_AT_low_pc: addrx slot, zeroed
|
||||
out = binary.AppendUvarint(out, uint64(fn.Size))
|
||||
out = append(out, 1, 0x9c) // DW_AT_frame_base: block1, DW_OP_call_frame_cfa
|
||||
out = binary.LittleEndian.AppendUint32(out, 1)
|
||||
out = binary.AppendUvarint(out, uint64(fn.Line))
|
||||
if fn.Static {
|
||||
out = append(out, 0)
|
||||
} else {
|
||||
out = append(out, 1) // DW_AT_external
|
||||
}
|
||||
out = append(out, 0) // end of children
|
||||
|
||||
relocs := []goobjReloc{{
|
||||
off: int32(addrx), siz: 4, typ: relocDWTXTADDRU4(),
|
||||
pkg: pkgIdxNone, sym: uint32(fnNpIdx),
|
||||
}}
|
||||
return out, relocs
|
||||
}
|
||||
@@ -0,0 +1,214 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// TestDWSelectOpcode checks the special-opcode selection against
|
||||
// hand-computed values for the boundary cases: line deltas below, inside
|
||||
// and above the line range, and pc deltas at and beyond PC_RANGE (24).
|
||||
func TestDWSelectOpcode(t *testing.T) {
|
||||
cases := []struct {
|
||||
deltaPC uint64
|
||||
deltaLC int64
|
||||
want int64
|
||||
}{
|
||||
{0, 2, 17}, // the common single-instruction step
|
||||
{0, -4, 11}, // deltaLC == LINE_BASE
|
||||
{0, -5, 11}, // deltaLC below LINE_BASE: opcode adds nothing
|
||||
{0, 6, 20}, // deltaLC == LINE_BASE+LINE_RANGE, remainder via advance_line
|
||||
{4, 1, 56}, // the 4-byte loong64 instruction step
|
||||
{23, 1, 246}, // deltaPC == PC_RANGE-1
|
||||
{24, 1, 246}, // deltaPC == PC_RANGE: wraps past 255
|
||||
{25, 1, 246}, // deltaPC past PC_RANGE (the const_add_pc remainder adjusts it later)
|
||||
{100, 1, 246},
|
||||
{151, 10, 255}, // deltaPC-PC_RANGE == (1<<7)-1, large line delta
|
||||
{100, 10, 250}, // deltaPC-PC_RANGE not on a switch boundary
|
||||
{23, 10, 250}, // large line delta inside PC_RANGE
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := dwSelectOpcode(c.deltaPC, c.deltaLC); got != c.want {
|
||||
t.Errorf("dwSelectOpcode(%d, %d) = %d, want %d", c.deltaPC, c.deltaLC, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// decodeDWLineProgram decodes a .debug_line state-machine program (as
|
||||
// emitted by goobjDwarfLines) into (pc, line) rows.
|
||||
func decodeDWLineProgram(t *testing.T, b []byte) (pcs []uint64, lines []int64) {
|
||||
t.Helper()
|
||||
pc, line := uint64(0), int64(1)
|
||||
emit := func() {
|
||||
if len(pcs) == 0 || pcs[len(pcs)-1] != pc || lines[len(lines)-1] != line {
|
||||
pcs = append(pcs, pc)
|
||||
lines = append(lines, line)
|
||||
}
|
||||
}
|
||||
advancePC := func(delta uint64) { pc += delta }
|
||||
advanceLine := func(delta int64) { line += delta }
|
||||
for i := 0; i < len(b); {
|
||||
op := b[i]
|
||||
i++
|
||||
switch {
|
||||
case op == 0: // extended opcode
|
||||
ln, n := binary.Uvarint(b[i:])
|
||||
i += n
|
||||
sub := b[i]
|
||||
i++
|
||||
_ = ln
|
||||
switch sub {
|
||||
case 2: // DW_LNE_set_address: 8-byte address
|
||||
pc = binary.LittleEndian.Uint64(b[i:])
|
||||
i += 8
|
||||
case 1: // DW_LNE_end_sequence
|
||||
// terminates the sequence; no new row
|
||||
}
|
||||
case op == 2: // DW_LNS_advance_pc
|
||||
v, n := binary.Uvarint(b[i:])
|
||||
i += n
|
||||
advancePC(v)
|
||||
case op == 3: // DW_LNS_advance_line
|
||||
v, n := binary.Varint(b[i:])
|
||||
i += n
|
||||
advanceLine(v)
|
||||
case op == 8: // DW_LNS_const_add_pc
|
||||
advancePC(uint64(dwPCRange))
|
||||
case op == 9: // DW_LNS_fixed_advance_pc
|
||||
advancePC(uint64(binary.LittleEndian.Uint16(b[i:])))
|
||||
i += 2
|
||||
case op >= dwOpcodeBase: // special opcode
|
||||
advancePC(uint64((int64(op) - dwOpcodeBase) / dwLineRange))
|
||||
advanceLine((int64(op)-dwOpcodeBase)%dwLineRange + dwLineBase)
|
||||
emit()
|
||||
}
|
||||
}
|
||||
return pcs, lines
|
||||
}
|
||||
|
||||
// TestGoobjDwarfLinesRows checks the emitted line program's rows for
|
||||
// synthetic functions: a zero-frame function with one instruction per
|
||||
// line, a framed function (the prologue row is prepended on the TEXT
|
||||
// line), instructions sharing a line, and a function with a large pc gap
|
||||
// (the const_add_pc remainder path).
|
||||
func TestGoobjDwarfLinesRows(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
fn FuncLayout
|
||||
want [][2]int64 // (pc, line)
|
||||
}{
|
||||
{
|
||||
"one instruction per line",
|
||||
FuncLayout{Size: 20, Line: 2, Lines: []LineEntry{
|
||||
{0, 3}, {4, 4}, {8, 5}, {12, 6}, {16, 7},
|
||||
}},
|
||||
[][2]int64{{0, 3}, {4, 4}, {8, 5}, {12, 6}, {16, 7}},
|
||||
},
|
||||
{
|
||||
"framed: prologue row on the TEXT line",
|
||||
FuncLayout{Size: 24, Line: 2, Lines: []LineEntry{
|
||||
{12, 3}, {16, 4},
|
||||
}},
|
||||
[][2]int64{{0, 2}, {12, 3}, {16, 4}},
|
||||
},
|
||||
{
|
||||
"instructions sharing a line fold into one row",
|
||||
FuncLayout{Size: 16, Line: 2, Lines: []LineEntry{
|
||||
{0, 3}, {4, 3}, {8, 4}, {12, 4},
|
||||
}},
|
||||
[][2]int64{{0, 3}, {8, 4}},
|
||||
},
|
||||
{
|
||||
"large gap crosses PC_RANGE",
|
||||
FuncLayout{Size: 60, Line: 2, Lines: []LineEntry{
|
||||
{0, 3}, {40, 4},
|
||||
}},
|
||||
[][2]int64{{0, 3}, {40, 4}},
|
||||
},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
prog, relocs := goobjDwarfLines(c.fn, 0)
|
||||
if len(relocs) != 1 || relocs[0].off != 3 || relocs[0].siz != 8 || relocs[0].typ != relocAddr || relocs[0].sym != 0 {
|
||||
t.Fatalf("relocs = %+v", relocs)
|
||||
}
|
||||
pcs, lines := decodeDWLineProgram(t, prog)
|
||||
if len(pcs) != len(c.want) {
|
||||
t.Fatalf("rows = %d (%v / %v), want %d", len(pcs), pcs, lines, len(c.want))
|
||||
}
|
||||
for i, w := range c.want {
|
||||
if pcs[i] != uint64(w[0]) || lines[i] != w[1] {
|
||||
t.Errorf("row %d = (%d, %d), want (%d, %d)", i, pcs[i], lines[i], w[0], w[1])
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestGoobjDwarfInfo checks the subprogram DIE for an exported and a
|
||||
// static function: the abbrev, name, high_pc, frame base, decl file/line,
|
||||
// the external flag and the addrx relocation position.
|
||||
func TestGoobjDwarfInfo(t *testing.T) {
|
||||
fn := FuncLayout{Size: 20, Line: 2}
|
||||
die, relocs := goobjDwarfInfo(fn, "pkg.f", 3)
|
||||
want := []byte{
|
||||
0x03,
|
||||
'p', 'k', 'g', '.', 'f', 0,
|
||||
0, 0, 0, 0, // addrx slot at offset 7
|
||||
0x14, // high_pc: 20
|
||||
0x01, 0x9c, // frame_base
|
||||
0x01, 0, 0, 0, // decl_file 1
|
||||
0x02, // decl_line 2
|
||||
0x01, // external
|
||||
0x00, // end of children
|
||||
}
|
||||
if !bytes.Equal(die, want) {
|
||||
t.Errorf("DIE = %x, want %x", die, want)
|
||||
}
|
||||
if len(relocs) != 1 || relocs[0].off != 7 || relocs[0].siz != 4 || relocs[0].typ != relocDWTXTADDRU4() || relocs[0].sym != 3 {
|
||||
t.Errorf("relocs = %+v", relocs)
|
||||
}
|
||||
|
||||
// A static function carries no external flag and no package prefix.
|
||||
fn.Static = true
|
||||
die, _ = goobjDwarfInfo(fn, "f", 1)
|
||||
if die[len(die)-2] != 0 {
|
||||
t.Errorf("static external flag = %d, want 0", die[len(die)-2])
|
||||
}
|
||||
}
|
||||
|
||||
// TestDwPutPCLCDeltaRemainders checks the standard-opcode remainders:
|
||||
// const_add_pc and fixed_advance_pc after a special opcode.
|
||||
func TestDwPutPCLCDeltaRemainders(t *testing.T) {
|
||||
// deltaPC 25 past PC_RANGE: opcode 26 covers (1, 1), const_add_pc
|
||||
// covers the remaining 23 pc and 0 line.
|
||||
got := dwPutPCLCDelta(nil, 25, 1)
|
||||
if !bytes.Equal(got, []byte{8, 26}) {
|
||||
t.Errorf("25/1 = %x, want [8 1a]", got)
|
||||
}
|
||||
|
||||
// deltaPC 20000: opcode 246 covers 23, fixed_advance_pc covers the
|
||||
// remaining 19977.
|
||||
got = dwPutPCLCDelta(nil, 20000, 1)
|
||||
if got[0] != 9 || binary.LittleEndian.Uint16(got[1:]) != 19977 || got[3] != 246 {
|
||||
t.Errorf("20000/1 = %x, want fixed_advance_pc 19977 then 246", got)
|
||||
}
|
||||
|
||||
// Line remainder: deltaLC 10 leaves 5 past the opcode's reach, encoded
|
||||
// as advance_line 5 (zigzag 0x0a) before opcode 250.
|
||||
got = dwPutPCLCDelta(nil, 23, 10)
|
||||
if !bytes.Equal(got, []byte{3, 0x0a, 250}) {
|
||||
t.Errorf("23/10 = %x, want [03 0a fa]", got)
|
||||
}
|
||||
|
||||
// Negative line remainder: deltaLC -5 leaves advance_line -1 (zigzag
|
||||
// 0x01) after opcode 11.
|
||||
got = dwPutPCLCDelta(nil, 0, -5)
|
||||
if !bytes.Equal(got, []byte{3, 1, 11}) {
|
||||
t.Errorf("0/-5 = %x, want [03 01 0b]", got)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,330 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// exportPath returns the export file path for a given import path by running
|
||||
// "go list -export". The result is cached so repeated calls for the same
|
||||
// package are fast.
|
||||
func exportPath(importPath string) (string, error) {
|
||||
cmd := exec.Command("go", "list", "-json", "-export", importPath)
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("go list %s: %w", importPath, err)
|
||||
}
|
||||
// Quick JSON extraction: find "Export": "…"
|
||||
const key = `"Export": "`
|
||||
i := bytes.Index(out, []byte(key))
|
||||
if i < 0 {
|
||||
return "", fmt.Errorf("go list %s: no Export field", importPath)
|
||||
}
|
||||
start := i + len(key)
|
||||
end := bytes.IndexByte(out[start:], '"')
|
||||
if end < 0 {
|
||||
return "", fmt.Errorf("go list %s: malformed Export field", importPath)
|
||||
}
|
||||
return string(out[start : start+end]), nil
|
||||
}
|
||||
|
||||
// resolveExternalGOOBJ resolves a set of external symbol references into
|
||||
// (package index, symbol index) pairs suitable for GOOBJ emission.
|
||||
//
|
||||
// refs maps package import paths to the symbol names referenced from that
|
||||
// package. The returned pkgIdx maps each import path to its position in
|
||||
// the blkPkgIdx table (0-based), and symIdx gives each symbol's index within
|
||||
// its package.
|
||||
func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symIdx map[string]int, err error) {
|
||||
pkgIdx = make(map[string]int, len(refs))
|
||||
symIdx = make(map[string]int)
|
||||
|
||||
// Assign package indices in sorted order for determinism.
|
||||
packages := sortedPkgRefs(refs)
|
||||
|
||||
for i, pkg := range packages {
|
||||
pkgIdx[pkg.path] = i
|
||||
exp, err := exportPath(pkg.path)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
data, err := os.ReadFile(exp)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
gobj, err := extractGOOBJ(data)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("%s: %w", pkg.path, err)
|
||||
}
|
||||
for _, name := range pkg.syms {
|
||||
idx := gobj.findSymbol(pkg.path, name)
|
||||
if idx < 0 {
|
||||
return nil, nil, fmt.Errorf("symbol %s·%s not found in export data of %s", pkg.path, name, pkg.path)
|
||||
}
|
||||
symIdx[pkg.path+"·"+name] = idx
|
||||
}
|
||||
}
|
||||
return pkgIdx, symIdx, nil
|
||||
}
|
||||
|
||||
type pkgRef struct {
|
||||
path string
|
||||
syms []string
|
||||
}
|
||||
|
||||
func sortedPkgRefs(refs map[string][]string) []pkgRef {
|
||||
var pkgs []pkgRef
|
||||
for pkg, syms := range refs {
|
||||
pkgs = append(pkgs, pkgRef{pkg, syms})
|
||||
}
|
||||
// Simple insertion sort — the list is tiny (usually 1–3 packages).
|
||||
for i := 1; i < len(pkgs); i++ {
|
||||
for j := i; j > 0 && pkgs[j-1].path > pkgs[j].path; j-- {
|
||||
pkgs[j-1], pkgs[j] = pkgs[j], pkgs[j-1]
|
||||
}
|
||||
}
|
||||
return pkgs
|
||||
}
|
||||
|
||||
// extractGOOBJ finds the GOOBJ data in an ar archive and returns a parsed
|
||||
// goobjFile. The archive member _go_.o contains the "go object …\n!\n"
|
||||
// preamble followed by the GOOBJ payload; __.PKGDEF is the compiler export
|
||||
// data (type information) and is not the GOOBJ object.
|
||||
func extractGOOBJ(data []byte) (*goobjFile, error) {
|
||||
if len(data) < 8 || string(data[:8]) != "!<arch>\n" {
|
||||
return nil, fmt.Errorf("not an ar archive")
|
||||
}
|
||||
pos := 8
|
||||
for pos+60 <= len(data) {
|
||||
hdr := data[pos : pos+60]
|
||||
pos += 60
|
||||
|
||||
// Parse ar header fields.
|
||||
name := strings.TrimRight(string(hdr[:16]), " /")
|
||||
size := parseArDecimal(hdr[48:58])
|
||||
if size < 0 {
|
||||
return nil, fmt.Errorf("invalid ar header: bad size")
|
||||
}
|
||||
if pos+size > len(data) {
|
||||
return nil, fmt.Errorf("ar entry %q extends past end of file", name)
|
||||
}
|
||||
body := data[pos : pos+size]
|
||||
pos += size
|
||||
// ar pads to even bytes.
|
||||
if pos%2 != 0 {
|
||||
pos++
|
||||
}
|
||||
|
||||
if name == "_go_.o" {
|
||||
return parseGOOBJ(body)
|
||||
}
|
||||
}
|
||||
return nil, fmt.Errorf("archive contains no _go_.o member")
|
||||
}
|
||||
|
||||
// parseArDecimal parses a decimal number from a space-padded field.
|
||||
func parseArDecimal(b []byte) int {
|
||||
v := 0
|
||||
for _, c := range b {
|
||||
if c == ' ' {
|
||||
continue
|
||||
}
|
||||
if c < '0' || c > '9' {
|
||||
return -1
|
||||
}
|
||||
v = v*10 + int(c-'0')
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
// goobjFile is a parsed GOOBJ file: the string table and the symbol-definition
|
||||
// block.
|
||||
type goobjFile struct {
|
||||
strTab []byte // string table, at headerSize + n
|
||||
symdef []byte // blkSymdef raw block
|
||||
npdef []byte // blkNonpkgdef raw block
|
||||
}
|
||||
|
||||
// symbols returns all symbol names in definition order by scanning the
|
||||
// symdef and nonpkgdef blocks and resolving each name through the string
|
||||
// table. Package definitions (blkSymdef) use fully-qualified names like
|
||||
// "runtime.morestack"; non-package definitions (blkNonpkgdef) use bare
|
||||
// names like "morestack". This combined list matches the index the
|
||||
// linker expects for cross-package references.
|
||||
func (f *goobjFile) symbols() []string {
|
||||
return append(f.defNames(), f.npdefNames()...)
|
||||
}
|
||||
|
||||
// findSymbol returns the index of a symbol within the combined symbol list,
|
||||
// or -1 if not found. It first tries the fully-qualified name (pkg.name),
|
||||
// then the bare name.
|
||||
func (f *goobjFile) findSymbol(pkg, name string) int {
|
||||
qualified := pkg + "." + name
|
||||
syms := f.symbols()
|
||||
for i, s := range syms {
|
||||
if s == qualified {
|
||||
return i
|
||||
}
|
||||
}
|
||||
// Try bare name (for non-package definitions).
|
||||
for i, s := range syms {
|
||||
if s == name {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// defNames returns names from blkSymdef only.
|
||||
func (f *goobjFile) defNames() []string {
|
||||
return f.readSymNames(f.symdef)
|
||||
}
|
||||
|
||||
// npdefNames returns names from blkNonpkgdef.
|
||||
func (f *goobjFile) npdefNames() []string {
|
||||
return f.readSymNames(f.npdef)
|
||||
}
|
||||
|
||||
// readSymNames reads symbol names from a symdef/nonpkgdef block. Each record
|
||||
// is 21 bytes: nameLen (u32), nameOff (u32), abi (u16), typ, flag, flag2,
|
||||
// size (u32), align (u32). nameOff is an absolute offset into the string
|
||||
// table.
|
||||
func (f *goobjFile) readSymNames(block []byte) []string {
|
||||
const recSize = 21
|
||||
if len(block) < recSize {
|
||||
return nil
|
||||
}
|
||||
n := len(block) / recSize
|
||||
names := make([]string, 0, n)
|
||||
for i := range n {
|
||||
rec := block[i*recSize : (i+1)*recSize]
|
||||
nameLen := binary.LittleEndian.Uint32(rec[0:4])
|
||||
nameOff := binary.LittleEndian.Uint32(rec[4:8])
|
||||
// nameOff is an absolute offset into the GOOBJ payload. The string
|
||||
// table we have starts at goobjHeaderSize, so we subtract that.
|
||||
if nameOff < goobjHeaderSize {
|
||||
continue
|
||||
}
|
||||
relOff := nameOff - goobjHeaderSize
|
||||
if relOff >= uint32(len(f.strTab)) || relOff+nameLen > uint32(len(f.strTab)) {
|
||||
continue
|
||||
}
|
||||
names = append(names, string(f.strTab[relOff:relOff+nameLen]))
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
const goobjHeaderSize = 8 + 8 + 4 + 4*(blkEnd+1) // magic + fingerprint + flags + 19 block offsets
|
||||
|
||||
// parseGOOBJ parses a raw GOOBJ payload (the data after the "\n!\n" preamble).
|
||||
func parseGOOBJ(data []byte) (*goobjFile, error) {
|
||||
// Find the "\n!\n" separator.
|
||||
sep := []byte("\n!\n")
|
||||
i := bytes.Index(data, sep)
|
||||
if i < 0 {
|
||||
// Maybe the data has no preamble (e.g. a raw .o file).
|
||||
i = -3 // treat as if preamble starts before the data
|
||||
}
|
||||
payload := data[i+len(sep):]
|
||||
|
||||
if len(payload) < goobjHeaderSize {
|
||||
return nil, fmt.Errorf("GOOBJ payload too short (%d bytes)", len(payload))
|
||||
}
|
||||
if string(payload[:8]) != goobjMagic {
|
||||
return nil, fmt.Errorf("bad GOOBJ magic: %q", payload[:8])
|
||||
}
|
||||
|
||||
// Read block offsets. The header layout is:
|
||||
// [0:8] magic
|
||||
// [8:16] fingerprint
|
||||
// [16:20] flags
|
||||
// [20:96] 19 × uint32 offsets
|
||||
var offs [blkEnd + 1]uint32
|
||||
for i := 0; i <= blkEnd; i++ {
|
||||
offs[i] = binary.LittleEndian.Uint32(payload[20+4*i:])
|
||||
}
|
||||
// The string table lives at headerSize.
|
||||
strTabStart := uint32(goobjHeaderSize)
|
||||
|
||||
f := &goobjFile{
|
||||
strTab: payload[strTabStart:offs[0]],
|
||||
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
|
||||
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
|
||||
}
|
||||
return f, nil
|
||||
}
|
||||
|
||||
// blockSlice extracts a block from the payload using its offset pair.
|
||||
func blockSlice(payload []byte, offs [blkEnd + 1]uint32, start, end int) []byte {
|
||||
if start < 0 || end > blkEnd || offs[end] < offs[start] {
|
||||
return nil
|
||||
}
|
||||
beg := offs[start]
|
||||
fin := offs[end]
|
||||
if int(fin) > len(payload) || int(beg) > int(fin) {
|
||||
return nil
|
||||
}
|
||||
return payload[beg:fin]
|
||||
}
|
||||
|
||||
// resolveExternalSymbols is the high-level entry point for GOOBJ emission.
|
||||
// Given a list of external symbol names (e.g. ["runtime·morestack",
|
||||
// "runtime·g0"]), it returns the package-index table entries and a map from
|
||||
// full symbol name to GOOBJ {pkgIdx, symIdx}.
|
||||
//
|
||||
// The package table entries should be written into blkPkgIdx, and the
|
||||
// returned indices should replace pkgIdxSelf / placeholder values in the
|
||||
// relocation records.
|
||||
func resolveExternalSymbols(externals []string) (pkgTable []string, pkgIdxMap map[string]int, symIdxMap map[string]int, err error) {
|
||||
// Group references by package.
|
||||
refs := make(map[string]map[string]bool)
|
||||
for _, full := range externals {
|
||||
pkg, name := splitQualified(full)
|
||||
if refs[pkg] == nil {
|
||||
refs[pkg] = make(map[string]bool)
|
||||
}
|
||||
refs[pkg][name] = true
|
||||
}
|
||||
|
||||
// Convert maps to slices.
|
||||
r := make(map[string][]string, len(refs))
|
||||
for pkg, names := range refs {
|
||||
for name := range names {
|
||||
r[pkg] = append(r[pkg], name)
|
||||
}
|
||||
}
|
||||
|
||||
pkgIdx1, symIdx1, err := resolveExternalGOOBJ(r)
|
||||
if err != nil {
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
|
||||
// Build the package table in pkgIdx order.
|
||||
pkgTable = make([]string, len(pkgIdx1))
|
||||
for pkg, idx := range pkgIdx1 {
|
||||
pkgTable[idx] = pkg
|
||||
}
|
||||
|
||||
return pkgTable, pkgIdx1, symIdx1, nil
|
||||
}
|
||||
|
||||
// splitQualified splits a qualified Go symbol name (pkgpath·name) into its
|
||||
// package path and local name. The separator is the middle dot (U+00B7).
|
||||
// If no separator is found, the symbol is assumed to be in the current
|
||||
// package (empty pkg).
|
||||
func splitQualified(full string) (pkg, name string) {
|
||||
if idx := strings.IndexByte(full, '\u00b7'); idx >= 0 {
|
||||
return full[:idx], full[idx+len("\u00b7"):]
|
||||
}
|
||||
if before, after, ok := strings.Cut(full, "."); ok {
|
||||
return before, after
|
||||
}
|
||||
return "", full
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// TestReadRuntimeSymbols verifies the GOOBJ reader can extract and find
|
||||
// symbols from the runtime package's compiled archive.
|
||||
func TestReadRuntimeSymbols(t *testing.T) {
|
||||
exp, err := exportPath("runtime")
|
||||
if err != nil {
|
||||
t.Skipf("cannot find runtime export: %v (need Go toolchain)", err)
|
||||
}
|
||||
data, err := os.ReadFile(exp)
|
||||
if err != nil {
|
||||
t.Skipf("cannot read runtime export: %v", err)
|
||||
}
|
||||
gobj, err := extractGOOBJ(data)
|
||||
if err != nil {
|
||||
t.Fatalf("extractGOOBJ: %v", err)
|
||||
}
|
||||
|
||||
t.Logf("runtime: %d symbols", len(gobj.symbols()))
|
||||
|
||||
// Verify we can find well-known runtime symbols.
|
||||
for _, tc := range []struct{ pkg, name string }{
|
||||
{"runtime", "g0"},
|
||||
{"runtime", "morestack"},
|
||||
{"runtime", "newstack"},
|
||||
} {
|
||||
idx := gobj.findSymbol(tc.pkg, tc.name)
|
||||
if idx < 0 {
|
||||
t.Errorf("findSymbol(%q, %q) = -1", tc.pkg, tc.name)
|
||||
} else {
|
||||
t.Logf("findSymbol(%q, %q) = %d", tc.pkg, tc.name, idx)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestResolveExternalSymbols verifies end-to-end resolution of external
|
||||
// symbol references.
|
||||
func TestResolveExternalSymbols(t *testing.T) {
|
||||
if _, err := exec.LookPath("go"); err != nil {
|
||||
t.Skip("go toolchain not available")
|
||||
}
|
||||
|
||||
refs := map[string][]string{
|
||||
"runtime": {"g0"},
|
||||
}
|
||||
pkgIdx, symIdx, err := resolveExternalGOOBJ(refs)
|
||||
if err != nil {
|
||||
t.Fatalf("resolveExternalGOOBJ: %v", err)
|
||||
}
|
||||
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 0 {
|
||||
t.Errorf("pkgIdx = %v, want runtime→0", pkgIdx)
|
||||
}
|
||||
if _, ok := symIdx["runtime·g0"]; !ok {
|
||||
t.Errorf("symIdx missing runtime·g0, got %v", symIdx)
|
||||
}
|
||||
t.Logf("runtime·g0 → SymIdx=%d", symIdx["runtime·g0"])
|
||||
}
|
||||
@@ -0,0 +1,513 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// goobjView is a minimal parsed view of a GOOBJ payload, enough to check
|
||||
// the emitter's output block by block.
|
||||
type goobjView struct {
|
||||
t *testing.T
|
||||
b []byte
|
||||
offs [blkEnd + 1]uint32
|
||||
strOff uint32
|
||||
}
|
||||
|
||||
func openGoobj(t *testing.T, data []byte) *goobjView {
|
||||
t.Helper()
|
||||
i := bytes.Index(data, []byte(goobjMagic))
|
||||
if i < 0 {
|
||||
t.Fatal("no GOOBJ magic in output")
|
||||
}
|
||||
v := &goobjView{t: t, b: data[i:], strOff: uint32(i + 96)}
|
||||
for j := 0; j <= blkEnd; j++ {
|
||||
v.offs[j] = binary.LittleEndian.Uint32(v.b[20+4*j:])
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
func (v *goobjView) blk(i int) []byte { return v.b[v.offs[i]:v.offs[i+1]] }
|
||||
|
||||
func (v *goobjView) str(off, ln uint32) string {
|
||||
return string(v.b[off : off+ln])
|
||||
}
|
||||
|
||||
type goobjSymView struct {
|
||||
name string
|
||||
abi uint16
|
||||
typ uint8
|
||||
flag uint8
|
||||
flag2 uint8
|
||||
size uint32
|
||||
align uint32
|
||||
}
|
||||
|
||||
func (v *goobjView) syms(i int) []goobjSymView {
|
||||
var out []goobjSymView
|
||||
for x := v.blk(i); len(x) >= 21; x = x[21:] {
|
||||
le := binary.LittleEndian
|
||||
out = append(out, goobjSymView{
|
||||
name: v.str(le.Uint32(x[4:]), le.Uint32(x[0:])),
|
||||
abi: le.Uint16(x[8:]),
|
||||
typ: x[10],
|
||||
flag: x[11],
|
||||
flag2: x[12],
|
||||
size: le.Uint32(x[13:]),
|
||||
align: le.Uint32(x[17:]),
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// TestGOObjectStructure checks the emitted object's blocks against the
|
||||
// ground truth captured from go tool asm: the symbol tables, the FuncInfo
|
||||
// contents, the pc-value tables, the relocation and the aux wiring.
|
||||
func TestGOObjectStructure(t *testing.T) {
|
||||
f, errs := parser.Parse("t_amd64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·addq(SB), NOSPLIT, $0-24
|
||||
MOVQ a+0(FP), AX
|
||||
MOVQ b+8(FP), CX
|
||||
ADDQ CX, AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
|
||||
TEXT ·loadmask(SB), NOSPLIT, $0-8
|
||||
VMOVDQU mask<>(SB), X0
|
||||
VPMOVMSKB X0, AX
|
||||
MOVQ AX, ret+0(FP)
|
||||
RET
|
||||
|
||||
GLOBL mask<>(SB), RODATA, $16
|
||||
DATA mask<>+0(SB)/8, $0x0807060504030201
|
||||
DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
obj, err := img.GOObject("testpkg", "t_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObject: %v", err)
|
||||
}
|
||||
v := openGoobj(t, obj)
|
||||
|
||||
if flags := binary.LittleEndian.Uint32(v.b[16:]); flags != 4 {
|
||||
t.Errorf("flags = %#x, want ObjFlagFromAssembly (4)", flags)
|
||||
}
|
||||
|
||||
// Package defs: the static GLOBL, then per function the FuncInfo and the
|
||||
// two DWARF symbols (debug_line program, subprogram DIE).
|
||||
defs := v.syms(blkSymdef)
|
||||
if len(defs) != 7 {
|
||||
t.Fatalf("symdefs = %d, want 7", len(defs))
|
||||
}
|
||||
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
|
||||
t.Errorf("mask symbol = %+v", defs[0])
|
||||
}
|
||||
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
|
||||
t.Errorf("addq funcinfo symbol = %+v", defs[1])
|
||||
}
|
||||
if defs[2].name != "" || defs[2].typ != kindSDWARFLINES || defs[2].size == 0 {
|
||||
t.Errorf("addq lines symbol = %+v", defs[2])
|
||||
}
|
||||
if defs[3].name != "" || defs[3].typ != kindSDWARFFCN || defs[3].size == 0 {
|
||||
t.Errorf("addq DIE symbol = %+v", defs[3])
|
||||
}
|
||||
if defs[4].name != "" || defs[4].typ != kindSDATA || defs[4].size != 28 {
|
||||
t.Errorf("loadmask funcinfo symbol = %+v", defs[4])
|
||||
}
|
||||
|
||||
// Non-package defs: four pc tables and the function, per function.
|
||||
nps := v.syms(blkNonpkgdef)
|
||||
if len(nps) != 10 {
|
||||
t.Fatalf("nonpkgdefs = %d, want 10", len(nps))
|
||||
}
|
||||
fn := nps[4]
|
||||
if fn.name != "testpkg.addq" || fn.typ != kindSTEXT || fn.flag != symFlagNoSplit || fn.size != 19 {
|
||||
t.Errorf("addq symbol = %+v", fn)
|
||||
}
|
||||
for i, s := range []int{0, 1, 2, 3, 5, 6, 7, 8} {
|
||||
if nps[s].typ != kindSRODATA || nps[s].align != 1 || nps[s].name != "" {
|
||||
t.Errorf("pc table %d = %+v", i, nps[s])
|
||||
}
|
||||
}
|
||||
|
||||
// FuncInfo: args 24, FuncFlag Asm, one file, no inline tree.
|
||||
le := binary.LittleEndian
|
||||
data := v.blk(blkData)
|
||||
didx := v.blk(blkDataIdx)
|
||||
fi := data[16:44]
|
||||
if le.Uint32(fi[0:]) != 24 || le.Uint32(fi[4:]) != 0 || fi[8] != 0 || fi[9] != funcFlagAsm ||
|
||||
le.Uint32(fi[16:]) != 1 || le.Uint32(fi[20:]) != 0 || le.Uint32(fi[24:]) != 0 {
|
||||
t.Errorf("funcinfo bytes %x", fi)
|
||||
}
|
||||
|
||||
// The pc-value tables of addq (non-package indices 0–3, so global
|
||||
// indices 7–10): pcsp a flat zero over the whole function, pcinline a
|
||||
// flat -1, both with the pc delta in MinLC (1) units.
|
||||
pcsp := data[le.Uint32(didx[4*7:]):]
|
||||
if got := pcsp[:3]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
|
||||
t.Errorf("pcsp = %x, want 021300", got)
|
||||
}
|
||||
pcinl := data[le.Uint32(didx[4*10:]):]
|
||||
if got := pcinl[:3]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) {
|
||||
t.Errorf("pcinline = %x, want 001300", got)
|
||||
}
|
||||
|
||||
// Relocations: the four DWARF address references (two per function, in
|
||||
// definition order), then the loadmask code's R_PCREL against the
|
||||
// GLOBL, with the field in the function code left zero. The loadmask
|
||||
// code's offset comes from the data index (7 defs + 9 non-package).
|
||||
relocs := v.blk(blkReloc)
|
||||
if len(relocs) != 5*23 {
|
||||
t.Fatalf("relocs = %d bytes, want 5 entries", len(relocs))
|
||||
}
|
||||
// addq's DWARF references (defs 2 and 3) against the function, which
|
||||
// is non-package index 4.
|
||||
lr := relocs[:23]
|
||||
if int32(le.Uint32(lr[0:])) != 3 || lr[4] != 8 || le.Uint16(lr[5:]) != relocAddr ||
|
||||
le.Uint32(lr[15:]) != pkgIdxNone || le.Uint32(lr[19:]) != 4 {
|
||||
t.Errorf("addq lines reloc = %x", lr)
|
||||
}
|
||||
dr := relocs[23:46]
|
||||
if dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4() ||
|
||||
le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 {
|
||||
t.Errorf("addq DIE reloc = %x", dr)
|
||||
}
|
||||
cr := relocs[4*23:]
|
||||
off := int32(le.Uint32(cr[0:]))
|
||||
if off != 4 || cr[4] != 4 || le.Uint16(cr[5:]) != relocPCRel ||
|
||||
le.Uint64(cr[7:]) != 0 || le.Uint32(cr[15:]) != pkgIdxSelf || le.Uint32(cr[19:]) != 0 {
|
||||
t.Errorf("loadmask reloc = %x", cr)
|
||||
}
|
||||
lm := le.Uint32(didx[4*16:])
|
||||
code := data[lm : lm+18]
|
||||
if !bytes.Equal(code[4:8], []byte{0, 0, 0, 0}) {
|
||||
t.Errorf("relocated field = %x, want zeroed", code[4:8])
|
||||
}
|
||||
|
||||
// Aux wiring: FuncInfo, the two DWARF symbols (package symbols), then
|
||||
// the four pc tables (non-package symbols).
|
||||
auxs := v.blk(blkAux)
|
||||
if len(auxs) != 2*7*9 {
|
||||
t.Fatalf("aux = %d bytes, want 14 entries", len(auxs))
|
||||
}
|
||||
wantAux := []struct {
|
||||
typ uint8
|
||||
pkg uint32
|
||||
idx uint32
|
||||
}{
|
||||
{auxFuncInfo, pkgIdxSelf, 1},
|
||||
{auxDwarfInfo, pkgIdxSelf, 3},
|
||||
{auxDwarfLines, pkgIdxSelf, 2},
|
||||
{auxPcsp, pkgIdxNone, 0},
|
||||
{auxPcfile, pkgIdxNone, 1},
|
||||
{auxPcline, pkgIdxNone, 2},
|
||||
{auxPcinline, pkgIdxNone, 3},
|
||||
{auxFuncInfo, pkgIdxSelf, 4},
|
||||
{auxDwarfInfo, pkgIdxSelf, 6},
|
||||
{auxDwarfLines, pkgIdxSelf, 5},
|
||||
{auxPcsp, pkgIdxNone, 5},
|
||||
{auxPcfile, pkgIdxNone, 6},
|
||||
{auxPcline, pkgIdxNone, 7},
|
||||
{auxPcinline, pkgIdxNone, 8},
|
||||
}
|
||||
for i, w := range wantAux {
|
||||
e := auxs[i*9:]
|
||||
if e[0] != w.typ || le.Uint32(e[1:]) != w.pkg || le.Uint32(e[5:]) != w.idx {
|
||||
t.Errorf("aux[%d] = {%d,%d,%d}, want {%d,%d,%d}", i, e[0], le.Uint32(e[1:]), le.Uint32(e[5:]), w.typ, w.pkg, w.idx)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// decodePCValues decodes a pc-value table into (pc, value) steps. The
|
||||
// table ends with a final unsigned pc delta covering the rest of the
|
||||
// function, followed by a zero byte that carries no value delta.
|
||||
func decodePCValues(b []byte) (pcs, vals []int64) {
|
||||
val, n := binary.Varint(b)
|
||||
b = b[n:]
|
||||
val-- // the first delta is against the implicit -1
|
||||
var pc int64
|
||||
pcs = append(pcs, pc)
|
||||
vals = append(vals, val)
|
||||
for {
|
||||
pcd, n := binary.Uvarint(b)
|
||||
b = b[n:]
|
||||
if pcd == 0 { // zero pc delta terminates the table
|
||||
break
|
||||
}
|
||||
pc += int64(pcd)
|
||||
if len(b) == 1 && b[0] == 0 { // final coverage, no value change
|
||||
break
|
||||
}
|
||||
vd, n := binary.Varint(b)
|
||||
b = b[n:]
|
||||
val += vd
|
||||
pcs = append(pcs, pc)
|
||||
vals = append(vals, val)
|
||||
}
|
||||
return pcs, vals
|
||||
}
|
||||
|
||||
// TestGOObjectPcspFrame checks the pcsp table of a frame-pointer function:
|
||||
// the prologue raises the stack delta to 8+frame, the RET's epilogue
|
||||
// restores it to zero.
|
||||
func TestGOObjectPcspFrame(t *testing.T) {
|
||||
f, errs := parser.Parse("frame_amd64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·framed(SB), NOSPLIT, $8-0
|
||||
MOVQ BP, AX
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
pcs, vals := decodePCValues(pcspTable(fn, 1))
|
||||
// Prologue: PUSHQ BP (1 byte, +8), MOVQ SP, BP (3 bytes, no change),
|
||||
// SUBQ $8, SP (4 bytes, +16 in total); the RET's epilogue unwinds
|
||||
// ADDQ $8, SP (+8) then POPQ BP (0).
|
||||
wantPCs := []int64{0, 1, 8}
|
||||
wantVals := []int64{0, 8, 16}
|
||||
if len(pcs) < len(wantPCs) {
|
||||
t.Fatalf("pcsp pcs = %v vals = %v", pcs, vals)
|
||||
}
|
||||
for i := range wantPCs {
|
||||
if pcs[i] != wantPCs[i] || vals[i] != wantVals[i] {
|
||||
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d) — all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
|
||||
}
|
||||
}
|
||||
// The last two steps unwind the epilogue to zero.
|
||||
n := len(pcs)
|
||||
if vals[n-1] != 0 || vals[n-2] != 8 {
|
||||
t.Errorf("epilogue steps = %v %v, want …8, 0", pcs, vals)
|
||||
}
|
||||
// The table covers the whole function.
|
||||
if last := pcs[n-1]; last >= int64(fn.Size) {
|
||||
t.Errorf("last pc %d beyond function size %d", last, fn.Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectExternalRejected checks that a reference to a symbol no GLOBL
|
||||
// defines is reported: GOOBJ emission resolves only file-local symbols so
|
||||
// far.
|
||||
func TestGOObjectExternalRejected(t *testing.T) {
|
||||
f, errs := parser.Parse("ext_amd64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·useext(SB), NOSPLIT, $0-8
|
||||
MOVQ elsewhere(SB), AX
|
||||
MOVQ AX, ret+0(FP)
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
if _, err := img.GOObject("p", "ext_amd64.s"); err == nil || !strings.Contains(err.Error(), "external") {
|
||||
t.Errorf("error = %v, want an external-symbol error", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectLinkAndRun is the end-to-end check: assemble the test
|
||||
// functions to a GOOBJ, swap it into a go build in place of the toolchain's
|
||||
// assembly object, link, and run — the output must match the baseline
|
||||
// binary the Go assembler produced. Skipped when no Go toolchain is
|
||||
// available.
|
||||
func TestGOObjectLinkAndRun(t *testing.T) {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
t.Skip("no Go toolchain available")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
|
||||
const asmSrc = `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·addq(SB), NOSPLIT, $0-24
|
||||
MOVQ a+0(FP), AX
|
||||
MOVQ b+8(FP), CX
|
||||
ADDQ CX, AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
|
||||
TEXT ·loadmask(SB), NOSPLIT, $0-8
|
||||
VMOVDQU mask<>(SB), X0
|
||||
VPMOVMSKB X0, AX
|
||||
MOVQ AX, ret+0(FP)
|
||||
RET
|
||||
|
||||
GLOBL mask<>(SB), RODATA, $16
|
||||
DATA mask<>+0(SB)/8, $0x0807060504030201
|
||||
DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
||||
`
|
||||
const mainSrc = `package main
|
||||
|
||||
func addq(a, b int64) int64
|
||||
func loadmask() int64
|
||||
|
||||
func main() {
|
||||
println(addq(41, 1))
|
||||
println(loadmask())
|
||||
}
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module goobjtest\n\ngo 1.27\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Baseline build with the toolchain's assembler; keep the work
|
||||
// directory and the commands the build used.
|
||||
cmd := exec.Command(goBin, "build", "-x", "-work", "-o", "app", ".")
|
||||
cmd.Dir = dir
|
||||
buildLog, err := cmd.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||
}
|
||||
var work string
|
||||
var asmObj, pkgArch, linkLine string
|
||||
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
case strings.Contains(line, "/asm ") && strings.Contains(line, "-o ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
|
||||
asmObj = fieldAfter(line, "-o")
|
||||
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
|
||||
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
|
||||
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
|
||||
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||
linkLine = line
|
||||
}
|
||||
}
|
||||
if work == "" || asmObj == "" || pkgArch == "" || linkLine == "" {
|
||||
t.Fatalf("could not locate the build steps:\n%s", buildLog)
|
||||
}
|
||||
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
|
||||
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
|
||||
|
||||
// The baseline's answer.
|
||||
baseOut, err := exec.Command(filepath.Join(dir, "app")).CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("run baseline: %v\n%s", err, baseOut)
|
||||
}
|
||||
|
||||
// Assemble the same source with gasm and swap the object in.
|
||||
pf, perrs := parser.Parse(filepath.Join(dir, "main_amd64.s"), asmSrc)
|
||||
if len(perrs) > 0 {
|
||||
t.Fatalf("parse: %v", perrs)
|
||||
}
|
||||
img, err := AssembleFile(pf)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
obj, err := img.GOObject("main", filepath.Join(dir, "main_amd64.s"))
|
||||
if err != nil {
|
||||
t.Fatalf("GOObject: %v", err)
|
||||
}
|
||||
|
||||
// Rebuild the package archive with our object in place of the
|
||||
// toolchain's (go tool pack has no replace-in-place that dedupes, so
|
||||
// extract, substitute and repack). The archive member holding the
|
||||
// assembler's output is named after the asm object file, e.g.
|
||||
// main_amd64.o.
|
||||
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
|
||||
membersDir := filepath.Join(dir, "members")
|
||||
if err := os.MkdirAll(membersDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
extract.Dir = membersDir
|
||||
if out, err := extract.CombinedOutput(); err != nil {
|
||||
t.Fatalf("pack x: %v\n%s", err, out)
|
||||
}
|
||||
member := filepath.Join(membersDir, filepath.Base(asmObj))
|
||||
if err := os.Chmod(member, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(member, obj, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
|
||||
listOut, err := listCmd.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("pack t: %v\n%s", err, listOut)
|
||||
}
|
||||
newArch := filepath.Join(dir, "pkg.a")
|
||||
args := []string{"tool", "pack", "c", newArch}
|
||||
seen := map[string]bool{}
|
||||
for m := range strings.FieldsSeq(string(listOut)) {
|
||||
if seen[m] {
|
||||
continue
|
||||
}
|
||||
seen[m] = true
|
||||
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
args = append(args, filepath.Join(membersDir, m))
|
||||
}
|
||||
pack := exec.Command(goBin, args...)
|
||||
pack.Dir = membersDir
|
||||
if out, err := pack.CombinedOutput(); err != nil {
|
||||
t.Fatalf("pack c: %v\n%s", err, out)
|
||||
}
|
||||
|
||||
// Link with our archive. The link line carries a GOROOT assignment
|
||||
// and $WORK placeholders; run it through the shell with the
|
||||
// GOEXPERIMENT the toolchain expects (the linker compares the object
|
||||
// header against its own, experiments included).
|
||||
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
|
||||
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
|
||||
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
|
||||
link := exec.Command("sh", "-c", linkLine)
|
||||
link.Dir = dir
|
||||
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)))
|
||||
if out, err := link.CombinedOutput(); err != nil {
|
||||
t.Fatalf("link with gasm object: %v\n%s", err, out)
|
||||
}
|
||||
got, err := exec.Command(filepath.Join(dir, "app2")).CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("run gasm-linked binary: %v\n%s", err, got)
|
||||
}
|
||||
if !bytes.Equal(got, baseOut) {
|
||||
t.Errorf("gasm-linked output %q, want baseline %q", got, baseOut)
|
||||
}
|
||||
}
|
||||
|
||||
// fieldAfter returns the whitespace-delimited field following the first
|
||||
// occurrence of flag in line.
|
||||
func fieldAfter(line, flag string) string {
|
||||
fields := strings.Fields(line)
|
||||
for i, f := range fields {
|
||||
if f == flag && i+1 < len(fields) {
|
||||
return fields[i+1]
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"sync"
|
||||
)
|
||||
|
||||
// GOObjectAARCH64 emits a GOOBJ object file for AArch64. The layout is
|
||||
// the shared one in goobj.go — the toolchain preamble, the go120ld header
|
||||
// with its block offsets, the string table, the symbol definitions and the
|
||||
// reloc/aux/data index arrays — with the arm64 preamble, the MinLC of 4
|
||||
// for the pc-value deltas, and R_ADDRARM64 relocation types for the
|
||||
// ADRP+ADD/LDR/STR address pairs.
|
||||
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
|
||||
pre, err := toolchainObjectPreambleAARCH64()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
|
||||
if r.Kind == RelArm64Branch {
|
||||
return relocArm64Branch, 4
|
||||
}
|
||||
return relocArm64Addr, 4
|
||||
})
|
||||
}
|
||||
|
||||
// arm64 relocation types (cmd/internal/objabi).
|
||||
const (
|
||||
relocArm64Addr = 3 // R_ADDRARM64 — ADRP+ADD/LDR/STR pair
|
||||
relocArm64Branch = 9 // R_CALLARM64 — BL instruction
|
||||
)
|
||||
|
||||
// toolchainObjectPreambleAARCH64 returns the "go object ...\n!\n" header
|
||||
// the installed go tool asm writes for arm64, captured by assembling a
|
||||
// one-instruction probe.
|
||||
var (
|
||||
preambleAARCH64Once sync.Once
|
||||
preambleAARCH64 []byte
|
||||
preambleAARCH64Err error
|
||||
)
|
||||
|
||||
func toolchainObjectPreambleAARCH64() ([]byte, error) {
|
||||
preambleAARCH64Once.Do(func() {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
preambleAARCH64Err = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
|
||||
return
|
||||
}
|
||||
dir, err := os.MkdirTemp("", "gasm-preamble-arm64")
|
||||
if err != nil {
|
||||
preambleAARCH64Err = err
|
||||
return
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
src := filepath.Join(dir, "probe_arm64.s")
|
||||
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
|
||||
preambleAARCH64Err = err
|
||||
return
|
||||
}
|
||||
obj := filepath.Join(dir, "probe.o")
|
||||
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
|
||||
cmd.Env = append(os.Environ(), "GOARCH=arm64")
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
preambleAARCH64Err = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
|
||||
return
|
||||
}
|
||||
data, err := os.ReadFile(obj)
|
||||
if err != nil {
|
||||
preambleAARCH64Err = err
|
||||
return
|
||||
}
|
||||
i := bytes.Index(data, []byte("\n!\n"))
|
||||
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
|
||||
preambleAARCH64Err = fmt.Errorf("unrecognised assembler object layout")
|
||||
return
|
||||
}
|
||||
preambleAARCH64 = data[:i+3]
|
||||
})
|
||||
return preambleAARCH64, preambleAARCH64Err
|
||||
}
|
||||
@@ -0,0 +1,91 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"sync"
|
||||
)
|
||||
|
||||
// GOObjectLOONG64 emits a GOOBJ object file for LoongArch. The layout is
|
||||
// the shared one in goobj.go — the toolchain preamble, the go120ld header
|
||||
// with its block offsets, the string table, the symbol definitions and the
|
||||
// reloc/aux/data index arrays — with the loong64 preamble, the MinLC of 4
|
||||
// for the pc-value deltas, and R_LOONG64_ADDR_HI/LO relocation types for
|
||||
// the pcalau12i+addi.d address pairs.
|
||||
func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
|
||||
pre, err := toolchainObjectPreambleLOONG64()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
|
||||
// A pcalau12i+addi.d pair: the high part carries
|
||||
// R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO.
|
||||
if r.Kind == RelLoong64AddrLo {
|
||||
return relocLoong64AddrLo, 4
|
||||
}
|
||||
return relocLoong64AddrHi, 4
|
||||
})
|
||||
}
|
||||
|
||||
// Loong64 relocation types (cmd/internal/objabi). R_LOONG64_ADDR_HI
|
||||
// resolves the high 20 bits of a PC-relative address into pcalau12i;
|
||||
// R_LOONG64_ADDR_LO the low 12 bits into addi.d/ld/st.
|
||||
const (
|
||||
relocLoong64AddrHi = 77 // R_LOONG64_ADDR_HI
|
||||
relocLoong64AddrLo = 78 // R_LOONG64_ADDR_LO
|
||||
)
|
||||
|
||||
// toolchainObjectPreambleLOONG64 returns the "go object ...\n!\n" header
|
||||
// the installed go tool asm writes for loong64, captured by assembling a
|
||||
// one-instruction probe (see toolchainObjectPreamble).
|
||||
var (
|
||||
preambleLOONG64Once sync.Once
|
||||
preambleLOONG64 []byte
|
||||
preambleLOONG64Err error
|
||||
)
|
||||
|
||||
func toolchainObjectPreambleLOONG64() ([]byte, error) {
|
||||
preambleLOONG64Once.Do(func() {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
preambleLOONG64Err = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
|
||||
return
|
||||
}
|
||||
dir, err := os.MkdirTemp("", "gasm-preamble-loong64")
|
||||
if err != nil {
|
||||
preambleLOONG64Err = err
|
||||
return
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
src := filepath.Join(dir, "probe_loong64.s")
|
||||
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
|
||||
preambleLOONG64Err = err
|
||||
return
|
||||
}
|
||||
obj := filepath.Join(dir, "probe.o")
|
||||
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
|
||||
cmd.Env = append(os.Environ(), "GOARCH=loong64")
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
preambleLOONG64Err = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
|
||||
return
|
||||
}
|
||||
data, err := os.ReadFile(obj)
|
||||
if err != nil {
|
||||
preambleLOONG64Err = err
|
||||
return
|
||||
}
|
||||
i := bytes.Index(data, []byte("\n!\n"))
|
||||
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
|
||||
preambleLOONG64Err = fmt.Errorf("unrecognised assembler object layout")
|
||||
return
|
||||
}
|
||||
preambleLOONG64 = data[:i+3]
|
||||
})
|
||||
return preambleLOONG64, preambleLOONG64Err
|
||||
}
|
||||
@@ -0,0 +1,95 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"sync"
|
||||
)
|
||||
|
||||
// GOObjectRISCV emits a GOOBJ object file for RISC-V. The layout is the
|
||||
// shared one in goobj.go — the toolchain preamble, the go120ld header with
|
||||
// its block offsets, the string table, the symbol definitions and the
|
||||
// reloc/aux/data index arrays — with the RISC-V preamble, the MinLC of 2 for
|
||||
// the pc-value deltas, and the single R_RISCV_PCREL_ITYPE/STYPE relocation
|
||||
// per AUIPC pair, matching `go tool asm`'s model (each pair is one 8-byte
|
||||
// relocation, not the ELF HI20/LO12 pair).
|
||||
func (img *Image) GOObjectRISCV(pkgPath, srcPath string) ([]byte, error) {
|
||||
pre, err := toolchainObjectPreambleRISCV()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return img.emitGOObject(pkgPath, srcPath, pre, 2, func(r Reloc) (uint16, uint8) {
|
||||
switch r.Kind {
|
||||
case RelRISCVPCRELSType:
|
||||
return relocRISCVPcrelStype, 8
|
||||
case RelRISCVJal:
|
||||
return relocRISCVJal, 4
|
||||
default:
|
||||
return relocRISCVPcrelItype, 8
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// RISC-V relocation types (cmd/internal/objabi). The Go linker applies
|
||||
// R_RISCV_PCREL_ITYPE/STYPE to an AUIPC + I/S-type instruction pair as a
|
||||
// single 8-byte field; R_RISCV_JAL covers a single 4-byte J-type instruction.
|
||||
const (
|
||||
relocRISCVJal = 59 // R_RISCV_JAL
|
||||
relocRISCVPcrelItype = 62 // R_RISCV_PCREL_ITYPE
|
||||
relocRISCVPcrelStype = 63 // R_RISCV_PCREL_STYPE
|
||||
)
|
||||
|
||||
// toolchainObjectPreambleRISCV returns the "go object ...\n!\n" header
|
||||
// the installed go tool asm writes for riscv64, captured by assembling a
|
||||
// one-instruction probe (see toolchainObjectPreamble).
|
||||
var (
|
||||
preambleRISCVOnce sync.Once
|
||||
preambleRISCV []byte
|
||||
preambleRISCVErr error
|
||||
)
|
||||
|
||||
func toolchainObjectPreambleRISCV() ([]byte, error) {
|
||||
preambleRISCVOnce.Do(func() {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
preambleRISCVErr = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
|
||||
return
|
||||
}
|
||||
dir, err := os.MkdirTemp("", "gasm-preamble-riscv")
|
||||
if err != nil {
|
||||
preambleRISCVErr = err
|
||||
return
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
src := filepath.Join(dir, "probe_riscv64.s")
|
||||
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
|
||||
preambleRISCVErr = err
|
||||
return
|
||||
}
|
||||
obj := filepath.Join(dir, "probe.o")
|
||||
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
|
||||
cmd.Env = append(os.Environ(), "GOARCH=riscv64")
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
preambleRISCVErr = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
|
||||
return
|
||||
}
|
||||
data, err := os.ReadFile(obj)
|
||||
if err != nil {
|
||||
preambleRISCVErr = err
|
||||
return
|
||||
}
|
||||
i := bytes.Index(data, []byte("\n!\n"))
|
||||
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
|
||||
preambleRISCVErr = fmt.Errorf("unrecognised assembler object layout")
|
||||
return
|
||||
}
|
||||
preambleRISCV = data[:i+3]
|
||||
})
|
||||
return preambleRISCV, preambleRISCVErr
|
||||
}
|
||||
+321
-16
@@ -48,6 +48,57 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
|
||||
// Integer scalar XMM moves: MOVQ with an XMM operand is the SSE2
|
||||
// packed-quadword move, NOT a GPR move: mem→xmm encodes as F3 0F 7E
|
||||
// (reg = dst, no REX.W — the Go assembler's form), xmm→mem as
|
||||
// 66 0F D6 (rm = xmm). Register forms against a GPR use the MOVD
|
||||
// opcodes with REX.W instead: 66 REX.W 0F 6E (gpr→xmm) and
|
||||
// 66 REX.W 0F 7E (xmm→gpr); the memory opcodes with a register r/m
|
||||
// would be undefined forms. MOVL is the packed-dword move:
|
||||
// 66 0F 6E load, 66 0F 7E store, no REX.W. A GPR-move fallback would
|
||||
// silently emit REX.W 8B with the wrong operand meaning.
|
||||
_, srcVec := vecReg(src)
|
||||
dstReg, dstVec := vecReg(dst)
|
||||
if srcVec || dstVec {
|
||||
if dstVec {
|
||||
if g, ok := src.(Reg); ok && !g.isVec() {
|
||||
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x6E}, modrm: -1, sib: -1, rexW: size == 8}
|
||||
if err := setRM(i, dstReg, src, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1}
|
||||
if size == 4 {
|
||||
i.prefix = 0x66
|
||||
i.opcode = []byte{0x0F, 0x6E}
|
||||
}
|
||||
if err := setRM(i, dstReg, src, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
srcXMM, srcIsXMM := src.(Reg)
|
||||
if !srcIsXMM || !srcXMM.isVec() {
|
||||
return fmt.Errorf("MOV: store needs an XMM source")
|
||||
}
|
||||
if g, ok := dst.(Reg); ok && !g.isVec() {
|
||||
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1, rexW: size == 8}
|
||||
if err := setRM(i, srcXMM, dst, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
|
||||
if size == 4 {
|
||||
i.opcode = []byte{0x0F, 0x7E}
|
||||
}
|
||||
if err := setRM(i, srcXMM, dst, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
dstReg, dstIsReg := dst.(Reg)
|
||||
switch src := src.(type) {
|
||||
case Reg:
|
||||
@@ -78,9 +129,41 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
||||
}
|
||||
return e.emit(i)
|
||||
|
||||
case sbMem:
|
||||
if !dstIsReg {
|
||||
return fmt.Errorf("MOV: two memory operands")
|
||||
}
|
||||
// MOV r, r/m: reg=dst, rm=src(static symbol).
|
||||
i := newInstr(size, []byte{movRR(size)})
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
|
||||
case Imm:
|
||||
if dstIsReg {
|
||||
// MOV r, imm: 0xB0+reg (8-bit) / 0xB8+reg (16/32/64, imm64 for Q).
|
||||
v := int64(src)
|
||||
// The Go assembler compresses 64-bit moves whose immediate fits
|
||||
// a signed int32, choosing per sign:
|
||||
// v >= 0: B8+rd imm32 without REX.W (zero-extended by the
|
||||
// hardware, REX.B still emitted for R8-R15);
|
||||
// v < 0: REX.W C7 /0 imm32 (sign-extended — the plain B8+rd
|
||||
// form would zero-extend and corrupt the value).
|
||||
// Out-of-range immediates keep the B8+rd imm64 form.
|
||||
if size == 8 && v >= 0 && v <= (1<<31)-1 {
|
||||
i := newInstr(4, []byte{0xB8 + byte(dstReg.idx&7)})
|
||||
i.rexB = dstReg.idx >= 8
|
||||
i.imm = le32(v)
|
||||
return e.emit(i)
|
||||
}
|
||||
if size == 8 && v < 0 && v >= -(1<<31) {
|
||||
i := newInstr(8, []byte{0xC7})
|
||||
if err := setRMDigit(i, 0, dstReg, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = le32(v)
|
||||
return e.emit(i)
|
||||
}
|
||||
opBase := byte(0xB8)
|
||||
if size == 1 {
|
||||
opBase = 0xB0
|
||||
@@ -90,7 +173,7 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
||||
if dstReg.needsREX(size) {
|
||||
i.rexForced = true
|
||||
}
|
||||
i.imm = immediate(int64(src), size, true)
|
||||
i.imm = immediate(v, size, true)
|
||||
return e.emit(i)
|
||||
}
|
||||
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
|
||||
@@ -133,7 +216,13 @@ func (e *enc) encodeALU(op struct {
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
|
||||
// CMP never takes its immediate first: the Go assembler rejects
|
||||
// CMPL $0, AX outright (only CMPL AX, $0 is legal, unlike TEST and the
|
||||
// writing ALU ops whose immediate is naturally the source).
|
||||
if imm, ok := src.(Imm); ok {
|
||||
if op.digit == 7 {
|
||||
return fmt.Errorf("CMP immediate must be the second operand (reg, $imm)")
|
||||
}
|
||||
return e.encodeALUImm(op.digit, dst, int64(imm), size)
|
||||
}
|
||||
|
||||
@@ -224,6 +313,15 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
||||
i.imm = []byte{byte(int8(imm))}
|
||||
return e.emit(i)
|
||||
}
|
||||
// 0x81 /digit, imm16/imm32 — or the Go assembler's accumulator short
|
||||
// form (opcode+5, no ModR/M) when the destination is AX/AL, which it
|
||||
// prefers over the generic form exactly here.
|
||||
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
||||
accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit]
|
||||
i := newInstr(size, []byte{accOp})
|
||||
i.imm = immediate(imm, size, false)
|
||||
return e.emit(i)
|
||||
}
|
||||
// 0x81 /digit, imm16/imm32.
|
||||
i := newInstr(size, []byte{0x81})
|
||||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||||
@@ -241,7 +339,18 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
if imm, ok := src.(Imm); ok {
|
||||
// TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0.
|
||||
// TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0 — but the Go assembler
|
||||
// always uses the accumulator forms (A8/A9, no ModR/M) when the
|
||||
// register operand is AL/AX, whatever the immediate's width.
|
||||
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
||||
op := byte(0xA9)
|
||||
if size == 1 {
|
||||
op = 0xA8
|
||||
}
|
||||
i := newInstr(size, []byte{op})
|
||||
i.imm = immediate(int64(imm), size, false)
|
||||
return e.emit(i)
|
||||
}
|
||||
op := byte(0xF7)
|
||||
if size == 1 {
|
||||
op = 0xF6
|
||||
@@ -280,12 +389,13 @@ func (e *enc) encodeLea(ops []Operand, size int) error {
|
||||
if !ok {
|
||||
return fmt.Errorf("LEA: destination must be a register")
|
||||
}
|
||||
mem, ok := src.(Mem)
|
||||
if !ok {
|
||||
switch src.(type) {
|
||||
case Mem, sbMem:
|
||||
default:
|
||||
return fmt.Errorf("LEA: source must be a memory operand")
|
||||
}
|
||||
i := newInstr(size, []byte{0x8D})
|
||||
if err := setRM(i, dstReg, mem, size); err != nil {
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
@@ -585,31 +695,60 @@ func (e *enc) encodeSet(upper string, ops []Operand) error {
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- LZCNT / TZCNT ----------------------------------------------------------
|
||||
// --- bit scan / bit count ----------------------------------------------------
|
||||
|
||||
// encodeCount encodes LZCNT/TZCNT (leading / trailing zero count): F3 0F BD
|
||||
// or F3 0F BC, with reg = dst and rm = src. The size suffix selects the
|
||||
// operand width (LZCNTW/LZCNTL/LZCNTQ).
|
||||
// countOp maps the bit-scan and bit-count mnemonics to their opcode byte and
|
||||
// mandatory prefix. TZCNT/LZCNT/POPCNT are the F3-prefixed forms of the
|
||||
// same map as BSF/BSR's 0F BC/BD; POPCNT is F3 0F B8.
|
||||
var countOp = map[string]struct {
|
||||
op byte
|
||||
prefix byte
|
||||
}{
|
||||
"BSF": {0xBC, 0},
|
||||
"BSR": {0xBD, 0},
|
||||
"TZCNT": {0xBC, 0xF3},
|
||||
"LZCNT": {0xBD, 0xF3},
|
||||
"POPCNT": {0xB8, 0xF3},
|
||||
}
|
||||
|
||||
// encodeCount encodes the bit-scan and bit-count family — BSF (0F BC),
|
||||
// BSR (0F BD), TZCNT (F3 0F BC), LZCNT (F3 0F BD) and POPCNT (F3 0F B8) —
|
||||
// with reg = dst and rm = src. The size suffix selects the operand width
|
||||
// (BSFQ, TZCNTL, …). Note BSF/BSR leave the destination undefined when the
|
||||
// source is zero (unlike their F3-prefixed counterparts); callers must
|
||||
// guard non-zero inputs themselves.
|
||||
func (e *enc) encodeCount(base string, ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||||
}
|
||||
op := byte(0xBD)
|
||||
if base == "TZCNT" {
|
||||
op = 0xBC
|
||||
}
|
||||
spec := countOp[base]
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s destination must be a register", base)
|
||||
}
|
||||
i := newInstr(size, []byte{0x0F, op})
|
||||
i.prefix = 0xF3
|
||||
i := newInstr(size, []byte{0x0F, spec.op})
|
||||
i.prefix = spec.prefix
|
||||
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeBswap encodes BSWAP: the single register operand is encoded in the
|
||||
// opcode byte (0F C8+r), with REX.B for R8-R15 and REX.W for the quad form.
|
||||
func (e *enc) encodeBswap(ops []Operand, size int) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("BSWAP expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
reg, ok := ops[0].(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("BSWAP operand must be a register")
|
||||
}
|
||||
i := newInstr(size, []byte{0x0F, 0xC8 + byte(reg.idx&7)})
|
||||
i.rexB = reg.idx >= 8
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- mixed-width sign/zero-extending moves -----------------------------------
|
||||
|
||||
// movExtendOp maps Go's mixed-width move names to their opcode and destination
|
||||
@@ -649,6 +788,172 @@ func (e *enc) encodeMovExtend(base string, ops []Operand) error {
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- legacy SSE moves --------------------------------------------------------
|
||||
|
||||
// sseMove describes a legacy (non-VEX) SSE move: a mandatory prefix plus a
|
||||
// load opcode (reg = destination, rm = source) and a store opcode (the
|
||||
// reverse). The Plan 9 names MOVOU/MOVO are the integer unaligned/aligned
|
||||
// octa moves (MOVDQU/MOVDQA), not the packed-single ones.
|
||||
type sseMove struct {
|
||||
prefix byte // 0, 0x66, 0xF2 or 0xF3
|
||||
load byte
|
||||
store byte
|
||||
}
|
||||
|
||||
var sseMoveTable = map[string]sseMove{
|
||||
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU — unaligned octa
|
||||
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA — aligned octa
|
||||
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
|
||||
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
|
||||
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
|
||||
"MOVAPD": {0x66, 0x28, 0x29}, // aligned packed double
|
||||
"MOVSD": {0xF2, 0x10, 0x11}, // scalar double
|
||||
"MOVSS": {0xF3, 0x10, 0x11}, // scalar single
|
||||
}
|
||||
|
||||
// encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the
|
||||
// load form (reg = destination), matching the Go assembler.
|
||||
func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("SSE move expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
srcReg, srcVec := vecReg(src)
|
||||
dstReg, dstVec := vecReg(dst)
|
||||
op := m.store
|
||||
var reg Reg
|
||||
var rm Operand
|
||||
switch {
|
||||
case srcVec && dstVec:
|
||||
op = m.load
|
||||
reg, rm = dstReg, src
|
||||
case srcVec:
|
||||
if _, ok := dst.(Mem); !ok {
|
||||
return fmt.Errorf("SSE move: invalid destination operand")
|
||||
}
|
||||
reg, rm = srcReg, dst
|
||||
case dstVec:
|
||||
if _, ok := src.(Mem); !ok {
|
||||
return fmt.Errorf("SSE move: invalid source operand")
|
||||
}
|
||||
op = m.load
|
||||
reg, rm = dstReg, src
|
||||
default:
|
||||
return fmt.Errorf("SSE move needs a vector register operand")
|
||||
}
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, reg, rm, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- legacy SSE packed binary and shuffles -----------------------------------
|
||||
|
||||
// sseBin describes a legacy (non-VEX) SSE packed/scalar binary op: an
|
||||
// optional mandatory prefix plus the 0F-prefixed opcode (0F38 for the
|
||||
// SSSE3 integer shuffles). Plan 9 asm lists the source operand first, so
|
||||
// MULPS X0, X1 computes X1 = X1 * X0.
|
||||
type sseBin struct {
|
||||
prefix byte // 0, 0x66, 0xF2 or 0xF3
|
||||
op byte
|
||||
map38 bool // opcode lives under 0F38 instead of 0F
|
||||
}
|
||||
|
||||
var sseBinTable = map[string]sseBin{
|
||||
"ADDPS": {0, 0x58, false}, "ADDPD": {0x66, 0x58, false},
|
||||
"MULPS": {0, 0x59, false}, "MULPD": {0x66, 0x59, false},
|
||||
"SUBPS": {0, 0x5C, false}, "SUBPD": {0x66, 0x5C, false},
|
||||
"DIVPS": {0, 0x5E, false}, "DIVPD": {0x66, 0x5E, false},
|
||||
"ANDPS": {0, 0x54, false}, "ANDPD": {0x66, 0x54, false},
|
||||
"ORPS": {0, 0x56, false}, "ORPD": {0x66, 0x56, false},
|
||||
"XORPS": {0, 0x57, false}, "XORPD": {0x66, 0x57, false},
|
||||
"MINPS": {0, 0x5D, false}, "MINPD": {0x66, 0x5D, false},
|
||||
"MAXPS": {0, 0x5F, false}, "MAXPD": {0x66, 0x5F, false},
|
||||
"ADDSS": {0xF3, 0x58, false}, "ADDSD": {0xF2, 0x58, false},
|
||||
"MULSS": {0xF3, 0x59, false}, "MULSD": {0xF2, 0x59, false},
|
||||
"SUBSS": {0xF3, 0x5C, false}, "SUBSD": {0xF2, 0x5C, false},
|
||||
"DIVSS": {0xF3, 0x5E, false}, "DIVSD": {0xF2, 0x5E, false},
|
||||
"MINSS": {0xF3, 0x5D, false}, "MINSD": {0xF2, 0x5D, false},
|
||||
"MAXSS": {0xF3, 0x5F, false}, "MAXSD": {0xF2, 0x5F, false},
|
||||
"UNPCKLPS": {0, 0x14, false}, "UNPCKHPS": {0, 0x15, false},
|
||||
"UNPCKLPD": {0x66, 0x14, false}, "UNPCKHPD": {0x66, 0x15, false},
|
||||
"CVTSS2SD": {0xF3, 0x5A, false}, "CVTSD2SS": {0xF2, 0x5A, false},
|
||||
"CVTPS2PD": {0, 0x5A, false}, "CVTPD2PS": {0x66, 0x5A, false},
|
||||
// SSE2 packed integers (reg = reg op rm) and the SSSE3 byte shuffle.
|
||||
"PXOR": {0x66, 0xEF, false},
|
||||
"POR": {0x66, 0xEB, false},
|
||||
"PAND": {0x66, 0xDB, false},
|
||||
"PANDN": {0x66, 0xDF, false},
|
||||
"PADDB": {0x66, 0xFC, false}, "PADDW": {0x66, 0xFD, false},
|
||||
"PADDD": {0x66, 0xFE, false}, "PADDQ": {0x66, 0xD4, false},
|
||||
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
|
||||
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
|
||||
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
|
||||
"PCMPEQD": {0x66, 0x76, false},
|
||||
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
|
||||
"PCMPGTD": {0x66, 0x66, false},
|
||||
"PSHUFB": {0x66, 0x00, true},
|
||||
}
|
||||
|
||||
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
|
||||
// (PSHUFD/PSHUFHW/PSHUFLW also carry the packed-int 0x66/F3/F2 prefixes).
|
||||
type sseShuf struct {
|
||||
prefix byte
|
||||
op byte
|
||||
}
|
||||
|
||||
var sseShufTable = map[string]sseShuf{
|
||||
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
|
||||
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
|
||||
}
|
||||
|
||||
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
|
||||
func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("SSE binary expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("SSE binary destination must be a vector register")
|
||||
}
|
||||
opcode := []byte{0x0F, m.op}
|
||||
if m.map38 {
|
||||
opcode = []byte{0x0F, 0x38, m.op}
|
||||
}
|
||||
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dstReg, src, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeSSEShuf encodes an imm8 shuffle: SHUFPS $imm, src, dst.
|
||||
func (e *enc) encodeSSEShuf(m sseShuf, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("SSE shuffle expects 3 operands, got %d", len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("SSE shuffle needs an imm8 first operand")
|
||||
}
|
||||
if imm < -128 || imm > 255 {
|
||||
return fmt.Errorf("SSE shuffle imm8 %d out of range", imm)
|
||||
}
|
||||
src, dst := ops[1], ops[2]
|
||||
dstReg, ok2 := dst.(Reg)
|
||||
if !ok2 || !dstReg.isVec() {
|
||||
return fmt.Errorf("SSE shuffle destination must be a vector register")
|
||||
}
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dstReg, src, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{byte(int8(imm))}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- CVTSL2SD / CVTSQ2SD -----------------------------------------------------
|
||||
|
||||
// encodeCvtsi2sd encodes a signed integer to scalar double conversion
|
||||
|
||||
@@ -0,0 +1,159 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build integration
|
||||
|
||||
// Package asm integration tests against the production go-libraries kernels.
|
||||
// These are excluded from the default test run (go test ./...) so that the
|
||||
// coverage numbers are identical locally and in CI, where go-libraries is
|
||||
// not checked out. Run them explicitly with: go test -tags=integration ./asm/
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"os"
|
||||
"testing"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel —
|
||||
// all functions plus the file-local mask24 constant — and checks that every
|
||||
// static-symbol load resolves to the right bytes in the image.
|
||||
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
||||
path := "../../go-libraries/go-flac/avx2_amd64.s"
|
||||
if _, err := os.Stat(path); err != nil {
|
||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
||||
}
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f, errs := parser.Parse(path, string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
if len(img.Funcs) != 17 {
|
||||
t.Errorf("functions = %d, want 17", len(img.Funcs))
|
||||
}
|
||||
|
||||
// mask24 as the DATA directives define it.
|
||||
mask := []byte{
|
||||
0x00, 0x01, 0x02, 0x80, 0x03, 0x04, 0x05, 0x80,
|
||||
0x06, 0x07, 0x08, 0x80, 0x09, 0x0a, 0x0b, 0x80,
|
||||
}
|
||||
image := img.Bytes()
|
||||
if got := image[img.Symbols["mask24"] : img.Symbols["mask24"]+16]; !bytes.Equal(got, mask) {
|
||||
t.Errorf("mask24 contents %x, want %x", got, mask)
|
||||
}
|
||||
|
||||
// Every VMOVDQU mask24<>(SB), X15 (c5 7a 6f 3d + rel32, i.e. a VMOVDQU
|
||||
// with a RIP-relative r/m) must land on the mask bytes within the image.
|
||||
loads := 0
|
||||
for _, fn := range img.Funcs {
|
||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
for pc := 0; pc < len(code); {
|
||||
inst, err := x86asm.Decode(code[pc:], 64)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: decode at +%d: %v", fn.Name, pc, err)
|
||||
}
|
||||
// mod=00, rm=101 → RIP-relative.
|
||||
if inst.Op == x86asm.VMOVDQU && inst.Len == 8 && code[pc+3]&0xC7 == 0x05 {
|
||||
rel := int32(uint32(code[pc+4]) | uint32(code[pc+5])<<8 | uint32(code[pc+6])<<16 | uint32(code[pc+7])<<24)
|
||||
target := fn.Offset + pc + 8 + int(rel)
|
||||
if !bytes.Equal(image[target:target+16], mask) {
|
||||
t.Errorf("%s: mask load at +%d lands on %x, want %x", fn.Name, pc, image[target:target+16], mask)
|
||||
}
|
||||
loads++
|
||||
}
|
||||
pc += inst.Len
|
||||
}
|
||||
}
|
||||
if loads != 2 {
|
||||
t.Errorf("mask loads found = %d, want 2", loads)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
|
||||
// kernel — all functions plus the file-global idx16 constant — and checks
|
||||
// that the static-symbol load resolves to the right bytes in the image.
|
||||
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
|
||||
path := "../../go-libraries/go-flac/avx512_amd64.s"
|
||||
if _, err := os.Stat(path); err != nil {
|
||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
||||
}
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f, errs := parser.Parse(path, string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
if len(img.Funcs) != 10 {
|
||||
t.Errorf("functions = %d, want 10", len(img.Funcs))
|
||||
}
|
||||
|
||||
// idx16 as the DATA directives define it: dwords 1..16.
|
||||
idx := make([]byte, 0, 64)
|
||||
for i := 1; i <= 16; i++ {
|
||||
idx = append(idx, byte(i), 0, 0, 0)
|
||||
}
|
||||
image := img.Bytes()
|
||||
base := img.Symbols["idx16"]
|
||||
if base == 0 {
|
||||
t.Fatal("idx16 not laid out")
|
||||
}
|
||||
if got := image[base : base+64]; hexCompact(got) != hexCompact(idx) {
|
||||
t.Errorf("idx16 contents %x, want %x", got, idx)
|
||||
}
|
||||
|
||||
// The VMOVDQU32 idx16(SB), Z13 load (62 71 7e 48 6f 2d + rel32) must
|
||||
// resolve to idx16 within the image.
|
||||
loads := 0
|
||||
for _, fn := range img.Funcs {
|
||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
pat := []byte{0x62, 0x71, 0x7e, 0x48, 0x6f, 0x2d}
|
||||
for pos := 0; ; {
|
||||
i := indexOf(code[pos:], pat)
|
||||
if i < 0 {
|
||||
break
|
||||
}
|
||||
i += pos
|
||||
rel := int32(uint32(code[i+6]) | uint32(code[i+7])<<8 | uint32(code[i+8])<<16 | uint32(code[i+9])<<24)
|
||||
target := fn.Offset + i + 10 + int(rel)
|
||||
if target != base {
|
||||
t.Errorf("%s: idx16 load at +%d targets 0x%x, want 0x%x", fn.Name, i, target, base)
|
||||
}
|
||||
loads++
|
||||
pos = i + 10
|
||||
}
|
||||
}
|
||||
if loads != 1 {
|
||||
t.Errorf("idx16 loads found = %d, want 1", loads)
|
||||
}
|
||||
}
|
||||
|
||||
// indexOf returns the index of the first occurrence of pat in b, or -1.
|
||||
func indexOf(b, pat []byte) int {
|
||||
for i := 0; i+len(pat) <= len(b); i++ {
|
||||
j := 0
|
||||
for j < len(pat) && b[i+j] == pat[j] {
|
||||
j++
|
||||
}
|
||||
if j == len(pat) {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
@@ -0,0 +1,326 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
|
||||
// the symbol tables, the function code bytes and the relocation wiring.
|
||||
func TestGOObjectLOONG64Structure(t *testing.T) {
|
||||
f, errs := parser.Parse("k_loong64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVV a+0(FP), R4
|
||||
MOVV b+8(FP), R5
|
||||
ADDV R5, R4, R4
|
||||
MOVV R4, ret+16(FP)
|
||||
RET
|
||||
|
||||
GLOBL ·table<>(SB), RODATA, $8
|
||||
DATA ·table<>+0(SB)/8, $0x1122334455667788
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileLOONG64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||
}
|
||||
obj, err := img.GOObjectLOONG64("testpkg", "k_loong64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObjectLOONG64: %v", err)
|
||||
}
|
||||
v := openGoobj(t, obj)
|
||||
|
||||
// Package defs: the static GLOBL, then the FuncInfo and the two DWARF
|
||||
// symbols (debug_line program, subprogram DIE).
|
||||
defs := v.syms(blkSymdef)
|
||||
if len(defs) != 4 {
|
||||
t.Fatalf("symdefs = %d, want 4", len(defs))
|
||||
}
|
||||
if defs[0].name != "table" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 8 {
|
||||
t.Errorf("table symbol = %+v", defs[0])
|
||||
}
|
||||
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
|
||||
t.Errorf("funcinfo symbol = %+v", defs[1])
|
||||
}
|
||||
if defs[2].name != "" || defs[2].typ != kindSDWARFLINES || defs[2].size == 0 {
|
||||
t.Errorf("lines symbol = %+v", defs[2])
|
||||
}
|
||||
if defs[3].name != "" || defs[3].typ != kindSDWARFFCN || defs[3].size == 0 {
|
||||
t.Errorf("DIE symbol = %+v", defs[3])
|
||||
}
|
||||
|
||||
// Non-package defs: four pc tables and the function.
|
||||
nps := v.syms(blkNonpkgdef)
|
||||
if len(nps) != 5 {
|
||||
t.Fatalf("nonpkgdefs = %d, want 5", len(nps))
|
||||
}
|
||||
fn := nps[4]
|
||||
if fn.name != "testpkg.add" || fn.typ != kindSTEXT || fn.flag != symFlagNoSplit || fn.size != 20 {
|
||||
t.Errorf("add symbol = %+v", fn)
|
||||
}
|
||||
|
||||
// The function code: 20 bytes, the ground-truth encoding. It sits
|
||||
// after the GLOBL, FuncInfo, two DWARF symbols and four pc tables.
|
||||
dataIdx := v.blk(blkDataIdx)
|
||||
dataBlk := v.blk(blkData)
|
||||
le := binary.LittleEndian
|
||||
dOff := le.Uint32(dataIdx[8*4:])
|
||||
code := dataBlk[dOff : dOff+20]
|
||||
want := []byte{
|
||||
0x64, 0x20, 0xc0, 0x28, // ld.d r4, 8(r3)
|
||||
0x65, 0x40, 0xc0, 0x28, // ld.d r5, 16(r3)
|
||||
0x84, 0x94, 0x10, 0x00, // add.d r4, r4, r5
|
||||
0x64, 0x60, 0xc0, 0x29, // st.d r4, 24(r3)
|
||||
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
|
||||
}
|
||||
for i := range want {
|
||||
if code[i] != want[i] {
|
||||
t.Fatalf("code byte %d = %02x, want %02x", i, code[i], want[i])
|
||||
}
|
||||
}
|
||||
|
||||
// The debug_line program: LNE_set_address (the R_ADDR relocation
|
||||
// carries the function address), then one row per line change — the
|
||||
// TEXT is on line 4 (a leading blank line precedes the include), the
|
||||
// instructions on lines 5–9 — an advance to the 20-byte end and an
|
||||
// end-of-sequence.
|
||||
linesOff := le.Uint32(dataIdx[4*2:])
|
||||
lines := dataBlk[linesOff : linesOff+21]
|
||||
wantLines := []byte{
|
||||
0x00, 0x09, 0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, // LNE_set_address
|
||||
0x13, // pc 0, line 5
|
||||
0x38, // pc 4, line 6
|
||||
0x38, // pc 8, line 7
|
||||
0x38, // pc 12, line 8
|
||||
0x38, // pc 16, line 9
|
||||
0x02, 0x04, // advance_pc to 20
|
||||
0x00, 0x01, 0x01, // end_sequence
|
||||
}
|
||||
for i := range wantLines {
|
||||
if lines[i] != wantLines[i] {
|
||||
t.Fatalf("lines byte %d = %02x, want %02x", i, lines[i], wantLines[i])
|
||||
}
|
||||
}
|
||||
|
||||
// The subprogram DIE: abbrev 3 (FUNCTION), the qualified name, the
|
||||
// addrx low_pc slot (R_DWTXTADDR_U4), the size as high_pc, the
|
||||
// call-frame-CFA frame base, decl file/line and the external flag.
|
||||
dieOff := le.Uint32(dataIdx[4*3:])
|
||||
die := dataBlk[dieOff : dieOff+27]
|
||||
wantDie := []byte{
|
||||
0x03,
|
||||
't', 'e', 's', 't', 'p', 'k', 'g', '.', 'a', 'd', 'd', 0,
|
||||
0x00, 0x00, 0x00, 0x00, // low_pc: addrx slot
|
||||
0x14, // high_pc: 20
|
||||
0x01, 0x9c, // frame_base: DW_OP_call_frame_cfa
|
||||
0x01, 0x00, 0x00, 0x00, // decl_file: 1
|
||||
0x04, // decl_line: 4
|
||||
0x01, // external
|
||||
0x00, // end of children
|
||||
}
|
||||
for i := range wantDie {
|
||||
if die[i] != wantDie[i] {
|
||||
t.Fatalf("DIE byte %d = %02x, want %02x", i, die[i], wantDie[i])
|
||||
}
|
||||
}
|
||||
|
||||
// The DWARF symbols carry the function-address references: R_ADDR for
|
||||
// the line program's set_address, R_DWTXTADDR_U4 for the DIE's addrx
|
||||
// slot, both against the function's non-package index. The reloc
|
||||
// index counts relocations, not bytes.
|
||||
relocIdx := v.blk(blkRelocIdx)
|
||||
relocs := v.blk(blkReloc)
|
||||
if le.Uint32(relocIdx[4*2:]) != 0 || le.Uint32(relocIdx[4*3:]) != 1 || le.Uint32(relocIdx[4*4:]) != 2 {
|
||||
t.Fatalf("dwarf reloc index ranges: %d %d %d", le.Uint32(relocIdx[4*2:]), le.Uint32(relocIdx[4*3:]), le.Uint32(relocIdx[4*4:]))
|
||||
}
|
||||
lr := relocs[:23]
|
||||
if int32(le.Uint32(lr[0:])) != 3 || lr[4] != 8 || le.Uint16(lr[5:]) != relocAddr ||
|
||||
le.Uint32(lr[15:]) != pkgIdxNone || le.Uint32(lr[19:]) != 4 {
|
||||
t.Errorf("lines reloc = %x", lr)
|
||||
}
|
||||
dr := relocs[23:46]
|
||||
if int32(le.Uint32(dr[0:])) != 13 || dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4() ||
|
||||
le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 {
|
||||
t.Errorf("die reloc = %x", dr)
|
||||
}
|
||||
|
||||
// The pc-value deltas are in MinLC (4) units: the flat pcsp covers
|
||||
// the whole 20-byte function with a delta of 5.
|
||||
pcspOff := le.Uint32(dataIdx[4*4:])
|
||||
if got := dataBlk[pcspOff : pcspOff+3]; !bytes.Equal(got, []byte{0x02, 0x05, 0x00}) {
|
||||
t.Errorf("pcsp = %x, want 020500", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectLOONG64Link cross-compiles a Go program with the gasm-produced
|
||||
// object substituted into the package archive, proving cmd/link accepts the
|
||||
// emitted GOOBJ. The binary is not executed (no LoongArch host or qemu).
|
||||
// Skipped when no Go toolchain is available.
|
||||
func TestGOObjectLOONG64Link(t *testing.T) {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
t.Skip("no Go toolchain available")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
asmSrc := `#include "textflag.h"
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVV a+0(FP), R4
|
||||
MOVV b+8(FP), R5
|
||||
ADDV R5, R4, R4
|
||||
MOVV R4, ret+16(FP)
|
||||
RET
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main_loong64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
mainSrc := `package main
|
||||
|
||||
func add(a, b int64) int64
|
||||
|
||||
func main() {
|
||||
if add(20, 22) != 42 {
|
||||
panic("bad add")
|
||||
}
|
||||
}
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module l64link\n\ngo 1.21\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Capture the cross build (GOARCH=loong64): the package archive and the
|
||||
// link line.
|
||||
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||
build.Dir = dir
|
||||
build.Env = append(os.Environ(), "GOARCH=loong64")
|
||||
buildLog, err := build.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||
}
|
||||
var pkgArch, work, linkLine, asmObj string
|
||||
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_loong64.s") && !strings.Contains(line, "-gensymabis"):
|
||||
asmObj = fieldAfter(line, "-o")
|
||||
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
|
||||
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
|
||||
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
|
||||
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||
linkLine = line
|
||||
}
|
||||
}
|
||||
if pkgArch == "" || linkLine == "" || asmObj == "" {
|
||||
t.Skip("could not locate the archive, asm output or link line in the build log")
|
||||
}
|
||||
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
|
||||
// The archive member holding the assembler's output is named after the
|
||||
// asm object file (main_loong64.o), as cmd/go packs it with `pack r`.
|
||||
asmMember := filepath.Base(strings.ReplaceAll(asmObj, "$WORK", work))
|
||||
|
||||
// Assemble the same source with gasm and swap the object in.
|
||||
pf, perrs := parser.Parse(filepath.Join(dir, "main_loong64.s"), asmSrc)
|
||||
if len(perrs) > 0 {
|
||||
t.Fatalf("parse: %v", perrs)
|
||||
}
|
||||
pimg, err := AssembleFileLOONG64(pf)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||
}
|
||||
obj, err := pimg.GOObjectLOONG64("main", filepath.Join(dir, "main_loong64.s"))
|
||||
if err != nil {
|
||||
t.Fatalf("GOObjectLOONG64: %v", err)
|
||||
}
|
||||
|
||||
// Extract the archive, substitute the object member, repack.
|
||||
membersDir := filepath.Join(dir, "members")
|
||||
if err := os.MkdirAll(membersDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
|
||||
extract.Dir = membersDir
|
||||
extract.Env = append(os.Environ(), "GOARCH=loong64")
|
||||
if out, err := extract.CombinedOutput(); err != nil {
|
||||
t.Fatalf("pack x: %v\n%s", err, out)
|
||||
}
|
||||
// Substitute the gasm object for the assembler's archive member (pack
|
||||
// extracts members read-only).
|
||||
member := filepath.Join(membersDir, asmMember)
|
||||
if err := os.Chmod(member, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(member, obj, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
|
||||
listCmd.Env = append(os.Environ(), "GOARCH=loong64")
|
||||
listOut, err := listCmd.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("pack t: %v\n%s", err, listOut)
|
||||
}
|
||||
newArch := filepath.Join(dir, "pkg.a")
|
||||
args := []string{"tool", "pack", "c", newArch}
|
||||
seen := map[string]bool{}
|
||||
for m := range strings.FieldsSeq(string(listOut)) {
|
||||
if seen[m] {
|
||||
continue
|
||||
}
|
||||
seen[m] = true
|
||||
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
args = append(args, filepath.Join(membersDir, m))
|
||||
}
|
||||
pack := exec.Command(goBin, args...)
|
||||
pack.Dir = membersDir
|
||||
pack.Env = append(os.Environ(), "GOARCH=loong64")
|
||||
if out, err := pack.CombinedOutput(); err != nil {
|
||||
t.Fatalf("pack c: %v\n%s", err, out)
|
||||
}
|
||||
|
||||
// Re-link with our archive in place of the toolchain's. The link line
|
||||
// carries a GOROOT assignment and $WORK placeholders; run it through the
|
||||
// shell with the GOEXPERIMENT and GOARCH the toolchain expects (the
|
||||
// linker compares the object header against its own, experiments
|
||||
// included).
|
||||
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
|
||||
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
|
||||
link := exec.Command("sh", "-c", linkLine)
|
||||
link.Dir = dir
|
||||
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
|
||||
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)), "GOARCH=loong64")
|
||||
if out, err := link.CombinedOutput(); err != nil {
|
||||
t.Fatalf("link with gasm object: %v\n%s", err, out)
|
||||
}
|
||||
|
||||
// The binary is not executed: there is no LoongArch host or qemu here.
|
||||
// The link itself and the symbol table prove cmd/link accepted the gasm
|
||||
// object and laid out the function.
|
||||
nm := exec.Command(goBin, "tool", "nm", filepath.Join(dir, "app2"))
|
||||
nm.Env = append(os.Environ(), "GOARCH=loong64")
|
||||
nmOut, err := nm.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("nm gasm-linked binary: %v\n%s", err, nmOut)
|
||||
}
|
||||
if !strings.Contains(string(nmOut), "main.add") {
|
||||
t.Errorf("main.add not found in linked binary:\n%s", nmOut)
|
||||
}
|
||||
}
|
||||
+518
@@ -0,0 +1,518 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
)
|
||||
|
||||
// Image is an assembled file: the function bodies laid out in source order,
|
||||
// followed by the file's static data section (GLOBL/DATA). References to
|
||||
// file-local static symbols are encoded RIP-relative and resolved within the
|
||||
// image, so the raw bytes are self-consistent and executable at any base
|
||||
// address; references to external symbols are recorded as relocations
|
||||
// (Funcs[i].Relocs, Externals) and left unresolved — the object-file
|
||||
// emitters turn them into linker relocations.
|
||||
type Image struct {
|
||||
Code []byte // concatenated function bodies
|
||||
Data []byte // static data section
|
||||
Funcs []FuncLayout // function positions, in source order
|
||||
Symbols map[string]int // static symbol → byte offset within the image
|
||||
DataSyms []DataSymbol // GLOBL symbols, in layout order
|
||||
Externals []string // referenced but undefined symbols, sorted
|
||||
}
|
||||
|
||||
// FuncLayout describes one assembled function within an Image.
|
||||
type FuncLayout struct {
|
||||
Name string
|
||||
Pkg string // explicit package prefix ("" = the current package)
|
||||
Static bool // the <> marker: file-local, not exported
|
||||
Offset int // start offset within the image (== offset within Code)
|
||||
Size int
|
||||
Args int // declared argument/result area (the TEXT size suffix)
|
||||
Frame int // local frame size (the TEXT $framesize)
|
||||
NoSplit bool // the NOSPLIT flag
|
||||
SPWrite bool // the SPWRITE flag: writes an arbitrary value to SP
|
||||
Line int // source line of the TEXT directive
|
||||
Labels map[string]int // local labels, function-relative
|
||||
Relocs []Reloc // static-symbol references, in emission order
|
||||
Spadj []SpadjStep // stack-adjustment boundaries, ascending by PC
|
||||
Lines []LineEntry // source-line table: byte offset → source line
|
||||
}
|
||||
|
||||
// SpadjStep is one stack-adjustment boundary: Value is the SP delta from the
|
||||
// entry state in effect from PC (function-relative) until the next step.
|
||||
type SpadjStep struct {
|
||||
PC int
|
||||
Value int
|
||||
}
|
||||
|
||||
// LineEntry maps a byte offset (function-relative) to a source line number.
|
||||
type LineEntry struct {
|
||||
Offset int
|
||||
Line int
|
||||
}
|
||||
|
||||
// LineAt returns the source line number for the given function-relative byte
|
||||
// offset, using a binary search on the line table. Returns 0 if the offset
|
||||
// is before the first instruction or the table is empty.
|
||||
func (fl *FuncLayout) LineAt(offset int) int {
|
||||
if len(fl.Lines) == 0 {
|
||||
return 0
|
||||
}
|
||||
// Binary search: find the last entry with Offset <= offset.
|
||||
lo, hi := 0, len(fl.Lines)-1
|
||||
for lo < hi {
|
||||
mid := (lo + hi + 1) / 2
|
||||
if fl.Lines[mid].Offset <= offset {
|
||||
lo = mid
|
||||
} else {
|
||||
hi = mid - 1
|
||||
}
|
||||
}
|
||||
if fl.Lines[lo].Offset <= offset {
|
||||
return fl.Lines[lo].Line
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// RelocKind Reloc is one static-symbol reference within a function body: the disp32
|
||||
// field at Off (function-relative) must reach the symbol plus Addend,
|
||||
// measured from After, the address just past the instruction. An External
|
||||
// relocation names a symbol no GLOBL in the file defines; the object-file
|
||||
// emitters carry it into the output's relocation table.
|
||||
// RelocKind discriminates the type of relocation needed.
|
||||
type RelocKind int
|
||||
|
||||
const (
|
||||
RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64)
|
||||
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
|
||||
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
|
||||
RelRISCVJal // R_RISCV_JAL (J-type call)
|
||||
RelPCRelAbs // 32-bit absolute (R_RISCV_32)
|
||||
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
|
||||
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
|
||||
RelArm64Addr // R_ADDRARM64 (ADRP + ADD/LDR/STR pair)
|
||||
RelArm64Branch // R_CALLARM64 (BL instruction)
|
||||
)
|
||||
|
||||
type Reloc struct {
|
||||
Off int
|
||||
After int
|
||||
Name string
|
||||
Addend int64
|
||||
External bool
|
||||
Kind RelocKind
|
||||
}
|
||||
|
||||
// DataSymbol describes one GLOBL symbol laid out in the data section.
|
||||
type DataSymbol struct {
|
||||
Name string
|
||||
Pkg string // explicit package prefix ("" = the current package)
|
||||
Offset int // byte offset within Data
|
||||
Size int
|
||||
Static bool // the <> marker: file-local, not exported
|
||||
Rodata bool // the RODATA flag: read-only data
|
||||
Dupok bool // the DUPOK flag: duplicate-OK
|
||||
}
|
||||
|
||||
// Bytes returns the whole image: code, then data.
|
||||
func (img *Image) Bytes() []byte {
|
||||
out := make([]byte, 0, len(img.Code)+len(img.Data))
|
||||
out = append(out, img.Code...)
|
||||
return append(out, img.Data...)
|
||||
}
|
||||
|
||||
// AssembleFile assembles every TEXT function of a parsed file and lays out
|
||||
// its static symbols (GLOBL/DATA) in a data section behind the code. Each
|
||||
// reference to a file-local static symbol becomes a RIP-relative load whose
|
||||
// displacement is resolved against that layout; a reference to a symbol no
|
||||
// GLOBL defines is recorded as an external relocation (Externals) with its
|
||||
// displacement left zero — the object-file emitters resolve it at link
|
||||
// time, while the raw image (Bytes) cannot represent it.
|
||||
func AssembleFile(f *ast.File) (*Image, error) {
|
||||
dataSyms, err := collectData(f)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
known := make(map[string]bool, len(dataSyms))
|
||||
for _, d := range dataSyms {
|
||||
known[d.name] = true
|
||||
}
|
||||
link := &linkInfo{symbols: known, allowExternal: true}
|
||||
|
||||
img := &Image{Symbols: map[string]int{}}
|
||||
type asmFunc struct {
|
||||
name string
|
||||
patches []sbPatch
|
||||
}
|
||||
var funcs []asmFunc
|
||||
for _, d := range f.Decls {
|
||||
t, ok := d.(*ast.Text)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, patches, labels, steps, lines, err := assemble(t, link)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||
}
|
||||
fl := FuncLayout{
|
||||
Name: t.Name.Name,
|
||||
Pkg: t.Name.Pkg,
|
||||
Static: t.Name.Static,
|
||||
Offset: len(img.Code),
|
||||
Size: len(code),
|
||||
Frame: frameSize(t),
|
||||
Args: argsSize(t),
|
||||
Line: t.Pos().Line,
|
||||
Labels: labels,
|
||||
Lines: lines,
|
||||
}
|
||||
for _, f := range t.Flags {
|
||||
switch f {
|
||||
case "NOSPLIT":
|
||||
fl.NoSplit = true
|
||||
case "SPWRITE":
|
||||
fl.SPWrite = true
|
||||
}
|
||||
}
|
||||
for _, s := range steps {
|
||||
fl.Spadj = append(fl.Spadj, SpadjStep{PC: s.pc, Value: s.value})
|
||||
}
|
||||
img.Funcs = append(img.Funcs, fl)
|
||||
img.Code = append(img.Code, code...)
|
||||
funcs = append(funcs, asmFunc{name: t.Name.Name, patches: patches})
|
||||
}
|
||||
|
||||
// Lay out the data section behind the code, each symbol 16-aligned.
|
||||
dataStart := len(img.Code)
|
||||
for _, d := range dataSyms {
|
||||
if pos := dataStart + len(img.Data); pos != align16(pos) {
|
||||
img.Data = append(img.Data, make([]byte, align16(pos)-pos)...)
|
||||
}
|
||||
img.Symbols[d.name] = dataStart + len(img.Data)
|
||||
img.DataSyms = append(img.DataSyms, DataSymbol{
|
||||
Name: d.name,
|
||||
Pkg: d.pkg,
|
||||
Offset: len(img.Data),
|
||||
Size: len(d.buf),
|
||||
Static: d.static,
|
||||
Rodata: d.rodata,
|
||||
Dupok: d.dupok,
|
||||
})
|
||||
img.Data = append(img.Data, d.buf...)
|
||||
}
|
||||
|
||||
// Resolve the RIP-relative displacements of file-local references now
|
||||
// that every address is known, and record every reference (resolved or
|
||||
// external) for the object-file emitters.
|
||||
externals := map[string]bool{}
|
||||
for i, fn := range funcs {
|
||||
base := img.Funcs[i].Offset
|
||||
code := img.Code[base : base+img.Funcs[i].Size]
|
||||
for _, p := range fn.patches {
|
||||
reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend}
|
||||
if imgOff, ok := img.Symbols[p.name]; ok {
|
||||
rel := int64(imgOff) + p.addend - int64(base+p.after)
|
||||
if rel < -1<<31 || rel >= 1<<31 {
|
||||
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
|
||||
}
|
||||
copy(code[p.off:p.off+4], le32(rel))
|
||||
} else {
|
||||
reloc.External = true
|
||||
externals[p.name] = true
|
||||
}
|
||||
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
|
||||
}
|
||||
}
|
||||
for name := range externals {
|
||||
img.Externals = append(img.Externals, name)
|
||||
}
|
||||
sort.Strings(img.Externals)
|
||||
return img, nil
|
||||
}
|
||||
|
||||
// AssembleFileRISCV assembles every TEXT function of a parsed RISC-V file
|
||||
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
|
||||
// code. SB references in the code are encoded as AUIPC pairs with zero
|
||||
// immediates; the object-file emitters record relocations for the linker.
|
||||
func AssembleFileRISCV(f *ast.File) (*Image, error) {
|
||||
dataSyms, err := collectData(f)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
img := &Image{Symbols: map[string]int{}}
|
||||
for _, d := range f.Decls {
|
||||
t, ok := d.(*ast.Text)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, labels, relocs, lines, spadj, err := assembleRISCV(t)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||
}
|
||||
fl := FuncLayout{
|
||||
Name: t.Name.Name,
|
||||
Pkg: t.Name.Pkg,
|
||||
Static: t.Name.Static,
|
||||
Offset: len(img.Code),
|
||||
Size: len(code),
|
||||
Frame: frameSize(t),
|
||||
Args: argsSize(t),
|
||||
Line: t.Pos().Line,
|
||||
Labels: labels,
|
||||
Lines: lines,
|
||||
Spadj: spadj,
|
||||
Relocs: relocs,
|
||||
}
|
||||
for _, f := range t.Flags {
|
||||
switch f {
|
||||
case "NOSPLIT":
|
||||
fl.NoSplit = true
|
||||
case "SPWRITE":
|
||||
fl.SPWrite = true
|
||||
}
|
||||
}
|
||||
img.Funcs = append(img.Funcs, fl)
|
||||
img.Code = append(img.Code, code...)
|
||||
}
|
||||
|
||||
// Lay out the data section behind the code, 16-aligned.
|
||||
dataStart := len(img.Code)
|
||||
for _, d := range dataSyms {
|
||||
pos := dataStart + len(img.Data)
|
||||
for pos%16 != 0 {
|
||||
img.Data = append(img.Data, 0)
|
||||
pos++
|
||||
}
|
||||
img.Symbols[d.name] = pos
|
||||
img.Data = append(img.Data, d.buf...)
|
||||
img.DataSyms = append(img.DataSyms, DataSymbol{
|
||||
Name: d.name,
|
||||
Pkg: d.pkg,
|
||||
Offset: len(img.Data) - len(d.buf), // relative to the data section
|
||||
Size: d.size,
|
||||
Static: d.static,
|
||||
Rodata: d.rodata,
|
||||
Dupok: d.dupok,
|
||||
})
|
||||
}
|
||||
|
||||
markExternals(img, dataSyms)
|
||||
return img, nil
|
||||
}
|
||||
|
||||
// AssembleFileLOONG64 assembles every TEXT function of a parsed loong64 file
|
||||
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
|
||||
// code. SB references in the code are encoded as pcalau12i pairs with zero
|
||||
// immediates; the object-file emitters record R_LOONG64_ADDR_HI/LO
|
||||
// relocations for the linker.
|
||||
func AssembleFileLOONG64(f *ast.File) (*Image, error) {
|
||||
dataSyms, err := collectData(f)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
img := &Image{Symbols: map[string]int{}}
|
||||
for _, d := range f.Decls {
|
||||
t, ok := d.(*ast.Text)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, labels, relocs, lines, spadj, err := assembleLOONG64(t)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||
}
|
||||
fl := FuncLayout{
|
||||
Name: t.Name.Name,
|
||||
Pkg: t.Name.Pkg,
|
||||
Static: t.Name.Static,
|
||||
Offset: len(img.Code),
|
||||
Size: len(code),
|
||||
Frame: frameSize(t),
|
||||
Args: argsSize(t),
|
||||
Line: t.Pos().Line,
|
||||
Labels: labels,
|
||||
Lines: lines,
|
||||
Spadj: spadj,
|
||||
Relocs: relocs,
|
||||
}
|
||||
for _, f := range t.Flags {
|
||||
switch f {
|
||||
case "NOSPLIT":
|
||||
fl.NoSplit = true
|
||||
case "SPWRITE":
|
||||
fl.SPWrite = true
|
||||
}
|
||||
}
|
||||
img.Funcs = append(img.Funcs, fl)
|
||||
img.Code = append(img.Code, code...)
|
||||
}
|
||||
|
||||
// Lay out the data section behind the code, 16-aligned.
|
||||
dataStart := len(img.Code)
|
||||
for _, d := range dataSyms {
|
||||
pos := dataStart + len(img.Data)
|
||||
for pos%16 != 0 {
|
||||
img.Data = append(img.Data, 0)
|
||||
pos++
|
||||
}
|
||||
img.Symbols[d.name] = pos
|
||||
img.Data = append(img.Data, d.buf...)
|
||||
img.DataSyms = append(img.DataSyms, DataSymbol{
|
||||
Name: d.name,
|
||||
Pkg: d.pkg,
|
||||
Offset: len(img.Data) - len(d.buf), // relative to the data section
|
||||
Size: d.size,
|
||||
Static: d.static,
|
||||
Rodata: d.rodata,
|
||||
Dupok: d.dupok,
|
||||
})
|
||||
}
|
||||
|
||||
markExternals(img, dataSyms)
|
||||
return img, nil
|
||||
}
|
||||
|
||||
// markExternals identifies relocations that reference symbols not defined in
|
||||
// the file (neither a GLOBL/DATA symbol nor a TEXT function) and records them
|
||||
// as external. The non-amd64 architectures emit relocations for every SB
|
||||
// reference; this post-processing step distinguishes file-local from external.
|
||||
func markExternals(img *Image, dataSyms []dataSym) {
|
||||
known := make(map[string]bool, len(dataSyms)+len(img.Funcs))
|
||||
for _, d := range dataSyms {
|
||||
known[d.name] = true
|
||||
}
|
||||
for _, fn := range img.Funcs {
|
||||
known[fn.Name] = true
|
||||
}
|
||||
externals := map[string]bool{}
|
||||
for i := range img.Funcs {
|
||||
for j := range img.Funcs[i].Relocs {
|
||||
r := &img.Funcs[i].Relocs[j]
|
||||
if !known[r.Name] {
|
||||
r.External = true
|
||||
externals[r.Name] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
for name := range externals {
|
||||
img.Externals = append(img.Externals, name)
|
||||
}
|
||||
sort.Strings(img.Externals)
|
||||
}
|
||||
|
||||
// dataSym is one GLOBL symbol and its DATA initialiser.
|
||||
type dataSym struct {
|
||||
name string
|
||||
pkg string
|
||||
buf []byte
|
||||
size int
|
||||
static bool
|
||||
rodata bool
|
||||
dupok bool
|
||||
}
|
||||
|
||||
// collectData gathers the file's static symbols (GLOBL) and their initial
|
||||
// contents (DATA) into byte buffers, in declaration order.
|
||||
func collectData(f *ast.File) ([]dataSym, error) {
|
||||
index := map[string]int{}
|
||||
var syms []dataSym
|
||||
for _, d := range f.Decls {
|
||||
switch dd := d.(type) {
|
||||
case *ast.Globl:
|
||||
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
||||
continue
|
||||
}
|
||||
name := dd.Name.Name
|
||||
if _, dup := index[name]; dup {
|
||||
return nil, fmt.Errorf("duplicate GLOBL %q", name)
|
||||
}
|
||||
size := 0
|
||||
if dd.Size != nil && dd.Size.Imm.HasVal {
|
||||
size = int(dd.Size.Imm.Val)
|
||||
}
|
||||
index[name] = len(syms)
|
||||
ds := dataSym{
|
||||
name: name,
|
||||
pkg: dd.Name.Pkg,
|
||||
buf: make([]byte, size),
|
||||
size: size,
|
||||
static: dd.Name.Static,
|
||||
}
|
||||
for _, f := range dd.Flags {
|
||||
switch f {
|
||||
case "RODATA":
|
||||
ds.rodata = true
|
||||
case "DUPOK":
|
||||
ds.dupok = true
|
||||
case "1":
|
||||
ds.dupok = true
|
||||
case "8":
|
||||
ds.rodata = true
|
||||
case "9":
|
||||
ds.dupok = true
|
||||
ds.rodata = true
|
||||
}
|
||||
}
|
||||
syms = append(syms, ds)
|
||||
|
||||
case *ast.Data:
|
||||
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
||||
continue
|
||||
}
|
||||
i, ok := index[dd.Name.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
|
||||
}
|
||||
if dd.Value == nil || !dd.Value.Imm.HasVal {
|
||||
return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name)
|
||||
}
|
||||
w := dd.Width
|
||||
switch w {
|
||||
case 1, 2, 4, 8:
|
||||
default:
|
||||
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
|
||||
}
|
||||
off := dd.Name.Offset
|
||||
buf := syms[i].buf
|
||||
if off < 0 || off+int64(w) > int64(len(buf)) {
|
||||
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
|
||||
}
|
||||
v := dd.Value.Imm.Val
|
||||
if dd.Value.Imm.Neg {
|
||||
v = -v
|
||||
}
|
||||
for j := range w {
|
||||
buf[off+int64(j)] = byte(v >> (8 * j))
|
||||
}
|
||||
}
|
||||
}
|
||||
return syms, nil
|
||||
}
|
||||
|
||||
// align16 rounds n up to the next multiple of 16.
|
||||
func align16(n int) int {
|
||||
return (n + 15) &^ 15
|
||||
}
|
||||
|
||||
// frameSize returns the local frame size declared on the TEXT directive.
|
||||
func frameSize(t *ast.Text) int {
|
||||
if t.Frame != nil && t.Frame.Imm.HasVal {
|
||||
return int(t.Frame.Imm.Val)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// argsSize returns the argument/result area declared on the TEXT directive.
|
||||
func argsSize(t *ast.Text) int {
|
||||
if t.Args != nil && t.Args.Imm.HasVal {
|
||||
return int(t.Args.Imm.Val)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
@@ -0,0 +1,130 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestAssembleFileStaticData checks the whole-image layout — code, padding
|
||||
// and the data section — and that the RIP-relative displacements of static
|
||||
// symbol loads resolve to the right bytes.
|
||||
func TestAssembleFileStaticData(t *testing.T) {
|
||||
f, errs := parser.Parse("d_amd64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·load(SB), NOSPLIT, $0
|
||||
VMOVDQU mask<>(SB), X15
|
||||
MOVL small<>(SB), AX
|
||||
RET
|
||||
GLOBL mask<>(SB), RODATA, $16
|
||||
DATA mask<>+0(SB)/4, $0x80020100
|
||||
DATA mask<>+4(SB)/4, $0x80050403
|
||||
DATA mask<>+8(SB)/4, $0x80080706
|
||||
DATA mask<>+12(SB)/4, $0x800B0A09
|
||||
GLOBL small<>(SB), RODATA, $4
|
||||
DATA small<>+0(SB)/4, $0x1234
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
|
||||
// Code (15 bytes) + 1 pad byte to align the data section to 16:
|
||||
// VMOVDQU mask<>(SB), X15 c5 7a 6f 3d 08 00 00 00 (disp = 16 − 8)
|
||||
// MOVL small<>(SB), AX 8b 05 12 00 00 00 (disp = 32 − 14)
|
||||
// RET c3
|
||||
// Data: pad, mask (16 bytes), small (4 bytes).
|
||||
want := "c57a6f3d080000008b0512000000c300" +
|
||||
"000102800304058006070880090a0b80" +
|
||||
"34120000"
|
||||
if got := strings.ReplaceAll(hexBytes(img.Bytes()), " ", ""); got != want {
|
||||
t.Errorf("image bytes:\n got %s\n want %s", got, want)
|
||||
}
|
||||
if img.Symbols["mask"] != 16 || img.Symbols["small"] != 32 {
|
||||
t.Errorf("symbol offsets = %v, want mask=16 small=32", img.Symbols)
|
||||
}
|
||||
if len(img.Funcs) != 1 || img.Funcs[0].Name != "load" || img.Funcs[0].Size != 15 {
|
||||
t.Errorf("funcs = %+v", img.Funcs)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleFileErrors checks the static-symbol error paths.
|
||||
func TestAssembleFileErrors(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
src string
|
||||
want string // substring of the error
|
||||
}{
|
||||
{
|
||||
"undefined symbol",
|
||||
`
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
VMOVDQU nope<>(SB), X0
|
||||
RET
|
||||
`,
|
||||
"undefined symbol",
|
||||
},
|
||||
{
|
||||
"DATA without GLOBL",
|
||||
`
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
RET
|
||||
DATA orphan<>+0(SB)/4, $1
|
||||
`,
|
||||
"no matching GLOBL",
|
||||
},
|
||||
{
|
||||
"DATA exceeds size",
|
||||
`
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
RET
|
||||
GLOBL tiny<>(SB), RODATA, $4
|
||||
DATA tiny<>+0(SB)/8, $1
|
||||
`,
|
||||
"exceeds GLOBL size",
|
||||
},
|
||||
{
|
||||
"DATA bad width",
|
||||
`
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
RET
|
||||
GLOBL odd<>(SB), RODATA, $4
|
||||
DATA odd<>+0(SB)/3, $1
|
||||
`,
|
||||
"invalid width",
|
||||
},
|
||||
}
|
||||
for _, c := range cases {
|
||||
f, errs := parser.Parse("e_amd64.s", c.src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("%s: parse: %v", c.name, errs)
|
||||
}
|
||||
if _, err := AssembleFile(f); err == nil || !strings.Contains(err.Error(), c.want) {
|
||||
t.Errorf("%s: error %v, want substring %q", c.name, err, c.want)
|
||||
}
|
||||
}
|
||||
|
||||
// A static-symbol operand is unresolvable in single-function assembly.
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
MOVQ x<>(SB), AX
|
||||
RET
|
||||
GLOBL x<>(SB), RODATA, $8
|
||||
DATA x<>+0(SB)/4, $1
|
||||
`)
|
||||
if _, _, err := Assemble(fn); err == nil || !strings.Contains(err.Error(), "file-level assembly") {
|
||||
t.Errorf("single-function SB: error %v, want a file-level-assembly error", err)
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,595 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
// loong64 (LoongArch) instruction encoding.
|
||||
//
|
||||
// The encoder is data-driven: each mnemonic maps to an instruction format and
|
||||
// an opcode constant, and the format selects the bit layout. The opcode
|
||||
// constants and formats are transcribed from the Go toolchain's own loong64
|
||||
// backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm`
|
||||
// exactly — the ground-truth oracle for the verify suite.
|
||||
//
|
||||
// All LoongArch instructions are 32 bits, little-endian. The formats used
|
||||
// here (per the LoongArch Volume I specification):
|
||||
//
|
||||
// 3R opcode[31:15] | rk[4:0] | rj[4:0] | rd[4:0]
|
||||
// 2R opcode[31:15] | rj[4:0] | rd[4:0]
|
||||
// 2RI12 opcode[31:22] | si12[11:0] | rj[4:0] | rd[4:0]
|
||||
// 2RI14 opcode[31:18] | si14[13:0] | rj[4:0] | rd[4:0]
|
||||
// 2RI16 opcode[31:22] | si16[15:0] | rj[4:0] | rd[4:0]
|
||||
// 2RI20 opcode[31:25] | si20[19:0] | rd[4:0]
|
||||
// 1RI21 opcode[31:26] | si21[20:0] | rj[4:0] (BEQZ/BNEZ, B*Z, BC*Z)
|
||||
// B/BL opcode[31:26] | offs[25:0]
|
||||
// 4R opcode[31:20] | r1[4:0] | r2[4:0] | r3[4:0] | r4[4:0]
|
||||
// IRIR opcode[31:22] | msb[4:0] | rj[4:0] | lsb[4:0] | rd[4:0]
|
||||
// 3RI2 opcode[31:17] | sa2[1:0] | rk[4:0] | rj[4:0] | rd[4:0]
|
||||
//
|
||||
// The opcode constants are pre-positioned (they include the zero bit ranges
|
||||
// of the immediate and register fields), mirroring the toolchain's OP_*
|
||||
// helpers, so each l64* function only ORs its fields in.
|
||||
|
||||
import "maps"
|
||||
|
||||
// loong64RegNum returns the 5-bit register number for a LoongArch register
|
||||
// name: R0–R31 (integer), F0–F31 (floating point), FCC0–FCC7 (condition
|
||||
// flags), FCSR0–FCSR31 (control/status) and the ABI aliases the runtime's
|
||||
// assembly uses. Returns -1 for an unrecognised name.
|
||||
func loong64RegNum(name string) int {
|
||||
switch name {
|
||||
case "R0", "ZERO":
|
||||
return 0
|
||||
case "R1", "RA", "LINK":
|
||||
return 1
|
||||
case "R2", "TP":
|
||||
return 2
|
||||
case "R3", "SP":
|
||||
return 3
|
||||
case "R4", "A0":
|
||||
return 4
|
||||
case "R5", "A1":
|
||||
return 5
|
||||
case "R6", "A2":
|
||||
return 6
|
||||
case "R7", "A3":
|
||||
return 7
|
||||
case "R8", "A4":
|
||||
return 8
|
||||
case "R9", "A5":
|
||||
return 9
|
||||
case "R10", "A6":
|
||||
return 10
|
||||
case "R11", "A7":
|
||||
return 11
|
||||
case "R12", "T0":
|
||||
return 12
|
||||
case "R13", "T1":
|
||||
return 13
|
||||
case "R14", "T2":
|
||||
return 14
|
||||
case "R15", "T3":
|
||||
return 15
|
||||
case "R16", "T4":
|
||||
return 16
|
||||
case "R17", "T5":
|
||||
return 17
|
||||
case "R18", "T6":
|
||||
return 18
|
||||
case "R19", "T7":
|
||||
return 19
|
||||
case "R20", "T8":
|
||||
return 20
|
||||
case "R21":
|
||||
return 21
|
||||
case "R22", "G", "g", "FP":
|
||||
return 22
|
||||
case "R23", "S0":
|
||||
return 23
|
||||
case "R24", "S1":
|
||||
return 24
|
||||
case "R25", "S2":
|
||||
return 25
|
||||
case "R26", "S3":
|
||||
return 26
|
||||
case "R27", "S4":
|
||||
return 27
|
||||
case "R28", "S5":
|
||||
return 28
|
||||
case "R29", "S6", "CTXT":
|
||||
return 29
|
||||
case "R30", "S7", "TMP":
|
||||
return 30
|
||||
case "R31", "S8":
|
||||
return 31
|
||||
}
|
||||
// F0–F31, FCC0–FCC7, FCSR0–FCSR31.
|
||||
if len(name) >= 4 && name[:4] == "FCSR" {
|
||||
return loong64RegSpecial(name[4:], 31)
|
||||
}
|
||||
if len(name) >= 3 && name[:3] == "FCC" {
|
||||
return loong64RegSpecial(name[3:], 7)
|
||||
}
|
||||
if len(name) < 2 {
|
||||
return -1
|
||||
}
|
||||
prefix, digits := name[:1], name[1:]
|
||||
if digits[0] < '0' || digits[0] > '9' {
|
||||
return -1
|
||||
}
|
||||
n := 0
|
||||
for i := 0; i < len(digits); i++ {
|
||||
if digits[i] < '0' || digits[i] > '9' {
|
||||
return -1
|
||||
}
|
||||
n = n*10 + int(digits[i]-'0')
|
||||
}
|
||||
if prefix == "F" && n <= 31 {
|
||||
return n
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// loong64RegSpecial parses a numbered FCC/FCSR register.
|
||||
func loong64RegSpecial(digits string, max int) int {
|
||||
if digits == "" {
|
||||
return -1
|
||||
}
|
||||
n := 0
|
||||
for i := 0; i < len(digits); i++ {
|
||||
if digits[i] < '0' || digits[i] > '9' {
|
||||
return -1
|
||||
}
|
||||
n = n*10 + int(digits[i]-'0')
|
||||
}
|
||||
if n <= max {
|
||||
return n
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// ---- format helpers ----
|
||||
|
||||
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
|
||||
func l64rrr(op uint32, rk, rj, rd int) uint32 {
|
||||
return op | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
||||
}
|
||||
|
||||
// l64rr encodes a 2R instruction: op | rj<<5 | rd.
|
||||
func l64rr(op uint32, rj, rd int) uint32 {
|
||||
return op | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
||||
}
|
||||
|
||||
// l64irr encodes a 2RI12 instruction: op | si12<<10 | rj<<5 | rd.
|
||||
func l64irr(op uint32, imm, rj, rd int) uint32 {
|
||||
return op | (uint32(imm)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
||||
}
|
||||
|
||||
// l64irr14 encodes a 2RI14 instruction: op | si14<<10 | rj<<5 | rd.
|
||||
func l64irr14(op uint32, imm, rj, rd int) uint32 {
|
||||
return op | (uint32(imm)&0x3FFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
||||
}
|
||||
|
||||
// l64irr16 encodes a 2RI16 instruction: op | si16<<10 | rj<<5 | rd.
|
||||
func l64irr16(op uint32, imm, rj, rd int) uint32 {
|
||||
return op | (uint32(imm)&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
||||
}
|
||||
|
||||
// l64ir encodes a 2RI20 instruction: op | si20<<5 | rd.
|
||||
func l64ir(op uint32, imm, rd int) uint32 {
|
||||
return op | (uint32(imm)&0xFFFFF)<<5 | uint32(rd&0x1f)
|
||||
}
|
||||
|
||||
// l64bbl encodes a B/BL instruction: op | offs[25:0], where offs is the
|
||||
// 4-byte-aligned word distance (the toolchain stores the shifted value).
|
||||
func l64bbl(op uint32, offs int) uint32 {
|
||||
return op | (uint32(offs)&0xFFFF)<<10 | (uint32(offs)>>16)&0x3FF
|
||||
}
|
||||
|
||||
// l64ir21 encodes a 1RI21 branch (BEQZ/BNEZ, BLTZ/BGEZ/BLEZ/BGTZ, BFPT/BFPF):
|
||||
// op | si21[15:0]<<10 | rj<<5 | si21[20:16].
|
||||
func l64ir21(op uint32, offs, rj int) uint32 {
|
||||
v := uint32(offs)
|
||||
return op | (v&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | (v>>16)&0x1F
|
||||
}
|
||||
|
||||
// l64rrrr encodes a 4R instruction: op | r1<<15 | r2<<10 | r3<<5 | r4.
|
||||
func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
|
||||
return op | uint32(r1&0x1f)<<15 | uint32(r2&0x1f)<<10 | uint32(r3&0x1f)<<5 | uint32(r4&0x1f)
|
||||
}
|
||||
|
||||
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
|
||||
// The msb/lsb fields are 6 bits wide (0–63) and are validated by the caller.
|
||||
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
|
||||
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
|
||||
}
|
||||
|
||||
// l64irrr encodes a 3RI2 instruction (ALSL): op | sa<<15 | rk<<10 | rj<<5 | rd.
|
||||
func l64irrr(op uint32, sa, rk, rj, rd int) uint32 {
|
||||
return op | uint32(sa&0x3)<<15 | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
||||
}
|
||||
|
||||
// l64i15 encodes a no-operand system instruction with a 15-bit code field
|
||||
// (SYSCALL, BREAK, DBAR): op | code[14:0].
|
||||
func l64i15(op uint32, code int) uint32 {
|
||||
return op | uint32(code)&0x7FFF
|
||||
}
|
||||
|
||||
// l64irr5i encodes PRELD: op | offs<<10 | rj<<5 | hint.
|
||||
func l64irr5i(op uint32, offs, rj, hint int) uint32 {
|
||||
return op | (uint32(offs)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(hint&0x1f)
|
||||
}
|
||||
|
||||
// l64wordLE encodes a uint32 as 4 little-endian bytes.
|
||||
func l64wordLE(w uint32) []byte {
|
||||
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
|
||||
}
|
||||
|
||||
// l64WordsLE concatenates one or more instruction words as little-endian bytes.
|
||||
func l64WordsLE(ws ...uint32) []byte {
|
||||
var out []byte
|
||||
for _, w := range ws {
|
||||
out = append(out, l64wordLE(w)...)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// ---- instruction formats ----
|
||||
|
||||
type l64Format uint8
|
||||
|
||||
const (
|
||||
l64Frrr l64Format = iota // 3R (integer and FP arithmetic)
|
||||
l64Frr // 2R
|
||||
l64Firr // 2RI12 (arithmetic with 12-bit immediate)
|
||||
l64Firr14 // 2RI14 (ldptr/stptr)
|
||||
l64Firr16 // 2RI16 (addu16i.d)
|
||||
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
|
||||
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub)
|
||||
l64Firir // bstrins/bstrpick
|
||||
l64Firrr // alsl
|
||||
l64Fi15 // syscall/break/dbar
|
||||
l64Fam // atomic (3R with the AM field order)
|
||||
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
|
||||
l64Fshift // 2RI12 with a 5/6-bit shift immediate
|
||||
l64Fpreld // preld (2RI12 + 5-bit hint)
|
||||
)
|
||||
|
||||
// l64Enc is one instruction's encoding: its bit layout (format) and the
|
||||
// opcode constant, positioned at its exact bit range.
|
||||
type l64Enc struct {
|
||||
format l64Format
|
||||
op uint32
|
||||
}
|
||||
|
||||
// l64DualEnc holds both forms of a dual-form mnemonic: the 3R register form
|
||||
// and the 2RI12 immediate form (which is a shift for the shift mnemonics).
|
||||
type l64DualEnc struct {
|
||||
rrr uint32 // 3R register form
|
||||
imm uint32 // 2RI12 immediate form
|
||||
shift bool // the immediate form is a 5/6-bit shift amount
|
||||
}
|
||||
|
||||
// l64DualTable maps the dual-form arithmetic/logic mnemonics to both
|
||||
// encodings; the assembler picks by operand kind.
|
||||
var l64DualTable = map[string]l64DualEnc{}
|
||||
|
||||
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
|
||||
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered
|
||||
// yet; the base integer, memory and floating-point ISA is complete.
|
||||
var l64InstrTable = map[string]l64Enc{}
|
||||
|
||||
func init() {
|
||||
// 3R — integer.
|
||||
rrr := map[string]uint32{
|
||||
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
|
||||
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
|
||||
"SGT": 0x24 << 15, "SGTU": 0x25 << 15,
|
||||
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15,
|
||||
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
|
||||
"ORN": 0x2c << 15, "ANDN": 0x2d << 15,
|
||||
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
|
||||
"SLLV": 0x31 << 15, "SRLV": 0x32 << 15, "SRAV": 0x33 << 15,
|
||||
"ROTR": 0x36 << 15, "ROTRV": 0x37 << 15,
|
||||
"MUL": 0x38 << 15, "MULW": 0x38 << 15, "MULH": 0x39 << 15, "MULHU": 0x3a << 15,
|
||||
"MULV": 0x3b << 15, "MULVU": 0x3b << 15, "MULHV": 0x3c << 15, "MULHVU": 0x3d << 15,
|
||||
"MULWVW": 0x3e << 15, "MULWVWU": 0x3f << 15,
|
||||
"DIV": 0x40 << 15, "DIVW": 0x40 << 15, "REM": 0x41 << 15, "REMW": 0x41 << 15,
|
||||
"DIVU": 0x42 << 15, "DIVWU": 0x42 << 15, "REMU": 0x43 << 15, "REMWU": 0x43 << 15,
|
||||
"DIVV": 0x44 << 15, "REMV": 0x45 << 15, "DIVVU": 0x46 << 15, "REMVU": 0x47 << 15,
|
||||
"CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15,
|
||||
"CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15,
|
||||
}
|
||||
// 3R — floating point.
|
||||
rrr["MULF"] = 0x209 << 15
|
||||
rrr["MULD"] = 0x20a << 15
|
||||
rrr["DIVF"] = 0x20d << 15
|
||||
rrr["DIVD"] = 0x20e << 15
|
||||
rrr["SUBF"] = 0x205 << 15
|
||||
rrr["SUBD"] = 0x206 << 15
|
||||
rrr["ADDF"] = 0x201 << 15
|
||||
rrr["ADDD"] = 0x202 << 15
|
||||
rrr["CMPEQF"] = 0x0c1<<20 | 0x4<<15
|
||||
rrr["CMPEQD"] = 0x0c2<<20 | 0x4<<15
|
||||
rrr["CMPGED"] = 0x0c2<<20 | 0x7<<15
|
||||
rrr["CMPGEF"] = 0x0c1<<20 | 0x7<<15
|
||||
rrr["CMPGTD"] = 0x0c2<<20 | 0x3<<15
|
||||
rrr["CMPGTF"] = 0x0c1<<20 | 0x3<<15
|
||||
rrr["FMINF"] = 0x215 << 15
|
||||
rrr["FMIND"] = 0x216 << 15
|
||||
rrr["FMAXF"] = 0x211 << 15
|
||||
rrr["FMAXD"] = 0x212 << 15
|
||||
rrr["FMAXAF"] = 0x219 << 15
|
||||
rrr["FMAXAD"] = 0x21a << 15
|
||||
rrr["FMINAF"] = 0x21d << 15
|
||||
rrr["FMINAD"] = 0x21e << 15
|
||||
rrr["FSCALEBF"] = 0x221 << 15
|
||||
rrr["FSCALEBD"] = 0x222 << 15
|
||||
rrr["FCOPYSGF"] = 0x225 << 15
|
||||
rrr["FCOPYSGD"] = 0x226 << 15
|
||||
for m, op := range rrr {
|
||||
l64InstrTable[m] = l64Enc{format: l64Frrr, op: op}
|
||||
}
|
||||
|
||||
// 2R.
|
||||
rr := map[string]uint32{
|
||||
"CLOW": 0x4 << 10, "CLZW": 0x5 << 10, "CTOW": 0x6 << 10, "CTZW": 0x7 << 10,
|
||||
"CLOV": 0x8 << 10, "CLZV": 0x9 << 10, "CTOV": 0xa << 10, "CTZV": 0xb << 10,
|
||||
"REVB2H": 0xc << 10, "REVB4H": 0xd << 10, "REVB2W": 0xe << 10, "REVBV": 0xf << 10,
|
||||
"REVH2W": 0x10 << 10, "REVHV": 0x11 << 10,
|
||||
"BITREV4B": 0x12 << 10, "BITREV8B": 0x13 << 10, "BITREVW": 0x14 << 10, "BITREVV": 0x15 << 10,
|
||||
"EXTWH": 0x16 << 10, "EXTWB": 0x17 << 10, "CPUCFG": 0x1b << 10,
|
||||
"TRUNCFV": 0x46a9 << 10, "TRUNCDV": 0x46aa << 10, "TRUNCFW": 0x46a1 << 10, "TRUNCDW": 0x46a2 << 10,
|
||||
"MOVWF": 0x4744 << 10, "MOVVF": 0x4746 << 10, "MOVWD": 0x4748 << 10, "MOVVD": 0x474a << 10,
|
||||
"MOVFW": 0x46c1 << 10, "MOVDW": 0x46c2 << 10, "MOVFV": 0x46c9 << 10, "MOVDV": 0x46ca << 10,
|
||||
"FRINTF": 0x4791 << 10, "FRINTD": 0x4792 << 10,
|
||||
"MOVDF": 0x4646 << 10, "MOVFD": 0x4649 << 10,
|
||||
"ABSF": 0x4501 << 10, "ABSD": 0x4502 << 10,
|
||||
"MOVF": 0x4525 << 10, "MOVD": 0x4526 << 10,
|
||||
"NEGF": 0x4505 << 10, "NEGD": 0x4506 << 10,
|
||||
"SQRTF": 0x4511 << 10, "SQRTD": 0x4512 << 10,
|
||||
"FLOGBF": 0x4509 << 10, "FLOGBD": 0x450a << 10,
|
||||
"FCLASSF": 0x450d << 10, "FCLASSD": 0x450e << 10,
|
||||
"FTINTRMWF": 0x4681 << 10, "FTINTRMWD": 0x4682 << 10,
|
||||
"FTINTRMVF": 0x4689 << 10, "FTINTRMVD": 0x468a << 10,
|
||||
"FTINTRPWF": 0x4691 << 10, "FTINTRPWD": 0x4692 << 10,
|
||||
"FTINTRPVF": 0x4699 << 10, "FTINTRPVD": 0x469a << 10,
|
||||
"FTINTRZWF": 0x46a1 << 10, "FTINTRZWD": 0x46a2 << 10,
|
||||
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
|
||||
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
|
||||
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
|
||||
}
|
||||
for m, op := range rr {
|
||||
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
|
||||
}
|
||||
// RDTIME is a 2R instruction with rd and rj in swapped positions.
|
||||
l64InstrTable["RDTIMELW"] = l64Enc{format: l64Frdtime, op: 0x18 << 10}
|
||||
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
|
||||
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
|
||||
|
||||
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
|
||||
// selected by the operand kind; the shift mnemonics pair the 3R form
|
||||
// with a 5/6-bit shift immediate.
|
||||
maps.Copy(l64DualTable, map[string]l64DualEnc{
|
||||
"ADD": {rrr: 0x20 << 15, imm: 0x00a << 22},
|
||||
"ADDW": {rrr: 0x20 << 15, imm: 0x00a << 22},
|
||||
"ADDV": {rrr: 0x21 << 15, imm: 0x00b << 22},
|
||||
"ADDVU": {rrr: 0x21 << 15, imm: 0x00b << 22},
|
||||
"AND": {rrr: 0x29 << 15, imm: 0x00d << 22},
|
||||
"OR": {rrr: 0x2a << 15, imm: 0x00e << 22},
|
||||
"XOR": {rrr: 0x2b << 15, imm: 0x00f << 22},
|
||||
"SGT": {rrr: 0x24 << 15, imm: 0x008 << 22},
|
||||
"SGTU": {rrr: 0x25 << 15, imm: 0x009 << 22},
|
||||
"SLL": {rrr: 0x2e << 15, imm: 0x00081 << 15, shift: true},
|
||||
"SRL": {rrr: 0x2f << 15, imm: 0x00089 << 15, shift: true},
|
||||
"SRA": {rrr: 0x30 << 15, imm: 0x00091 << 15, shift: true},
|
||||
"ROTR": {rrr: 0x36 << 15, imm: 0x00099 << 15, shift: true},
|
||||
"SLLV": {rrr: 0x31 << 15, imm: 0x0041 << 16, shift: true},
|
||||
"SRLV": {rrr: 0x32 << 15, imm: 0x0045 << 16, shift: true},
|
||||
"SRAV": {rrr: 0x33 << 15, imm: 0x0049 << 16, shift: true},
|
||||
"ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true},
|
||||
})
|
||||
|
||||
// 2RI12 — pure immediate arithmetic (LU52ID has no register form).
|
||||
l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22}
|
||||
// ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16.
|
||||
l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26}
|
||||
|
||||
// 2RI14 — LL/SC are aliased by the Go assembler to the pointer loads and
|
||||
// stores (ldptr/stptr), with the offset scaled by 4.
|
||||
l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w
|
||||
l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d
|
||||
l64InstrTable["SC"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w
|
||||
l64InstrTable["SCW"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w
|
||||
l64InstrTable["SCV"] = l64Enc{format: l64Firr14, op: 0x23 << 24} // sc.d
|
||||
l64InstrTable["LL"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w)
|
||||
l64InstrTable["LLW"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w)
|
||||
l64InstrTable["LLV"] = l64Enc{format: l64Firr14, op: 0x22 << 24} // ldptr.d (ll.d)
|
||||
|
||||
// 2RI20.
|
||||
l64InstrTable["LU12IW"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
||||
l64InstrTable["LU32ID"] = l64Enc{format: l64Fir20, op: 0x0b << 25}
|
||||
l64InstrTable["PCALAU12I"] = l64Enc{format: l64Fir20, op: 0x0d << 25}
|
||||
l64InstrTable["PCADDU12I"] = l64Enc{format: l64Fir20, op: 0x0e << 25}
|
||||
// LUI is the Plan 9 spelling of lu12i.w.
|
||||
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
||||
|
||||
// 4R — fused multiply-add.
|
||||
rrrr := map[string]uint32{
|
||||
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
|
||||
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
|
||||
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
|
||||
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
|
||||
}
|
||||
for m, op := range rrrr {
|
||||
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
|
||||
}
|
||||
|
||||
// IRIR — bit-field insert/extract.
|
||||
irir := map[string]uint32{
|
||||
"BSTRINSW": 0x3<<21 | 0x0<<15,
|
||||
"BSTRINSV": 0x2 << 22,
|
||||
"BSTRPICKW": 0x3<<21 | 0x1<<15,
|
||||
"BSTRPICKV": 0x3 << 22,
|
||||
}
|
||||
for m, op := range irir {
|
||||
l64InstrTable[m] = l64Enc{format: l64Firir, op: op}
|
||||
}
|
||||
|
||||
// 3RI2 — ALSL.
|
||||
irrr := map[string]uint32{
|
||||
"ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17,
|
||||
}
|
||||
for m, op := range irrr {
|
||||
l64InstrTable[m] = l64Enc{format: l64Firrr, op: op}
|
||||
}
|
||||
|
||||
// 0-operand system instructions.
|
||||
l64InstrTable["SYSCALL"] = l64Enc{format: l64Fi15, op: 0x56 << 15}
|
||||
l64InstrTable["BREAK"] = l64Enc{format: l64Fi15, op: 0x54 << 15}
|
||||
l64InstrTable["DBAR"] = l64Enc{format: l64Fi15, op: 0x70e4 << 15}
|
||||
|
||||
// PRELD.
|
||||
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
|
||||
|
||||
// Atomics — 3R with the AM field order (rk=value, rj=address, rd=result).
|
||||
am := map[string]uint32{
|
||||
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
|
||||
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
|
||||
"AMCASB": 0x070B0 << 15, "AMCASH": 0x070B1 << 15,
|
||||
"AMCASW": 0x070B2 << 15, "AMCASV": 0x070B3 << 15,
|
||||
"AMADDW": 0x070C2 << 15, "AMADDV": 0x070C3 << 15,
|
||||
"AMANDW": 0x070C4 << 15, "AMANDV": 0x070C5 << 15,
|
||||
"AMORW": 0x070C6 << 15, "AMORV": 0x070C7 << 15,
|
||||
"AMXORW": 0x070C8 << 15, "AMXORV": 0x070C9 << 15,
|
||||
"AMMAXW": 0x070CA << 15, "AMMAXV": 0x070CB << 15,
|
||||
"AMMINW": 0x070CC << 15, "AMMINV": 0x070CD << 15,
|
||||
"AMMAXWU": 0x070CE << 15, "AMMAXVU": 0x070CF << 15,
|
||||
"AMMINWU": 0x070D0 << 15, "AMMINVU": 0x070D1 << 15,
|
||||
"AMSWAPDBB": 0x070BC << 15, "AMSWAPDBH": 0x070BD << 15,
|
||||
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
|
||||
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
|
||||
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
|
||||
}
|
||||
for m, op := range am {
|
||||
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
|
||||
}
|
||||
}
|
||||
|
||||
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
|
||||
// register move between the integer and floating-point register banks — the
|
||||
// MOVW/MOVV specials the Go assembler accepts.
|
||||
var l64FpMovTable = map[string]uint32{
|
||||
"MOVV.R.F": 0x452a << 10, // movgr2fr.d
|
||||
"MOVV.R.FCC": 0x4536 << 10, // movgr2cf
|
||||
"MOVV.R.FCSR": 0x4530 << 10, // movgr2fcsr
|
||||
"MOVV.F.R": 0x452e << 10, // movfr2gr.d
|
||||
"MOVV.F.FCC": 0x4534 << 10, // movfr2cf
|
||||
"MOVV.FCC.R": 0x4537 << 10, // movcf2gr
|
||||
"MOVV.FCC.F": 0x4535 << 10, // movcf2fr
|
||||
"MOVV.FCSR.R": 0x4532 << 10, // movfcsr2gr
|
||||
"MOVW.R.F": 0x4529 << 10, // movgr2fr.w
|
||||
"MOVW.F.R": 0x452d << 10, // movfr2gr.s
|
||||
}
|
||||
|
||||
// l64branchTable holds the 16-bit branch and jump encodings (2RI16).
|
||||
var l64branchTable = map[string]uint32{
|
||||
"BEQ": 0x16 << 26,
|
||||
"BNE": 0x17 << 26,
|
||||
"BLT": 0x18 << 26,
|
||||
"BGE": 0x19 << 26,
|
||||
"BLTU": 0x1a << 26,
|
||||
"BGEU": 0x1b << 26,
|
||||
"JIRL": 0x13 << 26,
|
||||
}
|
||||
|
||||
// l64branch21Table holds the single-register branches with 21-bit offsets:
|
||||
// the negative opcode constants the toolchain uses for the short forms.
|
||||
var l64branch21Table = map[string]uint32{
|
||||
"BEQZ": 0x10 << 26, // beq r0, rj → beqz
|
||||
"BNEZ": 0x11 << 26, // bne r0, rj → bnez
|
||||
"BLTZ": 0x18 << 26, // blt rj, r0 → bltz
|
||||
"BGEZ": 0x19 << 26, // bge rj, r0 → bgez
|
||||
"BGTZ": 0x18 << 26, // blt r0, rj → bgtz
|
||||
"BLEZ": 0x19 << 26, // bge r0, rj → blez
|
||||
"BFPT": 0x12<<26 | 0x1<<8,
|
||||
"BFPF": 0x12<<26 | 0x0<<8,
|
||||
}
|
||||
|
||||
// l64jumpTable maps the jump pseudo-instructions and their aliases to the
|
||||
// B/BL opcode constants.
|
||||
var l64jumpTable = map[string]uint32{
|
||||
"JMP": 0x14 << 26, // b
|
||||
"B": 0x14 << 26, // b
|
||||
"JAL": 0x15 << 26, // bl
|
||||
"CALL": 0x15 << 26, // bl
|
||||
"BL": 0x15 << 26, // bl
|
||||
}
|
||||
|
||||
// l64loadStoreTable maps the MOV width mnemonics to their load and store
|
||||
// 2RI12 opcodes. The load opcode is the negated store opcode, exactly as
|
||||
// the toolchain derives it.
|
||||
var l64loadStoreTable = map[string]struct{ ld, st uint32 }{
|
||||
"MOVB": {0x0a0 << 22, 0x0a4 << 22},
|
||||
"MOVH": {0x0a1 << 22, 0x0a5 << 22},
|
||||
"MOVW": {0x0a2 << 22, 0x0a6 << 22},
|
||||
"MOVV": {0x0a3 << 22, 0x0a7 << 22},
|
||||
"MOVBU": {0x0a8 << 22, 0x0a4 << 22},
|
||||
"MOVHU": {0x0a9 << 22, 0x0a5 << 22},
|
||||
"MOVWU": {0x0aa << 22, 0x0a6 << 22},
|
||||
"MOVF": {0x0ac << 22, 0x0ad << 22},
|
||||
"MOVD": {0x0ae << 22, 0x0af << 22},
|
||||
}
|
||||
|
||||
// l64movRegTable maps a register-to-register MOV mnemonic to its expansion,
|
||||
// matching the toolchain's case-1 encoding: MOVB → ext.w.b, MOVH → ext.w.h,
|
||||
// MOVW → sll.w, MOVV → or, MOVBU → andi. MOVHU/MOVWU expand to bstrpick.d
|
||||
// and are handled separately in the assembler.
|
||||
type l64MovRegEnc struct {
|
||||
rr bool // 2R format (ext.w.b/ext.w.h)
|
||||
op uint32 // opcode constant (rr forms) or 3R/2RI12 opcode
|
||||
imm int // 2RI12 immediate for MOVBU's andi
|
||||
}
|
||||
|
||||
var l64movRegTable = map[string]l64MovRegEnc{
|
||||
"MOVB": {true, 0x17 << 10, 0}, // ext.w.b rd, rj
|
||||
"MOVH": {true, 0x16 << 10, 0}, // ext.w.h rd, rj
|
||||
"MOVW": {false, 0x2e << 15, 0}, // sll.w rd, rj, r0
|
||||
"MOVV": {false, 0x2a << 15, 0}, // or rd, rj, r0
|
||||
"MOVBU": {false, 0x00d << 22, 0xff}, // andi rd, rj, $0xff
|
||||
}
|
||||
|
||||
// l64movFpRegTable maps a floating-point register move mnemonic to its 2R
|
||||
// opcode (fmov.s / fmov.d), used when both operands are F registers.
|
||||
var l64movFpRegTable = map[string]uint32{
|
||||
"MOVF": 0x4525 << 10,
|
||||
"MOVD": 0x4526 << 10,
|
||||
}
|
||||
|
||||
// l64RegClass discriminates integer (R), floating-point (F) and condition
|
||||
// (FCC) registers for the MOV pseudo-instruction's register-move encoding.
|
||||
type l64RegClass int
|
||||
|
||||
const (
|
||||
l64ClsNone l64RegClass = iota
|
||||
l64ClsGR
|
||||
l64ClsFP
|
||||
l64ClsFCC
|
||||
l64ClsFCSR
|
||||
)
|
||||
|
||||
// loong64RegClass reports the register class of a register operand name.
|
||||
func loong64RegClass(name string) l64RegClass {
|
||||
switch {
|
||||
case name == "":
|
||||
return l64ClsNone
|
||||
case len(name) >= 3 && name[:3] == "FCC":
|
||||
return l64ClsFCC
|
||||
case len(name) >= 4 && name[:4] == "FCSR":
|
||||
return l64ClsFCSR
|
||||
case name[0] == 'F':
|
||||
return l64ClsFP
|
||||
default:
|
||||
return l64ClsGR
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,293 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// firstTextLOONG64 parses assembly source and returns the first TEXT body.
|
||||
func firstTextLOONG64(t *testing.T, src string) *ast.Text {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse("f_loong64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
for _, d := range f.Decls {
|
||||
if fn, ok := d.(*ast.Text); ok {
|
||||
return fn
|
||||
}
|
||||
}
|
||||
t.Fatal("no TEXT found")
|
||||
return nil
|
||||
}
|
||||
|
||||
// assembleLOONG64Helper assembles one TEXT function and returns its bytes.
|
||||
func assembleLOONG64Helper(t *testing.T, fn *ast.Text) []byte {
|
||||
t.Helper()
|
||||
code, _, _, _, _, err := assembleLOONG64(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
return code
|
||||
}
|
||||
|
||||
// wantWords checks that code matches the expected little-endian words.
|
||||
func wantWords(t *testing.T, code []byte, want ...uint32) {
|
||||
t.Helper()
|
||||
got := make([]uint32, 0, len(code)/4)
|
||||
for i := 0; i+4 <= len(code); i += 4 {
|
||||
got = append(got, binary.LittleEndian.Uint32(code[i:]))
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestLOONG64_add(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVV a+0(FP), R4
|
||||
MOVV b+8(FP), R5
|
||||
ADDV R5, R4, R4
|
||||
MOVV R4, ret+16(FP)
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
// 5 instructions: two ld.d, add.d, st.d, jirl r0, r1, 0.
|
||||
wantWords(t, code,
|
||||
0x28C02064, // ld.d r4, 8(r3)
|
||||
0x28C04065, // ld.d r5, 16(r3)
|
||||
0x00109484, // add.d r4, r4, r5
|
||||
0x29C06064, // st.d r4, 24(r3)
|
||||
0x4C000020, // jirl r0, r1, 0
|
||||
)
|
||||
}
|
||||
|
||||
func TestLOONG64_arithmetic(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·arith(SB), NOSPLIT, $0
|
||||
ADDV R4, R5, R6
|
||||
SUBV R7, R8, R9
|
||||
MULV R10, R11, R12
|
||||
DIVV R13, R14, R15
|
||||
AND R16, R17, R18
|
||||
OR R18, R19, R20
|
||||
XOR R20, R21, R2
|
||||
SLLV R2, R23, R24
|
||||
SRLV R24, R25, R26
|
||||
SRAV R26, R27, R28
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x001090A6, // add.d r6, r5, r4
|
||||
0x00119D09, // sub.d r9, r8, r7
|
||||
0x001DA96C, // mul.d r12, r11, r10
|
||||
0x002235CF, // div.d r15, r14, r13
|
||||
0x0014C232, // and r18, r17, r16
|
||||
0x00154A74, // or r20, r19, r18
|
||||
0x0015D2A2, // xor r2, r21, r20
|
||||
0x00188AF8, // sll.d r24, r23, r2
|
||||
0x0019633A, // srl.d r26, r25, r24
|
||||
0x0019EB7C, // sra.d r28, r27, r26
|
||||
0x4C000020, // jirl r0, r1, 0
|
||||
)
|
||||
}
|
||||
|
||||
func TestLOONG64_immediates(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·imm(SB), NOSPLIT, $0
|
||||
ADDV $42, R4, R5
|
||||
ADDV $-8, R6
|
||||
AND $0xff, R7, R8
|
||||
OR $1, R9, R10
|
||||
SGT $100, R13, R14
|
||||
SLLV $4, R15, R16
|
||||
MOVV $0x12345, R17
|
||||
MOVV $0, R18
|
||||
MOVW $0, R19
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x02C0A885, // addi.d r5, r4, 42
|
||||
0x02FFE0C6, // addi.d r6, r6, -8
|
||||
0x0343FCE8, // andi r8, r7, 0xff
|
||||
0x0380052A, // ori r10, r9, 1
|
||||
0x020191AE, // slti r14, r13, 100
|
||||
0x004111F0, // slli.d r16, r15, 4
|
||||
0x14000251, // lu12i.w r17, 0x12
|
||||
0x038D1631, // ori r17, r17, 0x345
|
||||
0x00150012, // or r18, r0, r0
|
||||
0x00170013, // sll.w r19, r0, r0
|
||||
0x4C000020, // jirl r0, r1, 0
|
||||
)
|
||||
}
|
||||
|
||||
func TestLOONG64_loadStore(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·mem(SB), NOSPLIT, $0
|
||||
MOVV (R4), R5
|
||||
MOVV R5, (R6)
|
||||
MOVW 8(R7), R8
|
||||
MOVB R9, -4(R10)
|
||||
MOVV (R11)(R12), R13
|
||||
MOVV R14, (R15)(R16)
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x28C00085, // ld.d r5, 0(r4)
|
||||
0x29C000C5, // st.d r5, 0(r6)
|
||||
0x288020E8, // ld.w r8, 8(r7)
|
||||
0x293FF149, // st.b r9, -4(r10)
|
||||
0x380C316D, // ldx.d r13, r11, r12
|
||||
0x381C41EE, // stx.d r14, r15, r16
|
||||
0x4C000020, // jirl r0, r1, 0
|
||||
)
|
||||
}
|
||||
|
||||
func TestLOONG64_branches(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·br(SB), NOSPLIT, $0
|
||||
BEQ R4, R5, done
|
||||
BNE R6, R7, skip
|
||||
BLT R8, R9, done
|
||||
BGE R10, R11, done
|
||||
BLTU R12, R13, done
|
||||
BGEU R14, R15, done
|
||||
skip:
|
||||
JMP done
|
||||
done:
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
// skip is at 0x18 (6 words), done at 0x1c.
|
||||
wantWords(t, code,
|
||||
0x58001C85, // beq r5, r4, +7
|
||||
0x5C0018C7, // bne r7, r6, +6
|
||||
0x60001509, // blt r9, r8, +5
|
||||
0x6400114B, // bge r11, r10, +4
|
||||
0x68000D8D, // bltu r13, r12, +3
|
||||
0x6C0009CF, // bgeu r15, r14, +2
|
||||
0x50000400, // b done (+1, chain-folded through skip)
|
||||
0x4C000020, // jirl r0, r1, 0
|
||||
)
|
||||
}
|
||||
|
||||
func TestLOONG64_frame(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $32-8
|
||||
MOVV R4, R5
|
||||
MOVV arg+0(FP), R6
|
||||
MOVV R7, local-8(SP)
|
||||
MOVV local-8(SP), R8
|
||||
MOVV R9, ret+0(FP)
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
// autosize = align8(32+8) = 40; prologue stores LR at -40(SP),
|
||||
// opens the frame, stores LR again at 0(SP). The function is a leaf
|
||||
// (no calls), so the epilogue skips the LR restore. FP args are at
|
||||
// autosize+8; SP locals at autosize+offset.
|
||||
wantWords(t, code,
|
||||
0x29FF6061, // st.d r1, -40(r3)
|
||||
0x02FF6063, // addi.d r3, r3, -40
|
||||
0x29C00061, // st.d r1, 0(r3)
|
||||
0x00150085, // or r5, r4, r0
|
||||
0x28C0C066, // ld.d r6, 48(r3) arg+0(FP) → 0+40+8
|
||||
0x29C08067, // st.d r7, 32(r3) local-8(SP) → 40-8
|
||||
0x28C08068, // ld.d r8, 32(r3)
|
||||
0x29C0C069, // st.d r9, 48(r3) ret+0(FP) → 0+40+8
|
||||
0x02C0A063, // addi.d r3, r3, 40
|
||||
0x4C000020, // jirl r0, r1, 0
|
||||
)
|
||||
}
|
||||
|
||||
func TestLOONG64_jumpChain(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·jc(SB), NOSPLIT, $0
|
||||
JMP a
|
||||
a:
|
||||
JMP b
|
||||
b:
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x50000800, // b +2 (a, chain-folded to b)
|
||||
0x50000400, // b +1 (b)
|
||||
0x4C000020, // jirl r0, r1, 0
|
||||
)
|
||||
}
|
||||
|
||||
func TestLOONG64_dconClasses(t *testing.T) {
|
||||
cases := []struct {
|
||||
v int64
|
||||
word int // expected word count
|
||||
}{
|
||||
{0x123456789, 3}, // lu12i.w + ori + lu32i.d
|
||||
{-1, 2}, // addi.d + lu52i.d (the MOV path handles -1 earlier)
|
||||
{0x1000000000000, 2}, // addi.w + lu32i.d
|
||||
{0x123456789abcdef0, 4}, // full sequence
|
||||
{0xFFFFFFFFF, 2}, // lu12i.w + ori
|
||||
{0x1234567800000000, 3}, // addi.w + lu32i.d + lu52i.d
|
||||
}
|
||||
for _, c := range cases {
|
||||
if n := len(l64DconMovWords(0, c.v)); n != c.word {
|
||||
t.Errorf("0x%x: %d words, want %d", c.v, n, c.word)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestLOONG64_regNames(t *testing.T) {
|
||||
cases := map[string]int{
|
||||
"R0": 0, "R31": 31, "F0": 0, "F31": 31, "FCC0": 0, "FCC7": 7,
|
||||
"FCSR0": 0, "FCSR3": 3, "ZERO": 0, "RA": 1, "SP": 3, "g": 22, "G": 22,
|
||||
"R32": -1, "FCC8": -1, "X0": -1, "R": -1, "TMP": 30, "CTXT": 29,
|
||||
}
|
||||
for name, want := range cases {
|
||||
if got := loong64RegNum(name); got != want {
|
||||
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
|
||||
// A spot-check that assembleLOONG64 emits the same bytes the Go
|
||||
// toolchain does for a small kernel (the full comparison lives in
|
||||
// verify's TestGroundTruthLOONG64).
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·k(SB), NOSPLIT, $0-0
|
||||
ADDV R4, R5, R6
|
||||
MOVV $0x100000, R7
|
||||
BEQ R6, R7, done
|
||||
JMP done
|
||||
done:
|
||||
RET
|
||||
`
|
||||
fn := firstTextLOONG64(t, src)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
want := []byte{
|
||||
0xa6, 0x90, 0x10, 0x00, // add.d r6, r5, r4
|
||||
0x07, 0x20, 0x00, 0x14, // lu12i.w r7, 0x100
|
||||
0xc7, 0x08, 0x00, 0x58, // beq r7, r6, +2 (done)
|
||||
0x00, 0x04, 0x00, 0x50, // b +1 (done)
|
||||
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
|
||||
}
|
||||
if !bytes.Equal(code, want) {
|
||||
t.Errorf("code = % x\nwant % x", code, want)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,129 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
)
|
||||
|
||||
// Loong64 frame mapping, matching the Go toolchain's loong64 backend.
|
||||
//
|
||||
// Go's loong64 functions have no frame pointer: FP and SP are synthetic
|
||||
// registers resolved against the hardware stack pointer (R3) and the frame
|
||||
// size. The return address lives in R1 (the link register).
|
||||
//
|
||||
// The autosize is the real stack adjustment: the declared local frame plus
|
||||
// the 8 bytes for the saved link register, rounded up to a multiple of 8
|
||||
// (the toolchain aligns frames with `if autosize&4 != 0 { autosize += 4 }`).
|
||||
// A leaf function (no calls) with a zero frame gets no prologue at all.
|
||||
//
|
||||
// Prologue (autosize > 0), byte-identical to the toolchain:
|
||||
//
|
||||
// MOVV R1, -autosize(R3) // save LR below the new SP (traceback-safe)
|
||||
// ADDV $-autosize, R3 // open the frame
|
||||
// MOVV R1, 0(R3) // save LR again at SP (signal-safety)
|
||||
//
|
||||
// Epilogue: MOVV 0(R3), R1; ADDV $autosize, R3 (non-leaf only for the LR
|
||||
// restore); the RET's jirl r0, r1, 0 follows.
|
||||
|
||||
// loong64FrameInfo holds the frame layout derived from a TEXT directive.
|
||||
type loong64FrameInfo struct {
|
||||
autosize int // the real SP adjustment (locals + saved LR, aligned)
|
||||
frame int // the declared $framesize
|
||||
args int // the declared -argsize
|
||||
noSplit bool // the NOSPLIT flag
|
||||
leaf bool // no call instructions in the body
|
||||
}
|
||||
|
||||
// loong64ComputeFrame derives the frame layout for a TEXT function.
|
||||
func loong64ComputeFrame(t *ast.Text) loong64FrameInfo {
|
||||
fi := loong64FrameInfo{
|
||||
frame: frameSize(t),
|
||||
args: argsSize(t),
|
||||
}
|
||||
for _, f := range t.Flags {
|
||||
if f == "NOSPLIT" {
|
||||
fi.noSplit = true
|
||||
}
|
||||
}
|
||||
fi.leaf = loong64IsLeaf(t)
|
||||
if fi.frame != 0 {
|
||||
fi.autosize = fi.frame + 8 // space for the saved LR
|
||||
if fi.autosize&4 != 0 {
|
||||
fi.autosize += 4
|
||||
}
|
||||
} else if !fi.leaf {
|
||||
// A zero-frame non-leaf function still opens an 8-byte frame for LR.
|
||||
fi.autosize = 8
|
||||
}
|
||||
return fi
|
||||
}
|
||||
|
||||
// loong64IsLeaf reports whether a function contains no call instructions
|
||||
// (JAL/BL/CALL), matching the toolchain's LEAF mark, which drives the frame
|
||||
// and the epilogue shape.
|
||||
func loong64IsLeaf(t *ast.Text) bool {
|
||||
for _, stmt := range t.Body {
|
||||
in, ok := stmt.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
switch strings.ToUpper(in.Mnemonic.Text) {
|
||||
case "JAL", "CALL", "BL":
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// loong64Prologue returns the prologue bytes for a loong64 function.
|
||||
func loong64Prologue(fi loong64FrameInfo) []byte {
|
||||
if fi.autosize == 0 {
|
||||
return nil
|
||||
}
|
||||
addiD := l64DualTable["ADDV"].imm
|
||||
return l64WordsLE(
|
||||
l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, 3, 1), // MOVV R1, -autosize(R3)
|
||||
l64irr(addiD, -fi.autosize, 3, 3), // ADDV $-autosize, R3
|
||||
l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1), // MOVV R1, 0(R3)
|
||||
)
|
||||
}
|
||||
|
||||
// loong64Return returns the bytes for a RET: the epilogue (restore LR and
|
||||
// deallocate the frame when present) followed by jirl r0, r1, 0.
|
||||
func loong64Return(fi loong64FrameInfo) []byte {
|
||||
var ws []uint32
|
||||
if fi.autosize != 0 {
|
||||
if !fi.leaf {
|
||||
// MOVV 0(R3), R1 — restore the link register.
|
||||
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1))
|
||||
}
|
||||
// ADDV $autosize, R3 — close the frame.
|
||||
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
|
||||
}
|
||||
// jirl r0, r1, 0 — return.
|
||||
ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0))
|
||||
return l64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// loong64ResolvePseudo translates a pseudo-register memory reference into a
|
||||
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
|
||||
// x-N(SP) → (autosize - N)(SP). Returns base = -1 for an unresolvable
|
||||
// reference (SB: static data, handled by the relocation path).
|
||||
func loong64ResolvePseudo(sym *ast.Symbol, fi loong64FrameInfo) (base int, off int32) {
|
||||
if sym == nil {
|
||||
return -1, 0
|
||||
}
|
||||
switch sym.Pseudo {
|
||||
case "FP":
|
||||
return 3, int32(sym.Offset) + int32(fi.autosize) + 8
|
||||
case "SP":
|
||||
return 3, int32(fi.autosize) + int32(sym.Offset)
|
||||
case "SB":
|
||||
return -1, int32(sym.Offset)
|
||||
}
|
||||
return -1, 0
|
||||
}
|
||||
@@ -0,0 +1,367 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestLOONG64_sys exercises the no-operand system instructions and the
|
||||
// bare-data pseudo-instructions. The words match `go tool asm`
|
||||
// (GOARCH=loong64) for the same source.
|
||||
func TestLOONG64_sys(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·sys(SB), NOSPLIT, $0
|
||||
NOOP
|
||||
UNDEF
|
||||
WORD $0x12345678
|
||||
SYSCALL $0x10
|
||||
BREAK $0x20
|
||||
DBAR $1
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x03400000, // andi r0, r0, 0 (NOOP)
|
||||
0x002A0000, // break 0 (UNDEF)
|
||||
0x12345678, // WORD
|
||||
0x002B0010, // syscall 0x10
|
||||
0x002A0020, // break 0x20
|
||||
0x38720001, // dbar 1
|
||||
0x4C000020, // jirl r0, r1, 0
|
||||
)
|
||||
}
|
||||
|
||||
// TestLOONG64_branches21 exercises the single-register branch forms: the
|
||||
// 21-bit BEQZ/BNEZ/BLTZ/BGEZ and the rd-field BGTZ/BLEZ.
|
||||
func TestLOONG64_branches21(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·b21(SB), NOSPLIT, $0
|
||||
BEQZ R4, done
|
||||
BNEZ R5, done
|
||||
BLTZ R6, done
|
||||
BGEZ R7, done
|
||||
BGTZ R8, done
|
||||
BLEZ R9, done
|
||||
done:
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x40001880, // beqz r4, +6
|
||||
0x440014A0, // bnez r5, +5
|
||||
0x600010C0, // bltz r6, +4
|
||||
0x64000CE0, // bgez r7, +3
|
||||
0x60000808, // bgtz r8, +2 (register in the rd field)
|
||||
0x64000409, // blez r9, +1
|
||||
0x4C000020, // jirl r0, r1, 0
|
||||
)
|
||||
}
|
||||
|
||||
// TestLOONG64_fma exercises the four fused multiply-add forms (4 and 3
|
||||
// operand spellings).
|
||||
func TestLOONG64_fma(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·fma(SB), NOSPLIT, $0
|
||||
FMADDD F0, F1, F2, F3
|
||||
FMSUBD F4, F5, F6
|
||||
FNMADDD F7, F8, F9, F10
|
||||
FNMSUBD F11, F12, F13
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x08200443, // fmadd.d f3, f2, f1, f0
|
||||
0x086214C6, // fmsub.d f6, f5, f5, f4
|
||||
0x08A3A12A, // fnmadd.d f10, f9, f8, f7
|
||||
0x08E5B1AD, // fnmsub.d f13, f12, f12, f11
|
||||
0x4C000020,
|
||||
)
|
||||
}
|
||||
|
||||
// TestLOONG64_bitops exercises BSTRINS/BSTRPICK (the 6-bit msb/lsb fields)
|
||||
// and ALSL (the sa−1 shift field).
|
||||
func TestLOONG64_bitops(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·bits(SB), NOSPLIT, $0
|
||||
BSTRINSW $3, R4, $0, R5
|
||||
BSTRINSV $3, R4, $1, R6
|
||||
BSTRPICKW $3, R4, $0, R5
|
||||
BSTRPICKV $6, R7, $0, R8
|
||||
ALSLW $1, R4, R5, R6
|
||||
ALSLW $4, R7, R8, R9
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x00630085, // bstrins.w r5, r4, $3, $0
|
||||
0x00830486, // bstrins.d r6, r4, $3, $1
|
||||
0x00638085, // bstrpick.w r5, r4, $3, $0
|
||||
0x00C600E8, // bstrpick.d r8, r7, $6, $0
|
||||
0x00041486, // alsl.w r6, r5, r4, $1 (sa-1)
|
||||
0x0005A0E9, // alsl.w r9, r8, r7, $4
|
||||
0x4C000020,
|
||||
)
|
||||
}
|
||||
|
||||
// TestLOONG64_ptr exercises the 14-bit-offset memory forms (LL/SC/MOVWP/
|
||||
// MOVVP with the offset scaled by 4) and PRELD.
|
||||
func TestLOONG64_ptr(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·ptr(SB), NOSPLIT, $0
|
||||
LLW 8(R14), R15
|
||||
SCW R16, -4(R17)
|
||||
MOVWP 16(R18), R19
|
||||
MOVVP R20, 24(R21)
|
||||
PRELD 32(R22), $0
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x200009CF, // ll.w r15, 8(r14)
|
||||
0x21FFFE30, // sc.w r16, -4(r17)
|
||||
0x24001253, // ldptr.w r19, 16(r18)
|
||||
0x27001AB4, // stptr.d r20, 24(r21)
|
||||
0x2AC082C0, // preld 32(r22), 0
|
||||
0x4C000020,
|
||||
)
|
||||
}
|
||||
|
||||
// TestLOONG64_atomics exercises the AM* read-modify-write forms and
|
||||
// RDTIME, plus the MOVV FP→GP move.
|
||||
func TestLOONG64_atomics(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·atoms(SB), NOSPLIT, $0
|
||||
AMADDW R4, (R5), R6
|
||||
RDTIMED R7, R8
|
||||
MOVV F1, R2
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x386110A6, // amadd.w r6, r5, r4
|
||||
0x000068E8, // rdtime.d r8, r7
|
||||
0x0114B822, // movfr2gr.d r2, f1
|
||||
0x4C000020,
|
||||
)
|
||||
}
|
||||
|
||||
// TestLOONG64_lu52 exercises the LU52I.D immediate form (a gasm extension
|
||||
// the toolchain reaches only through its MOVV expansion).
|
||||
func TestLOONG64_lu52(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·lu52(SB), NOSPLIT, $0
|
||||
LU52ID $0x345, R10
|
||||
LU52ID $0x123, R11, R12
|
||||
ADDV16 $0x10000, R13
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
wantWords(t, code,
|
||||
0x030D154A, // lu52i.d r10, r10, 0x345
|
||||
0x03048D6C, // lu52i.d r12, r11, 0x123
|
||||
0x100005AD, // addu16i.d r13, r13, 0x10000>>16
|
||||
0x4C000020,
|
||||
)
|
||||
}
|
||||
|
||||
// TestLOONG64_sbRefs checks the static-symbol reference forms through the
|
||||
// full file assembly: each pcalau12i+addi.d/ld/st pair carries the
|
||||
// R_LOONG64_ADDR_HI/LO relocation pair, and the immediate fields are left
|
||||
// zero for the linker.
|
||||
func TestLOONG64_sbRefs(t *testing.T) {
|
||||
f, errs := parser.Parse("sb_loong64.s", `#include "textflag.h"
|
||||
TEXT ·sb(SB), NOSPLIT, $0
|
||||
MOVV $·table(SB), R4
|
||||
MOVV ·table+8(SB), R5
|
||||
MOVV R6, ·table(SB)
|
||||
RET
|
||||
|
||||
GLOBL ·table(SB), RODATA, $8
|
||||
DATA ·table+0(SB)/8, $42
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileLOONG64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
if fn.Size != 28 {
|
||||
t.Fatalf("function size = %d, want 28", fn.Size)
|
||||
}
|
||||
var hi, lo int
|
||||
// The three references: $·table (0), ·table+8 (8), ·table (0).
|
||||
wantAdd := []int64{0, 0, 8, 8, 0, 0}
|
||||
for i, r := range fn.Relocs {
|
||||
wantKind := RelLoong64AddrHi
|
||||
wantOff := (i / 2) * 8
|
||||
if i%2 == 1 {
|
||||
wantKind = RelLoong64AddrLo
|
||||
wantOff += 4
|
||||
}
|
||||
if r.Kind != wantKind || r.Off != wantOff || r.Name != "table" || r.Addend != wantAdd[i] {
|
||||
t.Errorf("reloc %d = {kind %v off %d name %q addend %d}", i, r.Kind, r.Off, r.Name, r.Addend)
|
||||
}
|
||||
if r.Kind == RelLoong64AddrHi {
|
||||
hi++
|
||||
} else {
|
||||
lo++
|
||||
}
|
||||
}
|
||||
if hi != 3 || lo != 3 {
|
||||
t.Errorf("relocs = %d hi + %d lo, want 3 + 3", hi, lo)
|
||||
}
|
||||
// The image carries the zero-immediate pair encodings (the linker
|
||||
// fills the immediate fields from the relocations).
|
||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
wantWords(t, code,
|
||||
0x1A000004, // pcalau12i r4, 0
|
||||
0x02C00084, // addi.d r4, r4, 0
|
||||
0x1A00001E, // pcalau12i r30, 0
|
||||
0x28C003C5, // ld.d r5, 0(r30)
|
||||
0x1A00001E, // pcalau12i r30, 0
|
||||
0x29C003C6, // st.d r6, 0(r30)
|
||||
0x4C000020, // jirl r0, r1, 0
|
||||
)
|
||||
}
|
||||
|
||||
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
|
||||
// invalid register operands and operand-count mismatches.
|
||||
func TestLOONG64_errors(t *testing.T) {
|
||||
cases := []string{
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
JMP nowhere
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
BEQZ X0, done
|
||||
done:
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
ADDV R4
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
FMADDD F0, F1
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
AMADDW R4, R5
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
WORD
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
PRELD 32(R4)
|
||||
RET
|
||||
`,
|
||||
`TEXT ·e(SB), NOSPLIT, $0
|
||||
ALSLW $5, R4, R5, R6
|
||||
RET
|
||||
`,
|
||||
}
|
||||
for i, src := range cases {
|
||||
fn := firstTextLOONG64(t, src)
|
||||
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||
t.Errorf("case %d: expected an error, got none", i)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestLOONG64_pcsp checks the stack-adjustment table of a framed function:
|
||||
// the prologue raises the SP delta by autosize (in effect from the third
|
||||
// instruction) and the RET's epilogue restores it to zero, with the pc deltas
|
||||
// in MinLC (4) units — byte-identical to `go tool asm`.
|
||||
func TestLOONG64_pcsp(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
src string
|
||||
want []byte
|
||||
}{
|
||||
{
|
||||
"leaf",
|
||||
`#include "textflag.h"
|
||||
TEXT ·leaf(SB), NOSPLIT, $8-0
|
||||
MOVV R4, R5
|
||||
RET
|
||||
`,
|
||||
[]byte{0x02, 0x02, 0x20, 0x03, 0x1f, 0x01, 0x00},
|
||||
},
|
||||
{
|
||||
"nonleaf",
|
||||
`#include "textflag.h"
|
||||
TEXT ·nonleaf(SB), NOSPLIT, $8-0
|
||||
MOVV R4, R5
|
||||
JAL (R12)
|
||||
RET
|
||||
`,
|
||||
[]byte{0x02, 0x02, 0x20, 0x05, 0x1f, 0x01, 0x00},
|
||||
},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
f, errs := parser.Parse("pcsp_loong64.s", c.src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileLOONG64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||
}
|
||||
if got := pcspTable(img.Funcs[0], 4); !bytes.Equal(got, c.want) {
|
||||
t.Errorf("pcsp = % x, want % x", got, c.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestLOONG64_sbRefsUndefined checks that a reference to a symbol no GLOBL
|
||||
// defines assembles into a relocation and is rejected at object emission.
|
||||
func TestLOONG64_sbRefsUndefined(t *testing.T) {
|
||||
f, errs := parser.Parse("sb_loong64.s", `#include "textflag.h"
|
||||
TEXT ·sb(SB), NOSPLIT, $0
|
||||
MOVV missing(SB), R4
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileLOONG64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||
}
|
||||
if len(img.Funcs[0].Relocs) != 2 {
|
||||
t.Fatalf("relocs = %d, want the HI/LO pair", len(img.Funcs[0].Relocs))
|
||||
}
|
||||
if _, err := img.GOObjectLOONG64("p", "sb_loong64.s"); err == nil {
|
||||
t.Error("expected an unknown-symbol error at emission")
|
||||
}
|
||||
}
|
||||
|
||||
// TestLOONG64_movImmToFp checks the immediate-to-FP move forms.
|
||||
func TestLOONG64_movImmToFp(t *testing.T) {
|
||||
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||
TEXT ·fpmov(SB), NOSPLIT, $0
|
||||
MOVV $0x1, F0
|
||||
MOVW $0x2, F4
|
||||
RET
|
||||
`)
|
||||
code := assembleLOONG64Helper(t, fn)
|
||||
want := []byte{
|
||||
0x00, 0x04, 0x80, 0x03, // ori f0, r0, 1
|
||||
0x04, 0x08, 0x80, 0x03, // ori f4, r0, 2
|
||||
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
|
||||
}
|
||||
if !bytes.Equal(code, want) {
|
||||
t.Errorf("code = % x\nwant % x", code, want)
|
||||
}
|
||||
}
|
||||
+10
-3
@@ -37,7 +37,14 @@ func Idx(base, index Reg, scale int, disp int64, size int) Mem {
|
||||
return Mem{Base: base, Index: index, Scale: scale, Disp: disp, Size: size, HasBase: true, HasIndex: true}
|
||||
}
|
||||
|
||||
// Rip builds a RIP-relative memory operand (RIP)+disp.
|
||||
func Rip(disp int64, size int) Mem {
|
||||
return Mem{Disp: disp, Size: size}
|
||||
// sbMem is a memory operand that references a static (SB) symbol. It encodes
|
||||
// as a RIP-relative reference with a placeholder displacement; the encoder
|
||||
// records a patch site so the file-level layout can fill in the true rel32
|
||||
// once the symbol's address is known.
|
||||
type sbMem struct {
|
||||
size int
|
||||
name string // static symbol name (the GLOBL identifier)
|
||||
addend int64 // byte offset within the symbol
|
||||
}
|
||||
|
||||
func (sbMem) isOperand() {}
|
||||
|
||||
+73
-59
@@ -7,6 +7,8 @@
|
||||
// by round-tripping through golang.org/x/arch's decoder in the tests.
|
||||
package asm
|
||||
|
||||
import "maps"
|
||||
|
||||
import "strings"
|
||||
|
||||
// Reg is an x86-64 register. In Plan 9 assembly the classic names (AX, BX, …)
|
||||
@@ -14,19 +16,24 @@ import "strings"
|
||||
// so the encoder keys off the register's index and lets the mnemonic supply the
|
||||
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
|
||||
// occupy indices 4–7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
|
||||
// those indices but require one.
|
||||
// those indices but require one. The mask flag marks the AVX-512 opmask
|
||||
// registers K0–K7.
|
||||
type Reg struct {
|
||||
idx int
|
||||
size int // informational width implied by the name; the mnemonic decides
|
||||
high bool // AH/CH/DH/BH
|
||||
mask bool // K0–K7 opmask register
|
||||
}
|
||||
|
||||
// Index returns the register number (0–15).
|
||||
// Index returns the register number (0–15 for GPRs, 0–31 for vectors).
|
||||
func (r Reg) Index() int { return r.idx }
|
||||
|
||||
// Size returns the width in bytes implied by the register's name.
|
||||
func (r Reg) Size() int { return r.size }
|
||||
|
||||
// IsMask reports whether r is an AVX-512 opmask register (K0–K7).
|
||||
func (r Reg) IsMask() bool { return r.mask }
|
||||
|
||||
func (r Reg) isOperand() {}
|
||||
|
||||
// needsREX reports whether this register forces a REX prefix at the given
|
||||
@@ -41,45 +48,45 @@ func (r Reg) needsREX(opSize int) bool {
|
||||
|
||||
// Register constants (the size is the width the name implies).
|
||||
var (
|
||||
AL = Reg{0, 1, false}
|
||||
CL = Reg{1, 1, false}
|
||||
DL = Reg{2, 1, false}
|
||||
BL = Reg{3, 1, false}
|
||||
AH = Reg{4, 1, true}
|
||||
CH = Reg{5, 1, true}
|
||||
DH = Reg{6, 1, true}
|
||||
BH = Reg{7, 1, true}
|
||||
SPL = Reg{4, 1, false}
|
||||
BPL = Reg{5, 1, false}
|
||||
SIL = Reg{6, 1, false}
|
||||
DIL = Reg{7, 1, false}
|
||||
AL = Reg{idx: 0, size: 1}
|
||||
CL = Reg{idx: 1, size: 1}
|
||||
DL = Reg{idx: 2, size: 1}
|
||||
BL = Reg{idx: 3, size: 1}
|
||||
AH = Reg{idx: 4, size: 1, high: true}
|
||||
CH = Reg{idx: 5, size: 1, high: true}
|
||||
DH = Reg{idx: 6, size: 1, high: true}
|
||||
BH = Reg{idx: 7, size: 1, high: true}
|
||||
SPL = Reg{idx: 4, size: 1}
|
||||
BPL = Reg{idx: 5, size: 1}
|
||||
SIL = Reg{idx: 6, size: 1}
|
||||
DIL = Reg{idx: 7, size: 1}
|
||||
|
||||
AX = Reg{0, 2, false}
|
||||
CX = Reg{1, 2, false}
|
||||
DX = Reg{2, 2, false}
|
||||
BX = Reg{3, 2, false}
|
||||
SP = Reg{4, 2, false}
|
||||
BP = Reg{5, 2, false}
|
||||
SI = Reg{6, 2, false}
|
||||
DI = Reg{7, 2, false}
|
||||
AX = Reg{idx: 0, size: 2}
|
||||
CX = Reg{idx: 1, size: 2}
|
||||
DX = Reg{idx: 2, size: 2}
|
||||
BX = Reg{idx: 3, size: 2}
|
||||
_ = Reg{idx: 4, size: 2}
|
||||
_ = Reg{idx: 5, size: 2}
|
||||
SI = Reg{idx: 6, size: 2}
|
||||
DI = Reg{idx: 7, size: 2}
|
||||
|
||||
EAX = Reg{0, 4, false}
|
||||
ECX = Reg{1, 4, false}
|
||||
EDX = Reg{2, 4, false}
|
||||
EBX = Reg{3, 4, false}
|
||||
ESP = Reg{4, 4, false}
|
||||
EBP = Reg{5, 4, false}
|
||||
ESI = Reg{6, 4, false}
|
||||
EDI = Reg{7, 4, false}
|
||||
_ = Reg{idx: 0, size: 4}
|
||||
_ = Reg{idx: 1, size: 4}
|
||||
_ = Reg{idx: 2, size: 4}
|
||||
_ = Reg{idx: 3, size: 4}
|
||||
_ = Reg{idx: 4, size: 4}
|
||||
_ = Reg{idx: 5, size: 4}
|
||||
_ = Reg{idx: 6, size: 4}
|
||||
_ = Reg{idx: 7, size: 4}
|
||||
|
||||
RAX = Reg{0, 8, false}
|
||||
RCX = Reg{1, 8, false}
|
||||
RDX = Reg{2, 8, false}
|
||||
RBX = Reg{3, 8, false}
|
||||
RSP = Reg{4, 8, false}
|
||||
RBP = Reg{5, 8, false}
|
||||
RSI = Reg{6, 8, false}
|
||||
RDI = Reg{7, 8, false}
|
||||
_ = Reg{idx: 0, size: 8}
|
||||
_ = Reg{idx: 1, size: 8}
|
||||
_ = Reg{idx: 2, size: 8}
|
||||
_ = Reg{idx: 3, size: 8}
|
||||
_ = Reg{idx: 4, size: 8}
|
||||
_ = Reg{idx: 5, size: 8}
|
||||
_ = Reg{idx: 6, size: 8}
|
||||
_ = Reg{idx: 7, size: 8}
|
||||
)
|
||||
|
||||
// regByName maps an assembly register name (case-insensitive) to a Reg.
|
||||
@@ -91,58 +98,65 @@ func buildRegByName() map[string]Reg {
|
||||
// 64-bit: RAX..RDI, R8..R15.
|
||||
r64 := []string{"RAX", "RCX", "RDX", "RBX", "RSP", "RBP", "RSI", "RDI"}
|
||||
for i, n := range r64 {
|
||||
m[n] = Reg{i, 8, false}
|
||||
m[n] = Reg{idx: i, size: 8}
|
||||
}
|
||||
for i := 8; i <= 15; i++ {
|
||||
m["R"+itoa(i)] = Reg{i, 8, false}
|
||||
m["R"+itoa(i)] = Reg{idx: i, size: 8}
|
||||
}
|
||||
|
||||
// 32-bit: EAX..EDI, R8D..R15D.
|
||||
e32 := []string{"EAX", "ECX", "EDX", "EBX", "ESP", "EBP", "ESI", "EDI"}
|
||||
for i, n := range e32 {
|
||||
m[n] = Reg{i, 4, false}
|
||||
m[n] = Reg{idx: i, size: 4}
|
||||
}
|
||||
for i := 8; i <= 15; i++ {
|
||||
m["R"+itoa(i)+"D"] = Reg{i, 4, false}
|
||||
m["R"+itoa(i)+"D"] = Reg{idx: i, size: 4}
|
||||
}
|
||||
|
||||
// 16-bit: AX..DI, R8W..R15W.
|
||||
w16 := []string{"AX", "CX", "DX", "BX", "SP", "BP", "SI", "DI"}
|
||||
for i, n := range w16 {
|
||||
m[n] = Reg{i, 2, false}
|
||||
m[n] = Reg{idx: i, size: 2}
|
||||
}
|
||||
for i := 8; i <= 15; i++ {
|
||||
m["R"+itoa(i)+"W"] = Reg{i, 2, false}
|
||||
m["R"+itoa(i)+"W"] = Reg{idx: i, size: 2}
|
||||
}
|
||||
|
||||
// 8-bit: AL..BH, SPL..DIL, R8B..R15B.
|
||||
for n, r := range map[string]Reg{
|
||||
maps.Copy(m, map[string]Reg{
|
||||
"AL": AL, "CL": CL, "DL": DL, "BL": BL,
|
||||
"AH": AH, "CH": CH, "DH": DH, "BH": BH,
|
||||
"SPL": SPL, "BPL": BPL, "SIL": SIL, "DIL": DIL,
|
||||
} {
|
||||
m[n] = r
|
||||
}
|
||||
})
|
||||
for i := 8; i <= 15; i++ {
|
||||
m["R"+itoa(i)+"B"] = Reg{i, 1, false}
|
||||
m["R"+itoa(i)+"B"] = Reg{idx: i, size: 1}
|
||||
}
|
||||
|
||||
// Vector: X0..X15 (128-bit, encoded size 16), Y0..Y15 (256-bit, size 32).
|
||||
// Z (512-bit) and K (mask) registers arrive with EVEX/AVX-512 support.
|
||||
for i := 0; i <= 15; i++ {
|
||||
m["X"+itoa(i)] = Reg{i, 16, false}
|
||||
m["Y"+itoa(i)] = Reg{i, 32, false}
|
||||
// Vector: X0..X31 (128-bit, size 16), Y0..Y31 (256-bit, size 32),
|
||||
// Z0..Z31 (512-bit, size 64). Indices 16–31 are only encodable in EVEX
|
||||
// (AVX-512) instructions; the encoder validates that through its tables.
|
||||
for i := 0; i <= 31; i++ {
|
||||
m["X"+itoa(i)] = Reg{idx: i, size: 16}
|
||||
m["Y"+itoa(i)] = Reg{idx: i, size: 32}
|
||||
m["Z"+itoa(i)] = Reg{idx: i, size: 64}
|
||||
}
|
||||
// Opmask: K0..K7.
|
||||
for i := 0; i <= 7; i++ {
|
||||
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// isVec reports whether r is an XMM/YMM vector register.
|
||||
func (r Reg) isVec() bool { return r.size == 16 || r.size == 32 }
|
||||
// isVec reports whether r is an XMM/YMM/ZMM vector register.
|
||||
func (r Reg) isVec() bool { return r.size == 16 || r.size == 32 || r.size == 64 }
|
||||
|
||||
// vecLenBit returns the VEX.L bit for a vector register (X=0/128-bit,
|
||||
// Y=1/256-bit).
|
||||
// vecLenBit returns the vector-length field for a vector register:
|
||||
// 0 (128-bit, VEX.L / EVEX.L'L=00), 1 (256-bit) or 2 (512-bit, EVEX only).
|
||||
func (r Reg) vecLenBit() int {
|
||||
if r.size == 32 {
|
||||
switch r.size {
|
||||
case 64:
|
||||
return 2
|
||||
case 32:
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,564 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
// RISC-V register encoding: maps register names to their 5-bit numbers.
|
||||
// The Go assembler uses the standard RISC-V ABI naming.
|
||||
|
||||
// riscvRegNum returns the 5-bit register number for a RISC-V register name.
|
||||
// Returns -1 if the register is not recognized.
|
||||
func riscvRegNum(name string) int {
|
||||
switch name {
|
||||
// Numbered integer registers.
|
||||
case "X0", "ZERO":
|
||||
return 0
|
||||
case "X1", "RA", "LR":
|
||||
return 1
|
||||
case "X2", "SP":
|
||||
return 2
|
||||
case "X3", "GP":
|
||||
return 3
|
||||
case "X4", "TP":
|
||||
return 4
|
||||
case "X5", "T0":
|
||||
return 5
|
||||
case "X6", "T1":
|
||||
return 6
|
||||
case "X7", "T2":
|
||||
return 7
|
||||
case "X8", "S0", "FP":
|
||||
return 8
|
||||
case "X9", "S1":
|
||||
return 9
|
||||
case "X10", "A0":
|
||||
return 10
|
||||
case "X11", "A1":
|
||||
return 11
|
||||
case "X12", "A2":
|
||||
return 12
|
||||
case "X13", "A3":
|
||||
return 13
|
||||
case "X14", "A4":
|
||||
return 14
|
||||
case "X15", "A5":
|
||||
return 15
|
||||
case "X16", "A6":
|
||||
return 16
|
||||
case "X17", "A7":
|
||||
return 17
|
||||
case "X18", "S2":
|
||||
return 18
|
||||
case "X19", "S3":
|
||||
return 19
|
||||
case "X20", "S4":
|
||||
return 20
|
||||
case "X21", "S5":
|
||||
return 21
|
||||
case "X22", "S6":
|
||||
return 22
|
||||
case "X23", "S7":
|
||||
return 23
|
||||
case "X24", "S8":
|
||||
return 24
|
||||
case "X25", "S9":
|
||||
return 25
|
||||
case "X26", "S10":
|
||||
return 26
|
||||
case "X27", "S11":
|
||||
return 27
|
||||
case "X28", "T3":
|
||||
return 28
|
||||
case "X29", "T4":
|
||||
return 29
|
||||
case "X30", "T5":
|
||||
return 30
|
||||
case "X31", "T6", "TMP":
|
||||
return 31
|
||||
// Floating-point registers (F0-F31).
|
||||
case "F0", "FT0":
|
||||
return 0
|
||||
case "F1", "FT1":
|
||||
return 1
|
||||
case "F2", "FT2":
|
||||
return 2
|
||||
case "F3", "FT3":
|
||||
return 3
|
||||
case "F4", "FT4":
|
||||
return 4
|
||||
case "F5", "FT5":
|
||||
return 5
|
||||
case "F6", "FT6":
|
||||
return 6
|
||||
case "F7", "FT7":
|
||||
return 7
|
||||
case "F8", "FS0":
|
||||
return 8
|
||||
case "F9", "FS1":
|
||||
return 9
|
||||
case "F10", "FA0":
|
||||
return 10
|
||||
case "F11", "FA1":
|
||||
return 11
|
||||
case "F12", "FA2":
|
||||
return 12
|
||||
case "F13", "FA3":
|
||||
return 13
|
||||
case "F14", "FA4":
|
||||
return 14
|
||||
case "F15", "FA5":
|
||||
return 15
|
||||
case "F16", "FA6":
|
||||
return 16
|
||||
case "F17", "FA7":
|
||||
return 17
|
||||
case "F18", "FS2":
|
||||
return 18
|
||||
case "F19", "FS3":
|
||||
return 19
|
||||
case "F20", "FS4":
|
||||
return 20
|
||||
case "F21", "FS5":
|
||||
return 21
|
||||
case "F22", "FS6":
|
||||
return 22
|
||||
case "F23", "FS7":
|
||||
return 23
|
||||
case "F24", "FS8":
|
||||
return 24
|
||||
case "F25", "FS9":
|
||||
return 25
|
||||
case "F26", "FS10":
|
||||
return 26
|
||||
case "F27", "FS11":
|
||||
return 27
|
||||
case "F28", "FT8":
|
||||
return 28
|
||||
case "F29", "FT9":
|
||||
return 29
|
||||
case "F30", "FT10":
|
||||
return 30
|
||||
case "F31", "FT11":
|
||||
return 31
|
||||
default:
|
||||
return -1
|
||||
}
|
||||
}
|
||||
|
||||
// RISC-V instruction encoding parameters.
|
||||
type riscvEnc struct {
|
||||
opcode uint32 // bits [6:0]
|
||||
funct3 uint32 // bits [14:12]
|
||||
funct7 uint32 // bits [31:25]
|
||||
}
|
||||
|
||||
// riscvInstrTable maps RISC-V mnemonics to their encoding.
|
||||
var riscvInstrTable = map[string]riscvEnc{
|
||||
// RV64I — R-type arithmetic/logic.
|
||||
"ADD": {0x33, 0x0, 0x00},
|
||||
"SUB": {0x33, 0x0, 0x20},
|
||||
"SLL": {0x33, 0x1, 0x00},
|
||||
"SLT": {0x33, 0x2, 0x00},
|
||||
"SLTU": {0x33, 0x3, 0x00},
|
||||
"XOR": {0x33, 0x4, 0x00},
|
||||
"SRL": {0x33, 0x5, 0x00},
|
||||
"SRA": {0x33, 0x5, 0x20},
|
||||
"OR": {0x33, 0x6, 0x00},
|
||||
"AND": {0x33, 0x7, 0x00},
|
||||
// RV64I — 32-bit variants (W suffix).
|
||||
"ADDW": {0x3B, 0x0, 0x00},
|
||||
"SUBW": {0x3B, 0x0, 0x20},
|
||||
"SLLW": {0x3B, 0x1, 0x00},
|
||||
"SRLW": {0x3B, 0x5, 0x00},
|
||||
"SRAW": {0x3B, 0x5, 0x20},
|
||||
// RV64I — I-type shift-immediate (shamt in rs2 field).
|
||||
"SLLI": {0x13, 0x1, 0x00},
|
||||
"SRLI": {0x13, 0x5, 0x00},
|
||||
"SRAI": {0x13, 0x5, 0x20},
|
||||
"SLLIW": {0x1B, 0x1, 0x00},
|
||||
"SRLIW": {0x1B, 0x5, 0x00},
|
||||
"SRAIW": {0x1B, 0x5, 0x20},
|
||||
// RV64M — multiply/divide.
|
||||
"MUL": {0x33, 0x0, 0x01},
|
||||
"MULH": {0x33, 0x1, 0x01},
|
||||
"MULHSU": {0x33, 0x2, 0x01},
|
||||
"MULHU": {0x33, 0x3, 0x01},
|
||||
"DIV": {0x33, 0x4, 0x01},
|
||||
"DIVU": {0x33, 0x5, 0x01},
|
||||
"REM": {0x33, 0x6, 0x01},
|
||||
"REMU": {0x33, 0x7, 0x01},
|
||||
// RV64M — 32-bit variants.
|
||||
"MULW": {0x3B, 0x0, 0x01},
|
||||
"DIVW": {0x3B, 0x4, 0x01},
|
||||
"DIVUW": {0x3B, 0x5, 0x01},
|
||||
"REMW": {0x3B, 0x6, 0x01},
|
||||
"REMUW": {0x3B, 0x7, 0x01},
|
||||
// RV64I — I-type arithmetic.
|
||||
"ADDI": {0x13, 0x0, 0x00},
|
||||
"ADDIW": {0x1B, 0x0, 0x00},
|
||||
"SLTI": {0x13, 0x2, 0x00},
|
||||
"SLTIU": {0x13, 0x3, 0x00},
|
||||
"XORI": {0x13, 0x4, 0x00},
|
||||
"ORI": {0x13, 0x6, 0x00},
|
||||
"ANDI": {0x13, 0x7, 0x00},
|
||||
// Loads (I-type).
|
||||
"LB": {0x03, 0x0, 0x00},
|
||||
"LH": {0x03, 0x1, 0x00},
|
||||
"LW": {0x03, 0x2, 0x00},
|
||||
"LD": {0x03, 0x3, 0x00},
|
||||
"LBU": {0x03, 0x4, 0x00},
|
||||
"LHU": {0x03, 0x5, 0x00},
|
||||
"LWU": {0x03, 0x6, 0x00},
|
||||
// Stores (S-type).
|
||||
"SB": {0x23, 0x0, 0x00},
|
||||
"SH": {0x23, 0x1, 0x00},
|
||||
"SW": {0x23, 0x2, 0x00},
|
||||
"SD": {0x23, 0x3, 0x00},
|
||||
// Branches (B-type).
|
||||
"BEQ": {0x63, 0x0, 0x00},
|
||||
"BNE": {0x63, 0x1, 0x00},
|
||||
"BLT": {0x63, 0x4, 0x00},
|
||||
"BGE": {0x63, 0x5, 0x00},
|
||||
"BLTU": {0x63, 0x6, 0x00},
|
||||
"BGEU": {0x63, 0x7, 0x00},
|
||||
// U-type.
|
||||
"LUI": {0x37, 0x0, 0x00},
|
||||
"AUIPC": {0x17, 0x0, 0x00},
|
||||
// System.
|
||||
"ECALL": {0x73, 0x0, 0x00},
|
||||
"EBREAK": {0x73, 0x0, 0x00},
|
||||
"FENCE": {0x0F, 0x0, 0x00},
|
||||
// JALR — indirect jump/call (I-type).
|
||||
"JALR": {0x67, 0x0, 0x00},
|
||||
|
||||
// RV64A — atomics (AMO opcode 0x2F).
|
||||
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
|
||||
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
|
||||
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
|
||||
"AMOADDW": {0x2F, 0x2, 0x00 << 2},
|
||||
"AMOADDD": {0x2F, 0x3, 0x00 << 2},
|
||||
"AMOANDW": {0x2F, 0x2, 0x0C << 2},
|
||||
"AMOANDD": {0x2F, 0x3, 0x0C << 2},
|
||||
"AMOORW": {0x2F, 0x2, 0x06 << 2},
|
||||
"AMOORD": {0x2F, 0x3, 0x06 << 2},
|
||||
"AMOXORW": {0x2F, 0x2, 0x04 << 2},
|
||||
"AMOXORD": {0x2F, 0x3, 0x04 << 2},
|
||||
"AMOMAXW": {0x2F, 0x2, 0x14 << 2},
|
||||
"AMOMAXD": {0x2F, 0x3, 0x14 << 2},
|
||||
"AMOMINW": {0x2F, 0x2, 0x10 << 2},
|
||||
"AMOMIND": {0x2F, 0x3, 0x10 << 2},
|
||||
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2},
|
||||
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2},
|
||||
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
|
||||
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
|
||||
|
||||
// RV64F/D — floating-point arithmetic.
|
||||
"FADDS": {0x53, 0x0, 0x00},
|
||||
"FSUBS": {0x53, 0x0, 0x04},
|
||||
"FMULS": {0x53, 0x0, 0x08},
|
||||
"FDIVS": {0x53, 0x0, 0x0C},
|
||||
"FADDD": {0x53, 0x0, 0x01},
|
||||
"FSUBD": {0x53, 0x0, 0x05},
|
||||
"FMULD": {0x53, 0x0, 0x09},
|
||||
"FDIVD": {0x53, 0x0, 0x0D},
|
||||
"FSQRTS": {0x53, 0x0, 0x2C},
|
||||
"FSQRTD": {0x53, 0x0, 0x2D},
|
||||
// FP loads/stores.
|
||||
"FLW": {0x07, 0x2, 0x00},
|
||||
"FLD": {0x07, 0x3, 0x00},
|
||||
"FSW": {0x27, 0x2, 0x00},
|
||||
"FSD": {0x27, 0x3, 0x00},
|
||||
// FP min/max.
|
||||
"FMINS": {0x53, 0x0, 0x14},
|
||||
"FMAXS": {0x53, 0x1, 0x14},
|
||||
"FMIND": {0x53, 0x0, 0x15},
|
||||
"FMAXD": {0x53, 0x1, 0x15},
|
||||
|
||||
// RV64A — load-reserved / store-conditional (funct5 0x02 / 0x03).
|
||||
"LRW": {0x2F, 0x2, 0x02 << 2},
|
||||
"LRD": {0x2F, 0x3, 0x02 << 2},
|
||||
"SCW": {0x2F, 0x2, 0x03 << 2},
|
||||
"SCD": {0x2F, 0x3, 0x03 << 2},
|
||||
|
||||
// FP compare — result in integer register (funct7 0x50/0x51).
|
||||
"FEQS": {0x53, 0x2, 0x50},
|
||||
"FLTS": {0x53, 0x1, 0x50},
|
||||
"FLES": {0x53, 0x0, 0x50},
|
||||
"FEQD": {0x53, 0x2, 0x51},
|
||||
"FLTD": {0x53, 0x1, 0x51},
|
||||
"FLED": {0x53, 0x0, 0x51},
|
||||
}
|
||||
|
||||
// riscvRType encodes an R-type instruction: funct7 | rs2 | rs1 | funct3 | rd | opcode.
|
||||
func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// riscvAMOType encodes an atomic (AMO) instruction.
|
||||
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode.
|
||||
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2).
|
||||
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value
|
||||
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// FP conversion instructions (FCVT, FMV). These use the rs2 field to
|
||||
// encode the conversion type rather than a register, so they are handled
|
||||
// separately from the general instruction table.
|
||||
type riscvCvtEnc struct {
|
||||
funct7 uint32 // bits [31:25]
|
||||
rs2 uint32 // conversion-type code in bits [24:20]
|
||||
opcode uint32 // always 0x53 (OP-FP)
|
||||
}
|
||||
|
||||
var riscvCvtTable = map[string]riscvCvtEnc{
|
||||
// float → int (rs2 selects the integer width/sign).
|
||||
"FCVTWS": {0x60, 0x0, 0x53}, // float32 → int32
|
||||
"FCVTWUS": {0x60, 0x1, 0x53}, // float32 → uint32
|
||||
"FCVTLS": {0x60, 0x2, 0x53}, // float32 → int64
|
||||
"FCVTLUS": {0x60, 0x3, 0x53}, // float32 → uint64
|
||||
"FCVTWD": {0x61, 0x0, 0x53}, // float64 → int32
|
||||
"FCVTWUD": {0x61, 0x1, 0x53}, // float64 → uint32
|
||||
"FCVTLD": {0x61, 0x2, 0x53}, // float64 → int64
|
||||
"FCVTLUD": {0x61, 0x3, 0x53}, // float64 → uint64
|
||||
// int → float (rs2 selects the integer width/sign).
|
||||
"FCVTSW": {0x68, 0x0, 0x53}, // int32 → float32
|
||||
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
|
||||
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
|
||||
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
|
||||
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
|
||||
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
|
||||
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
|
||||
"FCVTDLU": {0x69, 0x3, 0x53}, // uint64 → float64
|
||||
// float → float width conversion.
|
||||
"FCVTSD": {0x20, 0x1, 0x53}, // float64 → float32
|
||||
"FCVTDS": {0x21, 0x0, 0x53}, // float32 → float64
|
||||
// Bit moves between integer and FP registers (no conversion).
|
||||
"FMVXD": {0x71, 0x0, 0x53}, // float64 → int64 (bit move)
|
||||
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
|
||||
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
|
||||
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
|
||||
}
|
||||
|
||||
// riscvCvtType encodes an FP conversion instruction.
|
||||
// Layout: funct7 | rs2(convtype) | rs1 | funct3(0) | rd | opcode.
|
||||
func riscvCvtType(enc riscvCvtEnc, rd, rs1 int) uint32 {
|
||||
return (enc.funct7 << 25) | (enc.rs2 << 20) | (uint32(rs1) << 15) |
|
||||
(uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// R4-type fused multiply-add instructions (FMADD/FMSUB/FNMSUB/FNMADD).
|
||||
// These take 4 register operands: rs1, rs2, rs3, rd.
|
||||
// Layout: rs3 | fmt | rs2 | rs1 | rm | rd | opcode.
|
||||
type riscvFmaEnc struct {
|
||||
fmt uint32 // bits [26:25]: 0x0 = single, 0x1 = double
|
||||
opcode uint32 // bits [6:0]
|
||||
}
|
||||
|
||||
var riscvFmaTable = map[string]riscvFmaEnc{
|
||||
"FMADDS": {0x0, 0x43}, // rd = rs1*rs2 + rs3
|
||||
"FMADDD": {0x1, 0x43},
|
||||
"FMSUBS": {0x0, 0x47}, // rd = rs1*rs2 - rs3
|
||||
"FMSUBD": {0x1, 0x47},
|
||||
"FNMSUBS": {0x0, 0x4B}, // rd = -(rs1*rs2) + rs3
|
||||
"FNMSUBD": {0x1, 0x4B},
|
||||
"FNMADDS": {0x0, 0x4F}, // rd = -(rs1*rs2) - rs3
|
||||
"FNMADDD": {0x1, 0x4F},
|
||||
}
|
||||
|
||||
// riscvFmaType encodes an R4-type fused multiply-add instruction.
|
||||
func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 {
|
||||
return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) |
|
||||
(uint32(rs1) << 15) | (0x0 << 12) /* rm=dynamic */ | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// CSR (Control and Status Register) instructions.
|
||||
// Format: csr[11:0] | rs1/zimm | funct3 | rd | opcode (0x73).
|
||||
type riscvCsrEnc struct {
|
||||
funct3 uint32 // bits [14:12]
|
||||
imm bool // true for CSRRWI/CSRRSI/CSRRCI (5-bit uimm variant)
|
||||
}
|
||||
|
||||
var riscvCsrTable = map[string]riscvCsrEnc{
|
||||
"CSRRW": {0x1, false}, // rd=CSR, CSR=rs1
|
||||
"CSRRS": {0x2, false}, // rd=CSR, CSR |= rs1
|
||||
"CSRRC": {0x3, false}, // rd=CSR, CSR &= ~rs1
|
||||
"CSRRWI": {0x5, true}, // rd=CSR, CSR=uimm
|
||||
"CSRRSI": {0x6, true}, // rd=CSR, CSR |= uimm
|
||||
"CSRRCI": {0x7, true}, // rd=CSR, CSR &= ~uimm
|
||||
}
|
||||
|
||||
// riscvCsrType encodes a CSR instruction.
|
||||
// csr is the 12-bit CSR address; src is either a register number or a 5-bit
|
||||
// unsigned immediate (depending on enc.imm).
|
||||
func riscvCsrType(enc riscvCsrEnc, rd, src int, csr int32) uint32 {
|
||||
return (uint32(csr&0xFFF) << 20) | (uint32(src&0x1F) << 15) |
|
||||
(enc.funct3 << 12) | (uint32(rd) << 7) | 0x73
|
||||
}
|
||||
|
||||
// riscvIType encodes an I-type instruction: imm[11:0] | rs1 | funct3 | rd | opcode.
|
||||
func riscvIType(enc riscvEnc, rd, rs1 int, imm int32) uint32 {
|
||||
return (uint32(imm&0xFFF) << 20) | (uint32(rs1) << 15) |
|
||||
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// riscvSType encodes an S-type instruction: imm[11:5] | rs2 | rs1 | funct3 | imm[4:0] | opcode.
|
||||
func riscvSType(enc riscvEnc, rs1, rs2 int, imm int32) uint32 {
|
||||
immU := uint32(imm) & 0xFFF
|
||||
return ((immU >> 5) << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||
(enc.funct3 << 12) | ((immU & 0x1F) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// riscvBType encodes a B-type instruction (branches).
|
||||
func riscvBType(enc riscvEnc, rs1, rs2 int, offset int32) uint32 {
|
||||
imm := uint32(offset) & 0x1FFE // bits [12:1], bit 0 is always 0
|
||||
return (((imm >> 12) & 1) << 31) | // imm[12]
|
||||
(((imm >> 5) & 0x3F) << 25) | // imm[10:5]
|
||||
(uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||
(enc.funct3 << 12) |
|
||||
(((imm >> 1) & 0xF) << 8) | // imm[4:1]
|
||||
(((imm >> 11) & 1) << 7) | // imm[11]
|
||||
enc.opcode
|
||||
}
|
||||
|
||||
// riscvUType encodes a U-type instruction: imm[31:12] | rd | opcode.
|
||||
func riscvUType(enc riscvEnc, rd int, imm int32) uint32 {
|
||||
return (uint32(imm) & 0xFFFFF000) | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// riscvJType encodes a J-type instruction (JAL).
|
||||
func riscvJType(rd int, offset int32) uint32 {
|
||||
imm := uint32(offset) & 0x1FFFFE // bits [20:1]
|
||||
return (((imm >> 20) & 1) << 31) | // imm[20]
|
||||
(((imm >> 1) & 0x3FF) << 21) | // imm[10:1]
|
||||
(((imm >> 11) & 1) << 20) | // imm[11]
|
||||
(((imm >> 12) & 0xFF) << 12) | // imm[19:12]
|
||||
(uint32(rd) << 7) |
|
||||
0x6F // JAL opcode
|
||||
}
|
||||
|
||||
// ---- RVC (compressed) encoding helpers ----
|
||||
|
||||
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
|
||||
// prime register field used by compressed instructions (x8–x15).
|
||||
func isRVCIntReg(r int) bool { return r >= 8 && r <= 15 }
|
||||
|
||||
// rvcReg3 returns the 3-bit encoding for registers x8–x15 (0–7).
|
||||
func rvcReg3(r int) uint32 { return uint32(r - 8) }
|
||||
|
||||
// rvcCR encodes a CR-type (register) compressed instruction.
|
||||
// Format: funct4 | rd/rs1 | rs2 | op=2.
|
||||
func rvcCR(funct4, rd, rs2 uint32) uint16 {
|
||||
return uint16((funct4 << 12) | (rd << 7) | (rs2 << 2) | 0x2)
|
||||
}
|
||||
|
||||
// rvcCI encodes a CI-type (immediate) compressed instruction.
|
||||
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW — linear 6-bit immediate.
|
||||
func rvcCI(funct3, rd uint32, imm uint32) uint16 {
|
||||
return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x1)
|
||||
}
|
||||
|
||||
// rvcSLLI encodes C.SLLI, which shares funct3=0 with C.ADDI but lives in the
|
||||
// op=10 quadrant (unlike C.ADDI's op=01).
|
||||
func rvcSLLI(rd, shamt uint32) uint16 {
|
||||
return uint16(((shamt>>5)&1)<<12 | (rd << 7) | (shamt&0x1F)<<2 | 0x2)
|
||||
}
|
||||
|
||||
// encodeRVCPattern extracts the bits listed in pattern (MSB first) from imm
|
||||
// into a packed value, matching cmd/internal/obj/riscv's encodeBitPattern.
|
||||
func encodeRVCPattern(imm uint32, pattern []int) uint32 {
|
||||
packed := uint32(0)
|
||||
for _, bit := range pattern {
|
||||
packed = packed<<1 | (imm>>bit)&1
|
||||
}
|
||||
return packed
|
||||
}
|
||||
|
||||
// rvcLSP encodes a stack-relative compressed load (op=10 quadrant): C.LWSP
|
||||
// (funct3=2, 4-byte scale), C.LDSP (funct3=3) or C.FLDSP (funct3=1, 8-byte
|
||||
// scale). offset is the full byte offset.
|
||||
func rvcLSP(funct3, rd uint32, offset uint32) uint16 {
|
||||
pattern := []int{5, 4, 3, 8, 7, 6}
|
||||
if funct3 == 0x2 {
|
||||
pattern = []int{5, 4, 3, 2, 7, 6}
|
||||
}
|
||||
packed := uint32(0)
|
||||
for i, b := range pattern {
|
||||
packed |= ((offset >> b) & 1) << (5 - i)
|
||||
}
|
||||
return uint16((funct3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x2)
|
||||
}
|
||||
|
||||
// rvcSSP encodes a stack-relative compressed store (op=10 quadrant): C.SWSP
|
||||
// (funct3=6, 4-byte scale), C.SDSP (funct3=7) or C.FSDSP (funct3=5, 8-byte
|
||||
// scale). offset is the full byte offset.
|
||||
func rvcSSP(funct3, rs2 uint32, offset uint32) uint16 {
|
||||
pattern := []int{5, 4, 3, 8, 7, 6}
|
||||
if funct3 == 0x6 {
|
||||
pattern = []int{5, 4, 3, 2, 7, 6}
|
||||
}
|
||||
packed := uint32(0)
|
||||
for i, b := range pattern {
|
||||
packed |= ((offset >> b) & 1) << (5 - i)
|
||||
}
|
||||
return uint16((funct3 << 13) | (packed << 7) | (rs2 << 2) | 0x2)
|
||||
}
|
||||
|
||||
// rvcCL encodes a register-relative compressed load (op=00 quadrant): C.LW
|
||||
// (funct3=2), C.LD (funct3=3) or C.FLD (funct3=1). imm is the full byte
|
||||
// offset; the immediate bits are extracted per the RISC-V CL format.
|
||||
func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 {
|
||||
pattern := []int{5, 4, 3, 7, 6}
|
||||
if funct3 == 0x2 {
|
||||
pattern = []int{5, 4, 3, 2, 6}
|
||||
}
|
||||
packed := encodeRVCPattern(imm, pattern)
|
||||
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rd << 2))
|
||||
}
|
||||
|
||||
// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
|
||||
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
|
||||
// offset; the immediate bits are extracted per the RISC-V CS format.
|
||||
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
|
||||
pattern := []int{5, 3, 7, 6}
|
||||
if funct3 == 0x6 {
|
||||
pattern = []int{5, 3, 2, 6}
|
||||
}
|
||||
packed := encodeRVCPattern(imm, pattern)
|
||||
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
|
||||
}
|
||||
|
||||
// rvcCIW encodes a CIW-type compressed immediate wide instruction: C.ADDI4SPN
|
||||
// (funct3=0). imm is the raw byte offset.
|
||||
func rvcCIW(funct3, rd uint32, imm uint32) uint16 {
|
||||
packed := encodeRVCPattern(imm, []int{5, 4, 9, 8, 7, 6, 2, 3})
|
||||
return uint16((funct3 << 13) | (packed << 5) | (rd << 2))
|
||||
}
|
||||
|
||||
// rvcCA encodes a CA-type (arithmetic) compressed instruction.
|
||||
// Format: funct6[15:10] | rd'/rs1'[9:7] | funct2[6:5] | rs2'[4:2] | op=01.
|
||||
func rvcCA(funct6, funct2, rd, rs2 uint32) uint16 {
|
||||
return uint16((funct6 << 10) | (rd << 7) | (funct2 << 5) | (rs2 << 2) | 0x1)
|
||||
}
|
||||
|
||||
// rvcCBShift encodes a CB-type shift/immediate compressed instruction
|
||||
// (C.SRLI, C.SRAI, C.ANDI). rd is the 3-bit prime-register index; imm is
|
||||
// the 6-bit shamt/immediate; funct2 selects the operation (0=SRLI, 1=SRAI,
|
||||
// 2=ANDI).
|
||||
func rvcCBShift(funct2, rd, imm uint32) uint16 {
|
||||
return uint16((0x4 << 13) | ((imm>>5)&1)<<12 | (funct2 << 10) | (rd << 7) | (imm&0x1F)<<2 | 0x1)
|
||||
}
|
||||
|
||||
// rvcADDI16SP encodes C.ADDI16SP: ADDI rd, imm, rd for the stack pointer
|
||||
// with a 10-bit signed, 16-byte-scaled immediate. imm is the raw byte
|
||||
// offset; the immediate bits are extracted in the order [9|4|6|8:7|5].
|
||||
func rvcADDI16SP(rd uint32, imm int32) uint16 {
|
||||
u := uint32(imm)
|
||||
packed := uint32(0)
|
||||
for _, bit := range []uint{9, 4, 6, 8, 7, 5} {
|
||||
packed = packed<<1 | (u>>bit)&1
|
||||
}
|
||||
return uint16((0x3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x1)
|
||||
}
|
||||
@@ -0,0 +1,763 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// firstTextRISCV parses assembly source and returns the first TEXT function body.
|
||||
func firstTextRISCV(t *testing.T, src string) *ast.Text {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse("f_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
for _, d := range f.Decls {
|
||||
if fn, ok := d.(*ast.Text); ok {
|
||||
return fn
|
||||
}
|
||||
}
|
||||
t.Fatal("no TEXT found")
|
||||
return nil
|
||||
}
|
||||
|
||||
// assembleRISCVHelper assembles one TEXT function and returns its code bytes.
|
||||
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
|
||||
t.Helper()
|
||||
code, _, _, _, _, err := assembleRISCV(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
return code
|
||||
}
|
||||
|
||||
func TestRISCV_add(t *testing.T) {
|
||||
// func add(a, b int64) int64
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOV a+0(FP), X10
|
||||
MOV b+8(FP), X11
|
||||
ADD X11, X10, X10
|
||||
MOV X10, ret+16(FP)
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// should be 12 bytes with RVC: C.LDSP + C.LDSP + ADD + C.SDSP + C.JR
|
||||
_ = code
|
||||
if len(code) == 0 {
|
||||
t.Error("empty output")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_arithmetic(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·arith(SB), NOSPLIT, $0
|
||||
ADD X10, X11, X12
|
||||
SUB X12, X13, X14
|
||||
MUL X14, X15, X16
|
||||
DIV X16, X17, X18
|
||||
REM X18, X19, X20
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 5 R-type instructions + RET = 5*4 + 4 = 24
|
||||
if len(code) != 24 {
|
||||
t.Errorf("expected 24 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_loadStore(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·mem(SB), NOSPLIT, $0
|
||||
LD (X10), X11
|
||||
SD X11, (X12)
|
||||
LW (X13), X14
|
||||
SW X14, (X15)
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// Four register-relative loads/stores compress (2B each) + JALR (4B) = 12.
|
||||
if len(code) != 12 {
|
||||
t.Errorf("expected 12 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_immediate(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·imm(SB), NOSPLIT, $0
|
||||
ADDI $42, X10, X11
|
||||
ANDI $0xFF, X11, X12
|
||||
ORI $1, X12, X13
|
||||
XORI $0, X13, X14
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 4 I-type + JALR = 4*4 + 4 = 20
|
||||
if len(code) != 20 {
|
||||
t.Errorf("expected 20 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_branches(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·br(SB), NOSPLIT, $0
|
||||
ADDI $1, X10, X10
|
||||
loop:
|
||||
BEQ X10, X11, done
|
||||
ADDI $1, X10, X10
|
||||
JMP loop
|
||||
done:
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
_ = code
|
||||
if len(code) == 0 {
|
||||
t.Error("empty output")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_MOV_imm_small(t *testing.T) {
|
||||
// MOV $42, rd → ADDI (fits in 12 bits, but not C.LI's 6-bit immediate).
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·small(SB), NOSPLIT, $0
|
||||
MOV $42, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// ADDI (4B) + JALR (4B) = 8
|
||||
if len(code) != 8 {
|
||||
t.Errorf("expected 8 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_MOV_imm_large(t *testing.T) {
|
||||
// MOV $0x12345, rd → C.LUI $18 (2B) + ADDIW $837 (4B).
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·large(SB), NOSPLIT, $0
|
||||
MOV $0x12345, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.LUI (2B) + ADDIW (4B) + JALR (4B) = 10
|
||||
if len(code) != 10 {
|
||||
t.Errorf("expected 10 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_MOV_reg(t *testing.T) {
|
||||
// MOV rs, rd → ADDI $0, rs, rd, compresses to C.MV
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·reg(SB), NOSPLIT, $0
|
||||
MOV X10, X11
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.MV (2B) + JALR (4B) = 6
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes, got %d (% x)", len(code), code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_MOV_frame(t *testing.T) {
|
||||
// MOV name+off(FP), rd → load with frame mapping
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·frame(SB), NOSPLIT, $0-8
|
||||
MOV a+0(FP), X10
|
||||
MOV X10, ret+0(FP)
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.LDSP (2B) + C.SDSP (2B) + JALR (4B) = 8
|
||||
if len(code) != 8 {
|
||||
t.Errorf("expected 8 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_loadStore(t *testing.T) {
|
||||
// Verify that loads/stores from SP are compressed.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·rvcstore(SB), NOSPLIT, $0
|
||||
LD 0(SP), X10
|
||||
SD X10, 8(SP)
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.LDSP (2B) + C.SDSP (2B) + JALR (4B) = 8
|
||||
if len(code) != 8 {
|
||||
t.Errorf("expected 8 bytes, got %d (% x)", len(code), code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_atomics(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·amo(SB), NOSPLIT, $0
|
||||
AMOADDD X10, (X11), X12
|
||||
LRD (X13), X14
|
||||
SCD X15, (X16), X17
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 3 AMO instructions (4B each) + JALR (4B) = 16
|
||||
if len(code) != 16 {
|
||||
t.Errorf("expected 16 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_fpArith(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·fpadd(SB), NOSPLIT, $0
|
||||
FADDD F10, F11, F12
|
||||
FSUBD F12, F13, F14
|
||||
FMULD F14, F15, F16
|
||||
FDIVD F16, F17, F18
|
||||
FSQRTD F18, F19
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 5 FP instructions (4B each) + JALR (4B) = 24
|
||||
if len(code) != 24 {
|
||||
t.Errorf("expected 24 bytes, got %d (%d)", len(code), len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_csr(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·csrtest(SB), NOSPLIT, $0
|
||||
CSRRS $0x300, X0, X10
|
||||
CSRRW $0x305, X10, X11
|
||||
CSRRSI $0x304, $5, X12
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 3 CSR instructions (4B each) + JALR (4B) = 16
|
||||
if len(code) != 16 {
|
||||
t.Errorf("expected 16 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_fma(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·fmatest(SB), NOSPLIT, $0
|
||||
FMADDD F10, F11, F12, F13
|
||||
FMSUBD F13, F14, F15, F16
|
||||
FNMSUBD F16, F17, F18, F19
|
||||
FNMADDD F19, F10, F11, F12
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 4 FMA instructions (4B each) + JALR (4B) = 20
|
||||
if len(code) != 20 {
|
||||
t.Errorf("expected 20 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_conversions(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cvt(SB), NOSPLIT, $0
|
||||
FCVTDL X10, F10
|
||||
FCVTLD F10, X11
|
||||
FMVXD F10, X12
|
||||
FMVDX X12, F11
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 4 conversion instructions (4B each) + JALR (4B) = 20
|
||||
if len(code) != 20 {
|
||||
t.Errorf("expected 20 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_fpCmp(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cmp(SB), NOSPLIT, $0
|
||||
FEQD F10, F11, X10
|
||||
FLTD F12, F13, X11
|
||||
FLED F14, F15, X12
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 3 FP compare (4B each) + JALR (4B) = 16
|
||||
if len(code) != 16 {
|
||||
t.Errorf("expected 16 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_forwardBranch(t *testing.T) {
|
||||
// Forward label reference — must not fail.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·fwd(SB), NOSPLIT, $0
|
||||
ADDI $1, X10, X10
|
||||
BEQ X10, X11, done
|
||||
ADDI $1, X10, X10
|
||||
done:
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
_ = code
|
||||
if len(code) == 0 {
|
||||
t.Error("empty output")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_ADDI(t *testing.T) {
|
||||
// ADDI where rd=rs1 and small imm → C.ADDI
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·caddi(SB), NOSPLIT, $0
|
||||
ADDI $5, X10, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.ADDI (2B) + JALR (4B) = 6
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_LI(t *testing.T) {
|
||||
// ADDI X0, $imm, rd → C.LI
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cli(SB), NOSPLIT, $0
|
||||
ADDI $7, X0, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.LI (2B) + JALR (4B) = 6
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_LUI(t *testing.T) {
|
||||
// LUI rd, small nonzero imm → C.LUI
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·clui(SB), NOSPLIT, $0
|
||||
LUI X10, $1
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.LUI (2B) + JALR (4B) = 6
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_AssembleFile(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOV a+0(FP), X10
|
||||
RET
|
||||
|
||||
TEXT ·sub(SB), NOSPLIT, $0
|
||||
SUB X10, X11, X12
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("t_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
if len(img.Funcs) != 2 {
|
||||
t.Fatalf("expected 2 functions, got %d", len(img.Funcs))
|
||||
}
|
||||
// func add: C.LDSP(2) + JALR(4) = 6
|
||||
if img.Funcs[0].Size != 6 {
|
||||
t.Errorf("add: expected 6 bytes, got %d", img.Funcs[0].Size)
|
||||
}
|
||||
// func sub: SUB(4) + JALR(4) = 8
|
||||
if img.Funcs[1].Size != 8 {
|
||||
t.Errorf("sub: expected 8 bytes, got %d", img.Funcs[1].Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_encodings(t *testing.T) {
|
||||
// Smoke test that all known RISC-V mnemonics encode successfully.
|
||||
tests := []struct {
|
||||
name, src string
|
||||
wantBytes int
|
||||
}{
|
||||
{"ADD", "ADD X10, X11, X12\nRET\n", 8},
|
||||
{"SUBW", "SUBW X10, X11, X12\nRET\n", 8},
|
||||
{"MUL", "MUL X10, X11, X12\nRET\n", 8},
|
||||
{"DIVW", "DIVW X10, X11, X12\nRET\n", 8},
|
||||
{"REMUW", "REMUW X10, X11, X12\nRET\n", 8},
|
||||
{"ADDIW", "ADDIW $5, X10, X11\nRET\n", 8},
|
||||
{"SLLI", "SLLI $3, X10, X11\nRET\n", 8}, // ADDI+SLLI? No, SLLI uses I-type
|
||||
{"SRLI", "SRLI $2, X10, X11\nRET\n", 8},
|
||||
{"SRAI", "SRAI $1, X10, X11\nRET\n", 8},
|
||||
{"LB", "LB (X10), X11\nRET\n", 8},
|
||||
{"LBU", "LBU (X10), X11\nRET\n", 8},
|
||||
{"LH", "LH (X10), X11\nRET\n", 8},
|
||||
{"LHU", "LHU (X10), X11\nRET\n", 8},
|
||||
{"LWU", "LWU (X10), X11\nRET\n", 8},
|
||||
{"SB", "SB X10, (X11)\nRET\n", 8},
|
||||
{"SH", "SH X10, (X11)\nRET\n", 8},
|
||||
{"SW", "SW X10, (X11)\nRET\n", 6},
|
||||
{"LUI", "LUI X10, $0x12345\nRET\n", 8},
|
||||
{"AUIPC", "AUIPC X10, $0\nRET\n", 8},
|
||||
{"FLW", "FLW (X10), F10\nRET\n", 8},
|
||||
{"FSW", "FSW F10, (X11)\nRET\n", 8},
|
||||
{"FADDS", "FADDS F10, F11, F12\nRET\n", 8},
|
||||
{"FMINS", "FMINS F10, F11, F12\nRET\n", 8},
|
||||
{"FMAXD", "FMAXD F10, F11, F12\nRET\n", 8},
|
||||
{"FCVTSD", "FCVTSD F10, F11\nRET\n", 8},
|
||||
{"FCVTDS", "FCVTDS F10, F11\nRET\n", 8},
|
||||
{"FMVXW", "FMVXW F10, X10\nRET\n", 8},
|
||||
{"FMADD_S", "FMADDS F10, F11, F12, F13\nRET\n", 8},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·`+tt.name+`(SB), NOSPLIT, $0
|
||||
`+tt.src)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
if len(code) != tt.wantBytes {
|
||||
t.Errorf("expected %d bytes, got %d", tt.wantBytes, len(code))
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_branch(t *testing.T) {
|
||||
// Branches are never RVC-compressed (no C.BEQZ/C.BNEZ), matching go tool asm.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cbeqz(SB), NOSPLIT, $0
|
||||
ADDI $1, X10, X10
|
||||
BEQ X10, X0, done
|
||||
ADDI $1, X10, X10
|
||||
done:
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.ADDI(2) + BEQ(4) + C.ADDI(2) + JALR(4) = 12
|
||||
if len(code) != 12 {
|
||||
t.Errorf("expected 12 bytes with uncompressed BEQ, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CJ(t *testing.T) {
|
||||
// JMP target → JAL X0 (never compressed to C.J), matching go tool asm.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cj(SB), NOSPLIT, $0
|
||||
JMP done
|
||||
done:
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// JAL(4) + JALR(4) = 8
|
||||
if len(code) != 8 {
|
||||
t.Errorf("expected 8 bytes with uncompressed JMP, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CADD(t *testing.T) {
|
||||
// ADD where rd==rs1 and both in prime regs → C.ADD.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cadd(SB), NOSPLIT, $0
|
||||
ADD X10, X11, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.ADD(2) + JALR(4) = 6
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes with C.ADD, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CADD_commute(t *testing.T) {
|
||||
// ADD where rd==rs2 (commutative swap) → C.ADD.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cadd2(SB), NOSPLIT, $0
|
||||
ADD X11, X10, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.ADD(2) + JALR(4) = 6
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes with C.ADD (commuted), got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CSUB(t *testing.T) {
|
||||
// SUB rs2, rs1, rd → C.SUB when rd == rs1 and both in prime regs.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·csub(SB), NOSPLIT, $0
|
||||
SUB X11, X10, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// SUB X11, X10, X10 → rs2=X11, rs1=X10, rd=X10; rd==rs1 → C.SUB (2B) + JALR (4B) = 6.
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes with C.SUB, got %d (% x)", len(code), code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CXOR(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cxor(SB), NOSPLIT, $0
|
||||
XOR X10, X11, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes with C.XOR, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_COR(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cor(SB), NOSPLIT, $0
|
||||
OR X10, X11, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes with C.OR, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CAND(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cand(SB), NOSPLIT, $0
|
||||
AND X10, X11, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes with C.AND, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CFLDSP(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cfldsp(SB), NOSPLIT, $0-8
|
||||
FLD a+0(FP), F10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.FLDSP(2) + JALR(4) = 6
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes with C.FLDSP, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CFSDSP(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cfsdsp(SB), NOSPLIT, $0-8
|
||||
FSD F10, ret+0(FP)
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.FSDSP(2) + JALR(4) = 6
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes with C.FSDSP, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_SB_addr(t *testing.T) {
|
||||
// MOV $sym<>(SB), rd → AUIPC + ADDI (8 bytes for SB).
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·sbaddr(SB), NOSPLIT, $0
|
||||
MOV $answer<>(SB), X10
|
||||
RET
|
||||
GLOBL answer<>(SB), RODATA, $8
|
||||
DATA answer<>+0(SB)/8, $42
|
||||
`
|
||||
f, errs := parser.Parse("t_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
// AUIPC(4) + ADDI(4) + JALR(4) = 12
|
||||
if img.Funcs[0].Size != 12 {
|
||||
t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_SB_store(t *testing.T) {
|
||||
// MOV rd, sym<>(SB) → AUIPC + SD (8 bytes for SB).
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·sbstore(SB), NOSPLIT, $0
|
||||
MOV X10, result<>(SB)
|
||||
RET
|
||||
GLOBL result<>(SB), NOPTR, $8
|
||||
`
|
||||
f, errs := parser.Parse("t_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
// AUIPC X31(4) + SD X10,0(X31)(4) + JALR(4) = 12
|
||||
if img.Funcs[0].Size != 12 {
|
||||
t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_ELF(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·simple(SB), NOSPLIT, $0
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("t_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
obj, err := img.ELFRISCVObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFRISCVObject: %v", err)
|
||||
}
|
||||
if len(obj) < 4 || obj[0] != 0x7f || obj[1] != 'E' || obj[2] != 'L' || obj[3] != 'F' {
|
||||
t.Fatal("not a valid ELF file")
|
||||
}
|
||||
if len(obj) >= 20 {
|
||||
machine := uint16(obj[18]) | uint16(obj[19])<<8
|
||||
if machine != 243 {
|
||||
t.Errorf("e_machine = %d, want 243 (EM_RISCV)", machine)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_ELF_withData(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·get(SB), NOSPLIT, $0
|
||||
RET
|
||||
GLOBL val<>(SB), RODATA, $4
|
||||
DATA val<>+0(SB)/4, $7
|
||||
`
|
||||
f, errs := parser.Parse("t_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
if len(img.DataSyms) != 1 {
|
||||
t.Fatalf("expected 1 data symbol, got %d", len(img.DataSyms))
|
||||
}
|
||||
if img.DataSyms[0].Name != "val" {
|
||||
t.Errorf("data symbol name = %q, want val", img.DataSyms[0].Name)
|
||||
}
|
||||
if img.DataSyms[0].Size != 4 {
|
||||
t.Errorf("data symbol size = %d, want 4", img.DataSyms[0].Size)
|
||||
}
|
||||
obj, err := img.ELFRISCVObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFRISCVObject: %v", err)
|
||||
}
|
||||
_ = obj
|
||||
}
|
||||
|
||||
func TestRISCV_SB_load(t *testing.T) {
|
||||
// MOV sym<>(SB), rd → AUIPC + LD (8 bytes for SB).
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·sbload(SB), NOSPLIT, $0
|
||||
MOV answer<>(SB), X10
|
||||
RET
|
||||
GLOBL answer<>(SB), RODATA, $8
|
||||
DATA answer<>+0(SB)/8, $42
|
||||
`
|
||||
f, errs := parser.Parse("t_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
// AUIPC(4) + LD(4) + JALR(4) = 12
|
||||
if img.Funcs[0].Size != 12 {
|
||||
t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_system_instrs(t *testing.T) {
|
||||
// Test FENCE, ECALL, EBREAK encoding.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·sys(SB), NOSPLIT, $0
|
||||
FENCE
|
||||
ECALL
|
||||
EBREAK
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// FENCE(4) + ECALL(4) + C.EBREAK(2) + JALR(4) = 14
|
||||
if len(code) != 14 {
|
||||
t.Errorf("expected 14 bytes, got %d (% x)", len(code), code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_MOV_sym_FP_error(t *testing.T) {
|
||||
// MOV $sym(FP), rd should return an error (unsupported).
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·badfp(SB), NOSPLIT, $0
|
||||
MOV $arg(FP), X10
|
||||
RET
|
||||
`)
|
||||
_, _, _, _, _, err := assembleRISCV(fn)
|
||||
if err == nil {
|
||||
t.Error("expected error for MOV $arg(FP), got nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_CALL(t *testing.T) {
|
||||
// CALL sym(SB) → JAL X1, sym(SB) with a single R_RISCV_JAL relocation.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·calltest(SB), NOSPLIT, $0
|
||||
CALL ext(SB)
|
||||
RET
|
||||
`)
|
||||
code, _, relocs, _, _, err := assembleRISCV(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
// prologue (8) + JAL (4) + epilogue+JALR (8) = 20
|
||||
if len(code) != 20 {
|
||||
t.Fatalf("expected 20 bytes with CALL sym(SB), got %d", len(code))
|
||||
}
|
||||
if len(relocs) != 1 {
|
||||
t.Fatalf("relocs = %d, want 1", len(relocs))
|
||||
}
|
||||
r := relocs[0]
|
||||
if r.Kind != RelRISCVJal || r.Name != "ext" || r.Off != 8 || r.After != 12 || r.Addend != 0 {
|
||||
t.Errorf("reloc = {kind %v off %d after %d name %q addend %d}", r.Kind, r.Off, r.After, r.Name, r.Addend)
|
||||
}
|
||||
// The JAL instruction itself is JAL X1, 0 at function offset 8.
|
||||
wantJAL := wordLE(riscvJType(1, 0))
|
||||
if !bytes.Equal(code[8:12], wantJAL) {
|
||||
t.Errorf("JAL = % x, want % x", code[8:12], wantJAL)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_CALL_local_error(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·calllocal(SB), NOSPLIT, $0
|
||||
CALL sub
|
||||
sub:
|
||||
RET
|
||||
`)
|
||||
_, _, _, _, _, err := assembleRISCV(fn)
|
||||
if err == nil {
|
||||
t.Error("expected error for CALL to local label, got nil")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,182 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
)
|
||||
|
||||
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
|
||||
//
|
||||
// Go's riscv64 functions have no hardware frame pointer: FP and SP are
|
||||
// synthetic registers resolved against the hardware stack pointer (X2) and
|
||||
// the frame size. The return address lives in the link register (X1, RA/LR).
|
||||
//
|
||||
// The autosize is the real stack adjustment: the declared local frame plus
|
||||
// the 8 bytes for the saved link register (the toolchain's FixedFrameSize).
|
||||
// A leaf function with a zero frame gets no prologue at all.
|
||||
//
|
||||
// Prologue (autosize > 0), byte-identical to the toolchain:
|
||||
//
|
||||
// MOV LR, -autosize(SP) // save LR below the new SP (traceback-safe)
|
||||
// ADDI $-autosize, SP, SP // open the frame
|
||||
// MOV LR, 0(SP) // save LR again at SP (signal-safety)
|
||||
//
|
||||
// Epilogue (autosize > 0): MOV 0(SP), LR; ADDI $autosize, SP, SP; the RET's
|
||||
// uncompressed JALR X0, 0(X1) follows. The toolchain restores LR on every
|
||||
// frame, leaf or not.
|
||||
|
||||
// riscvFrameInfo holds the frame layout derived from a TEXT directive.
|
||||
type riscvFrameInfo struct {
|
||||
autosize int // the real SP adjustment (locals + saved LR)
|
||||
}
|
||||
|
||||
// riscvComputeFrame derives the frame layout for a TEXT function.
|
||||
func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
|
||||
frame := frameSize(t)
|
||||
if frame != 0 || !riscvIsLeaf(t) {
|
||||
// FixedFrameSize = 8: space for the saved link register. A
|
||||
// zero-frame non-leaf function still opens an 8-byte frame for LR.
|
||||
return riscvFrameInfo{autosize: frame + 8}
|
||||
}
|
||||
return riscvFrameInfo{}
|
||||
}
|
||||
|
||||
// riscvIsLeaf reports whether a function contains no call instructions.
|
||||
// CALL always links; JAL/JALR link only when their destination register is
|
||||
// the link register (X1), matching cmd/internal/obj/riscv's containsCall.
|
||||
func riscvIsLeaf(t *ast.Text) bool {
|
||||
for _, stmt := range t.Body {
|
||||
in, ok := stmt.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
switch strings.ToUpper(in.Mnemonic.Text) {
|
||||
case "CALL":
|
||||
return false
|
||||
case "JAL":
|
||||
// JAL rd, target — a call only when rd is the link register.
|
||||
if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 {
|
||||
return false
|
||||
}
|
||||
case "JALR":
|
||||
// JALR rs1, rd — a call when rd is X1; JALR offset(rs1) always
|
||||
// links to X1.
|
||||
if len(in.Operands) == 1 {
|
||||
return false
|
||||
}
|
||||
if len(in.Operands) >= 2 && regFromOperand(in.Operands[1]) == 1 {
|
||||
return false
|
||||
}
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// riscvPrologue returns the prologue bytes for a RISC-V function, matching
|
||||
// the toolchain's compression: the SP adjustment compresses to C.ADDI when
|
||||
// the immediate fits, and the second LR save compresses to C.SDSP.
|
||||
func riscvPrologue(fi riscvFrameInfo) []byte {
|
||||
if fi.autosize == 0 {
|
||||
return nil
|
||||
}
|
||||
var out []byte
|
||||
// MOV LR, -autosize(SP) — SD X1, -autosize(X2). The negative offset is
|
||||
// not compressible to C.SDSP (unsigned), so it stays 4 bytes.
|
||||
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
|
||||
// ADDI $-autosize, SP, SP — open the frame (C.ADDI when it fits).
|
||||
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
|
||||
// MOV LR, 0(SP) — SD X1, 0(X2) → C.SDSP X1, 0.
|
||||
c := rvcSSP(0x7, 1, 0)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
return out
|
||||
}
|
||||
|
||||
// riscvReturn returns the bytes for a RET: the epilogue (restore LR and
|
||||
// deallocate the frame when present) followed by the uncompressed JALR X0,
|
||||
// 0(X1) the toolchain emits for RET (it never compresses RET to C.JR).
|
||||
func riscvReturn(fi riscvFrameInfo) []byte {
|
||||
var out []byte
|
||||
if fi.autosize != 0 {
|
||||
// MOV 0(SP), LR — LD X1, 0(X2) → C.LDSP X1, 0.
|
||||
c := rvcLSP(0x3, 1, 0)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
// ADDI $autosize, SP, SP — close the frame (C.ADDI when it fits).
|
||||
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
|
||||
}
|
||||
// JALR X0, 0(X1).
|
||||
return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...)
|
||||
}
|
||||
|
||||
// riscvSPAdjust emits an ADDI rd, imm, rd for the stack pointer (rd = rs1 =
|
||||
// X2), compressed to C.ADDI16SP when the immediate is a nonzero 16-byte
|
||||
// multiple, else C.ADDI when it fits 6-bit signed.
|
||||
func riscvSPAdjust(imm int32) []byte {
|
||||
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
|
||||
c := rvcADDI16SP(2, imm)
|
||||
return []byte{byte(c), byte(c >> 8)}
|
||||
}
|
||||
if riscvFitsCAddi(imm) {
|
||||
c := rvcCI(0x0, 2, uint32(imm)&0x3F)
|
||||
return []byte{byte(c), byte(c >> 8)}
|
||||
}
|
||||
return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 2, 2, imm))
|
||||
}
|
||||
|
||||
// riscvFitsCAddi reports whether imm compresses to C.ADDI (a nonzero 6-bit
|
||||
// signed immediate).
|
||||
func riscvFitsCAddi(imm int32) bool {
|
||||
return imm != 0 && imm >= -32 && imm <= 31
|
||||
}
|
||||
|
||||
// riscvPrologueSpadjPC returns the function-relative byte offset where the
|
||||
// prologue has finished decrementing SP (the delta becomes autosize).
|
||||
func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
|
||||
if fi.autosize == 0 {
|
||||
return 0
|
||||
}
|
||||
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
||||
return 4 + riscvSPAdjustLen(int32(-fi.autosize))
|
||||
}
|
||||
|
||||
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
||||
// (but not including) the final JALR — the point where SP is restored.
|
||||
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
|
||||
if fi.autosize == 0 {
|
||||
return 0
|
||||
}
|
||||
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
||||
return 2 + riscvSPAdjustLen(int32(fi.autosize))
|
||||
}
|
||||
|
||||
func riscvSPAdjustLen(imm int32) int {
|
||||
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
|
||||
return 2
|
||||
}
|
||||
if riscvFitsCAddi(imm) {
|
||||
return 2
|
||||
}
|
||||
return 4
|
||||
}
|
||||
|
||||
// riscvResolvePseudo translates a pseudo-register memory reference into a
|
||||
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
|
||||
// x+N(SP) → (N + autosize)(SP). Returns base = -1 for an unresolvable
|
||||
// reference (SB: static data, handled by the relocation path).
|
||||
func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32) {
|
||||
if sym == nil {
|
||||
return -1, 0
|
||||
}
|
||||
switch sym.Pseudo {
|
||||
case "FP":
|
||||
return 2, int32(sym.Offset) + int32(fi.autosize) + 8
|
||||
case "SP":
|
||||
return 2, int32(fi.autosize) + int32(sym.Offset)
|
||||
case "SB":
|
||||
return -1, int32(sym.Offset)
|
||||
}
|
||||
return -1, 0
|
||||
}
|
||||
@@ -0,0 +1,81 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestRISCVFrameSpadjAndLines checks that a framed function records its
|
||||
// stack-adjustment boundaries and source-line table, the inputs the GOOBJ
|
||||
// emitter turns into the pcsp/pcfile/pcline tables.
|
||||
func TestRISCVFrameSpadjAndLines(t *testing.T) {
|
||||
f, errs := parser.Parse("frame_riscv64.s", `#include "textflag.h"
|
||||
|
||||
TEXT ·framed(SB), NOSPLIT, $16-16
|
||||
MOV a+0(FP), X10
|
||||
MOV b+8(FP), X11
|
||||
ADD X11, X10, X10
|
||||
MOV X10, ret+16(FP)
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
if fn.Size != 24 {
|
||||
t.Fatalf("size = %d, want 24", fn.Size)
|
||||
}
|
||||
|
||||
// autosize = 16 + 8 = 24; the prologue boundary is just past its C.ADDI
|
||||
// (SD 4 + C.ADDI 2 = 6), and the RET restores SP just past its C.ADDI
|
||||
// (RET starts at 16; C.LDSP 2 + C.ADDI 2 = 20).
|
||||
wantSpadj := []SpadjStep{{PC: 6, Value: 24}, {PC: 20, Value: 0}}
|
||||
if len(fn.Spadj) != len(wantSpadj) {
|
||||
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
|
||||
}
|
||||
for i := range wantSpadj {
|
||||
if fn.Spadj[i] != wantSpadj[i] {
|
||||
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
|
||||
}
|
||||
}
|
||||
|
||||
// One line entry per instruction, in emission order.
|
||||
wantLines := []LineEntry{
|
||||
{Offset: 8, Line: 4},
|
||||
{Offset: 10, Line: 5},
|
||||
{Offset: 12, Line: 6},
|
||||
{Offset: 14, Line: 7},
|
||||
{Offset: 16, Line: 8},
|
||||
}
|
||||
if len(fn.Lines) != len(wantLines) {
|
||||
t.Fatalf("lines = %v, want %v", fn.Lines, wantLines)
|
||||
}
|
||||
for i := range wantLines {
|
||||
if fn.Lines[i] != wantLines[i] {
|
||||
t.Errorf("lines[%d] = %v, want %v", i, fn.Lines[i], wantLines[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestRISCVRegAliases checks the Go ABI register aliases that the toolchain
|
||||
// defines: LR is the link register (X1) and TMP is the assembler scratch
|
||||
// register (X31/T6).
|
||||
func TestRISCVRegAliases(t *testing.T) {
|
||||
for name, want := range map[string]int{
|
||||
"X1": 1, "RA": 1, "LR": 1,
|
||||
"X31": 31, "T6": 31, "TMP": 31,
|
||||
"X2": 2, "SP": 2,
|
||||
} {
|
||||
if got := riscvRegNum(name); got != want {
|
||||
t.Errorf("riscvRegNum(%q) = %d, want %d", name, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,346 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"debug/elf"
|
||||
"encoding/binary"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
|
||||
// instruction carrying an R_RISCV_JAL relocation (4-byte field) in both the
|
||||
// GOOBJ and ELF object emitters.
|
||||
func TestGOObjectRISCVCallReloc(t *testing.T) {
|
||||
f, errs := parser.Parse("k_riscv64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·c(SB), NOSPLIT, $0-0
|
||||
CALL callee<>(SB)
|
||||
RET
|
||||
|
||||
GLOBL callee<>(SB), RODATA, $8
|
||||
DATA callee<>+0(SB)/8, $42
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
if len(fn.Relocs) != 1 {
|
||||
t.Fatalf("relocs = %d, want 1", len(fn.Relocs))
|
||||
}
|
||||
r := fn.Relocs[0]
|
||||
if r.Kind != RelRISCVJal || r.Off != 8 || r.After != 12 || r.Name != "callee" || r.Addend != 0 || r.External {
|
||||
t.Errorf("reloc = {kind %v off %d after %d name %q addend %d external %v}", r.Kind, r.Off, r.After, r.Name, r.Addend, r.External)
|
||||
}
|
||||
|
||||
obj, err := img.GOObjectRISCV("testpkg", "k_riscv64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObjectRISCV: %v", err)
|
||||
}
|
||||
v := openGoobj(t, obj)
|
||||
relocIdx := v.blk(blkRelocIdx)
|
||||
relocs := v.blk(blkReloc)
|
||||
// The function is the last non-package symbol: 4 package defs, then the
|
||||
// 4 pc tables and the function.
|
||||
first := int(binary.LittleEndian.Uint32(relocIdx[(4+4)*4:]))
|
||||
if (first+1)*23 > len(relocs) {
|
||||
t.Fatalf("reloc block too short: first=%d len=%d", first, len(relocs))
|
||||
}
|
||||
e := relocs[first*23:]
|
||||
le := binary.LittleEndian
|
||||
if int32(le.Uint32(e[0:])) != 8 || e[4] != 4 || le.Uint16(e[5:]) != relocRISCVJal || le.Uint32(e[15:]) != pkgIdxSelf || le.Uint32(e[19:]) != 0 {
|
||||
t.Errorf("GOOBJ reloc = off %d size %d type %d pkg %d sym %d", int32(le.Uint32(e[0:])), e[4], le.Uint16(e[5:]), le.Uint32(e[15:]), le.Uint32(e[19:]))
|
||||
}
|
||||
|
||||
// The ELF object must carry a single R_RISCV_JAL relocation in .rela.text.
|
||||
elfObj, err := img.ELFRISCVObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFRISCVObject: %v", err)
|
||||
}
|
||||
if !hasELFRISCVJAL(t, elfObj) {
|
||||
t.Error("ELF object missing R_RISCV_JAL relocation")
|
||||
}
|
||||
}
|
||||
func TestGOObjectRISCVStructure(t *testing.T) {
|
||||
f, errs := parser.Parse("k_riscv64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·sb(SB), NOSPLIT, $0-0
|
||||
MOV $answer<>(SB), X10
|
||||
MOV answer<>(SB), X11
|
||||
MOV X12, answer<>(SB)
|
||||
RET
|
||||
|
||||
GLOBL answer<>(SB), RODATA, $8
|
||||
DATA answer<>+0(SB)/8, $42
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
if fn.Size != 28 {
|
||||
t.Fatalf("function size = %d, want 28", fn.Size)
|
||||
}
|
||||
if len(fn.Relocs) != 3 {
|
||||
t.Fatalf("relocs = %d, want 3", len(fn.Relocs))
|
||||
}
|
||||
wantKind := []RelocKind{RelRISCVPCRELIType, RelRISCVPCRELIType, RelRISCVPCRELSType}
|
||||
wantOff := []int{0, 8, 16}
|
||||
for i, r := range fn.Relocs {
|
||||
if r.Kind != wantKind[i] || r.Off != wantOff[i] || r.After != r.Off+8 || r.Name != "answer" || r.Addend != 0 {
|
||||
t.Errorf("reloc %d = {kind %v off %d after %d name %q addend %d}", i, r.Kind, r.Off, r.After, r.Name, r.Addend)
|
||||
}
|
||||
}
|
||||
|
||||
obj, err := img.GOObjectRISCV("testpkg", "k_riscv64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObjectRISCV: %v", err)
|
||||
}
|
||||
v := openGoobj(t, obj)
|
||||
|
||||
// Package defs: the static GLOBL, the FuncInfo, then the two DWARF
|
||||
// symbols.
|
||||
defs := v.syms(blkSymdef)
|
||||
if len(defs) != 4 {
|
||||
t.Fatalf("symdefs = %d, want 4", len(defs))
|
||||
}
|
||||
if defs[0].name != "answer" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 8 {
|
||||
t.Errorf("answer symbol = %+v", defs[0])
|
||||
}
|
||||
if defs[2].typ != kindSDWARFLINES || defs[3].typ != kindSDWARFFCN {
|
||||
t.Errorf("dwarf symbols = %+v, %+v", defs[2], defs[3])
|
||||
}
|
||||
|
||||
// The three code relocations, in definition order: ITYPE, ITYPE, STYPE,
|
||||
// each 8 bytes wide against the GLOBL (package symbol 0).
|
||||
relocIdx := v.blk(blkRelocIdx)
|
||||
relocs := v.blk(blkReloc)
|
||||
if len(relocs) != 5*23 {
|
||||
t.Fatalf("relocs = %d bytes, want 5 entries", len(relocs))
|
||||
}
|
||||
// The function is the last non-package symbol; its relocs start after
|
||||
// the DWARF symbols' (defs 2 and 3 each carry one).
|
||||
le := binary.LittleEndian
|
||||
first := int(le.Uint32(relocIdx[4*(4+4):]))
|
||||
wantType := []uint16{relocRISCVPcrelItype, relocRISCVPcrelItype, relocRISCVPcrelStype}
|
||||
wantOffAbs := []int{0, 8, 16}
|
||||
for i := range 3 {
|
||||
e := relocs[(first+i)*23:]
|
||||
if int32(le.Uint32(e[0:])) != int32(wantOffAbs[i]) || e[4] != 8 || le.Uint16(e[5:]) != wantType[i] ||
|
||||
le.Uint32(e[15:]) != pkgIdxSelf || le.Uint32(e[19:]) != 0 {
|
||||
t.Errorf("reloc %d = off %d size %d type %d pkg %d sym %d", i, int32(le.Uint32(e[0:])), e[4], le.Uint16(e[5:]), le.Uint32(e[15:]), le.Uint32(e[19:]))
|
||||
}
|
||||
}
|
||||
|
||||
// The function code: three AUIPC+second-instruction pairs with zero
|
||||
// immediates, then the uncompressed JALR X0, 0(X1) the toolchain emits
|
||||
// for RET.
|
||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
want := append(wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 10, 0)), wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 10, 10, 0))...)
|
||||
want = append(want, wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 11, 0))...)
|
||||
want = append(want, wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 11, 11, 0))...)
|
||||
want = append(want, wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 31, 0))...)
|
||||
want = append(want, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 12, 0))...)
|
||||
want = append(want, 0x67, 0x80, 0x00, 0x00) // JALR X0, 0(X1)
|
||||
if !bytes.Equal(code, want) {
|
||||
t.Errorf("code = % x\nwant % x", code, want)
|
||||
}
|
||||
|
||||
// The same bytes must survive into the object's data block intact: the
|
||||
// linker patches only the immediate fields of the AUIPC pairs, so the
|
||||
// opcode/register bits of every instruction must not be zeroed.
|
||||
dataIdx := v.blk(blkDataIdx)
|
||||
dataBlk := v.blk(blkData)
|
||||
dOff := int(le.Uint32(dataIdx[8*4:])) // the function is the last symbol
|
||||
emitted := dataBlk[dOff : dOff+fn.Size]
|
||||
if !bytes.Equal(emitted, want) {
|
||||
t.Errorf("emitted data = % x\nwant % x", emitted, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectRISCVLink cross-compiles a Go program with the gasm-produced
|
||||
// object substituted into the package archive, proving cmd/link accepts the
|
||||
// emitted RISC-V GOOBJ. The binary is not executed (no riscv64 host or
|
||||
// qemu). Skipped when no Go toolchain is available.
|
||||
func TestGOObjectRISCVLink(t *testing.T) {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
t.Skip("no Go toolchain available")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
asmSrc := `#include "textflag.h"
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOV a+0(FP), X10
|
||||
MOV b+8(FP), X11
|
||||
ADD X11, X10, X10
|
||||
MOV X10, ret+16(FP)
|
||||
RET
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main_riscv64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
mainSrc := `package main
|
||||
|
||||
func add(a, b int64) int64
|
||||
|
||||
func main() {
|
||||
if add(20, 22) != 42 {
|
||||
panic("bad add")
|
||||
}
|
||||
}
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module rvlink\n\ngo 1.21\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||
build.Dir = dir
|
||||
build.Env = append(os.Environ(), "GOARCH=riscv64")
|
||||
buildLog, err := build.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||
}
|
||||
var pkgArch, work, linkLine, asmObj string
|
||||
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_riscv64.s") && !strings.Contains(line, "-gensymabis"):
|
||||
asmObj = fieldAfter(line, "-o")
|
||||
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
|
||||
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
|
||||
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
|
||||
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||
linkLine = line
|
||||
}
|
||||
}
|
||||
if pkgArch == "" || linkLine == "" || asmObj == "" {
|
||||
t.Skip("could not locate the archive, asm output or link line in the build log")
|
||||
}
|
||||
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
|
||||
asmMember := filepath.Base(strings.ReplaceAll(asmObj, "$WORK", work))
|
||||
|
||||
pf, perrs := parser.Parse(filepath.Join(dir, "main_riscv64.s"), asmSrc)
|
||||
if len(perrs) > 0 {
|
||||
t.Fatalf("parse: %v", perrs)
|
||||
}
|
||||
pimg, err := AssembleFileRISCV(pf)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
obj, err := pimg.GOObjectRISCV("main", filepath.Join(dir, "main_riscv64.s"))
|
||||
if err != nil {
|
||||
t.Fatalf("GOObjectRISCV: %v", err)
|
||||
}
|
||||
|
||||
membersDir := filepath.Join(dir, "members")
|
||||
if err := os.MkdirAll(membersDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
|
||||
extract.Dir = membersDir
|
||||
extract.Env = append(os.Environ(), "GOARCH=riscv64")
|
||||
if out, err := extract.CombinedOutput(); err != nil {
|
||||
t.Fatalf("pack x: %v\n%s", err, out)
|
||||
}
|
||||
member := filepath.Join(membersDir, asmMember)
|
||||
if err := os.Chmod(member, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(member, obj, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
|
||||
listCmd.Env = append(os.Environ(), "GOARCH=riscv64")
|
||||
listOut, err := listCmd.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("pack t: %v\n%s", err, listOut)
|
||||
}
|
||||
newArch := filepath.Join(dir, "pkg.a")
|
||||
args := []string{"tool", "pack", "c", newArch}
|
||||
seen := map[string]bool{}
|
||||
for m := range strings.FieldsSeq(string(listOut)) {
|
||||
if seen[m] {
|
||||
continue
|
||||
}
|
||||
seen[m] = true
|
||||
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
args = append(args, filepath.Join(membersDir, m))
|
||||
}
|
||||
pack := exec.Command(goBin, args...)
|
||||
pack.Dir = membersDir
|
||||
pack.Env = append(os.Environ(), "GOARCH=riscv64")
|
||||
if out, err := pack.CombinedOutput(); err != nil {
|
||||
t.Fatalf("pack c: %v\n%s", err, out)
|
||||
}
|
||||
|
||||
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
|
||||
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
|
||||
link := exec.Command("sh", "-c", linkLine)
|
||||
link.Dir = dir
|
||||
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
|
||||
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)), "GOARCH=riscv64")
|
||||
if out, err := link.CombinedOutput(); err != nil {
|
||||
t.Fatalf("link with gasm object: %v\n%s", err, out)
|
||||
}
|
||||
|
||||
nm := exec.Command(goBin, "tool", "nm", filepath.Join(dir, "app2"))
|
||||
nm.Env = append(os.Environ(), "GOARCH=riscv64")
|
||||
nmOut, err := nm.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("nm gasm-linked binary: %v\n%s", err, nmOut)
|
||||
}
|
||||
if !strings.Contains(string(nmOut), "main.add") {
|
||||
t.Errorf("main.add not found in linked binary:\n%s", nmOut)
|
||||
}
|
||||
}
|
||||
|
||||
// hasELFRISCVJAL reports whether the ELF object carries an R_RISCV_JAL
|
||||
// relocation in its .rela.text section.
|
||||
func hasELFRISCVJAL(t *testing.T, data []byte) bool {
|
||||
t.Helper()
|
||||
f, err := elf.NewFile(bytes.NewReader(data))
|
||||
if err != nil {
|
||||
t.Fatalf("parse ELF: %v", err)
|
||||
}
|
||||
defer f.Close()
|
||||
rela := f.Section(".rela.text")
|
||||
if rela == nil {
|
||||
return false
|
||||
}
|
||||
b, err := rela.Data()
|
||||
if err != nil {
|
||||
t.Fatalf(".rela.text data: %v", err)
|
||||
}
|
||||
const rRISCVJAL = 17
|
||||
for i := 0; i+24 <= len(b); i += 24 {
|
||||
info := binary.LittleEndian.Uint64(b[i+8:])
|
||||
if uint32(info) == rRISCVJAL {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
+182
-4
@@ -39,6 +39,17 @@ const (
|
||||
// source lives in the reg field, the destination in r/m — the PEXTR-style
|
||||
// layout. VEXTRACTI128 and VEXTRACTF128 use this shape.
|
||||
vexExtract
|
||||
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
||||
// in ModRM.reg and the destination in r/m — the layout of the EVEX
|
||||
// narrowing stores (VPMOVDW, VPMOVQD).
|
||||
vexRMRev
|
||||
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
||||
// vector length follows the source: the packed-double → dword
|
||||
// conversions (VCVTPD2DQ/VCVTTPD2DQ and their X/Y spellings) narrow into
|
||||
// an XMM destination, so the L bit rides with the wider source. The
|
||||
// mnemonic's spelling fixes the length (X = 128, Y = 256), which also
|
||||
// covers a memory source. ModRM.reg = dst, ModRM.rm = src, no vvvv.
|
||||
vexRMSrcLen
|
||||
// vexZero is the no-operand form (VZEROUPPER).
|
||||
vexZero
|
||||
)
|
||||
@@ -80,14 +91,38 @@ var vexTable = map[string]vexSpec{
|
||||
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
|
||||
|
||||
// VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic.
|
||||
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
|
||||
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
|
||||
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
|
||||
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
|
||||
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
|
||||
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
|
||||
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
|
||||
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
|
||||
// VEX.128/256.0F.WIG — packed single-precision arithmetic.
|
||||
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
|
||||
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
|
||||
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
|
||||
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3},
|
||||
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3},
|
||||
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3},
|
||||
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
|
||||
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
|
||||
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
|
||||
// VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed
|
||||
// opcodes with an F2 pp).
|
||||
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
|
||||
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
|
||||
"VMULSD": {1, 0x59, 0, 3, -1, vexNDS3},
|
||||
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
|
||||
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
|
||||
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
|
||||
// VEX.128.F3.0F.WIG — scalar single-precision arithmetic (the packed
|
||||
// opcodes with an F3 pp).
|
||||
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
|
||||
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
|
||||
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3},
|
||||
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
|
||||
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
|
||||
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
||||
// VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form).
|
||||
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
||||
|
||||
@@ -95,12 +130,35 @@ var vexTable = map[string]vexSpec{
|
||||
// no vvvv).
|
||||
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
||||
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
||||
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM},
|
||||
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM},
|
||||
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM},
|
||||
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
||||
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
|
||||
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM},
|
||||
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM},
|
||||
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM},
|
||||
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
|
||||
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
||||
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
||||
"VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM},
|
||||
"VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM},
|
||||
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
|
||||
// (reg=dst, rm=src, no vvvv; the length follows the destination).
|
||||
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
|
||||
// VEX.128/256.0F.WIG — signed dword to packed single conversion
|
||||
// (reg=dst, rm=src, no vvvv, no mandatory prefix).
|
||||
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
|
||||
// VEX.128/256.0F.WIG — packed single to packed double conversion
|
||||
// (reg=dst, rm=src; the destination is the wide operand and sets the
|
||||
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
|
||||
// the Go assembler emits the instruction with pp = 00, and gasm follows
|
||||
// the Go assembler's bytes — its machine code is the oracle, not the
|
||||
// manual.
|
||||
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
|
||||
// VEX.128.F2.0F.WIG — duplicate the low double of each 128-bit lane
|
||||
// (reg=dst, rm=src, no vvvv; the length follows the destination).
|
||||
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
|
||||
// VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src).
|
||||
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
|
||||
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
|
||||
@@ -128,9 +186,75 @@ var vexTable = map[string]vexSpec{
|
||||
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
|
||||
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
|
||||
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
|
||||
// VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst:
|
||||
// reg=src, rm=XMM/memory dst, imm8 — the extract layout).
|
||||
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
|
||||
|
||||
// VEX.128.0F.W0 — no operands.
|
||||
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
||||
|
||||
// VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
||||
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
||||
|
||||
// VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst,
|
||||
// rm=scalar memory; SD is 256-bit only).
|
||||
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
||||
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
||||
// VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width
|
||||
// source).
|
||||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
||||
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
|
||||
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
||||
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
||||
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y
|
||||
// spellings: the destination is always XMM and the spelling fixes the
|
||||
// source length (X = 128, Y = 256).
|
||||
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||
|
||||
// VEX scalar conversions between vector and general-purpose registers.
|
||||
// Vector to GPR (two operands: vec/mem source, GPR destination, vvvv
|
||||
// unused; the length follows the source).
|
||||
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM},
|
||||
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM},
|
||||
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM},
|
||||
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM},
|
||||
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM},
|
||||
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM},
|
||||
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM},
|
||||
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM},
|
||||
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
||||
// vector source in vvvv, vector destination in reg).
|
||||
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3},
|
||||
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3},
|
||||
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
|
||||
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
|
||||
|
||||
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
|
||||
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
||||
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
|
||||
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
|
||||
|
||||
// VEX.F2.0F — packed double to packed dword conversions, truncating and
|
||||
// non-truncating. The destination is always XMM; the X/Y spellings fix
|
||||
// the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen.
|
||||
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
||||
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
||||
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||
}
|
||||
|
||||
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
|
||||
// the packed-double → dword conversions) to its fixed vector length:
|
||||
// X = 128 (L = 0), Y = 256 (L = 1). The spelling fixes the length even for
|
||||
// a memory source, matching the Go assembler's ytab.
|
||||
var vexSrcLen = map[string]int{
|
||||
"VCVTPD2DQX": 0,
|
||||
"VCVTPD2DQY": 1,
|
||||
"VCVTTPD2DQX": 0,
|
||||
"VCVTTPD2DQY": 1,
|
||||
"VCVTPD2PSX": 0,
|
||||
"VCVTPD2PSY": 1,
|
||||
}
|
||||
|
||||
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
|
||||
@@ -175,6 +299,11 @@ var vexMoveTable = map[string]vexMoveSpec{
|
||||
// VEX.128.F2.0F.WIG — scalar double move, memory operands only (the
|
||||
// register form takes three operands and is not supported yet).
|
||||
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
|
||||
// VEX.128.F3.0F.WIG — scalar single move, memory operands only.
|
||||
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
|
||||
// VEX.128/256 — aligned packed moves.
|
||||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
|
||||
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
|
||||
}
|
||||
|
||||
// isVex reports whether the mnemonic is a VEX-encoded instruction we handle.
|
||||
@@ -188,6 +317,13 @@ func isVex(mnemUpper string) bool {
|
||||
|
||||
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
|
||||
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
||||
// Vector register indices 16–31 exist only in EVEX encodings; fail
|
||||
// loudly rather than silently truncating the index.
|
||||
for _, op := range ops {
|
||||
if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 {
|
||||
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
||||
}
|
||||
}
|
||||
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
||||
return e.encodeVexMove(mnemUpper, ms, ops)
|
||||
}
|
||||
@@ -216,6 +352,8 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
||||
return e.encodeVexNDS3Imm(spec, ops)
|
||||
case vexExtract:
|
||||
return e.encodeVexExtract(spec, ops)
|
||||
case vexRMSrcLen:
|
||||
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
||||
case vexZero:
|
||||
return e.encodeVexZero(mnemUpper, spec, ops)
|
||||
}
|
||||
@@ -281,6 +419,32 @@ func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
|
||||
return e.emitVexFields(spec, l, regField, rBit, 15, src)
|
||||
}
|
||||
|
||||
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
|
||||
// the destination always XMM and the VEX.L bit following the source — fixed
|
||||
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
|
||||
// the source is memory.
|
||||
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("VEX destination must be a vector register")
|
||||
}
|
||||
ll, ok := vexSrcLen[mnem]
|
||||
if !ok {
|
||||
return fmt.Errorf("no fixed vector length for %s", mnem)
|
||||
}
|
||||
regField := dstReg.idx & 7
|
||||
rBit := 0
|
||||
if dstReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
// An unused vvvv field must be stored as all ones (v̄vvv = 1111).
|
||||
return e.emitVexFields(spec, ll, regField, rBit, 15, src)
|
||||
}
|
||||
|
||||
// encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst.
|
||||
// The destination is carried in VEX.vvvv, the source in ModRM.rm, and the
|
||||
// shift kind in the ModRM.reg /digit.
|
||||
@@ -507,7 +671,8 @@ func vecReg(op Operand) (Reg, bool) {
|
||||
|
||||
// vecOrMem reports whether op is a vector register or a memory reference.
|
||||
func vecOrMem(op Operand) bool {
|
||||
if _, ok := op.(Mem); ok {
|
||||
switch op.(type) {
|
||||
case Mem, sbMem:
|
||||
return true
|
||||
}
|
||||
r, ok := op.(Reg)
|
||||
@@ -518,7 +683,7 @@ func vecOrMem(op Operand) bool {
|
||||
// acceptable: memory always is, a GPR only for VMOVD/VMOVQ.
|
||||
func validMoveOther(ms vexMoveSpec, op Operand) bool {
|
||||
switch o := op.(type) {
|
||||
case Mem:
|
||||
case Mem, sbMem:
|
||||
return true
|
||||
case Reg:
|
||||
return ms.gprOK && !o.isVec()
|
||||
@@ -530,9 +695,13 @@ func validMoveOther(ms vexMoveSpec, op Operand) bool {
|
||||
// the given precomputed fields. It is shared by every register/rm VEX form;
|
||||
// immediate bytes are appended by the caller.
|
||||
func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Operand) error {
|
||||
if l > 1 {
|
||||
return fmt.Errorf("ZMM operand requires an EVEX instruction")
|
||||
}
|
||||
var modrm, sib int
|
||||
var disp []byte
|
||||
var xBit, bBit int
|
||||
var sb *sbRef
|
||||
switch r := rm.(type) {
|
||||
case Reg:
|
||||
modrm = 0xC0 | regField<<3 | (r.idx & 7)
|
||||
@@ -546,6 +715,12 @@ func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Ope
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
case sbMem:
|
||||
// RIP-relative static-symbol reference; disp32 patched at link time.
|
||||
modrm = regField<<3 | 0x05
|
||||
sib = -1
|
||||
disp = le32(0)
|
||||
sb = &sbRef{name: r.name, addend: r.addend}
|
||||
default:
|
||||
return fmt.Errorf("invalid VEX r/m operand")
|
||||
}
|
||||
@@ -561,6 +736,9 @@ func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Ope
|
||||
if sib >= 0 {
|
||||
e.out = append(e.out, byte(sib))
|
||||
}
|
||||
if sb != nil {
|
||||
e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend})
|
||||
}
|
||||
e.out = append(e.out, disp...)
|
||||
return nil
|
||||
}
|
||||
|
||||
+113
-67
@@ -39,11 +39,14 @@ func TestVexNDS3(t *testing.T) {
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(% x): %v", mnem, code, err)
|
||||
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
||||
continue
|
||||
}
|
||||
if inst.Op.String() != mnem {
|
||||
t.Errorf("%s: decoded as %s (% x)", mnem, inst.Op.String(), code)
|
||||
// The decoder folds the Plan 9 L/Q GPR-width spellings (VCVTSI2SDL/
|
||||
// SDQ, SSL/SSQ) onto the base name; the W bit carries the width.
|
||||
got := inst.Op.String()
|
||||
if got != mnem && !(len(mnem) > len(got) && mnem[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s (% x)", mnem, got, code)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -156,82 +159,121 @@ func TestVexShiftImm(t *testing.T) {
|
||||
// as well as every new operand form.
|
||||
func TestVexGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
wantOp string // decoded mnemonic, when it differs from mnem (the X/Y spellings)
|
||||
}{
|
||||
// Three-operand NDS form.
|
||||
{"VPADDQ Y8,Y9,Y8", "VPADDQ", []Operand{vreg(t, "Y8"), vreg(t, "Y9"), vreg(t, "Y8")}, "c44135d4c0"},
|
||||
{"VPADDQ X9,X8,X8", "VPADDQ", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c44139d4c1"},
|
||||
{"VPXOR X7,X7,X7", "VPXOR", []Operand{vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7")}, "c5c1efff"},
|
||||
{"VPSHUFB Y1,Y2,Y3", "VPSHUFB", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d00d9"},
|
||||
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9"},
|
||||
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec"},
|
||||
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9"},
|
||||
{"VPADDQ Y8,Y9,Y8", "VPADDQ", []Operand{vreg(t, "Y8"), vreg(t, "Y9"), vreg(t, "Y8")}, "c44135d4c0", ""},
|
||||
{"VPADDQ X9,X8,X8", "VPADDQ", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c44139d4c1", ""},
|
||||
{"VPXOR X7,X7,X7", "VPXOR", []Operand{vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7")}, "c5c1efff", ""},
|
||||
{"VPSHUFB Y1,Y2,Y3", "VPSHUFB", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d00d9", ""},
|
||||
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""},
|
||||
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""},
|
||||
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""},
|
||||
// Floating point (packed and scalar) and FMA — same NDS form, the pp
|
||||
// bits and map select the operation.
|
||||
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1"},
|
||||
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9"},
|
||||
{"VMULPD Y12,Y12,Y12", "VMULPD", []Operand{vreg(t, "Y12"), vreg(t, "Y12"), vreg(t, "Y12")}, "c4411d59e4"},
|
||||
{"VXORPD Y8,Y8,Y8", "VXORPD", []Operand{vreg(t, "Y8"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d57c0"},
|
||||
{"VUNPCKHPD X8,X8,X9", "VUNPCKHPD", []Operand{vreg(t, "X8"), vreg(t, "X8"), vreg(t, "X9")}, "c4413915c8"},
|
||||
{"VADDSD X9,X8,X8", "VADDSD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c4413b58c1"},
|
||||
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8"},
|
||||
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6"},
|
||||
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807"},
|
||||
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""},
|
||||
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""},
|
||||
{"VMULPD Y12,Y12,Y12", "VMULPD", []Operand{vreg(t, "Y12"), vreg(t, "Y12"), vreg(t, "Y12")}, "c4411d59e4", ""},
|
||||
{"VXORPD Y8,Y8,Y8", "VXORPD", []Operand{vreg(t, "Y8"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d57c0", ""},
|
||||
{"VUNPCKHPD X8,X8,X9", "VUNPCKHPD", []Operand{vreg(t, "X8"), vreg(t, "X8"), vreg(t, "X9")}, "c4413915c8", ""},
|
||||
{"VADDSD X9,X8,X8", "VADDSD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c4413b58c1", ""},
|
||||
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
|
||||
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
|
||||
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
|
||||
// Two-operand reg/rm form (v̄vvv must be 1111).
|
||||
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0"},
|
||||
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306"},
|
||||
{"VPBROADCASTD X0,Y15", "VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, "c4627d58f8"},
|
||||
{"VCVTDQ2PD X12,Y12", "VCVTDQ2PD", []Operand{vreg(t, "X12"), vreg(t, "Y12")}, "c4417ee6e4"},
|
||||
{"VCVTDQ2PD (SI),Y4", "VCVTDQ2PD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y4")}, "c5fee626"},
|
||||
{"VPMOVMSKB X11,AX", "VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, "c4c179d7c3"},
|
||||
{"VMOVMSKPS Y7,AX", "VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, "c5fc50c7"},
|
||||
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
|
||||
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
|
||||
{"VPBROADCASTD X0,Y15", "VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, "c4627d58f8", ""},
|
||||
{"VCVTDQ2PD X12,Y12", "VCVTDQ2PD", []Operand{vreg(t, "X12"), vreg(t, "Y12")}, "c4417ee6e4", ""},
|
||||
{"VCVTDQ2PD (SI),Y4", "VCVTDQ2PD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y4")}, "c5fee626", ""},
|
||||
{"VPMOVMSKB X11,AX", "VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, "c4c179d7c3", ""},
|
||||
{"VMOVMSKPS Y7,AX", "VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, "c5fc50c7", ""},
|
||||
// Immediate shifts.
|
||||
{"VPSLLD $1,Y3,Y4", "VPSLLD", []Operand{Imm(1), vreg(t, "Y3"), vreg(t, "Y4")}, "c5dd72f301"},
|
||||
{"VPSRLQ $2,Y5,Y6", "VPSRLQ", []Operand{Imm(2), vreg(t, "Y5"), vreg(t, "Y6")}, "c5cd73d502"},
|
||||
{"VPSLLD $1,Y3,Y4", "VPSLLD", []Operand{Imm(1), vreg(t, "Y3"), vreg(t, "Y4")}, "c5dd72f301", ""},
|
||||
{"VPSRLQ $2,Y5,Y6", "VPSRLQ", []Operand{Imm(2), vreg(t, "Y5"), vreg(t, "Y6")}, "c5cd73d502", ""},
|
||||
// Variable-count shifts: the count lives in an XMM register or memory
|
||||
// and the instruction takes the NDS form.
|
||||
{"VPSRLQ X0,Y8,Y8", "VPSRLQ", []Operand{vreg(t, "X0"), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd3c0"},
|
||||
{"VPSRLQ (AX),Y8,Y8", "VPSRLQ", []Operand{Ptr(AX, 0, 16), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd300"},
|
||||
{"VPSLLD X0,Y1,Y2", "VPSLLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f2d0"},
|
||||
{"VPSRLD X0,Y1,Y2", "VPSRLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5d2d0"},
|
||||
{"VPSRAD X0,Y1,Y2", "VPSRAD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5e2d0"},
|
||||
{"VPSLLQ X0,Y1,Y2", "VPSLLQ", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f3d0"},
|
||||
{"VPSRLQ X0,Y8,Y8", "VPSRLQ", []Operand{vreg(t, "X0"), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd3c0", ""},
|
||||
{"VPSRLQ (AX),Y8,Y8", "VPSRLQ", []Operand{Ptr(AX, 0, 16), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd300", ""},
|
||||
{"VPSLLD X0,Y1,Y2", "VPSLLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f2d0", ""},
|
||||
{"VPSRLD X0,Y1,Y2", "VPSRLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5d2d0", ""},
|
||||
{"VPSRAD X0,Y1,Y2", "VPSRAD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5e2d0", ""},
|
||||
{"VPSLLQ X0,Y1,Y2", "VPSLLQ", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f3d0", ""},
|
||||
// Immediate shuffle (reg=dst, rm=src, imm8).
|
||||
{"VPSHUFD $0xEE,X8,X9", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")}, "c4417970c8ee"},
|
||||
{"VPSHUFD $0xEE,Y1,Y2", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "Y1"), vreg(t, "Y2")}, "c5fd70d1ee"},
|
||||
{"VPERMQ $0x1B,Y1,Y2", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e3fd00d11b"},
|
||||
{"VPERMQ $0x1B,Y11,Y12", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y11"), vreg(t, "Y12")}, "c443fd00e31b"},
|
||||
{"VPSHUFD $0xEE,X8,X9", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")}, "c4417970c8ee", ""},
|
||||
{"VPSHUFD $0xEE,Y1,Y2", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "Y1"), vreg(t, "Y2")}, "c5fd70d1ee", ""},
|
||||
{"VPERMQ $0x1B,Y1,Y2", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e3fd00d11b", ""},
|
||||
{"VPERMQ $0x1B,Y11,Y12", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y11"), vreg(t, "Y12")}, "c443fd00e31b", ""},
|
||||
// Three-operand + immediate (reg=dst, vvvv=src1, rm=src2, imm8).
|
||||
{"VSHUFPD $1,X1,X2,X3", "VSHUFPD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e9c6d901"},
|
||||
{"VSHUFPD $1,Y1,Y2,Y3", "VSHUFPD", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5edc6d901"},
|
||||
{"VPERM2I128 $0x31,Y1,Y2,Y3", "VPERM2I128", []Operand{Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e36d46d931"},
|
||||
{"VINSERTI128 $1,X5,Y1,Y2", "VINSERTI128", []Operand{Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37538d501"},
|
||||
{"VSHUFPD $1,X1,X2,X3", "VSHUFPD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e9c6d901", ""},
|
||||
{"VSHUFPD $1,Y1,Y2,Y3", "VSHUFPD", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5edc6d901", ""},
|
||||
{"VPERM2I128 $0x31,Y1,Y2,Y3", "VPERM2I128", []Operand{Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e36d46d931", ""},
|
||||
{"VINSERTI128 $1,X5,Y1,Y2", "VINSERTI128", []Operand{Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37538d501", ""},
|
||||
// Lane extract (reg=YMM source, rm=XMM/memory destination, imm8).
|
||||
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101"},
|
||||
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701"},
|
||||
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101"},
|
||||
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""},
|
||||
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""},
|
||||
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""},
|
||||
// Moves — each direction picks its own opcode and VEX.W.
|
||||
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e"},
|
||||
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f"},
|
||||
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca"},
|
||||
{"VMOVUPD (DI),Y14", "VMOVUPD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y14")}, "c57d1037"},
|
||||
{"VMOVUPD Y14,(DI)", "VMOVUPD", []Operand{vreg(t, "Y14"), Ptr(DI, 0, 32)}, "c57d1137"},
|
||||
{"VMOVUPD X1,X2", "VMOVUPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f911ca"},
|
||||
{"VMOVQ X8,AX", "VMOVQ", []Operand{vreg(t, "X8"), AX}, "c461f97ec0"},
|
||||
{"VMOVQ AX,X9", "VMOVQ", []Operand{AX, vreg(t, "X9")}, "c461f96ec8"},
|
||||
{"VMOVQ X8,(DI)", "VMOVQ", []Operand{vreg(t, "X8"), Ptr(DI, 0, 8)}, "c461f97e07"},
|
||||
{"VMOVQ (SI),X9", "VMOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X9")}, "c461f96e0e"},
|
||||
{"VMOVQ X8,X2", "VMOVQ", []Operand{vreg(t, "X8"), vreg(t, "X2")}, "c579d6c2"},
|
||||
{"VMOVQ X2,X8", "VMOVQ", []Operand{vreg(t, "X2"), vreg(t, "X8")}, "c4c179d6d0"},
|
||||
{"VMOVD X0,(SI)", "VMOVD", []Operand{vreg(t, "X0"), Ptr(SI, 0, 4)}, "c5f97e06"},
|
||||
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0"},
|
||||
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006"},
|
||||
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106"},
|
||||
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""},
|
||||
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""},
|
||||
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""},
|
||||
{"VMOVUPD (DI),Y14", "VMOVUPD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y14")}, "c57d1037", ""},
|
||||
{"VMOVUPD Y14,(DI)", "VMOVUPD", []Operand{vreg(t, "Y14"), Ptr(DI, 0, 32)}, "c57d1137", ""},
|
||||
{"VMOVUPD X1,X2", "VMOVUPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f911ca", ""},
|
||||
{"VMOVQ X8,AX", "VMOVQ", []Operand{vreg(t, "X8"), AX}, "c461f97ec0", ""},
|
||||
{"VMOVQ AX,X9", "VMOVQ", []Operand{AX, vreg(t, "X9")}, "c461f96ec8", ""},
|
||||
{"VMOVQ X8,(DI)", "VMOVQ", []Operand{vreg(t, "X8"), Ptr(DI, 0, 8)}, "c461f97e07", ""},
|
||||
{"VMOVQ (SI),X9", "VMOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X9")}, "c461f96e0e", ""},
|
||||
{"VMOVQ X8,X2", "VMOVQ", []Operand{vreg(t, "X8"), vreg(t, "X2")}, "c579d6c2", ""},
|
||||
{"VMOVQ X2,X8", "VMOVQ", []Operand{vreg(t, "X2"), vreg(t, "X8")}, "c4c179d6d0", ""},
|
||||
{"VMOVD X0,(SI)", "VMOVD", []Operand{vreg(t, "X0"), Ptr(SI, 0, 4)}, "c5f97e06", ""},
|
||||
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""},
|
||||
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""},
|
||||
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""},
|
||||
// Packed double arithmetic and unpack — the NDS form, the opcode
|
||||
// selects the operation.
|
||||
{"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""},
|
||||
{"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""},
|
||||
{"VMINPD Y1,Y2,Y3", "VMINPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5dd9", ""},
|
||||
{"VMAXPD X4,X5,X6", "VMAXPD", []Operand{vreg(t, "X4"), vreg(t, "X5"), vreg(t, "X6")}, "c5d15ff4", ""},
|
||||
{"VUNPCKLPD X1,X2,X3", "VUNPCKLPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e914d9", ""},
|
||||
{"VUNPCKLPD Y1,Y2,Y3", "VUNPCKLPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed14d9", ""},
|
||||
{"VSUBPD (AX),X1,X2", "VSUBPD", []Operand{Ptr(AX, 0, 16), vreg(t, "X1"), vreg(t, "X2")}, "c5f15c10", ""},
|
||||
// Scalar double and single arithmetic (F2 / F3 pp, 128-bit only).
|
||||
{"VSUBSD X1,X2,X3", "VSUBSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5eb5cd9", ""},
|
||||
{"VDIVSD X7,X1,X2", "VDIVSD", []Operand{vreg(t, "X7"), vreg(t, "X1"), vreg(t, "X2")}, "c5f35ed7", ""},
|
||||
{"VMINSD X1,X2,X3", "VMINSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5eb5dd9", ""},
|
||||
{"VMAXSD X3,X4,X5", "VMAXSD", []Operand{vreg(t, "X3"), vreg(t, "X4"), vreg(t, "X5")}, "c5db5feb", ""},
|
||||
{"VADDSS X1,X2,X3", "VADDSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea58d9", ""},
|
||||
{"VSUBSS X1,X2,X3", "VSUBSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5cd9", ""},
|
||||
{"VMULSS X9,X10,X11", "VMULSS", []Operand{vreg(t, "X9"), vreg(t, "X10"), vreg(t, "X11")}, "c4412a59d9", ""},
|
||||
{"VDIVSS X1,X2,X3", "VDIVSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5ed9", ""},
|
||||
{"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""},
|
||||
{"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""},
|
||||
{"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""},
|
||||
// VMOVDDUP — duplicate the low double (reg=dst, rm=src, F2 pp).
|
||||
{"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""},
|
||||
{"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""},
|
||||
{"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""},
|
||||
// Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3
|
||||
// prefix — see the table comment), DQ→PD.
|
||||
{"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""},
|
||||
{"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""},
|
||||
{"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""},
|
||||
{"VCVTPS2PD X1,Y2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c5fc5ad1", ""},
|
||||
// PD→DQ conversions: the X/Y spellings fix the source length and the
|
||||
// destination is always XMM; the decoder reports the base mnemonic.
|
||||
{"VCVTPD2DQX X1,X2", "VCVTPD2DQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fbe6d1", "VCVTPD2DQ"},
|
||||
{"VCVTPD2DQY Y1,X2", "VCVTPD2DQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5ffe6d1", "VCVTPD2DQ"},
|
||||
{"VCVTTPD2DQX X3,X4", "VCVTTPD2DQX", []Operand{vreg(t, "X3"), vreg(t, "X4")}, "c5f9e6e3", "VCVTTPD2DQ"},
|
||||
{"VCVTTPD2DQY Y5,X6", "VCVTTPD2DQY", []Operand{vreg(t, "Y5"), vreg(t, "X6")}, "c5fde6f5", "VCVTTPD2DQ"},
|
||||
{"VCVTPD2DQY (AX),X1", "VCVTPD2DQY", []Operand{Ptr(AX, 0, 32), vreg(t, "X1")}, "c5ffe608", "VCVTPD2DQ"},
|
||||
// No-operand.
|
||||
{"VZEROUPPER", "VZEROUPPER", nil, "c5f877"},
|
||||
{"VZEROUPPER", "VZEROUPPER", nil, "c5f877", ""},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
@@ -251,7 +293,11 @@ func TestVexGroundTruth(t *testing.T) {
|
||||
if inst.Len != len(code) {
|
||||
t.Errorf("%s: Decode consumed %d of %d bytes", c.name, inst.Len, len(code))
|
||||
}
|
||||
if inst.Op.String() != c.mnem {
|
||||
wantOp := c.wantOp
|
||||
if wantOp == "" {
|
||||
wantOp = c.mnem
|
||||
}
|
||||
if inst.Op.String() != wantOp {
|
||||
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
|
||||
}
|
||||
}
|
||||
|
||||
+1
-3
@@ -123,10 +123,8 @@ type Symbol struct {
|
||||
// OpKind classifies an operand syntactically.
|
||||
type OpKind int
|
||||
|
||||
// Operand kinds.
|
||||
const (
|
||||
OpInvalid OpKind = iota
|
||||
OpImmediate // $value
|
||||
OpImmediate = iota // $value
|
||||
OpAddr // register, memory reference, symbol or label
|
||||
)
|
||||
|
||||
|
||||
+2
-2
@@ -54,11 +54,11 @@ func TestStmtPositions(t *testing.T) {
|
||||
// TestInterfaces confirms the node types satisfy their interfaces, so callers
|
||||
// can range over Decls and Stmts.
|
||||
func TestInterfaces(t *testing.T) {
|
||||
var decls []Decl = []Decl{&Include{}, &Preproc{}, &Text{}, &Globl{}, &Data{}}
|
||||
var decls = []Decl{&Include{}, &Preproc{}, &Text{}, &Globl{}, &Data{}}
|
||||
if len(decls) != 5 {
|
||||
t.Fatal("decl interface set")
|
||||
}
|
||||
var stmts []Stmt = []Stmt{&Label{}, &Instr{}}
|
||||
var stmts = []Stmt{&Label{}, &Instr{}}
|
||||
if len(stmts) != 2 {
|
||||
t.Fatal("stmt interface set")
|
||||
}
|
||||
|
||||
@@ -0,0 +1,307 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"runtime"
|
||||
"slices"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// cmdAuditInstructions cross-checks a gasm encoder against the Go toolchain's
|
||||
// own assembler, probed black-box: every mnemonic in the gasm table is offered
|
||||
// to go tool asm in its bare form, and a mnemonic counts as known to Go when
|
||||
// the error is anything but "unrecognized instruction" (a wrong-shape error
|
||||
// still proves the mnemonic exists in Go's tables). The audit answers three
|
||||
// questions at a glance:
|
||||
//
|
||||
// - which mnemonics gasm can encode that go tool asm does not know
|
||||
// (superset encodings, usable only through the gasm goobj path);
|
||||
// - which mnemonics the architecture table knows but the encoder cannot
|
||||
// emit yet (the implementation backlog);
|
||||
// - which mnemonics go tool asm knows that gasm cannot encode (feature
|
||||
// gaps).
|
||||
//
|
||||
// The amd64 derived families (Jcc, CMOVcc, SETcc) exist on both sides by
|
||||
// construction and are excluded from the diff; the other architectures list
|
||||
// their conditional branches outright.
|
||||
func cmdAuditInstructions(args []string) error {
|
||||
fs := newCommand("audit-instructions", "gasm audit-instructions [amd64|arm64|riscv64|loong64]", `
|
||||
Compare the gasm encoder for the given architecture (default amd64) against
|
||||
go tool asm and print the diff: superset encodings (gasm-only, shippable via
|
||||
gasm asm --format goobj), known-but-unencodable names (the backlog) and go-
|
||||
only names (feature gaps). The Go side is probed black-box with a battery
|
||||
of bare mnemonics, so the audit tracks whatever toolchain `+"`go env GOROOT`"+`
|
||||
provides.
|
||||
`)
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return err
|
||||
}
|
||||
archName := "amd64"
|
||||
switch n := len(fs.Args()); {
|
||||
case n > 1:
|
||||
return fmt.Errorf("audit-instructions takes at most one architecture argument")
|
||||
case n == 1:
|
||||
archName = strings.ToLower(fs.Arg(0))
|
||||
}
|
||||
a, err := auditArch(archName)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
tab := arch.ForArch(a)
|
||||
var names []string
|
||||
seen := map[string]bool{}
|
||||
for _, in := range tab.Instructions() {
|
||||
name := strings.ToUpper(in.Name)
|
||||
if a == arch.AMD64 && derivedFamily(name) || seen[name] {
|
||||
continue
|
||||
}
|
||||
seen[name] = true
|
||||
names = append(names, name)
|
||||
}
|
||||
|
||||
goKnown, err := probeGoAsm(goarchName(a), names)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
var superset, backlog, shared []string
|
||||
for _, name := range names {
|
||||
switch {
|
||||
case !gasmEncodable(a, name):
|
||||
backlog = append(backlog, name)
|
||||
case !goKnown[name]:
|
||||
superset = append(superset, name)
|
||||
default:
|
||||
shared = append(shared, name)
|
||||
}
|
||||
}
|
||||
// GO-ONLY is not enumerable by probing: Go's table is only visible
|
||||
// through names we already know, so nothing can be reported there.
|
||||
|
||||
slices.Sort(superset)
|
||||
slices.Sort(backlog)
|
||||
slices.Sort(shared)
|
||||
|
||||
w := os.Stdout
|
||||
fmt.Fprintf(w, "gasm table (%s, families excluded): %d mnemonics\n", archName, len(names))
|
||||
fmt.Fprintf(w, "gasm encodable: %d go tool asm recognized: %d\n", len(shared)+len(superset), countTrue(goKnown))
|
||||
fmt.Fprintf(w, "shared: %d\n", len(shared))
|
||||
fmt.Fprintf(w, "\nSuperset encodings (gasm-only; ship via gasm asm --format goobj):\n")
|
||||
for _, n := range superset {
|
||||
fmt.Fprintf(w, " %s\n", n)
|
||||
}
|
||||
fmt.Fprintf(w, "\nKnown but not encodable (backlog):\n")
|
||||
for _, n := range backlog {
|
||||
fmt.Fprintf(w, " %s\n", n)
|
||||
}
|
||||
fmt.Fprintf(w, "\nGo-only names cannot be enumerated by probing; extend the gasm\n")
|
||||
fmt.Fprintf(w, "table from the Go release notes when a new instruction family ships.\n")
|
||||
return nil
|
||||
}
|
||||
|
||||
// auditArch resolves the audit's architecture argument.
|
||||
func auditArch(name string) (arch.Arch, error) {
|
||||
switch strings.ToLower(name) {
|
||||
case "amd64":
|
||||
return arch.AMD64, nil
|
||||
case "arm64":
|
||||
return arch.ARM64, nil
|
||||
case "riscv64", "riscv":
|
||||
return arch.RISCV, nil
|
||||
case "loong64", "loong":
|
||||
return arch.LOONG64, nil
|
||||
}
|
||||
return arch.Unknown, fmt.Errorf("unknown architecture %q: want amd64, arm64, riscv64 or loong64", name)
|
||||
}
|
||||
|
||||
// goarchName maps an arch identifier onto its GOARCH spelling.
|
||||
func goarchName(a arch.Arch) string {
|
||||
switch a {
|
||||
case arch.ARM64:
|
||||
return "arm64"
|
||||
case arch.RISCV:
|
||||
return "riscv64"
|
||||
case arch.LOONG64:
|
||||
return "loong64"
|
||||
}
|
||||
return "amd64"
|
||||
}
|
||||
|
||||
func countTrue(m map[string]bool) int {
|
||||
n := 0
|
||||
for _, v := range m {
|
||||
if v {
|
||||
n++
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// derivedFamily reports whether a mnemonic belongs to a family both
|
||||
// assemblers construct from condition codes rather than list exhaustively
|
||||
// (JEQ/CMOVLGT/SETNE and friends). Such names never probe cleanly, so
|
||||
// including them in the diff would be noise. amd64 only: the other
|
||||
// architectures list their conditional branches outright.
|
||||
func derivedFamily(name string) bool {
|
||||
if strings.HasPrefix(name, "J") && name != "JMP" && name != "JMPQ" {
|
||||
return true
|
||||
}
|
||||
if strings.HasPrefix(name, "CMOV") || strings.HasPrefix(name, "SET") {
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
var unrecognizedRe = regexp.MustCompile(`unrecognized instruction`)
|
||||
|
||||
// probeGoAsm feeds every mnemonic to go tool asm in one generated file and
|
||||
// classifies the diagnostics. "Unrecognized instruction" is a parse-stage
|
||||
// verdict on the mnemonic alone, so a single bare-instruction probe per
|
||||
// mnemonic decides recognition; the combined file still reports every line's
|
||||
// error even when others fail.
|
||||
func probeGoAsm(goarch string, names []string) (map[string]bool, error) {
|
||||
dir, err := os.MkdirTemp("", "gasm-audit")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
|
||||
var sb strings.Builder
|
||||
sb.WriteString("TEXT ·probe(SB), 4, $0\n\tRET\n")
|
||||
lineMnemonic := map[int]string{}
|
||||
line := 3
|
||||
for _, name := range names {
|
||||
fmt.Fprintf(&sb, "TEXT ·p%s%d(SB), 4, $0\n", sanitize(name), line)
|
||||
sb.WriteString("\t" + name + "\n\tRET\n")
|
||||
lineMnemonic[line+1] = name // the instruction line, after TEXT
|
||||
line += 3
|
||||
}
|
||||
probePath := filepath.Join(dir, "probe.s")
|
||||
if err := os.WriteFile(probePath, []byte(sb.String()), 0o644); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
toolDir, err := exec.Command("go", "env", "GOTOOLDIR").Output()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("go env GOTOOLDIR: %w", err)
|
||||
}
|
||||
asmBin := filepath.Join(strings.TrimSpace(string(toolDir)), "asm")
|
||||
if _, err := os.Stat(asmBin); err != nil {
|
||||
return nil, fmt.Errorf("go tool asm not found at %s", asmBin)
|
||||
}
|
||||
cmd := exec.Command(asmBin, "-p", "probe", "-o", filepath.Join(dir, "probe.o"), probePath)
|
||||
cmd.Env = append(os.Environ(), "GOARCH="+goarch, "GOOS="+runtime.GOOS)
|
||||
out, _ := cmd.CombinedOutput()
|
||||
|
||||
result := map[string]bool{}
|
||||
for _, name := range names {
|
||||
result[name] = true // no news = the name parsed fine
|
||||
}
|
||||
reParse := regexp.MustCompile(`probe\.s:(\d+):`)
|
||||
for l := range strings.SplitSeq(string(out), "\n") {
|
||||
m := reParse.FindStringSubmatch(l)
|
||||
if m == nil {
|
||||
continue
|
||||
}
|
||||
lineNo, err := strconv.Atoi(m[1])
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if name, ok := lineMnemonic[lineNo]; ok && unrecognizedRe.MatchString(l) {
|
||||
result[name] = false
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// probeShapes lists representative operand shapes for the encodability
|
||||
// probe. The assemblers report an unknown mnemonic and a known mnemonic
|
||||
// with no supported form alike ("unsupported <arch> instruction"), so only
|
||||
// a shape that assembles cleanly counts, and the backlog over-approximates:
|
||||
// a name whose real forms the battery misses lands there. amd64 keeps its
|
||||
// exact table-driven check.
|
||||
func probeShapes(a arch.Arch) []string {
|
||||
switch a {
|
||||
case arch.ARM64:
|
||||
return []string{
|
||||
"X0, X1, X2", "X0, X1", "X0", "$1, X0", "X0, (X1)", "(X0), X1",
|
||||
"X0, (X1, 8)", "(SP), X0", "F0, F1, F2", "F0, F1", "F0",
|
||||
"V0.B16, V1.B16, V2.B16", "p2", "X0, p2", "X0, X1, p2",
|
||||
// The conditional select family spells the condition first
|
||||
// and takes R register spellings.
|
||||
"EQ, R0, R1, R2", "EQ, R0, R1", "EQ, R0",
|
||||
"GE, F0, F1, F2", "NE, F0, F1, $0",
|
||||
}
|
||||
case arch.RISCV:
|
||||
return []string{
|
||||
"X5, X6, X7", "X5, X6", "X5", "$1, X5", "X5, (X6)", "$1, X5, X6",
|
||||
"(X5), X6", "F0, F1, F2", "F0, F1", "p2", "X1, p2", "X0, p2",
|
||||
"X5, X6, p2", "p2(SB)",
|
||||
}
|
||||
case arch.LOONG64:
|
||||
return []string{
|
||||
"R4, R5, R6", "R4, R5", "R4", "$1, R4", "R4, (R5)", "(R4), R5",
|
||||
"F0, F1, F2", "F0, F1", "p2", "R1, p2", "R4, p2",
|
||||
"$1, R4, R5, R6", "$65536, R4", "R4, R5, p2", "p2(SB)",
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// gasmEncodable reports whether the gasm encoder for a can emit the
|
||||
// mnemonic, decided by trial assembly over the shape battery.
|
||||
func gasmEncodable(a arch.Arch, name string) bool {
|
||||
switch a {
|
||||
case arch.ARM64, arch.RISCV, arch.LOONG64:
|
||||
default:
|
||||
return asm.Encodable(name)
|
||||
}
|
||||
for _, shape := range probeShapes(a) {
|
||||
if gasmAssembles(a, name, shape) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// gasmAssembles reports whether a one-instruction probe file containing name
|
||||
// with the given operand shape assembles without error.
|
||||
func gasmAssembles(a arch.Arch, name, shape string) bool {
|
||||
src := "TEXT ·p(SB), NOSPLIT, $0\n\t" + name
|
||||
if shape != "" {
|
||||
src += " " + shape
|
||||
}
|
||||
src += "\n\tRET\np2:\n\tRET\n"
|
||||
f, errs := parser.Parse("probe.s", src)
|
||||
if len(errs) > 0 {
|
||||
return false
|
||||
}
|
||||
var err error
|
||||
switch a {
|
||||
case arch.ARM64:
|
||||
_, err = asm.AssembleFileARM64(f)
|
||||
case arch.RISCV:
|
||||
_, err = asm.AssembleFileRISCV(f)
|
||||
case arch.LOONG64:
|
||||
_, err = asm.AssembleFileLOONG64(f)
|
||||
}
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// sanitize makes a mnemonic safe for use in a Go symbol name.
|
||||
func sanitize(name string) string {
|
||||
return strings.NewReplacer(".", "_", "$", "_").Replace(name)
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||
)
|
||||
|
||||
func TestDerivedFamily(t *testing.T) {
|
||||
for _, n := range []string{"JEQ", "JLT", "JCC", "CMOVLGT", "SETNE", "SETA"} {
|
||||
if !derivedFamily(n) {
|
||||
t.Errorf("derivedFamily(%q) = false, want true", n)
|
||||
}
|
||||
}
|
||||
for _, n := range []string{"JMP", "ADDQ", "VPGATHERDD", "MOVBE", "PSHUFB"} {
|
||||
if derivedFamily(n) {
|
||||
t.Errorf("derivedFamily(%q) = true, want false", n)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestSanitize(t *testing.T) {
|
||||
if got := sanitize("VPCMP.UB"); got != "VPCMP_UB" {
|
||||
t.Errorf("sanitize: got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAuditArch(t *testing.T) {
|
||||
for in, want := range map[string]arch.Arch{
|
||||
"amd64": arch.AMD64, "arm64": arch.ARM64,
|
||||
"riscv64": arch.RISCV, "riscv": arch.RISCV,
|
||||
"loong64": arch.LOONG64, "LOONG": arch.LOONG64,
|
||||
} {
|
||||
got, err := auditArch(in)
|
||||
if err != nil || got != want {
|
||||
t.Errorf("auditArch(%q) = %v, %v; want %v", in, got, err, want)
|
||||
}
|
||||
}
|
||||
if _, err := auditArch("mips"); err == nil {
|
||||
t.Error("auditArch(mips) must fail")
|
||||
}
|
||||
}
|
||||
|
||||
func TestGasmEncodable(t *testing.T) {
|
||||
cases := []struct {
|
||||
a arch.Arch
|
||||
yes string
|
||||
no string
|
||||
}{
|
||||
{arch.AMD64, "ADDQ", "NOSUCHMNEMONIC"},
|
||||
{arch.ARM64, "ADD", "NOSUCHMNEMONIC"},
|
||||
{arch.RISCV, "ADD", "NOSUCHMNEMONIC"},
|
||||
{arch.LOONG64, "ADDV", "NOSUCHMNEMONIC"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if !gasmEncodable(c.a, c.yes) {
|
||||
t.Errorf("%s: %s should be encodable", c.a, c.yes)
|
||||
}
|
||||
if gasmEncodable(c.a, c.no) {
|
||||
t.Errorf("%s: %s should not be encodable", c.a, c.no)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,331 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/debug"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
||||
)
|
||||
|
||||
func cmdDebug(args []string) int {
|
||||
fs := newCommand("debug", "gasm debug <file.s> --func <name>", `
|
||||
Interactive debugger for JIT-assembled functions. Launches the
|
||||
function in a traced subprocess (ptrace), then provides a REPL for
|
||||
single-stepping, breakpoints, register and memory inspection.
|
||||
|
||||
REPL commands:
|
||||
break <label|addr> [if <reg> <op> <val>]
|
||||
set a breakpoint, optionally conditional on a
|
||||
register comparison (reg-reg or reg-immediate)
|
||||
delete <label|addr> remove a breakpoint
|
||||
info break list all breakpoints
|
||||
step [n], s single-step n instructions (default 1)
|
||||
next, n step over a CALL
|
||||
finish, fin run until the function returns
|
||||
continue, c run until a breakpoint, watchpoint or exit
|
||||
disas [n], u disassemble n instructions at PC
|
||||
regs print general-purpose and vector registers
|
||||
where show source line and nearest label at PC
|
||||
stack show stack near RSP (return address + ABI0 args)
|
||||
bt, backtrace backtrace (current frame + return address)
|
||||
x [addr] [len] hex-dump memory (default: current PC, 64 bytes)
|
||||
w <addr> <val...> write bytes to memory
|
||||
set <reg> <value> set a register
|
||||
watch <addr> [r|w] [size]
|
||||
set a hardware watchpoint (write by default)
|
||||
unwatch [<slot>] clear one watchpoint, or all without an argument
|
||||
labels, l list function labels and offsets
|
||||
help, h, ? show command help
|
||||
quit, q kill the debuggee and exit
|
||||
`)
|
||||
funcName := fs.String("func", "", "function to debug")
|
||||
argsFile := fs.String("args", "", "file containing the ABI0 argument block")
|
||||
bufSpec := fs.String("buf", "", "buffer specification: name:size:pattern[,name:size:pattern...] where pattern is zero, ones, seq, or hex")
|
||||
script := fs.String("script", "", "run REPL commands from a file (one per line) and exit; '-' reads stdin")
|
||||
cover := fs.Bool("cover", false, "run to completion with a breakpoint on every instruction and report which executed and how often")
|
||||
timeout := fs.Duration("timeout", 0, "kill the debuggee after this duration (e.g. 30s); for headless --script runs")
|
||||
fs.Parse(args)
|
||||
|
||||
// --- Debuggee mode (internal, spawned by the debugger) ---
|
||||
if os.Getenv("GASM_DEBUG_TARGET") != "" {
|
||||
tmpDir := os.Getenv("GASM_DEBUG_TMP")
|
||||
if tmpDir == "" || fs.NArg() < 1 || *funcName == "" || *argsFile == "" {
|
||||
fmt.Fprintln(os.Stderr, "gasm debug: internal debuggee mode")
|
||||
return 2
|
||||
}
|
||||
if err := debug.RunTarget(fs.Arg(0), *funcName, *argsFile, tmpDir); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// --- Debugger mode (interactive REPL) ---
|
||||
if fs.NArg() < 1 || *funcName == "" {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm debug <file.s> --func <name>")
|
||||
return 2
|
||||
}
|
||||
path := fs.Arg(0)
|
||||
|
||||
// The watchdog is armed before anything can block: ptrace attach and a
|
||||
// continued kernel loop both hang the run when the environment forbids
|
||||
// tracing or the kernel loops forever, and neither is interruptible from
|
||||
// the inside.
|
||||
if *timeout > 0 {
|
||||
go func() {
|
||||
time.Sleep(*timeout)
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: timeout (%s) — killing the debuggee\n", *timeout)
|
||||
os.Exit(3)
|
||||
}()
|
||||
}
|
||||
|
||||
// Load the kernel to extract function metadata and labels.
|
||||
k, err := verify.Load(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer k.Close()
|
||||
|
||||
fl, err := k.Func(*funcName)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
// Build the label list for the REPL.
|
||||
var labels []debug.Label
|
||||
for name, off := range fl.Labels {
|
||||
labels = append(labels, debug.Label{Name: name, Offset: off})
|
||||
}
|
||||
sort.Slice(labels, func(i, j int) bool { return labels[i].Offset < labels[j].Offset })
|
||||
|
||||
// Launch the debuggee with the argument block.
|
||||
var argBlock []byte
|
||||
var bufAddrs []uint64
|
||||
var sess *debug.Session
|
||||
if *bufSpec != "" {
|
||||
// Parse the function signature to determine argument layout.
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
sig, ok := verify.ExtractFuncSig(src, *funcName)
|
||||
if !ok {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: no // func signature found for %s\n", *funcName)
|
||||
return 1
|
||||
}
|
||||
layout := verify.ArgLayout(sig)
|
||||
|
||||
// Parse the buffer spec to get buffer names.
|
||||
bufNames := parseBufNames(*bufSpec)
|
||||
|
||||
// Allocate buffers in the debuggee.
|
||||
argBlock = make([]byte, fl.Args)
|
||||
sess, bufAddrs, err = debug.LaunchWithBuffers("", path, *funcName, argBlock, *bufSpec)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
// Construct the argument block with buffer pointers at the correct positions.
|
||||
bufIdx := 0
|
||||
for _, arg := range layout {
|
||||
if !arg.IsPtr {
|
||||
continue
|
||||
}
|
||||
// Find the buffer that matches this argument.
|
||||
for i, name := range bufNames {
|
||||
if i < len(bufAddrs) && (name == arg.Name || strings.HasPrefix(arg.Name, name)) {
|
||||
addr := bufAddrs[i]
|
||||
off := arg.Offset
|
||||
if off+8 <= len(argBlock) {
|
||||
argBlock[off] = byte(addr)
|
||||
argBlock[off+1] = byte(addr >> 8)
|
||||
argBlock[off+2] = byte(addr >> 16)
|
||||
argBlock[off+3] = byte(addr >> 24)
|
||||
argBlock[off+4] = byte(addr >> 32)
|
||||
argBlock[off+5] = byte(addr >> 40)
|
||||
argBlock[off+6] = byte(addr >> 48)
|
||||
argBlock[off+7] = byte(addr >> 56)
|
||||
}
|
||||
// For slices, also set the length and capacity.
|
||||
if strings.HasPrefix(arg.Typ, "[]") && off+24 <= len(argBlock) {
|
||||
// Find the buffer size from the spec.
|
||||
size := parseBufSize(*bufSpec, name)
|
||||
// Length at offset+8, capacity at offset+16.
|
||||
for j := range 8 {
|
||||
argBlock[off+8+j] = byte(size >> (j * 8))
|
||||
argBlock[off+16+j] = byte(size >> (j * 8))
|
||||
}
|
||||
}
|
||||
bufIdx++
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
_ = bufIdx
|
||||
} else {
|
||||
argBlock = make([]byte, fl.Args)
|
||||
sess, err = debug.Launch("", path, *funcName, argBlock)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
}
|
||||
defer sess.Kill()
|
||||
|
||||
bm := debug.NewBreakpoints(sess)
|
||||
fmt.Printf("gasm debug: %s in %s (pid %d)\n", *funcName, path, sess.Pid())
|
||||
|
||||
// Convert the line table for the command loop.
|
||||
var srcLines []debug.SourceLine
|
||||
for _, le := range fl.Lines {
|
||||
srcLines = append(srcLines, debug.SourceLine{Offset: le.Offset, Line: le.Line})
|
||||
}
|
||||
|
||||
// Coverage mode: pre-register a breakpoint on every instruction (walked
|
||||
// by length through the function body while the debuggee is stopped) and
|
||||
// let the kernel run to completion. Each trap counts a hit for that
|
||||
// instruction, so the final report shows exactly which instructions
|
||||
// executed and how often, with the label-level view derived from it.
|
||||
// Expect the run to slow to ptrace speed: one trap per executed
|
||||
// instruction.
|
||||
if *cover {
|
||||
base := sess.CodeBase() + uint64(fl.Offset)
|
||||
type coverInstr struct {
|
||||
off uint64
|
||||
text string
|
||||
}
|
||||
var instrs []coverInstr
|
||||
for off := uint64(0); off < uint64(fl.Size); {
|
||||
text, ln, err := sess.Disassemble(base + off)
|
||||
if err != nil || ln == 0 {
|
||||
break
|
||||
}
|
||||
instrs = append(instrs, coverInstr{off: off, text: text})
|
||||
off += uint64(ln)
|
||||
}
|
||||
for _, in := range instrs {
|
||||
if _, err := bm.SetWithCond(base+in.off, fmt.Sprintf("func+%#x", in.off), nil); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: cover: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
}
|
||||
fmt.Printf("gasm debug: coverage run over %d instructions\n", len(instrs))
|
||||
for {
|
||||
for _, bp := range bm.All() {
|
||||
bm.Reinsert(bp.Addr)
|
||||
}
|
||||
if err := sess.Continue(); err != nil {
|
||||
break // debuggee finished or died
|
||||
}
|
||||
if sess.Exited() {
|
||||
break
|
||||
}
|
||||
regs, rerr := sess.GetRegs()
|
||||
if rerr != nil {
|
||||
break
|
||||
}
|
||||
// HandleTrap restores the original byte, rewinds PC and counts
|
||||
// the hit on the breakpoint itself. Single-step over the
|
||||
// restored instruction so the reinsertion at the top of the
|
||||
// loop cannot re-trap on the same breakpoint.
|
||||
if bp := bm.HandleTrap(®s); bp != nil {
|
||||
if err := sess.Step(); err != nil {
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
hits := map[uint64]int{}
|
||||
traps := 0
|
||||
for _, bp := range bm.All() {
|
||||
if n := bp.Hits(); n > 0 {
|
||||
hits[bp.Addr-base] = n
|
||||
traps += n
|
||||
}
|
||||
}
|
||||
var hit []string
|
||||
var missed []string
|
||||
for _, l := range labels {
|
||||
if hits[uint64(l.Offset)] > 0 {
|
||||
hit = append(hit, l.Name)
|
||||
} else {
|
||||
missed = append(missed, l.Name)
|
||||
}
|
||||
}
|
||||
sort.Strings(hit)
|
||||
sort.Strings(missed)
|
||||
fmt.Printf("coverage: %d/%d instructions executed (%d traps)\n", len(hits), len(instrs), traps)
|
||||
fmt.Printf("coverage: %d/%d labels reached\n", len(hit), len(labels))
|
||||
for _, l := range hit {
|
||||
fmt.Printf(" covered %s\n", l)
|
||||
}
|
||||
for _, l := range missed {
|
||||
fmt.Printf(" MISSED %s\n", l)
|
||||
}
|
||||
fmt.Println("executed instructions:")
|
||||
for _, in := range instrs {
|
||||
if n := hits[in.off]; n > 0 {
|
||||
fmt.Printf(" func+%#04x %4dx %s\n", in.off, n, in.text)
|
||||
}
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// Headless mode: run the script through the normal command loop and
|
||||
// exit. The watchdog armed above covers launch, continue and step.
|
||||
var in io.Reader = os.Stdin
|
||||
if *script != "" {
|
||||
if *script == "-" {
|
||||
in = os.Stdin
|
||||
} else {
|
||||
f, err := os.Open(*script)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer f.Close()
|
||||
in = f
|
||||
}
|
||||
}
|
||||
debug.REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, labels, srcLines, in)
|
||||
return 0
|
||||
}
|
||||
|
||||
// parseBufNames extracts buffer names from a buffer specification.
|
||||
// Format: name:size:pattern[,name:size:pattern...]
|
||||
func parseBufNames(spec string) []string {
|
||||
var names []string
|
||||
for part := range strings.SplitSeq(spec, ",") {
|
||||
fields := strings.SplitN(part, ":", 3)
|
||||
if len(fields) >= 1 && fields[0] != "" {
|
||||
names = append(names, fields[0])
|
||||
}
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
// parseBufSize extracts the size of a named buffer from a buffer specification.
|
||||
func parseBufSize(spec, name string) int {
|
||||
for part := range strings.SplitSeq(spec, ",") {
|
||||
fields := strings.SplitN(part, ":", 3)
|
||||
if len(fields) >= 2 && fields[0] == name {
|
||||
var size int
|
||||
fmt.Sscanf(fields[1], "%d", &size)
|
||||
return size
|
||||
}
|
||||
}
|
||||
return 0
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build !linux
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
)
|
||||
|
||||
func cmdDebug(args []string) int {
|
||||
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux (ptrace)")
|
||||
return 1
|
||||
}
|
||||
+1509
-59
File diff suppressed because it is too large
Load Diff
+125
-5
@@ -7,8 +7,11 @@ import (
|
||||
"bytes"
|
||||
"io"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strings"
|
||||
"syscall"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -52,6 +55,54 @@ func capture(fn func() int) (stdout, stderr string, code int) {
|
||||
return string(ob), string(eb), code
|
||||
}
|
||||
|
||||
// TestCmdFmtRecursive checks the go-fmt-style directory mode: with no
|
||||
// arguments every .s file below the working directory is formatted in place
|
||||
// ("." and "_" directories skipped), changed files are listed, and a second
|
||||
// run is a no-op.
|
||||
func TestCmdFmtRecursive(t *testing.T) {
|
||||
tmp := t.TempDir()
|
||||
t.Chdir(tmp)
|
||||
unformatted := []byte("TEXT ·f(SB),NOSPLIT,$0\nRET\n")
|
||||
write := func(path string) {
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(path, unformatted, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
write("a_amd64.s")
|
||||
write(filepath.Join("sub", "b_amd64.s"))
|
||||
write(filepath.Join("_refs", "c_amd64.s"))
|
||||
write(filepath.Join(".git", "d_amd64.s"))
|
||||
|
||||
out, errOut, code := capture(func() int { return cmdFmt(nil) })
|
||||
if code != 0 {
|
||||
t.Fatalf("code = %d (%s)", code, errOut)
|
||||
}
|
||||
if out != "a_amd64.s\n"+filepath.Join("sub", "b_amd64.s")+"\n" {
|
||||
t.Errorf("listed files unexpected:\n%s", out)
|
||||
}
|
||||
for _, p := range []string{"a_amd64.s", filepath.Join("sub", "b_amd64.s")} {
|
||||
b, _ := os.ReadFile(p)
|
||||
if !strings.Contains(string(b), "\tRET") {
|
||||
t.Errorf("%s not formatted in place:\n%s", p, b)
|
||||
}
|
||||
}
|
||||
for _, p := range []string{filepath.Join("_refs", "c_amd64.s"), filepath.Join(".git", "d_amd64.s")} {
|
||||
b, _ := os.ReadFile(p)
|
||||
if string(b) != string(unformatted) {
|
||||
t.Errorf("%s must not be touched:\n%s", p, b)
|
||||
}
|
||||
}
|
||||
|
||||
// Second pass: everything is canonical, nothing is listed.
|
||||
out, _, code = capture(func() int { return cmdFmt(nil) })
|
||||
if code != 0 || out != "" {
|
||||
t.Errorf("second pass: code=%d out=%q, want a no-op", code, out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdTokens(t *testing.T) {
|
||||
path := writeTemp(t, "f_amd64.s", clean)
|
||||
out, _, code := capture(func() int { return cmdTokens([]string{path}) })
|
||||
@@ -152,15 +203,32 @@ func TestCmdFmtWrite(t *testing.T) {
|
||||
func TestUsage(t *testing.T) {
|
||||
var b bytes.Buffer
|
||||
usage(&b)
|
||||
if !strings.Contains(b.String(), "gasm") {
|
||||
t.Errorf("usage text unexpected:\n%s", b.String())
|
||||
out := b.String()
|
||||
for _, want := range []string{
|
||||
"gasm", "Commands:", "Flags:", "--help", "--version",
|
||||
"tokens", "parse", "fmt", "lint", "asm", "lsp", "version",
|
||||
} {
|
||||
if !strings.Contains(out, want) {
|
||||
t.Errorf("usage text missing %q:\n%s", want, out)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdVersion(t *testing.T) {
|
||||
out, _, code := capture(func() int { return cmdVersion() })
|
||||
if code != 0 {
|
||||
t.Fatalf("code = %d", code)
|
||||
}
|
||||
if !strings.Contains(out, version) {
|
||||
t.Errorf("version output %q does not mention %q", out, version)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdArgErrors(t *testing.T) {
|
||||
// Missing file arguments produce a usage error (code 2).
|
||||
if _, _, code := capture(func() int { return cmdFmt(nil) }); code != 2 {
|
||||
t.Errorf("cmdFmt() code = %d, want 2", code)
|
||||
// A missing path is an error (code 1); cmdFmt with no arguments is the
|
||||
// recursive mode now, covered by TestCmdFmtRecursive.
|
||||
if _, _, code := capture(func() int { return cmdFmt([]string{"no/such/path"}) }); code != 1 {
|
||||
t.Errorf("cmdFmt(missing path) code = %d, want 1", code)
|
||||
}
|
||||
if _, _, code := capture(func() int { return cmdLint(nil) }); code != 2 {
|
||||
t.Errorf("cmdLint() code = %d, want 2", code)
|
||||
@@ -172,3 +240,55 @@ func TestCmdArgErrors(t *testing.T) {
|
||||
t.Errorf("cmdParse() code = %d, want 2", code)
|
||||
}
|
||||
}
|
||||
|
||||
// TestVerifySmokeCrashIsolation checks that a function faulting on its
|
||||
// zeroed smoke arguments is reported as CRASH by a child process instead of
|
||||
// killing `gasm verify` itself.
|
||||
func TestVerifySmokeCrashIsolation(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("builds the gasm binary")
|
||||
}
|
||||
if runtime.GOARCH != "amd64" {
|
||||
t.Skip("amd64 JIT only")
|
||||
}
|
||||
bin := filepath.Join(t.TempDir(), "gasm")
|
||||
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
|
||||
t.Fatalf("build gasm: %v\n%s", err, out)
|
||||
}
|
||||
src := filepath.Join(t.TempDir(), "crash_amd64.s")
|
||||
kernel := "#include \"textflag.h\"\n" +
|
||||
"\n" +
|
||||
"// func Fault(x []byte) int\n" +
|
||||
"TEXT ·Fault(SB), NOSPLIT, $0-32\n" +
|
||||
"\tMOVQ\tx+0(FP), AX\n" +
|
||||
"\tMOVQ\t(AX), AX // faults on the zeroed nil pointer\n" +
|
||||
"\tMOVQ\tAX, ret+24(FP)\n" +
|
||||
"\tRET\n"
|
||||
if err := os.WriteFile(src, []byte(kernel), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
cmd := exec.Command(bin, "verify", "-smoke", src)
|
||||
out, err := cmd.CombinedOutput()
|
||||
if err == nil {
|
||||
t.Fatalf("expected a failure report, got success:\n%s", out)
|
||||
}
|
||||
if exitErr, ok := err.(*exec.ExitError); ok {
|
||||
if ws, ok := exitErr.Sys().(syscall.WaitStatus); ok && ws.Signaled() {
|
||||
t.Fatalf("verify died from %v — the crash was not isolated:\n%s", ws.Signal(), out)
|
||||
}
|
||||
}
|
||||
if !strings.Contains(string(out), "CRASH") {
|
||||
t.Errorf("output does not report CRASH:\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSweepCheckLines(t *testing.T) {
|
||||
out := []byte("crash_amd64.s: 1 functions JIT-loaded\n" +
|
||||
" Fault: 21 bytes, args=32, frame=0 NOSPLIT\n" +
|
||||
" smoke: OK\n" +
|
||||
" abi: clean (10 varied inputs)\n")
|
||||
want := " smoke: OK\n abi: clean (10 varied inputs)"
|
||||
if got := sweepCheckLines(out); got != want {
|
||||
t.Errorf("sweepCheckLines = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,345 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
gasmast "sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
gasmparser "sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// cmdScaffold generates a differential test skeleton for every kernel in a
|
||||
// file: a Go test that seeds random states, drives both the assembly kernel
|
||||
// and a caller-provided portable reference, and compares the outputs
|
||||
// byte-for-byte. The lesson this encodes: a pipeline-level fuzz cannot see
|
||||
// an unwired kernel — only a direct-call differential against the portable
|
||||
// specification can, so every kernel ships with one.
|
||||
//
|
||||
// The generated file follows two conventions the caller fills in:
|
||||
// - the assembly symbols resolve because the test lives in the kernel's
|
||||
// own package (the //go:noescape declarations reference them);
|
||||
// - each kernel gets a <name>Portable Go function the author implements as
|
||||
// the specification, and the test fails on the first divergent byte.
|
||||
func cmdScaffold(args []string) error {
|
||||
fs := newCommand("scaffold", "gasm scaffold differential <file.s>", `
|
||||
Print a differential test skeleton for every // func signature in FILE.
|
||||
The test seeds random states, drives the kernel and a portable reference
|
||||
(<name>Portable), and compares outputs byte-for-byte. Write the reference
|
||||
bodies, place the file in the kernel's package, and run it in CI.
|
||||
`)
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return err
|
||||
}
|
||||
rest := fs.Args()
|
||||
// The first positional word is the scaffold style; "differential" is the
|
||||
// only one today.
|
||||
if len(rest) > 0 && rest[0] == "differential" {
|
||||
rest = rest[1:]
|
||||
}
|
||||
if len(rest) != 1 {
|
||||
return fmt.Errorf("usage: gasm scaffold differential <file.s>")
|
||||
}
|
||||
path := rest[0]
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
f, errs := gasmparser.Parse(path, string(src))
|
||||
if len(errs) > 0 {
|
||||
return fmt.Errorf("parse: %v", errs[0])
|
||||
}
|
||||
|
||||
var out strings.Builder
|
||||
out.WriteString(headerComment)
|
||||
out.WriteString("package " + packageName + "\n\n")
|
||||
out.WriteString("import (\n\t\"bytes\"\n\t\"math/rand\"\n\t\"testing\"\n)\n\n")
|
||||
out.WriteString(generatedHelpers)
|
||||
|
||||
kernels := 0
|
||||
for _, d := range f.Decls {
|
||||
txt, ok := d.(*gasmast.Text)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
params, results, ok := parseSig(txt.Doc)
|
||||
if !ok || len(params) == 0 {
|
||||
continue
|
||||
}
|
||||
kernels++
|
||||
name := txt.Name.Name
|
||||
fmt.Fprintf(&out, "// %sPortable is the specification %s is pinned against:\n", name, name)
|
||||
fmt.Fprintf(&out, "// fill in a straightforward implementation of the same contract.\n")
|
||||
fmt.Fprintf(&out, "func %sPortable(%s) (%s) {\n\tpanic(\"implement the portable specification\")\n}\n\n", name, paramDecl(params), resultDecl(results))
|
||||
|
||||
fmt.Fprintf(&out, "func Test%sDifferential(t *testing.T) {\n", strings.ToUpper(name[:1])+name[1:])
|
||||
fmt.Fprintf(&out, "\trng := rand.New(rand.NewSource(1))\n")
|
||||
fmt.Fprintf(&out, "\tfor range 1000 {\n")
|
||||
// Seed two independent argument sets per iteration: the kernel runs
|
||||
// on set A, the portable reference on set B, so in-place writes
|
||||
// through pointer/slice arguments cannot contaminate the other side.
|
||||
var sliceNames []string
|
||||
seen := map[string]bool{}
|
||||
aArgs := make([]string, 0, len(params))
|
||||
bArgs := make([]string, 0, len(params))
|
||||
for _, p := range params {
|
||||
a, b, slices := genParamSeed(&out, p, seen)
|
||||
aArgs = append(aArgs, a)
|
||||
bArgs = append(bArgs, b)
|
||||
sliceNames = append(sliceNames, slices...)
|
||||
}
|
||||
fmt.Fprintf(&out, "\t\tgot := %s(%s)\n", name, strings.Join(aArgs, ", "))
|
||||
fmt.Fprintf(&out, "\t\twant := %sPortable(%s)\n", name, strings.Join(bArgs, ", "))
|
||||
fmt.Fprintf(&out, "\t\tif !bytes.Equal(outputBytes(got), outputBytes(want)) {\n")
|
||||
fmt.Fprintf(&out, "\t\t\tt.Fatalf(\"kernel diverges from the portable spec (seed 1, deterministic)\")\n")
|
||||
fmt.Fprintf(&out, "\t\t}\n")
|
||||
for _, s := range sliceNames {
|
||||
fmt.Fprintf(&out, "\t\tif !bytes.Equal(outputBytes(%sA), outputBytes(%sB)) {\n", s, s)
|
||||
fmt.Fprintf(&out, "\t\t\tt.Fatalf(\"kernel mutated %%q differently (seed 1, deterministic)\", %q)\n", s)
|
||||
fmt.Fprintf(&out, "\t\t}\n")
|
||||
}
|
||||
fmt.Fprintf(&out, "\t}\n}\n\n")
|
||||
}
|
||||
if kernels == 0 {
|
||||
return fmt.Errorf("%s: no // func signatures found; add one doc comment per kernel", path)
|
||||
}
|
||||
os.Stdout.WriteString(out.String())
|
||||
return nil
|
||||
}
|
||||
|
||||
const packageName = "yourpkg"
|
||||
|
||||
const headerComment = `// Code generated by gasm scaffold differential; EDIT THE PANICS.
|
||||
// Each Test*Differential drives the assembly kernel and its portable
|
||||
// reference over the same random states and compares the outputs.
|
||||
// Place this file in the kernel's own package so the symbols resolve.
|
||||
|
||||
`
|
||||
|
||||
// sigParam is one parsed // func parameter.
|
||||
type sigParam struct {
|
||||
Names []string
|
||||
Type string
|
||||
}
|
||||
|
||||
type sigResult struct {
|
||||
Names []string
|
||||
Type string
|
||||
}
|
||||
|
||||
// parseSig parses the // func signature of a doc comment.
|
||||
func parseSig(doc string) ([]sigParam, []sigResult, bool) {
|
||||
var line string
|
||||
for l := range strings.SplitSeq(doc, "\n") {
|
||||
if t := strings.TrimSpace(l); strings.HasPrefix(t, "func ") {
|
||||
line = t
|
||||
break
|
||||
}
|
||||
}
|
||||
if line == "" {
|
||||
return nil, nil, false
|
||||
}
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "sig.go", "package p\n"+line+" {}\n", 0)
|
||||
if err != nil {
|
||||
return nil, nil, false
|
||||
}
|
||||
fd, ok := f.Decls[0].(*ast.FuncDecl)
|
||||
if !ok || fd.Type == nil {
|
||||
return nil, nil, false
|
||||
}
|
||||
var params []sigParam
|
||||
for _, field := range fd.Type.Params.List {
|
||||
typ := exprString(field.Type)
|
||||
if len(field.Names) == 0 {
|
||||
params = append(params, sigParam{Names: []string{""}, Type: typ})
|
||||
continue
|
||||
}
|
||||
// Shared names (`L, result *byte`) expand to one entry per name:
|
||||
// every name is a separate argument at the call site.
|
||||
for _, n := range field.Names {
|
||||
params = append(params, sigParam{Names: []string{n.Name}, Type: typ})
|
||||
}
|
||||
}
|
||||
var results []sigResult
|
||||
if fd.Type.Results != nil {
|
||||
for _, field := range fd.Type.Results.List {
|
||||
results = append(results, sigResult{Names: identNames(field.Names), Type: exprString(field.Type)})
|
||||
}
|
||||
}
|
||||
return params, results, true
|
||||
}
|
||||
|
||||
func identNames(idents []*ast.Ident) []string {
|
||||
var out []string
|
||||
for _, id := range idents {
|
||||
out = append(out, id.Name)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func exprString(e ast.Expr) string {
|
||||
switch t := e.(type) {
|
||||
case *ast.Ident:
|
||||
return t.Name
|
||||
case *ast.StarExpr:
|
||||
return "*" + exprString(t.X)
|
||||
case *ast.SelectorExpr:
|
||||
return exprString(t.X) + "." + t.Sel.Name
|
||||
case *ast.ArrayType:
|
||||
if t.Len == nil {
|
||||
return "[]" + exprString(t.Elt)
|
||||
}
|
||||
return "[N]" + exprString(t.Elt)
|
||||
}
|
||||
return "interface{}"
|
||||
}
|
||||
|
||||
// paramDecl renders a parameter list for the portable reference signature.
|
||||
func paramDecl(params []sigParam) string {
|
||||
var parts []string
|
||||
for _, p := range params {
|
||||
if len(p.Names) == 0 {
|
||||
parts = append(parts, p.Type)
|
||||
continue
|
||||
}
|
||||
for _, n := range p.Names {
|
||||
parts = append(parts, n+" "+p.Type)
|
||||
}
|
||||
}
|
||||
return strings.Join(parts, ", ")
|
||||
}
|
||||
|
||||
// resultDecl renders a result list; unnamed results keep bare types.
|
||||
func resultDecl(results []sigResult) string {
|
||||
if len(results) == 0 {
|
||||
return ""
|
||||
}
|
||||
var parts []string
|
||||
for _, r := range results {
|
||||
parts = append(parts, r.Type)
|
||||
}
|
||||
return strings.Join(parts, ", ")
|
||||
}
|
||||
|
||||
// genParamSeed emits the seeding statements for one parameter and returns
|
||||
// the kernel-side (A) and reference-side (B) argument expressions, plus the
|
||||
// names of any slice variables written in place (compared after the calls).
|
||||
func genParamSeed(out *strings.Builder, p sigParam, seen map[string]bool) (aArg, bArg string, slices []string) {
|
||||
name := p.Names[0]
|
||||
elem := strings.TrimPrefix(p.Type, "*")
|
||||
isSlice := strings.HasPrefix(p.Type, "[]")
|
||||
if isSlice {
|
||||
elem = strings.TrimPrefix(p.Type, "[]")
|
||||
}
|
||||
switch {
|
||||
case isSlice:
|
||||
v := uniqueName(seen, name)
|
||||
fmt.Fprintf(out, "\t\t%sA := make([]%s, 1+rng.Intn(512))\n", v, elem)
|
||||
fmt.Fprintf(out, "\t\t%sB := make([]%s, len(%sA))\n", v, elem, v)
|
||||
fmt.Fprintf(out, "\t\tfor i := range %sA {\n", v)
|
||||
fmt.Fprintf(out, "\t\t\tw%s := %s(rng.Intn(256))\n", v, goCast(elem))
|
||||
fmt.Fprintf(out, "\t\t\t%sA[i] = w%s\n", v, v)
|
||||
fmt.Fprintf(out, "\t\t\t%sB[i] = w%s\n", v, v)
|
||||
fmt.Fprintf(out, "\t\t}\n")
|
||||
return v, v, []string{v}
|
||||
case strings.HasPrefix(p.Type, "*"):
|
||||
v := uniqueName(seen, name)
|
||||
fmt.Fprintf(out, "\t\tvar %sA, %sB %s\n", v, v, elem)
|
||||
fmt.Fprintf(out, "\t\tw%s := %s(rng.Intn(256))\n", v, goCast(elem))
|
||||
fmt.Fprintf(out, "\t\t%sA = w%s\n", v, v)
|
||||
fmt.Fprintf(out, "\t\t%sB = w%s\n", v, v)
|
||||
return "&" + v + "A", "&" + v + "B", nil
|
||||
default:
|
||||
v := uniqueName(seen, name)
|
||||
fmt.Fprintf(out, "\t\tw%s := %s(rng.Intn(512))\n", v, goCast(""))
|
||||
fmt.Fprintf(out, "\t\tvar %sA, %sB %s = w%s, w%s\n", v, v, p.Type, v, v)
|
||||
return v + "A", v + "B", nil
|
||||
}
|
||||
}
|
||||
|
||||
// uniqueName de-duplicates seeded variable names when one kernel takes two
|
||||
// parameters of the same name (impossible in Go) or a name repeats across
|
||||
// kernels in one file.
|
||||
func uniqueName(seen map[string]bool, base string) string {
|
||||
if base == "" {
|
||||
base = "arg"
|
||||
}
|
||||
if !seen[base] {
|
||||
seen[base] = true
|
||||
return base
|
||||
}
|
||||
for i := 2; ; i++ {
|
||||
cand := fmt.Sprintf("%s%d", base, i)
|
||||
if !seen[cand] {
|
||||
seen[cand] = true
|
||||
return cand
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// goCast returns the conversion turning rng.Intn into the element type.
|
||||
func goCast(elem string) string {
|
||||
switch elem {
|
||||
case "byte", "uint8":
|
||||
return "byte"
|
||||
case "int8":
|
||||
return "int8"
|
||||
case "uint16":
|
||||
return "uint16"
|
||||
case "int16":
|
||||
return "int16"
|
||||
case "uint32":
|
||||
return "uint32"
|
||||
case "int32":
|
||||
return "int32"
|
||||
case "uint64":
|
||||
return "uint64"
|
||||
default:
|
||||
return "int"
|
||||
}
|
||||
}
|
||||
|
||||
// generatedHelpers is emitted into every generated test file: outputBytes
|
||||
// narrows returned slices and scalars to a byte form for the comparison.
|
||||
// It lives in the template, not in this binary, because only the generated
|
||||
// file ever calls it.
|
||||
const generatedHelpers = `// outputBytes narrows a returned slice or scalar to bytes for the
|
||||
// comparison; extend the switch when a kernel returns a wider type.
|
||||
func outputBytes(v any) []byte {
|
||||
switch t := v.(type) {
|
||||
case []byte:
|
||||
return t
|
||||
case []int32:
|
||||
b := make([]byte, 4*len(t))
|
||||
for i, x := range t {
|
||||
b[i*4] = byte(x)
|
||||
b[i*4+1] = byte(x >> 8)
|
||||
b[i*4+2] = byte(x >> 16)
|
||||
b[i*4+3] = byte(x >> 24)
|
||||
}
|
||||
return b
|
||||
case []uint16:
|
||||
b := make([]byte, 2*len(t))
|
||||
for i, x := range t {
|
||||
b[i*2] = byte(x)
|
||||
b[i*2+1] = byte(x >> 8)
|
||||
}
|
||||
return b
|
||||
case int:
|
||||
b := make([]byte, 8)
|
||||
for i := range 8 {
|
||||
b[i] = byte(uint64(t) >> (8 * i))
|
||||
}
|
||||
return b
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
`
|
||||
@@ -0,0 +1,253 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux
|
||||
|
||||
package debug
|
||||
|
||||
import "strings"
|
||||
|
||||
import "fmt"
|
||||
|
||||
// Breakpoint is one INT3 breakpoint in the debuggee.
|
||||
type Breakpoint struct {
|
||||
Addr uint64 // absolute address in the debuggee
|
||||
Label string // source label ("" for raw addresses)
|
||||
Orig byte // original byte at Addr (restored on removal)
|
||||
Enabled bool
|
||||
Cond *Condition // optional condition (nil = unconditional)
|
||||
hits int
|
||||
}
|
||||
|
||||
// Condition is a register-comparison condition evaluated when a breakpoint
|
||||
// is hit. Supports three forms:
|
||||
// - register vs constant: <reg> <op> <value>
|
||||
// - register vs register: <reg> <op> <reg2>
|
||||
// - register vs memory: <reg> <op> *<addr>
|
||||
type Condition struct {
|
||||
Reg string // register name (rax, rbx, rip, rsp, ...)
|
||||
Op string // comparison operator: ==, !=, <, >, <=, >=
|
||||
Value uint64 // constant value (when Reg2 == "" and MemAddr == 0)
|
||||
Reg2 string // second register name (for register-register comparison)
|
||||
MemAddr uint64 // memory address (for register-memory comparison, prefixed with *)
|
||||
}
|
||||
|
||||
// Eval checks the condition against the current registers.
|
||||
func (c *Condition) Eval(regs *Regs) bool {
|
||||
actual, ok := regs.RegValue(c.Reg)
|
||||
if !ok {
|
||||
return true // unknown register — don't block
|
||||
}
|
||||
var expected uint64
|
||||
switch {
|
||||
case c.Reg2 != "":
|
||||
// Register-register comparison.
|
||||
v, ok := regs.RegValue(c.Reg2)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
expected = v
|
||||
case c.MemAddr != 0:
|
||||
// Register-memory comparison — requires a Session, not available here.
|
||||
// Fall back to treating as constant (the caller should resolve).
|
||||
expected = c.Value
|
||||
default:
|
||||
expected = c.Value
|
||||
}
|
||||
switch c.Op {
|
||||
case "==", "=":
|
||||
return actual == expected
|
||||
case "!=":
|
||||
return actual != expected
|
||||
case "<":
|
||||
return actual < expected
|
||||
case ">":
|
||||
return actual > expected
|
||||
case "<=":
|
||||
return actual <= expected
|
||||
case ">=":
|
||||
return actual >= expected
|
||||
default:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// Breakpoints manages the set of breakpoints for a Session.
|
||||
// Breakpoints manages software breakpoints for a debuggee.
|
||||
type Breakpoints struct {
|
||||
t tracer
|
||||
bps map[uint64]*Breakpoint
|
||||
}
|
||||
|
||||
// NewBreakpoints creates a new breakpoint manager.
|
||||
func NewBreakpoints(t tracer) *Breakpoints {
|
||||
return &Breakpoints{t: t, bps: make(map[uint64]*Breakpoint)}
|
||||
}
|
||||
|
||||
// Set installs a breakpoint at addr (replaces any existing one).
|
||||
func (bm *Breakpoints) Set(addr uint64, label string) (*Breakpoint, error) {
|
||||
return bm.SetWithCond(addr, label, nil)
|
||||
}
|
||||
|
||||
// SetWithCond installs a breakpoint with an optional condition.
|
||||
func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (*Breakpoint, error) {
|
||||
if bp, ok := bm.bps[addr]; ok {
|
||||
bp.Enabled = true
|
||||
bp.Cond = cond
|
||||
return bp, nil
|
||||
}
|
||||
// Read the original bytes.
|
||||
word, err := bm.t.Peek(addr)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
orig := byte(word)
|
||||
// Patch with the breakpoint instruction, preserving the rest of the word.
|
||||
mask := uint64(0)
|
||||
for range breakpointInsn {
|
||||
mask = (mask << 8) | 0xFF
|
||||
}
|
||||
patched := (word &^ mask) | breakpointWord(breakpointInsn)
|
||||
if err := bm.t.Poke(addr, patched); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
bp := &Breakpoint{Addr: addr, Label: label, Orig: orig, Enabled: true, Cond: cond}
|
||||
bm.bps[addr] = bp
|
||||
return bp, nil
|
||||
}
|
||||
|
||||
// Info returns a formatted list of all breakpoints.
|
||||
func (bm *Breakpoints) Info() string {
|
||||
if len(bm.bps) == 0 {
|
||||
return "no breakpoints set\n"
|
||||
}
|
||||
var result strings.Builder
|
||||
i := 0
|
||||
for _, bp := range bm.bps {
|
||||
i++
|
||||
status := "enabled"
|
||||
if !bp.Enabled {
|
||||
status = "disabled"
|
||||
}
|
||||
label := bp.Label
|
||||
if label == "" {
|
||||
label = fmt.Sprintf("%#x", bp.Addr)
|
||||
}
|
||||
cond := ""
|
||||
if bp.Cond != nil {
|
||||
cond = fmt.Sprintf(" if %s %s %#x", bp.Cond.Reg, bp.Cond.Op, bp.Cond.Value)
|
||||
}
|
||||
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond))
|
||||
}
|
||||
return result.String()
|
||||
}
|
||||
|
||||
// Clear removes the breakpoint at addr, restoring the original byte.
|
||||
func (bm *Breakpoints) Clear(addr uint64) error {
|
||||
bp, ok := bm.bps[addr]
|
||||
if !ok {
|
||||
return fmt.Errorf("debug: no breakpoint at %#x", addr)
|
||||
}
|
||||
word, err := bm.t.Peek(addr)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
||||
if err := bm.t.Poke(addr, restored); err != nil {
|
||||
return err
|
||||
}
|
||||
delete(bm.bps, addr)
|
||||
return nil
|
||||
}
|
||||
|
||||
// ClearAll removes all breakpoints.
|
||||
func (bm *Breakpoints) ClearAll() error {
|
||||
for addr := range bm.bps {
|
||||
if err := bm.Clear(addr); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// At returns the breakpoint at addr, if any.
|
||||
func (bm *Breakpoints) At(addr uint64) *Breakpoint {
|
||||
return bm.bps[addr]
|
||||
}
|
||||
|
||||
// All returns all breakpoints.
|
||||
func (bm *Breakpoints) All() []*Breakpoint {
|
||||
out := make([]*Breakpoint, 0, len(bm.bps))
|
||||
for _, bp := range bm.bps {
|
||||
out = append(out, bp)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// HandleTrap is called after the debuggee stops on SIGTRAP. It checks
|
||||
// whether the trap was caused by one of our breakpoints (PC-adjust matches
|
||||
// a breakpoint address), restores the original byte, rewinds PC, and
|
||||
// returns the breakpoint that was hit (or nil if it was a single-step).
|
||||
// Hits returns how many times the breakpoint has been hit.
|
||||
func (bp *Breakpoint) Hits() int { return bp.hits }
|
||||
|
||||
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
|
||||
// After a breakpoint trap, PC points past the breakpoint instruction.
|
||||
trapAddr := regs.GetPC() - uint64(breakpointPCAdjust)
|
||||
bp, ok := bm.bps[trapAddr]
|
||||
if !ok || !bp.Enabled {
|
||||
return nil // single-step trap or unknown
|
||||
}
|
||||
// Check the condition (if any).
|
||||
if bp.Cond != nil && !bp.Cond.Eval(regs) {
|
||||
// Condition not met — restore the byte but do NOT rewind RIP.
|
||||
// The process continues from the next instruction (past the INT3).
|
||||
word, err := bm.t.Peek(trapAddr)
|
||||
if err == nil {
|
||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
||||
bm.t.Poke(trapAddr, restored)
|
||||
}
|
||||
// RIP is already past the INT3 (trapAddr + 1). Don't rewind.
|
||||
return nil
|
||||
}
|
||||
bp.hits++
|
||||
// Restore the original byte.
|
||||
word, err := bm.t.Peek(trapAddr)
|
||||
if err == nil {
|
||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
||||
bm.t.Poke(trapAddr, restored)
|
||||
}
|
||||
// Rewind PC to re-execute the original instruction.
|
||||
regs.SetPC(trapAddr)
|
||||
bm.t.SetRegs(regs)
|
||||
return bp
|
||||
}
|
||||
|
||||
// Reinsert re-inserts the breakpoint at addr after a single-step past it.
|
||||
// Called after Step() when we want the breakpoint to fire again on the
|
||||
// next Continue().
|
||||
func (bm *Breakpoints) Reinsert(addr uint64) error {
|
||||
bp, ok := bm.bps[addr]
|
||||
if !ok || !bp.Enabled {
|
||||
return nil
|
||||
}
|
||||
word, err := bm.t.Peek(addr)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
mask := uint64(0)
|
||||
for range breakpointInsn {
|
||||
mask = (mask << 8) | 0xFF
|
||||
}
|
||||
patched := (word &^ mask) | breakpointWord(breakpointInsn)
|
||||
return bm.t.Poke(addr, patched)
|
||||
}
|
||||
|
||||
// breakpointWord converts the breakpoint instruction bytes to a uint64.
|
||||
func breakpointWord(insn []byte) uint64 {
|
||||
var w uint64
|
||||
for i, b := range insn {
|
||||
w |= uint64(b) << (i * 8)
|
||||
}
|
||||
return w
|
||||
}
|
||||
@@ -0,0 +1,326 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestConditionEval(t *testing.T) {
|
||||
regs := &Regs{
|
||||
RAX: 42,
|
||||
RBX: 0,
|
||||
RCX: 100,
|
||||
RIP: 0x1000,
|
||||
RSP: 0x2000,
|
||||
R8: 8,
|
||||
R15: 15,
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
cond Condition
|
||||
want bool
|
||||
}{
|
||||
{Condition{Reg: "rax", Op: "==", Value: 42}, true},
|
||||
{Condition{Reg: "rax", Op: "==", Value: 43}, false},
|
||||
{Condition{Reg: "rax", Op: "!=", Value: 43}, true},
|
||||
{Condition{Reg: "rax", Op: "!=", Value: 42}, false},
|
||||
{Condition{Reg: "rax", Op: "<", Value: 50}, true},
|
||||
{Condition{Reg: "rax", Op: "<", Value: 40}, false},
|
||||
{Condition{Reg: "rax", Op: ">", Value: 40}, true},
|
||||
{Condition{Reg: "rax", Op: ">", Value: 50}, false},
|
||||
{Condition{Reg: "rax", Op: "<=", Value: 42}, true},
|
||||
{Condition{Reg: "rax", Op: ">=", Value: 42}, true},
|
||||
{Condition{Reg: "rbx", Op: "==", Value: 0}, true},
|
||||
{Condition{Reg: "rcx", Op: ">", Value: 50}, true},
|
||||
{Condition{Reg: "rip", Op: "==", Value: 0x1000}, true},
|
||||
{Condition{Reg: "rsp", Op: ">", Value: 0x1000}, true},
|
||||
{Condition{Reg: "r8", Op: "==", Value: 8}, true},
|
||||
{Condition{Reg: "r15", Op: "==", Value: 15}, true},
|
||||
{Condition{Reg: "eax", Op: "==", Value: 42}, true}, // 32-bit alias
|
||||
{Condition{Reg: "ax", Op: "==", Value: 42}, true}, // 16-bit alias
|
||||
{Condition{Reg: "unknown", Op: "==", Value: 0}, true}, // unknown reg → don't block
|
||||
{Condition{Reg: "rax", Op: "??", Value: 0}, true}, // unknown op → don't block
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := tt.cond.Eval(regs)
|
||||
if got != tt.want {
|
||||
t.Errorf("Condition{%q %q %d}.Eval() = %v, want %v",
|
||||
tt.cond.Reg, tt.cond.Op, tt.cond.Value, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestLineAt(t *testing.T) {
|
||||
lines := []SourceLine{
|
||||
{Offset: 0, Line: 5},
|
||||
{Offset: 5, Line: 6},
|
||||
{Offset: 10, Line: 7},
|
||||
{Offset: 15, Line: 8},
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
offset int
|
||||
want int
|
||||
}{
|
||||
{0, 5},
|
||||
{1, 5},
|
||||
{4, 5},
|
||||
{5, 6},
|
||||
{7, 6},
|
||||
{10, 7},
|
||||
{12, 7},
|
||||
{15, 8},
|
||||
{20, 8},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := lineAt(lines, tt.offset)
|
||||
if got != tt.want {
|
||||
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
|
||||
}
|
||||
}
|
||||
|
||||
// Empty table.
|
||||
if lineAt(nil, 5) != 0 {
|
||||
t.Error("lineAt(nil, 5) should return 0")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOffsetForLine(t *testing.T) {
|
||||
lines := []SourceLine{
|
||||
{Offset: 0, Line: 5},
|
||||
{Offset: 5, Line: 6},
|
||||
{Offset: 10, Line: 7},
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
line int
|
||||
want int
|
||||
}{
|
||||
{5, 0},
|
||||
{6, 5},
|
||||
{7, 10},
|
||||
{99, -1}, // not found
|
||||
{0, -1}, // not found
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := offsetForLine(lines, tt.line)
|
||||
if got != tt.want {
|
||||
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDecodeRflags(t *testing.T) {
|
||||
tests := []struct {
|
||||
flags uint64
|
||||
want string
|
||||
}{
|
||||
{0x202, "IF"}, // only IF set (bit 9)
|
||||
{0x246, "PF ZF IF"}, // PF(2) + ZF(6) + IF(9)
|
||||
{0x001, "CF"}, // carry flag
|
||||
{0x080, "SF"}, // sign flag
|
||||
{0x800, "OF"}, // overflow flag
|
||||
{0x000, "none"}, // no flags
|
||||
{0x202 | 0x001, "CF IF"}, // CF + IF
|
||||
{0x3F7, "CF PF AF ZF SF TF IF"}, // all arithmetic flags
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := decodeRflags(tt.flags)
|
||||
if got != tt.want {
|
||||
t.Errorf("decodeRflags(%#x) = %q, want %q", tt.flags, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestNearestLabel(t *testing.T) {
|
||||
labels := []Label{
|
||||
{Name: "start", Offset: 0},
|
||||
{Name: "loop", Offset: 10},
|
||||
{Name: "done", Offset: 20},
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
offset int
|
||||
want string
|
||||
}{
|
||||
{0, "start"},
|
||||
{5, "start"},
|
||||
{10, "loop"},
|
||||
{15, "loop"},
|
||||
{20, "done"},
|
||||
{25, "done"},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := nearestLabel(labels, tt.offset)
|
||||
if got != tt.want {
|
||||
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointsSetAndClear(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
|
||||
// Set a breakpoint at address 0x1000.
|
||||
bp, err := bm.Set(0x1000, "test")
|
||||
if err != nil {
|
||||
t.Fatalf("Set: %v", err)
|
||||
}
|
||||
if !bp.Enabled {
|
||||
t.Error("breakpoint not enabled")
|
||||
}
|
||||
if bp.Label != "test" {
|
||||
t.Errorf("label = %q, want test", bp.Label)
|
||||
}
|
||||
|
||||
// Verify Peek was called.
|
||||
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
|
||||
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
|
||||
}
|
||||
|
||||
// Verify Poke wrote INT3.
|
||||
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
|
||||
t.Errorf("pokes = %v", tr.pokes)
|
||||
}
|
||||
|
||||
// At should find it.
|
||||
if bm.At(0x1000) == nil {
|
||||
t.Error("At(0x1000) returned nil")
|
||||
}
|
||||
|
||||
// All should return it.
|
||||
all := bm.All()
|
||||
if len(all) != 1 {
|
||||
t.Errorf("All() = %d breakpoints, want 1", len(all))
|
||||
}
|
||||
|
||||
// Clear it.
|
||||
if err := bm.Clear(0x1000); err != nil {
|
||||
t.Fatalf("Clear: %v", err)
|
||||
}
|
||||
if bm.At(0x1000) != nil {
|
||||
t.Error("At(0x1000) after Clear should be nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointsSetWithCond(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
|
||||
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
|
||||
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
|
||||
if err != nil {
|
||||
t.Fatalf("SetWithCond: %v", err)
|
||||
}
|
||||
if bp.Cond == nil || bp.Cond.Value != 42 {
|
||||
t.Error("condition not set")
|
||||
}
|
||||
|
||||
// Re-setting the same address should update the condition.
|
||||
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
|
||||
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
|
||||
if err != nil {
|
||||
t.Fatalf("SetWithCond (update): %v", err)
|
||||
}
|
||||
if bp2.Cond.Value != 100 {
|
||||
t.Error("condition not updated")
|
||||
}
|
||||
// Should have only 1 Peek (first Set), second is update (no Peek needed).
|
||||
if len(tr.peeks) != 1 {
|
||||
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointsClearAll(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
|
||||
bm.Set(0x1000, "a")
|
||||
bm.Set(0x2000, "b")
|
||||
bm.Set(0x3000, "c")
|
||||
|
||||
if len(bm.All()) != 3 {
|
||||
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
|
||||
}
|
||||
|
||||
bm.ClearAll()
|
||||
if len(bm.All()) != 0 {
|
||||
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointInfo(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
bm.Set(0x4000, "info_test")
|
||||
|
||||
info := bm.Info()
|
||||
if info == "" {
|
||||
t.Error("Info returned empty string")
|
||||
}
|
||||
if !strings.Contains(info, "info_test") {
|
||||
t.Errorf("Info %q does not contain label", info)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWatchpointSlotTracking(t *testing.T) {
|
||||
wpSlots = [4]bool{} // reset
|
||||
s := &Session{}
|
||||
|
||||
// All four slots are free initially.
|
||||
for i := range 4 {
|
||||
if s.IsWatchpointSlotUsed(i) {
|
||||
t.Errorf("slot %d should be free initially", i)
|
||||
}
|
||||
}
|
||||
if got := s.FindFreeWatchpointSlot(); got != 0 {
|
||||
t.Errorf("FindFreeWatchpointSlot() = %d, want 0", got)
|
||||
}
|
||||
|
||||
// Manually mark slots 0 and 2 as used (simulating successful SetWatchpoint).
|
||||
wpSlots[0] = true
|
||||
wpSlots[2] = true
|
||||
|
||||
if !s.IsWatchpointSlotUsed(0) {
|
||||
t.Error("slot 0 should be in use")
|
||||
}
|
||||
if s.IsWatchpointSlotUsed(1) {
|
||||
t.Error("slot 1 should be free")
|
||||
}
|
||||
if !s.IsWatchpointSlotUsed(2) {
|
||||
t.Error("slot 2 should be in use")
|
||||
}
|
||||
if s.IsWatchpointSlotUsed(3) {
|
||||
t.Error("slot 3 should be free")
|
||||
}
|
||||
if got := s.FindFreeWatchpointSlot(); got != 1 {
|
||||
t.Errorf("FindFreeWatchpointSlot() = %d, want 1", got)
|
||||
}
|
||||
|
||||
// Out-of-range slot queries return false.
|
||||
if s.IsWatchpointSlotUsed(-1) {
|
||||
t.Error("slot -1 should be reported as free (out of range)")
|
||||
}
|
||||
if s.IsWatchpointSlotUsed(4) {
|
||||
t.Error("slot 4 should be reported as free (out of range)")
|
||||
}
|
||||
|
||||
// Mark all slots used: FindFreeWatchpointSlot returns -1.
|
||||
for i := range 4 {
|
||||
wpSlots[i] = true
|
||||
}
|
||||
if got := s.FindFreeWatchpointSlot(); got != -1 {
|
||||
t.Errorf("FindFreeWatchpointSlot() with all slots used = %d, want -1", got)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
)
|
||||
|
||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||
// memory and returns its text representation and length in bytes.
|
||||
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||
// Read up to 15 bytes (max x86 instruction length).
|
||||
mem, err := s.ReadMemory(addr, 15)
|
||||
if err != nil {
|
||||
// Try a shorter read if we're near a page boundary.
|
||||
mem, err = s.ReadMemory(addr, 1)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
}
|
||||
inst, err := x86asm.Decode(mem, 64)
|
||||
if err != nil {
|
||||
return "???", 1, nil
|
||||
}
|
||||
text := x86asm.IntelSyntax(inst, addr, nil)
|
||||
return text, inst.Len, nil
|
||||
}
|
||||
|
||||
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||
// them as a formatted string with addresses and byte offsets.
|
||||
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||
var result strings.Builder
|
||||
pc := addr
|
||||
for range n {
|
||||
text, length, err := s.Disassemble(pc)
|
||||
if err != nil {
|
||||
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
|
||||
break
|
||||
}
|
||||
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
|
||||
if length == 0 {
|
||||
length = 1
|
||||
}
|
||||
pc += uint64(length)
|
||||
}
|
||||
return result.String()
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && arm64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"golang.org/x/arch/arm64/arm64asm"
|
||||
)
|
||||
|
||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||
// memory and returns its text representation and length in bytes.
|
||||
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||
mem, err := s.ReadMemory(addr, 4)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
inst, err := arm64asm.Decode(mem)
|
||||
if err != nil {
|
||||
return "???", 4, nil
|
||||
}
|
||||
text := arm64asm.GoSyntax(inst, addr, nil, nil)
|
||||
return text, 4, nil
|
||||
}
|
||||
|
||||
// DisassembleN decodes up to n instructions starting at addr.
|
||||
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||
var result string
|
||||
pc := addr
|
||||
for range n {
|
||||
text, length, err := s.Disassemble(pc)
|
||||
if err != nil {
|
||||
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
|
||||
break
|
||||
}
|
||||
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
|
||||
if length == 0 {
|
||||
length = 4
|
||||
}
|
||||
pc += uint64(length)
|
||||
}
|
||||
return result
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && loong64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"golang.org/x/arch/loong64/loong64asm"
|
||||
)
|
||||
|
||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||
// memory and returns its text representation and length in bytes.
|
||||
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||
mem, err := s.ReadMemory(addr, 4)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
inst, err := loong64asm.Decode(mem)
|
||||
if err != nil {
|
||||
return "???", 4, nil
|
||||
}
|
||||
text := loong64asm.GoSyntax(inst, addr, nil)
|
||||
return text, 4, nil
|
||||
}
|
||||
|
||||
// DisassembleN decodes up to n instructions starting at addr.
|
||||
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||
var result string
|
||||
pc := addr
|
||||
for range n {
|
||||
text, length, err := s.Disassemble(pc)
|
||||
if err != nil {
|
||||
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
|
||||
break
|
||||
}
|
||||
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
|
||||
if length == 0 {
|
||||
length = 4
|
||||
}
|
||||
pc += uint64(length)
|
||||
}
|
||||
return result
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && riscv64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"golang.org/x/arch/riscv64/riscv64asm"
|
||||
)
|
||||
|
||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||
// memory and returns its text representation and length in bytes.
|
||||
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||
mem, err := s.ReadMemory(addr, 4)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
inst, err := riscv64asm.Decode(mem)
|
||||
if err != nil {
|
||||
return "???", 4, nil
|
||||
}
|
||||
text := riscv64asm.GoSyntax(inst, addr, nil, nil)
|
||||
return text, inst.Len, nil
|
||||
}
|
||||
|
||||
// DisassembleN decodes up to n instructions starting at addr.
|
||||
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||
var result string
|
||||
pc := addr
|
||||
for range n {
|
||||
text, length, err := s.Disassemble(pc)
|
||||
if err != nil {
|
||||
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
|
||||
break
|
||||
}
|
||||
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
|
||||
if length == 0 {
|
||||
length = 4
|
||||
}
|
||||
pc += uint64(length)
|
||||
}
|
||||
return result
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package debug
|
||||
|
||||
import "fmt"
|
||||
|
||||
func printRegs(regs *Regs, codeBase, funcOff uint64) {
|
||||
fmt.Printf(" RIP = %#016x (func+%#x)\n", regs.RIP, regs.RIP-codeBase-funcOff)
|
||||
fmt.Printf(" RSP = %#016x RBP = %#016x\n", regs.RSP, regs.RBP)
|
||||
fmt.Printf(" RAX = %#016x RBX = %#016x\n", regs.RAX, regs.RBX)
|
||||
fmt.Printf(" RCX = %#016x RDX = %#016x\n", regs.RCX, regs.RDX)
|
||||
fmt.Printf(" RSI = %#016x RDI = %#016x\n", regs.RSI, regs.RDI)
|
||||
fmt.Printf(" R8 = %#016x R9 = %#016x\n", regs.R8, regs.R9)
|
||||
fmt.Printf(" R10 = %#016x R11 = %#016x\n", regs.R10, regs.R11)
|
||||
fmt.Printf(" R12 = %#016x R13 = %#016x\n", regs.R12, regs.R13)
|
||||
fmt.Printf(" R14 = %#016x R15 = %#016x\n", regs.R14, regs.R15)
|
||||
fmt.Printf(" RFLAGS = %#x [%s]\n", regs.RFLAGS, decodeRflags(regs.RFLAGS))
|
||||
}
|
||||
|
||||
func printVectorRegs(v *VectorRegs) {
|
||||
fmt.Println("\n Vector registers (YMM):")
|
||||
for i := 0; i < 16; i += 2 {
|
||||
fmt.Printf(" YMM%-2d = ", i)
|
||||
printYMM(v.YMM[i][:])
|
||||
fmt.Printf(" YMM%-2d = ", i+1)
|
||||
printYMM(v.YMM[i+1][:])
|
||||
fmt.Println()
|
||||
}
|
||||
}
|
||||
|
||||
func printYMM(b []byte) {
|
||||
for j := 0; j < 32; j += 4 {
|
||||
v := uint32(b[j]) | uint32(b[j+1])<<8 | uint32(b[j+2])<<16 | uint32(b[j+3])<<24
|
||||
fmt.Printf("%08x ", v)
|
||||
}
|
||||
}
|
||||
|
||||
func decodeRflags(f uint64) string {
|
||||
var flags string
|
||||
if f&1 != 0 {
|
||||
flags += "CF "
|
||||
}
|
||||
if f&(1<<2) != 0 {
|
||||
flags += "PF "
|
||||
}
|
||||
if f&(1<<4) != 0 {
|
||||
flags += "AF "
|
||||
}
|
||||
if f&(1<<6) != 0 {
|
||||
flags += "ZF "
|
||||
}
|
||||
if f&(1<<7) != 0 {
|
||||
flags += "SF "
|
||||
}
|
||||
if f&(1<<8) != 0 {
|
||||
flags += "TF "
|
||||
}
|
||||
if f&(1<<9) != 0 {
|
||||
flags += "IF "
|
||||
}
|
||||
if f&(1<<10) != 0 {
|
||||
flags += "DF "
|
||||
}
|
||||
if f&(1<<11) != 0 {
|
||||
flags += "OF "
|
||||
}
|
||||
if flags == "" {
|
||||
return "none"
|
||||
}
|
||||
return flags[:len(flags)-1]
|
||||
}
|
||||
|
||||
// archReturnAddr reads the return address from the stack (amd64 ABI0 convention).
|
||||
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
|
||||
return s.Peek(regs.GetSP())
|
||||
}
|
||||
|
||||
// archSPLabel returns the SP register name for display.
|
||||
func archSPLabel() string { return "RSP" }
|
||||
@@ -0,0 +1,45 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && arm64
|
||||
|
||||
package debug
|
||||
|
||||
import "fmt"
|
||||
|
||||
func printRegs(regs *Regs, codeBase, funcOff uint64) {
|
||||
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
|
||||
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.SP, regs.X29)
|
||||
fmt.Printf(" LR = %#016x\n", regs.X30)
|
||||
fmt.Printf(" X0 = %#016x X1 = %#016x\n", regs.X0, regs.X1)
|
||||
fmt.Printf(" X2 = %#016x X3 = %#016x\n", regs.X2, regs.X3)
|
||||
fmt.Printf(" X4 = %#016x X5 = %#016x\n", regs.X4, regs.X5)
|
||||
fmt.Printf(" X6 = %#016x X7 = %#016x\n", regs.X6, regs.X7)
|
||||
fmt.Printf(" X8 = %#016x X9 = %#016x\n", regs.X8, regs.X9)
|
||||
fmt.Printf(" X10 = %#016x X11 = %#016x\n", regs.X10, regs.X11)
|
||||
fmt.Printf(" X12 = %#016x X13 = %#016x\n", regs.X12, regs.X13)
|
||||
fmt.Printf(" X14 = %#016x X15 = %#016x\n", regs.X14, regs.X15)
|
||||
fmt.Printf(" X16 = %#016x X17 = %#016x\n", regs.X16, regs.X17)
|
||||
fmt.Printf(" X18 = %#016x X19 = %#016x\n", regs.X18, regs.X19)
|
||||
fmt.Printf(" X20 = %#016x X21 = %#016x\n", regs.X20, regs.X21)
|
||||
fmt.Printf(" X22 = %#016x X23 = %#016x\n", regs.X22, regs.X23)
|
||||
fmt.Printf(" X24 = %#016x X25 = %#016x\n", regs.X24, regs.X25)
|
||||
fmt.Printf(" X26 = %#016x X27 = %#016x\n", regs.X26, regs.X27)
|
||||
fmt.Printf(" X28 = %#016x PSTATE = %#x\n", regs.X28, regs.PSTATE)
|
||||
}
|
||||
|
||||
func printVectorRegs(v *VectorRegs) {
|
||||
fmt.Println("\n Vector registers (V0-V31):")
|
||||
for i := 0; i < 32; i += 2 {
|
||||
fmt.Printf(" V%-2d = %016x%016x\n", i, v.V[i][8], v.V[i][0])
|
||||
fmt.Printf(" V%-2d = %016x%016x\n", i+1, v.V[i+1][8], v.V[i+1][0])
|
||||
}
|
||||
}
|
||||
|
||||
// archReturnAddr reads the return address from LR (arm64 convention).
|
||||
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
|
||||
return regs.X30, nil
|
||||
}
|
||||
|
||||
// archSPLabel returns the SP register name for display.
|
||||
func archSPLabel() string { return "SP" }
|
||||
@@ -0,0 +1,44 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && loong64
|
||||
|
||||
package debug
|
||||
|
||||
import "fmt"
|
||||
|
||||
func printRegs(regs *Regs, codeBase, funcOff uint64) {
|
||||
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.R31, regs.R31-codeBase-funcOff)
|
||||
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.R3, regs.R21)
|
||||
fmt.Printf(" RA = %#016x\n", regs.R1)
|
||||
fmt.Printf(" A0 = %#016x A1 = %#016x\n", regs.R4, regs.R5)
|
||||
fmt.Printf(" A2 = %#016x A3 = %#016x\n", regs.R6, regs.R7)
|
||||
fmt.Printf(" A4 = %#016x A5 = %#016x\n", regs.R8, regs.R9)
|
||||
fmt.Printf(" A6 = %#016x A7 = %#016x\n", regs.R10, regs.R11)
|
||||
fmt.Printf(" T0 = %#016x T1 = %#016x\n", regs.R12, regs.R13)
|
||||
fmt.Printf(" T2 = %#016x T3 = %#016x\n", regs.R14, regs.R15)
|
||||
fmt.Printf(" T4 = %#016x T5 = %#016x\n", regs.R16, regs.R17)
|
||||
fmt.Printf(" T6 = %#016x T7 = %#016x\n", regs.R18, regs.R19)
|
||||
fmt.Printf(" T8 = %#016x\n", regs.R20)
|
||||
fmt.Printf(" S0 = %#016x S1 = %#016x\n", regs.R22, regs.R23)
|
||||
fmt.Printf(" S2 = %#016x S3 = %#016x\n", regs.R24, regs.R25)
|
||||
fmt.Printf(" S4 = %#016x S5 = %#016x\n", regs.R26, regs.R27)
|
||||
fmt.Printf(" S6 = %#016x S7 = %#016x\n", regs.R28, regs.R29)
|
||||
fmt.Printf(" S8 = %#016x\n", regs.R30)
|
||||
}
|
||||
|
||||
func printVectorRegs(v *VectorRegs) {
|
||||
fmt.Println("\n FP registers (F0-F31):")
|
||||
for i := 0; i < 32; i += 2 {
|
||||
fmt.Printf(" F%-2d = %#018x F%-2d = %#018x\n", i, v.F[i], i+1, v.F[i+1])
|
||||
}
|
||||
fmt.Printf(" FCC = %#016x FCSR = %#x\n", v.FCC, v.FCSR)
|
||||
}
|
||||
|
||||
// archReturnAddr reads the return address from RA (loong64 convention).
|
||||
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
|
||||
return regs.R1, nil
|
||||
}
|
||||
|
||||
// archSPLabel returns the SP register name for display.
|
||||
func archSPLabel() string { return "SP" }
|
||||
@@ -0,0 +1,44 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && riscv64
|
||||
|
||||
package debug
|
||||
|
||||
import "fmt"
|
||||
|
||||
func printRegs(regs *Regs, codeBase, funcOff uint64) {
|
||||
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
|
||||
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.Sp, regs.S0)
|
||||
fmt.Printf(" RA = %#016x\n", regs.Ra)
|
||||
fmt.Printf(" A0 = %#016x A1 = %#016x\n", regs.A0, regs.A1)
|
||||
fmt.Printf(" A2 = %#016x A3 = %#016x\n", regs.A2, regs.A3)
|
||||
fmt.Printf(" A4 = %#016x A5 = %#016x\n", regs.A4, regs.A5)
|
||||
fmt.Printf(" A6 = %#016x A7 = %#016x\n", regs.A6, regs.A7)
|
||||
fmt.Printf(" T0 = %#016x T1 = %#016x\n", regs.T0, regs.T1)
|
||||
fmt.Printf(" T2 = %#016x T3 = %#016x\n", regs.T2, regs.T3)
|
||||
fmt.Printf(" T4 = %#016x T5 = %#016x\n", regs.T4, regs.T5)
|
||||
fmt.Printf(" T6 = %#016x\n", regs.T6)
|
||||
fmt.Printf(" S1 = %#016x S2 = %#016x\n", regs.S1, regs.S2)
|
||||
fmt.Printf(" S3 = %#016x S4 = %#016x\n", regs.S3, regs.S4)
|
||||
fmt.Printf(" S5 = %#016x S6 = %#016x\n", regs.S5, regs.S6)
|
||||
fmt.Printf(" S7 = %#016x S8 = %#016x\n", regs.S7, regs.S8)
|
||||
fmt.Printf(" S9 = %#016x S10 = %#016x\n", regs.S9, regs.S10)
|
||||
fmt.Printf(" S11 = %#016x\n", regs.S11)
|
||||
}
|
||||
|
||||
func printVectorRegs(v *VectorRegs) {
|
||||
fmt.Println("\n FP registers (F0-F31):")
|
||||
for i := 0; i < 32; i += 2 {
|
||||
fmt.Printf(" F%-2d = %#018x F%-2d = %#018x\n", i, v.F[i], i+1, v.F[i+1])
|
||||
}
|
||||
fmt.Printf(" FCSR = %#x\n", v.FCSR)
|
||||
}
|
||||
|
||||
// archReturnAddr reads the return address from RA (riscv64 convention).
|
||||
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
|
||||
return regs.Ra, nil
|
||||
}
|
||||
|
||||
// archSPLabel returns the SP register name for display.
|
||||
func archSPLabel() string { return "SP" }
|
||||
@@ -0,0 +1,126 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
||||
)
|
||||
|
||||
// buildGasm produces the gasm binary the debugger spawns as its debuggee.
|
||||
func buildGasm(t *testing.T) string {
|
||||
t.Helper()
|
||||
if p := os.Getenv("GASM_TEST_BIN"); p != "" {
|
||||
return p
|
||||
}
|
||||
bin := filepath.Join(t.TempDir(), "gasm")
|
||||
cmd := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm")
|
||||
out, err := cmd.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("build gasm: %v: %s", err, out)
|
||||
}
|
||||
return bin
|
||||
}
|
||||
|
||||
// TestLaunchAndBreakpoint drives a real ptrace session end to end: launch the
|
||||
// debuggee, break on the first instruction of the function and expect a
|
||||
// breakpoint trap instead of a clean exit.
|
||||
func TestLaunchAndBreakpoint(t *testing.T) {
|
||||
if runtime.GOARCH != "amd64" {
|
||||
t.Skip("runs only on amd64 hosts")
|
||||
}
|
||||
// The tracer is the OS thread that forked the debuggee (PTRACE_TRACEME
|
||||
// binds the relation to that thread); every ptrace request must come
|
||||
// from the same thread, so pin the test goroutine to one thread.
|
||||
runtime.LockOSThread()
|
||||
defer runtime.UnlockOSThread()
|
||||
bin := buildGasm(t)
|
||||
|
||||
const kernelPath = "../testdata/verify/basic_amd64.s"
|
||||
k, err := verify.Load(kernelPath)
|
||||
if err != nil {
|
||||
t.Fatalf("Load: %v", err)
|
||||
}
|
||||
t.Cleanup(k.Close)
|
||||
fl, err := k.Func("wideCopy")
|
||||
if err != nil {
|
||||
t.Fatalf("Func: %v", err)
|
||||
}
|
||||
|
||||
sess, err := Launch(bin, kernelPath, "wideCopy", make([]byte, fl.Args))
|
||||
if err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
t.Cleanup(sess.Kill)
|
||||
|
||||
bm := NewBreakpoints(sess)
|
||||
entry := sess.CodeBase() + uint64(fl.Offset)
|
||||
if _, err := bm.Set(entry, "entry"); err != nil {
|
||||
t.Fatalf("Set: %v", err)
|
||||
}
|
||||
|
||||
// The INT3 must be visible in the debuggee's memory.
|
||||
word, err := sess.Peek(entry)
|
||||
if err != nil {
|
||||
t.Fatalf("Peek: %v", err)
|
||||
}
|
||||
if b := word & 0xFF; b != 0xCC {
|
||||
t.Fatalf("int3 not patched: first byte %#02x at %#x", b, entry)
|
||||
}
|
||||
|
||||
// The debuggee raises a second SIGSTOP after the launch barrier (the
|
||||
// child's RunTarget marks its entry), so like the REPL and the cover
|
||||
// mode the test keeps resuming until the breakpoint trap arrives.
|
||||
for range 10 {
|
||||
if err := sess.Continue(); err != nil {
|
||||
st, _ := os.ReadFile(fmt.Sprintf("/proc/%d/stat", sess.Pid()))
|
||||
status, _ := os.ReadFile(fmt.Sprintf("/proc/%d/status", sess.Pid()))
|
||||
t.Fatalf("Continue: %v\nstate: %s\n%s", err, fieldName(st), statusDump(status))
|
||||
}
|
||||
if sess.Exited() {
|
||||
t.Fatal("debuggee exited instead of trapping on the breakpoint")
|
||||
}
|
||||
regs, err := sess.GetRegs()
|
||||
if err != nil {
|
||||
t.Fatalf("GetRegs: %v", err)
|
||||
}
|
||||
if bp := bm.HandleTrap(®s); bp != nil {
|
||||
if bp.Addr != entry {
|
||||
t.Fatalf("trap at %#x, want %#x", bp.Addr, entry)
|
||||
}
|
||||
return // trap on the entry breakpoint: the whole flow works
|
||||
}
|
||||
}
|
||||
t.Fatal("no breakpoint trap after 10 resumes")
|
||||
}
|
||||
|
||||
func fieldName(stat []byte) string {
|
||||
f := strings.Split(string(stat), " ")
|
||||
if len(f) > 2 {
|
||||
return "state=" + f[2]
|
||||
}
|
||||
return "no stat"
|
||||
}
|
||||
|
||||
func statusDump(b []byte) string {
|
||||
var out []string
|
||||
for l := range strings.SplitSeq(string(b), "\n") {
|
||||
if strings.HasPrefix(l, "State") || strings.HasPrefix(l, "Pid") ||
|
||||
strings.HasPrefix(l, "PPid") || strings.HasPrefix(l, "TracerPid") ||
|
||||
strings.HasPrefix(l, "Threads") || strings.HasPrefix(l, "SigPnd") ||
|
||||
strings.HasPrefix(l, "SigBlk") || strings.HasPrefix(l, "SigIgn") {
|
||||
out = append(out, l)
|
||||
}
|
||||
}
|
||||
return strings.Join(out, "\n")
|
||||
}
|
||||
@@ -0,0 +1,328 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strings"
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Session is a ptrace debugging session controlling one debuggee process.
|
||||
type Session struct {
|
||||
pid int
|
||||
cmd *exec.Cmd
|
||||
stopped bool
|
||||
exited bool
|
||||
codeBase uint64 // base address of the JIT code in the debuggee
|
||||
}
|
||||
|
||||
// Launch starts the debuggee subprocess (gasm debug --target ...) and
|
||||
// attaches to it via ptrace.
|
||||
func Launch(gasmBin, asmPath, funcName string, args []byte) (*Session, error) {
|
||||
sess, _, err := LaunchWithBuffers(gasmBin, asmPath, funcName, args, "")
|
||||
return sess, err
|
||||
}
|
||||
|
||||
// LaunchWithBuffers is like Launch but also allocates buffers in the debuggee.
|
||||
//
|
||||
// It pins the calling goroutine to its OS thread and leaves it pinned: the
|
||||
// debuggee's PTRACE_TRACEME binds the tracer relation to the forking thread,
|
||||
// and every ptrace request on the session must come from that same thread.
|
||||
// All Session methods must therefore be called from the goroutine that
|
||||
// launched the session (the REPL and coverage loops do exactly that).
|
||||
func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec string) (*Session, []uint64, error) {
|
||||
runtime.LockOSThread() // ptrace requests must stay on the forking thread
|
||||
self, err := os.Executable()
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("debug: cannot find gasm binary: %w", err)
|
||||
}
|
||||
if gasmBin != "" {
|
||||
self = gasmBin
|
||||
}
|
||||
|
||||
tmpDir, err := os.MkdirTemp("", "gasm-debug-*")
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("debug: tempdir: %w", err)
|
||||
}
|
||||
argsFile := filepath.Join(tmpDir, "args.bin")
|
||||
if err := os.WriteFile(argsFile, args, 0o644); err != nil {
|
||||
os.RemoveAll(tmpDir)
|
||||
return nil, nil, fmt.Errorf("debug: write args: %w", err)
|
||||
}
|
||||
|
||||
if bufSpec != "" {
|
||||
if err := os.WriteFile(filepath.Join(tmpDir, "bufspec"), []byte(bufSpec), 0o644); err != nil {
|
||||
os.RemoveAll(tmpDir)
|
||||
return nil, nil, fmt.Errorf("debug: write bufspec: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
cmd := exec.Command(self, "debug", "--func", funcName, "--args", argsFile, asmPath)
|
||||
cmd.Env = append(os.Environ(), "GASM_DEBUG_TARGET=1", "GASM_DEBUG_TMP="+tmpDir)
|
||||
cmd.Stdout = nil
|
||||
cmd.Stderr = os.Stderr
|
||||
cmd.SysProcAttr = &syscall.SysProcAttr{}
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
os.RemoveAll(tmpDir)
|
||||
return nil, nil, fmt.Errorf("debug: start debuggee: %w", err)
|
||||
}
|
||||
|
||||
s := &Session{pid: cmd.Process.Pid, cmd: cmd}
|
||||
|
||||
readyFile := filepath.Join(tmpDir, "ready")
|
||||
for range 500 {
|
||||
if _, err := os.Stat(readyFile); err == nil {
|
||||
break
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
|
||||
// The debuggee parks itself with SIGSTOP once the JIT code is mapped.
|
||||
// A Go tracee also reports SIGURG preemption as signal-delivery-stops,
|
||||
// so the wait loops until a stop the debugger cares about instead of
|
||||
// assuming the first event is the SIGSTOP.
|
||||
if _, err := s.waitStopped(); err != nil {
|
||||
cmd.Process.Kill()
|
||||
os.RemoveAll(tmpDir)
|
||||
return nil, nil, fmt.Errorf("debug: wait for debuggee: %w", err)
|
||||
}
|
||||
s.stopped = true
|
||||
|
||||
// The debuggee reports its JIT mapping in the codebase file; that is the
|
||||
// exact region the kernel was written to. Scanning /proc/pid/maps for
|
||||
// any RWX region is only the fallback.
|
||||
if data, err := os.ReadFile(filepath.Join(tmpDir, "codebase")); err == nil {
|
||||
fmt.Sscanf(string(data), "%d", &s.codeBase)
|
||||
}
|
||||
if s.codeBase == 0 {
|
||||
s.codeBase = findRWXMapping(s.pid)
|
||||
}
|
||||
|
||||
var bufAddrs []uint64
|
||||
if bufSpec != "" {
|
||||
addrFile := filepath.Join(tmpDir, "bufaddrs")
|
||||
if data, err := os.ReadFile(addrFile); err == nil {
|
||||
for line := range strings.SplitSeq(strings.TrimSpace(string(data)), "\n") {
|
||||
var addr uint64
|
||||
if _, err := fmt.Sscanf(line, "%d", &addr); err == nil {
|
||||
bufAddrs = append(bufAddrs, addr)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return s, bufAddrs, nil
|
||||
}
|
||||
|
||||
// wait waits for the debuggee to stop and returns the wait status.
|
||||
func (s *Session) wait() error {
|
||||
var ws syscall.WaitStatus
|
||||
_, err := syscall.Wait4(s.pid, &ws, 0, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if ws.Exited() {
|
||||
s.exited = true
|
||||
return fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
|
||||
}
|
||||
s.stopped = true
|
||||
return nil
|
||||
}
|
||||
|
||||
// waitStopped consumes ptrace-stop events until one the debugger cares
|
||||
// about arrives: SIGTRAP (a breakpoint or a completed single-step) or the
|
||||
// debuggee's own SIGSTOP. A Go tracee's runtime raises SIGURG for
|
||||
// asynchronous preemption, and every signal on a traced thread surfaces as
|
||||
// a signal-delivery-stop, so those are suppressed and the tracee resumed
|
||||
// without them. Runtime noise is why a single wait can return in the
|
||||
// middle of runtime code and a resume can then fail: the event stream must
|
||||
// be drained by the tracer.
|
||||
func (s *Session) waitStopped() (syscall.Signal, error) {
|
||||
for {
|
||||
var ws syscall.WaitStatus
|
||||
if _, err := syscall.Wait4(s.pid, &ws, syscall.WUNTRACED, nil); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
if ws.Exited() {
|
||||
s.exited = true
|
||||
return 0, fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
|
||||
}
|
||||
if ws.Signaled() {
|
||||
s.exited = true
|
||||
return 0, fmt.Errorf("debuggee killed by signal %v", ws.Signal())
|
||||
}
|
||||
switch sig := ws.StopSignal(); sig {
|
||||
case syscall.SIGTRAP, syscall.SIGSTOP:
|
||||
s.stopped = true
|
||||
return sig, nil
|
||||
default:
|
||||
// Runtime noise (SIGURG preemption and friends): resume the
|
||||
// tracee without delivering the signal.
|
||||
if _, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_CONT),
|
||||
uintptr(s.pid),
|
||||
0, 0, 0, 0,
|
||||
); errno != 0 {
|
||||
return 0, fmt.Errorf("debug: PTRACE_CONT: %w", errno)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Peek reads a word (8 bytes) from the debuggee's memory at addr.
|
||||
func (s *Session) Peek(addr uint64) (uint64, error) {
|
||||
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
|
||||
}
|
||||
defer mem.Close()
|
||||
buf := make([]byte, 8)
|
||||
if _, err := mem.ReadAt(buf, int64(addr)); err != nil {
|
||||
return 0, fmt.Errorf("debug: read mem %#x: %w", addr, err)
|
||||
}
|
||||
return uint64(buf[0]) | uint64(buf[1])<<8 | uint64(buf[2])<<16 | uint64(buf[3])<<24 |
|
||||
uint64(buf[4])<<32 | uint64(buf[5])<<40 | uint64(buf[6])<<48 | uint64(buf[7])<<56, nil
|
||||
}
|
||||
|
||||
// Poke writes a word (8 bytes) to the debuggee's memory at addr.
|
||||
func (s *Session) Poke(addr, val uint64) error {
|
||||
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_WRONLY, 0)
|
||||
if err != nil {
|
||||
return fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
|
||||
}
|
||||
defer mem.Close()
|
||||
buf := []byte{byte(val), byte(val >> 8), byte(val >> 16), byte(val >> 24),
|
||||
byte(val >> 32), byte(val >> 40), byte(val >> 48), byte(val >> 56)}
|
||||
if _, err := mem.WriteAt(buf, int64(addr)); err != nil {
|
||||
return fmt.Errorf("debug: write mem %#x: %w", addr, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ReadMemory reads len bytes from the debuggee's memory at addr.
|
||||
func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) {
|
||||
out := make([]byte, length)
|
||||
for i := 0; i < length; i += 8 {
|
||||
word, err := s.Peek(addr + uint64(i))
|
||||
if err != nil {
|
||||
return out[:i], err
|
||||
}
|
||||
for j := 0; j < 8 && i+j < length; j++ {
|
||||
out[i+j] = byte(word >> (8 * j))
|
||||
}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// WriteMemory writes bytes to the debuggee's memory at addr.
|
||||
func (s *Session) WriteMemory(addr uint64, data []byte) error {
|
||||
for i := 0; i < len(data); i += 8 {
|
||||
end := min(i+8, len(data))
|
||||
var word uint64
|
||||
for j := 0; j < end-i; j++ {
|
||||
word |= uint64(data[i+j]) << (8 * j)
|
||||
}
|
||||
if end-i < 8 {
|
||||
existing, err := s.Peek(addr + uint64(i))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
mask := ^((uint64(1) << (8 * (end - i))) - 1)
|
||||
word = (existing & mask) | word
|
||||
}
|
||||
if err := s.Poke(addr+uint64(i), word); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Step executes a single instruction in the debuggee.
|
||||
func (s *Session) Step() error {
|
||||
if s.exited {
|
||||
return fmt.Errorf("debug: debuggee has exited")
|
||||
}
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_SINGLESTEP),
|
||||
uintptr(s.pid),
|
||||
0, 0, 0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return fmt.Errorf("debug: PTRACE_SINGLESTEP: %w", errno)
|
||||
}
|
||||
_, err := s.waitStopped()
|
||||
return err
|
||||
}
|
||||
|
||||
// Continue resumes execution until the next breakpoint or exit.
|
||||
func (s *Session) Continue() error {
|
||||
if s.exited {
|
||||
return fmt.Errorf("debug: debuggee has exited")
|
||||
}
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_CONT),
|
||||
uintptr(s.pid),
|
||||
0, 0, 0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return fmt.Errorf("debug: PTRACE_CONT: %w", errno)
|
||||
}
|
||||
_, err := s.waitStopped()
|
||||
return err
|
||||
}
|
||||
|
||||
// Exited returns true if the debuggee has terminated.
|
||||
func (s *Session) Exited() bool { return s.exited }
|
||||
|
||||
// Pid returns the debuggee's process ID.
|
||||
func (s *Session) Pid() int { return s.pid }
|
||||
|
||||
// CodeBase returns the base address of the JIT code in the debuggee.
|
||||
func (s *Session) CodeBase() uint64 { return s.codeBase }
|
||||
|
||||
// Kill terminates the debuggee.
|
||||
func (s *Session) Kill() {
|
||||
if !s.exited {
|
||||
syscall.Kill(s.pid, syscall.SIGKILL)
|
||||
syscall.Wait4(s.pid, nil, 0, nil)
|
||||
s.exited = true
|
||||
}
|
||||
if s.cmd != nil && s.cmd.Process != nil {
|
||||
s.cmd.Wait()
|
||||
}
|
||||
}
|
||||
|
||||
// findRWXMapping reads /proc/pid/maps and returns the base address of the
|
||||
// first read-write-execute mapping (the JIT code region).
|
||||
func findRWXMapping(pid int) uint64 {
|
||||
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
for line := range strings.SplitSeq(string(data), "\n") {
|
||||
fields := strings.Fields(line)
|
||||
if len(fields) < 2 {
|
||||
continue
|
||||
}
|
||||
perms := fields[1]
|
||||
if len(perms) >= 3 && perms[0] == 'r' && perms[1] == 'w' && perms[2] == 'x' {
|
||||
var start uint64
|
||||
fmt.Sscanf(fields[0], "%x-", &start)
|
||||
return start
|
||||
}
|
||||
}
|
||||
return 0
|
||||
}
|
||||
@@ -0,0 +1,98 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// GetRegs reads the general-purpose registers of the stopped debuggee.
|
||||
func (s *Session) GetRegs() (Regs, error) {
|
||||
var regs Regs
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_GETREGS),
|
||||
uintptr(s.pid),
|
||||
0,
|
||||
uintptr(unsafe.Pointer(®s)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return regs, fmt.Errorf("debug: PTRACE_GETREGS: %w", errno)
|
||||
}
|
||||
return regs, nil
|
||||
}
|
||||
|
||||
// SetRegs writes the general-purpose registers of the stopped debuggee.
|
||||
func (s *Session) SetRegs(regs *Regs) error {
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_SETREGS),
|
||||
uintptr(s.pid),
|
||||
0,
|
||||
uintptr(unsafe.Pointer(regs)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return fmt.Errorf("debug: PTRACE_SETREGS: %w", errno)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// FPRegs holds the x87 FPU and SSE (XMM) register state from PTRACE_GETFPREGS.
|
||||
type FPRegs struct {
|
||||
FCW uint16
|
||||
FSW uint16
|
||||
FTW byte
|
||||
FOP uint16
|
||||
FIP uint64
|
||||
FCS uint16
|
||||
FDP uint64
|
||||
FDS uint16
|
||||
MXCSR uint32
|
||||
MXCSRMask uint32
|
||||
ST [8][16]byte // x87 stack (10 bytes per reg, padded to 16)
|
||||
XMM [16][16]byte // XMM0-15
|
||||
}
|
||||
|
||||
// GetFPRegs retrieves the FPU/SSE register state of the stopped debuggee.
|
||||
func (s *Session) GetFPRegs() (FPRegs, error) {
|
||||
var fp FPRegs
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_GETFPREGS),
|
||||
uintptr(s.pid),
|
||||
0,
|
||||
uintptr(unsafe.Pointer(&fp)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return fp, fmt.Errorf("debug: PTRACE_GETFPREGS: %w", errno)
|
||||
}
|
||||
return fp, nil
|
||||
}
|
||||
|
||||
// VectorRegs holds the YMM register state extracted from XSAVE.
|
||||
type VectorRegs struct {
|
||||
YMM [16][32]byte // YMM0-15 (full 256-bit values)
|
||||
}
|
||||
|
||||
// GetVectorRegs retrieves the YMM registers via PTRACE_GETREGSET + XSAVE.
|
||||
func (s *Session) GetVectorRegs() (VectorRegs, error) {
|
||||
var v VectorRegs
|
||||
fp, err := s.GetFPRegs()
|
||||
if err != nil {
|
||||
return v, err
|
||||
}
|
||||
for i := range 16 {
|
||||
for j := range 16 {
|
||||
v.YMM[i][j] = fp.XMM[i][j]
|
||||
}
|
||||
}
|
||||
return v, nil
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && arm64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// GetRegs reads the general-purpose registers of the stopped debuggee.
|
||||
func (s *Session) GetRegs() (Regs, error) {
|
||||
var regs Regs
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_GETREGS),
|
||||
uintptr(s.pid),
|
||||
0,
|
||||
uintptr(unsafe.Pointer(®s)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return regs, fmt.Errorf("debug: PTRACE_GETREGS: %w", errno)
|
||||
}
|
||||
return regs, nil
|
||||
}
|
||||
|
||||
// SetRegs writes the general-purpose registers of the stopped debuggee.
|
||||
func (s *Session) SetRegs(regs *Regs) error {
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_SETREGS),
|
||||
uintptr(s.pid),
|
||||
0,
|
||||
uintptr(unsafe.Pointer(regs)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return fmt.Errorf("debug: PTRACE_SETREGS: %w", errno)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ntPrFPREG is the NT_PRFPREG note type (ELF NT ver): the FP register set.
|
||||
const ntPrFPREG = 0x2
|
||||
|
||||
// FPRegs holds the arm64 FP/NEON register state, matching the kernel's
|
||||
// user_fpsimd_struct layout (32 128-bit V registers, then FPSR and FPCR).
|
||||
type FPRegs struct {
|
||||
V [32][16]byte // V0-V31 (128-bit NEON/FP registers)
|
||||
FPSR uint32
|
||||
FPCR uint32
|
||||
}
|
||||
|
||||
// GetFPRegs retrieves the FP register state via PTRACE_GETREGSET with
|
||||
// NT_PRFPREG (this architecture has no PTRACE_GETFPREGS request).
|
||||
func (s *Session) GetFPRegs() (FPRegs, error) {
|
||||
var fp FPRegs
|
||||
iovec := syscall.Iovec{
|
||||
Base: (*byte)(unsafe.Pointer(&fp)),
|
||||
Len: uint64(unsafe.Sizeof(fp)),
|
||||
}
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_GETREGSET),
|
||||
uintptr(s.pid),
|
||||
uintptr(ntPrFPREG),
|
||||
uintptr(unsafe.Pointer(&iovec)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return fp, fmt.Errorf("debug: PTRACE_GETREGSET (NT_PRFPREG): %w", errno)
|
||||
}
|
||||
return fp, nil
|
||||
}
|
||||
|
||||
// VectorRegs holds the full SIMD register state.
|
||||
type VectorRegs struct {
|
||||
V [32][16]byte // V0-V31 (128-bit)
|
||||
}
|
||||
|
||||
// GetVectorRegs retrieves the SIMD registers.
|
||||
func (s *Session) GetVectorRegs() (VectorRegs, error) {
|
||||
var v VectorRegs
|
||||
fp, err := s.GetFPRegs()
|
||||
if err != nil {
|
||||
return v, err
|
||||
}
|
||||
copy(v.V[:][:], fp.V[:][:])
|
||||
return v, nil
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && loong64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// GetRegs reads the general-purpose registers of the stopped debuggee.
|
||||
func (s *Session) GetRegs() (Regs, error) {
|
||||
var regs Regs
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_GETREGS),
|
||||
uintptr(s.pid),
|
||||
0,
|
||||
uintptr(unsafe.Pointer(®s)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return regs, fmt.Errorf("debug: PTRACE_GETREGS: %w", errno)
|
||||
}
|
||||
return regs, nil
|
||||
}
|
||||
|
||||
// SetRegs writes the general-purpose registers of the stopped debuggee.
|
||||
func (s *Session) SetRegs(regs *Regs) error {
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_SETREGS),
|
||||
uintptr(s.pid),
|
||||
0,
|
||||
uintptr(unsafe.Pointer(regs)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return fmt.Errorf("debug: PTRACE_SETREGS: %w", errno)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ntPrFPREG is the NT_PRFPREG note type (ELF NT ver): the FP register set.
|
||||
const ntPrFPREG = 0x2
|
||||
|
||||
// FPRegs holds the LoongArch FP register state, matching the kernel's
|
||||
// user_fp_struct layout (32 64-bit FP registers, the fcc condition flags,
|
||||
// and fcsr).
|
||||
type FPRegs struct {
|
||||
F [32]uint64 // F0-F31 (64-bit FP registers)
|
||||
FCC uint64 // eight per-register condition flags, packed
|
||||
FCSR uint32
|
||||
}
|
||||
|
||||
// GetFPRegs retrieves the FP register state via PTRACE_GETREGSET with
|
||||
// NT_PRFPREG (this architecture has no PTRACE_GETFPREGS request).
|
||||
func (s *Session) GetFPRegs() (FPRegs, error) {
|
||||
var fp FPRegs
|
||||
iovec := syscall.Iovec{
|
||||
Base: (*byte)(unsafe.Pointer(&fp)),
|
||||
Len: uint64(unsafe.Sizeof(fp)),
|
||||
}
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_GETREGSET),
|
||||
uintptr(s.pid),
|
||||
uintptr(ntPrFPREG),
|
||||
uintptr(unsafe.Pointer(&iovec)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return fp, fmt.Errorf("debug: PTRACE_GETREGSET (NT_PRFPREG): %w", errno)
|
||||
}
|
||||
return fp, nil
|
||||
}
|
||||
|
||||
// VectorRegs holds the FP register state shown by the regs command
|
||||
// (the scalar FP subset: 32 64-bit registers, fcc and fcsr; the LSX/LASX
|
||||
// vector files are not read yet).
|
||||
type VectorRegs struct {
|
||||
F [32]uint64
|
||||
FCC uint64
|
||||
FCSR uint32
|
||||
}
|
||||
|
||||
// GetVectorRegs retrieves the FP registers.
|
||||
func (s *Session) GetVectorRegs() (VectorRegs, error) {
|
||||
fp, err := s.GetFPRegs()
|
||||
if err != nil {
|
||||
return VectorRegs{}, err
|
||||
}
|
||||
return VectorRegs{F: fp.F, FCC: fp.FCC, FCSR: fp.FCSR}, nil
|
||||
}
|
||||
@@ -0,0 +1,93 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && riscv64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// GetRegs reads the general-purpose registers of the stopped debuggee.
|
||||
func (s *Session) GetRegs() (Regs, error) {
|
||||
var regs Regs
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_GETREGS),
|
||||
uintptr(s.pid),
|
||||
0,
|
||||
uintptr(unsafe.Pointer(®s)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return regs, fmt.Errorf("debug: PTRACE_GETREGS: %w", errno)
|
||||
}
|
||||
return regs, nil
|
||||
}
|
||||
|
||||
// SetRegs writes the general-purpose registers of the stopped debuggee.
|
||||
func (s *Session) SetRegs(regs *Regs) error {
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_SETREGS),
|
||||
uintptr(s.pid),
|
||||
0,
|
||||
uintptr(unsafe.Pointer(regs)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return fmt.Errorf("debug: PTRACE_SETREGS: %w", errno)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ntPrFPREG is the NT_PRFPREG note type (ELF NT ver): the FP register set.
|
||||
const ntPrFPREG = 0x2
|
||||
|
||||
// FPRegs holds the RISC-V FP register state, matching the kernel's
|
||||
// user_fp_struct layout (32 64-bit FP registers plus fcsr).
|
||||
type FPRegs struct {
|
||||
F [32]uint64 // F0-F31 (64-bit FP registers)
|
||||
FCSR uint32
|
||||
}
|
||||
|
||||
// GetFPRegs retrieves the FP register state via PTRACE_GETREGSET with
|
||||
// NT_PRFPREG (this architecture has no PTRACE_GETFPREGS request).
|
||||
func (s *Session) GetFPRegs() (FPRegs, error) {
|
||||
var fp FPRegs
|
||||
iovec := syscall.Iovec{
|
||||
Base: (*byte)(unsafe.Pointer(&fp)),
|
||||
Len: uint64(unsafe.Sizeof(fp)),
|
||||
}
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_GETREGSET),
|
||||
uintptr(s.pid),
|
||||
uintptr(ntPrFPREG),
|
||||
uintptr(unsafe.Pointer(&iovec)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return fp, fmt.Errorf("debug: PTRACE_GETREGSET (NT_PRFPREG): %w", errno)
|
||||
}
|
||||
return fp, nil
|
||||
}
|
||||
|
||||
// VectorRegs holds the FP register state shown by the regs command
|
||||
// (riscv64 has 32 64-bit FP registers and fcsr).
|
||||
type VectorRegs struct {
|
||||
F [32]uint64
|
||||
FCSR uint32
|
||||
}
|
||||
|
||||
// GetVectorRegs retrieves the FP registers.
|
||||
func (s *Session) GetVectorRegs() (VectorRegs, error) {
|
||||
fp, err := s.GetFPRegs()
|
||||
if err != nil {
|
||||
return VectorRegs{}, err
|
||||
}
|
||||
return VectorRegs{F: fp.F, FCSR: fp.FCSR}, nil
|
||||
}
|
||||
@@ -0,0 +1,95 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package debug
|
||||
|
||||
// Regs holds the full general-purpose register set of a traced process
|
||||
// (the Linux amd64 user_regs_struct layout).
|
||||
type Regs struct {
|
||||
R15 uint64
|
||||
R14 uint64
|
||||
R13 uint64
|
||||
R12 uint64
|
||||
RBP uint64
|
||||
RBX uint64
|
||||
R11 uint64
|
||||
R10 uint64
|
||||
R9 uint64
|
||||
R8 uint64
|
||||
RAX uint64
|
||||
RCX uint64
|
||||
RDX uint64
|
||||
RSI uint64
|
||||
RDI uint64
|
||||
OrigRAX uint64
|
||||
RIP uint64
|
||||
CS uint64
|
||||
RFLAGS uint64
|
||||
RSP uint64
|
||||
SS uint64
|
||||
FSBase uint64
|
||||
GSBase uint64
|
||||
DS uint64
|
||||
ES uint64
|
||||
FS uint64
|
||||
GS uint64
|
||||
}
|
||||
|
||||
// GetPC returns the program counter.
|
||||
func (r *Regs) GetPC() uint64 { return r.RIP }
|
||||
|
||||
// SetPC sets the program counter.
|
||||
func (r *Regs) SetPC(pc uint64) { r.RIP = pc }
|
||||
|
||||
// GetSP returns the stack pointer.
|
||||
func (r *Regs) GetSP() uint64 { return r.RSP }
|
||||
|
||||
// RegValue returns the value of the named register, or false if unknown.
|
||||
func (r *Regs) RegValue(name string) (uint64, bool) {
|
||||
switch name {
|
||||
case "rax", "eax", "ax", "al":
|
||||
return r.RAX, true
|
||||
case "rbx", "ebx", "bx", "bl":
|
||||
return r.RBX, true
|
||||
case "rcx", "ecx", "cx", "cl":
|
||||
return r.RCX, true
|
||||
case "rdx", "edx", "dx", "dl":
|
||||
return r.RDX, true
|
||||
case "rsi", "esi", "si":
|
||||
return r.RSI, true
|
||||
case "rdi", "edi", "di":
|
||||
return r.RDI, true
|
||||
case "rbp", "ebp", "bp":
|
||||
return r.RBP, true
|
||||
case "rsp", "esp", "sp":
|
||||
return r.RSP, true
|
||||
case "r8":
|
||||
return r.R8, true
|
||||
case "r9":
|
||||
return r.R9, true
|
||||
case "r10":
|
||||
return r.R10, true
|
||||
case "r11":
|
||||
return r.R11, true
|
||||
case "r12":
|
||||
return r.R12, true
|
||||
case "r13":
|
||||
return r.R13, true
|
||||
case "r14":
|
||||
return r.R14, true
|
||||
case "r15":
|
||||
return r.R15, true
|
||||
case "rip", "eip":
|
||||
return r.RIP, true
|
||||
default:
|
||||
return 0, false
|
||||
}
|
||||
}
|
||||
|
||||
// breakpointInsn is the software breakpoint instruction.
|
||||
var breakpointInsn = []byte{0xCC} // INT3
|
||||
|
||||
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
|
||||
const breakpointPCAdjust = 1
|
||||
@@ -0,0 +1,134 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && arm64
|
||||
|
||||
package debug
|
||||
|
||||
// Regs holds the full general-purpose register set of a traced process
|
||||
// (the Linux arm64 user_pt_regs layout).
|
||||
type Regs struct {
|
||||
X0 uint64
|
||||
X1 uint64
|
||||
X2 uint64
|
||||
X3 uint64
|
||||
X4 uint64
|
||||
X5 uint64
|
||||
X6 uint64
|
||||
X7 uint64
|
||||
X8 uint64
|
||||
X9 uint64
|
||||
X10 uint64
|
||||
X11 uint64
|
||||
X12 uint64
|
||||
X13 uint64
|
||||
X14 uint64
|
||||
X15 uint64
|
||||
X16 uint64
|
||||
X17 uint64
|
||||
X18 uint64
|
||||
X19 uint64
|
||||
X20 uint64
|
||||
X21 uint64
|
||||
X22 uint64
|
||||
X23 uint64
|
||||
X24 uint64
|
||||
X25 uint64
|
||||
X26 uint64
|
||||
X27 uint64
|
||||
X28 uint64
|
||||
X29 uint64 // FP (frame pointer)
|
||||
X30 uint64 // LR (link register)
|
||||
SP uint64
|
||||
PC uint64
|
||||
PSTATE uint64
|
||||
}
|
||||
|
||||
// PC returns the program counter.
|
||||
func (r *Regs) GetPC() uint64 { return r.PC }
|
||||
|
||||
// SetPC sets the program counter.
|
||||
func (r *Regs) SetPC(pc uint64) { r.PC = pc }
|
||||
|
||||
// GetSP returns the stack pointer.
|
||||
func (r *Regs) GetSP() uint64 { return r.SP }
|
||||
|
||||
// RegValue returns the value of the named register, or false if unknown.
|
||||
func (r *Regs) RegValue(name string) (uint64, bool) {
|
||||
switch name {
|
||||
case "x0":
|
||||
return r.X0, true
|
||||
case "x1":
|
||||
return r.X1, true
|
||||
case "x2":
|
||||
return r.X2, true
|
||||
case "x3":
|
||||
return r.X3, true
|
||||
case "x4":
|
||||
return r.X4, true
|
||||
case "x5":
|
||||
return r.X5, true
|
||||
case "x6":
|
||||
return r.X6, true
|
||||
case "x7":
|
||||
return r.X7, true
|
||||
case "x8":
|
||||
return r.X8, true
|
||||
case "x9":
|
||||
return r.X9, true
|
||||
case "x10":
|
||||
return r.X10, true
|
||||
case "x11":
|
||||
return r.X11, true
|
||||
case "x12":
|
||||
return r.X12, true
|
||||
case "x13":
|
||||
return r.X13, true
|
||||
case "x14":
|
||||
return r.X14, true
|
||||
case "x15":
|
||||
return r.X15, true
|
||||
case "x16":
|
||||
return r.X16, true
|
||||
case "x17":
|
||||
return r.X17, true
|
||||
case "x18":
|
||||
return r.X18, true
|
||||
case "x19":
|
||||
return r.X19, true
|
||||
case "x20":
|
||||
return r.X20, true
|
||||
case "x21":
|
||||
return r.X21, true
|
||||
case "x22":
|
||||
return r.X22, true
|
||||
case "x23":
|
||||
return r.X23, true
|
||||
case "x24":
|
||||
return r.X24, true
|
||||
case "x25":
|
||||
return r.X25, true
|
||||
case "x26":
|
||||
return r.X26, true
|
||||
case "x27":
|
||||
return r.X27, true
|
||||
case "x28":
|
||||
return r.X28, true
|
||||
case "x29", "fp":
|
||||
return r.X29, true
|
||||
case "x30", "lr":
|
||||
return r.X30, true
|
||||
case "sp":
|
||||
return r.SP, true
|
||||
case "pc":
|
||||
return r.PC, true
|
||||
default:
|
||||
return 0, false
|
||||
}
|
||||
}
|
||||
|
||||
// breakpointInsn is the software breakpoint instruction (BRK #0).
|
||||
var breakpointInsn = []byte{0x00, 0x00, 0x20, 0xD4} // BRK #0
|
||||
|
||||
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
|
||||
const breakpointPCAdjust = 4
|
||||
@@ -0,0 +1,130 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && loong64
|
||||
|
||||
package debug
|
||||
|
||||
// Regs holds the full general-purpose register set of a traced process
|
||||
// (the Linux loong64 user_pt_regs layout).
|
||||
type Regs struct {
|
||||
R0 uint64 // zero
|
||||
R1 uint64 // RA (return address)
|
||||
R2 uint64 // TP (thread pointer)
|
||||
R3 uint64 // SP (stack pointer)
|
||||
R4 uint64 // A0
|
||||
R5 uint64 // A1
|
||||
R6 uint64 // A2
|
||||
R7 uint64 // A3
|
||||
R8 uint64 // A4
|
||||
R9 uint64 // A5
|
||||
R10 uint64 // A6
|
||||
R11 uint64 // A7
|
||||
R12 uint64 // T0
|
||||
R13 uint64 // T1
|
||||
R14 uint64 // T2
|
||||
R15 uint64 // T3
|
||||
R16 uint64 // T4
|
||||
R17 uint64 // T5
|
||||
R18 uint64 // T6
|
||||
R19 uint64 // T7
|
||||
R20 uint64 // T8
|
||||
R21 uint64 // FP (frame pointer)
|
||||
R22 uint64 // S0
|
||||
R23 uint64 // S1
|
||||
R24 uint64 // S2
|
||||
R25 uint64 // S3
|
||||
R26 uint64 // S4
|
||||
R27 uint64 // S5
|
||||
R28 uint64 // S6
|
||||
R29 uint64 // S7
|
||||
R30 uint64 // S8
|
||||
R31 uint64 // PC
|
||||
}
|
||||
|
||||
// GetPC returns the program counter.
|
||||
func (r *Regs) GetPC() uint64 { return r.R31 }
|
||||
|
||||
// SetPC sets the program counter.
|
||||
func (r *Regs) SetPC(pc uint64) { r.R31 = pc }
|
||||
|
||||
// GetSP returns the stack pointer.
|
||||
func (r *Regs) GetSP() uint64 { return r.R3 }
|
||||
|
||||
// RegValue returns the value of the named register, or false if unknown.
|
||||
func (r *Regs) RegValue(name string) (uint64, bool) {
|
||||
switch name {
|
||||
case "r0", "zero":
|
||||
return r.R0, true
|
||||
case "r1", "ra":
|
||||
return r.R1, true
|
||||
case "r2", "tp":
|
||||
return r.R2, true
|
||||
case "r3", "sp":
|
||||
return r.R3, true
|
||||
case "r4", "a0":
|
||||
return r.R4, true
|
||||
case "r5", "a1":
|
||||
return r.R5, true
|
||||
case "r6", "a2":
|
||||
return r.R6, true
|
||||
case "r7", "a3":
|
||||
return r.R7, true
|
||||
case "r8", "a4":
|
||||
return r.R8, true
|
||||
case "r9", "a5":
|
||||
return r.R9, true
|
||||
case "r10", "a6":
|
||||
return r.R10, true
|
||||
case "r11", "a7":
|
||||
return r.R11, true
|
||||
case "r12", "t0":
|
||||
return r.R12, true
|
||||
case "r13", "t1":
|
||||
return r.R13, true
|
||||
case "r14", "t2":
|
||||
return r.R14, true
|
||||
case "r15", "t3":
|
||||
return r.R15, true
|
||||
case "r16", "t4":
|
||||
return r.R16, true
|
||||
case "r17", "t5":
|
||||
return r.R17, true
|
||||
case "r18", "t6":
|
||||
return r.R18, true
|
||||
case "r19", "t7":
|
||||
return r.R19, true
|
||||
case "r20", "t8":
|
||||
return r.R20, true
|
||||
case "r21", "fp":
|
||||
return r.R21, true
|
||||
case "r22", "s0":
|
||||
return r.R22, true
|
||||
case "r23", "s1":
|
||||
return r.R23, true
|
||||
case "r24", "s2":
|
||||
return r.R24, true
|
||||
case "r25", "s3":
|
||||
return r.R25, true
|
||||
case "r26", "s4":
|
||||
return r.R26, true
|
||||
case "r27", "s5":
|
||||
return r.R27, true
|
||||
case "r28", "s6":
|
||||
return r.R28, true
|
||||
case "r29", "s7":
|
||||
return r.R29, true
|
||||
case "r30", "s8":
|
||||
return r.R30, true
|
||||
case "r31", "pc":
|
||||
return r.R31, true
|
||||
default:
|
||||
return 0, false
|
||||
}
|
||||
}
|
||||
|
||||
// breakpointInsn is the software breakpoint instruction (BRK $0).
|
||||
var breakpointInsn = []byte{0x05, 0x00, 0x2a, 0x00} // break 0
|
||||
|
||||
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
|
||||
const breakpointPCAdjust = 4
|
||||
@@ -0,0 +1,130 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && riscv64
|
||||
|
||||
package debug
|
||||
|
||||
// Regs holds the full general-purpose register set of a traced process
|
||||
// (the Linux riscv64 user_regs_struct layout).
|
||||
type Regs struct {
|
||||
PC uint64
|
||||
Ra uint64 // x1 (return address)
|
||||
Sp uint64 // x2
|
||||
Gp uint64 // x3
|
||||
Tp uint64 // x4
|
||||
T0 uint64 // x5
|
||||
T1 uint64 // x6
|
||||
T2 uint64 // x7
|
||||
S0 uint64 // x8 (frame pointer)
|
||||
S1 uint64 // x9
|
||||
A0 uint64 // x10
|
||||
A1 uint64 // x11
|
||||
A2 uint64 // x12
|
||||
A3 uint64 // x13
|
||||
A4 uint64 // x14
|
||||
A5 uint64 // x15
|
||||
A6 uint64 // x16
|
||||
A7 uint64 // x17
|
||||
S2 uint64 // x18
|
||||
S3 uint64 // x19
|
||||
S4 uint64 // x20
|
||||
S5 uint64 // x21
|
||||
S6 uint64 // x22
|
||||
S7 uint64 // x23
|
||||
S8 uint64 // x24
|
||||
S9 uint64 // x25
|
||||
S10 uint64 // x26
|
||||
S11 uint64 // x27
|
||||
T3 uint64 // x28
|
||||
T4 uint64 // x29
|
||||
T5 uint64 // x30
|
||||
T6 uint64 // x31
|
||||
}
|
||||
|
||||
// GetPC returns the program counter.
|
||||
func (r *Regs) GetPC() uint64 { return r.PC }
|
||||
|
||||
// SetPC sets the program counter.
|
||||
func (r *Regs) SetPC(pc uint64) { r.PC = pc }
|
||||
|
||||
// GetSP returns the stack pointer.
|
||||
func (r *Regs) GetSP() uint64 { return r.Sp }
|
||||
|
||||
// RegValue returns the value of the named register, or false if unknown.
|
||||
func (r *Regs) RegValue(name string) (uint64, bool) {
|
||||
switch name {
|
||||
case "pc":
|
||||
return r.PC, true
|
||||
case "ra", "x1":
|
||||
return r.Ra, true
|
||||
case "sp", "x2":
|
||||
return r.Sp, true
|
||||
case "gp", "x3":
|
||||
return r.Gp, true
|
||||
case "tp", "x4":
|
||||
return r.Tp, true
|
||||
case "t0", "x5":
|
||||
return r.T0, true
|
||||
case "t1", "x6":
|
||||
return r.T1, true
|
||||
case "t2", "x7":
|
||||
return r.T2, true
|
||||
case "s0", "fp", "x8":
|
||||
return r.S0, true
|
||||
case "s1", "x9":
|
||||
return r.S1, true
|
||||
case "a0", "x10":
|
||||
return r.A0, true
|
||||
case "a1", "x11":
|
||||
return r.A1, true
|
||||
case "a2", "x12":
|
||||
return r.A2, true
|
||||
case "a3", "x13":
|
||||
return r.A3, true
|
||||
case "a4", "x14":
|
||||
return r.A4, true
|
||||
case "a5", "x15":
|
||||
return r.A5, true
|
||||
case "a6", "x16":
|
||||
return r.A6, true
|
||||
case "a7", "x17":
|
||||
return r.A7, true
|
||||
case "s2", "x18":
|
||||
return r.S2, true
|
||||
case "s3", "x19":
|
||||
return r.S3, true
|
||||
case "s4", "x20":
|
||||
return r.S4, true
|
||||
case "s5", "x21":
|
||||
return r.S5, true
|
||||
case "s6", "x22":
|
||||
return r.S6, true
|
||||
case "s7", "x23":
|
||||
return r.S7, true
|
||||
case "s8", "x24":
|
||||
return r.S8, true
|
||||
case "s9", "x25":
|
||||
return r.S9, true
|
||||
case "s10", "x26":
|
||||
return r.S10, true
|
||||
case "s11", "x27":
|
||||
return r.S11, true
|
||||
case "t3", "x28":
|
||||
return r.T3, true
|
||||
case "t4", "x29":
|
||||
return r.T4, true
|
||||
case "t5", "x30":
|
||||
return r.T5, true
|
||||
case "t6", "x31":
|
||||
return r.T6, true
|
||||
default:
|
||||
return 0, false
|
||||
}
|
||||
}
|
||||
|
||||
// breakpointInsn is the software breakpoint instruction (EBREAK).
|
||||
var breakpointInsn = []byte{0x73, 0x00, 0x10, 0x00} // ebreak
|
||||
|
||||
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
|
||||
const breakpointPCAdjust = 4
|
||||
+572
@@ -0,0 +1,572 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"fmt"
|
||||
"io"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Label is a named address within the debugged function.
|
||||
type Label struct {
|
||||
Name string
|
||||
Offset int // function-relative offset
|
||||
}
|
||||
|
||||
// SourceLine maps a byte offset to a source line number.
|
||||
type SourceLine struct {
|
||||
Offset int
|
||||
Line int
|
||||
}
|
||||
|
||||
// REPL runs the interactive debugger loop, reading commands from in (pass
|
||||
// os.Stdin interactively, or a bytes.Reader/script file for headless runs).
|
||||
func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, argsSize int, labels []Label, lines []SourceLine, in io.Reader) {
|
||||
entryAddr := codeBase + uint64(funcOffset)
|
||||
|
||||
fmt.Printf("stopped at function entry: %#x (%d bytes)\n", entryAddr, funcSize)
|
||||
fmt.Println("commands: break <label|addr> | step [n] | continue | disas [n] | regs | where | x <addr> [len] | w <addr> <val...> | labels | quit")
|
||||
|
||||
scanner := bufio.NewScanner(in)
|
||||
|
||||
for {
|
||||
fmt.Print("(gasm) ")
|
||||
if !scanner.Scan() {
|
||||
break
|
||||
}
|
||||
line := strings.TrimSpace(scanner.Text())
|
||||
if line == "" {
|
||||
continue
|
||||
}
|
||||
parts := strings.Fields(line)
|
||||
cmd := parts[0]
|
||||
|
||||
switch cmd {
|
||||
case "q", "quit":
|
||||
s.Kill()
|
||||
return
|
||||
|
||||
case "regs":
|
||||
regs, err := s.GetRegs()
|
||||
if err != nil {
|
||||
fmt.Println(err)
|
||||
continue
|
||||
}
|
||||
printRegs(®s, codeBase, uint64(funcOffset))
|
||||
vregs, err := s.GetVectorRegs()
|
||||
if err != nil {
|
||||
fmt.Printf(" (vector regs unavailable: %v)\n", err)
|
||||
} else {
|
||||
printVectorRegs(&vregs)
|
||||
}
|
||||
|
||||
case "step", "s":
|
||||
n := 1
|
||||
if len(parts) > 1 {
|
||||
n, _ = strconv.Atoi(parts[1])
|
||||
}
|
||||
for range n {
|
||||
if s.Exited() {
|
||||
fmt.Println("debuggee exited")
|
||||
break
|
||||
}
|
||||
if err := s.Step(); err != nil {
|
||||
fmt.Println(err)
|
||||
break
|
||||
}
|
||||
}
|
||||
if !s.Exited() {
|
||||
regs, _ := s.GetRegs()
|
||||
pc := regs.GetPC()
|
||||
text, _, _ := s.Disassemble(pc)
|
||||
fmt.Printf("=> %#x (func+%#x): %s\n", pc, pc-codeBase-uint64(funcOffset), text)
|
||||
}
|
||||
|
||||
case "next", "n":
|
||||
regs, _ := s.GetRegs()
|
||||
pc := regs.GetPC()
|
||||
text, instLen, _ := s.Disassemble(pc)
|
||||
if strings.HasPrefix(strings.ToLower(text), "call") || strings.HasPrefix(strings.ToLower(text), "bl") {
|
||||
afterAddr := pc + uint64(instLen)
|
||||
_, err := bm.Set(afterAddr, "(next)")
|
||||
if err != nil {
|
||||
fmt.Printf("cannot set next breakpoint: %v\n", err)
|
||||
continue
|
||||
}
|
||||
for _, b := range bm.All() {
|
||||
bm.Reinsert(b.Addr)
|
||||
}
|
||||
if err := s.Continue(); err != nil {
|
||||
fmt.Println(err)
|
||||
bm.Clear(afterAddr)
|
||||
continue
|
||||
}
|
||||
bm.HandleTrap(®s)
|
||||
bm.Clear(afterAddr)
|
||||
} else {
|
||||
if err := s.Step(); err != nil {
|
||||
fmt.Println(err)
|
||||
continue
|
||||
}
|
||||
}
|
||||
if !s.Exited() {
|
||||
regs, _ := s.GetRegs()
|
||||
pc := regs.GetPC()
|
||||
text, _, _ := s.Disassemble(pc)
|
||||
fmt.Printf("=> %#x (func+%#x): %s\n", pc, pc-codeBase-uint64(funcOffset), text)
|
||||
}
|
||||
|
||||
case "finish", "fin":
|
||||
regs, _ := s.GetRegs()
|
||||
retAddr, err := archReturnAddr(s, ®s)
|
||||
if err != nil {
|
||||
fmt.Printf("cannot read return address: %v\n", err)
|
||||
continue
|
||||
}
|
||||
_, err = bm.Set(retAddr, "(finish)")
|
||||
if err != nil {
|
||||
fmt.Printf("cannot set finish breakpoint: %v\n", err)
|
||||
continue
|
||||
}
|
||||
for _, b := range bm.All() {
|
||||
bm.Reinsert(b.Addr)
|
||||
}
|
||||
if err := s.Continue(); err != nil {
|
||||
fmt.Println(err)
|
||||
bm.Clear(retAddr)
|
||||
continue
|
||||
}
|
||||
if !s.Exited() {
|
||||
bm.HandleTrap(®s)
|
||||
}
|
||||
bm.Clear(retAddr)
|
||||
if s.Exited() {
|
||||
fmt.Println("debuggee exited")
|
||||
} else {
|
||||
regs, _ := s.GetRegs()
|
||||
fmt.Printf("finished, now at %#x\n", regs.GetPC())
|
||||
}
|
||||
|
||||
case "continue", "c":
|
||||
if s.Exited() {
|
||||
fmt.Println("debuggee exited")
|
||||
continue
|
||||
}
|
||||
for {
|
||||
for _, bp := range bm.All() {
|
||||
bm.Reinsert(bp.Addr)
|
||||
}
|
||||
if err := s.Continue(); err != nil {
|
||||
fmt.Println(err)
|
||||
break
|
||||
}
|
||||
if s.Exited() {
|
||||
fmt.Println("debuggee exited")
|
||||
break
|
||||
}
|
||||
reason, wpAddr := s.StopInfo()
|
||||
if reason == StopWatchpoint {
|
||||
fmt.Printf("watchpoint hit at %#x\n", wpAddr)
|
||||
break
|
||||
}
|
||||
regs, _ := s.GetRegs()
|
||||
if bp := bm.HandleTrap(®s); bp != nil {
|
||||
// Execute the instruction under the restored breakpoint
|
||||
// so the next continue cannot re-trap on the same
|
||||
// breakpoint; the process parks right after it.
|
||||
if err := s.Step(); err != nil {
|
||||
fmt.Println(err)
|
||||
break
|
||||
}
|
||||
name := bp.Label
|
||||
if name == "" {
|
||||
name = fmt.Sprintf("%#x", bp.Addr)
|
||||
}
|
||||
fmt.Printf("breakpoint hit: %s (func+%#x)\n", name, bp.Addr-codeBase-uint64(funcOffset))
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
case "break", "b":
|
||||
if len(parts) < 2 {
|
||||
fmt.Println("usage: break <label|addr|line> [if <reg> <op> <val>]")
|
||||
continue
|
||||
}
|
||||
var addr uint64
|
||||
var label string
|
||||
if lineNum, err := strconv.Atoi(parts[1]); err == nil && lineNum > 0 {
|
||||
off := offsetForLine(lines, lineNum)
|
||||
if off < 0 {
|
||||
fmt.Printf("no instruction at line %d\n", lineNum)
|
||||
continue
|
||||
}
|
||||
addr = codeBase + uint64(funcOffset) + uint64(off)
|
||||
label = fmt.Sprintf("line %d", lineNum)
|
||||
} else {
|
||||
addr, label = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||
}
|
||||
if addr == 0 {
|
||||
fmt.Printf("unknown label, address, or line: %s\n", parts[1])
|
||||
continue
|
||||
}
|
||||
var cond *Condition
|
||||
if len(parts) >= 6 && parts[2] == "if" {
|
||||
reg := strings.ToLower(parts[3])
|
||||
op := parts[4]
|
||||
operand := parts[5]
|
||||
if val, err := strconv.ParseUint(operand, 0, 64); err == nil {
|
||||
cond = &Condition{Reg: reg, Op: op, Value: val}
|
||||
} else {
|
||||
cond = &Condition{Reg: reg, Op: op, Reg2: strings.ToLower(operand)}
|
||||
}
|
||||
} else if len(parts) >= 4 && parts[2] == "if" {
|
||||
fmt.Println("usage: break <label|addr> if <reg> <op> <value|reg>")
|
||||
continue
|
||||
}
|
||||
bp, err := bm.SetWithCond(addr, label, cond)
|
||||
if err != nil {
|
||||
fmt.Println(err)
|
||||
continue
|
||||
}
|
||||
condStr := ""
|
||||
if cond != nil {
|
||||
condStr = fmt.Sprintf(" if %s %s %#x", cond.Reg, cond.Op, cond.Value)
|
||||
}
|
||||
fmt.Printf("breakpoint set: %s at %#x (func+%#x)%s\n", bp.Label, bp.Addr, bp.Addr-codeBase-uint64(funcOffset), condStr)
|
||||
|
||||
case "info":
|
||||
if len(parts) < 2 {
|
||||
fmt.Println("usage: info break")
|
||||
continue
|
||||
}
|
||||
switch parts[1] {
|
||||
case "break", "breakpoints", "b":
|
||||
fmt.Print(bm.Info())
|
||||
default:
|
||||
fmt.Printf("unknown info target: %s\n", parts[1])
|
||||
}
|
||||
|
||||
case "delete", "d":
|
||||
if len(parts) < 2 {
|
||||
fmt.Println("usage: delete <label|addr>")
|
||||
continue
|
||||
}
|
||||
addr, _ := resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||
if addr == 0 {
|
||||
fmt.Printf("unknown: %s\n", parts[1])
|
||||
continue
|
||||
}
|
||||
if err := bm.Clear(addr); err != nil {
|
||||
fmt.Println(err)
|
||||
} else {
|
||||
fmt.Println("breakpoint removed")
|
||||
}
|
||||
|
||||
case "x":
|
||||
regs, _ := s.GetRegs()
|
||||
addr := regs.GetPC()
|
||||
length := 64
|
||||
if len(parts) > 1 {
|
||||
addr, _ = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||
}
|
||||
if len(parts) > 2 {
|
||||
length, _ = strconv.Atoi(parts[2])
|
||||
}
|
||||
mem, err := s.ReadMemory(addr, length)
|
||||
if err != nil {
|
||||
fmt.Println(err)
|
||||
continue
|
||||
}
|
||||
hexDump(addr, mem)
|
||||
|
||||
case "w":
|
||||
if len(parts) < 3 {
|
||||
fmt.Println("usage: w <addr> <byte|0x...> [byte...]")
|
||||
continue
|
||||
}
|
||||
addr, _ := resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||
if addr == 0 {
|
||||
fmt.Printf("unknown address: %s\n", parts[1])
|
||||
continue
|
||||
}
|
||||
var bytes []byte
|
||||
for _, arg := range parts[2:] {
|
||||
v, err := strconv.ParseUint(arg, 0, 64)
|
||||
if err != nil {
|
||||
fmt.Printf("invalid value: %s\n", arg)
|
||||
continue
|
||||
}
|
||||
if v > 255 {
|
||||
for j := range 8 {
|
||||
bytes = append(bytes, byte(v>>(8*j)))
|
||||
}
|
||||
} else {
|
||||
bytes = append(bytes, byte(v))
|
||||
}
|
||||
}
|
||||
if len(bytes) > 0 {
|
||||
if err := s.WriteMemory(addr, bytes); err != nil {
|
||||
fmt.Println(err)
|
||||
} else {
|
||||
fmt.Printf("wrote %d bytes at %#x\n", len(bytes), addr)
|
||||
}
|
||||
}
|
||||
|
||||
case "set":
|
||||
if len(parts) < 3 {
|
||||
fmt.Println("usage: set <reg> <value>")
|
||||
continue
|
||||
}
|
||||
val, err := strconv.ParseUint(parts[2], 0, 64)
|
||||
if err != nil {
|
||||
fmt.Printf("invalid value: %s\n", parts[2])
|
||||
continue
|
||||
}
|
||||
if err := s.SetReg(strings.ToLower(parts[1]), val); err != nil {
|
||||
fmt.Printf("set: %v\n", err)
|
||||
} else {
|
||||
fmt.Printf("%s = %#x\n", parts[1], val)
|
||||
}
|
||||
|
||||
case "labels", "l":
|
||||
sorted := make([]Label, len(labels))
|
||||
copy(sorted, labels)
|
||||
sort.Slice(sorted, func(i, j int) bool { return sorted[i].Offset < sorted[j].Offset })
|
||||
for _, l := range sorted {
|
||||
fmt.Printf(" func+%#04x %s\n", l.Offset, l.Name)
|
||||
}
|
||||
|
||||
case "disas", "u":
|
||||
n := 5
|
||||
if len(parts) > 1 {
|
||||
n, _ = strconv.Atoi(parts[1])
|
||||
if n <= 0 {
|
||||
n = 5
|
||||
}
|
||||
}
|
||||
regs, _ := s.GetRegs()
|
||||
fmt.Print(s.DisassembleN(regs.GetPC(), n))
|
||||
|
||||
case "where":
|
||||
regs, _ := s.GetRegs()
|
||||
funcOff := int(regs.GetPC() - codeBase - uint64(funcOffset))
|
||||
line := lineAt(lines, funcOff)
|
||||
label := nearestLabel(labels, funcOff)
|
||||
fmt.Printf(" func+%#x", funcOff)
|
||||
if label != "" {
|
||||
fmt.Printf(" (near %s)", label)
|
||||
}
|
||||
if line > 0 {
|
||||
fmt.Printf(" line %d", line)
|
||||
}
|
||||
fmt.Println()
|
||||
|
||||
case "help", "h", "?":
|
||||
fmt.Printf(` break <label|addr> [if <reg> <op> <val>] set a breakpoint
|
||||
delete <label|addr> remove a breakpoint
|
||||
info break list all breakpoints
|
||||
watch <addr> [r|w] [size] set a hardware watchpoint (write by default)
|
||||
unwatch [<slot>] clear one or all watchpoints
|
||||
step [n], s single-step n instructions
|
||||
next, n step over CALL/BL
|
||||
continue, c run until breakpoint or exit
|
||||
disas [n], u disassemble n instructions at PC
|
||||
regs print registers
|
||||
where show source line and nearest label
|
||||
stack show stack near %s (args + return address)
|
||||
x [addr] [len] hex-dump memory
|
||||
w <addr> <val...> write bytes to memory
|
||||
labels, l list function labels
|
||||
help, h, ? this help
|
||||
quit, q kill debuggee and exit`, archSPLabel())
|
||||
|
||||
case "stack":
|
||||
regs, _ := s.GetRegs()
|
||||
sp := regs.GetSP()
|
||||
retAddr, _ := archReturnAddr(s, ®s)
|
||||
fmt.Printf(" [%s] return addr = %#x\n", archSPLabel(), retAddr)
|
||||
if argsSize > 0 {
|
||||
fmt.Printf(" args (%d bytes at %s+8):\n", argsSize, archSPLabel())
|
||||
argBytes, err := s.ReadMemory(sp+8, argsSize)
|
||||
if err == nil {
|
||||
for i := 0; i < argsSize; i += 8 {
|
||||
var v uint64
|
||||
for j := 0; j < 8 && i+j < len(argBytes); j++ {
|
||||
v |= uint64(argBytes[i+j]) << (8 * j)
|
||||
}
|
||||
fmt.Printf(" [%+3d] %#016x\n", i+8, v)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
case "bt", "backtrace":
|
||||
regs, _ := s.GetRegs()
|
||||
funcOff := int(regs.GetPC() - codeBase - uint64(funcOffset))
|
||||
line := lineAt(lines, funcOff)
|
||||
label := nearestLabel(labels, funcOff)
|
||||
fmt.Printf(" #0 func+%#x", funcOff)
|
||||
if label != "" {
|
||||
fmt.Printf(" (%s)", label)
|
||||
}
|
||||
if line > 0 {
|
||||
fmt.Printf(" [line %d]", line)
|
||||
}
|
||||
fmt.Println()
|
||||
retAddr, _ := archReturnAddr(s, ®s)
|
||||
fmt.Printf(" #1 return to %#x\n", retAddr)
|
||||
|
||||
case "watch":
|
||||
if len(parts) < 2 {
|
||||
fmt.Println("usage: watch <addr> [r|w] [size]")
|
||||
continue
|
||||
}
|
||||
addr, _ := resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||
if addr == 0 {
|
||||
fmt.Printf("unknown address: %s\n", parts[1])
|
||||
continue
|
||||
}
|
||||
typ := WatchWrite
|
||||
size := 8
|
||||
if len(parts) > 2 {
|
||||
switch parts[2] {
|
||||
case "r":
|
||||
typ = WatchRead
|
||||
case "w":
|
||||
typ = WatchWrite
|
||||
}
|
||||
}
|
||||
if len(parts) > 3 {
|
||||
size, _ = strconv.Atoi(parts[3])
|
||||
}
|
||||
slot := s.FindFreeWatchpointSlot()
|
||||
if slot < 0 {
|
||||
fmt.Println("no free watchpoint slots (use 'unwatch <slot>' to clear one)")
|
||||
continue
|
||||
}
|
||||
if err := s.SetWatchpoint(slot, addr, typ, size); err != nil {
|
||||
fmt.Printf("watch: %v\n", err)
|
||||
} else {
|
||||
typStr := "w"
|
||||
if typ == WatchRead {
|
||||
typStr = "r"
|
||||
}
|
||||
fmt.Printf("watchpoint %d set: %#x (%s, %d bytes)\n", slot, addr, typStr, size)
|
||||
}
|
||||
|
||||
case "unwatch":
|
||||
if len(parts) >= 2 {
|
||||
slot, err := strconv.Atoi(parts[1])
|
||||
if err != nil || slot < 0 || slot > 3 {
|
||||
fmt.Println("usage: unwatch [<slot>]")
|
||||
continue
|
||||
}
|
||||
if err := s.ClearWatchpoint(slot); err != nil {
|
||||
fmt.Printf("unwatch: %v\n", err)
|
||||
} else {
|
||||
fmt.Printf("watchpoint %d cleared\n", slot)
|
||||
}
|
||||
} else {
|
||||
if err := s.ClearAllWatchpoints(); err != nil {
|
||||
fmt.Printf("unwatch: %v\n", err)
|
||||
} else {
|
||||
fmt.Println("all watchpoints cleared")
|
||||
}
|
||||
}
|
||||
|
||||
default:
|
||||
fmt.Printf("unknown command: %s\n", cmd)
|
||||
}
|
||||
}
|
||||
s.Kill()
|
||||
}
|
||||
|
||||
func hexDump(addr uint64, data []byte) {
|
||||
for i := 0; i < len(data); i += 16 {
|
||||
end := min(i+16, len(data))
|
||||
fmt.Printf(" %#08x:", addr+uint64(i))
|
||||
for j := i; j < i+16; j++ {
|
||||
if j < end {
|
||||
fmt.Printf(" %02x", data[j])
|
||||
} else {
|
||||
fmt.Print(" ")
|
||||
}
|
||||
}
|
||||
fmt.Print(" ")
|
||||
for j := i; j < end; j++ {
|
||||
if data[j] >= 0x20 && data[j] < 0x7f {
|
||||
fmt.Printf("%c", data[j])
|
||||
} else {
|
||||
fmt.Print(".")
|
||||
}
|
||||
}
|
||||
fmt.Println()
|
||||
}
|
||||
}
|
||||
|
||||
func resolveAddr(s string, codeBase, funcOff uint64, labels []Label) (uint64, string) {
|
||||
if strings.HasPrefix(s, "0x") || strings.HasPrefix(s, "0X") {
|
||||
v, err := strconv.ParseUint(s, 0, 64)
|
||||
if err == nil {
|
||||
return v, ""
|
||||
}
|
||||
}
|
||||
if strings.HasPrefix(s, "+") {
|
||||
off, err := strconv.ParseUint(s[1:], 0, 64)
|
||||
if err == nil {
|
||||
return codeBase + funcOff + off, fmt.Sprintf("func+%#x", off)
|
||||
}
|
||||
}
|
||||
for _, l := range labels {
|
||||
if l.Name == s {
|
||||
return codeBase + funcOff + uint64(l.Offset), l.Name
|
||||
}
|
||||
}
|
||||
return 0, ""
|
||||
}
|
||||
|
||||
func lineAt(lines []SourceLine, offset int) int {
|
||||
if len(lines) == 0 {
|
||||
return 0
|
||||
}
|
||||
lo, hi := 0, len(lines)-1
|
||||
for lo < hi {
|
||||
mid := (lo + hi + 1) / 2
|
||||
if lines[mid].Offset <= offset {
|
||||
lo = mid
|
||||
} else {
|
||||
hi = mid - 1
|
||||
}
|
||||
}
|
||||
if lines[lo].Offset <= offset {
|
||||
return lines[lo].Line
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func offsetForLine(lines []SourceLine, line int) int {
|
||||
for _, le := range lines {
|
||||
if le.Line == line {
|
||||
return le.Offset
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
func nearestLabel(labels []Label, offset int) string {
|
||||
best := ""
|
||||
bestOff := -1
|
||||
for _, l := range labels {
|
||||
if l.Offset <= offset && l.Offset > bestOff {
|
||||
best = l.Name
|
||||
bestOff = l.Offset
|
||||
}
|
||||
}
|
||||
return best
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"syscall"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// StopReason describes why the debuggee stopped.
|
||||
type StopReason int
|
||||
|
||||
const (
|
||||
StopNone StopReason = iota
|
||||
StopBreakpoint // software breakpoint hit
|
||||
StopWatchpoint // hardware watchpoint triggered
|
||||
StopSingleStep // single-step completed
|
||||
StopSignal // stopped by a signal
|
||||
StopExited // process exited
|
||||
)
|
||||
|
||||
// siginfo_t layout (Linux): si_signo, si_errno, si_code, then union.
|
||||
// The si_addr field is at offset 16 on all supported architectures.
|
||||
type siginfoT struct {
|
||||
SiSigno int32
|
||||
SiErrno int32
|
||||
SiCode int32
|
||||
_pad [125]byte
|
||||
}
|
||||
|
||||
const (
|
||||
trapBRKPT = 1 // software breakpoint
|
||||
trapHWBRKPT = 4 // hardware watchpoint
|
||||
)
|
||||
|
||||
// StopInfo returns the reason the debuggee stopped and the faulting address
|
||||
// (for watchpoints, the watched address that was accessed).
|
||||
func (s *Session) StopInfo() (StopReason, uint64) {
|
||||
if s.exited {
|
||||
return StopExited, 0
|
||||
}
|
||||
var info siginfoT
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_GETSIGINFO),
|
||||
uintptr(s.pid),
|
||||
0,
|
||||
uintptr(unsafe.Pointer(&info)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return StopNone, 0
|
||||
}
|
||||
if info.SiSigno != int32(syscall.SIGTRAP) {
|
||||
return StopSignal, uint64(info.SiCode)
|
||||
}
|
||||
switch info.SiCode {
|
||||
case trapBRKPT:
|
||||
return StopBreakpoint, 0
|
||||
case trapHWBRKPT:
|
||||
addr := *(*uint64)(unsafe.Add(unsafe.Pointer(&info), 16))
|
||||
return StopWatchpoint, addr
|
||||
default:
|
||||
return StopSingleStep, 0
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package debug
|
||||
|
||||
import "fmt"
|
||||
|
||||
// SetReg modifies a register value in the debuggee.
|
||||
func (s *Session) SetReg(name string, value uint64) error {
|
||||
regs, err := s.GetRegs()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
switch name {
|
||||
case "rax", "eax", "ax", "al":
|
||||
regs.RAX = value
|
||||
case "rbx", "ebx", "bx", "bl":
|
||||
regs.RBX = value
|
||||
case "rcx", "ecx", "cx", "cl":
|
||||
regs.RCX = value
|
||||
case "rdx", "edx", "dx", "dl":
|
||||
regs.RDX = value
|
||||
case "rsi", "esi", "si":
|
||||
regs.RSI = value
|
||||
case "rdi", "edi", "di":
|
||||
regs.RDI = value
|
||||
case "rbp", "ebp", "bp":
|
||||
regs.RBP = value
|
||||
case "rsp", "esp", "sp":
|
||||
regs.RSP = value
|
||||
case "r8":
|
||||
regs.R8 = value
|
||||
case "r9":
|
||||
regs.R9 = value
|
||||
case "r10":
|
||||
regs.R10 = value
|
||||
case "r11":
|
||||
regs.R11 = value
|
||||
case "r12":
|
||||
regs.R12 = value
|
||||
case "r13":
|
||||
regs.R13 = value
|
||||
case "r14":
|
||||
regs.R14 = value
|
||||
case "r15":
|
||||
regs.R15 = value
|
||||
case "rip", "eip":
|
||||
regs.RIP = value
|
||||
default:
|
||||
return fmt.Errorf("debug: unknown register %q", name)
|
||||
}
|
||||
return s.SetRegs(®s)
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && arm64
|
||||
|
||||
package debug
|
||||
|
||||
import "fmt"
|
||||
|
||||
// SetReg modifies a register value in the debuggee.
|
||||
func (s *Session) SetReg(name string, value uint64) error {
|
||||
regs, err := s.GetRegs()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
switch name {
|
||||
case "x0":
|
||||
regs.X0 = value
|
||||
case "x1":
|
||||
regs.X1 = value
|
||||
case "x2":
|
||||
regs.X2 = value
|
||||
case "x3":
|
||||
regs.X3 = value
|
||||
case "x4":
|
||||
regs.X4 = value
|
||||
case "x5":
|
||||
regs.X5 = value
|
||||
case "x6":
|
||||
regs.X6 = value
|
||||
case "x7":
|
||||
regs.X7 = value
|
||||
case "x8":
|
||||
regs.X8 = value
|
||||
case "x9":
|
||||
regs.X9 = value
|
||||
case "x10":
|
||||
regs.X10 = value
|
||||
case "x11":
|
||||
regs.X11 = value
|
||||
case "x12":
|
||||
regs.X12 = value
|
||||
case "x13":
|
||||
regs.X13 = value
|
||||
case "x14":
|
||||
regs.X14 = value
|
||||
case "x15":
|
||||
regs.X15 = value
|
||||
case "x16":
|
||||
regs.X16 = value
|
||||
case "x17":
|
||||
regs.X17 = value
|
||||
case "x18":
|
||||
regs.X18 = value
|
||||
case "x19":
|
||||
regs.X19 = value
|
||||
case "x20":
|
||||
regs.X20 = value
|
||||
case "x21":
|
||||
regs.X21 = value
|
||||
case "x22":
|
||||
regs.X22 = value
|
||||
case "x23":
|
||||
regs.X23 = value
|
||||
case "x24":
|
||||
regs.X24 = value
|
||||
case "x25":
|
||||
regs.X25 = value
|
||||
case "x26":
|
||||
regs.X26 = value
|
||||
case "x27":
|
||||
regs.X27 = value
|
||||
case "x28":
|
||||
regs.X28 = value
|
||||
case "x29", "fp":
|
||||
regs.X29 = value
|
||||
case "x30", "lr":
|
||||
regs.X30 = value
|
||||
case "sp":
|
||||
regs.SP = value
|
||||
case "pc":
|
||||
regs.PC = value
|
||||
default:
|
||||
return fmt.Errorf("debug: unknown register %q", name)
|
||||
}
|
||||
return s.SetRegs(®s)
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && loong64
|
||||
|
||||
package debug
|
||||
|
||||
import "fmt"
|
||||
|
||||
// SetReg modifies a register value in the debuggee.
|
||||
func (s *Session) SetReg(name string, value uint64) error {
|
||||
regs, err := s.GetRegs()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
switch name {
|
||||
case "r0", "zero":
|
||||
regs.R0 = value
|
||||
case "r1", "ra":
|
||||
regs.R1 = value
|
||||
case "r2", "tp":
|
||||
regs.R2 = value
|
||||
case "r3", "sp":
|
||||
regs.R3 = value
|
||||
case "r4", "a0":
|
||||
regs.R4 = value
|
||||
case "r5", "a1":
|
||||
regs.R5 = value
|
||||
case "r6", "a2":
|
||||
regs.R6 = value
|
||||
case "r7", "a3":
|
||||
regs.R7 = value
|
||||
case "r8", "a4":
|
||||
regs.R8 = value
|
||||
case "r9", "a5":
|
||||
regs.R9 = value
|
||||
case "r10", "a6":
|
||||
regs.R10 = value
|
||||
case "r11", "a7":
|
||||
regs.R11 = value
|
||||
case "r12", "t0":
|
||||
regs.R12 = value
|
||||
case "r13", "t1":
|
||||
regs.R13 = value
|
||||
case "r14", "t2":
|
||||
regs.R14 = value
|
||||
case "r15", "t3":
|
||||
regs.R15 = value
|
||||
case "r16", "t4":
|
||||
regs.R16 = value
|
||||
case "r17", "t5":
|
||||
regs.R17 = value
|
||||
case "r18", "t6":
|
||||
regs.R18 = value
|
||||
case "r19", "t7":
|
||||
regs.R19 = value
|
||||
case "r20", "t8":
|
||||
regs.R20 = value
|
||||
case "r21", "fp":
|
||||
regs.R21 = value
|
||||
case "r22", "s0":
|
||||
regs.R22 = value
|
||||
case "r23", "s1":
|
||||
regs.R23 = value
|
||||
case "r24", "s2":
|
||||
regs.R24 = value
|
||||
case "r25", "s3":
|
||||
regs.R25 = value
|
||||
case "r26", "s4":
|
||||
regs.R26 = value
|
||||
case "r27", "s5":
|
||||
regs.R27 = value
|
||||
case "r28", "s6":
|
||||
regs.R28 = value
|
||||
case "r29", "s7":
|
||||
regs.R29 = value
|
||||
case "r30", "s8":
|
||||
regs.R30 = value
|
||||
case "r31", "pc":
|
||||
regs.R31 = value
|
||||
default:
|
||||
return fmt.Errorf("debug: unknown register %q", name)
|
||||
}
|
||||
return s.SetRegs(®s)
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && riscv64
|
||||
|
||||
package debug
|
||||
|
||||
import "fmt"
|
||||
|
||||
// SetReg modifies a register value in the debuggee.
|
||||
func (s *Session) SetReg(name string, value uint64) error {
|
||||
regs, err := s.GetRegs()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
switch name {
|
||||
case "pc":
|
||||
regs.PC = value
|
||||
case "ra", "x1":
|
||||
regs.Ra = value
|
||||
case "sp", "x2":
|
||||
regs.Sp = value
|
||||
case "gp", "x3":
|
||||
regs.Gp = value
|
||||
case "tp", "x4":
|
||||
regs.Tp = value
|
||||
case "t0", "x5":
|
||||
regs.T0 = value
|
||||
case "t1", "x6":
|
||||
regs.T1 = value
|
||||
case "t2", "x7":
|
||||
regs.T2 = value
|
||||
case "s0", "fp", "x8":
|
||||
regs.S0 = value
|
||||
case "s1", "x9":
|
||||
regs.S1 = value
|
||||
case "a0", "x10":
|
||||
regs.A0 = value
|
||||
case "a1", "x11":
|
||||
regs.A1 = value
|
||||
case "a2", "x12":
|
||||
regs.A2 = value
|
||||
case "a3", "x13":
|
||||
regs.A3 = value
|
||||
case "a4", "x14":
|
||||
regs.A4 = value
|
||||
case "a5", "x15":
|
||||
regs.A5 = value
|
||||
case "a6", "x16":
|
||||
regs.A6 = value
|
||||
case "a7", "x17":
|
||||
regs.A7 = value
|
||||
case "s2", "x18":
|
||||
regs.S2 = value
|
||||
case "s3", "x19":
|
||||
regs.S3 = value
|
||||
case "s4", "x20":
|
||||
regs.S4 = value
|
||||
case "s5", "x21":
|
||||
regs.S5 = value
|
||||
case "s6", "x22":
|
||||
regs.S6 = value
|
||||
case "s7", "x23":
|
||||
regs.S7 = value
|
||||
case "s8", "x24":
|
||||
regs.S8 = value
|
||||
case "s9", "x25":
|
||||
regs.S9 = value
|
||||
case "s10", "x26":
|
||||
regs.S10 = value
|
||||
case "s11", "x27":
|
||||
regs.S11 = value
|
||||
case "t3", "x28":
|
||||
regs.T3 = value
|
||||
case "t4", "x29":
|
||||
regs.T4 = value
|
||||
case "t5", "x30":
|
||||
regs.T5 = value
|
||||
case "t6", "x31":
|
||||
regs.T6 = value
|
||||
default:
|
||||
return fmt.Errorf("debug: unknown register %q", name)
|
||||
}
|
||||
return s.SetRegs(®s)
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user