Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
01dcc3b86e | ||
|
|
b08885bd31 | ||
|
|
c05c53452f | ||
|
|
681a449c01 | ||
|
|
31a2cee382 | ||
|
|
3bc7c18bc3 | ||
|
|
f0512a4e1c | ||
|
|
373c09f725 | ||
|
|
b78b6c5004 | ||
|
|
9b5c878f9e | ||
|
|
31ee8e7941 | ||
|
|
2d1176e045 | ||
|
|
2a27a3a52b | ||
|
|
cde7d0f96a | ||
|
|
d7ee1b78d4 | ||
|
|
30c53565a7 | ||
|
|
ebd8ab8a3c | ||
|
|
176d856f67 | ||
|
|
eace06bbd6 | ||
|
|
f97bea61c5 | ||
|
|
19b37569c0 | ||
|
|
23b3d3e152 | ||
|
|
b0c62be8ce | ||
|
|
ece0d3f127 | ||
|
|
ad6e3360df | ||
|
|
49566de7fb | ||
|
|
6228d77566 | ||
|
|
c0e280ee3c | ||
|
|
2d45dbf7ff | ||
|
|
8ddac0135e | ||
|
|
f58a4fe51d | ||
|
|
5cb7e3e231 | ||
|
|
36bbc0c13b | ||
|
|
32afa3449f | ||
|
|
2ab6b9eb84 | ||
|
|
f41a86b660 | ||
|
|
7721353d44 | ||
|
|
243b087116 | ||
|
|
eee7a6d4a4 | ||
|
|
801fb963c9 | ||
|
|
a2acc9b5a3 | ||
|
|
f860bf8ce6 | ||
|
|
e7df5e5225 | ||
|
|
d114b3412c | ||
|
|
c77d68018c | ||
|
|
89d633f4bb | ||
|
|
f20e0bf1e7 | ||
|
|
1a45b66139 | ||
|
|
382efe538a | ||
|
|
f8d28a42ba | ||
|
|
51a2854d7f |
@@ -0,0 +1,157 @@
|
|||||||
|
# Release — gasm binaries. Runs on version tags (v0.28.0) pushed to main.
|
||||||
|
name: Release
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
tags: ["v*"]
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
build:
|
||||||
|
runs-on: fedora
|
||||||
|
strategy:
|
||||||
|
fail-fast: false
|
||||||
|
matrix:
|
||||||
|
include:
|
||||||
|
- goos: linux
|
||||||
|
goarch: amd64
|
||||||
|
- goos: linux
|
||||||
|
goarch: arm64
|
||||||
|
- goos: linux
|
||||||
|
goarch: riscv64
|
||||||
|
- goos: linux
|
||||||
|
goarch: loong64
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
|
- uses: actions/setup-go@v6
|
||||||
|
with:
|
||||||
|
go-version: "1.26"
|
||||||
|
|
||||||
|
- name: Download dependencies
|
||||||
|
run: go mod download
|
||||||
|
|
||||||
|
- name: Validate tag and build
|
||||||
|
id: build
|
||||||
|
env:
|
||||||
|
VERSION: ${{ gitea.ref_name }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
if ! echo "$VERSION" | grep -qE '^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$'; then
|
||||||
|
echo "ERROR: expected a semver tag like v1.2.3, got: '$VERSION'"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
VERSION_NO_V="${VERSION#v}"
|
||||||
|
echo "version_no_v=${VERSION_NO_V}" >> "$GITEA_OUTPUT"
|
||||||
|
|
||||||
|
mkdir -p bin
|
||||||
|
GOOS=${{ matrix.goos }} GOARCH=${{ matrix.goarch }} CGO_ENABLED=0 \
|
||||||
|
go build -ldflags "-s -w -X main.version=${VERSION_NO_V}" \
|
||||||
|
-o "bin/gasm-${VERSION_NO_V}-${{ matrix.goos }}-${{ matrix.goarch }}" \
|
||||||
|
./cmd/gasm
|
||||||
|
|
||||||
|
- name: Upload artifact
|
||||||
|
uses: actions/upload-artifact@v3
|
||||||
|
with:
|
||||||
|
name: gasm-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||||
|
path: bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||||
|
if-no-files-found: error
|
||||||
|
|
||||||
|
- name: Smoke test
|
||||||
|
if: matrix.goos == 'linux' && matrix.goarch == 'amd64'
|
||||||
|
run: |
|
||||||
|
chmod +x bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||||
|
./bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }} --version
|
||||||
|
|
||||||
|
release:
|
||||||
|
runs-on: fedora
|
||||||
|
needs: build
|
||||||
|
permissions:
|
||||||
|
releases: write
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
|
- name: Download all artifacts
|
||||||
|
uses: actions/download-artifact@v3
|
||||||
|
with:
|
||||||
|
path: dist
|
||||||
|
|
||||||
|
- name: Extract CHANGELOG section
|
||||||
|
env:
|
||||||
|
VERSION: ${{ gitea.ref_name }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
VERSION_NO_V="${VERSION#v}"
|
||||||
|
|
||||||
|
sed -n "/^## \[${VERSION_NO_V}\] /,/^## \[/p" CHANGELOG.md \
|
||||||
|
| sed '$d' \
|
||||||
|
| tail -n +2 \
|
||||||
|
> release-body.md
|
||||||
|
|
||||||
|
if [ ! -s release-body.md ]; then
|
||||||
|
echo "ERROR: no CHANGELOG section found for ${VERSION_NO_V}"
|
||||||
|
echo "Expected a heading like: ## [${VERSION_NO_V}] — YYYY-MM-DD"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
- name: Create release
|
||||||
|
env:
|
||||||
|
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||||
|
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
||||||
|
GITEA_REPOSITORY: ${{ gitea.repository }}
|
||||||
|
GITEA_REF_NAME: ${{ gitea.ref_name }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
BODY=$(sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' -e 's/\t/\\t/g' -e 's/\r//g' release-body.md | sed ':a;N;$!ba;s/\n/\\n/g')
|
||||||
|
BODY="\"${BODY}\""
|
||||||
|
|
||||||
|
response=$(curl -sS -w '\n%{http_code}' \
|
||||||
|
-H "Authorization: token ${GITEA_TOKEN}" \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-X POST \
|
||||||
|
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases" \
|
||||||
|
-d "{\"tag_name\":\"${GITEA_REF_NAME}\",\"name\":\"${GITEA_REF_NAME}\",\"body\":${BODY},\"draft\":false,\"prerelease\":false}")
|
||||||
|
|
||||||
|
http_code=$(echo "$response" | tail -1)
|
||||||
|
payload=$(echo "$response" | sed '$d')
|
||||||
|
|
||||||
|
echo "HTTP ${http_code}"
|
||||||
|
if [ "$http_code" != "201" ]; then
|
||||||
|
echo "Failed to create release: ${payload}"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
RELEASE_ID=$(echo "$payload" | grep -oE '"id"[[:space:]]*:[[:space:]]*[0-9]+' | head -1 | grep -oE '[0-9]+')
|
||||||
|
echo "Created release ID=${RELEASE_ID}"
|
||||||
|
printf '%s' "${RELEASE_ID}" > release-id.txt
|
||||||
|
|
||||||
|
- name: Upload assets
|
||||||
|
env:
|
||||||
|
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||||
|
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
||||||
|
GITEA_REPOSITORY: ${{ gitea.repository }}
|
||||||
|
GITEA_REF_NAME: ${{ gitea.ref_name }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
RELEASE_ID=$(cat release-id.txt)
|
||||||
|
|
||||||
|
for binary in dist/gasm-*/gasm-*; do
|
||||||
|
[ -f "$binary" ] || continue
|
||||||
|
fname=$(basename "$binary")
|
||||||
|
echo "Uploading ${fname}..."
|
||||||
|
http_code=$(curl -sS -o /dev/null -w '%{http_code}' \
|
||||||
|
-H "Authorization: token ${GITEA_TOKEN}" \
|
||||||
|
-H "Content-Type: application/octet-stream" \
|
||||||
|
-X POST \
|
||||||
|
--data-binary "@${binary}" \
|
||||||
|
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases/${RELEASE_ID}/assets?name=${fname}")
|
||||||
|
echo " HTTP ${http_code}"
|
||||||
|
if [ "$http_code" != "201" ]; then
|
||||||
|
echo "Failed to upload ${fname}"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
echo "Release ${GITEA_REF_NAME} is live."
|
||||||
@@ -0,0 +1,96 @@
|
|||||||
|
# Test — gasm-devkit. Runs on push and pull request to development.
|
||||||
|
name: Test
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: [development]
|
||||||
|
pull_request:
|
||||||
|
branches: [development]
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
vet:
|
||||||
|
runs-on: fedora
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
|
- uses: actions/setup-go@v6
|
||||||
|
with:
|
||||||
|
go-version: "1.26"
|
||||||
|
|
||||||
|
- name: Download dependencies
|
||||||
|
run: go mod download
|
||||||
|
|
||||||
|
- name: gofmt
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
unformatted=$(gofmt -l .)
|
||||||
|
if [ -n "$unformatted" ]; then
|
||||||
|
echo "These files need gofmt:"
|
||||||
|
echo "$unformatted"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
- name: go vet
|
||||||
|
run: go vet ./...
|
||||||
|
|
||||||
|
test:
|
||||||
|
runs-on: fedora
|
||||||
|
needs: vet
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
|
- uses: actions/setup-go@v6
|
||||||
|
with:
|
||||||
|
go-version: "1.26"
|
||||||
|
|
||||||
|
- name: Download dependencies
|
||||||
|
run: go mod download
|
||||||
|
|
||||||
|
- name: Install gcc
|
||||||
|
run: dnf install -y gcc
|
||||||
|
|
||||||
|
- name: go test -race
|
||||||
|
run: go test -race -count=1 ./...
|
||||||
|
|
||||||
|
- name: Coverage gate — 80 % minimum
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
# Exclude packages inherently untestable without hardware:
|
||||||
|
# debug — interactive ptrace, requires a live process
|
||||||
|
# cmd/gasm — CLI glue, covered by integration tests
|
||||||
|
go test -coverprofile=coverage.out \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/arch \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/asm \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/ast \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/format \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/lexer \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/lint \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/lsp \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/parser \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/token \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/verify
|
||||||
|
coverage=$(go tool cover -func=coverage.out | awk '/^total:/ { gsub("%", "", $3); print $3 }')
|
||||||
|
echo "Total coverage: ${coverage}%"
|
||||||
|
if awk -v c="$coverage" 'BEGIN { exit !(c+0 < 80) }'; then
|
||||||
|
echo "ERROR: coverage ${coverage}% is below the 80% threshold"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
build:
|
||||||
|
runs-on: fedora
|
||||||
|
needs: test
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
|
- uses: actions/setup-go@v6
|
||||||
|
with:
|
||||||
|
go-version: "1.26"
|
||||||
|
|
||||||
|
- name: Download dependencies
|
||||||
|
run: go mod download
|
||||||
|
|
||||||
|
- name: Build
|
||||||
|
run: go build -ldflags="-s -w" -o bin/gasm ./cmd/gasm
|
||||||
|
|
||||||
|
- name: Smoke test
|
||||||
|
run: ./bin/gasm --version
|
||||||
@@ -10,3 +10,6 @@ coverage.out
|
|||||||
# Editor detritus
|
# Editor detritus
|
||||||
*.swp
|
*.swp
|
||||||
.DS_Store
|
.DS_Store
|
||||||
|
|
||||||
|
# Scratch / temporary work
|
||||||
|
_scratch/
|
||||||
|
|||||||
@@ -0,0 +1,119 @@
|
|||||||
|
# AGENTS.md — gasm-devkit
|
||||||
|
|
||||||
|
Repository rules for AI agents and contributors. Read before modifying any
|
||||||
|
code in this repository.
|
||||||
|
|
||||||
|
## AI Contribution Policy
|
||||||
|
|
||||||
|
AI agents may assist with code, documentation, tests, and review in this
|
||||||
|
repository. All AI-assisted changes must:
|
||||||
|
|
||||||
|
- Follow the code style and conventions in this file.
|
||||||
|
- Include the trailer `Assisted-by: <model-name>` in every commit message.
|
||||||
|
- Not commit directly to `main` — work on `development`.
|
||||||
|
- Pass the full Definition of Done before any commit.
|
||||||
|
|
||||||
|
## Workflow
|
||||||
|
|
||||||
|
- **Branching.** `development` is the working branch. `main` is
|
||||||
|
release-only: merge from `development`, then tag. Never commit directly
|
||||||
|
to `main`.
|
||||||
|
- **Release procedure.**
|
||||||
|
1. Bump `version` in `justfile` and `cmd/gasm/main.go`.
|
||||||
|
2. Update `CHANGELOG.md` with a new `## [X.Y.Z] — YYYY-MM-DD` section.
|
||||||
|
3. Update `README.md` and `docs/ARCHITECTURE.md` if user-visible
|
||||||
|
behaviour changed.
|
||||||
|
4. Run the Definition of Done (below).
|
||||||
|
5. Commit on `development`.
|
||||||
|
6. `git checkout main && git merge --ff-only development`.
|
||||||
|
7. `git tag vX.Y.Z`.
|
||||||
|
8. `git checkout development`.
|
||||||
|
9. `GOBIN=~/.local/bin just install-bin`.
|
||||||
|
|
||||||
|
## Commit Messages
|
||||||
|
|
||||||
|
Conventional Commits, subject line only, imperative mood, lowercase after
|
||||||
|
the colon:
|
||||||
|
|
||||||
|
```
|
||||||
|
feat(asm): add EVEX gather and scatter with VSIB addressing
|
||||||
|
```
|
||||||
|
|
||||||
|
Allowed types: `feat`, `fix`, `docs`, `style`, `refactor`, `perf`, `test`,
|
||||||
|
`chore`, `ci`, `build`, `revert`.
|
||||||
|
|
||||||
|
Every commit ends with exactly one trailer, using the model that
|
||||||
|
assisted with the change:
|
||||||
|
|
||||||
|
```
|
||||||
|
Assisted-by: <model-name>
|
||||||
|
```
|
||||||
|
|
||||||
|
Replace `<model-name>` with the actual model (e.g. `DeepSeek V4 Pro`).
|
||||||
|
|
||||||
|
No body, no footers, no trailing period on the subject.
|
||||||
|
|
||||||
|
## Code Style
|
||||||
|
|
||||||
|
Language: Go 1.26 (`toolchain go1.26.5`).
|
||||||
|
|
||||||
|
### Formatter
|
||||||
|
|
||||||
|
`gofmt` — zero diff. Run `just fmt` before committing.
|
||||||
|
|
||||||
|
### Linter
|
||||||
|
|
||||||
|
`go vet` — zero warnings. Run `just build` before committing.
|
||||||
|
|
||||||
|
### Tests
|
||||||
|
|
||||||
|
`go test -race -count=1 ./...` — all green, coverage ≥ 80 % (hard gate,
|
||||||
|
enforced by `just test`).
|
||||||
|
|
||||||
|
### Dependencies
|
||||||
|
|
||||||
|
- **Production code:** standard library only. No third-party imports in
|
||||||
|
shipped code.
|
||||||
|
- **Test code:** `golang.org/x/arch` is the sole test dependency (decode
|
||||||
|
oracle for round-trip validation). It is never linked into the binary.
|
||||||
|
- **No cgo, no C, no external toolchains, no JavaScript.**
|
||||||
|
|
||||||
|
### Error Handling
|
||||||
|
|
||||||
|
Explicit `if err != nil`. Wrap with `fmt.Errorf("context: %w", err)`.
|
||||||
|
No panics outside `main`. The one exception: the JIT trampoline's
|
||||||
|
`recover`-guarded decoder hot path, which converts bounds panics to
|
||||||
|
sentinel errors.
|
||||||
|
|
||||||
|
### Assembly
|
||||||
|
|
||||||
|
Plan 9 syntax (Go's assembler dialect). Hand-written — no code generators
|
||||||
|
except `_gen/gen.go` for instruction tables (which parses the Go
|
||||||
|
toolchain source). Every instruction table is committed; no runtime
|
||||||
|
dependency on the Go toolchain.
|
||||||
|
|
||||||
|
### File Naming
|
||||||
|
|
||||||
|
- `_amd64.s`, `_arm64.s`, `_riscv64.s`, `_loong64.s` for
|
||||||
|
architecture-specific assembly.
|
||||||
|
- `_linux_amd64.go` for platform-specific Go files.
|
||||||
|
- `_test.go` suffix for test files.
|
||||||
|
|
||||||
|
## Definition of Done
|
||||||
|
|
||||||
|
A task is not complete until all of these pass:
|
||||||
|
|
||||||
|
1. `just build` — `go vet` + `gofmt` check, zero errors, zero warnings.
|
||||||
|
2. `just test` — full suite with `-race`, coverage ≥ 80 %.
|
||||||
|
3. `just fmt` — produces no diff.
|
||||||
|
4. Diagnostics — zero warnings across the project.
|
||||||
|
5. Non-trivial changes reviewed.
|
||||||
|
|
||||||
|
## Licence
|
||||||
|
|
||||||
|
BSD-3-Clause. Every source file carries the SPDX header:
|
||||||
|
|
||||||
|
```
|
||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
```
|
||||||
+929
@@ -0,0 +1,929 @@
|
|||||||
|
# Changelog
|
||||||
|
|
||||||
|
All notable changes to gasm-devkit are documented here.
|
||||||
|
|
||||||
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
||||||
|
and this project adheres to [Conventional Commits](https://www.conventionalcommits.org/).
|
||||||
|
|
||||||
|
## [development]
|
||||||
|
|
||||||
|
Unreleased changes on the `development` branch.
|
||||||
|
|
||||||
|
## [0.30.0] — 2026-08-13
|
||||||
|
|
||||||
|
The LoongArch encoder (Phase 5) ships with ELF64 and GOOBJ emission, verified
|
||||||
|
byte-for-byte against `GOARCH=loong64 go tool asm` and linked into a real
|
||||||
|
`go build`; the shared GOOBJ emitter now writes the per-function DWARF symbols
|
||||||
|
the linker's DWARF pass reads. The RISC-V encoder reaches byte-for-byte parity
|
||||||
|
with `go tool asm`: the frame model, operand ordering, RVC compression,
|
||||||
|
large-immediate and `MOV $imm` materialisation, branch/jump encodings, and
|
||||||
|
`CALL sym(SB)` (now a `JAL` with an `R_RISCV_JAL` relocation). The debugger
|
||||||
|
tracks four hardware watchpoint slots, and the toolkit is Linux-only.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **LoongArch encoder (Phase 5).** `gasm asm` can now assemble `_loong64.s`
|
||||||
|
files: the full LoongArch64 instruction set with the dual-form arithmetic
|
||||||
|
mnemonics, the 16/21-bit branch families, the MOV pseudo-instruction and
|
||||||
|
its immediate-constant expansions, FP/SP frame mapping, SB/global symbol
|
||||||
|
references (pcalau12i pairs) and ELF64 emission
|
||||||
|
(`gasm asm --format elf`). Ground-truth verification against
|
||||||
|
`GOARCH=loong64 go tool asm` matches byte-for-byte; GOOBJ emission
|
||||||
|
(`gasm asm --format goobj`) is proven end-to-end by linking the object
|
||||||
|
into a cross-compiled `go build`.
|
||||||
|
- **GOOBJ DWARF symbols.** The GOOBJ emitters now write the per-function
|
||||||
|
DWARF symbols the linker requires (the subprogram DIE and the `.debug_line`
|
||||||
|
program, byte-identical to `cmd/asm`'s), and the pc-value table deltas are
|
||||||
|
in the architecture's MinLC units as the runtime expects — the amd64 link
|
||||||
|
test now genuinely substitutes the gasm object, and the amd64/loong64
|
||||||
|
end-to-end GOOBJ link tests pass.
|
||||||
|
- **RISC-V GOOBJ emission via the shared emitter.** RISC-V GOOBJ output is
|
||||||
|
now written by the same shared emitter as amd64 and LoongArch, modelling
|
||||||
|
each AUIPC + second-instruction pair as a single R_RISCV_PCREL_ITYPE/STYPE
|
||||||
|
relocation (the layout `cmd/asm` writes, not the ELF HI20/LO12 pair), so the
|
||||||
|
object links into a cross-compiled `go build` for `GOARCH=riscv64`. An
|
||||||
|
end-to-end link test substitutes the gasm object and reads the symbol back
|
||||||
|
with `go tool nm`; the rewrite also corrects the relocation `after` field.
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **RISC-V frame model and RVC encodings.** The riscv64 frame layout now
|
||||||
|
matches `go tool asm`: the prologue/epilogue save and restore the link
|
||||||
|
register (LR) instead of S0, with the correct autosize (locals + 8) and the
|
||||||
|
RVC-compressed prologue/epilogue instructions; `RET` emits the uncompressed
|
||||||
|
`JALR X0, 0(X1)` the toolchain writes; the `C.ADDI`/`C.LI`/`C.LUI`/`C.ADDIW`
|
||||||
|
opcode bit and the `C.ADD` CR-type encoding are fixed; and the `LR`/`TMP`
|
||||||
|
register aliases now resolve to X1 and X31. The pcsp/pcfile/pcline tables
|
||||||
|
are populated from the recorded stack-adjustment and source-line data, and
|
||||||
|
a byte-exact ground-truth test compares framed and leaf functions against
|
||||||
|
`GOARCH=riscv64 go tool asm`.
|
||||||
|
- **RISC-V operand ordering and RVC compression.** R-type instructions now
|
||||||
|
take `rs2, rs1, rd` and I-type arithmetic instructions take `imm12, rs1,
|
||||||
|
rd`, matching the Go assembler's documented operand order (previously both
|
||||||
|
were reversed, so non-commutative R-type instructions such as `SUB` encoded
|
||||||
|
the wrong operation). The two-operand ternary forms (`ADD rs2, rd`,
|
||||||
|
`ADDI $imm, rd`, `SLLI $shamt, rd`) are now accepted. RVC compression is
|
||||||
|
completed for `C.ADDI16SP`, `C.SLLI`, `C.SRLI`, `C.SRAI`, `C.ANDI`,
|
||||||
|
`C.NOP`, `C.EBREAK`, `C.MV` (from `ADDI`/`ADD`) and the commutative
|
||||||
|
`AND`/`OR`/`XOR` forms; the byte-exact ground-truth test now covers these.
|
||||||
|
- **RISC-V compressed loads/stores and word arithmetic.** RVC compression
|
||||||
|
now also covers the register-relative `C.LW`/`C.SW`/`C.LD`/`C.SD`/
|
||||||
|
`C.FLD`/`C.FSD` forms (in addition to the stack-relative `C.LWSP`/`C.SWSP`/
|
||||||
|
`C.LDSP`/`C.SDSP`), plus `C.ADDI4SPN`, `C.ADDW` and `C.SUBW`. The
|
||||||
|
byte-exact ground-truth test exercises these against `GOARCH=riscv64
|
||||||
|
go tool asm`.
|
||||||
|
- **RISC-V large-immediate materialisation.** `ADDI`/`ANDI`/`ORI`/`XORI`
|
||||||
|
with a 32-bit immediate that does not fit 12 bits now expand exactly as
|
||||||
|
`cmd/asm`: two `ADDI`s for the small `ADDI` split range, and
|
||||||
|
`LUI`+`ADDIW`+`<op>` otherwise, with the `LUI` and `ADDIW` compressed to
|
||||||
|
`C.LUI`/`C.ADDIW` when their immediate fits six signed bits. The
|
||||||
|
byte-exact ground-truth test covers positive, negative, and out-of-range
|
||||||
|
immediates against `GOARCH=riscv64 go tool asm`.
|
||||||
|
- **RISC-V `MOV $imm, rd` materialisation.** The immediate-loading
|
||||||
|
pseudo-instruction now uses the toolchain's `Split32BitImmediate` split
|
||||||
|
(previously it rounded the upper 20 bits, producing wrong results for
|
||||||
|
negative and bit-11-set immediates) and compresses the emitted
|
||||||
|
`ADDI`/`LUI`/`ADDIW` to `C.LI`/`C.LUI`/`C.ADDIW` when their immediate
|
||||||
|
fits six signed bits. A byte-exact ground-truth test covers zero, small,
|
||||||
|
negative, and 32-bit immediates against `GOARCH=riscv64 go tool asm`.
|
||||||
|
- **RISC-V branch/jump compression.** `JMP`/`JAL` were being compressed to
|
||||||
|
`C.J` and `BEQ`/`BNE` (with `X0`) to `C.BEQZ`/`C.BNEZ`, but `go tool asm`
|
||||||
|
never emits these compressed forms. They now emit the 32-bit `JAL` and
|
||||||
|
branch encodings the toolchain writes; the dead `C.J`/`C.BEQZ`/`C.BNEZ`
|
||||||
|
encoders were removed, and the `C.LUI` direct-instruction compression now
|
||||||
|
uses the correct six-bit signed range. A byte-exact ground-truth test
|
||||||
|
covers the branch family and jumps against `GOARCH=riscv64 go tool asm`.
|
||||||
|
- **RISC-V `CALL sym(SB)`.** The call pseudo-instruction now emits the
|
||||||
|
toolchain's `JAL X1, sym(SB)` with a single `R_RISCV_JAL` relocation
|
||||||
|
(previously it emitted an `AUIPC`+`JALR` pair against a local branch
|
||||||
|
label, a form `go tool asm` rejects). The GOOBJ and ELF emitters now map
|
||||||
|
that relocation (Go objabi 59 / ELF `R_RISCV_JAL` 17, a 4-byte field), and
|
||||||
|
relocation offsets are recorded relative to the function start (including
|
||||||
|
the prologue). A byte-exact ground-truth test covers a call against
|
||||||
|
`GOARCH=riscv64 go tool asm`.
|
||||||
|
- **Debugger watchpoint slots.** `gasm debug`'s `watch` command always used
|
||||||
|
hardware watchpoint slot 0, so a second `watch` call silently overwrote
|
||||||
|
the first. Watchpoint slots are now tracked in the `Session` (DR0–DR3);
|
||||||
|
`watch` picks the first free slot and reports an error if all four are in
|
||||||
|
use, and `unwatch <slot>` clears one (no argument clears all).
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **Linux only.** The toolkit, its CI and the released binaries are now
|
||||||
|
Linux-only; cross-compiled to linux/{amd64,arm64,riscv64,loong64}.
|
||||||
|
- **Phase 4 closed.** README's "Remaining" list for the debugger is gone;
|
||||||
|
disassembly at PC, memory-write, watchpoints, and source-line mapping are
|
||||||
|
all shipped.
|
||||||
|
|
||||||
|
## [0.29.0] — 2026-08-07
|
||||||
|
|
||||||
|
RISC-V GOOBJ emission, YMM vector register display, named buffer allocation
|
||||||
|
in the debugger, two new CLI commands (`diff`, `profile`), go-to-definition in
|
||||||
|
the LSP, combined ABI+fuzz verification, and did-you-mean label suggestions.
|
||||||
|
A `--map` flag for `diff` and `--call`/`--buf` flags for `verify` extend the
|
||||||
|
new CLI commands. A signature-parser fix corrects grouped Go parameters.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **RISC-V GOOBJ emission** — `gasm asm --format goobj` for RISC-V produces
|
||||||
|
linkable Go objects with funcdata, pc-value tables, and RISC-V relocation
|
||||||
|
types (same format as amd64 GOOBJ, with the RISC-V architecture marker).
|
||||||
|
- **`gasm diff`** — compare the machine code of two assembly files byte-for-byte;
|
||||||
|
shows which functions differ and the first few differing bytes.
|
||||||
|
- **`gasm profile`** — show the basic-block structure of each function: labels,
|
||||||
|
offsets, frame size, and NOSPLIT flag.
|
||||||
|
- **LSP go-to-definition** — `textDocument/definition` navigates from a label
|
||||||
|
reference to its definition.
|
||||||
|
- **did-you-mean** — when the RISC-V assembler encounters an undefined label, it
|
||||||
|
suggests the closest existing label using Levenshtein distance.
|
||||||
|
- **YMM vector register display** — `regs` in the debugger now shows YMM
|
||||||
|
registers via `PTRACE_GETFPREGS` (falls back to XMM when XSAVE is unavailable).
|
||||||
|
- **Named buffer allocation** — `gasm debug --buf name:size:pattern` allocates
|
||||||
|
buffers in the debuggee filled with `zero`, `ones`, `seq`, or a hex pattern;
|
||||||
|
buffer pointers are placed into the argument block at the matching positions.
|
||||||
|
- **Crash input storage** — `FuzzResult.CrashInput` stores the input that caused
|
||||||
|
a crash or mismatch for reproducibility.
|
||||||
|
- **ABI + fuzz combined** — `gasm verify --fuzz` now runs ABI checks (sentinel
|
||||||
|
registers, canary, stack bounds) alongside differential fuzz testing.
|
||||||
|
- **`gasm diff --map`** — compare functions whose names differ between files
|
||||||
|
(e.g. `--map wideCopyAVX2=wideCopyAVX512` pairs two variants regardless
|
||||||
|
of suffix). Unmapped functions fall back to the original name match.
|
||||||
|
- **`gasm verify --call`** — invoke a single function with user-supplied buffers
|
||||||
|
(`--buf name:size:pattern`) instead of the smoke/abi/fuzz sweeps. Patterns:
|
||||||
|
`zero`, `ones`, `seq`, or a hex blob. Useful for partial functions (e.g.
|
||||||
|
decoders) that crash on random input but should succeed on valid data.
|
||||||
|
The arg block is printed before and after the call, showing return values.
|
||||||
|
- **`gasm verify --ground-truth`** now documented in `--help` (was already a flag,
|
||||||
|
just missing from the help text).
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Signature parser** — grouped Go parameters like `dst, src []byte` are now
|
||||||
|
parsed correctly (both get type `[]byte`). Previously the first name was
|
||||||
|
treated as its own type (`dst` with size 8), causing wrong ABI0 arg-block
|
||||||
|
layout in both `verify --call` and the fuzzer.
|
||||||
|
- **Flaky JIT tests** — `runtime.KeepAlive` guards and package-level buffers
|
||||||
|
prevent GC from collecting heap objects whose addresses were passed to JIT
|
||||||
|
code via `unsafe.Pointer`; all verify tests pass 100/100 under `-race`.
|
||||||
|
|
||||||
|
### Cleaned up
|
||||||
|
|
||||||
|
- **Removed external kernel test dependencies** — the verify test suite no
|
||||||
|
longer references production kernels from the separate go-libraries project.
|
||||||
|
The remaining test suite uses only `testdata/verify/*.s` kernels, which are
|
||||||
|
part of this repository. Coverage is identical locally and in CI (80.3 %).
|
||||||
|
|
||||||
|
### Verified
|
||||||
|
|
||||||
|
- `gasm diff` detects byte-level differences; `--map` pairs differently-named
|
||||||
|
functions for comparison.
|
||||||
|
- `gasm verify --call` invokes functions with user-supplied buffers; the arg
|
||||||
|
block is printed before and after the call, showing return values.
|
||||||
|
- LSP go-to-definition resolves labels across functions and files.
|
||||||
|
|
||||||
|
## [0.28.0] — 2026-08-03
|
||||||
|
|
||||||
|
RISC-V encoder: full RV64IMAFDC instruction set with RVC compression, MOV
|
||||||
|
pseudo-instruction, SB/global symbol references, ELF64 object emission, and
|
||||||
|
ground-truth verification against `GOARCH=riscv64 go tool asm`.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **RISC-V encoder** — RV64I, RV64M, RV64A, RV64F/D, FMA, CSR, JALR.
|
||||||
|
- **MOV pseudo-instruction** — load, store, reg-to-reg, immediate, frame mapping.
|
||||||
|
- **RVC compression** — 22 compressed instruction types (C.LDSP, C.SDSP, C.FLDSP,
|
||||||
|
C.FSDSP, C.ADDI, C.LI, C.LUI, C.ADDIW, C.MV, C.ADD, C.SUB, C.XOR, C.OR, C.AND,
|
||||||
|
C.SLLI, C.SRLI, C.SRAI, C.ANDI, C.BEQZ, C.BNEZ, C.J, C.JR).
|
||||||
|
- **SB/global symbols** — `MOV $sym(SB)`, `MOV sym(SB)`, `MOV rd, sym(SB)`
|
||||||
|
encoded as AUIPC pairs with R_RISCV_PCREL_HI20/LO12 relocations.
|
||||||
|
- **GLOBL/DATA** — data section layout in `AssembleFileRISCV`.
|
||||||
|
- **ELF64 emission** — `gasm asm --format elf` produces EM_RISCV objects
|
||||||
|
(.text, .data, .symtab, .rela.text).
|
||||||
|
- **`gasm verify --ground-truth`** — byte-exact comparison against
|
||||||
|
`GOARCH=riscv64 go tool asm`.
|
||||||
|
- **`gasm verify --profile`** — function layout listing for RISC-V.
|
||||||
|
- **CALL** — AUIPC + JALR pair encoding.
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- Parser: bare-number offset before `(SP)` no longer misidentified as pseudo.
|
||||||
|
- MOV: `MOV $sym(FP/SP), rd` now returns an explicit error instead of silent fallback.
|
||||||
|
- RVC: C.LDSP/C.SDSP/FLDSP/FSDSP immediate encoding now matches Go toolchain
|
||||||
|
(bit-interleaved format).
|
||||||
|
|
||||||
|
### Verified
|
||||||
|
|
||||||
|
- 118 RISC-V tests, asm coverage 83.3%.
|
||||||
|
- Ground-truth: C.LDSP, C.SDSP, C.FLDSP, C.FSDSP byte-exact vs Go toolchain.
|
||||||
|
|
||||||
|
## [0.27.0] — 2026-08-01
|
||||||
|
|
||||||
|
Subprocess isolation for `--fuzz`: each function is fuzzed in its own child
|
||||||
|
process, so a partial function (decoder) that faults on random garbage is
|
||||||
|
reported as "CRASH (partial function, use --ground-truth)" without killing
|
||||||
|
the parent. CRASH is informational (exit 0); only MISMATCH is an error.
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- `gasm verify --fuzz` no longer crashes the process on partial functions.
|
||||||
|
|
||||||
|
## [0.26.0] — 2026-07-31
|
||||||
|
|
||||||
|
Universal differential fuzzing: `gasm verify --fuzz` needs no hand-written
|
||||||
|
reference. It parses the `// func` signature from the assembly source,
|
||||||
|
generates typed random inputs (slices with random content, ints, pointers to
|
||||||
|
fixed arrays), JIT-executes BOTH the gasm-assembled and the go-tool-asm-
|
||||||
|
assembled versions with independent buffer copies, and compares the result
|
||||||
|
area bit-for-bit.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `verify`: `FuzzFunc` / `ExtractSignatures` / `parseFuncSig` — universal
|
||||||
|
differential fuzz driven by the conventional `// func` comment. Each
|
||||||
|
version gets its own buffer set (deep copy) so functions that write to
|
||||||
|
their arguments (histogram increments) don't corrupt the other's input.
|
||||||
|
- `gasm verify --fuzz [-n N]`: runs the differential fuzz for every function
|
||||||
|
with a parseable signature. Total functions (wideCopy, pack16, decorrelate,
|
||||||
|
analyze, autocorr) pass; partial functions (decoders that fault on malformed
|
||||||
|
input) should use `--ground-truth` instead.
|
||||||
|
|
||||||
|
### Known limitation
|
||||||
|
|
||||||
|
`--fuzz` crashes the process for partial functions (e.g. LZ4 decoders) whose
|
||||||
|
over-copy paths read past the buffer on random garbage input. Subprocess
|
||||||
|
isolation (fork per function) is planned. Use `--ground-truth` for decoders.
|
||||||
|
|
||||||
|
## [0.25.0] — 2026-07-30
|
||||||
|
|
||||||
|
Universal ground-truth verification: `gasm verify --ground-truth` assembles
|
||||||
|
any `.s` file with both gasm and `go tool asm`, then compares the machine
|
||||||
|
code byte-for-byte per function (relocation sites masked). No hand-written
|
||||||
|
reference needed — the Go toolchain IS the oracle.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `verify`: `GroundTruth` — shells out to `go tool asm`, parses the GOOBJ
|
||||||
|
output (minimal reader: block offsets, nonpkg symbol table, data index)
|
||||||
|
and returns per-function code bytes.
|
||||||
|
- `gasm verify --ground-truth`: compares gasm's output against the Go
|
||||||
|
assembler's, reporting MATCH/MISMATCH per function with the first
|
||||||
|
differing byte. Relocation disp32 fields (static-symbol references the
|
||||||
|
linker fills) are masked before comparison.
|
||||||
|
- Verified: go-lz4 AVX2 2/2, go-flac AVX2 17/17 functions byte-identical.
|
||||||
|
|
||||||
|
## [0.24.0] — 2026-07-29
|
||||||
|
|
||||||
|
The full analyze family and stereo PCM decode are now differentially tested.
|
||||||
|
15 of 17 go-flac AVX2 kernels have bit-for-bit differential coverage; the
|
||||||
|
two remaining (autocorrAVX2 — FMA reassociation, lpcResidualAVX2 — complex
|
||||||
|
multi-arg) are deferred.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `verify`: `analyzeO3RangeAVX2` and `analyzeO4RangeAVX2` differential tests
|
||||||
|
(200 iterations each, same harness as O1/O2/Res).
|
||||||
|
- `verify`: `decodeStereo16AVX2` differential test (500 random interleaved
|
||||||
|
stereo PCM buffers, both channels compared sample-by-sample).
|
||||||
|
|
||||||
|
## [0.23.0] — 2026-07-28
|
||||||
|
|
||||||
|
The analyze family and 24-bit PCM decode join the differential suite.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `verify`: `analyzeO2RangeAVX2` and `analyzeResRangeAVX2` differential
|
||||||
|
tests (200 iterations each, shared harness with O1: zigzag fold, partial
|
||||||
|
sum, overflow flag and Len32 histogram).
|
||||||
|
- `verify`: `decodeMono24AVX2` differential test (500 random 24-bit PCM
|
||||||
|
buffers, sign-extension compared sample-by-sample).
|
||||||
|
|
||||||
|
## [0.22.0] — 2026-07-27
|
||||||
|
|
||||||
|
The remaining go-flac encoder kernels join the differential suite.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `verify`: `analyzeO1RangeAVX2` differential test (300 random partitions:
|
||||||
|
zigzag fold, partial sum, overflow flag and the 32-bin Len32 histogram
|
||||||
|
compared element-by-element against the portable Go reference).
|
||||||
|
- `verify`: `fastStereoSumsAVX2` differential test (300 random stereo
|
||||||
|
frames: the four zigzag-fold entropy sums compared against the scalar
|
||||||
|
loop).
|
||||||
|
|
||||||
|
### Verified
|
||||||
|
|
||||||
|
- `gasm fmt` doc-comment indentation confirmed correct: comments before
|
||||||
|
every TEXT are at column 0 (the RET-detection logic handles multi-exit
|
||||||
|
functions).
|
||||||
|
|
||||||
|
## [0.21.0] — 2026-07-26
|
||||||
|
|
||||||
|
Differential testing extended to all four production kernels and the CLI
|
||||||
|
exposes the full dynamic-analysis toolkit.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `verify`: go-flac AVX2 differential tests — `decodeMono16AVX2` (500
|
||||||
|
random PCM buffers), `pack16AVX2` (500 random int32→int16 packings) and
|
||||||
|
all four decorrelation kernels (200 iterations each: left-side, side-right,
|
||||||
|
mid-side, interleave) compared bit-for-bit against the portable Go
|
||||||
|
references.
|
||||||
|
- `verify`: go-lz4 AVX-512 differential tests — `decodeBlockAVX512` (3 000
|
||||||
|
fuzzed LZ4 blocks + known answers) and `wideCopyAVX512` (0–1024 bytes)
|
||||||
|
against the same portable oracle as the AVX2 suite.
|
||||||
|
- `gasm verify --abi`: runs each NOSPLIT function with sentinel registers
|
||||||
|
and a red-zone canary, reporting violations.
|
||||||
|
- `gasm verify --profile`: lists the static basic-block count per function.
|
||||||
|
|
||||||
|
## [0.20.0] — 2026-07-25
|
||||||
|
|
||||||
|
Coverage profiling: the third pillar of Phase 3. Static basic-block
|
||||||
|
enumeration from the assembler's label map, combined with multi-input path
|
||||||
|
diversity measurement — how many observationally distinct execution paths a
|
||||||
|
test corpus exercises.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `verify`: `Kernel.Blocks` / `Kernel.BlockCount` — enumerate basic blocks
|
||||||
|
from the assembler's local-label map (every jump target is a block
|
||||||
|
boundary; the function entry is always a block). `decodeBlockAVX2` has
|
||||||
|
27 blocks.
|
||||||
|
- `verify`: `Kernel.ProfilePaths` — run the function with a corpus of
|
||||||
|
argument blocks and collect distinct output fingerprints (the result
|
||||||
|
words); reports path diversity as a lower bound on code coverage.
|
||||||
|
|
||||||
|
### Note
|
||||||
|
|
||||||
|
INT3-based per-block hit counting was prototyped but deferred: Go's runtime
|
||||||
|
signal management (sigaltstack, handler re-installation) makes raw
|
||||||
|
rt_sigaction handlers fragile in a Go process. The static + path-diversity
|
||||||
|
approach delivers the project's goal (proving the SIMD path and tail handling
|
||||||
|
execute) without fighting the runtime.
|
||||||
|
|
||||||
|
## [0.19.0] — 2026-07-24
|
||||||
|
|
||||||
|
Runtime ABI checks: the second pillar of Phase 3. The JIT trampoline now
|
||||||
|
has an ABI-checking variant that sets sentinels in the callee-saved registers
|
||||||
|
(BP, R14) before entering the assembled function and verifies they survive on
|
||||||
|
return, plus a red-zone canary (128 bytes below SP filled with 0xA5) that
|
||||||
|
detects any illegal write below the stack pointer.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `verify`: `CallChecked` / `Kernel.CallFuncChecked` — ABI-checking JIT call
|
||||||
|
with sentinel registers and red-zone canary; returns an `ABIReport`
|
||||||
|
(BPClobbered, R14Clobbered, RedZoneHit).
|
||||||
|
- `verify`: the raw `leaveJITCheckedRaw` trampoline — a TEXT symbol with no
|
||||||
|
ABIInternal wrapper (address obtained via GLOBL/DATA), so the JIT
|
||||||
|
function's RET lands directly in the check code and sees the registers
|
||||||
|
exactly as the function left them.
|
||||||
|
- Tests: deliberate BP/R14 clobberers detected; both go-lz4 kernels
|
||||||
|
confirmed ABI-clean (BP preserved, R14 preserved, red zone intact).
|
||||||
|
|
||||||
|
## [0.18.0] — 2026-07-23
|
||||||
|
|
||||||
|
Differential testing: the JIT-assembled go-lz4 `decodeBlockAVX2` kernel is
|
||||||
|
fuzzed against a portable Go reference — 5 000 valid LZ4 blocks compared
|
||||||
|
bit-for-bit, plus 2 000 hostile (random garbage) inputs with matching error
|
||||||
|
codes. This is the automated form of the project's bit-identical contract.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `verify`: differential fuzz tests — a random LZ4 block generator produces
|
||||||
|
valid blocks (literals, overlapping matches, extension bytes) and the
|
||||||
|
JIT-assembled kernel's output is compared byte-for-byte against a portable
|
||||||
|
Go decoder; a hostile-input suite confirms error-code agreement on random
|
||||||
|
garbage (no crashes, same classification).
|
||||||
|
|
||||||
|
## [0.17.0] — 2026-07-22
|
||||||
|
|
||||||
|
Phase 3 begins: dynamic analysis. A JIT execution substrate that assembles
|
||||||
|
Plan 9 amd64 kernels into executable memory and calls them directly — pure Go
|
||||||
|
(stdlib only, `syscall.Mmap` + an assembly trampoline), no cgo, no external
|
||||||
|
toolchain.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `verify` package: JIT infrastructure — `Map` copies machine code into a
|
||||||
|
W^X memory mapping, `Call` invokes it through an ABI0 trampoline that
|
||||||
|
switches to a prepared stack and back. `Load`/`LoadSource`/`LoadAST`
|
||||||
|
parse, assemble and map a `.s` file in one step; `Kernel.CallFunc`
|
||||||
|
marshals the argument block and returns results.
|
||||||
|
- `gasm verify` subcommand: assembles a file, JIT-loads it and reports the
|
||||||
|
available functions; with `-smoke`, calls each NOSPLIT function with
|
||||||
|
zeroed arguments to confirm the trampoline works end-to-end.
|
||||||
|
- Integration tests: the go-lz4 `decodeBlockAVX2` and `wideCopyAVX2`
|
||||||
|
kernels (699 and 146 bytes) assemble, map and execute correctly —
|
||||||
|
known-answer LZ4 blocks decode bit-for-bit, wide copies of 0–1024 bytes
|
||||||
|
match, malformed input returns the correct error codes.
|
||||||
|
|
||||||
|
### Verified
|
||||||
|
|
||||||
|
- `just test` (race, 84.6 % total coverage, verify 82.2 %).
|
||||||
|
- `gasm verify` on both go-lz4 kernels: all functions JIT-load and
|
||||||
|
smoke-test clean.
|
||||||
|
|
||||||
|
## [0.16.0] — 2026-07-21
|
||||||
|
|
||||||
|
The scalar conversions between vector and general-purpose registers — the
|
||||||
|
last of the amd64 EVEX instruction set.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `asm`: the GPR-interchanging conversions, byte for byte against the Go
|
||||||
|
assembler (28 ground-truth cases including memory sources and extended
|
||||||
|
GPRs): vector to GPR — the signed and truncated VCVT{,T}S{D,S}2SI{,Q}
|
||||||
|
in both VEX and EVEX, and the unsigned VCVT{,T}S{D,S}2USI{L,Q}
|
||||||
|
(EVEX only); GPR to vector — VCVTSI2SD{L,Q}/VCVTSI2SS{L,Q} (VEX and
|
||||||
|
EVEX) and VCVTUSI2SD{L,Q}/VCVTUSI2SS{L,Q} (EVEX only), whose preserved
|
||||||
|
vector source sits in vvvv (three Plan 9 operands).
|
||||||
|
|
||||||
|
## [0.15.0] — 2026-07-20
|
||||||
|
|
||||||
|
The last of the EVEX conversions and narrowing/extending moves — the EVEX
|
||||||
|
instruction set is now complete save for the GPR-interchanging forms.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `asm`: the unsigned and truncating conversions — VCVTPD2PS (and the X/Y
|
||||||
|
spellings, whose length the spelling fixes), VCVTPD2UDQ (X/Y),
|
||||||
|
VCVTTPD2UDQ (X/Y), VCVTTPD2UQQ, VCVTPS2UDQ, VCVTTPS2UDQ, VCVTPS2UQQ,
|
||||||
|
VCVTTPS2UQQ, VCVTTPD2QQ, VCVTTPS2QQ, VCVTUQQ2PD, VCVTUQQ2PS (X/Y) and
|
||||||
|
VCVTQQ2PS X/Y.
|
||||||
|
- `asm`: the remaining sign/zero-extending moves (VPMOVSXBD/BQ/WQ and
|
||||||
|
VPMOVZXBD/BQ/WD/WQ, VEX and EVEX) and the complete signed and unsigned
|
||||||
|
narrowing stores (VPMOVS{DB,QB,DW,QW,QD,WB}, VPMOVUS{DB,QB,DW,QW,QD,WB},
|
||||||
|
VPMOVDB, VPMOVQW).
|
||||||
|
- `asm`: the mask/vector conversions (VPMOVM2B/W/D/Q and VPMOVB2M/W2M/
|
||||||
|
D2M/Q2M), whose K register is a genuine operand rather than a mask and
|
||||||
|
which therefore take no masking suffixes.
|
||||||
|
|
||||||
|
## [0.14.0] — 2026-07-19
|
||||||
|
|
||||||
|
The floating-point helper and conversion tail of the AVX-512 set, plus
|
||||||
|
gather and scatter with VSIB addressing — every encoding verified byte for
|
||||||
|
byte against the Go assembler.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `asm`: the floating-point helpers — reciprocals and reciprocal square
|
||||||
|
roots (VRCP14/VRSQRT14 PD/PS/SD/SS), exponents and mantissas (VGETEXP*,
|
||||||
|
VGETMANT*), scaling by powers of two (VSCALEF*), rounding (VRNDSCALE*),
|
||||||
|
reduction (VREDUCE*), immediate fixup (VFIXUPIMM*) and range selection
|
||||||
|
(VRANGE*), and floating-point class tests (VFPCLASSPD/PS X/Y/Z and
|
||||||
|
VFPCLASSSD/SS — a new immediate form whose reg field carries the opmask
|
||||||
|
destination).
|
||||||
|
- `asm`: **gather and scatter with VSIB addressing.** The gathers take
|
||||||
|
both Go spellings: the VEX form with a vector mask register (OP mask,
|
||||||
|
vsib, dst) and the EVEX form with an explicit K mask (OP vsib, K, dst),
|
||||||
|
where the EVEX L'L field follows the VSIB index register rather than the
|
||||||
|
data register (a ZMM index with an YMM destination encodes L'L = 10, as
|
||||||
|
the Go assembler emits). The scatters (VSCATTER*/VPSCATTER*) are EVEX
|
||||||
|
only (OP src, K, vsib). All eight gather and eight scatter widths.
|
||||||
|
- `asm`: the remaining conversions — VCVTQQ2PS (the 512-bit source sets
|
||||||
|
the length), VCVTPD2QQ/UQQ, VCVTPS2QQ, VCVTUDQ2PD/PS, the half-precision
|
||||||
|
VCVTPH2PS and VCVTPS2PH (the extract layout with an immediate).
|
||||||
|
|
||||||
|
## [0.13.0] — 2026-07-18
|
||||||
|
|
||||||
|
The wider AVX-512 set: ternary logic, permutes, compares, expand/compress,
|
||||||
|
the opmask instructions and the EVEX rounding/SAE/broadcast suffixes — every
|
||||||
|
encoding verified byte for byte against the Go assembler.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `asm`: the wider EVEX/AVX-512 set, across roughly sixty new ground-truth
|
||||||
|
cases: ternary logic (VPTERNLOGD/Q), the lane shuffles/inserts/extracts
|
||||||
|
(VSHUF{F,I}{32,64}X{2,4}, the VINSERT*/VEXTRACT* {F,I}{32,64}X{2,4,8}
|
||||||
|
family, VPALIGNR), compares with an opmask destination (VCMPPD/PS/SD/SS —
|
||||||
|
a new NDS-plus-immediate form with the K register in the reg field), the
|
||||||
|
permutes (VPERMB/W, VPERMI2/T2 D/Q/PD), the wider integer families
|
||||||
|
(VPMADDWD/UBSW, VPMULHUW, VPACKSSWB/USWB/SSDW/USDW, VPABS B/W/D/Q, the
|
||||||
|
VPROL*/VPROR* rotates, the word shifts and the EVEX W1 qword shifts),
|
||||||
|
expand/compress (VEXPANDPD/PS, VPEXPANDD/Q, VCOMPRESSPD/PS, VPCOMPRESSD/
|
||||||
|
Q), the broadcasts (VPBROADCASTB/W from a GPR or memory, VBROADCASTSS/
|
||||||
|
SD), the opmask-register instructions (KAND/KOR/KXNOR/KADD/KUNPCK/KNOT/
|
||||||
|
KSHIFTL/KORTEST B/W/D/Q and KMOVQ, whose width the L/W/pp bits select),
|
||||||
|
the packed single arithmetic (VADD/VSUB/VMUL/VDIV/VMIN/VMAX PS), the
|
||||||
|
aligned moves (VMOVAPS/APD, VMOVDQA32/64, VMOVSS), the replicating moves
|
||||||
|
(VMOVSLDUP/VMOVSHDUP), the conversions (VCVTPS2DQ, VCVTTPS2DQ) and the
|
||||||
|
remaining extending and narrowing moves (VPMOVSXBW, VPMOVZXBW, VPMOVWB,
|
||||||
|
VPMOVQB).
|
||||||
|
- `asm`: the EVEX mnemonic suffixes the Go assembler accepts — the rounding
|
||||||
|
modes `.RN_SAE`, `.RD_SAE`, `.RU_SAE`, `.RZ_SAE` (the EVEX b bit with the
|
||||||
|
rounding control in L'L), suppress-all-exceptions `.SAE`, and memory
|
||||||
|
broadcast `.BCST` (the b bit, the vector length preserved, disp8×N scaled
|
||||||
|
by the element size) — each combinable with the `.Z` zeroing suffix,
|
||||||
|
validated against the Go assembler's bytes, and rejected on instructions
|
||||||
|
that do not support them.
|
||||||
|
|
||||||
|
## [0.12.0] — 2026-07-17
|
||||||
|
|
||||||
|
GOOBJ emission: gasm-assembled functions drop into a `go build` without the
|
||||||
|
Go assembler.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `asm`: **GOOBJ object output.** `gasm asm --format goobj -p <pkgpath>`
|
||||||
|
writes the Go toolchain's own object format — the one `cmd/link` consumes
|
||||||
|
directly: the functions as non-package symbols qualified with the package
|
||||||
|
path (exactly as `cmd/asm` records assembly symbols), the `GLOBL` data,
|
||||||
|
one serialized `FuncInfo` per function (argument/frame sizes, the asm
|
||||||
|
func flag, the start line, the file table) and the four pc-value tables
|
||||||
|
(`pcsp`, `pcfile`, `pcline`, `pcinline`). The `pcsp` table carries the
|
||||||
|
real stack deltas: the assembler now tracks every stack-adjustment
|
||||||
|
boundary through the prologue (`PUSHQ BP`, `SUBQ $frame, SP`) and each
|
||||||
|
`RET`'s epilogue, so frame-pointer functions unwind correctly. The
|
||||||
|
object preamble — the version-and-experiment header the linker compares
|
||||||
|
verbatim — is captured from the installed `go tool asm`, so the output is
|
||||||
|
always consistent with the toolchain that links it.
|
||||||
|
- `asm`: relocations against file-local `GLOBL` symbols become `R_PCREL`
|
||||||
|
entries in the GOOBJ output, with the instruction's displacement field
|
||||||
|
left zero for the linker to fill (as `cmd/asm` leaves it).
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- `parser`: 64-bit `DATA` literals above `MaxInt64`
|
||||||
|
(`DATA mask<>+8(SB)/8, $0x800f…`) parse as unsigned and keep their bit
|
||||||
|
pattern, instead of being rejected as non-integer.
|
||||||
|
|
||||||
|
### Verified
|
||||||
|
|
||||||
|
- End-to-end: a gasm-emitted GOOBJ swapped into a `go build` in place of
|
||||||
|
the toolchain's assembly object links and runs with output identical to
|
||||||
|
the baseline binary (stack-argument calls and a `GLOBL` relocation
|
||||||
|
resolved by the Go linker). All 17 go-flac AVX2 kernel functions emit as
|
||||||
|
a GOOBJ that `go tool nm` reads back with every symbol intact.
|
||||||
|
|
||||||
|
## [0.11.0] — 2026-07-16
|
||||||
|
|
||||||
|
Linkable object output: external symbols and relocatable ELF / Mach-O
|
||||||
|
objects.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `asm`: **object-file emission.** `gasm asm --format elf` writes an
|
||||||
|
ELF64 relocatable object and `--format macho` a Mach-O x86-64
|
||||||
|
`MH_OBJECT`: a code section (`.text` / `__TEXT,__text`) and a data
|
||||||
|
section (`.data` / `__DATA,__data`), a symbol table with one symbol per
|
||||||
|
`TEXT` and `GLOBL` (file-local `<>` symbols local, the rest global), and
|
||||||
|
one PC-relative relocation per static-symbol reference
|
||||||
|
(`R_X86_64_PC32` / `X86_64_RELOC_SIGNED`, the −4 addend the form needs).
|
||||||
|
The ELF output is verified end-to-end: a gasm-emitted object links with
|
||||||
|
a C driver and runs, resolving both a file-local constant and an
|
||||||
|
external symbol; the Mach-O output is verified structurally with
|
||||||
|
`debug/macho`.
|
||||||
|
- `asm`: **external symbol references.** A reference to a symbol no
|
||||||
|
`GLOBL` in the file defines no longer aborts assembly — it is recorded
|
||||||
|
as an external relocation (`Image.Externals`, `FuncLayout.Relocs`) and
|
||||||
|
becomes an undefined global symbol in the object output. The raw image
|
||||||
|
format (`--format raw`, the default) still reports them: only an object
|
||||||
|
file can represent a reference the linker must resolve.
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- `gasm asm` takes a `--format raw|elf|macho` flag selecting what `-o`
|
||||||
|
writes; without `--format` the behaviour is unchanged (the concatenated
|
||||||
|
image).
|
||||||
|
|
||||||
|
## [0.10.0] — 2026-07-15
|
||||||
|
|
||||||
|
The EVEX floating-point and conversion set: the packed-double arithmetic,
|
||||||
|
the scalar SD/SS forms, VMOVDDUP and the width-changing conversions, each
|
||||||
|
verified byte for byte against the Go assembler.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `asm`: the rest of the common EVEX/VEX floating-point set — packed double
|
||||||
|
arithmetic (VSUBPD, VDIVPD, VMINPD, VMAXPD, VUNPCKLPD and the EVEX form of
|
||||||
|
VUNPCKHPD), the scalar double and single operations (VSUBSD, VDIVSD,
|
||||||
|
VMINSD, VMAXSD and the full VADDSS/VSUBSS/VMULSS/VDIVSS/VMINSS/VMAXSS
|
||||||
|
family in both VEX and EVEX — the EVEX scalar forms exist for masked and
|
||||||
|
zeroing use), and VMOVDDUP (lane duplication, VEX and EVEX).
|
||||||
|
- `asm`: the width-changing conversions — VCVTDQ2PS and VCVTPS2PD (VEX and
|
||||||
|
EVEX; the destination sets the length for PS→PD), the EVEX form of
|
||||||
|
VCVTDQ2PD, and the packed-double → dword family: VCVTPD2DQ/VCVTTPD2DQ
|
||||||
|
(EVEX-512 only, a ZMM source and an XMM destination) and their X/Y
|
||||||
|
spellings (VCVTPD2DQX/Y, VCVTTPD2DQX/Y), whose length follows the wider
|
||||||
|
source — a new operand form, since the destination is always XMM while
|
||||||
|
VEX.L / EVEX.L'L ride with the source (fixed by the spelling even for a
|
||||||
|
memory source).
|
||||||
|
- `asm`: masking and zeroing on every new form — the scalar SD/SS
|
||||||
|
arithmetic, the unpacks, VMOVDDUP and the conversions all accept the
|
||||||
|
explicit K1–K7 operand and the `.Z` suffix the way Go writes them.
|
||||||
|
|
||||||
|
### Documented
|
||||||
|
|
||||||
|
- VCVTPS2PD follows the Go assembler's encoding, which omits the F3
|
||||||
|
mandatory prefix (VEX.pp / EVEX.pp = 00) that Intel's maps prescribe; the
|
||||||
|
Go toolchain's machine code is the project's byte-for-byte oracle, and
|
||||||
|
gasm reproduces it exactly (and round-trips through the x86 decoder, which
|
||||||
|
shares the convention).
|
||||||
|
|
||||||
|
### Verified
|
||||||
|
|
||||||
|
- 58 new ground-truth cases — every instruction extracted from the Go
|
||||||
|
toolchain's own assembly (go build + an executable-segment dump), checked
|
||||||
|
byte for byte and round-tripped through the decoder, covering disp8×N for
|
||||||
|
the scalar (×8/×4), duplication (×8/×32/×64) and conversion (×8/×16/×32)
|
||||||
|
memory operands, the 5-bit register fields and the masked/zeroing P2
|
||||||
|
byte. All four go-flac/go-lz4 kernels still assemble byte-identically
|
||||||
|
and lint clean.
|
||||||
|
|
||||||
|
## [0.9.0] — 2026-07-14
|
||||||
|
|
||||||
|
AVX-512 masking and a wider EVEX integer set.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `asm`: **EVEX masking** the way Go writes it — an explicit `K1`–`K7`
|
||||||
|
operand placed among the operands (merging mask), and a `.Z` mnemonic
|
||||||
|
suffix for zeroing (`VPADDD.Z Z1, Z2, K2, Z3`). Supported across the NDS,
|
||||||
|
reg/rm, immediate-shift, align, extract, convert and move forms, including
|
||||||
|
masked comparisons with a K destination (`VPCMPEQD Z0, Z3, K2, K1`). K0 is
|
||||||
|
rejected as an explicit mask, and `.Z` without a mask is an error, matching
|
||||||
|
the Go assembler.
|
||||||
|
- `asm`: the common AVX-512 F/BW integer set — VPADDB/W, VPSUBB/W, VPANDD/Q,
|
||||||
|
VPANDND/Q, VPMULLW, VPAVGB/W, the signed/unsigned min/max family for
|
||||||
|
B/W/D/Q elements, the variable shifts VPSLLVD/Q, VPSRLVD/Q, VPSRAVD/Q, the
|
||||||
|
EVEX forms of VPSHUFD/VPSHUFB, and the VMOVDQU8/VMOVDQU16 move aliases.
|
||||||
|
Register indices 16–31 encode correctly (the mod=11 quirk carries rm[4]
|
||||||
|
in X̄). All verified byte for byte against the Go assembler.
|
||||||
|
- `lint`: masked EVEX forms (`.Z` suffix, K operands) are recognised by
|
||||||
|
`unknown-instruction` and exempted from `operand-count`.
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- `asm`: EVEX register–register operands with indices 16–31 encoded rm[4]
|
||||||
|
into B̄ instead of X̄ (the EVEX mod=11 extension quirk), producing wrong
|
||||||
|
prefix bytes for X16+/Y16+ r/m operands.
|
||||||
|
|
||||||
|
## [0.8.0] — 2026-07-13
|
||||||
|
|
||||||
|
Standard CLI ergonomics.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `gasm --help` prints a proper top-level help (description, commands,
|
||||||
|
flags, examples), and every subcommand now answers `-h`/`--help` with its
|
||||||
|
own usage block (usage line, description, flag defaults), exiting 0. An
|
||||||
|
unknown command points at `gasm --help` instead of dumping the whole usage.
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- The version is primarily available as the standard `gasm --version` / `-V`
|
||||||
|
flag; the `gasm version` spelling remains as an alias.
|
||||||
|
|
||||||
|
## [0.7.0] — 2026-07-12
|
||||||
|
|
||||||
|
The formatter behaves like `go fmt` and canonicalises block separation.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `gasm fmt` now works like `go fmt`: with no arguments — or with a directory
|
||||||
|
argument — it reformats every `.s` file below it in place and lists the
|
||||||
|
changed files, skipping `.` and `_` directories (`.git`, `_refs`, …).
|
||||||
|
Explicit file arguments keep the `-w` / standard-output behaviour.
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- `format`: canonical blank-line layout — a new block (a label, `TEXT` or
|
||||||
|
`GLOBL`) is preceded by exactly one blank line, neither more nor less.
|
||||||
|
Comments leading a block stay with it (the blank line goes before them),
|
||||||
|
stacked labels share their block, the function's first label keeps hugging
|
||||||
|
its `TEXT`, and runs of blank lines collapse to one. The output remains
|
||||||
|
idempotent and round-trips through the parser. All four go-flac/go-lz4
|
||||||
|
kernels were reformatted with this release and remain byte-identical when
|
||||||
|
assembled.
|
||||||
|
|
||||||
|
## [0.6.0] — 2026-07-11
|
||||||
|
|
||||||
|
Calibrated to the Go ABI: `register-clobber` stops reporting legal code, and
|
||||||
|
the encoder learns the legacy SSE moves.
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- `lint`: **`register-clobber` is now calibrated to the Go ABI**
|
||||||
|
(`cmd/compile/abi-internal.md`), not the platform ABI. Go's stack-based
|
||||||
|
ABI0 has no System V style callee-saved registers — amd64 `BX`, `R12`–`R15`
|
||||||
|
and the arm64/riscv64/loong64 scratch sets are caller-saved or permanent
|
||||||
|
scratch, and hand-written kernels may clobber them freely. The rule now
|
||||||
|
audits only the registers Go fixes across calls: the frame pointer and the
|
||||||
|
goroutine pointer (amd64 `BP`/`R14`, arm64 `R18`/`R28`/`R29`, riscv64
|
||||||
|
`X27`, loong64 `R22`), and the goroutine pointer is reported only when the
|
||||||
|
function can reach the runtime (is not `NOSPLIT` or makes a call) — the
|
||||||
|
ABI0 transition restores it on those paths, and NOSPLIT call-free leaves
|
||||||
|
may use it, exactly as the runtime's own assembly does. Both go-flac
|
||||||
|
kernels now lint with zero diagnostics.
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- `lint`: the liveness analysis took the destination operand to be the
|
||||||
|
*first* operand on arm64, riscv64 and loong64; Plan 9 spelling puts it last
|
||||||
|
on every architecture Go supports. The def/use and save/restore
|
||||||
|
classification on those architectures was inverted.
|
||||||
|
- `format`: a comment that follows a `RET` (typically the next function's doc
|
||||||
|
comment) is no longer indented as if it were still inside the finished
|
||||||
|
function body.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `asm`: the legacy (non-VEX) SSE moves — `MOVOU`/`MOVO` (the Plan 9 names
|
||||||
|
for MOVDQU/MOVDQA), `MOVUPS`/`MOVAPS`/`MOVUPD`/`MOVAPD` and the scalar
|
||||||
|
`MOVSD`/`MOVSS` — and `VMOVDQU64` in the EVEX set. All verified byte for
|
||||||
|
byte against the Go assembler.
|
||||||
|
|
||||||
|
## [0.5.0] — 2026-07-10
|
||||||
|
|
||||||
|
EVEX / AVX-512: the go-flac AVX-512 kernel now assembles, byte-identically to
|
||||||
|
the Go toolchain, completing the production-kernel coverage.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `asm`: **EVEX (AVX-512) encoding** — the four-byte EVEX prefix with the
|
||||||
|
5-bit register fields (Z0–Z31, X/Y 16–31, with the reg-r/m X̄ quirk and
|
||||||
|
V'̄ shared between vvvv and the SIB index), opmask registers (K0–K7) as
|
||||||
|
operands and as mask destinations, and the compressed disp8×N displacement
|
||||||
|
(the multiplier follows the memory operand's size, as the Go assembler's
|
||||||
|
opcode tables prescribe). Covers every AVX-512 instruction the go-flac
|
||||||
|
kernels use: VPXORD/Q, VPADDD, VPSUBD/Q, VPUNPCK*DQ, VPMULLD/Q, VPERMD,
|
||||||
|
VPSLLD/VPSRAD/VPSRAQ, VALIGND, VPCMPEQD (K destination), VMOVDQU32,
|
||||||
|
VMOVUPD, VCVTQQ2PD, VPMOVSXDQ, the narrowing stores VPMOVDW/VPMOVQD, the
|
||||||
|
extracts VEXTRACTI64X4/VEXTRACTF64X4, VFMADD231PD, VADDPD, VMULPD, the
|
||||||
|
broadcasts VPBROADCASTD/Q (GPR and memory sources take different opcodes)
|
||||||
|
and the mask moves KMOVW/KTESTW. Masking/zeroing suffixes are out of scope
|
||||||
|
— the kernels use neither.
|
||||||
|
- `asm`: `AssembleFile` now accepts file-defined global (`non-<>`) symbols
|
||||||
|
too; a reference is external only when no `GLOBL` in the file defines it.
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- `asm`: registers X16–Y31 force the EVEX encoding of dual-form mnemonics;
|
||||||
|
previously a `VPBROADCASTD AX, Y30` fell into the VEX encoder, which cannot
|
||||||
|
represent indices above 15 and silently truncated them.
|
||||||
|
- `asm`: the VEX encoder now rejects vector register indices 16–31 instead of
|
||||||
|
encoding a truncated (wrong) register.
|
||||||
|
|
||||||
|
### Verified
|
||||||
|
|
||||||
|
- All 10 functions of the go-flac `avx512_amd64.s` kernel assemble
|
||||||
|
byte-identically to the Go toolchain's machine code (the disp32 of the one
|
||||||
|
`VMOVDQU32 idx16(SB), Z13` load is linker-filled in Go and resolved within
|
||||||
|
gasm's own image — checked to reach the right constant bytes). The AVX2
|
||||||
|
kernel's 17 functions remain byte-identical.
|
||||||
|
|
||||||
|
## [0.4.0] — 2026-07-09
|
||||||
|
|
||||||
|
The standalone assembler reaches the whole go-flac AVX2 kernel: static
|
||||||
|
symbols assemble, and all 17 kernel functions now match the Go toolchain's
|
||||||
|
machine code byte for byte.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `asm`: **file-level assembly** — `AssembleFile` turns a parsed file into an
|
||||||
|
`Image`: the function bodies in source order followed by a data section
|
||||||
|
built from the file's `GLOBL`/`DATA` directives (each symbol 16-aligned).
|
||||||
|
- `asm`: **static-symbol (`SB`) operands** — `mask<>(SB)` references encode as
|
||||||
|
RIP-relative loads with a patched disp32, resolved against the image layout
|
||||||
|
so the output is self-consistent and position-independent. External
|
||||||
|
(non-file-local) symbols are rejected with a clear error: they need
|
||||||
|
object-file emission.
|
||||||
|
- `gasm asm` prints the data section and symbol map alongside the functions
|
||||||
|
and writes the whole image (code + data) with `-o`.
|
||||||
|
|
||||||
|
### Verified
|
||||||
|
|
||||||
|
- All 17 functions of the go-flac `avx2_amd64.s` kernel assemble
|
||||||
|
byte-identically to the Go toolchain's machine code; the only differing
|
||||||
|
bytes are the displacements of the two `VMOVDQU mask24<>(SB), X15` loads,
|
||||||
|
which the Go linker fills at link time and gasm resolves within its own
|
||||||
|
image (checked to reach the right constant bytes).
|
||||||
|
|
||||||
|
## [0.3.0] — 2026-07-08
|
||||||
|
|
||||||
|
The assembler reaches byte-identical parity with the Go toolchain on the
|
||||||
|
production go-flac AVX2 kernels: every one of the 15 kernel functions that
|
||||||
|
avoid global symbols now assembles to exactly the Go assembler's bytes (the
|
||||||
|
two holdouts load a file-local constant through `SB` and wait on relocation
|
||||||
|
support).
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `asm`: the scalar instruction families the kernels use — `CMOVcc` and
|
||||||
|
`SETcc` (conditions spelled exactly like the jumps), `LZCNT`/`TZCNT`
|
||||||
|
(legacy `F3 0F BD/BC`), the sign/zero-extending moves (`MOVBLZX`, `MOVBQZX`,
|
||||||
|
`MOVWLZX`, `MOVWQZX`, `MOVWLSX`, `MOVLQSX`), `CVTSL2SD`/`CVTSQ2SD` (the
|
||||||
|
legacy SSE encoding, as the Go assembler emits it), the traditional
|
||||||
|
three-operand `IMUL3{W,L,Q}`, and the variable-count vector shifts
|
||||||
|
(`VPSRLQ X0, Y8, Y8` — the count in an XMM register or memory takes the
|
||||||
|
ordinary NDS form).
|
||||||
|
- `asm`: **jump relaxation** — jumps start in the short (rel8) form and
|
||||||
|
expand to rel32 when the settled displacement does not fit, iterating the
|
||||||
|
layout to a fixed point (CALL is always rel32).
|
||||||
|
- `asm`: **jump-to-jump folding** — a conditional jump to a label whose only
|
||||||
|
instruction is an unconditional jump is redirected to the ultimate
|
||||||
|
target, replicating the Go toolchain's linker, which chases such chains
|
||||||
|
before it encodes branches.
|
||||||
|
- `parser`: leading negative displacements with a base and index
|
||||||
|
(`LEAQ -4(DX)(R9*4), R9`) parse into a fully populated address.
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- `asm`: `CMP` with a register or memory operand computed **second − first**
|
||||||
|
instead of first − second, silently inverting every condition that followed
|
||||||
|
(`CMPQ SI, R10; JGE` tested R10 ≥ SI). The encoding now always records
|
||||||
|
first − second — `CMP r/m, r` with the first operand in r/m, `CMP r, r/m`
|
||||||
|
with the first operand in reg — and is byte-identical to the Go assembler.
|
||||||
|
- `asm`: register-to-register `MOV` now uses the `r/m ← r` opcode (reg =
|
||||||
|
source), the Go assembler's choice; the output is byte-identical.
|
||||||
|
|
||||||
|
## [0.2.0] — 2026-07-07
|
||||||
|
|
||||||
|
The Phase 2 assembler grows the SIMD set: shuffles, extract/insert, permute
|
||||||
|
and the moves, on top of the Phase 1 VEX forms.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `asm`: four new VEX (AVX/AVX2) operand forms, each validated by round-trip
|
||||||
|
decoding through `golang.org/x/arch` **and** byte-for-byte against the
|
||||||
|
machine code the real Go assembler emits:
|
||||||
|
- the immediate shuffle (`VPSHUFD`, `VPERMQ`),
|
||||||
|
- the three-operand-plus-immediate form (`VSHUFPD`, `VPERM2I128`,
|
||||||
|
`VINSERTI128`),
|
||||||
|
- the lane extract (`VEXTRACTI128`, `VEXTRACTF128` — the YMM source occupies
|
||||||
|
the ModRM.reg field, the XMM/memory destination the r/m field),
|
||||||
|
- the direction-sensitive moves (`VMOVDQU`, `VMOVUPD`, `VMOVD`, `VMOVQ`,
|
||||||
|
`VMOVSD` — each direction picks its own opcode and VEX.W; a vector→vector
|
||||||
|
move uses the store-form layout, matching the Go assembler),
|
||||||
|
- the no-operand `VZEROUPPER`, and `VPERMD` in the NDS form,
|
||||||
|
- the floating-point and FMA set (`VADDPD`, `VMULPD`, `VXORPD`,
|
||||||
|
`VUNPCKHPD`, the scalar `VADDSD`/`VMULSD`, `VCVTDQ2PD`, `VFMADD231PD`).
|
||||||
|
With the scalar set and the earlier NDS / reg-rm / immediate-shift forms,
|
||||||
|
the encoder now covers every integer, shuffle and FP instruction the
|
||||||
|
go-flac AVX2 kernels use.
|
||||||
|
- `asm`: `CMP` accepts the immediate in the second operand position
|
||||||
|
(`CMPL CX, $31`) — the spelling the Go assembler accepts — encoding it
|
||||||
|
identically to the immediate-first form.
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- `asm`: an unused VEX.vvvv field is now stored as `1111` (v̄vvv = 1111), as
|
||||||
|
the hardware requires — the previous value (`0000`) made the two-operand
|
||||||
|
reg/rm forms (VPMOVSXWD, VPBROADCASTD, VMOVMSKPS, …) raise #UD on real CPUs
|
||||||
|
and differ from the Go assembler's bytes. The round-trip decoder ignores
|
||||||
|
the field on these instructions, which is why the byte-for-byte Go
|
||||||
|
comparison (added this release) is now part of the test suite.
|
||||||
|
|
||||||
|
## [0.1.0] — 2026-07-06
|
||||||
|
|
||||||
|
Initial release — the Phase 1 foundation.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- `token`, `lexer`, `ast`, `parser`: a hand-written, error-tolerant front end
|
||||||
|
for Plan 9 assembly. The lexer splices C-preprocessor line continuations
|
||||||
|
(`\` before a newline) so multi-line `#define` macros parse as one opaque
|
||||||
|
directive. Validated against the production AVX2/AVX-512 kernels in
|
||||||
|
`go-libraries/go-flac` and the Go runtime's `src/runtime/*.s` for all four
|
||||||
|
architectures, with zero parse errors.
|
||||||
|
- `arch`: register files and **complete** instruction tables for amd64,
|
||||||
|
arm64, riscv64 and loong64, with the middle-dot symbol separator and static
|
||||||
|
(`<>`) symbols. Instruction names are generated from the Go toolchain's own
|
||||||
|
assembler source (`just gen`) — the `anames` opcode lists plus the common
|
||||||
|
opcodes and the per-architecture front-end aliases (arm64 `B`/`BL`, the
|
||||||
|
`.P`/`.W` addressing suffixes, loong64 `JAL`, the x86 conditional-jump
|
||||||
|
spellings) — so every mnemonic the real assembler accepts is recognised.
|
||||||
|
- `lint`: conservative rules — `unknown-instruction`, `operand-count`,
|
||||||
|
`undefined-label`, `duplicate-label`, `missing-ret`,
|
||||||
|
`missing-textflag-include`, `abi-argsize` and `unreachable-code`. Macro
|
||||||
|
invocations are recognised (in-file `#define` names and underscore
|
||||||
|
identifiers) and the label/RET heuristics are suppressed in macro-using
|
||||||
|
files. `abi-argsize` parses the `// func` signature with the Go parser and
|
||||||
|
checks the declared TEXT argument size against Go's ABI0 layout;
|
||||||
|
`unreachable-code` flags dead code after `RET`, suppressed where reachability
|
||||||
|
is undecidable (PC-relative jumps, register-indirect branches, `#ifdef`).
|
||||||
|
Register liveness is computed by dataflow over the control-flow graph (basic
|
||||||
|
blocks, def/use, iterative backward iteration) and drives `register-clobber`,
|
||||||
|
an audit that flags a callee-saved register written but never saved/restored.
|
||||||
|
`funcdata-pcdata` validates the structure of `FUNCDATA`/`PCDATA` directives.
|
||||||
|
Zero error-severity diagnostics across the 90-file Go runtime corpus and the
|
||||||
|
production go-flac kernels (the `register-clobber` audit additionally reports
|
||||||
|
the go-flac kernels' unsaved callee-saved register use for review).
|
||||||
|
- `format`: an idempotent canonical formatter (operand spacing and per-function
|
||||||
|
mnemonic alignment) that preserves comments and round-trips through the
|
||||||
|
parser.
|
||||||
|
- `lsp`: a Language Server Protocol server over stdio providing completion,
|
||||||
|
hover documentation, document symbols, publish-diagnostics and semantic-token
|
||||||
|
highlighting.
|
||||||
|
- `asm`: a standalone amd64 (x86-64) assembler — an instruction encoder (REX/
|
||||||
|
ModR-M/SIB/displacement/immediate plus the scalar instruction set, and VEX/
|
||||||
|
AVX2 SIMD across three operand forms — NDS, reg/rm and immediate-shift —
|
||||||
|
covering the bulk of the integer SIMD set) validated by round-trip decoding
|
||||||
|
against `golang.org/x/arch`, and an assembler that drives the parser's AST
|
||||||
|
into the encoder with local-label resolution and `FP`/`SP` frame mapping
|
||||||
|
(plus Go prologue/epilogue generation), producing output byte-identical to the
|
||||||
|
Go assembler for the supported operand forms.
|
||||||
|
- `cmd/gasm`: the `gasm` binary with `tokens`, `parse`, `fmt`, `lint`, `asm`
|
||||||
|
and `lsp` subcommands.
|
||||||
|
- `_gen`: the generator that rebuilds the architecture instruction tables from
|
||||||
|
the Go toolchain source (`just gen`).
|
||||||
+101
@@ -0,0 +1,101 @@
|
|||||||
|
# Contributing to gasm-devkit
|
||||||
|
|
||||||
|
## Prerequisites
|
||||||
|
|
||||||
|
- Go 1.26 or later (`toolchain go1.26.5`)
|
||||||
|
- `just` command runner
|
||||||
|
- A Linux host on amd64, arm64, riscv64 or loong64
|
||||||
|
|
||||||
|
## Development Setup
|
||||||
|
|
||||||
|
```sh
|
||||||
|
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
|
||||||
|
cd gasm-devkit
|
||||||
|
just install # download module dependencies
|
||||||
|
just build # go vet + gofmt check
|
||||||
|
just test # full test suite with race detector
|
||||||
|
```
|
||||||
|
|
||||||
|
## Commands
|
||||||
|
|
||||||
|
Every just recipe:
|
||||||
|
|
||||||
|
| Recipe | What it does |
|
||||||
|
|--------|-------------|
|
||||||
|
| `just` | List all recipes |
|
||||||
|
| `just install` | `go mod download` |
|
||||||
|
| `just build` | `go vet ./...` + `gofmt -l .` check — zero errors required |
|
||||||
|
| `just test` | `go test -race -count=1 -coverprofile=coverage.out ./...` + 80 % coverage gate |
|
||||||
|
| `just fmt` | `gofmt -w .` |
|
||||||
|
| `just run -- lint file.s` | Run the CLI with `go run` (args after `--`) |
|
||||||
|
| `just install-bin` | Install `gasm` into `$GOBIN` with the release version stamped |
|
||||||
|
| `just gen` | Regenerate `arch/*_gen.go` instruction tables from the Go toolchain |
|
||||||
|
| `just uninstall` | Remove build artefacts (`coverage.out`, `gasm`, `*.test`) |
|
||||||
|
|
||||||
|
## Running a Single Test
|
||||||
|
|
||||||
|
```sh
|
||||||
|
go test -run TestVexGroundTruth ./asm/
|
||||||
|
go test -run TestDifferentialLZ4Fuzz ./verify/
|
||||||
|
```
|
||||||
|
|
||||||
|
## Testing the Debugger
|
||||||
|
|
||||||
|
The interactive debugger (`gasm debug`) requires a compiled binary —
|
||||||
|
`go run` does not work for the child process. Install first:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
just install-bin
|
||||||
|
gasm debug --func add testdata/verify/basic_amd64.s
|
||||||
|
```
|
||||||
|
|
||||||
|
## Code Style
|
||||||
|
|
||||||
|
See [AGENTS.md](AGENTS.md) for the full style guide. Key points:
|
||||||
|
|
||||||
|
- `gofmt` — zero diff.
|
||||||
|
- `go vet` — zero warnings.
|
||||||
|
- Standard library only in production code; `golang.org/x/arch` in tests.
|
||||||
|
- No cgo, no C, no JavaScript.
|
||||||
|
- Hand-written Plan 9 assembly; tables generated only via `_gen/gen.go`.
|
||||||
|
|
||||||
|
## Branches and Releases
|
||||||
|
|
||||||
|
- `development` is the working branch.
|
||||||
|
- `main` is release-only: `git merge --ff-only development`, then `git tag vX.Y.Z`.
|
||||||
|
- Conventional Commits: `feat(asm): add EVEX gather and scatter`.
|
||||||
|
- Every commit ends with `Assisted-by: <model-name>`.
|
||||||
|
|
||||||
|
## CI
|
||||||
|
|
||||||
|
CI runs on every push to `development` and on pull requests:
|
||||||
|
|
||||||
|
- **Test** (`test.yml`) — `gofmt` check, `go vet`, `go test -race` and the
|
||||||
|
80 % coverage gate.
|
||||||
|
- **Release** (`release.yml`) — cross-compiles release binaries for
|
||||||
|
linux/{amd64,arm64,riscv64,loong64} on version tags and publishes them.
|
||||||
|
|
||||||
|
The Definition of Done (`just build` + `just test` + `just fmt`) must still
|
||||||
|
pass locally before pushing.
|
||||||
|
|
||||||
|
## AI-Assisted Contributions
|
||||||
|
|
||||||
|
AI agents may assist with code, documentation, tests, and review. All
|
||||||
|
AI-assisted changes must:
|
||||||
|
|
||||||
|
- Include the trailer `Assisted-by: <model-name>` in the commit message
|
||||||
|
(e.g. `Assisted-by: DeepSeek V4 Pro`).
|
||||||
|
- Follow the [AGENTS.md](AGENTS.md) rules.
|
||||||
|
- Pass the Definition of Done before committing.
|
||||||
|
|
||||||
|
Attribute agent authorship in issues and pull requests on one trailing
|
||||||
|
line:
|
||||||
|
|
||||||
|
```
|
||||||
|
_Assisted-by: Qwen 3.8 Max_
|
||||||
|
```
|
||||||
|
|
||||||
|
## Questions
|
||||||
|
|
||||||
|
Open an issue at
|
||||||
|
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit/issues).
|
||||||
@@ -0,0 +1,370 @@
|
|||||||
|
# gasm-devkit
|
||||||
|
|
||||||
|
Developer tooling for **GAsm** — Go's built-in Plan 9 assembler.
|
||||||
|
|
||||||
|
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
|
||||||
|
|
||||||
|
Go ships an assembler but no tooling for it. There is no syntax highlighting,
|
||||||
|
no autocomplete, no linter, no static analyser, no formatter, no standalone
|
||||||
|
assembler and no debugger for `.s` files. Developers write assembly blind,
|
||||||
|
validate it by benchmark, and debug it by print statement.
|
||||||
|
|
||||||
|
gasm-devkit is the missing toolkit. It is a single, self-contained binary —
|
||||||
|
`gasm` — that brings proper developer tooling to Plan 9 assembly:
|
||||||
|
|
||||||
|
```
|
||||||
|
gasm tokens dump the lexical token stream
|
||||||
|
gasm parse parse and report syntax errors
|
||||||
|
gasm fmt canonicalise formatting (gofmt for assembly)
|
||||||
|
gasm lint static checks
|
||||||
|
gasm lsp language server (completion, hover, symbols, diagnostics, highlighting)
|
||||||
|
gasm asm standalone assembler (Phase 2)
|
||||||
|
gasm verify dynamic analysis & verification (Phase 3)
|
||||||
|
gasm debug source-level debugger (Phase 4)
|
||||||
|
gasm diff compare machine code of two .s files
|
||||||
|
gasm profile show basic-block structure of functions
|
||||||
|
```
|
||||||
|
|
||||||
|
> **Status: Phase 4 — done, Phase 5 underway.** Phase 1 (the language
|
||||||
|
> foundation, linter, formatter and language server) shipped in v0.1.0;
|
||||||
|
> Phase 2 (the standalone assembler — the full amd64 instruction set plus
|
||||||
|
> ELF and GOOBJ object emission) in v0.12.0; Phase 3 (dynamic
|
||||||
|
> analysis — JIT execution, differential testing, ABI checks and coverage
|
||||||
|
> profiling) in v0.25.0; Phase 4 (interactive debugger — ptrace-based,
|
||||||
|
> breakpoints, watchpoints, stepping, vector register display, named buffer
|
||||||
|
> allocation) in v0.27.0; RISC-V encoder (RV64IMAFDC + RVC, ELF emission,
|
||||||
|
> ground-truth, GOOBJ) in v0.28.0–v0.29.0; LoongArch encoder (the full
|
||||||
|
> instruction set with the MOV expansions, ELF and GOOBJ emission, and
|
||||||
|
> ground-truth verification) after v0.29.0. See [Roadmap](#roadmap).
|
||||||
|
|
||||||
|
## Architecture support
|
||||||
|
|
||||||
|
gasm-devkit targets every architecture Go's assembler speaks. The instruction
|
||||||
|
tables are **generated from the Go toolchain's own assembler source**
|
||||||
|
(`cmd/internal/obj/<arch>`), so gasm-devkit recognises *every* mnemonic the
|
||||||
|
real assembler accepts — not a hand-maintained subset that drifts and rots.
|
||||||
|
|
||||||
|
| Architecture | GOARCH | File suffix | Instructions recognised |
|
||||||
|
|--------------|-------------|----------------|------------------------------------|
|
||||||
|
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
|
||||||
|
| ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes |
|
||||||
|
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes |
|
||||||
|
| LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes |
|
||||||
|
|
||||||
|
"Common opcodes" are the instructions shared by every architecture (`RET`,
|
||||||
|
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, …). AMD64 additionally
|
||||||
|
carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
|
||||||
|
…) that the assembler accepts as aliases. Regenerating the tables is one
|
||||||
|
command — `just gen` — and requires only a Go installation; the committed
|
||||||
|
output has no runtime dependency on the toolchain.
|
||||||
|
|
||||||
|
## Supported Platforms
|
||||||
|
|
||||||
|
The toolkit runs on Linux. All four Linux architectures are supported as
|
||||||
|
hosts — amd64, arm64, riscv64 and loong64 — and the release matrix
|
||||||
|
cross-compiles the same four targets.
|
||||||
|
|
||||||
|
**FreeBSD support is planned for a future release.**
|
||||||
|
|
||||||
|
## Roadmap
|
||||||
|
|
||||||
|
The work is delivered in four phases. Each phase is completed and hardened
|
||||||
|
before the next begins. The ordering follows a dependency chain: understand
|
||||||
|
the code statically (Phase 1), make it runnable (Phase 2), then run it and
|
||||||
|
observe or control it (Phases 3–4).
|
||||||
|
|
||||||
|
### Phase 1 — language foundation, editor tooling and static analysis · *done*
|
||||||
|
|
||||||
|
Everything needed to read, understand, check, format and highlight GAsm —
|
||||||
|
without executing it.
|
||||||
|
|
||||||
|
| Capability | Status |
|
||||||
|
|------------|--------|
|
||||||
|
| Lexer — permissive, position-aware scanner for all four architectures | done |
|
||||||
|
| Parser — line-oriented, error-tolerant, full AST with source positions | done |
|
||||||
|
| Instruction + register tables for amd64, arm64, riscv64, loong64 (generated, complete) | done |
|
||||||
|
| Linter — `unknown-instruction`, `operand-count`, `undefined-label`, `duplicate-label`, `missing-ret`, `missing-textflag-include`, `abi-argsize`, `unreachable-code`, `register-clobber`, `funcdata-pcdata` | done |
|
||||||
|
| Formatter — idempotent, comment-preserving, per-function alignment; a `RET` terminates the body for indentation, so the next function's doc comment stays at column 0; exactly one blank line before every block (label, `TEXT`, `GLOBL`) and runs of blanks collapsed; directory / no-argument mode reformats every `.s` in place, `go fmt`-style | done |
|
||||||
|
| Language server — completion, hover, document symbols, diagnostics, semantic-token highlighting | done |
|
||||||
|
| CLI — `gasm tokens / parse / fmt / lint / lsp` | done |
|
||||||
|
| Real-world validation against production AVX2 / AVX-512 kernels | done |
|
||||||
|
| Lint hardening — zero false positives across the Go runtime corpus (90 files, all four architectures): macro-invocation handling, branch aliases (`B`/`BL`/`JAL`), addressing suffixes (`.P`/`.W`), terminal `UNDEF` | done |
|
||||||
|
| Static analysis — `abi-argsize` (argument/result area computed from the `// func` signature under Go's ABI0 layout and checked against the TEXT declaration) and `unreachable-code` (dead code after `RET`, suppressed where reachability is undecidable: PC-relative jumps, register-indirect branches, `#ifdef`) | done |
|
||||||
|
| Static analysis — register liveness (CFG construction + per-instruction def/use + iterative backward dataflow) driving `register-clobber`, calibrated to the **Go ABI** (not System V): flags writes to the registers Go fixes across calls — the frame pointer and the goroutine pointer (`R14` on amd64, `R28`/`R29` on arm64, `X27` on riscv64, `R22` on loong64, plus the OS-reserved `R18` on arm64) — that are never saved/restored; the goroutine pointer is reported only when the function can reach the runtime (not `NOSPLIT`, or makes calls), matching how the runtime's own assembly uses it. `funcdata-pcdata` structural validation of `FUNCDATA`/`PCDATA` operands and indices | done |
|
||||||
|
|
||||||
|
> **Limitation — macros.** gasm-devkit reads `.s` source as written; it does
|
||||||
|
> **not** run the C preprocessor, so `#define` macros are not expanded. Files
|
||||||
|
> that use macros (the runtime's `asm_*.s`, `race_*.s`, `sys_*.s`, …) parse
|
||||||
|
> cleanly, and macro *invocations* are recognised and never flagged, but the
|
||||||
|
> `undefined-label` and `missing-ret` heuristics are suppressed in macro-using
|
||||||
|
> files because labels a macro defines are invisible without expansion. Full
|
||||||
|
> macro expansion is future work (it pairs naturally with the Phase 2
|
||||||
|
> assembler). Hand-written, macro-free kernels — such as everything in
|
||||||
|
> `go-libraries` — are analysed in full.
|
||||||
|
|
||||||
|
### Phase 2 — standalone assembler · *done*
|
||||||
|
|
||||||
|
Assembly without the Go toolchain in the loop.
|
||||||
|
|
||||||
|
- **`gasm asm`:** a standalone assembler that turns a `.s` file into machine
|
||||||
|
code directly — pure Go, no `go build`, no external toolchain. Useful for
|
||||||
|
fast iteration, for environments without a full Go installation, and as the
|
||||||
|
execution substrate that Phases 3 and 4 build on.
|
||||||
|
|
||||||
|
Done so far:
|
||||||
|
|
||||||
|
- An amd64 (x86-64) **instruction encoder** — REX/ModR-M/SIB/displacement/
|
||||||
|
immediate machinery and the scalar instruction set (MOV, the ALU group, TEST,
|
||||||
|
LEA, INC/DEC/NEG/NOT, shifts, IMUL and IMUL3, PUSH/POP, JMP/CALL/Jcc,
|
||||||
|
CMOVcc, SETcc, LZCNT/TZCNT, the sign/zero-extending moves — MOVBLZX and
|
||||||
|
friends, MOVLQSX — and CVTSL2SD/CVTSQ2SD), validated by round-tripping
|
||||||
|
every encoding through `golang.org/x/arch`'s decoder and byte-for-byte
|
||||||
|
against the Go assembler.
|
||||||
|
- An **assembler** that drives the parser's AST into the encoder with local-
|
||||||
|
label resolution — jumps start in the short (rel8) form and expand to rel32
|
||||||
|
when the displacement does not fit, and jump-to-jump chains are folded the
|
||||||
|
way the Go toolchain folds them — so `gasm asm <file>` emits machine code
|
||||||
|
for each `TEXT` function.
|
||||||
|
- **File-level assembly with static data** — `GLOBL`/`DATA` symbols are laid
|
||||||
|
out in a data section behind the code and references to them (`mask<>(SB)`)
|
||||||
|
are encoded RIP-relative with the displacement resolved within the image,
|
||||||
|
so the output is self-consistent and position-independent. References to
|
||||||
|
symbols no `GLOBL` in the file defines are recorded as relocations and
|
||||||
|
carried into the object-file output.
|
||||||
|
- **GOOBJ emission** — `gasm asm --format goobj -p <pkgpath>` writes the Go
|
||||||
|
toolchain's own object format (the one `cmd/link` consumes directly), so
|
||||||
|
gasm-assembled kernels drop into a `go build` without the Go assembler:
|
||||||
|
the functions as non-package symbols, `GLOBL` data, one `FuncInfo` per
|
||||||
|
function and the pc-value tables (`pcsp` with the real prologue/epilogue
|
||||||
|
stack deltas, `pcfile`, `pcline`, `pcinline`). Verified end-to-end by
|
||||||
|
swapping a gasm-emitted object into a `go build` in place of the
|
||||||
|
toolchain's, linking and running — bit-identical behaviour.
|
||||||
|
- **Object-file emission** — `gasm asm --format elf` writes a relocatable
|
||||||
|
object (a `.text` and a `.data` section, a symbol table — file-local `<>`
|
||||||
|
symbols local, the rest global — and one `R_X86_64_PC32` relocation per
|
||||||
|
static-symbol reference) that links with the system toolchain: external
|
||||||
|
references resolve against undefined symbols, file-local ones against the
|
||||||
|
data section. Verified end-to-end by linking a gasm-emitted object with
|
||||||
|
a C driver and running it. RISC-V uses the equivalent `R_RISCV_PCREL_HI20`
|
||||||
|
/ `R_RISCV_PCREL_LO12_I` pair for AUIPC+JAL/JALR sequences.
|
||||||
|
- **`FP`/`SP` frame mapping** — the pseudo-registers are translated onto the
|
||||||
|
hardware stack pointer (`x+N(FP)` → `(N+8)(SP)` for a zero frame, `(N+frame+
|
||||||
|
16)(SP)` with a frame pointer; locals via `x-N(SP)`), and the Go-style
|
||||||
|
prologue/epilogue is generated for functions with a frame. The output is
|
||||||
|
**byte-identical to the Go assembler** for these cases (verified against
|
||||||
|
`go tool objdump`).
|
||||||
|
- **SIMD (VEX / AVX2)** — the VEX prefix machinery (2-byte C5 and 3-byte C4)
|
||||||
|
with XMM/YMM vector registers, validated by round-trip decoding **and**
|
||||||
|
byte-for-byte against the Go assembler's machine code, across eight operand
|
||||||
|
forms: the three-operand NDS form (VPADDD/Q, VPSUBD/Q, VPXOR, VPOR, VPAND/N,
|
||||||
|
VPCMPEQD, VPCMPGTQ, VPUNPCK*, VPMULLD, VPMULDQ, VPSHUFB, VPACKSSDW,
|
||||||
|
VPERMD), the two-operand reg/rm form (VPMOVSXWD/DQ, VPMOVZXDQ,
|
||||||
|
VPBROADCASTD/Q, VPMOVMSKB, VMOVMSKPS, VCVTDQ2PD), the immediate-shift and
|
||||||
|
variable-count shifts (VPSLLD/Q, VPSRAD, VPSRLD/Q with an immediate or an
|
||||||
|
XMM/memory count), the immediate shuffle (VPSHUFD, VPERMQ), the
|
||||||
|
three-operand-plus-immediate form (VSHUFPD, VPERM2I128, VINSERTI128), the
|
||||||
|
lane extract (VEXTRACTI128, VEXTRACTF128), the direction-sensitive moves
|
||||||
|
(VMOVDQU, VMOVUPD, VMOVD, VMOVQ, VMOVSD), the no-operand VZEROUPPER, and
|
||||||
|
the floating-point set: the packed double arithmetic
|
||||||
|
(VADDPD/VSUBPD/VMULPD/VDIVPD/VMINPD/VMAXPD), the unpacks
|
||||||
|
(VUNPCKHPD/VUNPCKLPD), the scalar SD and SS operations, VMOVDDUP, the
|
||||||
|
width-changing conversions (VCVTDQ2PS, VCVTPS2PD, VCVTDQ2PD and the
|
||||||
|
VCVTPD2DQX/Y / VCVTTPD2DQX/Y spellings, whose VEX.L follows the wider
|
||||||
|
source) and VFMADD231PD.
|
||||||
|
- **SIMD (EVEX / AVX-512)** — the four-byte EVEX prefix with the 5-bit
|
||||||
|
register fields (Z0–Z31, X/Y 16–31), opmask registers (K0–K7 as operands
|
||||||
|
and mask destinations, KMOVW, KTESTW) and the compressed disp8×N
|
||||||
|
displacement, covering every AVX-512 instruction the go-flac kernels use:
|
||||||
|
VPXORD/Q, VPADDD, VPSUBD/Q, VPUNPCK*DQ, VPMULLD/Q, VPERMD, VPSLLD/VPSRAD/
|
||||||
|
VPSRAQ, VALIGND, VPCMPEQD (with a K destination), VMOVDQU32, VMOVUPD,
|
||||||
|
VCVTQQ2PD, VPMOVSXDQ, the narrowing stores VPMOVDW/VPMOVQD, the lane
|
||||||
|
extracts VEXTRACTI64X4/VEXTRACTF64X4, VFMADD231PD, VADDPD, VMULPD,
|
||||||
|
VMOVDQU64 and the broadcasts VPBROADCASTD/Q from a GPR or memory, plus the
|
||||||
|
wider AVX-512 F/BW integer set (VPADDB/W, VPSUBB/W, VPANDD/Q/ND/NQ, VPMULLW,
|
||||||
|
VPMIN*/VPMAX* for B/W/D/Q elements, signed and unsigned, VPAVGB/W, the variable
|
||||||
|
shifts VPSLLV*/VPSRLV*/VPSRAV*, VMOVDQU8/16), the common floating-point
|
||||||
|
and conversion set (the packed double and single arithmetic
|
||||||
|
VADD/VSUB/VMUL/VDIV/VMIN/VMAX PD and PS, the scalar SD/SS operations —
|
||||||
|
whose EVEX forms exist for masked and zeroing use — the VUNPCK{L,H}PD
|
||||||
|
unpacks, VMOVDDUP, VMOVSLDUP/VMOVSHDUP and the VCVT* conversions), and
|
||||||
|
the wider AVX-512 set: ternary logic (VPTERNLOGD/Q), lane shuffles,
|
||||||
|
inserts and extracts (VSHUF{F,I}{32,64}X{2,4}, the VINSERT*/VEXTRACT*
|
||||||
|
{F,I}{32,64}X{2,4,8} family, VPALIGNR), compares with an opmask
|
||||||
|
destination (VCMPPD/PS/SD/SS), the permutes (VPERMB/W, VPERMI2/T2
|
||||||
|
D/Q/PD), the wider integer families (VPMADDWD/UBSW, VPMULHUW, VPACK*,
|
||||||
|
VPABS*, the VPROL*/VPROR* rotates and the word shifts), expand/compress
|
||||||
|
(VEXPAND*/VCOMPRESS*, VPEXPAND*/VPCOMPRESS*), the broadcasts
|
||||||
|
(VPBROADCASTB/W, VBROADCASTSS/SD), the opmask instructions (KAND/KOR/
|
||||||
|
KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST, KMOVQ), the aligned moves
|
||||||
|
(VMOVAPS/APD, VMOVDQA32/64, VMOVSS) and the remaining extending and
|
||||||
|
narrowing moves, the floating-point helper and conversion tail
|
||||||
|
(VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*, VSCALEF*, VRNDSCALE*,
|
||||||
|
VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with a K destination, and the
|
||||||
|
VCVT* conversions VCVTQQ2PS, VCVTPD2QQ/UQQ, VCVTPS2QQ, VCVTUDQ2PD/PS,
|
||||||
|
VCVTPH2PS, VCVTPS2PH), and gather/scatter with VSIB addressing
|
||||||
|
(VGATHER*/VPGATHER* in both the VEX mask-register spelling and the EVEX
|
||||||
|
K-mask spelling — where the L'L field follows the VSIB index — plus
|
||||||
|
VSCATTER*/VPSCATTER*). The EVEX mnemonic suffixes the Go assembler
|
||||||
|
accepts are honoured: rounding modes (.RN_SAE, .RD_SAE, .RU_SAE,
|
||||||
|
.RZ_SAE), suppress-all-exceptions (.SAE) and memory broadcast (.BCST,
|
||||||
|
with the element-sized disp8×N), each combinable with the .Z zeroing
|
||||||
|
suffix. Masking is supported the way
|
||||||
|
Go writes it — an explicit K1–K7 operand placed among the operands, and a
|
||||||
|
`.Z` mnemonic suffix for zeroing.
|
||||||
|
- **Legacy SSE moves** — `MOVOU`/`MOVO` (the Plan 9 names for MOVDQU/MOVDQA),
|
||||||
|
`MOVUPS`/`MOVAPS`/`MOVUPD`/`MOVAPD` and the scalar `MOVSD`/`MOVSS`.
|
||||||
|
- **Both go-flac kernels — all 17 AVX2 and all 10 AVX-512 functions —
|
||||||
|
assemble byte-identically to the Go toolchain's machine code**; the only
|
||||||
|
differing bytes are the displacements of the static-constant loads, which
|
||||||
|
the Go linker fills at link time and gasm resolves within its own image
|
||||||
|
(verified to reach the right constant bytes).
|
||||||
|
|
||||||
|
Remaining for Phase 2:
|
||||||
|
|
||||||
|
- External (cross-package) symbol references in the GOOBJ output —
|
||||||
|
**deferred** with a recorded decision and three options; see
|
||||||
|
[`docs/DEFERRED.md`](docs/DEFERRED.md). Single-package objects (no
|
||||||
|
cross-package references) work today, which covers the production
|
||||||
|
kernels. With that item deferred, the amd64 instruction set — scalar,
|
||||||
|
VEX/AVX2 and the full EVEX/AVX-512 set including GPR-interchanging
|
||||||
|
conversions — is complete, and RISC-V encoding (RV64IMAFDC + RVC)
|
||||||
|
including ELF and GOOBJ emission is complete.
|
||||||
|
|
||||||
|
### Phase 3 — dynamic analysis · *done*
|
||||||
|
|
||||||
|
Run the code and check what static analysis cannot. The oracle is the
|
||||||
|
portable Go implementation every kernel is derived from.
|
||||||
|
|
||||||
|
- **`gasm verify`:**
|
||||||
|
- **JIT execution substrate** — *done.* Assemble the kernel, map it into
|
||||||
|
executable memory (`syscall.Mmap`, W^X) and call it through an ABI0
|
||||||
|
trampoline; pure Go, no cgo, no external toolchain.
|
||||||
|
- **Differential testing** — *done.* The JIT-assembled kernel is fuzzed
|
||||||
|
against a portable Go reference, comparing the result bit-for-bit;
|
||||||
|
the automated form of the project's bit-identical contract.
|
||||||
|
- **Runtime ABI checks** — *done.* The ABI-checking trampoline sets
|
||||||
|
sentinels in BP and R14, verifies they survive the call, and fills a
|
||||||
|
128-byte red-zone canary below SP.
|
||||||
|
- **Coverage / basic-block profiling** — *done.* Static block enumeration
|
||||||
|
from the assembler's label map plus multi-input path-diversity
|
||||||
|
measurement: how many observationally distinct execution paths a
|
||||||
|
test corpus exercises.
|
||||||
|
|
||||||
|
### Phase 4 — debugger · *done*
|
||||||
|
|
||||||
|
- **`gasm debug`:** single-step a GAsm function, inspect registers (including
|
||||||
|
YMM vector registers), set breakpoints and watchpoints on addresses, write
|
||||||
|
memory, allocate and fill named buffers, disassemble at PC, and trace the
|
||||||
|
source-line mapping — the interactive counterpart to Phase 3's execution
|
||||||
|
substrate.
|
||||||
|
- ptrace-based debuggee subprocess (PTRACE_TRACEME + LockOSThread), entry
|
||||||
|
breakpoint (auto-run to function start), single-step, register inspection
|
||||||
|
(GPR + YMM/XMM via PTRACE_GETFPREGS), label resolution, breakpoint
|
||||||
|
management via `/proc/pid/mem`, named buffer allocation with pattern
|
||||||
|
filling (`--buf`), interactive REPL with conditional breakpoints, four
|
||||||
|
hardware watchpoints (DR0–DR3), step-over-CALL, run-to-return, backtrace,
|
||||||
|
memory read/write, disassembly at PC (x86asm), and source-line ↔ offset
|
||||||
|
mapping.
|
||||||
|
|
||||||
|
### Phase 5 — the other architectures · *in progress*
|
||||||
|
|
||||||
|
- **RISC-V encoding — done.** RV64IMAFDC instruction set, RVC compression,
|
||||||
|
MOV pseudo-instruction, SB/global symbols (AUIPC pairs), ELF64 and GOOBJ
|
||||||
|
emission, and ground-truth verification against `go tool asm`.
|
||||||
|
- **LoongArch encoding — done.** The LoongArch64 instruction set with the
|
||||||
|
MOV pseudo-instruction and its immediate-constant expansions, FP/SP frame
|
||||||
|
handling, SB/global symbol references (pcalau12i pairs), ELF64 and GOOBJ
|
||||||
|
emission, and ground-truth verification against `go tool asm` — the emitted
|
||||||
|
GOOBJ links into a real `go build` for `GOARCH=loong64`.
|
||||||
|
- **Remaining:** arm64 encoding, plus the same encode-and-verify treatment
|
||||||
|
(instruction tables already generated from the toolchain).
|
||||||
|
|
||||||
|
## Principles
|
||||||
|
|
||||||
|
- **Pure Go and GAsm only.** No C, no cgo, no external toolchains, no native
|
||||||
|
binaries, no JavaScript runtimes. The parser is hand-written; there is no
|
||||||
|
parser generator.
|
||||||
|
- **Self-contained.** The toolkit's production code depends only on the
|
||||||
|
standard library; one binary, no runtime data files. The single module
|
||||||
|
dependency, `golang.org/x/arch`, is used **only in tests** to validate the
|
||||||
|
instruction encoder by round-trip decoding — it is never linked into the
|
||||||
|
`gasm` binary.
|
||||||
|
- **Linux-only.** Runs natively on amd64, arm64, riscv64 and loong64 Linux
|
||||||
|
hosts; the release matrix cross-compiles the same four targets. Latest
|
||||||
|
stable Go only.
|
||||||
|
- **No vendor lock-in.** The integration surface is the Language Server
|
||||||
|
Protocol and a command-line interface — both open standards. No cloud
|
||||||
|
service, no proprietary API, no dependence on any one editor's internals.
|
||||||
|
- **Complete and verifiable.** Instruction coverage is generated from the
|
||||||
|
assembler's own source and regenerated on demand, so it cannot silently fall
|
||||||
|
behind the toolchain.
|
||||||
|
|
||||||
|
## Components
|
||||||
|
|
||||||
|
| Package | Purpose |
|
||||||
|
|---------|---------|
|
||||||
|
| `token` | Lexical token kinds and source positions. |
|
||||||
|
| `lexer` | Hand-written scanner for Plan 9 assembly. |
|
||||||
|
| `ast` | The abstract syntax tree. |
|
||||||
|
| `parser` | Line-oriented, error-tolerant parser producing the AST. |
|
||||||
|
| `arch` | amd64, arm64, riscv64 and loong64 register files and instruction tables. |
|
||||||
|
| `lint` | Conservative static checks. |
|
||||||
|
| `format` | A canonical formatter — `gofmt` for assembly. |
|
||||||
|
| `asm` | The standalone assembler: amd64, RISC-V and LoongArch encoders, linker, object-file emitters (ELF, GOOBJ). |
|
||||||
|
| `verify` | JIT execution substrate for dynamic analysis, combined ABI+fuzz differential testing (Phase 3). |
|
||||||
|
| `debug` | Interactive ptrace debugger with GPR/YMM register display and named buffer allocation (Phase 4). |
|
||||||
|
| `lsp` | Language Server Protocol server. |
|
||||||
|
| `cmd/gasm` | The `gasm` binary tying it all together. |
|
||||||
|
| `_gen` | The generator that rebuilds the instruction tables from the Go toolchain. |
|
||||||
|
|
||||||
|
See [`docs/ARCHITECTURE.md`](docs/ARCHITECTURE.md) for the design rationale and
|
||||||
|
data flow, and [`docs/DEFERRED.md`](docs/DEFERRED.md) for design decisions
|
||||||
|
deliberately postponed (with the analysis needed to pick them up again).
|
||||||
|
|
||||||
|
## Quick start
|
||||||
|
|
||||||
|
```sh
|
||||||
|
just install # download dependencies (there are none)
|
||||||
|
just build # go vet + gofmt check — zero errors, zero warnings
|
||||||
|
just test # full suite, race detector, 80 % coverage gate
|
||||||
|
just fmt # gofmt the tree
|
||||||
|
just gen # regenerate the instruction tables from the Go toolchain
|
||||||
|
```
|
||||||
|
|
||||||
|
Install the binary and use it:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
just install-bin # installs gasm into $GOBIN
|
||||||
|
|
||||||
|
gasm --help # overview of commands and flags
|
||||||
|
gasm tokens kernel_amd64.s # dump the token stream
|
||||||
|
gasm parse kernel_amd64.s # parse, report syntax errors
|
||||||
|
gasm fmt -w kernel_amd64.s # canonicalise in place
|
||||||
|
gasm fmt # reformat every .s below here, like go fmt
|
||||||
|
gasm lint *.s # static checks
|
||||||
|
gasm asm --format elf -o k.o k.s # assemble to a linkable ELF object
|
||||||
|
gasm verify kernel_amd64.s # JIT-load and report functions
|
||||||
|
gasm verify --ground-truth k.s # byte-for-byte vs go tool asm
|
||||||
|
gasm verify --call decodeBlockAVX2 --buf src:64:hex...,dst:256:zero k.s
|
||||||
|
gasm debug --func name k.s # interactive debugger
|
||||||
|
gasm diff a.s b.s # compare machine code byte-for-byte
|
||||||
|
gasm diff --map wideCopyAVX2=wideCopyAVX512 avx2.s avx512.s
|
||||||
|
gasm profile k.s # show basic-block structure
|
||||||
|
```
|
||||||
|
|
||||||
|
See [CONTRIBUTING.md](CONTRIBUTING.md) for the full development workflow,
|
||||||
|
[docs/cli.md](docs/cli.md) for the command reference, and
|
||||||
|
[docs/development.md](docs/development.md) for setup and recipes.
|
||||||
|
|
||||||
|
## Editor integration
|
||||||
|
|
||||||
|
`gasm lsp` speaks the Language Server Protocol over standard input/output, so
|
||||||
|
any LSP-capable editor can use it — point your editor's LSP client at the
|
||||||
|
binary and associate it with `.s` files. Syntax highlighting is delivered as
|
||||||
|
**LSP semantic tokens**, so no editor-specific grammar is required. The server
|
||||||
|
infers the target architecture from the file-name suffix
|
||||||
|
(`_amd64.s` / `_arm64.s` / `_riscv64.s` / `_loong64.s`).
|
||||||
|
|
||||||
|
## Licence
|
||||||
|
|
||||||
|
BSD-3-Clause — the same licence as Go itself. See [`LICENSE`](LICENSE).
|
||||||
+8
-2
@@ -86,7 +86,10 @@ func filterCommon(names []string) []string {
|
|||||||
func writeCommon(names []string) error {
|
func writeCommon(names []string) error {
|
||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
||||||
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n\n")
|
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
|
||||||
|
b.WriteString("//\n")
|
||||||
|
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
||||||
|
b.WriteString("// SPDX-License-Identifier: BSD-3-Clause\n\n")
|
||||||
b.WriteString("package arch\n\n")
|
b.WriteString("package arch\n\n")
|
||||||
b.WriteString("// commonGeneratedInstrs is the set of opcodes shared by every architecture\n")
|
b.WriteString("// commonGeneratedInstrs is the set of opcodes shared by every architecture\n")
|
||||||
b.WriteString("// (RET, JMP, NOP, CALL, TEXT, FUNCDATA, PCDATA, …).\n")
|
b.WriteString("// (RET, JMP, NOP, CALL, TEXT, FUNCDATA, PCDATA, …).\n")
|
||||||
@@ -154,7 +157,10 @@ func stringLit(elt ast.Expr) string {
|
|||||||
func writeGen(arch, sub string, names []string) error {
|
func writeGen(arch, sub string, names []string) error {
|
||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
||||||
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n\n")
|
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
|
||||||
|
b.WriteString("//\n")
|
||||||
|
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
||||||
|
b.WriteString("// SPDX-License-Identifier: BSD-3-Clause\n\n")
|
||||||
b.WriteString("package arch\n\n")
|
b.WriteString("package arch\n\n")
|
||||||
b.WriteString("// " + arch + "GeneratedInstrs is the complete set of " + arch +
|
b.WriteString("// " + arch + "GeneratedInstrs is the complete set of " + arch +
|
||||||
" mnemonics accepted by\n// Go's Plan 9 assembler.\n")
|
" mnemonics accepted by\n// Go's Plan 9 assembler.\n")
|
||||||
|
|||||||
@@ -1,5 +1,8 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
|
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
|
||||||
|
//
|
||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
package arch
|
package arch
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,8 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
|
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
|
||||||
|
//
|
||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
package arch
|
package arch
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,8 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/util.go from the Go toolchain.
|
// Source: cmd/internal/obj/util.go from the Go toolchain.
|
||||||
|
//
|
||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
package arch
|
package arch
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,8 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
|
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
|
||||||
|
//
|
||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
package arch
|
package arch
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,8 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
|
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
|
||||||
|
//
|
||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
package arch
|
package arch
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,23 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
|
)
|
||||||
|
|
||||||
|
// assembleARM64 is a stub. The arm64 (AArch64) instruction encoder is not yet
|
||||||
|
// implemented — the instruction tables, register files and operand-count
|
||||||
|
// metadata are in place (package arch), and the lexer, parser, formatter and
|
||||||
|
// linter already handle arm64 source files.
|
||||||
|
func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, error) {
|
||||||
|
return nil, nil, nil, fmt.Errorf("arm64 instruction encoding is not yet implemented")
|
||||||
|
}
|
||||||
|
|
||||||
|
// AssembleFileARM64 is a stub, returning the same error as assembleARM64.
|
||||||
|
func AssembleFileARM64(f *ast.File) (*Image, error) {
|
||||||
|
return nil, fmt.Errorf("arm64 instruction encoding is not yet implemented")
|
||||||
|
}
|
||||||
+8
-6
@@ -23,7 +23,7 @@ import (
|
|||||||
// operands require relocations and are not yet supported; the SIMD (VEX/AVX2)
|
// operands require relocations and are not yet supported; the SIMD (VEX/AVX2)
|
||||||
// integer and shuffle/extract/permute/move set is in.
|
// integer and shuffle/extract/permute/move set is in.
|
||||||
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
|
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
|
||||||
code, _, labels, _, err := assemble(t, nil)
|
code, _, labels, _, _, err := assemble(t, nil)
|
||||||
return code, labels, err
|
return code, labels, err
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -60,7 +60,7 @@ type spadjStep struct {
|
|||||||
// assemble encodes a TEXT body, returning the machine code, the static-symbol
|
// assemble encodes a TEXT body, returning the machine code, the static-symbol
|
||||||
// patch sites (for the file-level layout to resolve), the label table and the
|
// patch sites (for the file-level layout to resolve), the label table and the
|
||||||
// stack-adjustment boundaries.
|
// stack-adjustment boundaries.
|
||||||
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, error) {
|
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
|
||||||
fi := computeFrame(t)
|
fi := computeFrame(t)
|
||||||
chain := jumpChain(t)
|
chain := jumpChain(t)
|
||||||
resolve := func(name string) string {
|
resolve := func(name string) string {
|
||||||
@@ -84,7 +84,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
case *ast.Instr:
|
case *ast.Instr:
|
||||||
sz, err := instrSize(s, fi, long[i], link)
|
sz, err := instrSize(s, fi, long[i], link)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||||
}
|
}
|
||||||
sizes[i] = sz
|
sizes[i] = sz
|
||||||
pcs[i] = pos
|
pcs[i] = pos
|
||||||
@@ -125,6 +125,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
out := append([]byte(nil), fi.prologue...)
|
out := append([]byte(nil), fi.prologue...)
|
||||||
var patches []sbPatch
|
var patches []sbPatch
|
||||||
var steps []spadjStep
|
var steps []spadjStep
|
||||||
|
var lines []LineEntry
|
||||||
if fi.useFP {
|
if fi.useFP {
|
||||||
// PUSHQ BP saves the return-address-relative base (+8); the MOVQ
|
// PUSHQ BP saves the return-address-relative base (+8); the MOVQ
|
||||||
// changes nothing; SUBQ $size, SP completes the frame.
|
// changes nothing; SUBQ $size, SP completes the frame.
|
||||||
@@ -150,16 +151,17 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
}
|
}
|
||||||
code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
|
code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||||
}
|
}
|
||||||
if len(code) != sizes[i] {
|
if len(code) != sizes[i] {
|
||||||
return nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
||||||
}
|
}
|
||||||
patches = append(patches, ps...)
|
patches = append(patches, ps...)
|
||||||
|
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
|
||||||
out = append(out, code...)
|
out = append(out, code...)
|
||||||
pos += len(code)
|
pos += len(code)
|
||||||
}
|
}
|
||||||
return out, patches, offsets, steps, nil
|
return out, patches, offsets, steps, lines, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// jumpChain precomputes jump-to-jump folding: a label whose first instruction
|
// jumpChain precomputes jump-to-jump folding: a label whose first instruction
|
||||||
|
|||||||
@@ -0,0 +1,223 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
)
|
||||||
|
|
||||||
|
// LoongArch ELF64 relocatable object emission.
|
||||||
|
|
||||||
|
const (
|
||||||
|
emLOONGARCH = 258 // EM_LOONGARCH
|
||||||
|
|
||||||
|
// LoongArch relocation types (the ELF psABI).
|
||||||
|
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
|
||||||
|
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
|
||||||
|
)
|
||||||
|
|
||||||
|
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
|
||||||
|
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
|
||||||
|
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
|
||||||
|
// optional .rela.text.
|
||||||
|
func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||||
|
le := binary.LittleEndian
|
||||||
|
|
||||||
|
const (
|
||||||
|
secText = 1
|
||||||
|
secData = 2
|
||||||
|
)
|
||||||
|
|
||||||
|
// Build symbol table.
|
||||||
|
var locals, globals []elfSym
|
||||||
|
for _, fn := range img.Funcs {
|
||||||
|
s := elfSym{
|
||||||
|
name: objectName(fn.Pkg, fn.Name),
|
||||||
|
info: sttFunc,
|
||||||
|
shndx: secText,
|
||||||
|
value: uint64(fn.Offset),
|
||||||
|
size: uint64(fn.Size),
|
||||||
|
}
|
||||||
|
if fn.Static {
|
||||||
|
locals = append(locals, s)
|
||||||
|
} else {
|
||||||
|
s.info |= stbGlobal << stInfoShift
|
||||||
|
globals = append(globals, s)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
s := elfSym{
|
||||||
|
name: objectName(d.Pkg, d.Name),
|
||||||
|
info: sttObject,
|
||||||
|
shndx: secData,
|
||||||
|
value: uint64(d.Offset),
|
||||||
|
size: uint64(d.Size),
|
||||||
|
}
|
||||||
|
if d.Static {
|
||||||
|
locals = append(locals, s)
|
||||||
|
} else {
|
||||||
|
s.info |= stbGlobal << stInfoShift
|
||||||
|
globals = append(globals, s)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, name := range img.Externals {
|
||||||
|
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
|
||||||
|
}
|
||||||
|
syms := []elfSym{
|
||||||
|
{},
|
||||||
|
{name: ".text", info: sttSection, shndx: secText},
|
||||||
|
{name: ".data", info: sttSection, shndx: secData},
|
||||||
|
}
|
||||||
|
syms = append(syms, locals...)
|
||||||
|
shInfo := len(syms)
|
||||||
|
syms = append(syms, globals...)
|
||||||
|
symIdx := map[string]int{}
|
||||||
|
for i, s := range syms {
|
||||||
|
symIdx[s.name] = i
|
||||||
|
}
|
||||||
|
|
||||||
|
// Build relocations. Each SB reference is a pcalau12i pair:
|
||||||
|
// pcalau12i rd, 0 → R_LARCH_PCALA_HI20
|
||||||
|
// addi.d/ld/st → R_LARCH_PCALA_LO12
|
||||||
|
type elfRela struct {
|
||||||
|
off uint64
|
||||||
|
typ uint32
|
||||||
|
sym int
|
||||||
|
addend int64
|
||||||
|
}
|
||||||
|
var relas []elfRela
|
||||||
|
for _, fn := range img.Funcs {
|
||||||
|
for _, r := range fn.Relocs {
|
||||||
|
idx, ok := symIdx[r.Name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||||
|
}
|
||||||
|
typ := uint32(rLarchPCALAHI20)
|
||||||
|
if r.Kind == RelLoong64AddrLo {
|
||||||
|
typ = rLarchPCALALO12
|
||||||
|
}
|
||||||
|
relas = append(relas, elfRela{
|
||||||
|
off: uint64(fn.Offset + r.Off),
|
||||||
|
typ: typ,
|
||||||
|
sym: idx,
|
||||||
|
addend: r.Addend - int64(r.After-r.Off),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// String tables.
|
||||||
|
stNames := newElfStrtab()
|
||||||
|
for _, s := range syms {
|
||||||
|
stNames.add(s.name)
|
||||||
|
}
|
||||||
|
stSections := newElfStrtab()
|
||||||
|
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||||
|
stSections.add(n)
|
||||||
|
}
|
||||||
|
|
||||||
|
hasRela := len(relas) > 0
|
||||||
|
nSections := 6
|
||||||
|
if hasRela {
|
||||||
|
nSections = 7
|
||||||
|
}
|
||||||
|
secSymtab, secStrtab := 3, 4
|
||||||
|
secShstr := nSections - 1
|
||||||
|
|
||||||
|
// Layout.
|
||||||
|
var out []byte
|
||||||
|
out = append(out, make([]byte, 64)...)
|
||||||
|
|
||||||
|
align := func(n int) {
|
||||||
|
for len(out)%n != 0 {
|
||||||
|
out = append(out, 0)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
align(16)
|
||||||
|
textOff := len(out)
|
||||||
|
out = append(out, img.Code...)
|
||||||
|
|
||||||
|
align(16)
|
||||||
|
dataOff := len(out)
|
||||||
|
out = append(out, img.Data...)
|
||||||
|
|
||||||
|
align(8)
|
||||||
|
symtabOff := len(out)
|
||||||
|
for _, s := range syms {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
|
||||||
|
b[4] = s.info
|
||||||
|
b[5] = 0
|
||||||
|
le.PutUint16(b[6:], s.shndx)
|
||||||
|
le.PutUint64(b[8:], s.value)
|
||||||
|
le.PutUint64(b[16:], s.size)
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
|
||||||
|
strtabOff := len(out)
|
||||||
|
out = append(out, stNames.bytes()...)
|
||||||
|
|
||||||
|
var relaOff int
|
||||||
|
if hasRela {
|
||||||
|
align(8)
|
||||||
|
relaOff = len(out)
|
||||||
|
for _, r := range relas {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint64(b[0:], r.off)
|
||||||
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||||
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
shstrOff := len(out)
|
||||||
|
out = append(out, stSections.bytes()...)
|
||||||
|
|
||||||
|
align(8)
|
||||||
|
shoff := len(out)
|
||||||
|
|
||||||
|
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
|
||||||
|
var b [64]byte
|
||||||
|
le.PutUint32(b[0:], uint32(stSections.at(name)))
|
||||||
|
le.PutUint32(b[4:], uint32(typ))
|
||||||
|
le.PutUint64(b[8:], flags)
|
||||||
|
le.PutUint64(b[16:], 0)
|
||||||
|
le.PutUint64(b[24:], uint64(off))
|
||||||
|
le.PutUint64(b[32:], uint64(size))
|
||||||
|
le.PutUint32(b[40:], uint32(link))
|
||||||
|
le.PutUint32(b[44:], uint32(info))
|
||||||
|
le.PutUint64(b[48:], alignV)
|
||||||
|
le.PutUint64(b[56:], entsize)
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
|
||||||
|
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
|
||||||
|
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
|
||||||
|
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
|
||||||
|
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
|
||||||
|
if hasRela {
|
||||||
|
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||||
|
}
|
||||||
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
|
|
||||||
|
// ELF header.
|
||||||
|
hdr := out[:64]
|
||||||
|
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
|
||||||
|
le.PutUint16(hdr[16:], etREL)
|
||||||
|
le.PutUint16(hdr[18:], emLOONGARCH)
|
||||||
|
le.PutUint32(hdr[20:], elfVersion)
|
||||||
|
le.PutUint64(hdr[24:], 0)
|
||||||
|
le.PutUint64(hdr[32:], 0)
|
||||||
|
le.PutUint64(hdr[40:], uint64(shoff))
|
||||||
|
le.PutUint32(hdr[48:], 0)
|
||||||
|
le.PutUint16(hdr[52:], 64)
|
||||||
|
le.PutUint16(hdr[54:], 0)
|
||||||
|
le.PutUint16(hdr[56:], 0)
|
||||||
|
le.PutUint16(hdr[58:], 64)
|
||||||
|
le.PutUint16(hdr[60:], uint16(nSections))
|
||||||
|
le.PutUint16(hdr[62:], uint16(secShstr))
|
||||||
|
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,200 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"debug/elf"
|
||||||
|
"encoding/binary"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
|
||||||
|
// relocatable object: sections, the symbol table (bindings, types, values,
|
||||||
|
// sizes) and the .rela.text relocation pair for the static-symbol load,
|
||||||
|
// parsed back with debug/elf.
|
||||||
|
func TestELFLOONG64Object(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("k_loong64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOVV a+0(FP), R4
|
||||||
|
MOVV b+8(FP), R5
|
||||||
|
ADDV R5, R4, R4
|
||||||
|
MOVV R4, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·getanswer(SB), NOSPLIT, $0-8
|
||||||
|
MOVV answer<>(SB), R4
|
||||||
|
MOVV R4, ret+0(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
GLOBL answer<>(SB), RODATA, $8
|
||||||
|
DATA answer<>+0(SB)/8, $42
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileLOONG64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFLOONG64Object()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFLOONG64Object: %v", err)
|
||||||
|
}
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
|
||||||
|
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_LOONGARCH {
|
||||||
|
t.Errorf("type/machine = %v/%v, want ET_REL/EM_LOONGARCH", ef.Type, ef.Machine)
|
||||||
|
}
|
||||||
|
|
||||||
|
text := ef.Section(".text")
|
||||||
|
data := ef.Section(".data")
|
||||||
|
if text == nil || data == nil {
|
||||||
|
t.Fatal("missing .text or .data section")
|
||||||
|
}
|
||||||
|
if text.Flags&elf.SHF_EXECINSTR == 0 || text.Flags&elf.SHF_ALLOC == 0 {
|
||||||
|
t.Errorf(".text flags = %v", text.Flags)
|
||||||
|
}
|
||||||
|
if data.Flags&elf.SHF_WRITE == 0 {
|
||||||
|
t.Errorf(".data flags = %v", data.Flags)
|
||||||
|
}
|
||||||
|
textData, err := text.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if !bytes.Equal(textData, img.Code) {
|
||||||
|
t.Errorf(".text contents differ from the image code")
|
||||||
|
}
|
||||||
|
dataData, err := data.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("symbols: %v", err)
|
||||||
|
}
|
||||||
|
byName := map[string]elf.Symbol{}
|
||||||
|
for _, s := range syms {
|
||||||
|
byName[s.Name] = s
|
||||||
|
}
|
||||||
|
wantSym := func(name string, bind elf.SymBind, typ elf.SymType, section elf.SectionIndex, size uint64) {
|
||||||
|
t.Helper()
|
||||||
|
s, ok := byName[name]
|
||||||
|
if !ok {
|
||||||
|
t.Errorf("symbol %q not found", name)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if elf.ST_BIND(s.Info) != bind || elf.ST_TYPE(s.Info) != typ {
|
||||||
|
t.Errorf("%s: bind/type = %v/%v, want %v/%v", name, elf.ST_BIND(s.Info), elf.ST_TYPE(s.Info), bind, typ)
|
||||||
|
}
|
||||||
|
if s.Section != section {
|
||||||
|
t.Errorf("%s: section = %v, want %v", name, s.Section, section)
|
||||||
|
}
|
||||||
|
if s.Size != size {
|
||||||
|
t.Errorf("%s: size = %d, want %d", name, s.Size, size)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if ef.Sections[1].Name != ".text" || ef.Sections[2].Name != ".data" {
|
||||||
|
t.Fatalf("section layout = %s, %s; want .text, .data", ef.Sections[1].Name, ef.Sections[2].Name)
|
||||||
|
}
|
||||||
|
textIdx := elf.SectionIndex(1)
|
||||||
|
dataIdx := elf.SectionIndex(2)
|
||||||
|
wantSym("add", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 20)
|
||||||
|
wantSym("getanswer", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 16)
|
||||||
|
wantSym("answer", elf.STB_LOCAL, elf.STT_OBJECT, dataIdx, 8)
|
||||||
|
|
||||||
|
// The data section carries 16-byte alignment padding; the answer
|
||||||
|
// symbol sits at its padded offset.
|
||||||
|
ans := byName["answer"]
|
||||||
|
if ans.Value+8 > uint64(len(dataData)) {
|
||||||
|
t.Fatalf("answer value %d outside .data (%d bytes)", ans.Value, len(dataData))
|
||||||
|
}
|
||||||
|
if got := dataData[ans.Value : ans.Value+8]; !bytes.Equal(got, []byte{42, 0, 0, 0, 0, 0, 0, 0}) {
|
||||||
|
t.Errorf("answer data = % x, want $42", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Relocations: the static-symbol load is a pcalau12i+ld.d pair, so one
|
||||||
|
// R_LARCH_PCALA_HI20 and one R_LARCH_PCALA_LO12, both against the local
|
||||||
|
// data symbol. debug/elf does not surface rela entries, so read the
|
||||||
|
// section directly.
|
||||||
|
relaSec := ef.Section(".rela.text")
|
||||||
|
if relaSec == nil {
|
||||||
|
t.Fatal("missing .rela.text")
|
||||||
|
}
|
||||||
|
raw, err := relaSec.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if len(raw)%24 != 0 || len(raw)/24 != 2 {
|
||||||
|
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
|
||||||
|
}
|
||||||
|
le := binary.LittleEndian
|
||||||
|
for i := 0; i < 2; i++ {
|
||||||
|
e := raw[i*24 : (i+1)*24]
|
||||||
|
off := le.Uint64(e[0:])
|
||||||
|
info := le.Uint64(e[8:])
|
||||||
|
typ := info & 0xffffffff
|
||||||
|
sym := int(info >> 32)
|
||||||
|
if i == 0 && (typ != uint64(elf.R_LARCH_PCALA_HI20) || off != 20) {
|
||||||
|
t.Errorf("reloc %d: type %d off %d, want R_LARCH_PCALA_HI20 at 20", i, typ, off)
|
||||||
|
}
|
||||||
|
if i == 1 && (typ != uint64(elf.R_LARCH_PCALA_LO12) || off != 24) {
|
||||||
|
t.Errorf("reloc %d: type %d off %d, want R_LARCH_PCALA_LO12 at 24", i, typ, off)
|
||||||
|
}
|
||||||
|
if sym != 3 { // NULL, .text, .data, then the first local: answer
|
||||||
|
t.Errorf("reloc %d: symbol index %d, want 3 (answer)", i, sym)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestELFLOONG64ObjectNoRelocations checks a file with no static-symbol
|
||||||
|
// references emits a valid object without a .rela.text section.
|
||||||
|
func TestELFLOONG64ObjectNoRelocations(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("n_loong64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·nop(SB), NOSPLIT, $0
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileLOONG64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFLOONG64Object()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFLOONG64Object: %v", err)
|
||||||
|
}
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
if ef.Section(".rela.text") != nil {
|
||||||
|
t.Error("unexpected .rela.text section")
|
||||||
|
}
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
found := false
|
||||||
|
for _, s := range syms {
|
||||||
|
if s.Name == "nop" && elf.ST_TYPE(s.Info) == elf.STT_FUNC {
|
||||||
|
found = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !found {
|
||||||
|
t.Error("function symbol nop not found")
|
||||||
|
}
|
||||||
|
}
|
||||||
+236
@@ -0,0 +1,236 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
)
|
||||||
|
|
||||||
|
// RISC-V ELF64 relocatable object emission.
|
||||||
|
|
||||||
|
const (
|
||||||
|
emRISCV = 243 // EM_RISCV
|
||||||
|
|
||||||
|
// RISC-V relocation types.
|
||||||
|
rRISCV32 = 1
|
||||||
|
rRISCVJAL = 17 // R_RISCV_JAL
|
||||||
|
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
|
||||||
|
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
|
||||||
|
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
|
||||||
|
)
|
||||||
|
|
||||||
|
// ELFRISCVObject returns the image as an ELF64 relocatable object file for
|
||||||
|
// RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64
|
||||||
|
// ELF emission: .text, .data, .symtab, .strtab and optional .rela.text.
|
||||||
|
func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||||
|
le := binary.LittleEndian
|
||||||
|
|
||||||
|
const (
|
||||||
|
secText = 1
|
||||||
|
secData = 2
|
||||||
|
)
|
||||||
|
|
||||||
|
// Build symbol table.
|
||||||
|
var locals, globals []elfSym
|
||||||
|
for _, fn := range img.Funcs {
|
||||||
|
s := elfSym{
|
||||||
|
name: objectName(fn.Pkg, fn.Name),
|
||||||
|
info: sttFunc,
|
||||||
|
shndx: secText,
|
||||||
|
value: uint64(fn.Offset),
|
||||||
|
size: uint64(fn.Size),
|
||||||
|
}
|
||||||
|
if fn.Static {
|
||||||
|
locals = append(locals, s)
|
||||||
|
} else {
|
||||||
|
s.info |= stbGlobal << stInfoShift
|
||||||
|
globals = append(globals, s)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
s := elfSym{
|
||||||
|
name: objectName(d.Pkg, d.Name),
|
||||||
|
info: sttObject,
|
||||||
|
shndx: secData,
|
||||||
|
value: uint64(d.Offset),
|
||||||
|
size: uint64(d.Size),
|
||||||
|
}
|
||||||
|
if d.Static {
|
||||||
|
locals = append(locals, s)
|
||||||
|
} else {
|
||||||
|
s.info |= stbGlobal << stInfoShift
|
||||||
|
globals = append(globals, s)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, name := range img.Externals {
|
||||||
|
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
|
||||||
|
}
|
||||||
|
syms := []elfSym{
|
||||||
|
{},
|
||||||
|
{name: ".text", info: sttSection, shndx: secText},
|
||||||
|
{name: ".data", info: sttSection, shndx: secData},
|
||||||
|
}
|
||||||
|
syms = append(syms, locals...)
|
||||||
|
shInfo := len(syms)
|
||||||
|
syms = append(syms, globals...)
|
||||||
|
symIdx := map[string]int{}
|
||||||
|
for i, s := range syms {
|
||||||
|
symIdx[s.name] = i
|
||||||
|
}
|
||||||
|
|
||||||
|
// Build relocations. Each SB reference is an AUIPC + second-instruction
|
||||||
|
// pair carrying a single relocation kind; the ELF writer expands it into
|
||||||
|
// the R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I/S pair the psABI expects.
|
||||||
|
// The HI20 carries the symbol addend; the LO12 addend is zero, matching
|
||||||
|
// cmd/link's own ELF conversion (the LO12 resolves against the HI20's
|
||||||
|
// AUIPC location).
|
||||||
|
type elfRela struct {
|
||||||
|
off uint64
|
||||||
|
typ uint32
|
||||||
|
sym int
|
||||||
|
addend int64
|
||||||
|
}
|
||||||
|
var relas []elfRela
|
||||||
|
for _, fn := range img.Funcs {
|
||||||
|
for _, r := range fn.Relocs {
|
||||||
|
idx, ok := symIdx[r.Name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||||
|
}
|
||||||
|
switch r.Kind {
|
||||||
|
case RelRISCVPCRELIType:
|
||||||
|
relas = append(relas,
|
||||||
|
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
||||||
|
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: idx, addend: 0},
|
||||||
|
)
|
||||||
|
case RelRISCVPCRELSType:
|
||||||
|
relas = append(relas,
|
||||||
|
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
|
||||||
|
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: idx, addend: 0},
|
||||||
|
)
|
||||||
|
case RelRISCVJal:
|
||||||
|
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVJAL, sym: idx, addend: r.Addend})
|
||||||
|
case RelPCRelAbs:
|
||||||
|
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCV32, sym: idx, addend: r.Addend})
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// String tables.
|
||||||
|
stNames := newElfStrtab()
|
||||||
|
for _, s := range syms {
|
||||||
|
stNames.add(s.name)
|
||||||
|
}
|
||||||
|
stSections := newElfStrtab()
|
||||||
|
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||||
|
stSections.add(n)
|
||||||
|
}
|
||||||
|
|
||||||
|
hasRela := len(relas) > 0
|
||||||
|
nSections := 6
|
||||||
|
if hasRela {
|
||||||
|
nSections = 7
|
||||||
|
}
|
||||||
|
secSymtab, secStrtab := 3, 4
|
||||||
|
secShstr := nSections - 1
|
||||||
|
|
||||||
|
// Layout.
|
||||||
|
var out []byte
|
||||||
|
out = append(out, make([]byte, 64)...)
|
||||||
|
|
||||||
|
align := func(n int) {
|
||||||
|
for len(out)%n != 0 {
|
||||||
|
out = append(out, 0)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
align(16)
|
||||||
|
textOff := len(out)
|
||||||
|
out = append(out, img.Code...)
|
||||||
|
|
||||||
|
align(16)
|
||||||
|
dataOff := len(out)
|
||||||
|
out = append(out, img.Data...)
|
||||||
|
|
||||||
|
align(8)
|
||||||
|
symtabOff := len(out)
|
||||||
|
for _, s := range syms {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
|
||||||
|
b[4] = s.info
|
||||||
|
b[5] = 0
|
||||||
|
le.PutUint16(b[6:], s.shndx)
|
||||||
|
le.PutUint64(b[8:], s.value)
|
||||||
|
le.PutUint64(b[16:], s.size)
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
|
||||||
|
strtabOff := len(out)
|
||||||
|
out = append(out, stNames.bytes()...)
|
||||||
|
|
||||||
|
var relaOff int
|
||||||
|
if hasRela {
|
||||||
|
align(8)
|
||||||
|
relaOff = len(out)
|
||||||
|
for _, r := range relas {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint64(b[0:], r.off)
|
||||||
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||||
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
shstrOff := len(out)
|
||||||
|
out = append(out, stSections.bytes()...)
|
||||||
|
|
||||||
|
align(8)
|
||||||
|
shoff := len(out)
|
||||||
|
|
||||||
|
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
|
||||||
|
var b [64]byte
|
||||||
|
le.PutUint32(b[0:], uint32(stSections.at(name)))
|
||||||
|
le.PutUint32(b[4:], uint32(typ))
|
||||||
|
le.PutUint64(b[8:], flags)
|
||||||
|
le.PutUint64(b[16:], 0)
|
||||||
|
le.PutUint64(b[24:], uint64(off))
|
||||||
|
le.PutUint64(b[32:], uint64(size))
|
||||||
|
le.PutUint32(b[40:], uint32(link))
|
||||||
|
le.PutUint32(b[44:], uint32(info))
|
||||||
|
le.PutUint64(b[48:], alignV)
|
||||||
|
le.PutUint64(b[56:], entsize)
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
|
||||||
|
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
|
||||||
|
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
|
||||||
|
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
|
||||||
|
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
|
||||||
|
if hasRela {
|
||||||
|
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||||
|
}
|
||||||
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
|
|
||||||
|
// ELF header.
|
||||||
|
hdr := out[:64]
|
||||||
|
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
|
||||||
|
le.PutUint16(hdr[16:], etREL)
|
||||||
|
le.PutUint16(hdr[18:], emRISCV)
|
||||||
|
le.PutUint32(hdr[20:], elfVersion)
|
||||||
|
le.PutUint64(hdr[24:], 0)
|
||||||
|
le.PutUint64(hdr[32:], 0)
|
||||||
|
le.PutUint64(hdr[40:], uint64(shoff))
|
||||||
|
le.PutUint32(hdr[48:], 0)
|
||||||
|
le.PutUint16(hdr[52:], 64)
|
||||||
|
le.PutUint16(hdr[54:], 0)
|
||||||
|
le.PutUint16(hdr[56:], 0)
|
||||||
|
le.PutUint16(hdr[58:], 64)
|
||||||
|
le.PutUint16(hdr[60:], uint16(nSections))
|
||||||
|
le.PutUint16(hdr[62:], uint16(secShstr))
|
||||||
|
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
@@ -4,13 +4,10 @@
|
|||||||
package asm
|
package asm
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"os"
|
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"golang.org/x/arch/x86/x86asm"
|
"golang.org/x/arch/x86/x86asm"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestEvexGroundTruth checks the EVEX (AVX-512) encodings byte for byte
|
// TestEvexGroundTruth checks the EVEX (AVX-512) encodings byte for byte
|
||||||
@@ -649,71 +646,6 @@ func TestEvexErrors(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
|
|
||||||
// kernel — all functions plus the file-global idx16 constant — and checks
|
|
||||||
// that the static-symbol load resolves to the right bytes in the image.
|
|
||||||
// Skipped when the sibling repository is not checked out.
|
|
||||||
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
|
|
||||||
path := "../../go-libraries/go-flac/avx512_amd64.s"
|
|
||||||
if _, err := os.Stat(path); err != nil {
|
|
||||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
|
||||||
}
|
|
||||||
src, err := os.ReadFile(path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
f, errs := parser.Parse(path, string(src))
|
|
||||||
if len(errs) > 0 {
|
|
||||||
t.Fatalf("parse: %v", errs)
|
|
||||||
}
|
|
||||||
img, err := AssembleFile(f)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("AssembleFile: %v", err)
|
|
||||||
}
|
|
||||||
if len(img.Funcs) != 10 {
|
|
||||||
t.Errorf("functions = %d, want 10", len(img.Funcs))
|
|
||||||
}
|
|
||||||
|
|
||||||
// idx16 as the DATA directives define it: dwords 1..16.
|
|
||||||
idx := make([]byte, 0, 64)
|
|
||||||
for i := 1; i <= 16; i++ {
|
|
||||||
idx = append(idx, byte(i), 0, 0, 0)
|
|
||||||
}
|
|
||||||
image := img.Bytes()
|
|
||||||
base := img.Symbols["idx16"]
|
|
||||||
if base == 0 {
|
|
||||||
t.Fatal("idx16 not laid out")
|
|
||||||
}
|
|
||||||
if got := image[base : base+64]; hexCompact(got) != hexCompact(idx) {
|
|
||||||
t.Errorf("idx16 contents %x, want %x", got, idx)
|
|
||||||
}
|
|
||||||
|
|
||||||
// The VMOVDQU32 idx16(SB), Z13 load (62 71 7e 48 6f 2d + rel32) must
|
|
||||||
// resolve to idx16 within the image.
|
|
||||||
loads := 0
|
|
||||||
for _, fn := range img.Funcs {
|
|
||||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
|
||||||
pat := []byte{0x62, 0x71, 0x7e, 0x48, 0x6f, 0x2d}
|
|
||||||
for pos := 0; ; {
|
|
||||||
i := indexOf(code[pos:], pat)
|
|
||||||
if i < 0 {
|
|
||||||
break
|
|
||||||
}
|
|
||||||
i += pos
|
|
||||||
rel := int32(uint32(code[i+6]) | uint32(code[i+7])<<8 | uint32(code[i+8])<<16 | uint32(code[i+9])<<24)
|
|
||||||
target := fn.Offset + i + 10 + int(rel)
|
|
||||||
if target != base {
|
|
||||||
t.Errorf("%s: idx16 load at +%d targets 0x%x, want 0x%x", fn.Name, i, target, base)
|
|
||||||
}
|
|
||||||
loads++
|
|
||||||
pos = i + 10
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if loads != 1 {
|
|
||||||
t.Errorf("idx16 loads found = %d, want 1", loads)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// hexCompact renders bytes as a lowercase hex string without separators.
|
// hexCompact renders bytes as a lowercase hex string without separators.
|
||||||
func hexCompact(b []byte) string {
|
func hexCompact(b []byte) string {
|
||||||
const hexdig = "0123456789abcdef"
|
const hexdig = "0123456789abcdef"
|
||||||
@@ -724,17 +656,3 @@ func hexCompact(b []byte) string {
|
|||||||
}
|
}
|
||||||
return string(out)
|
return string(out)
|
||||||
}
|
}
|
||||||
|
|
||||||
// indexOf returns the index of the first occurrence of pat in b, or -1.
|
|
||||||
func indexOf(b, pat []byte) int {
|
|
||||||
for i := 0; i+len(pat) <= len(b); i++ {
|
|
||||||
j := 0
|
|
||||||
for j < len(pat) && b[i+j] == pat[j] {
|
|
||||||
j++
|
|
||||||
}
|
|
||||||
if j == len(pat) {
|
|
||||||
return i
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return -1
|
|
||||||
}
|
|
||||||
|
|||||||
+214
-86
@@ -22,9 +22,15 @@ import (
|
|||||||
//
|
//
|
||||||
// The object carries what the linker requires of an assembly object: the
|
// The object carries what the linker requires of an assembly object: the
|
||||||
// functions (non-package symbols, as cmd/asm emits them), the GLOBL data,
|
// functions (non-package symbols, as cmd/asm emits them), the GLOBL data,
|
||||||
// one FuncInfo per function, and the pc-value tables (pcsp, pcfile,
|
// one FuncInfo per function, the per-function DWARF symbols (the
|
||||||
// pcline, pcinline). DWARF and the implicit funcdata symbols are omitted;
|
// .debug_line program and the subprogram DIE, which the linker's DWARF
|
||||||
// the linker fills their defaults.
|
// pass reads verbatim), and the pc-value tables (pcsp, pcfile, pcline,
|
||||||
|
// pcinline). The implicit funcdata symbols are omitted; the linker fills
|
||||||
|
// their defaults.
|
||||||
|
//
|
||||||
|
// emitGOObject is architecture-agnostic; the per-architecture GOObject*
|
||||||
|
// methods supply the toolchain preamble, the MinLC (pc-value delta unit)
|
||||||
|
// and the relocation-type mapping for code relocations.
|
||||||
|
|
||||||
// GOOBJ block indices (cmd/internal/goobj).
|
// GOOBJ block indices (cmd/internal/goobj).
|
||||||
const (
|
const (
|
||||||
@@ -51,9 +57,11 @@ const (
|
|||||||
|
|
||||||
// Symbol kinds used by assembly objects (cmd/internal/objabi).
|
// Symbol kinds used by assembly objects (cmd/internal/objabi).
|
||||||
const (
|
const (
|
||||||
kindSTEXT = 1
|
kindSTEXT = 1
|
||||||
kindSRODATA = 3
|
kindSRODATA = 3
|
||||||
kindSDATA = 7
|
kindSDATA = 7
|
||||||
|
kindSDWARFFCN = 14
|
||||||
|
kindSDWARFLINES = 20
|
||||||
)
|
)
|
||||||
|
|
||||||
// Symbol flags (cmd/internal/goobj).
|
// Symbol flags (cmd/internal/goobj).
|
||||||
@@ -66,11 +74,13 @@ const (
|
|||||||
|
|
||||||
// Aux entry types (cmd/internal/goobj).
|
// Aux entry types (cmd/internal/goobj).
|
||||||
const (
|
const (
|
||||||
auxFuncInfo = 1
|
auxFuncInfo = 1
|
||||||
auxPcsp = 7
|
auxDwarfInfo = 3
|
||||||
auxPcfile = 8
|
auxDwarfLines = 6
|
||||||
auxPcline = 9
|
auxPcsp = 7
|
||||||
auxPcinline = 10
|
auxPcfile = 8
|
||||||
|
auxPcline = 9
|
||||||
|
auxPcinline = 10
|
||||||
)
|
)
|
||||||
|
|
||||||
// FuncInfo flags (internal/abi).
|
// FuncInfo flags (internal/abi).
|
||||||
@@ -80,7 +90,11 @@ const (
|
|||||||
)
|
)
|
||||||
|
|
||||||
// Relocation types (cmd/internal/objabi).
|
// Relocation types (cmd/internal/objabi).
|
||||||
const relocPCRel = 14
|
const (
|
||||||
|
relocPCRel = 14 // R_PCREL
|
||||||
|
relocAddr = 1 // R_ADDR
|
||||||
|
relocDWTXTADDRU4 = 106 // R_DWTXTADDR_U4
|
||||||
|
)
|
||||||
|
|
||||||
// Special package indices for symbol references.
|
// Special package indices for symbol references.
|
||||||
const (
|
const (
|
||||||
@@ -110,6 +124,13 @@ func (s goSym) append(b []byte, strOff map[string]uint32) []byte {
|
|||||||
return binary.LittleEndian.AppendUint32(b, s.align)
|
return binary.LittleEndian.AppendUint32(b, s.align)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// dwarfRelocSet attaches emitter-generated relocations (the DWARF
|
||||||
|
// lines/info symbols' address references) to a definition index.
|
||||||
|
type dwarfRelocSet struct {
|
||||||
|
si int
|
||||||
|
relocs []goobjReloc
|
||||||
|
}
|
||||||
|
|
||||||
// GOObject returns the image as a GOOBJ object file for the given package
|
// GOObject returns the image as a GOOBJ object file for the given package
|
||||||
// path (the linker qualifies the exported symbols with it, the way cmd/asm
|
// path (the linker qualifies the exported symbols with it, the way cmd/asm
|
||||||
// does with its -p flag). srcPath names the source file recorded in the
|
// does with its -p flag). srcPath names the source file recorded in the
|
||||||
@@ -117,19 +138,90 @@ func (s goSym) append(b []byte, strOff map[string]uint32) []byte {
|
|||||||
// captured from the installed go tool asm, so the output links with the
|
// captured from the installed go tool asm, so the output links with the
|
||||||
// toolchain it was produced on — exactly like a real assembly object.
|
// toolchain it was produced on — exactly like a real assembly object.
|
||||||
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
|
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
|
||||||
if pkgPath == "" {
|
|
||||||
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
|
|
||||||
}
|
|
||||||
pre, err := toolchainObjectPreamble()
|
pre, err := toolchainObjectPreamble()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
// amd64: MinLC 1, R_PCREL for the code relocations.
|
||||||
|
return img.emitGOObject(pkgPath, srcPath, pre, 1, func(Reloc) (uint16, uint8) { return relocPCRel, 4 })
|
||||||
|
}
|
||||||
|
|
||||||
// The symbol tables. Package definitions: the GLOBL symbols, then one
|
// emitGOObject assembles the GOOBJ payload for any architecture. pre is
|
||||||
// anonymous FuncInfo symbol per function. Non-package definitions: the
|
// the toolchain's object preamble; minLC is the architecture's minimum
|
||||||
// pc-value tables and the functions themselves, as cmd/asm lays them
|
// instruction length, the unit of the pc-value table deltas; relocField
|
||||||
// out. defIdx maps a GLOBL's bare name to its definition index for the
|
// maps a code relocation to its objabi relocation type and the width of
|
||||||
// relocations; fnNpIdx maps a function to its non-package index.
|
// the instruction field the linker writes.
|
||||||
|
func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, relocField func(Reloc) (uint16, uint8)) ([]byte, error) {
|
||||||
|
if pkgPath == "" {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
|
||||||
|
}
|
||||||
|
|
||||||
|
// The non-package definitions first — the DWARF symbols reference the
|
||||||
|
// functions by these indices: per function the four pc-value tables
|
||||||
|
// and the function itself, as cmd/asm lays them out.
|
||||||
|
type npSym struct {
|
||||||
|
sym goSym
|
||||||
|
data []byte
|
||||||
|
}
|
||||||
|
var nps []npSym
|
||||||
|
type pcRefs struct{ sp, file, line, inl int }
|
||||||
|
pcIdx := make([]pcRefs, len(img.Funcs))
|
||||||
|
fnNpIdx := make([]int, len(img.Funcs))
|
||||||
|
for i, fn := range img.Funcs {
|
||||||
|
tables := []struct {
|
||||||
|
data []byte
|
||||||
|
dst *int
|
||||||
|
}{
|
||||||
|
{pcspTable(fn, minLC), &pcIdx[i].sp},
|
||||||
|
{pcValueFlat(0, fn.Size, minLC), &pcIdx[i].file},
|
||||||
|
{pcValueFlat(int32(fn.Line), fn.Size, minLC), &pcIdx[i].line},
|
||||||
|
{pcValueFlat(-1, fn.Size, minLC), &pcIdx[i].inl},
|
||||||
|
}
|
||||||
|
for _, t := range tables {
|
||||||
|
*t.dst = len(nps)
|
||||||
|
nps = append(nps, npSym{
|
||||||
|
sym: goSym{typ: kindSRODATA, size: uint32(len(t.data)), align: 1},
|
||||||
|
data: t.data,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
name := fn.Name
|
||||||
|
abi := uint16(0)
|
||||||
|
if fn.Static {
|
||||||
|
abi = symABIStatic
|
||||||
|
} else {
|
||||||
|
name = pkgPath + "." + name
|
||||||
|
}
|
||||||
|
flag := uint8(0)
|
||||||
|
if fn.NoSplit {
|
||||||
|
flag |= symFlagNoSplit
|
||||||
|
}
|
||||||
|
fnNpIdx[i] = len(nps)
|
||||||
|
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
|
||||||
|
for _, r := range fn.Relocs {
|
||||||
|
// Only the amd64 encoder resolves file-local static symbols
|
||||||
|
// into a disp32 field at assemble time; GOOBJ must leave that
|
||||||
|
// field zero for the linker to fill. The RISC-V and LoongArch
|
||||||
|
// encoders emit zero immediates with a relocation instead, and
|
||||||
|
// their relocations cover whole AUIPC/pcalau12i pairs, so
|
||||||
|
// zeroing r.Off would erase the opcode/register bits the linker
|
||||||
|
// preserves when it patches only the immediate.
|
||||||
|
if r.Kind != RelPCRel32 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if r.Off >= 0 && r.Off+4 <= len(code) {
|
||||||
|
code[r.Off], code[r.Off+1], code[r.Off+2], code[r.Off+3] = 0, 0, 0, 0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
nps = append(nps, npSym{
|
||||||
|
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
|
||||||
|
data: code,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// The package definitions: the GLOBL symbols, then, per function, the
|
||||||
|
// FuncInfo and the two DWARF symbols (the .debug_line program and the
|
||||||
|
// subprogram DIE). defIdx maps a GLOBL's bare name to its definition
|
||||||
|
// index for the code relocations.
|
||||||
var defs []goSym
|
var defs []goSym
|
||||||
var defData [][]byte
|
var defData [][]byte
|
||||||
defIdx := map[string]int{}
|
defIdx := map[string]int{}
|
||||||
@@ -155,74 +247,82 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
|
|||||||
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
|
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
|
||||||
}
|
}
|
||||||
fnFiIdx := make([]int, len(img.Funcs))
|
fnFiIdx := make([]int, len(img.Funcs))
|
||||||
for i := range img.Funcs {
|
fnLinesIdx := make([]int, len(img.Funcs))
|
||||||
data := marshalFuncInfo(img.Funcs[i])
|
fnDIEIdx := make([]int, len(img.Funcs))
|
||||||
|
var dwarfRelocs []dwarfRelocSet
|
||||||
|
for i, fn := range img.Funcs {
|
||||||
|
data := marshalFuncInfo(fn)
|
||||||
fnFiIdx[i] = len(defs)
|
fnFiIdx[i] = len(defs)
|
||||||
defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))})
|
defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))})
|
||||||
defData = append(defData, data)
|
defData = append(defData, data)
|
||||||
}
|
|
||||||
|
|
||||||
type npSym struct {
|
|
||||||
sym goSym
|
|
||||||
data []byte
|
|
||||||
}
|
|
||||||
var nps []npSym
|
|
||||||
type pcRefs struct{ sp, file, line, inl int }
|
|
||||||
pcIdx := make([]pcRefs, len(img.Funcs))
|
|
||||||
fnNpIdx := make([]int, len(img.Funcs))
|
|
||||||
for i, fn := range img.Funcs {
|
|
||||||
tables := []struct {
|
|
||||||
data []byte
|
|
||||||
dst *int
|
|
||||||
}{
|
|
||||||
{pcspTable(fn), &pcIdx[i].sp},
|
|
||||||
{pcValueFlat(0, fn.Size), &pcIdx[i].file},
|
|
||||||
{pcValueFlat(int32(fn.Line), fn.Size), &pcIdx[i].line},
|
|
||||||
{pcValueFlat(-1, fn.Size), &pcIdx[i].inl},
|
|
||||||
}
|
|
||||||
for _, t := range tables {
|
|
||||||
*t.dst = len(nps)
|
|
||||||
nps = append(nps, npSym{
|
|
||||||
sym: goSym{typ: kindSRODATA, size: uint32(len(t.data)), align: 1},
|
|
||||||
data: t.data,
|
|
||||||
})
|
|
||||||
}
|
|
||||||
name := fn.Name
|
name := fn.Name
|
||||||
abi := uint16(0)
|
if !fn.Static {
|
||||||
if fn.Static {
|
|
||||||
abi = symABIStatic
|
|
||||||
} else {
|
|
||||||
name = pkgPath + "." + name
|
name = pkgPath + "." + name
|
||||||
}
|
}
|
||||||
flag := uint8(0)
|
|
||||||
if fn.NoSplit {
|
// The DWARF symbols: the .debug_line state-machine program and the
|
||||||
flag |= symFlagNoSplit
|
// subprogram DIE, both referencing the function by its non-package
|
||||||
|
// index (package definitions, like cmd/asm's).
|
||||||
|
lines, lrel := goobjDwarfLines(fn, fnNpIdx[i])
|
||||||
|
fnLinesIdx[i] = len(defs)
|
||||||
|
defs = append(defs, goSym{typ: kindSDWARFLINES, size: uint32(len(lines))})
|
||||||
|
defData = append(defData, lines)
|
||||||
|
die, drel := goobjDwarfInfo(fn, name, fnNpIdx[i])
|
||||||
|
fnDIEIdx[i] = len(defs)
|
||||||
|
defs = append(defs, goSym{typ: kindSDWARFFCN, size: uint32(len(die))})
|
||||||
|
defData = append(defData, die)
|
||||||
|
dwarfRelocs = append(dwarfRelocs,
|
||||||
|
dwarfRelocSet{si: fnLinesIdx[i], relocs: lrel},
|
||||||
|
dwarfRelocSet{si: fnDIEIdx[i], relocs: drel},
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Resolve external symbol references (cross-package). Build the
|
||||||
|
// package index table and determine each external symbol's SymIdx
|
||||||
|
// by reading the target package's export data.
|
||||||
|
var extPkgTable []string
|
||||||
|
var extPkgIdx map[string]int
|
||||||
|
var extSymIdx map[string]int
|
||||||
|
if len(img.Externals) > 0 {
|
||||||
|
var err error
|
||||||
|
extPkgTable, extPkgIdx, extSymIdx, err = resolveExternalSymbols(img.Externals)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: resolving external symbols: %w", err)
|
||||||
}
|
}
|
||||||
fnNpIdx[i] = len(nps)
|
|
||||||
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
|
|
||||||
for _, r := range fn.Relocs {
|
|
||||||
// The linker writes the resolved displacement into the field;
|
|
||||||
// leave it zero, as cmd/asm's object does.
|
|
||||||
if r.Off >= 0 && r.Off+4 <= len(code) {
|
|
||||||
code[r.Off], code[r.Off+1], code[r.Off+2], code[r.Off+3] = 0, 0, 0, 0
|
|
||||||
}
|
|
||||||
}
|
|
||||||
nps = append(nps, npSym{
|
|
||||||
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
|
|
||||||
data: code,
|
|
||||||
})
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// Relocations, per defined symbol in definition order (package defs,
|
// Relocations, per defined symbol in definition order (package defs,
|
||||||
// then non-package defs). Only file-local GLOBL references resolve;
|
// then non-package defs).
|
||||||
// external symbols need the import machinery of a later increment.
|
|
||||||
nsyms := len(defs) + len(nps)
|
nsyms := len(defs) + len(nps)
|
||||||
symRelocs := make([][]byte, nsyms) // flat 23-byte records
|
symRelocs := make([][]byte, nsyms) // flat 23-byte records
|
||||||
for i, fn := range img.Funcs {
|
for i, fn := range img.Funcs {
|
||||||
si := len(defs) + fnNpIdx[i]
|
si := len(defs) + fnNpIdx[i]
|
||||||
for _, r := range fn.Relocs {
|
for _, r := range fn.Relocs {
|
||||||
|
typ, size := relocField(r)
|
||||||
if r.External {
|
if r.External {
|
||||||
return nil, fmt.Errorf("GOOBJ emission: external symbol %q is not supported yet", r.Name)
|
// Split package-qualified name: "runtime·morestack" → runtime, morestack.
|
||||||
|
pkg, name := splitQualified(r.Name)
|
||||||
|
if pkg == "" {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: external symbol %q has no package prefix", r.Name)
|
||||||
|
}
|
||||||
|
pIdx, ok := extPkgIdx[pkg]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: package %q not resolved", pkg)
|
||||||
|
}
|
||||||
|
sIdx, ok := extSymIdx[pkg+"·"+name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: symbol %s·%s not resolved", pkg, name)
|
||||||
|
}
|
||||||
|
var rec [23]byte
|
||||||
|
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
|
||||||
|
rec[4] = size // field width
|
||||||
|
binary.LittleEndian.PutUint16(rec[5:], typ)
|
||||||
|
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
|
||||||
|
binary.LittleEndian.PutUint32(rec[15:], uint32(pIdx))
|
||||||
|
binary.LittleEndian.PutUint32(rec[19:], uint32(sIdx))
|
||||||
|
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
||||||
|
continue
|
||||||
}
|
}
|
||||||
di, ok := defIdx[r.Name]
|
di, ok := defIdx[r.Name]
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -230,17 +330,30 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
|
|||||||
}
|
}
|
||||||
var rec [23]byte
|
var rec [23]byte
|
||||||
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
|
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
|
||||||
rec[4] = 4 // field width
|
rec[4] = size // field width
|
||||||
binary.LittleEndian.PutUint16(rec[5:], relocPCRel)
|
binary.LittleEndian.PutUint16(rec[5:], typ)
|
||||||
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
|
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
|
||||||
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
|
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
|
||||||
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
|
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
|
||||||
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// The DWARF symbols' own relocations (the function address references).
|
||||||
|
for _, ds := range dwarfRelocs {
|
||||||
|
for _, r := range ds.relocs {
|
||||||
|
var rec [23]byte
|
||||||
|
binary.LittleEndian.PutUint32(rec[0:], uint32(r.off))
|
||||||
|
rec[4] = r.siz
|
||||||
|
binary.LittleEndian.PutUint16(rec[5:], r.typ)
|
||||||
|
binary.LittleEndian.PutUint64(rec[7:], uint64(r.add))
|
||||||
|
binary.LittleEndian.PutUint32(rec[15:], r.pkg)
|
||||||
|
binary.LittleEndian.PutUint32(rec[19:], r.sym)
|
||||||
|
symRelocs[ds.si] = append(symRelocs[ds.si], rec[:]...)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Aux entries per function: FuncInfo, then the four pc tables.
|
// Aux entries per function: FuncInfo, the DWARF symbols, then the four
|
||||||
// References into the non-package table use pkgIdxNone.
|
// pc tables. References into the non-package table use pkgIdxNone.
|
||||||
symAux := make([][]byte, nsyms)
|
symAux := make([][]byte, nsyms)
|
||||||
for i := range img.Funcs {
|
for i := range img.Funcs {
|
||||||
si := len(defs) + fnNpIdx[i]
|
si := len(defs) + fnNpIdx[i]
|
||||||
@@ -252,10 +365,14 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
|
|||||||
symAux[si] = append(symAux[si], rec[:]...)
|
symAux[si] = append(symAux[si], rec[:]...)
|
||||||
}
|
}
|
||||||
aux(auxFuncInfo, pkgIdxSelf, uint32(fnFiIdx[i]))
|
aux(auxFuncInfo, pkgIdxSelf, uint32(fnFiIdx[i]))
|
||||||
aux(auxPcsp, pkgIdxNone, uint32(len(defs)+pcIdx[i].sp))
|
aux(auxDwarfInfo, pkgIdxSelf, uint32(fnDIEIdx[i]))
|
||||||
aux(auxPcfile, pkgIdxNone, uint32(len(defs)+pcIdx[i].file))
|
aux(auxDwarfLines, pkgIdxSelf, uint32(fnLinesIdx[i]))
|
||||||
aux(auxPcline, pkgIdxNone, uint32(len(defs)+pcIdx[i].line))
|
// The pc-table references are 0-based within the non-package
|
||||||
aux(auxPcinline, pkgIdxNone, uint32(len(defs)+pcIdx[i].inl))
|
// definitions; the loader adds the package-definition count itself.
|
||||||
|
aux(auxPcsp, pkgIdxNone, uint32(pcIdx[i].sp))
|
||||||
|
aux(auxPcfile, pkgIdxNone, uint32(pcIdx[i].file))
|
||||||
|
aux(auxPcline, pkgIdxNone, uint32(pcIdx[i].line))
|
||||||
|
aux(auxPcinline, pkgIdxNone, uint32(pcIdx[i].inl))
|
||||||
}
|
}
|
||||||
|
|
||||||
// The string table. Absolute offsets: it starts right after the
|
// The string table. Absolute offsets: it starts right after the
|
||||||
@@ -291,7 +408,16 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
|
|||||||
for _, s := range nps {
|
for _, s := range nps {
|
||||||
npdefBlk = s.sym.append(npdefBlk, strOff)
|
npdefBlk = s.sym.append(npdefBlk, strOff)
|
||||||
}
|
}
|
||||||
pkgIdxBlk := stringRef(nil, "") // index 0: the dummy invalid package
|
|
||||||
|
// Package index table: index 0 is the dummy invalid package.
|
||||||
|
// External packages follow, in pkgIdx order.
|
||||||
|
for _, pkg := range extPkgTable {
|
||||||
|
addStr(pkg)
|
||||||
|
}
|
||||||
|
pkgIdxBlk := stringRef(nil, "") // index 0: dummy
|
||||||
|
for _, pkg := range extPkgTable {
|
||||||
|
pkgIdxBlk = stringRef(pkgIdxBlk, pkg)
|
||||||
|
}
|
||||||
fileBlk := stringRef(nil, srcPath)
|
fileBlk := stringRef(nil, srcPath)
|
||||||
|
|
||||||
var relocBlk, auxBlk, dataBlk []byte
|
var relocBlk, auxBlk, dataBlk []byte
|
||||||
@@ -374,19 +500,21 @@ func marshalFuncInfo(fn FuncLayout) []byte {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// pcValueFlat encodes a pc-value table holding v over the whole function.
|
// pcValueFlat encodes a pc-value table holding v over the whole function.
|
||||||
func pcValueFlat(v int32, size int) []byte {
|
// The pc deltas are in MinLC units (the runtime scales them by the
|
||||||
|
// architecture's minimum instruction length).
|
||||||
|
func pcValueFlat(v int32, size, minLC int) []byte {
|
||||||
// The table is delta-encoded from an implicit value of -1: a varint
|
// The table is delta-encoded from an implicit value of -1: a varint
|
||||||
// value delta, an unsigned pc delta to the end, and a zero terminator.
|
// value delta, an unsigned pc delta to the end, and a zero terminator.
|
||||||
out := binary.AppendVarint(nil, int64(v)+1)
|
out := binary.AppendVarint(nil, int64(v)+1)
|
||||||
out = binary.AppendUvarint(out, uint64(size))
|
out = binary.AppendUvarint(out, uint64(size/minLC))
|
||||||
return append(out, 0)
|
return append(out, 0)
|
||||||
}
|
}
|
||||||
|
|
||||||
// pcspTable encodes the stack-adjustment table: the SP delta in effect at
|
// pcspTable encodes the stack-adjustment table: the SP delta in effect at
|
||||||
// every pc, from the function's prologue and epilogue boundaries.
|
// every pc, from the function's prologue and epilogue boundaries.
|
||||||
func pcspTable(fn FuncLayout) []byte {
|
func pcspTable(fn FuncLayout, minLC int) []byte {
|
||||||
if len(fn.Spadj) == 0 {
|
if len(fn.Spadj) == 0 {
|
||||||
return pcValueFlat(0, fn.Size)
|
return pcValueFlat(0, fn.Size, minLC)
|
||||||
}
|
}
|
||||||
pts := make([]SpadjStep, 0, len(fn.Spadj)+1)
|
pts := make([]SpadjStep, 0, len(fn.Spadj)+1)
|
||||||
pts = append(pts, SpadjStep{PC: 0, Value: 0})
|
pts = append(pts, SpadjStep{PC: 0, Value: 0})
|
||||||
@@ -394,11 +522,11 @@ func pcspTable(fn FuncLayout) []byte {
|
|||||||
out := binary.AppendVarint(nil, int64(pts[0].Value)+1)
|
out := binary.AppendVarint(nil, int64(pts[0].Value)+1)
|
||||||
cur, old := pts[0].PC, pts[0].Value
|
cur, old := pts[0].PC, pts[0].Value
|
||||||
for _, p := range pts[1:] {
|
for _, p := range pts[1:] {
|
||||||
out = binary.AppendUvarint(out, uint64(p.PC-cur))
|
out = binary.AppendUvarint(out, uint64((p.PC-cur)/minLC))
|
||||||
out = binary.AppendVarint(out, int64(p.Value-old))
|
out = binary.AppendVarint(out, int64(p.Value-old))
|
||||||
cur, old = p.PC, p.Value
|
cur, old = p.PC, p.Value
|
||||||
}
|
}
|
||||||
out = binary.AppendUvarint(out, uint64(fn.Size-cur))
|
out = binary.AppendUvarint(out, uint64((fn.Size-cur)/minLC))
|
||||||
return append(out, 0)
|
return append(out, 0)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,188 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
)
|
||||||
|
|
||||||
|
// This file generates the per-function DWARF symbols the linker's DWARF
|
||||||
|
// pass requires of an assembly object, byte-identical to what cmd/asm
|
||||||
|
// emits: the .debug_line state-machine program (SDWARFLINES) and the
|
||||||
|
// subprogram DIE (SDWARFFCN). The linker copies the DIE and line-program
|
||||||
|
// bytes verbatim into .debug_info and .debug_line, fixing up their
|
||||||
|
// relocations, so the formats here must match cmd/internal/dwarf's
|
||||||
|
// DW_ABRV_FUNCTION and generateDebugLinesSymbol exactly.
|
||||||
|
//
|
||||||
|
// DWARF5 is assumed throughout (the toolchain's default on Linux and the
|
||||||
|
// other non-Darwin targets gasm supports).
|
||||||
|
|
||||||
|
// Line-program parameters (cmd/internal/obj/dwarf.go).
|
||||||
|
const (
|
||||||
|
dwLineBase = -4
|
||||||
|
dwLineRange = 10
|
||||||
|
dwOpcodeBase = 11
|
||||||
|
dwPCRange = (255 - dwOpcodeBase) / dwLineRange
|
||||||
|
)
|
||||||
|
|
||||||
|
// goobjReloc is one relocation attached to an emitter-generated symbol
|
||||||
|
// (the DWARF lines/info symbols), in goobj's on-disk encoding fields.
|
||||||
|
type goobjReloc struct {
|
||||||
|
off int32
|
||||||
|
siz uint8
|
||||||
|
typ uint16
|
||||||
|
add int64
|
||||||
|
pkg uint32
|
||||||
|
sym uint32
|
||||||
|
}
|
||||||
|
|
||||||
|
// goobjDwarfLines builds the function's .debug_line state-machine program:
|
||||||
|
// an LNE_set_address extended opcode establishing the function's start
|
||||||
|
// address (carrying the R_ADDR relocation), one row per source line
|
||||||
|
// change across the function's instructions, an advance to the end of the
|
||||||
|
// function and an end-of-sequence opcode. The linker appends these bytes
|
||||||
|
// after the unit's line header, so they must start with the address and
|
||||||
|
// leave the state machine terminated.
|
||||||
|
func goobjDwarfLines(fn FuncLayout, fnNpIdx int) ([]byte, []goobjReloc) {
|
||||||
|
// Rows: the prologue, if any, then the body instructions (fn.Lines
|
||||||
|
// covers the body only). The first body offset > 0 means a prologue
|
||||||
|
// precedes it; the toolchain reports the prologue on the TEXT line.
|
||||||
|
pts := make([]LineEntry, 0, len(fn.Lines)+1)
|
||||||
|
if len(fn.Lines) == 0 || fn.Lines[0].Offset > 0 {
|
||||||
|
pts = append(pts, LineEntry{Offset: 0, Line: fn.Line})
|
||||||
|
}
|
||||||
|
pts = append(pts, fn.Lines...)
|
||||||
|
|
||||||
|
out := []byte{0, 9, 2, 0, 0, 0, 0, 0, 0, 0, 0} // LNE_set_address, address zeroed
|
||||||
|
relocs := []goobjReloc{{
|
||||||
|
off: 3, siz: 8, typ: relocAddr,
|
||||||
|
pkg: pkgIdxNone, sym: uint32(fnNpIdx),
|
||||||
|
}}
|
||||||
|
|
||||||
|
// The state machine starts at line 1, pc 0 (function-relative); the
|
||||||
|
// implicit initial pc is the function entry, so the first pc delta is
|
||||||
|
// against 0.
|
||||||
|
line := int64(1)
|
||||||
|
pc := uint64(0)
|
||||||
|
for _, p := range pts {
|
||||||
|
if p.Line == 0 || uint64(p.Offset) < pc {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Rows mark source-line changes only; the pc delta is measured from
|
||||||
|
// the previous row, not the previous instruction.
|
||||||
|
if int64(p.Line) == line {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
deltaPC := uint64(p.Offset) - pc
|
||||||
|
deltaLC := int64(p.Line) - line
|
||||||
|
out = dwPutPCLCDelta(out, deltaPC, deltaLC)
|
||||||
|
line, pc = int64(p.Line), uint64(p.Offset)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Cover the rest of the function and close the sequence.
|
||||||
|
if end := uint64(fn.Size) - pc; end > 0 {
|
||||||
|
out = append(out, 2) // DW_LNS_advance_pc
|
||||||
|
out = binary.AppendUvarint(out, end)
|
||||||
|
}
|
||||||
|
out = append(out, 0, 1, 1) // LNE_end_sequence
|
||||||
|
return out, relocs
|
||||||
|
}
|
||||||
|
|
||||||
|
// dwPutPCLCDelta encodes one (pcDelta, lineDelta) step as the shortest
|
||||||
|
// special opcode plus any standard-opcode remainder, exactly like
|
||||||
|
// cmd/internal/obj's putpclcdelta.
|
||||||
|
func dwPutPCLCDelta(b []byte, deltaPC uint64, deltaLC int64) []byte {
|
||||||
|
opcode := dwSelectOpcode(deltaPC, deltaLC)
|
||||||
|
deltaPC -= uint64((opcode - dwOpcodeBase) / dwLineRange)
|
||||||
|
deltaLC -= (opcode-dwOpcodeBase)%dwLineRange + dwLineBase
|
||||||
|
|
||||||
|
// The remainder: standard opcodes first, then the special opcode
|
||||||
|
// (which emits the row).
|
||||||
|
if deltaPC != 0 {
|
||||||
|
switch {
|
||||||
|
case deltaPC <= uint64(dwPCRange):
|
||||||
|
opcode -= dwLineRange * int64(uint64(dwPCRange)-deltaPC)
|
||||||
|
b = append(b, 8) // DW_LNS_const_add_pc
|
||||||
|
case (1<<14) <= deltaPC && deltaPC < (1<<16):
|
||||||
|
b = append(b, 9) // DW_LNS_fixed_advance_pc
|
||||||
|
b = binary.LittleEndian.AppendUint16(b, uint16(deltaPC))
|
||||||
|
default:
|
||||||
|
b = append(b, 2) // DW_LNS_advance_pc
|
||||||
|
b = binary.AppendUvarint(b, deltaPC)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if deltaLC != 0 {
|
||||||
|
b = append(b, 3) // DW_LNS_advance_line
|
||||||
|
b = binary.AppendVarint(b, deltaLC)
|
||||||
|
}
|
||||||
|
return append(b, byte(opcode))
|
||||||
|
}
|
||||||
|
|
||||||
|
// dwSelectOpcode picks the special opcode for (deltaPC, deltaLC) per
|
||||||
|
// cmd/internal/obj's putpclcdelta selection logic.
|
||||||
|
func dwSelectOpcode(deltaPC uint64, deltaLC int64) int64 {
|
||||||
|
switch {
|
||||||
|
case deltaLC < dwLineBase:
|
||||||
|
if deltaPC >= uint64(dwPCRange) {
|
||||||
|
return dwOpcodeBase + dwLineRange*dwPCRange
|
||||||
|
}
|
||||||
|
return dwOpcodeBase + dwLineRange*int64(deltaPC)
|
||||||
|
case deltaLC < dwLineBase+dwLineRange:
|
||||||
|
if deltaPC >= uint64(dwPCRange) {
|
||||||
|
op := int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*dwPCRange
|
||||||
|
if op > 255 {
|
||||||
|
op -= dwLineRange
|
||||||
|
}
|
||||||
|
return op
|
||||||
|
}
|
||||||
|
return int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*int64(deltaPC)
|
||||||
|
default:
|
||||||
|
if deltaPC <= uint64(dwPCRange) {
|
||||||
|
op := int64(dwOpcodeBase) + (dwLineRange - 1) + dwLineRange*int64(deltaPC)
|
||||||
|
if op > 255 {
|
||||||
|
op = 255
|
||||||
|
}
|
||||||
|
return op
|
||||||
|
}
|
||||||
|
switch deltaPC - uint64(dwPCRange) {
|
||||||
|
case uint64(dwPCRange), (1 << 7) - 1, (1 << 16) - 1, (1 << 21) - 1,
|
||||||
|
(1 << 28) - 1, (1 << 35) - 1, (1 << 42) - 1, (1 << 49) - 1,
|
||||||
|
(1 << 56) - 1, (1 << 63) - 1:
|
||||||
|
return 255
|
||||||
|
default:
|
||||||
|
// 250: the toolchain's "249" comment is stale.
|
||||||
|
return dwOpcodeBase + dwLineRange*dwPCRange - 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// goobjDwarfInfo builds the function's DWARF5 subprogram DIE (abbrev
|
||||||
|
// DW_ABRV_FUNCTION): name, low_pc as a .debug_addr index (the
|
||||||
|
// R_DWTXTADDR_U4 relocation), high_pc as the size, the call-frame-CFA
|
||||||
|
// frame base, the decl file/line and the external flag. name is the
|
||||||
|
// symbol's object name (package-qualified unless static).
|
||||||
|
func goobjDwarfInfo(fn FuncLayout, name string, fnNpIdx int) ([]byte, []goobjReloc) {
|
||||||
|
out := []byte{3} // DW_ABRV_FUNCTION
|
||||||
|
out = append(out, name...)
|
||||||
|
out = append(out, 0)
|
||||||
|
|
||||||
|
addrx := len(out)
|
||||||
|
out = append(out, 0, 0, 0, 0) // DW_AT_low_pc: addrx slot, zeroed
|
||||||
|
out = binary.AppendUvarint(out, uint64(fn.Size))
|
||||||
|
out = append(out, 1, 0x9c) // DW_AT_frame_base: block1, DW_OP_call_frame_cfa
|
||||||
|
out = binary.LittleEndian.AppendUint32(out, 1)
|
||||||
|
out = binary.AppendUvarint(out, uint64(fn.Line))
|
||||||
|
if fn.Static {
|
||||||
|
out = append(out, 0)
|
||||||
|
} else {
|
||||||
|
out = append(out, 1) // DW_AT_external
|
||||||
|
}
|
||||||
|
out = append(out, 0) // end of children
|
||||||
|
|
||||||
|
relocs := []goobjReloc{{
|
||||||
|
off: int32(addrx), siz: 4, typ: relocDWTXTADDRU4,
|
||||||
|
pkg: pkgIdxNone, sym: uint32(fnNpIdx),
|
||||||
|
}}
|
||||||
|
return out, relocs
|
||||||
|
}
|
||||||
@@ -0,0 +1,214 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/binary"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestDWSelectOpcode checks the special-opcode selection against
|
||||||
|
// hand-computed values for the boundary cases: line deltas below, inside
|
||||||
|
// and above the line range, and pc deltas at and beyond PC_RANGE (24).
|
||||||
|
func TestDWSelectOpcode(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
deltaPC uint64
|
||||||
|
deltaLC int64
|
||||||
|
want int64
|
||||||
|
}{
|
||||||
|
{0, 2, 17}, // the common single-instruction step
|
||||||
|
{0, -4, 11}, // deltaLC == LINE_BASE
|
||||||
|
{0, -5, 11}, // deltaLC below LINE_BASE: opcode adds nothing
|
||||||
|
{0, 6, 20}, // deltaLC == LINE_BASE+LINE_RANGE, remainder via advance_line
|
||||||
|
{4, 1, 56}, // the 4-byte loong64 instruction step
|
||||||
|
{23, 1, 246}, // deltaPC == PC_RANGE-1
|
||||||
|
{24, 1, 246}, // deltaPC == PC_RANGE: wraps past 255
|
||||||
|
{25, 1, 246}, // deltaPC past PC_RANGE (the const_add_pc remainder adjusts it later)
|
||||||
|
{100, 1, 246},
|
||||||
|
{151, 10, 255}, // deltaPC-PC_RANGE == (1<<7)-1, large line delta
|
||||||
|
{100, 10, 250}, // deltaPC-PC_RANGE not on a switch boundary
|
||||||
|
{23, 10, 250}, // large line delta inside PC_RANGE
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
if got := dwSelectOpcode(c.deltaPC, c.deltaLC); got != c.want {
|
||||||
|
t.Errorf("dwSelectOpcode(%d, %d) = %d, want %d", c.deltaPC, c.deltaLC, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// decodeDWLineProgram decodes a .debug_line state-machine program (as
|
||||||
|
// emitted by goobjDwarfLines) into (pc, line) rows.
|
||||||
|
func decodeDWLineProgram(t *testing.T, b []byte) (pcs []uint64, lines []int64) {
|
||||||
|
t.Helper()
|
||||||
|
pc, line := uint64(0), int64(1)
|
||||||
|
emit := func() {
|
||||||
|
if len(pcs) == 0 || pcs[len(pcs)-1] != pc || lines[len(lines)-1] != line {
|
||||||
|
pcs = append(pcs, pc)
|
||||||
|
lines = append(lines, line)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
advancePC := func(delta uint64) { pc += delta }
|
||||||
|
advanceLine := func(delta int64) { line += delta }
|
||||||
|
for i := 0; i < len(b); {
|
||||||
|
op := b[i]
|
||||||
|
i++
|
||||||
|
switch {
|
||||||
|
case op == 0: // extended opcode
|
||||||
|
ln, n := binary.Uvarint(b[i:])
|
||||||
|
i += n
|
||||||
|
sub := b[i]
|
||||||
|
i++
|
||||||
|
_ = ln
|
||||||
|
switch sub {
|
||||||
|
case 2: // DW_LNE_set_address: 8-byte address
|
||||||
|
pc = binary.LittleEndian.Uint64(b[i:])
|
||||||
|
i += 8
|
||||||
|
case 1: // DW_LNE_end_sequence
|
||||||
|
// terminates the sequence; no new row
|
||||||
|
}
|
||||||
|
case op == 2: // DW_LNS_advance_pc
|
||||||
|
v, n := binary.Uvarint(b[i:])
|
||||||
|
i += n
|
||||||
|
advancePC(v)
|
||||||
|
case op == 3: // DW_LNS_advance_line
|
||||||
|
v, n := binary.Varint(b[i:])
|
||||||
|
i += n
|
||||||
|
advanceLine(v)
|
||||||
|
case op == 8: // DW_LNS_const_add_pc
|
||||||
|
advancePC(uint64(dwPCRange))
|
||||||
|
case op == 9: // DW_LNS_fixed_advance_pc
|
||||||
|
advancePC(uint64(binary.LittleEndian.Uint16(b[i:])))
|
||||||
|
i += 2
|
||||||
|
case op >= dwOpcodeBase: // special opcode
|
||||||
|
advancePC(uint64((int64(op) - dwOpcodeBase) / dwLineRange))
|
||||||
|
advanceLine((int64(op)-dwOpcodeBase)%dwLineRange + dwLineBase)
|
||||||
|
emit()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return pcs, lines
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGoobjDwarfLinesRows checks the emitted line program's rows for
|
||||||
|
// synthetic functions: a zero-frame function with one instruction per
|
||||||
|
// line, a framed function (the prologue row is prepended on the TEXT
|
||||||
|
// line), instructions sharing a line, and a function with a large pc gap
|
||||||
|
// (the const_add_pc remainder path).
|
||||||
|
func TestGoobjDwarfLinesRows(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
fn FuncLayout
|
||||||
|
want [][2]int64 // (pc, line)
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
"one instruction per line",
|
||||||
|
FuncLayout{Size: 20, Line: 2, Lines: []LineEntry{
|
||||||
|
{0, 3}, {4, 4}, {8, 5}, {12, 6}, {16, 7},
|
||||||
|
}},
|
||||||
|
[][2]int64{{0, 3}, {4, 4}, {8, 5}, {12, 6}, {16, 7}},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"framed: prologue row on the TEXT line",
|
||||||
|
FuncLayout{Size: 24, Line: 2, Lines: []LineEntry{
|
||||||
|
{12, 3}, {16, 4},
|
||||||
|
}},
|
||||||
|
[][2]int64{{0, 2}, {12, 3}, {16, 4}},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"instructions sharing a line fold into one row",
|
||||||
|
FuncLayout{Size: 16, Line: 2, Lines: []LineEntry{
|
||||||
|
{0, 3}, {4, 3}, {8, 4}, {12, 4},
|
||||||
|
}},
|
||||||
|
[][2]int64{{0, 3}, {8, 4}},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"large gap crosses PC_RANGE",
|
||||||
|
FuncLayout{Size: 60, Line: 2, Lines: []LineEntry{
|
||||||
|
{0, 3}, {40, 4},
|
||||||
|
}},
|
||||||
|
[][2]int64{{0, 3}, {40, 4}},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
prog, relocs := goobjDwarfLines(c.fn, 0)
|
||||||
|
if len(relocs) != 1 || relocs[0].off != 3 || relocs[0].siz != 8 || relocs[0].typ != relocAddr || relocs[0].sym != 0 {
|
||||||
|
t.Fatalf("relocs = %+v", relocs)
|
||||||
|
}
|
||||||
|
pcs, lines := decodeDWLineProgram(t, prog)
|
||||||
|
if len(pcs) != len(c.want) {
|
||||||
|
t.Fatalf("rows = %d (%v / %v), want %d", len(pcs), pcs, lines, len(c.want))
|
||||||
|
}
|
||||||
|
for i, w := range c.want {
|
||||||
|
if pcs[i] != uint64(w[0]) || lines[i] != w[1] {
|
||||||
|
t.Errorf("row %d = (%d, %d), want (%d, %d)", i, pcs[i], lines[i], w[0], w[1])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGoobjDwarfInfo checks the subprogram DIE for an exported and a
|
||||||
|
// static function: the abbrev, name, high_pc, frame base, decl file/line,
|
||||||
|
// the external flag and the addrx relocation position.
|
||||||
|
func TestGoobjDwarfInfo(t *testing.T) {
|
||||||
|
fn := FuncLayout{Size: 20, Line: 2}
|
||||||
|
die, relocs := goobjDwarfInfo(fn, "pkg.f", 3)
|
||||||
|
want := []byte{
|
||||||
|
0x03,
|
||||||
|
'p', 'k', 'g', '.', 'f', 0,
|
||||||
|
0, 0, 0, 0, // addrx slot at offset 7
|
||||||
|
0x14, // high_pc: 20
|
||||||
|
0x01, 0x9c, // frame_base
|
||||||
|
0x01, 0, 0, 0, // decl_file 1
|
||||||
|
0x02, // decl_line 2
|
||||||
|
0x01, // external
|
||||||
|
0x00, // end of children
|
||||||
|
}
|
||||||
|
if !bytes.Equal(die, want) {
|
||||||
|
t.Errorf("DIE = %x, want %x", die, want)
|
||||||
|
}
|
||||||
|
if len(relocs) != 1 || relocs[0].off != 7 || relocs[0].siz != 4 || relocs[0].typ != relocDWTXTADDRU4 || relocs[0].sym != 3 {
|
||||||
|
t.Errorf("relocs = %+v", relocs)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A static function carries no external flag and no package prefix.
|
||||||
|
fn.Static = true
|
||||||
|
die, _ = goobjDwarfInfo(fn, "f", 1)
|
||||||
|
if die[len(die)-2] != 0 {
|
||||||
|
t.Errorf("static external flag = %d, want 0", die[len(die)-2])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDwPutPCLCDeltaRemainders checks the standard-opcode remainders:
|
||||||
|
// const_add_pc and fixed_advance_pc after a special opcode.
|
||||||
|
func TestDwPutPCLCDeltaRemainders(t *testing.T) {
|
||||||
|
// deltaPC 25 past PC_RANGE: opcode 26 covers (1, 1), const_add_pc
|
||||||
|
// covers the remaining 23 pc and 0 line.
|
||||||
|
got := dwPutPCLCDelta(nil, 25, 1)
|
||||||
|
if !bytes.Equal(got, []byte{8, 26}) {
|
||||||
|
t.Errorf("25/1 = %x, want [8 1a]", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
// deltaPC 20000: opcode 246 covers 23, fixed_advance_pc covers the
|
||||||
|
// remaining 19977.
|
||||||
|
got = dwPutPCLCDelta(nil, 20000, 1)
|
||||||
|
if got[0] != 9 || binary.LittleEndian.Uint16(got[1:]) != 19977 || got[3] != 246 {
|
||||||
|
t.Errorf("20000/1 = %x, want fixed_advance_pc 19977 then 246", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Line remainder: deltaLC 10 leaves 5 past the opcode's reach, encoded
|
||||||
|
// as advance_line 5 (zigzag 0x0a) before opcode 250.
|
||||||
|
got = dwPutPCLCDelta(nil, 23, 10)
|
||||||
|
if !bytes.Equal(got, []byte{3, 0x0a, 250}) {
|
||||||
|
t.Errorf("23/10 = %x, want [03 0a fa]", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Negative line remainder: deltaLC -5 leaves advance_line -1 (zigzag
|
||||||
|
// 0x01) after opcode 11.
|
||||||
|
got = dwPutPCLCDelta(nil, 0, -5)
|
||||||
|
if !bytes.Equal(got, []byte{3, 1, 11}) {
|
||||||
|
t.Errorf("0/-5 = %x, want [03 01 0b]", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,348 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// readGOOBJSymbols reads the GOOBJ symbol definitions from a compiled Go
|
||||||
|
// package's export file. The file is an ar archive containing a __.PKGDEF
|
||||||
|
// member whose payload is the "go object ...\n!\n" preamble followed by the
|
||||||
|
// GOOBJ data. The function returns the symbol names in definition order
|
||||||
|
// (the order they appear in blkSymdef), which matches the SymIdx the linker
|
||||||
|
// expects for cross-package references.
|
||||||
|
func readGOOBJSymbols(exportPath string) ([]string, error) {
|
||||||
|
data, err := os.ReadFile(exportPath)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
goobj, err := extractGOOBJ(data)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("%s: %w", exportPath, err)
|
||||||
|
}
|
||||||
|
return goobj.symbols(), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// exportPath returns the export file path for a given import path by running
|
||||||
|
// "go list -export". The result is cached so repeated calls for the same
|
||||||
|
// package are fast.
|
||||||
|
func exportPath(importPath string) (string, error) {
|
||||||
|
cmd := exec.Command("go", "list", "-json", "-export", importPath)
|
||||||
|
out, err := cmd.Output()
|
||||||
|
if err != nil {
|
||||||
|
return "", fmt.Errorf("go list %s: %w", importPath, err)
|
||||||
|
}
|
||||||
|
// Quick JSON extraction: find "Export": "…"
|
||||||
|
const key = `"Export": "`
|
||||||
|
i := bytes.Index(out, []byte(key))
|
||||||
|
if i < 0 {
|
||||||
|
return "", fmt.Errorf("go list %s: no Export field", importPath)
|
||||||
|
}
|
||||||
|
start := i + len(key)
|
||||||
|
end := bytes.IndexByte(out[start:], '"')
|
||||||
|
if end < 0 {
|
||||||
|
return "", fmt.Errorf("go list %s: malformed Export field", importPath)
|
||||||
|
}
|
||||||
|
return string(out[start : start+end]), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// resolveExternalGOOBJ resolves a set of external symbol references into
|
||||||
|
// (package index, symbol index) pairs suitable for GOOBJ emission.
|
||||||
|
//
|
||||||
|
// refs maps package import paths to the symbol names referenced from that
|
||||||
|
// package. The returned pkgIdx maps each import path to its position in
|
||||||
|
// the blkPkgIdx table (0-based), and symIdx gives each symbol's index within
|
||||||
|
// its package.
|
||||||
|
func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symIdx map[string]int, err error) {
|
||||||
|
pkgIdx = make(map[string]int, len(refs))
|
||||||
|
symIdx = make(map[string]int)
|
||||||
|
|
||||||
|
// Assign package indices in sorted order for determinism.
|
||||||
|
packages := sortedPkgRefs(refs)
|
||||||
|
|
||||||
|
for i, pkg := range packages {
|
||||||
|
pkgIdx[pkg.path] = i
|
||||||
|
exp, err := exportPath(pkg.path)
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, err
|
||||||
|
}
|
||||||
|
data, err := os.ReadFile(exp)
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, err
|
||||||
|
}
|
||||||
|
gobj, err := extractGOOBJ(data)
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, fmt.Errorf("%s: %w", pkg.path, err)
|
||||||
|
}
|
||||||
|
for _, name := range pkg.syms {
|
||||||
|
idx := gobj.findSymbol(pkg.path, name)
|
||||||
|
if idx < 0 {
|
||||||
|
return nil, nil, fmt.Errorf("symbol %s·%s not found in export data of %s", pkg.path, name, pkg.path)
|
||||||
|
}
|
||||||
|
symIdx[pkg.path+"·"+name] = idx
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return pkgIdx, symIdx, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type pkgRef struct {
|
||||||
|
path string
|
||||||
|
syms []string
|
||||||
|
}
|
||||||
|
|
||||||
|
func sortedPkgRefs(refs map[string][]string) []pkgRef {
|
||||||
|
var pkgs []pkgRef
|
||||||
|
for pkg, syms := range refs {
|
||||||
|
pkgs = append(pkgs, pkgRef{pkg, syms})
|
||||||
|
}
|
||||||
|
// Simple insertion sort — the list is tiny (usually 1–3 packages).
|
||||||
|
for i := 1; i < len(pkgs); i++ {
|
||||||
|
for j := i; j > 0 && pkgs[j-1].path > pkgs[j].path; j-- {
|
||||||
|
pkgs[j-1], pkgs[j] = pkgs[j], pkgs[j-1]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return pkgs
|
||||||
|
}
|
||||||
|
|
||||||
|
// extractGOOBJ finds the GOOBJ data in an ar archive and returns a parsed
|
||||||
|
// goobjFile. The archive member _go_.o contains the "go object …\n!\n"
|
||||||
|
// preamble followed by the GOOBJ payload; __.PKGDEF is the compiler export
|
||||||
|
// data (type information) and is not the GOOBJ object.
|
||||||
|
func extractGOOBJ(data []byte) (*goobjFile, error) {
|
||||||
|
if len(data) < 8 || string(data[:8]) != "!<arch>\n" {
|
||||||
|
return nil, fmt.Errorf("not an ar archive")
|
||||||
|
}
|
||||||
|
pos := 8
|
||||||
|
for pos+60 <= len(data) {
|
||||||
|
hdr := data[pos : pos+60]
|
||||||
|
pos += 60
|
||||||
|
|
||||||
|
// Parse ar header fields.
|
||||||
|
name := strings.TrimRight(string(hdr[:16]), " /")
|
||||||
|
size := parseArDecimal(hdr[48:58])
|
||||||
|
if size < 0 {
|
||||||
|
return nil, fmt.Errorf("invalid ar header: bad size")
|
||||||
|
}
|
||||||
|
if pos+size > len(data) {
|
||||||
|
return nil, fmt.Errorf("ar entry %q extends past end of file", name)
|
||||||
|
}
|
||||||
|
body := data[pos : pos+size]
|
||||||
|
pos += size
|
||||||
|
// ar pads to even bytes.
|
||||||
|
if pos%2 != 0 {
|
||||||
|
pos++
|
||||||
|
}
|
||||||
|
|
||||||
|
if name == "_go_.o" {
|
||||||
|
return parseGOOBJ(body)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil, fmt.Errorf("archive contains no _go_.o member")
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseArDecimal parses a decimal number from a space-padded field.
|
||||||
|
func parseArDecimal(b []byte) int {
|
||||||
|
v := 0
|
||||||
|
for _, c := range b {
|
||||||
|
if c == ' ' {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if c < '0' || c > '9' {
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
v = v*10 + int(c-'0')
|
||||||
|
}
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
|
||||||
|
// goobjFile is a parsed GOOBJ file: the string table and the symbol-definition
|
||||||
|
// block.
|
||||||
|
type goobjFile struct {
|
||||||
|
strTab []byte // string table, at headerSize + n
|
||||||
|
symdef []byte // blkSymdef raw block
|
||||||
|
npdef []byte // blkNonpkgdef raw block
|
||||||
|
}
|
||||||
|
|
||||||
|
// symbols returns all symbol names in definition order by scanning the
|
||||||
|
// symdef and nonpkgdef blocks and resolving each name through the string
|
||||||
|
// table. Package definitions (blkSymdef) use fully-qualified names like
|
||||||
|
// "runtime.morestack"; non-package definitions (blkNonpkgdef) use bare
|
||||||
|
// names like "morestack". This combined list matches the index the
|
||||||
|
// linker expects for cross-package references.
|
||||||
|
func (f *goobjFile) symbols() []string {
|
||||||
|
return append(f.defNames(), f.npdefNames()...)
|
||||||
|
}
|
||||||
|
|
||||||
|
// findSymbol returns the index of a symbol within the combined symbol list,
|
||||||
|
// or -1 if not found. It first tries the fully-qualified name (pkg.name),
|
||||||
|
// then the bare name.
|
||||||
|
func (f *goobjFile) findSymbol(pkg, name string) int {
|
||||||
|
qualified := pkg + "." + name
|
||||||
|
syms := f.symbols()
|
||||||
|
for i, s := range syms {
|
||||||
|
if s == qualified {
|
||||||
|
return i
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Try bare name (for non-package definitions).
|
||||||
|
for i, s := range syms {
|
||||||
|
if s == name {
|
||||||
|
return i
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
|
||||||
|
// defNames returns names from blkSymdef only.
|
||||||
|
func (f *goobjFile) defNames() []string {
|
||||||
|
return f.readSymNames(f.symdef)
|
||||||
|
}
|
||||||
|
|
||||||
|
// npdefNames returns names from blkNonpkgdef.
|
||||||
|
func (f *goobjFile) npdefNames() []string {
|
||||||
|
return f.readSymNames(f.npdef)
|
||||||
|
}
|
||||||
|
|
||||||
|
// readSymNames reads symbol names from a symdef/nonpkgdef block. Each record
|
||||||
|
// is 21 bytes: nameLen (u32), nameOff (u32), abi (u16), typ, flag, flag2,
|
||||||
|
// size (u32), align (u32). nameOff is an absolute offset into the string
|
||||||
|
// table.
|
||||||
|
func (f *goobjFile) readSymNames(block []byte) []string {
|
||||||
|
const recSize = 21
|
||||||
|
if len(block) < recSize {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
n := len(block) / recSize
|
||||||
|
names := make([]string, 0, n)
|
||||||
|
for i := 0; i < n; i++ {
|
||||||
|
rec := block[i*recSize : (i+1)*recSize]
|
||||||
|
nameLen := binary.LittleEndian.Uint32(rec[0:4])
|
||||||
|
nameOff := binary.LittleEndian.Uint32(rec[4:8])
|
||||||
|
// nameOff is an absolute offset into the GOOBJ payload. The string
|
||||||
|
// table we have starts at goobjHeaderSize, so we subtract that.
|
||||||
|
if nameOff < goobjHeaderSize {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
relOff := nameOff - goobjHeaderSize
|
||||||
|
if relOff >= uint32(len(f.strTab)) || relOff+nameLen > uint32(len(f.strTab)) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
names = append(names, string(f.strTab[relOff:relOff+nameLen]))
|
||||||
|
}
|
||||||
|
return names
|
||||||
|
}
|
||||||
|
|
||||||
|
const goobjHeaderSize = 8 + 8 + 4 + 4*(blkEnd+1) // magic + fingerprint + flags + 19 block offsets
|
||||||
|
|
||||||
|
// parseGOOBJ parses a raw GOOBJ payload (the data after the "\n!\n" preamble).
|
||||||
|
func parseGOOBJ(data []byte) (*goobjFile, error) {
|
||||||
|
// Find the "\n!\n" separator.
|
||||||
|
sep := []byte("\n!\n")
|
||||||
|
i := bytes.Index(data, sep)
|
||||||
|
if i < 0 {
|
||||||
|
// Maybe the data has no preamble (e.g. a raw .o file).
|
||||||
|
i = -3 // treat as if preamble starts before the data
|
||||||
|
}
|
||||||
|
payload := data[i+len(sep):]
|
||||||
|
|
||||||
|
if len(payload) < goobjHeaderSize {
|
||||||
|
return nil, fmt.Errorf("GOOBJ payload too short (%d bytes)", len(payload))
|
||||||
|
}
|
||||||
|
if string(payload[:8]) != goobjMagic {
|
||||||
|
return nil, fmt.Errorf("bad GOOBJ magic: %q", payload[:8])
|
||||||
|
}
|
||||||
|
|
||||||
|
// Read block offsets. The header layout is:
|
||||||
|
// [0:8] magic
|
||||||
|
// [8:16] fingerprint
|
||||||
|
// [16:20] flags
|
||||||
|
// [20:96] 19 × uint32 offsets
|
||||||
|
var offs [blkEnd + 1]uint32
|
||||||
|
for i := 0; i <= blkEnd; i++ {
|
||||||
|
offs[i] = binary.LittleEndian.Uint32(payload[20+4*i:])
|
||||||
|
}
|
||||||
|
// The string table lives at headerSize.
|
||||||
|
strTabStart := uint32(goobjHeaderSize)
|
||||||
|
|
||||||
|
f := &goobjFile{
|
||||||
|
strTab: payload[strTabStart:offs[0]],
|
||||||
|
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
|
||||||
|
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
|
||||||
|
}
|
||||||
|
return f, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// blockSlice extracts a block from the payload using its offset pair.
|
||||||
|
func blockSlice(payload []byte, offs [blkEnd + 1]uint32, start, end int) []byte {
|
||||||
|
if start < 0 || end > blkEnd || offs[end] < offs[start] {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
beg := offs[start]
|
||||||
|
fin := offs[end]
|
||||||
|
if int(fin) > len(payload) || int(beg) > int(fin) {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return payload[beg:fin]
|
||||||
|
}
|
||||||
|
|
||||||
|
// resolveExternalSymbols is the high-level entry point for GOOBJ emission.
|
||||||
|
// Given a list of external symbol names (e.g. ["runtime·morestack",
|
||||||
|
// "runtime·g0"]), it returns the package-index table entries and a map from
|
||||||
|
// full symbol name to GOOBJ {pkgIdx, symIdx}.
|
||||||
|
//
|
||||||
|
// The package table entries should be written into blkPkgIdx, and the
|
||||||
|
// returned indices should replace pkgIdxSelf / placeholder values in the
|
||||||
|
// relocation records.
|
||||||
|
func resolveExternalSymbols(externals []string) (pkgTable []string, pkgIdxMap map[string]int, symIdxMap map[string]int, err error) {
|
||||||
|
// Group references by package.
|
||||||
|
refs := make(map[string]map[string]bool)
|
||||||
|
for _, full := range externals {
|
||||||
|
pkg, name := splitQualified(full)
|
||||||
|
if refs[pkg] == nil {
|
||||||
|
refs[pkg] = make(map[string]bool)
|
||||||
|
}
|
||||||
|
refs[pkg][name] = true
|
||||||
|
}
|
||||||
|
|
||||||
|
// Convert maps to slices.
|
||||||
|
r := make(map[string][]string, len(refs))
|
||||||
|
for pkg, names := range refs {
|
||||||
|
for name := range names {
|
||||||
|
r[pkg] = append(r[pkg], name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pkgIdx1, symIdx1, err := resolveExternalGOOBJ(r)
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
// Build the package table in pkgIdx order.
|
||||||
|
pkgTable = make([]string, len(pkgIdx1))
|
||||||
|
for pkg, idx := range pkgIdx1 {
|
||||||
|
pkgTable[idx] = pkg
|
||||||
|
}
|
||||||
|
|
||||||
|
return pkgTable, pkgIdx1, symIdx1, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// splitQualified splits a qualified Go symbol name (pkgpath·name) into its
|
||||||
|
// package path and local name. The separator is the middle dot (U+00B7).
|
||||||
|
// If no separator is found, the symbol is assumed to be in the current
|
||||||
|
// package (empty pkg).
|
||||||
|
func splitQualified(full string) (pkg, name string) {
|
||||||
|
if idx := strings.IndexByte(full, '\u00b7'); idx >= 0 {
|
||||||
|
return full[:idx], full[idx+len("\u00b7"):]
|
||||||
|
}
|
||||||
|
if idx := strings.IndexByte(full, '.'); idx >= 0 {
|
||||||
|
return full[:idx], full[idx+1:]
|
||||||
|
}
|
||||||
|
return "", full
|
||||||
|
}
|
||||||
@@ -0,0 +1,66 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestReadRuntimeSymbols verifies the GOOBJ reader can extract and find
|
||||||
|
// symbols from the runtime package's compiled archive.
|
||||||
|
func TestReadRuntimeSymbols(t *testing.T) {
|
||||||
|
exp, err := exportPath("runtime")
|
||||||
|
if err != nil {
|
||||||
|
t.Skipf("cannot find runtime export: %v (need Go toolchain)", err)
|
||||||
|
}
|
||||||
|
data, err := os.ReadFile(exp)
|
||||||
|
if err != nil {
|
||||||
|
t.Skipf("cannot read runtime export: %v", err)
|
||||||
|
}
|
||||||
|
gobj, err := extractGOOBJ(data)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("extractGOOBJ: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
t.Logf("runtime: %d symbols", len(gobj.symbols()))
|
||||||
|
|
||||||
|
// Verify we can find well-known runtime symbols.
|
||||||
|
for _, tc := range []struct{ pkg, name string }{
|
||||||
|
{"runtime", "g0"},
|
||||||
|
{"runtime", "morestack"},
|
||||||
|
{"runtime", "newstack"},
|
||||||
|
} {
|
||||||
|
idx := gobj.findSymbol(tc.pkg, tc.name)
|
||||||
|
if idx < 0 {
|
||||||
|
t.Errorf("findSymbol(%q, %q) = -1", tc.pkg, tc.name)
|
||||||
|
} else {
|
||||||
|
t.Logf("findSymbol(%q, %q) = %d", tc.pkg, tc.name, idx)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestResolveExternalSymbols verifies end-to-end resolution of external
|
||||||
|
// symbol references.
|
||||||
|
func TestResolveExternalSymbols(t *testing.T) {
|
||||||
|
if _, err := exec.LookPath("go"); err != nil {
|
||||||
|
t.Skip("go toolchain not available")
|
||||||
|
}
|
||||||
|
|
||||||
|
refs := map[string][]string{
|
||||||
|
"runtime": {"g0"},
|
||||||
|
}
|
||||||
|
pkgIdx, symIdx, err := resolveExternalGOOBJ(refs)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("resolveExternalGOOBJ: %v", err)
|
||||||
|
}
|
||||||
|
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 0 {
|
||||||
|
t.Errorf("pkgIdx = %v, want runtime→0", pkgIdx)
|
||||||
|
}
|
||||||
|
if _, ok := symIdx["runtime·g0"]; !ok {
|
||||||
|
t.Errorf("symIdx missing runtime·g0, got %v", symIdx)
|
||||||
|
}
|
||||||
|
t.Logf("runtime·g0 → SymIdx=%d", symIdx["runtime·g0"])
|
||||||
|
}
|
||||||
+74
-38
@@ -111,17 +111,26 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
|||||||
t.Errorf("flags = %#x, want ObjFlagFromAssembly (4)", flags)
|
t.Errorf("flags = %#x, want ObjFlagFromAssembly (4)", flags)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Package defs: the static GLOBL, then one anonymous FuncInfo per
|
// Package defs: the static GLOBL, then per function the FuncInfo and the
|
||||||
// function.
|
// two DWARF symbols (debug_line program, subprogram DIE).
|
||||||
defs := v.syms(blkSymdef)
|
defs := v.syms(blkSymdef)
|
||||||
if len(defs) != 3 {
|
if len(defs) != 7 {
|
||||||
t.Fatalf("symdefs = %d, want 3", len(defs))
|
t.Fatalf("symdefs = %d, want 7", len(defs))
|
||||||
}
|
}
|
||||||
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
|
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
|
||||||
t.Errorf("mask symbol = %+v", defs[0])
|
t.Errorf("mask symbol = %+v", defs[0])
|
||||||
}
|
}
|
||||||
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
|
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
|
||||||
t.Errorf("funcinfo symbol = %+v", defs[1])
|
t.Errorf("addq funcinfo symbol = %+v", defs[1])
|
||||||
|
}
|
||||||
|
if defs[2].name != "" || defs[2].typ != kindSDWARFLINES || defs[2].size == 0 {
|
||||||
|
t.Errorf("addq lines symbol = %+v", defs[2])
|
||||||
|
}
|
||||||
|
if defs[3].name != "" || defs[3].typ != kindSDWARFFCN || defs[3].size == 0 {
|
||||||
|
t.Errorf("addq DIE symbol = %+v", defs[3])
|
||||||
|
}
|
||||||
|
if defs[4].name != "" || defs[4].typ != kindSDATA || defs[4].size != 28 {
|
||||||
|
t.Errorf("loadmask funcinfo symbol = %+v", defs[4])
|
||||||
}
|
}
|
||||||
|
|
||||||
// Non-package defs: four pc tables and the function, per function.
|
// Non-package defs: four pc tables and the function, per function.
|
||||||
@@ -142,45 +151,62 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
|||||||
// FuncInfo: args 24, FuncFlag Asm, one file, no inline tree.
|
// FuncInfo: args 24, FuncFlag Asm, one file, no inline tree.
|
||||||
le := binary.LittleEndian
|
le := binary.LittleEndian
|
||||||
data := v.blk(blkData)
|
data := v.blk(blkData)
|
||||||
|
didx := v.blk(blkDataIdx)
|
||||||
fi := data[16:44]
|
fi := data[16:44]
|
||||||
if le.Uint32(fi[0:]) != 24 || le.Uint32(fi[4:]) != 0 || fi[8] != 0 || fi[9] != funcFlagAsm ||
|
if le.Uint32(fi[0:]) != 24 || le.Uint32(fi[4:]) != 0 || fi[8] != 0 || fi[9] != funcFlagAsm ||
|
||||||
le.Uint32(fi[16:]) != 1 || le.Uint32(fi[20:]) != 0 || le.Uint32(fi[24:]) != 0 {
|
le.Uint32(fi[16:]) != 1 || le.Uint32(fi[20:]) != 0 || le.Uint32(fi[24:]) != 0 {
|
||||||
t.Errorf("funcinfo bytes %x", fi)
|
t.Errorf("funcinfo bytes %x", fi)
|
||||||
}
|
}
|
||||||
|
|
||||||
// pcsp: a flat zero over the whole function (zero-frame NOSPLIT).
|
// The pc-value tables of addq (non-package indices 0–3, so global
|
||||||
if got := data[72:75]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
|
// indices 7–10): pcsp a flat zero over the whole function, pcinline a
|
||||||
|
// flat -1, both with the pc delta in MinLC (1) units.
|
||||||
|
pcsp := data[le.Uint32(didx[4*7:]):]
|
||||||
|
if got := pcsp[:3]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
|
||||||
t.Errorf("pcsp = %x, want 021300", got)
|
t.Errorf("pcsp = %x, want 021300", got)
|
||||||
}
|
}
|
||||||
// pcinline: a flat -1.
|
pcinl := data[le.Uint32(didx[4*10:]):]
|
||||||
if got := data[81:84]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) {
|
if got := pcinl[:3]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) {
|
||||||
t.Errorf("pcinline = %x, want 001300", got)
|
t.Errorf("pcinline = %x, want 001300", got)
|
||||||
}
|
}
|
||||||
|
|
||||||
// The one relocation: R_PCREL, four bytes wide, against the GLOBL,
|
// Relocations: the four DWARF address references (two per function, in
|
||||||
// with the field in the function code left zero. The loadmask code's
|
// definition order), then the loadmask code's R_PCREL against the
|
||||||
// offset comes from the data index (symbol 3 defs + 9 non-package).
|
// GLOBL, with the field in the function code left zero. The loadmask
|
||||||
|
// code's offset comes from the data index (7 defs + 9 non-package).
|
||||||
relocs := v.blk(blkReloc)
|
relocs := v.blk(blkReloc)
|
||||||
if len(relocs) != 23 {
|
if len(relocs) != 5*23 {
|
||||||
t.Fatalf("relocs = %d bytes, want one 23-byte entry", len(relocs))
|
t.Fatalf("relocs = %d bytes, want 5 entries", len(relocs))
|
||||||
}
|
}
|
||||||
off := int32(le.Uint32(relocs[0:]))
|
// addq's DWARF references (defs 2 and 3) against the function, which
|
||||||
if off != 4 || relocs[4] != 4 || le.Uint16(relocs[5:]) != relocPCRel ||
|
// is non-package index 4.
|
||||||
le.Uint64(relocs[7:]) != 0 || le.Uint32(relocs[15:]) != pkgIdxSelf || le.Uint32(relocs[19:]) != 0 {
|
lr := relocs[:23]
|
||||||
t.Errorf("reloc = %x", relocs)
|
if int32(le.Uint32(lr[0:])) != 3 || lr[4] != 8 || le.Uint16(lr[5:]) != relocAddr ||
|
||||||
|
le.Uint32(lr[15:]) != pkgIdxNone || le.Uint32(lr[19:]) != 4 {
|
||||||
|
t.Errorf("addq lines reloc = %x", lr)
|
||||||
}
|
}
|
||||||
didx := v.blk(blkDataIdx)
|
dr := relocs[23:46]
|
||||||
lm := le.Uint32(didx[4*(3+9):])
|
if dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4 ||
|
||||||
|
le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 {
|
||||||
|
t.Errorf("addq DIE reloc = %x", dr)
|
||||||
|
}
|
||||||
|
cr := relocs[4*23:]
|
||||||
|
off := int32(le.Uint32(cr[0:]))
|
||||||
|
if off != 4 || cr[4] != 4 || le.Uint16(cr[5:]) != relocPCRel ||
|
||||||
|
le.Uint64(cr[7:]) != 0 || le.Uint32(cr[15:]) != pkgIdxSelf || le.Uint32(cr[19:]) != 0 {
|
||||||
|
t.Errorf("loadmask reloc = %x", cr)
|
||||||
|
}
|
||||||
|
lm := le.Uint32(didx[4*16:])
|
||||||
code := data[lm : lm+18]
|
code := data[lm : lm+18]
|
||||||
if !bytes.Equal(code[4:8], []byte{0, 0, 0, 0}) {
|
if !bytes.Equal(code[4:8], []byte{0, 0, 0, 0}) {
|
||||||
t.Errorf("relocated field = %x, want zeroed", code[4:8])
|
t.Errorf("relocated field = %x, want zeroed", code[4:8])
|
||||||
}
|
}
|
||||||
|
|
||||||
// Aux wiring: FuncInfo (package symbol), then the four pc tables
|
// Aux wiring: FuncInfo, the two DWARF symbols (package symbols), then
|
||||||
// (non-package symbols).
|
// the four pc tables (non-package symbols).
|
||||||
auxs := v.blk(blkAux)
|
auxs := v.blk(blkAux)
|
||||||
if len(auxs) != 2*5*9 {
|
if len(auxs) != 2*7*9 {
|
||||||
t.Fatalf("aux = %d bytes, want 10 entries", len(auxs))
|
t.Fatalf("aux = %d bytes, want 14 entries", len(auxs))
|
||||||
}
|
}
|
||||||
wantAux := []struct {
|
wantAux := []struct {
|
||||||
typ uint8
|
typ uint8
|
||||||
@@ -188,15 +214,19 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
|||||||
idx uint32
|
idx uint32
|
||||||
}{
|
}{
|
||||||
{auxFuncInfo, pkgIdxSelf, 1},
|
{auxFuncInfo, pkgIdxSelf, 1},
|
||||||
{auxPcsp, pkgIdxNone, uint32(len(defs) + 0)},
|
{auxDwarfInfo, pkgIdxSelf, 3},
|
||||||
{auxPcfile, pkgIdxNone, uint32(len(defs) + 1)},
|
{auxDwarfLines, pkgIdxSelf, 2},
|
||||||
{auxPcline, pkgIdxNone, uint32(len(defs) + 2)},
|
{auxPcsp, pkgIdxNone, 0},
|
||||||
{auxPcinline, pkgIdxNone, uint32(len(defs) + 3)},
|
{auxPcfile, pkgIdxNone, 1},
|
||||||
{auxFuncInfo, pkgIdxSelf, 2},
|
{auxPcline, pkgIdxNone, 2},
|
||||||
{auxPcsp, pkgIdxNone, uint32(len(defs) + 5)},
|
{auxPcinline, pkgIdxNone, 3},
|
||||||
{auxPcfile, pkgIdxNone, uint32(len(defs) + 6)},
|
{auxFuncInfo, pkgIdxSelf, 4},
|
||||||
{auxPcline, pkgIdxNone, uint32(len(defs) + 7)},
|
{auxDwarfInfo, pkgIdxSelf, 6},
|
||||||
{auxPcinline, pkgIdxNone, uint32(len(defs) + 8)},
|
{auxDwarfLines, pkgIdxSelf, 5},
|
||||||
|
{auxPcsp, pkgIdxNone, 5},
|
||||||
|
{auxPcfile, pkgIdxNone, 6},
|
||||||
|
{auxPcline, pkgIdxNone, 7},
|
||||||
|
{auxPcinline, pkgIdxNone, 8},
|
||||||
}
|
}
|
||||||
for i, w := range wantAux {
|
for i, w := range wantAux {
|
||||||
e := auxs[i*9:]
|
e := auxs[i*9:]
|
||||||
@@ -253,7 +283,7 @@ TEXT ·framed(SB), NOSPLIT, $8-0
|
|||||||
t.Fatalf("AssembleFile: %v", err)
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
}
|
}
|
||||||
fn := img.Funcs[0]
|
fn := img.Funcs[0]
|
||||||
pcs, vals := decodePCValues(pcspTable(fn))
|
pcs, vals := decodePCValues(pcspTable(fn, 1))
|
||||||
// Prologue: PUSHQ BP (1 byte, +8), MOVQ SP, BP (3 bytes, no change),
|
// Prologue: PUSHQ BP (1 byte, +8), MOVQ SP, BP (3 bytes, no change),
|
||||||
// SUBQ $8, SP (4 bytes, +16 in total); the RET's epilogue unwinds
|
// SUBQ $8, SP (4 bytes, +16 in total); the RET's epilogue unwinds
|
||||||
// ADDQ $8, SP (+8) then POPQ BP (0).
|
// ADDQ $8, SP (+8) then POPQ BP (0).
|
||||||
@@ -401,13 +431,12 @@ func main() {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("GOObject: %v", err)
|
t.Fatalf("GOObject: %v", err)
|
||||||
}
|
}
|
||||||
if err := os.WriteFile(asmObj, obj, 0o644); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Rebuild the package archive with our object in place of the
|
// Rebuild the package archive with our object in place of the
|
||||||
// toolchain's (go tool pack has no replace-in-place that dedupes, so
|
// toolchain's (go tool pack has no replace-in-place that dedupes, so
|
||||||
// extract, substitute and repack).
|
// extract, substitute and repack). The archive member holding the
|
||||||
|
// assembler's output is named after the asm object file, e.g.
|
||||||
|
// main_amd64.o.
|
||||||
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
|
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
|
||||||
membersDir := filepath.Join(dir, "members")
|
membersDir := filepath.Join(dir, "members")
|
||||||
if err := os.MkdirAll(membersDir, 0o755); err != nil {
|
if err := os.MkdirAll(membersDir, 0o755); err != nil {
|
||||||
@@ -417,6 +446,13 @@ func main() {
|
|||||||
if out, err := extract.CombinedOutput(); err != nil {
|
if out, err := extract.CombinedOutput(); err != nil {
|
||||||
t.Fatalf("pack x: %v\n%s", err, out)
|
t.Fatalf("pack x: %v\n%s", err, out)
|
||||||
}
|
}
|
||||||
|
member := filepath.Join(membersDir, filepath.Base(asmObj))
|
||||||
|
if err := os.Chmod(member, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(member, obj, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
|
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
|
||||||
listOut, err := listCmd.CombinedOutput()
|
listOut, err := listCmd.CombinedOutput()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
@@ -0,0 +1,91 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"sync"
|
||||||
|
)
|
||||||
|
|
||||||
|
// GOObjectLOONG64 emits a GOOBJ object file for LoongArch. The layout is
|
||||||
|
// the shared one in goobj.go — the toolchain preamble, the go120ld header
|
||||||
|
// with its block offsets, the string table, the symbol definitions and the
|
||||||
|
// reloc/aux/data index arrays — with the loong64 preamble, the MinLC of 4
|
||||||
|
// for the pc-value deltas, and R_LOONG64_ADDR_HI/LO relocation types for
|
||||||
|
// the pcalau12i+addi.d address pairs.
|
||||||
|
func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
|
||||||
|
pre, err := toolchainObjectPreambleLOONG64()
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
|
||||||
|
// A pcalau12i+addi.d pair: the high part carries
|
||||||
|
// R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO.
|
||||||
|
if r.Kind == RelLoong64AddrLo {
|
||||||
|
return relocLoong64AddrLo, 4
|
||||||
|
}
|
||||||
|
return relocLoong64AddrHi, 4
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Loong64 relocation types (cmd/internal/objabi). R_LOONG64_ADDR_HI
|
||||||
|
// resolves the high 20 bits of a PC-relative address into pcalau12i;
|
||||||
|
// R_LOONG64_ADDR_LO the low 12 bits into addi.d/ld/st.
|
||||||
|
const (
|
||||||
|
relocLoong64AddrHi = 77 // R_LOONG64_ADDR_HI
|
||||||
|
relocLoong64AddrLo = 78 // R_LOONG64_ADDR_LO
|
||||||
|
)
|
||||||
|
|
||||||
|
// toolchainObjectPreambleLOONG64 returns the "go object ...\n!\n" header
|
||||||
|
// the installed go tool asm writes for loong64, captured by assembling a
|
||||||
|
// one-instruction probe (see toolchainObjectPreamble).
|
||||||
|
var (
|
||||||
|
preambleLOONG64Once sync.Once
|
||||||
|
preambleLOONG64 []byte
|
||||||
|
preambleLOONG64Err error
|
||||||
|
)
|
||||||
|
|
||||||
|
func toolchainObjectPreambleLOONG64() ([]byte, error) {
|
||||||
|
preambleLOONG64Once.Do(func() {
|
||||||
|
goBin, err := exec.LookPath("go")
|
||||||
|
if err != nil {
|
||||||
|
preambleLOONG64Err = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
dir, err := os.MkdirTemp("", "gasm-preamble-loong64")
|
||||||
|
if err != nil {
|
||||||
|
preambleLOONG64Err = err
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer os.RemoveAll(dir)
|
||||||
|
src := filepath.Join(dir, "probe_loong64.s")
|
||||||
|
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
|
||||||
|
preambleLOONG64Err = err
|
||||||
|
return
|
||||||
|
}
|
||||||
|
obj := filepath.Join(dir, "probe.o")
|
||||||
|
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
|
||||||
|
cmd.Env = append(os.Environ(), "GOARCH=loong64")
|
||||||
|
if out, err := cmd.CombinedOutput(); err != nil {
|
||||||
|
preambleLOONG64Err = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
data, err := os.ReadFile(obj)
|
||||||
|
if err != nil {
|
||||||
|
preambleLOONG64Err = err
|
||||||
|
return
|
||||||
|
}
|
||||||
|
i := bytes.Index(data, []byte("\n!\n"))
|
||||||
|
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
|
||||||
|
preambleLOONG64Err = fmt.Errorf("unrecognised assembler object layout")
|
||||||
|
return
|
||||||
|
}
|
||||||
|
preambleLOONG64 = data[:i+3]
|
||||||
|
})
|
||||||
|
return preambleLOONG64, preambleLOONG64Err
|
||||||
|
}
|
||||||
@@ -0,0 +1,95 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"sync"
|
||||||
|
)
|
||||||
|
|
||||||
|
// GOObjectRISCV emits a GOOBJ object file for RISC-V. The layout is the
|
||||||
|
// shared one in goobj.go — the toolchain preamble, the go120ld header with
|
||||||
|
// its block offsets, the string table, the symbol definitions and the
|
||||||
|
// reloc/aux/data index arrays — with the RISC-V preamble, the MinLC of 2 for
|
||||||
|
// the pc-value deltas, and the single R_RISCV_PCREL_ITYPE/STYPE relocation
|
||||||
|
// per AUIPC pair, matching `go tool asm`'s model (each pair is one 8-byte
|
||||||
|
// relocation, not the ELF HI20/LO12 pair).
|
||||||
|
func (img *Image) GOObjectRISCV(pkgPath, srcPath string) ([]byte, error) {
|
||||||
|
pre, err := toolchainObjectPreambleRISCV()
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return img.emitGOObject(pkgPath, srcPath, pre, 2, func(r Reloc) (uint16, uint8) {
|
||||||
|
switch r.Kind {
|
||||||
|
case RelRISCVPCRELSType:
|
||||||
|
return relocRISCVPcrelStype, 8
|
||||||
|
case RelRISCVJal:
|
||||||
|
return relocRISCVJal, 4
|
||||||
|
default:
|
||||||
|
return relocRISCVPcrelItype, 8
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// RISC-V relocation types (cmd/internal/objabi). The Go linker applies
|
||||||
|
// R_RISCV_PCREL_ITYPE/STYPE to an AUIPC + I/S-type instruction pair as a
|
||||||
|
// single 8-byte field; R_RISCV_JAL covers a single 4-byte J-type instruction.
|
||||||
|
const (
|
||||||
|
relocRISCVJal = 59 // R_RISCV_JAL
|
||||||
|
relocRISCVPcrelItype = 62 // R_RISCV_PCREL_ITYPE
|
||||||
|
relocRISCVPcrelStype = 63 // R_RISCV_PCREL_STYPE
|
||||||
|
)
|
||||||
|
|
||||||
|
// toolchainObjectPreambleRISCV returns the "go object ...\n!\n" header
|
||||||
|
// the installed go tool asm writes for riscv64, captured by assembling a
|
||||||
|
// one-instruction probe (see toolchainObjectPreamble).
|
||||||
|
var (
|
||||||
|
preambleRISCVOnce sync.Once
|
||||||
|
preambleRISCV []byte
|
||||||
|
preambleRISCVErr error
|
||||||
|
)
|
||||||
|
|
||||||
|
func toolchainObjectPreambleRISCV() ([]byte, error) {
|
||||||
|
preambleRISCVOnce.Do(func() {
|
||||||
|
goBin, err := exec.LookPath("go")
|
||||||
|
if err != nil {
|
||||||
|
preambleRISCVErr = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
dir, err := os.MkdirTemp("", "gasm-preamble-riscv")
|
||||||
|
if err != nil {
|
||||||
|
preambleRISCVErr = err
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer os.RemoveAll(dir)
|
||||||
|
src := filepath.Join(dir, "probe_riscv64.s")
|
||||||
|
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
|
||||||
|
preambleRISCVErr = err
|
||||||
|
return
|
||||||
|
}
|
||||||
|
obj := filepath.Join(dir, "probe.o")
|
||||||
|
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
|
||||||
|
cmd.Env = append(os.Environ(), "GOARCH=riscv64")
|
||||||
|
if out, err := cmd.CombinedOutput(); err != nil {
|
||||||
|
preambleRISCVErr = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
data, err := os.ReadFile(obj)
|
||||||
|
if err != nil {
|
||||||
|
preambleRISCVErr = err
|
||||||
|
return
|
||||||
|
}
|
||||||
|
i := bytes.Index(data, []byte("\n!\n"))
|
||||||
|
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
|
||||||
|
preambleRISCVErr = fmt.Errorf("unrecognised assembler object layout")
|
||||||
|
return
|
||||||
|
}
|
||||||
|
preambleRISCV = data[:i+3]
|
||||||
|
})
|
||||||
|
return preambleRISCV, preambleRISCVErr
|
||||||
|
}
|
||||||
@@ -0,0 +1,159 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build integration
|
||||||
|
|
||||||
|
// Package asm integration tests against the production go-libraries kernels.
|
||||||
|
// These are excluded from the default test run (go test ./...) so that the
|
||||||
|
// coverage numbers are identical locally and in CI, where go-libraries is
|
||||||
|
// not checked out. Run them explicitly with: go test -tags=integration ./asm/
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"os"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"golang.org/x/arch/x86/x86asm"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel —
|
||||||
|
// all functions plus the file-local mask24 constant — and checks that every
|
||||||
|
// static-symbol load resolves to the right bytes in the image.
|
||||||
|
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
||||||
|
path := "../../go-libraries/go-flac/avx2_amd64.s"
|
||||||
|
if _, err := os.Stat(path); err != nil {
|
||||||
|
t.Skip("go-libraries repository not present next to gasm-devkit")
|
||||||
|
}
|
||||||
|
src, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
f, errs := parser.Parse(path, string(src))
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
if len(img.Funcs) != 17 {
|
||||||
|
t.Errorf("functions = %d, want 17", len(img.Funcs))
|
||||||
|
}
|
||||||
|
|
||||||
|
// mask24 as the DATA directives define it.
|
||||||
|
mask := []byte{
|
||||||
|
0x00, 0x01, 0x02, 0x80, 0x03, 0x04, 0x05, 0x80,
|
||||||
|
0x06, 0x07, 0x08, 0x80, 0x09, 0x0a, 0x0b, 0x80,
|
||||||
|
}
|
||||||
|
image := img.Bytes()
|
||||||
|
if got := image[img.Symbols["mask24"] : img.Symbols["mask24"]+16]; !bytes.Equal(got, mask) {
|
||||||
|
t.Errorf("mask24 contents %x, want %x", got, mask)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Every VMOVDQU mask24<>(SB), X15 (c5 7a 6f 3d + rel32, i.e. a VMOVDQU
|
||||||
|
// with a RIP-relative r/m) must land on the mask bytes within the image.
|
||||||
|
loads := 0
|
||||||
|
for _, fn := range img.Funcs {
|
||||||
|
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||||
|
for pc := 0; pc < len(code); {
|
||||||
|
inst, err := x86asm.Decode(code[pc:], 64)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("%s: decode at +%d: %v", fn.Name, pc, err)
|
||||||
|
}
|
||||||
|
// mod=00, rm=101 → RIP-relative.
|
||||||
|
if inst.Op == x86asm.VMOVDQU && inst.Len == 8 && code[pc+3]&0xC7 == 0x05 {
|
||||||
|
rel := int32(uint32(code[pc+4]) | uint32(code[pc+5])<<8 | uint32(code[pc+6])<<16 | uint32(code[pc+7])<<24)
|
||||||
|
target := fn.Offset + pc + 8 + int(rel)
|
||||||
|
if !bytes.Equal(image[target:target+16], mask) {
|
||||||
|
t.Errorf("%s: mask load at +%d lands on %x, want %x", fn.Name, pc, image[target:target+16], mask)
|
||||||
|
}
|
||||||
|
loads++
|
||||||
|
}
|
||||||
|
pc += inst.Len
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if loads != 2 {
|
||||||
|
t.Errorf("mask loads found = %d, want 2", loads)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
|
||||||
|
// kernel — all functions plus the file-global idx16 constant — and checks
|
||||||
|
// that the static-symbol load resolves to the right bytes in the image.
|
||||||
|
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
|
||||||
|
path := "../../go-libraries/go-flac/avx512_amd64.s"
|
||||||
|
if _, err := os.Stat(path); err != nil {
|
||||||
|
t.Skip("go-libraries repository not present next to gasm-devkit")
|
||||||
|
}
|
||||||
|
src, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
f, errs := parser.Parse(path, string(src))
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
if len(img.Funcs) != 10 {
|
||||||
|
t.Errorf("functions = %d, want 10", len(img.Funcs))
|
||||||
|
}
|
||||||
|
|
||||||
|
// idx16 as the DATA directives define it: dwords 1..16.
|
||||||
|
idx := make([]byte, 0, 64)
|
||||||
|
for i := 1; i <= 16; i++ {
|
||||||
|
idx = append(idx, byte(i), 0, 0, 0)
|
||||||
|
}
|
||||||
|
image := img.Bytes()
|
||||||
|
base := img.Symbols["idx16"]
|
||||||
|
if base == 0 {
|
||||||
|
t.Fatal("idx16 not laid out")
|
||||||
|
}
|
||||||
|
if got := image[base : base+64]; hexCompact(got) != hexCompact(idx) {
|
||||||
|
t.Errorf("idx16 contents %x, want %x", got, idx)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The VMOVDQU32 idx16(SB), Z13 load (62 71 7e 48 6f 2d + rel32) must
|
||||||
|
// resolve to idx16 within the image.
|
||||||
|
loads := 0
|
||||||
|
for _, fn := range img.Funcs {
|
||||||
|
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||||
|
pat := []byte{0x62, 0x71, 0x7e, 0x48, 0x6f, 0x2d}
|
||||||
|
for pos := 0; ; {
|
||||||
|
i := indexOf(code[pos:], pat)
|
||||||
|
if i < 0 {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
i += pos
|
||||||
|
rel := int32(uint32(code[i+6]) | uint32(code[i+7])<<8 | uint32(code[i+8])<<16 | uint32(code[i+9])<<24)
|
||||||
|
target := fn.Offset + i + 10 + int(rel)
|
||||||
|
if target != base {
|
||||||
|
t.Errorf("%s: idx16 load at +%d targets 0x%x, want 0x%x", fn.Name, i, target, base)
|
||||||
|
}
|
||||||
|
loads++
|
||||||
|
pos = i + 10
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if loads != 1 {
|
||||||
|
t.Errorf("idx16 loads found = %d, want 1", loads)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// indexOf returns the index of the first occurrence of pat in b, or -1.
|
||||||
|
func indexOf(b, pat []byte) int {
|
||||||
|
for i := 0; i+len(pat) <= len(b); i++ {
|
||||||
|
j := 0
|
||||||
|
for j < len(pat) && b[i+j] == pat[j] {
|
||||||
|
j++
|
||||||
|
}
|
||||||
|
if j == len(pat) {
|
||||||
|
return i
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return -1
|
||||||
|
}
|
||||||
@@ -0,0 +1,326 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/binary"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
|
||||||
|
// the symbol tables, the function code bytes and the relocation wiring.
|
||||||
|
func TestGOObjectLOONG64Structure(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("k_loong64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOVV a+0(FP), R4
|
||||||
|
MOVV b+8(FP), R5
|
||||||
|
ADDV R5, R4, R4
|
||||||
|
MOVV R4, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
GLOBL ·table<>(SB), RODATA, $8
|
||||||
|
DATA ·table<>+0(SB)/8, $0x1122334455667788
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileLOONG64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.GOObjectLOONG64("testpkg", "k_loong64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObjectLOONG64: %v", err)
|
||||||
|
}
|
||||||
|
v := openGoobj(t, obj)
|
||||||
|
|
||||||
|
// Package defs: the static GLOBL, then the FuncInfo and the two DWARF
|
||||||
|
// symbols (debug_line program, subprogram DIE).
|
||||||
|
defs := v.syms(blkSymdef)
|
||||||
|
if len(defs) != 4 {
|
||||||
|
t.Fatalf("symdefs = %d, want 4", len(defs))
|
||||||
|
}
|
||||||
|
if defs[0].name != "table" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 8 {
|
||||||
|
t.Errorf("table symbol = %+v", defs[0])
|
||||||
|
}
|
||||||
|
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
|
||||||
|
t.Errorf("funcinfo symbol = %+v", defs[1])
|
||||||
|
}
|
||||||
|
if defs[2].name != "" || defs[2].typ != kindSDWARFLINES || defs[2].size == 0 {
|
||||||
|
t.Errorf("lines symbol = %+v", defs[2])
|
||||||
|
}
|
||||||
|
if defs[3].name != "" || defs[3].typ != kindSDWARFFCN || defs[3].size == 0 {
|
||||||
|
t.Errorf("DIE symbol = %+v", defs[3])
|
||||||
|
}
|
||||||
|
|
||||||
|
// Non-package defs: four pc tables and the function.
|
||||||
|
nps := v.syms(blkNonpkgdef)
|
||||||
|
if len(nps) != 5 {
|
||||||
|
t.Fatalf("nonpkgdefs = %d, want 5", len(nps))
|
||||||
|
}
|
||||||
|
fn := nps[4]
|
||||||
|
if fn.name != "testpkg.add" || fn.typ != kindSTEXT || fn.flag != symFlagNoSplit || fn.size != 20 {
|
||||||
|
t.Errorf("add symbol = %+v", fn)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The function code: 20 bytes, the ground-truth encoding. It sits
|
||||||
|
// after the GLOBL, FuncInfo, two DWARF symbols and four pc tables.
|
||||||
|
dataIdx := v.blk(blkDataIdx)
|
||||||
|
dataBlk := v.blk(blkData)
|
||||||
|
le := binary.LittleEndian
|
||||||
|
dOff := le.Uint32(dataIdx[8*4:])
|
||||||
|
code := dataBlk[dOff : dOff+20]
|
||||||
|
want := []byte{
|
||||||
|
0x64, 0x20, 0xc0, 0x28, // ld.d r4, 8(r3)
|
||||||
|
0x65, 0x40, 0xc0, 0x28, // ld.d r5, 16(r3)
|
||||||
|
0x84, 0x94, 0x10, 0x00, // add.d r4, r4, r5
|
||||||
|
0x64, 0x60, 0xc0, 0x29, // st.d r4, 24(r3)
|
||||||
|
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if code[i] != want[i] {
|
||||||
|
t.Fatalf("code byte %d = %02x, want %02x", i, code[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The debug_line program: LNE_set_address (the R_ADDR relocation
|
||||||
|
// carries the function address), then one row per line change — the
|
||||||
|
// TEXT is on line 4 (a leading blank line precedes the include), the
|
||||||
|
// instructions on lines 5–9 — an advance to the 20-byte end and an
|
||||||
|
// end-of-sequence.
|
||||||
|
linesOff := le.Uint32(dataIdx[4*2:])
|
||||||
|
lines := dataBlk[linesOff : linesOff+21]
|
||||||
|
wantLines := []byte{
|
||||||
|
0x00, 0x09, 0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, // LNE_set_address
|
||||||
|
0x13, // pc 0, line 5
|
||||||
|
0x38, // pc 4, line 6
|
||||||
|
0x38, // pc 8, line 7
|
||||||
|
0x38, // pc 12, line 8
|
||||||
|
0x38, // pc 16, line 9
|
||||||
|
0x02, 0x04, // advance_pc to 20
|
||||||
|
0x00, 0x01, 0x01, // end_sequence
|
||||||
|
}
|
||||||
|
for i := range wantLines {
|
||||||
|
if lines[i] != wantLines[i] {
|
||||||
|
t.Fatalf("lines byte %d = %02x, want %02x", i, lines[i], wantLines[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The subprogram DIE: abbrev 3 (FUNCTION), the qualified name, the
|
||||||
|
// addrx low_pc slot (R_DWTXTADDR_U4), the size as high_pc, the
|
||||||
|
// call-frame-CFA frame base, decl file/line and the external flag.
|
||||||
|
dieOff := le.Uint32(dataIdx[4*3:])
|
||||||
|
die := dataBlk[dieOff : dieOff+27]
|
||||||
|
wantDie := []byte{
|
||||||
|
0x03,
|
||||||
|
't', 'e', 's', 't', 'p', 'k', 'g', '.', 'a', 'd', 'd', 0,
|
||||||
|
0x00, 0x00, 0x00, 0x00, // low_pc: addrx slot
|
||||||
|
0x14, // high_pc: 20
|
||||||
|
0x01, 0x9c, // frame_base: DW_OP_call_frame_cfa
|
||||||
|
0x01, 0x00, 0x00, 0x00, // decl_file: 1
|
||||||
|
0x04, // decl_line: 4
|
||||||
|
0x01, // external
|
||||||
|
0x00, // end of children
|
||||||
|
}
|
||||||
|
for i := range wantDie {
|
||||||
|
if die[i] != wantDie[i] {
|
||||||
|
t.Fatalf("DIE byte %d = %02x, want %02x", i, die[i], wantDie[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The DWARF symbols carry the function-address references: R_ADDR for
|
||||||
|
// the line program's set_address, R_DWTXTADDR_U4 for the DIE's addrx
|
||||||
|
// slot, both against the function's non-package index. The reloc
|
||||||
|
// index counts relocations, not bytes.
|
||||||
|
relocIdx := v.blk(blkRelocIdx)
|
||||||
|
relocs := v.blk(blkReloc)
|
||||||
|
if le.Uint32(relocIdx[4*2:]) != 0 || le.Uint32(relocIdx[4*3:]) != 1 || le.Uint32(relocIdx[4*4:]) != 2 {
|
||||||
|
t.Fatalf("dwarf reloc index ranges: %d %d %d", le.Uint32(relocIdx[4*2:]), le.Uint32(relocIdx[4*3:]), le.Uint32(relocIdx[4*4:]))
|
||||||
|
}
|
||||||
|
lr := relocs[:23]
|
||||||
|
if int32(le.Uint32(lr[0:])) != 3 || lr[4] != 8 || le.Uint16(lr[5:]) != relocAddr ||
|
||||||
|
le.Uint32(lr[15:]) != pkgIdxNone || le.Uint32(lr[19:]) != 4 {
|
||||||
|
t.Errorf("lines reloc = %x", lr)
|
||||||
|
}
|
||||||
|
dr := relocs[23:46]
|
||||||
|
if int32(le.Uint32(dr[0:])) != 13 || dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4 ||
|
||||||
|
le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 {
|
||||||
|
t.Errorf("die reloc = %x", dr)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The pc-value deltas are in MinLC (4) units: the flat pcsp covers
|
||||||
|
// the whole 20-byte function with a delta of 5.
|
||||||
|
pcspOff := le.Uint32(dataIdx[4*4:])
|
||||||
|
if got := dataBlk[pcspOff : pcspOff+3]; !bytes.Equal(got, []byte{0x02, 0x05, 0x00}) {
|
||||||
|
t.Errorf("pcsp = %x, want 020500", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGOObjectLOONG64Link cross-compiles a Go program with the gasm-produced
|
||||||
|
// object substituted into the package archive, proving cmd/link accepts the
|
||||||
|
// emitted GOOBJ. The binary is not executed (no LoongArch host or qemu).
|
||||||
|
// Skipped when no Go toolchain is available.
|
||||||
|
func TestGOObjectLOONG64Link(t *testing.T) {
|
||||||
|
goBin, err := exec.LookPath("go")
|
||||||
|
if err != nil {
|
||||||
|
t.Skip("no Go toolchain available")
|
||||||
|
}
|
||||||
|
dir := t.TempDir()
|
||||||
|
asmSrc := `#include "textflag.h"
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOVV a+0(FP), R4
|
||||||
|
MOVV b+8(FP), R5
|
||||||
|
ADDV R5, R4, R4
|
||||||
|
MOVV R4, ret+16(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main_loong64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
mainSrc := `package main
|
||||||
|
|
||||||
|
func add(a, b int64) int64
|
||||||
|
|
||||||
|
func main() {
|
||||||
|
if add(20, 22) != 42 {
|
||||||
|
panic("bad add")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module l64link\n\ngo 1.21\n"), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Capture the cross build (GOARCH=loong64): the package archive and the
|
||||||
|
// link line.
|
||||||
|
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||||
|
build.Dir = dir
|
||||||
|
build.Env = append(os.Environ(), "GOARCH=loong64")
|
||||||
|
buildLog, err := build.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||||
|
}
|
||||||
|
var pkgArch, work, linkLine, asmObj string
|
||||||
|
for _, line := range strings.Split(string(buildLog), "\n") {
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(line, "WORK="):
|
||||||
|
work = strings.TrimPrefix(line, "WORK=")
|
||||||
|
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_loong64.s") && !strings.Contains(line, "-gensymabis"):
|
||||||
|
asmObj = fieldAfter(line, "-o")
|
||||||
|
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
|
||||||
|
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
|
||||||
|
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
|
||||||
|
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||||
|
linkLine = line
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if pkgArch == "" || linkLine == "" || asmObj == "" {
|
||||||
|
t.Skip("could not locate the archive, asm output or link line in the build log")
|
||||||
|
}
|
||||||
|
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
|
||||||
|
// The archive member holding the assembler's output is named after the
|
||||||
|
// asm object file (main_loong64.o), as cmd/go packs it with `pack r`.
|
||||||
|
asmMember := filepath.Base(strings.ReplaceAll(asmObj, "$WORK", work))
|
||||||
|
|
||||||
|
// Assemble the same source with gasm and swap the object in.
|
||||||
|
pf, perrs := parser.Parse(filepath.Join(dir, "main_loong64.s"), asmSrc)
|
||||||
|
if len(perrs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", perrs)
|
||||||
|
}
|
||||||
|
pimg, err := AssembleFileLOONG64(pf)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := pimg.GOObjectLOONG64("main", filepath.Join(dir, "main_loong64.s"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObjectLOONG64: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Extract the archive, substitute the object member, repack.
|
||||||
|
membersDir := filepath.Join(dir, "members")
|
||||||
|
if err := os.MkdirAll(membersDir, 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
|
||||||
|
extract.Dir = membersDir
|
||||||
|
extract.Env = append(os.Environ(), "GOARCH=loong64")
|
||||||
|
if out, err := extract.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("pack x: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
// Substitute the gasm object for the assembler's archive member (pack
|
||||||
|
// extracts members read-only).
|
||||||
|
member := filepath.Join(membersDir, asmMember)
|
||||||
|
if err := os.Chmod(member, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(member, obj, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
|
||||||
|
listCmd.Env = append(os.Environ(), "GOARCH=loong64")
|
||||||
|
listOut, err := listCmd.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("pack t: %v\n%s", err, listOut)
|
||||||
|
}
|
||||||
|
newArch := filepath.Join(dir, "pkg.a")
|
||||||
|
args := []string{"tool", "pack", "c", newArch}
|
||||||
|
seen := map[string]bool{}
|
||||||
|
for _, m := range strings.Fields(string(listOut)) {
|
||||||
|
if seen[m] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
seen[m] = true
|
||||||
|
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
args = append(args, filepath.Join(membersDir, m))
|
||||||
|
}
|
||||||
|
pack := exec.Command(goBin, args...)
|
||||||
|
pack.Dir = membersDir
|
||||||
|
pack.Env = append(os.Environ(), "GOARCH=loong64")
|
||||||
|
if out, err := pack.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("pack c: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Re-link with our archive in place of the toolchain's. The link line
|
||||||
|
// carries a GOROOT assignment and $WORK placeholders; run it through the
|
||||||
|
// shell with the GOEXPERIMENT and GOARCH the toolchain expects (the
|
||||||
|
// linker compares the object header against its own, experiments
|
||||||
|
// included).
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
|
||||||
|
link := exec.Command("sh", "-c", linkLine)
|
||||||
|
link.Dir = dir
|
||||||
|
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
|
||||||
|
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)), "GOARCH=loong64")
|
||||||
|
if out, err := link.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("link with gasm object: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The binary is not executed: there is no LoongArch host or qemu here.
|
||||||
|
// The link itself and the symbol table prove cmd/link accepted the gasm
|
||||||
|
// object and laid out the function.
|
||||||
|
nm := exec.Command(goBin, "tool", "nm", filepath.Join(dir, "app2"))
|
||||||
|
nm.Env = append(os.Environ(), "GOARCH=loong64")
|
||||||
|
nmOut, err := nm.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("nm gasm-linked binary: %v\n%s", err, nmOut)
|
||||||
|
}
|
||||||
|
if !strings.Contains(string(nmOut), "main.add") {
|
||||||
|
t.Errorf("main.add not found in linked binary:\n%s", nmOut)
|
||||||
|
}
|
||||||
|
}
|
||||||
+189
-1
@@ -41,6 +41,7 @@ type FuncLayout struct {
|
|||||||
Labels map[string]int // local labels, function-relative
|
Labels map[string]int // local labels, function-relative
|
||||||
Relocs []Reloc // static-symbol references, in emission order
|
Relocs []Reloc // static-symbol references, in emission order
|
||||||
Spadj []SpadjStep // stack-adjustment boundaries, ascending by PC
|
Spadj []SpadjStep // stack-adjustment boundaries, ascending by PC
|
||||||
|
Lines []LineEntry // source-line table: byte offset → source line
|
||||||
}
|
}
|
||||||
|
|
||||||
// SpadjStep is one stack-adjustment boundary: Value is the SP delta from the
|
// SpadjStep is one stack-adjustment boundary: Value is the SP delta from the
|
||||||
@@ -50,17 +51,60 @@ type SpadjStep struct {
|
|||||||
Value int
|
Value int
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// LineEntry maps a byte offset (function-relative) to a source line number.
|
||||||
|
type LineEntry struct {
|
||||||
|
Offset int
|
||||||
|
Line int
|
||||||
|
}
|
||||||
|
|
||||||
|
// LineAt returns the source line number for the given function-relative byte
|
||||||
|
// offset, using a binary search on the line table. Returns 0 if the offset
|
||||||
|
// is before the first instruction or the table is empty.
|
||||||
|
func (fl *FuncLayout) LineAt(offset int) int {
|
||||||
|
if len(fl.Lines) == 0 {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
// Binary search: find the last entry with Offset <= offset.
|
||||||
|
lo, hi := 0, len(fl.Lines)-1
|
||||||
|
for lo < hi {
|
||||||
|
mid := (lo + hi + 1) / 2
|
||||||
|
if fl.Lines[mid].Offset <= offset {
|
||||||
|
lo = mid
|
||||||
|
} else {
|
||||||
|
hi = mid - 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if fl.Lines[lo].Offset <= offset {
|
||||||
|
return fl.Lines[lo].Line
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
// Reloc is one static-symbol reference within a function body: the disp32
|
// Reloc is one static-symbol reference within a function body: the disp32
|
||||||
// field at Off (function-relative) must reach the symbol plus Addend,
|
// field at Off (function-relative) must reach the symbol plus Addend,
|
||||||
// measured from After, the address just past the instruction. An External
|
// measured from After, the address just past the instruction. An External
|
||||||
// relocation names a symbol no GLOBL in the file defines; the object-file
|
// relocation names a symbol no GLOBL in the file defines; the object-file
|
||||||
// emitters carry it into the output's relocation table.
|
// emitters carry it into the output's relocation table.
|
||||||
|
// RelocKind discriminates the type of relocation needed.
|
||||||
|
type RelocKind int
|
||||||
|
|
||||||
|
const (
|
||||||
|
RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64)
|
||||||
|
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
|
||||||
|
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
|
||||||
|
RelRISCVJal // R_RISCV_JAL (J-type call)
|
||||||
|
RelPCRelAbs // 32-bit absolute (R_RISCV_32)
|
||||||
|
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
|
||||||
|
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
|
||||||
|
)
|
||||||
|
|
||||||
type Reloc struct {
|
type Reloc struct {
|
||||||
Off int
|
Off int
|
||||||
After int
|
After int
|
||||||
Name string
|
Name string
|
||||||
Addend int64
|
Addend int64
|
||||||
External bool
|
External bool
|
||||||
|
Kind RelocKind
|
||||||
}
|
}
|
||||||
|
|
||||||
// DataSymbol describes one GLOBL symbol laid out in the data section.
|
// DataSymbol describes one GLOBL symbol laid out in the data section.
|
||||||
@@ -110,7 +154,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
|||||||
if !ok {
|
if !ok {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
code, patches, labels, steps, err := assemble(t, link)
|
code, patches, labels, steps, lines, err := assemble(t, link)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||||
}
|
}
|
||||||
@@ -124,6 +168,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
|||||||
Args: argsSize(t),
|
Args: argsSize(t),
|
||||||
Line: t.Pos().Line,
|
Line: t.Pos().Line,
|
||||||
Labels: labels,
|
Labels: labels,
|
||||||
|
Lines: lines,
|
||||||
}
|
}
|
||||||
for _, f := range t.Flags {
|
for _, f := range t.Flags {
|
||||||
switch f {
|
switch f {
|
||||||
@@ -189,11 +234,153 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
|||||||
return img, nil
|
return img, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// AssembleFileRISCV assembles every TEXT function of a parsed RISC-V file
|
||||||
|
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
|
||||||
|
// code. SB references in the code are encoded as AUIPC pairs with zero
|
||||||
|
// immediates; the object-file emitters record relocations for the linker.
|
||||||
|
func AssembleFileRISCV(f *ast.File) (*Image, error) {
|
||||||
|
dataSyms, err := collectData(f)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
img := &Image{Symbols: map[string]int{}}
|
||||||
|
for _, d := range f.Decls {
|
||||||
|
t, ok := d.(*ast.Text)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
code, labels, relocs, lines, spadj, err := assembleRISCV(t)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||||
|
}
|
||||||
|
fl := FuncLayout{
|
||||||
|
Name: t.Name.Name,
|
||||||
|
Pkg: t.Name.Pkg,
|
||||||
|
Static: t.Name.Static,
|
||||||
|
Offset: len(img.Code),
|
||||||
|
Size: len(code),
|
||||||
|
Frame: frameSize(t),
|
||||||
|
Args: argsSize(t),
|
||||||
|
Line: t.Pos().Line,
|
||||||
|
Labels: labels,
|
||||||
|
Lines: lines,
|
||||||
|
Spadj: spadj,
|
||||||
|
Relocs: relocs,
|
||||||
|
}
|
||||||
|
for _, f := range t.Flags {
|
||||||
|
switch f {
|
||||||
|
case "NOSPLIT":
|
||||||
|
fl.NoSplit = true
|
||||||
|
case "SPWRITE":
|
||||||
|
fl.SPWrite = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
img.Funcs = append(img.Funcs, fl)
|
||||||
|
img.Code = append(img.Code, code...)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Lay out the data section behind the code, 16-aligned.
|
||||||
|
dataStart := len(img.Code)
|
||||||
|
for _, d := range dataSyms {
|
||||||
|
pos := dataStart + len(img.Data)
|
||||||
|
for pos%16 != 0 {
|
||||||
|
img.Data = append(img.Data, 0)
|
||||||
|
pos++
|
||||||
|
}
|
||||||
|
img.Symbols[d.name] = pos
|
||||||
|
img.Data = append(img.Data, d.buf...)
|
||||||
|
img.DataSyms = append(img.DataSyms, DataSymbol{
|
||||||
|
Name: d.name,
|
||||||
|
Pkg: d.pkg,
|
||||||
|
Offset: len(img.Data) - len(d.buf), // relative to the data section
|
||||||
|
Size: d.size,
|
||||||
|
Static: d.static,
|
||||||
|
Rodata: d.rodata,
|
||||||
|
Dupok: d.dupok,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
return img, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// AssembleFileLOONG64 assembles every TEXT function of a parsed loong64 file
|
||||||
|
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
|
||||||
|
// code. SB references in the code are encoded as pcalau12i pairs with zero
|
||||||
|
// immediates; the object-file emitters record R_LOONG64_ADDR_HI/LO
|
||||||
|
// relocations for the linker.
|
||||||
|
func AssembleFileLOONG64(f *ast.File) (*Image, error) {
|
||||||
|
dataSyms, err := collectData(f)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
img := &Image{Symbols: map[string]int{}}
|
||||||
|
for _, d := range f.Decls {
|
||||||
|
t, ok := d.(*ast.Text)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
code, labels, relocs, lines, spadj, err := assembleLOONG64(t)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||||
|
}
|
||||||
|
fl := FuncLayout{
|
||||||
|
Name: t.Name.Name,
|
||||||
|
Pkg: t.Name.Pkg,
|
||||||
|
Static: t.Name.Static,
|
||||||
|
Offset: len(img.Code),
|
||||||
|
Size: len(code),
|
||||||
|
Frame: frameSize(t),
|
||||||
|
Args: argsSize(t),
|
||||||
|
Line: t.Pos().Line,
|
||||||
|
Labels: labels,
|
||||||
|
Lines: lines,
|
||||||
|
Spadj: spadj,
|
||||||
|
Relocs: relocs,
|
||||||
|
}
|
||||||
|
for _, f := range t.Flags {
|
||||||
|
switch f {
|
||||||
|
case "NOSPLIT":
|
||||||
|
fl.NoSplit = true
|
||||||
|
case "SPWRITE":
|
||||||
|
fl.SPWrite = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
img.Funcs = append(img.Funcs, fl)
|
||||||
|
img.Code = append(img.Code, code...)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Lay out the data section behind the code, 16-aligned.
|
||||||
|
dataStart := len(img.Code)
|
||||||
|
for _, d := range dataSyms {
|
||||||
|
pos := dataStart + len(img.Data)
|
||||||
|
for pos%16 != 0 {
|
||||||
|
img.Data = append(img.Data, 0)
|
||||||
|
pos++
|
||||||
|
}
|
||||||
|
img.Symbols[d.name] = pos
|
||||||
|
img.Data = append(img.Data, d.buf...)
|
||||||
|
img.DataSyms = append(img.DataSyms, DataSymbol{
|
||||||
|
Name: d.name,
|
||||||
|
Pkg: d.pkg,
|
||||||
|
Offset: len(img.Data) - len(d.buf), // relative to the data section
|
||||||
|
Size: d.size,
|
||||||
|
Static: d.static,
|
||||||
|
Rodata: d.rodata,
|
||||||
|
Dupok: d.dupok,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
return img, nil
|
||||||
|
}
|
||||||
|
|
||||||
// dataSym is one GLOBL symbol and its DATA initialiser.
|
// dataSym is one GLOBL symbol and its DATA initialiser.
|
||||||
type dataSym struct {
|
type dataSym struct {
|
||||||
name string
|
name string
|
||||||
pkg string
|
pkg string
|
||||||
buf []byte
|
buf []byte
|
||||||
|
size int
|
||||||
static bool
|
static bool
|
||||||
rodata bool
|
rodata bool
|
||||||
dupok bool
|
dupok bool
|
||||||
@@ -223,6 +410,7 @@ func collectData(f *ast.File) ([]dataSym, error) {
|
|||||||
name: name,
|
name: name,
|
||||||
pkg: dd.Name.Pkg,
|
pkg: dd.Name.Pkg,
|
||||||
buf: make([]byte, size),
|
buf: make([]byte, size),
|
||||||
|
size: size,
|
||||||
static: dd.Name.Static,
|
static: dd.Name.Static,
|
||||||
}
|
}
|
||||||
for _, f := range dd.Flags {
|
for _, f := range dd.Flags {
|
||||||
|
|||||||
@@ -4,13 +4,9 @@
|
|||||||
package asm
|
package asm
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
|
||||||
"os"
|
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"golang.org/x/arch/x86/x86asm"
|
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -132,65 +128,3 @@ DATA x<>+0(SB)/4, $1
|
|||||||
t.Errorf("single-function SB: error %v, want a file-level-assembly error", err)
|
t.Errorf("single-function SB: error %v, want a file-level-assembly error", err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel —
|
|
||||||
// all functions plus the file-local mask24 constant — and checks that every
|
|
||||||
// static-symbol load resolves to the right bytes in the image. Skipped when
|
|
||||||
// the sibling repository is not checked out.
|
|
||||||
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
|
||||||
path := "../../go-libraries/go-flac/avx2_amd64.s"
|
|
||||||
if _, err := os.Stat(path); err != nil {
|
|
||||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
|
||||||
}
|
|
||||||
src, err := os.ReadFile(path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
f, errs := parser.Parse(path, string(src))
|
|
||||||
if len(errs) > 0 {
|
|
||||||
t.Fatalf("parse: %v", errs)
|
|
||||||
}
|
|
||||||
img, err := AssembleFile(f)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("AssembleFile: %v", err)
|
|
||||||
}
|
|
||||||
if len(img.Funcs) != 17 {
|
|
||||||
t.Errorf("functions = %d, want 17", len(img.Funcs))
|
|
||||||
}
|
|
||||||
|
|
||||||
// mask24 as the DATA directives define it.
|
|
||||||
mask := []byte{
|
|
||||||
0x00, 0x01, 0x02, 0x80, 0x03, 0x04, 0x05, 0x80,
|
|
||||||
0x06, 0x07, 0x08, 0x80, 0x09, 0x0a, 0x0b, 0x80,
|
|
||||||
}
|
|
||||||
image := img.Bytes()
|
|
||||||
if got := image[img.Symbols["mask24"] : img.Symbols["mask24"]+16]; !bytes.Equal(got, mask) {
|
|
||||||
t.Errorf("mask24 contents %x, want %x", got, mask)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Every VMOVDQU mask24<>(SB), X15 (c5 7a 6f 3d + rel32, i.e. a VMOVDQU
|
|
||||||
// with a RIP-relative r/m) must land on the mask bytes within the image.
|
|
||||||
loads := 0
|
|
||||||
for _, fn := range img.Funcs {
|
|
||||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
|
||||||
for pc := 0; pc < len(code); {
|
|
||||||
inst, err := x86asm.Decode(code[pc:], 64)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("%s: decode at +%d: %v", fn.Name, pc, err)
|
|
||||||
}
|
|
||||||
// mod=00, rm=101 → RIP-relative.
|
|
||||||
if inst.Op == x86asm.VMOVDQU && inst.Len == 8 && code[pc+3]&0xC7 == 0x05 {
|
|
||||||
rel := int32(uint32(code[pc+4]) | uint32(code[pc+5])<<8 | uint32(code[pc+6])<<16 | uint32(code[pc+7])<<24)
|
|
||||||
target := fn.Offset + pc + 8 + int(rel)
|
|
||||||
if !bytes.Equal(image[target:target+16], mask) {
|
|
||||||
t.Errorf("%s: mask load at +%d lands on %x, want %x", fn.Name, pc, image[target:target+16], mask)
|
|
||||||
}
|
|
||||||
loads++
|
|
||||||
}
|
|
||||||
pc += inst.Len
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if loads != 2 {
|
|
||||||
t.Errorf("mask loads found = %d, want 2", loads)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,595 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
// loong64 (LoongArch) instruction encoding.
|
||||||
|
//
|
||||||
|
// The encoder is data-driven: each mnemonic maps to an instruction format and
|
||||||
|
// an opcode constant, and the format selects the bit layout. The opcode
|
||||||
|
// constants and formats are transcribed from the Go toolchain's own loong64
|
||||||
|
// backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm`
|
||||||
|
// exactly — the ground-truth oracle for the verify suite.
|
||||||
|
//
|
||||||
|
// All LoongArch instructions are 32 bits, little-endian. The formats used
|
||||||
|
// here (per the LoongArch Volume I specification):
|
||||||
|
//
|
||||||
|
// 3R opcode[31:15] | rk[4:0] | rj[4:0] | rd[4:0]
|
||||||
|
// 2R opcode[31:15] | rj[4:0] | rd[4:0]
|
||||||
|
// 2RI12 opcode[31:22] | si12[11:0] | rj[4:0] | rd[4:0]
|
||||||
|
// 2RI14 opcode[31:18] | si14[13:0] | rj[4:0] | rd[4:0]
|
||||||
|
// 2RI16 opcode[31:22] | si16[15:0] | rj[4:0] | rd[4:0]
|
||||||
|
// 2RI20 opcode[31:25] | si20[19:0] | rd[4:0]
|
||||||
|
// 1RI21 opcode[31:26] | si21[20:0] | rj[4:0] (BEQZ/BNEZ, B*Z, BC*Z)
|
||||||
|
// B/BL opcode[31:26] | offs[25:0]
|
||||||
|
// 4R opcode[31:20] | r1[4:0] | r2[4:0] | r3[4:0] | r4[4:0]
|
||||||
|
// IRIR opcode[31:22] | msb[4:0] | rj[4:0] | lsb[4:0] | rd[4:0]
|
||||||
|
// 3RI2 opcode[31:17] | sa2[1:0] | rk[4:0] | rj[4:0] | rd[4:0]
|
||||||
|
//
|
||||||
|
// The opcode constants are pre-positioned (they include the zero bit ranges
|
||||||
|
// of the immediate and register fields), mirroring the toolchain's OP_*
|
||||||
|
// helpers, so each l64* function only ORs its fields in.
|
||||||
|
|
||||||
|
// loong64RegNum returns the 5-bit register number for a LoongArch register
|
||||||
|
// name: R0–R31 (integer), F0–F31 (floating point), FCC0–FCC7 (condition
|
||||||
|
// flags), FCSR0–FCSR31 (control/status) and the ABI aliases the runtime's
|
||||||
|
// assembly uses. Returns -1 for an unrecognised name.
|
||||||
|
func loong64RegNum(name string) int {
|
||||||
|
switch name {
|
||||||
|
case "R0", "ZERO":
|
||||||
|
return 0
|
||||||
|
case "R1", "RA", "LINK":
|
||||||
|
return 1
|
||||||
|
case "R2", "TP":
|
||||||
|
return 2
|
||||||
|
case "R3", "SP":
|
||||||
|
return 3
|
||||||
|
case "R4", "A0":
|
||||||
|
return 4
|
||||||
|
case "R5", "A1":
|
||||||
|
return 5
|
||||||
|
case "R6", "A2":
|
||||||
|
return 6
|
||||||
|
case "R7", "A3":
|
||||||
|
return 7
|
||||||
|
case "R8", "A4":
|
||||||
|
return 8
|
||||||
|
case "R9", "A5":
|
||||||
|
return 9
|
||||||
|
case "R10", "A6":
|
||||||
|
return 10
|
||||||
|
case "R11", "A7":
|
||||||
|
return 11
|
||||||
|
case "R12", "T0":
|
||||||
|
return 12
|
||||||
|
case "R13", "T1":
|
||||||
|
return 13
|
||||||
|
case "R14", "T2":
|
||||||
|
return 14
|
||||||
|
case "R15", "T3":
|
||||||
|
return 15
|
||||||
|
case "R16", "T4":
|
||||||
|
return 16
|
||||||
|
case "R17", "T5":
|
||||||
|
return 17
|
||||||
|
case "R18", "T6":
|
||||||
|
return 18
|
||||||
|
case "R19", "T7":
|
||||||
|
return 19
|
||||||
|
case "R20", "T8":
|
||||||
|
return 20
|
||||||
|
case "R21":
|
||||||
|
return 21
|
||||||
|
case "R22", "G", "g", "FP":
|
||||||
|
return 22
|
||||||
|
case "R23", "S0":
|
||||||
|
return 23
|
||||||
|
case "R24", "S1":
|
||||||
|
return 24
|
||||||
|
case "R25", "S2":
|
||||||
|
return 25
|
||||||
|
case "R26", "S3":
|
||||||
|
return 26
|
||||||
|
case "R27", "S4":
|
||||||
|
return 27
|
||||||
|
case "R28", "S5":
|
||||||
|
return 28
|
||||||
|
case "R29", "S6", "CTXT":
|
||||||
|
return 29
|
||||||
|
case "R30", "S7", "TMP":
|
||||||
|
return 30
|
||||||
|
case "R31", "S8":
|
||||||
|
return 31
|
||||||
|
}
|
||||||
|
// F0–F31, FCC0–FCC7, FCSR0–FCSR31.
|
||||||
|
if len(name) >= 4 && name[:4] == "FCSR" {
|
||||||
|
return loong64RegSpecial(name[4:], "FCSR", 31)
|
||||||
|
}
|
||||||
|
if len(name) >= 3 && name[:3] == "FCC" {
|
||||||
|
return loong64RegSpecial(name[3:], "FCC", 7)
|
||||||
|
}
|
||||||
|
if len(name) < 2 {
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
prefix, digits := name[:1], name[1:]
|
||||||
|
if digits[0] < '0' || digits[0] > '9' {
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
n := 0
|
||||||
|
for i := 0; i < len(digits); i++ {
|
||||||
|
if digits[i] < '0' || digits[i] > '9' {
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
n = n*10 + int(digits[i]-'0')
|
||||||
|
}
|
||||||
|
if prefix == "F" && n <= 31 {
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
|
||||||
|
// loong64RegSpecial parses a numbered FCC/FCSR register.
|
||||||
|
func loong64RegSpecial(digits, prefix string, max int) int {
|
||||||
|
if digits == "" {
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
n := 0
|
||||||
|
for i := 0; i < len(digits); i++ {
|
||||||
|
if digits[i] < '0' || digits[i] > '9' {
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
n = n*10 + int(digits[i]-'0')
|
||||||
|
}
|
||||||
|
if n <= max {
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- format helpers ----
|
||||||
|
|
||||||
|
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
|
||||||
|
func l64rrr(op uint32, rk, rj, rd int) uint32 {
|
||||||
|
return op | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64rr encodes a 2R instruction: op | rj<<5 | rd.
|
||||||
|
func l64rr(op uint32, rj, rd int) uint32 {
|
||||||
|
return op | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64irr encodes a 2RI12 instruction: op | si12<<10 | rj<<5 | rd.
|
||||||
|
func l64irr(op uint32, imm, rj, rd int) uint32 {
|
||||||
|
return op | (uint32(imm)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64irr14 encodes a 2RI14 instruction: op | si14<<10 | rj<<5 | rd.
|
||||||
|
func l64irr14(op uint32, imm, rj, rd int) uint32 {
|
||||||
|
return op | (uint32(imm)&0x3FFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64irr16 encodes a 2RI16 instruction: op | si16<<10 | rj<<5 | rd.
|
||||||
|
func l64irr16(op uint32, imm, rj, rd int) uint32 {
|
||||||
|
return op | (uint32(imm)&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64ir encodes a 2RI20 instruction: op | si20<<5 | rd.
|
||||||
|
func l64ir(op uint32, imm, rd int) uint32 {
|
||||||
|
return op | (uint32(imm)&0xFFFFF)<<5 | uint32(rd&0x1f)
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64bbl encodes a B/BL instruction: op | offs[25:0], where offs is the
|
||||||
|
// 4-byte-aligned word distance (the toolchain stores the shifted value).
|
||||||
|
func l64bbl(op uint32, offs int) uint32 {
|
||||||
|
return op | (uint32(offs)&0xFFFF)<<10 | (uint32(offs)>>16)&0x3FF
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64ir21 encodes a 1RI21 branch (BEQZ/BNEZ, BLTZ/BGEZ/BLEZ/BGTZ, BFPT/BFPF):
|
||||||
|
// op | si21[15:0]<<10 | rj<<5 | si21[20:16].
|
||||||
|
func l64ir21(op uint32, offs, rj int) uint32 {
|
||||||
|
v := uint32(offs)
|
||||||
|
return op | (v&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | (v>>16)&0x1F
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64rrrr encodes a 4R instruction: op | r1<<15 | r2<<10 | r3<<5 | r4.
|
||||||
|
func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
|
||||||
|
return op | uint32(r1&0x1f)<<15 | uint32(r2&0x1f)<<10 | uint32(r3&0x1f)<<5 | uint32(r4&0x1f)
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
|
||||||
|
// The msb/lsb fields are 6 bits wide (0–63) and are validated by the caller.
|
||||||
|
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
|
||||||
|
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64irrr encodes a 3RI2 instruction (ALSL): op | sa<<15 | rk<<10 | rj<<5 | rd.
|
||||||
|
func l64irrr(op uint32, sa, rk, rj, rd int) uint32 {
|
||||||
|
return op | uint32(sa&0x3)<<15 | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64i15 encodes a no-operand system instruction with a 15-bit code field
|
||||||
|
// (SYSCALL, BREAK, DBAR): op | code[14:0].
|
||||||
|
func l64i15(op uint32, code int) uint32 {
|
||||||
|
return op | uint32(code)&0x7FFF
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64irr5i encodes PRELD: op | offs<<10 | rj<<5 | hint.
|
||||||
|
func l64irr5i(op uint32, offs, rj, hint int) uint32 {
|
||||||
|
return op | (uint32(offs)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(hint&0x1f)
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64wordLE encodes a uint32 as 4 little-endian bytes.
|
||||||
|
func l64wordLE(w uint32) []byte {
|
||||||
|
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64WordsLE concatenates one or more instruction words as little-endian bytes.
|
||||||
|
func l64WordsLE(ws ...uint32) []byte {
|
||||||
|
var out []byte
|
||||||
|
for _, w := range ws {
|
||||||
|
out = append(out, l64wordLE(w)...)
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- instruction formats ----
|
||||||
|
|
||||||
|
type l64Format uint8
|
||||||
|
|
||||||
|
const (
|
||||||
|
l64Frrr l64Format = iota // 3R (integer and FP arithmetic)
|
||||||
|
l64Frr // 2R
|
||||||
|
l64Firr // 2RI12 (arithmetic with 12-bit immediate)
|
||||||
|
l64Firr14 // 2RI14 (ldptr/stptr)
|
||||||
|
l64Firr16 // 2RI16 (addu16i.d)
|
||||||
|
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
|
||||||
|
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub)
|
||||||
|
l64Firir // bstrins/bstrpick
|
||||||
|
l64Firrr // alsl
|
||||||
|
l64Fi15 // syscall/break/dbar
|
||||||
|
l64Fam // atomic (3R with the AM field order)
|
||||||
|
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
|
||||||
|
l64Fshift // 2RI12 with a 5/6-bit shift immediate
|
||||||
|
l64Fpreld // preld (2RI12 + 5-bit hint)
|
||||||
|
)
|
||||||
|
|
||||||
|
// l64Enc is one instruction's encoding: its bit layout (format) and the
|
||||||
|
// opcode constant, positioned at its exact bit range.
|
||||||
|
type l64Enc struct {
|
||||||
|
format l64Format
|
||||||
|
op uint32
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64DualEnc holds both forms of a dual-form mnemonic: the 3R register form
|
||||||
|
// and the 2RI12 immediate form (which is a shift for the shift mnemonics).
|
||||||
|
type l64DualEnc struct {
|
||||||
|
rrr uint32 // 3R register form
|
||||||
|
imm uint32 // 2RI12 immediate form
|
||||||
|
shift bool // the immediate form is a 5/6-bit shift amount
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64DualTable maps the dual-form arithmetic/logic mnemonics to both
|
||||||
|
// encodings; the assembler picks by operand kind.
|
||||||
|
var l64DualTable = map[string]l64DualEnc{}
|
||||||
|
|
||||||
|
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
|
||||||
|
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered
|
||||||
|
// yet; the base integer, memory and floating-point ISA is complete.
|
||||||
|
var l64InstrTable = map[string]l64Enc{}
|
||||||
|
|
||||||
|
func init() {
|
||||||
|
// 3R — integer.
|
||||||
|
rrr := map[string]uint32{
|
||||||
|
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
|
||||||
|
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
|
||||||
|
"SGT": 0x24 << 15, "SGTU": 0x25 << 15,
|
||||||
|
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15,
|
||||||
|
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
|
||||||
|
"ORN": 0x2c << 15, "ANDN": 0x2d << 15,
|
||||||
|
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
|
||||||
|
"SLLV": 0x31 << 15, "SRLV": 0x32 << 15, "SRAV": 0x33 << 15,
|
||||||
|
"ROTR": 0x36 << 15, "ROTRV": 0x37 << 15,
|
||||||
|
"MUL": 0x38 << 15, "MULW": 0x38 << 15, "MULH": 0x39 << 15, "MULHU": 0x3a << 15,
|
||||||
|
"MULV": 0x3b << 15, "MULVU": 0x3b << 15, "MULHV": 0x3c << 15, "MULHVU": 0x3d << 15,
|
||||||
|
"MULWVW": 0x3e << 15, "MULWVWU": 0x3f << 15,
|
||||||
|
"DIV": 0x40 << 15, "DIVW": 0x40 << 15, "REM": 0x41 << 15, "REMW": 0x41 << 15,
|
||||||
|
"DIVU": 0x42 << 15, "DIVWU": 0x42 << 15, "REMU": 0x43 << 15, "REMWU": 0x43 << 15,
|
||||||
|
"DIVV": 0x44 << 15, "REMV": 0x45 << 15, "DIVVU": 0x46 << 15, "REMVU": 0x47 << 15,
|
||||||
|
"CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15,
|
||||||
|
"CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15,
|
||||||
|
}
|
||||||
|
// 3R — floating point.
|
||||||
|
rrr["MULF"] = 0x209 << 15
|
||||||
|
rrr["MULD"] = 0x20a << 15
|
||||||
|
rrr["DIVF"] = 0x20d << 15
|
||||||
|
rrr["DIVD"] = 0x20e << 15
|
||||||
|
rrr["SUBF"] = 0x205 << 15
|
||||||
|
rrr["SUBD"] = 0x206 << 15
|
||||||
|
rrr["ADDF"] = 0x201 << 15
|
||||||
|
rrr["ADDD"] = 0x202 << 15
|
||||||
|
rrr["CMPEQF"] = 0x0c1<<20 | 0x4<<15
|
||||||
|
rrr["CMPEQD"] = 0x0c2<<20 | 0x4<<15
|
||||||
|
rrr["CMPGED"] = 0x0c2<<20 | 0x7<<15
|
||||||
|
rrr["CMPGEF"] = 0x0c1<<20 | 0x7<<15
|
||||||
|
rrr["CMPGTD"] = 0x0c2<<20 | 0x3<<15
|
||||||
|
rrr["CMPGTF"] = 0x0c1<<20 | 0x3<<15
|
||||||
|
rrr["FMINF"] = 0x215 << 15
|
||||||
|
rrr["FMIND"] = 0x216 << 15
|
||||||
|
rrr["FMAXF"] = 0x211 << 15
|
||||||
|
rrr["FMAXD"] = 0x212 << 15
|
||||||
|
rrr["FMAXAF"] = 0x219 << 15
|
||||||
|
rrr["FMAXAD"] = 0x21a << 15
|
||||||
|
rrr["FMINAF"] = 0x21d << 15
|
||||||
|
rrr["FMINAD"] = 0x21e << 15
|
||||||
|
rrr["FSCALEBF"] = 0x221 << 15
|
||||||
|
rrr["FSCALEBD"] = 0x222 << 15
|
||||||
|
rrr["FCOPYSGF"] = 0x225 << 15
|
||||||
|
rrr["FCOPYSGD"] = 0x226 << 15
|
||||||
|
for m, op := range rrr {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Frrr, op: op}
|
||||||
|
}
|
||||||
|
|
||||||
|
// 2R.
|
||||||
|
rr := map[string]uint32{
|
||||||
|
"CLOW": 0x4 << 10, "CLZW": 0x5 << 10, "CTOW": 0x6 << 10, "CTZW": 0x7 << 10,
|
||||||
|
"CLOV": 0x8 << 10, "CLZV": 0x9 << 10, "CTOV": 0xa << 10, "CTZV": 0xb << 10,
|
||||||
|
"REVB2H": 0xc << 10, "REVB4H": 0xd << 10, "REVB2W": 0xe << 10, "REVBV": 0xf << 10,
|
||||||
|
"REVH2W": 0x10 << 10, "REVHV": 0x11 << 10,
|
||||||
|
"BITREV4B": 0x12 << 10, "BITREV8B": 0x13 << 10, "BITREVW": 0x14 << 10, "BITREVV": 0x15 << 10,
|
||||||
|
"EXTWH": 0x16 << 10, "EXTWB": 0x17 << 10, "CPUCFG": 0x1b << 10,
|
||||||
|
"TRUNCFV": 0x46a9 << 10, "TRUNCDV": 0x46aa << 10, "TRUNCFW": 0x46a1 << 10, "TRUNCDW": 0x46a2 << 10,
|
||||||
|
"MOVWF": 0x4744 << 10, "MOVVF": 0x4746 << 10, "MOVWD": 0x4748 << 10, "MOVVD": 0x474a << 10,
|
||||||
|
"MOVFW": 0x46c1 << 10, "MOVDW": 0x46c2 << 10, "MOVFV": 0x46c9 << 10, "MOVDV": 0x46ca << 10,
|
||||||
|
"FRINTF": 0x4791 << 10, "FRINTD": 0x4792 << 10,
|
||||||
|
"MOVDF": 0x4646 << 10, "MOVFD": 0x4649 << 10,
|
||||||
|
"ABSF": 0x4501 << 10, "ABSD": 0x4502 << 10,
|
||||||
|
"MOVF": 0x4525 << 10, "MOVD": 0x4526 << 10,
|
||||||
|
"NEGF": 0x4505 << 10, "NEGD": 0x4506 << 10,
|
||||||
|
"SQRTF": 0x4511 << 10, "SQRTD": 0x4512 << 10,
|
||||||
|
"FLOGBF": 0x4509 << 10, "FLOGBD": 0x450a << 10,
|
||||||
|
"FCLASSF": 0x450d << 10, "FCLASSD": 0x450e << 10,
|
||||||
|
"FTINTRMWF": 0x4681 << 10, "FTINTRMWD": 0x4682 << 10,
|
||||||
|
"FTINTRMVF": 0x4689 << 10, "FTINTRMVD": 0x468a << 10,
|
||||||
|
"FTINTRPWF": 0x4691 << 10, "FTINTRPWD": 0x4692 << 10,
|
||||||
|
"FTINTRPVF": 0x4699 << 10, "FTINTRPVD": 0x469a << 10,
|
||||||
|
"FTINTRZWF": 0x46a1 << 10, "FTINTRZWD": 0x46a2 << 10,
|
||||||
|
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
|
||||||
|
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
|
||||||
|
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
|
||||||
|
}
|
||||||
|
for m, op := range rr {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
|
||||||
|
}
|
||||||
|
// RDTIME is a 2R instruction with rd and rj in swapped positions.
|
||||||
|
l64InstrTable["RDTIMELW"] = l64Enc{format: l64Frdtime, op: 0x18 << 10}
|
||||||
|
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
|
||||||
|
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
|
||||||
|
|
||||||
|
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
|
||||||
|
// selected by the operand kind; the shift mnemonics pair the 3R form
|
||||||
|
// with a 5/6-bit shift immediate.
|
||||||
|
for m, e := range map[string]l64DualEnc{
|
||||||
|
"ADD": {rrr: 0x20 << 15, imm: 0x00a << 22},
|
||||||
|
"ADDW": {rrr: 0x20 << 15, imm: 0x00a << 22},
|
||||||
|
"ADDV": {rrr: 0x21 << 15, imm: 0x00b << 22},
|
||||||
|
"ADDVU": {rrr: 0x21 << 15, imm: 0x00b << 22},
|
||||||
|
"AND": {rrr: 0x29 << 15, imm: 0x00d << 22},
|
||||||
|
"OR": {rrr: 0x2a << 15, imm: 0x00e << 22},
|
||||||
|
"XOR": {rrr: 0x2b << 15, imm: 0x00f << 22},
|
||||||
|
"SGT": {rrr: 0x24 << 15, imm: 0x008 << 22},
|
||||||
|
"SGTU": {rrr: 0x25 << 15, imm: 0x009 << 22},
|
||||||
|
"SLL": {rrr: 0x2e << 15, imm: 0x00081 << 15, shift: true},
|
||||||
|
"SRL": {rrr: 0x2f << 15, imm: 0x00089 << 15, shift: true},
|
||||||
|
"SRA": {rrr: 0x30 << 15, imm: 0x00091 << 15, shift: true},
|
||||||
|
"ROTR": {rrr: 0x36 << 15, imm: 0x00099 << 15, shift: true},
|
||||||
|
"SLLV": {rrr: 0x31 << 15, imm: 0x0041 << 16, shift: true},
|
||||||
|
"SRLV": {rrr: 0x32 << 15, imm: 0x0045 << 16, shift: true},
|
||||||
|
"SRAV": {rrr: 0x33 << 15, imm: 0x0049 << 16, shift: true},
|
||||||
|
"ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true},
|
||||||
|
} {
|
||||||
|
l64DualTable[m] = e
|
||||||
|
}
|
||||||
|
|
||||||
|
// 2RI12 — pure immediate arithmetic (LU52ID has no register form).
|
||||||
|
l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22}
|
||||||
|
// ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16.
|
||||||
|
l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26}
|
||||||
|
|
||||||
|
// 2RI14 — LL/SC are aliased by the Go assembler to the pointer loads and
|
||||||
|
// stores (ldptr/stptr), with the offset scaled by 4.
|
||||||
|
l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w
|
||||||
|
l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d
|
||||||
|
l64InstrTable["SC"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w
|
||||||
|
l64InstrTable["SCW"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w
|
||||||
|
l64InstrTable["SCV"] = l64Enc{format: l64Firr14, op: 0x23 << 24} // sc.d
|
||||||
|
l64InstrTable["LL"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w)
|
||||||
|
l64InstrTable["LLW"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w)
|
||||||
|
l64InstrTable["LLV"] = l64Enc{format: l64Firr14, op: 0x22 << 24} // ldptr.d (ll.d)
|
||||||
|
|
||||||
|
// 2RI20.
|
||||||
|
l64InstrTable["LU12IW"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
||||||
|
l64InstrTable["LU32ID"] = l64Enc{format: l64Fir20, op: 0x0b << 25}
|
||||||
|
l64InstrTable["PCALAU12I"] = l64Enc{format: l64Fir20, op: 0x0d << 25}
|
||||||
|
l64InstrTable["PCADDU12I"] = l64Enc{format: l64Fir20, op: 0x0e << 25}
|
||||||
|
// LUI is the Plan 9 spelling of lu12i.w.
|
||||||
|
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
||||||
|
|
||||||
|
// 4R — fused multiply-add.
|
||||||
|
rrrr := map[string]uint32{
|
||||||
|
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
|
||||||
|
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
|
||||||
|
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
|
||||||
|
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
|
||||||
|
}
|
||||||
|
for m, op := range rrrr {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
|
||||||
|
}
|
||||||
|
|
||||||
|
// IRIR — bit-field insert/extract.
|
||||||
|
irir := map[string]uint32{
|
||||||
|
"BSTRINSW": 0x3<<21 | 0x0<<15,
|
||||||
|
"BSTRINSV": 0x2 << 22,
|
||||||
|
"BSTRPICKW": 0x3<<21 | 0x1<<15,
|
||||||
|
"BSTRPICKV": 0x3 << 22,
|
||||||
|
}
|
||||||
|
for m, op := range irir {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Firir, op: op}
|
||||||
|
}
|
||||||
|
|
||||||
|
// 3RI2 — ALSL.
|
||||||
|
irrr := map[string]uint32{
|
||||||
|
"ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17,
|
||||||
|
}
|
||||||
|
for m, op := range irrr {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Firrr, op: op}
|
||||||
|
}
|
||||||
|
|
||||||
|
// 0-operand system instructions.
|
||||||
|
l64InstrTable["SYSCALL"] = l64Enc{format: l64Fi15, op: 0x56 << 15}
|
||||||
|
l64InstrTable["BREAK"] = l64Enc{format: l64Fi15, op: 0x54 << 15}
|
||||||
|
l64InstrTable["DBAR"] = l64Enc{format: l64Fi15, op: 0x70e4 << 15}
|
||||||
|
|
||||||
|
// PRELD.
|
||||||
|
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
|
||||||
|
|
||||||
|
// Atomics — 3R with the AM field order (rk=value, rj=address, rd=result).
|
||||||
|
am := map[string]uint32{
|
||||||
|
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
|
||||||
|
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
|
||||||
|
"AMCASB": 0x070B0 << 15, "AMCASH": 0x070B1 << 15,
|
||||||
|
"AMCASW": 0x070B2 << 15, "AMCASV": 0x070B3 << 15,
|
||||||
|
"AMADDW": 0x070C2 << 15, "AMADDV": 0x070C3 << 15,
|
||||||
|
"AMANDW": 0x070C4 << 15, "AMANDV": 0x070C5 << 15,
|
||||||
|
"AMORW": 0x070C6 << 15, "AMORV": 0x070C7 << 15,
|
||||||
|
"AMXORW": 0x070C8 << 15, "AMXORV": 0x070C9 << 15,
|
||||||
|
"AMMAXW": 0x070CA << 15, "AMMAXV": 0x070CB << 15,
|
||||||
|
"AMMINW": 0x070CC << 15, "AMMINV": 0x070CD << 15,
|
||||||
|
"AMMAXWU": 0x070CE << 15, "AMMAXVU": 0x070CF << 15,
|
||||||
|
"AMMINWU": 0x070D0 << 15, "AMMINVU": 0x070D1 << 15,
|
||||||
|
"AMSWAPDBB": 0x070BC << 15, "AMSWAPDBH": 0x070BD << 15,
|
||||||
|
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
|
||||||
|
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
|
||||||
|
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
|
||||||
|
}
|
||||||
|
for m, op := range am {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
|
||||||
|
// register move between the integer and floating-point register banks — the
|
||||||
|
// MOVW/MOVV specials the Go assembler accepts.
|
||||||
|
var l64FpMovTable = map[string]uint32{
|
||||||
|
"MOVV.R.F": 0x452a << 10, // movgr2fr.d
|
||||||
|
"MOVV.R.FCC": 0x4536 << 10, // movgr2cf
|
||||||
|
"MOVV.R.FCSR": 0x4530 << 10, // movgr2fcsr
|
||||||
|
"MOVV.F.R": 0x452e << 10, // movfr2gr.d
|
||||||
|
"MOVV.F.FCC": 0x4534 << 10, // movfr2cf
|
||||||
|
"MOVV.FCC.R": 0x4537 << 10, // movcf2gr
|
||||||
|
"MOVV.FCC.F": 0x4535 << 10, // movcf2fr
|
||||||
|
"MOVV.FCSR.R": 0x4532 << 10, // movfcsr2gr
|
||||||
|
"MOVW.R.F": 0x4529 << 10, // movgr2fr.w
|
||||||
|
"MOVW.F.R": 0x452d << 10, // movfr2gr.s
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64branchTable holds the 16-bit branch and jump encodings (2RI16).
|
||||||
|
var l64branchTable = map[string]uint32{
|
||||||
|
"BEQ": 0x16 << 26,
|
||||||
|
"BNE": 0x17 << 26,
|
||||||
|
"BLT": 0x18 << 26,
|
||||||
|
"BGE": 0x19 << 26,
|
||||||
|
"BLTU": 0x1a << 26,
|
||||||
|
"BGEU": 0x1b << 26,
|
||||||
|
"JIRL": 0x13 << 26,
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64branch21Table holds the single-register branches with 21-bit offsets:
|
||||||
|
// the negative opcode constants the toolchain uses for the short forms.
|
||||||
|
var l64branch21Table = map[string]uint32{
|
||||||
|
"BEQZ": 0x10 << 26, // beq r0, rj → beqz
|
||||||
|
"BNEZ": 0x11 << 26, // bne r0, rj → bnez
|
||||||
|
"BLTZ": 0x18 << 26, // blt rj, r0 → bltz
|
||||||
|
"BGEZ": 0x19 << 26, // bge rj, r0 → bgez
|
||||||
|
"BGTZ": 0x18 << 26, // blt r0, rj → bgtz
|
||||||
|
"BLEZ": 0x19 << 26, // bge r0, rj → blez
|
||||||
|
"BFPT": 0x12<<26 | 0x1<<8,
|
||||||
|
"BFPF": 0x12<<26 | 0x0<<8,
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64jumpTable maps the jump pseudo-instructions and their aliases to the
|
||||||
|
// B/BL opcode constants.
|
||||||
|
var l64jumpTable = map[string]uint32{
|
||||||
|
"JMP": 0x14 << 26, // b
|
||||||
|
"B": 0x14 << 26, // b
|
||||||
|
"JAL": 0x15 << 26, // bl
|
||||||
|
"CALL": 0x15 << 26, // bl
|
||||||
|
"BL": 0x15 << 26, // bl
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64loadStoreTable maps the MOV width mnemonics to their load and store
|
||||||
|
// 2RI12 opcodes. The load opcode is the negated store opcode, exactly as
|
||||||
|
// the toolchain derives it.
|
||||||
|
var l64loadStoreTable = map[string]struct{ ld, st uint32 }{
|
||||||
|
"MOVB": {0x0a0 << 22, 0x0a4 << 22},
|
||||||
|
"MOVH": {0x0a1 << 22, 0x0a5 << 22},
|
||||||
|
"MOVW": {0x0a2 << 22, 0x0a6 << 22},
|
||||||
|
"MOVV": {0x0a3 << 22, 0x0a7 << 22},
|
||||||
|
"MOVBU": {0x0a8 << 22, 0x0a4 << 22},
|
||||||
|
"MOVHU": {0x0a9 << 22, 0x0a5 << 22},
|
||||||
|
"MOVWU": {0x0aa << 22, 0x0a6 << 22},
|
||||||
|
"MOVF": {0x0ac << 22, 0x0ad << 22},
|
||||||
|
"MOVD": {0x0ae << 22, 0x0af << 22},
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64movRegTable maps a register-to-register MOV mnemonic to its expansion,
|
||||||
|
// matching the toolchain's case-1 encoding: MOVB → ext.w.b, MOVH → ext.w.h,
|
||||||
|
// MOVW → sll.w, MOVV → or, MOVBU → andi. MOVHU/MOVWU expand to bstrpick.d
|
||||||
|
// and are handled separately in the assembler.
|
||||||
|
type l64MovRegEnc struct {
|
||||||
|
rr bool // 2R format (ext.w.b/ext.w.h)
|
||||||
|
op uint32 // opcode constant (rr forms) or 3R/2RI12 opcode
|
||||||
|
imm int // 2RI12 immediate for MOVBU's andi
|
||||||
|
}
|
||||||
|
|
||||||
|
var l64movRegTable = map[string]l64MovRegEnc{
|
||||||
|
"MOVB": {true, 0x17 << 10, 0}, // ext.w.b rd, rj
|
||||||
|
"MOVH": {true, 0x16 << 10, 0}, // ext.w.h rd, rj
|
||||||
|
"MOVW": {false, 0x2e << 15, 0}, // sll.w rd, rj, r0
|
||||||
|
"MOVV": {false, 0x2a << 15, 0}, // or rd, rj, r0
|
||||||
|
"MOVBU": {false, 0x00d << 22, 0xff}, // andi rd, rj, $0xff
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64movFpRegTable maps a floating-point register move mnemonic to its 2R
|
||||||
|
// opcode (fmov.s / fmov.d), used when both operands are F registers.
|
||||||
|
var l64movFpRegTable = map[string]uint32{
|
||||||
|
"MOVF": 0x4525 << 10,
|
||||||
|
"MOVD": 0x4526 << 10,
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64RegClass discriminates integer (R), floating-point (F) and condition
|
||||||
|
// (FCC) registers for the MOV pseudo-instruction's register-move encoding.
|
||||||
|
type l64RegClass int
|
||||||
|
|
||||||
|
const (
|
||||||
|
l64ClsNone l64RegClass = iota
|
||||||
|
l64ClsGR
|
||||||
|
l64ClsFP
|
||||||
|
l64ClsFCC
|
||||||
|
l64ClsFCSR
|
||||||
|
)
|
||||||
|
|
||||||
|
// loong64RegClass reports the register class of a register operand name.
|
||||||
|
func loong64RegClass(name string) l64RegClass {
|
||||||
|
switch {
|
||||||
|
case name == "":
|
||||||
|
return l64ClsNone
|
||||||
|
case len(name) >= 3 && name[:3] == "FCC":
|
||||||
|
return l64ClsFCC
|
||||||
|
case len(name) >= 4 && name[:4] == "FCSR":
|
||||||
|
return l64ClsFCSR
|
||||||
|
case name[0] == 'F':
|
||||||
|
return l64ClsFP
|
||||||
|
default:
|
||||||
|
return l64ClsGR
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,293 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/binary"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// firstTextLOONG64 parses assembly source and returns the first TEXT body.
|
||||||
|
func firstTextLOONG64(t *testing.T, src string) *ast.Text {
|
||||||
|
t.Helper()
|
||||||
|
f, errs := parser.Parse("f_loong64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
for _, d := range f.Decls {
|
||||||
|
if fn, ok := d.(*ast.Text); ok {
|
||||||
|
return fn
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatal("no TEXT found")
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// assembleLOONG64Helper assembles one TEXT function and returns its bytes.
|
||||||
|
func assembleLOONG64Helper(t *testing.T, fn *ast.Text) []byte {
|
||||||
|
t.Helper()
|
||||||
|
code, _, _, _, _, err := assembleLOONG64(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
return code
|
||||||
|
}
|
||||||
|
|
||||||
|
// wantWords checks that code matches the expected little-endian words.
|
||||||
|
func wantWords(t *testing.T, code []byte, want ...uint32) {
|
||||||
|
t.Helper()
|
||||||
|
got := make([]uint32, 0, len(code)/4)
|
||||||
|
for i := 0; i+4 <= len(code); i += 4 {
|
||||||
|
got = append(got, binary.LittleEndian.Uint32(code[i:]))
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLOONG64_add(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOVV a+0(FP), R4
|
||||||
|
MOVV b+8(FP), R5
|
||||||
|
ADDV R5, R4, R4
|
||||||
|
MOVV R4, ret+16(FP)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
// 5 instructions: two ld.d, add.d, st.d, jirl r0, r1, 0.
|
||||||
|
wantWords(t, code,
|
||||||
|
0x28C02064, // ld.d r4, 8(r3)
|
||||||
|
0x28C04065, // ld.d r5, 16(r3)
|
||||||
|
0x00109484, // add.d r4, r4, r5
|
||||||
|
0x29C06064, // st.d r4, 24(r3)
|
||||||
|
0x4C000020, // jirl r0, r1, 0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLOONG64_arithmetic(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·arith(SB), NOSPLIT, $0
|
||||||
|
ADDV R4, R5, R6
|
||||||
|
SUBV R7, R8, R9
|
||||||
|
MULV R10, R11, R12
|
||||||
|
DIVV R13, R14, R15
|
||||||
|
AND R16, R17, R18
|
||||||
|
OR R18, R19, R20
|
||||||
|
XOR R20, R21, R2
|
||||||
|
SLLV R2, R23, R24
|
||||||
|
SRLV R24, R25, R26
|
||||||
|
SRAV R26, R27, R28
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x001090A6, // add.d r6, r5, r4
|
||||||
|
0x00119D09, // sub.d r9, r8, r7
|
||||||
|
0x001DA96C, // mul.d r12, r11, r10
|
||||||
|
0x002235CF, // div.d r15, r14, r13
|
||||||
|
0x0014C232, // and r18, r17, r16
|
||||||
|
0x00154A74, // or r20, r19, r18
|
||||||
|
0x0015D2A2, // xor r2, r21, r20
|
||||||
|
0x00188AF8, // sll.d r24, r23, r2
|
||||||
|
0x0019633A, // srl.d r26, r25, r24
|
||||||
|
0x0019EB7C, // sra.d r28, r27, r26
|
||||||
|
0x4C000020, // jirl r0, r1, 0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLOONG64_immediates(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·imm(SB), NOSPLIT, $0
|
||||||
|
ADDV $42, R4, R5
|
||||||
|
ADDV $-8, R6
|
||||||
|
AND $0xff, R7, R8
|
||||||
|
OR $1, R9, R10
|
||||||
|
SGT $100, R13, R14
|
||||||
|
SLLV $4, R15, R16
|
||||||
|
MOVV $0x12345, R17
|
||||||
|
MOVV $0, R18
|
||||||
|
MOVW $0, R19
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x02C0A885, // addi.d r5, r4, 42
|
||||||
|
0x02FFE0C6, // addi.d r6, r6, -8
|
||||||
|
0x0343FCE8, // andi r8, r7, 0xff
|
||||||
|
0x0380052A, // ori r10, r9, 1
|
||||||
|
0x020191AE, // slti r14, r13, 100
|
||||||
|
0x004111F0, // slli.d r16, r15, 4
|
||||||
|
0x14000251, // lu12i.w r17, 0x12
|
||||||
|
0x038D1631, // ori r17, r17, 0x345
|
||||||
|
0x00150012, // or r18, r0, r0
|
||||||
|
0x00170013, // sll.w r19, r0, r0
|
||||||
|
0x4C000020, // jirl r0, r1, 0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLOONG64_loadStore(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·mem(SB), NOSPLIT, $0
|
||||||
|
MOVV (R4), R5
|
||||||
|
MOVV R5, (R6)
|
||||||
|
MOVW 8(R7), R8
|
||||||
|
MOVB R9, -4(R10)
|
||||||
|
MOVV (R11)(R12), R13
|
||||||
|
MOVV R14, (R15)(R16)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x28C00085, // ld.d r5, 0(r4)
|
||||||
|
0x29C000C5, // st.d r5, 0(r6)
|
||||||
|
0x288020E8, // ld.w r8, 8(r7)
|
||||||
|
0x293FF149, // st.b r9, -4(r10)
|
||||||
|
0x380C316D, // ldx.d r13, r11, r12
|
||||||
|
0x381C41EE, // stx.d r14, r15, r16
|
||||||
|
0x4C000020, // jirl r0, r1, 0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLOONG64_branches(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·br(SB), NOSPLIT, $0
|
||||||
|
BEQ R4, R5, done
|
||||||
|
BNE R6, R7, skip
|
||||||
|
BLT R8, R9, done
|
||||||
|
BGE R10, R11, done
|
||||||
|
BLTU R12, R13, done
|
||||||
|
BGEU R14, R15, done
|
||||||
|
skip:
|
||||||
|
JMP done
|
||||||
|
done:
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
// skip is at 0x18 (6 words), done at 0x1c.
|
||||||
|
wantWords(t, code,
|
||||||
|
0x58001C85, // beq r5, r4, +7
|
||||||
|
0x5C0018C7, // bne r7, r6, +6
|
||||||
|
0x60001509, // blt r9, r8, +5
|
||||||
|
0x6400114B, // bge r11, r10, +4
|
||||||
|
0x68000D8D, // bltu r13, r12, +3
|
||||||
|
0x6C0009CF, // bgeu r15, r14, +2
|
||||||
|
0x50000400, // b done (+1, chain-folded through skip)
|
||||||
|
0x4C000020, // jirl r0, r1, 0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLOONG64_frame(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·f(SB), NOSPLIT, $32-8
|
||||||
|
MOVV R4, R5
|
||||||
|
MOVV arg+0(FP), R6
|
||||||
|
MOVV R7, local-8(SP)
|
||||||
|
MOVV local-8(SP), R8
|
||||||
|
MOVV R9, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
// autosize = align8(32+8) = 40; prologue stores LR at -40(SP),
|
||||||
|
// opens the frame, stores LR again at 0(SP). The function is a leaf
|
||||||
|
// (no calls), so the epilogue skips the LR restore. FP args are at
|
||||||
|
// autosize+8; SP locals at autosize+offset.
|
||||||
|
wantWords(t, code,
|
||||||
|
0x29FF6061, // st.d r1, -40(r3)
|
||||||
|
0x02FF6063, // addi.d r3, r3, -40
|
||||||
|
0x29C00061, // st.d r1, 0(r3)
|
||||||
|
0x00150085, // or r5, r4, r0
|
||||||
|
0x28C0C066, // ld.d r6, 48(r3) arg+0(FP) → 0+40+8
|
||||||
|
0x29C08067, // st.d r7, 32(r3) local-8(SP) → 40-8
|
||||||
|
0x28C08068, // ld.d r8, 32(r3)
|
||||||
|
0x29C0C069, // st.d r9, 48(r3) ret+0(FP) → 0+40+8
|
||||||
|
0x02C0A063, // addi.d r3, r3, 40
|
||||||
|
0x4C000020, // jirl r0, r1, 0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLOONG64_jumpChain(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·jc(SB), NOSPLIT, $0
|
||||||
|
JMP a
|
||||||
|
a:
|
||||||
|
JMP b
|
||||||
|
b:
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x50000800, // b +2 (a, chain-folded to b)
|
||||||
|
0x50000400, // b +1 (b)
|
||||||
|
0x4C000020, // jirl r0, r1, 0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLOONG64_dconClasses(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
v int64
|
||||||
|
word int // expected word count
|
||||||
|
}{
|
||||||
|
{0x123456789, 3}, // lu12i.w + ori + lu32i.d
|
||||||
|
{-1, 2}, // addi.d + lu52i.d (the MOV path handles -1 earlier)
|
||||||
|
{0x1000000000000, 2}, // addi.w + lu32i.d
|
||||||
|
{0x123456789abcdef0, 4}, // full sequence
|
||||||
|
{0xFFFFFFFFF, 2}, // lu12i.w + ori
|
||||||
|
{0x1234567800000000, 3}, // addi.w + lu32i.d + lu52i.d
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
if n := len(l64DconMovWords(0, c.v)); n != c.word {
|
||||||
|
t.Errorf("0x%x: %d words, want %d", c.v, n, c.word)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLOONG64_regNames(t *testing.T) {
|
||||||
|
cases := map[string]int{
|
||||||
|
"R0": 0, "R31": 31, "F0": 0, "F31": 31, "FCC0": 0, "FCC7": 7,
|
||||||
|
"FCSR0": 0, "FCSR3": 3, "ZERO": 0, "RA": 1, "SP": 3, "g": 22, "G": 22,
|
||||||
|
"R32": -1, "FCC8": -1, "X0": -1, "R": -1, "TMP": 30, "CTXT": 29,
|
||||||
|
}
|
||||||
|
for name, want := range cases {
|
||||||
|
if got := loong64RegNum(name); got != want {
|
||||||
|
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
|
||||||
|
// A spot-check that assembleLOONG64 emits the same bytes the Go
|
||||||
|
// toolchain does for a small kernel (the full comparison lives in
|
||||||
|
// verify's TestGroundTruthLOONG64).
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·k(SB), NOSPLIT, $0-0
|
||||||
|
ADDV R4, R5, R6
|
||||||
|
MOVV $0x100000, R7
|
||||||
|
BEQ R6, R7, done
|
||||||
|
JMP done
|
||||||
|
done:
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
fn := firstTextLOONG64(t, src)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
want := []byte{
|
||||||
|
0xa6, 0x90, 0x10, 0x00, // add.d r6, r5, r4
|
||||||
|
0x07, 0x20, 0x00, 0x14, // lu12i.w r7, 0x100
|
||||||
|
0xc7, 0x08, 0x00, 0x58, // beq r7, r6, +2 (done)
|
||||||
|
0x00, 0x04, 0x00, 0x50, // b +1 (done)
|
||||||
|
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
|
||||||
|
}
|
||||||
|
if !bytes.Equal(code, want) {
|
||||||
|
t.Errorf("code = % x\nwant % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,129 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Loong64 frame mapping, matching the Go toolchain's loong64 backend.
|
||||||
|
//
|
||||||
|
// Go's loong64 functions have no frame pointer: FP and SP are synthetic
|
||||||
|
// registers resolved against the hardware stack pointer (R3) and the frame
|
||||||
|
// size. The return address lives in R1 (the link register).
|
||||||
|
//
|
||||||
|
// The autosize is the real stack adjustment: the declared local frame plus
|
||||||
|
// the 8 bytes for the saved link register, rounded up to a multiple of 8
|
||||||
|
// (the toolchain aligns frames with `if autosize&4 != 0 { autosize += 4 }`).
|
||||||
|
// A leaf function (no calls) with a zero frame gets no prologue at all.
|
||||||
|
//
|
||||||
|
// Prologue (autosize > 0), byte-identical to the toolchain:
|
||||||
|
//
|
||||||
|
// MOVV R1, -autosize(R3) // save LR below the new SP (traceback-safe)
|
||||||
|
// ADDV $-autosize, R3 // open the frame
|
||||||
|
// MOVV R1, 0(R3) // save LR again at SP (signal-safety)
|
||||||
|
//
|
||||||
|
// Epilogue: MOVV 0(R3), R1; ADDV $autosize, R3 (non-leaf only for the LR
|
||||||
|
// restore); the RET's jirl r0, r1, 0 follows.
|
||||||
|
|
||||||
|
// loong64FrameInfo holds the frame layout derived from a TEXT directive.
|
||||||
|
type loong64FrameInfo struct {
|
||||||
|
autosize int // the real SP adjustment (locals + saved LR, aligned)
|
||||||
|
frame int // the declared $framesize
|
||||||
|
args int // the declared -argsize
|
||||||
|
noSplit bool // the NOSPLIT flag
|
||||||
|
leaf bool // no call instructions in the body
|
||||||
|
}
|
||||||
|
|
||||||
|
// loong64ComputeFrame derives the frame layout for a TEXT function.
|
||||||
|
func loong64ComputeFrame(t *ast.Text) loong64FrameInfo {
|
||||||
|
fi := loong64FrameInfo{
|
||||||
|
frame: frameSize(t),
|
||||||
|
args: argsSize(t),
|
||||||
|
}
|
||||||
|
for _, f := range t.Flags {
|
||||||
|
if f == "NOSPLIT" {
|
||||||
|
fi.noSplit = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
fi.leaf = loong64IsLeaf(t)
|
||||||
|
if fi.frame != 0 {
|
||||||
|
fi.autosize = fi.frame + 8 // space for the saved LR
|
||||||
|
if fi.autosize&4 != 0 {
|
||||||
|
fi.autosize += 4
|
||||||
|
}
|
||||||
|
} else if !fi.leaf {
|
||||||
|
// A zero-frame non-leaf function still opens an 8-byte frame for LR.
|
||||||
|
fi.autosize = 8
|
||||||
|
}
|
||||||
|
return fi
|
||||||
|
}
|
||||||
|
|
||||||
|
// loong64IsLeaf reports whether a function contains no call instructions
|
||||||
|
// (JAL/BL/CALL), matching the toolchain's LEAF mark, which drives the frame
|
||||||
|
// and the epilogue shape.
|
||||||
|
func loong64IsLeaf(t *ast.Text) bool {
|
||||||
|
for _, stmt := range t.Body {
|
||||||
|
in, ok := stmt.(*ast.Instr)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
switch strings.ToUpper(in.Mnemonic.Text) {
|
||||||
|
case "JAL", "CALL", "BL":
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
// loong64Prologue returns the prologue bytes for a loong64 function.
|
||||||
|
func loong64Prologue(fi loong64FrameInfo) []byte {
|
||||||
|
if fi.autosize == 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
addiD := l64DualTable["ADDV"].imm
|
||||||
|
return l64WordsLE(
|
||||||
|
l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, 3, 1), // MOVV R1, -autosize(R3)
|
||||||
|
l64irr(addiD, -fi.autosize, 3, 3), // ADDV $-autosize, R3
|
||||||
|
l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1), // MOVV R1, 0(R3)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// loong64Return returns the bytes for a RET: the epilogue (restore LR and
|
||||||
|
// deallocate the frame when present) followed by jirl r0, r1, 0.
|
||||||
|
func loong64Return(fi loong64FrameInfo) []byte {
|
||||||
|
var ws []uint32
|
||||||
|
if fi.autosize != 0 {
|
||||||
|
if !fi.leaf {
|
||||||
|
// MOVV 0(R3), R1 — restore the link register.
|
||||||
|
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1))
|
||||||
|
}
|
||||||
|
// ADDV $autosize, R3 — close the frame.
|
||||||
|
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
|
||||||
|
}
|
||||||
|
// jirl r0, r1, 0 — return.
|
||||||
|
ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0))
|
||||||
|
return l64WordsLE(ws...)
|
||||||
|
}
|
||||||
|
|
||||||
|
// loong64ResolvePseudo translates a pseudo-register memory reference into a
|
||||||
|
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
|
||||||
|
// x-N(SP) → (autosize - N)(SP). Returns base = -1 for an unresolvable
|
||||||
|
// reference (SB: static data, handled by the relocation path).
|
||||||
|
func loong64ResolvePseudo(sym *ast.Symbol, fi loong64FrameInfo) (base int, off int32) {
|
||||||
|
if sym == nil {
|
||||||
|
return -1, 0
|
||||||
|
}
|
||||||
|
switch sym.Pseudo {
|
||||||
|
case "FP":
|
||||||
|
return 3, int32(sym.Offset) + int32(fi.autosize) + 8
|
||||||
|
case "SP":
|
||||||
|
return 3, int32(fi.autosize) + int32(sym.Offset)
|
||||||
|
case "SB":
|
||||||
|
return -1, int32(sym.Offset)
|
||||||
|
}
|
||||||
|
return -1, 0
|
||||||
|
}
|
||||||
@@ -0,0 +1,367 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestLOONG64_sys exercises the no-operand system instructions and the
|
||||||
|
// bare-data pseudo-instructions. The words match `go tool asm`
|
||||||
|
// (GOARCH=loong64) for the same source.
|
||||||
|
func TestLOONG64_sys(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·sys(SB), NOSPLIT, $0
|
||||||
|
NOOP
|
||||||
|
UNDEF
|
||||||
|
WORD $0x12345678
|
||||||
|
SYSCALL $0x10
|
||||||
|
BREAK $0x20
|
||||||
|
DBAR $1
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x03400000, // andi r0, r0, 0 (NOOP)
|
||||||
|
0x002A0000, // break 0 (UNDEF)
|
||||||
|
0x12345678, // WORD
|
||||||
|
0x002B0010, // syscall 0x10
|
||||||
|
0x002A0020, // break 0x20
|
||||||
|
0x38720001, // dbar 1
|
||||||
|
0x4C000020, // jirl r0, r1, 0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_branches21 exercises the single-register branch forms: the
|
||||||
|
// 21-bit BEQZ/BNEZ/BLTZ/BGEZ and the rd-field BGTZ/BLEZ.
|
||||||
|
func TestLOONG64_branches21(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·b21(SB), NOSPLIT, $0
|
||||||
|
BEQZ R4, done
|
||||||
|
BNEZ R5, done
|
||||||
|
BLTZ R6, done
|
||||||
|
BGEZ R7, done
|
||||||
|
BGTZ R8, done
|
||||||
|
BLEZ R9, done
|
||||||
|
done:
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x40001880, // beqz r4, +6
|
||||||
|
0x440014A0, // bnez r5, +5
|
||||||
|
0x600010C0, // bltz r6, +4
|
||||||
|
0x64000CE0, // bgez r7, +3
|
||||||
|
0x60000808, // bgtz r8, +2 (register in the rd field)
|
||||||
|
0x64000409, // blez r9, +1
|
||||||
|
0x4C000020, // jirl r0, r1, 0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_fma exercises the four fused multiply-add forms (4 and 3
|
||||||
|
// operand spellings).
|
||||||
|
func TestLOONG64_fma(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·fma(SB), NOSPLIT, $0
|
||||||
|
FMADDD F0, F1, F2, F3
|
||||||
|
FMSUBD F4, F5, F6
|
||||||
|
FNMADDD F7, F8, F9, F10
|
||||||
|
FNMSUBD F11, F12, F13
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x08200443, // fmadd.d f3, f2, f1, f0
|
||||||
|
0x086214C6, // fmsub.d f6, f5, f5, f4
|
||||||
|
0x08A3A12A, // fnmadd.d f10, f9, f8, f7
|
||||||
|
0x08E5B1AD, // fnmsub.d f13, f12, f12, f11
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_bitops exercises BSTRINS/BSTRPICK (the 6-bit msb/lsb fields)
|
||||||
|
// and ALSL (the sa−1 shift field).
|
||||||
|
func TestLOONG64_bitops(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·bits(SB), NOSPLIT, $0
|
||||||
|
BSTRINSW $3, R4, $0, R5
|
||||||
|
BSTRINSV $3, R4, $1, R6
|
||||||
|
BSTRPICKW $3, R4, $0, R5
|
||||||
|
BSTRPICKV $6, R7, $0, R8
|
||||||
|
ALSLW $1, R4, R5, R6
|
||||||
|
ALSLW $4, R7, R8, R9
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x00630085, // bstrins.w r5, r4, $3, $0
|
||||||
|
0x00830486, // bstrins.d r6, r4, $3, $1
|
||||||
|
0x00638085, // bstrpick.w r5, r4, $3, $0
|
||||||
|
0x00C600E8, // bstrpick.d r8, r7, $6, $0
|
||||||
|
0x00041486, // alsl.w r6, r5, r4, $1 (sa-1)
|
||||||
|
0x0005A0E9, // alsl.w r9, r8, r7, $4
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_ptr exercises the 14-bit-offset memory forms (LL/SC/MOVWP/
|
||||||
|
// MOVVP with the offset scaled by 4) and PRELD.
|
||||||
|
func TestLOONG64_ptr(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·ptr(SB), NOSPLIT, $0
|
||||||
|
LLW 8(R14), R15
|
||||||
|
SCW R16, -4(R17)
|
||||||
|
MOVWP 16(R18), R19
|
||||||
|
MOVVP R20, 24(R21)
|
||||||
|
PRELD 32(R22), $0
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x200009CF, // ll.w r15, 8(r14)
|
||||||
|
0x21FFFE30, // sc.w r16, -4(r17)
|
||||||
|
0x24001253, // ldptr.w r19, 16(r18)
|
||||||
|
0x27001AB4, // stptr.d r20, 24(r21)
|
||||||
|
0x2AC082C0, // preld 32(r22), 0
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_atomics exercises the AM* read-modify-write forms and
|
||||||
|
// RDTIME, plus the MOVV FP→GP move.
|
||||||
|
func TestLOONG64_atomics(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·atoms(SB), NOSPLIT, $0
|
||||||
|
AMADDW R4, (R5), R6
|
||||||
|
RDTIMED R7, R8
|
||||||
|
MOVV F1, R2
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x386110A6, // amadd.w r6, r5, r4
|
||||||
|
0x000068E8, // rdtime.d r8, r7
|
||||||
|
0x0114B822, // movfr2gr.d r2, f1
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_lu52 exercises the LU52I.D immediate form (a gasm extension
|
||||||
|
// the toolchain reaches only through its MOVV expansion).
|
||||||
|
func TestLOONG64_lu52(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·lu52(SB), NOSPLIT, $0
|
||||||
|
LU52ID $0x345, R10
|
||||||
|
LU52ID $0x123, R11, R12
|
||||||
|
ADDV16 $0x10000, R13
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x030D154A, // lu52i.d r10, r10, 0x345
|
||||||
|
0x03048D6C, // lu52i.d r12, r11, 0x123
|
||||||
|
0x100005AD, // addu16i.d r13, r13, 0x10000>>16
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_sbRefs checks the static-symbol reference forms through the
|
||||||
|
// full file assembly: each pcalau12i+addi.d/ld/st pair carries the
|
||||||
|
// R_LOONG64_ADDR_HI/LO relocation pair, and the immediate fields are left
|
||||||
|
// zero for the linker.
|
||||||
|
func TestLOONG64_sbRefs(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("sb_loong64.s", `#include "textflag.h"
|
||||||
|
TEXT ·sb(SB), NOSPLIT, $0
|
||||||
|
MOVV $·table(SB), R4
|
||||||
|
MOVV ·table+8(SB), R5
|
||||||
|
MOVV R6, ·table(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
GLOBL ·table(SB), RODATA, $8
|
||||||
|
DATA ·table+0(SB)/8, $42
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileLOONG64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||||
|
}
|
||||||
|
fn := img.Funcs[0]
|
||||||
|
if fn.Size != 28 {
|
||||||
|
t.Fatalf("function size = %d, want 28", fn.Size)
|
||||||
|
}
|
||||||
|
var hi, lo int
|
||||||
|
// The three references: $·table (0), ·table+8 (8), ·table (0).
|
||||||
|
wantAdd := []int64{0, 0, 8, 8, 0, 0}
|
||||||
|
for i, r := range fn.Relocs {
|
||||||
|
wantKind := RelLoong64AddrHi
|
||||||
|
wantOff := (i / 2) * 8
|
||||||
|
if i%2 == 1 {
|
||||||
|
wantKind = RelLoong64AddrLo
|
||||||
|
wantOff += 4
|
||||||
|
}
|
||||||
|
if r.Kind != wantKind || r.Off != wantOff || r.Name != "table" || r.Addend != wantAdd[i] {
|
||||||
|
t.Errorf("reloc %d = {kind %v off %d name %q addend %d}", i, r.Kind, r.Off, r.Name, r.Addend)
|
||||||
|
}
|
||||||
|
if r.Kind == RelLoong64AddrHi {
|
||||||
|
hi++
|
||||||
|
} else {
|
||||||
|
lo++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if hi != 3 || lo != 3 {
|
||||||
|
t.Errorf("relocs = %d hi + %d lo, want 3 + 3", hi, lo)
|
||||||
|
}
|
||||||
|
// The image carries the zero-immediate pair encodings (the linker
|
||||||
|
// fills the immediate fields from the relocations).
|
||||||
|
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||||
|
wantWords(t, code,
|
||||||
|
0x1A000004, // pcalau12i r4, 0
|
||||||
|
0x02C00084, // addi.d r4, r4, 0
|
||||||
|
0x1A00001E, // pcalau12i r30, 0
|
||||||
|
0x28C003C5, // ld.d r5, 0(r30)
|
||||||
|
0x1A00001E, // pcalau12i r30, 0
|
||||||
|
0x29C003C6, // st.d r6, 0(r30)
|
||||||
|
0x4C000020, // jirl r0, r1, 0
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
|
||||||
|
// invalid register operands and operand-count mismatches.
|
||||||
|
func TestLOONG64_errors(t *testing.T) {
|
||||||
|
cases := []string{
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
JMP nowhere
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
BEQZ X0, done
|
||||||
|
done:
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
ADDV R4
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
FMADDD F0, F1
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
AMADDW R4, R5
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
WORD
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
PRELD 32(R4)
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
ALSLW $5, R4, R5, R6
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
}
|
||||||
|
for i, src := range cases {
|
||||||
|
fn := firstTextLOONG64(t, src)
|
||||||
|
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||||
|
t.Errorf("case %d: expected an error, got none", i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_pcsp checks the stack-adjustment table of a framed function:
|
||||||
|
// the prologue raises the SP delta by autosize (in effect from the third
|
||||||
|
// instruction) and the RET's epilogue restores it to zero, with the pc deltas
|
||||||
|
// in MinLC (4) units — byte-identical to `go tool asm`.
|
||||||
|
func TestLOONG64_pcsp(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
src string
|
||||||
|
want []byte
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
"leaf",
|
||||||
|
`#include "textflag.h"
|
||||||
|
TEXT ·leaf(SB), NOSPLIT, $8-0
|
||||||
|
MOVV R4, R5
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
[]byte{0x02, 0x02, 0x20, 0x03, 0x1f, 0x01, 0x00},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"nonleaf",
|
||||||
|
`#include "textflag.h"
|
||||||
|
TEXT ·nonleaf(SB), NOSPLIT, $8-0
|
||||||
|
MOVV R4, R5
|
||||||
|
JAL (R12)
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
[]byte{0x02, 0x02, 0x20, 0x05, 0x1f, 0x01, 0x00},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("pcsp_loong64.s", c.src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileLOONG64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||||
|
}
|
||||||
|
if got := pcspTable(img.Funcs[0], 4); !bytes.Equal(got, c.want) {
|
||||||
|
t.Errorf("pcsp = % x, want % x", got, c.want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_sbRefsUndefined checks that a reference to a symbol no GLOBL
|
||||||
|
// defines assembles into a relocation and is rejected at object emission.
|
||||||
|
func TestLOONG64_sbRefsUndefined(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("sb_loong64.s", `#include "textflag.h"
|
||||||
|
TEXT ·sb(SB), NOSPLIT, $0
|
||||||
|
MOVV missing(SB), R4
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileLOONG64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||||
|
}
|
||||||
|
if len(img.Funcs[0].Relocs) != 2 {
|
||||||
|
t.Fatalf("relocs = %d, want the HI/LO pair", len(img.Funcs[0].Relocs))
|
||||||
|
}
|
||||||
|
if _, err := img.GOObjectLOONG64("p", "sb_loong64.s"); err == nil {
|
||||||
|
t.Error("expected an unknown-symbol error at emission")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_movImmToFp checks the immediate-to-FP move forms.
|
||||||
|
func TestLOONG64_movImmToFp(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·fpmov(SB), NOSPLIT, $0
|
||||||
|
MOVV $0x1, F0
|
||||||
|
MOVW $0x2, F4
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
want := []byte{
|
||||||
|
0x00, 0x04, 0x80, 0x03, // ori f0, r0, 1
|
||||||
|
0x04, 0x08, 0x80, 0x03, // ori f4, r0, 2
|
||||||
|
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
|
||||||
|
}
|
||||||
|
if !bytes.Equal(code, want) {
|
||||||
|
t.Errorf("code = % x\nwant % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
-258
@@ -1,258 +0,0 @@
|
|||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
||||||
// SPDX-License-Identifier: BSD-3-Clause
|
|
||||||
|
|
||||||
package asm
|
|
||||||
|
|
||||||
import (
|
|
||||||
"encoding/binary"
|
|
||||||
"fmt"
|
|
||||||
)
|
|
||||||
|
|
||||||
// This file emits Mach-O x86-64 objects (MH_OBJECT) from an assembled
|
|
||||||
// Image, in the shape the Darwin assembler produces: one unnamed segment
|
|
||||||
// carrying a __TEXT,__text and a __DATA,__data section laid out back to
|
|
||||||
// back at addresses zero and len(code), a symbol table (locals first, then
|
|
||||||
// exported definitions, then undefined externals) and one relocation entry
|
|
||||||
// per static-symbol reference, of type X86_64_RELOC_SIGNED.
|
|
||||||
//
|
|
||||||
// The image's own address space carries straight over — the data section
|
|
||||||
// starts immediately after the code, and the layout padding already lives
|
|
||||||
// inside Image.Data — so every symbol keeps its image address as its
|
|
||||||
// n_value, and a local (non-external) relocation leaves the displacement
|
|
||||||
// the assembler resolved in place: the linker only adjusts it by the
|
|
||||||
// section's final movement.
|
|
||||||
|
|
||||||
// Mach-O constants.
|
|
||||||
const (
|
|
||||||
machoMagic64 = 0xfeedfacf
|
|
||||||
machoCPUamd64 = 0x01000007 // CPU_TYPE_X86_64
|
|
||||||
machoCPUSubAll = 3 // CPU_SUBTYPE_X86_64_ALL
|
|
||||||
machoObj = 1 // MH_OBJECT
|
|
||||||
|
|
||||||
machoSegment64 = 0x19 // LC_SEGMENT_64
|
|
||||||
machoSymtab = 0x2 // LC_SYMTAB
|
|
||||||
|
|
||||||
machoSectTextFlags = 0x80000400 // S_ATTR_PURE_INSTRUCTIONS | S_ATTR_SOME_INSTRUCTIONS
|
|
||||||
|
|
||||||
nUndf = 0x00 // undefined symbol
|
|
||||||
nSect = 0x0e // defined in section number n_sect
|
|
||||||
nExt = 0x01 // external (exported or undefined-global) bit
|
|
||||||
|
|
||||||
x8664RelocSigned = 1
|
|
||||||
)
|
|
||||||
|
|
||||||
// MachOObject returns the image as a Mach-O x86-64 relocatable object
|
|
||||||
// (MH_OBJECT), the shape the Darwin toolchain links. Symbol names follow
|
|
||||||
// the same rules as the ELF output. Every static-symbol reference becomes
|
|
||||||
// an X86_64_RELOC_SIGNED relocation: external references against their
|
|
||||||
// undefined symbol, file-local ones against the __DATA section with the
|
|
||||||
// resolved displacement carried in the instruction bytes.
|
|
||||||
func (img *Image) MachOObject() ([]byte, error) {
|
|
||||||
le := binary.LittleEndian
|
|
||||||
|
|
||||||
// Section ordinals (1-based, as Mach-O numbers them).
|
|
||||||
const (
|
|
||||||
sectText = 1
|
|
||||||
sectData = 2
|
|
||||||
)
|
|
||||||
|
|
||||||
// Object address space: code at 0, data immediately after (the layout
|
|
||||||
// padding is already part of img.Data, so image addresses are object
|
|
||||||
// addresses).
|
|
||||||
textAddr := uint64(0)
|
|
||||||
dataAddr := uint64(len(img.Code))
|
|
||||||
vmsize := dataAddr + uint64(len(img.Data))
|
|
||||||
|
|
||||||
// The code, with external displacements primed to addend − 4: the
|
|
||||||
// linker adds the symbol's address to the field as it stands. Local
|
|
||||||
// displacements stay as the assembler resolved them.
|
|
||||||
code := append([]byte(nil), img.Code...)
|
|
||||||
for _, fn := range img.Funcs {
|
|
||||||
for _, r := range fn.Relocs {
|
|
||||||
if r.External {
|
|
||||||
// Prime the field to the addend measured from the patch
|
|
||||||
// site: the assembler records it from the instruction end,
|
|
||||||
// After − Off bytes past the field.
|
|
||||||
copy(code[fn.Offset+r.Off:], le32(r.Addend-int64(r.After-r.Off)))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Symbols: locals first, then exported definitions, then undefined
|
|
||||||
// externals — the order the classic link editor expects.
|
|
||||||
type machoSym struct {
|
|
||||||
name string
|
|
||||||
typ byte
|
|
||||||
sect byte
|
|
||||||
value uint64
|
|
||||||
}
|
|
||||||
var locals, globals, undefs []machoSym
|
|
||||||
for _, fn := range img.Funcs {
|
|
||||||
s := machoSym{name: objectName(fn.Pkg, fn.Name), typ: nSect, sect: sectText, value: textAddr + uint64(fn.Offset)}
|
|
||||||
if fn.Static {
|
|
||||||
locals = append(locals, s)
|
|
||||||
} else {
|
|
||||||
s.typ |= nExt
|
|
||||||
globals = append(globals, s)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for _, d := range img.DataSyms {
|
|
||||||
s := machoSym{name: objectName(d.Pkg, d.Name), typ: nSect, sect: sectData, value: dataAddr + uint64(d.Offset)}
|
|
||||||
if d.Static {
|
|
||||||
locals = append(locals, s)
|
|
||||||
} else {
|
|
||||||
s.typ |= nExt
|
|
||||||
globals = append(globals, s)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for _, name := range img.Externals {
|
|
||||||
undefs = append(undefs, machoSym{name: name, typ: nUndf | nExt})
|
|
||||||
}
|
|
||||||
syms := append(append(locals, globals...), undefs...)
|
|
||||||
symIdx := map[string]int{}
|
|
||||||
for i, s := range syms {
|
|
||||||
symIdx[s.name] = i
|
|
||||||
}
|
|
||||||
|
|
||||||
// Relocations, attached to the __text section.
|
|
||||||
type machoReloc struct {
|
|
||||||
addr uint32
|
|
||||||
symnum uint32
|
|
||||||
extern bool
|
|
||||||
}
|
|
||||||
var relocs []machoReloc
|
|
||||||
for _, fn := range img.Funcs {
|
|
||||||
for _, r := range fn.Relocs {
|
|
||||||
rel := machoReloc{addr: uint32(fn.Offset + r.Off)}
|
|
||||||
if r.External {
|
|
||||||
idx, ok := symIdx[r.Name]
|
|
||||||
if !ok {
|
|
||||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
|
||||||
}
|
|
||||||
rel.symnum = uint32(idx)
|
|
||||||
rel.extern = true
|
|
||||||
} else {
|
|
||||||
// Section-relative: r_symbolnum carries the section number
|
|
||||||
// and the resolved displacement stays in the bytes.
|
|
||||||
rel.symnum = sectData
|
|
||||||
}
|
|
||||||
relocs = append(relocs, rel)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// The string table opens with the conventional " \0".
|
|
||||||
strtab := []byte{' ', 0}
|
|
||||||
strOff := map[string]int{}
|
|
||||||
for _, s := range syms {
|
|
||||||
if _, ok := strOff[s.name]; ok {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
strOff[s.name] = len(strtab)
|
|
||||||
strtab = append(strtab, s.name...)
|
|
||||||
strtab = append(strtab, 0)
|
|
||||||
}
|
|
||||||
|
|
||||||
// File layout: header, the two load commands, section data (code,
|
|
||||||
// data), the relocation table, the symbol table, the string table.
|
|
||||||
const (
|
|
||||||
hdrSize = 32
|
|
||||||
segCmdSize = 72 + 2*80 // segment command with two sections
|
|
||||||
symCmdSize = 24
|
|
||||||
)
|
|
||||||
sizeofcmds := segCmdSize + symCmdSize
|
|
||||||
dataOff := hdrSize + sizeofcmds
|
|
||||||
reloff := dataOff + len(code) + len(img.Data)
|
|
||||||
symoff := reloff + 8*len(relocs)
|
|
||||||
stroff := symoff + 16*len(syms)
|
|
||||||
|
|
||||||
out := make([]byte, stroff+len(strtab))
|
|
||||||
|
|
||||||
// mach_header_64.
|
|
||||||
le.PutUint32(out[0:], machoMagic64)
|
|
||||||
le.PutUint32(out[4:], machoCPUamd64)
|
|
||||||
le.PutUint32(out[8:], machoCPUSubAll)
|
|
||||||
le.PutUint32(out[12:], machoObj)
|
|
||||||
le.PutUint32(out[16:], 2) // ncmds
|
|
||||||
le.PutUint32(out[20:], uint32(sizeofcmds))
|
|
||||||
le.PutUint32(out[24:], 0) // flags
|
|
||||||
le.PutUint32(out[28:], 0) // reserved
|
|
||||||
|
|
||||||
// LC_SEGMENT_64 with the two sections.
|
|
||||||
p := hdrSize
|
|
||||||
le.PutUint32(out[p:], machoSegment64)
|
|
||||||
le.PutUint32(out[p+4:], segCmdSize)
|
|
||||||
// segname: the empty string, zero-padded to 16 bytes.
|
|
||||||
le.PutUint64(out[p+8:], 0)
|
|
||||||
le.PutUint64(out[p+16:], 0)
|
|
||||||
le.PutUint64(out[p+24:], 0) // vmaddr
|
|
||||||
le.PutUint64(out[p+32:], vmsize)
|
|
||||||
le.PutUint64(out[p+40:], uint64(dataOff))
|
|
||||||
le.PutUint64(out[p+48:], vmsize)
|
|
||||||
le.PutUint32(out[p+56:], 7) // maxprot rwx
|
|
||||||
le.PutUint32(out[p+60:], 7) // initprot rwx
|
|
||||||
le.PutUint32(out[p+64:], 2) // nsects
|
|
||||||
le.PutUint32(out[p+68:], 0) // flags
|
|
||||||
|
|
||||||
// __TEXT,__text
|
|
||||||
s := p + 72
|
|
||||||
copy(out[s:], "__text")
|
|
||||||
copy(out[s+16:], "__TEXT")
|
|
||||||
le.PutUint64(out[s+32:], textAddr)
|
|
||||||
le.PutUint64(out[s+40:], uint64(len(code)))
|
|
||||||
le.PutUint32(out[s+48:], uint32(dataOff))
|
|
||||||
le.PutUint32(out[s+52:], 4) // align 2^4
|
|
||||||
le.PutUint32(out[s+56:], uint32(reloff))
|
|
||||||
le.PutUint32(out[s+60:], uint32(len(relocs)))
|
|
||||||
le.PutUint32(out[s+64:], machoSectTextFlags)
|
|
||||||
|
|
||||||
// __DATA,__data
|
|
||||||
s += 80
|
|
||||||
copy(out[s:], "__data")
|
|
||||||
copy(out[s+16:], "__DATA")
|
|
||||||
le.PutUint64(out[s+32:], dataAddr)
|
|
||||||
le.PutUint64(out[s+40:], uint64(len(img.Data)))
|
|
||||||
le.PutUint32(out[s+48:], uint32(dataOff+len(code)))
|
|
||||||
le.PutUint32(out[s+52:], 4) // align 2^4
|
|
||||||
|
|
||||||
// LC_SYMTAB.
|
|
||||||
p = hdrSize + segCmdSize
|
|
||||||
le.PutUint32(out[p:], machoSymtab)
|
|
||||||
le.PutUint32(out[p+4:], symCmdSize)
|
|
||||||
le.PutUint32(out[p+8:], uint32(symoff))
|
|
||||||
le.PutUint32(out[p+12:], uint32(len(syms)))
|
|
||||||
le.PutUint32(out[p+16:], uint32(stroff))
|
|
||||||
le.PutUint32(out[p+20:], uint32(len(strtab)))
|
|
||||||
|
|
||||||
// Section data.
|
|
||||||
copy(out[dataOff:], code)
|
|
||||||
copy(out[dataOff+len(code):], img.Data)
|
|
||||||
|
|
||||||
// Relocation entries.
|
|
||||||
for i, r := range relocs {
|
|
||||||
e := out[reloff+i*8:]
|
|
||||||
le.PutUint32(e[0:], r.addr)
|
|
||||||
bits := r.symnum & 0x00ffffff
|
|
||||||
bits |= 1 << 24 // r_pcrel
|
|
||||||
bits |= 2 << 25 // r_length = 4 bytes
|
|
||||||
if r.extern {
|
|
||||||
bits |= 1 << 27 // r_extern
|
|
||||||
}
|
|
||||||
bits |= x8664RelocSigned << 28
|
|
||||||
le.PutUint32(e[4:], bits)
|
|
||||||
}
|
|
||||||
|
|
||||||
// nlist_64 entries.
|
|
||||||
for i, s := range syms {
|
|
||||||
e := out[symoff+i*16:]
|
|
||||||
le.PutUint32(e[0:], uint32(strOff[s.name]))
|
|
||||||
e[4] = s.typ
|
|
||||||
e[5] = s.sect
|
|
||||||
le.PutUint16(e[6:], 0) // n_desc
|
|
||||||
le.PutUint64(e[8:], s.value)
|
|
||||||
}
|
|
||||||
|
|
||||||
// String table.
|
|
||||||
copy(out[stroff:], strtab)
|
|
||||||
|
|
||||||
return out, nil
|
|
||||||
}
|
|
||||||
@@ -1,127 +0,0 @@
|
|||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
||||||
// SPDX-License-Identifier: BSD-3-Clause
|
|
||||||
|
|
||||||
package asm
|
|
||||||
|
|
||||||
import (
|
|
||||||
"bytes"
|
|
||||||
"debug/macho"
|
|
||||||
"encoding/binary"
|
|
||||||
"testing"
|
|
||||||
)
|
|
||||||
|
|
||||||
// TestMachOObject checks the structure of the emitted MH_OBJECT: the two
|
|
||||||
// sections and their addresses, the symbol table (types, sections, values)
|
|
||||||
// and the __text relocation entries, parsed back with debug/macho. No
|
|
||||||
// Darwin toolchain is available on the test hosts, so the check is
|
|
||||||
// structural — the ELF output carries the end-to-end link-and-run proof of
|
|
||||||
// the shared symbol and relocation model.
|
|
||||||
func TestMachOObject(t *testing.T) {
|
|
||||||
img := elfTestImage(t)
|
|
||||||
obj, err := img.MachOObject()
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("MachOObject: %v", err)
|
|
||||||
}
|
|
||||||
f, err := macho.NewFile(bytes.NewReader(obj))
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("parse emitted object: %v", err)
|
|
||||||
}
|
|
||||||
defer f.Close()
|
|
||||||
|
|
||||||
if f.Type != macho.TypeObj {
|
|
||||||
t.Errorf("file type = %v, want MH_OBJECT", f.Type)
|
|
||||||
}
|
|
||||||
if f.Cpu != macho.CpuAmd64 {
|
|
||||||
t.Errorf("cpu = %v, want CpuAmd64", f.Cpu)
|
|
||||||
}
|
|
||||||
|
|
||||||
text := f.Section("__text")
|
|
||||||
data := f.Section("__data")
|
|
||||||
if text == nil || data == nil {
|
|
||||||
t.Fatal("missing __text or __data section")
|
|
||||||
}
|
|
||||||
if text.Addr != 0 || text.Size != uint64(len(img.Code)) {
|
|
||||||
t.Errorf("__text addr/size = %#x/%d, want 0/%d", text.Addr, text.Size, len(img.Code))
|
|
||||||
}
|
|
||||||
if data.Addr != uint64(len(img.Code)) {
|
|
||||||
t.Errorf("__data addr = %#x, want %#x", data.Addr, len(img.Code))
|
|
||||||
}
|
|
||||||
|
|
||||||
// Symbol table: locals, exported definitions, undefined externals.
|
|
||||||
syms := f.Symtab.Syms
|
|
||||||
byName := map[string]macho.Symbol{}
|
|
||||||
for _, s := range syms {
|
|
||||||
byName[s.Name] = s
|
|
||||||
}
|
|
||||||
wantSym := func(name string, typ, sect uint8, value uint64) {
|
|
||||||
t.Helper()
|
|
||||||
s, ok := byName[name]
|
|
||||||
if !ok {
|
|
||||||
t.Errorf("symbol %q not found", name)
|
|
||||||
return
|
|
||||||
}
|
|
||||||
if s.Type != typ || s.Sect != sect || s.Value != value {
|
|
||||||
t.Errorf("%s: type/sect/value = %#x/%d/%#x, want %#x/%d/%#x",
|
|
||||||
name, s.Type, s.Sect, s.Value, typ, sect, value)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const (
|
|
||||||
defined = nSect | nExt
|
|
||||||
local = nSect
|
|
||||||
undefined = nUndf | nExt
|
|
||||||
)
|
|
||||||
wantSym("addq", defined, 1, 0)
|
|
||||||
wantSym("getanswer", defined, 1, 5)
|
|
||||||
wantSym("useextern", defined, 1, 13)
|
|
||||||
answer := byName["answer"]
|
|
||||||
if answer.Type != local || answer.Sect != 2 {
|
|
||||||
t.Errorf("answer: type/sect = %#x/%d, want %#x/2", answer.Type, answer.Sect, local)
|
|
||||||
}
|
|
||||||
wantSym("extvar", undefined, 0, 0)
|
|
||||||
|
|
||||||
// Relocations: both X86_64_RELOC_SIGNED, PC-relative, 4 bytes wide.
|
|
||||||
// The local one carries its section number in Value, the external one
|
|
||||||
// its symbol number.
|
|
||||||
if len(text.Relocs) != 2 {
|
|
||||||
t.Fatalf("__text relocs = %d, want 2", len(text.Relocs))
|
|
||||||
}
|
|
||||||
var sawLocal, sawExternal bool
|
|
||||||
for _, r := range text.Relocs {
|
|
||||||
if !r.Pcrel || r.Len != 2 || r.Type != x8664RelocSigned {
|
|
||||||
t.Errorf("reloc at %#x: pcrel/len/type = %v/%d/%d", r.Addr, r.Pcrel, r.Len, r.Type)
|
|
||||||
}
|
|
||||||
switch {
|
|
||||||
case r.Extern:
|
|
||||||
if name := syms[r.Value].Name; name != "extvar" {
|
|
||||||
t.Errorf("external reloc at %#x names %q, want extvar", r.Addr, name)
|
|
||||||
}
|
|
||||||
sawExternal = true
|
|
||||||
default:
|
|
||||||
if r.Value != 2 { // __data, the second section
|
|
||||||
t.Errorf("local reloc at %#x: section %d, want 2 (__data)", r.Addr, r.Value)
|
|
||||||
}
|
|
||||||
sawLocal = true
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if !sawLocal || !sawExternal {
|
|
||||||
t.Errorf("relocs seen: local=%v external=%v, want both", sawLocal, sawExternal)
|
|
||||||
}
|
|
||||||
|
|
||||||
// The __text bytes are the image code, with the external displacement
|
|
||||||
// primed to addend − 4 and the local one left resolved.
|
|
||||||
textData, err := text.Data()
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
want := append([]byte(nil), img.Code...)
|
|
||||||
for _, fn := range img.Funcs {
|
|
||||||
for _, r := range fn.Relocs {
|
|
||||||
if r.Name == "extvar" {
|
|
||||||
binary.LittleEndian.PutUint32(want[fn.Offset+r.Off:], 0xfffffffc) // −4
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if !bytes.Equal(textData, want) {
|
|
||||||
t.Errorf("__text bytes %x, want %x", textData, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,564 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
// RISC-V register encoding: maps register names to their 5-bit numbers.
|
||||||
|
// The Go assembler uses the standard RISC-V ABI naming.
|
||||||
|
|
||||||
|
// riscvRegNum returns the 5-bit register number for a RISC-V register name.
|
||||||
|
// Returns -1 if the register is not recognized.
|
||||||
|
func riscvRegNum(name string) int {
|
||||||
|
switch name {
|
||||||
|
// Numbered integer registers.
|
||||||
|
case "X0", "ZERO":
|
||||||
|
return 0
|
||||||
|
case "X1", "RA", "LR":
|
||||||
|
return 1
|
||||||
|
case "X2", "SP":
|
||||||
|
return 2
|
||||||
|
case "X3", "GP":
|
||||||
|
return 3
|
||||||
|
case "X4", "TP":
|
||||||
|
return 4
|
||||||
|
case "X5", "T0":
|
||||||
|
return 5
|
||||||
|
case "X6", "T1":
|
||||||
|
return 6
|
||||||
|
case "X7", "T2":
|
||||||
|
return 7
|
||||||
|
case "X8", "S0", "FP":
|
||||||
|
return 8
|
||||||
|
case "X9", "S1":
|
||||||
|
return 9
|
||||||
|
case "X10", "A0":
|
||||||
|
return 10
|
||||||
|
case "X11", "A1":
|
||||||
|
return 11
|
||||||
|
case "X12", "A2":
|
||||||
|
return 12
|
||||||
|
case "X13", "A3":
|
||||||
|
return 13
|
||||||
|
case "X14", "A4":
|
||||||
|
return 14
|
||||||
|
case "X15", "A5":
|
||||||
|
return 15
|
||||||
|
case "X16", "A6":
|
||||||
|
return 16
|
||||||
|
case "X17", "A7":
|
||||||
|
return 17
|
||||||
|
case "X18", "S2":
|
||||||
|
return 18
|
||||||
|
case "X19", "S3":
|
||||||
|
return 19
|
||||||
|
case "X20", "S4":
|
||||||
|
return 20
|
||||||
|
case "X21", "S5":
|
||||||
|
return 21
|
||||||
|
case "X22", "S6":
|
||||||
|
return 22
|
||||||
|
case "X23", "S7":
|
||||||
|
return 23
|
||||||
|
case "X24", "S8":
|
||||||
|
return 24
|
||||||
|
case "X25", "S9":
|
||||||
|
return 25
|
||||||
|
case "X26", "S10":
|
||||||
|
return 26
|
||||||
|
case "X27", "S11":
|
||||||
|
return 27
|
||||||
|
case "X28", "T3":
|
||||||
|
return 28
|
||||||
|
case "X29", "T4":
|
||||||
|
return 29
|
||||||
|
case "X30", "T5":
|
||||||
|
return 30
|
||||||
|
case "X31", "T6", "TMP":
|
||||||
|
return 31
|
||||||
|
// Floating-point registers (F0-F31).
|
||||||
|
case "F0", "FT0":
|
||||||
|
return 0
|
||||||
|
case "F1", "FT1":
|
||||||
|
return 1
|
||||||
|
case "F2", "FT2":
|
||||||
|
return 2
|
||||||
|
case "F3", "FT3":
|
||||||
|
return 3
|
||||||
|
case "F4", "FT4":
|
||||||
|
return 4
|
||||||
|
case "F5", "FT5":
|
||||||
|
return 5
|
||||||
|
case "F6", "FT6":
|
||||||
|
return 6
|
||||||
|
case "F7", "FT7":
|
||||||
|
return 7
|
||||||
|
case "F8", "FS0":
|
||||||
|
return 8
|
||||||
|
case "F9", "FS1":
|
||||||
|
return 9
|
||||||
|
case "F10", "FA0":
|
||||||
|
return 10
|
||||||
|
case "F11", "FA1":
|
||||||
|
return 11
|
||||||
|
case "F12", "FA2":
|
||||||
|
return 12
|
||||||
|
case "F13", "FA3":
|
||||||
|
return 13
|
||||||
|
case "F14", "FA4":
|
||||||
|
return 14
|
||||||
|
case "F15", "FA5":
|
||||||
|
return 15
|
||||||
|
case "F16", "FA6":
|
||||||
|
return 16
|
||||||
|
case "F17", "FA7":
|
||||||
|
return 17
|
||||||
|
case "F18", "FS2":
|
||||||
|
return 18
|
||||||
|
case "F19", "FS3":
|
||||||
|
return 19
|
||||||
|
case "F20", "FS4":
|
||||||
|
return 20
|
||||||
|
case "F21", "FS5":
|
||||||
|
return 21
|
||||||
|
case "F22", "FS6":
|
||||||
|
return 22
|
||||||
|
case "F23", "FS7":
|
||||||
|
return 23
|
||||||
|
case "F24", "FS8":
|
||||||
|
return 24
|
||||||
|
case "F25", "FS9":
|
||||||
|
return 25
|
||||||
|
case "F26", "FS10":
|
||||||
|
return 26
|
||||||
|
case "F27", "FS11":
|
||||||
|
return 27
|
||||||
|
case "F28", "FT8":
|
||||||
|
return 28
|
||||||
|
case "F29", "FT9":
|
||||||
|
return 29
|
||||||
|
case "F30", "FT10":
|
||||||
|
return 30
|
||||||
|
case "F31", "FT11":
|
||||||
|
return 31
|
||||||
|
default:
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// RISC-V instruction encoding parameters.
|
||||||
|
type riscvEnc struct {
|
||||||
|
opcode uint32 // bits [6:0]
|
||||||
|
funct3 uint32 // bits [14:12]
|
||||||
|
funct7 uint32 // bits [31:25]
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvInstrTable maps RISC-V mnemonics to their encoding.
|
||||||
|
var riscvInstrTable = map[string]riscvEnc{
|
||||||
|
// RV64I — R-type arithmetic/logic.
|
||||||
|
"ADD": {0x33, 0x0, 0x00},
|
||||||
|
"SUB": {0x33, 0x0, 0x20},
|
||||||
|
"SLL": {0x33, 0x1, 0x00},
|
||||||
|
"SLT": {0x33, 0x2, 0x00},
|
||||||
|
"SLTU": {0x33, 0x3, 0x00},
|
||||||
|
"XOR": {0x33, 0x4, 0x00},
|
||||||
|
"SRL": {0x33, 0x5, 0x00},
|
||||||
|
"SRA": {0x33, 0x5, 0x20},
|
||||||
|
"OR": {0x33, 0x6, 0x00},
|
||||||
|
"AND": {0x33, 0x7, 0x00},
|
||||||
|
// RV64I — 32-bit variants (W suffix).
|
||||||
|
"ADDW": {0x3B, 0x0, 0x00},
|
||||||
|
"SUBW": {0x3B, 0x0, 0x20},
|
||||||
|
"SLLW": {0x3B, 0x1, 0x00},
|
||||||
|
"SRLW": {0x3B, 0x5, 0x00},
|
||||||
|
"SRAW": {0x3B, 0x5, 0x20},
|
||||||
|
// RV64I — I-type shift-immediate (shamt in rs2 field).
|
||||||
|
"SLLI": {0x13, 0x1, 0x00},
|
||||||
|
"SRLI": {0x13, 0x5, 0x00},
|
||||||
|
"SRAI": {0x13, 0x5, 0x20},
|
||||||
|
"SLLIW": {0x1B, 0x1, 0x00},
|
||||||
|
"SRLIW": {0x1B, 0x5, 0x00},
|
||||||
|
"SRAIW": {0x1B, 0x5, 0x20},
|
||||||
|
// RV64M — multiply/divide.
|
||||||
|
"MUL": {0x33, 0x0, 0x01},
|
||||||
|
"MULH": {0x33, 0x1, 0x01},
|
||||||
|
"MULHSU": {0x33, 0x2, 0x01},
|
||||||
|
"MULHU": {0x33, 0x3, 0x01},
|
||||||
|
"DIV": {0x33, 0x4, 0x01},
|
||||||
|
"DIVU": {0x33, 0x5, 0x01},
|
||||||
|
"REM": {0x33, 0x6, 0x01},
|
||||||
|
"REMU": {0x33, 0x7, 0x01},
|
||||||
|
// RV64M — 32-bit variants.
|
||||||
|
"MULW": {0x3B, 0x0, 0x01},
|
||||||
|
"DIVW": {0x3B, 0x4, 0x01},
|
||||||
|
"DIVUW": {0x3B, 0x5, 0x01},
|
||||||
|
"REMW": {0x3B, 0x6, 0x01},
|
||||||
|
"REMUW": {0x3B, 0x7, 0x01},
|
||||||
|
// RV64I — I-type arithmetic.
|
||||||
|
"ADDI": {0x13, 0x0, 0x00},
|
||||||
|
"ADDIW": {0x1B, 0x0, 0x00},
|
||||||
|
"SLTI": {0x13, 0x2, 0x00},
|
||||||
|
"SLTIU": {0x13, 0x3, 0x00},
|
||||||
|
"XORI": {0x13, 0x4, 0x00},
|
||||||
|
"ORI": {0x13, 0x6, 0x00},
|
||||||
|
"ANDI": {0x13, 0x7, 0x00},
|
||||||
|
// Loads (I-type).
|
||||||
|
"LB": {0x03, 0x0, 0x00},
|
||||||
|
"LH": {0x03, 0x1, 0x00},
|
||||||
|
"LW": {0x03, 0x2, 0x00},
|
||||||
|
"LD": {0x03, 0x3, 0x00},
|
||||||
|
"LBU": {0x03, 0x4, 0x00},
|
||||||
|
"LHU": {0x03, 0x5, 0x00},
|
||||||
|
"LWU": {0x03, 0x6, 0x00},
|
||||||
|
// Stores (S-type).
|
||||||
|
"SB": {0x23, 0x0, 0x00},
|
||||||
|
"SH": {0x23, 0x1, 0x00},
|
||||||
|
"SW": {0x23, 0x2, 0x00},
|
||||||
|
"SD": {0x23, 0x3, 0x00},
|
||||||
|
// Branches (B-type).
|
||||||
|
"BEQ": {0x63, 0x0, 0x00},
|
||||||
|
"BNE": {0x63, 0x1, 0x00},
|
||||||
|
"BLT": {0x63, 0x4, 0x00},
|
||||||
|
"BGE": {0x63, 0x5, 0x00},
|
||||||
|
"BLTU": {0x63, 0x6, 0x00},
|
||||||
|
"BGEU": {0x63, 0x7, 0x00},
|
||||||
|
// U-type.
|
||||||
|
"LUI": {0x37, 0x0, 0x00},
|
||||||
|
"AUIPC": {0x17, 0x0, 0x00},
|
||||||
|
// System.
|
||||||
|
"ECALL": {0x73, 0x0, 0x00},
|
||||||
|
"EBREAK": {0x73, 0x0, 0x00},
|
||||||
|
"FENCE": {0x0F, 0x0, 0x00},
|
||||||
|
// JALR — indirect jump/call (I-type).
|
||||||
|
"JALR": {0x67, 0x0, 0x00},
|
||||||
|
|
||||||
|
// RV64A — atomics (AMO opcode 0x2F).
|
||||||
|
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
|
||||||
|
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
|
||||||
|
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
|
||||||
|
"AMOADDW": {0x2F, 0x2, 0x00 << 2},
|
||||||
|
"AMOADDD": {0x2F, 0x3, 0x00 << 2},
|
||||||
|
"AMOANDW": {0x2F, 0x2, 0x0C << 2},
|
||||||
|
"AMOANDD": {0x2F, 0x3, 0x0C << 2},
|
||||||
|
"AMOORW": {0x2F, 0x2, 0x06 << 2},
|
||||||
|
"AMOORD": {0x2F, 0x3, 0x06 << 2},
|
||||||
|
"AMOXORW": {0x2F, 0x2, 0x04 << 2},
|
||||||
|
"AMOXORD": {0x2F, 0x3, 0x04 << 2},
|
||||||
|
"AMOMAXW": {0x2F, 0x2, 0x14 << 2},
|
||||||
|
"AMOMAXD": {0x2F, 0x3, 0x14 << 2},
|
||||||
|
"AMOMINW": {0x2F, 0x2, 0x10 << 2},
|
||||||
|
"AMOMIND": {0x2F, 0x3, 0x10 << 2},
|
||||||
|
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2},
|
||||||
|
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2},
|
||||||
|
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
|
||||||
|
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
|
||||||
|
|
||||||
|
// RV64F/D — floating-point arithmetic.
|
||||||
|
"FADDS": {0x53, 0x0, 0x00},
|
||||||
|
"FSUBS": {0x53, 0x0, 0x04},
|
||||||
|
"FMULS": {0x53, 0x0, 0x08},
|
||||||
|
"FDIVS": {0x53, 0x0, 0x0C},
|
||||||
|
"FADDD": {0x53, 0x0, 0x01},
|
||||||
|
"FSUBD": {0x53, 0x0, 0x05},
|
||||||
|
"FMULD": {0x53, 0x0, 0x09},
|
||||||
|
"FDIVD": {0x53, 0x0, 0x0D},
|
||||||
|
"FSQRTS": {0x53, 0x0, 0x2C},
|
||||||
|
"FSQRTD": {0x53, 0x0, 0x2D},
|
||||||
|
// FP loads/stores.
|
||||||
|
"FLW": {0x07, 0x2, 0x00},
|
||||||
|
"FLD": {0x07, 0x3, 0x00},
|
||||||
|
"FSW": {0x27, 0x2, 0x00},
|
||||||
|
"FSD": {0x27, 0x3, 0x00},
|
||||||
|
// FP min/max.
|
||||||
|
"FMINS": {0x53, 0x0, 0x14},
|
||||||
|
"FMAXS": {0x53, 0x1, 0x14},
|
||||||
|
"FMIND": {0x53, 0x0, 0x15},
|
||||||
|
"FMAXD": {0x53, 0x1, 0x15},
|
||||||
|
|
||||||
|
// RV64A — load-reserved / store-conditional (funct5 0x02 / 0x03).
|
||||||
|
"LRW": {0x2F, 0x2, 0x02 << 2},
|
||||||
|
"LRD": {0x2F, 0x3, 0x02 << 2},
|
||||||
|
"SCW": {0x2F, 0x2, 0x03 << 2},
|
||||||
|
"SCD": {0x2F, 0x3, 0x03 << 2},
|
||||||
|
|
||||||
|
// FP compare — result in integer register (funct7 0x50/0x51).
|
||||||
|
"FEQS": {0x53, 0x2, 0x50},
|
||||||
|
"FLTS": {0x53, 0x1, 0x50},
|
||||||
|
"FLES": {0x53, 0x0, 0x50},
|
||||||
|
"FEQD": {0x53, 0x2, 0x51},
|
||||||
|
"FLTD": {0x53, 0x1, 0x51},
|
||||||
|
"FLED": {0x53, 0x0, 0x51},
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvRType encodes an R-type instruction: funct7 | rs2 | rs1 | funct3 | rd | opcode.
|
||||||
|
func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||||
|
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||||
|
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvAMOType encodes an atomic (AMO) instruction.
|
||||||
|
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode.
|
||||||
|
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2).
|
||||||
|
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||||
|
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value
|
||||||
|
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||||
|
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||||
|
}
|
||||||
|
|
||||||
|
// FP conversion instructions (FCVT, FMV). These use the rs2 field to
|
||||||
|
// encode the conversion type rather than a register, so they are handled
|
||||||
|
// separately from the general instruction table.
|
||||||
|
type riscvCvtEnc struct {
|
||||||
|
funct7 uint32 // bits [31:25]
|
||||||
|
rs2 uint32 // conversion-type code in bits [24:20]
|
||||||
|
opcode uint32 // always 0x53 (OP-FP)
|
||||||
|
}
|
||||||
|
|
||||||
|
var riscvCvtTable = map[string]riscvCvtEnc{
|
||||||
|
// float → int (rs2 selects the integer width/sign).
|
||||||
|
"FCVTWS": {0x60, 0x0, 0x53}, // float32 → int32
|
||||||
|
"FCVTWUS": {0x60, 0x1, 0x53}, // float32 → uint32
|
||||||
|
"FCVTLS": {0x60, 0x2, 0x53}, // float32 → int64
|
||||||
|
"FCVTLUS": {0x60, 0x3, 0x53}, // float32 → uint64
|
||||||
|
"FCVTWD": {0x61, 0x0, 0x53}, // float64 → int32
|
||||||
|
"FCVTWUD": {0x61, 0x1, 0x53}, // float64 → uint32
|
||||||
|
"FCVTLD": {0x61, 0x2, 0x53}, // float64 → int64
|
||||||
|
"FCVTLUD": {0x61, 0x3, 0x53}, // float64 → uint64
|
||||||
|
// int → float (rs2 selects the integer width/sign).
|
||||||
|
"FCVTSW": {0x68, 0x0, 0x53}, // int32 → float32
|
||||||
|
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
|
||||||
|
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
|
||||||
|
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
|
||||||
|
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
|
||||||
|
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
|
||||||
|
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
|
||||||
|
"FCVTDLU": {0x69, 0x3, 0x53}, // uint64 → float64
|
||||||
|
// float → float width conversion.
|
||||||
|
"FCVTSD": {0x20, 0x1, 0x53}, // float64 → float32
|
||||||
|
"FCVTDS": {0x21, 0x0, 0x53}, // float32 → float64
|
||||||
|
// Bit moves between integer and FP registers (no conversion).
|
||||||
|
"FMVXD": {0x71, 0x0, 0x53}, // float64 → int64 (bit move)
|
||||||
|
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
|
||||||
|
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
|
||||||
|
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvCvtType encodes an FP conversion instruction.
|
||||||
|
// Layout: funct7 | rs2(convtype) | rs1 | funct3(0) | rd | opcode.
|
||||||
|
func riscvCvtType(enc riscvCvtEnc, rd, rs1 int) uint32 {
|
||||||
|
return (enc.funct7 << 25) | (enc.rs2 << 20) | (uint32(rs1) << 15) |
|
||||||
|
(uint32(rd) << 7) | enc.opcode
|
||||||
|
}
|
||||||
|
|
||||||
|
// R4-type fused multiply-add instructions (FMADD/FMSUB/FNMSUB/FNMADD).
|
||||||
|
// These take 4 register operands: rs1, rs2, rs3, rd.
|
||||||
|
// Layout: rs3 | fmt | rs2 | rs1 | rm | rd | opcode.
|
||||||
|
type riscvFmaEnc struct {
|
||||||
|
fmt uint32 // bits [26:25]: 0x0 = single, 0x1 = double
|
||||||
|
opcode uint32 // bits [6:0]
|
||||||
|
}
|
||||||
|
|
||||||
|
var riscvFmaTable = map[string]riscvFmaEnc{
|
||||||
|
"FMADDS": {0x0, 0x43}, // rd = rs1*rs2 + rs3
|
||||||
|
"FMADDD": {0x1, 0x43},
|
||||||
|
"FMSUBS": {0x0, 0x47}, // rd = rs1*rs2 - rs3
|
||||||
|
"FMSUBD": {0x1, 0x47},
|
||||||
|
"FNMSUBS": {0x0, 0x4B}, // rd = -(rs1*rs2) + rs3
|
||||||
|
"FNMSUBD": {0x1, 0x4B},
|
||||||
|
"FNMADDS": {0x0, 0x4F}, // rd = -(rs1*rs2) - rs3
|
||||||
|
"FNMADDD": {0x1, 0x4F},
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvFmaType encodes an R4-type fused multiply-add instruction.
|
||||||
|
func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 {
|
||||||
|
return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) |
|
||||||
|
(uint32(rs1) << 15) | (0x0 << 12) /* rm=dynamic */ | (uint32(rd) << 7) | enc.opcode
|
||||||
|
}
|
||||||
|
|
||||||
|
// CSR (Control and Status Register) instructions.
|
||||||
|
// Format: csr[11:0] | rs1/zimm | funct3 | rd | opcode (0x73).
|
||||||
|
type riscvCsrEnc struct {
|
||||||
|
funct3 uint32 // bits [14:12]
|
||||||
|
imm bool // true for CSRRWI/CSRRSI/CSRRCI (5-bit uimm variant)
|
||||||
|
}
|
||||||
|
|
||||||
|
var riscvCsrTable = map[string]riscvCsrEnc{
|
||||||
|
"CSRRW": {0x1, false}, // rd=CSR, CSR=rs1
|
||||||
|
"CSRRS": {0x2, false}, // rd=CSR, CSR |= rs1
|
||||||
|
"CSRRC": {0x3, false}, // rd=CSR, CSR &= ~rs1
|
||||||
|
"CSRRWI": {0x5, true}, // rd=CSR, CSR=uimm
|
||||||
|
"CSRRSI": {0x6, true}, // rd=CSR, CSR |= uimm
|
||||||
|
"CSRRCI": {0x7, true}, // rd=CSR, CSR &= ~uimm
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvCsrType encodes a CSR instruction.
|
||||||
|
// csr is the 12-bit CSR address; src is either a register number or a 5-bit
|
||||||
|
// unsigned immediate (depending on enc.imm).
|
||||||
|
func riscvCsrType(enc riscvCsrEnc, rd, src int, csr int32) uint32 {
|
||||||
|
return (uint32(csr&0xFFF) << 20) | (uint32(src&0x1F) << 15) |
|
||||||
|
(enc.funct3 << 12) | (uint32(rd) << 7) | 0x73
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvIType encodes an I-type instruction: imm[11:0] | rs1 | funct3 | rd | opcode.
|
||||||
|
func riscvIType(enc riscvEnc, rd, rs1 int, imm int32) uint32 {
|
||||||
|
return (uint32(imm&0xFFF) << 20) | (uint32(rs1) << 15) |
|
||||||
|
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvSType encodes an S-type instruction: imm[11:5] | rs2 | rs1 | funct3 | imm[4:0] | opcode.
|
||||||
|
func riscvSType(enc riscvEnc, rs1, rs2 int, imm int32) uint32 {
|
||||||
|
immU := uint32(imm) & 0xFFF
|
||||||
|
return ((immU >> 5) << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||||
|
(enc.funct3 << 12) | ((immU & 0x1F) << 7) | enc.opcode
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvBType encodes a B-type instruction (branches).
|
||||||
|
func riscvBType(enc riscvEnc, rs1, rs2 int, offset int32) uint32 {
|
||||||
|
imm := uint32(offset) & 0x1FFE // bits [12:1], bit 0 is always 0
|
||||||
|
return (((imm >> 12) & 1) << 31) | // imm[12]
|
||||||
|
(((imm >> 5) & 0x3F) << 25) | // imm[10:5]
|
||||||
|
(uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||||
|
(enc.funct3 << 12) |
|
||||||
|
(((imm >> 1) & 0xF) << 8) | // imm[4:1]
|
||||||
|
(((imm >> 11) & 1) << 7) | // imm[11]
|
||||||
|
enc.opcode
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvUType encodes a U-type instruction: imm[31:12] | rd | opcode.
|
||||||
|
func riscvUType(enc riscvEnc, rd int, imm int32) uint32 {
|
||||||
|
return (uint32(imm) & 0xFFFFF000) | (uint32(rd) << 7) | enc.opcode
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvJType encodes a J-type instruction (JAL).
|
||||||
|
func riscvJType(rd int, offset int32) uint32 {
|
||||||
|
imm := uint32(offset) & 0x1FFFFE // bits [20:1]
|
||||||
|
return (((imm >> 20) & 1) << 31) | // imm[20]
|
||||||
|
(((imm >> 1) & 0x3FF) << 21) | // imm[10:1]
|
||||||
|
(((imm >> 11) & 1) << 20) | // imm[11]
|
||||||
|
(((imm >> 12) & 0xFF) << 12) | // imm[19:12]
|
||||||
|
(uint32(rd) << 7) |
|
||||||
|
0x6F // JAL opcode
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- RVC (compressed) encoding helpers ----
|
||||||
|
|
||||||
|
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
|
||||||
|
// prime register field used by compressed instructions (x8–x15).
|
||||||
|
func isRVCIntReg(r int) bool { return r >= 8 && r <= 15 }
|
||||||
|
|
||||||
|
// rvcReg3 returns the 3-bit encoding for registers x8–x15 (0–7).
|
||||||
|
func rvcReg3(r int) uint32 { return uint32(r - 8) }
|
||||||
|
|
||||||
|
// rvcCR encodes a CR-type (register) compressed instruction.
|
||||||
|
// Format: funct4 | rd/rs1 | rs2 | op=2.
|
||||||
|
func rvcCR(funct4, rd, rs2 uint32) uint16 {
|
||||||
|
return uint16((funct4 << 12) | (rd << 7) | (rs2 << 2) | 0x2)
|
||||||
|
}
|
||||||
|
|
||||||
|
// rvcCI encodes a CI-type (immediate) compressed instruction.
|
||||||
|
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW — linear 6-bit immediate.
|
||||||
|
func rvcCI(funct3, rd uint32, imm uint32) uint16 {
|
||||||
|
return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x1)
|
||||||
|
}
|
||||||
|
|
||||||
|
// rvcSLLI encodes C.SLLI, which shares funct3=0 with C.ADDI but lives in the
|
||||||
|
// op=10 quadrant (unlike C.ADDI's op=01).
|
||||||
|
func rvcSLLI(rd, shamt uint32) uint16 {
|
||||||
|
return uint16(((shamt>>5)&1)<<12 | (rd << 7) | (shamt&0x1F)<<2 | 0x2)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeRVCPattern extracts the bits listed in pattern (MSB first) from imm
|
||||||
|
// into a packed value, matching cmd/internal/obj/riscv's encodeBitPattern.
|
||||||
|
func encodeRVCPattern(imm uint32, pattern []int) uint32 {
|
||||||
|
packed := uint32(0)
|
||||||
|
for _, bit := range pattern {
|
||||||
|
packed = packed<<1 | (imm>>bit)&1
|
||||||
|
}
|
||||||
|
return packed
|
||||||
|
}
|
||||||
|
|
||||||
|
// rvcLSP encodes a stack-relative compressed load (op=10 quadrant): C.LWSP
|
||||||
|
// (funct3=2, 4-byte scale), C.LDSP (funct3=3) or C.FLDSP (funct3=1, 8-byte
|
||||||
|
// scale). offset is the full byte offset.
|
||||||
|
func rvcLSP(funct3, rd uint32, offset uint32) uint16 {
|
||||||
|
pattern := []int{5, 4, 3, 8, 7, 6}
|
||||||
|
if funct3 == 0x2 {
|
||||||
|
pattern = []int{5, 4, 3, 2, 7, 6}
|
||||||
|
}
|
||||||
|
packed := uint32(0)
|
||||||
|
for i, b := range pattern {
|
||||||
|
packed |= ((offset >> b) & 1) << (5 - i)
|
||||||
|
}
|
||||||
|
return uint16((funct3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x2)
|
||||||
|
}
|
||||||
|
|
||||||
|
// rvcSSP encodes a stack-relative compressed store (op=10 quadrant): C.SWSP
|
||||||
|
// (funct3=6, 4-byte scale), C.SDSP (funct3=7) or C.FSDSP (funct3=5, 8-byte
|
||||||
|
// scale). offset is the full byte offset.
|
||||||
|
func rvcSSP(funct3, rs2 uint32, offset uint32) uint16 {
|
||||||
|
pattern := []int{5, 4, 3, 8, 7, 6}
|
||||||
|
if funct3 == 0x6 {
|
||||||
|
pattern = []int{5, 4, 3, 2, 7, 6}
|
||||||
|
}
|
||||||
|
packed := uint32(0)
|
||||||
|
for i, b := range pattern {
|
||||||
|
packed |= ((offset >> b) & 1) << (5 - i)
|
||||||
|
}
|
||||||
|
return uint16((funct3 << 13) | (packed << 7) | (rs2 << 2) | 0x2)
|
||||||
|
}
|
||||||
|
|
||||||
|
// rvcCL encodes a register-relative compressed load (op=00 quadrant): C.LW
|
||||||
|
// (funct3=2), C.LD (funct3=3) or C.FLD (funct3=1). imm is the full byte
|
||||||
|
// offset; the immediate bits are extracted per the RISC-V CL format.
|
||||||
|
func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 {
|
||||||
|
pattern := []int{5, 4, 3, 7, 6}
|
||||||
|
if funct3 == 0x2 {
|
||||||
|
pattern = []int{5, 4, 3, 2, 6}
|
||||||
|
}
|
||||||
|
packed := encodeRVCPattern(imm, pattern)
|
||||||
|
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rd << 2))
|
||||||
|
}
|
||||||
|
|
||||||
|
// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
|
||||||
|
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
|
||||||
|
// offset; the immediate bits are extracted per the RISC-V CS format.
|
||||||
|
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
|
||||||
|
pattern := []int{5, 3, 7, 6}
|
||||||
|
if funct3 == 0x6 {
|
||||||
|
pattern = []int{5, 3, 2, 6}
|
||||||
|
}
|
||||||
|
packed := encodeRVCPattern(imm, pattern)
|
||||||
|
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
|
||||||
|
}
|
||||||
|
|
||||||
|
// rvcCIW encodes a CIW-type compressed immediate wide instruction: C.ADDI4SPN
|
||||||
|
// (funct3=0). imm is the raw byte offset.
|
||||||
|
func rvcCIW(funct3, rd uint32, imm uint32) uint16 {
|
||||||
|
packed := encodeRVCPattern(imm, []int{5, 4, 9, 8, 7, 6, 2, 3})
|
||||||
|
return uint16((funct3 << 13) | (packed << 5) | (rd << 2))
|
||||||
|
}
|
||||||
|
|
||||||
|
// rvcCA encodes a CA-type (arithmetic) compressed instruction.
|
||||||
|
// Format: funct6[15:10] | rd'/rs1'[9:7] | funct2[6:5] | rs2'[4:2] | op=01.
|
||||||
|
func rvcCA(funct6, funct2, rd, rs2 uint32) uint16 {
|
||||||
|
return uint16((funct6 << 10) | (rd << 7) | (funct2 << 5) | (rs2 << 2) | 0x1)
|
||||||
|
}
|
||||||
|
|
||||||
|
// rvcCBShift encodes a CB-type shift/immediate compressed instruction
|
||||||
|
// (C.SRLI, C.SRAI, C.ANDI). rd is the 3-bit prime-register index; imm is
|
||||||
|
// the 6-bit shamt/immediate; funct2 selects the operation (0=SRLI, 1=SRAI,
|
||||||
|
// 2=ANDI).
|
||||||
|
func rvcCBShift(funct2, rd, imm uint32) uint16 {
|
||||||
|
return uint16((0x4 << 13) | ((imm>>5)&1)<<12 | (funct2 << 10) | (rd << 7) | (imm&0x1F)<<2 | 0x1)
|
||||||
|
}
|
||||||
|
|
||||||
|
// rvcADDI16SP encodes C.ADDI16SP: ADDI rd, imm, rd for the stack pointer
|
||||||
|
// with a 10-bit signed, 16-byte-scaled immediate. imm is the raw byte
|
||||||
|
// offset; the immediate bits are extracted in the order [9|4|6|8:7|5].
|
||||||
|
func rvcADDI16SP(rd uint32, imm int32) uint16 {
|
||||||
|
u := uint32(imm)
|
||||||
|
packed := uint32(0)
|
||||||
|
for _, bit := range []uint{9, 4, 6, 8, 7, 5} {
|
||||||
|
packed = packed<<1 | (u>>bit)&1
|
||||||
|
}
|
||||||
|
return uint16((0x3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x1)
|
||||||
|
}
|
||||||
@@ -0,0 +1,763 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// firstTextRISCV parses assembly source and returns the first TEXT function body.
|
||||||
|
func firstTextRISCV(t *testing.T, src string) *ast.Text {
|
||||||
|
t.Helper()
|
||||||
|
f, errs := parser.Parse("f_riscv64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
for _, d := range f.Decls {
|
||||||
|
if fn, ok := d.(*ast.Text); ok {
|
||||||
|
return fn
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatal("no TEXT found")
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// assembleRISCVHelper assembles one TEXT function and returns its code bytes.
|
||||||
|
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
|
||||||
|
t.Helper()
|
||||||
|
code, _, _, _, _, err := assembleRISCV(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
return code
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_add(t *testing.T) {
|
||||||
|
// func add(a, b int64) int64
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
MOV b+8(FP), X11
|
||||||
|
ADD X11, X10, X10
|
||||||
|
MOV X10, ret+16(FP)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// should be 12 bytes with RVC: C.LDSP + C.LDSP + ADD + C.SDSP + C.JR
|
||||||
|
_ = code
|
||||||
|
if len(code) == 0 {
|
||||||
|
t.Error("empty output")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_arithmetic(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·arith(SB), NOSPLIT, $0
|
||||||
|
ADD X10, X11, X12
|
||||||
|
SUB X12, X13, X14
|
||||||
|
MUL X14, X15, X16
|
||||||
|
DIV X16, X17, X18
|
||||||
|
REM X18, X19, X20
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// 5 R-type instructions + RET = 5*4 + 4 = 24
|
||||||
|
if len(code) != 24 {
|
||||||
|
t.Errorf("expected 24 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_loadStore(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·mem(SB), NOSPLIT, $0
|
||||||
|
LD (X10), X11
|
||||||
|
SD X11, (X12)
|
||||||
|
LW (X13), X14
|
||||||
|
SW X14, (X15)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// Four register-relative loads/stores compress (2B each) + JALR (4B) = 12.
|
||||||
|
if len(code) != 12 {
|
||||||
|
t.Errorf("expected 12 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_immediate(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·imm(SB), NOSPLIT, $0
|
||||||
|
ADDI $42, X10, X11
|
||||||
|
ANDI $0xFF, X11, X12
|
||||||
|
ORI $1, X12, X13
|
||||||
|
XORI $0, X13, X14
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// 4 I-type + JALR = 4*4 + 4 = 20
|
||||||
|
if len(code) != 20 {
|
||||||
|
t.Errorf("expected 20 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_branches(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·br(SB), NOSPLIT, $0
|
||||||
|
ADDI $1, X10, X10
|
||||||
|
loop:
|
||||||
|
BEQ X10, X11, done
|
||||||
|
ADDI $1, X10, X10
|
||||||
|
JMP loop
|
||||||
|
done:
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
_ = code
|
||||||
|
if len(code) == 0 {
|
||||||
|
t.Error("empty output")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_MOV_imm_small(t *testing.T) {
|
||||||
|
// MOV $42, rd → ADDI (fits in 12 bits, but not C.LI's 6-bit immediate).
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·small(SB), NOSPLIT, $0
|
||||||
|
MOV $42, X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// ADDI (4B) + JALR (4B) = 8
|
||||||
|
if len(code) != 8 {
|
||||||
|
t.Errorf("expected 8 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_MOV_imm_large(t *testing.T) {
|
||||||
|
// MOV $0x12345, rd → C.LUI $18 (2B) + ADDIW $837 (4B).
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·large(SB), NOSPLIT, $0
|
||||||
|
MOV $0x12345, X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// C.LUI (2B) + ADDIW (4B) + JALR (4B) = 10
|
||||||
|
if len(code) != 10 {
|
||||||
|
t.Errorf("expected 10 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_MOV_reg(t *testing.T) {
|
||||||
|
// MOV rs, rd → ADDI $0, rs, rd, compresses to C.MV
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·reg(SB), NOSPLIT, $0
|
||||||
|
MOV X10, X11
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// C.MV (2B) + JALR (4B) = 6
|
||||||
|
if len(code) != 6 {
|
||||||
|
t.Errorf("expected 6 bytes, got %d (% x)", len(code), code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_MOV_frame(t *testing.T) {
|
||||||
|
// MOV name+off(FP), rd → load with frame mapping
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·frame(SB), NOSPLIT, $0-8
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
MOV X10, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// C.LDSP (2B) + C.SDSP (2B) + JALR (4B) = 8
|
||||||
|
if len(code) != 8 {
|
||||||
|
t.Errorf("expected 8 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_loadStore(t *testing.T) {
|
||||||
|
// Verify that loads/stores from SP are compressed.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·rvcstore(SB), NOSPLIT, $0
|
||||||
|
LD 0(SP), X10
|
||||||
|
SD X10, 8(SP)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// C.LDSP (2B) + C.SDSP (2B) + JALR (4B) = 8
|
||||||
|
if len(code) != 8 {
|
||||||
|
t.Errorf("expected 8 bytes, got %d (% x)", len(code), code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_atomics(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·amo(SB), NOSPLIT, $0
|
||||||
|
AMOADDD X10, (X11), X12
|
||||||
|
LRD (X13), X14
|
||||||
|
SCD X15, (X16), X17
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// 3 AMO instructions (4B each) + JALR (4B) = 16
|
||||||
|
if len(code) != 16 {
|
||||||
|
t.Errorf("expected 16 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_fpArith(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·fpadd(SB), NOSPLIT, $0
|
||||||
|
FADDD F10, F11, F12
|
||||||
|
FSUBD F12, F13, F14
|
||||||
|
FMULD F14, F15, F16
|
||||||
|
FDIVD F16, F17, F18
|
||||||
|
FSQRTD F18, F19
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// 5 FP instructions (4B each) + JALR (4B) = 24
|
||||||
|
if len(code) != 24 {
|
||||||
|
t.Errorf("expected 24 bytes, got %d (%d)", len(code), len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_csr(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·csrtest(SB), NOSPLIT, $0
|
||||||
|
CSRRS $0x300, X0, X10
|
||||||
|
CSRRW $0x305, X10, X11
|
||||||
|
CSRRSI $0x304, $5, X12
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// 3 CSR instructions (4B each) + JALR (4B) = 16
|
||||||
|
if len(code) != 16 {
|
||||||
|
t.Errorf("expected 16 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_fma(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·fmatest(SB), NOSPLIT, $0
|
||||||
|
FMADDD F10, F11, F12, F13
|
||||||
|
FMSUBD F13, F14, F15, F16
|
||||||
|
FNMSUBD F16, F17, F18, F19
|
||||||
|
FNMADDD F19, F10, F11, F12
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// 4 FMA instructions (4B each) + JALR (4B) = 20
|
||||||
|
if len(code) != 20 {
|
||||||
|
t.Errorf("expected 20 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_conversions(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·cvt(SB), NOSPLIT, $0
|
||||||
|
FCVTDL X10, F10
|
||||||
|
FCVTLD F10, X11
|
||||||
|
FMVXD F10, X12
|
||||||
|
FMVDX X12, F11
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// 4 conversion instructions (4B each) + JALR (4B) = 20
|
||||||
|
if len(code) != 20 {
|
||||||
|
t.Errorf("expected 20 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_fpCmp(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·cmp(SB), NOSPLIT, $0
|
||||||
|
FEQD F10, F11, X10
|
||||||
|
FLTD F12, F13, X11
|
||||||
|
FLED F14, F15, X12
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// 3 FP compare (4B each) + JALR (4B) = 16
|
||||||
|
if len(code) != 16 {
|
||||||
|
t.Errorf("expected 16 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_forwardBranch(t *testing.T) {
|
||||||
|
// Forward label reference — must not fail.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·fwd(SB), NOSPLIT, $0
|
||||||
|
ADDI $1, X10, X10
|
||||||
|
BEQ X10, X11, done
|
||||||
|
ADDI $1, X10, X10
|
||||||
|
done:
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
_ = code
|
||||||
|
if len(code) == 0 {
|
||||||
|
t.Error("empty output")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_ADDI(t *testing.T) {
|
||||||
|
// ADDI where rd=rs1 and small imm → C.ADDI
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·caddi(SB), NOSPLIT, $0
|
||||||
|
ADDI $5, X10, X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// C.ADDI (2B) + JALR (4B) = 6
|
||||||
|
if len(code) != 6 {
|
||||||
|
t.Errorf("expected 6 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_LI(t *testing.T) {
|
||||||
|
// ADDI X0, $imm, rd → C.LI
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·cli(SB), NOSPLIT, $0
|
||||||
|
ADDI $7, X0, X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// C.LI (2B) + JALR (4B) = 6
|
||||||
|
if len(code) != 6 {
|
||||||
|
t.Errorf("expected 6 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_LUI(t *testing.T) {
|
||||||
|
// LUI rd, small nonzero imm → C.LUI
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·clui(SB), NOSPLIT, $0
|
||||||
|
LUI X10, $1
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// C.LUI (2B) + JALR (4B) = 6
|
||||||
|
if len(code) != 6 {
|
||||||
|
t.Errorf("expected 6 bytes, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_AssembleFile(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·sub(SB), NOSPLIT, $0
|
||||||
|
SUB X10, X11, X12
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("t_riscv64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
if len(img.Funcs) != 2 {
|
||||||
|
t.Fatalf("expected 2 functions, got %d", len(img.Funcs))
|
||||||
|
}
|
||||||
|
// func add: C.LDSP(2) + JALR(4) = 6
|
||||||
|
if img.Funcs[0].Size != 6 {
|
||||||
|
t.Errorf("add: expected 6 bytes, got %d", img.Funcs[0].Size)
|
||||||
|
}
|
||||||
|
// func sub: SUB(4) + JALR(4) = 8
|
||||||
|
if img.Funcs[1].Size != 8 {
|
||||||
|
t.Errorf("sub: expected 8 bytes, got %d", img.Funcs[1].Size)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_encodings(t *testing.T) {
|
||||||
|
// Smoke test that all known RISC-V mnemonics encode successfully.
|
||||||
|
tests := []struct {
|
||||||
|
name, src string
|
||||||
|
wantBytes int
|
||||||
|
}{
|
||||||
|
{"ADD", "ADD X10, X11, X12\nRET\n", 8},
|
||||||
|
{"SUBW", "SUBW X10, X11, X12\nRET\n", 8},
|
||||||
|
{"MUL", "MUL X10, X11, X12\nRET\n", 8},
|
||||||
|
{"DIVW", "DIVW X10, X11, X12\nRET\n", 8},
|
||||||
|
{"REMUW", "REMUW X10, X11, X12\nRET\n", 8},
|
||||||
|
{"ADDIW", "ADDIW $5, X10, X11\nRET\n", 8},
|
||||||
|
{"SLLI", "SLLI $3, X10, X11\nRET\n", 8}, // ADDI+SLLI? No, SLLI uses I-type
|
||||||
|
{"SRLI", "SRLI $2, X10, X11\nRET\n", 8},
|
||||||
|
{"SRAI", "SRAI $1, X10, X11\nRET\n", 8},
|
||||||
|
{"LB", "LB (X10), X11\nRET\n", 8},
|
||||||
|
{"LBU", "LBU (X10), X11\nRET\n", 8},
|
||||||
|
{"LH", "LH (X10), X11\nRET\n", 8},
|
||||||
|
{"LHU", "LHU (X10), X11\nRET\n", 8},
|
||||||
|
{"LWU", "LWU (X10), X11\nRET\n", 8},
|
||||||
|
{"SB", "SB X10, (X11)\nRET\n", 8},
|
||||||
|
{"SH", "SH X10, (X11)\nRET\n", 8},
|
||||||
|
{"SW", "SW X10, (X11)\nRET\n", 6},
|
||||||
|
{"LUI", "LUI X10, $0x12345\nRET\n", 8},
|
||||||
|
{"AUIPC", "AUIPC X10, $0\nRET\n", 8},
|
||||||
|
{"FLW", "FLW (X10), F10\nRET\n", 8},
|
||||||
|
{"FSW", "FSW F10, (X11)\nRET\n", 8},
|
||||||
|
{"FADDS", "FADDS F10, F11, F12\nRET\n", 8},
|
||||||
|
{"FMINS", "FMINS F10, F11, F12\nRET\n", 8},
|
||||||
|
{"FMAXD", "FMAXD F10, F11, F12\nRET\n", 8},
|
||||||
|
{"FCVTSD", "FCVTSD F10, F11\nRET\n", 8},
|
||||||
|
{"FCVTDS", "FCVTDS F10, F11\nRET\n", 8},
|
||||||
|
{"FMVXW", "FMVXW F10, X10\nRET\n", 8},
|
||||||
|
{"FMADD_S", "FMADDS F10, F11, F12, F13\nRET\n", 8},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·`+tt.name+`(SB), NOSPLIT, $0
|
||||||
|
`+tt.src)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
if len(code) != tt.wantBytes {
|
||||||
|
t.Errorf("expected %d bytes, got %d", tt.wantBytes, len(code))
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_branch(t *testing.T) {
|
||||||
|
// Branches are never RVC-compressed (no C.BEQZ/C.BNEZ), matching go tool asm.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·cbeqz(SB), NOSPLIT, $0
|
||||||
|
ADDI $1, X10, X10
|
||||||
|
BEQ X10, X0, done
|
||||||
|
ADDI $1, X10, X10
|
||||||
|
done:
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// C.ADDI(2) + BEQ(4) + C.ADDI(2) + JALR(4) = 12
|
||||||
|
if len(code) != 12 {
|
||||||
|
t.Errorf("expected 12 bytes with uncompressed BEQ, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_CJ(t *testing.T) {
|
||||||
|
// JMP target → JAL X0 (never compressed to C.J), matching go tool asm.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·cj(SB), NOSPLIT, $0
|
||||||
|
JMP done
|
||||||
|
done:
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// JAL(4) + JALR(4) = 8
|
||||||
|
if len(code) != 8 {
|
||||||
|
t.Errorf("expected 8 bytes with uncompressed JMP, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_CADD(t *testing.T) {
|
||||||
|
// ADD where rd==rs1 and both in prime regs → C.ADD.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·cadd(SB), NOSPLIT, $0
|
||||||
|
ADD X10, X11, X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// C.ADD(2) + JALR(4) = 6
|
||||||
|
if len(code) != 6 {
|
||||||
|
t.Errorf("expected 6 bytes with C.ADD, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_CADD_commute(t *testing.T) {
|
||||||
|
// ADD where rd==rs2 (commutative swap) → C.ADD.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·cadd2(SB), NOSPLIT, $0
|
||||||
|
ADD X11, X10, X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// C.ADD(2) + JALR(4) = 6
|
||||||
|
if len(code) != 6 {
|
||||||
|
t.Errorf("expected 6 bytes with C.ADD (commuted), got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_CSUB(t *testing.T) {
|
||||||
|
// SUB rs2, rs1, rd → C.SUB when rd == rs1 and both in prime regs.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·csub(SB), NOSPLIT, $0
|
||||||
|
SUB X11, X10, X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// SUB X11, X10, X10 → rs2=X11, rs1=X10, rd=X10; rd==rs1 → C.SUB (2B) + JALR (4B) = 6.
|
||||||
|
if len(code) != 6 {
|
||||||
|
t.Errorf("expected 6 bytes with C.SUB, got %d (% x)", len(code), code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_CXOR(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·cxor(SB), NOSPLIT, $0
|
||||||
|
XOR X10, X11, X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
if len(code) != 6 {
|
||||||
|
t.Errorf("expected 6 bytes with C.XOR, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_COR(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·cor(SB), NOSPLIT, $0
|
||||||
|
OR X10, X11, X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
if len(code) != 6 {
|
||||||
|
t.Errorf("expected 6 bytes with C.OR, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_CAND(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·cand(SB), NOSPLIT, $0
|
||||||
|
AND X10, X11, X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
if len(code) != 6 {
|
||||||
|
t.Errorf("expected 6 bytes with C.AND, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_CFLDSP(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·cfldsp(SB), NOSPLIT, $0-8
|
||||||
|
FLD a+0(FP), F10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// C.FLDSP(2) + JALR(4) = 6
|
||||||
|
if len(code) != 6 {
|
||||||
|
t.Errorf("expected 6 bytes with C.FLDSP, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_RVC_CFSDSP(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·cfsdsp(SB), NOSPLIT, $0-8
|
||||||
|
FSD F10, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// C.FSDSP(2) + JALR(4) = 6
|
||||||
|
if len(code) != 6 {
|
||||||
|
t.Errorf("expected 6 bytes with C.FSDSP, got %d", len(code))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_SB_addr(t *testing.T) {
|
||||||
|
// MOV $sym<>(SB), rd → AUIPC + ADDI (8 bytes for SB).
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·sbaddr(SB), NOSPLIT, $0
|
||||||
|
MOV $answer<>(SB), X10
|
||||||
|
RET
|
||||||
|
GLOBL answer<>(SB), RODATA, $8
|
||||||
|
DATA answer<>+0(SB)/8, $42
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("t_riscv64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
// AUIPC(4) + ADDI(4) + JALR(4) = 12
|
||||||
|
if img.Funcs[0].Size != 12 {
|
||||||
|
t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_SB_store(t *testing.T) {
|
||||||
|
// MOV rd, sym<>(SB) → AUIPC + SD (8 bytes for SB).
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·sbstore(SB), NOSPLIT, $0
|
||||||
|
MOV X10, result<>(SB)
|
||||||
|
RET
|
||||||
|
GLOBL result<>(SB), NOPTR, $8
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("t_riscv64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
// AUIPC X31(4) + SD X10,0(X31)(4) + JALR(4) = 12
|
||||||
|
if img.Funcs[0].Size != 12 {
|
||||||
|
t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_ELF(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·simple(SB), NOSPLIT, $0
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("t_riscv64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFRISCVObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFRISCVObject: %v", err)
|
||||||
|
}
|
||||||
|
if len(obj) < 4 || obj[0] != 0x7f || obj[1] != 'E' || obj[2] != 'L' || obj[3] != 'F' {
|
||||||
|
t.Fatal("not a valid ELF file")
|
||||||
|
}
|
||||||
|
if len(obj) >= 20 {
|
||||||
|
machine := uint16(obj[18]) | uint16(obj[19])<<8
|
||||||
|
if machine != 243 {
|
||||||
|
t.Errorf("e_machine = %d, want 243 (EM_RISCV)", machine)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_ELF_withData(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·get(SB), NOSPLIT, $0
|
||||||
|
RET
|
||||||
|
GLOBL val<>(SB), RODATA, $4
|
||||||
|
DATA val<>+0(SB)/4, $7
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("t_riscv64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
if len(img.DataSyms) != 1 {
|
||||||
|
t.Fatalf("expected 1 data symbol, got %d", len(img.DataSyms))
|
||||||
|
}
|
||||||
|
if img.DataSyms[0].Name != "val" {
|
||||||
|
t.Errorf("data symbol name = %q, want val", img.DataSyms[0].Name)
|
||||||
|
}
|
||||||
|
if img.DataSyms[0].Size != 4 {
|
||||||
|
t.Errorf("data symbol size = %d, want 4", img.DataSyms[0].Size)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFRISCVObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFRISCVObject: %v", err)
|
||||||
|
}
|
||||||
|
_ = obj
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_SB_load(t *testing.T) {
|
||||||
|
// MOV sym<>(SB), rd → AUIPC + LD (8 bytes for SB).
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·sbload(SB), NOSPLIT, $0
|
||||||
|
MOV answer<>(SB), X10
|
||||||
|
RET
|
||||||
|
GLOBL answer<>(SB), RODATA, $8
|
||||||
|
DATA answer<>+0(SB)/8, $42
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("t_riscv64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
// AUIPC(4) + LD(4) + JALR(4) = 12
|
||||||
|
if img.Funcs[0].Size != 12 {
|
||||||
|
t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_system_instrs(t *testing.T) {
|
||||||
|
// Test FENCE, ECALL, EBREAK encoding.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·sys(SB), NOSPLIT, $0
|
||||||
|
FENCE
|
||||||
|
ECALL
|
||||||
|
EBREAK
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// FENCE(4) + ECALL(4) + C.EBREAK(2) + JALR(4) = 14
|
||||||
|
if len(code) != 14 {
|
||||||
|
t.Errorf("expected 14 bytes, got %d (% x)", len(code), code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_MOV_sym_FP_error(t *testing.T) {
|
||||||
|
// MOV $sym(FP), rd should return an error (unsupported).
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·badfp(SB), NOSPLIT, $0
|
||||||
|
MOV $arg(FP), X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
_, _, _, _, _, err := assembleRISCV(fn)
|
||||||
|
if err == nil {
|
||||||
|
t.Error("expected error for MOV $arg(FP), got nil")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_CALL(t *testing.T) {
|
||||||
|
// CALL sym(SB) → JAL X1, sym(SB) with a single R_RISCV_JAL relocation.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·calltest(SB), NOSPLIT, $0
|
||||||
|
CALL ext(SB)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, relocs, _, _, err := assembleRISCV(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
// prologue (8) + JAL (4) + epilogue+JALR (8) = 20
|
||||||
|
if len(code) != 20 {
|
||||||
|
t.Fatalf("expected 20 bytes with CALL sym(SB), got %d", len(code))
|
||||||
|
}
|
||||||
|
if len(relocs) != 1 {
|
||||||
|
t.Fatalf("relocs = %d, want 1", len(relocs))
|
||||||
|
}
|
||||||
|
r := relocs[0]
|
||||||
|
if r.Kind != RelRISCVJal || r.Name != "ext" || r.Off != 8 || r.After != 12 || r.Addend != 0 {
|
||||||
|
t.Errorf("reloc = {kind %v off %d after %d name %q addend %d}", r.Kind, r.Off, r.After, r.Name, r.Addend)
|
||||||
|
}
|
||||||
|
// The JAL instruction itself is JAL X1, 0 at function offset 8.
|
||||||
|
wantJAL := wordLE(riscvJType(1, 0))
|
||||||
|
if !bytes.Equal(code[8:12], wantJAL) {
|
||||||
|
t.Errorf("JAL = % x, want % x", code[8:12], wantJAL)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_CALL_local_error(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·calllocal(SB), NOSPLIT, $0
|
||||||
|
CALL sub
|
||||||
|
sub:
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
_, _, _, _, _, err := assembleRISCV(fn)
|
||||||
|
if err == nil {
|
||||||
|
t.Error("expected error for CALL to local label, got nil")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,182 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
|
)
|
||||||
|
|
||||||
|
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
|
||||||
|
//
|
||||||
|
// Go's riscv64 functions have no hardware frame pointer: FP and SP are
|
||||||
|
// synthetic registers resolved against the hardware stack pointer (X2) and
|
||||||
|
// the frame size. The return address lives in the link register (X1, RA/LR).
|
||||||
|
//
|
||||||
|
// The autosize is the real stack adjustment: the declared local frame plus
|
||||||
|
// the 8 bytes for the saved link register (the toolchain's FixedFrameSize).
|
||||||
|
// A leaf function with a zero frame gets no prologue at all.
|
||||||
|
//
|
||||||
|
// Prologue (autosize > 0), byte-identical to the toolchain:
|
||||||
|
//
|
||||||
|
// MOV LR, -autosize(SP) // save LR below the new SP (traceback-safe)
|
||||||
|
// ADDI $-autosize, SP, SP // open the frame
|
||||||
|
// MOV LR, 0(SP) // save LR again at SP (signal-safety)
|
||||||
|
//
|
||||||
|
// Epilogue (autosize > 0): MOV 0(SP), LR; ADDI $autosize, SP, SP; the RET's
|
||||||
|
// uncompressed JALR X0, 0(X1) follows. The toolchain restores LR on every
|
||||||
|
// frame, leaf or not.
|
||||||
|
|
||||||
|
// riscvFrameInfo holds the frame layout derived from a TEXT directive.
|
||||||
|
type riscvFrameInfo struct {
|
||||||
|
autosize int // the real SP adjustment (locals + saved LR)
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvComputeFrame derives the frame layout for a TEXT function.
|
||||||
|
func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
|
||||||
|
frame := frameSize(t)
|
||||||
|
if frame != 0 || !riscvIsLeaf(t) {
|
||||||
|
// FixedFrameSize = 8: space for the saved link register. A
|
||||||
|
// zero-frame non-leaf function still opens an 8-byte frame for LR.
|
||||||
|
return riscvFrameInfo{autosize: frame + 8}
|
||||||
|
}
|
||||||
|
return riscvFrameInfo{}
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvIsLeaf reports whether a function contains no call instructions.
|
||||||
|
// CALL always links; JAL/JALR link only when their destination register is
|
||||||
|
// the link register (X1), matching cmd/internal/obj/riscv's containsCall.
|
||||||
|
func riscvIsLeaf(t *ast.Text) bool {
|
||||||
|
for _, stmt := range t.Body {
|
||||||
|
in, ok := stmt.(*ast.Instr)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
switch strings.ToUpper(in.Mnemonic.Text) {
|
||||||
|
case "CALL":
|
||||||
|
return false
|
||||||
|
case "JAL":
|
||||||
|
// JAL rd, target — a call only when rd is the link register.
|
||||||
|
if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
case "JALR":
|
||||||
|
// JALR rs1, rd — a call when rd is X1; JALR offset(rs1) always
|
||||||
|
// links to X1.
|
||||||
|
if len(in.Operands) == 1 {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
if len(in.Operands) >= 2 && regFromOperand(in.Operands[1]) == 1 {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvPrologue returns the prologue bytes for a RISC-V function, matching
|
||||||
|
// the toolchain's compression: the SP adjustment compresses to C.ADDI when
|
||||||
|
// the immediate fits, and the second LR save compresses to C.SDSP.
|
||||||
|
func riscvPrologue(fi riscvFrameInfo) []byte {
|
||||||
|
if fi.autosize == 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
var out []byte
|
||||||
|
// MOV LR, -autosize(SP) — SD X1, -autosize(X2). The negative offset is
|
||||||
|
// not compressible to C.SDSP (unsigned), so it stays 4 bytes.
|
||||||
|
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
|
||||||
|
// ADDI $-autosize, SP, SP — open the frame (C.ADDI when it fits).
|
||||||
|
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
|
||||||
|
// MOV LR, 0(SP) — SD X1, 0(X2) → C.SDSP X1, 0.
|
||||||
|
c := rvcSSP(0x7, 1, 0)
|
||||||
|
out = append(out, byte(c), byte(c>>8))
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvReturn returns the bytes for a RET: the epilogue (restore LR and
|
||||||
|
// deallocate the frame when present) followed by the uncompressed JALR X0,
|
||||||
|
// 0(X1) the toolchain emits for RET (it never compresses RET to C.JR).
|
||||||
|
func riscvReturn(fi riscvFrameInfo) []byte {
|
||||||
|
var out []byte
|
||||||
|
if fi.autosize != 0 {
|
||||||
|
// MOV 0(SP), LR — LD X1, 0(X2) → C.LDSP X1, 0.
|
||||||
|
c := rvcLSP(0x3, 1, 0)
|
||||||
|
out = append(out, byte(c), byte(c>>8))
|
||||||
|
// ADDI $autosize, SP, SP — close the frame (C.ADDI when it fits).
|
||||||
|
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
|
||||||
|
}
|
||||||
|
// JALR X0, 0(X1).
|
||||||
|
return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...)
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvSPAdjust emits an ADDI rd, imm, rd for the stack pointer (rd = rs1 =
|
||||||
|
// X2), compressed to C.ADDI16SP when the immediate is a nonzero 16-byte
|
||||||
|
// multiple, else C.ADDI when it fits 6-bit signed.
|
||||||
|
func riscvSPAdjust(imm int32) []byte {
|
||||||
|
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
|
||||||
|
c := rvcADDI16SP(2, imm)
|
||||||
|
return []byte{byte(c), byte(c >> 8)}
|
||||||
|
}
|
||||||
|
if riscvFitsCAddi(imm) {
|
||||||
|
c := rvcCI(0x0, 2, uint32(imm)&0x3F)
|
||||||
|
return []byte{byte(c), byte(c >> 8)}
|
||||||
|
}
|
||||||
|
return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 2, 2, imm))
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvFitsCAddi reports whether imm compresses to C.ADDI (a nonzero 6-bit
|
||||||
|
// signed immediate).
|
||||||
|
func riscvFitsCAddi(imm int32) bool {
|
||||||
|
return imm != 0 && imm >= -32 && imm <= 31
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvPrologueSpadjPC returns the function-relative byte offset where the
|
||||||
|
// prologue has finished decrementing SP (the delta becomes autosize).
|
||||||
|
func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
|
||||||
|
if fi.autosize == 0 {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
||||||
|
return 4 + riscvSPAdjustLen(int32(-fi.autosize))
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
||||||
|
// (but not including) the final JALR — the point where SP is restored.
|
||||||
|
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
|
||||||
|
if fi.autosize == 0 {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
|
||||||
|
return 2 + riscvSPAdjustLen(int32(fi.autosize))
|
||||||
|
}
|
||||||
|
|
||||||
|
func riscvSPAdjustLen(imm int32) int {
|
||||||
|
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if riscvFitsCAddi(imm) {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
return 4
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvResolvePseudo translates a pseudo-register memory reference into a
|
||||||
|
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
|
||||||
|
// x+N(SP) → (N + autosize)(SP). Returns base = -1 for an unresolvable
|
||||||
|
// reference (SB: static data, handled by the relocation path).
|
||||||
|
func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32) {
|
||||||
|
if sym == nil {
|
||||||
|
return -1, 0
|
||||||
|
}
|
||||||
|
switch sym.Pseudo {
|
||||||
|
case "FP":
|
||||||
|
return 2, int32(sym.Offset) + int32(fi.autosize) + 8
|
||||||
|
case "SP":
|
||||||
|
return 2, int32(fi.autosize) + int32(sym.Offset)
|
||||||
|
case "SB":
|
||||||
|
return -1, int32(sym.Offset)
|
||||||
|
}
|
||||||
|
return -1, 0
|
||||||
|
}
|
||||||
@@ -0,0 +1,81 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestRISCVFrameSpadjAndLines checks that a framed function records its
|
||||||
|
// stack-adjustment boundaries and source-line table, the inputs the GOOBJ
|
||||||
|
// emitter turns into the pcsp/pcfile/pcline tables.
|
||||||
|
func TestRISCVFrameSpadjAndLines(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("frame_riscv64.s", `#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·framed(SB), NOSPLIT, $16-16
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
MOV b+8(FP), X11
|
||||||
|
ADD X11, X10, X10
|
||||||
|
MOV X10, ret+16(FP)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
fn := img.Funcs[0]
|
||||||
|
if fn.Size != 24 {
|
||||||
|
t.Fatalf("size = %d, want 24", fn.Size)
|
||||||
|
}
|
||||||
|
|
||||||
|
// autosize = 16 + 8 = 24; the prologue boundary is just past its C.ADDI
|
||||||
|
// (SD 4 + C.ADDI 2 = 6), and the RET restores SP just past its C.ADDI
|
||||||
|
// (RET starts at 16; C.LDSP 2 + C.ADDI 2 = 20).
|
||||||
|
wantSpadj := []SpadjStep{{PC: 6, Value: 24}, {PC: 20, Value: 0}}
|
||||||
|
if len(fn.Spadj) != len(wantSpadj) {
|
||||||
|
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
|
||||||
|
}
|
||||||
|
for i := range wantSpadj {
|
||||||
|
if fn.Spadj[i] != wantSpadj[i] {
|
||||||
|
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// One line entry per instruction, in emission order.
|
||||||
|
wantLines := []LineEntry{
|
||||||
|
{Offset: 8, Line: 4},
|
||||||
|
{Offset: 10, Line: 5},
|
||||||
|
{Offset: 12, Line: 6},
|
||||||
|
{Offset: 14, Line: 7},
|
||||||
|
{Offset: 16, Line: 8},
|
||||||
|
}
|
||||||
|
if len(fn.Lines) != len(wantLines) {
|
||||||
|
t.Fatalf("lines = %v, want %v", fn.Lines, wantLines)
|
||||||
|
}
|
||||||
|
for i := range wantLines {
|
||||||
|
if fn.Lines[i] != wantLines[i] {
|
||||||
|
t.Errorf("lines[%d] = %v, want %v", i, fn.Lines[i], wantLines[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCVRegAliases checks the Go ABI register aliases that the toolchain
|
||||||
|
// defines: LR is the link register (X1) and TMP is the assembler scratch
|
||||||
|
// register (X31/T6).
|
||||||
|
func TestRISCVRegAliases(t *testing.T) {
|
||||||
|
for name, want := range map[string]int{
|
||||||
|
"X1": 1, "RA": 1, "LR": 1,
|
||||||
|
"X31": 31, "T6": 31, "TMP": 31,
|
||||||
|
"X2": 2, "SP": 2,
|
||||||
|
} {
|
||||||
|
if got := riscvRegNum(name); got != want {
|
||||||
|
t.Errorf("riscvRegNum(%q) = %d, want %d", name, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,346 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"debug/elf"
|
||||||
|
"encoding/binary"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
|
||||||
|
// instruction carrying an R_RISCV_JAL relocation (4-byte field) in both the
|
||||||
|
// GOOBJ and ELF object emitters.
|
||||||
|
func TestGOObjectRISCVCallReloc(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("k_riscv64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·c(SB), NOSPLIT, $0-0
|
||||||
|
CALL callee<>(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
GLOBL callee<>(SB), RODATA, $8
|
||||||
|
DATA callee<>+0(SB)/8, $42
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
fn := img.Funcs[0]
|
||||||
|
if len(fn.Relocs) != 1 {
|
||||||
|
t.Fatalf("relocs = %d, want 1", len(fn.Relocs))
|
||||||
|
}
|
||||||
|
r := fn.Relocs[0]
|
||||||
|
if r.Kind != RelRISCVJal || r.Off != 8 || r.After != 12 || r.Name != "callee" || r.Addend != 0 || r.External {
|
||||||
|
t.Errorf("reloc = {kind %v off %d after %d name %q addend %d external %v}", r.Kind, r.Off, r.After, r.Name, r.Addend, r.External)
|
||||||
|
}
|
||||||
|
|
||||||
|
obj, err := img.GOObjectRISCV("testpkg", "k_riscv64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObjectRISCV: %v", err)
|
||||||
|
}
|
||||||
|
v := openGoobj(t, obj)
|
||||||
|
relocIdx := v.blk(blkRelocIdx)
|
||||||
|
relocs := v.blk(blkReloc)
|
||||||
|
// The function is the last non-package symbol: 4 package defs, then the
|
||||||
|
// 4 pc tables and the function.
|
||||||
|
first := int(binary.LittleEndian.Uint32(relocIdx[(4+4)*4:]))
|
||||||
|
if (first+1)*23 > len(relocs) {
|
||||||
|
t.Fatalf("reloc block too short: first=%d len=%d", first, len(relocs))
|
||||||
|
}
|
||||||
|
e := relocs[first*23:]
|
||||||
|
le := binary.LittleEndian
|
||||||
|
if int32(le.Uint32(e[0:])) != 8 || e[4] != 4 || le.Uint16(e[5:]) != relocRISCVJal || le.Uint32(e[15:]) != pkgIdxSelf || le.Uint32(e[19:]) != 0 {
|
||||||
|
t.Errorf("GOOBJ reloc = off %d size %d type %d pkg %d sym %d", int32(le.Uint32(e[0:])), e[4], le.Uint16(e[5:]), le.Uint32(e[15:]), le.Uint32(e[19:]))
|
||||||
|
}
|
||||||
|
|
||||||
|
// The ELF object must carry a single R_RISCV_JAL relocation in .rela.text.
|
||||||
|
elfObj, err := img.ELFRISCVObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFRISCVObject: %v", err)
|
||||||
|
}
|
||||||
|
if !hasELFRISCVJAL(t, elfObj) {
|
||||||
|
t.Error("ELF object missing R_RISCV_JAL relocation")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
func TestGOObjectRISCVStructure(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("k_riscv64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·sb(SB), NOSPLIT, $0-0
|
||||||
|
MOV $answer<>(SB), X10
|
||||||
|
MOV answer<>(SB), X11
|
||||||
|
MOV X12, answer<>(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
GLOBL answer<>(SB), RODATA, $8
|
||||||
|
DATA answer<>+0(SB)/8, $42
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
fn := img.Funcs[0]
|
||||||
|
if fn.Size != 28 {
|
||||||
|
t.Fatalf("function size = %d, want 28", fn.Size)
|
||||||
|
}
|
||||||
|
if len(fn.Relocs) != 3 {
|
||||||
|
t.Fatalf("relocs = %d, want 3", len(fn.Relocs))
|
||||||
|
}
|
||||||
|
wantKind := []RelocKind{RelRISCVPCRELIType, RelRISCVPCRELIType, RelRISCVPCRELSType}
|
||||||
|
wantOff := []int{0, 8, 16}
|
||||||
|
for i, r := range fn.Relocs {
|
||||||
|
if r.Kind != wantKind[i] || r.Off != wantOff[i] || r.After != r.Off+8 || r.Name != "answer" || r.Addend != 0 {
|
||||||
|
t.Errorf("reloc %d = {kind %v off %d after %d name %q addend %d}", i, r.Kind, r.Off, r.After, r.Name, r.Addend)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
obj, err := img.GOObjectRISCV("testpkg", "k_riscv64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObjectRISCV: %v", err)
|
||||||
|
}
|
||||||
|
v := openGoobj(t, obj)
|
||||||
|
|
||||||
|
// Package defs: the static GLOBL, the FuncInfo, then the two DWARF
|
||||||
|
// symbols.
|
||||||
|
defs := v.syms(blkSymdef)
|
||||||
|
if len(defs) != 4 {
|
||||||
|
t.Fatalf("symdefs = %d, want 4", len(defs))
|
||||||
|
}
|
||||||
|
if defs[0].name != "answer" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 8 {
|
||||||
|
t.Errorf("answer symbol = %+v", defs[0])
|
||||||
|
}
|
||||||
|
if defs[2].typ != kindSDWARFLINES || defs[3].typ != kindSDWARFFCN {
|
||||||
|
t.Errorf("dwarf symbols = %+v, %+v", defs[2], defs[3])
|
||||||
|
}
|
||||||
|
|
||||||
|
// The three code relocations, in definition order: ITYPE, ITYPE, STYPE,
|
||||||
|
// each 8 bytes wide against the GLOBL (package symbol 0).
|
||||||
|
relocIdx := v.blk(blkRelocIdx)
|
||||||
|
relocs := v.blk(blkReloc)
|
||||||
|
if len(relocs) != 5*23 {
|
||||||
|
t.Fatalf("relocs = %d bytes, want 5 entries", len(relocs))
|
||||||
|
}
|
||||||
|
// The function is the last non-package symbol; its relocs start after
|
||||||
|
// the DWARF symbols' (defs 2 and 3 each carry one).
|
||||||
|
le := binary.LittleEndian
|
||||||
|
first := int(le.Uint32(relocIdx[4*(4+4):]))
|
||||||
|
wantType := []uint16{relocRISCVPcrelItype, relocRISCVPcrelItype, relocRISCVPcrelStype}
|
||||||
|
wantOffAbs := []int{0, 8, 16}
|
||||||
|
for i := 0; i < 3; i++ {
|
||||||
|
e := relocs[(first+i)*23:]
|
||||||
|
if int32(le.Uint32(e[0:])) != int32(wantOffAbs[i]) || e[4] != 8 || le.Uint16(e[5:]) != wantType[i] ||
|
||||||
|
le.Uint32(e[15:]) != pkgIdxSelf || le.Uint32(e[19:]) != 0 {
|
||||||
|
t.Errorf("reloc %d = off %d size %d type %d pkg %d sym %d", i, int32(le.Uint32(e[0:])), e[4], le.Uint16(e[5:]), le.Uint32(e[15:]), le.Uint32(e[19:]))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The function code: three AUIPC+second-instruction pairs with zero
|
||||||
|
// immediates, then the uncompressed JALR X0, 0(X1) the toolchain emits
|
||||||
|
// for RET.
|
||||||
|
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||||
|
want := append(wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 10, 0)), wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 10, 10, 0))...)
|
||||||
|
want = append(want, wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 11, 0))...)
|
||||||
|
want = append(want, wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 11, 11, 0))...)
|
||||||
|
want = append(want, wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 31, 0))...)
|
||||||
|
want = append(want, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 12, 0))...)
|
||||||
|
want = append(want, 0x67, 0x80, 0x00, 0x00) // JALR X0, 0(X1)
|
||||||
|
if !bytes.Equal(code, want) {
|
||||||
|
t.Errorf("code = % x\nwant % x", code, want)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The same bytes must survive into the object's data block intact: the
|
||||||
|
// linker patches only the immediate fields of the AUIPC pairs, so the
|
||||||
|
// opcode/register bits of every instruction must not be zeroed.
|
||||||
|
dataIdx := v.blk(blkDataIdx)
|
||||||
|
dataBlk := v.blk(blkData)
|
||||||
|
dOff := int(le.Uint32(dataIdx[8*4:])) // the function is the last symbol
|
||||||
|
emitted := dataBlk[dOff : dOff+fn.Size]
|
||||||
|
if !bytes.Equal(emitted, want) {
|
||||||
|
t.Errorf("emitted data = % x\nwant % x", emitted, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGOObjectRISCVLink cross-compiles a Go program with the gasm-produced
|
||||||
|
// object substituted into the package archive, proving cmd/link accepts the
|
||||||
|
// emitted RISC-V GOOBJ. The binary is not executed (no riscv64 host or
|
||||||
|
// qemu). Skipped when no Go toolchain is available.
|
||||||
|
func TestGOObjectRISCVLink(t *testing.T) {
|
||||||
|
goBin, err := exec.LookPath("go")
|
||||||
|
if err != nil {
|
||||||
|
t.Skip("no Go toolchain available")
|
||||||
|
}
|
||||||
|
dir := t.TempDir()
|
||||||
|
asmSrc := `#include "textflag.h"
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
MOV b+8(FP), X11
|
||||||
|
ADD X11, X10, X10
|
||||||
|
MOV X10, ret+16(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main_riscv64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
mainSrc := `package main
|
||||||
|
|
||||||
|
func add(a, b int64) int64
|
||||||
|
|
||||||
|
func main() {
|
||||||
|
if add(20, 22) != 42 {
|
||||||
|
panic("bad add")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module rvlink\n\ngo 1.21\n"), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||||
|
build.Dir = dir
|
||||||
|
build.Env = append(os.Environ(), "GOARCH=riscv64")
|
||||||
|
buildLog, err := build.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||||
|
}
|
||||||
|
var pkgArch, work, linkLine, asmObj string
|
||||||
|
for _, line := range strings.Split(string(buildLog), "\n") {
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(line, "WORK="):
|
||||||
|
work = strings.TrimPrefix(line, "WORK=")
|
||||||
|
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_riscv64.s") && !strings.Contains(line, "-gensymabis"):
|
||||||
|
asmObj = fieldAfter(line, "-o")
|
||||||
|
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
|
||||||
|
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
|
||||||
|
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
|
||||||
|
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||||
|
linkLine = line
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if pkgArch == "" || linkLine == "" || asmObj == "" {
|
||||||
|
t.Skip("could not locate the archive, asm output or link line in the build log")
|
||||||
|
}
|
||||||
|
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
|
||||||
|
asmMember := filepath.Base(strings.ReplaceAll(asmObj, "$WORK", work))
|
||||||
|
|
||||||
|
pf, perrs := parser.Parse(filepath.Join(dir, "main_riscv64.s"), asmSrc)
|
||||||
|
if len(perrs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", perrs)
|
||||||
|
}
|
||||||
|
pimg, err := AssembleFileRISCV(pf)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := pimg.GOObjectRISCV("main", filepath.Join(dir, "main_riscv64.s"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObjectRISCV: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
membersDir := filepath.Join(dir, "members")
|
||||||
|
if err := os.MkdirAll(membersDir, 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
|
||||||
|
extract.Dir = membersDir
|
||||||
|
extract.Env = append(os.Environ(), "GOARCH=riscv64")
|
||||||
|
if out, err := extract.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("pack x: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
member := filepath.Join(membersDir, asmMember)
|
||||||
|
if err := os.Chmod(member, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(member, obj, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
|
||||||
|
listCmd.Env = append(os.Environ(), "GOARCH=riscv64")
|
||||||
|
listOut, err := listCmd.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("pack t: %v\n%s", err, listOut)
|
||||||
|
}
|
||||||
|
newArch := filepath.Join(dir, "pkg.a")
|
||||||
|
args := []string{"tool", "pack", "c", newArch}
|
||||||
|
seen := map[string]bool{}
|
||||||
|
for _, m := range strings.Fields(string(listOut)) {
|
||||||
|
if seen[m] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
seen[m] = true
|
||||||
|
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
args = append(args, filepath.Join(membersDir, m))
|
||||||
|
}
|
||||||
|
pack := exec.Command(goBin, args...)
|
||||||
|
pack.Dir = membersDir
|
||||||
|
pack.Env = append(os.Environ(), "GOARCH=riscv64")
|
||||||
|
if out, err := pack.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("pack c: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
|
||||||
|
link := exec.Command("sh", "-c", linkLine)
|
||||||
|
link.Dir = dir
|
||||||
|
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
|
||||||
|
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)), "GOARCH=riscv64")
|
||||||
|
if out, err := link.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("link with gasm object: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
nm := exec.Command(goBin, "tool", "nm", filepath.Join(dir, "app2"))
|
||||||
|
nm.Env = append(os.Environ(), "GOARCH=riscv64")
|
||||||
|
nmOut, err := nm.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("nm gasm-linked binary: %v\n%s", err, nmOut)
|
||||||
|
}
|
||||||
|
if !strings.Contains(string(nmOut), "main.add") {
|
||||||
|
t.Errorf("main.add not found in linked binary:\n%s", nmOut)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// hasELFRISCVJAL reports whether the ELF object carries an R_RISCV_JAL
|
||||||
|
// relocation in its .rela.text section.
|
||||||
|
func hasELFRISCVJAL(t *testing.T, data []byte) bool {
|
||||||
|
t.Helper()
|
||||||
|
f, err := elf.NewFile(bytes.NewReader(data))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse ELF: %v", err)
|
||||||
|
}
|
||||||
|
defer f.Close()
|
||||||
|
rela := f.Section(".rela.text")
|
||||||
|
if rela == nil {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
b, err := rela.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf(".rela.text data: %v", err)
|
||||||
|
}
|
||||||
|
const rRISCVJAL = 17
|
||||||
|
for i := 0; i+24 <= len(b); i += 24 {
|
||||||
|
info := binary.LittleEndian.Uint64(b[i+8:])
|
||||||
|
if uint32(info) == rRISCVJAL {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
@@ -0,0 +1,193 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build linux && amd64
|
||||||
|
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"sort"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/debug"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
||||||
|
)
|
||||||
|
|
||||||
|
func cmdDebug(args []string) int {
|
||||||
|
fs := newCommand("debug", "gasm debug <file.s> --func <name>", `
|
||||||
|
Interactive debugger for JIT-assembled amd64 functions. Launches the
|
||||||
|
function in a traced subprocess (ptrace), then provides a REPL for
|
||||||
|
single-stepping, breakpoints, register and memory inspection.
|
||||||
|
|
||||||
|
REPL commands:
|
||||||
|
break <label|addr> set a breakpoint at a label or absolute address
|
||||||
|
step [n] single-step n instructions (default 1)
|
||||||
|
continue run until next breakpoint or exit
|
||||||
|
regs print general-purpose registers
|
||||||
|
x [addr] [len] hex-dump memory (default: current PC, 64 bytes)
|
||||||
|
labels list function labels and offsets
|
||||||
|
quit kill the debuggee and exit
|
||||||
|
`)
|
||||||
|
target := fs.Bool("target", false, "") // hidden: debuggee subprocess mode
|
||||||
|
funcName := fs.String("func", "", "function to debug")
|
||||||
|
argsFile := fs.String("args", "", "file containing the ABI0 argument block")
|
||||||
|
bufSpec := fs.String("buf", "", "buffer specification: name:size:pattern[,name:size:pattern...] where pattern is zero, ones, seq, or hex")
|
||||||
|
fs.Parse(args)
|
||||||
|
|
||||||
|
// --- Debuggee mode (internal, spawned by the debugger) ---
|
||||||
|
if *target {
|
||||||
|
tmpDir := os.Getenv("GASM_DEBUG_TMP")
|
||||||
|
if tmpDir == "" || fs.NArg() < 1 || *funcName == "" || *argsFile == "" {
|
||||||
|
fmt.Fprintln(os.Stderr, "gasm debug --target: internal mode")
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if err := debug.RunTarget(fs.Arg(0), *funcName, *argsFile, tmpDir); err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- Debugger mode (interactive REPL) ---
|
||||||
|
if fs.NArg() < 1 || *funcName == "" {
|
||||||
|
fmt.Fprintln(os.Stderr, "usage: gasm debug <file.s> --func <name>")
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
path := fs.Arg(0)
|
||||||
|
|
||||||
|
// Load the kernel to extract function metadata and labels.
|
||||||
|
k, err := verify.Load(path)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
defer k.Close()
|
||||||
|
|
||||||
|
fl, err := k.Func(*funcName)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
// Build the label list for the REPL.
|
||||||
|
var labels []debug.Label
|
||||||
|
for name, off := range fl.Labels {
|
||||||
|
labels = append(labels, debug.Label{Name: name, Offset: off})
|
||||||
|
}
|
||||||
|
sort.Slice(labels, func(i, j int) bool { return labels[i].Offset < labels[j].Offset })
|
||||||
|
|
||||||
|
// Launch the debuggee with the argument block.
|
||||||
|
var argBlock []byte
|
||||||
|
var bufAddrs []uint64
|
||||||
|
var sess *debug.Session
|
||||||
|
if *bufSpec != "" {
|
||||||
|
// Parse the function signature to determine argument layout.
|
||||||
|
src, err := readSource(path)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
sig, ok := verify.ExtractFuncSig(src, *funcName)
|
||||||
|
if !ok {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm debug: no // func signature found for %s\n", *funcName)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
layout := verify.ArgLayout(sig)
|
||||||
|
|
||||||
|
// Parse the buffer spec to get buffer names.
|
||||||
|
bufNames := parseBufNames(*bufSpec)
|
||||||
|
|
||||||
|
// Allocate buffers in the debuggee.
|
||||||
|
argBlock = make([]byte, fl.Args)
|
||||||
|
sess, bufAddrs, err = debug.LaunchWithBuffers("", path, *funcName, argBlock, *bufSpec)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
// Construct the argument block with buffer pointers at the correct positions.
|
||||||
|
bufIdx := 0
|
||||||
|
for _, arg := range layout {
|
||||||
|
if !arg.IsPtr {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Find the buffer that matches this argument.
|
||||||
|
for i, name := range bufNames {
|
||||||
|
if i < len(bufAddrs) && (name == arg.Name || strings.HasPrefix(arg.Name, name)) {
|
||||||
|
addr := bufAddrs[i]
|
||||||
|
off := arg.Offset
|
||||||
|
if off+8 <= len(argBlock) {
|
||||||
|
argBlock[off] = byte(addr)
|
||||||
|
argBlock[off+1] = byte(addr >> 8)
|
||||||
|
argBlock[off+2] = byte(addr >> 16)
|
||||||
|
argBlock[off+3] = byte(addr >> 24)
|
||||||
|
argBlock[off+4] = byte(addr >> 32)
|
||||||
|
argBlock[off+5] = byte(addr >> 40)
|
||||||
|
argBlock[off+6] = byte(addr >> 48)
|
||||||
|
argBlock[off+7] = byte(addr >> 56)
|
||||||
|
}
|
||||||
|
// For slices, also set the length and capacity.
|
||||||
|
if strings.HasPrefix(arg.Typ, "[]") && off+24 <= len(argBlock) {
|
||||||
|
// Find the buffer size from the spec.
|
||||||
|
size := parseBufSize(*bufSpec, name)
|
||||||
|
// Length at offset+8, capacity at offset+16.
|
||||||
|
for j := 0; j < 8; j++ {
|
||||||
|
argBlock[off+8+j] = byte(size >> (j * 8))
|
||||||
|
argBlock[off+16+j] = byte(size >> (j * 8))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
bufIdx++
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ = bufIdx
|
||||||
|
} else {
|
||||||
|
argBlock = make([]byte, fl.Args)
|
||||||
|
sess, err = debug.Launch("", path, *funcName, argBlock)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
defer sess.Kill()
|
||||||
|
|
||||||
|
bm := debug.NewBreakpoints(sess)
|
||||||
|
fmt.Printf("gasm debug: %s in %s (pid %d)\n", *funcName, path, sess.Pid())
|
||||||
|
|
||||||
|
// Convert the line table for the REPL.
|
||||||
|
var srcLines []debug.SourceLine
|
||||||
|
for _, le := range fl.Lines {
|
||||||
|
srcLines = append(srcLines, debug.SourceLine{Offset: le.Offset, Line: le.Line})
|
||||||
|
}
|
||||||
|
debug.REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, labels, srcLines)
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseBufNames extracts buffer names from a buffer specification.
|
||||||
|
// Format: name:size:pattern[,name:size:pattern...]
|
||||||
|
func parseBufNames(spec string) []string {
|
||||||
|
var names []string
|
||||||
|
for _, part := range strings.Split(spec, ",") {
|
||||||
|
fields := strings.SplitN(part, ":", 3)
|
||||||
|
if len(fields) >= 1 && fields[0] != "" {
|
||||||
|
names = append(names, fields[0])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return names
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseBufSize extracts the size of a named buffer from a buffer specification.
|
||||||
|
func parseBufSize(spec, name string) int {
|
||||||
|
for _, part := range strings.Split(spec, ",") {
|
||||||
|
fields := strings.SplitN(part, ":", 3)
|
||||||
|
if len(fields) >= 2 && fields[0] == name {
|
||||||
|
var size int
|
||||||
|
fmt.Sscanf(fields[1], "%d", &size)
|
||||||
|
return size
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build !(linux && amd64)
|
||||||
|
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
)
|
||||||
|
|
||||||
|
func cmdDebug(args []string) int {
|
||||||
|
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires linux/amd64 (ptrace)")
|
||||||
|
return 1
|
||||||
|
}
|
||||||
+842
-61
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,248 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import "fmt"
|
||||||
|
|
||||||
|
// Breakpoint is one INT3 breakpoint in the debuggee.
|
||||||
|
type Breakpoint struct {
|
||||||
|
Addr uint64 // absolute address in the debuggee
|
||||||
|
Label string // source label ("" for raw addresses)
|
||||||
|
Orig byte // original byte at Addr (restored on removal)
|
||||||
|
Enabled bool
|
||||||
|
Cond *Condition // optional condition (nil = unconditional)
|
||||||
|
hits int
|
||||||
|
}
|
||||||
|
|
||||||
|
// Condition is a simple register-comparison condition evaluated when a
|
||||||
|
// breakpoint is hit. Format: <reg> <op> <value>.
|
||||||
|
type Condition struct {
|
||||||
|
Reg string // register name (rax, rbx, rip, rsp, ...)
|
||||||
|
Op string // comparison operator: ==, !=, <, >, <=, >=
|
||||||
|
Value uint64
|
||||||
|
}
|
||||||
|
|
||||||
|
// Eval checks the condition against the current registers.
|
||||||
|
func (c *Condition) Eval(regs *Regs) bool {
|
||||||
|
var actual uint64
|
||||||
|
switch c.Reg {
|
||||||
|
case "rax", "eax", "ax", "al":
|
||||||
|
actual = regs.RAX
|
||||||
|
case "rbx", "ebx", "bx", "bl":
|
||||||
|
actual = regs.RBX
|
||||||
|
case "rcx", "ecx", "cx", "cl":
|
||||||
|
actual = regs.RCX
|
||||||
|
case "rdx", "edx", "dx", "dl":
|
||||||
|
actual = regs.RDX
|
||||||
|
case "rsi", "esi", "si":
|
||||||
|
actual = regs.RSI
|
||||||
|
case "rdi", "edi", "di":
|
||||||
|
actual = regs.RDI
|
||||||
|
case "rbp", "ebp", "bp":
|
||||||
|
actual = regs.RBP
|
||||||
|
case "rsp", "esp", "sp":
|
||||||
|
actual = regs.RSP
|
||||||
|
case "r8":
|
||||||
|
actual = regs.R8
|
||||||
|
case "r9":
|
||||||
|
actual = regs.R9
|
||||||
|
case "r10":
|
||||||
|
actual = regs.R10
|
||||||
|
case "r11":
|
||||||
|
actual = regs.R11
|
||||||
|
case "r12":
|
||||||
|
actual = regs.R12
|
||||||
|
case "r13":
|
||||||
|
actual = regs.R13
|
||||||
|
case "r14":
|
||||||
|
actual = regs.R14
|
||||||
|
case "r15":
|
||||||
|
actual = regs.R15
|
||||||
|
case "rip", "eip":
|
||||||
|
actual = regs.RIP
|
||||||
|
default:
|
||||||
|
return true // unknown register — don't block
|
||||||
|
}
|
||||||
|
switch c.Op {
|
||||||
|
case "==", "=":
|
||||||
|
return actual == c.Value
|
||||||
|
case "!=":
|
||||||
|
return actual != c.Value
|
||||||
|
case "<":
|
||||||
|
return actual < c.Value
|
||||||
|
case ">":
|
||||||
|
return actual > c.Value
|
||||||
|
case "<=":
|
||||||
|
return actual <= c.Value
|
||||||
|
case ">=":
|
||||||
|
return actual >= c.Value
|
||||||
|
default:
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Breakpoints manages the set of breakpoints for a Session.
|
||||||
|
// Breakpoints manages software breakpoints for a debuggee.
|
||||||
|
type Breakpoints struct {
|
||||||
|
t tracer
|
||||||
|
bps map[uint64]*Breakpoint
|
||||||
|
}
|
||||||
|
|
||||||
|
// NewBreakpoints creates a new breakpoint manager.
|
||||||
|
func NewBreakpoints(t tracer) *Breakpoints {
|
||||||
|
return &Breakpoints{t: t, bps: make(map[uint64]*Breakpoint)}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Set installs a breakpoint at addr (replaces any existing one).
|
||||||
|
func (bm *Breakpoints) Set(addr uint64, label string) (*Breakpoint, error) {
|
||||||
|
return bm.SetWithCond(addr, label, nil)
|
||||||
|
}
|
||||||
|
|
||||||
|
// SetWithCond installs a breakpoint with an optional condition.
|
||||||
|
func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (*Breakpoint, error) {
|
||||||
|
if bp, ok := bm.bps[addr]; ok {
|
||||||
|
bp.Enabled = true
|
||||||
|
bp.Cond = cond
|
||||||
|
return bp, nil
|
||||||
|
}
|
||||||
|
// Read the original byte.
|
||||||
|
word, err := bm.t.Peek(addr)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
orig := byte(word)
|
||||||
|
// Patch with INT3 (0xCC), preserving the rest of the word.
|
||||||
|
patched := (word &^ 0xFF) | 0xCC
|
||||||
|
if err := bm.t.Poke(addr, patched); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
bp := &Breakpoint{Addr: addr, Label: label, Orig: orig, Enabled: true, Cond: cond}
|
||||||
|
bm.bps[addr] = bp
|
||||||
|
return bp, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Hits returns the number of times the breakpoint has been hit.
|
||||||
|
func (bp *Breakpoint) Hits() int {
|
||||||
|
return bp.hits
|
||||||
|
}
|
||||||
|
|
||||||
|
// Info returns a formatted list of all breakpoints.
|
||||||
|
func (bm *Breakpoints) Info() string {
|
||||||
|
if len(bm.bps) == 0 {
|
||||||
|
return "no breakpoints set\n"
|
||||||
|
}
|
||||||
|
result := ""
|
||||||
|
i := 0
|
||||||
|
for _, bp := range bm.bps {
|
||||||
|
i++
|
||||||
|
status := "enabled"
|
||||||
|
if !bp.Enabled {
|
||||||
|
status = "disabled"
|
||||||
|
}
|
||||||
|
label := bp.Label
|
||||||
|
if label == "" {
|
||||||
|
label = fmt.Sprintf("%#x", bp.Addr)
|
||||||
|
}
|
||||||
|
cond := ""
|
||||||
|
if bp.Cond != nil {
|
||||||
|
cond = fmt.Sprintf(" if %s %s %#x", bp.Cond.Reg, bp.Cond.Op, bp.Cond.Value)
|
||||||
|
}
|
||||||
|
result += fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond)
|
||||||
|
}
|
||||||
|
return result
|
||||||
|
}
|
||||||
|
|
||||||
|
// Clear removes the breakpoint at addr, restoring the original byte.
|
||||||
|
func (bm *Breakpoints) Clear(addr uint64) error {
|
||||||
|
bp, ok := bm.bps[addr]
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("debug: no breakpoint at %#x", addr)
|
||||||
|
}
|
||||||
|
word, err := bm.t.Peek(addr)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
||||||
|
if err := bm.t.Poke(addr, restored); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
delete(bm.bps, addr)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// ClearAll removes all breakpoints.
|
||||||
|
func (bm *Breakpoints) ClearAll() error {
|
||||||
|
for addr := range bm.bps {
|
||||||
|
if err := bm.Clear(addr); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// At returns the breakpoint at addr, if any.
|
||||||
|
func (bm *Breakpoints) At(addr uint64) *Breakpoint {
|
||||||
|
return bm.bps[addr]
|
||||||
|
}
|
||||||
|
|
||||||
|
// All returns all breakpoints.
|
||||||
|
func (bm *Breakpoints) All() []*Breakpoint {
|
||||||
|
out := make([]*Breakpoint, 0, len(bm.bps))
|
||||||
|
for _, bp := range bm.bps {
|
||||||
|
out = append(out, bp)
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// HandleTrap is called after the debuggee stops on SIGTRAP. It checks
|
||||||
|
// whether the trap was caused by one of our breakpoints (RIP-1 matches
|
||||||
|
// a breakpoint address), restores the original byte, rewinds RIP, and
|
||||||
|
// returns the breakpoint that was hit (or nil if it was a single-step).
|
||||||
|
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
|
||||||
|
// After INT3, RIP points to the byte AFTER the 0xCC.
|
||||||
|
trapAddr := regs.RIP - 1
|
||||||
|
bp, ok := bm.bps[trapAddr]
|
||||||
|
if !ok || !bp.Enabled {
|
||||||
|
return nil // single-step trap or unknown
|
||||||
|
}
|
||||||
|
// Check the condition (if any).
|
||||||
|
if bp.Cond != nil && !bp.Cond.Eval(regs) {
|
||||||
|
// Condition not met — restore the byte but do NOT rewind RIP.
|
||||||
|
// The process continues from the next instruction (past the INT3).
|
||||||
|
word, err := bm.t.Peek(trapAddr)
|
||||||
|
if err == nil {
|
||||||
|
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
||||||
|
bm.t.Poke(trapAddr, restored)
|
||||||
|
}
|
||||||
|
// RIP is already past the INT3 (trapAddr + 1). Don't rewind.
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
bp.hits++
|
||||||
|
// Restore the original byte.
|
||||||
|
word, err := bm.t.Peek(trapAddr)
|
||||||
|
if err == nil {
|
||||||
|
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
||||||
|
bm.t.Poke(trapAddr, restored)
|
||||||
|
}
|
||||||
|
// Rewind RIP to re-execute the original instruction.
|
||||||
|
regs.RIP = trapAddr
|
||||||
|
bm.t.SetRegs(regs)
|
||||||
|
return bp
|
||||||
|
}
|
||||||
|
|
||||||
|
// Reinsert re-inserts the breakpoint at addr after a single-step past it.
|
||||||
|
// Called after Step() when we want the breakpoint to fire again on the
|
||||||
|
// next Continue().
|
||||||
|
func (bm *Breakpoints) Reinsert(addr uint64) error {
|
||||||
|
bp, ok := bm.bps[addr]
|
||||||
|
if !ok || !bp.Enabled {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
word, err := bm.t.Peek(addr)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
patched := (word &^ 0xFF) | 0xCC
|
||||||
|
return bm.t.Poke(addr, patched)
|
||||||
|
}
|
||||||
@@ -0,0 +1,323 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestConditionEval(t *testing.T) {
|
||||||
|
regs := &Regs{
|
||||||
|
RAX: 42,
|
||||||
|
RBX: 0,
|
||||||
|
RCX: 100,
|
||||||
|
RIP: 0x1000,
|
||||||
|
RSP: 0x2000,
|
||||||
|
R8: 8,
|
||||||
|
R15: 15,
|
||||||
|
}
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
cond Condition
|
||||||
|
want bool
|
||||||
|
}{
|
||||||
|
{Condition{Reg: "rax", Op: "==", Value: 42}, true},
|
||||||
|
{Condition{Reg: "rax", Op: "==", Value: 43}, false},
|
||||||
|
{Condition{Reg: "rax", Op: "!=", Value: 43}, true},
|
||||||
|
{Condition{Reg: "rax", Op: "!=", Value: 42}, false},
|
||||||
|
{Condition{Reg: "rax", Op: "<", Value: 50}, true},
|
||||||
|
{Condition{Reg: "rax", Op: "<", Value: 40}, false},
|
||||||
|
{Condition{Reg: "rax", Op: ">", Value: 40}, true},
|
||||||
|
{Condition{Reg: "rax", Op: ">", Value: 50}, false},
|
||||||
|
{Condition{Reg: "rax", Op: "<=", Value: 42}, true},
|
||||||
|
{Condition{Reg: "rax", Op: ">=", Value: 42}, true},
|
||||||
|
{Condition{Reg: "rbx", Op: "==", Value: 0}, true},
|
||||||
|
{Condition{Reg: "rcx", Op: ">", Value: 50}, true},
|
||||||
|
{Condition{Reg: "rip", Op: "==", Value: 0x1000}, true},
|
||||||
|
{Condition{Reg: "rsp", Op: ">", Value: 0x1000}, true},
|
||||||
|
{Condition{Reg: "r8", Op: "==", Value: 8}, true},
|
||||||
|
{Condition{Reg: "r15", Op: "==", Value: 15}, true},
|
||||||
|
{Condition{Reg: "eax", Op: "==", Value: 42}, true}, // 32-bit alias
|
||||||
|
{Condition{Reg: "ax", Op: "==", Value: 42}, true}, // 16-bit alias
|
||||||
|
{Condition{Reg: "unknown", Op: "==", Value: 0}, true}, // unknown reg → don't block
|
||||||
|
{Condition{Reg: "rax", Op: "??", Value: 0}, true}, // unknown op → don't block
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
got := tt.cond.Eval(regs)
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("Condition{%q %q %d}.Eval() = %v, want %v",
|
||||||
|
tt.cond.Reg, tt.cond.Op, tt.cond.Value, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLineAt(t *testing.T) {
|
||||||
|
lines := []SourceLine{
|
||||||
|
{Offset: 0, Line: 5},
|
||||||
|
{Offset: 5, Line: 6},
|
||||||
|
{Offset: 10, Line: 7},
|
||||||
|
{Offset: 15, Line: 8},
|
||||||
|
}
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
offset int
|
||||||
|
want int
|
||||||
|
}{
|
||||||
|
{0, 5},
|
||||||
|
{1, 5},
|
||||||
|
{4, 5},
|
||||||
|
{5, 6},
|
||||||
|
{7, 6},
|
||||||
|
{10, 7},
|
||||||
|
{12, 7},
|
||||||
|
{15, 8},
|
||||||
|
{20, 8},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
got := lineAt(lines, tt.offset)
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Empty table.
|
||||||
|
if lineAt(nil, 5) != 0 {
|
||||||
|
t.Error("lineAt(nil, 5) should return 0")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestOffsetForLine(t *testing.T) {
|
||||||
|
lines := []SourceLine{
|
||||||
|
{Offset: 0, Line: 5},
|
||||||
|
{Offset: 5, Line: 6},
|
||||||
|
{Offset: 10, Line: 7},
|
||||||
|
}
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
line int
|
||||||
|
want int
|
||||||
|
}{
|
||||||
|
{5, 0},
|
||||||
|
{6, 5},
|
||||||
|
{7, 10},
|
||||||
|
{99, -1}, // not found
|
||||||
|
{0, -1}, // not found
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
got := offsetForLine(lines, tt.line)
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDecodeRflags(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
flags uint64
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{0x202, "IF"}, // only IF set (bit 9)
|
||||||
|
{0x246, "PF ZF IF"}, // PF(2) + ZF(6) + IF(9)
|
||||||
|
{0x001, "CF"}, // carry flag
|
||||||
|
{0x080, "SF"}, // sign flag
|
||||||
|
{0x800, "OF"}, // overflow flag
|
||||||
|
{0x000, "none"}, // no flags
|
||||||
|
{0x202 | 0x001, "CF IF"}, // CF + IF
|
||||||
|
{0x3F7, "CF PF AF ZF SF TF IF"}, // all arithmetic flags
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
got := decodeRflags(tt.flags)
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("decodeRflags(%#x) = %q, want %q", tt.flags, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestNearestLabel(t *testing.T) {
|
||||||
|
labels := []Label{
|
||||||
|
{Name: "start", Offset: 0},
|
||||||
|
{Name: "loop", Offset: 10},
|
||||||
|
{Name: "done", Offset: 20},
|
||||||
|
}
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
offset int
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{0, "start"},
|
||||||
|
{5, "start"},
|
||||||
|
{10, "loop"},
|
||||||
|
{15, "loop"},
|
||||||
|
{20, "done"},
|
||||||
|
{25, "done"},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
got := nearestLabel(labels, tt.offset)
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBreakpointsSetAndClear(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
|
||||||
|
// Set a breakpoint at address 0x1000.
|
||||||
|
bp, err := bm.Set(0x1000, "test")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Set: %v", err)
|
||||||
|
}
|
||||||
|
if !bp.Enabled {
|
||||||
|
t.Error("breakpoint not enabled")
|
||||||
|
}
|
||||||
|
if bp.Label != "test" {
|
||||||
|
t.Errorf("label = %q, want test", bp.Label)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Verify Peek was called.
|
||||||
|
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
|
||||||
|
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Verify Poke wrote INT3.
|
||||||
|
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
|
||||||
|
t.Errorf("pokes = %v", tr.pokes)
|
||||||
|
}
|
||||||
|
|
||||||
|
// At should find it.
|
||||||
|
if bm.At(0x1000) == nil {
|
||||||
|
t.Error("At(0x1000) returned nil")
|
||||||
|
}
|
||||||
|
|
||||||
|
// All should return it.
|
||||||
|
all := bm.All()
|
||||||
|
if len(all) != 1 {
|
||||||
|
t.Errorf("All() = %d breakpoints, want 1", len(all))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Clear it.
|
||||||
|
if err := bm.Clear(0x1000); err != nil {
|
||||||
|
t.Fatalf("Clear: %v", err)
|
||||||
|
}
|
||||||
|
if bm.At(0x1000) != nil {
|
||||||
|
t.Error("At(0x1000) after Clear should be nil")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBreakpointsSetWithCond(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
|
||||||
|
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
|
||||||
|
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("SetWithCond: %v", err)
|
||||||
|
}
|
||||||
|
if bp.Cond == nil || bp.Cond.Value != 42 {
|
||||||
|
t.Error("condition not set")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Re-setting the same address should update the condition.
|
||||||
|
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
|
||||||
|
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("SetWithCond (update): %v", err)
|
||||||
|
}
|
||||||
|
if bp2.Cond.Value != 100 {
|
||||||
|
t.Error("condition not updated")
|
||||||
|
}
|
||||||
|
// Should have only 1 Peek (first Set), second is update (no Peek needed).
|
||||||
|
if len(tr.peeks) != 1 {
|
||||||
|
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBreakpointsClearAll(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
|
||||||
|
bm.Set(0x1000, "a")
|
||||||
|
bm.Set(0x2000, "b")
|
||||||
|
bm.Set(0x3000, "c")
|
||||||
|
|
||||||
|
if len(bm.All()) != 3 {
|
||||||
|
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
|
||||||
|
}
|
||||||
|
|
||||||
|
bm.ClearAll()
|
||||||
|
if len(bm.All()) != 0 {
|
||||||
|
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBreakpointInfo(t *testing.T) {
|
||||||
|
tr := newMockTracer()
|
||||||
|
bm := NewBreakpoints(tr)
|
||||||
|
bm.Set(0x4000, "info_test")
|
||||||
|
|
||||||
|
info := bm.Info()
|
||||||
|
if info == "" {
|
||||||
|
t.Error("Info returned empty string")
|
||||||
|
}
|
||||||
|
if !strings.Contains(info, "info_test") {
|
||||||
|
t.Errorf("Info %q does not contain label", info)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestWatchpointSlotTracking(t *testing.T) {
|
||||||
|
s := &Session{}
|
||||||
|
|
||||||
|
// All four slots are free initially.
|
||||||
|
for i := 0; i < 4; i++ {
|
||||||
|
if s.IsWatchpointSlotUsed(i) {
|
||||||
|
t.Errorf("slot %d should be free initially", i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if got := s.FindFreeWatchpointSlot(); got != 0 {
|
||||||
|
t.Errorf("FindFreeWatchpointSlot() = %d, want 0", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Manually mark slots 0 and 2 as used (simulating successful SetWatchpoint).
|
||||||
|
s.wpSlots[0] = true
|
||||||
|
s.wpSlots[2] = true
|
||||||
|
|
||||||
|
if !s.IsWatchpointSlotUsed(0) {
|
||||||
|
t.Error("slot 0 should be in use")
|
||||||
|
}
|
||||||
|
if s.IsWatchpointSlotUsed(1) {
|
||||||
|
t.Error("slot 1 should be free")
|
||||||
|
}
|
||||||
|
if !s.IsWatchpointSlotUsed(2) {
|
||||||
|
t.Error("slot 2 should be in use")
|
||||||
|
}
|
||||||
|
if s.IsWatchpointSlotUsed(3) {
|
||||||
|
t.Error("slot 3 should be free")
|
||||||
|
}
|
||||||
|
if got := s.FindFreeWatchpointSlot(); got != 1 {
|
||||||
|
t.Errorf("FindFreeWatchpointSlot() = %d, want 1", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Out-of-range slot queries return false.
|
||||||
|
if s.IsWatchpointSlotUsed(-1) {
|
||||||
|
t.Error("slot -1 should be reported as free (out of range)")
|
||||||
|
}
|
||||||
|
if s.IsWatchpointSlotUsed(4) {
|
||||||
|
t.Error("slot 4 should be reported as free (out of range)")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Mark all slots used: FindFreeWatchpointSlot returns -1.
|
||||||
|
for i := 0; i < 4; i++ {
|
||||||
|
s.wpSlots[i] = true
|
||||||
|
}
|
||||||
|
if got := s.FindFreeWatchpointSlot(); got != -1 {
|
||||||
|
t.Errorf("FindFreeWatchpointSlot() with all slots used = %d, want -1", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build linux && amd64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
|
||||||
|
"golang.org/x/arch/x86/x86asm"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||||
|
// memory and returns its text representation and length in bytes.
|
||||||
|
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||||
|
// Read up to 15 bytes (max x86 instruction length).
|
||||||
|
mem, err := s.ReadMemory(addr, 15)
|
||||||
|
if err != nil {
|
||||||
|
// Try a shorter read if we're near a page boundary.
|
||||||
|
mem, err = s.ReadMemory(addr, 1)
|
||||||
|
if err != nil {
|
||||||
|
return "", 0, err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
inst, err := x86asm.Decode(mem, 64)
|
||||||
|
if err != nil {
|
||||||
|
return "???", 1, nil
|
||||||
|
}
|
||||||
|
text := x86asm.IntelSyntax(inst, addr, nil)
|
||||||
|
return text, inst.Len, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||||
|
// them as a formatted string with addresses and byte offsets.
|
||||||
|
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||||
|
var result string
|
||||||
|
pc := addr
|
||||||
|
for i := 0; i < n; i++ {
|
||||||
|
text, length, err := s.Disassemble(pc)
|
||||||
|
if err != nil {
|
||||||
|
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
|
||||||
|
break
|
||||||
|
}
|
||||||
|
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
|
||||||
|
if length == 0 {
|
||||||
|
length = 1
|
||||||
|
}
|
||||||
|
pc += uint64(length)
|
||||||
|
}
|
||||||
|
return result
|
||||||
|
}
|
||||||
@@ -0,0 +1,412 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build linux && amd64
|
||||||
|
|
||||||
|
// Package debug implements the interactive debugger for gasm (Phase 4):
|
||||||
|
// single-stepping, breakpoints, register and memory inspection for
|
||||||
|
// JIT-assembled Plan 9 amd64 functions, controlled via ptrace.
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"syscall"
|
||||||
|
"time"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Session is a ptrace debugging session controlling one debuggee process.
|
||||||
|
type Session struct {
|
||||||
|
pid int
|
||||||
|
cmd *exec.Cmd
|
||||||
|
stopped bool
|
||||||
|
exited bool
|
||||||
|
codeBase uint64 // base address of the JIT code in the debuggee
|
||||||
|
wpSlots [4]bool // watchpoint slot occupancy (DR0-DR3)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Launch starts the debuggee subprocess (gasm debug --target ...) and
|
||||||
|
// attaches to it via ptrace. The debuggee assembles the file, maps the
|
||||||
|
// JIT code, calls PTRACE_TRACEME and raises SIGSTOP; Launch waits for
|
||||||
|
// that initial stop and returns a ready Session.
|
||||||
|
func Launch(gasmBin, asmPath, funcName string, args []byte) (*Session, error) {
|
||||||
|
sess, _, err := LaunchWithBuffers(gasmBin, asmPath, funcName, args, "")
|
||||||
|
return sess, err
|
||||||
|
}
|
||||||
|
|
||||||
|
// LaunchWithBuffers is like Launch but also allocates buffers in the debuggee
|
||||||
|
// based on the buffer specification. Returns the Session and the buffer
|
||||||
|
// addresses (in the order they appear in the spec).
|
||||||
|
func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec string) (*Session, []uint64, error) {
|
||||||
|
self, err := os.Executable()
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, fmt.Errorf("debug: cannot find gasm binary: %w", err)
|
||||||
|
}
|
||||||
|
if gasmBin != "" {
|
||||||
|
self = gasmBin
|
||||||
|
}
|
||||||
|
|
||||||
|
// Write the arg block to a temp file (the child reads it).
|
||||||
|
tmpDir, err := os.MkdirTemp("", "gasm-debug-*")
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, fmt.Errorf("debug: tempdir: %w", err)
|
||||||
|
}
|
||||||
|
argsFile := filepath.Join(tmpDir, "args.bin")
|
||||||
|
if err := os.WriteFile(argsFile, args, 0o644); err != nil {
|
||||||
|
os.RemoveAll(tmpDir)
|
||||||
|
return nil, nil, fmt.Errorf("debug: write args: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Write the buffer spec if present.
|
||||||
|
if bufSpec != "" {
|
||||||
|
if err := os.WriteFile(filepath.Join(tmpDir, "bufspec"), []byte(bufSpec), 0o644); err != nil {
|
||||||
|
os.RemoveAll(tmpDir)
|
||||||
|
return nil, nil, fmt.Errorf("debug: write bufspec: %w", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
cmd := exec.Command(self, "debug", "--target", "--func", funcName, "--args", argsFile, asmPath)
|
||||||
|
cmd.Env = append(os.Environ(), "GASM_DEBUG_TMP="+tmpDir)
|
||||||
|
cmd.Stdout = nil // output goes to the debugger, not the terminal
|
||||||
|
cmd.Stderr = os.Stderr
|
||||||
|
cmd.SysProcAttr = &syscall.SysProcAttr{}
|
||||||
|
|
||||||
|
if err := cmd.Start(); err != nil {
|
||||||
|
os.RemoveAll(tmpDir)
|
||||||
|
return nil, nil, fmt.Errorf("debug: start debuggee: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
s := &Session{pid: cmd.Process.Pid, cmd: cmd}
|
||||||
|
|
||||||
|
// Wait for the child to signal readiness and stop. The child calls
|
||||||
|
// PTRACE_TRACEME then SIGSTOP, so Wait4 with WUNTRACED observes the
|
||||||
|
// ptrace-stop directly (no PTRACE_ATTACH needed).
|
||||||
|
readyFile := filepath.Join(tmpDir, "ready")
|
||||||
|
for i := 0; i < 500; i++ {
|
||||||
|
if _, err := os.Stat(readyFile); err == nil {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
time.Sleep(5 * time.Millisecond)
|
||||||
|
}
|
||||||
|
var ws syscall.WaitStatus
|
||||||
|
if _, err := syscall.Wait4(s.pid, &ws, syscall.WUNTRACED, nil); err != nil {
|
||||||
|
cmd.Process.Kill()
|
||||||
|
os.RemoveAll(tmpDir)
|
||||||
|
return nil, nil, fmt.Errorf("debug: wait for stop: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Wait for the debuggee to reach the function entry point.
|
||||||
|
entryFile := filepath.Join(tmpDir, "entry")
|
||||||
|
for i := 0; i < 500; i++ {
|
||||||
|
if _, err := os.Stat(entryFile); err == nil {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
time.Sleep(5 * time.Millisecond)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Continue the debuggee to the entry point.
|
||||||
|
if err := s.Continue(); err != nil {
|
||||||
|
return nil, nil, fmt.Errorf("debug: continue to entry: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Wait for the entry stop.
|
||||||
|
if _, err := syscall.Wait4(s.pid, &ws, syscall.WUNTRACED, nil); err != nil {
|
||||||
|
return nil, nil, fmt.Errorf("debug: wait for entry: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
s.stopped = true
|
||||||
|
|
||||||
|
// Read the code base from /proc/pid/maps (find the RWX mapping).
|
||||||
|
s.codeBase = findRWXMapping(s.pid)
|
||||||
|
if s.codeBase == 0 {
|
||||||
|
// Fallback: try the file the child wrote.
|
||||||
|
baseFile := filepath.Join(tmpDir, "codebase")
|
||||||
|
if data, err := os.ReadFile(baseFile); err == nil {
|
||||||
|
fmt.Sscanf(string(data), "%d", &s.codeBase)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Read buffer addresses if buffers were allocated.
|
||||||
|
var bufAddrs []uint64
|
||||||
|
if bufSpec != "" {
|
||||||
|
addrFile := filepath.Join(tmpDir, "bufaddrs")
|
||||||
|
if data, err := os.ReadFile(addrFile); err == nil {
|
||||||
|
for _, line := range strings.Split(strings.TrimSpace(string(data)), "\n") {
|
||||||
|
var addr uint64
|
||||||
|
if _, err := fmt.Sscanf(line, "%d", &addr); err == nil {
|
||||||
|
bufAddrs = append(bufAddrs, addr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return s, bufAddrs, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// wait waits for the debuggee to stop and returns the wait status.
|
||||||
|
func (s *Session) wait() error {
|
||||||
|
var ws syscall.WaitStatus
|
||||||
|
_, err := syscall.Wait4(s.pid, &ws, 0, nil)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if ws.Exited() {
|
||||||
|
s.exited = true
|
||||||
|
return fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
|
||||||
|
}
|
||||||
|
s.stopped = true
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetRegs reads the general-purpose registers of the stopped debuggee.
|
||||||
|
func (s *Session) GetRegs() (Regs, error) {
|
||||||
|
var regs Regs
|
||||||
|
_, _, errno := syscall.Syscall6(
|
||||||
|
syscall.SYS_PTRACE,
|
||||||
|
uintptr(syscall.PTRACE_GETREGS),
|
||||||
|
uintptr(s.pid),
|
||||||
|
0,
|
||||||
|
uintptr(unsafe.Pointer(®s)),
|
||||||
|
0, 0,
|
||||||
|
)
|
||||||
|
if errno != 0 {
|
||||||
|
return regs, fmt.Errorf("debug: PTRACE_GETREGS: %w", errno)
|
||||||
|
}
|
||||||
|
return regs, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// SetRegs writes the general-purpose registers of the stopped debuggee.
|
||||||
|
func (s *Session) SetRegs(regs *Regs) error {
|
||||||
|
_, _, errno := syscall.Syscall6(
|
||||||
|
syscall.SYS_PTRACE,
|
||||||
|
uintptr(syscall.PTRACE_SETREGS),
|
||||||
|
uintptr(s.pid),
|
||||||
|
0,
|
||||||
|
uintptr(unsafe.Pointer(regs)),
|
||||||
|
0, 0,
|
||||||
|
)
|
||||||
|
if errno != 0 {
|
||||||
|
return fmt.Errorf("debug: PTRACE_SETREGS: %w", errno)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// FPRegs holds the x87 FPU and SSE (XMM) register state from PTRACE_GETFPREGS.
|
||||||
|
type FPRegs struct {
|
||||||
|
FCW uint16
|
||||||
|
FSW uint16
|
||||||
|
FTW byte
|
||||||
|
FOP uint16
|
||||||
|
FIP uint64
|
||||||
|
FCS uint16
|
||||||
|
FDP uint64
|
||||||
|
FDS uint16
|
||||||
|
MXCSR uint32
|
||||||
|
MXCSRMask uint32
|
||||||
|
ST [8][16]byte // x87 stack (10 bytes per reg, padded to 16)
|
||||||
|
XMM [16][16]byte // XMM0-15
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetFPRegs retrieves the FPU/SSE register state of the stopped debuggee.
|
||||||
|
func (s *Session) GetFPRegs() (FPRegs, error) {
|
||||||
|
var fp FPRegs
|
||||||
|
_, _, errno := syscall.Syscall6(
|
||||||
|
syscall.SYS_PTRACE,
|
||||||
|
uintptr(syscall.PTRACE_GETFPREGS),
|
||||||
|
uintptr(s.pid),
|
||||||
|
0,
|
||||||
|
uintptr(unsafe.Pointer(&fp)),
|
||||||
|
0, 0,
|
||||||
|
)
|
||||||
|
if errno != 0 {
|
||||||
|
return fp, fmt.Errorf("debug: PTRACE_GETFPREGS: %w", errno)
|
||||||
|
}
|
||||||
|
return fp, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// VectorRegs holds the YMM register state extracted from XSAVE.
|
||||||
|
type VectorRegs struct {
|
||||||
|
YMM [16][32]byte // YMM0-15 (full 256-bit values)
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetVectorRegs retrieves the YMM registers via PTRACE_GETREGSET + XSAVE.
|
||||||
|
// Falls back to XMM if XSAVE is unavailable.
|
||||||
|
func (s *Session) GetVectorRegs() (VectorRegs, error) {
|
||||||
|
var v VectorRegs
|
||||||
|
fp, err := s.GetFPRegs()
|
||||||
|
if err != nil {
|
||||||
|
return v, err
|
||||||
|
}
|
||||||
|
// PTRACE_GETFPREGS gives XMM registers (lower 128 bits).
|
||||||
|
// For YMM we'd need XSAVE; for now, copy XMM and zero the upper half.
|
||||||
|
for i := 0; i < 16; i++ {
|
||||||
|
for j := 0; j < 16; j++ {
|
||||||
|
v.YMM[i][j] = fp.XMM[i][j]
|
||||||
|
}
|
||||||
|
// Upper 128 bits would come from XSAVE, not available via GETFPREGS.
|
||||||
|
}
|
||||||
|
return v, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Peek reads a word (8 bytes) from the debuggee's memory at addr.
|
||||||
|
// Uses /proc/pid/mem which works reliably with Go's multi-threaded runtime.
|
||||||
|
func (s *Session) Peek(addr uint64) (uint64, error) {
|
||||||
|
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
|
||||||
|
if err != nil {
|
||||||
|
return 0, fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
|
||||||
|
}
|
||||||
|
defer mem.Close()
|
||||||
|
buf := make([]byte, 8)
|
||||||
|
if _, err := mem.ReadAt(buf, int64(addr)); err != nil {
|
||||||
|
return 0, fmt.Errorf("debug: read mem %#x: %w", addr, err)
|
||||||
|
}
|
||||||
|
return uint64(buf[0]) | uint64(buf[1])<<8 | uint64(buf[2])<<16 | uint64(buf[3])<<24 |
|
||||||
|
uint64(buf[4])<<32 | uint64(buf[5])<<40 | uint64(buf[6])<<48 | uint64(buf[7])<<56, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Poke writes a word (8 bytes) to the debuggee's memory at addr.
|
||||||
|
func (s *Session) Poke(addr, val uint64) error {
|
||||||
|
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_WRONLY, 0)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
|
||||||
|
}
|
||||||
|
defer mem.Close()
|
||||||
|
buf := []byte{byte(val), byte(val >> 8), byte(val >> 16), byte(val >> 24),
|
||||||
|
byte(val >> 32), byte(val >> 40), byte(val >> 48), byte(val >> 56)}
|
||||||
|
if _, err := mem.WriteAt(buf, int64(addr)); err != nil {
|
||||||
|
return fmt.Errorf("debug: write mem %#x: %w", addr, err)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// ReadMemory reads len bytes from the debuggee's memory at addr.
|
||||||
|
func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) {
|
||||||
|
out := make([]byte, length)
|
||||||
|
for i := 0; i < length; i += 8 {
|
||||||
|
word, err := s.Peek(addr + uint64(i))
|
||||||
|
if err != nil {
|
||||||
|
return out[:i], err
|
||||||
|
}
|
||||||
|
for j := 0; j < 8 && i+j < length; j++ {
|
||||||
|
out[i+j] = byte(word >> (8 * j))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// WriteMemory writes bytes to the debuggee's memory at addr.
|
||||||
|
func (s *Session) WriteMemory(addr uint64, data []byte) error {
|
||||||
|
for i := 0; i < len(data); i += 8 {
|
||||||
|
end := i + 8
|
||||||
|
if end > len(data) {
|
||||||
|
end = len(data)
|
||||||
|
}
|
||||||
|
var word uint64
|
||||||
|
for j := 0; j < end-i; j++ {
|
||||||
|
word |= uint64(data[i+j]) << (8 * j)
|
||||||
|
}
|
||||||
|
// For partial writes, read-modify-write the existing word.
|
||||||
|
if end-i < 8 {
|
||||||
|
existing, err := s.Peek(addr + uint64(i))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
// Clear the bytes we're overwriting and merge.
|
||||||
|
mask := ^((uint64(1) << (8 * (end - i))) - 1)
|
||||||
|
word = (existing & mask) | word
|
||||||
|
}
|
||||||
|
if err := s.Poke(addr+uint64(i), word); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Step executes a single instruction in the debuggee.
|
||||||
|
func (s *Session) Step() error {
|
||||||
|
if s.exited {
|
||||||
|
return fmt.Errorf("debug: debuggee has exited")
|
||||||
|
}
|
||||||
|
_, _, errno := syscall.Syscall6(
|
||||||
|
syscall.SYS_PTRACE,
|
||||||
|
uintptr(syscall.PTRACE_SINGLESTEP),
|
||||||
|
uintptr(s.pid),
|
||||||
|
0, 0, 0, 0,
|
||||||
|
)
|
||||||
|
if errno != 0 {
|
||||||
|
return fmt.Errorf("debug: PTRACE_SINGLESTEP: %w", errno)
|
||||||
|
}
|
||||||
|
return s.wait()
|
||||||
|
}
|
||||||
|
|
||||||
|
// Continue resumes execution until the next breakpoint or exit.
|
||||||
|
func (s *Session) Continue() error {
|
||||||
|
if s.exited {
|
||||||
|
return fmt.Errorf("debug: debuggee has exited")
|
||||||
|
}
|
||||||
|
_, _, errno := syscall.Syscall6(
|
||||||
|
syscall.SYS_PTRACE,
|
||||||
|
uintptr(syscall.PTRACE_CONT),
|
||||||
|
uintptr(s.pid),
|
||||||
|
0, 0, 0, 0,
|
||||||
|
)
|
||||||
|
if errno != 0 {
|
||||||
|
return fmt.Errorf("debug: PTRACE_CONT: %w", errno)
|
||||||
|
}
|
||||||
|
return s.wait()
|
||||||
|
}
|
||||||
|
|
||||||
|
// Exited returns true if the debuggee has terminated.
|
||||||
|
func (s *Session) Exited() bool {
|
||||||
|
return s.exited
|
||||||
|
}
|
||||||
|
|
||||||
|
// Pid returns the debuggee's process ID.
|
||||||
|
func (s *Session) Pid() int {
|
||||||
|
return s.pid
|
||||||
|
}
|
||||||
|
|
||||||
|
// CodeBase returns the base address of the JIT code in the debuggee.
|
||||||
|
func (s *Session) CodeBase() uint64 {
|
||||||
|
return s.codeBase
|
||||||
|
}
|
||||||
|
|
||||||
|
// Kill terminates the debuggee.
|
||||||
|
func (s *Session) Kill() {
|
||||||
|
if !s.exited {
|
||||||
|
syscall.Kill(s.pid, syscall.SIGKILL)
|
||||||
|
syscall.Wait4(s.pid, nil, 0, nil)
|
||||||
|
s.exited = true
|
||||||
|
}
|
||||||
|
if s.cmd != nil && s.cmd.Process != nil {
|
||||||
|
s.cmd.Wait()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// findRWXMapping reads /proc/pid/maps and returns the base address of the
|
||||||
|
// first read-write-execute mapping (the JIT code region).
|
||||||
|
func findRWXMapping(pid int) uint64 {
|
||||||
|
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
|
||||||
|
if err != nil {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
for _, line := range strings.Split(string(data), "\n") {
|
||||||
|
// Format: addr-addr perms offset dev inode pathname
|
||||||
|
fields := strings.Fields(line)
|
||||||
|
if len(fields) < 2 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
perms := fields[1]
|
||||||
|
if len(perms) >= 3 && perms[0] == 'r' && perms[1] == 'w' && perms[2] == 'x' {
|
||||||
|
// Parse the start address.
|
||||||
|
var start uint64
|
||||||
|
fmt.Sscanf(fields[0], "%x-", &start)
|
||||||
|
return start
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
+662
@@ -0,0 +1,662 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build linux && amd64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bufio"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"sort"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Label is a named address within the debugged function.
|
||||||
|
type Label struct {
|
||||||
|
Name string
|
||||||
|
Offset int // function-relative offset
|
||||||
|
}
|
||||||
|
|
||||||
|
// SourceLine maps a byte offset to a source line number.
|
||||||
|
type SourceLine struct {
|
||||||
|
Offset int
|
||||||
|
Line int
|
||||||
|
}
|
||||||
|
|
||||||
|
// REPL runs the interactive debugger loop. On entry, the debuggee is
|
||||||
|
// stopped in the Go runtime (after PTRACE_TRACEME + SIGSTOP). The REPL
|
||||||
|
// sets a temporary breakpoint at the function entry, continues to it, and
|
||||||
|
// then presents the prompt — so the user starts debugging at the first
|
||||||
|
// instruction of the assembled function.
|
||||||
|
func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, argsSize int, labels []Label, lines []SourceLine) {
|
||||||
|
entryAddr := codeBase + uint64(funcOffset)
|
||||||
|
|
||||||
|
// The debuggee is already stopped at the function entry point.
|
||||||
|
fmt.Printf("stopped at function entry: %#x (%d bytes)\n", entryAddr, funcSize)
|
||||||
|
fmt.Println("commands: break <label|addr> | step [n] | continue | disas [n] | regs | where | x <addr> [len] | w <addr> <val...> | labels | quit")
|
||||||
|
|
||||||
|
scanner := bufio.NewScanner(os.Stdin)
|
||||||
|
|
||||||
|
for {
|
||||||
|
fmt.Print("(gasm) ")
|
||||||
|
if !scanner.Scan() {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
line := strings.TrimSpace(scanner.Text())
|
||||||
|
if line == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
parts := strings.Fields(line)
|
||||||
|
cmd := parts[0]
|
||||||
|
|
||||||
|
switch cmd {
|
||||||
|
case "q", "quit":
|
||||||
|
s.Kill()
|
||||||
|
return
|
||||||
|
|
||||||
|
case "regs":
|
||||||
|
regs, err := s.GetRegs()
|
||||||
|
if err != nil {
|
||||||
|
fmt.Println(err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
printRegs(®s, codeBase, uint64(funcOffset))
|
||||||
|
// Also show vector registers.
|
||||||
|
vregs, err := s.GetVectorRegs()
|
||||||
|
if err != nil {
|
||||||
|
fmt.Printf(" (vector regs unavailable: %v)\n", err)
|
||||||
|
} else {
|
||||||
|
printVectorRegs(&vregs)
|
||||||
|
}
|
||||||
|
|
||||||
|
case "step", "s":
|
||||||
|
n := 1
|
||||||
|
if len(parts) > 1 {
|
||||||
|
n, _ = strconv.Atoi(parts[1])
|
||||||
|
}
|
||||||
|
for i := 0; i < n; i++ {
|
||||||
|
if s.Exited() {
|
||||||
|
fmt.Println("debuggee exited")
|
||||||
|
break
|
||||||
|
}
|
||||||
|
if err := s.Step(); err != nil {
|
||||||
|
fmt.Println(err)
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !s.Exited() {
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
text, _, _ := s.Disassemble(regs.RIP)
|
||||||
|
fmt.Printf("=> %#x (func+%#x): %s\n", regs.RIP, regs.RIP-codeBase-uint64(funcOffset), text)
|
||||||
|
}
|
||||||
|
|
||||||
|
case "next", "n":
|
||||||
|
// Step over: if the current instruction is a CALL, set a
|
||||||
|
// breakpoint after it and continue; otherwise single-step.
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
text, instLen, _ := s.Disassemble(regs.RIP)
|
||||||
|
if strings.HasPrefix(strings.ToLower(text), "call") {
|
||||||
|
// Set a temporary breakpoint after the CALL.
|
||||||
|
afterAddr := regs.RIP + uint64(instLen)
|
||||||
|
bp, err := bm.Set(afterAddr, "(next)")
|
||||||
|
if err != nil {
|
||||||
|
fmt.Printf("cannot set next breakpoint: %v\n", err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Continue until the breakpoint.
|
||||||
|
for _, b := range bm.All() {
|
||||||
|
bm.Reinsert(b.Addr)
|
||||||
|
}
|
||||||
|
if err := s.Continue(); err != nil {
|
||||||
|
fmt.Println(err)
|
||||||
|
bm.Clear(afterAddr)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
bm.HandleTrap(®s)
|
||||||
|
bm.Clear(afterAddr)
|
||||||
|
_ = bp
|
||||||
|
} else {
|
||||||
|
// Not a CALL — just single-step.
|
||||||
|
if err := s.Step(); err != nil {
|
||||||
|
fmt.Println(err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !s.Exited() {
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
text, _, _ := s.Disassemble(regs.RIP)
|
||||||
|
fmt.Printf("=> %#x (func+%#x): %s\n", regs.RIP, regs.RIP-codeBase-uint64(funcOffset), text)
|
||||||
|
}
|
||||||
|
|
||||||
|
case "finish", "fin":
|
||||||
|
// Run until the current function returns.
|
||||||
|
// For NOSPLIT frame=0: return address is at [RSP].
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
retAddr, err := s.Peek(regs.RSP)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Printf("cannot read return address: %v\n", err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Set a temporary breakpoint at the return address.
|
||||||
|
bp, err := bm.Set(retAddr, "(finish)")
|
||||||
|
if err != nil {
|
||||||
|
fmt.Printf("cannot set finish breakpoint: %v\n", err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Continue until the breakpoint.
|
||||||
|
for _, b := range bm.All() {
|
||||||
|
bm.Reinsert(b.Addr)
|
||||||
|
}
|
||||||
|
if err := s.Continue(); err != nil {
|
||||||
|
fmt.Println(err)
|
||||||
|
bm.Clear(retAddr)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !s.Exited() {
|
||||||
|
bm.HandleTrap(®s)
|
||||||
|
}
|
||||||
|
bm.Clear(retAddr)
|
||||||
|
_ = bp
|
||||||
|
if s.Exited() {
|
||||||
|
fmt.Println("debuggee exited")
|
||||||
|
} else {
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
fmt.Printf("finished, now at %#x\n", regs.RIP)
|
||||||
|
}
|
||||||
|
|
||||||
|
case "continue", "c":
|
||||||
|
if s.Exited() {
|
||||||
|
fmt.Println("debuggee exited")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Loop: continue until a breakpoint fires (condition met) or exit.
|
||||||
|
for {
|
||||||
|
// Re-insert all breakpoints before continuing.
|
||||||
|
for _, bp := range bm.All() {
|
||||||
|
bm.Reinsert(bp.Addr)
|
||||||
|
}
|
||||||
|
if err := s.Continue(); err != nil {
|
||||||
|
fmt.Println(err)
|
||||||
|
break
|
||||||
|
}
|
||||||
|
if s.Exited() {
|
||||||
|
fmt.Println("debuggee exited")
|
||||||
|
break
|
||||||
|
}
|
||||||
|
// Check for watchpoint hits.
|
||||||
|
reason, wpAddr := s.StopInfo()
|
||||||
|
if reason == StopWatchpoint {
|
||||||
|
fmt.Printf("watchpoint hit at %#x\n", wpAddr)
|
||||||
|
break
|
||||||
|
}
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
if bp := bm.HandleTrap(®s); bp != nil {
|
||||||
|
name := bp.Label
|
||||||
|
if name == "" {
|
||||||
|
name = fmt.Sprintf("%#x", bp.Addr)
|
||||||
|
}
|
||||||
|
fmt.Printf("breakpoint hit: %s (func+%#x)\n", name, bp.Addr-codeBase-uint64(funcOffset))
|
||||||
|
break
|
||||||
|
}
|
||||||
|
// Condition not met (or single-step trap) — re-insert and continue.
|
||||||
|
}
|
||||||
|
|
||||||
|
case "break", "b":
|
||||||
|
if len(parts) < 2 {
|
||||||
|
fmt.Println("usage: break <label|addr|line> [if <reg> <op> <val>]")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Try as a line number first.
|
||||||
|
var addr uint64
|
||||||
|
var label string
|
||||||
|
if lineNum, err := strconv.Atoi(parts[1]); err == nil && lineNum > 0 {
|
||||||
|
// Find the byte offset for this line.
|
||||||
|
off := offsetForLine(lines, lineNum)
|
||||||
|
if off < 0 {
|
||||||
|
fmt.Printf("no instruction at line %d\n", lineNum)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
addr = codeBase + uint64(funcOffset) + uint64(off)
|
||||||
|
label = fmt.Sprintf("line %d", lineNum)
|
||||||
|
} else {
|
||||||
|
addr, label = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||||
|
}
|
||||||
|
if addr == 0 {
|
||||||
|
fmt.Printf("unknown label, address, or line: %s\n", parts[1])
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Parse optional condition: "if <reg> <op> <value>"
|
||||||
|
var cond *Condition
|
||||||
|
if len(parts) >= 6 && parts[2] == "if" {
|
||||||
|
val, err := strconv.ParseUint(parts[5], 0, 64)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Printf("invalid condition value: %s\n", parts[5])
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
cond = &Condition{Reg: strings.ToLower(parts[3]), Op: parts[4], Value: val}
|
||||||
|
} else if len(parts) >= 4 && parts[2] == "if" {
|
||||||
|
fmt.Println("usage: break <label|addr> if <reg> <op> <value>")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
bp, err := bm.SetWithCond(addr, label, cond)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Println(err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
condStr := ""
|
||||||
|
if cond != nil {
|
||||||
|
condStr = fmt.Sprintf(" if %s %s %#x", cond.Reg, cond.Op, cond.Value)
|
||||||
|
}
|
||||||
|
fmt.Printf("breakpoint set: %s at %#x (func+%#x)%s\n", bp.Label, bp.Addr, bp.Addr-codeBase-uint64(funcOffset), condStr)
|
||||||
|
|
||||||
|
case "info":
|
||||||
|
if len(parts) < 2 {
|
||||||
|
fmt.Println("usage: info break")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
switch parts[1] {
|
||||||
|
case "break", "breakpoints", "b":
|
||||||
|
fmt.Print(bm.Info())
|
||||||
|
default:
|
||||||
|
fmt.Printf("unknown info target: %s\n", parts[1])
|
||||||
|
}
|
||||||
|
|
||||||
|
case "delete", "d":
|
||||||
|
if len(parts) < 2 {
|
||||||
|
fmt.Println("usage: delete <label|addr>")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
addr, _ := resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||||
|
if addr == 0 {
|
||||||
|
fmt.Printf("unknown: %s\n", parts[1])
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if err := bm.Clear(addr); err != nil {
|
||||||
|
fmt.Println(err)
|
||||||
|
} else {
|
||||||
|
fmt.Println("breakpoint removed")
|
||||||
|
}
|
||||||
|
|
||||||
|
case "x":
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
addr := regs.RIP // default: current PC
|
||||||
|
length := 64
|
||||||
|
if len(parts) > 1 {
|
||||||
|
addr, _ = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||||
|
}
|
||||||
|
if len(parts) > 2 {
|
||||||
|
length, _ = strconv.Atoi(parts[2])
|
||||||
|
}
|
||||||
|
mem, err := s.ReadMemory(addr, length)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Println(err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
hexDump(addr, mem)
|
||||||
|
|
||||||
|
case "w":
|
||||||
|
if len(parts) < 3 {
|
||||||
|
fmt.Println("usage: w <addr> <byte|0x...> [byte...]")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
addr, _ := resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||||
|
if addr == 0 {
|
||||||
|
fmt.Printf("unknown address: %s\n", parts[1])
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
var bytes []byte
|
||||||
|
for _, arg := range parts[2:] {
|
||||||
|
v, err := strconv.ParseUint(arg, 0, 64)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Printf("invalid value: %s\n", arg)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// Write as 8-byte word if it looks like a large value, else single byte.
|
||||||
|
if v > 255 {
|
||||||
|
for j := 0; j < 8; j++ {
|
||||||
|
bytes = append(bytes, byte(v>>(8*j)))
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
bytes = append(bytes, byte(v))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(bytes) > 0 {
|
||||||
|
if err := s.WriteMemory(addr, bytes); err != nil {
|
||||||
|
fmt.Println(err)
|
||||||
|
} else {
|
||||||
|
fmt.Printf("wrote %d bytes at %#x\n", len(bytes), addr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
case "set":
|
||||||
|
if len(parts) < 3 {
|
||||||
|
fmt.Println("usage: set <reg> <value>")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
val, err := strconv.ParseUint(parts[2], 0, 64)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Printf("invalid value: %s\n", parts[2])
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if err := s.SetReg(strings.ToLower(parts[1]), val); err != nil {
|
||||||
|
fmt.Printf("set: %v\n", err)
|
||||||
|
} else {
|
||||||
|
fmt.Printf("%s = %#x\n", parts[1], val)
|
||||||
|
}
|
||||||
|
|
||||||
|
case "labels", "l":
|
||||||
|
sorted := make([]Label, len(labels))
|
||||||
|
copy(sorted, labels)
|
||||||
|
sort.Slice(sorted, func(i, j int) bool { return sorted[i].Offset < sorted[j].Offset })
|
||||||
|
for _, l := range sorted {
|
||||||
|
fmt.Printf(" func+%#04x %s\n", l.Offset, l.Name)
|
||||||
|
}
|
||||||
|
|
||||||
|
case "disas", "u":
|
||||||
|
n := 5
|
||||||
|
if len(parts) > 1 {
|
||||||
|
n, _ = strconv.Atoi(parts[1])
|
||||||
|
if n <= 0 {
|
||||||
|
n = 5
|
||||||
|
}
|
||||||
|
}
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
fmt.Print(s.DisassembleN(regs.RIP, n))
|
||||||
|
|
||||||
|
case "where":
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
funcOff := int(regs.RIP - codeBase - uint64(funcOffset))
|
||||||
|
line := lineAt(lines, funcOff)
|
||||||
|
label := nearestLabel(labels, funcOff)
|
||||||
|
fmt.Printf(" func+%#x", funcOff)
|
||||||
|
if label != "" {
|
||||||
|
fmt.Printf(" (near %s)", label)
|
||||||
|
}
|
||||||
|
if line > 0 {
|
||||||
|
fmt.Printf(" line %d", line)
|
||||||
|
}
|
||||||
|
fmt.Println()
|
||||||
|
|
||||||
|
case "help", "h", "?":
|
||||||
|
fmt.Println(` break <label|addr> [if <reg> <op> <val>] set a breakpoint
|
||||||
|
delete <label|addr> remove a breakpoint
|
||||||
|
info break list all breakpoints
|
||||||
|
watch <addr> [r|w] [size] set a hardware watchpoint (write by default)
|
||||||
|
unwatch [<slot>] clear one or all watchpoints
|
||||||
|
step [n], s single-step n instructions
|
||||||
|
next, n step over CALL
|
||||||
|
continue, c run until breakpoint or exit
|
||||||
|
disas [n], u disassemble n instructions at PC
|
||||||
|
regs print registers and RFLAGS
|
||||||
|
where show source line and nearest label
|
||||||
|
stack show stack near RSP (args + return address)
|
||||||
|
x [addr] [len] hex-dump memory
|
||||||
|
w <addr> <val...> write bytes to memory
|
||||||
|
labels, l list function labels
|
||||||
|
help, h, ? this help
|
||||||
|
quit, q kill debuggee and exit`)
|
||||||
|
|
||||||
|
case "stack":
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
// For NOSPLIT frame=0: [RSP] = return address, [RSP+8..] = args.
|
||||||
|
retAddr, _ := s.Peek(regs.RSP)
|
||||||
|
fmt.Printf(" [RSP] return addr = %#x\n", retAddr)
|
||||||
|
if argsSize > 0 {
|
||||||
|
fmt.Printf(" args (%d bytes at RSP+8):\n", argsSize)
|
||||||
|
argBytes, err := s.ReadMemory(regs.RSP+8, argsSize)
|
||||||
|
if err == nil {
|
||||||
|
for i := 0; i < argsSize; i += 8 {
|
||||||
|
var v uint64
|
||||||
|
for j := 0; j < 8 && i+j < len(argBytes); j++ {
|
||||||
|
v |= uint64(argBytes[i+j]) << (8 * j)
|
||||||
|
}
|
||||||
|
fmt.Printf(" [%+3d] %#016x\n", i+8, v)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
case "bt", "backtrace":
|
||||||
|
regs, _ := s.GetRegs()
|
||||||
|
funcOff := int(regs.RIP - codeBase - uint64(funcOffset))
|
||||||
|
line := lineAt(lines, funcOff)
|
||||||
|
label := nearestLabel(labels, funcOff)
|
||||||
|
fmt.Printf(" #0 func+%#x", funcOff)
|
||||||
|
if label != "" {
|
||||||
|
fmt.Printf(" (%s)", label)
|
||||||
|
}
|
||||||
|
if line > 0 {
|
||||||
|
fmt.Printf(" [line %d]", line)
|
||||||
|
}
|
||||||
|
fmt.Println()
|
||||||
|
retAddr, _ := s.Peek(regs.RSP)
|
||||||
|
fmt.Printf(" #1 return to %#x\n", retAddr)
|
||||||
|
|
||||||
|
case "watch":
|
||||||
|
if len(parts) < 2 {
|
||||||
|
fmt.Println("usage: watch <addr> [r|w] [size]")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
addr, _ := resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||||
|
if addr == 0 {
|
||||||
|
fmt.Printf("unknown address: %s\n", parts[1])
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
typ := WatchWrite
|
||||||
|
size := 8
|
||||||
|
if len(parts) > 2 {
|
||||||
|
switch parts[2] {
|
||||||
|
case "r":
|
||||||
|
typ = WatchRead
|
||||||
|
case "w":
|
||||||
|
typ = WatchWrite
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(parts) > 3 {
|
||||||
|
size, _ = strconv.Atoi(parts[3])
|
||||||
|
}
|
||||||
|
slot := s.FindFreeWatchpointSlot()
|
||||||
|
if slot < 0 {
|
||||||
|
fmt.Println("no free watchpoint slots (use 'unwatch <slot>' to clear one)")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if err := s.SetWatchpoint(slot, addr, typ, size); err != nil {
|
||||||
|
fmt.Printf("watch: %v\n", err)
|
||||||
|
} else {
|
||||||
|
typStr := "w"
|
||||||
|
if typ == WatchRead {
|
||||||
|
typStr = "r"
|
||||||
|
}
|
||||||
|
fmt.Printf("watchpoint %d set: %#x (%s, %d bytes)\n", slot, addr, typStr, size)
|
||||||
|
}
|
||||||
|
|
||||||
|
case "unwatch":
|
||||||
|
if len(parts) >= 2 {
|
||||||
|
slot, err := strconv.Atoi(parts[1])
|
||||||
|
if err != nil || slot < 0 || slot > 3 {
|
||||||
|
fmt.Println("usage: unwatch [<slot>]")
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if err := s.ClearWatchpoint(slot); err != nil {
|
||||||
|
fmt.Printf("unwatch: %v\n", err)
|
||||||
|
} else {
|
||||||
|
fmt.Printf("watchpoint %d cleared\n", slot)
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
if err := s.ClearAllWatchpoints(); err != nil {
|
||||||
|
fmt.Printf("unwatch: %v\n", err)
|
||||||
|
} else {
|
||||||
|
fmt.Println("all watchpoints cleared")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
default:
|
||||||
|
fmt.Printf("unknown command: %s\n", cmd)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
s.Kill()
|
||||||
|
}
|
||||||
|
|
||||||
|
func printRegs(regs *Regs, codeBase, funcOff uint64) {
|
||||||
|
fmt.Printf(" RIP = %#016x (func+%#x)\n", regs.RIP, regs.RIP-codeBase-funcOff)
|
||||||
|
fmt.Printf(" RSP = %#016x RBP = %#016x\n", regs.RSP, regs.RBP)
|
||||||
|
fmt.Printf(" RAX = %#016x RBX = %#016x\n", regs.RAX, regs.RBX)
|
||||||
|
fmt.Printf(" RCX = %#016x RDX = %#016x\n", regs.RCX, regs.RDX)
|
||||||
|
fmt.Printf(" RSI = %#016x RDI = %#016x\n", regs.RSI, regs.RDI)
|
||||||
|
fmt.Printf(" R8 = %#016x R9 = %#016x\n", regs.R8, regs.R9)
|
||||||
|
fmt.Printf(" R10 = %#016x R11 = %#016x\n", regs.R10, regs.R11)
|
||||||
|
fmt.Printf(" R12 = %#016x R13 = %#016x\n", regs.R12, regs.R13)
|
||||||
|
fmt.Printf(" R14 = %#016x R15 = %#016x\n", regs.R14, regs.R15)
|
||||||
|
fmt.Printf(" RFLAGS = %#x [%s]\n", regs.RFLAGS, decodeRflags(regs.RFLAGS))
|
||||||
|
}
|
||||||
|
|
||||||
|
// printVectorRegs displays the YMM registers.
|
||||||
|
func printVectorRegs(v *VectorRegs) {
|
||||||
|
fmt.Println("\n Vector registers (YMM):")
|
||||||
|
for i := 0; i < 16; i += 2 {
|
||||||
|
fmt.Printf(" YMM%-2d = ", i)
|
||||||
|
printYMM(v.YMM[i][:])
|
||||||
|
fmt.Printf(" YMM%-2d = ", i+1)
|
||||||
|
printYMM(v.YMM[i+1][:])
|
||||||
|
fmt.Println()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func printYMM(b []byte) {
|
||||||
|
// Show as 8 32-bit values.
|
||||||
|
for j := 0; j < 32; j += 4 {
|
||||||
|
v := uint32(b[j]) | uint32(b[j+1])<<8 | uint32(b[j+2])<<16 | uint32(b[j+3])<<24
|
||||||
|
fmt.Printf("%08x ", v)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func decodeRflags(f uint64) string {
|
||||||
|
var flags string
|
||||||
|
if f&1 != 0 {
|
||||||
|
flags += "CF "
|
||||||
|
}
|
||||||
|
if f&(1<<2) != 0 {
|
||||||
|
flags += "PF "
|
||||||
|
}
|
||||||
|
if f&(1<<4) != 0 {
|
||||||
|
flags += "AF "
|
||||||
|
}
|
||||||
|
if f&(1<<6) != 0 {
|
||||||
|
flags += "ZF "
|
||||||
|
}
|
||||||
|
if f&(1<<7) != 0 {
|
||||||
|
flags += "SF "
|
||||||
|
}
|
||||||
|
if f&(1<<8) != 0 {
|
||||||
|
flags += "TF "
|
||||||
|
}
|
||||||
|
if f&(1<<9) != 0 {
|
||||||
|
flags += "IF "
|
||||||
|
}
|
||||||
|
if f&(1<<10) != 0 {
|
||||||
|
flags += "DF "
|
||||||
|
}
|
||||||
|
if f&(1<<11) != 0 {
|
||||||
|
flags += "OF "
|
||||||
|
}
|
||||||
|
if flags == "" {
|
||||||
|
return "none"
|
||||||
|
}
|
||||||
|
return flags[:len(flags)-1] // trim trailing space
|
||||||
|
}
|
||||||
|
|
||||||
|
func hexDump(addr uint64, data []byte) {
|
||||||
|
for i := 0; i < len(data); i += 16 {
|
||||||
|
end := i + 16
|
||||||
|
if end > len(data) {
|
||||||
|
end = len(data)
|
||||||
|
}
|
||||||
|
fmt.Printf(" %#08x:", addr+uint64(i))
|
||||||
|
for j := i; j < i+16; j++ {
|
||||||
|
if j < end {
|
||||||
|
fmt.Printf(" %02x", data[j])
|
||||||
|
} else {
|
||||||
|
fmt.Print(" ")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
fmt.Print(" ")
|
||||||
|
for j := i; j < end; j++ {
|
||||||
|
if data[j] >= 0x20 && data[j] < 0x7f {
|
||||||
|
fmt.Printf("%c", data[j])
|
||||||
|
} else {
|
||||||
|
fmt.Print(".")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
fmt.Println()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func resolveAddr(s string, codeBase, funcOff uint64, labels []Label) (uint64, string) {
|
||||||
|
// Try as a hex address.
|
||||||
|
if strings.HasPrefix(s, "0x") || strings.HasPrefix(s, "0X") {
|
||||||
|
v, err := strconv.ParseUint(s, 0, 64)
|
||||||
|
if err == nil {
|
||||||
|
return v, ""
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Try as func+offset.
|
||||||
|
if strings.HasPrefix(s, "+") {
|
||||||
|
off, err := strconv.ParseUint(s[1:], 0, 64)
|
||||||
|
if err == nil {
|
||||||
|
return codeBase + funcOff + off, fmt.Sprintf("func+%#x", off)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Try as a label name.
|
||||||
|
for _, l := range labels {
|
||||||
|
if l.Name == s {
|
||||||
|
return codeBase + funcOff + uint64(l.Offset), l.Name
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return 0, ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// lineAt returns the source line for a given function-relative offset.
|
||||||
|
func lineAt(lines []SourceLine, offset int) int {
|
||||||
|
if len(lines) == 0 {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
lo, hi := 0, len(lines)-1
|
||||||
|
for lo < hi {
|
||||||
|
mid := (lo + hi + 1) / 2
|
||||||
|
if lines[mid].Offset <= offset {
|
||||||
|
lo = mid
|
||||||
|
} else {
|
||||||
|
hi = mid - 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if lines[lo].Offset <= offset {
|
||||||
|
return lines[lo].Line
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// offsetForLine returns the byte offset for a given source line number.
|
||||||
|
// Returns -1 if no instruction is at that line.
|
||||||
|
func offsetForLine(lines []SourceLine, line int) int {
|
||||||
|
for _, le := range lines {
|
||||||
|
if le.Line == line {
|
||||||
|
return le.Offset
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
|
||||||
|
// nearestLabel returns the name of the label at or just before the offset.
|
||||||
|
func nearestLabel(labels []Label, offset int) string {
|
||||||
|
best := ""
|
||||||
|
bestOff := -1
|
||||||
|
for _, l := range labels {
|
||||||
|
if l.Offset <= offset && l.Offset > bestOff {
|
||||||
|
best = l.Name
|
||||||
|
bestOff = l.Offset
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return best
|
||||||
|
}
|
||||||
@@ -0,0 +1,117 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build linux && amd64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"syscall"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// StopReason describes why the debuggee stopped.
|
||||||
|
type StopReason int
|
||||||
|
|
||||||
|
const (
|
||||||
|
StopNone StopReason = iota
|
||||||
|
StopBreakpoint // INT3 breakpoint hit
|
||||||
|
StopWatchpoint // hardware watchpoint triggered
|
||||||
|
StopSingleStep // single-step completed
|
||||||
|
StopSignal // stopped by a signal
|
||||||
|
StopExited // process exited
|
||||||
|
)
|
||||||
|
|
||||||
|
// siginfo_t layout (Linux amd64): si_signo, si_errno, si_code, then union.
|
||||||
|
type siginfoT struct {
|
||||||
|
SiSigno int32
|
||||||
|
SiErrno int32
|
||||||
|
SiCode int32
|
||||||
|
_pad [125]byte
|
||||||
|
}
|
||||||
|
|
||||||
|
const (
|
||||||
|
trapBRKPT = 1 // INT3 breakpoint
|
||||||
|
trapHWBRKPT = 4 // hardware watchpoint
|
||||||
|
)
|
||||||
|
|
||||||
|
// StopInfo returns the reason the debuggee stopped and the faulting address
|
||||||
|
// (for watchpoints, the watched address that was accessed).
|
||||||
|
func (s *Session) StopInfo() (StopReason, uint64) {
|
||||||
|
if s.exited {
|
||||||
|
return StopExited, 0
|
||||||
|
}
|
||||||
|
var info siginfoT
|
||||||
|
_, _, errno := syscall.Syscall6(
|
||||||
|
syscall.SYS_PTRACE,
|
||||||
|
uintptr(syscall.PTRACE_GETSIGINFO),
|
||||||
|
uintptr(s.pid),
|
||||||
|
0,
|
||||||
|
uintptr(unsafe.Pointer(&info)),
|
||||||
|
0, 0,
|
||||||
|
)
|
||||||
|
if errno != 0 {
|
||||||
|
return StopNone, 0
|
||||||
|
}
|
||||||
|
if info.SiSigno != int32(syscall.SIGTRAP) {
|
||||||
|
return StopSignal, uint64(info.SiCode)
|
||||||
|
}
|
||||||
|
switch info.SiCode {
|
||||||
|
case trapBRKPT:
|
||||||
|
return StopBreakpoint, 0
|
||||||
|
case trapHWBRKPT:
|
||||||
|
// The faulting address is in si_addr (offset 16 in siginfo_t on amd64).
|
||||||
|
addr := *(*uint64)(unsafe.Pointer(uintptr(unsafe.Pointer(&info)) + 16))
|
||||||
|
return StopWatchpoint, addr
|
||||||
|
default:
|
||||||
|
return StopSingleStep, 0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// SetReg modifies a register value in the debuggee.
|
||||||
|
func (s *Session) SetReg(name string, value uint64) error {
|
||||||
|
regs, err := s.GetRegs()
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
switch name {
|
||||||
|
case "rax", "eax", "ax", "al":
|
||||||
|
regs.RAX = value
|
||||||
|
case "rbx", "ebx", "bx", "bl":
|
||||||
|
regs.RBX = value
|
||||||
|
case "rcx", "ecx", "cx", "cl":
|
||||||
|
regs.RCX = value
|
||||||
|
case "rdx", "edx", "dx", "dl":
|
||||||
|
regs.RDX = value
|
||||||
|
case "rsi", "esi", "si":
|
||||||
|
regs.RSI = value
|
||||||
|
case "rdi", "edi", "di":
|
||||||
|
regs.RDI = value
|
||||||
|
case "rbp", "ebp", "bp":
|
||||||
|
regs.RBP = value
|
||||||
|
case "rsp", "esp", "sp":
|
||||||
|
regs.RSP = value
|
||||||
|
case "r8":
|
||||||
|
regs.R8 = value
|
||||||
|
case "r9":
|
||||||
|
regs.R9 = value
|
||||||
|
case "r10":
|
||||||
|
regs.R10 = value
|
||||||
|
case "r11":
|
||||||
|
regs.R11 = value
|
||||||
|
case "r12":
|
||||||
|
regs.R12 = value
|
||||||
|
case "r13":
|
||||||
|
regs.R13 = value
|
||||||
|
case "r14":
|
||||||
|
regs.R14 = value
|
||||||
|
case "r15":
|
||||||
|
regs.R15 = value
|
||||||
|
case "rip", "eip":
|
||||||
|
regs.RIP = value
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("debug: unknown register %q", name)
|
||||||
|
}
|
||||||
|
return s.SetRegs(®s)
|
||||||
|
}
|
||||||
@@ -0,0 +1,227 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build linux && amd64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/hex"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"runtime"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
"syscall"
|
||||||
|
"unsafe"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
||||||
|
)
|
||||||
|
|
||||||
|
// RunTarget is the debuggee entry point (gasm debug --target). It
|
||||||
|
// assembles the file, maps the JIT code, registers itself for ptrace,
|
||||||
|
// stops, and then executes the named function. The parent debugger
|
||||||
|
// controls execution from there.
|
||||||
|
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
|
||||||
|
// Parse and assemble.
|
||||||
|
src, err := os.ReadFile(asmPath)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("debug target: %w", err)
|
||||||
|
}
|
||||||
|
file, errs := parser.Parse(asmPath, string(src))
|
||||||
|
if len(errs) > 0 {
|
||||||
|
return fmt.Errorf("debug target: parse: %v", errs[0])
|
||||||
|
}
|
||||||
|
img, err := asm.AssembleFile(file)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("debug target: assemble: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Find the function.
|
||||||
|
var fl *asm.FuncLayout
|
||||||
|
for i := range img.Funcs {
|
||||||
|
if img.Funcs[i].Name == funcName {
|
||||||
|
fl = &img.Funcs[i]
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if fl == nil {
|
||||||
|
return fmt.Errorf("debug target: function %q not found", funcName)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Map the entire image RWX (we need write access for breakpoints).
|
||||||
|
code := img.Bytes()
|
||||||
|
exec, err := mapRWX(code)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("debug target: mmap: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Write the code base address for the parent.
|
||||||
|
codeBase := uintptr(unsafe.Pointer(&exec[0]))
|
||||||
|
if err := os.WriteFile(tmpDir+"/codebase", []byte(fmt.Sprintf("%d", codeBase)), 0o644); err != nil {
|
||||||
|
return fmt.Errorf("debug target: write codebase: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Write function metadata (offset, size, args) for the parent.
|
||||||
|
meta := fmt.Sprintf("%d %d %d", fl.Offset, fl.Size, fl.Args)
|
||||||
|
os.WriteFile(tmpDir+"/funcmeta", []byte(meta), 0o644)
|
||||||
|
|
||||||
|
// Write label table for breakpoint resolution.
|
||||||
|
labelsFile, _ := os.Create(tmpDir + "/labels")
|
||||||
|
if labelsFile != nil {
|
||||||
|
for label, off := range fl.Labels {
|
||||||
|
fmt.Fprintf(labelsFile, "%s %d\n", label, off)
|
||||||
|
}
|
||||||
|
labelsFile.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
// Read the argument block.
|
||||||
|
args, err := os.ReadFile(argsFile)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("debug target: read args: %w", err)
|
||||||
|
}
|
||||||
|
if len(args) < fl.Args {
|
||||||
|
padded := make([]byte, fl.Args)
|
||||||
|
copy(padded, args)
|
||||||
|
args = padded
|
||||||
|
}
|
||||||
|
|
||||||
|
// Read buffer specification if present.
|
||||||
|
bufSpecFile := tmpDir + "/bufspec"
|
||||||
|
if bufSpec, err := os.ReadFile(bufSpecFile); err == nil && len(bufSpec) > 0 {
|
||||||
|
args, err = setupBuffers(string(bufSpec), args, fl.Args, tmpDir)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("debug target: setup buffers: %w", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Lock this goroutine to the current OS thread so the parent's
|
||||||
|
// ptrace (attached to this thread) controls the JIT execution.
|
||||||
|
runtime.LockOSThread()
|
||||||
|
|
||||||
|
// Request tracing by the parent, then stop. PTRACE_TRACEME makes
|
||||||
|
// the subsequent SIGSTOP a ptrace-stop (not a group-stop), giving
|
||||||
|
// the parent full control from the start.
|
||||||
|
if _, _, errno := syscall.Syscall(syscall.SYS_PTRACE, uintptr(syscall.PTRACE_TRACEME), 0, 0); errno != 0 {
|
||||||
|
return fmt.Errorf("debug target: PTRACE_TRACEME: %v", errno)
|
||||||
|
}
|
||||||
|
os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644)
|
||||||
|
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
|
||||||
|
|
||||||
|
// --- Execution resumes here after the parent continues us ---
|
||||||
|
|
||||||
|
// Stop at the function entry point so the debugger can set breakpoints.
|
||||||
|
// The parent will continue us when ready.
|
||||||
|
os.WriteFile(tmpDir+"/entry", []byte("ok"), 0o644)
|
||||||
|
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
|
||||||
|
|
||||||
|
// Prepare the ABI0 stack and call the function.
|
||||||
|
fnAddr := codeBase + uintptr(fl.Offset)
|
||||||
|
stackArgs := make([]byte, fl.Args)
|
||||||
|
copy(stackArgs, args)
|
||||||
|
|
||||||
|
_, callErr := verify.Call(fnAddr, stackArgs)
|
||||||
|
if callErr != nil {
|
||||||
|
// The function returned an error (shouldn't happen for valid code).
|
||||||
|
os.Exit(1)
|
||||||
|
}
|
||||||
|
os.Exit(0)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// mapRWX maps code into a read-write-execute region (needed for
|
||||||
|
// breakpoint patching via ptrace POKETEXT, though ptrace can write
|
||||||
|
// to any mapping regardless of permissions).
|
||||||
|
func mapRWX(code []byte) ([]byte, error) {
|
||||||
|
const pageSize = 4096
|
||||||
|
size := (len(code) + pageSize - 1) &^ (pageSize - 1)
|
||||||
|
mem, err := syscall.Mmap(-1, 0, size,
|
||||||
|
syscall.PROT_READ|syscall.PROT_WRITE|syscall.PROT_EXEC,
|
||||||
|
syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
copy(mem, code)
|
||||||
|
return mem, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// setupBuffers allocates buffers in the debuggee's memory and updates the
|
||||||
|
// argument block with pointers to them.
|
||||||
|
// Format: name:size:pattern[,name:size:pattern...]
|
||||||
|
// Patterns: zero, ones, seq, or hex (e.g. "deadbeef").
|
||||||
|
func setupBuffers(spec string, args []byte, argSize int, tmpDir string) ([]byte, error) {
|
||||||
|
// Parse the buffer spec.
|
||||||
|
type bufSpec struct {
|
||||||
|
name string
|
||||||
|
size int
|
||||||
|
pattern string
|
||||||
|
}
|
||||||
|
var specs []bufSpec
|
||||||
|
for _, part := range strings.Split(spec, ",") {
|
||||||
|
fields := strings.SplitN(part, ":", 3)
|
||||||
|
if len(fields) != 3 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
size, err := strconv.Atoi(fields[1])
|
||||||
|
if err != nil || size <= 0 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
specs = append(specs, bufSpec{name: fields[0], size: size, pattern: fields[2]})
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(specs) == 0 {
|
||||||
|
return args, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Allocate buffers and write their addresses to a file for the parent.
|
||||||
|
var bufAddrs []uint64
|
||||||
|
for _, s := range specs {
|
||||||
|
buf, err := syscall.Mmap(-1, 0, s.size,
|
||||||
|
syscall.PROT_READ|syscall.PROT_WRITE,
|
||||||
|
syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("mmap buffer %s: %w", s.name, err)
|
||||||
|
}
|
||||||
|
fillBuffer(buf, s.pattern)
|
||||||
|
bufAddrs = append(bufAddrs, uint64(uintptr(unsafe.Pointer(&buf[0]))))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Write buffer addresses to a file for the parent to read.
|
||||||
|
addrFile, err := os.Create(tmpDir + "/bufaddrs")
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
for _, addr := range bufAddrs {
|
||||||
|
fmt.Fprintf(addrFile, "%d\n", addr)
|
||||||
|
}
|
||||||
|
addrFile.Close()
|
||||||
|
|
||||||
|
// For now, return the args unchanged. The parent will read bufaddrs
|
||||||
|
// and construct the final argument block with the correct pointers.
|
||||||
|
return args, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// fillBuffer fills a buffer with the specified pattern.
|
||||||
|
func fillBuffer(buf []byte, pattern string) {
|
||||||
|
switch pattern {
|
||||||
|
case "zero":
|
||||||
|
// Already zeroed by mmap.
|
||||||
|
case "ones":
|
||||||
|
for i := range buf {
|
||||||
|
buf[i] = 0xFF
|
||||||
|
}
|
||||||
|
case "seq":
|
||||||
|
for i := range buf {
|
||||||
|
buf[i] = byte(i)
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
// Try to parse as hex.
|
||||||
|
if data, err := hex.DecodeString(pattern); err == nil && len(data) > 0 {
|
||||||
|
for i := range buf {
|
||||||
|
buf[i] = data[i%len(data)]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,90 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
// Regs holds the full general-purpose register set of a traced process
|
||||||
|
// (the Linux amd64 user_regs_struct layout).
|
||||||
|
type Regs struct {
|
||||||
|
R15 uint64
|
||||||
|
R14 uint64
|
||||||
|
R13 uint64
|
||||||
|
R12 uint64
|
||||||
|
RBP uint64
|
||||||
|
RBX uint64
|
||||||
|
R11 uint64
|
||||||
|
R10 uint64
|
||||||
|
R9 uint64
|
||||||
|
R8 uint64
|
||||||
|
RAX uint64
|
||||||
|
RCX uint64
|
||||||
|
RDX uint64
|
||||||
|
RSI uint64
|
||||||
|
RDI uint64
|
||||||
|
OrigRAX uint64
|
||||||
|
RIP uint64
|
||||||
|
CS uint64
|
||||||
|
RFLAGS uint64
|
||||||
|
RSP uint64
|
||||||
|
SS uint64
|
||||||
|
FSBase uint64
|
||||||
|
GSBase uint64
|
||||||
|
DS uint64
|
||||||
|
ES uint64
|
||||||
|
FS uint64
|
||||||
|
GS uint64
|
||||||
|
}
|
||||||
|
|
||||||
|
// tracer abstracts the minimal ptrace operations needed by the breakpoint
|
||||||
|
// manager and the stop-information helpers. The live implementation is
|
||||||
|
// *Session (ptrace_linux_amd64.go); tests supply a mock.
|
||||||
|
type tracer interface {
|
||||||
|
Peek(addr uint64) (uint64, error)
|
||||||
|
Poke(addr uint64, val uint64) error
|
||||||
|
SetRegs(regs *Regs) error
|
||||||
|
Pid() int
|
||||||
|
}
|
||||||
|
|
||||||
|
// mockTracer records Peek/Poke calls and provides fake register state.
|
||||||
|
type mockTracer struct {
|
||||||
|
mem map[uint64]byte
|
||||||
|
peeks []uint64
|
||||||
|
pokes []struct {
|
||||||
|
addr uint64
|
||||||
|
val uint64
|
||||||
|
}
|
||||||
|
regs *Regs
|
||||||
|
}
|
||||||
|
|
||||||
|
func newMockTracer() *mockTracer {
|
||||||
|
return &mockTracer{
|
||||||
|
mem: make(map[uint64]byte),
|
||||||
|
regs: &Regs{},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (m *mockTracer) Peek(addr uint64) (uint64, error) {
|
||||||
|
m.peeks = append(m.peeks, addr)
|
||||||
|
var val uint64
|
||||||
|
for i := uint64(0); i < 8; i++ {
|
||||||
|
val |= uint64(m.mem[addr+i]) << (i * 8)
|
||||||
|
}
|
||||||
|
return val, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (m *mockTracer) Poke(addr uint64, val uint64) error {
|
||||||
|
m.pokes = append(m.pokes, struct {
|
||||||
|
addr uint64
|
||||||
|
val uint64
|
||||||
|
}{addr, val})
|
||||||
|
for i := uint64(0); i < 8; i++ {
|
||||||
|
m.mem[addr+i] = byte(val >> (i * 8))
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (m *mockTracer) SetRegs(regs *Regs) error {
|
||||||
|
m.regs = regs
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
func (m *mockTracer) Pid() int { return 42 }
|
||||||
@@ -0,0 +1,175 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build linux && amd64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"syscall"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Hardware watchpoint support via x86-64 debug registers (DR0-DR3, DR7).
|
||||||
|
//
|
||||||
|
// DR0-DR3 hold the watched addresses. DR7 is the control register:
|
||||||
|
// bits 0,2,4,6: local enable for DR0-DR3
|
||||||
|
// bits 16-17,20-21,24-25,28-29: R/W type (00=exec, 01=write, 11=read/write)
|
||||||
|
// bits 18-19,22-23,26-27,30-31: length (00=1, 01=2, 10=8, 11=4)
|
||||||
|
|
||||||
|
// WatchpointType selects what triggers the watchpoint.
|
||||||
|
type WatchpointType int
|
||||||
|
|
||||||
|
const (
|
||||||
|
WatchWrite WatchpointType = 1 // trigger on write
|
||||||
|
WatchRead WatchpointType = 3 // trigger on read or write
|
||||||
|
)
|
||||||
|
|
||||||
|
// FindFreeWatchpointSlot returns the index of the first free watchpoint slot
|
||||||
|
// (0-3), or -1 if all four hardware watchpoints are in use.
|
||||||
|
func (s *Session) FindFreeWatchpointSlot() int {
|
||||||
|
for i := 0; i < 4; i++ {
|
||||||
|
if !s.wpSlots[i] {
|
||||||
|
return i
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
|
||||||
|
// IsWatchpointSlotUsed reports whether slot (0-3) currently holds a watchpoint.
|
||||||
|
func (s *Session) IsWatchpointSlotUsed(slot int) bool {
|
||||||
|
if slot < 0 || slot > 3 {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
return s.wpSlots[slot]
|
||||||
|
}
|
||||||
|
|
||||||
|
// SetWatchpoint installs a hardware watchpoint on the given address.
|
||||||
|
// slot is 0-3 (four hardware watchpoints available); the slot must be free.
|
||||||
|
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
|
||||||
|
if slot < 0 || slot > 3 {
|
||||||
|
return fmt.Errorf("debug: watchpoint slot must be 0-3")
|
||||||
|
}
|
||||||
|
if s.wpSlots[slot] {
|
||||||
|
return fmt.Errorf("debug: watchpoint slot %d already in use", slot)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Determine the length encoding.
|
||||||
|
var lenBits uint64
|
||||||
|
switch size {
|
||||||
|
case 1:
|
||||||
|
lenBits = 0
|
||||||
|
case 2:
|
||||||
|
lenBits = 1
|
||||||
|
case 4:
|
||||||
|
lenBits = 3
|
||||||
|
case 8:
|
||||||
|
lenBits = 2
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Write the watched address to DR0-DR3.
|
||||||
|
var drAddr uintptr
|
||||||
|
switch slot {
|
||||||
|
case 0:
|
||||||
|
drAddr = 0x0 // DR0 offset in user_regs_struct
|
||||||
|
case 1:
|
||||||
|
drAddr = 0x8 // DR1
|
||||||
|
case 2:
|
||||||
|
drAddr = 0x10 // DR2
|
||||||
|
case 3:
|
||||||
|
drAddr = 0x18 // DR3
|
||||||
|
}
|
||||||
|
|
||||||
|
// PTRACE_POKEUSER writes to the debuggee's user area (includes debug regs).
|
||||||
|
if err := ptracePokeUser(s.pid, drAddr, addr); err != nil {
|
||||||
|
return fmt.Errorf("debug: set DR%d: %w", slot, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Read the current DR7, set the enable and type bits, write it back.
|
||||||
|
dr7, err := ptracePeekUser(s.pid, 0x38) // DR7 offset
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("debug: read DR7: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
enableBit := uint64(1) << (2 * slot) // local enable
|
||||||
|
rwBits := uint64(typ) << (16 + 4*slot) // R/W type
|
||||||
|
lenField := lenBits << (18 + 4*slot) // length
|
||||||
|
|
||||||
|
// Clear the existing bits for this slot, then set the new ones.
|
||||||
|
mask := ^((uint64(1) << (2 * slot)) | (uint64(3) << (16 + 4*slot)) | (uint64(3) << (18 + 4*slot)))
|
||||||
|
dr7 = (dr7 & mask) | enableBit | rwBits | lenField
|
||||||
|
|
||||||
|
if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
|
||||||
|
return fmt.Errorf("debug: set DR7: %w", err)
|
||||||
|
}
|
||||||
|
s.wpSlots[slot] = true
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// ClearWatchpoint removes a hardware watchpoint.
|
||||||
|
func (s *Session) ClearWatchpoint(slot int) error {
|
||||||
|
if slot < 0 || slot > 3 {
|
||||||
|
return fmt.Errorf("debug: watchpoint slot must be 0-3")
|
||||||
|
}
|
||||||
|
if !s.wpSlots[slot] {
|
||||||
|
return fmt.Errorf("debug: watchpoint slot %d is not in use", slot)
|
||||||
|
}
|
||||||
|
// Read DR7, clear the enable bit for this slot.
|
||||||
|
dr7, err := ptracePeekUser(s.pid, 0x38)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
dr7 &^= uint64(1) << (2 * slot) // disable
|
||||||
|
if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
s.wpSlots[slot] = false
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// ClearAllWatchpoints removes all hardware watchpoints.
|
||||||
|
func (s *Session) ClearAllWatchpoints() error {
|
||||||
|
for slot := 0; slot < 4; slot++ {
|
||||||
|
if s.wpSlots[slot] {
|
||||||
|
if err := s.ClearWatchpoint(slot); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// ptracePokeUser writes a value to the debuggee's user area at the given offset.
|
||||||
|
func ptracePokeUser(pid int, offset uintptr, val uint64) error {
|
||||||
|
const ptracePokeuser = 6 // PTRACE_POKEUSER
|
||||||
|
_, _, errno := syscall.Syscall6(
|
||||||
|
syscall.SYS_PTRACE,
|
||||||
|
uintptr(ptracePokeuser),
|
||||||
|
uintptr(pid),
|
||||||
|
offset,
|
||||||
|
uintptr(val),
|
||||||
|
0, 0,
|
||||||
|
)
|
||||||
|
if errno != 0 {
|
||||||
|
return errno
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// ptracePeekUser reads a value from the debuggee's user area at the given offset.
|
||||||
|
func ptracePeekUser(pid int, offset uintptr) (uint64, error) {
|
||||||
|
const ptracePeekuser = 3 // PTRACE_PEEKUSER
|
||||||
|
val, _, errno := syscall.Syscall6(
|
||||||
|
syscall.SYS_PTRACE,
|
||||||
|
uintptr(ptracePeekuser),
|
||||||
|
uintptr(pid),
|
||||||
|
offset,
|
||||||
|
0, 0, 0,
|
||||||
|
)
|
||||||
|
if errno != 0 {
|
||||||
|
return 0, errno
|
||||||
|
}
|
||||||
|
return uint64(val), nil
|
||||||
|
}
|
||||||
+64
-7
@@ -2,6 +2,8 @@
|
|||||||
|
|
||||||
How gasm-devkit is put together and why.
|
How gasm-devkit is put together and why.
|
||||||
|
|
||||||
|
Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
|
||||||
|
|
||||||
## Design goals
|
## Design goals
|
||||||
|
|
||||||
1. **A real AST, not a grammar hack.** The linter, analyser, assembler and
|
1. **A real AST, not a grammar hack.** The linter, analyser, assembler and
|
||||||
@@ -192,6 +194,24 @@ with the Plan 9 operand order (source first) mapped onto the x86 encoding.
|
|||||||
Every encoding is validated by decoding it again with `golang.org/x/arch` — the
|
Every encoding is validated by decoding it again with `golang.org/x/arch` — the
|
||||||
one module dependency, used in tests only and never linked into the binary.
|
one module dependency, used in tests only and never linked into the binary.
|
||||||
|
|
||||||
|
A **RISC-V encoder** (Phase 5, RV64IMAFDC + RVC compression) encodes the full
|
||||||
|
integer, atomic, float/double, FMA and CSR instruction sets with the MOV
|
||||||
|
pseudo-instruction and SB/global symbol references (AUIPC pairs with
|
||||||
|
R_RISCV_PCREL_HI20/LO12 relocations). The encoder compresses eligible
|
||||||
|
instructions to 16-bit RVC forms and is validated byte-for-byte against
|
||||||
|
`GOARCH=riscv64 go tool asm`.
|
||||||
|
|
||||||
|
A **LoongArch encoder** (Phase 5, LoongArch64) encodes the integer and
|
||||||
|
floating-point instruction sets with the dual-form arithmetic mnemonics (3R
|
||||||
|
vs 2RI12), the 16/21-bit branch families, the MOV pseudo-instruction and its
|
||||||
|
constant materialisation (the dcon classification driving lu12i.w/ori/lu32i.d/
|
||||||
|
lu52i.d expansions), the FP/SP frame mapping (autosize = align8(frame+8),
|
||||||
|
prologue storing the link register before and after the SP decrement) and
|
||||||
|
SB/global symbol references (pcalau12i pairs with R_LOONG64_ADDR_HI/LO
|
||||||
|
relocations). Like the RISC-V encoder it is validated byte-for-byte against
|
||||||
|
`GOARCH=loong64 go tool asm`, and its GOOBJ output is proven end-to-end by
|
||||||
|
substituting it into a cross-compiled `go build` and linking with `cmd/link`.
|
||||||
|
|
||||||
On top of the encoder, `Assemble` walks a parsed `TEXT` body, converts each
|
On top of the encoder, `Assemble` walks a parsed `TEXT` body, converts each
|
||||||
operand to an encoder operand, and lays the instructions out so local labels
|
operand to an encoder operand, and lays the instructions out so local labels
|
||||||
resolve to relative jump offsets: jumps start in the short (rel8) form and
|
resolve to relative jump offsets: jumps start in the short (rel8) form and
|
||||||
@@ -266,8 +286,8 @@ RIP-relative loads whose displacements point inside the resulting image, so
|
|||||||
the bytes are self-consistent at any base address. References to symbols no
|
the bytes are self-consistent at any base address. References to symbols no
|
||||||
`GLOBL` defines are kept as relocations on the function layout, and the
|
`GLOBL` defines are kept as relocations on the function layout, and the
|
||||||
object-file emitters turn the whole image into a linkable object: the ELF
|
object-file emitters turn the whole image into a linkable object: the ELF
|
||||||
and Mach-O writers (`gasm asm --format elf|macho`) lay the code and data out
|
writer (`gasm asm --format elf`) lays the code and data out as `.text`/`.data`
|
||||||
as `.text`/`.data` (or `__text`/`__data`) sections, export a symbol per
|
sections, exports a symbol per
|
||||||
`TEXT` and `GLOBL` (the `<>` ones local, the rest global) and emit one
|
`TEXT` and `GLOBL` (the `<>` ones local, the rest global) and emit one
|
||||||
PC-relative relocation per static-symbol reference — undefined external
|
PC-relative relocation per static-symbol reference — undefined external
|
||||||
symbols included, so the output links with the system toolchain. The GOOBJ
|
symbols included, so the output links with the system toolchain. The GOOBJ
|
||||||
@@ -279,10 +299,22 @@ boundaries, plus flat `pcfile`, `pcline` and `pcinline` tables — so a
|
|||||||
gasm-assembled object drops into a `go build` in place of the toolchain's.
|
gasm-assembled object drops into a `go build` in place of the toolchain's.
|
||||||
The object preamble (the version-and-experiment header the linker compares
|
The object preamble (the version-and-experiment header the linker compares
|
||||||
verbatim) is captured from the installed `go tool asm`, so the output is
|
verbatim) is captured from the installed `go tool asm`, so the output is
|
||||||
always consistent with the toolchain that links it. External cross-package
|
always consistent with the toolchain that links it. RISC-V and LoongArch
|
||||||
references and the implicit funcdata/DWARF symbols remain future work (the
|
GOOBJ emission share this emitter: the loong64 marker with
|
||||||
linker fills the latter's defaults); the rest of Phase 2 is those, the
|
R_LOONG64_ADDR_HI/LO relocation types, and the riscv64 marker with a single
|
||||||
remaining EVEX forms and the other architectures.
|
R_RISCV_PCREL_ITYPE/STYPE relocation per AUIPC pair (plus `R_RISCV_JAL` for
|
||||||
|
`CALL sym(SB)`) — the model `cmd/asm`
|
||||||
|
writes, not the ELF HI20/LO12 pair — and both link into a real `go build` for
|
||||||
|
their `GOARCH`. Per function, the emitter also writes the two DWARF
|
||||||
|
symbols the linker's DWARF pass reads verbatim — the subprogram DIE
|
||||||
|
(`SDWARFFCN`) and the `.debug_line` state-machine program (`SDWARFLINES`),
|
||||||
|
both built the way `cmd/asm` builds them (the DIE carries the
|
||||||
|
R_DWTXTADDR_U4 address reference; the line program one row per source-line
|
||||||
|
change, in the same special-opcode encoding) — and the pc-value deltas are
|
||||||
|
in the architecture's MinLC units, as the runtime's `pcvalue` expects.
|
||||||
|
External cross-package references remain future work (the amd64 and RISC-V
|
||||||
|
paths resolve them; LoongArch does not yet); the rest of Phase 2 is those
|
||||||
|
and the remaining EVEX forms.
|
||||||
|
|
||||||
### `verify`
|
### `verify`
|
||||||
|
|
||||||
@@ -307,7 +339,32 @@ assembler’s `Image.Bytes()` provides the code-and-data concatenation.
|
|||||||
|
|
||||||
The `gasm verify` CLI subcommand exposes this: it loads a file, reports the
|
The `gasm verify` CLI subcommand exposes this: it loads a file, reports the
|
||||||
available functions and (with `-smoke`) calls each NOSPLIT function with zeroed
|
available functions and (with `-smoke`) calls each NOSPLIT function with zeroed
|
||||||
arguments to confirm the trampoline round-trips.
|
arguments to confirm the trampoline round-trips. `gasm verify --fuzz` combines
|
||||||
|
ABI checks (sentinel registers, canary, stack bounds) with differential fuzz
|
||||||
|
testing, comparing the JIT-assembled kernel against the portable Go reference
|
||||||
|
bit-for-bit while verifying the ABI contract on every iteration. When a fuzz
|
||||||
|
iteration crashes or mismatches, `FuzzResult.CrashInput` stores the exact input
|
||||||
|
for reproducibility. `gasm verify --call <func> --buf name:size:pattern`
|
||||||
|
invokes a single function with user-supplied buffers (patterns: zero, ones,
|
||||||
|
seq, or hex), printing the ABI0 argument block before and after the call —
|
||||||
|
useful for partial functions (e.g. decoders) that crash on random input but
|
||||||
|
should succeed on valid data. `gasm verify --ground-truth` compares the
|
||||||
|
assembled machine code byte-for-byte against `go tool asm` (relocation sites
|
||||||
|
masked), reporting any encoding drift.
|
||||||
|
|
||||||
|
### `debug`
|
||||||
|
|
||||||
|
The interactive debugger (Phase 4, linux/amd64). It launches the target
|
||||||
|
function in a child process that maps the JIT code, calls
|
||||||
|
`PTRACE_TRACEME`, and stops; the parent attaches via ptrace and controls
|
||||||
|
execution. Breakpoints are patched as INT3 bytes through `/proc/pid/mem`
|
||||||
|
(PTRACE_PEEKTEXT is unreliable with Go's multi-threaded runtime).
|
||||||
|
The child pins its goroutine to the OS thread with `runtime.LockOSThread`
|
||||||
|
so the traced thread is the one executing JIT code. The REPL provides
|
||||||
|
single-step, register inspection (GPR + YMM/XMM via `PTRACE_GETFPREGS`),
|
||||||
|
label resolution, named buffer allocation with pattern filling
|
||||||
|
(`--buf name:size:pattern` — zero, ones, seq, or hex), and breakpoint
|
||||||
|
management.
|
||||||
|
|
||||||
## Extension points
|
## Extension points
|
||||||
|
|
||||||
|
|||||||
+20
-42
@@ -8,49 +8,27 @@ why, the options on the table, and the trigger that should reopen it.
|
|||||||
|
|
||||||
## GOOBJ external (cross-package) symbol references
|
## GOOBJ external (cross-package) symbol references
|
||||||
|
|
||||||
**Status:** deferred (v0.15.0, 2026-08-02). The GOOBJ emitter resolves only
|
**Status:** resolved (v0.29.0+, 2026-08-07).
|
||||||
symbols defined in the file being assembled; a reference to any other symbol
|
|
||||||
is rejected.
|
|
||||||
|
|
||||||
**Why it is deferred.** GOOBJ symbol references are *positional*: a
|
**Approach taken.** Instead of parsing the compiler's iexport data (which
|
||||||
reference is a `{PkgIdx, SymIdx}` pair, where `SymIdx` is the index of the
|
would have required either `golang.org/x/tools` or an in-house parser), the
|
||||||
symbol in the *referenced package's* symbol-definition table. That ordering
|
resolver reads the **GOOBJ data directly** from the target package's `.a`
|
||||||
is not derivable from the reference site — it lives in the referenced
|
archive. The `.a` file contains a `_go_.o` member whose GOOBJ format is the
|
||||||
package's gc export data (the iexport binary format, which evolves with the
|
same one gasm writes — the parser reuses the same layout (`blkSymdef`,
|
||||||
toolchain). `cmd/asm` reads it with `cmd/internal` readers gasm cannot
|
`blkNonpkgdef`, the string table), so no new dependency was needed.
|
||||||
import, so emitting external references means either parsing export data
|
|
||||||
ourselves or taking a dependency that does.
|
|
||||||
|
|
||||||
**What works today.** Single-package objects: every symbol the file defines
|
**How it works.**
|
||||||
(as `TEXT` or `GLOBL`, static or exported) and every reference to them.
|
|
||||||
This covers the production use case — the go-flac / go-lz4 kernels carry no
|
|
||||||
`FUNCDATA`/`PCDATA`, hence no references into `runtime`, and the Go side
|
|
||||||
references the assembly symbols, never the reverse. Such a package builds
|
|
||||||
with its assembly object replaced by a gasm-emitted one.
|
|
||||||
|
|
||||||
**The options, when we return.**
|
1. `go list -json -export <pkg>` finds the target package's `.a` file.
|
||||||
|
2. `extractGOOBJ` reads the ar archive, finds the `_go_.o` member, skips
|
||||||
|
the `"go object …\n!\n"` preamble and parses the GOOBJ header.
|
||||||
|
3. `goobjFile.symbols()` walks `blkSymdef` and `blkNonpkgdef` in definition
|
||||||
|
order — the same order the linker uses — to build the symbol → index
|
||||||
|
mapping.
|
||||||
|
4. `resolveExternalSymbols` wires the resolved `{PkgIdx, SymIdx}` into the
|
||||||
|
GOOBJ emission.
|
||||||
|
|
||||||
1. **`golang.org/x/tools/go/gcexportdata` as a production dependency.**
|
The resolver is invoked automatically when `img.Externals` is non-empty; it
|
||||||
The straightforward path: read each imported package's export file
|
runs `go list` as a subprocess (consistent with `toolchainObjectPreamble`
|
||||||
(paths from `-importcfg` or `go list -export`), assign symbol indices in
|
which already calls `go tool asm`). All symbol data is cached per package
|
||||||
its symbol order, write `PkgIndex`/`Autolib` entries (fingerprints from
|
for the lifetime of the GOOBJ emission.
|
||||||
the export files' build IDs) and positional references. Robust across
|
|
||||||
toolchain versions — `x/tools` tracks the format. **Cost:** the first
|
|
||||||
production dependency beyond the standard library, an explicit deviation
|
|
||||||
from the "production code depends only on the standard library"
|
|
||||||
principle in the README. Requires the user's explicit agreement.
|
|
||||||
2. **A minimal iexport parser of our own.** Preserves self-containment.
|
|
||||||
Substantial effort and inherently fragile: the format is an internal
|
|
||||||
contract that changes with Go releases, so the parser needs a
|
|
||||||
version-gated fallback and regression tests against several toolchains.
|
|
||||||
3. **Shell out to the toolchain for symbol metadata.** Consistent with the
|
|
||||||
existing GOOBJ preamble probe (which already runs `go tool asm`), but no
|
|
||||||
toolchain command exposes a package's symbols *in definition-index
|
|
||||||
order* — `go tool nm` sorts differently — so this does not solve the
|
|
||||||
core problem on its own; it would only feed option 1 or 2.
|
|
||||||
|
|
||||||
**Trigger to reopen.** An assembly file that needs a cross-package
|
|
||||||
reference — in practice `FUNCDATA $…, runtime·…(SB)` (stack maps / GC
|
|
||||||
metadata written in assembly), or any kernel that calls into another
|
|
||||||
package directly. Until then, option 3's limitation is moot and the
|
|
||||||
single-package emitter suffices.
|
|
||||||
|
|||||||
-79
@@ -1,79 +0,0 @@
|
|||||||
# Using gasm-devkit with Zed
|
|
||||||
|
|
||||||
This document is deliberately blunt, because the situation is a genuine
|
|
||||||
conflict between two of the project's own commitments, and papering over it
|
|
||||||
would be dishonest.
|
|
||||||
|
|
||||||
## The conflict
|
|
||||||
|
|
||||||
gasm-devkit is **pure Go, no C, no cgo, no JavaScript runtimes, no native
|
|
||||||
binaries, no vendor lock-in, no platform-specific IDE internals.**
|
|
||||||
|
|
||||||
Zed's extension model, as verified against Zed's own documentation, is:
|
|
||||||
|
|
||||||
- Extensions are written in **Rust** and compiled to **WebAssembly**
|
|
||||||
(`wasm32-wasip2`).
|
|
||||||
- Syntax highlighting is provided by **Tree-sitter** grammars, which are
|
|
||||||
**C** compiled to WebAssembly with the wasi-sdk, from a grammar written in a
|
|
||||||
**JavaScript** DSL.
|
|
||||||
- A *new* language cannot be registered through configuration alone. Defining
|
|
||||||
a language requires an extension, and every language extension must name a
|
|
||||||
Tree-sitter grammar. (Zed's `lsp` settings section configures
|
|
||||||
already-registered servers; it does not register an arbitrary external binary
|
|
||||||
for a brand-new language.)
|
|
||||||
|
|
||||||
There is therefore **no pure-Go path into Zed's extension host.** This is a
|
|
||||||
property of Zed, not of gasm-devkit: no language tooling author can feed Zed a
|
|
||||||
pure-Go highlighting grammar, because Zed's highlighting engine is Tree-sitter
|
|
||||||
and its plugin runtime is Rust/WASM.
|
|
||||||
|
|
||||||
## What gasm-devkit gives Zed regardless
|
|
||||||
|
|
||||||
The toolkit's integration surface is the **Language Server Protocol**, an open
|
|
||||||
standard. Through `gasm lsp` it provides, with zero editor-specific code:
|
|
||||||
|
|
||||||
- autocomplete (instructions, registers, pseudo-registers, labels),
|
|
||||||
- hover documentation,
|
|
||||||
- diagnostics (the linter, pushed as you type),
|
|
||||||
- document outline (functions and labels),
|
|
||||||
- **syntax highlighting, delivered as LSP semantic tokens.**
|
|
||||||
|
|
||||||
That last point matters: Zed can render highlighting entirely from LSP semantic
|
|
||||||
tokens (`"semantic_tokens": "full"` replaces Tree-sitter highlighting for a
|
|
||||||
language). So the highlighting *capability* exists in pure Go; what Zed needs
|
|
||||||
is merely to be told that `.s` files are a language served by `gasm lsp`.
|
|
||||||
|
|
||||||
## The honest options
|
|
||||||
|
|
||||||
1. **Use an editor that registers an external LSP by configuration.**
|
|
||||||
Neovim, Helix, VS Code and Sublime all let you associate `.s` with the
|
|
||||||
`gasm lsp` binary and use its semantic tokens — no Rust, no C, no lock-in.
|
|
||||||
This is the option that satisfies every stated constraint with no
|
|
||||||
exception.
|
|
||||||
|
|
||||||
2. **Treat a Zed adapter as one quarantined exception.** A minimal Zed
|
|
||||||
extension — a few lines of Rust that register the language and launch
|
|
||||||
`gasm lsp` — plus either a Tree-sitter grammar or `"full"` semantic tokens
|
|
||||||
for highlighting. Crucially, this adapter is the *editor's plugin format*;
|
|
||||||
it is sandboxed inside Zed and never linked into, compiled into, or shipped
|
|
||||||
with the Go toolkit. gasm-devkit itself stays pure Go. But producing it
|
|
||||||
uses the Rust/wasi-sdk/Tree-sitter toolchain, which the project constraints
|
|
||||||
forbid — so it must be a conscious, explicit decision, not a silent one.
|
|
||||||
|
|
||||||
The author's philosophy — digital sovereignty, no dependency on toolchains he
|
|
||||||
does not control — is the tie-breaker, and it is a value judgement rather than
|
|
||||||
a technical one. gasm-devkit is built so that **either** choice keeps the
|
|
||||||
toolkit itself clean: the pure-Go core and the LSP are the product; a Zed
|
|
||||||
adapter, if ever wanted, is a thin, separable leaf.
|
|
||||||
|
|
||||||
## Wiring the LSP (editor-agnostic)
|
|
||||||
|
|
||||||
Run the server and point an LSP client at it:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
go run ./cmd/gasm lsp # or: go install ./cmd/gasm && gasm lsp
|
|
||||||
```
|
|
||||||
|
|
||||||
Associate the command with `*.s` (and `*_amd64.s` / `*_arm64.s`) in whichever
|
|
||||||
editor you use. The server infers the target architecture from the file-name
|
|
||||||
suffix and selects the amd64 or arm64 instruction tables accordingly.
|
|
||||||
+158
@@ -0,0 +1,158 @@
|
|||||||
|
# CLI Reference
|
||||||
|
|
||||||
|
Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
|
||||||
|
|
||||||
|
`gasm` is a single binary with subcommands. Run `gasm --help` for an
|
||||||
|
overview, or `gasm <command> -h` for a command's usage and flags.
|
||||||
|
|
||||||
|
## Global Flags
|
||||||
|
|
||||||
|
| Flag | Description |
|
||||||
|
|------|-------------|
|
||||||
|
| `-h`, `--help` | Show help |
|
||||||
|
| `-V`, `--version` | Print the version |
|
||||||
|
|
||||||
|
## `gasm tokens <file>`
|
||||||
|
|
||||||
|
Print the lexical token stream of FILE: position, token kind, and text,
|
||||||
|
one token per line. FILE may be `-` to read standard input.
|
||||||
|
|
||||||
|
## `gasm parse <file>`
|
||||||
|
|
||||||
|
Parse FILE and report syntax errors on stderr. On success, prints how
|
||||||
|
many declarations and TEXT functions the file contains.
|
||||||
|
|
||||||
|
## `gasm fmt [-w] [path...]`
|
||||||
|
|
||||||
|
Canonicalise the formatting of Plan 9 assembly sources: indentation,
|
||||||
|
operand spacing, per-function mnemonic alignment, and blank-line layout.
|
||||||
|
|
||||||
|
| Flag | Description |
|
||||||
|
|------|-------------|
|
||||||
|
| `-w` | Write result to the source file (default: print to stdout) |
|
||||||
|
|
||||||
|
With no arguments, or with a directory argument, every `.s` file below
|
||||||
|
it is reformatted in place and the names of changed files are listed
|
||||||
|
(`go fmt` style). `.` and `_` directories are skipped.
|
||||||
|
|
||||||
|
## `gasm lint <file...>`
|
||||||
|
|
||||||
|
Run static checks and print diagnostics as
|
||||||
|
`file:line:col: severity: message [code]`. Exit status is non-zero when
|
||||||
|
an error-severity diagnostic is found.
|
||||||
|
|
||||||
|
| Flag | Description |
|
||||||
|
|------|-------------|
|
||||||
|
| `-disable` | Comma-separated rule codes to disable |
|
||||||
|
|
||||||
|
Rules: `unknown-instruction`, `operand-count`, `undefined-label`,
|
||||||
|
`duplicate-label`, `missing-ret`, `missing-textflag-include`,
|
||||||
|
`abi-argsize`, `unreachable-code`, `register-clobber`,
|
||||||
|
`funcdata-pcdata`.
|
||||||
|
|
||||||
|
## `gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] <file>`
|
||||||
|
|
||||||
|
Assemble FILE (amd64) to machine code.
|
||||||
|
|
||||||
|
| Flag | Description |
|
||||||
|
|------|-------------|
|
||||||
|
| `--format` | Output format: `raw` (default), `elf`, `goobj` |
|
||||||
|
| `-p` | Package path (required for `--format goobj`) |
|
||||||
|
| `-o` | Write output to file (default: hex dump to stdout) |
|
||||||
|
|
||||||
|
## `gasm verify [flags] <file.s>`
|
||||||
|
|
||||||
|
Assemble FILE, map it into executable memory, and run dynamic checks.
|
||||||
|
|
||||||
|
| Flag | Description |
|
||||||
|
|------|-------------|
|
||||||
|
| `--ground-truth` | Compare machine code byte-for-byte against `go tool asm` |
|
||||||
|
| `--fuzz` | Differential fuzz: JIT both gasm and go-tool-asm, compare outputs |
|
||||||
|
| `-n` | Fuzz iterations per function (default: 1000) |
|
||||||
|
| `--abi` | Run ABI-checking calls (sentinel registers + red zone) |
|
||||||
|
| `--abi-n` | Number of ABI check iterations with varied inputs (default: 100) |
|
||||||
|
| `--profile` | List basic-block structure per function |
|
||||||
|
| `--smoke` | Call each NOSPLIT function with zeroed args |
|
||||||
|
| `--call <func>` | Invoke a single function with `--buf` instead of the sweeps |
|
||||||
|
| `--buf <spec>` | Buffer spec for `--call`: `name:size:pattern[,name:size:pattern]` |
|
||||||
|
| `--repeat <n>` | Number of times to repeat a `--call` invocation (default: 1) |
|
||||||
|
|
||||||
|
The `--fuzz` mode runs each function in a subprocess; a partial function
|
||||||
|
(e.g. a decoder that faults on malformed input) is reported as
|
||||||
|
`CRASH` without killing the parent. Use `--call` with `--buf` to invoke
|
||||||
|
partial functions with valid data instead.
|
||||||
|
|
||||||
|
The `--call` mode parses the `// func` signature, allocates the requested
|
||||||
|
buffers (`zero`, `ones`, `seq`, or a hex blob), builds the ABI0 argument
|
||||||
|
block with buffer pointers/lengths/capacities at the matching parameter
|
||||||
|
offsets, and prints the arg block before and after the call — showing
|
||||||
|
return values and any output written to the buffers.
|
||||||
|
|
||||||
|
## `gasm debug --func <name> [--buf spec] <file.s>`
|
||||||
|
|
||||||
|
Interactive debugger for JIT-assembled amd64 functions. Requires a
|
||||||
|
compiled binary on `$PATH` (not `go run`).
|
||||||
|
|
||||||
|
| Flag | Description |
|
||||||
|
|------|-------------|
|
||||||
|
| `--func` | Function to debug (required) |
|
||||||
|
| `--buf` | Buffer spec: `name:size:pattern[,name:size:pattern]` |
|
||||||
|
|
||||||
|
REPL commands:
|
||||||
|
|
||||||
|
| Command | Description |
|
||||||
|
|---------|-------------|
|
||||||
|
| `break <label\|addr> [if <reg> <op> <val>]` | Set a breakpoint, optionally conditional |
|
||||||
|
| `delete <label\|addr>` | Remove a breakpoint |
|
||||||
|
| `info break` | List all breakpoints |
|
||||||
|
| `step [n]`, `s` | Single-step n instructions |
|
||||||
|
| `next`, `n` | Step over CALL |
|
||||||
|
| `finish`, `fin` | Run until the function returns |
|
||||||
|
| `continue`, `c` | Run until breakpoint, watchpoint or exit |
|
||||||
|
| `disas [n]`, `u` | Disassemble n instructions at PC |
|
||||||
|
| `regs` | Print general-purpose + YMM/XMM vector registers |
|
||||||
|
| `where` | Show source line and nearest label at PC |
|
||||||
|
| `stack` | Show stack near RSP (return address + ABI0 args) |
|
||||||
|
| `bt`, `backtrace` | Backtrace (current frame + return address) |
|
||||||
|
| `x [addr] [len]` | Hex-dump memory |
|
||||||
|
| `w <addr> <val...>` | Write bytes to memory |
|
||||||
|
| `set <reg> <value>` | Set a register |
|
||||||
|
| `watch <addr> [r\|w] [size]` | Set a hardware watchpoint (write by default) |
|
||||||
|
| `unwatch [<slot>]` | Clear one or all watchpoints |
|
||||||
|
| `labels`, `l` | List function labels and offsets |
|
||||||
|
| `help`, `h`, `?` | Show command help |
|
||||||
|
| `quit`, `q` | Kill the debuggee and exit |
|
||||||
|
|
||||||
|
## `gasm diff [--map old=new,...] <file1.s> <file2.s>`
|
||||||
|
|
||||||
|
Compare the machine code produced by assembling two files. Shows which
|
||||||
|
functions differ and the first few differing bytes. Useful for verifying
|
||||||
|
that two implementations produce identical code, or for tracking encoding
|
||||||
|
changes between Go assembler versions.
|
||||||
|
|
||||||
|
| Flag | Description |
|
||||||
|
|------|-------------|
|
||||||
|
| `--map` | Comma-separated `old=new` pairs to match functions with different names |
|
||||||
|
|
||||||
|
Without `--map`, functions are paired by exact name. With `--map`, a
|
||||||
|
function named `old` in the first file is compared against the function
|
||||||
|
named `new` in the second file (e.g. `--map wideCopyAVX2=wideCopyAVX512`
|
||||||
|
pairs AVX2 and AVX-512 variants regardless of suffix).
|
||||||
|
|
||||||
|
## `gasm profile <file.s>`
|
||||||
|
|
||||||
|
Show the basic-block structure of functions in an assembly file. Lists
|
||||||
|
each function's labels, their offsets, and the block boundaries. This is
|
||||||
|
the static structure; for runtime execution counts, use `gasm verify
|
||||||
|
--fuzz` which exercises the code paths.
|
||||||
|
|
||||||
|
## `gasm lsp`
|
||||||
|
|
||||||
|
Run the language server over standard input/output (JSON-RPC 2.0 with
|
||||||
|
Content-Length framing). Point an LSP-capable editor at the binary and
|
||||||
|
associate it with `.s` files. The target architecture is inferred from
|
||||||
|
the file-name suffix (`_amd64.s`, `_arm64.s`, `_riscv64.s`,
|
||||||
|
`_loong64.s`).
|
||||||
|
|
||||||
|
Provides: completion, hover, document symbols, diagnostics, and
|
||||||
|
semantic-token highlighting.
|
||||||
@@ -0,0 +1,108 @@
|
|||||||
|
# Development Guide
|
||||||
|
|
||||||
|
Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
|
||||||
|
|
||||||
|
## Prerequisites
|
||||||
|
|
||||||
|
- **Go** 1.26+ with `toolchain go1.26.5`
|
||||||
|
- **just** — the command runner; every task below is a just recipe
|
||||||
|
- No external dependencies beyond the Go toolchain
|
||||||
|
|
||||||
|
## Quick Start
|
||||||
|
|
||||||
|
```sh
|
||||||
|
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
|
||||||
|
cd gasm-devkit
|
||||||
|
just install # go mod download
|
||||||
|
just build # go vet + gofmt — must pass with zero output
|
||||||
|
just test # full suite, race detector, 80 % coverage gate
|
||||||
|
```
|
||||||
|
|
||||||
|
## Just Recipes
|
||||||
|
|
||||||
|
### `just build`
|
||||||
|
|
||||||
|
Runs `go vet ./...` and checks `gofmt -l .` produces no output. This is
|
||||||
|
the minimum bar before any commit.
|
||||||
|
|
||||||
|
### `just test`
|
||||||
|
|
||||||
|
```sh
|
||||||
|
go test -race -count=1 -coverprofile=coverage.out ./...
|
||||||
|
```
|
||||||
|
|
||||||
|
Plus an `awk` gate that fails if total coverage is below 80 %.
|
||||||
|
|
||||||
|
### `just fmt`
|
||||||
|
|
||||||
|
```sh
|
||||||
|
gofmt -w .
|
||||||
|
```
|
||||||
|
|
||||||
|
Run after editing any Go source. The output must be idempotent.
|
||||||
|
|
||||||
|
### `just run -- <args>`
|
||||||
|
|
||||||
|
Runs the CLI via `go run` with the version string stamped:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
just run -- lint kernel_amd64.s
|
||||||
|
just run -- fmt -w kernel_amd64.s
|
||||||
|
just run -- verify --ground-truth kernel_amd64.s
|
||||||
|
```
|
||||||
|
|
||||||
|
### `just install-bin`
|
||||||
|
|
||||||
|
Installs the `gasm` binary into `$GOBIN` with the release version
|
||||||
|
embedded via `-ldflags "-X main.version=..."`.
|
||||||
|
|
||||||
|
### `just gen`
|
||||||
|
|
||||||
|
Regenerates the architecture instruction tables in `arch/` by parsing
|
||||||
|
the Go toolchain's own assembler source
|
||||||
|
(`$GOROOT/src/cmd/internal/obj/<arch>/anames.go`). Requires a Go
|
||||||
|
installation. Output is committed — no runtime dependency on the
|
||||||
|
toolchain.
|
||||||
|
|
||||||
|
### `just uninstall`
|
||||||
|
|
||||||
|
Removes `coverage.out`, the `gasm` binary, and `*.test` artefacts.
|
||||||
|
|
||||||
|
## Running Individual Tests
|
||||||
|
|
||||||
|
```sh
|
||||||
|
go test -run TestVexGroundTruth ./asm/
|
||||||
|
go test -run TestDifferentialLZ4Fuzz ./verify/
|
||||||
|
go test -run TestFLACDecorrelate ./verify/
|
||||||
|
go test -run TestGOObjectLinkAndRun ./asm/
|
||||||
|
```
|
||||||
|
|
||||||
|
## Debugger Note
|
||||||
|
|
||||||
|
`gasm debug` spawns a child process from the binary on `$PATH`. It does
|
||||||
|
not work with `go run` — install first:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
just install-bin
|
||||||
|
gasm debug --func decodeBlockAVX2 path/to/kernel_amd64.s
|
||||||
|
```
|
||||||
|
|
||||||
|
## Project Layout
|
||||||
|
|
||||||
|
```
|
||||||
|
cmd/gasm/ CLI entry point (subcommands)
|
||||||
|
token/ Lexical token kinds and positions
|
||||||
|
lexer/ Hand-written scanner
|
||||||
|
ast/ Abstract syntax tree
|
||||||
|
parser/ Line-oriented parser
|
||||||
|
arch/ Register and instruction tables (generated)
|
||||||
|
lint/ Static analysis rules
|
||||||
|
format/ Canonical formatter
|
||||||
|
lsp/ Language Server Protocol server
|
||||||
|
asm/ Standalone assembler, encoder, object emitters
|
||||||
|
verify/ JIT execution, differential testing, ABI checks
|
||||||
|
debug/ Interactive ptrace debugger (linux/amd64)
|
||||||
|
_gen/ Instruction table generator
|
||||||
|
testdata/ Test fixtures
|
||||||
|
docs/ Architecture, development, CLI reference
|
||||||
|
```
|
||||||
@@ -5,7 +5,6 @@ package format
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
@@ -172,11 +171,9 @@ func TestIdempotent(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestRoundTrip checks that formatting produces source that still parses
|
// TestRoundTrip checks that formatting produces source that still parses
|
||||||
// cleanly, on the fixture and on the real go-flac kernels when present.
|
// cleanly on the in-repository fixture.
|
||||||
func TestRoundTrip(t *testing.T) {
|
func TestRoundTrip(t *testing.T) {
|
||||||
files := []string{"../testdata/sample_amd64.s"}
|
files := []string{"../testdata/sample_amd64.s"}
|
||||||
real, _ := filepath.Glob("../../go-libraries/go-*/*.s")
|
|
||||||
files = append(files, real...)
|
|
||||||
for _, path := range files {
|
for _, path := range files {
|
||||||
src, err := os.ReadFile(path)
|
src, err := os.ReadFile(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
@@ -0,0 +1,40 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build integration
|
||||||
|
|
||||||
|
// Package format integration tests against the production go-libraries kernels.
|
||||||
|
// Excluded from the default test run so coverage is identical locally and in CI.
|
||||||
|
// Run explicitly with: go test -tags=integration ./format/
|
||||||
|
package format
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestRoundTripRealGoLibraries checks that formatting the production kernels
|
||||||
|
// still produces source that parses cleanly.
|
||||||
|
func TestRoundTripRealGoLibraries(t *testing.T) {
|
||||||
|
real, _ := filepath.Glob("../../go-libraries/go-*/*.s")
|
||||||
|
if len(real) == 0 {
|
||||||
|
t.Skip("go-libraries repository not present")
|
||||||
|
}
|
||||||
|
for _, path := range real {
|
||||||
|
src, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
formatted := Source(path, string(src))
|
||||||
|
if _, errs := parser.Parse(path, formatted); len(errs) > 0 {
|
||||||
|
t.Errorf("formatted %s no longer parses: %v", path, errs)
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(formatted) == "" {
|
||||||
|
t.Errorf("formatted %s is empty", path)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -3,7 +3,7 @@
|
|||||||
|
|
||||||
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
|
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
|
||||||
|
|
||||||
version := "0.22.0"
|
version := "0.30.0"
|
||||||
|
|
||||||
default:
|
default:
|
||||||
@just --list
|
@just --list
|
||||||
@@ -18,8 +18,21 @@ build:
|
|||||||
@test -z "$(gofmt -l .)" || { echo "gofmt diff:"; gofmt -l .; exit 1; }
|
@test -z "$(gofmt -l .)" || { echo "gofmt diff:"; gofmt -l .; exit 1; }
|
||||||
|
|
||||||
# Full test suite + race detector + 80 % coverage gate.
|
# Full test suite + race detector + 80 % coverage gate.
|
||||||
|
# The coverage gate matches CI: it excludes packages that need hardware or
|
||||||
|
# are CLI glue (debug, cmd/gasm), so the number is identical locally and in CI.
|
||||||
test:
|
test:
|
||||||
go test -race -count=1 -coverprofile=coverage.out ./...
|
go test -race -count=1 ./...
|
||||||
|
go test -count=1 -coverprofile=coverage.out \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/arch \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/asm \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/ast \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/format \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/lexer \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/lint \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/lsp \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/parser \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/token \
|
||||||
|
sourcedock.dev/petrbalvin/gasm-devkit/verify
|
||||||
go tool cover -func=coverage.out | awk '/^total:/{gsub("%","",$3);if($3+0<80){print "coverage "$3"% < 80%";exit 1}print "coverage "$3"%"}'
|
go tool cover -func=coverage.out | awk '/^total:/{gsub("%","",$3);if($3+0<80){print "coverage "$3"% < 80%";exit 1}print "coverage "$3"%"}'
|
||||||
|
|
||||||
# Format all Go sources.
|
# Format all Go sources.
|
||||||
|
|||||||
@@ -0,0 +1,44 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build integration
|
||||||
|
|
||||||
|
// Package lint integration tests against the production go-libraries kernels.
|
||||||
|
// Excluded from the default test run so coverage is identical locally and in CI.
|
||||||
|
// Run explicitly with: go test -tags=integration ./lint/
|
||||||
|
package lint
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestRealGoLibrariesHasNoErrors asserts that the production go-flac kernels
|
||||||
|
// lint free of errors.
|
||||||
|
func TestRealGoLibrariesHasNoErrors(t *testing.T) {
|
||||||
|
matches, _ := filepath.Glob("../../go-libraries/go-*/*.s")
|
||||||
|
if len(matches) == 0 {
|
||||||
|
t.Skip("go-libraries repository not present")
|
||||||
|
}
|
||||||
|
for _, path := range matches {
|
||||||
|
src, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
f, errs := parser.Parse(path, string(src))
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse %s: %v", path, errs)
|
||||||
|
}
|
||||||
|
a := arch.FromFilename(path)
|
||||||
|
diags := File(f, Config{Arch: a})
|
||||||
|
for _, d := range diags {
|
||||||
|
if d.Severity == Error {
|
||||||
|
t.Errorf("%s: %s %s: %s", filepath.Base(path), d.Pos, d.Code, d.Message)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -5,7 +5,6 @@ package lint
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||||
@@ -234,29 +233,3 @@ TEXT ·f(SB), NOSPLIT, $0
|
|||||||
t.Fatalf("label rules should be suppressed in macro files: %+v", diags)
|
t.Fatalf("label rules should be suppressed in macro files: %+v", diags)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestRealGoLibrariesHasNoErrors asserts that the production go-flac kernels
|
|
||||||
// lint free of errors. Skipped when the sibling repository is absent.
|
|
||||||
func TestRealGoLibrariesHasNoErrors(t *testing.T) {
|
|
||||||
matches, _ := filepath.Glob("../../go-libraries/go-*/*.s")
|
|
||||||
if len(matches) == 0 {
|
|
||||||
t.Skip("go-libraries repository not present")
|
|
||||||
}
|
|
||||||
for _, path := range matches {
|
|
||||||
src, err := os.ReadFile(path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
f, errs := parser.Parse(path, string(src))
|
|
||||||
if len(errs) > 0 {
|
|
||||||
t.Fatalf("parse %s: %v", path, errs)
|
|
||||||
}
|
|
||||||
a := arch.FromFilename(path)
|
|
||||||
diags := File(f, Config{Arch: a})
|
|
||||||
for _, d := range diags {
|
|
||||||
if d.Severity == Error {
|
|
||||||
t.Errorf("%s: %s %s: %s", filepath.Base(path), d.Pos, d.Code, d.Message)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -90,6 +90,41 @@ func (s *Server) hover(p hoverParams) *Hover {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// definition returns the location of the label definition for a label reference.
|
||||||
|
func (s *Server) definition(p definitionParams) []Location {
|
||||||
|
text := s.docs[p.TextDocument.URI]
|
||||||
|
word, _ := wordAt(text, p.Position)
|
||||||
|
if word == "" {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse the document to find label definitions.
|
||||||
|
f, errs := parser.Parse(uriPath(p.TextDocument.URI), text)
|
||||||
|
if f == nil || len(errs) > 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Find the label definition.
|
||||||
|
for _, d := range f.Decls {
|
||||||
|
if t, ok := d.(*ast.Text); ok {
|
||||||
|
for _, stmt := range t.Body {
|
||||||
|
if lbl, ok := stmt.(*ast.Label); ok {
|
||||||
|
if lbl.Name.Text == word {
|
||||||
|
return []Location{{
|
||||||
|
URI: p.TextDocument.URI,
|
||||||
|
Range: Range{
|
||||||
|
Start: Position{Line: lbl.Name.Pos.Line - 1, Character: lbl.Name.Pos.Column - 1},
|
||||||
|
End: Position{Line: lbl.Name.Pos.Line - 1, Character: lbl.Name.Pos.Column - 1 + len(word)},
|
||||||
|
},
|
||||||
|
}}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
// documentSymbols returns functions and their labels, plus global symbols.
|
// documentSymbols returns functions and their labels, plus global symbols.
|
||||||
func (s *Server) documentSymbols(p documentSymbolParams) []DocumentSymbol {
|
func (s *Server) documentSymbols(p documentSymbolParams) []DocumentSymbol {
|
||||||
text := s.docs[p.TextDocument.URI]
|
text := s.docs[p.TextDocument.URI]
|
||||||
|
|||||||
@@ -152,6 +152,11 @@ type hoverParams struct {
|
|||||||
Position Position `json:"position"`
|
Position Position `json:"position"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
type definitionParams struct {
|
||||||
|
TextDocument textDocumentIdentifier `json:"textDocument"`
|
||||||
|
Position Position `json:"position"`
|
||||||
|
}
|
||||||
|
|
||||||
// Hover is the hover response.
|
// Hover is the hover response.
|
||||||
type Hover struct {
|
type Hover struct {
|
||||||
Contents markupContent `json:"contents"`
|
Contents markupContent `json:"contents"`
|
||||||
|
|||||||
@@ -170,6 +170,11 @@ func (s *Server) dispatch(msg *rpcMessage) (exit bool) {
|
|||||||
json.Unmarshal(msg.Params, &p)
|
json.Unmarshal(msg.Params, &p)
|
||||||
s.respond(msg.ID, s.hover(p))
|
s.respond(msg.ID, s.hover(p))
|
||||||
|
|
||||||
|
case "textDocument/definition":
|
||||||
|
var p definitionParams
|
||||||
|
json.Unmarshal(msg.Params, &p)
|
||||||
|
s.respond(msg.ID, s.definition(p))
|
||||||
|
|
||||||
case "textDocument/documentSymbol":
|
case "textDocument/documentSymbol":
|
||||||
var p documentSymbolParams
|
var p documentSymbolParams
|
||||||
json.Unmarshal(msg.Params, &p)
|
json.Unmarshal(msg.Params, &p)
|
||||||
|
|||||||
@@ -0,0 +1,39 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build integration
|
||||||
|
|
||||||
|
// Package parser integration tests against the production go-libraries kernels.
|
||||||
|
// Excluded from the default test run so coverage is identical locally and in CI.
|
||||||
|
// Run explicitly with: go test -tags=integration ./parser/
|
||||||
|
package parser
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestParseRealGoLibraries parses every .s file in the sibling go-libraries
|
||||||
|
// repository when it is checked out, asserting a clean, error-free parse.
|
||||||
|
func TestParseRealGoLibraries(t *testing.T) {
|
||||||
|
matches, _ := filepath.Glob("../../go-libraries/go-*/*.s")
|
||||||
|
if len(matches) == 0 {
|
||||||
|
t.Skip("go-libraries repository not present next to gasm-devkit")
|
||||||
|
}
|
||||||
|
for _, path := range matches {
|
||||||
|
src, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read %s: %v", path, err)
|
||||||
|
}
|
||||||
|
file, errs := Parse(path, string(src))
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Errorf("parse %s: %v", path, errs)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if len(texts(file)) == 0 {
|
||||||
|
t.Errorf("parse %s: no TEXT functions found", path)
|
||||||
|
}
|
||||||
|
t.Logf("%s: %d decls, %d functions", filepath.Base(path), len(file.Decls), len(texts(file)))
|
||||||
|
}
|
||||||
|
}
|
||||||
+12
-3
@@ -298,10 +298,14 @@ func parseSymbolPrefix(g []token.Token) (*ast.Symbol, int) {
|
|||||||
sym.Static = true
|
sym.Static = true
|
||||||
i += 2
|
i += 2
|
||||||
}
|
}
|
||||||
if i < len(g) && g[i].Kind == token.Plus {
|
if i < len(g) && (g[i].Kind == token.Plus || g[i].Kind == token.Minus) {
|
||||||
|
neg := g[i].Kind == token.Minus
|
||||||
i++
|
i++
|
||||||
if i < len(g) && g[i].Kind == token.Number {
|
if i < len(g) && g[i].Kind == token.Number {
|
||||||
sym.Offset, sym.HasOff = parseInt(g[i].Text), true
|
sym.Offset, sym.HasOff = parseInt(g[i].Text), true
|
||||||
|
if neg {
|
||||||
|
sym.Offset = -sym.Offset
|
||||||
|
}
|
||||||
i++
|
i++
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -398,10 +402,15 @@ func parseAddress(g []token.Token) ast.Address {
|
|||||||
return addr
|
return addr
|
||||||
}
|
}
|
||||||
// Symbol-with-pseudo form: name[<>][+off](PSEUDO).
|
// Symbol-with-pseudo form: name[<>][+off](PSEUDO).
|
||||||
|
// When the prefix is not a valid symbol name (e.g. a bare number like
|
||||||
|
// 0(SP) in RISC-V), sym is nil and we fall through to regular memory
|
||||||
|
// operand parsing instead of returning an empty address.
|
||||||
if idx := findPseudoParen(g); idx >= 0 {
|
if idx := findPseudoParen(g); idx >= 0 {
|
||||||
sym, _ := parseSymbolPrefix(g[:idx+3])
|
sym, _ := parseSymbolPrefix(g[:idx+3])
|
||||||
addr.Sym = sym
|
if sym != nil {
|
||||||
return addr
|
addr.Sym = sym
|
||||||
|
return addr
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
i := 0
|
i := 0
|
||||||
|
|||||||
@@ -5,7 +5,6 @@ package parser
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
@@ -254,28 +253,3 @@ func TestDataWidthAndStatic(t *testing.T) {
|
|||||||
t.Errorf("mask24 DATA should be static, got %+v", datas[2].Name)
|
t.Errorf("mask24 DATA should be static, got %+v", datas[2].Name)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestParseRealGoLibraries parses every .s file in the sibling go-libraries
|
|
||||||
// repository when it is checked out, asserting a clean, error-free parse. It
|
|
||||||
// is skipped when the repository is not present.
|
|
||||||
func TestParseRealGoLibraries(t *testing.T) {
|
|
||||||
matches, _ := filepath.Glob("../../go-libraries/go-*/*.s")
|
|
||||||
if len(matches) == 0 {
|
|
||||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
|
||||||
}
|
|
||||||
for _, path := range matches {
|
|
||||||
src, err := os.ReadFile(path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("read %s: %v", path, err)
|
|
||||||
}
|
|
||||||
file, errs := Parse(path, string(src))
|
|
||||||
if len(errs) > 0 {
|
|
||||||
t.Errorf("parse %s: %v", path, errs)
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
if len(texts(file)) == 0 {
|
|
||||||
t.Errorf("parse %s: no TEXT functions found", path)
|
|
||||||
}
|
|
||||||
t.Logf("%s: %d decls, %d functions", filepath.Base(path), len(file.Decls), len(texts(file)))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
Vendored
+12
@@ -0,0 +1,12 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func add(a, b int64) int64
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
MOV b+8(FP), X11
|
||||||
|
ADD X11, X10, X10
|
||||||
|
MOV X10, ret+16(FP)
|
||||||
|
RET
|
||||||
Vendored
+20
@@ -0,0 +1,20 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func atomicAdd(ptr *int64, val int64) int64
|
||||||
|
TEXT ·atomicAdd(SB), NOSPLIT, $0-24
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
MOV b+8(FP), X11
|
||||||
|
AMOADDD X11, (X10), X12
|
||||||
|
MOV X12, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func fpAdd(a, b float64) float64
|
||||||
|
TEXT ·fpAdd(SB), NOSPLIT, $0-24
|
||||||
|
FLD a+0(FP), F10
|
||||||
|
FLD b+8(FP), F11
|
||||||
|
FADDD F10, F11, F12
|
||||||
|
FSD F12, ret+16(FP)
|
||||||
|
RET
|
||||||
Vendored
+26
@@ -0,0 +1,26 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func readCSR(csr int64) int64
|
||||||
|
TEXT ·readCSR(SB), NOSPLIT, $0-16
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
CSRRS $0x300, X0, X11
|
||||||
|
MOV X11, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func setCSRBit(csr, bit int64) int64
|
||||||
|
TEXT ·setCSRBit(SB), NOSPLIT, $0-24
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
MOV b+8(FP), X11
|
||||||
|
CSRRS $0x304, X11, X12
|
||||||
|
MOV X12, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func writeCSR(val int64) int64
|
||||||
|
TEXT ·writeCSR(SB), NOSPLIT, $0-16
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
CSRRW $0x305, X10, X11
|
||||||
|
MOV X11, ret+8(FP)
|
||||||
|
RET
|
||||||
Vendored
+22
@@ -0,0 +1,22 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func fma(a, b, c float64) float64
|
||||||
|
TEXT ·fma(SB), NOSPLIT, $0-32
|
||||||
|
FLD a+0(FP), F10
|
||||||
|
FLD b+8(FP), F11
|
||||||
|
FLD c+16(FP), F12
|
||||||
|
FMADDD F10, F11, F12, F13
|
||||||
|
FSD F13, ret+24(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func fms(a, b, c float64) float64
|
||||||
|
TEXT ·fms(SB), NOSPLIT, $0-32
|
||||||
|
FLD a+0(FP), F10
|
||||||
|
FLD b+8(FP), F11
|
||||||
|
FLD c+16(FP), F12
|
||||||
|
FMSUBD F10, F11, F12, F13
|
||||||
|
FSD F13, ret+24(FP)
|
||||||
|
RET
|
||||||
Vendored
+36
@@ -0,0 +1,36 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func casLoop(ptr *int64, old, new int64) bool
|
||||||
|
TEXT ·casLoop(SB), NOSPLIT, $0-32
|
||||||
|
cas_retry:
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
LRD (X10), X11
|
||||||
|
MOV b+8(FP), X12
|
||||||
|
BNE X11, X12, cas_fail
|
||||||
|
MOV c+16(FP), X13
|
||||||
|
SCD X13, (X10), X14
|
||||||
|
BNE X14, X0, cas_retry
|
||||||
|
ADDI X0, $1, X15
|
||||||
|
MOV X15, ret+24(FP)
|
||||||
|
RET
|
||||||
|
cas_fail:
|
||||||
|
MOV X0, ret+24(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func intToFloat(x int64) float64
|
||||||
|
TEXT ·intToFloat(SB), NOSPLIT, $0-16
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
FCVTDL X10, F10
|
||||||
|
FSD F10, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func compare(a, b float64) bool
|
||||||
|
TEXT ·compare(SB), NOSPLIT, $0-24
|
||||||
|
FLD a+0(FP), F10
|
||||||
|
FLD b+8(FP), F11
|
||||||
|
FLTD F10, F11, X10
|
||||||
|
MOV X10, ret+16(FP)
|
||||||
|
RET
|
||||||
Vendored
+54
@@ -0,0 +1,54 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// add returns a + b.
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOVV a+0(FP), R4
|
||||||
|
MOVV b+8(FP), R5
|
||||||
|
ADDV R5, R4, R4
|
||||||
|
MOVV R4, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// arith exercises the 3R integer and FP set.
|
||||||
|
TEXT ·arith(SB), NOSPLIT, $0-0
|
||||||
|
ADDV R4, R5, R6
|
||||||
|
SUBV R7, R8, R9
|
||||||
|
MULV R10, R11, R12
|
||||||
|
DIVV R13, R14, R15
|
||||||
|
AND R16, R17, R18
|
||||||
|
OR R18, R19, R20
|
||||||
|
XOR R20, R21, R2
|
||||||
|
SLLV R2, R23, R24
|
||||||
|
SRLV R24, R25, R26
|
||||||
|
SRAV R26, R27, R28
|
||||||
|
RET
|
||||||
|
|
||||||
|
// imm exercises the immediate forms.
|
||||||
|
TEXT ·imm(SB), NOSPLIT, $0-0
|
||||||
|
ADDV $42, R4, R5
|
||||||
|
ADDV $-8, R6
|
||||||
|
AND $0xff, R7, R8
|
||||||
|
OR $1, R9, R10
|
||||||
|
XOR $0, R11, R12
|
||||||
|
SGT $100, R13, R14
|
||||||
|
SLLV $4, R15, R16
|
||||||
|
MOVV $0x12345, R17
|
||||||
|
RET
|
||||||
|
|
||||||
|
// branch exercises conditional and unconditional control flow.
|
||||||
|
TEXT ·branch(SB), NOSPLIT, $0-0
|
||||||
|
BEQ R4, R5, done
|
||||||
|
BNE R6, R7, skip
|
||||||
|
BLT R8, R9, done
|
||||||
|
BGE R10, R11, done
|
||||||
|
BLTU R12, R13, done
|
||||||
|
BGEU R14, R15, done
|
||||||
|
skip:
|
||||||
|
JMP loop
|
||||||
|
loop:
|
||||||
|
JAL skip
|
||||||
|
RET
|
||||||
|
done:
|
||||||
|
RET
|
||||||
Vendored
+18
@@ -0,0 +1,18 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·framed(SB), NOSPLIT, $16-16
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
MOV b+8(FP), X11
|
||||||
|
ADD X11, X10, X10
|
||||||
|
MOV X10, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·leaf(SB), NOSPLIT, $0-16
|
||||||
|
MOV a+0(FP), X10
|
||||||
|
MOV b+8(FP), X11
|
||||||
|
ADD X11, X10, X10
|
||||||
|
MOV X10, ret+16(FP)
|
||||||
|
RET
|
||||||
Vendored
+35
@@ -0,0 +1,35 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·branches(SB), NOSPLIT, $0
|
||||||
|
ADDI $1, X10, X10
|
||||||
|
BEQ X10, X11, beq_done
|
||||||
|
ADDI $2, X10, X10
|
||||||
|
beq_done:
|
||||||
|
BNE X10, X11, bne_done
|
||||||
|
ADDI $3, X10, X10
|
||||||
|
bne_done:
|
||||||
|
BLT X10, X11, blt_done
|
||||||
|
ADDI $4, X10, X10
|
||||||
|
blt_done:
|
||||||
|
BGE X10, X11, bge_done
|
||||||
|
ADDI $5, X10, X10
|
||||||
|
bge_done:
|
||||||
|
BLTU X10, X11, bltu_done
|
||||||
|
ADDI $6, X10, X10
|
||||||
|
bltu_done:
|
||||||
|
BGEU X10, X11, bgeu_done
|
||||||
|
ADDI $7, X10, X10
|
||||||
|
bgeu_done:
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·jumps(SB), NOSPLIT, $0
|
||||||
|
JMP done
|
||||||
|
ADDI $1, X10, X10
|
||||||
|
done:
|
||||||
|
JAL X11, skip
|
||||||
|
ADDI $2, X10, X10
|
||||||
|
skip:
|
||||||
|
RET
|
||||||
Vendored
+8
@@ -0,0 +1,8 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·call(SB), NOSPLIT, $0
|
||||||
|
CALL callee(SB)
|
||||||
|
RET
|
||||||
Vendored
+143
@@ -0,0 +1,143 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// fp exercises the floating-point set: 3R arithmetic, 2R unary, compares
|
||||||
|
// into FCC, fused multiply-add and the register moves.
|
||||||
|
TEXT ·fp(SB), NOSPLIT, $0-0
|
||||||
|
ADDD F4, F5, F6
|
||||||
|
SUBD F7, F8, F9
|
||||||
|
MULD F9, F10, F11
|
||||||
|
DIVD F11, F12, F13
|
||||||
|
MULF F13, F14, F15
|
||||||
|
ADDF F15, F16, F17
|
||||||
|
SQRTD F17, F18
|
||||||
|
SQRTF F18, F19
|
||||||
|
ABSD F19, F20
|
||||||
|
NEGD F20, F21
|
||||||
|
MOVD F21, F22
|
||||||
|
CMPEQD F22, F23, FCC0
|
||||||
|
CMPGTF F23, F24, FCC1
|
||||||
|
CMPGED F24, F25, FCC2
|
||||||
|
FMADDD F0, F1, F2, F3
|
||||||
|
FMSUBF F3, F4, F5, F6
|
||||||
|
FNMADDD F6, F7, F8, F9
|
||||||
|
FNMSUBF F9, F10, F11, F12
|
||||||
|
FMAXD F12, F13, F14
|
||||||
|
FMINF F14, F15, F16
|
||||||
|
FMAXAD F16, F17, F18
|
||||||
|
FMINAF F18, F19, F20
|
||||||
|
FSCALEBF F20, F21, F22
|
||||||
|
FCOPYSGD F22, F23, F24
|
||||||
|
MOVV F25, R25
|
||||||
|
MOVV R26, F27
|
||||||
|
MOVW R28, F29
|
||||||
|
MOVW F30, R31
|
||||||
|
RET
|
||||||
|
|
||||||
|
// mov forms: register moves, immediates (12/32/64-bit), memory with FP/SP
|
||||||
|
// pseudo-registers and the register-indexed forms.
|
||||||
|
TEXT ·mov(SB), NOSPLIT, $0-16
|
||||||
|
MOVV R4, R5
|
||||||
|
MOVW R6, R7
|
||||||
|
MOVB R8, R9
|
||||||
|
MOVBU R10, R11
|
||||||
|
MOVHU R12, R13
|
||||||
|
MOVWU R14, R15
|
||||||
|
MOVV $42, R16
|
||||||
|
MOVV $0x12345, R17
|
||||||
|
MOVV $0x100000, R18
|
||||||
|
MOVW $-100, R19
|
||||||
|
MOVV $0x123456789, R20
|
||||||
|
MOVV a+0(FP), R21
|
||||||
|
MOVV R23, b+8(FP)
|
||||||
|
MOVW c+16(FP), R24
|
||||||
|
MOVV (R24)(R25), R26
|
||||||
|
MOVV R27, (R28)(R29)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// frame exercises the prologue/epilogue of a function with a real frame.
|
||||||
|
TEXT ·frame(SB), NOSPLIT, $32-8
|
||||||
|
MOVV R4, R5
|
||||||
|
MOVV arg+0(FP), R6
|
||||||
|
MOVV R7, local-8(SP)
|
||||||
|
MOVV local-8(SP), R8
|
||||||
|
MOVV R9, ret+0(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// branches21 exercises the single-register and zero-register branch forms
|
||||||
|
// with 21-bit offsets.
|
||||||
|
TEXT ·branches21(SB), NOSPLIT, $0-0
|
||||||
|
BEQ R0, R4, l1
|
||||||
|
BEQ R5, R0, l2
|
||||||
|
BNE R0, R6, l3
|
||||||
|
BNE R7, R0, l4
|
||||||
|
BLTZ R8, l5
|
||||||
|
BGEZ R9, l6
|
||||||
|
BLEZ R10, l7
|
||||||
|
BGTZ R11, l8
|
||||||
|
JMP l9
|
||||||
|
l1:
|
||||||
|
JMP l10
|
||||||
|
l2:
|
||||||
|
JMP l11
|
||||||
|
l3:
|
||||||
|
JMP l12
|
||||||
|
l4:
|
||||||
|
JMP l13
|
||||||
|
l5:
|
||||||
|
JMP l14
|
||||||
|
l6:
|
||||||
|
JMP l15
|
||||||
|
l7:
|
||||||
|
JMP l16
|
||||||
|
l8:
|
||||||
|
JMP l16
|
||||||
|
l9:
|
||||||
|
MOVV R1, R2
|
||||||
|
l10:
|
||||||
|
LL (R12), R13
|
||||||
|
LLV (R14), R15
|
||||||
|
SC R16, (R17)
|
||||||
|
SCV R18, (R19)
|
||||||
|
RDTIMED R20, R21
|
||||||
|
SYSCALL
|
||||||
|
DBAR
|
||||||
|
RET
|
||||||
|
l11:
|
||||||
|
JAL (R30)
|
||||||
|
RET
|
||||||
|
l12:
|
||||||
|
BSTRINSV $7, R4, $0, R5
|
||||||
|
BSTRPICKV $63, R6, $32, R7
|
||||||
|
ALSLV $2, R8, R9, R10
|
||||||
|
ADDV16 $65536, R11, R12
|
||||||
|
RET
|
||||||
|
l13:
|
||||||
|
MOVV $0xffffffffffffffff, R13
|
||||||
|
RET
|
||||||
|
l14:
|
||||||
|
CPUCFG R14, R14
|
||||||
|
RET
|
||||||
|
l15:
|
||||||
|
NOR R15, R16, R17
|
||||||
|
ORN R18, R19, R20
|
||||||
|
ANDN R21, R24, R25
|
||||||
|
RET
|
||||||
|
l16:
|
||||||
|
MOVB R26, (R27)
|
||||||
|
MOVB (R28), R29
|
||||||
|
RET
|
||||||
|
|
||||||
|
// sbdata loads and stores a static symbol with relocations (the relocation
|
||||||
|
// fields are masked before comparison).
|
||||||
|
GLOBL ·table(SB), RODATA, $16
|
||||||
|
DATA ·table+0(SB)/8, $0x1122334455667788
|
||||||
|
DATA ·table+8(SB)/8, $0x8877665544332211
|
||||||
|
|
||||||
|
TEXT ·sbdata(SB), NOSPLIT, $0-0
|
||||||
|
MOVV $·table(SB), R4
|
||||||
|
MOVV ·table(SB), R5
|
||||||
|
MOVV R6, ·table+8(SB)
|
||||||
|
RET
|
||||||
Vendored
+12
@@ -0,0 +1,12 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·largeimm(SB), NOSPLIT, $0
|
||||||
|
ADDI $2048, X5
|
||||||
|
ADDI $4095, X5, X6
|
||||||
|
ANDI $4095, X5, X6
|
||||||
|
ORI $-4096, X5, X6
|
||||||
|
XORI $0x12345, X5, X6
|
||||||
|
RET
|
||||||
Vendored
+24
@@ -0,0 +1,24 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·ldst(SB), NOSPLIT, $0
|
||||||
|
LD (X8), X9
|
||||||
|
SD X9, (X8)
|
||||||
|
LW (X8), X9
|
||||||
|
SW X9, (X8)
|
||||||
|
LD 8(X2), X10
|
||||||
|
SD X10, 16(X2)
|
||||||
|
LW 4(X2), X11
|
||||||
|
SW X11, 8(X2)
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·addi4spn(SB), NOSPLIT, $0
|
||||||
|
ADDI $16, X2, X8
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·wordarith(SB), NOSPLIT, $0
|
||||||
|
ADDW X9, X8, X8
|
||||||
|
SUBW X9, X8, X8
|
||||||
|
RET
|
||||||
Vendored
+19
@@ -0,0 +1,19 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·movimm(SB), NOSPLIT, $0
|
||||||
|
MOV $0, X10
|
||||||
|
MOV $5, X10
|
||||||
|
MOV $42, X10
|
||||||
|
MOV $-1, X10
|
||||||
|
MOV $-2048, X10
|
||||||
|
MOV $-2049, X10
|
||||||
|
MOV $2047, X10
|
||||||
|
MOV $2048, X10
|
||||||
|
MOV $4095, X10
|
||||||
|
MOV $-4096, X10
|
||||||
|
MOV $0x12345, X10
|
||||||
|
MOV $2147483647, X10
|
||||||
|
RET
|
||||||
Vendored
+32
@@ -0,0 +1,32 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·shifts(SB), NOSPLIT, $0
|
||||||
|
SLLI $3, X10, X10
|
||||||
|
SRLI $2, X10, X10
|
||||||
|
SRAI $1, X10, X10
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·logic(SB), NOSPLIT, $0
|
||||||
|
AND X11, X10, X10
|
||||||
|
OR X11, X10, X10
|
||||||
|
XOR X11, X10, X10
|
||||||
|
ANDI $7, X10, X10
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·mv(SB), NOSPLIT, $0
|
||||||
|
ADDI $0, X11, X10
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·nop(SB), NOSPLIT, $0
|
||||||
|
ADDI $0, X0
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·bigframe(SB), NOSPLIT, $24-0
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·ebreak(SB), NOSPLIT, $0
|
||||||
|
EBREAK
|
||||||
|
RET
|
||||||
@@ -5,7 +5,6 @@ package verify
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"testing"
|
"testing"
|
||||||
"unsafe"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
func loadABIKernel(t *testing.T) *Kernel {
|
func loadABIKernel(t *testing.T) *Kernel {
|
||||||
@@ -79,55 +78,6 @@ func TestABIR14Clobbered(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestABILZ4Kernels verifies that the production go-lz4 kernels are ABI-clean:
|
|
||||||
// they preserve BP and R14 and do not write into the red zone.
|
|
||||||
func TestABILZ4Kernels(t *testing.T) {
|
|
||||||
k := loadLZ4Kernel(t)
|
|
||||||
|
|
||||||
// wideCopyAVX2 with a real copy.
|
|
||||||
src := make([]byte, 128)
|
|
||||||
for i := range src {
|
|
||||||
src[i] = byte(i)
|
|
||||||
}
|
|
||||||
dst := make([]byte, 128)
|
|
||||||
|
|
||||||
args := make([]byte, 48)
|
|
||||||
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
|
|
||||||
PutUint64(args, 8, 128)
|
|
||||||
PutUint64(args, 16, 128)
|
|
||||||
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
|
||||||
PutUint64(args, 32, 128)
|
|
||||||
PutUint64(args, 40, 128)
|
|
||||||
|
|
||||||
_, report, err := k.CallFuncChecked("wideCopyAVX2", args)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("CallFuncChecked(wideCopyAVX2): %v", err)
|
|
||||||
}
|
|
||||||
if !report.OK() {
|
|
||||||
t.Errorf("wideCopyAVX2: %s", report)
|
|
||||||
}
|
|
||||||
|
|
||||||
// decodeBlockAVX2 with a simple block.
|
|
||||||
decSrc := []byte{0x50, 'H', 'e', 'l', 'l', 'o'}
|
|
||||||
decDst := make([]byte, 64)
|
|
||||||
|
|
||||||
decArgs := make([]byte, 64)
|
|
||||||
PutPtr(decArgs, 0, unsafe.Pointer(&decSrc[0]))
|
|
||||||
PutUint64(decArgs, 8, uint64(len(decSrc)))
|
|
||||||
PutUint64(decArgs, 16, uint64(cap(decSrc)))
|
|
||||||
PutPtr(decArgs, 24, unsafe.Pointer(&decDst[0]))
|
|
||||||
PutUint64(decArgs, 32, uint64(len(decDst)))
|
|
||||||
PutUint64(decArgs, 40, uint64(cap(decDst)))
|
|
||||||
|
|
||||||
_, report, err = k.CallFuncChecked("decodeBlockAVX2", decArgs)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("CallFuncChecked(decodeBlockAVX2): %v", err)
|
|
||||||
}
|
|
||||||
if !report.OK() {
|
|
||||||
t.Errorf("decodeBlockAVX2: %s", report)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCallFuncCheckedErrors(t *testing.T) {
|
func TestCallFuncCheckedErrors(t *testing.T) {
|
||||||
k := loadABIKernel(t)
|
k := loadABIKernel(t)
|
||||||
|
|
||||||
|
|||||||
@@ -1,144 +0,0 @@
|
|||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
||||||
// SPDX-License-Identifier: BSD-3-Clause
|
|
||||||
|
|
||||||
package verify
|
|
||||||
|
|
||||||
import (
|
|
||||||
"bytes"
|
|
||||||
"math/rand"
|
|
||||||
"os"
|
|
||||||
"testing"
|
|
||||||
"unsafe"
|
|
||||||
)
|
|
||||||
|
|
||||||
const lz4AVX512Path = "../../go-libraries/go-lz4/avx512_amd64.s"
|
|
||||||
|
|
||||||
func loadLZ4AVX512Kernel(t *testing.T) *Kernel {
|
|
||||||
t.Helper()
|
|
||||||
if _, err := os.Stat(lz4AVX512Path); err != nil {
|
|
||||||
t.Skipf("sibling kernel not available: %v", err)
|
|
||||||
}
|
|
||||||
k, err := Load(lz4AVX512Path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("Load(%s): %v", lz4AVX512Path, err)
|
|
||||||
}
|
|
||||||
t.Cleanup(k.Close)
|
|
||||||
return k
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestAVX512DecodeKnownAnswers(t *testing.T) {
|
|
||||||
k := loadLZ4AVX512Kernel(t)
|
|
||||||
|
|
||||||
tests := []struct {
|
|
||||||
name string
|
|
||||||
src []byte
|
|
||||||
wantN int
|
|
||||||
wantCode int
|
|
||||||
}{
|
|
||||||
{"literals_only", []byte{0x50, 'H', 'e', 'l', 'l', 'o'}, 5, 0},
|
|
||||||
{"literals_and_match", []byte{0x54, 'A', 'A', 'A', 'A', 'A', 0x05, 0x00, 0x30, 'B', 'B', 'B'}, 16, 0},
|
|
||||||
{"overlapping", []byte{0x14, 'X', 0x01, 0x00, 0x10, 'Y'}, 10, 0},
|
|
||||||
{"malformed", []byte{0x50, 'H', 'e'}, 0, 1},
|
|
||||||
{"zero_offset", []byte{0x14, 'X', 0x00, 0x00}, 0, 2},
|
|
||||||
}
|
|
||||||
for _, tt := range tests {
|
|
||||||
t.Run(tt.name, func(t *testing.T) {
|
|
||||||
dst := make([]byte, 64)
|
|
||||||
args := make([]byte, 64)
|
|
||||||
PutPtr(args, 0, unsafe.Pointer(&tt.src[0]))
|
|
||||||
PutUint64(args, 8, uint64(len(tt.src)))
|
|
||||||
PutUint64(args, 16, uint64(cap(tt.src)))
|
|
||||||
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
|
||||||
PutUint64(args, 32, uint64(len(dst)))
|
|
||||||
PutUint64(args, 40, uint64(cap(dst)))
|
|
||||||
|
|
||||||
out, err := k.CallFunc("decodeBlockAVX512", args)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("CallFunc: %v", err)
|
|
||||||
}
|
|
||||||
n := int(GetUint64(out, 48))
|
|
||||||
code := int(GetUint64(out, 56))
|
|
||||||
if n != tt.wantN || code != tt.wantCode {
|
|
||||||
t.Errorf("got (n=%d, code=%d), want (n=%d, code=%d)", n, code, tt.wantN, tt.wantCode)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestAVX512DifferentialFuzz(t *testing.T) {
|
|
||||||
k := loadLZ4AVX512Kernel(t)
|
|
||||||
rng := rand.New(rand.NewSource(77))
|
|
||||||
|
|
||||||
for i := 0; i < 3000; i++ {
|
|
||||||
wantSize := 1 + rng.Intn(4096)
|
|
||||||
src := genLZ4Block(rng, wantSize)
|
|
||||||
dstSize := wantSize + 64
|
|
||||||
|
|
||||||
goDst := make([]byte, dstSize)
|
|
||||||
goN, goCode := decodeBlockGo(src, goDst)
|
|
||||||
|
|
||||||
jitDst := make([]byte, dstSize)
|
|
||||||
args := make([]byte, 64)
|
|
||||||
if len(src) > 0 {
|
|
||||||
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
|
||||||
}
|
|
||||||
PutUint64(args, 8, uint64(len(src)))
|
|
||||||
PutUint64(args, 16, uint64(cap(src)))
|
|
||||||
if dstSize > 0 {
|
|
||||||
PutPtr(args, 24, unsafe.Pointer(&jitDst[0]))
|
|
||||||
}
|
|
||||||
PutUint64(args, 32, uint64(dstSize))
|
|
||||||
PutUint64(args, 40, uint64(cap(jitDst)))
|
|
||||||
|
|
||||||
out, err := k.CallFunc("decodeBlockAVX512", args)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("iter %d: %v", i, err)
|
|
||||||
}
|
|
||||||
jitN := int(GetUint64(out, 48))
|
|
||||||
jitCode := int(GetUint64(out, 56))
|
|
||||||
|
|
||||||
if jitCode != goCode {
|
|
||||||
t.Fatalf("iter %d: code mismatch: JIT=%d Go=%d", i, jitCode, goCode)
|
|
||||||
}
|
|
||||||
if jitCode != 0 {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
if jitN != goN {
|
|
||||||
t.Fatalf("iter %d: n mismatch: JIT=%d Go=%d", i, jitN, goN)
|
|
||||||
}
|
|
||||||
if !bytes.Equal(jitDst[:jitN], goDst[:goN]) {
|
|
||||||
t.Fatalf("iter %d: output mismatch (n=%d)", i, jitN)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestAVX512WideCopy(t *testing.T) {
|
|
||||||
k := loadLZ4AVX512Kernel(t)
|
|
||||||
|
|
||||||
sizes := []int{0, 1, 31, 32, 63, 64, 65, 127, 128, 256, 1024}
|
|
||||||
for _, n := range sizes {
|
|
||||||
src := make([]byte, n)
|
|
||||||
for i := range src {
|
|
||||||
src[i] = byte(i*11 + 3)
|
|
||||||
}
|
|
||||||
dst := make([]byte, n)
|
|
||||||
|
|
||||||
args := make([]byte, 48)
|
|
||||||
if n > 0 {
|
|
||||||
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
|
|
||||||
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
|
||||||
}
|
|
||||||
PutUint64(args, 8, uint64(n))
|
|
||||||
PutUint64(args, 16, uint64(n))
|
|
||||||
PutUint64(args, 32, uint64(n))
|
|
||||||
PutUint64(args, 40, uint64(n))
|
|
||||||
|
|
||||||
_, err := k.CallFunc("wideCopyAVX512", args)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("wideCopyAVX512(n=%d): %v", n, err)
|
|
||||||
}
|
|
||||||
if !bytes.Equal(dst, src) {
|
|
||||||
t.Errorf("wideCopyAVX512(n=%d): mismatch", n)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,150 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"encoding/hex"
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// BufSpec is one buffer allocation request parsed from the user's --buf spec.
|
||||||
|
type BufSpec struct {
|
||||||
|
Name string
|
||||||
|
Size int // declared slice length and capacity
|
||||||
|
Pattern string // "zero", "ones", "seq", or a hex blob
|
||||||
|
}
|
||||||
|
|
||||||
|
// ParseBufSpec parses a "name:size:pattern[,name:size:pattern]" spec string
|
||||||
|
// into individual buffer specs. Empty input yields an empty slice.
|
||||||
|
func ParseBufSpec(spec string) ([]BufSpec, error) {
|
||||||
|
if spec == "" {
|
||||||
|
return nil, nil
|
||||||
|
}
|
||||||
|
var out []BufSpec
|
||||||
|
for _, part := range strings.Split(spec, ",") {
|
||||||
|
fields := strings.SplitN(part, ":", 3)
|
||||||
|
if len(fields) != 3 {
|
||||||
|
return nil, fmt.Errorf("verify: invalid buffer spec %q (expected name:size:pattern)", part)
|
||||||
|
}
|
||||||
|
var size int
|
||||||
|
if _, err := fmt.Sscanf(fields[1], "%d", &size); err != nil || size <= 0 {
|
||||||
|
return nil, fmt.Errorf("verify: invalid buffer size %q in %q", fields[1], part)
|
||||||
|
}
|
||||||
|
out = append(out, BufSpec{Name: fields[0], Size: size, Pattern: fields[2]})
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// allocatedBuf is one live buffer in a pool.
|
||||||
|
type allocatedBuf struct {
|
||||||
|
spec BufSpec
|
||||||
|
data []byte // Size + safetyMargin bytes; the first Size are the live region
|
||||||
|
}
|
||||||
|
|
||||||
|
// safetyMargin is the extra bytes allocated past the declared size so SIMD
|
||||||
|
// over-reads and functions that read slightly past len never touch unmapped
|
||||||
|
// memory. Matches the margin used by the fuzz generator.
|
||||||
|
const safetyMargin = 8192
|
||||||
|
|
||||||
|
// BufPool is a set of allocated buffers held alive for the duration of one or
|
||||||
|
// more calls. Buffers live on the Go heap (the JIT call is in-process); the
|
||||||
|
// pool keeps the backing slices referenced so the GC does not collect them
|
||||||
|
// before the call returns.
|
||||||
|
type BufPool struct {
|
||||||
|
bufs []allocatedBuf
|
||||||
|
}
|
||||||
|
|
||||||
|
// Alloc allocates and fills the buffers described by specs. The returned
|
||||||
|
// pool must be kept alive until every call using it has returned.
|
||||||
|
func (p *BufPool) Alloc(specs []BufSpec) error {
|
||||||
|
for _, s := range specs {
|
||||||
|
data := make([]byte, s.Size+safetyMargin)
|
||||||
|
fillBuffer(data, s.Pattern)
|
||||||
|
p.bufs = append(p.bufs, allocatedBuf{spec: s, data: data})
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Close releases the pool. No-op for Go-heap buffers, but keeps the API
|
||||||
|
// symmetric with debug's mmap-backed pool.
|
||||||
|
func (p *BufPool) Close() {
|
||||||
|
p.bufs = nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// findByName returns the buffer with the given spec name, if any.
|
||||||
|
func (p *BufPool) findByName(name string) *allocatedBuf {
|
||||||
|
for i := range p.bufs {
|
||||||
|
if p.bufs[i].spec.Name == name {
|
||||||
|
return &p.bufs[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// BuildArgs constructs an ABI0 argument block of argSize bytes for the given
|
||||||
|
// layout, placing each buffer's pointer/length/capacity at the matching
|
||||||
|
// parameter offset. Parameters whose names match a buffer spec get the
|
||||||
|
// buffer address; non-pointer parameters and unmatched pointers are zeroed.
|
||||||
|
//
|
||||||
|
// Matching is by exact name, then by prefix (a buffer named "src" matches a
|
||||||
|
// parameter named "src" or "srcBuf"), mirroring the debug allocator.
|
||||||
|
func (p *BufPool) BuildArgs(layout []ArgOffset, argSize int) []byte {
|
||||||
|
args := make([]byte, argSize)
|
||||||
|
for _, a := range layout {
|
||||||
|
if !a.IsPtr {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
buf := p.matchBuf(a.Name)
|
||||||
|
if buf == nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if a.Offset+8 <= len(args) {
|
||||||
|
binary.LittleEndian.PutUint64(args[a.Offset:a.Offset+8], uint64(uintptr(unsafe.Pointer(&buf.data[0]))))
|
||||||
|
}
|
||||||
|
if strings.HasPrefix(a.Typ, "[]") && a.Offset+24 <= len(args) {
|
||||||
|
binary.LittleEndian.PutUint64(args[a.Offset+8:a.Offset+16], uint64(buf.spec.Size))
|
||||||
|
binary.LittleEndian.PutUint64(args[a.Offset+16:a.Offset+24], uint64(buf.spec.Size))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return args
|
||||||
|
}
|
||||||
|
|
||||||
|
// matchBuf finds a buffer matching the parameter name (exact, then prefix).
|
||||||
|
func (p *BufPool) matchBuf(name string) *allocatedBuf {
|
||||||
|
if b := p.findByName(name); b != nil {
|
||||||
|
return b
|
||||||
|
}
|
||||||
|
for i := range p.bufs {
|
||||||
|
if strings.HasPrefix(name, p.bufs[i].spec.Name) {
|
||||||
|
return &p.bufs[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// fillBuffer fills buf with the named pattern: "zero" (no-op, already zeroed),
|
||||||
|
// "ones" (0xFF), "seq" (i mod 256), or a hex blob repeated to fill.
|
||||||
|
func fillBuffer(buf []byte, pattern string) {
|
||||||
|
switch pattern {
|
||||||
|
case "zero":
|
||||||
|
// Already zeroed by make.
|
||||||
|
case "ones":
|
||||||
|
for i := range buf {
|
||||||
|
buf[i] = 0xFF
|
||||||
|
}
|
||||||
|
case "seq":
|
||||||
|
for i := range buf {
|
||||||
|
buf[i] = byte(i)
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
if data, err := hex.DecodeString(pattern); err == nil && len(data) > 0 {
|
||||||
|
for i := range buf {
|
||||||
|
buf[i] = data[i%len(data)]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,158 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestParseParamsExported(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
input string
|
||||||
|
names []string
|
||||||
|
types []string
|
||||||
|
isPtr []bool
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
input: "dst []byte, src []byte",
|
||||||
|
names: []string{"dst", "src"},
|
||||||
|
types: []string{"[]byte", "[]byte"},
|
||||||
|
isPtr: []bool{true, true},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
input: "dst, src []byte",
|
||||||
|
names: []string{"dst", "src"},
|
||||||
|
types: []string{"[]byte", "[]byte"},
|
||||||
|
isPtr: []bool{true, true},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
input: "a, b int",
|
||||||
|
names: []string{"a", "b"},
|
||||||
|
types: []string{"int", "int"},
|
||||||
|
isPtr: []bool{false, false},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
input: "src []byte, dst []byte",
|
||||||
|
names: []string{"src", "dst"},
|
||||||
|
types: []string{"[]byte", "[]byte"},
|
||||||
|
isPtr: []bool{true, true},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
input: "swin []int32, dstP []uint32, hist *[32]uint16",
|
||||||
|
names: []string{"swin", "dstP", "hist"},
|
||||||
|
types: []string{"[]int32", "[]uint32", "*[32]uint16"},
|
||||||
|
isPtr: []bool{true, true, true},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
input: "n int, code int",
|
||||||
|
names: []string{"n", "code"},
|
||||||
|
types: []string{"int", "int"},
|
||||||
|
isPtr: []bool{false, false},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
params := parseParamsExported(tt.input)
|
||||||
|
if len(params) != len(tt.names) {
|
||||||
|
t.Errorf("parseParamsExported(%q): got %d params, want %d", tt.input, len(params), len(tt.names))
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for i, p := range params {
|
||||||
|
if p.Name != tt.names[i] {
|
||||||
|
t.Errorf("parseParamsExported(%q)[%d].Name = %q, want %q", tt.input, i, p.Name, tt.names[i])
|
||||||
|
}
|
||||||
|
if p.Typ != tt.types[i] {
|
||||||
|
t.Errorf("parseParamsExported(%q)[%d].Typ = %q, want %q", tt.input, i, p.Typ, tt.types[i])
|
||||||
|
}
|
||||||
|
if p.IsPointer() != tt.isPtr[i] {
|
||||||
|
t.Errorf("parseParamsExported(%q)[%d].IsPointer() = %v, want %v", tt.input, i, p.IsPointer(), tt.isPtr[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestParseBufSpec(t *testing.T) {
|
||||||
|
t.Run("empty", func(t *testing.T) {
|
||||||
|
specs, err := ParseBufSpec("")
|
||||||
|
if err != nil || len(specs) != 0 {
|
||||||
|
t.Errorf("ParseBufSpec(\"\") = %v, %v; want nil, nil", specs, err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("single", func(t *testing.T) {
|
||||||
|
specs, err := ParseBufSpec("dst:64:zero")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if len(specs) != 1 || specs[0].Name != "dst" || specs[0].Size != 64 || specs[0].Pattern != "zero" {
|
||||||
|
t.Errorf("ParseBufSpec(\"dst:64:zero\") = %+v; want [{dst 64 zero}]", specs)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("multiple", func(t *testing.T) {
|
||||||
|
specs, err := ParseBufSpec("dst:64:zero,src:128:seq")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if len(specs) != 2 {
|
||||||
|
t.Fatalf("got %d specs, want 2", len(specs))
|
||||||
|
}
|
||||||
|
if specs[0].Name != "dst" || specs[1].Name != "src" {
|
||||||
|
t.Errorf("names = %s, %s; want dst, src", specs[0].Name, specs[1].Name)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("invalid", func(t *testing.T) {
|
||||||
|
_, err := ParseBufSpec("bad")
|
||||||
|
if err == nil {
|
||||||
|
t.Error("ParseBufSpec(\"bad\") should error")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("zero-size", func(t *testing.T) {
|
||||||
|
_, err := ParseBufSpec("dst:0:zero")
|
||||||
|
if err == nil {
|
||||||
|
t.Error("ParseBufSpec(\"dst:0:zero\") should error on zero size")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBufPoolBuildArgs(t *testing.T) {
|
||||||
|
specs, err := ParseBufSpec("dst:64:seq,src:128:zero")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var pool BufPool
|
||||||
|
if err := pool.Alloc(specs); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
defer pool.Close()
|
||||||
|
|
||||||
|
// Layout for wideCopyAVX2(dst, src []byte): dst at 0, src at 24.
|
||||||
|
layout := []ArgOffset{
|
||||||
|
{Name: "dst", Typ: "[]byte", Offset: 0, Size: 24, IsPtr: true},
|
||||||
|
{Name: "src", Typ: "[]byte", Offset: 24, Size: 24, IsPtr: true},
|
||||||
|
}
|
||||||
|
args := pool.BuildArgs(layout, 48)
|
||||||
|
|
||||||
|
// dst.ptr should be non-zero.
|
||||||
|
if args[0] == 0 && args[1] == 0 && args[2] == 0 && args[3] == 0 {
|
||||||
|
t.Error("dst.ptr is zero; expected a buffer address")
|
||||||
|
}
|
||||||
|
// dst.len should be 64 (0x40).
|
||||||
|
if args[8] != 0x40 {
|
||||||
|
t.Errorf("dst.len = %d, want 64", args[8])
|
||||||
|
}
|
||||||
|
// dst.cap should be 64.
|
||||||
|
if args[16] != 0x40 {
|
||||||
|
t.Errorf("dst.cap = %d, want 64", args[16])
|
||||||
|
}
|
||||||
|
// src.ptr should be non-zero.
|
||||||
|
if args[24] == 0 && args[25] == 0 && args[26] == 0 && args[27] == 0 {
|
||||||
|
t.Error("src.ptr is zero; expected a buffer address")
|
||||||
|
}
|
||||||
|
// src.len should be 128 (0x80).
|
||||||
|
if args[32] != 0x80 {
|
||||||
|
t.Errorf("src.len = %d, want 128", args[32])
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -5,7 +5,6 @@ package verify
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"testing"
|
"testing"
|
||||||
"unsafe"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
func TestBlocks(t *testing.T) {
|
func TestBlocks(t *testing.T) {
|
||||||
@@ -24,62 +23,3 @@ func TestBlocks(t *testing.T) {
|
|||||||
}
|
}
|
||||||
t.Logf("sum blocks: %v", blocks)
|
t.Logf("sum blocks: %v", blocks)
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestBlockCount(t *testing.T) {
|
|
||||||
k := loadLZ4Kernel(t)
|
|
||||||
|
|
||||||
n, err := k.BlockCount("decodeBlockAVX2")
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("BlockCount: %v", err)
|
|
||||||
}
|
|
||||||
// The decoder has many labels (dec_loop, dec_malformed, etc.).
|
|
||||||
if n < 10 {
|
|
||||||
t.Errorf("decodeBlockAVX2: expected at least 10 blocks, got %d", n)
|
|
||||||
}
|
|
||||||
t.Logf("decodeBlockAVX2: %d basic blocks", n)
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestProfilePaths(t *testing.T) {
|
|
||||||
k := loadLZ4Kernel(t)
|
|
||||||
|
|
||||||
// Build a corpus of varied LZ4 blocks.
|
|
||||||
var argSets [][]byte
|
|
||||||
blocks := []struct {
|
|
||||||
src []byte
|
|
||||||
dstSize int
|
|
||||||
}{
|
|
||||||
{[]byte{0x00}, 16}, // empty
|
|
||||||
{[]byte{0x50, 'H', 'e', 'l', 'l', 'o'}, 16}, // literals only
|
|
||||||
{[]byte{0x54, 'A', 'A', 'A', 'A', 'A', 5, 0, 0x30, 'B', 'B', 'B'}, 32}, // match
|
|
||||||
{[]byte{0x14, 'X', 1, 0, 0x10, 'Y'}, 16}, // overlapping
|
|
||||||
{[]byte{0x50, 'H'}, 16}, // malformed
|
|
||||||
{[]byte{0x14, 'X', 0, 0}, 16}, // zero offset
|
|
||||||
}
|
|
||||||
for _, b := range blocks {
|
|
||||||
args := make([]byte, 64)
|
|
||||||
if len(b.src) > 0 {
|
|
||||||
PutPtr(args, 0, unsafe.Pointer(&b.src[0]))
|
|
||||||
}
|
|
||||||
PutUint64(args, 8, uint64(len(b.src)))
|
|
||||||
PutUint64(args, 16, uint64(cap(b.src)))
|
|
||||||
dst := make([]byte, b.dstSize)
|
|
||||||
if len(dst) > 0 {
|
|
||||||
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
|
||||||
}
|
|
||||||
PutUint64(args, 32, uint64(len(dst)))
|
|
||||||
PutUint64(args, 40, uint64(cap(dst)))
|
|
||||||
argSets = append(argSets, args)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Result offsets: n+48 and code+56.
|
|
||||||
paths, err := k.ProfilePaths("decodeBlockAVX2", argSets, []int{48, 56})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("ProfilePaths: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// We expect at least 3 distinct paths: success (various n), malformed, zero offset.
|
|
||||||
if len(paths) < 3 {
|
|
||||||
t.Errorf("expected at least 3 distinct paths, got %d", len(paths))
|
|
||||||
}
|
|
||||||
t.Logf("decodeBlockAVX2: %d distinct output paths from %d inputs", len(paths), len(argSets))
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -1,295 +0,0 @@
|
|||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
||||||
// SPDX-License-Identifier: BSD-3-Clause
|
|
||||||
|
|
||||||
package verify
|
|
||||||
|
|
||||||
import (
|
|
||||||
"bytes"
|
|
||||||
"math/rand"
|
|
||||||
"testing"
|
|
||||||
"unsafe"
|
|
||||||
)
|
|
||||||
|
|
||||||
// decodeBlockGo is a minimal portable LZ4 block decoder used as the
|
|
||||||
// differential-testing oracle. It mirrors the contract of
|
|
||||||
// go-lz4's decodeBlockGo: (bytesWritten, code) where code is
|
|
||||||
// 0 = ok, 1 = malformed, 2 = zero offset.
|
|
||||||
func decodeBlockGo(src, dst []byte) (int, int) {
|
|
||||||
if len(src) == 0 {
|
|
||||||
return 0, 1
|
|
||||||
}
|
|
||||||
si, di := 0, 0
|
|
||||||
for {
|
|
||||||
if si >= len(src) {
|
|
||||||
return 0, 1 // truncated: no token
|
|
||||||
}
|
|
||||||
token := int(src[si])
|
|
||||||
si++
|
|
||||||
|
|
||||||
// Literals.
|
|
||||||
lLen := token >> 4
|
|
||||||
if lLen == 15 {
|
|
||||||
for {
|
|
||||||
if si >= len(src) {
|
|
||||||
return 0, 1
|
|
||||||
}
|
|
||||||
b := int(src[si])
|
|
||||||
si++
|
|
||||||
lLen += b
|
|
||||||
if b != 255 {
|
|
||||||
break
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if si+lLen > len(src) {
|
|
||||||
return 0, 1 // truncated literals
|
|
||||||
}
|
|
||||||
if di+lLen > len(dst) {
|
|
||||||
return 0, 1 // destination overflow
|
|
||||||
}
|
|
||||||
copy(dst[di:di+lLen], src[si:si+lLen])
|
|
||||||
di += lLen
|
|
||||||
si += lLen
|
|
||||||
|
|
||||||
// End of block.
|
|
||||||
if si >= len(src) {
|
|
||||||
return di, 0
|
|
||||||
}
|
|
||||||
|
|
||||||
// Match offset.
|
|
||||||
if si+2 > len(src) {
|
|
||||||
return 0, 1
|
|
||||||
}
|
|
||||||
offset := int(src[si]) | int(src[si+1])<<8
|
|
||||||
si += 2
|
|
||||||
if offset == 0 {
|
|
||||||
return 0, 2
|
|
||||||
}
|
|
||||||
|
|
||||||
// Match length.
|
|
||||||
mLen := token & 15
|
|
||||||
if mLen == 15 {
|
|
||||||
for {
|
|
||||||
if si >= len(src) {
|
|
||||||
return 0, 1
|
|
||||||
}
|
|
||||||
b := int(src[si])
|
|
||||||
si++
|
|
||||||
mLen += b
|
|
||||||
if b != 255 {
|
|
||||||
break
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
mLen += 4
|
|
||||||
|
|
||||||
// Copy match (overlapping-safe).
|
|
||||||
if di-offset < 0 {
|
|
||||||
return 0, 1 // offset reaches before dst start
|
|
||||||
}
|
|
||||||
if di+mLen > len(dst) {
|
|
||||||
return 0, 1 // destination overflow
|
|
||||||
}
|
|
||||||
for i := 0; i < mLen; i++ {
|
|
||||||
dst[di+i] = dst[di-offset+i]
|
|
||||||
}
|
|
||||||
di += mLen
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// genLZ4Block generates a random valid LZ4 block that decompresses into
|
|
||||||
// approximately wantSize bytes. The block is always well-formed (ends with
|
|
||||||
// a literals-only sequence).
|
|
||||||
func genLZ4Block(rng *rand.Rand, wantSize int) []byte {
|
|
||||||
var block []byte
|
|
||||||
produced := 0
|
|
||||||
for produced < wantSize {
|
|
||||||
remaining := wantSize - produced
|
|
||||||
|
|
||||||
// Decide: emit a literals+match sequence or the final literals.
|
|
||||||
if remaining <= 8 || rng.Intn(4) == 0 {
|
|
||||||
// Final literals-only sequence.
|
|
||||||
lLen := remaining
|
|
||||||
if lLen > 60 {
|
|
||||||
lLen = 1 + rng.Intn(60)
|
|
||||||
}
|
|
||||||
block = appendToken(block, lLen, 0)
|
|
||||||
for i := 0; i < lLen; i++ {
|
|
||||||
block = append(block, byte(rng.Intn(256)))
|
|
||||||
}
|
|
||||||
produced += lLen
|
|
||||||
break
|
|
||||||
}
|
|
||||||
|
|
||||||
// Literals + match.
|
|
||||||
lLen := rng.Intn(min(16, remaining))
|
|
||||||
if produced+lLen == 0 {
|
|
||||||
lLen = 1 // must have at least 1 literal before the first match
|
|
||||||
}
|
|
||||||
mLenRaw := rng.Intn(12) // match length = mLenRaw + 4
|
|
||||||
mLen := mLenRaw + 4
|
|
||||||
if produced+mLen > remaining {
|
|
||||||
mLen = remaining - produced
|
|
||||||
if mLen < 4 {
|
|
||||||
// Not enough room for a match; emit final literals.
|
|
||||||
lLen = remaining
|
|
||||||
block = appendToken(block, lLen, 0)
|
|
||||||
for i := 0; i < lLen; i++ {
|
|
||||||
block = append(block, byte(rng.Intn(256)))
|
|
||||||
}
|
|
||||||
break
|
|
||||||
}
|
|
||||||
mLenRaw = mLen - 4
|
|
||||||
}
|
|
||||||
|
|
||||||
block = appendToken(block, lLen, mLenRaw)
|
|
||||||
for i := 0; i < lLen; i++ {
|
|
||||||
block = append(block, byte(rng.Intn(256)))
|
|
||||||
}
|
|
||||||
produced += lLen
|
|
||||||
|
|
||||||
// Offset: must be <= produced (can't reference before start).
|
|
||||||
maxOff := produced
|
|
||||||
if maxOff > 65535 {
|
|
||||||
maxOff = 65535
|
|
||||||
}
|
|
||||||
offset := 1 + rng.Intn(maxOff)
|
|
||||||
block = append(block, byte(offset), byte(offset>>8))
|
|
||||||
produced += mLen
|
|
||||||
}
|
|
||||||
return block
|
|
||||||
}
|
|
||||||
|
|
||||||
// appendToken appends a token (and extension bytes if needed) for the given
|
|
||||||
// literal and match lengths.
|
|
||||||
func appendToken(block []byte, lLen, mLenRaw int) []byte {
|
|
||||||
lit4 := lLen
|
|
||||||
if lit4 > 15 {
|
|
||||||
lit4 = 15
|
|
||||||
}
|
|
||||||
ml4 := mLenRaw
|
|
||||||
if ml4 > 15 {
|
|
||||||
ml4 = 15
|
|
||||||
}
|
|
||||||
block = append(block, byte(lit4<<4|ml4))
|
|
||||||
// Literal extension bytes.
|
|
||||||
rem := lLen - 15
|
|
||||||
for rem >= 255 {
|
|
||||||
block = append(block, 255)
|
|
||||||
rem -= 255
|
|
||||||
}
|
|
||||||
if lLen >= 15 {
|
|
||||||
block = append(block, byte(rem))
|
|
||||||
}
|
|
||||||
// Match extension bytes.
|
|
||||||
rem = mLenRaw - 15
|
|
||||||
for rem >= 255 {
|
|
||||||
block = append(block, 255)
|
|
||||||
rem -= 255
|
|
||||||
}
|
|
||||||
if mLenRaw >= 15 {
|
|
||||||
block = append(block, byte(rem))
|
|
||||||
}
|
|
||||||
return block
|
|
||||||
}
|
|
||||||
|
|
||||||
func min(a, b int) int {
|
|
||||||
if a < b {
|
|
||||||
return a
|
|
||||||
}
|
|
||||||
return b
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestDifferentialLZ4Fuzz drives the JIT-assembled decodeBlockAVX2 with
|
|
||||||
// random valid LZ4 blocks and compares the output bit-for-bit against the
|
|
||||||
// portable Go reference.
|
|
||||||
func TestDifferentialLZ4Fuzz(t *testing.T) {
|
|
||||||
k := loadLZ4Kernel(t)
|
|
||||||
|
|
||||||
const iterations = 5000
|
|
||||||
rng := rand.New(rand.NewSource(42))
|
|
||||||
|
|
||||||
for i := 0; i < iterations; i++ {
|
|
||||||
wantSize := 1 + rng.Intn(4096)
|
|
||||||
src := genLZ4Block(rng, wantSize)
|
|
||||||
dstSize := wantSize + 64 // generous destination
|
|
||||||
|
|
||||||
// Go reference.
|
|
||||||
goDst := make([]byte, dstSize)
|
|
||||||
goN, goCode := decodeBlockGo(src, goDst)
|
|
||||||
|
|
||||||
// JIT kernel.
|
|
||||||
jitDst := make([]byte, dstSize)
|
|
||||||
jitN, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
|
|
||||||
|
|
||||||
if jitCode != goCode {
|
|
||||||
t.Fatalf("iter %d: code mismatch: JIT=%d, Go=%d (src len=%d)",
|
|
||||||
i, jitCode, goCode, len(src))
|
|
||||||
}
|
|
||||||
if jitCode != 0 {
|
|
||||||
continue // both agree it's malformed/zero-offset
|
|
||||||
}
|
|
||||||
if jitN != goN {
|
|
||||||
t.Fatalf("iter %d: n mismatch: JIT=%d, Go=%d (src len=%d)",
|
|
||||||
i, jitN, goN, len(src))
|
|
||||||
}
|
|
||||||
if !bytes.Equal(jitDst[:jitN], goDst[:goN]) {
|
|
||||||
t.Fatalf("iter %d: output mismatch (n=%d, src len=%d)", i, jitN, len(src))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestDifferentialLZ4Hostile drives the kernel with random garbage to check
|
|
||||||
// that error codes agree with the Go reference (no crashes, same classification).
|
|
||||||
func TestDifferentialLZ4Hostile(t *testing.T) {
|
|
||||||
k := loadLZ4Kernel(t)
|
|
||||||
|
|
||||||
const iterations = 2000
|
|
||||||
rng := rand.New(rand.NewSource(99))
|
|
||||||
|
|
||||||
for i := 0; i < iterations; i++ {
|
|
||||||
srcLen := rng.Intn(128)
|
|
||||||
src := make([]byte, srcLen)
|
|
||||||
rng.Read(src)
|
|
||||||
dstSize := rng.Intn(512)
|
|
||||||
dst := make([]byte, dstSize)
|
|
||||||
|
|
||||||
// Go reference.
|
|
||||||
goDst := make([]byte, dstSize)
|
|
||||||
copy(goDst, dst)
|
|
||||||
_, goCode := decodeBlockGo(src, goDst)
|
|
||||||
|
|
||||||
// JIT kernel.
|
|
||||||
jitDst := make([]byte, dstSize)
|
|
||||||
copy(jitDst, dst)
|
|
||||||
_, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
|
|
||||||
|
|
||||||
if jitCode != goCode {
|
|
||||||
t.Fatalf("iter %d: hostile code mismatch: JIT=%d, Go=%d (srcLen=%d, dstSize=%d)",
|
|
||||||
i, jitCode, goCode, srcLen, dstSize)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// callDecodeBlockAVX2Raw is like callDecodeBlockAVX2 but accepts explicit
|
|
||||||
// dst size (for hostile tests where dst may be smaller than the output).
|
|
||||||
func callDecodeBlockAVX2Raw(t *testing.T, k *Kernel, src, dst []byte) (int, int) {
|
|
||||||
t.Helper()
|
|
||||||
args := make([]byte, 64)
|
|
||||||
if len(src) > 0 {
|
|
||||||
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
|
||||||
}
|
|
||||||
PutUint64(args, 8, uint64(len(src)))
|
|
||||||
PutUint64(args, 16, uint64(cap(src)))
|
|
||||||
if len(dst) > 0 {
|
|
||||||
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
|
||||||
}
|
|
||||||
PutUint64(args, 32, uint64(len(dst)))
|
|
||||||
PutUint64(args, 40, uint64(cap(dst)))
|
|
||||||
|
|
||||||
out, err := k.CallFunc("decodeBlockAVX2", args)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("CallFunc(decodeBlockAVX2): %v", err)
|
|
||||||
}
|
|
||||||
return int(GetUint64(out, 48)), int(GetUint64(out, 56))
|
|
||||||
}
|
|
||||||
@@ -1,330 +0,0 @@
|
|||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
||||||
// SPDX-License-Identifier: BSD-3-Clause
|
|
||||||
|
|
||||||
package verify
|
|
||||||
|
|
||||||
import (
|
|
||||||
"bytes"
|
|
||||||
"math/rand"
|
|
||||||
"os"
|
|
||||||
"testing"
|
|
||||||
"unsafe"
|
|
||||||
)
|
|
||||||
|
|
||||||
const flacKernelPath = "../../go-libraries/go-flac/avx2_amd64.s"
|
|
||||||
|
|
||||||
func loadFLACKernel(t *testing.T) *Kernel {
|
|
||||||
t.Helper()
|
|
||||||
if _, err := os.Stat(flacKernelPath); err != nil {
|
|
||||||
t.Skipf("sibling kernel not available: %v", err)
|
|
||||||
}
|
|
||||||
k, err := Load(flacKernelPath)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("Load(%s): %v", flacKernelPath, err)
|
|
||||||
}
|
|
||||||
t.Cleanup(k.Close)
|
|
||||||
return k
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- Portable Go references (from go-flac/simd.go) ---
|
|
||||||
|
|
||||||
func decodeMono16Go(src []byte, dst []int32) {
|
|
||||||
for i := 0; i < len(dst); i++ {
|
|
||||||
dst[i] = int32(int16(uint16(src[2*i]) | uint16(src[2*i+1])<<8))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func pack16Go(dst []byte, src []int32) {
|
|
||||||
for i, v := range src {
|
|
||||||
dst[2*i] = byte(v)
|
|
||||||
dst[2*i+1] = byte(v >> 8)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func decorrelateLeftSideGo(left, right, out []int32) {
|
|
||||||
for i := range left {
|
|
||||||
l := left[i]
|
|
||||||
out[2*i] = l
|
|
||||||
out[2*i+1] = l - right[i]
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func decorrelateSideRightGo(left, right, out []int32) {
|
|
||||||
for i := range left {
|
|
||||||
side := left[i]
|
|
||||||
rch := right[i]
|
|
||||||
out[2*i] = rch + side
|
|
||||||
out[2*i+1] = rch
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func decorrelateMidSideGo(left, right, out []int32) {
|
|
||||||
for i := range left {
|
|
||||||
mid := left[i]
|
|
||||||
side := right[i]
|
|
||||||
mid2 := mid<<1 | (side & 1)
|
|
||||||
out[2*i] = (mid2 + side) >> 1
|
|
||||||
out[2*i+1] = (mid2 - side) >> 1
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func decorrelateInterleaveGo(left, right, out []int32) {
|
|
||||||
for i := range left {
|
|
||||||
out[2*i] = left[i]
|
|
||||||
out[2*i+1] = right[i]
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func analyzeO1RangeGo(swin []int32, dstP []uint32, hist *[32]uint16) (partSum uint64, overflow bool) {
|
|
||||||
swin = swin[:len(dstP)+1]
|
|
||||||
for j := 0; j+1 < len(swin); j++ {
|
|
||||||
r := swin[j+1] - swin[j]
|
|
||||||
if r == -2147483648 { // math.MinInt32
|
|
||||||
overflow = true
|
|
||||||
}
|
|
||||||
f := uint32(r<<1) ^ uint32(r>>31)
|
|
||||||
dstP[j] = f
|
|
||||||
partSum += uint64(f)
|
|
||||||
bl := 0
|
|
||||||
for v := f; v > 0; v >>= 1 {
|
|
||||||
bl++
|
|
||||||
}
|
|
||||||
if bl > 31 {
|
|
||||||
bl = 31
|
|
||||||
}
|
|
||||||
hist[bl]++
|
|
||||||
}
|
|
||||||
return
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- Differential tests ---
|
|
||||||
|
|
||||||
func TestFLACDecodeMono16(t *testing.T) {
|
|
||||||
k := loadFLACKernel(t)
|
|
||||||
rng := rand.New(rand.NewSource(7))
|
|
||||||
|
|
||||||
for iter := 0; iter < 500; iter++ {
|
|
||||||
n := rng.Intn(256)
|
|
||||||
src := make([]byte, 2*n)
|
|
||||||
rng.Read(src)
|
|
||||||
|
|
||||||
goDst := make([]int32, n)
|
|
||||||
decodeMono16Go(src, goDst)
|
|
||||||
|
|
||||||
jitDst := make([]int32, n)
|
|
||||||
args := make([]byte, 48)
|
|
||||||
if len(src) > 0 {
|
|
||||||
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
|
||||||
}
|
|
||||||
PutUint64(args, 8, uint64(len(src)))
|
|
||||||
PutUint64(args, 16, uint64(cap(src)))
|
|
||||||
if n > 0 {
|
|
||||||
PutPtr(args, 24, unsafe.Pointer(&jitDst[0]))
|
|
||||||
}
|
|
||||||
PutUint64(args, 32, uint64(n))
|
|
||||||
PutUint64(args, 40, uint64(cap(jitDst)))
|
|
||||||
|
|
||||||
_, err := k.CallFunc("decodeMono16AVX2", args)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("iter %d: %v", iter, err)
|
|
||||||
}
|
|
||||||
for i := range goDst {
|
|
||||||
if jitDst[i] != goDst[i] {
|
|
||||||
t.Fatalf("iter %d: mismatch at [%d]: JIT=%d Go=%d", iter, i, jitDst[i], goDst[i])
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestFLACPack16(t *testing.T) {
|
|
||||||
k := loadFLACKernel(t)
|
|
||||||
rng := rand.New(rand.NewSource(13))
|
|
||||||
|
|
||||||
for iter := 0; iter < 500; iter++ {
|
|
||||||
n := rng.Intn(256)
|
|
||||||
src := make([]int32, n)
|
|
||||||
for i := range src {
|
|
||||||
src[i] = int32(rng.Intn(65536) - 32768)
|
|
||||||
}
|
|
||||||
|
|
||||||
goDst := make([]byte, 2*n)
|
|
||||||
pack16Go(goDst, src)
|
|
||||||
|
|
||||||
jitDst := make([]byte, 2*n)
|
|
||||||
args := make([]byte, 48)
|
|
||||||
if len(jitDst) > 0 {
|
|
||||||
PutPtr(args, 0, unsafe.Pointer(&jitDst[0]))
|
|
||||||
}
|
|
||||||
PutUint64(args, 8, uint64(len(jitDst)))
|
|
||||||
PutUint64(args, 16, uint64(cap(jitDst)))
|
|
||||||
if n > 0 {
|
|
||||||
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
|
||||||
}
|
|
||||||
PutUint64(args, 32, uint64(n))
|
|
||||||
PutUint64(args, 40, uint64(cap(src)))
|
|
||||||
|
|
||||||
_, err := k.CallFunc("pack16AVX2", args)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("iter %d: %v", iter, err)
|
|
||||||
}
|
|
||||||
if !bytes.Equal(jitDst, goDst) {
|
|
||||||
t.Fatalf("iter %d: output mismatch (n=%d)", iter, n)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestFLACDecorrelate(t *testing.T) {
|
|
||||||
k := loadFLACKernel(t)
|
|
||||||
rng := rand.New(rand.NewSource(21))
|
|
||||||
|
|
||||||
kernels := []struct {
|
|
||||||
name string
|
|
||||||
ref func(left, right, out []int32)
|
|
||||||
}{
|
|
||||||
{"decorrelateLeftSideAVX2", decorrelateLeftSideGo},
|
|
||||||
{"decorrelateSideRightAVX2", decorrelateSideRightGo},
|
|
||||||
{"decorrelateMidSideAVX2", decorrelateMidSideGo},
|
|
||||||
{"decorrelateInterleaveAVX2", decorrelateInterleaveGo},
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, kk := range kernels {
|
|
||||||
t.Run(kk.name, func(t *testing.T) {
|
|
||||||
for iter := 0; iter < 200; iter++ {
|
|
||||||
n := rng.Intn(128)
|
|
||||||
left := make([]int32, n)
|
|
||||||
right := make([]int32, n)
|
|
||||||
for i := range left {
|
|
||||||
left[i] = int32(rng.Intn(1<<24) - 1<<23)
|
|
||||||
right[i] = int32(rng.Intn(1<<24) - 1<<23)
|
|
||||||
}
|
|
||||||
|
|
||||||
goOut := make([]int32, 2*n)
|
|
||||||
kk.ref(left, right, goOut)
|
|
||||||
|
|
||||||
jitOut := make([]int32, 2*n)
|
|
||||||
args := make([]byte, 72)
|
|
||||||
if n > 0 {
|
|
||||||
PutPtr(args, 0, unsafe.Pointer(&left[0]))
|
|
||||||
PutPtr(args, 24, unsafe.Pointer(&right[0]))
|
|
||||||
PutPtr(args, 48, unsafe.Pointer(&jitOut[0]))
|
|
||||||
}
|
|
||||||
PutUint64(args, 8, uint64(n))
|
|
||||||
PutUint64(args, 16, uint64(cap(left)))
|
|
||||||
PutUint64(args, 32, uint64(n))
|
|
||||||
PutUint64(args, 40, uint64(cap(right)))
|
|
||||||
PutUint64(args, 56, uint64(2*n))
|
|
||||||
PutUint64(args, 64, uint64(cap(jitOut)))
|
|
||||||
|
|
||||||
_, err := k.CallFunc(kk.name, args)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("iter %d: %v", iter, err)
|
|
||||||
}
|
|
||||||
for i := range goOut {
|
|
||||||
if jitOut[i] != goOut[i] {
|
|
||||||
t.Fatalf("iter %d: mismatch at [%d]: JIT=%d Go=%d", iter, i, jitOut[i], goOut[i])
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestFLACAnalyzeO1Range(t *testing.T) {
|
|
||||||
k := loadFLACKernel(t)
|
|
||||||
rng := rand.New(rand.NewSource(33))
|
|
||||||
|
|
||||||
for iter := 0; iter < 300; iter++ {
|
|
||||||
n := 1 + rng.Intn(128) // partition size
|
|
||||||
swin := make([]int32, n+1)
|
|
||||||
for i := range swin {
|
|
||||||
swin[i] = int32(rng.Intn(1<<20) - 1<<19)
|
|
||||||
}
|
|
||||||
|
|
||||||
goDstP := make([]uint32, n)
|
|
||||||
var goHist [32]uint16
|
|
||||||
goSum, goOvf := analyzeO1RangeGo(swin, goDstP, &goHist)
|
|
||||||
|
|
||||||
jitDstP := make([]uint32, n)
|
|
||||||
var jitHist [32]uint16
|
|
||||||
args := make([]byte, 72) // 65 rounded up
|
|
||||||
PutPtr(args, 0, unsafe.Pointer(&swin[0]))
|
|
||||||
PutUint64(args, 8, uint64(len(swin)))
|
|
||||||
PutUint64(args, 16, uint64(cap(swin)))
|
|
||||||
PutPtr(args, 24, unsafe.Pointer(&jitDstP[0]))
|
|
||||||
PutUint64(args, 32, uint64(n))
|
|
||||||
PutUint64(args, 40, uint64(cap(jitDstP)))
|
|
||||||
PutPtr(args, 48, unsafe.Pointer(&jitHist[0]))
|
|
||||||
|
|
||||||
out, err := k.CallFunc("analyzeO1RangeAVX2", args)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("iter %d: %v", iter, err)
|
|
||||||
}
|
|
||||||
jitSum := GetUint64(out, 56)
|
|
||||||
jitOvf := out[64] != 0
|
|
||||||
|
|
||||||
if jitSum != goSum {
|
|
||||||
t.Fatalf("iter %d: partSum mismatch: JIT=%d Go=%d", iter, jitSum, goSum)
|
|
||||||
}
|
|
||||||
if jitOvf != goOvf {
|
|
||||||
t.Fatalf("iter %d: overflow mismatch: JIT=%v Go=%v", iter, jitOvf, goOvf)
|
|
||||||
}
|
|
||||||
for i := range goDstP {
|
|
||||||
if jitDstP[i] != goDstP[i] {
|
|
||||||
t.Fatalf("iter %d: dstP[%d] mismatch: JIT=%d Go=%d", iter, i, jitDstP[i], goDstP[i])
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if jitHist != goHist {
|
|
||||||
t.Fatalf("iter %d: hist mismatch: JIT=%v Go=%v", iter, jitHist, goHist)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestFLACFastStereoSums(t *testing.T) {
|
|
||||||
k := loadFLACKernel(t)
|
|
||||||
rng := rand.New(rand.NewSource(44))
|
|
||||||
|
|
||||||
for iter := 0; iter < 300; iter++ {
|
|
||||||
n := 1 + rng.Intn(256)
|
|
||||||
left := make([]int32, n)
|
|
||||||
right := make([]int32, n)
|
|
||||||
for i := range left {
|
|
||||||
left[i] = int32(rng.Intn(1<<24) - 1<<23)
|
|
||||||
right[i] = int32(rng.Intn(1<<24) - 1<<23)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Go reference: compute the four sums.
|
|
||||||
var goSums [4]uint64
|
|
||||||
for i := 0; i < n; i++ {
|
|
||||||
l := left[i]
|
|
||||||
r := right[i]
|
|
||||||
side := l - r
|
|
||||||
mid := (l + r) >> 1
|
|
||||||
goSums[0] += foldAbs(l) + foldAbs(r)
|
|
||||||
goSums[1] += foldAbs(l) + foldAbs(side)
|
|
||||||
goSums[2] += foldAbs(side) + foldAbs(r)
|
|
||||||
goSums[3] += foldAbs(mid) + foldAbs(side)
|
|
||||||
}
|
|
||||||
|
|
||||||
var jitSums [4]uint64
|
|
||||||
args := make([]byte, 56)
|
|
||||||
PutPtr(args, 0, unsafe.Pointer(&left[0]))
|
|
||||||
PutUint64(args, 8, uint64(n))
|
|
||||||
PutUint64(args, 16, uint64(cap(left)))
|
|
||||||
PutPtr(args, 24, unsafe.Pointer(&right[0]))
|
|
||||||
PutUint64(args, 32, uint64(n))
|
|
||||||
PutUint64(args, 40, uint64(cap(right)))
|
|
||||||
PutPtr(args, 48, unsafe.Pointer(&jitSums[0]))
|
|
||||||
|
|
||||||
_, err := k.CallFunc("fastStereoSumsAVX2", args)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("iter %d: %v", iter, err)
|
|
||||||
}
|
|
||||||
if jitSums != goSums {
|
|
||||||
t.Fatalf("iter %d: sums mismatch:\n JIT=%v\n Go =%v", iter, jitSums, goSums)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func foldAbs(v int32) uint64 {
|
|
||||||
return uint64(uint32(v<<1) ^ uint32(v>>31))
|
|
||||||
}
|
|
||||||
+387
@@ -0,0 +1,387 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"math/rand"
|
||||||
|
"regexp"
|
||||||
|
"runtime"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// FuzzResult reports the outcome of a differential fuzz campaign for one
|
||||||
|
// function.
|
||||||
|
type FuzzResult struct {
|
||||||
|
Func string
|
||||||
|
Iterations int
|
||||||
|
Matches int
|
||||||
|
Mismatches int
|
||||||
|
FirstFail string // description of the first mismatch ("" if none)
|
||||||
|
CrashInput []byte // input that caused the last crash/mismatch (nil if none)
|
||||||
|
}
|
||||||
|
|
||||||
|
// OK returns true when all iterations matched.
|
||||||
|
func (r FuzzResult) OK() bool { return r.Mismatches == 0 }
|
||||||
|
|
||||||
|
// String returns a human-readable summary.
|
||||||
|
func (r FuzzResult) String() string {
|
||||||
|
if r.OK() {
|
||||||
|
return fmt.Sprintf("%s: %d/%d iterations match", r.Func, r.Matches, r.Iterations)
|
||||||
|
}
|
||||||
|
s := fmt.Sprintf("%s: %d/%d match, %d MISMATCH — %s",
|
||||||
|
r.Func, r.Matches, r.Iterations, r.Mismatches, r.FirstFail)
|
||||||
|
if len(r.CrashInput) > 0 {
|
||||||
|
s += fmt.Sprintf("\n input: %x", r.CrashInput)
|
||||||
|
}
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
|
||||||
|
// funcSig is a parsed // func signature from the assembly source.
|
||||||
|
type funcSig struct {
|
||||||
|
name string
|
||||||
|
params []param
|
||||||
|
results []param
|
||||||
|
}
|
||||||
|
|
||||||
|
type param struct {
|
||||||
|
name string
|
||||||
|
typ string // "[]byte", "[]int32", "int", "*[32]uint16", etc.
|
||||||
|
}
|
||||||
|
|
||||||
|
// funcSigRe matches the conventional "// func name(...)" comment.
|
||||||
|
var funcSigRe = regexp.MustCompile(`^//\s*func\s+(\w+)\(([^)]*)\)\s*(.*)$`)
|
||||||
|
|
||||||
|
// parseFuncSig extracts the function signature from a "// func ..." comment.
|
||||||
|
func parseFuncSig(comment string) (funcSig, bool) {
|
||||||
|
m := funcSigRe.FindStringSubmatch(strings.TrimSpace(comment))
|
||||||
|
if m == nil {
|
||||||
|
return funcSig{}, false
|
||||||
|
}
|
||||||
|
sig := funcSig{name: m[1]}
|
||||||
|
sig.params = parseParams(m[2])
|
||||||
|
// Results may be "(a int, b int)" or "int" or "(int, error)".
|
||||||
|
res := strings.TrimSpace(m[3])
|
||||||
|
res = strings.TrimPrefix(res, "(")
|
||||||
|
res = strings.TrimSuffix(res, ")")
|
||||||
|
if res != "" {
|
||||||
|
sig.results = parseParams(res)
|
||||||
|
}
|
||||||
|
return sig, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseParams splits "a []byte, b []int32" into typed parameters, handling
|
||||||
|
// shared types ("a, b []int32").
|
||||||
|
func parseParams(s string) []param {
|
||||||
|
s = strings.TrimSpace(s)
|
||||||
|
if s == "" {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
fields := strings.Split(s, ",")
|
||||||
|
// First pass: extract the type from each field (if present).
|
||||||
|
types := make([]string, len(fields))
|
||||||
|
for i, field := range fields {
|
||||||
|
parts := strings.Fields(strings.TrimSpace(field))
|
||||||
|
if len(parts) >= 2 {
|
||||||
|
types[i] = parts[len(parts)-1]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Propagate types backward: a field without a type inherits from the next
|
||||||
|
// field that has one (e.g. "dst" inherits "[]byte" from "src []byte").
|
||||||
|
for i := range fields {
|
||||||
|
if types[i] == "" {
|
||||||
|
for j := i + 1; j < len(fields); j++ {
|
||||||
|
if types[j] != "" {
|
||||||
|
types[i] = types[j]
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
var out []param
|
||||||
|
for i, field := range fields {
|
||||||
|
field = strings.TrimSpace(field)
|
||||||
|
if field == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
parts := strings.Fields(field)
|
||||||
|
typ := types[i]
|
||||||
|
if typ == "" {
|
||||||
|
out = append(out, param{typ: parts[0]})
|
||||||
|
} else {
|
||||||
|
out = append(out, param{name: parts[0], typ: typ})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExtractSignatures scans assembly source for "// func name(...)" comments
|
||||||
|
// that immediately precede a TEXT directive, and returns the parsed
|
||||||
|
// signatures keyed by the function's short name.
|
||||||
|
func ExtractSignatures(src string) map[string]funcSig {
|
||||||
|
lines := strings.Split(src, "\n")
|
||||||
|
sigs := make(map[string]funcSig)
|
||||||
|
var comments []string
|
||||||
|
for _, line := range lines {
|
||||||
|
trimmed := strings.TrimSpace(line)
|
||||||
|
if strings.HasPrefix(trimmed, "//") {
|
||||||
|
comments = append(comments, trimmed)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if strings.HasPrefix(trimmed, "TEXT") {
|
||||||
|
// Search the comment block for the // func line.
|
||||||
|
for _, c := range comments {
|
||||||
|
if sig, ok := parseFuncSig(c); ok {
|
||||||
|
sigs[sig.name] = sig
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
comments = nil
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if trimmed != "" {
|
||||||
|
comments = nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return sigs
|
||||||
|
}
|
||||||
|
|
||||||
|
// FuzzFunc runs a differential fuzz campaign: it JIT-executes both the
|
||||||
|
// gasm-assembled and the go-tool-asm-assembled versions of the named
|
||||||
|
// function with random inputs derived from the // func signature, and
|
||||||
|
// compares the output argument area bit-for-bit.
|
||||||
|
//
|
||||||
|
// The signature comment must appear immediately above the TEXT directive
|
||||||
|
// in the source (the conventional Go assembly layout).
|
||||||
|
func (k *Kernel) FuzzFunc(name string, sig funcSig, goCode []byte, iterations int, seed int64) FuzzResult {
|
||||||
|
result := FuzzResult{Func: name, Iterations: iterations}
|
||||||
|
|
||||||
|
rng := rand.New(rand.NewSource(seed))
|
||||||
|
|
||||||
|
// Map the Go-assembled code into a second executable region.
|
||||||
|
goExec, err := Map(goCode)
|
||||||
|
if err != nil {
|
||||||
|
result.Mismatches = iterations
|
||||||
|
result.FirstFail = fmt.Sprintf("map go code: %v", err)
|
||||||
|
return result
|
||||||
|
}
|
||||||
|
defer goExec.Unmap()
|
||||||
|
|
||||||
|
fl, err := k.Func(name)
|
||||||
|
if err != nil {
|
||||||
|
result.Mismatches = iterations
|
||||||
|
result.FirstFail = err.Error()
|
||||||
|
return result
|
||||||
|
}
|
||||||
|
|
||||||
|
for i := 0; i < iterations; i++ {
|
||||||
|
// Generate inputs and build TWO independent arg blocks (one per
|
||||||
|
// version) so that functions which write to their arguments
|
||||||
|
// (e.g. histogram increments) don't corrupt the other's input.
|
||||||
|
gasmArgs, goArgs, bufs := genDualArgs(rng, sig, fl.Args)
|
||||||
|
|
||||||
|
// Save the current input for crash diagnostics.
|
||||||
|
result.CrashInput = gasmArgs
|
||||||
|
|
||||||
|
// Call the gasm version.
|
||||||
|
gasmOut, err := k.CallFunc(name, gasmArgs)
|
||||||
|
if err != nil {
|
||||||
|
result.Mismatches++
|
||||||
|
if result.FirstFail == "" {
|
||||||
|
result.FirstFail = fmt.Sprintf("iter %d: gasm call: %v", i, err)
|
||||||
|
}
|
||||||
|
runtime.KeepAlive(bufs)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
// Call the Go version (same function, independent buffers).
|
||||||
|
goOut, err := Call(goExec.FuncAddr(0), goArgs)
|
||||||
|
if err != nil {
|
||||||
|
result.Mismatches++
|
||||||
|
if result.FirstFail == "" {
|
||||||
|
result.FirstFail = fmt.Sprintf("iter %d: go call: %v", i, err)
|
||||||
|
}
|
||||||
|
runtime.KeepAlive(bufs)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
// Compare only the result area (after all input parameters).
|
||||||
|
// Pointers in the arg block differ (separate buffers), so we
|
||||||
|
// compare from resultOff to the end.
|
||||||
|
resultOff := paramsSize(sig)
|
||||||
|
gasmRes := gasmOut[resultOff:]
|
||||||
|
goRes := goOut[resultOff:]
|
||||||
|
if !equalBytes(gasmRes, goRes) {
|
||||||
|
result.Mismatches++
|
||||||
|
if result.FirstFail == "" {
|
||||||
|
result.FirstFail = fmt.Sprintf("iter %d: output mismatch at result offset %d", i, resultOff)
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
result.Matches++
|
||||||
|
}
|
||||||
|
runtime.KeepAlive(bufs)
|
||||||
|
}
|
||||||
|
return result
|
||||||
|
}
|
||||||
|
|
||||||
|
// genDualArgs generates two independent ABI0 argument blocks (for gasm and
|
||||||
|
// go) with identical logical content but separate backing buffers, so that
|
||||||
|
// functions which write to their arguments don't corrupt the other's input.
|
||||||
|
func genDualArgs(rng *rand.Rand, sig funcSig, argSize int) (gasmArgs, goArgs []byte, bufs [][]byte) {
|
||||||
|
gasmArgs = make([]byte, argSize)
|
||||||
|
goArgs = make([]byte, argSize)
|
||||||
|
off := 0
|
||||||
|
sliceIdx := 0
|
||||||
|
|
||||||
|
for _, p := range sig.params {
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(p.typ, "[]"):
|
||||||
|
elemSize := elemSizeFor(p.typ)
|
||||||
|
n := 1 + rng.Intn(127)
|
||||||
|
var declaredLen int
|
||||||
|
if sliceIdx == 0 {
|
||||||
|
declaredLen = n
|
||||||
|
} else {
|
||||||
|
declaredLen = n + 512
|
||||||
|
}
|
||||||
|
// Allocate a buffer comfortably larger than declaredLen*elemSize so
|
||||||
|
// that SIMD over-reads and functions that write slightly past len
|
||||||
|
// (e.g. decoders that trust len(src)) never touch unmapped memory.
|
||||||
|
bufBytes := declaredLen*elemSize + 8192
|
||||||
|
// Two independent buffers with identical random content.
|
||||||
|
buf1 := make([]byte, bufBytes)
|
||||||
|
buf2 := make([]byte, bufBytes)
|
||||||
|
rng.Read(buf1[:n*elemSize])
|
||||||
|
copy(buf2, buf1)
|
||||||
|
bufs = append(bufs, buf1, buf2)
|
||||||
|
putPtr(gasmArgs, off, unsafe.Pointer(&buf1[0]))
|
||||||
|
putPtr(goArgs, off, unsafe.Pointer(&buf2[0]))
|
||||||
|
// len and cap both equal declaredLen — the buffer is guaranteed
|
||||||
|
// to hold at least declaredLen elements plus safety margin.
|
||||||
|
putU64(gasmArgs, off+8, uint64(declaredLen))
|
||||||
|
putU64(gasmArgs, off+16, uint64(declaredLen))
|
||||||
|
putU64(goArgs, off+8, uint64(declaredLen))
|
||||||
|
putU64(goArgs, off+16, uint64(declaredLen))
|
||||||
|
off += 24
|
||||||
|
sliceIdx++
|
||||||
|
|
||||||
|
case strings.HasPrefix(p.typ, "*["):
|
||||||
|
nElem := arrayLen(p.typ)
|
||||||
|
elem := elemSizeFor("[]" + p.typ[strings.Index(p.typ, "]")+1:])
|
||||||
|
size := nElem * elem
|
||||||
|
if size < 8 {
|
||||||
|
size = 8
|
||||||
|
}
|
||||||
|
buf1 := make([]byte, size)
|
||||||
|
buf2 := make([]byte, size)
|
||||||
|
rng.Read(buf1)
|
||||||
|
copy(buf2, buf1)
|
||||||
|
bufs = append(bufs, buf1, buf2)
|
||||||
|
putPtr(gasmArgs, off, unsafe.Pointer(&buf1[0]))
|
||||||
|
putPtr(goArgs, off, unsafe.Pointer(&buf2[0]))
|
||||||
|
off += 8
|
||||||
|
|
||||||
|
case p.typ == "int" || p.typ == "uint" || p.typ == "int64" || p.typ == "uint64":
|
||||||
|
v := uint64(rng.Intn(256))
|
||||||
|
putU64(gasmArgs, off, v)
|
||||||
|
putU64(goArgs, off, v)
|
||||||
|
off += 8
|
||||||
|
|
||||||
|
default:
|
||||||
|
v := rng.Uint64()
|
||||||
|
putU64(gasmArgs, off, v)
|
||||||
|
putU64(goArgs, off, v)
|
||||||
|
off += 8
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return gasmArgs, goArgs, bufs
|
||||||
|
}
|
||||||
|
|
||||||
|
func elemSizeFor(sliceType string) int {
|
||||||
|
switch strings.TrimPrefix(sliceType, "[]") {
|
||||||
|
case "byte", "uint8", "int8":
|
||||||
|
return 1
|
||||||
|
case "uint16", "int16":
|
||||||
|
return 2
|
||||||
|
case "uint32", "int32", "float32":
|
||||||
|
return 4
|
||||||
|
case "uint64", "int64", "float64":
|
||||||
|
return 8
|
||||||
|
default:
|
||||||
|
return 8
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// paramsSize returns the ABI0 stack size occupied by the input parameters.
|
||||||
|
func paramsSize(sig funcSig) int {
|
||||||
|
size := 0
|
||||||
|
for _, p := range sig.params {
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(p.typ, "[]"):
|
||||||
|
size += 24 // slice header
|
||||||
|
case strings.HasPrefix(p.typ, "*["):
|
||||||
|
size += 8 // pointer
|
||||||
|
case p.typ == "bool":
|
||||||
|
size += 1
|
||||||
|
default:
|
||||||
|
size += 8 // int, uint, etc.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return size
|
||||||
|
}
|
||||||
|
|
||||||
|
func arrayLen(typ string) int {
|
||||||
|
// "*[32]uint16" → 32
|
||||||
|
start := strings.Index(typ, "[")
|
||||||
|
end := strings.Index(typ, "]")
|
||||||
|
if start < 0 || end < 0 || end <= start {
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
n, _ := strconv.Atoi(typ[start+1 : end])
|
||||||
|
if n <= 0 {
|
||||||
|
n = 1
|
||||||
|
}
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
|
||||||
|
func putPtr(buf []byte, off int, p unsafe.Pointer) {
|
||||||
|
if off+8 <= len(buf) {
|
||||||
|
u64 := uint64(uintptr(p))
|
||||||
|
buf[off] = byte(u64)
|
||||||
|
buf[off+1] = byte(u64 >> 8)
|
||||||
|
buf[off+2] = byte(u64 >> 16)
|
||||||
|
buf[off+3] = byte(u64 >> 24)
|
||||||
|
buf[off+4] = byte(u64 >> 32)
|
||||||
|
buf[off+5] = byte(u64 >> 40)
|
||||||
|
buf[off+6] = byte(u64 >> 48)
|
||||||
|
buf[off+7] = byte(u64 >> 56)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func putU64(buf []byte, off int, v uint64) {
|
||||||
|
if off+8 <= len(buf) {
|
||||||
|
buf[off] = byte(v)
|
||||||
|
buf[off+1] = byte(v >> 8)
|
||||||
|
buf[off+2] = byte(v >> 16)
|
||||||
|
buf[off+3] = byte(v >> 24)
|
||||||
|
buf[off+4] = byte(v >> 32)
|
||||||
|
buf[off+5] = byte(v >> 40)
|
||||||
|
buf[off+6] = byte(v >> 48)
|
||||||
|
buf[off+7] = byte(v >> 56)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func equalBytes(a, b []byte) bool {
|
||||||
|
if len(a) != len(b) {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
for i := range a {
|
||||||
|
if a[i] != b[i] {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true
|
||||||
|
}
|
||||||
@@ -0,0 +1,167 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestFuzzResultString(t *testing.T) {
|
||||||
|
t.Run("ok", func(t *testing.T) {
|
||||||
|
r := FuzzResult{Func: "add", Iterations: 100, Matches: 100}
|
||||||
|
s := r.String()
|
||||||
|
if s != "add: 100/100 iterations match" {
|
||||||
|
t.Errorf("String() = %q", s)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("mismatch", func(t *testing.T) {
|
||||||
|
r := FuzzResult{Func: "mul", Iterations: 100, Matches: 95, Mismatches: 5, FirstFail: "iter 23"}
|
||||||
|
s := r.String()
|
||||||
|
if s != "mul: 95/100 match, 5 MISMATCH — iter 23" {
|
||||||
|
t.Errorf("String() = %q", s)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("crash", func(t *testing.T) {
|
||||||
|
r := FuzzResult{Func: "dec", Iterations: 100, Matches: 99, Mismatches: 1, FirstFail: "SIGSEGV", CrashInput: []byte{0x01, 0x02}}
|
||||||
|
s := r.String()
|
||||||
|
if s != "dec: 99/100 match, 1 MISMATCH — SIGSEGV\n input: 0102" {
|
||||||
|
t.Errorf("String() = %q", s)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestArrayLen(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
typ string
|
||||||
|
want int
|
||||||
|
}{
|
||||||
|
{"*[32]uint16", 32},
|
||||||
|
{"*[16]int32", 16},
|
||||||
|
{"bad", 1},
|
||||||
|
{"*[]", 1},
|
||||||
|
{"*[0x]", 1},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
if got := arrayLen(tt.typ); got != tt.want {
|
||||||
|
t.Errorf("arrayLen(%q) = %d, want %d", tt.typ, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestElemSizeFor(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
typ string
|
||||||
|
want int
|
||||||
|
}{
|
||||||
|
{"[]byte", 1}, {"[]uint8", 1}, {"[]int8", 1},
|
||||||
|
{"[]uint16", 2}, {"[]int16", 2},
|
||||||
|
{"[]uint32", 4}, {"[]int32", 4}, {"[]float32", 4},
|
||||||
|
{"[]uint64", 8}, {"[]int64", 8}, {"[]float64", 8},
|
||||||
|
{"[]unknown", 8},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
if got := elemSizeFor(tt.typ); got != tt.want {
|
||||||
|
t.Errorf("elemSizeFor(%q) = %d, want %d", tt.typ, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEqualBytes(t *testing.T) {
|
||||||
|
if !equalBytes([]byte{1, 2, 3}, []byte{1, 2, 3}) {
|
||||||
|
t.Error("expected equal")
|
||||||
|
}
|
||||||
|
if equalBytes([]byte{1, 2}, []byte{1, 2, 3}) {
|
||||||
|
t.Error("different length: expected not equal")
|
||||||
|
}
|
||||||
|
if equalBytes([]byte{1, 2, 3}, []byte{1, 2, 4}) {
|
||||||
|
t.Error("different content: expected not equal")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestParamsSize(t *testing.T) {
|
||||||
|
sig := funcSig{
|
||||||
|
name: "test",
|
||||||
|
params: []param{{name: "a", typ: "[]byte"}, {name: "b", typ: "int"}},
|
||||||
|
results: []param{{name: "n", typ: "int"}},
|
||||||
|
}
|
||||||
|
if got := paramsSize(sig); got != 32 {
|
||||||
|
t.Errorf("paramsSize = %d, want 32 (24 for slice + 8 for int)", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBlockCount(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
n, err := k.BlockCount("sum")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("BlockCount(sum): %v", err)
|
||||||
|
}
|
||||||
|
if n < 2 {
|
||||||
|
t.Errorf("sum: expected at least 2 blocks, got %d", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestFillBuffer(t *testing.T) {
|
||||||
|
t.Run("zero", func(t *testing.T) {
|
||||||
|
// fillBuffer("zero") is a no-op — relies on make already zeroing.
|
||||||
|
buf := make([]byte, 16)
|
||||||
|
fillBuffer(buf, "zero")
|
||||||
|
for _, b := range buf {
|
||||||
|
if b != 0 {
|
||||||
|
t.Error("zero pattern: make should produce zeroed buffer")
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
t.Run("ones", func(t *testing.T) {
|
||||||
|
buf := make([]byte, 16)
|
||||||
|
fillBuffer(buf, "ones")
|
||||||
|
for _, b := range buf {
|
||||||
|
if b != 0xFF {
|
||||||
|
t.Error("ones pattern should fill with 0xFF")
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
t.Run("seq", func(t *testing.T) {
|
||||||
|
buf := make([]byte, 256)
|
||||||
|
fillBuffer(buf, "seq")
|
||||||
|
for i, b := range buf {
|
||||||
|
if b != byte(i) {
|
||||||
|
t.Errorf("seq[%d] = %d, want %d", i, b, i)
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
t.Run("hex", func(t *testing.T) {
|
||||||
|
buf := make([]byte, 6)
|
||||||
|
fillBuffer(buf, "deadbeef")
|
||||||
|
want := []byte{0xDE, 0xAD, 0xBE, 0xEF, 0xDE, 0xAD}
|
||||||
|
for i, b := range buf {
|
||||||
|
if b != want[i] {
|
||||||
|
t.Errorf("hex[%d] = %02x, want %02x", i, b, want[i])
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestFuzzFuncChecked(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
sig := funcSig{
|
||||||
|
name: "sum",
|
||||||
|
params: []param{{name: "data", typ: "[]int64"}},
|
||||||
|
results: []param{{name: "r", typ: "int64"}},
|
||||||
|
}
|
||||||
|
result := k.FuzzFuncChecked("sum", sig, 10, 0)
|
||||||
|
if !result.OK() {
|
||||||
|
t.Errorf("FuzzFuncChecked(sum): %s", result)
|
||||||
|
}
|
||||||
|
// Test with non-existent function — should report failure.
|
||||||
|
result = k.FuzzFuncChecked("nope", sig, 10, 0)
|
||||||
|
if result.OK() {
|
||||||
|
t.Error("FuzzFuncChecked(nope): expected failure")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,98 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestExtractSignatures(t *testing.T) {
|
||||||
|
src := `// func add(a int64, b int64) int64
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func wideCopy(dst []byte, src []byte)
|
||||||
|
TEXT ·wideCopy(SB), NOSPLIT, $0-48
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
sigs := ExtractSignatures(src)
|
||||||
|
if len(sigs) != 2 {
|
||||||
|
t.Fatalf("expected 2 signatures, got %d: %v", len(sigs), sigs)
|
||||||
|
}
|
||||||
|
add, ok := sigs["add"]
|
||||||
|
if !ok {
|
||||||
|
t.Fatal("add not found")
|
||||||
|
}
|
||||||
|
if len(add.params) != 2 {
|
||||||
|
t.Errorf("add params: got %d, want 2", len(add.params))
|
||||||
|
}
|
||||||
|
wc, ok := sigs["wideCopy"]
|
||||||
|
if !ok {
|
||||||
|
t.Fatal("wideCopy not found")
|
||||||
|
}
|
||||||
|
if len(wc.params) != 2 {
|
||||||
|
t.Errorf("wideCopy params: got %d, want 2", len(wc.params))
|
||||||
|
}
|
||||||
|
if wc.params[0].typ != "[]byte" {
|
||||||
|
t.Errorf("wideCopy param[0].typ = %q, want []byte", wc.params[0].typ)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestFuzzWideCopy(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
|
||||||
|
gt, err := GroundTruth("../testdata/verify/basic_amd64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GroundTruth: %v", err)
|
||||||
|
}
|
||||||
|
goCode, ok := gt["wideCopy"]
|
||||||
|
if !ok {
|
||||||
|
t.Skip("wideCopy not in ground truth")
|
||||||
|
}
|
||||||
|
|
||||||
|
sig := funcSig{
|
||||||
|
name: "wideCopy",
|
||||||
|
params: []param{
|
||||||
|
{name: "dst", typ: "[]byte"},
|
||||||
|
{name: "src", typ: "[]byte"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
res := k.FuzzFunc("wideCopy", sig, goCode, 200, 42)
|
||||||
|
if !res.OK() {
|
||||||
|
t.Errorf("wideCopy fuzz: %s", res)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestParseFuncSig(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
comment string
|
||||||
|
name string
|
||||||
|
nParams int
|
||||||
|
}{
|
||||||
|
{"// func add(a int64, b int64) int64", "add", 2},
|
||||||
|
{"// func wideCopy(dst []byte, src []byte)", "wideCopy", 2},
|
||||||
|
{"// func analyzeO1RangeAVX2(swin []int32, dstP []uint32, hist *[32]uint16) (partSum uint64, overflow bool)", "analyzeO1RangeAVX2", 3},
|
||||||
|
{"// not a func", "", 0},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
sig, ok := parseFuncSig(tt.comment)
|
||||||
|
if tt.name == "" {
|
||||||
|
if ok {
|
||||||
|
t.Errorf("parseFuncSig(%q): expected not ok", tt.comment)
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !ok {
|
||||||
|
t.Errorf("parseFuncSig(%q): expected ok", tt.comment)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if sig.name != tt.name {
|
||||||
|
t.Errorf("parseFuncSig(%q).name = %q, want %q", tt.comment, sig.name, tt.name)
|
||||||
|
}
|
||||||
|
if len(sig.params) != tt.nParams {
|
||||||
|
t.Errorf("parseFuncSig(%q): %d params, want %d", tt.comment, len(sig.params), tt.nParams)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,195 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"runtime"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// GroundTruth assembles the given .s file with the Go toolchain's own
|
||||||
|
// assembler and returns the machine code bytes for each TEXT function,
|
||||||
|
// keyed by the function's short name (the part after the middle dot).
|
||||||
|
// This is the universal oracle: any file that `go tool asm` accepts can
|
||||||
|
// be verified, with no hand-written reference.
|
||||||
|
//
|
||||||
|
// For RISC-V sources the assembler is invoked with GOARCH=riscv64;
|
||||||
|
// the caller must set the architecture via GroundTruthArch.
|
||||||
|
func GroundTruth(path string) (map[string][]byte, error) {
|
||||||
|
return groundTruthArch(path, "")
|
||||||
|
}
|
||||||
|
|
||||||
|
// GroundTruthRISCV assembles the given .s file with the Go toolchain in
|
||||||
|
// RISC-V cross-assembly mode (GOARCH=riscv64).
|
||||||
|
func GroundTruthRISCV(path string) (map[string][]byte, error) {
|
||||||
|
return groundTruthArch(path, "riscv64")
|
||||||
|
}
|
||||||
|
|
||||||
|
// GroundTruthLOONG64 assembles the given .s file with the Go toolchain in
|
||||||
|
// LoongArch cross-assembly mode (GOARCH=loong64).
|
||||||
|
func GroundTruthLOONG64(path string) (map[string][]byte, error) {
|
||||||
|
return groundTruthArch(path, "loong64")
|
||||||
|
}
|
||||||
|
|
||||||
|
func groundTruthArch(path, goarch string) (map[string][]byte, error) {
|
||||||
|
goroot := runtime.GOROOT()
|
||||||
|
asmBin := filepath.Join(goroot, "pkg", "tool", runtime.GOOS+"_"+runtime.GOARCH, "asm")
|
||||||
|
if _, err := os.Stat(asmBin); err != nil {
|
||||||
|
return nil, fmt.Errorf("verify: go tool asm not found at %s: %w", asmBin, err)
|
||||||
|
}
|
||||||
|
includeDir := filepath.Join(goroot, "pkg", "include")
|
||||||
|
|
||||||
|
tmpDir, err := os.MkdirTemp("", "gasm-verify-*")
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("verify: tempdir: %w", err)
|
||||||
|
}
|
||||||
|
defer os.RemoveAll(tmpDir)
|
||||||
|
objPath := filepath.Join(tmpDir, "out.o")
|
||||||
|
|
||||||
|
base := filepath.Base(path)
|
||||||
|
pkg := strings.TrimSuffix(base, ".s")
|
||||||
|
pkg = strings.TrimSuffix(pkg, "_amd64")
|
||||||
|
pkg = strings.TrimSuffix(pkg, "_riscv64")
|
||||||
|
pkg = strings.TrimSuffix(pkg, "_loong64")
|
||||||
|
|
||||||
|
cmd := exec.Command(asmBin, "-I", includeDir, "-p", pkg, "-o", objPath, path)
|
||||||
|
if goarch != "" {
|
||||||
|
cmd.Env = append(os.Environ(), "GOARCH="+goarch)
|
||||||
|
}
|
||||||
|
if out, err := cmd.CombinedOutput(); err != nil {
|
||||||
|
return nil, fmt.Errorf("verify: go tool asm (%s): %w\n%s", goarch, err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
objData, err := os.ReadFile(objPath)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("verify: read object: %w", err)
|
||||||
|
}
|
||||||
|
return extractGOOBJCode(objData)
|
||||||
|
}
|
||||||
|
|
||||||
|
// GOOBJ block indices (cmd/internal/goobj).
|
||||||
|
const (
|
||||||
|
blkAutolib = iota
|
||||||
|
blkPkgIdx
|
||||||
|
blkFile
|
||||||
|
blkSymdef
|
||||||
|
blkHashed64def
|
||||||
|
blkHasheddef
|
||||||
|
blkNonpkgdef
|
||||||
|
blkNonpkgref
|
||||||
|
blkRefFlags
|
||||||
|
blkHash64
|
||||||
|
blkHash
|
||||||
|
blkRelocIdx
|
||||||
|
blkAuxIdx
|
||||||
|
blkDataIdx
|
||||||
|
blkReloc
|
||||||
|
blkAux
|
||||||
|
blkData
|
||||||
|
blkRefName
|
||||||
|
blkEnd
|
||||||
|
)
|
||||||
|
|
||||||
|
const goobjMagic = "\x00go120ld"
|
||||||
|
|
||||||
|
// extractGOOBJCode parses a GOOBJ payload and returns the code bytes for
|
||||||
|
// each non-package STEXT symbol (the functions).
|
||||||
|
func extractGOOBJCode(data []byte) (map[string][]byte, error) {
|
||||||
|
// Find the GOOBJ header (after the "go object ..." preamble).
|
||||||
|
i := bytes.Index(data, []byte(goobjMagic))
|
||||||
|
if i < 0 {
|
||||||
|
return nil, fmt.Errorf("verify: no GOOBJ magic in object file")
|
||||||
|
}
|
||||||
|
b := data[i:]
|
||||||
|
le := binary.LittleEndian
|
||||||
|
|
||||||
|
// Read block offsets (20 bytes into the header: 4 magic + 8 go version
|
||||||
|
// + 8 experiment = 20, then blkEnd+1 uint32 offsets).
|
||||||
|
var offs [blkEnd + 1]uint32
|
||||||
|
for j := 0; j <= blkEnd; j++ {
|
||||||
|
offs[j] = le.Uint32(b[20+4*j:])
|
||||||
|
}
|
||||||
|
blk := func(idx int) []byte { return b[offs[idx]:offs[idx+1]] }
|
||||||
|
|
||||||
|
// Parse non-package symbol definitions (blkNonpkgdef): each entry is
|
||||||
|
// 21 bytes: [nameLen:4][nameOff:4][abi:2][type:1][flag:1][flag2:1][size:4][align:4].
|
||||||
|
const symSize = 21
|
||||||
|
nonpkg := blk(blkNonpkgdef)
|
||||||
|
nSyms := len(nonpkg) / symSize
|
||||||
|
|
||||||
|
// Data index (blkDataIdx): one uint32 per defined symbol across ALL
|
||||||
|
// definition blocks (blkSymdef + blkHashed64def + blkHasheddef +
|
||||||
|
// blkNonpkgdef), in that order. We need the offset for the nonpkg
|
||||||
|
// symbols, which come last.
|
||||||
|
dataIdx := blk(blkDataIdx)
|
||||||
|
dataBlk := blk(blkData)
|
||||||
|
|
||||||
|
// Count symbols in the preceding definition blocks.
|
||||||
|
preceding := 0
|
||||||
|
for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef} {
|
||||||
|
preceding += len(blk(bi)) / symSize
|
||||||
|
}
|
||||||
|
|
||||||
|
// Symbol name offsets in the GOOBJ symbol table are absolute byte
|
||||||
|
// offsets from the start of the GOOBJ payload (the magic).
|
||||||
|
readStr := func(off, ln uint32) string {
|
||||||
|
if int(off+ln) > len(b) {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
return string(b[off : off+ln])
|
||||||
|
}
|
||||||
|
|
||||||
|
result := make(map[string][]byte)
|
||||||
|
const kindSTEXT = 1
|
||||||
|
for s := 0; s < nSyms; s++ {
|
||||||
|
x := nonpkg[s*symSize:]
|
||||||
|
nameLen := le.Uint32(x[0:])
|
||||||
|
nameOff := le.Uint32(x[4:])
|
||||||
|
typ := x[10]
|
||||||
|
size := le.Uint32(x[13:])
|
||||||
|
|
||||||
|
if typ != kindSTEXT || size == 0 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
name := readStr(nameOff, nameLen)
|
||||||
|
// Strip the package prefix (everything up to and including the
|
||||||
|
// last middle dot or period-dot).
|
||||||
|
name = stripPkg(name)
|
||||||
|
|
||||||
|
// Data offset from the index (nonpkg symbols follow the preceding blocks).
|
||||||
|
diIdx := preceding + s
|
||||||
|
if (diIdx+1)*4 > len(dataIdx) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
dOff := le.Uint32(dataIdx[diIdx*4:])
|
||||||
|
if int(dOff+size) > len(dataBlk) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
code := make([]byte, size)
|
||||||
|
copy(code, dataBlk[dOff:dOff+size])
|
||||||
|
result[name] = code
|
||||||
|
}
|
||||||
|
return result, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// stripPkg removes the package path prefix from a symbol name, leaving
|
||||||
|
// just the function name. "pkg/path·FuncName" → "FuncName".
|
||||||
|
func stripPkg(name string) string {
|
||||||
|
if i := strings.LastIndex(name, "\u00B7"); i >= 0 {
|
||||||
|
return name[i+len("\u00B7"):]
|
||||||
|
}
|
||||||
|
if i := strings.LastIndex(name, "\"."); i >= 0 {
|
||||||
|
return name[i+2:]
|
||||||
|
}
|
||||||
|
if i := strings.LastIndex(name, "."); i >= 0 {
|
||||||
|
return name[i+1:]
|
||||||
|
}
|
||||||
|
return name
|
||||||
|
}
|
||||||
@@ -0,0 +1,77 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestGroundTruthBasic(t *testing.T) {
|
||||||
|
// Use the simple test kernel — it assembles with go tool asm.
|
||||||
|
gt, err := GroundTruth("../testdata/verify/basic_amd64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GroundTruth: %v", err)
|
||||||
|
}
|
||||||
|
if len(gt) == 0 {
|
||||||
|
t.Fatal("no functions extracted from ground truth")
|
||||||
|
}
|
||||||
|
// The "add" function should be present and non-empty.
|
||||||
|
code, ok := gt["add"]
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("function 'add' not found in ground truth; got: %v", keys(gt))
|
||||||
|
}
|
||||||
|
if len(code) == 0 {
|
||||||
|
t.Fatal("add: zero-length code")
|
||||||
|
}
|
||||||
|
t.Logf("ground truth functions: %v", keys(gt))
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestGroundTruthComparison(t *testing.T) {
|
||||||
|
// Assemble with gasm and compare against go tool asm.
|
||||||
|
k, err := Load("../testdata/verify/basic_amd64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
defer k.Close()
|
||||||
|
|
||||||
|
gt, err := GroundTruth("../testdata/verify/basic_amd64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GroundTruth: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, name := range k.FuncNames() {
|
||||||
|
fl, _ := k.Func(name)
|
||||||
|
gasmCode := k.Image().Code[fl.Offset : fl.Offset+fl.Size]
|
||||||
|
goCode, ok := gt[name]
|
||||||
|
if !ok {
|
||||||
|
t.Errorf("%s: not in ground truth", name)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if len(gasmCode) != len(goCode) {
|
||||||
|
t.Errorf("%s: size mismatch: gasm=%d go=%d", name, len(gasmCode), len(goCode))
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for i := range gasmCode {
|
||||||
|
if gasmCode[i] != goCode[i] {
|
||||||
|
t.Errorf("%s: byte %d differs: gasm=%02x go=%02x", name, i, gasmCode[i], goCode[i])
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestGroundTruthBadFile(t *testing.T) {
|
||||||
|
_, err := GroundTruth("/nonexistent/file_amd64.s")
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error for nonexistent file")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func keys(m map[string][]byte) []string {
|
||||||
|
out := make([]string, 0, len(m))
|
||||||
|
for k := range m {
|
||||||
|
out = append(out, k)
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
@@ -5,6 +5,7 @@ package verify
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
|
"runtime"
|
||||||
"testing"
|
"testing"
|
||||||
"unsafe"
|
"unsafe"
|
||||||
)
|
)
|
||||||
@@ -68,6 +69,7 @@ func TestJITSum(t *testing.T) {
|
|||||||
PutUint64(args, 16, uint64(cap(tt.data)))
|
PutUint64(args, 16, uint64(cap(tt.data)))
|
||||||
|
|
||||||
out, err := k.CallFunc("sum", args)
|
out, err := k.CallFunc("sum", args)
|
||||||
|
runtime.KeepAlive(tt.data)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("CallFunc(sum, %v): %v", tt.data, err)
|
t.Fatalf("CallFunc(sum, %v): %v", tt.data, err)
|
||||||
}
|
}
|
||||||
@@ -111,6 +113,8 @@ func TestJITWideCopy(t *testing.T) {
|
|||||||
PutUint64(args, 40, uint64(tt.n)) // src_cap
|
PutUint64(args, 40, uint64(tt.n)) // src_cap
|
||||||
|
|
||||||
_, err := k.CallFunc("wideCopy", args)
|
_, err := k.CallFunc("wideCopy", args)
|
||||||
|
runtime.KeepAlive(dst)
|
||||||
|
runtime.KeepAlive(src)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("CallFunc(wideCopy): %v", err)
|
t.Fatalf("CallFunc(wideCopy): %v", err)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,110 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestGroundTruthLOONG64 assembles the loong64 test kernels with gasm and
|
||||||
|
// compares them byte-for-byte against `go tool asm` (GOARCH=loong64). The
|
||||||
|
// relocation fields of static-symbol references are masked before the
|
||||||
|
// comparison, since the toolchain leaves them zero for the linker.
|
||||||
|
func TestGroundTruthLOONG64(t *testing.T) {
|
||||||
|
for _, path := range []string{
|
||||||
|
"../testdata/verify/basic_loong64.s",
|
||||||
|
"../testdata/verify/fp_loong64.s",
|
||||||
|
} {
|
||||||
|
t.Run(path, func(t *testing.T) {
|
||||||
|
src, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read: %v", err)
|
||||||
|
}
|
||||||
|
f, errs := parser.Parse(path, string(src))
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := asm.AssembleFileLOONG64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||||
|
}
|
||||||
|
gt, err := GroundTruthLOONG64(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GroundTruthLOONG64: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
matched := 0
|
||||||
|
for _, fn := range img.Funcs {
|
||||||
|
gasmCode := maskRelocs(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
|
||||||
|
goCode, ok := gt[fn.Name]
|
||||||
|
if !ok {
|
||||||
|
t.Errorf("%s: not in ground truth (%d functions)", fn.Name, len(gt))
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
goCode = maskRelocs(goCode, fn.Relocs)
|
||||||
|
if !bytes.Equal(gasmCode, goCode) {
|
||||||
|
t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\n%s", fn.Name, len(gasmCode), len(goCode), diffHex(gasmCode, goCode))
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
matched++
|
||||||
|
t.Logf("%s: MATCH (%d bytes)", fn.Name, fn.Size)
|
||||||
|
}
|
||||||
|
if matched == 0 {
|
||||||
|
t.Fatal("no functions matched")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// maskRelocs zeroes the 4-byte immediate fields of the relocation sites.
|
||||||
|
func maskRelocs(code []byte, relocs []asm.Reloc) []byte {
|
||||||
|
for _, r := range relocs {
|
||||||
|
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||||
|
code[j] = 0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return code
|
||||||
|
}
|
||||||
|
|
||||||
|
func diffHex(a, b []byte) string {
|
||||||
|
var out bytes.Buffer
|
||||||
|
n := len(a)
|
||||||
|
if len(b) > n {
|
||||||
|
n = len(b)
|
||||||
|
}
|
||||||
|
for i := 0; i < n; i += 4 {
|
||||||
|
ab, bb := "??", "??"
|
||||||
|
if i+4 <= len(a) {
|
||||||
|
ab = fmt.Sprintf("%02x%02x%02x%02x", a[i], a[i+1], a[i+2], a[i+3])
|
||||||
|
} else if i < len(a) {
|
||||||
|
var sb strings.Builder
|
||||||
|
for j := i; j < len(a); j++ {
|
||||||
|
fmt.Fprintf(&sb, "%02x", a[j])
|
||||||
|
}
|
||||||
|
ab = sb.String()
|
||||||
|
}
|
||||||
|
if i+4 <= len(b) {
|
||||||
|
bb = fmt.Sprintf("%02x%02x%02x%02x", b[i], b[i+1], b[i+2], b[i+3])
|
||||||
|
} else if i < len(b) {
|
||||||
|
var sb strings.Builder
|
||||||
|
for j := i; j < len(b); j++ {
|
||||||
|
fmt.Fprintf(&sb, "%02x", b[j])
|
||||||
|
}
|
||||||
|
bb = sb.String()
|
||||||
|
}
|
||||||
|
mark := " "
|
||||||
|
if ab != bb {
|
||||||
|
mark = "!"
|
||||||
|
}
|
||||||
|
fmt.Fprintf(&out, "%04x: %s %s %s\n", i, ab, bb, mark)
|
||||||
|
}
|
||||||
|
return out.String()
|
||||||
|
}
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user