Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2ab6b9eb84 | ||
|
|
f41a86b660 | ||
|
|
7721353d44 | ||
|
|
243b087116 | ||
|
|
eee7a6d4a4 | ||
|
|
801fb963c9 | ||
|
|
a2acc9b5a3 | ||
|
|
f860bf8ce6 | ||
|
|
e7df5e5225 | ||
|
|
d114b3412c | ||
|
|
c77d68018c | ||
|
|
89d633f4bb | ||
|
|
f20e0bf1e7 | ||
|
|
1a45b66139 | ||
|
|
382efe538a | ||
|
|
f8d28a42ba | ||
|
|
51a2854d7f | ||
|
|
c234c3dd5b | ||
|
|
9262990ce5 | ||
|
|
d9f6167a4d | ||
|
|
f5088c52fc | ||
|
|
f52e23f1bc | ||
|
|
c9775c2b95 | ||
|
|
5af12e15ac | ||
|
|
db8e3fc160 | ||
|
|
1312122a99 | ||
|
|
11f962fbcc | ||
|
|
ee68859beb | ||
|
|
0920edb092 | ||
|
|
900c9772b1 | ||
|
|
b914c0e390 | ||
|
|
0f3146ff2c | ||
|
|
9370f9c3ee | ||
|
|
e98680597d | ||
|
|
1a01870695 | ||
|
|
458cfb626e | ||
|
|
56ecc39539 | ||
|
|
a82f575aee |
@@ -0,0 +1,157 @@
|
||||
# Release — gasm binaries. Runs on version tags (v0.28.0) pushed to main.
|
||||
name: Release
|
||||
|
||||
on:
|
||||
push:
|
||||
tags: ["v*"]
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: fedora
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- goos: linux
|
||||
goarch: amd64
|
||||
- goos: linux
|
||||
goarch: arm64
|
||||
- goos: linux
|
||||
goarch: riscv64
|
||||
- goos: linux
|
||||
goarch: loong64
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.26"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
|
||||
- name: Validate tag and build
|
||||
id: build
|
||||
env:
|
||||
VERSION: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
if ! echo "$VERSION" | grep -qE '^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$'; then
|
||||
echo "ERROR: expected a semver tag like v1.2.3, got: '$VERSION'"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
VERSION_NO_V="${VERSION#v}"
|
||||
echo "version_no_v=${VERSION_NO_V}" >> "$GITEA_OUTPUT"
|
||||
|
||||
mkdir -p bin
|
||||
GOOS=${{ matrix.goos }} GOARCH=${{ matrix.goarch }} CGO_ENABLED=0 \
|
||||
go build -ldflags "-s -w -X main.version=${VERSION_NO_V}" \
|
||||
-o "bin/gasm-${VERSION_NO_V}-${{ matrix.goos }}-${{ matrix.goarch }}" \
|
||||
./cmd/gasm
|
||||
|
||||
- name: Upload artifact
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: gasm-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||
path: bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||
if-no-files-found: error
|
||||
|
||||
- name: Smoke test
|
||||
if: matrix.goos == 'linux' && matrix.goarch == 'amd64'
|
||||
run: |
|
||||
chmod +x bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
|
||||
./bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }} --version
|
||||
|
||||
release:
|
||||
runs-on: fedora
|
||||
needs: build
|
||||
permissions:
|
||||
releases: write
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
path: dist
|
||||
|
||||
- name: Extract CHANGELOG section
|
||||
env:
|
||||
VERSION: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
VERSION_NO_V="${VERSION#v}"
|
||||
|
||||
sed -n "/^## \[${VERSION_NO_V}\] /,/^## \[/p" CHANGELOG.md \
|
||||
| sed '$d' \
|
||||
| tail -n +2 \
|
||||
> release-body.md
|
||||
|
||||
if [ ! -s release-body.md ]; then
|
||||
echo "ERROR: no CHANGELOG section found for ${VERSION_NO_V}"
|
||||
echo "Expected a heading like: ## [${VERSION_NO_V}] — YYYY-MM-DD"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Create release
|
||||
env:
|
||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
||||
GITEA_REPOSITORY: ${{ gitea.repository }}
|
||||
GITEA_REF_NAME: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
BODY=$(sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' -e 's/\t/\\t/g' -e 's/\r//g' release-body.md | sed ':a;N;$!ba;s/\n/\\n/g')
|
||||
BODY="\"${BODY}\""
|
||||
|
||||
response=$(curl -sS -w '\n%{http_code}' \
|
||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
||||
-H "Content-Type: application/json" \
|
||||
-X POST \
|
||||
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases" \
|
||||
-d "{\"tag_name\":\"${GITEA_REF_NAME}\",\"name\":\"${GITEA_REF_NAME}\",\"body\":${BODY},\"draft\":false,\"prerelease\":false}")
|
||||
|
||||
http_code=$(echo "$response" | tail -1)
|
||||
payload=$(echo "$response" | sed '$d')
|
||||
|
||||
echo "HTTP ${http_code}"
|
||||
if [ "$http_code" != "201" ]; then
|
||||
echo "Failed to create release: ${payload}"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
RELEASE_ID=$(echo "$payload" | grep -oE '"id"[[:space:]]*:[[:space:]]*[0-9]+' | head -1 | grep -oE '[0-9]+')
|
||||
echo "Created release ID=${RELEASE_ID}"
|
||||
printf '%s' "${RELEASE_ID}" > release-id.txt
|
||||
|
||||
- name: Upload assets
|
||||
env:
|
||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||
GITEA_SERVER_URL: ${{ gitea.server_url }}
|
||||
GITEA_REPOSITORY: ${{ gitea.repository }}
|
||||
GITEA_REF_NAME: ${{ gitea.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
RELEASE_ID=$(cat release-id.txt)
|
||||
|
||||
for binary in dist/gasm-*/gasm-*; do
|
||||
[ -f "$binary" ] || continue
|
||||
fname=$(basename "$binary")
|
||||
echo "Uploading ${fname}..."
|
||||
http_code=$(curl -sS -o /dev/null -w '%{http_code}' \
|
||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
||||
-H "Content-Type: application/octet-stream" \
|
||||
-X POST \
|
||||
--data-binary "@${binary}" \
|
||||
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases/${RELEASE_ID}/assets?name=${fname}")
|
||||
echo " HTTP ${http_code}"
|
||||
if [ "$http_code" != "201" ]; then
|
||||
echo "Failed to upload ${fname}"
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
|
||||
echo "Release ${GITEA_REF_NAME} is live."
|
||||
@@ -0,0 +1,96 @@
|
||||
# Test — gasm-devkit. Runs on push and pull request to development.
|
||||
name: Test
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [development]
|
||||
pull_request:
|
||||
branches: [development]
|
||||
|
||||
jobs:
|
||||
vet:
|
||||
runs-on: fedora
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.26"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
|
||||
- name: gofmt
|
||||
run: |
|
||||
set -euo pipefail
|
||||
unformatted=$(gofmt -l .)
|
||||
if [ -n "$unformatted" ]; then
|
||||
echo "These files need gofmt:"
|
||||
echo "$unformatted"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: go vet
|
||||
run: go vet ./...
|
||||
|
||||
test:
|
||||
runs-on: fedora
|
||||
needs: vet
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.26"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
|
||||
- name: Install gcc
|
||||
run: dnf install -y gcc
|
||||
|
||||
- name: go test -race
|
||||
run: go test -race -count=1 ./...
|
||||
|
||||
- name: Coverage gate — 80 % minimum
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# Exclude packages inherently untestable without hardware:
|
||||
# debug — interactive ptrace, requires a live process
|
||||
# cmd/gasm — CLI glue, covered by integration tests
|
||||
go test -coverprofile=coverage.out \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/arch \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/asm \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/ast \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/format \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/lexer \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/lint \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/lsp \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/parser \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/token \
|
||||
sourcedock.dev/petrbalvin/gasm-devkit/verify
|
||||
coverage=$(go tool cover -func=coverage.out | awk '/^total:/ { gsub("%", "", $3); print $3 }')
|
||||
echo "Total coverage: ${coverage}%"
|
||||
if awk -v c="$coverage" 'BEGIN { exit !(c+0 < 80) }'; then
|
||||
echo "ERROR: coverage ${coverage}% is below the 80% threshold"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
build:
|
||||
runs-on: fedora
|
||||
needs: test
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version: "1.26"
|
||||
|
||||
- name: Download dependencies
|
||||
run: go mod download
|
||||
|
||||
- name: Build
|
||||
run: go build -ldflags="-s -w" -o bin/gasm ./cmd/gasm
|
||||
|
||||
- name: Smoke test
|
||||
run: ./bin/gasm --version
|
||||
@@ -10,3 +10,6 @@ coverage.out
|
||||
# Editor detritus
|
||||
*.swp
|
||||
.DS_Store
|
||||
|
||||
# Scratch / temporary work
|
||||
_scratch/
|
||||
|
||||
@@ -0,0 +1,119 @@
|
||||
# AGENTS.md — gasm-devkit
|
||||
|
||||
Repository rules for AI agents and contributors. Read before modifying any
|
||||
code in this repository.
|
||||
|
||||
## AI Contribution Policy
|
||||
|
||||
AI agents may assist with code, documentation, tests, and review in this
|
||||
repository. All AI-assisted changes must:
|
||||
|
||||
- Follow the code style and conventions in this file.
|
||||
- Include the trailer `Assisted-by: <model-name>` in every commit message.
|
||||
- Not commit directly to `main` — work on `development`.
|
||||
- Pass the full Definition of Done before any commit.
|
||||
|
||||
## Workflow
|
||||
|
||||
- **Branching.** `development` is the working branch. `main` is
|
||||
release-only: merge from `development`, then tag. Never commit directly
|
||||
to `main`.
|
||||
- **Release procedure.**
|
||||
1. Bump `version` in `justfile` and `cmd/gasm/main.go`.
|
||||
2. Update `CHANGELOG.md` with a new `## [X.Y.Z] — YYYY-MM-DD` section.
|
||||
3. Update `README.md` and `docs/ARCHITECTURE.md` if user-visible
|
||||
behaviour changed.
|
||||
4. Run the Definition of Done (below).
|
||||
5. Commit on `development`.
|
||||
6. `git checkout main && git merge --ff-only development`.
|
||||
7. `git tag vX.Y.Z`.
|
||||
8. `git checkout development`.
|
||||
9. `GOBIN=~/.local/bin just install-bin`.
|
||||
|
||||
## Commit Messages
|
||||
|
||||
Conventional Commits, subject line only, imperative mood, lowercase after
|
||||
the colon:
|
||||
|
||||
```
|
||||
feat(asm): add EVEX gather and scatter with VSIB addressing
|
||||
```
|
||||
|
||||
Allowed types: `feat`, `fix`, `docs`, `style`, `refactor`, `perf`, `test`,
|
||||
`chore`, `ci`, `build`, `revert`.
|
||||
|
||||
Every commit ends with exactly one trailer, using the model that
|
||||
assisted with the change:
|
||||
|
||||
```
|
||||
Assisted-by: <model-name>
|
||||
```
|
||||
|
||||
Replace `<model-name>` with the actual model (e.g. `DeepSeek V4 Pro`).
|
||||
|
||||
No body, no footers, no trailing period on the subject.
|
||||
|
||||
## Code Style
|
||||
|
||||
Language: Go 1.26 (`toolchain go1.26.5`).
|
||||
|
||||
### Formatter
|
||||
|
||||
`gofmt` — zero diff. Run `just fmt` before committing.
|
||||
|
||||
### Linter
|
||||
|
||||
`go vet` — zero warnings. Run `just build` before committing.
|
||||
|
||||
### Tests
|
||||
|
||||
`go test -race -count=1 ./...` — all green, coverage ≥ 80 % (hard gate,
|
||||
enforced by `just test`).
|
||||
|
||||
### Dependencies
|
||||
|
||||
- **Production code:** standard library only. No third-party imports in
|
||||
shipped code.
|
||||
- **Test code:** `golang.org/x/arch` is the sole test dependency (decode
|
||||
oracle for round-trip validation). It is never linked into the binary.
|
||||
- **No cgo, no C, no external toolchains, no JavaScript.**
|
||||
|
||||
### Error Handling
|
||||
|
||||
Explicit `if err != nil`. Wrap with `fmt.Errorf("context: %w", err)`.
|
||||
No panics outside `main`. The one exception: the JIT trampoline's
|
||||
`recover`-guarded decoder hot path, which converts bounds panics to
|
||||
sentinel errors.
|
||||
|
||||
### Assembly
|
||||
|
||||
Plan 9 syntax (Go's assembler dialect). Hand-written — no code generators
|
||||
except `_gen/gen.go` for instruction tables (which parses the Go
|
||||
toolchain source). Every instruction table is committed; no runtime
|
||||
dependency on the Go toolchain.
|
||||
|
||||
### File Naming
|
||||
|
||||
- `_amd64.s`, `_arm64.s`, `_riscv64.s`, `_loong64.s` for
|
||||
architecture-specific assembly.
|
||||
- `_linux_amd64.go` for platform-specific Go files.
|
||||
- `_test.go` suffix for test files.
|
||||
|
||||
## Definition of Done
|
||||
|
||||
A task is not complete until all of these pass:
|
||||
|
||||
1. `just build` — `go vet` + `gofmt` check, zero errors, zero warnings.
|
||||
2. `just test` — full suite with `-race`, coverage ≥ 80 %.
|
||||
3. `just fmt` — produces no diff.
|
||||
4. Diagnostics — zero warnings across the project.
|
||||
5. Non-trivial changes reviewed.
|
||||
|
||||
## Licence
|
||||
|
||||
BSD-3-Clause. Every source file carries the SPDX header:
|
||||
|
||||
```
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
```
|
||||
+757
@@ -0,0 +1,757 @@
|
||||
# Changelog
|
||||
|
||||
All notable changes to gasm-devkit are documented here.
|
||||
|
||||
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
||||
and this project adheres to [Conventional Commits](https://www.conventionalcommits.org/).
|
||||
|
||||
## [development]
|
||||
|
||||
Unreleased changes on the `development` branch.
|
||||
|
||||
## [0.28.0] — 2026-08-03
|
||||
|
||||
RISC-V encoder: full RV64IMAFDC instruction set with RVC compression, MOV
|
||||
pseudo-instruction, SB/global symbol references, ELF64 object emission, and
|
||||
ground-truth verification against `GOARCH=riscv64 go tool asm`.
|
||||
|
||||
### Added
|
||||
|
||||
- **RISC-V encoder** — RV64I, RV64M, RV64A, RV64F/D, FMA, CSR, JALR.
|
||||
- **MOV pseudo-instruction** — load, store, reg-to-reg, immediate, frame mapping.
|
||||
- **RVC compression** — 22 compressed instruction types (C.LDSP, C.SDSP, C.FLDSP,
|
||||
C.FSDSP, C.ADDI, C.LI, C.LUI, C.ADDIW, C.MV, C.ADD, C.SUB, C.XOR, C.OR, C.AND,
|
||||
C.SLLI, C.SRLI, C.SRAI, C.ANDI, C.BEQZ, C.BNEZ, C.J, C.JR).
|
||||
- **SB/global symbols** — `MOV $sym(SB)`, `MOV sym(SB)`, `MOV rd, sym(SB)`
|
||||
encoded as AUIPC pairs with R_RISCV_PCREL_HI20/LO12 relocations.
|
||||
- **GLOBL/DATA** — data section layout in `AssembleFileRISCV`.
|
||||
- **ELF64 emission** — `gasm asm --format elf` produces EM_RISCV objects
|
||||
(.text, .data, .symtab, .rela.text).
|
||||
- **`gasm verify --ground-truth`** — byte-exact comparison against
|
||||
`GOARCH=riscv64 go tool asm`.
|
||||
- **`gasm verify --profile`** — function layout listing for RISC-V.
|
||||
- **CALL** — AUIPC + JALR pair encoding.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Parser: bare-number offset before `(SP)` no longer misidentified as pseudo.
|
||||
- MOV: `MOV $sym(FP/SP), rd` now returns an explicit error instead of silent fallback.
|
||||
- RVC: C.LDSP/C.SDSP/FLDSP/FSDSP immediate encoding now matches Go toolchain
|
||||
(bit-interleaved format).
|
||||
|
||||
### Verified
|
||||
|
||||
- 118 RISC-V tests, asm coverage 83.3%.
|
||||
- Ground-truth: C.LDSP, C.SDSP, C.FLDSP, C.FSDSP byte-exact vs Go toolchain.
|
||||
|
||||
## [0.27.0] — 2026-08-01
|
||||
|
||||
Subprocess isolation for `--fuzz`: each function is fuzzed in its own child
|
||||
process, so a partial function (decoder) that faults on random garbage is
|
||||
reported as "CRASH (partial function, use --ground-truth)" without killing
|
||||
the parent. CRASH is informational (exit 0); only MISMATCH is an error.
|
||||
|
||||
### Fixed
|
||||
|
||||
- `gasm verify --fuzz` no longer crashes the process on partial functions.
|
||||
|
||||
## [0.26.0] — 2026-07-31
|
||||
|
||||
Universal differential fuzzing: `gasm verify --fuzz` needs no hand-written
|
||||
reference. It parses the `// func` signature from the assembly source,
|
||||
generates typed random inputs (slices with random content, ints, pointers to
|
||||
fixed arrays), JIT-executes BOTH the gasm-assembled and the go-tool-asm-
|
||||
assembled versions with independent buffer copies, and compares the result
|
||||
area bit-for-bit.
|
||||
|
||||
### Added
|
||||
|
||||
- `verify`: `FuzzFunc` / `ExtractSignatures` / `parseFuncSig` — universal
|
||||
differential fuzz driven by the conventional `// func` comment. Each
|
||||
version gets its own buffer set (deep copy) so functions that write to
|
||||
their arguments (histogram increments) don't corrupt the other's input.
|
||||
- `gasm verify --fuzz [-n N]`: runs the differential fuzz for every function
|
||||
with a parseable signature. Total functions (wideCopy, pack16, decorrelate,
|
||||
analyze, autocorr) pass; partial functions (decoders that fault on malformed
|
||||
input) should use `--ground-truth` instead.
|
||||
|
||||
### Known limitation
|
||||
|
||||
`--fuzz` crashes the process for partial functions (e.g. LZ4 decoders) whose
|
||||
over-copy paths read past the buffer on random garbage input. Subprocess
|
||||
isolation (fork per function) is planned. Use `--ground-truth` for decoders.
|
||||
|
||||
## [0.25.0] — 2026-07-30
|
||||
|
||||
Universal ground-truth verification: `gasm verify --ground-truth` assembles
|
||||
any `.s` file with both gasm and `go tool asm`, then compares the machine
|
||||
code byte-for-byte per function (relocation sites masked). No hand-written
|
||||
reference needed — the Go toolchain IS the oracle.
|
||||
|
||||
### Added
|
||||
|
||||
- `verify`: `GroundTruth` — shells out to `go tool asm`, parses the GOOBJ
|
||||
output (minimal reader: block offsets, nonpkg symbol table, data index)
|
||||
and returns per-function code bytes.
|
||||
- `gasm verify --ground-truth`: compares gasm's output against the Go
|
||||
assembler's, reporting MATCH/MISMATCH per function with the first
|
||||
differing byte. Relocation disp32 fields (static-symbol references the
|
||||
linker fills) are masked before comparison.
|
||||
- Verified: go-lz4 AVX2 2/2, go-flac AVX2 17/17 functions byte-identical.
|
||||
|
||||
## [0.24.0] — 2026-07-29
|
||||
|
||||
The full analyze family and stereo PCM decode are now differentially tested.
|
||||
15 of 17 go-flac AVX2 kernels have bit-for-bit differential coverage; the
|
||||
two remaining (autocorrAVX2 — FMA reassociation, lpcResidualAVX2 — complex
|
||||
multi-arg) are deferred.
|
||||
|
||||
### Added
|
||||
|
||||
- `verify`: `analyzeO3RangeAVX2` and `analyzeO4RangeAVX2` differential tests
|
||||
(200 iterations each, same harness as O1/O2/Res).
|
||||
- `verify`: `decodeStereo16AVX2` differential test (500 random interleaved
|
||||
stereo PCM buffers, both channels compared sample-by-sample).
|
||||
|
||||
## [0.23.0] — 2026-07-28
|
||||
|
||||
The analyze family and 24-bit PCM decode join the differential suite.
|
||||
|
||||
### Added
|
||||
|
||||
- `verify`: `analyzeO2RangeAVX2` and `analyzeResRangeAVX2` differential
|
||||
tests (200 iterations each, shared harness with O1: zigzag fold, partial
|
||||
sum, overflow flag and Len32 histogram).
|
||||
- `verify`: `decodeMono24AVX2` differential test (500 random 24-bit PCM
|
||||
buffers, sign-extension compared sample-by-sample).
|
||||
|
||||
## [0.22.0] — 2026-07-27
|
||||
|
||||
The remaining go-flac encoder kernels join the differential suite.
|
||||
|
||||
### Added
|
||||
|
||||
- `verify`: `analyzeO1RangeAVX2` differential test (300 random partitions:
|
||||
zigzag fold, partial sum, overflow flag and the 32-bin Len32 histogram
|
||||
compared element-by-element against the portable Go reference).
|
||||
- `verify`: `fastStereoSumsAVX2` differential test (300 random stereo
|
||||
frames: the four zigzag-fold entropy sums compared against the scalar
|
||||
loop).
|
||||
|
||||
### Verified
|
||||
|
||||
- `gasm fmt` doc-comment indentation confirmed correct: comments before
|
||||
every TEXT are at column 0 (the RET-detection logic handles multi-exit
|
||||
functions).
|
||||
|
||||
## [0.21.0] — 2026-07-26
|
||||
|
||||
Differential testing extended to all four production kernels and the CLI
|
||||
exposes the full dynamic-analysis toolkit.
|
||||
|
||||
### Added
|
||||
|
||||
- `verify`: go-flac AVX2 differential tests — `decodeMono16AVX2` (500
|
||||
random PCM buffers), `pack16AVX2` (500 random int32→int16 packings) and
|
||||
all four decorrelation kernels (200 iterations each: left-side, side-right,
|
||||
mid-side, interleave) compared bit-for-bit against the portable Go
|
||||
references.
|
||||
- `verify`: go-lz4 AVX-512 differential tests — `decodeBlockAVX512` (3 000
|
||||
fuzzed LZ4 blocks + known answers) and `wideCopyAVX512` (0–1024 bytes)
|
||||
against the same portable oracle as the AVX2 suite.
|
||||
- `gasm verify --abi`: runs each NOSPLIT function with sentinel registers
|
||||
and a red-zone canary, reporting violations.
|
||||
- `gasm verify --profile`: lists the static basic-block count per function.
|
||||
|
||||
## [0.20.0] — 2026-07-25
|
||||
|
||||
Coverage profiling: the third pillar of Phase 3. Static basic-block
|
||||
enumeration from the assembler's label map, combined with multi-input path
|
||||
diversity measurement — how many observationally distinct execution paths a
|
||||
test corpus exercises.
|
||||
|
||||
### Added
|
||||
|
||||
- `verify`: `Kernel.Blocks` / `Kernel.BlockCount` — enumerate basic blocks
|
||||
from the assembler's local-label map (every jump target is a block
|
||||
boundary; the function entry is always a block). `decodeBlockAVX2` has
|
||||
27 blocks.
|
||||
- `verify`: `Kernel.ProfilePaths` — run the function with a corpus of
|
||||
argument blocks and collect distinct output fingerprints (the result
|
||||
words); reports path diversity as a lower bound on code coverage.
|
||||
|
||||
### Note
|
||||
|
||||
INT3-based per-block hit counting was prototyped but deferred: Go's runtime
|
||||
signal management (sigaltstack, handler re-installation) makes raw
|
||||
rt_sigaction handlers fragile in a Go process. The static + path-diversity
|
||||
approach delivers the project's goal (proving the SIMD path and tail handling
|
||||
execute) without fighting the runtime.
|
||||
|
||||
## [0.19.0] — 2026-07-24
|
||||
|
||||
Runtime ABI checks: the second pillar of Phase 3. The JIT trampoline now
|
||||
has an ABI-checking variant that sets sentinels in the callee-saved registers
|
||||
(BP, R14) before entering the assembled function and verifies they survive on
|
||||
return, plus a red-zone canary (128 bytes below SP filled with 0xA5) that
|
||||
detects any illegal write below the stack pointer.
|
||||
|
||||
### Added
|
||||
|
||||
- `verify`: `CallChecked` / `Kernel.CallFuncChecked` — ABI-checking JIT call
|
||||
with sentinel registers and red-zone canary; returns an `ABIReport`
|
||||
(BPClobbered, R14Clobbered, RedZoneHit).
|
||||
- `verify`: the raw `leaveJITCheckedRaw` trampoline — a TEXT symbol with no
|
||||
ABIInternal wrapper (address obtained via GLOBL/DATA), so the JIT
|
||||
function's RET lands directly in the check code and sees the registers
|
||||
exactly as the function left them.
|
||||
- Tests: deliberate BP/R14 clobberers detected; both go-lz4 kernels
|
||||
confirmed ABI-clean (BP preserved, R14 preserved, red zone intact).
|
||||
|
||||
## [0.18.0] — 2026-07-23
|
||||
|
||||
Differential testing: the JIT-assembled go-lz4 `decodeBlockAVX2` kernel is
|
||||
fuzzed against a portable Go reference — 5 000 valid LZ4 blocks compared
|
||||
bit-for-bit, plus 2 000 hostile (random garbage) inputs with matching error
|
||||
codes. This is the automated form of the project's bit-identical contract.
|
||||
|
||||
### Added
|
||||
|
||||
- `verify`: differential fuzz tests — a random LZ4 block generator produces
|
||||
valid blocks (literals, overlapping matches, extension bytes) and the
|
||||
JIT-assembled kernel's output is compared byte-for-byte against a portable
|
||||
Go decoder; a hostile-input suite confirms error-code agreement on random
|
||||
garbage (no crashes, same classification).
|
||||
|
||||
## [0.17.0] — 2026-07-22
|
||||
|
||||
Phase 3 begins: dynamic analysis. A JIT execution substrate that assembles
|
||||
Plan 9 amd64 kernels into executable memory and calls them directly — pure Go
|
||||
(stdlib only, `syscall.Mmap` + an assembly trampoline), no cgo, no external
|
||||
toolchain.
|
||||
|
||||
### Added
|
||||
|
||||
- `verify` package: JIT infrastructure — `Map` copies machine code into a
|
||||
W^X memory mapping, `Call` invokes it through an ABI0 trampoline that
|
||||
switches to a prepared stack and back. `Load`/`LoadSource`/`LoadAST`
|
||||
parse, assemble and map a `.s` file in one step; `Kernel.CallFunc`
|
||||
marshals the argument block and returns results.
|
||||
- `gasm verify` subcommand: assembles a file, JIT-loads it and reports the
|
||||
available functions; with `-smoke`, calls each NOSPLIT function with
|
||||
zeroed arguments to confirm the trampoline works end-to-end.
|
||||
- Integration tests: the go-lz4 `decodeBlockAVX2` and `wideCopyAVX2`
|
||||
kernels (699 and 146 bytes) assemble, map and execute correctly —
|
||||
known-answer LZ4 blocks decode bit-for-bit, wide copies of 0–1024 bytes
|
||||
match, malformed input returns the correct error codes.
|
||||
|
||||
### Verified
|
||||
|
||||
- `just test` (race, 84.6 % total coverage, verify 82.2 %).
|
||||
- `gasm verify` on both go-lz4 kernels: all functions JIT-load and
|
||||
smoke-test clean.
|
||||
|
||||
## [0.16.0] — 2026-07-21
|
||||
|
||||
The scalar conversions between vector and general-purpose registers — the
|
||||
last of the amd64 EVEX instruction set.
|
||||
|
||||
### Added
|
||||
|
||||
- `asm`: the GPR-interchanging conversions, byte for byte against the Go
|
||||
assembler (28 ground-truth cases including memory sources and extended
|
||||
GPRs): vector to GPR — the signed and truncated VCVT{,T}S{D,S}2SI{,Q}
|
||||
in both VEX and EVEX, and the unsigned VCVT{,T}S{D,S}2USI{L,Q}
|
||||
(EVEX only); GPR to vector — VCVTSI2SD{L,Q}/VCVTSI2SS{L,Q} (VEX and
|
||||
EVEX) and VCVTUSI2SD{L,Q}/VCVTUSI2SS{L,Q} (EVEX only), whose preserved
|
||||
vector source sits in vvvv (three Plan 9 operands).
|
||||
|
||||
## [0.15.0] — 2026-07-20
|
||||
|
||||
The last of the EVEX conversions and narrowing/extending moves — the EVEX
|
||||
instruction set is now complete save for the GPR-interchanging forms.
|
||||
|
||||
### Added
|
||||
|
||||
- `asm`: the unsigned and truncating conversions — VCVTPD2PS (and the X/Y
|
||||
spellings, whose length the spelling fixes), VCVTPD2UDQ (X/Y),
|
||||
VCVTTPD2UDQ (X/Y), VCVTTPD2UQQ, VCVTPS2UDQ, VCVTTPS2UDQ, VCVTPS2UQQ,
|
||||
VCVTTPS2UQQ, VCVTTPD2QQ, VCVTTPS2QQ, VCVTUQQ2PD, VCVTUQQ2PS (X/Y) and
|
||||
VCVTQQ2PS X/Y.
|
||||
- `asm`: the remaining sign/zero-extending moves (VPMOVSXBD/BQ/WQ and
|
||||
VPMOVZXBD/BQ/WD/WQ, VEX and EVEX) and the complete signed and unsigned
|
||||
narrowing stores (VPMOVS{DB,QB,DW,QW,QD,WB}, VPMOVUS{DB,QB,DW,QW,QD,WB},
|
||||
VPMOVDB, VPMOVQW).
|
||||
- `asm`: the mask/vector conversions (VPMOVM2B/W/D/Q and VPMOVB2M/W2M/
|
||||
D2M/Q2M), whose K register is a genuine operand rather than a mask and
|
||||
which therefore take no masking suffixes.
|
||||
|
||||
## [0.14.0] — 2026-07-19
|
||||
|
||||
The floating-point helper and conversion tail of the AVX-512 set, plus
|
||||
gather and scatter with VSIB addressing — every encoding verified byte for
|
||||
byte against the Go assembler.
|
||||
|
||||
### Added
|
||||
|
||||
- `asm`: the floating-point helpers — reciprocals and reciprocal square
|
||||
roots (VRCP14/VRSQRT14 PD/PS/SD/SS), exponents and mantissas (VGETEXP*,
|
||||
VGETMANT*), scaling by powers of two (VSCALEF*), rounding (VRNDSCALE*),
|
||||
reduction (VREDUCE*), immediate fixup (VFIXUPIMM*) and range selection
|
||||
(VRANGE*), and floating-point class tests (VFPCLASSPD/PS X/Y/Z and
|
||||
VFPCLASSSD/SS — a new immediate form whose reg field carries the opmask
|
||||
destination).
|
||||
- `asm`: **gather and scatter with VSIB addressing.** The gathers take
|
||||
both Go spellings: the VEX form with a vector mask register (OP mask,
|
||||
vsib, dst) and the EVEX form with an explicit K mask (OP vsib, K, dst),
|
||||
where the EVEX L'L field follows the VSIB index register rather than the
|
||||
data register (a ZMM index with an YMM destination encodes L'L = 10, as
|
||||
the Go assembler emits). The scatters (VSCATTER*/VPSCATTER*) are EVEX
|
||||
only (OP src, K, vsib). All eight gather and eight scatter widths.
|
||||
- `asm`: the remaining conversions — VCVTQQ2PS (the 512-bit source sets
|
||||
the length), VCVTPD2QQ/UQQ, VCVTPS2QQ, VCVTUDQ2PD/PS, the half-precision
|
||||
VCVTPH2PS and VCVTPS2PH (the extract layout with an immediate).
|
||||
|
||||
## [0.13.0] — 2026-07-18
|
||||
|
||||
The wider AVX-512 set: ternary logic, permutes, compares, expand/compress,
|
||||
the opmask instructions and the EVEX rounding/SAE/broadcast suffixes — every
|
||||
encoding verified byte for byte against the Go assembler.
|
||||
|
||||
### Added
|
||||
|
||||
- `asm`: the wider EVEX/AVX-512 set, across roughly sixty new ground-truth
|
||||
cases: ternary logic (VPTERNLOGD/Q), the lane shuffles/inserts/extracts
|
||||
(VSHUF{F,I}{32,64}X{2,4}, the VINSERT*/VEXTRACT* {F,I}{32,64}X{2,4,8}
|
||||
family, VPALIGNR), compares with an opmask destination (VCMPPD/PS/SD/SS —
|
||||
a new NDS-plus-immediate form with the K register in the reg field), the
|
||||
permutes (VPERMB/W, VPERMI2/T2 D/Q/PD), the wider integer families
|
||||
(VPMADDWD/UBSW, VPMULHUW, VPACKSSWB/USWB/SSDW/USDW, VPABS B/W/D/Q, the
|
||||
VPROL*/VPROR* rotates, the word shifts and the EVEX W1 qword shifts),
|
||||
expand/compress (VEXPANDPD/PS, VPEXPANDD/Q, VCOMPRESSPD/PS, VPCOMPRESSD/
|
||||
Q), the broadcasts (VPBROADCASTB/W from a GPR or memory, VBROADCASTSS/
|
||||
SD), the opmask-register instructions (KAND/KOR/KXNOR/KADD/KUNPCK/KNOT/
|
||||
KSHIFTL/KORTEST B/W/D/Q and KMOVQ, whose width the L/W/pp bits select),
|
||||
the packed single arithmetic (VADD/VSUB/VMUL/VDIV/VMIN/VMAX PS), the
|
||||
aligned moves (VMOVAPS/APD, VMOVDQA32/64, VMOVSS), the replicating moves
|
||||
(VMOVSLDUP/VMOVSHDUP), the conversions (VCVTPS2DQ, VCVTTPS2DQ) and the
|
||||
remaining extending and narrowing moves (VPMOVSXBW, VPMOVZXBW, VPMOVWB,
|
||||
VPMOVQB).
|
||||
- `asm`: the EVEX mnemonic suffixes the Go assembler accepts — the rounding
|
||||
modes `.RN_SAE`, `.RD_SAE`, `.RU_SAE`, `.RZ_SAE` (the EVEX b bit with the
|
||||
rounding control in L'L), suppress-all-exceptions `.SAE`, and memory
|
||||
broadcast `.BCST` (the b bit, the vector length preserved, disp8×N scaled
|
||||
by the element size) — each combinable with the `.Z` zeroing suffix,
|
||||
validated against the Go assembler's bytes, and rejected on instructions
|
||||
that do not support them.
|
||||
|
||||
## [0.12.0] — 2026-07-17
|
||||
|
||||
GOOBJ emission: gasm-assembled functions drop into a `go build` without the
|
||||
Go assembler.
|
||||
|
||||
### Added
|
||||
|
||||
- `asm`: **GOOBJ object output.** `gasm asm --format goobj -p <pkgpath>`
|
||||
writes the Go toolchain's own object format — the one `cmd/link` consumes
|
||||
directly: the functions as non-package symbols qualified with the package
|
||||
path (exactly as `cmd/asm` records assembly symbols), the `GLOBL` data,
|
||||
one serialized `FuncInfo` per function (argument/frame sizes, the asm
|
||||
func flag, the start line, the file table) and the four pc-value tables
|
||||
(`pcsp`, `pcfile`, `pcline`, `pcinline`). The `pcsp` table carries the
|
||||
real stack deltas: the assembler now tracks every stack-adjustment
|
||||
boundary through the prologue (`PUSHQ BP`, `SUBQ $frame, SP`) and each
|
||||
`RET`'s epilogue, so frame-pointer functions unwind correctly. The
|
||||
object preamble — the version-and-experiment header the linker compares
|
||||
verbatim — is captured from the installed `go tool asm`, so the output is
|
||||
always consistent with the toolchain that links it.
|
||||
- `asm`: relocations against file-local `GLOBL` symbols become `R_PCREL`
|
||||
entries in the GOOBJ output, with the instruction's displacement field
|
||||
left zero for the linker to fill (as `cmd/asm` leaves it).
|
||||
|
||||
### Fixed
|
||||
|
||||
- `parser`: 64-bit `DATA` literals above `MaxInt64`
|
||||
(`DATA mask<>+8(SB)/8, $0x800f…`) parse as unsigned and keep their bit
|
||||
pattern, instead of being rejected as non-integer.
|
||||
|
||||
### Verified
|
||||
|
||||
- End-to-end: a gasm-emitted GOOBJ swapped into a `go build` in place of
|
||||
the toolchain's assembly object links and runs with output identical to
|
||||
the baseline binary (stack-argument calls and a `GLOBL` relocation
|
||||
resolved by the Go linker). All 17 go-flac AVX2 kernel functions emit as
|
||||
a GOOBJ that `go tool nm` reads back with every symbol intact.
|
||||
|
||||
## [0.11.0] — 2026-07-16
|
||||
|
||||
Linkable object output: external symbols and relocatable ELF / Mach-O
|
||||
objects.
|
||||
|
||||
### Added
|
||||
|
||||
- `asm`: **object-file emission.** `gasm asm --format elf` writes an
|
||||
ELF64 relocatable object and `--format macho` a Mach-O x86-64
|
||||
`MH_OBJECT`: a code section (`.text` / `__TEXT,__text`) and a data
|
||||
section (`.data` / `__DATA,__data`), a symbol table with one symbol per
|
||||
`TEXT` and `GLOBL` (file-local `<>` symbols local, the rest global), and
|
||||
one PC-relative relocation per static-symbol reference
|
||||
(`R_X86_64_PC32` / `X86_64_RELOC_SIGNED`, the −4 addend the form needs).
|
||||
The ELF output is verified end-to-end: a gasm-emitted object links with
|
||||
a C driver and runs, resolving both a file-local constant and an
|
||||
external symbol; the Mach-O output is verified structurally with
|
||||
`debug/macho`.
|
||||
- `asm`: **external symbol references.** A reference to a symbol no
|
||||
`GLOBL` in the file defines no longer aborts assembly — it is recorded
|
||||
as an external relocation (`Image.Externals`, `FuncLayout.Relocs`) and
|
||||
becomes an undefined global symbol in the object output. The raw image
|
||||
format (`--format raw`, the default) still reports them: only an object
|
||||
file can represent a reference the linker must resolve.
|
||||
|
||||
### Changed
|
||||
|
||||
- `gasm asm` takes a `--format raw|elf|macho` flag selecting what `-o`
|
||||
writes; without `--format` the behaviour is unchanged (the concatenated
|
||||
image).
|
||||
|
||||
## [0.10.0] — 2026-07-15
|
||||
|
||||
The EVEX floating-point and conversion set: the packed-double arithmetic,
|
||||
the scalar SD/SS forms, VMOVDDUP and the width-changing conversions, each
|
||||
verified byte for byte against the Go assembler.
|
||||
|
||||
### Added
|
||||
|
||||
- `asm`: the rest of the common EVEX/VEX floating-point set — packed double
|
||||
arithmetic (VSUBPD, VDIVPD, VMINPD, VMAXPD, VUNPCKLPD and the EVEX form of
|
||||
VUNPCKHPD), the scalar double and single operations (VSUBSD, VDIVSD,
|
||||
VMINSD, VMAXSD and the full VADDSS/VSUBSS/VMULSS/VDIVSS/VMINSS/VMAXSS
|
||||
family in both VEX and EVEX — the EVEX scalar forms exist for masked and
|
||||
zeroing use), and VMOVDDUP (lane duplication, VEX and EVEX).
|
||||
- `asm`: the width-changing conversions — VCVTDQ2PS and VCVTPS2PD (VEX and
|
||||
EVEX; the destination sets the length for PS→PD), the EVEX form of
|
||||
VCVTDQ2PD, and the packed-double → dword family: VCVTPD2DQ/VCVTTPD2DQ
|
||||
(EVEX-512 only, a ZMM source and an XMM destination) and their X/Y
|
||||
spellings (VCVTPD2DQX/Y, VCVTTPD2DQX/Y), whose length follows the wider
|
||||
source — a new operand form, since the destination is always XMM while
|
||||
VEX.L / EVEX.L'L ride with the source (fixed by the spelling even for a
|
||||
memory source).
|
||||
- `asm`: masking and zeroing on every new form — the scalar SD/SS
|
||||
arithmetic, the unpacks, VMOVDDUP and the conversions all accept the
|
||||
explicit K1–K7 operand and the `.Z` suffix the way Go writes them.
|
||||
|
||||
### Documented
|
||||
|
||||
- VCVTPS2PD follows the Go assembler's encoding, which omits the F3
|
||||
mandatory prefix (VEX.pp / EVEX.pp = 00) that Intel's maps prescribe; the
|
||||
Go toolchain's machine code is the project's byte-for-byte oracle, and
|
||||
gasm reproduces it exactly (and round-trips through the x86 decoder, which
|
||||
shares the convention).
|
||||
|
||||
### Verified
|
||||
|
||||
- 58 new ground-truth cases — every instruction extracted from the Go
|
||||
toolchain's own assembly (go build + an executable-segment dump), checked
|
||||
byte for byte and round-tripped through the decoder, covering disp8×N for
|
||||
the scalar (×8/×4), duplication (×8/×32/×64) and conversion (×8/×16/×32)
|
||||
memory operands, the 5-bit register fields and the masked/zeroing P2
|
||||
byte. All four go-flac/go-lz4 kernels still assemble byte-identically
|
||||
and lint clean.
|
||||
|
||||
## [0.9.0] — 2026-07-14
|
||||
|
||||
AVX-512 masking and a wider EVEX integer set.
|
||||
|
||||
### Added
|
||||
|
||||
- `asm`: **EVEX masking** the way Go writes it — an explicit `K1`–`K7`
|
||||
operand placed among the operands (merging mask), and a `.Z` mnemonic
|
||||
suffix for zeroing (`VPADDD.Z Z1, Z2, K2, Z3`). Supported across the NDS,
|
||||
reg/rm, immediate-shift, align, extract, convert and move forms, including
|
||||
masked comparisons with a K destination (`VPCMPEQD Z0, Z3, K2, K1`). K0 is
|
||||
rejected as an explicit mask, and `.Z` without a mask is an error, matching
|
||||
the Go assembler.
|
||||
- `asm`: the common AVX-512 F/BW integer set — VPADDB/W, VPSUBB/W, VPANDD/Q,
|
||||
VPANDND/Q, VPMULLW, VPAVGB/W, the signed/unsigned min/max family for
|
||||
B/W/D/Q elements, the variable shifts VPSLLVD/Q, VPSRLVD/Q, VPSRAVD/Q, the
|
||||
EVEX forms of VPSHUFD/VPSHUFB, and the VMOVDQU8/VMOVDQU16 move aliases.
|
||||
Register indices 16–31 encode correctly (the mod=11 quirk carries rm[4]
|
||||
in X̄). All verified byte for byte against the Go assembler.
|
||||
- `lint`: masked EVEX forms (`.Z` suffix, K operands) are recognised by
|
||||
`unknown-instruction` and exempted from `operand-count`.
|
||||
|
||||
### Fixed
|
||||
|
||||
- `asm`: EVEX register–register operands with indices 16–31 encoded rm[4]
|
||||
into B̄ instead of X̄ (the EVEX mod=11 extension quirk), producing wrong
|
||||
prefix bytes for X16+/Y16+ r/m operands.
|
||||
|
||||
## [0.8.0] — 2026-07-13
|
||||
|
||||
Standard CLI ergonomics.
|
||||
|
||||
### Added
|
||||
|
||||
- `gasm --help` prints a proper top-level help (description, commands,
|
||||
flags, examples), and every subcommand now answers `-h`/`--help` with its
|
||||
own usage block (usage line, description, flag defaults), exiting 0. An
|
||||
unknown command points at `gasm --help` instead of dumping the whole usage.
|
||||
|
||||
### Changed
|
||||
|
||||
- The version is primarily available as the standard `gasm --version` / `-V`
|
||||
flag; the `gasm version` spelling remains as an alias.
|
||||
|
||||
## [0.7.0] — 2026-07-12
|
||||
|
||||
The formatter behaves like `go fmt` and canonicalises block separation.
|
||||
|
||||
### Added
|
||||
|
||||
- `gasm fmt` now works like `go fmt`: with no arguments — or with a directory
|
||||
argument — it reformats every `.s` file below it in place and lists the
|
||||
changed files, skipping `.` and `_` directories (`.git`, `_refs`, …).
|
||||
Explicit file arguments keep the `-w` / standard-output behaviour.
|
||||
|
||||
### Changed
|
||||
|
||||
- `format`: canonical blank-line layout — a new block (a label, `TEXT` or
|
||||
`GLOBL`) is preceded by exactly one blank line, neither more nor less.
|
||||
Comments leading a block stay with it (the blank line goes before them),
|
||||
stacked labels share their block, the function's first label keeps hugging
|
||||
its `TEXT`, and runs of blank lines collapse to one. The output remains
|
||||
idempotent and round-trips through the parser. All four go-flac/go-lz4
|
||||
kernels were reformatted with this release and remain byte-identical when
|
||||
assembled.
|
||||
|
||||
## [0.6.0] — 2026-07-11
|
||||
|
||||
Calibrated to the Go ABI: `register-clobber` stops reporting legal code, and
|
||||
the encoder learns the legacy SSE moves.
|
||||
|
||||
### Changed
|
||||
|
||||
- `lint`: **`register-clobber` is now calibrated to the Go ABI**
|
||||
(`cmd/compile/abi-internal.md`), not the platform ABI. Go's stack-based
|
||||
ABI0 has no System V style callee-saved registers — amd64 `BX`, `R12`–`R15`
|
||||
and the arm64/riscv64/loong64 scratch sets are caller-saved or permanent
|
||||
scratch, and hand-written kernels may clobber them freely. The rule now
|
||||
audits only the registers Go fixes across calls: the frame pointer and the
|
||||
goroutine pointer (amd64 `BP`/`R14`, arm64 `R18`/`R28`/`R29`, riscv64
|
||||
`X27`, loong64 `R22`), and the goroutine pointer is reported only when the
|
||||
function can reach the runtime (is not `NOSPLIT` or makes a call) — the
|
||||
ABI0 transition restores it on those paths, and NOSPLIT call-free leaves
|
||||
may use it, exactly as the runtime's own assembly does. Both go-flac
|
||||
kernels now lint with zero diagnostics.
|
||||
|
||||
### Fixed
|
||||
|
||||
- `lint`: the liveness analysis took the destination operand to be the
|
||||
*first* operand on arm64, riscv64 and loong64; Plan 9 spelling puts it last
|
||||
on every architecture Go supports. The def/use and save/restore
|
||||
classification on those architectures was inverted.
|
||||
- `format`: a comment that follows a `RET` (typically the next function's doc
|
||||
comment) is no longer indented as if it were still inside the finished
|
||||
function body.
|
||||
|
||||
### Added
|
||||
|
||||
- `asm`: the legacy (non-VEX) SSE moves — `MOVOU`/`MOVO` (the Plan 9 names
|
||||
for MOVDQU/MOVDQA), `MOVUPS`/`MOVAPS`/`MOVUPD`/`MOVAPD` and the scalar
|
||||
`MOVSD`/`MOVSS` — and `VMOVDQU64` in the EVEX set. All verified byte for
|
||||
byte against the Go assembler.
|
||||
|
||||
## [0.5.0] — 2026-07-10
|
||||
|
||||
EVEX / AVX-512: the go-flac AVX-512 kernel now assembles, byte-identically to
|
||||
the Go toolchain, completing the production-kernel coverage.
|
||||
|
||||
### Added
|
||||
|
||||
- `asm`: **EVEX (AVX-512) encoding** — the four-byte EVEX prefix with the
|
||||
5-bit register fields (Z0–Z31, X/Y 16–31, with the reg-r/m X̄ quirk and
|
||||
V'̄ shared between vvvv and the SIB index), opmask registers (K0–K7) as
|
||||
operands and as mask destinations, and the compressed disp8×N displacement
|
||||
(the multiplier follows the memory operand's size, as the Go assembler's
|
||||
opcode tables prescribe). Covers every AVX-512 instruction the go-flac
|
||||
kernels use: VPXORD/Q, VPADDD, VPSUBD/Q, VPUNPCK*DQ, VPMULLD/Q, VPERMD,
|
||||
VPSLLD/VPSRAD/VPSRAQ, VALIGND, VPCMPEQD (K destination), VMOVDQU32,
|
||||
VMOVUPD, VCVTQQ2PD, VPMOVSXDQ, the narrowing stores VPMOVDW/VPMOVQD, the
|
||||
extracts VEXTRACTI64X4/VEXTRACTF64X4, VFMADD231PD, VADDPD, VMULPD, the
|
||||
broadcasts VPBROADCASTD/Q (GPR and memory sources take different opcodes)
|
||||
and the mask moves KMOVW/KTESTW. Masking/zeroing suffixes are out of scope
|
||||
— the kernels use neither.
|
||||
- `asm`: `AssembleFile` now accepts file-defined global (`non-<>`) symbols
|
||||
too; a reference is external only when no `GLOBL` in the file defines it.
|
||||
|
||||
### Fixed
|
||||
|
||||
- `asm`: registers X16–Y31 force the EVEX encoding of dual-form mnemonics;
|
||||
previously a `VPBROADCASTD AX, Y30` fell into the VEX encoder, which cannot
|
||||
represent indices above 15 and silently truncated them.
|
||||
- `asm`: the VEX encoder now rejects vector register indices 16–31 instead of
|
||||
encoding a truncated (wrong) register.
|
||||
|
||||
### Verified
|
||||
|
||||
- All 10 functions of the go-flac `avx512_amd64.s` kernel assemble
|
||||
byte-identically to the Go toolchain's machine code (the disp32 of the one
|
||||
`VMOVDQU32 idx16(SB), Z13` load is linker-filled in Go and resolved within
|
||||
gasm's own image — checked to reach the right constant bytes). The AVX2
|
||||
kernel's 17 functions remain byte-identical.
|
||||
|
||||
## [0.4.0] — 2026-07-09
|
||||
|
||||
The standalone assembler reaches the whole go-flac AVX2 kernel: static
|
||||
symbols assemble, and all 17 kernel functions now match the Go toolchain's
|
||||
machine code byte for byte.
|
||||
|
||||
### Added
|
||||
|
||||
- `asm`: **file-level assembly** — `AssembleFile` turns a parsed file into an
|
||||
`Image`: the function bodies in source order followed by a data section
|
||||
built from the file's `GLOBL`/`DATA` directives (each symbol 16-aligned).
|
||||
- `asm`: **static-symbol (`SB`) operands** — `mask<>(SB)` references encode as
|
||||
RIP-relative loads with a patched disp32, resolved against the image layout
|
||||
so the output is self-consistent and position-independent. External
|
||||
(non-file-local) symbols are rejected with a clear error: they need
|
||||
object-file emission.
|
||||
- `gasm asm` prints the data section and symbol map alongside the functions
|
||||
and writes the whole image (code + data) with `-o`.
|
||||
|
||||
### Verified
|
||||
|
||||
- All 17 functions of the go-flac `avx2_amd64.s` kernel assemble
|
||||
byte-identically to the Go toolchain's machine code; the only differing
|
||||
bytes are the displacements of the two `VMOVDQU mask24<>(SB), X15` loads,
|
||||
which the Go linker fills at link time and gasm resolves within its own
|
||||
image (checked to reach the right constant bytes).
|
||||
|
||||
## [0.3.0] — 2026-07-08
|
||||
|
||||
The assembler reaches byte-identical parity with the Go toolchain on the
|
||||
production go-flac AVX2 kernels: every one of the 15 kernel functions that
|
||||
avoid global symbols now assembles to exactly the Go assembler's bytes (the
|
||||
two holdouts load a file-local constant through `SB` and wait on relocation
|
||||
support).
|
||||
|
||||
### Added
|
||||
|
||||
- `asm`: the scalar instruction families the kernels use — `CMOVcc` and
|
||||
`SETcc` (conditions spelled exactly like the jumps), `LZCNT`/`TZCNT`
|
||||
(legacy `F3 0F BD/BC`), the sign/zero-extending moves (`MOVBLZX`, `MOVBQZX`,
|
||||
`MOVWLZX`, `MOVWQZX`, `MOVWLSX`, `MOVLQSX`), `CVTSL2SD`/`CVTSQ2SD` (the
|
||||
legacy SSE encoding, as the Go assembler emits it), the traditional
|
||||
three-operand `IMUL3{W,L,Q}`, and the variable-count vector shifts
|
||||
(`VPSRLQ X0, Y8, Y8` — the count in an XMM register or memory takes the
|
||||
ordinary NDS form).
|
||||
- `asm`: **jump relaxation** — jumps start in the short (rel8) form and
|
||||
expand to rel32 when the settled displacement does not fit, iterating the
|
||||
layout to a fixed point (CALL is always rel32).
|
||||
- `asm`: **jump-to-jump folding** — a conditional jump to a label whose only
|
||||
instruction is an unconditional jump is redirected to the ultimate
|
||||
target, replicating the Go toolchain's linker, which chases such chains
|
||||
before it encodes branches.
|
||||
- `parser`: leading negative displacements with a base and index
|
||||
(`LEAQ -4(DX)(R9*4), R9`) parse into a fully populated address.
|
||||
|
||||
### Fixed
|
||||
|
||||
- `asm`: `CMP` with a register or memory operand computed **second − first**
|
||||
instead of first − second, silently inverting every condition that followed
|
||||
(`CMPQ SI, R10; JGE` tested R10 ≥ SI). The encoding now always records
|
||||
first − second — `CMP r/m, r` with the first operand in r/m, `CMP r, r/m`
|
||||
with the first operand in reg — and is byte-identical to the Go assembler.
|
||||
- `asm`: register-to-register `MOV` now uses the `r/m ← r` opcode (reg =
|
||||
source), the Go assembler's choice; the output is byte-identical.
|
||||
|
||||
## [0.2.0] — 2026-07-07
|
||||
|
||||
The Phase 2 assembler grows the SIMD set: shuffles, extract/insert, permute
|
||||
and the moves, on top of the Phase 1 VEX forms.
|
||||
|
||||
### Added
|
||||
|
||||
- `asm`: four new VEX (AVX/AVX2) operand forms, each validated by round-trip
|
||||
decoding through `golang.org/x/arch` **and** byte-for-byte against the
|
||||
machine code the real Go assembler emits:
|
||||
- the immediate shuffle (`VPSHUFD`, `VPERMQ`),
|
||||
- the three-operand-plus-immediate form (`VSHUFPD`, `VPERM2I128`,
|
||||
`VINSERTI128`),
|
||||
- the lane extract (`VEXTRACTI128`, `VEXTRACTF128` — the YMM source occupies
|
||||
the ModRM.reg field, the XMM/memory destination the r/m field),
|
||||
- the direction-sensitive moves (`VMOVDQU`, `VMOVUPD`, `VMOVD`, `VMOVQ`,
|
||||
`VMOVSD` — each direction picks its own opcode and VEX.W; a vector→vector
|
||||
move uses the store-form layout, matching the Go assembler),
|
||||
- the no-operand `VZEROUPPER`, and `VPERMD` in the NDS form,
|
||||
- the floating-point and FMA set (`VADDPD`, `VMULPD`, `VXORPD`,
|
||||
`VUNPCKHPD`, the scalar `VADDSD`/`VMULSD`, `VCVTDQ2PD`, `VFMADD231PD`).
|
||||
With the scalar set and the earlier NDS / reg-rm / immediate-shift forms,
|
||||
the encoder now covers every integer, shuffle and FP instruction the
|
||||
go-flac AVX2 kernels use.
|
||||
- `asm`: `CMP` accepts the immediate in the second operand position
|
||||
(`CMPL CX, $31`) — the spelling the Go assembler accepts — encoding it
|
||||
identically to the immediate-first form.
|
||||
|
||||
### Fixed
|
||||
|
||||
- `asm`: an unused VEX.vvvv field is now stored as `1111` (v̄vvv = 1111), as
|
||||
the hardware requires — the previous value (`0000`) made the two-operand
|
||||
reg/rm forms (VPMOVSXWD, VPBROADCASTD, VMOVMSKPS, …) raise #UD on real CPUs
|
||||
and differ from the Go assembler's bytes. The round-trip decoder ignores
|
||||
the field on these instructions, which is why the byte-for-byte Go
|
||||
comparison (added this release) is now part of the test suite.
|
||||
|
||||
## [0.1.0] — 2026-07-06
|
||||
|
||||
Initial release — the Phase 1 foundation.
|
||||
|
||||
### Added
|
||||
|
||||
- `token`, `lexer`, `ast`, `parser`: a hand-written, error-tolerant front end
|
||||
for Plan 9 assembly. The lexer splices C-preprocessor line continuations
|
||||
(`\` before a newline) so multi-line `#define` macros parse as one opaque
|
||||
directive. Validated against the production AVX2/AVX-512 kernels in
|
||||
`go-libraries/go-flac` and the Go runtime's `src/runtime/*.s` for all four
|
||||
architectures, with zero parse errors.
|
||||
- `arch`: register files and **complete** instruction tables for amd64,
|
||||
arm64, riscv64 and loong64, with the middle-dot symbol separator and static
|
||||
(`<>`) symbols. Instruction names are generated from the Go toolchain's own
|
||||
assembler source (`just gen`) — the `anames` opcode lists plus the common
|
||||
opcodes and the per-architecture front-end aliases (arm64 `B`/`BL`, the
|
||||
`.P`/`.W` addressing suffixes, loong64 `JAL`, the x86 conditional-jump
|
||||
spellings) — so every mnemonic the real assembler accepts is recognised.
|
||||
- `lint`: conservative rules — `unknown-instruction`, `operand-count`,
|
||||
`undefined-label`, `duplicate-label`, `missing-ret`,
|
||||
`missing-textflag-include`, `abi-argsize` and `unreachable-code`. Macro
|
||||
invocations are recognised (in-file `#define` names and underscore
|
||||
identifiers) and the label/RET heuristics are suppressed in macro-using
|
||||
files. `abi-argsize` parses the `// func` signature with the Go parser and
|
||||
checks the declared TEXT argument size against Go's ABI0 layout;
|
||||
`unreachable-code` flags dead code after `RET`, suppressed where reachability
|
||||
is undecidable (PC-relative jumps, register-indirect branches, `#ifdef`).
|
||||
Register liveness is computed by dataflow over the control-flow graph (basic
|
||||
blocks, def/use, iterative backward iteration) and drives `register-clobber`,
|
||||
an audit that flags a callee-saved register written but never saved/restored.
|
||||
`funcdata-pcdata` validates the structure of `FUNCDATA`/`PCDATA` directives.
|
||||
Zero error-severity diagnostics across the 90-file Go runtime corpus and the
|
||||
production go-flac kernels (the `register-clobber` audit additionally reports
|
||||
the go-flac kernels' unsaved callee-saved register use for review).
|
||||
- `format`: an idempotent canonical formatter (operand spacing and per-function
|
||||
mnemonic alignment) that preserves comments and round-trips through the
|
||||
parser.
|
||||
- `lsp`: a Language Server Protocol server over stdio providing completion,
|
||||
hover documentation, document symbols, publish-diagnostics and semantic-token
|
||||
highlighting.
|
||||
- `asm`: a standalone amd64 (x86-64) assembler — an instruction encoder (REX/
|
||||
ModR-M/SIB/displacement/immediate plus the scalar instruction set, and VEX/
|
||||
AVX2 SIMD across three operand forms — NDS, reg/rm and immediate-shift —
|
||||
covering the bulk of the integer SIMD set) validated by round-trip decoding
|
||||
against `golang.org/x/arch`, and an assembler that drives the parser's AST
|
||||
into the encoder with local-label resolution and `FP`/`SP` frame mapping
|
||||
(plus Go prologue/epilogue generation), producing output byte-identical to the
|
||||
Go assembler for the supported operand forms.
|
||||
- `cmd/gasm`: the `gasm` binary with `tokens`, `parse`, `fmt`, `lint`, `asm`
|
||||
and `lsp` subcommands.
|
||||
- `_gen`: the generator that rebuilds the architecture instruction tables from
|
||||
the Go toolchain source (`just gen`).
|
||||
@@ -0,0 +1,94 @@
|
||||
# Contributing to gasm-devkit
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Go 1.26 or later (`toolchain go1.26.5`)
|
||||
- `just` command runner
|
||||
- A Linux, FreeBSD, or macOS host on amd64 or arm64
|
||||
|
||||
## Development Setup
|
||||
|
||||
```sh
|
||||
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
|
||||
cd gasm-devkit
|
||||
just install # download module dependencies
|
||||
just build # go vet + gofmt check
|
||||
just test # full test suite with race detector
|
||||
```
|
||||
|
||||
## Commands
|
||||
|
||||
Every just recipe:
|
||||
|
||||
| Recipe | What it does |
|
||||
|--------|-------------|
|
||||
| `just` | List all recipes |
|
||||
| `just install` | `go mod download` |
|
||||
| `just build` | `go vet ./...` + `gofmt -l .` check — zero errors required |
|
||||
| `just test` | `go test -race -count=1 -coverprofile=coverage.out ./...` + 80 % coverage gate |
|
||||
| `just fmt` | `gofmt -w .` |
|
||||
| `just run -- lint file.s` | Run the CLI with `go run` (args after `--`) |
|
||||
| `just install-bin` | Install `gasm` into `$GOBIN` with the release version stamped |
|
||||
| `just gen` | Regenerate `arch/*_gen.go` instruction tables from the Go toolchain |
|
||||
| `just uninstall` | Remove build artefacts (`coverage.out`, `gasm`, `*.test`) |
|
||||
|
||||
## Running a Single Test
|
||||
|
||||
```sh
|
||||
go test -run TestVexGroundTruth ./asm/
|
||||
go test -run TestDifferentialLZ4Fuzz ./verify/
|
||||
```
|
||||
|
||||
## Testing the Debugger
|
||||
|
||||
The interactive debugger (`gasm debug`) requires a compiled binary —
|
||||
`go run` does not work for the child process. Install first:
|
||||
|
||||
```sh
|
||||
just install-bin
|
||||
gasm debug --func add testdata/verify/basic_amd64.s
|
||||
```
|
||||
|
||||
## Code Style
|
||||
|
||||
See [AGENTS.md](AGENTS.md) for the full style guide. Key points:
|
||||
|
||||
- `gofmt` — zero diff.
|
||||
- `go vet` — zero warnings.
|
||||
- Standard library only in production code; `golang.org/x/arch` in tests.
|
||||
- No cgo, no C, no JavaScript.
|
||||
- Hand-written Plan 9 assembly; tables generated only via `_gen/gen.go`.
|
||||
|
||||
## Branches and Releases
|
||||
|
||||
- `development` is the working branch.
|
||||
- `main` is release-only: `git merge --ff-only development`, then `git tag vX.Y.Z`.
|
||||
- Conventional Commits: `feat(asm): add EVEX gather and scatter`.
|
||||
- Every commit ends with `Assisted-by: <model-name>`.
|
||||
|
||||
## CI
|
||||
|
||||
There is no CI pipeline in this repository. The Definition of Done
|
||||
(`just build` + `just test` + `just fmt`) is enforced locally.
|
||||
|
||||
## AI-Assisted Contributions
|
||||
|
||||
AI agents may assist with code, documentation, tests, and review. All
|
||||
AI-assisted changes must:
|
||||
|
||||
- Include the trailer `Assisted-by: <model-name>` in the commit message
|
||||
(e.g. `Assisted-by: DeepSeek V4 Pro`).
|
||||
- Follow the [AGENTS.md](AGENTS.md) rules.
|
||||
- Pass the Definition of Done before committing.
|
||||
|
||||
Attribute agent authorship in issues and pull requests on one trailing
|
||||
line:
|
||||
|
||||
```
|
||||
_Assisted-by: DeepSeek V4 Pro_
|
||||
```
|
||||
|
||||
## Questions
|
||||
|
||||
Open an issue at
|
||||
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit/issues).
|
||||
@@ -0,0 +1,354 @@
|
||||
# gasm-devkit
|
||||
|
||||
Developer tooling for **GAsm** — Go's built-in Plan 9 assembler.
|
||||
|
||||
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
|
||||
|
||||
Go ships an assembler but no tooling for it. There is no syntax highlighting,
|
||||
no autocomplete, no linter, no static analyser, no formatter, no standalone
|
||||
assembler and no debugger for `.s` files. Developers write assembly blind,
|
||||
validate it by benchmark, and debug it by print statement.
|
||||
|
||||
gasm-devkit is the missing toolkit. It is a single, self-contained binary —
|
||||
`gasm` — that brings proper developer tooling to Plan 9 assembly:
|
||||
|
||||
```
|
||||
gasm tokens dump the lexical token stream
|
||||
gasm parse parse and report syntax errors
|
||||
gasm fmt canonicalise formatting (gofmt for assembly)
|
||||
gasm lint static checks
|
||||
gasm lsp language server (completion, hover, symbols, diagnostics, highlighting)
|
||||
gasm asm standalone assembler (Phase 2)
|
||||
gasm verify dynamic analysis & verification (Phase 3)
|
||||
gasm debug source-level debugger (Phase 4)
|
||||
```
|
||||
|
||||
> **Status: Phase 4 — done.** Phase 1 (the language foundation,
|
||||
> linter, formatter and language server) shipped in v0.1.0; Phase 2 (the
|
||||
> standalone assembler — the full amd64 instruction set plus ELF, Mach-O
|
||||
> and GOOBJ object emission) in v0.12.0; Phase 3 (dynamic analysis —
|
||||
> JIT execution, differential testing, ABI checks and coverage profiling)
|
||||
> in v0.25.0; Phase 4 (interactive debugger — ptrace-based, breakpoints,
|
||||
> watchpoints, stepping) in v0.27.0; RISC-V encoder (RV64IMAFDC + RVC,
|
||||
> ELF emission, ground-truth) in v0.28.0. See [Roadmap](#roadmap).
|
||||
|
||||
## Architecture support
|
||||
|
||||
gasm-devkit targets every architecture Go's assembler speaks. The instruction
|
||||
tables are **generated from the Go toolchain's own assembler source**
|
||||
(`cmd/internal/obj/<arch>`), so gasm-devkit recognises *every* mnemonic the
|
||||
real assembler accepts — not a hand-maintained subset that drifts and rots.
|
||||
|
||||
| Architecture | GOARCH | File suffix | Instructions recognised |
|
||||
|--------------|-------------|----------------|------------------------------------|
|
||||
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
|
||||
| ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes |
|
||||
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes |
|
||||
| LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes |
|
||||
|
||||
"Common opcodes" are the instructions shared by every architecture (`RET`,
|
||||
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, …). AMD64 additionally
|
||||
carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
|
||||
…) that the assembler accepts as aliases. Regenerating the tables is one
|
||||
command — `just gen` — and requires only a Go installation; the committed
|
||||
output has no runtime dependency on the toolchain.
|
||||
|
||||
The target *architectures* above are what the toolkit analyses. The toolkit
|
||||
itself is portable Go and builds on Linux, FreeBSD and macOS, on amd64 and
|
||||
arm64 hosts.
|
||||
|
||||
## Roadmap
|
||||
|
||||
The work is delivered in four phases. Each phase is completed and hardened
|
||||
before the next begins. The ordering follows a dependency chain: understand
|
||||
the code statically (Phase 1), make it runnable (Phase 2), then run it and
|
||||
observe or control it (Phases 3–4).
|
||||
|
||||
### Phase 1 — language foundation, editor tooling and static analysis · *done*
|
||||
|
||||
Everything needed to read, understand, check, format and highlight GAsm —
|
||||
without executing it.
|
||||
|
||||
| Capability | Status |
|
||||
|------------|--------|
|
||||
| Lexer — permissive, position-aware scanner for all four architectures | done |
|
||||
| Parser — line-oriented, error-tolerant, full AST with source positions | done |
|
||||
| Instruction + register tables for amd64, arm64, riscv64, loong64 (generated, complete) | done |
|
||||
| Linter — `unknown-instruction`, `operand-count`, `undefined-label`, `duplicate-label`, `missing-ret`, `missing-textflag-include`, `abi-argsize`, `unreachable-code`, `register-clobber`, `funcdata-pcdata` | done |
|
||||
| Formatter — idempotent, comment-preserving, per-function alignment; a `RET` terminates the body for indentation, so the next function's doc comment stays at column 0; exactly one blank line before every block (label, `TEXT`, `GLOBL`) and runs of blanks collapsed; directory / no-argument mode reformats every `.s` in place, `go fmt`-style | done |
|
||||
| Language server — completion, hover, document symbols, diagnostics, semantic-token highlighting | done |
|
||||
| CLI — `gasm tokens / parse / fmt / lint / lsp` | done |
|
||||
| Real-world validation against production AVX2 / AVX-512 kernels | done |
|
||||
| Lint hardening — zero false positives across the Go runtime corpus (90 files, all four architectures): macro-invocation handling, branch aliases (`B`/`BL`/`JAL`), addressing suffixes (`.P`/`.W`), terminal `UNDEF` | done |
|
||||
| Static analysis — `abi-argsize` (argument/result area computed from the `// func` signature under Go's ABI0 layout and checked against the TEXT declaration) and `unreachable-code` (dead code after `RET`, suppressed where reachability is undecidable: PC-relative jumps, register-indirect branches, `#ifdef`) | done |
|
||||
| Static analysis — register liveness (CFG construction + per-instruction def/use + iterative backward dataflow) driving `register-clobber`, calibrated to the **Go ABI** (not System V): flags writes to the registers Go fixes across calls — the frame pointer and the goroutine pointer (`R14` on amd64, `R28`/`R29` on arm64, `X27` on riscv64, `R22` on loong64, plus the OS-reserved `R18` on arm64) — that are never saved/restored; the goroutine pointer is reported only when the function can reach the runtime (not `NOSPLIT`, or makes calls), matching how the runtime's own assembly uses it. `funcdata-pcdata` structural validation of `FUNCDATA`/`PCDATA` operands and indices | done |
|
||||
|
||||
> **Limitation — macros.** gasm-devkit reads `.s` source as written; it does
|
||||
> **not** run the C preprocessor, so `#define` macros are not expanded. Files
|
||||
> that use macros (the runtime's `asm_*.s`, `race_*.s`, `sys_*.s`, …) parse
|
||||
> cleanly, and macro *invocations* are recognised and never flagged, but the
|
||||
> `undefined-label` and `missing-ret` heuristics are suppressed in macro-using
|
||||
> files because labels a macro defines are invisible without expansion. Full
|
||||
> macro expansion is future work (it pairs naturally with the Phase 2
|
||||
> assembler). Hand-written, macro-free kernels — such as everything in
|
||||
> `go-libraries` — are analysed in full.
|
||||
|
||||
### Phase 2 — standalone assembler · *done*
|
||||
|
||||
Assembly without the Go toolchain in the loop.
|
||||
|
||||
- **`gasm asm`:** a standalone assembler that turns a `.s` file into machine
|
||||
code directly — pure Go, no `go build`, no external toolchain. Useful for
|
||||
fast iteration, for environments without a full Go installation, and as the
|
||||
execution substrate that Phases 3 and 4 build on.
|
||||
|
||||
Done so far:
|
||||
|
||||
- An amd64 (x86-64) **instruction encoder** — REX/ModR-M/SIB/displacement/
|
||||
immediate machinery and the scalar instruction set (MOV, the ALU group, TEST,
|
||||
LEA, INC/DEC/NEG/NOT, shifts, IMUL and IMUL3, PUSH/POP, JMP/CALL/Jcc,
|
||||
CMOVcc, SETcc, LZCNT/TZCNT, the sign/zero-extending moves — MOVBLZX and
|
||||
friends, MOVLQSX — and CVTSL2SD/CVTSQ2SD), validated by round-tripping
|
||||
every encoding through `golang.org/x/arch`'s decoder and byte-for-byte
|
||||
against the Go assembler.
|
||||
- An **assembler** that drives the parser's AST into the encoder with local-
|
||||
label resolution — jumps start in the short (rel8) form and expand to rel32
|
||||
when the displacement does not fit, and jump-to-jump chains are folded the
|
||||
way the Go toolchain folds them — so `gasm asm <file>` emits machine code
|
||||
for each `TEXT` function.
|
||||
- **File-level assembly with static data** — `GLOBL`/`DATA` symbols are laid
|
||||
out in a data section behind the code and references to them (`mask<>(SB)`)
|
||||
are encoded RIP-relative with the displacement resolved within the image,
|
||||
so the output is self-consistent and position-independent. References to
|
||||
symbols no `GLOBL` in the file defines are recorded as relocations and
|
||||
carried into the object-file output.
|
||||
- **GOOBJ emission** — `gasm asm --format goobj -p <pkgpath>` writes the Go
|
||||
toolchain's own object format (the one `cmd/link` consumes directly), so
|
||||
gasm-assembled kernels drop into a `go build` without the Go assembler:
|
||||
the functions as non-package symbols, `GLOBL` data, one `FuncInfo` per
|
||||
function and the pc-value tables (`pcsp` with the real prologue/epilogue
|
||||
stack deltas, `pcfile`, `pcline`, `pcinline`). Verified end-to-end by
|
||||
swapping a gasm-emitted object into a `go build` in place of the
|
||||
toolchain's, linking and running — bit-identical behaviour.
|
||||
- **Object-file emission** — `gasm asm --format elf` / `--format macho`
|
||||
writes a relocatable object (a `.text` and a `.data` section, a symbol
|
||||
table — file-local `<>` symbols local, the rest global — and one
|
||||
`R_X86_64_PC32` / `X86_64_RELOC_SIGNED` relocation per static-symbol
|
||||
reference) that links with the system toolchain: external references
|
||||
resolve against undefined symbols, file-local ones against the data
|
||||
section. Verified end-to-end by linking a gasm-emitted object with a C
|
||||
driver and running it.
|
||||
- **`FP`/`SP` frame mapping** — the pseudo-registers are translated onto the
|
||||
hardware stack pointer (`x+N(FP)` → `(N+8)(SP)` for a zero frame, `(N+frame+
|
||||
16)(SP)` with a frame pointer; locals via `x-N(SP)`), and the Go-style
|
||||
prologue/epilogue is generated for functions with a frame. The output is
|
||||
**byte-identical to the Go assembler** for these cases (verified against
|
||||
`go tool objdump`).
|
||||
- **SIMD (VEX / AVX2)** — the VEX prefix machinery (2-byte C5 and 3-byte C4)
|
||||
with XMM/YMM vector registers, validated by round-trip decoding **and**
|
||||
byte-for-byte against the Go assembler's machine code, across eight operand
|
||||
forms: the three-operand NDS form (VPADDD/Q, VPSUBD/Q, VPXOR, VPOR, VPAND/N,
|
||||
VPCMPEQD, VPCMPGTQ, VPUNPCK*, VPMULLD, VPMULDQ, VPSHUFB, VPACKSSDW,
|
||||
VPERMD), the two-operand reg/rm form (VPMOVSXWD/DQ, VPMOVZXDQ,
|
||||
VPBROADCASTD/Q, VPMOVMSKB, VMOVMSKPS, VCVTDQ2PD), the immediate-shift and
|
||||
variable-count shifts (VPSLLD/Q, VPSRAD, VPSRLD/Q with an immediate or an
|
||||
XMM/memory count), the immediate shuffle (VPSHUFD, VPERMQ), the
|
||||
three-operand-plus-immediate form (VSHUFPD, VPERM2I128, VINSERTI128), the
|
||||
lane extract (VEXTRACTI128, VEXTRACTF128), the direction-sensitive moves
|
||||
(VMOVDQU, VMOVUPD, VMOVD, VMOVQ, VMOVSD), the no-operand VZEROUPPER, and
|
||||
the floating-point set: the packed double arithmetic
|
||||
(VADDPD/VSUBPD/VMULPD/VDIVPD/VMINPD/VMAXPD), the unpacks
|
||||
(VUNPCKHPD/VUNPCKLPD), the scalar SD and SS operations, VMOVDDUP, the
|
||||
width-changing conversions (VCVTDQ2PS, VCVTPS2PD, VCVTDQ2PD and the
|
||||
VCVTPD2DQX/Y / VCVTTPD2DQX/Y spellings, whose VEX.L follows the wider
|
||||
source) and VFMADD231PD.
|
||||
- **SIMD (EVEX / AVX-512)** — the four-byte EVEX prefix with the 5-bit
|
||||
register fields (Z0–Z31, X/Y 16–31), opmask registers (K0–K7 as operands
|
||||
and mask destinations, KMOVW, KTESTW) and the compressed disp8×N
|
||||
displacement, covering every AVX-512 instruction the go-flac kernels use:
|
||||
VPXORD/Q, VPADDD, VPSUBD/Q, VPUNPCK*DQ, VPMULLD/Q, VPERMD, VPSLLD/VPSRAD/
|
||||
VPSRAQ, VALIGND, VPCMPEQD (with a K destination), VMOVDQU32, VMOVUPD,
|
||||
VCVTQQ2PD, VPMOVSXDQ, the narrowing stores VPMOVDW/VPMOVQD, the lane
|
||||
extracts VEXTRACTI64X4/VEXTRACTF64X4, VFMADD231PD, VADDPD, VMULPD,
|
||||
VMOVDQU64 and the broadcasts VPBROADCASTD/Q from a GPR or memory, plus the
|
||||
wider AVX-512 F/BW integer set (VPADDB/W, VPSUBB/W, VPANDD/Q/ND/NQ, VPMULLW,
|
||||
VPMIN*/VPMAX* for B/W/D/Q elements, signed and unsigned, VPAVGB/W, the variable
|
||||
shifts VPSLLV*/VPSRLV*/VPSRAV*, VMOVDQU8/16), the common floating-point
|
||||
and conversion set (the packed double and single arithmetic
|
||||
VADD/VSUB/VMUL/VDIV/VMIN/VMAX PD and PS, the scalar SD/SS operations —
|
||||
whose EVEX forms exist for masked and zeroing use — the VUNPCK{L,H}PD
|
||||
unpacks, VMOVDDUP, VMOVSLDUP/VMOVSHDUP and the VCVT* conversions), and
|
||||
the wider AVX-512 set: ternary logic (VPTERNLOGD/Q), lane shuffles,
|
||||
inserts and extracts (VSHUF{F,I}{32,64}X{2,4}, the VINSERT*/VEXTRACT*
|
||||
{F,I}{32,64}X{2,4,8} family, VPALIGNR), compares with an opmask
|
||||
destination (VCMPPD/PS/SD/SS), the permutes (VPERMB/W, VPERMI2/T2
|
||||
D/Q/PD), the wider integer families (VPMADDWD/UBSW, VPMULHUW, VPACK*,
|
||||
VPABS*, the VPROL*/VPROR* rotates and the word shifts), expand/compress
|
||||
(VEXPAND*/VCOMPRESS*, VPEXPAND*/VPCOMPRESS*), the broadcasts
|
||||
(VPBROADCASTB/W, VBROADCASTSS/SD), the opmask instructions (KAND/KOR/
|
||||
KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST, KMOVQ), the aligned moves
|
||||
(VMOVAPS/APD, VMOVDQA32/64, VMOVSS) and the remaining extending and
|
||||
narrowing moves, the floating-point helper and conversion tail
|
||||
(VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*, VSCALEF*, VRNDSCALE*,
|
||||
VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with a K destination, and the
|
||||
VCVT* conversions VCVTQQ2PS, VCVTPD2QQ/UQQ, VCVTPS2QQ, VCVTUDQ2PD/PS,
|
||||
VCVTPH2PS, VCVTPS2PH), and gather/scatter with VSIB addressing
|
||||
(VGATHER*/VPGATHER* in both the VEX mask-register spelling and the EVEX
|
||||
K-mask spelling — where the L'L field follows the VSIB index — plus
|
||||
VSCATTER*/VPSCATTER*). The EVEX mnemonic suffixes the Go assembler
|
||||
accepts are honoured: rounding modes (.RN_SAE, .RD_SAE, .RU_SAE,
|
||||
.RZ_SAE), suppress-all-exceptions (.SAE) and memory broadcast (.BCST,
|
||||
with the element-sized disp8×N), each combinable with the .Z zeroing
|
||||
suffix. Masking is supported the way
|
||||
Go writes it — an explicit K1–K7 operand placed among the operands, and a
|
||||
`.Z` mnemonic suffix for zeroing.
|
||||
- **Legacy SSE moves** — `MOVOU`/`MOVO` (the Plan 9 names for MOVDQU/MOVDQA),
|
||||
`MOVUPS`/`MOVAPS`/`MOVUPD`/`MOVAPD` and the scalar `MOVSD`/`MOVSS`.
|
||||
- **Both go-flac kernels — all 17 AVX2 and all 10 AVX-512 functions —
|
||||
assemble byte-identically to the Go toolchain's machine code**; the only
|
||||
differing bytes are the displacements of the static-constant loads, which
|
||||
the Go linker fills at link time and gasm resolves within its own image
|
||||
(verified to reach the right constant bytes).
|
||||
|
||||
Remaining for Phase 2:
|
||||
|
||||
- External (cross-package) symbol references in the GOOBJ output —
|
||||
**deferred** with a recorded decision and three options; see
|
||||
[`docs/DEFERRED.md`](docs/DEFERRED.md). Single-package objects (no
|
||||
cross-package references) work today, which covers the production
|
||||
kernels. With that item deferred, the amd64 instruction set — scalar,
|
||||
VEX/AVX2 and the full EVEX/AVX-512 set including GPR-interchanging
|
||||
conversions — is complete.
|
||||
|
||||
### Phase 3 — dynamic analysis · *done*
|
||||
|
||||
Run the code and check what static analysis cannot. The oracle is the
|
||||
portable Go implementation every kernel is derived from.
|
||||
|
||||
- **`gasm verify`:**
|
||||
- **JIT execution substrate** — *done.* Assemble the kernel, map it into
|
||||
executable memory (`syscall.Mmap`, W^X) and call it through an ABI0
|
||||
trampoline; pure Go, no cgo, no external toolchain. Both go-lz4 kernels
|
||||
(AVX2, 845 bytes total) JIT-load and execute correctly.
|
||||
- **Differential testing** — *done.* The JIT-assembled kernel is fuzzed
|
||||
with random valid LZ4 blocks and hostile garbage, comparing the result
|
||||
**bit-for-bit** against a portable Go reference; the automated form of
|
||||
the project's bit-identical contract.
|
||||
- **Runtime ABI checks** — *done.* The ABI-checking trampoline sets
|
||||
sentinels in BP and R14, verifies they survive the call, and fills a
|
||||
128-byte red-zone canary below SP; both go-lz4 kernels pass clean.
|
||||
- **Coverage / basic-block profiling** — *done.* Static block enumeration
|
||||
from the assembler's label map (27 blocks in `decodeBlockAVX2`) plus
|
||||
multi-input path-diversity measurement: how many observationally distinct
|
||||
execution paths a test corpus exercises.
|
||||
|
||||
### Phase 4 — debugger · *in progress*
|
||||
|
||||
- **`gasm debug`:** single-step a GAsm function, inspect registers, set
|
||||
breakpoints on labels, and hex-dump memory — the interactive counterpart
|
||||
to Phase 3's execution substrate.
|
||||
- **MVP** — *done.* ptrace-based debuggee subprocess (PTRACE_TRACEME +
|
||||
LockOSThread), entry breakpoint (auto-run to function start),
|
||||
single-step, register inspection, label resolution, breakpoint
|
||||
management via `/proc/pid/mem`, and an interactive REPL.
|
||||
- **Remaining:** disassembly at PC (x86asm decode), memory-write support,
|
||||
watchpoints, source-line mapping, and multi-platform support
|
||||
(FreeBSD/macOS ptrace variants).
|
||||
|
||||
### Phase 5 — the other architectures
|
||||
|
||||
- **Encoding for arm64, riscv64 and loong64.** The lexer, parser, linter
|
||||
and formatter already cover all four architectures; the assembler today
|
||||
encodes amd64 only. Phase 5 brings the same encode-and-verify treatment
|
||||
(instruction tables already generated from the toolchain, every encoding
|
||||
checked byte for byte against `go tool asm`) to the remaining three.
|
||||
|
||||
## Principles
|
||||
|
||||
- **Pure Go and GAsm only.** No C, no cgo, no external toolchains, no native
|
||||
binaries, no JavaScript runtimes. The parser is hand-written; there is no
|
||||
parser generator.
|
||||
- **Self-contained.** The toolkit's production code depends only on the
|
||||
standard library; one binary, no runtime data files. The single module
|
||||
dependency, `golang.org/x/arch`, is used **only in tests** to validate the
|
||||
instruction encoder by round-trip decoding — it is never linked into the
|
||||
`gasm` binary.
|
||||
- **Portable.** Builds and runs on Linux, FreeBSD and macOS; amd64 and arm64
|
||||
hosts. Latest stable Go only.
|
||||
- **No vendor lock-in.** The integration surface is the Language Server
|
||||
Protocol and a command-line interface — both open standards. No cloud
|
||||
service, no proprietary API, no dependence on any one editor's internals.
|
||||
- **Complete and verifiable.** Instruction coverage is generated from the
|
||||
assembler's own source and regenerated on demand, so it cannot silently fall
|
||||
behind the toolchain.
|
||||
|
||||
## Components
|
||||
|
||||
| Package | Purpose |
|
||||
|---------|---------|
|
||||
| `token` | Lexical token kinds and source positions. |
|
||||
| `lexer` | Hand-written scanner for Plan 9 assembly. |
|
||||
| `ast` | The abstract syntax tree. |
|
||||
| `parser` | Line-oriented, error-tolerant parser producing the AST. |
|
||||
| `arch` | amd64, arm64, riscv64 and loong64 register files and instruction tables. |
|
||||
| `lint` | Conservative static checks. |
|
||||
| `format` | A canonical formatter — `gofmt` for assembly. |
|
||||
| `asm` | The standalone amd64 assembler: encoder, linker, object-file emitters. |
|
||||
| `verify` | JIT execution substrate for dynamic analysis (Phase 3). |
|
||||
| `debug` | Interactive ptrace debugger for amd64 (Phase 4). |
|
||||
| `lsp` | Language Server Protocol server. |
|
||||
| `cmd/gasm` | The `gasm` binary tying it all together. |
|
||||
| `_gen` | The generator that rebuilds the instruction tables from the Go toolchain. |
|
||||
|
||||
See [`docs/ARCHITECTURE.md`](docs/ARCHITECTURE.md) for the design rationale and
|
||||
data flow, [`docs/ZED.md`](docs/ZED.md) for the editor-integration story, and
|
||||
[`docs/DEFERRED.md`](docs/DEFERRED.md) for design decisions deliberately
|
||||
postponed (with the analysis needed to pick them up again).
|
||||
|
||||
## Quick start
|
||||
|
||||
```sh
|
||||
just install # download dependencies (there are none)
|
||||
just build # go vet + gofmt check — zero errors, zero warnings
|
||||
just test # full suite, race detector, 80 % coverage gate
|
||||
just fmt # gofmt the tree
|
||||
just gen # regenerate the instruction tables from the Go toolchain
|
||||
```
|
||||
|
||||
Install the binary and use it:
|
||||
|
||||
```sh
|
||||
just install-bin # installs gasm into $GOBIN
|
||||
|
||||
gasm --help # overview of commands and flags
|
||||
gasm tokens kernel_amd64.s # dump the token stream
|
||||
gasm parse kernel_amd64.s # parse, report syntax errors
|
||||
gasm fmt -w kernel_amd64.s # canonicalise in place
|
||||
gasm fmt # reformat every .s below here, like go fmt
|
||||
gasm lint *.s # static checks
|
||||
gasm asm --format elf -o k.o k.s # assemble to a linkable ELF object
|
||||
gasm verify kernel_amd64.s # JIT-load and report functions
|
||||
gasm verify --ground-truth k.s # byte-for-byte vs go tool asm
|
||||
gasm debug --func name k.s # interactive debugger
|
||||
```
|
||||
|
||||
See [CONTRIBUTING.md](CONTRIBUTING.md) for the full development workflow,
|
||||
[docs/cli.md](docs/cli.md) for the command reference, and
|
||||
[docs/development.md](docs/development.md) for setup and recipes.
|
||||
|
||||
## Editor integration
|
||||
|
||||
`gasm lsp` speaks the Language Server Protocol over standard input/output, so
|
||||
any LSP-capable editor can use it — point your editor's LSP client at the
|
||||
binary and associate it with `.s` files. Syntax highlighting is delivered as
|
||||
**LSP semantic tokens**, so no editor-specific grammar is required. The server
|
||||
infers the target architecture from the file-name suffix
|
||||
(`_amd64.s` / `_arm64.s` / `_riscv64.s` / `_loong64.s`).
|
||||
|
||||
Zed users should read [`docs/ZED.md`](docs/ZED.md): Zed's native highlighting
|
||||
engine (Tree-sitter, C/WASM) cannot be fed from pure Go, so the pure-Go path
|
||||
into Zed is the language server and its semantic tokens.
|
||||
|
||||
## Licence
|
||||
|
||||
BSD-3-Clause — the same licence as Go itself. See [`LICENSE`](LICENSE).
|
||||
@@ -156,6 +156,16 @@ func (t *Table) Lookup(mnemonic string) (Instr, bool) {
|
||||
}
|
||||
}
|
||||
}
|
||||
// amd64 EVEX instructions take a .Z zeroing suffix (masking is written as
|
||||
// an explicit K operand rather than a suffix); strip it so the base
|
||||
// instruction is still recognised.
|
||||
if t.Arch == AMD64 {
|
||||
if base, ok := strings.CutSuffix(key, ".Z"); ok {
|
||||
if in, found := t.instrs[base]; found {
|
||||
return in, true
|
||||
}
|
||||
}
|
||||
}
|
||||
return Instr{}, false
|
||||
}
|
||||
|
||||
|
||||
+252
-49
@@ -13,55 +13,207 @@ import (
|
||||
// Assemble encodes the body of a TEXT function into x86-64 machine code,
|
||||
// resolving local labels to relative jump offsets and translating the FP/SP
|
||||
// pseudo-registers onto the hardware stack pointer (matching the Go
|
||||
// assembler's default frame-pointer behaviour). Jumps always use the 32-bit
|
||||
// relative form so instruction sizes are fixed and offsets resolve in a single
|
||||
// layout pass.
|
||||
// assembler's default frame-pointer behaviour). Jumps start in the short
|
||||
// (rel8) form and expand to rel32 when the settled displacement does not fit;
|
||||
// sizes only grow, so the layout reaches a fixed point in a few passes. CALL
|
||||
// has no short form and is always rel32.
|
||||
//
|
||||
// Supported operands: registers, memory (real base register), immediates,
|
||||
// FP/SP frame-relative operands, and local-label jumps. SB (global symbol)
|
||||
// operands require relocations and are not yet supported; the SIMD (VEX/AVX2)
|
||||
// integer and shuffle/extract/permute/move set is in.
|
||||
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
|
||||
fi := computeFrame(t)
|
||||
code, _, labels, _, _, err := assemble(t, nil)
|
||||
return code, labels, err
|
||||
}
|
||||
|
||||
// Pass 1: lay out instructions (including prologue/epilogue) to fix label
|
||||
// offsets.
|
||||
offsets := map[string]int{}
|
||||
// linkInfo carries file-level symbol context into a single-function assembly:
|
||||
// the set of static symbols a GLOBL in the same file defines. A nil link
|
||||
// rejects SB operands outright (single-function assembly cannot resolve
|
||||
// them). When allowExternal is set, a reference to a symbol no GLOBL in the
|
||||
// file defines is recorded as an external relocation instead of failing —
|
||||
// the object-file emitters resolve it at link time.
|
||||
type linkInfo struct {
|
||||
symbols map[string]bool
|
||||
allowExternal bool
|
||||
}
|
||||
|
||||
// sbPatch is a function-relative static-symbol relocation: the disp32 field
|
||||
// at off must become the symbol's address minus after, where after is the
|
||||
// function-relative address just past the instruction.
|
||||
type sbPatch struct {
|
||||
off int
|
||||
after int
|
||||
name string
|
||||
addend int64
|
||||
}
|
||||
|
||||
// spadjStep is one stack-adjustment boundary within a function: Value is the
|
||||
// SP delta from the entry state (just below the return address) in effect
|
||||
// from PC (function-relative) until the next step. The steps feed the
|
||||
// pcsp table of the object-file emitters.
|
||||
type spadjStep struct {
|
||||
pc int
|
||||
value int
|
||||
}
|
||||
|
||||
// assemble encodes a TEXT body, returning the machine code, the static-symbol
|
||||
// patch sites (for the file-level layout to resolve), the label table and the
|
||||
// stack-adjustment boundaries.
|
||||
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
|
||||
fi := computeFrame(t)
|
||||
chain := jumpChain(t)
|
||||
resolve := func(name string) string {
|
||||
if r, ok := chain[name]; ok {
|
||||
return r
|
||||
}
|
||||
return name
|
||||
}
|
||||
|
||||
// Layout: iterate jump sizes to a fixed point.
|
||||
long := make([]bool, len(t.Body))
|
||||
sizes := make([]int, len(t.Body))
|
||||
pos := len(fi.prologue)
|
||||
for i, stmt := range t.Body {
|
||||
switch s := stmt.(type) {
|
||||
case *ast.Label:
|
||||
offsets[s.Name.Text] = pos
|
||||
case *ast.Instr:
|
||||
sz, err := instrSize(s, fi)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
offsets := map[string]int{}
|
||||
pcs := make([]int, len(t.Body))
|
||||
for {
|
||||
pos := len(fi.prologue)
|
||||
for i, stmt := range t.Body {
|
||||
switch s := stmt.(type) {
|
||||
case *ast.Label:
|
||||
offsets[s.Name.Text] = pos
|
||||
case *ast.Instr:
|
||||
sz, err := instrSize(s, fi, long[i], link)
|
||||
if err != nil {
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
}
|
||||
sizes[i] = sz
|
||||
pcs[i] = pos
|
||||
pos += sz
|
||||
}
|
||||
sizes[i] = sz
|
||||
pos += sz
|
||||
}
|
||||
// Expand any short jump whose displacement no longer fits rel8.
|
||||
changed := false
|
||||
for i, stmt := range t.Body {
|
||||
s, ok := stmt.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
mnem := strings.ToUpper(s.Mnemonic.Text)
|
||||
if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] {
|
||||
continue
|
||||
}
|
||||
name, ok := labelName(s.Operands[0])
|
||||
if !ok {
|
||||
continue // reported during emission
|
||||
}
|
||||
target, ok := offsets[resolve(name)]
|
||||
if !ok {
|
||||
continue // reported during emission
|
||||
}
|
||||
rel := int64(target - (pcs[i] + jumpSize(mnem, false)))
|
||||
if !fits8(rel) {
|
||||
long[i] = true
|
||||
changed = true
|
||||
}
|
||||
}
|
||||
if !changed {
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2: emit.
|
||||
out := append([]byte(nil), fi.prologue...)
|
||||
pos = len(fi.prologue)
|
||||
var patches []sbPatch
|
||||
var steps []spadjStep
|
||||
var lines []LineEntry
|
||||
if fi.useFP {
|
||||
// PUSHQ BP saves the return-address-relative base (+8); the MOVQ
|
||||
// changes nothing; SUBQ $size, SP completes the frame.
|
||||
steps = append(steps,
|
||||
spadjStep{1, 8},
|
||||
spadjStep{len(fi.prologue), 8 + fi.size},
|
||||
)
|
||||
}
|
||||
pos := len(fi.prologue)
|
||||
for i, stmt := range t.Body {
|
||||
s, ok := stmt.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, err := encodeInstr(s, pos, offsets, fi)
|
||||
if strings.ToUpper(s.Mnemonic.Text) == "RET" && fi.useFP {
|
||||
// The RET's epilogue prefix unwinds: ADDQ $size, SP restores
|
||||
// the saved-BP-only stack, POPQ BP the entry state.
|
||||
epi := len(fi.epilogue)
|
||||
steps = append(steps,
|
||||
spadjStep{pos + epi - 1, 8},
|
||||
spadjStep{pos + epi, 0},
|
||||
)
|
||||
}
|
||||
code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
}
|
||||
if len(code) != sizes[i] {
|
||||
return nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
||||
}
|
||||
patches = append(patches, ps...)
|
||||
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
|
||||
out = append(out, code...)
|
||||
pos += len(code)
|
||||
}
|
||||
return out, offsets, nil
|
||||
return out, patches, offsets, steps, lines, nil
|
||||
}
|
||||
|
||||
// jumpChain precomputes jump-to-jump folding: a label whose first instruction
|
||||
// is an unconditional local jump redirects its own jumpers to the ultimate
|
||||
// target. The Go toolchain chases exactly these chains (the linker's xfol
|
||||
// pass) before it encodes branches, so matching its bytes requires the same
|
||||
// redirection.
|
||||
func jumpChain(t *ast.Text) map[string]string {
|
||||
// label → the target of its leading unconditional local JMP, if any.
|
||||
leadsTo := map[string]string{}
|
||||
for i, stmt := range t.Body {
|
||||
l, ok := stmt.(*ast.Label)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
// Stacked labels share an address: skip to the first instruction.
|
||||
j := i + 1
|
||||
for j < len(t.Body) {
|
||||
if _, isLabel := t.Body[j].(*ast.Label); !isLabel {
|
||||
break
|
||||
}
|
||||
j++
|
||||
}
|
||||
if j >= len(t.Body) {
|
||||
continue
|
||||
}
|
||||
in, ok := t.Body[j].(*ast.Instr)
|
||||
if !ok || strings.ToUpper(in.Mnemonic.Text) != "JMP" || len(in.Operands) != 1 {
|
||||
continue
|
||||
}
|
||||
if name, ok := labelName(in.Operands[0]); ok {
|
||||
leadsTo[l.Name.Text] = name
|
||||
}
|
||||
}
|
||||
// Chase each chain to its end, guarding against cycles.
|
||||
chain := map[string]string{}
|
||||
for name := range leadsTo {
|
||||
visited := map[string]bool{name: true}
|
||||
cur := name
|
||||
for {
|
||||
next, ok := leadsTo[cur]
|
||||
if !ok || visited[next] {
|
||||
break
|
||||
}
|
||||
visited[next] = true
|
||||
cur = next
|
||||
}
|
||||
if cur != name {
|
||||
chain[name] = cur
|
||||
}
|
||||
}
|
||||
return chain
|
||||
}
|
||||
|
||||
// frameInfo carries the frame layout derived from the TEXT directive.
|
||||
@@ -119,15 +271,15 @@ func addSP(size int) []byte { // ADDQ $size, SP
|
||||
return append([]byte{0x48, 0x81, 0xC4}, le32(int64(size))...)
|
||||
}
|
||||
|
||||
// instrSize returns the encoded length of an instruction (pass 1). encodeInstr
|
||||
// already includes the epilogue for a RET in a frame-pointer function; jumps use
|
||||
// a fixed rel32 size (no epilogue).
|
||||
func instrSize(s *ast.Instr, fi frameInfo) (int, error) {
|
||||
// instrSize returns the encoded length of an instruction (layout pass).
|
||||
// encodeInstr already includes the epilogue for a RET in a frame-pointer
|
||||
// function; jumps use their short or long form (never an epilogue).
|
||||
func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, error) {
|
||||
mnem := strings.ToUpper(s.Mnemonic.Text)
|
||||
if isJumpMnemonic(mnem) {
|
||||
return jumpSize(mnem), nil
|
||||
return jumpSize(mnem, long), nil
|
||||
}
|
||||
code, err := encodeInstr(s, 0, nil, fi)
|
||||
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
@@ -142,18 +294,27 @@ func isJumpMnemonic(mnem string) bool {
|
||||
return ok
|
||||
}
|
||||
|
||||
// jumpSize returns the fixed length of a rel32 jump instruction.
|
||||
func jumpSize(mnem string) int {
|
||||
if mnem == "JMP" || mnem == "CALL" {
|
||||
// jumpSize returns the length of a jump instruction in the requested form:
|
||||
// short (rel8) where available, otherwise the rel32 form. CALL is always
|
||||
// rel32.
|
||||
func jumpSize(mnem string, long bool) int {
|
||||
if mnem == "CALL" {
|
||||
return 5 // opcode + rel32
|
||||
}
|
||||
if !long {
|
||||
return 2 // opcode + rel8
|
||||
}
|
||||
if mnem == "JMP" {
|
||||
return 5 // E9 + rel32
|
||||
}
|
||||
return 6 // 0x0F 0x8x + rel32
|
||||
}
|
||||
|
||||
// encodeInstr encodes one instruction, resolving jump targets against offsets
|
||||
// (relative to pc, the instruction's own offset). A RET in a frame-pointer
|
||||
// function is prefixed with the epilogue.
|
||||
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo) ([]byte, error) {
|
||||
// function is prefixed with the epilogue. resolve, when non-nil, redirects a
|
||||
// jump label through the jump-to-jump chain before the offset lookup.
|
||||
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, error) {
|
||||
mnem := strings.ToUpper(s.Mnemonic.Text)
|
||||
|
||||
var prefix []byte
|
||||
@@ -162,37 +323,53 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo) ([]
|
||||
}
|
||||
|
||||
var code []byte
|
||||
var ps []sbPatch
|
||||
var err error
|
||||
if isJumpMnemonic(mnem) {
|
||||
code, err = encodeJump(s, mnem, pc+len(prefix), offsets)
|
||||
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
|
||||
} else {
|
||||
code, err = encodeNormal(s, fi)
|
||||
code, ps, err = encodeNormal(s, fi, link)
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
return nil, nil, err
|
||||
}
|
||||
return append(prefix, code...), nil
|
||||
// Anchor the patch fields at function-relative positions: off indexes the
|
||||
// disp32 field, after is the address just past the instruction.
|
||||
body := pc + len(prefix)
|
||||
for i := range ps {
|
||||
ps[i].off += body
|
||||
ps[i].after = body + len(code)
|
||||
}
|
||||
return append(prefix, code...), ps, nil
|
||||
}
|
||||
|
||||
func encodeNormal(s *ast.Instr, fi frameInfo) ([]byte, error) {
|
||||
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) {
|
||||
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
|
||||
if size == 0 {
|
||||
size = 8
|
||||
}
|
||||
ops := make([]Operand, len(s.Operands))
|
||||
for i, op := range s.Operands {
|
||||
o, err := operandFromAST(op, size, fi)
|
||||
o, err := operandFromAST(op, size, fi, link)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
return nil, nil, err
|
||||
}
|
||||
ops[i] = o
|
||||
}
|
||||
return Encode(s.Mnemonic.Text, ops...)
|
||||
e := &enc{}
|
||||
if err := e.encode(s.Mnemonic.Text, ops); err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
ps := make([]sbPatch, len(e.patches))
|
||||
for i, p := range e.patches {
|
||||
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
|
||||
}
|
||||
return e.out, ps, nil
|
||||
}
|
||||
|
||||
// encodeJump encodes a JMP/CALL/Jcc with a rel32 offset resolved from the
|
||||
// target label.
|
||||
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int) ([]byte, error) {
|
||||
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
|
||||
// target label, in the short (rel8) or long (rel32) form.
|
||||
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long bool, resolve func(string) string) ([]byte, error) {
|
||||
if len(s.Operands) != 1 {
|
||||
return nil, fmt.Errorf("jump expects 1 operand, got %d", len(s.Operands))
|
||||
}
|
||||
@@ -200,12 +377,25 @@ func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int) ([]by
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("jump target must be a local label")
|
||||
}
|
||||
if resolve != nil && mnem != "CALL" {
|
||||
name = resolve(name)
|
||||
}
|
||||
target, ok := offsets[name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("undefined label %q", name)
|
||||
}
|
||||
rel := int64(target - (pc + jumpSize(mnem)))
|
||||
rel := int64(target - (pc + jumpSize(mnem, long)))
|
||||
|
||||
if !long {
|
||||
if !fits8(rel) {
|
||||
return nil, fmt.Errorf("jump to %q does not fit the short form", name)
|
||||
}
|
||||
if mnem == "JMP" {
|
||||
return []byte{0xEB, byte(int8(rel))}, nil
|
||||
}
|
||||
cc, _ := condCode(mnem)
|
||||
return []byte{0x70 + byte(cc), byte(int8(rel))}, nil
|
||||
}
|
||||
switch mnem {
|
||||
case "JMP":
|
||||
return append([]byte{0xE9}, le32(rel)...), nil
|
||||
@@ -231,7 +421,7 @@ var spReg = Reg{idx: 4, size: 8}
|
||||
|
||||
// operandFromAST converts a parsed operand into an encoder Operand, applying
|
||||
// the frame translation to FP/SP pseudo-register operands.
|
||||
func operandFromAST(op *ast.Operand, size int, fi frameInfo) (Operand, error) {
|
||||
func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) {
|
||||
switch op.Kind {
|
||||
case ast.OpImmediate:
|
||||
if op.Imm.HasVal {
|
||||
@@ -257,9 +447,22 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo) (Operand, error) {
|
||||
off := fi.spAdjust + a.Sym.Offset
|
||||
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
|
||||
}
|
||||
// SB (global symbol) needs a relocation — not yet supported.
|
||||
// SB (global symbol): a symbol defined in the same file (GLOBL) is
|
||||
// encoded RIP-relative and resolved by the file-level layout;
|
||||
// anything not defined here needs object-file emission.
|
||||
if a.Sym != nil && a.Sym.Pseudo == "SB" {
|
||||
return nil, fmt.Errorf("SB (global symbol) operands need relocation support (pending)")
|
||||
if link == nil || link.symbols == nil {
|
||||
return nil, fmt.Errorf("symbol %q needs file-level assembly (AssembleFile)", a.Sym.Name)
|
||||
}
|
||||
if !link.symbols[a.Sym.Name] {
|
||||
if a.Sym.Static {
|
||||
return nil, fmt.Errorf("undefined symbol %q", a.Sym.Name)
|
||||
}
|
||||
if !link.allowExternal {
|
||||
return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name)
|
||||
}
|
||||
}
|
||||
return sbMem{size: size, name: a.Sym.Name, addend: a.Sym.Offset}, nil
|
||||
}
|
||||
|
||||
// Memory with a real base register: (base), off(base), (base)(index*scale).
|
||||
|
||||
@@ -244,3 +244,76 @@ TEXT ·hsum(SB), NOSPLIT, $0
|
||||
t.Errorf("VEX kernel mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleShortJumps checks that a tight loop settles on the short (rel8)
|
||||
// jump forms, byte for byte with the Go assembler.
|
||||
func TestAssembleShortJumps(t *testing.T) {
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·loop(SB), NOSPLIT, $0
|
||||
XORQ AX, AX
|
||||
l1:
|
||||
ADDQ $1, AX
|
||||
CMPQ AX, $10
|
||||
JLT l1
|
||||
RET
|
||||
`)
|
||||
code, _, err := Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble: %v", err)
|
||||
}
|
||||
// From the Go-assembled function:
|
||||
// XORQ AX, AX 4831c0
|
||||
// ADDQ $1, AX 4883c001
|
||||
// CMPQ AX, $10 4883f80a
|
||||
// JLT l1 7cf6 (short, rel8)
|
||||
// RET c3
|
||||
want := []byte{
|
||||
0x48, 0x31, 0xc0,
|
||||
0x48, 0x83, 0xc0, 0x01,
|
||||
0x48, 0x83, 0xf8, 0x0a,
|
||||
0x7c, 0xf6,
|
||||
0xc3,
|
||||
}
|
||||
if hexBytes(code) != hexBytes(want) {
|
||||
t.Errorf("short-jump mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleJumpFolding checks jump-to-jump folding: a conditional jump to a
|
||||
// label that only holds an unconditional jump is redirected to the ultimate
|
||||
// target, exactly as the Go toolchain does before it encodes branches.
|
||||
func TestAssembleJumpFolding(t *testing.T) {
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·fold(SB), NOSPLIT, $0
|
||||
XORQ AX, AX
|
||||
JGE done
|
||||
INCQ AX
|
||||
done:
|
||||
JMP end
|
||||
end:
|
||||
RET
|
||||
`)
|
||||
code, _, err := Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble: %v", err)
|
||||
}
|
||||
// From the Go-assembled function: the JGE skips past the done: trampoline
|
||||
// straight to end:
|
||||
// XORQ AX, AX 4831c0
|
||||
// JGE end 7d05 (folded past done)
|
||||
// INCQ AX 48ffc0
|
||||
// JMP end eb00
|
||||
// RET c3
|
||||
want := []byte{
|
||||
0x48, 0x31, 0xc0,
|
||||
0x7d, 0x05,
|
||||
0x48, 0xff, 0xc0,
|
||||
0xeb, 0x00,
|
||||
0xc3,
|
||||
}
|
||||
if hexBytes(code) != hexBytes(want) {
|
||||
t.Errorf("jump-folding mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
|
||||
+301
@@ -0,0 +1,301 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
)
|
||||
|
||||
// This file emits ELF64 relocatable objects (ET_REL) from an assembled
|
||||
// Image: a .text section holding the function bodies, a .data section
|
||||
// holding the GLOBL initialisers, a symbol table with one symbol per TEXT
|
||||
// and GLOBL (file-local <> symbols are STB_LOCAL, the rest STB_GLOBAL), and
|
||||
// a .rela.text relocation table — one R_X86_64_PC32 entry per static-symbol
|
||||
// reference, internal references resolving against the local data symbols
|
||||
// and external ones against undefined globals. The output links with the
|
||||
// system toolchain (cc/ld) the way a hand-assembled .o would.
|
||||
|
||||
// ELF constants (ELF64, little-endian, System V).
|
||||
const (
|
||||
elfClass64 = 2
|
||||
elfDataLSB = 1
|
||||
elfVersion = 1
|
||||
|
||||
etREL = 1 // relocatable object
|
||||
emX8664 = 62
|
||||
|
||||
shtNull = 0
|
||||
shtProgbits = 1
|
||||
shtSymtab = 2
|
||||
shtStrtab = 3
|
||||
shtRela = 4
|
||||
|
||||
shfWrite = 1
|
||||
shfAlloc = 2
|
||||
shfExecInstr = 4
|
||||
|
||||
stbLocal = 0
|
||||
stbGlobal = 1
|
||||
|
||||
sttNotype = 0
|
||||
sttObject = 1
|
||||
sttFunc = 2
|
||||
sttSection = 3
|
||||
stInfoShift = 4
|
||||
|
||||
shnUndef = 0
|
||||
|
||||
rX8664PC32 = 2
|
||||
)
|
||||
|
||||
// elfSym is one symbol-table entry in construction.
|
||||
type elfSym struct {
|
||||
name string
|
||||
info byte
|
||||
shndx uint16
|
||||
value uint64
|
||||
size uint64
|
||||
}
|
||||
|
||||
// ELFObject returns the image as an ELF64 relocatable object file, ready for
|
||||
// the system linker. Symbol names are the TEXT and GLOBL identifiers as
|
||||
// written (the middle dot stripped); a package prefix, when present, is
|
||||
// joined with a dot. Every static-symbol reference becomes an
|
||||
// R_X86_64_PC32 relocation, so the code is position-independent and links
|
||||
// at any address.
|
||||
func (img *Image) ELFObject() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
// Section indices: 0 NULL, 1 .text, 2 .data; the tables follow.
|
||||
const (
|
||||
secText = 1
|
||||
secData = 2
|
||||
)
|
||||
|
||||
// Build the symbol table: the null entry and the two section symbols
|
||||
// come first, then the local symbols (static TEXT and GLOBL), then the
|
||||
// globals (exported TEXT and GLOBL, and the undefined externals) — ELF
|
||||
// requires every local to precede every global, and sh_info records the
|
||||
// boundary. symIdx maps a symbol name to its index for the relocations.
|
||||
var locals, globals []elfSym
|
||||
for _, fn := range img.Funcs {
|
||||
s := elfSym{
|
||||
name: objectName(fn.Pkg, fn.Name),
|
||||
info: sttFunc,
|
||||
shndx: secText,
|
||||
value: uint64(fn.Offset),
|
||||
size: uint64(fn.Size),
|
||||
}
|
||||
if fn.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, d := range img.DataSyms {
|
||||
s := elfSym{
|
||||
name: objectName(d.Pkg, d.Name),
|
||||
info: sttObject,
|
||||
shndx: secData,
|
||||
value: uint64(d.Offset),
|
||||
size: uint64(d.Size),
|
||||
}
|
||||
if d.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, name := range img.Externals {
|
||||
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
|
||||
}
|
||||
syms := []elfSym{
|
||||
{}, // the mandatory null entry
|
||||
{name: ".text", info: sttSection, shndx: secText},
|
||||
{name: ".data", info: sttSection, shndx: secData},
|
||||
}
|
||||
syms = append(syms, locals...)
|
||||
shInfo := len(syms) // first global symbol
|
||||
syms = append(syms, globals...)
|
||||
symIdx := map[string]int{}
|
||||
for i, s := range syms {
|
||||
symIdx[s.name] = i
|
||||
}
|
||||
|
||||
// Build the relocations.
|
||||
type elfRela struct {
|
||||
off uint64
|
||||
sym int
|
||||
addend int64
|
||||
}
|
||||
var relas []elfRela
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
relas = append(relas, elfRela{
|
||||
off: uint64(fn.Offset + r.Off),
|
||||
sym: idx,
|
||||
// R_X86_64_PC32 computes S + A − P with P the patch site; the
|
||||
// assembler measures the symbol from the instruction end,
|
||||
// After − Off bytes past the field, so the addend carries
|
||||
// that distance with a negative sign.
|
||||
addend: r.Addend - int64(r.After-r.Off),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Serialise the string tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
stNames.add(s.name)
|
||||
}
|
||||
stSections := newElfStrtab()
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
// Section presence: .rela.text only when there are relocations.
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6 // NULL, .text, .data, .symtab, .strtab, .shstrtab
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Lay the file out: header, section data, section headers.
|
||||
var out []byte
|
||||
out = append(out, make([]byte, 64)...) // ELF header, filled last
|
||||
|
||||
align := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
|
||||
align(16)
|
||||
textOff := len(out)
|
||||
out = append(out, img.Code...)
|
||||
|
||||
align(16)
|
||||
dataOff := len(out)
|
||||
out = append(out, img.Data...)
|
||||
|
||||
align(8)
|
||||
symtabOff := len(out)
|
||||
for _, s := range syms {
|
||||
var b [24]byte
|
||||
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
|
||||
b[4] = s.info
|
||||
b[5] = 0 // st_other
|
||||
le.PutUint16(b[6:], s.shndx)
|
||||
le.PutUint64(b[8:], s.value)
|
||||
le.PutUint64(b[16:], s.size)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
for _, r := range relas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|rX8664PC32)
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
|
||||
align(8)
|
||||
shoff := len(out)
|
||||
|
||||
// Section headers.
|
||||
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
|
||||
var b [64]byte
|
||||
le.PutUint32(b[0:], uint32(stSections.at(name)))
|
||||
le.PutUint32(b[4:], uint32(typ))
|
||||
le.PutUint64(b[8:], flags)
|
||||
le.PutUint64(b[16:], 0) // sh_addr
|
||||
le.PutUint64(b[24:], uint64(off))
|
||||
le.PutUint64(b[32:], uint64(size))
|
||||
le.PutUint32(b[40:], uint32(link))
|
||||
le.PutUint32(b[44:], uint32(info))
|
||||
le.PutUint64(b[48:], alignV)
|
||||
le.PutUint64(b[56:], entsize)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
|
||||
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
|
||||
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
|
||||
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
|
||||
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
|
||||
// The ELF header.
|
||||
hdr := out[:64]
|
||||
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
|
||||
le.PutUint16(hdr[16:], etREL)
|
||||
le.PutUint16(hdr[18:], emX8664)
|
||||
le.PutUint32(hdr[20:], elfVersion)
|
||||
le.PutUint64(hdr[24:], 0) // e_entry
|
||||
le.PutUint64(hdr[32:], 0) // e_phoff
|
||||
le.PutUint64(hdr[40:], uint64(shoff)) // e_shoff
|
||||
le.PutUint32(hdr[48:], 0) // e_flags
|
||||
le.PutUint16(hdr[52:], 64) // e_ehsize
|
||||
le.PutUint16(hdr[54:], 0) // e_phentsize
|
||||
le.PutUint16(hdr[56:], 0) // e_phnum
|
||||
le.PutUint16(hdr[58:], 64) // e_shentsize
|
||||
le.PutUint16(hdr[60:], uint16(nSections))
|
||||
le.PutUint16(hdr[62:], uint16(secShstr))
|
||||
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// objectName renders a symbol's object-file name: the identifier as written,
|
||||
// with an explicit package prefix joined by a dot.
|
||||
func objectName(pkg, name string) string {
|
||||
if pkg == "" {
|
||||
return name
|
||||
}
|
||||
return pkg + "." + name
|
||||
}
|
||||
|
||||
// elfStrtab is an ELF string table under construction.
|
||||
type elfStrtab struct {
|
||||
buf []byte
|
||||
off map[string]int
|
||||
}
|
||||
|
||||
func newElfStrtab() *elfStrtab {
|
||||
return &elfStrtab{buf: []byte{0}, off: map[string]int{"": 0}}
|
||||
}
|
||||
|
||||
func (s *elfStrtab) add(name string) {
|
||||
if _, ok := s.off[name]; ok {
|
||||
return
|
||||
}
|
||||
s.off[name] = len(s.buf)
|
||||
s.buf = append(s.buf, name...)
|
||||
s.buf = append(s.buf, 0)
|
||||
}
|
||||
|
||||
func (s *elfStrtab) at(name string) int { return s.off[name] }
|
||||
|
||||
func (s *elfStrtab) bytes() []byte { return s.buf }
|
||||
+310
@@ -0,0 +1,310 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"debug/elf"
|
||||
"encoding/binary"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// The object-file tests share one source: two exported functions, one
|
||||
// file-local constant reached through a relocation, and one external symbol
|
||||
// the linker must resolve. The functions take their arguments in the System
|
||||
// V registers (not the Go stack ABI) so a C driver can call them directly.
|
||||
const elfTestSrc = `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·addq(SB), NOSPLIT, $0
|
||||
LEAQ (DI)(SI*1), AX
|
||||
RET
|
||||
|
||||
TEXT ·getanswer(SB), NOSPLIT, $0
|
||||
MOVQ answer<>(SB), AX
|
||||
RET
|
||||
|
||||
TEXT ·useextern(SB), NOSPLIT, $0
|
||||
MOVQ extvar(SB), AX
|
||||
RET
|
||||
|
||||
GLOBL answer<>(SB), RODATA, $8
|
||||
DATA answer<>+0(SB)/8, $42
|
||||
`
|
||||
|
||||
func elfTestImage(t *testing.T) *Image {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse("t_amd64.s", elfTestSrc)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
return img
|
||||
}
|
||||
|
||||
// TestAssembleFileExternals checks that a reference to a symbol no GLOBL
|
||||
// defines is recorded as an external relocation instead of failing — the
|
||||
// raw image leaves the displacement zero, the object emitters carry it.
|
||||
func TestAssembleFileExternals(t *testing.T) {
|
||||
img := elfTestImage(t)
|
||||
if len(img.Externals) != 1 || img.Externals[0] != "extvar" {
|
||||
t.Fatalf("Externals = %v, want [extvar]", img.Externals)
|
||||
}
|
||||
var ext, local int
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
if r.External {
|
||||
ext++
|
||||
if r.Name != "extvar" {
|
||||
t.Errorf("external reloc names %q, want extvar", r.Name)
|
||||
}
|
||||
} else {
|
||||
local++
|
||||
if r.Name != "answer" {
|
||||
t.Errorf("local reloc names %q, want answer", r.Name)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if ext != 1 || local != 1 {
|
||||
t.Errorf("relocs = %d external, %d local; want 1 and 1", ext, local)
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFObject checks the structure of the emitted ELF64 relocatable
|
||||
// object: sections, the symbol table (bindings, types, values, sizes) and
|
||||
// the .rela.text relocations, parsed back with debug/elf.
|
||||
func TestELFObject(t *testing.T) {
|
||||
img := elfTestImage(t)
|
||||
obj, err := img.ELFObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFObject: %v", err)
|
||||
}
|
||||
f, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
if f.Type != elf.ET_REL || f.Machine != elf.EM_X86_64 {
|
||||
t.Errorf("type/machine = %v/%v, want ET_REL/EM_X86_64", f.Type, f.Machine)
|
||||
}
|
||||
|
||||
text := f.Section(".text")
|
||||
data := f.Section(".data")
|
||||
if text == nil || data == nil {
|
||||
t.Fatal("missing .text or .data section")
|
||||
}
|
||||
if text.Flags&elf.SHF_EXECINSTR == 0 || text.Flags&elf.SHF_ALLOC == 0 {
|
||||
t.Errorf(".text flags = %v", text.Flags)
|
||||
}
|
||||
if data.Flags&elf.SHF_WRITE == 0 {
|
||||
t.Errorf(".data flags = %v", data.Flags)
|
||||
}
|
||||
textData, err := text.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !bytes.Equal(textData, img.Code) {
|
||||
t.Errorf(".text contents differ from the image code")
|
||||
}
|
||||
|
||||
syms, err := f.Symbols()
|
||||
if err != nil {
|
||||
t.Fatalf("symbols: %v", err)
|
||||
}
|
||||
byName := map[string]elf.Symbol{}
|
||||
for _, s := range syms {
|
||||
byName[s.Name] = s
|
||||
}
|
||||
wantSym := func(name string, bind elf.SymBind, typ elf.SymType, section elf.SectionIndex, size uint64) {
|
||||
t.Helper()
|
||||
s, ok := byName[name]
|
||||
if !ok {
|
||||
t.Errorf("symbol %q not found", name)
|
||||
return
|
||||
}
|
||||
if elf.ST_BIND(s.Info) != bind || elf.ST_TYPE(s.Info) != typ {
|
||||
t.Errorf("%s: bind/type = %v/%v, want %v/%v", name, elf.ST_BIND(s.Info), elf.ST_TYPE(s.Info), bind, typ)
|
||||
}
|
||||
if s.Section != section {
|
||||
t.Errorf("%s: section = %v, want %v", name, s.Section, section)
|
||||
}
|
||||
if s.Size != size {
|
||||
t.Errorf("%s: size = %d, want %d", name, s.Size, size)
|
||||
}
|
||||
}
|
||||
// The emitted layout is fixed: 0 NULL, 1 .text, 2 .data.
|
||||
if f.Sections[1].Name != ".text" || f.Sections[2].Name != ".data" {
|
||||
t.Fatalf("section layout = %s, %s; want .text, .data", f.Sections[1].Name, f.Sections[2].Name)
|
||||
}
|
||||
textIdx := elf.SectionIndex(1)
|
||||
dataIdx := elf.SectionIndex(2)
|
||||
wantSym("addq", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 5)
|
||||
wantSym("getanswer", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 8)
|
||||
wantSym("useextern", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 8)
|
||||
wantSym("answer", elf.STB_LOCAL, elf.STT_OBJECT, dataIdx, 8)
|
||||
wantSym("extvar", elf.STB_GLOBAL, elf.STT_NOTYPE, elf.SHN_UNDEF, 0)
|
||||
|
||||
// Relocations: one for the file-local constant (resolving against the
|
||||
// local data symbol) and one for the external (against the undefined
|
||||
// global), both R_X86_64_PC32 with the −4 addend the PC-relative form
|
||||
// needs. debug/elf does not surface rela entries, so read the section
|
||||
// directly.
|
||||
relaSec := f.Section(".rela.text")
|
||||
if relaSec == nil {
|
||||
t.Fatal("missing .rela.text")
|
||||
}
|
||||
raw, err := relaSec.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(raw)%24 != 0 || len(raw)/24 != 2 {
|
||||
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
|
||||
}
|
||||
// Symbol names straight from the raw tables: r_info carries an index
|
||||
// into .symtab including the null entry, which debug/elf's Symbols()
|
||||
// slice may not mirror.
|
||||
symtabRaw, err := f.Section(".symtab").Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
strtabRaw, err := f.Section(".strtab").Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
symName := func(idx int) string {
|
||||
stName := binary.LittleEndian.Uint32(symtabRaw[idx*24:])
|
||||
end := bytes.IndexByte(strtabRaw[stName:], 0)
|
||||
return string(strtabRaw[stName : int(stName)+end])
|
||||
}
|
||||
for i := 0; i < 2; i++ {
|
||||
e := raw[i*24 : (i+1)*24]
|
||||
off := binary.LittleEndian.Uint64(e[0:])
|
||||
info := binary.LittleEndian.Uint64(e[8:])
|
||||
addend := int64(binary.LittleEndian.Uint64(e[16:]))
|
||||
typ := info & 0xffffffff
|
||||
sym := int(info >> 32)
|
||||
if typ != uint64(elf.R_X86_64_PC32) {
|
||||
t.Errorf("reloc %d: type %d, want R_X86_64_PC32", i, typ)
|
||||
}
|
||||
if addend != -4 {
|
||||
t.Errorf("reloc %d: addend %d, want -4", i, addend)
|
||||
}
|
||||
if name := symName(sym); name != "answer" && name != "extvar" {
|
||||
t.Errorf("reloc %d: symbol %q, want answer or extvar", i, name)
|
||||
}
|
||||
// The relocation offset lands on the disp32 field: the four bytes
|
||||
// before a RET-terminated eight-byte MOVQ.
|
||||
if off+4 > uint64(len(textData)) {
|
||||
t.Errorf("reloc %d: offset %d outside .text", i, off)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFObjectNoRelocations checks a file with no static-symbol references
|
||||
// emits a valid object without a .rela.text section.
|
||||
func TestELFObjectNoRelocations(t *testing.T) {
|
||||
f, errs := parser.Parse("n_amd64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·nop(SB), NOSPLIT, $0
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
obj, err := img.ELFObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFObject: %v", err)
|
||||
}
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
if ef.Section(".rela.text") != nil {
|
||||
t.Error("unexpected .rela.text section")
|
||||
}
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
found := false
|
||||
for _, s := range syms {
|
||||
if s.Name == "nop" && elf.ST_TYPE(s.Info) == elf.STT_FUNC {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Error("function symbol nop not found")
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFLinkAndRun is the end-to-end check: assemble the test functions,
|
||||
// link the emitted object with a C driver that defines the external symbol,
|
||||
// and run the result. Skipped when no C compiler is available.
|
||||
func TestELFLinkAndRun(t *testing.T) {
|
||||
cc, err := exec.LookPath("cc")
|
||||
if err != nil {
|
||||
t.Skip("no C compiler available")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
|
||||
img := elfTestImage(t)
|
||||
obj, err := img.ELFObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFObject: %v", err)
|
||||
}
|
||||
objPath := filepath.Join(dir, "t.o")
|
||||
if err := os.WriteFile(objPath, obj, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
const driver = `
|
||||
#include <stdio.h>
|
||||
|
||||
long addq(long a, long b);
|
||||
long getanswer(void);
|
||||
long useextern(void);
|
||||
|
||||
long extvar = 7;
|
||||
|
||||
int main(void) {
|
||||
printf("%ld %ld %ld\n", addq(41, 1), getanswer(), useextern());
|
||||
return 0;
|
||||
}
|
||||
`
|
||||
driverPath := filepath.Join(dir, "driver.c")
|
||||
if err := os.WriteFile(driverPath, []byte(driver), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// -no-pie: the encoder emits R_X86_64_PC32 for external references,
|
||||
// which a position-independent executable would reject (it wants
|
||||
// PLT32/GOT relocations, a future increment).
|
||||
appPath := filepath.Join(dir, "app")
|
||||
out, err := exec.Command(cc, "-no-pie", "-o", appPath, driverPath, objPath).CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("link failed: %v\n%s", err, out)
|
||||
}
|
||||
run, err := exec.Command(appPath).CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("run failed: %v\n%s", err, run)
|
||||
}
|
||||
if got := string(run); got != "42 42 7\n" {
|
||||
t.Errorf("output %q, want \"42 42 7\\n\"", got)
|
||||
}
|
||||
}
|
||||
+232
@@ -0,0 +1,232 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
)
|
||||
|
||||
// RISC-V ELF64 relocatable object emission.
|
||||
|
||||
const (
|
||||
emRISCV = 243 // EM_RISCV
|
||||
|
||||
// RISC-V relocation types.
|
||||
rRISCV32 = 1
|
||||
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
|
||||
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
|
||||
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
|
||||
)
|
||||
|
||||
// ELFRISCVObject returns the image as an ELF64 relocatable object file for
|
||||
// RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64
|
||||
// ELF emission: .text, .data, .symtab, .strtab and optional .rela.text.
|
||||
func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
const (
|
||||
secText = 1
|
||||
secData = 2
|
||||
)
|
||||
|
||||
// Build symbol table.
|
||||
var locals, globals []elfSym
|
||||
for _, fn := range img.Funcs {
|
||||
s := elfSym{
|
||||
name: objectName(fn.Pkg, fn.Name),
|
||||
info: sttFunc,
|
||||
shndx: secText,
|
||||
value: uint64(fn.Offset),
|
||||
size: uint64(fn.Size),
|
||||
}
|
||||
if fn.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, d := range img.DataSyms {
|
||||
s := elfSym{
|
||||
name: objectName(d.Pkg, d.Name),
|
||||
info: sttObject,
|
||||
shndx: secData,
|
||||
value: uint64(d.Offset),
|
||||
size: uint64(d.Size),
|
||||
}
|
||||
if d.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.info |= stbGlobal << stInfoShift
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, name := range img.Externals {
|
||||
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
|
||||
}
|
||||
syms := []elfSym{
|
||||
{},
|
||||
{name: ".text", info: sttSection, shndx: secText},
|
||||
{name: ".data", info: sttSection, shndx: secData},
|
||||
}
|
||||
syms = append(syms, locals...)
|
||||
shInfo := len(syms)
|
||||
syms = append(syms, globals...)
|
||||
symIdx := map[string]int{}
|
||||
for i, s := range syms {
|
||||
symIdx[s.name] = i
|
||||
}
|
||||
|
||||
// Build relocations. Each SB reference produces a pair:
|
||||
// AUIPC rd, 0 → R_RISCV_PCREL_HI20
|
||||
// ADDI/LD/SD → R_RISCV_PCREL_LO12_I or _S
|
||||
// For now we record them as individual entries; at link time
|
||||
// the linker must pair HI20 with its matching LO12.
|
||||
type elfRela struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
sym int
|
||||
addend int64
|
||||
}
|
||||
var relas []elfRela
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
// Determine relocation type from the relocation kind.
|
||||
typ := uint32(rRISCVPCRELHI20) // default: AUIPC
|
||||
switch r.Kind {
|
||||
case RelPCRelLO12:
|
||||
typ = rRISCVPCRELLO12I
|
||||
case RelPCRelLO12S:
|
||||
typ = rRISCVPCRELLO12S
|
||||
case RelPCRelAbs:
|
||||
typ = rRISCV32
|
||||
}
|
||||
relas = append(relas, elfRela{
|
||||
off: uint64(fn.Offset + r.Off),
|
||||
typ: typ,
|
||||
sym: idx,
|
||||
addend: r.Addend - int64(r.After-r.Off),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// String tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
stNames.add(s.name)
|
||||
}
|
||||
stSections := newElfStrtab()
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Layout.
|
||||
var out []byte
|
||||
out = append(out, make([]byte, 64)...)
|
||||
|
||||
align := func(n int) {
|
||||
for len(out)%n != 0 {
|
||||
out = append(out, 0)
|
||||
}
|
||||
}
|
||||
|
||||
align(16)
|
||||
textOff := len(out)
|
||||
out = append(out, img.Code...)
|
||||
|
||||
align(16)
|
||||
dataOff := len(out)
|
||||
out = append(out, img.Data...)
|
||||
|
||||
align(8)
|
||||
symtabOff := len(out)
|
||||
for _, s := range syms {
|
||||
var b [24]byte
|
||||
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
|
||||
b[4] = s.info
|
||||
b[5] = 0
|
||||
le.PutUint16(b[6:], s.shndx)
|
||||
le.PutUint64(b[8:], s.value)
|
||||
le.PutUint64(b[16:], s.size)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
for _, r := range relas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
|
||||
align(8)
|
||||
shoff := len(out)
|
||||
|
||||
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
|
||||
var b [64]byte
|
||||
le.PutUint32(b[0:], uint32(stSections.at(name)))
|
||||
le.PutUint32(b[4:], uint32(typ))
|
||||
le.PutUint64(b[8:], flags)
|
||||
le.PutUint64(b[16:], 0)
|
||||
le.PutUint64(b[24:], uint64(off))
|
||||
le.PutUint64(b[32:], uint64(size))
|
||||
le.PutUint32(b[40:], uint32(link))
|
||||
le.PutUint32(b[44:], uint32(info))
|
||||
le.PutUint64(b[48:], alignV)
|
||||
le.PutUint64(b[56:], entsize)
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
|
||||
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
|
||||
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
|
||||
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
|
||||
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
|
||||
// ELF header.
|
||||
hdr := out[:64]
|
||||
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
|
||||
le.PutUint16(hdr[16:], etREL)
|
||||
le.PutUint16(hdr[18:], emRISCV)
|
||||
le.PutUint32(hdr[20:], elfVersion)
|
||||
le.PutUint64(hdr[24:], 0)
|
||||
le.PutUint64(hdr[32:], 0)
|
||||
le.PutUint64(hdr[40:], uint64(shoff))
|
||||
le.PutUint32(hdr[48:], 0)
|
||||
le.PutUint16(hdr[52:], 64)
|
||||
le.PutUint16(hdr[54:], 0)
|
||||
le.PutUint16(hdr[56:], 0)
|
||||
le.PutUint16(hdr[58:], 64)
|
||||
le.PutUint16(hdr[60:], uint16(nSections))
|
||||
le.PutUint16(hdr[62:], uint16(secShstr))
|
||||
|
||||
return out, nil
|
||||
}
|
||||
+94
-6
@@ -19,7 +19,16 @@ func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
|
||||
}
|
||||
|
||||
type enc struct {
|
||||
out []byte
|
||||
out []byte
|
||||
patches []encPatch // disp32 fields awaiting static-symbol resolution
|
||||
}
|
||||
|
||||
// encPatch marks a 4-byte displacement field in enc.out that must receive the
|
||||
// RIP-relative offset of a static symbol once the file layout is settled.
|
||||
type encPatch struct {
|
||||
off int
|
||||
name string
|
||||
addend int64
|
||||
}
|
||||
|
||||
func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
@@ -40,10 +49,27 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
return e.encodeJcc(cc, ops)
|
||||
}
|
||||
|
||||
// VEX (AVX/AVX2) instructions: the trailing B/W/L/Q/D is part of the
|
||||
// mnemonic, not a size suffix, so dispatch before splitSize.
|
||||
if isVex(upper) {
|
||||
return e.encodeVex(upper, ops)
|
||||
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
||||
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
|
||||
// before splitSize. EVEX suffixes (.Z, .SAE, rounding, .BCST) split
|
||||
// off the mnemonic too.
|
||||
base, sfx, err := parseEvexSuffix(upper)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
|
||||
return e.encodeVec(base, ops, sfx)
|
||||
}
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s: the suffix requires an EVEX instruction", mnem)
|
||||
}
|
||||
|
||||
// CMOVcc and SETcc carry the condition in the mnemonic (CMOVLGT, SETNE).
|
||||
if strings.HasPrefix(upper, "CMOV") {
|
||||
return e.encodeCmov(upper, ops)
|
||||
}
|
||||
if strings.HasPrefix(upper, "SET") {
|
||||
return e.encodeSet(upper, ops)
|
||||
}
|
||||
|
||||
base, size := splitSize(upper)
|
||||
@@ -63,12 +89,20 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
return e.encodeUnary(unaryOp[base], ops, size)
|
||||
case "SHL", "SHR", "SAR":
|
||||
return e.encodeShift(shiftOp[base], ops, size)
|
||||
case "IMUL":
|
||||
case "IMUL", "IMUL3":
|
||||
return e.encodeImul(ops, size)
|
||||
case "PUSH":
|
||||
return e.encodePushPop(ops, true)
|
||||
case "POP":
|
||||
return e.encodePushPop(ops, false)
|
||||
case "LZCNT", "TZCNT":
|
||||
return e.encodeCount(base, ops, size)
|
||||
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX":
|
||||
return e.encodeMovExtend(base, ops)
|
||||
case "CVTSL2SD", "CVTSQ2SD":
|
||||
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
|
||||
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||
}
|
||||
return fmt.Errorf("unsupported instruction %q", mnem)
|
||||
}
|
||||
@@ -91,6 +125,38 @@ func splitSize(upper string) (base string, size int) {
|
||||
return upper, 0
|
||||
}
|
||||
|
||||
// encodeVec dispatches a VEX/EVEX mnemonic to the right encoding: KMOVW has
|
||||
// its own direction-dependent opcodes; KTESTW is always VEX; everything else
|
||||
// takes EVEX when an operand demands it (a ZMM or K register, or an
|
||||
// EVEX-only mnemonic) and VEX otherwise.
|
||||
func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
|
||||
if gs, ok := gatherTable[upper]; ok {
|
||||
return e.encodeGather(upper, gs, ops, sfx)
|
||||
}
|
||||
if ss, ok := scatterTable[upper]; ok {
|
||||
return e.encodeScatter(upper, ss, ops, sfx)
|
||||
}
|
||||
if upper == "KMOVW" || upper == "KMOVQ" {
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
||||
}
|
||||
return e.encodeKmov(upper, ops)
|
||||
}
|
||||
if isKOp(upper) {
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
||||
}
|
||||
return e.encodeKOp(upper, ops)
|
||||
}
|
||||
if upper == "KTESTW" || (!evexRequired(upper, ops) && !sfx.evexOnly()) {
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s: the .Z suffix requires an EVEX instruction", upper)
|
||||
}
|
||||
return e.encodeVex(upper, ops)
|
||||
}
|
||||
return e.encodeEvex(upper, ops, sfx)
|
||||
}
|
||||
|
||||
// --- instruction components -------------------------------------------------
|
||||
|
||||
type instr struct {
|
||||
@@ -100,17 +166,29 @@ type instr struct {
|
||||
rexX bool
|
||||
rexB bool
|
||||
rexForced bool // REX needed even with all bits zero (8-bit low registers)
|
||||
prefix byte // legacy 0xF2/0xF3 prefix (0 = none); emitted after 0x66
|
||||
opcode []byte
|
||||
modrm int // -1 if absent
|
||||
sib int // -1 if absent
|
||||
disp []byte
|
||||
imm []byte
|
||||
sb *sbRef // static-symbol displacement in disp, awaiting resolution
|
||||
}
|
||||
|
||||
// sbRef records that an instruction's displacement refers to a static symbol
|
||||
// rather than holding a literal value.
|
||||
type sbRef struct {
|
||||
name string
|
||||
addend int64
|
||||
}
|
||||
|
||||
func (e *enc) emit(i *instr) error {
|
||||
if i.opSize16 {
|
||||
e.out = append(e.out, 0x66)
|
||||
}
|
||||
if i.prefix != 0 {
|
||||
e.out = append(e.out, i.prefix)
|
||||
}
|
||||
rex := byte(0)
|
||||
if i.rexW {
|
||||
rex |= 0x08
|
||||
@@ -134,6 +212,9 @@ func (e *enc) emit(i *instr) error {
|
||||
if i.sib >= 0 {
|
||||
e.out = append(e.out, byte(i.sib))
|
||||
}
|
||||
if i.sb != nil {
|
||||
e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend})
|
||||
}
|
||||
e.out = append(e.out, i.disp...)
|
||||
e.out = append(e.out, i.imm...)
|
||||
return nil
|
||||
@@ -181,6 +262,13 @@ func setRMReg(i *instr, regField int, rexR, regForced bool, rm Operand, opSize i
|
||||
return nil
|
||||
case Mem:
|
||||
return setMem(i, regField, r)
|
||||
case sbMem:
|
||||
// RIP-relative reference; the displacement is patched once the static
|
||||
// symbol's address is known.
|
||||
i.modrm = regField<<3 | 0x05 // mod=00, rm=101 → (RIP)+disp32
|
||||
i.disp = le32(0)
|
||||
i.sb = &sbRef{name: r.name, addend: r.addend}
|
||||
return nil
|
||||
default:
|
||||
return fmt.Errorf("invalid r/m operand %T", rm)
|
||||
}
|
||||
|
||||
+161
-1
@@ -4,6 +4,7 @@
|
||||
package asm
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
@@ -70,7 +71,7 @@ func TestALU(t *testing.T) {
|
||||
checkSyntax(t, "and rbx, 0x7", "ANDQ", Imm(7), BX)
|
||||
checkSyntax(t, "or rcx, rbx", "ORQ", BX, CX)
|
||||
checkSyntax(t, "xor rax, rax", "XORQ", AX, AX)
|
||||
checkSyntax(t, "cmp r10, rsi", "CMPQ", SI, Reg{idx: 10, size: 8})
|
||||
checkSyntax(t, "cmp rsi, r10", "CMPQ", SI, Reg{idx: 10, size: 8})
|
||||
checkSyntax(t, "add rbx, qword ptr [rax]", "ADDQ", Ptr(AX, 0, 8), BX)
|
||||
checkSyntax(t, "add qword ptr [rax], rbx", "ADDQ", BX, Ptr(AX, 0, 8))
|
||||
checkSyntax(t, "cmp rbx, -0x20", "CMPQ", Imm(-32), BX)
|
||||
@@ -126,6 +127,52 @@ func TestControl(t *testing.T) {
|
||||
checkOp(t, x86asm.JBE, "JLS", Imm(0))
|
||||
}
|
||||
|
||||
// TestSSEMoveGroundTruth checks the legacy (non-VEX) SSE moves byte for byte
|
||||
// against the Go assembler. wantOp is the decoder's name, which differs from
|
||||
// the Plan 9 spelling for the octa moves (MOVOU = MOVDQU, MOVO = MOVDQA).
|
||||
func TestSSEMoveGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
wantOp string
|
||||
}{
|
||||
{"MOVOU (SI),X1", "MOVOU", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "f30f6f0e", "MOVDQU"},
|
||||
{"MOVOU X3,(DI)", "MOVOU", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "f30f7f1f", "MOVDQU"},
|
||||
{"MOVOU X1,X2", "MOVOU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f30f6fd1", "MOVDQU"},
|
||||
{"MOVOU (SI)(BX*4),X9", "MOVOU", []Operand{Idx(SI, BX, 4, 0, 16), vreg(t, "X9")}, "f3440f6f0c9e", "MOVDQU"},
|
||||
{"MOVO (SI),X1", "MOVO", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "660f6f0e", "MOVDQA"},
|
||||
{"MOVO X3,(DI)", "MOVO", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "660f7f1f", "MOVDQA"},
|
||||
{"MOVUPS (SI),X1", "MOVUPS", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "0f100e", "MOVUPS"},
|
||||
{"MOVAPS X3,(DI)", "MOVAPS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "0f291f", "MOVAPS"},
|
||||
{"MOVUPD (SI),X1", "MOVUPD", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "660f100e", "MOVUPD"},
|
||||
{"MOVAPD X3,(DI)", "MOVAPD", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "660f291f", "MOVAPD"},
|
||||
{"MOVSD (SI),X1", "MOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X1")}, "f20f100e", "MOVSD_XMM"},
|
||||
{"MOVSD X1,X2", "MOVSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f20f10d1", "MOVSD_XMM"},
|
||||
{"MOVSS X3,(DI)", "MOVSS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 4)}, "f30f111f", "MOVSS"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
if inst.Op.String() != c.wantOp {
|
||||
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestGoFlacScalarTail encodes the scalar tail of an analyze kernel to confirm
|
||||
// the encoder handles a realistic instruction sequence.
|
||||
func TestGoFlacScalarTail(t *testing.T) {
|
||||
@@ -134,3 +181,116 @@ func TestGoFlacScalarTail(t *testing.T) {
|
||||
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
|
||||
checkSyntax(t, "and r10, -0x8", "ANDQ", Imm(-8), Reg{idx: 10, size: 8})
|
||||
}
|
||||
|
||||
// TestScalarGroundTruth checks the scalar instruction families the go-flac
|
||||
// kernels use beyond the basic set, byte for byte against the Go assembler's
|
||||
// machine code. wantOp is the x86 decoder's name, which differs from the
|
||||
// Plan 9 spelling for some of these (CMOVLGT → CMOVG, MOVBLZX → MOVZX, …).
|
||||
func TestScalarGroundTruth(t *testing.T) {
|
||||
r8 := Reg{idx: 8, size: 8}
|
||||
r9 := Reg{idx: 9, size: 8}
|
||||
r9w := Reg{idx: 9, size: 2}
|
||||
r8w := Reg{idx: 8, size: 2}
|
||||
r13 := Reg{idx: 13, size: 8}
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
wantOp string
|
||||
}{
|
||||
{"LZCNTL AX,CX", "LZCNTL", []Operand{AX, CX}, "f30fbdc8", "LZCNT"},
|
||||
{"LZCNTQ R8,R9", "LZCNTQ", []Operand{r8, r9}, "f34d0fbdc8", "LZCNT"},
|
||||
{"LZCNTW AX,CX", "LZCNTW", []Operand{AX, CX}, "66f30fbdc8", "LZCNT"},
|
||||
{"TZCNTL AX,CX", "TZCNTL", []Operand{AX, CX}, "f30fbcc8", "TZCNT"},
|
||||
{"CMOVLGT CX,AX", "CMOVLGT", []Operand{CX, AX}, "0f4fc1", "CMOVG"},
|
||||
{"CMOVLEQ CX,AX", "CMOVLEQ", []Operand{CX, AX}, "0f44c1", "CMOVE"},
|
||||
{"CMOVQGT R9,R8", "CMOVQGT", []Operand{r9, r8}, "4d0f4fc1", "CMOVG"},
|
||||
{"CMOVWLS R9W,R8W", "CMOVWLS", []Operand{r9w, r8w}, "66450f46c1", "CMOVBE"},
|
||||
{"SETNE AL", "SETNE", []Operand{AL}, "0f95c0", "SETNE"},
|
||||
{"SETNE (AX)", "SETNE", []Operand{Ptr(AX, 0, 1)}, "0f9500", "SETNE"},
|
||||
{"MOVBLZX AL,CX", "MOVBLZX", []Operand{AL, CX}, "0fb6c8", "MOVZX"},
|
||||
{"MOVBLZX (SI),CX", "MOVBLZX", []Operand{Ptr(SI, 0, 1), CX}, "0fb60e", "MOVZX"},
|
||||
{"MOVWLSX (SI)(AX*1),CX", "MOVWLSX", []Operand{Idx(SI, AX, 1, 0, 2), CX}, "0fbf0c06", "MOVSX"},
|
||||
{"MOVLQSX CX,R8", "MOVLQSX", []Operand{CX, r8}, "4c63c1", "MOVSXD"},
|
||||
{"MOVBQZX AL,R8", "MOVBQZX", []Operand{AL, r8}, "4c0fb6c0", "MOVZX"},
|
||||
{"MOVWLZX AX,CX", "MOVWLZX", []Operand{AX, CX}, "0fb7c8", "MOVZX"},
|
||||
{"MOVWQZX AX,R8", "MOVWQZX", []Operand{AX, r8}, "4c0fb7c0", "MOVZX"},
|
||||
{"CVTSL2SD R8,X13", "CVTSL2SD", []Operand{r8, vreg(t, "X13")}, "f2450f2ae8", "CVTSI2SD"},
|
||||
{"CVTSL2SD AX,X0", "CVTSL2SD", []Operand{AX, vreg(t, "X0")}, "f20f2ac0", "CVTSI2SD"},
|
||||
{"CVTSQ2SD R8,X13", "CVTSQ2SD", []Operand{r8, vreg(t, "X13")}, "f24d0f2ae8", "CVTSI2SD"},
|
||||
{"INCW (R13)(AX*2)", "INCW", []Operand{Idx(r13, AX, 2, 0, 2)}, "6641ff444500", "INC"},
|
||||
// The traditional three-operand IMUL spelling.
|
||||
{"IMUL3L $31,CX,DX", "IMUL3L", []Operand{Imm(31), CX, DX}, "6bd11f", "IMUL"},
|
||||
{"IMUL3L $256,CX,DX", "IMUL3L", []Operand{Imm(256), CX, DX}, "69d100010000", "IMUL"},
|
||||
{"IMUL3Q $7,R9,R8", "IMUL3Q", []Operand{Imm(7), r9, r8}, "4d6bc107", "IMUL"},
|
||||
{"IMUL3W $5,CX,DX", "IMUL3W", []Operand{Imm(5), CX, DX}, "666bd105", "IMUL"},
|
||||
// Negative displacement with base + index (regression: the parser
|
||||
// used to drop the whole address).
|
||||
{"LEAQ -4(DX)(R9*4),R9", "LEAQ", []Operand{Idx(DX, r9, 4, -4, 8), r9}, "4e8d4c8afc", "LEA"},
|
||||
{"LEAQ 16(SI)(BX*4),R10", "LEAQ", []Operand{Idx(SI, BX, 4, 16, 8), Reg{idx: 10, size: 8}}, "4c8d549e10", "LEA"},
|
||||
// Register-to-register MOV uses the r/m←r opcode (reg = source), the
|
||||
// Go assembler's choice.
|
||||
{"MOVQ BX,R10", "MOVQ", []Operand{BX, Reg{idx: 10, size: 8}}, "4989da", "MOV"},
|
||||
{"MOVQ AX,BX", "MOVQ", []Operand{AX, BX}, "4889c3", "MOV"},
|
||||
{"MOVL AX,BX", "MOVL", []Operand{AX, BX}, "89c3", "MOV"},
|
||||
{"MOVB AL,BL", "MOVB", []Operand{AL, BL}, "88c3", "MOV"},
|
||||
{"MOVW AX,BX", "MOVW", []Operand{AX, BX}, "6689c3", "MOV"},
|
||||
{"MOVQ R12,R13", "MOVQ", []Operand{Reg{idx: 12, size: 8}, Reg{idx: 13, size: 8}}, "4d89e5", "MOV"},
|
||||
// CMP must record first − second: with a register second operand the
|
||||
// first goes in r/m, with a memory second operand the first goes in reg.
|
||||
{"CMPQ SI,R10", "CMPQ", []Operand{SI, Reg{idx: 10, size: 8}}, "4c39d6", "CMP"},
|
||||
{"CMPQ SI,(AX)", "CMPQ", []Operand{SI, Ptr(AX, 0, 8)}, "483b30", "CMP"},
|
||||
{"CMPQ (AX),SI", "CMPQ", []Operand{Ptr(AX, 0, 8), SI}, "483930", "CMP"},
|
||||
{"CMPL CX,(AX)", "CMPL", []Operand{CX, Ptr(AX, 0, 4)}, "3b08", "CMP"},
|
||||
{"CMPB AL,(BX)", "CMPB", []Operand{AL, Ptr(BX, 0, 1)}, "3a03", "CMP"},
|
||||
{"CMPW AX,BX", "CMPW", []Operand{AX, BX}, "6639d8", "CMP"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := strings.ReplaceAll(hexBytes(code), " ", ""); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
if inst.Op.String() != c.wantOp {
|
||||
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestScalarErrors checks that malformed conditional / extend / convert
|
||||
// instructions are rejected.
|
||||
func TestScalarErrors(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"CMOV arity", "CMOVLGT", []Operand{AX}},
|
||||
{"CMOV bare", "CMOV", []Operand{AX, BX}},
|
||||
{"CMOV bad size", "CMOVBGT", []Operand{AX, BX}},
|
||||
{"CMOV bad condition", "CMOVLXX", []Operand{AX, BX}},
|
||||
{"CMOV mem dst", "CMOVLGT", []Operand{AX, Ptr(BX, 0, 4)}},
|
||||
{"SET arity", "SETNE", []Operand{AL, BL}},
|
||||
{"SET bad condition", "SETXX", []Operand{AL}},
|
||||
{"SET bare", "SET", []Operand{AL}},
|
||||
{"LZCNT arity", "LZCNTL", []Operand{AX}},
|
||||
{"LZCNT mem dst", "LZCNTL", []Operand{AX, Ptr(BX, 0, 4)}},
|
||||
{"MOVBLZX mem dst", "MOVBLZX", []Operand{AL, Ptr(BX, 0, 4)}},
|
||||
{"CVTSL2SD gpr dst", "CVTSL2SD", []Operand{AX, BX}},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", c.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+1570
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,740 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestEvexGroundTruth checks the EVEX (AVX-512) encodings byte for byte
|
||||
// against machine code extracted from the Go toolchain's assembly of the
|
||||
// same instructions, covering every operand shape the go-flac AVX-512
|
||||
// kernels use: NDS arithmetic, immediate and variable shifts, shuffles with
|
||||
// an immediate, lane extracts, narrowing stores, broadcasts from a GPR or
|
||||
// memory, mask destinations, mask moves, disp8×N compression and the 5-bit
|
||||
// register fields (X/Y 16–31, Z 0–31).
|
||||
func TestEvexGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// NDS integer arithmetic / logic.
|
||||
{"VPXORD Z12,Z12,Z12", "VPXORD", []Operand{vreg(t, "Z12"), vreg(t, "Z12"), vreg(t, "Z12")}, "62511d48efe4"},
|
||||
{"VPXORQ Z8,Z9,Z10", "VPXORQ", []Operand{vreg(t, "Z8"), vreg(t, "Z9"), vreg(t, "Z10")}, "6251b548efd0"},
|
||||
{"VPADDD Z1,Z0,Z0", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z0"), vreg(t, "Z0")}, "62f17d48fec1"},
|
||||
{"VPSUBQ Z8,Z11,Z11", "VPSUBQ", []Operand{vreg(t, "Z8"), vreg(t, "Z11"), vreg(t, "Z11")}, "6251a548fbd8"},
|
||||
{"VPUNPCKLDQ Z5,Z3,Z6", "VPUNPCKLDQ", []Operand{vreg(t, "Z5"), vreg(t, "Z3"), vreg(t, "Z6")}, "62f1654862f5"},
|
||||
{"VPUNPCKHDQ Z5,Z3,Z7", "VPUNPCKHDQ", []Operand{vreg(t, "Z5"), vreg(t, "Z3"), vreg(t, "Z7")}, "62f165486afd"},
|
||||
{"VPMULLQ Z9,Z10,Z10", "VPMULLQ", []Operand{vreg(t, "Z9"), vreg(t, "Z10"), vreg(t, "Z10")}, "6252ad4840d1"},
|
||||
{"VPMULLD Z13,Z11,Z2", "VPMULLD", []Operand{vreg(t, "Z13"), vreg(t, "Z11"), vreg(t, "Z2")}, "62d2254840d5"},
|
||||
{"VPERMD Z0,Z15,Z8", "VPERMD", []Operand{vreg(t, "Z0"), vreg(t, "Z15"), vreg(t, "Z8")}, "6272054836c0"},
|
||||
// Packed-double arithmetic (EVEX forms carry W=1).
|
||||
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
|
||||
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
|
||||
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
|
||||
// Align (NDS + imm8).
|
||||
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
|
||||
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
|
||||
// Shifts: immediate (/digit) and variable (XMM count).
|
||||
{"VPSRAD $31,Z3,Z5", "VPSRAD", []Operand{Imm(31), vreg(t, "Z3"), vreg(t, "Z5")}, "62f1554872e31f"},
|
||||
{"VPSLLD $1,Z3,Z4", "VPSLLD", []Operand{Imm(1), vreg(t, "Z3"), vreg(t, "Z4")}, "62f15d4872f301"},
|
||||
{"VPSRAQ X31,Z8,Z8", "VPSRAQ", []Operand{vreg(t, "X31"), vreg(t, "Z8"), vreg(t, "Z8")}, "6211bd48e2c7"},
|
||||
// Mask destinations (the K register occupies the reg field).
|
||||
{"VPCMPEQD Z0,Z3,K1", "VPCMPEQD", []Operand{vreg(t, "Z0"), vreg(t, "Z3"), vreg(t, "K1")}, "62f1654876c8"},
|
||||
{"VPCMPEQD Y30,Y11,K1", "VPCMPEQD", []Operand{vreg(t, "Y30"), vreg(t, "Y11"), vreg(t, "K1")}, "6291252876ce"},
|
||||
// Mask moves and test (VEX-encoded).
|
||||
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
|
||||
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
|
||||
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
|
||||
// Moves, incl. disp8×N (64 for a 512-bit operand).
|
||||
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
|
||||
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
|
||||
{"VMOVDQU32 16(SI)(R15*4),Z4", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 16, 64), vreg(t, "Z4")}, "62b17e486fa4be10000000"},
|
||||
{"VMOVDQU32 Z0,4(SI)(AX*1)", "VMOVDQU32", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f17e487f840604000000"},
|
||||
{"VMOVDQU32 Z3,(DI)(R15*4)", "VMOVDQU32", []Operand{vreg(t, "Z3"), Idx(DI, vreg(t, "R15"), 4, 0, 64)}, "62b17e487f1cbf"},
|
||||
// VMOVDQU64 — the W1 qword variant.
|
||||
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
|
||||
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
|
||||
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
|
||||
// The wider AVX-512 F/BW integer set.
|
||||
{"VPADDB Z1,Z2,Z3", "VPADDB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48fcd9"},
|
||||
{"VPSUBW Z1,Z2,Z3", "VPSUBW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48f9d9"},
|
||||
{"VPANDQ Z1,Z2,Z3", "VPANDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed48dbd9"},
|
||||
{"VPANDND Z1,Z2,Z3", "VPANDND", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48dfd9"},
|
||||
{"VPMULLW Z1,Z2,Z3", "VPMULLW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48d5d9"},
|
||||
{"VPMINUB Z1,Z2,Z3", "VPMINUB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48dad9"},
|
||||
{"VPMAXUQ Z1,Z2,Z3", "VPMAXUQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed483fd9"},
|
||||
{"VPAVGW Z1,Z2,Z3", "VPAVGW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48e3d9"},
|
||||
{"VPSLLVQ Z3,Z1,Z2", "VPSLLVQ", []Operand{vreg(t, "Z3"), vreg(t, "Z1"), vreg(t, "Z2")}, "62f2f54847d3"},
|
||||
{"VPSRAVQ Z3,Z1,Z2", "VPSRAVQ", []Operand{vreg(t, "Z3"), vreg(t, "Z1"), vreg(t, "Z2")}, "62f2f54846d3"},
|
||||
{"VPSHUFD $0x1B,Z1,Z2", "VPSHUFD", []Operand{Imm(0x1B), vreg(t, "Z1"), vreg(t, "Z2")}, "62f17d4870d11b"},
|
||||
{"VPSHUFB Z1,Z2,Z3", "VPSHUFB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4800d9"},
|
||||
{"VMOVDQU8 Z1,Z2", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17f487fca"},
|
||||
{"VMOVDQU16 Z1,Z2", "VMOVDQU16", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff487fca"},
|
||||
// Indices 16–31: rm[4] rides in X̄ for register operands.
|
||||
{"VPSHUFD $1,X16,X17", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}, "62a17d0870c801"},
|
||||
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
|
||||
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
|
||||
// Conversions and narrowing stores (reg = wide source).
|
||||
{"VCVTQQ2PD Z12,Z12", "VCVTQQ2PD", []Operand{vreg(t, "Z12"), vreg(t, "Z12")}, "6251fe48e6e4"},
|
||||
{"VCVTQQ2PD X13,X13", "VCVTQQ2PD", []Operand{vreg(t, "X13"), vreg(t, "X13")}, "6251fe08e6ed"},
|
||||
{"VPMOVSXDQ 32(SI),Z12", "VPMOVSXDQ", []Operand{Ptr(SI, 32, 32), vreg(t, "Z12")}, "62727d48256601"},
|
||||
{"VPMOVDW Z0,Y0", "VPMOVDW", []Operand{vreg(t, "Z0"), vreg(t, "Y0")}, "62f27e4833c0"},
|
||||
{"VPMOVQD Z11,Y11", "VPMOVQD", []Operand{vreg(t, "Z11"), vreg(t, "Y11")}, "62527e4835db"},
|
||||
// Lane extracts.
|
||||
{"VEXTRACTI64X4 $1,Z8,Y9", "VEXTRACTI64X4", []Operand{Imm(1), vreg(t, "Z8"), vreg(t, "Y9")}, "6253fd483bc101"},
|
||||
{"VEXTRACTF64X4 $1,Z10,Y11", "VEXTRACTF64X4", []Operand{Imm(1), vreg(t, "Z10"), vreg(t, "Y11")}, "6253fd481bd301"},
|
||||
// Broadcasts: GPR source (0x7C) vs memory source (0x58/0x59, disp8×4/8).
|
||||
{"VPBROADCASTD AX,Z15", "VPBROADCASTD", []Operand{AX, vreg(t, "Z15")}, "62727d487cf8"},
|
||||
{"VPBROADCASTD (SI),Z8", "VPBROADCASTD", []Operand{Ptr(SI, 0, 4), vreg(t, "Z8")}, "62727d485806"},
|
||||
{"VPBROADCASTD 4(SI),Z10", "VPBROADCASTD", []Operand{Ptr(SI, 4, 4), vreg(t, "Z10")}, "62727d48585601"},
|
||||
{"VPBROADCASTQ R8,X31", "VPBROADCASTQ", []Operand{vreg(t, "R8"), vreg(t, "X31")}, "6242fd087cf8"},
|
||||
{"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"},
|
||||
// Register indices 16–31 exist only in EVEX encodings.
|
||||
{"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"},
|
||||
// Packed double arithmetic / unpack (EVEX forms carry W=1).
|
||||
{"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"},
|
||||
{"VDIVPD Z4,Z5,Z6", "VDIVPD", []Operand{vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "Z6")}, "62f1d5485ef4"},
|
||||
{"VMINPD Z7,Z8,Z9", "VMINPD", []Operand{vreg(t, "Z7"), vreg(t, "Z8"), vreg(t, "Z9")}, "6271bd485dcf"},
|
||||
{"VMAXPD Z10,Z11,Z12", "VMAXPD", []Operand{vreg(t, "Z10"), vreg(t, "Z11"), vreg(t, "Z12")}, "6251a5485fe2"},
|
||||
{"VUNPCKLPD Z1,Z2,Z3", "VUNPCKLPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4814d9"},
|
||||
{"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"},
|
||||
{"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"},
|
||||
{"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"},
|
||||
// VMOVDDUP — duplicate the low double; disp8×N = 64 at 512 bits, and
|
||||
// X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄).
|
||||
{"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"},
|
||||
{"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"},
|
||||
{"VMOVDDUP X16,X17", "VMOVDDUP", []Operand{vreg(t, "X16"), vreg(t, "X17")}, "62a1ff0812c8"},
|
||||
// Conversions: DQ→PS, PS→PD (pp = 00, the Go assembler's choice),
|
||||
// DQ→PD (the destination sets the length).
|
||||
{"VCVTDQ2PS Z1,Z2", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"},
|
||||
{"VCVTPS2PD Y1,Z2", "VCVTPS2PD", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17c485ad1"},
|
||||
{"VCVTPS2PD 32(AX),Z2", "VCVTPS2PD", []Operand{Ptr(AX, 32, 32), vreg(t, "Z2")}, "62f17c485a5001"},
|
||||
{"VCVTDQ2PD Y1,Z2", "VCVTDQ2PD", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17e48e6d1"},
|
||||
// PD→DQ conversions: the source is the wide operand and fixes the
|
||||
// length (ZMM source → L'L = 10 even with an XMM destination; a
|
||||
// memory source takes the length the mnemonic's spelling implies).
|
||||
{"VCVTPD2DQ Z1,Y2", "VCVTPD2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff48e6d1"},
|
||||
{"VCVTPD2DQ 64(AX),Y2", "VCVTPD2DQ", []Operand{Ptr(AX, 64, 64), vreg(t, "Y2")}, "62f1ff48e65001"},
|
||||
{"VCVTTPD2DQ Z3,Y4", "VCVTTPD2DQ", []Operand{vreg(t, "Z3"), vreg(t, "Y4")}, "62f1fd48e6e3"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
want := strings.ReplaceAll(c.want, " ", "")
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
if inst.Len != len(code) {
|
||||
t.Errorf("%s: Decode consumed %d of %d bytes", c.name, inst.Len, len(code))
|
||||
}
|
||||
if inst.Op.String() != c.mnem {
|
||||
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexMasking checks the AVX-512 mask operand (K1–K7, placed freely among
|
||||
// the operands) and the .Z zeroing suffix, byte for byte against the Go
|
||||
// assembler.
|
||||
func TestEvexMasking(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// Masked arithmetic: K anywhere among the operands; .Z sets the z bit.
|
||||
{"VPADDD.Z merging+zeroing", "VPADDD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f16dcafed9"},
|
||||
{"VPADDD merging", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49fed9"},
|
||||
{"VADDPD.Z", "VADDPD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1edca58d9"},
|
||||
{"VPMINSD.Z", "VPMINSD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K5"), vreg(t, "Z3")}, "62f26dcd39d9"},
|
||||
{"VPMINSQ.Z", "VPMINSQ.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K5"), vreg(t, "Z3")}, "62f2edcd39d9"},
|
||||
// Masked immediate shift (K before the destination).
|
||||
{"VPSRAD.Z", "VPSRAD.Z", []Operand{Imm(1), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f165c972e201"},
|
||||
{"VPSLLD merge", "VPSLLD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1654a72f104"},
|
||||
// Masked align.
|
||||
{"VALIGND", "VALIGND", []Operand{Imm(12), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f36d4b03e10c"},
|
||||
// Masked conversion and extract.
|
||||
{"VCVTQQ2PD.Z", "VCVTQQ2PD.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1fecae6d9"},
|
||||
{"VEXTRACTI64X4", "VEXTRACTI64X4", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f3fd4a3bcb01"},
|
||||
// Masked moves: K sits between the register and memory operands.
|
||||
{"VMOVDQU8 store", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "K3"), Ptr(SI, 0, 64)}, "62f17f4b7f0e"},
|
||||
{"VMOVDQU32 load", "VMOVDQU32", []Operand{Ptr(SI, 0, 64), vreg(t, "K4"), vreg(t, "Z1")}, "62f17e4c6f0e"},
|
||||
{"VMOVDQU32 store", "VMOVDQU32", []Operand{vreg(t, "Z1"), vreg(t, "K4"), Ptr(DI, 0, 64)}, "62f17e4c7f0f"},
|
||||
// Masked comparison with a K destination: dst K1, mask K2.
|
||||
{"VPCMPEQD k-dst+mask", "VPCMPEQD", []Operand{vreg(t, "Z0"), vreg(t, "Z3"), vreg(t, "K2"), vreg(t, "K1")}, "62f1654a76c8"},
|
||||
// Masked floating point: packed double, the scalar SD/SS forms (which
|
||||
// exist under EVEX only for masked and zeroing use) and conversions.
|
||||
{"VSUBPD.Z", "VSUBPD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f1edcb5ce1"},
|
||||
{"VADDSD merge", "VADDSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K3"), vreg(t, "X4")}, "62f1ef0b58e1"},
|
||||
{"VSUBSD.Z", "VSUBSD.Z", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K5"), vreg(t, "X3")}, "62f1ef8d5cd9"},
|
||||
{"VADDSS merge", "VADDSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f16e0958d9"},
|
||||
{"VCVTPD2DQ merge", "VCVTPD2DQ", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f1ff4ae6d9"},
|
||||
{"VCVTTPD2DQ.Z", "VCVTTPD2DQ.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f1fdcae6d9"},
|
||||
{"VCVTDQ2PS.Z", "VCVTDQ2PS.Z", []Operand{vreg(t, "Z1"), vreg(t, "K4"), vreg(t, "Z2")}, "62f17ccc5bd1"},
|
||||
{"VCVTDQ2PD merge", "VCVTDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "K2"), vreg(t, "X3")}, "62f17e0ae6d9"},
|
||||
{"VCVTDQ2PD.Z", "VCVTDQ2PD.Z", []Operand{vreg(t, "Y1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f17ecae6d1"},
|
||||
{"VCVTPS2PD.Z", "VCVTPS2PD.Z", []Operand{vreg(t, "Y1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f17ccb5ad1"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
want := c.mnem
|
||||
if i := len(want) - 2; i > 0 && want[i:] == ".Z" {
|
||||
want = want[:i]
|
||||
}
|
||||
if inst.Op.String() != want {
|
||||
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
|
||||
}
|
||||
}
|
||||
|
||||
// Error cases.
|
||||
bad := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"zeroing without mask", "VPADDD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"K0 mask", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K0"), vreg(t, "Z3")}},
|
||||
{"two masks", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "K2"), vreg(t, "Z3")}},
|
||||
{".Z on VEX-only", "VPSHUFD.Z", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}},
|
||||
{"broadcast unsupported", "VPXORD.BCST", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"rounding unsupported", "VPXORD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"bcst with rounding", "VADDPD.BCST.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"Z not last", "VADDPD.Z.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"duplicate suffix", "VADDPD.Z.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"KMOVW.Z", "KMOVW.Z", []Operand{vreg(t, "K1"), vreg(t, "K2")}},
|
||||
}
|
||||
for _, c := range bad {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", c.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set — ternary
|
||||
// logic, lane shuffles/inserts/extracts, compares with a K destination,
|
||||
// permutes, the wider integer families, expand/compress, broadcasts,
|
||||
// rotates and word shifts, the opmask instructions, the EVEX suffixes
|
||||
// (rounding/SAE/broadcast) and the aligned/scalar moves — byte for byte
|
||||
// against the Go assembler.
|
||||
func TestEvexExtendedGroundTruth(t *testing.T) {
|
||||
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// Ternary logic and lane shuffles (NDS + imm8).
|
||||
{"VPTERNLOGD", "VPTERNLOGD", []Operand{Imm(0xE8), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4825d9e8"},
|
||||
{"VPTERNLOGQ", "VPTERNLOGQ", []Operand{Imm(0x96), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4825d996"},
|
||||
{"VSHUFI32X4", "VSHUFI32X4", []Operand{Imm(0x4E), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "62f36d2843d94e"},
|
||||
{"VSHUFF64X2", "VSHUFF64X2", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4823d901"},
|
||||
{"VPALIGNR", "VPALIGNR", []Operand{Imm(7), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d480fd907"},
|
||||
// Permutes.
|
||||
{"VPERMB", "VPERMB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d488dd9"},
|
||||
{"VPERMW", "VPERMW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed488dd9"},
|
||||
{"VPERMI2D", "VPERMI2D", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4876d9"},
|
||||
{"VPERMT2PD", "VPERMT2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed487fd9"},
|
||||
// Compare with a K destination (and an immediate predicate).
|
||||
{"VCMPPD", "VCMPPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3")}, "62f1ed48c2d904"},
|
||||
{"VCMPPS", "VCMPPS", []Operand{Imm(0), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "K4")}, "62f16c28c2e100"},
|
||||
{"VCMPSD", "VCMPSD", []Operand{Imm(17), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K5")}, "62f1ef08c2e911"},
|
||||
// Rounding / SAE / broadcast suffixes.
|
||||
{"VADDPD.RN_SAE", "VADDPD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed1858d9"},
|
||||
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
|
||||
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
|
||||
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
|
||||
// Packed single arithmetic (same opcodes, no mandatory prefix) —
|
||||
// ZMM, YMM and XMM widths, rounding and broadcast.
|
||||
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
|
||||
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
|
||||
{"VMAXPS", "VMAXPS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e85fd9"},
|
||||
{"VDIVPS.RD_SAE", "VDIVPS.RD_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c385ed9"},
|
||||
{"VADDPS.BCST", "VADDPS.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f174585810"},
|
||||
// Compress / expand.
|
||||
{"VCOMPRESSPD", "VCOMPRESSPD", []Operand{vreg(t, "Z1"), mem64(DI)}, "62f2fd488a0f"},
|
||||
{"VEXPANDPS", "VEXPANDPS", []Operand{mem64(SI), vreg(t, "Y2")}, "62f27d288816"},
|
||||
{"VPCOMPRESSD.Z", "VPCOMPRESSD.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), mem64(DI)}, "62f27dca8b0f"},
|
||||
// Broadcasts.
|
||||
{"VPBROADCASTB gpr", "VPBROADCASTB", []Operand{BX, vreg(t, "Z1")}, "62f27d487acb"},
|
||||
{"VPBROADCASTW mem", "VPBROADCASTW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d487910"},
|
||||
{"VBROADCASTSS", "VBROADCASTSS", []Operand{mem64(AX), vreg(t, "Y3")}, "c4e27d1818"},
|
||||
{"VBROADCASTSD", "VBROADCASTSD", []Operand{mem64(AX), vreg(t, "Z4")}, "62f2fd481920"},
|
||||
// Wider integer families.
|
||||
{"VPMADDWD", "VPMADDWD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48f5d9"},
|
||||
{"VPMADDUBSW", "VPMADDUBSW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4804d9"},
|
||||
{"VPMULHUW", "VPMULHUW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48e4d9"},
|
||||
{"VPSLLVW", "VPSLLVW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4812d9"},
|
||||
{"VPACKSSWB", "VPACKSSWB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d4863d9"},
|
||||
{"VPACKUSDW", "VPACKUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482bd9"},
|
||||
// Absolute values and replicating moves.
|
||||
{"VPABSD", "VPABSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d481ed1"},
|
||||
{"VPABSQ mem", "VPABSQ", []Operand{mem64(AX), vreg(t, "Z2")}, "62f2fd481f10"},
|
||||
{"VMOVSLDUP", "VMOVSLDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa12d1"},
|
||||
{"VMOVSHDUP", "VMOVSHDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e4816d1"},
|
||||
// Rotates and word/qword shifts.
|
||||
{"VPROLD", "VPROLD", []Operand{Imm(5), vreg(t, "Z1"), vreg(t, "Z2")}, "62f16d4872c905"},
|
||||
{"VPRORQ", "VPRORQ", []Operand{Imm(63), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4872c13f"},
|
||||
{"VPSLLW", "VPSLLW", []Operand{Imm(9), vreg(t, "X1"), vreg(t, "X2")}, "c5e971f109"},
|
||||
{"VPSRLQ", "VPSRLQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4873d103"},
|
||||
// Opmask instructions (VEX-encoded, the width in the L/W/pp bits).
|
||||
{"KANDW", "KANDW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec41d9"},
|
||||
{"KORD", "KORD", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d545f4"},
|
||||
{"KXNORQ", "KXNORQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec46d9"},
|
||||
{"KNOTB", "KNOTB", []Operand{vreg(t, "K4"), vreg(t, "K5")}, "c5f944ec"},
|
||||
{"KUNPCKBW", "KUNPCKBW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ed4bd9"},
|
||||
{"KSHIFTLW", "KSHIFTLW", []Operand{Imm(2), vreg(t, "K1"), vreg(t, "K2")}, "c4e3f932d102"},
|
||||
{"KADDQ", "KADDQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec4ad9"},
|
||||
{"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"},
|
||||
{"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"},
|
||||
{"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"},
|
||||
// Lane extract / insert.
|
||||
{"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"},
|
||||
{"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"},
|
||||
{"VINSERTF32X8", "VINSERTF32X8", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d481ad901"},
|
||||
{"VINSERTI64X4", "VINSERTI64X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed483ad901"},
|
||||
// Aligned moves and the scalar single move.
|
||||
{"VMOVAPS", "VMOVAPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4829ca"},
|
||||
{"VMOVDQA64 mem", "VMOVDQA64", []Operand{mem64(AX), vreg(t, "Z2")}, "62f1fd486f10"},
|
||||
{"VMOVSS mem", "VMOVSS", []Operand{mem64(AX), vreg(t, "X2")}, "c5fa1010"},
|
||||
// Conversions and extending/narrowing moves.
|
||||
{"VCVTPS2DQ", "VCVTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17d485bd1"},
|
||||
{"VCVTTPS2DQ", "VCVTTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e485bd1"},
|
||||
{"VPMOVZXBW", "VPMOVZXBW", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d30d1"},
|
||||
{"VPMOVSXBW mem", "VPMOVSXBW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d482010"},
|
||||
{"VPMOVWB", "VPMOVWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4830ca"},
|
||||
{"VPMOVQB", "VPMOVQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4832ca"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
want := c.mnem
|
||||
if i := strings.IndexByte(want, '.'); i > 0 {
|
||||
want = want[:i]
|
||||
}
|
||||
if inst.Op.String() != want {
|
||||
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
|
||||
// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef,
|
||||
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions —
|
||||
// plus gather/scatter with VSIB addressing, byte for byte against the Go
|
||||
// assembler.
|
||||
func TestEvexHelperGroundTruth(t *testing.T) {
|
||||
vsib := func(base, idx string, scale int) Operand {
|
||||
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
|
||||
}
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// Reciprocals and rsqrt (packed RM, scalar NDS).
|
||||
{"VRCP14PD", "VRCP14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484cd1"},
|
||||
{"VRCP14PS", "VRCP14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484cd1"},
|
||||
{"VRCP14SD", "VRCP14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084dd9"},
|
||||
{"VRCP14SS", "VRCP14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084dd9"},
|
||||
{"VRSQRT14PD", "VRSQRT14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484ed1"},
|
||||
{"VRSQRT14PS", "VRSQRT14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484ed1"},
|
||||
{"VRSQRT14SD", "VRSQRT14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084fd9"},
|
||||
{"VRSQRT14SS", "VRSQRT14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084fd9"},
|
||||
// Getexp (packed RM, scalar NDS).
|
||||
{"VGETEXPPD", "VGETEXPPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd4842d1"},
|
||||
{"VGETEXPPS", "VGETEXPPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d4842d1"},
|
||||
{"VGETEXPSD", "VGETEXPSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed0843d9"},
|
||||
{"VGETEXPSS", "VGETEXPSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d0843d9"},
|
||||
// Scalef (NDS).
|
||||
{"VSCALEFPD", "VSCALEFPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed482cd9"},
|
||||
{"VSCALEFPS", "VSCALEFPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482cd9"},
|
||||
{"VSCALEFSD", "VSCALEFSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed082dd9"},
|
||||
{"VSCALEFSS", "VSCALEFSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d082dd9"},
|
||||
// Rndscale / getmant / reduce (packed $imm,src,dst; scalar NDS+imm).
|
||||
{"VRNDSCALEPD", "VRNDSCALEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4809d104"},
|
||||
{"VRNDSCALEPS", "VRNDSCALEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4808d104"},
|
||||
{"VRNDSCALESD", "VRNDSCALESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed080bd904"},
|
||||
{"VRNDSCALESS", "VRNDSCALESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d080ad904"},
|
||||
{"VGETMANTPD", "VGETMANTPD", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4826d103"},
|
||||
{"VGETMANTPS", "VGETMANTPS", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4826d103"},
|
||||
{"VGETMANTSD", "VGETMANTSD", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0827d903"},
|
||||
{"VGETMANTSS", "VGETMANTSS", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0827d903"},
|
||||
{"VREDUCEPD", "VREDUCEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4856d104"},
|
||||
{"VREDUCEPS", "VREDUCEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4856d104"},
|
||||
{"VREDUCESD", "VREDUCESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0857d904"},
|
||||
{"VREDUCESS", "VREDUCESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0857d904"},
|
||||
// Fixupimm / range (NDS + imm8).
|
||||
{"VFIXUPIMMPD", "VFIXUPIMMPD", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4854d902"},
|
||||
{"VFIXUPIMMPS", "VFIXUPIMMPS", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4854d902"},
|
||||
{"VFIXUPIMMSD", "VFIXUPIMMSD", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0855d902"},
|
||||
{"VFIXUPIMMSS", "VFIXUPIMMSS", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0855d902"},
|
||||
{"VRANGEPD", "VRANGEPD", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4850d901"},
|
||||
{"VRANGEPS", "VRANGEPS", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4850d901"},
|
||||
{"VRANGESD", "VRANGESD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0851d901"},
|
||||
{"VRANGESS", "VRANGESS", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0851d901"},
|
||||
// FP class test ($imm, src, kdst; packed forms carry the length in
|
||||
// the X/Y/Z mnemonic suffix the decoder drops).
|
||||
{"VFPCLASSPDZ", "VFPCLASSPDZ", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "K2")}, "62f3fd4866d104"},
|
||||
{"VFPCLASSPSY", "VFPCLASSPSY", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "K2")}, "62f37d2866d104"},
|
||||
{"VFPCLASSSD", "VFPCLASSSD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f3fd0867d104"},
|
||||
{"VFPCLASSSS", "VFPCLASSSS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f37d0867d104"},
|
||||
// Gather: VEX spelling (mask register, VSIB, destination) and EVEX
|
||||
// spelling (VSIB, K mask, destination; L'L follows the VSIB index).
|
||||
{"VGATHERDPS vex", "VGATHERDPS", []Operand{vreg(t, "X2"), vsib("SI", "X1", 4), vreg(t, "X3")}, "c4e269921c8e"},
|
||||
{"VPGATHERDD vex", "VPGATHERDD", []Operand{vreg(t, "Y2"), vsib("SI", "Y1", 4), vreg(t, "Y3")}, "c4e26d901c8e"},
|
||||
{"VGATHERDPS evex", "VGATHERDPS", []Operand{vsib("SI", "X1", 4), vreg(t, "K2"), vreg(t, "X3")}, "62f27d0a921c8e"},
|
||||
{"VPGATHERQD evex", "VPGATHERQD", []Operand{vsib("SI", "Z1", 8), vreg(t, "K2"), vreg(t, "Y3")}, "62f27d4a911cce"},
|
||||
// Scatter (EVEX only: source, K mask, VSIB).
|
||||
{"VSCATTERDPS", "VSCATTERDPS", []Operand{vreg(t, "X3"), vreg(t, "K1"), vsib("SI", "X1", 4)}, "62f27d09a21c8e"},
|
||||
{"VSCATTERQPD", "VSCATTERQPD", []Operand{vreg(t, "Z3"), vreg(t, "K1"), vsib("SI", "Z1", 8)}, "62f2fd49a31cce"},
|
||||
// The remaining conversions.
|
||||
{"VCVTDQ2PS", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"},
|
||||
{"VCVTQQ2PS", "VCVTQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc485bd1"},
|
||||
{"VCVTPD2QQ", "VCVTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487bd1"},
|
||||
{"VCVTPS2QQ", "VCVTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487bd1"},
|
||||
{"VCVTUDQ2PD", "VCVTUDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "62f17e287ad1"},
|
||||
{"VCVTPH2PS", "VCVTPH2PS", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f27d4813d1"},
|
||||
{"VCVTPS2PH", "VCVTPS2PH", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "X2")}, "c4e37d1dca04"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
want := c.mnem
|
||||
got := inst.Op.String()
|
||||
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s", c.name, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexGprGroundTruth covers the scalar conversions between vector and
|
||||
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
|
||||
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
|
||||
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
|
||||
// for byte against the Go assembler, including memory sources and extended
|
||||
// GPRs.
|
||||
func TestEvexGprGroundTruth(t *testing.T) {
|
||||
mem := func(b Reg) Operand { return Ptr(b, 0, 8) }
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"VCVTSD2SI", "VCVTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2dc1"},
|
||||
{"VCVTSD2SIQ", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2dc1"},
|
||||
{"VCVTSS2SI", "VCVTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2dc1"},
|
||||
{"VCVTSS2SIQ", "VCVTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2dc1"},
|
||||
{"VCVTTSD2SI", "VCVTTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2cc1"},
|
||||
{"VCVTTSD2SIQ", "VCVTTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2cc1"},
|
||||
{"VCVTTSS2SI", "VCVTTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2cc1"},
|
||||
{"VCVTTSS2SIQ", "VCVTTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2cc1"},
|
||||
{"VCVTSD2USIL", "VCVTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0879c1"},
|
||||
{"VCVTSD2USIQ", "VCVTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0879c1"},
|
||||
{"VCVTSS2USIL", "VCVTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0879c1"},
|
||||
{"VCVTSS2USIQ", "VCVTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0879c1"},
|
||||
{"VCVTTSD2USIL", "VCVTTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0878c1"},
|
||||
{"VCVTTSD2USIQ", "VCVTTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0878c1"},
|
||||
{"VCVTTSS2USIL", "VCVTTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0878c1"},
|
||||
{"VCVTTSS2USIQ", "VCVTTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0878c1"},
|
||||
{"VCVTSI2SDL", "VCVTSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f32ad0"},
|
||||
{"VCVTSI2SDQ", "VCVTSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32ad0"},
|
||||
{"VCVTSI2SSL", "VCVTSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f22ad0"},
|
||||
{"VCVTSI2SSQ", "VCVTSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f22ad0"},
|
||||
{"VCVTUSI2SDL", "VCVTUSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f177087bd0"},
|
||||
{"VCVTUSI2SDQ", "VCVTUSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f7087bd0"},
|
||||
{"VCVTUSI2SSL", "VCVTUSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f176087bd0"},
|
||||
{"VCVTUSI2SSQ", "VCVTUSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f6087bd0"},
|
||||
{"VCVTSD2SI mem", "VCVTSD2SI", []Operand{mem(AX), BX}, "c5fb2d18"},
|
||||
{"VCVTSI2SDQ mem", "VCVTSI2SDQ", []Operand{mem(BX), vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32a13"},
|
||||
{"VCVTSD2SIQ hi gpr", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), vreg(t, "R9")}, "c461fb2dc9"},
|
||||
{"VCVTSI2SDQ hi gpr", "VCVTSI2SDQ", []Operand{vreg(t, "R10"), vreg(t, "X1"), vreg(t, "X2")}, "c4c1f32ad2"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
// The decoder does not distinguish the Plan 9 SIQ spelling (the
|
||||
// 64-bit GPR destination) from the base name; the W bit carries it.
|
||||
want := c.mnem
|
||||
got := inst.Op.String()
|
||||
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s", c.name, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexConversionGroundTruth covers the unsigned and truncating VCVT*
|
||||
// conversions, the remaining sign/zero-extending moves, the signed/unsigned
|
||||
// narrowing stores and the mask/vector conversions, byte for byte against
|
||||
// the Go assembler.
|
||||
func TestEvexConversionGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// Unsigned and truncating conversions.
|
||||
{"VCVTPD2PS", "VCVTPD2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fd485ad1"},
|
||||
{"VCVTPD2PSX", "VCVTPD2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f95ad1"},
|
||||
{"VCVTPD2PSY", "VCVTPD2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5fd5ad1"},
|
||||
{"VCVTPD2UDQ", "VCVTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4879d1"},
|
||||
{"VCVTPD2UDQX", "VCVTPD2UDQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc0879d1"},
|
||||
{"VCVTTPD2UDQ", "VCVTTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4878d1"},
|
||||
{"VCVTTPD2UDQY", "VCVTTPD2UDQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc2878d1"},
|
||||
{"VCVTTPD2UQQ", "VCVTTPD2UQQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd4878d1"},
|
||||
{"VCVTPS2UDQ", "VCVTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4879d1"},
|
||||
{"VCVTTPS2UDQ", "VCVTTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4878d1"},
|
||||
{"VCVTPS2UQQ", "VCVTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4879d1"},
|
||||
{"VCVTTPS2UQQ", "VCVTTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4878d1"},
|
||||
{"VCVTTPD2QQ", "VCVTTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487ad1"},
|
||||
{"VCVTTPS2QQ", "VCVTTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487ad1"},
|
||||
{"VCVTUQQ2PD", "VCVTUQQ2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487ad1"},
|
||||
{"VCVTUQQ2PS", "VCVTUQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff487ad1"},
|
||||
{"VCVTUQQ2PSX", "VCVTUQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1ff087ad1"},
|
||||
{"VCVTQQ2PSX", "VCVTQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc085bd1"},
|
||||
{"VCVTQQ2PSY", "VCVTQQ2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc285bd1"},
|
||||
// The remaining sign/zero-extending moves.
|
||||
{"VPMOVSXBD", "VPMOVSXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d21d1"},
|
||||
{"VPMOVSXBQ evex", "VPMOVSXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4822d1"},
|
||||
{"VPMOVSXWQ", "VPMOVSXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d24d1"},
|
||||
{"VPMOVSXWD", "VPMOVSXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d23d1"},
|
||||
{"VPMOVZXBD", "VPMOVZXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d31d1"},
|
||||
{"VPMOVZXBQ evex", "VPMOVZXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4832d1"},
|
||||
{"VPMOVZXWD", "VPMOVZXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d33d1"},
|
||||
{"VPMOVZXWQ", "VPMOVZXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d34d1"},
|
||||
// Signed narrowing stores.
|
||||
{"VPMOVSDB", "VPMOVSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4821ca"},
|
||||
{"VPMOVSDW", "VPMOVSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4823ca"},
|
||||
{"VPMOVSQB", "VPMOVSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4822ca"},
|
||||
{"VPMOVSQD", "VPMOVSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4825ca"},
|
||||
{"VPMOVSQW", "VPMOVSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4824ca"},
|
||||
{"VPMOVSWB", "VPMOVSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4820ca"},
|
||||
// Unsigned narrowing stores.
|
||||
{"VPMOVUSDB", "VPMOVUSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4811ca"},
|
||||
{"VPMOVUSDW", "VPMOVUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4813ca"},
|
||||
{"VPMOVUSQB", "VPMOVUSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4812ca"},
|
||||
{"VPMOVUSQD", "VPMOVUSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4815ca"},
|
||||
{"VPMOVUSQW", "VPMOVUSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4814ca"},
|
||||
{"VPMOVUSWB", "VPMOVUSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4810ca"},
|
||||
{"VPMOVDB", "VPMOVDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4831ca"},
|
||||
{"VPMOVQW", "VPMOVQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4834ca"},
|
||||
// Mask/vector conversions (the K register is an operand, not a
|
||||
// mask).
|
||||
{"VPMOVM2B", "VPMOVM2B", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0828d1"},
|
||||
{"VPMOVM2W", "VPMOVM2W", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f2fe0828d1"},
|
||||
{"VPMOVM2D", "VPMOVM2D", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0838d1"},
|
||||
{"VPMOVM2Q", "VPMOVM2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe4838d1"},
|
||||
{"VPMOVB2M", "VPMOVB2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f27e0829d1"},
|
||||
{"VPMOVW2M", "VPMOVW2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f2fe0829d1"},
|
||||
{"VPMOVD2M", "VPMOVD2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f27e4839d1"},
|
||||
{"VPMOVQ2M", "VPMOVQ2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f2fe4839d1"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
want := c.mnem
|
||||
got := inst.Op.String()
|
||||
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s", c.name, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexErrors checks the EVEX-specific error paths.
|
||||
func TestEvexErrors(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"NDS arity", "VPXORD", []Operand{vreg(t, "Z0"), vreg(t, "Z1")}},
|
||||
{"KMOVW arity", "KMOVW", []Operand{vreg(t, "K1")}},
|
||||
{"KMOVW no K", "KMOVW", []Operand{AX, CX}},
|
||||
{"VMOVUPD Z gpr", "VMOVUPD", []Operand{AX, vreg(t, "Z1")}},
|
||||
{"broadcast src", "VPBROADCASTD", []Operand{Imm(1), vreg(t, "Z1")}},
|
||||
{"VPMOVDW src", "VPMOVDW", []Operand{AX, vreg(t, "Y0")}},
|
||||
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
|
||||
// VEX-only mnemonics reject registers only EVEX can encode.
|
||||
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", c.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
|
||||
// kernel — all functions plus the file-global idx16 constant — and checks
|
||||
// that the static-symbol load resolves to the right bytes in the image.
|
||||
// Skipped when the sibling repository is not checked out.
|
||||
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
|
||||
path := "../../go-libraries/go-flac/avx512_amd64.s"
|
||||
if _, err := os.Stat(path); err != nil {
|
||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
||||
}
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f, errs := parser.Parse(path, string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
if len(img.Funcs) != 10 {
|
||||
t.Errorf("functions = %d, want 10", len(img.Funcs))
|
||||
}
|
||||
|
||||
// idx16 as the DATA directives define it: dwords 1..16.
|
||||
idx := make([]byte, 0, 64)
|
||||
for i := 1; i <= 16; i++ {
|
||||
idx = append(idx, byte(i), 0, 0, 0)
|
||||
}
|
||||
image := img.Bytes()
|
||||
base := img.Symbols["idx16"]
|
||||
if base == 0 {
|
||||
t.Fatal("idx16 not laid out")
|
||||
}
|
||||
if got := image[base : base+64]; hexCompact(got) != hexCompact(idx) {
|
||||
t.Errorf("idx16 contents %x, want %x", got, idx)
|
||||
}
|
||||
|
||||
// The VMOVDQU32 idx16(SB), Z13 load (62 71 7e 48 6f 2d + rel32) must
|
||||
// resolve to idx16 within the image.
|
||||
loads := 0
|
||||
for _, fn := range img.Funcs {
|
||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
pat := []byte{0x62, 0x71, 0x7e, 0x48, 0x6f, 0x2d}
|
||||
for pos := 0; ; {
|
||||
i := indexOf(code[pos:], pat)
|
||||
if i < 0 {
|
||||
break
|
||||
}
|
||||
i += pos
|
||||
rel := int32(uint32(code[i+6]) | uint32(code[i+7])<<8 | uint32(code[i+8])<<16 | uint32(code[i+9])<<24)
|
||||
target := fn.Offset + i + 10 + int(rel)
|
||||
if target != base {
|
||||
t.Errorf("%s: idx16 load at +%d targets 0x%x, want 0x%x", fn.Name, i, target, base)
|
||||
}
|
||||
loads++
|
||||
pos = i + 10
|
||||
}
|
||||
}
|
||||
if loads != 1 {
|
||||
t.Errorf("idx16 loads found = %d, want 1", loads)
|
||||
}
|
||||
}
|
||||
|
||||
// hexCompact renders bytes as a lowercase hex string without separators.
|
||||
func hexCompact(b []byte) string {
|
||||
const hexdig = "0123456789abcdef"
|
||||
out := make([]byte, len(b)*2)
|
||||
for i, c := range b {
|
||||
out[i*2] = hexdig[c>>4]
|
||||
out[i*2+1] = hexdig[c&0xf]
|
||||
}
|
||||
return string(out)
|
||||
}
|
||||
|
||||
// indexOf returns the index of the first occurrence of pat in b, or -1.
|
||||
func indexOf(b, pat []byte) int {
|
||||
for i := 0; i+len(pat) <= len(b); i++ {
|
||||
j := 0
|
||||
for j < len(pat) && b[i+j] == pat[j] {
|
||||
j++
|
||||
}
|
||||
if j == len(pat) {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
+453
@@ -0,0 +1,453 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"sync"
|
||||
)
|
||||
|
||||
// This file emits GOOBJ — the Go toolchain's object format, which cmd/link
|
||||
// consumes directly — so gasm-assembled functions drop into a go build
|
||||
// without the Go assembler. The layout follows cmd/internal/goobj: a
|
||||
// toolchain preamble ("go object ...\n!\n"), the go120ld header with its
|
||||
// block offsets, a string table, symbol definitions, the relocation /
|
||||
// aux / data index arrays, and the three blocks themselves.
|
||||
//
|
||||
// The object carries what the linker requires of an assembly object: the
|
||||
// functions (non-package symbols, as cmd/asm emits them), the GLOBL data,
|
||||
// one FuncInfo per function, and the pc-value tables (pcsp, pcfile,
|
||||
// pcline, pcinline). DWARF and the implicit funcdata symbols are omitted;
|
||||
// the linker fills their defaults.
|
||||
|
||||
// GOOBJ block indices (cmd/internal/goobj).
|
||||
const (
|
||||
blkAutolib = iota
|
||||
blkPkgIdx
|
||||
blkFile
|
||||
blkSymdef
|
||||
blkHashed64def
|
||||
blkHasheddef
|
||||
blkNonpkgdef
|
||||
blkNonpkgref
|
||||
blkRefFlags
|
||||
blkHash64
|
||||
blkHash
|
||||
blkRelocIdx
|
||||
blkAuxIdx
|
||||
blkDataIdx
|
||||
blkReloc
|
||||
blkAux
|
||||
blkData
|
||||
blkRefName
|
||||
blkEnd
|
||||
)
|
||||
|
||||
// Symbol kinds used by assembly objects (cmd/internal/objabi).
|
||||
const (
|
||||
kindSTEXT = 1
|
||||
kindSRODATA = 3
|
||||
kindSDATA = 7
|
||||
)
|
||||
|
||||
// Symbol flags (cmd/internal/goobj).
|
||||
const (
|
||||
symFlagDupok = 0x01
|
||||
symFlagNoSplit = 0x10
|
||||
symFlag2Link = 0x10 // asm objects flag every named symbol as linkname
|
||||
symABIStatic = 0xffff
|
||||
)
|
||||
|
||||
// Aux entry types (cmd/internal/goobj).
|
||||
const (
|
||||
auxFuncInfo = 1
|
||||
auxPcsp = 7
|
||||
auxPcfile = 8
|
||||
auxPcline = 9
|
||||
auxPcinline = 10
|
||||
)
|
||||
|
||||
// FuncInfo flags (internal/abi).
|
||||
const (
|
||||
funcFlagSPWrite = 2
|
||||
funcFlagAsm = 4
|
||||
)
|
||||
|
||||
// Relocation types (cmd/internal/objabi).
|
||||
const relocPCRel = 14
|
||||
|
||||
// Special package indices for symbol references.
|
||||
const (
|
||||
pkgIdxNone = 0x7fffffff
|
||||
pkgIdxSelf = 0x7ffffffb
|
||||
)
|
||||
|
||||
const goobjMagic = "\x00go120ld"
|
||||
|
||||
// goSym is one symbol definition under construction.
|
||||
type goSym struct {
|
||||
name string
|
||||
abi uint16
|
||||
typ uint8
|
||||
flag uint8
|
||||
flag2 uint8
|
||||
size uint32
|
||||
align uint32
|
||||
}
|
||||
|
||||
func (s goSym) append(b []byte, strOff map[string]uint32) []byte {
|
||||
b = binary.LittleEndian.AppendUint32(b, uint32(len(s.name)))
|
||||
b = binary.LittleEndian.AppendUint32(b, strOff[s.name])
|
||||
b = binary.LittleEndian.AppendUint16(b, s.abi)
|
||||
b = append(b, s.typ, s.flag, s.flag2)
|
||||
b = binary.LittleEndian.AppendUint32(b, s.size)
|
||||
return binary.LittleEndian.AppendUint32(b, s.align)
|
||||
}
|
||||
|
||||
// GOObject returns the image as a GOOBJ object file for the given package
|
||||
// path (the linker qualifies the exported symbols with it, the way cmd/asm
|
||||
// does with its -p flag). srcPath names the source file recorded in the
|
||||
// object's file table and line tables. The toolchain's object preamble is
|
||||
// captured from the installed go tool asm, so the output links with the
|
||||
// toolchain it was produced on — exactly like a real assembly object.
|
||||
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
|
||||
if pkgPath == "" {
|
||||
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
|
||||
}
|
||||
pre, err := toolchainObjectPreamble()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// The symbol tables. Package definitions: the GLOBL symbols, then one
|
||||
// anonymous FuncInfo symbol per function. Non-package definitions: the
|
||||
// pc-value tables and the functions themselves, as cmd/asm lays them
|
||||
// out. defIdx maps a GLOBL's bare name to its definition index for the
|
||||
// relocations; fnNpIdx maps a function to its non-package index.
|
||||
var defs []goSym
|
||||
var defData [][]byte
|
||||
defIdx := map[string]int{}
|
||||
for _, d := range img.DataSyms {
|
||||
name := d.Name
|
||||
if !d.Static {
|
||||
name = pkgPath + "." + name
|
||||
}
|
||||
typ := uint8(kindSDATA)
|
||||
if d.Rodata {
|
||||
typ = kindSRODATA
|
||||
}
|
||||
flag := uint8(0)
|
||||
if d.Dupok {
|
||||
flag = symFlagDupok
|
||||
}
|
||||
abi := uint16(0)
|
||||
if d.Static {
|
||||
abi = symABIStatic
|
||||
}
|
||||
defIdx[d.Name] = len(defs)
|
||||
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)})
|
||||
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
|
||||
}
|
||||
fnFiIdx := make([]int, len(img.Funcs))
|
||||
for i := range img.Funcs {
|
||||
data := marshalFuncInfo(img.Funcs[i])
|
||||
fnFiIdx[i] = len(defs)
|
||||
defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))})
|
||||
defData = append(defData, data)
|
||||
}
|
||||
|
||||
type npSym struct {
|
||||
sym goSym
|
||||
data []byte
|
||||
}
|
||||
var nps []npSym
|
||||
type pcRefs struct{ sp, file, line, inl int }
|
||||
pcIdx := make([]pcRefs, len(img.Funcs))
|
||||
fnNpIdx := make([]int, len(img.Funcs))
|
||||
for i, fn := range img.Funcs {
|
||||
tables := []struct {
|
||||
data []byte
|
||||
dst *int
|
||||
}{
|
||||
{pcspTable(fn), &pcIdx[i].sp},
|
||||
{pcValueFlat(0, fn.Size), &pcIdx[i].file},
|
||||
{pcValueFlat(int32(fn.Line), fn.Size), &pcIdx[i].line},
|
||||
{pcValueFlat(-1, fn.Size), &pcIdx[i].inl},
|
||||
}
|
||||
for _, t := range tables {
|
||||
*t.dst = len(nps)
|
||||
nps = append(nps, npSym{
|
||||
sym: goSym{typ: kindSRODATA, size: uint32(len(t.data)), align: 1},
|
||||
data: t.data,
|
||||
})
|
||||
}
|
||||
name := fn.Name
|
||||
abi := uint16(0)
|
||||
if fn.Static {
|
||||
abi = symABIStatic
|
||||
} else {
|
||||
name = pkgPath + "." + name
|
||||
}
|
||||
flag := uint8(0)
|
||||
if fn.NoSplit {
|
||||
flag |= symFlagNoSplit
|
||||
}
|
||||
fnNpIdx[i] = len(nps)
|
||||
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
|
||||
for _, r := range fn.Relocs {
|
||||
// The linker writes the resolved displacement into the field;
|
||||
// leave it zero, as cmd/asm's object does.
|
||||
if r.Off >= 0 && r.Off+4 <= len(code) {
|
||||
code[r.Off], code[r.Off+1], code[r.Off+2], code[r.Off+3] = 0, 0, 0, 0
|
||||
}
|
||||
}
|
||||
nps = append(nps, npSym{
|
||||
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
|
||||
data: code,
|
||||
})
|
||||
}
|
||||
|
||||
// Relocations, per defined symbol in definition order (package defs,
|
||||
// then non-package defs). Only file-local GLOBL references resolve;
|
||||
// external symbols need the import machinery of a later increment.
|
||||
nsyms := len(defs) + len(nps)
|
||||
symRelocs := make([][]byte, nsyms) // flat 23-byte records
|
||||
for i, fn := range img.Funcs {
|
||||
si := len(defs) + fnNpIdx[i]
|
||||
for _, r := range fn.Relocs {
|
||||
if r.External {
|
||||
return nil, fmt.Errorf("GOOBJ emission: external symbol %q is not supported yet", r.Name)
|
||||
}
|
||||
di, ok := defIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
|
||||
}
|
||||
var rec [23]byte
|
||||
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
|
||||
rec[4] = 4 // field width
|
||||
binary.LittleEndian.PutUint16(rec[5:], relocPCRel)
|
||||
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
|
||||
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
|
||||
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
|
||||
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
// Aux entries per function: FuncInfo, then the four pc tables.
|
||||
// References into the non-package table use pkgIdxNone.
|
||||
symAux := make([][]byte, nsyms)
|
||||
for i := range img.Funcs {
|
||||
si := len(defs) + fnNpIdx[i]
|
||||
aux := func(typ uint8, pkg, idx uint32) {
|
||||
var rec [9]byte
|
||||
rec[0] = typ
|
||||
binary.LittleEndian.PutUint32(rec[1:], pkg)
|
||||
binary.LittleEndian.PutUint32(rec[5:], idx)
|
||||
symAux[si] = append(symAux[si], rec[:]...)
|
||||
}
|
||||
aux(auxFuncInfo, pkgIdxSelf, uint32(fnFiIdx[i]))
|
||||
aux(auxPcsp, pkgIdxNone, uint32(len(defs)+pcIdx[i].sp))
|
||||
aux(auxPcfile, pkgIdxNone, uint32(len(defs)+pcIdx[i].file))
|
||||
aux(auxPcline, pkgIdxNone, uint32(len(defs)+pcIdx[i].line))
|
||||
aux(auxPcinline, pkgIdxNone, uint32(len(defs)+pcIdx[i].inl))
|
||||
}
|
||||
|
||||
// The string table. Absolute offsets: it starts right after the
|
||||
// 96-byte header (magic, fingerprint, flags, the 19 block offsets).
|
||||
const headerSize = 8 + 8 + 4 + 4*(blkEnd+1)
|
||||
strTab := []byte{}
|
||||
strOff := map[string]uint32{}
|
||||
addStr := func(s string) {
|
||||
if _, ok := strOff[s]; ok {
|
||||
return
|
||||
}
|
||||
strOff[s] = uint32(headerSize + len(strTab))
|
||||
strTab = append(strTab, s...)
|
||||
}
|
||||
addStr("")
|
||||
addStr(srcPath)
|
||||
for _, s := range defs {
|
||||
addStr(s.name)
|
||||
}
|
||||
for _, s := range nps {
|
||||
addStr(s.sym.name)
|
||||
}
|
||||
stringRef := func(b []byte, s string) []byte {
|
||||
b = binary.LittleEndian.AppendUint32(b, uint32(len(s)))
|
||||
return binary.LittleEndian.AppendUint32(b, strOff[s])
|
||||
}
|
||||
|
||||
// Serialise the block bodies.
|
||||
var symdefBlk, npdefBlk []byte
|
||||
for _, s := range defs {
|
||||
symdefBlk = s.append(symdefBlk, strOff)
|
||||
}
|
||||
for _, s := range nps {
|
||||
npdefBlk = s.sym.append(npdefBlk, strOff)
|
||||
}
|
||||
pkgIdxBlk := stringRef(nil, "") // index 0: the dummy invalid package
|
||||
fileBlk := stringRef(nil, srcPath)
|
||||
|
||||
var relocBlk, auxBlk, dataBlk []byte
|
||||
relocIdxBlk := make([]byte, 0, 4*(nsyms+1))
|
||||
auxIdxBlk := make([]byte, 0, 4*(nsyms+1))
|
||||
dataIdxBlk := make([]byte, 0, 4*(nsyms+1))
|
||||
var nr, na, nd uint32
|
||||
for si := 0; si < nsyms; si++ {
|
||||
relocIdxBlk = binary.LittleEndian.AppendUint32(relocIdxBlk, nr)
|
||||
auxIdxBlk = binary.LittleEndian.AppendUint32(auxIdxBlk, na)
|
||||
dataIdxBlk = binary.LittleEndian.AppendUint32(dataIdxBlk, nd)
|
||||
relocBlk = append(relocBlk, symRelocs[si]...)
|
||||
auxBlk = append(auxBlk, symAux[si]...)
|
||||
var d []byte
|
||||
if si < len(defData) {
|
||||
d = defData[si]
|
||||
} else {
|
||||
d = nps[si-len(defData)].data
|
||||
}
|
||||
dataBlk = append(dataBlk, d...)
|
||||
nr += uint32(len(symRelocs[si])) / 23
|
||||
na += uint32(len(symAux[si])) / 9
|
||||
nd += uint32(len(d))
|
||||
}
|
||||
relocIdxBlk = binary.LittleEndian.AppendUint32(relocIdxBlk, nr)
|
||||
auxIdxBlk = binary.LittleEndian.AppendUint32(auxIdxBlk, na)
|
||||
dataIdxBlk = binary.LittleEndian.AppendUint32(dataIdxBlk, nd)
|
||||
|
||||
blocks := [blkEnd][]byte{
|
||||
blkPkgIdx: pkgIdxBlk,
|
||||
blkFile: fileBlk,
|
||||
blkSymdef: symdefBlk,
|
||||
blkNonpkgdef: npdefBlk,
|
||||
blkRelocIdx: relocIdxBlk,
|
||||
blkAuxIdx: auxIdxBlk,
|
||||
blkDataIdx: dataIdxBlk,
|
||||
blkReloc: relocBlk,
|
||||
blkAux: auxBlk,
|
||||
blkData: dataBlk,
|
||||
}
|
||||
|
||||
// Assemble the payload: header (offsets filled once known), string
|
||||
// table, blocks in order.
|
||||
payload := make([]byte, headerSize)
|
||||
copy(payload, goobjMagic)
|
||||
// The fingerprint stays zero, as cmd/asm leaves it.
|
||||
binary.LittleEndian.PutUint32(payload[16:], 4) // ObjFlagFromAssembly
|
||||
off := uint32(headerSize + len(strTab))
|
||||
for i := 0; i < blkEnd; i++ {
|
||||
binary.LittleEndian.PutUint32(payload[20+4*i:], off)
|
||||
off += uint32(len(blocks[i]))
|
||||
}
|
||||
binary.LittleEndian.PutUint32(payload[20+4*blkEnd:], off)
|
||||
payload = append(payload, strTab...)
|
||||
for _, blk := range blocks {
|
||||
payload = append(payload, blk...)
|
||||
}
|
||||
|
||||
out := make([]byte, 0, len(pre)+len(payload))
|
||||
out = append(out, pre...)
|
||||
return append(out, payload...), nil
|
||||
}
|
||||
|
||||
// marshalFuncInfo serialises a function's goobj.FuncInfo: sizes, flags,
|
||||
// start line, the one-element file table and an empty inline tree.
|
||||
func marshalFuncInfo(fn FuncLayout) []byte {
|
||||
flag := uint8(funcFlagAsm)
|
||||
if fn.SPWrite {
|
||||
flag |= funcFlagSPWrite
|
||||
}
|
||||
b := make([]byte, 0, 28)
|
||||
b = binary.LittleEndian.AppendUint32(b, uint32(fn.Args))
|
||||
b = binary.LittleEndian.AppendUint32(b, uint32(fn.Frame))
|
||||
b = append(b, 0, flag, 0, 0) // FuncID normal, flags, padding
|
||||
b = binary.LittleEndian.AppendUint32(b, uint32(int32(fn.Line)))
|
||||
b = binary.LittleEndian.AppendUint32(b, 1) // one file
|
||||
b = binary.LittleEndian.AppendUint32(b, 0) // file index 0
|
||||
b = binary.LittleEndian.AppendUint32(b, 0) // no inline tree
|
||||
return b
|
||||
}
|
||||
|
||||
// pcValueFlat encodes a pc-value table holding v over the whole function.
|
||||
func pcValueFlat(v int32, size int) []byte {
|
||||
// The table is delta-encoded from an implicit value of -1: a varint
|
||||
// value delta, an unsigned pc delta to the end, and a zero terminator.
|
||||
out := binary.AppendVarint(nil, int64(v)+1)
|
||||
out = binary.AppendUvarint(out, uint64(size))
|
||||
return append(out, 0)
|
||||
}
|
||||
|
||||
// pcspTable encodes the stack-adjustment table: the SP delta in effect at
|
||||
// every pc, from the function's prologue and epilogue boundaries.
|
||||
func pcspTable(fn FuncLayout) []byte {
|
||||
if len(fn.Spadj) == 0 {
|
||||
return pcValueFlat(0, fn.Size)
|
||||
}
|
||||
pts := make([]SpadjStep, 0, len(fn.Spadj)+1)
|
||||
pts = append(pts, SpadjStep{PC: 0, Value: 0})
|
||||
pts = append(pts, fn.Spadj...)
|
||||
out := binary.AppendVarint(nil, int64(pts[0].Value)+1)
|
||||
cur, old := pts[0].PC, pts[0].Value
|
||||
for _, p := range pts[1:] {
|
||||
out = binary.AppendUvarint(out, uint64(p.PC-cur))
|
||||
out = binary.AppendVarint(out, int64(p.Value-old))
|
||||
cur, old = p.PC, p.Value
|
||||
}
|
||||
out = binary.AppendUvarint(out, uint64(fn.Size-cur))
|
||||
return append(out, 0)
|
||||
}
|
||||
|
||||
// toolchainObjectPreamble returns the "go object ...\n!\n" header the
|
||||
// installed go tool asm writes, captured by assembling a one-instruction
|
||||
// probe. The linker compares this string verbatim against its own, so it
|
||||
// must come from the toolchain itself, not be reconstructed.
|
||||
var (
|
||||
preambleOnce sync.Once
|
||||
preamble []byte
|
||||
preambleErr error
|
||||
)
|
||||
|
||||
func toolchainObjectPreamble() ([]byte, error) {
|
||||
preambleOnce.Do(func() {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
preambleErr = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
|
||||
return
|
||||
}
|
||||
dir, err := os.MkdirTemp("", "gasm-preamble")
|
||||
if err != nil {
|
||||
preambleErr = err
|
||||
return
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
src := filepath.Join(dir, "probe_amd64.s")
|
||||
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
|
||||
preambleErr = err
|
||||
return
|
||||
}
|
||||
obj := filepath.Join(dir, "probe.o")
|
||||
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
|
||||
cmd.Env = append(os.Environ(), "GOARCH=amd64")
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
preambleErr = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
|
||||
return
|
||||
}
|
||||
data, err := os.ReadFile(obj)
|
||||
if err != nil {
|
||||
preambleErr = err
|
||||
return
|
||||
}
|
||||
i := bytes.Index(data, []byte("\n!\n"))
|
||||
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
|
||||
preambleErr = fmt.Errorf("unrecognised assembler object layout")
|
||||
return
|
||||
}
|
||||
preamble = data[:i+3]
|
||||
})
|
||||
return preamble, preambleErr
|
||||
}
|
||||
@@ -0,0 +1,477 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// goobjView is a minimal parsed view of a GOOBJ payload, enough to check
|
||||
// the emitter's output block by block.
|
||||
type goobjView struct {
|
||||
t *testing.T
|
||||
b []byte
|
||||
offs [blkEnd + 1]uint32
|
||||
strOff uint32
|
||||
}
|
||||
|
||||
func openGoobj(t *testing.T, data []byte) *goobjView {
|
||||
t.Helper()
|
||||
i := bytes.Index(data, []byte(goobjMagic))
|
||||
if i < 0 {
|
||||
t.Fatal("no GOOBJ magic in output")
|
||||
}
|
||||
v := &goobjView{t: t, b: data[i:], strOff: uint32(i + 96)}
|
||||
for j := 0; j <= blkEnd; j++ {
|
||||
v.offs[j] = binary.LittleEndian.Uint32(v.b[20+4*j:])
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
func (v *goobjView) blk(i int) []byte { return v.b[v.offs[i]:v.offs[i+1]] }
|
||||
|
||||
func (v *goobjView) str(off, ln uint32) string {
|
||||
return string(v.b[off : off+ln])
|
||||
}
|
||||
|
||||
type goobjSymView struct {
|
||||
name string
|
||||
abi uint16
|
||||
typ uint8
|
||||
flag uint8
|
||||
flag2 uint8
|
||||
size uint32
|
||||
align uint32
|
||||
}
|
||||
|
||||
func (v *goobjView) syms(i int) []goobjSymView {
|
||||
var out []goobjSymView
|
||||
for x := v.blk(i); len(x) >= 21; x = x[21:] {
|
||||
le := binary.LittleEndian
|
||||
out = append(out, goobjSymView{
|
||||
name: v.str(le.Uint32(x[4:]), le.Uint32(x[0:])),
|
||||
abi: le.Uint16(x[8:]),
|
||||
typ: x[10],
|
||||
flag: x[11],
|
||||
flag2: x[12],
|
||||
size: le.Uint32(x[13:]),
|
||||
align: le.Uint32(x[17:]),
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// TestGOObjectStructure checks the emitted object's blocks against the
|
||||
// ground truth captured from go tool asm: the symbol tables, the FuncInfo
|
||||
// contents, the pc-value tables, the relocation and the aux wiring.
|
||||
func TestGOObjectStructure(t *testing.T) {
|
||||
f, errs := parser.Parse("t_amd64.s", `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·addq(SB), NOSPLIT, $0-24
|
||||
MOVQ a+0(FP), AX
|
||||
MOVQ b+8(FP), CX
|
||||
ADDQ CX, AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
|
||||
TEXT ·loadmask(SB), NOSPLIT, $0-8
|
||||
VMOVDQU mask<>(SB), X0
|
||||
VPMOVMSKB X0, AX
|
||||
MOVQ AX, ret+0(FP)
|
||||
RET
|
||||
|
||||
GLOBL mask<>(SB), RODATA, $16
|
||||
DATA mask<>+0(SB)/8, $0x0807060504030201
|
||||
DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
obj, err := img.GOObject("testpkg", "t_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObject: %v", err)
|
||||
}
|
||||
v := openGoobj(t, obj)
|
||||
|
||||
if flags := binary.LittleEndian.Uint32(v.b[16:]); flags != 4 {
|
||||
t.Errorf("flags = %#x, want ObjFlagFromAssembly (4)", flags)
|
||||
}
|
||||
|
||||
// Package defs: the static GLOBL, then one anonymous FuncInfo per
|
||||
// function.
|
||||
defs := v.syms(blkSymdef)
|
||||
if len(defs) != 3 {
|
||||
t.Fatalf("symdefs = %d, want 3", len(defs))
|
||||
}
|
||||
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
|
||||
t.Errorf("mask symbol = %+v", defs[0])
|
||||
}
|
||||
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
|
||||
t.Errorf("funcinfo symbol = %+v", defs[1])
|
||||
}
|
||||
|
||||
// Non-package defs: four pc tables and the function, per function.
|
||||
nps := v.syms(blkNonpkgdef)
|
||||
if len(nps) != 10 {
|
||||
t.Fatalf("nonpkgdefs = %d, want 10", len(nps))
|
||||
}
|
||||
fn := nps[4]
|
||||
if fn.name != "testpkg.addq" || fn.typ != kindSTEXT || fn.flag != symFlagNoSplit || fn.size != 19 {
|
||||
t.Errorf("addq symbol = %+v", fn)
|
||||
}
|
||||
for i, s := range []int{0, 1, 2, 3, 5, 6, 7, 8} {
|
||||
if nps[s].typ != kindSRODATA || nps[s].align != 1 || nps[s].name != "" {
|
||||
t.Errorf("pc table %d = %+v", i, nps[s])
|
||||
}
|
||||
}
|
||||
|
||||
// FuncInfo: args 24, FuncFlag Asm, one file, no inline tree.
|
||||
le := binary.LittleEndian
|
||||
data := v.blk(blkData)
|
||||
fi := data[16:44]
|
||||
if le.Uint32(fi[0:]) != 24 || le.Uint32(fi[4:]) != 0 || fi[8] != 0 || fi[9] != funcFlagAsm ||
|
||||
le.Uint32(fi[16:]) != 1 || le.Uint32(fi[20:]) != 0 || le.Uint32(fi[24:]) != 0 {
|
||||
t.Errorf("funcinfo bytes %x", fi)
|
||||
}
|
||||
|
||||
// pcsp: a flat zero over the whole function (zero-frame NOSPLIT).
|
||||
if got := data[72:75]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
|
||||
t.Errorf("pcsp = %x, want 021300", got)
|
||||
}
|
||||
// pcinline: a flat -1.
|
||||
if got := data[81:84]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) {
|
||||
t.Errorf("pcinline = %x, want 001300", got)
|
||||
}
|
||||
|
||||
// The one relocation: R_PCREL, four bytes wide, against the GLOBL,
|
||||
// with the field in the function code left zero. The loadmask code's
|
||||
// offset comes from the data index (symbol 3 defs + 9 non-package).
|
||||
relocs := v.blk(blkReloc)
|
||||
if len(relocs) != 23 {
|
||||
t.Fatalf("relocs = %d bytes, want one 23-byte entry", len(relocs))
|
||||
}
|
||||
off := int32(le.Uint32(relocs[0:]))
|
||||
if off != 4 || relocs[4] != 4 || le.Uint16(relocs[5:]) != relocPCRel ||
|
||||
le.Uint64(relocs[7:]) != 0 || le.Uint32(relocs[15:]) != pkgIdxSelf || le.Uint32(relocs[19:]) != 0 {
|
||||
t.Errorf("reloc = %x", relocs)
|
||||
}
|
||||
didx := v.blk(blkDataIdx)
|
||||
lm := le.Uint32(didx[4*(3+9):])
|
||||
code := data[lm : lm+18]
|
||||
if !bytes.Equal(code[4:8], []byte{0, 0, 0, 0}) {
|
||||
t.Errorf("relocated field = %x, want zeroed", code[4:8])
|
||||
}
|
||||
|
||||
// Aux wiring: FuncInfo (package symbol), then the four pc tables
|
||||
// (non-package symbols).
|
||||
auxs := v.blk(blkAux)
|
||||
if len(auxs) != 2*5*9 {
|
||||
t.Fatalf("aux = %d bytes, want 10 entries", len(auxs))
|
||||
}
|
||||
wantAux := []struct {
|
||||
typ uint8
|
||||
pkg uint32
|
||||
idx uint32
|
||||
}{
|
||||
{auxFuncInfo, pkgIdxSelf, 1},
|
||||
{auxPcsp, pkgIdxNone, uint32(len(defs) + 0)},
|
||||
{auxPcfile, pkgIdxNone, uint32(len(defs) + 1)},
|
||||
{auxPcline, pkgIdxNone, uint32(len(defs) + 2)},
|
||||
{auxPcinline, pkgIdxNone, uint32(len(defs) + 3)},
|
||||
{auxFuncInfo, pkgIdxSelf, 2},
|
||||
{auxPcsp, pkgIdxNone, uint32(len(defs) + 5)},
|
||||
{auxPcfile, pkgIdxNone, uint32(len(defs) + 6)},
|
||||
{auxPcline, pkgIdxNone, uint32(len(defs) + 7)},
|
||||
{auxPcinline, pkgIdxNone, uint32(len(defs) + 8)},
|
||||
}
|
||||
for i, w := range wantAux {
|
||||
e := auxs[i*9:]
|
||||
if e[0] != w.typ || le.Uint32(e[1:]) != w.pkg || le.Uint32(e[5:]) != w.idx {
|
||||
t.Errorf("aux[%d] = {%d,%d,%d}, want {%d,%d,%d}", i, e[0], le.Uint32(e[1:]), le.Uint32(e[5:]), w.typ, w.pkg, w.idx)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// decodePCValues decodes a pc-value table into (pc, value) steps. The
|
||||
// table ends with a final unsigned pc delta covering the rest of the
|
||||
// function, followed by a zero byte that carries no value delta.
|
||||
func decodePCValues(b []byte) (pcs, vals []int64) {
|
||||
val, n := binary.Varint(b)
|
||||
b = b[n:]
|
||||
val-- // the first delta is against the implicit -1
|
||||
var pc int64
|
||||
pcs = append(pcs, pc)
|
||||
vals = append(vals, val)
|
||||
for {
|
||||
pcd, n := binary.Uvarint(b)
|
||||
b = b[n:]
|
||||
if pcd == 0 { // zero pc delta terminates the table
|
||||
break
|
||||
}
|
||||
pc += int64(pcd)
|
||||
if len(b) == 1 && b[0] == 0 { // final coverage, no value change
|
||||
break
|
||||
}
|
||||
vd, n := binary.Varint(b)
|
||||
b = b[n:]
|
||||
val += vd
|
||||
pcs = append(pcs, pc)
|
||||
vals = append(vals, val)
|
||||
}
|
||||
return pcs, vals
|
||||
}
|
||||
|
||||
// TestGOObjectPcspFrame checks the pcsp table of a frame-pointer function:
|
||||
// the prologue raises the stack delta to 8+frame, the RET's epilogue
|
||||
// restores it to zero.
|
||||
func TestGOObjectPcspFrame(t *testing.T) {
|
||||
f, errs := parser.Parse("frame_amd64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·framed(SB), NOSPLIT, $8-0
|
||||
MOVQ BP, AX
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
pcs, vals := decodePCValues(pcspTable(fn))
|
||||
// Prologue: PUSHQ BP (1 byte, +8), MOVQ SP, BP (3 bytes, no change),
|
||||
// SUBQ $8, SP (4 bytes, +16 in total); the RET's epilogue unwinds
|
||||
// ADDQ $8, SP (+8) then POPQ BP (0).
|
||||
wantPCs := []int64{0, 1, 8}
|
||||
wantVals := []int64{0, 8, 16}
|
||||
if len(pcs) < len(wantPCs) {
|
||||
t.Fatalf("pcsp pcs = %v vals = %v", pcs, vals)
|
||||
}
|
||||
for i := range wantPCs {
|
||||
if pcs[i] != wantPCs[i] || vals[i] != wantVals[i] {
|
||||
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d) — all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
|
||||
}
|
||||
}
|
||||
// The last two steps unwind the epilogue to zero.
|
||||
n := len(pcs)
|
||||
if vals[n-1] != 0 || vals[n-2] != 8 {
|
||||
t.Errorf("epilogue steps = %v %v, want …8, 0", pcs, vals)
|
||||
}
|
||||
// The table covers the whole function.
|
||||
if last := pcs[n-1]; last >= int64(fn.Size) {
|
||||
t.Errorf("last pc %d beyond function size %d", last, fn.Size)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectExternalRejected checks that a reference to a symbol no GLOBL
|
||||
// defines is reported: GOOBJ emission resolves only file-local symbols so
|
||||
// far.
|
||||
func TestGOObjectExternalRejected(t *testing.T) {
|
||||
f, errs := parser.Parse("ext_amd64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·useext(SB), NOSPLIT, $0-8
|
||||
MOVQ elsewhere(SB), AX
|
||||
MOVQ AX, ret+0(FP)
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
if _, err := img.GOObject("p", "ext_amd64.s"); err == nil || !strings.Contains(err.Error(), "external") {
|
||||
t.Errorf("error = %v, want an external-symbol error", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectLinkAndRun is the end-to-end check: assemble the test
|
||||
// functions to a GOOBJ, swap it into a go build in place of the toolchain's
|
||||
// assembly object, link, and run — the output must match the baseline
|
||||
// binary the Go assembler produced. Skipped when no Go toolchain is
|
||||
// available.
|
||||
func TestGOObjectLinkAndRun(t *testing.T) {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
t.Skip("no Go toolchain available")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
|
||||
const asmSrc = `
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·addq(SB), NOSPLIT, $0-24
|
||||
MOVQ a+0(FP), AX
|
||||
MOVQ b+8(FP), CX
|
||||
ADDQ CX, AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
|
||||
TEXT ·loadmask(SB), NOSPLIT, $0-8
|
||||
VMOVDQU mask<>(SB), X0
|
||||
VPMOVMSKB X0, AX
|
||||
MOVQ AX, ret+0(FP)
|
||||
RET
|
||||
|
||||
GLOBL mask<>(SB), RODATA, $16
|
||||
DATA mask<>+0(SB)/8, $0x0807060504030201
|
||||
DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
|
||||
`
|
||||
const mainSrc = `package main
|
||||
|
||||
func addq(a, b int64) int64
|
||||
func loadmask() int64
|
||||
|
||||
func main() {
|
||||
println(addq(41, 1))
|
||||
println(loadmask())
|
||||
}
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module goobjtest\n\ngo 1.26\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Baseline build with the toolchain's assembler; keep the work
|
||||
// directory and the commands the build used.
|
||||
cmd := exec.Command(goBin, "build", "-x", "-work", "-o", "app", ".")
|
||||
cmd.Dir = dir
|
||||
buildLog, err := cmd.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||
}
|
||||
var work string
|
||||
var asmObj, pkgArch, linkLine string
|
||||
for _, line := range strings.Split(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
case strings.Contains(line, "/asm ") && strings.Contains(line, "-o ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
|
||||
asmObj = fieldAfter(line, "-o")
|
||||
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
|
||||
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
|
||||
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
|
||||
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||
linkLine = line
|
||||
}
|
||||
}
|
||||
if work == "" || asmObj == "" || pkgArch == "" || linkLine == "" {
|
||||
t.Fatalf("could not locate the build steps:\n%s", buildLog)
|
||||
}
|
||||
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
|
||||
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
|
||||
|
||||
// The baseline's answer.
|
||||
baseOut, err := exec.Command(filepath.Join(dir, "app")).CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("run baseline: %v\n%s", err, baseOut)
|
||||
}
|
||||
|
||||
// Assemble the same source with gasm and swap the object in.
|
||||
pf, perrs := parser.Parse(filepath.Join(dir, "main_amd64.s"), asmSrc)
|
||||
if len(perrs) > 0 {
|
||||
t.Fatalf("parse: %v", perrs)
|
||||
}
|
||||
img, err := AssembleFile(pf)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
obj, err := img.GOObject("main", filepath.Join(dir, "main_amd64.s"))
|
||||
if err != nil {
|
||||
t.Fatalf("GOObject: %v", err)
|
||||
}
|
||||
if err := os.WriteFile(asmObj, obj, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Rebuild the package archive with our object in place of the
|
||||
// toolchain's (go tool pack has no replace-in-place that dedupes, so
|
||||
// extract, substitute and repack).
|
||||
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
|
||||
membersDir := filepath.Join(dir, "members")
|
||||
if err := os.MkdirAll(membersDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
extract.Dir = membersDir
|
||||
if out, err := extract.CombinedOutput(); err != nil {
|
||||
t.Fatalf("pack x: %v\n%s", err, out)
|
||||
}
|
||||
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
|
||||
listOut, err := listCmd.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("pack t: %v\n%s", err, listOut)
|
||||
}
|
||||
newArch := filepath.Join(dir, "pkg.a")
|
||||
args := []string{"tool", "pack", "c", newArch}
|
||||
seen := map[string]bool{}
|
||||
for _, m := range strings.Fields(string(listOut)) {
|
||||
if seen[m] {
|
||||
continue
|
||||
}
|
||||
seen[m] = true
|
||||
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
args = append(args, filepath.Join(membersDir, m))
|
||||
}
|
||||
pack := exec.Command(goBin, args...)
|
||||
pack.Dir = membersDir
|
||||
if out, err := pack.CombinedOutput(); err != nil {
|
||||
t.Fatalf("pack c: %v\n%s", err, out)
|
||||
}
|
||||
|
||||
// Link with our archive. The link line carries a GOROOT assignment
|
||||
// and $WORK placeholders; run it through the shell with the
|
||||
// GOEXPERIMENT the toolchain expects (the linker compares the object
|
||||
// header against its own, experiments included).
|
||||
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
|
||||
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
|
||||
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
|
||||
link := exec.Command("sh", "-c", linkLine)
|
||||
link.Dir = dir
|
||||
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)))
|
||||
if out, err := link.CombinedOutput(); err != nil {
|
||||
t.Fatalf("link with gasm object: %v\n%s", err, out)
|
||||
}
|
||||
got, err := exec.Command(filepath.Join(dir, "app2")).CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("run gasm-linked binary: %v\n%s", err, got)
|
||||
}
|
||||
if !bytes.Equal(got, baseOut) {
|
||||
t.Errorf("gasm-linked output %q, want baseline %q", got, baseOut)
|
||||
}
|
||||
}
|
||||
|
||||
// fieldAfter returns the whitespace-delimited field following the first
|
||||
// occurrence of flag in line.
|
||||
func fieldAfter(line, flag string) string {
|
||||
fields := strings.Fields(line)
|
||||
for i, f := range fields {
|
||||
if f == flag && i+1 < len(fields) {
|
||||
return fields[i+1]
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
+256
-6
@@ -52,9 +52,10 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
||||
switch src := src.(type) {
|
||||
case Reg:
|
||||
if dstIsReg {
|
||||
// MOV r, r/m: 0x8A/0x8B, reg=dst, rm=src.
|
||||
i := newInstr(size, []byte{movRR(size)})
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
// MOV r/m, r: 0x88/0x89, reg=src, rm=dst — the form the Go
|
||||
// assembler emits for register-to-register moves.
|
||||
i := newInstr(size, []byte{movRM(size)})
|
||||
if err := setRM(i, src, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
@@ -77,6 +78,17 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
||||
}
|
||||
return e.emit(i)
|
||||
|
||||
case sbMem:
|
||||
if !dstIsReg {
|
||||
return fmt.Errorf("MOV: two memory operands")
|
||||
}
|
||||
// MOV r, r/m: reg=dst, rm=src(static symbol).
|
||||
i := newInstr(size, []byte{movRR(size)})
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
|
||||
case Imm:
|
||||
if dstIsReg {
|
||||
// MOV r, imm: 0xB0+reg (8-bit) / 0xB8+reg (16/32/64, imm64 for Q).
|
||||
@@ -147,9 +159,37 @@ func (e *enc) encodeALU(op struct {
|
||||
return e.encodeALUImm(op.digit, src, int64(imm), size)
|
||||
}
|
||||
|
||||
// CMP records first − second without writing anywhere, so the first
|
||||
// operand must land as the minuend; every other ALU op writes its second
|
||||
// operand and follows the forms below.
|
||||
cmp := op.rr == 0x39
|
||||
dstReg, dstIsReg := dst.(Reg)
|
||||
srcReg, srcIsReg := src.(Reg)
|
||||
switch {
|
||||
case cmp && dstIsReg:
|
||||
// CMP x, reg: OP r/m, r (0x38/0x39) with rm = first operand, reg =
|
||||
// second, matching the Go assembler.
|
||||
opc := op.rr
|
||||
if size == 1 {
|
||||
opc = op.rr - 1
|
||||
}
|
||||
i := newInstr(size, []byte{opc})
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
case cmp && srcIsReg:
|
||||
// CMP reg, mem: OP r, r/m (0x3A/0x3B) with reg = first operand, rm =
|
||||
// second.
|
||||
opc := op.rr + 2
|
||||
if size == 1 {
|
||||
opc = op.rr + 1
|
||||
}
|
||||
i := newInstr(size, []byte{opc})
|
||||
if err := setRM(i, srcReg, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
case srcIsReg:
|
||||
// OP r/m, r: reg=src, rm=dst (dst is a register or memory). This is the
|
||||
// form the Go assembler prefers when the source is a register.
|
||||
@@ -251,12 +291,13 @@ func (e *enc) encodeLea(ops []Operand, size int) error {
|
||||
if !ok {
|
||||
return fmt.Errorf("LEA: destination must be a register")
|
||||
}
|
||||
mem, ok := src.(Mem)
|
||||
if !ok {
|
||||
switch src.(type) {
|
||||
case Mem, sbMem:
|
||||
default:
|
||||
return fmt.Errorf("LEA: source must be a memory operand")
|
||||
}
|
||||
i := newInstr(size, []byte{0x8D})
|
||||
if err := setRM(i, dstReg, mem, size); err != nil {
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
@@ -497,3 +538,212 @@ func immediate(v int64, size int, full64 bool) []byte {
|
||||
return le32(v) // sign-extended imm32
|
||||
}
|
||||
}
|
||||
|
||||
// --- CMOVcc / SETcc ---------------------------------------------------------
|
||||
|
||||
// encodeCmov encodes a conditional move: CMOV + size (W/L/Q) + condition
|
||||
// (CMOVLGT, CMOVQEQ, …). The condition reads exactly like the Jcc spellings;
|
||||
// the instruction is 0F 40+cc with reg = dst, rm = src.
|
||||
func (e *enc) encodeCmov(upper string, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("CMOVcc expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
rest := upper[len("CMOV"):]
|
||||
if len(rest) < 2 {
|
||||
return fmt.Errorf("unsupported instruction %q", upper)
|
||||
}
|
||||
var size int
|
||||
switch rest[0] {
|
||||
case 'W':
|
||||
size = 2
|
||||
case 'L':
|
||||
size = 4
|
||||
case 'Q':
|
||||
size = 8
|
||||
default:
|
||||
return fmt.Errorf("unsupported instruction %q", upper)
|
||||
}
|
||||
cc, ok := jccMap[rest[1:]]
|
||||
if !ok {
|
||||
return fmt.Errorf("unsupported instruction %q", upper)
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("CMOVcc destination must be a register")
|
||||
}
|
||||
i := newInstr(size, []byte{0x0F, byte(0x40 + cc)})
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeSet encodes a conditional byte set: SET + condition (SETNE, SETEQ, …),
|
||||
// always a byte write — 0F 90+cc /0 into a register or memory operand.
|
||||
func (e *enc) encodeSet(upper string, ops []Operand) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("SETcc expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
cond := upper[len("SET"):]
|
||||
cc, ok := jccMap[cond]
|
||||
if !ok || cond == "" {
|
||||
return fmt.Errorf("unsupported instruction %q", upper)
|
||||
}
|
||||
i := &instr{opcode: []byte{0x0F, byte(0x90 + cc)}, modrm: -1, sib: -1}
|
||||
if err := setRMDigit(i, 0, ops[0], 1); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- LZCNT / TZCNT ----------------------------------------------------------
|
||||
|
||||
// encodeCount encodes LZCNT/TZCNT (leading / trailing zero count): F3 0F BD
|
||||
// or F3 0F BC, with reg = dst and rm = src. The size suffix selects the
|
||||
// operand width (LZCNTW/LZCNTL/LZCNTQ).
|
||||
func (e *enc) encodeCount(base string, ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||||
}
|
||||
op := byte(0xBD)
|
||||
if base == "TZCNT" {
|
||||
op = 0xBC
|
||||
}
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s destination must be a register", base)
|
||||
}
|
||||
i := newInstr(size, []byte{0x0F, op})
|
||||
i.prefix = 0xF3
|
||||
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- mixed-width sign/zero-extending moves -----------------------------------
|
||||
|
||||
// movExtendOp maps Go's mixed-width move names to their opcode and destination
|
||||
// width. The source is narrower than the destination, so the plain size-suffix
|
||||
// convention does not apply to these names.
|
||||
var movExtendOp = map[string]struct {
|
||||
op []byte
|
||||
dst64 bool
|
||||
}{
|
||||
"MOVBLZX": {[]byte{0x0F, 0xB6}, false}, // byte → long, zero-extend
|
||||
"MOVBQZX": {[]byte{0x0F, 0xB6}, true}, // byte → quad, zero-extend
|
||||
"MOVWLZX": {[]byte{0x0F, 0xB7}, false}, // word → long, zero-extend
|
||||
"MOVWQZX": {[]byte{0x0F, 0xB7}, true}, // word → quad, zero-extend
|
||||
"MOVWLSX": {[]byte{0x0F, 0xBF}, false}, // word → long, sign-extend
|
||||
"MOVLQSX": {[]byte{0x63}, true}, // long → quad, sign-extend (MOVSXD)
|
||||
}
|
||||
|
||||
// encodeMovExtend encodes a mixed-width extending move: reg = dst (the wider
|
||||
// operand), rm = src.
|
||||
func (e *enc) encodeMovExtend(base string, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||||
}
|
||||
spec := movExtendOp[base]
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s destination must be a register", base)
|
||||
}
|
||||
size := 4
|
||||
if spec.dst64 {
|
||||
size = 8
|
||||
}
|
||||
i := newInstr(size, spec.op)
|
||||
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- legacy SSE moves --------------------------------------------------------
|
||||
|
||||
// sseMove describes a legacy (non-VEX) SSE move: a mandatory prefix plus a
|
||||
// load opcode (reg = destination, rm = source) and a store opcode (the
|
||||
// reverse). The Plan 9 names MOVOU/MOVO are the integer unaligned/aligned
|
||||
// octa moves (MOVDQU/MOVDQA), not the packed-single ones.
|
||||
type sseMove struct {
|
||||
prefix byte // 0, 0x66, 0xF2 or 0xF3
|
||||
load byte
|
||||
store byte
|
||||
}
|
||||
|
||||
var sseMoveTable = map[string]sseMove{
|
||||
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU — unaligned octa
|
||||
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA — aligned octa
|
||||
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
|
||||
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
|
||||
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
|
||||
"MOVAPD": {0x66, 0x28, 0x29}, // aligned packed double
|
||||
"MOVSD": {0xF2, 0x10, 0x11}, // scalar double
|
||||
"MOVSS": {0xF3, 0x10, 0x11}, // scalar single
|
||||
}
|
||||
|
||||
// encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the
|
||||
// load form (reg = destination), matching the Go assembler.
|
||||
func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("SSE move expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
srcReg, srcVec := vecReg(src)
|
||||
dstReg, dstVec := vecReg(dst)
|
||||
op := m.store
|
||||
var reg Reg
|
||||
var rm Operand
|
||||
switch {
|
||||
case srcVec && dstVec:
|
||||
op = m.load
|
||||
reg, rm = dstReg, src
|
||||
case srcVec:
|
||||
if _, ok := dst.(Mem); !ok {
|
||||
return fmt.Errorf("SSE move: invalid destination operand")
|
||||
}
|
||||
reg, rm = srcReg, dst
|
||||
case dstVec:
|
||||
if _, ok := src.(Mem); !ok {
|
||||
return fmt.Errorf("SSE move: invalid source operand")
|
||||
}
|
||||
op = m.load
|
||||
reg, rm = dstReg, src
|
||||
default:
|
||||
return fmt.Errorf("SSE move needs a vector register operand")
|
||||
}
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, reg, rm, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- CVTSL2SD / CVTSQ2SD -----------------------------------------------------
|
||||
|
||||
// encodeCvtsi2sd encodes a signed integer to scalar double conversion
|
||||
// (CVTSL2SD from a 32-bit, CVTSQ2SD from a 64-bit source): F2 0F 2A with
|
||||
// reg = XMM dst, rm = GPR/memory src. The Go assembler emits the legacy SSE
|
||||
// encoding here, not the VEX form, so we match it byte for byte.
|
||||
func (e *enc) encodeCvtsi2sd(quad bool, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("CVTSx2SD expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("CVTSx2SD destination must be a vector register")
|
||||
}
|
||||
size := 4
|
||||
if quad {
|
||||
size = 8
|
||||
}
|
||||
i := newInstr(size, []byte{0x0F, 0x2A})
|
||||
i.prefix = 0xF2
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
+411
@@ -0,0 +1,411 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
)
|
||||
|
||||
// Image is an assembled file: the function bodies laid out in source order,
|
||||
// followed by the file's static data section (GLOBL/DATA). References to
|
||||
// file-local static symbols are encoded RIP-relative and resolved within the
|
||||
// image, so the raw bytes are self-consistent and executable at any base
|
||||
// address; references to external symbols are recorded as relocations
|
||||
// (Funcs[i].Relocs, Externals) and left unresolved — the object-file
|
||||
// emitters turn them into linker relocations.
|
||||
type Image struct {
|
||||
Code []byte // concatenated function bodies
|
||||
Data []byte // static data section
|
||||
Funcs []FuncLayout // function positions, in source order
|
||||
Symbols map[string]int // static symbol → byte offset within the image
|
||||
DataSyms []DataSymbol // GLOBL symbols, in layout order
|
||||
Externals []string // referenced but undefined symbols, sorted
|
||||
}
|
||||
|
||||
// FuncLayout describes one assembled function within an Image.
|
||||
type FuncLayout struct {
|
||||
Name string
|
||||
Pkg string // explicit package prefix ("" = the current package)
|
||||
Static bool // the <> marker: file-local, not exported
|
||||
Offset int // start offset within the image (== offset within Code)
|
||||
Size int
|
||||
Args int // declared argument/result area (the TEXT size suffix)
|
||||
Frame int // local frame size (the TEXT $framesize)
|
||||
NoSplit bool // the NOSPLIT flag
|
||||
SPWrite bool // the SPWRITE flag: writes an arbitrary value to SP
|
||||
Line int // source line of the TEXT directive
|
||||
Labels map[string]int // local labels, function-relative
|
||||
Relocs []Reloc // static-symbol references, in emission order
|
||||
Spadj []SpadjStep // stack-adjustment boundaries, ascending by PC
|
||||
Lines []LineEntry // source-line table: byte offset → source line
|
||||
}
|
||||
|
||||
// SpadjStep is one stack-adjustment boundary: Value is the SP delta from the
|
||||
// entry state in effect from PC (function-relative) until the next step.
|
||||
type SpadjStep struct {
|
||||
PC int
|
||||
Value int
|
||||
}
|
||||
|
||||
// LineEntry maps a byte offset (function-relative) to a source line number.
|
||||
type LineEntry struct {
|
||||
Offset int
|
||||
Line int
|
||||
}
|
||||
|
||||
// LineAt returns the source line number for the given function-relative byte
|
||||
// offset, using a binary search on the line table. Returns 0 if the offset
|
||||
// is before the first instruction or the table is empty.
|
||||
func (fl *FuncLayout) LineAt(offset int) int {
|
||||
if len(fl.Lines) == 0 {
|
||||
return 0
|
||||
}
|
||||
// Binary search: find the last entry with Offset <= offset.
|
||||
lo, hi := 0, len(fl.Lines)-1
|
||||
for lo < hi {
|
||||
mid := (lo + hi + 1) / 2
|
||||
if fl.Lines[mid].Offset <= offset {
|
||||
lo = mid
|
||||
} else {
|
||||
hi = mid - 1
|
||||
}
|
||||
}
|
||||
if fl.Lines[lo].Offset <= offset {
|
||||
return fl.Lines[lo].Line
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// Reloc is one static-symbol reference within a function body: the disp32
|
||||
// field at Off (function-relative) must reach the symbol plus Addend,
|
||||
// measured from After, the address just past the instruction. An External
|
||||
// relocation names a symbol no GLOBL in the file defines; the object-file
|
||||
// emitters carry it into the output's relocation table.
|
||||
// RelocKind discriminates the type of relocation needed.
|
||||
type RelocKind int
|
||||
|
||||
const (
|
||||
RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64)
|
||||
RelPCRelHI20 // R_RISCV_PCREL_HI20 (AUIPC)
|
||||
RelPCRelLO12 // R_RISCV_PCREL_LO12_I (ADDI, LD)
|
||||
RelPCRelLO12S // R_RISCV_PCREL_LO12_S (SD)
|
||||
RelPCRelAbs // 32-bit absolute (R_RISCV_32)
|
||||
)
|
||||
|
||||
type Reloc struct {
|
||||
Off int
|
||||
After int
|
||||
Name string
|
||||
Addend int64
|
||||
External bool
|
||||
Kind RelocKind
|
||||
}
|
||||
|
||||
// DataSymbol describes one GLOBL symbol laid out in the data section.
|
||||
type DataSymbol struct {
|
||||
Name string
|
||||
Pkg string // explicit package prefix ("" = the current package)
|
||||
Offset int // byte offset within Data
|
||||
Size int
|
||||
Static bool // the <> marker: file-local, not exported
|
||||
Rodata bool // the RODATA flag: read-only data
|
||||
Dupok bool // the DUPOK flag: duplicate-OK
|
||||
}
|
||||
|
||||
// Bytes returns the whole image: code, then data.
|
||||
func (img *Image) Bytes() []byte {
|
||||
out := make([]byte, 0, len(img.Code)+len(img.Data))
|
||||
out = append(out, img.Code...)
|
||||
return append(out, img.Data...)
|
||||
}
|
||||
|
||||
// AssembleFile assembles every TEXT function of a parsed file and lays out
|
||||
// its static symbols (GLOBL/DATA) in a data section behind the code. Each
|
||||
// reference to a file-local static symbol becomes a RIP-relative load whose
|
||||
// displacement is resolved against that layout; a reference to a symbol no
|
||||
// GLOBL defines is recorded as an external relocation (Externals) with its
|
||||
// displacement left zero — the object-file emitters resolve it at link
|
||||
// time, while the raw image (Bytes) cannot represent it.
|
||||
func AssembleFile(f *ast.File) (*Image, error) {
|
||||
dataSyms, err := collectData(f)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
known := make(map[string]bool, len(dataSyms))
|
||||
for _, d := range dataSyms {
|
||||
known[d.name] = true
|
||||
}
|
||||
link := &linkInfo{symbols: known, allowExternal: true}
|
||||
|
||||
img := &Image{Symbols: map[string]int{}}
|
||||
type asmFunc struct {
|
||||
name string
|
||||
patches []sbPatch
|
||||
}
|
||||
var funcs []asmFunc
|
||||
for _, d := range f.Decls {
|
||||
t, ok := d.(*ast.Text)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, patches, labels, steps, lines, err := assemble(t, link)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||
}
|
||||
fl := FuncLayout{
|
||||
Name: t.Name.Name,
|
||||
Pkg: t.Name.Pkg,
|
||||
Static: t.Name.Static,
|
||||
Offset: len(img.Code),
|
||||
Size: len(code),
|
||||
Frame: frameSize(t),
|
||||
Args: argsSize(t),
|
||||
Line: t.Pos().Line,
|
||||
Labels: labels,
|
||||
Lines: lines,
|
||||
}
|
||||
for _, f := range t.Flags {
|
||||
switch f {
|
||||
case "NOSPLIT":
|
||||
fl.NoSplit = true
|
||||
case "SPWRITE":
|
||||
fl.SPWrite = true
|
||||
}
|
||||
}
|
||||
for _, s := range steps {
|
||||
fl.Spadj = append(fl.Spadj, SpadjStep{PC: s.pc, Value: s.value})
|
||||
}
|
||||
img.Funcs = append(img.Funcs, fl)
|
||||
img.Code = append(img.Code, code...)
|
||||
funcs = append(funcs, asmFunc{name: t.Name.Name, patches: patches})
|
||||
}
|
||||
|
||||
// Lay out the data section behind the code, each symbol 16-aligned.
|
||||
dataStart := len(img.Code)
|
||||
for _, d := range dataSyms {
|
||||
if pos := dataStart + len(img.Data); pos != align16(pos) {
|
||||
img.Data = append(img.Data, make([]byte, align16(pos)-pos)...)
|
||||
}
|
||||
img.Symbols[d.name] = dataStart + len(img.Data)
|
||||
img.DataSyms = append(img.DataSyms, DataSymbol{
|
||||
Name: d.name,
|
||||
Pkg: d.pkg,
|
||||
Offset: len(img.Data),
|
||||
Size: len(d.buf),
|
||||
Static: d.static,
|
||||
Rodata: d.rodata,
|
||||
Dupok: d.dupok,
|
||||
})
|
||||
img.Data = append(img.Data, d.buf...)
|
||||
}
|
||||
|
||||
// Resolve the RIP-relative displacements of file-local references now
|
||||
// that every address is known, and record every reference (resolved or
|
||||
// external) for the object-file emitters.
|
||||
externals := map[string]bool{}
|
||||
for i, fn := range funcs {
|
||||
base := img.Funcs[i].Offset
|
||||
code := img.Code[base : base+img.Funcs[i].Size]
|
||||
for _, p := range fn.patches {
|
||||
reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend}
|
||||
if imgOff, ok := img.Symbols[p.name]; ok {
|
||||
rel := int64(imgOff) + p.addend - int64(base+p.after)
|
||||
if rel < -1<<31 || rel >= 1<<31 {
|
||||
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
|
||||
}
|
||||
copy(code[p.off:p.off+4], le32(rel))
|
||||
} else {
|
||||
reloc.External = true
|
||||
externals[p.name] = true
|
||||
}
|
||||
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
|
||||
}
|
||||
}
|
||||
for name := range externals {
|
||||
img.Externals = append(img.Externals, name)
|
||||
}
|
||||
sort.Strings(img.Externals)
|
||||
return img, nil
|
||||
}
|
||||
|
||||
// AssembleFileRISCV assembles every TEXT function of a parsed RISC-V file
|
||||
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
|
||||
// code. SB references in the code are encoded as AUIPC pairs with zero
|
||||
// immediates; the object-file emitters record relocations for the linker.
|
||||
func AssembleFileRISCV(f *ast.File) (*Image, error) {
|
||||
dataSyms, err := collectData(f)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
img := &Image{Symbols: map[string]int{}}
|
||||
for _, d := range f.Decls {
|
||||
t, ok := d.(*ast.Text)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, labels, relocs, err := assembleRISCV(t)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||
}
|
||||
fl := FuncLayout{
|
||||
Name: t.Name.Name,
|
||||
Pkg: t.Name.Pkg,
|
||||
Static: t.Name.Static,
|
||||
Offset: len(img.Code),
|
||||
Size: len(code),
|
||||
Frame: frameSize(t),
|
||||
Args: argsSize(t),
|
||||
Line: t.Pos().Line,
|
||||
Labels: labels,
|
||||
Relocs: relocs,
|
||||
}
|
||||
for _, f := range t.Flags {
|
||||
switch f {
|
||||
case "NOSPLIT":
|
||||
fl.NoSplit = true
|
||||
case "SPWRITE":
|
||||
fl.SPWrite = true
|
||||
}
|
||||
}
|
||||
img.Funcs = append(img.Funcs, fl)
|
||||
img.Code = append(img.Code, code...)
|
||||
}
|
||||
|
||||
// Lay out the data section behind the code, 16-aligned.
|
||||
dataStart := len(img.Code)
|
||||
for _, d := range dataSyms {
|
||||
pos := dataStart + len(img.Data)
|
||||
for pos%16 != 0 {
|
||||
img.Data = append(img.Data, 0)
|
||||
pos++
|
||||
}
|
||||
img.Symbols[d.name] = pos
|
||||
img.Data = append(img.Data, d.buf...)
|
||||
img.DataSyms = append(img.DataSyms, DataSymbol{
|
||||
Name: d.name,
|
||||
Pkg: d.pkg,
|
||||
Offset: pos,
|
||||
Size: d.size,
|
||||
Static: d.static,
|
||||
Rodata: d.rodata,
|
||||
Dupok: d.dupok,
|
||||
})
|
||||
}
|
||||
|
||||
return img, nil
|
||||
}
|
||||
|
||||
// dataSym is one GLOBL symbol and its DATA initialiser.
|
||||
type dataSym struct {
|
||||
name string
|
||||
pkg string
|
||||
buf []byte
|
||||
size int
|
||||
static bool
|
||||
rodata bool
|
||||
dupok bool
|
||||
}
|
||||
|
||||
// collectData gathers the file's static symbols (GLOBL) and their initial
|
||||
// contents (DATA) into byte buffers, in declaration order.
|
||||
func collectData(f *ast.File) ([]dataSym, error) {
|
||||
index := map[string]int{}
|
||||
var syms []dataSym
|
||||
for _, d := range f.Decls {
|
||||
switch dd := d.(type) {
|
||||
case *ast.Globl:
|
||||
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
||||
continue
|
||||
}
|
||||
name := dd.Name.Name
|
||||
if _, dup := index[name]; dup {
|
||||
return nil, fmt.Errorf("duplicate GLOBL %q", name)
|
||||
}
|
||||
size := 0
|
||||
if dd.Size != nil && dd.Size.Imm.HasVal {
|
||||
size = int(dd.Size.Imm.Val)
|
||||
}
|
||||
index[name] = len(syms)
|
||||
ds := dataSym{
|
||||
name: name,
|
||||
pkg: dd.Name.Pkg,
|
||||
buf: make([]byte, size),
|
||||
size: size,
|
||||
static: dd.Name.Static,
|
||||
}
|
||||
for _, f := range dd.Flags {
|
||||
switch f {
|
||||
case "RODATA":
|
||||
ds.rodata = true
|
||||
case "DUPOK":
|
||||
ds.dupok = true
|
||||
case "1":
|
||||
ds.dupok = true
|
||||
case "8":
|
||||
ds.rodata = true
|
||||
case "9":
|
||||
ds.dupok = true
|
||||
ds.rodata = true
|
||||
}
|
||||
}
|
||||
syms = append(syms, ds)
|
||||
|
||||
case *ast.Data:
|
||||
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
||||
continue
|
||||
}
|
||||
i, ok := index[dd.Name.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
|
||||
}
|
||||
if dd.Value == nil || !dd.Value.Imm.HasVal {
|
||||
return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name)
|
||||
}
|
||||
w := dd.Width
|
||||
switch w {
|
||||
case 1, 2, 4, 8:
|
||||
default:
|
||||
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
|
||||
}
|
||||
off := dd.Name.Offset
|
||||
buf := syms[i].buf
|
||||
if off < 0 || off+int64(w) > int64(len(buf)) {
|
||||
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
|
||||
}
|
||||
v := dd.Value.Imm.Val
|
||||
if dd.Value.Imm.Neg {
|
||||
v = -v
|
||||
}
|
||||
for j := 0; j < w; j++ {
|
||||
buf[off+int64(j)] = byte(v >> (8 * j))
|
||||
}
|
||||
}
|
||||
}
|
||||
return syms, nil
|
||||
}
|
||||
|
||||
// align16 rounds n up to the next multiple of 16.
|
||||
func align16(n int) int {
|
||||
return (n + 15) &^ 15
|
||||
}
|
||||
|
||||
// frameSize returns the local frame size declared on the TEXT directive.
|
||||
func frameSize(t *ast.Text) int {
|
||||
if t.Frame != nil && t.Frame.Imm.HasVal {
|
||||
return int(t.Frame.Imm.Val)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// argsSize returns the argument/result area declared on the TEXT directive.
|
||||
func argsSize(t *ast.Text) int {
|
||||
if t.Args != nil && t.Args.Imm.HasVal {
|
||||
return int(t.Args.Imm.Val)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
@@ -0,0 +1,196 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestAssembleFileStaticData checks the whole-image layout — code, padding
|
||||
// and the data section — and that the RIP-relative displacements of static
|
||||
// symbol loads resolve to the right bytes.
|
||||
func TestAssembleFileStaticData(t *testing.T) {
|
||||
f, errs := parser.Parse("d_amd64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·load(SB), NOSPLIT, $0
|
||||
VMOVDQU mask<>(SB), X15
|
||||
MOVL small<>(SB), AX
|
||||
RET
|
||||
GLOBL mask<>(SB), RODATA, $16
|
||||
DATA mask<>+0(SB)/4, $0x80020100
|
||||
DATA mask<>+4(SB)/4, $0x80050403
|
||||
DATA mask<>+8(SB)/4, $0x80080706
|
||||
DATA mask<>+12(SB)/4, $0x800B0A09
|
||||
GLOBL small<>(SB), RODATA, $4
|
||||
DATA small<>+0(SB)/4, $0x1234
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
|
||||
// Code (15 bytes) + 1 pad byte to align the data section to 16:
|
||||
// VMOVDQU mask<>(SB), X15 c5 7a 6f 3d 08 00 00 00 (disp = 16 − 8)
|
||||
// MOVL small<>(SB), AX 8b 05 12 00 00 00 (disp = 32 − 14)
|
||||
// RET c3
|
||||
// Data: pad, mask (16 bytes), small (4 bytes).
|
||||
want := "c57a6f3d080000008b0512000000c300" +
|
||||
"000102800304058006070880090a0b80" +
|
||||
"34120000"
|
||||
if got := strings.ReplaceAll(hexBytes(img.Bytes()), " ", ""); got != want {
|
||||
t.Errorf("image bytes:\n got %s\n want %s", got, want)
|
||||
}
|
||||
if img.Symbols["mask"] != 16 || img.Symbols["small"] != 32 {
|
||||
t.Errorf("symbol offsets = %v, want mask=16 small=32", img.Symbols)
|
||||
}
|
||||
if len(img.Funcs) != 1 || img.Funcs[0].Name != "load" || img.Funcs[0].Size != 15 {
|
||||
t.Errorf("funcs = %+v", img.Funcs)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleFileErrors checks the static-symbol error paths.
|
||||
func TestAssembleFileErrors(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
src string
|
||||
want string // substring of the error
|
||||
}{
|
||||
{
|
||||
"undefined symbol",
|
||||
`
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
VMOVDQU nope<>(SB), X0
|
||||
RET
|
||||
`,
|
||||
"undefined symbol",
|
||||
},
|
||||
{
|
||||
"DATA without GLOBL",
|
||||
`
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
RET
|
||||
DATA orphan<>+0(SB)/4, $1
|
||||
`,
|
||||
"no matching GLOBL",
|
||||
},
|
||||
{
|
||||
"DATA exceeds size",
|
||||
`
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
RET
|
||||
GLOBL tiny<>(SB), RODATA, $4
|
||||
DATA tiny<>+0(SB)/8, $1
|
||||
`,
|
||||
"exceeds GLOBL size",
|
||||
},
|
||||
{
|
||||
"DATA bad width",
|
||||
`
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
RET
|
||||
GLOBL odd<>(SB), RODATA, $4
|
||||
DATA odd<>+0(SB)/3, $1
|
||||
`,
|
||||
"invalid width",
|
||||
},
|
||||
}
|
||||
for _, c := range cases {
|
||||
f, errs := parser.Parse("e_amd64.s", c.src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("%s: parse: %v", c.name, errs)
|
||||
}
|
||||
if _, err := AssembleFile(f); err == nil || !strings.Contains(err.Error(), c.want) {
|
||||
t.Errorf("%s: error %v, want substring %q", c.name, err, c.want)
|
||||
}
|
||||
}
|
||||
|
||||
// A static-symbol operand is unresolvable in single-function assembly.
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
MOVQ x<>(SB), AX
|
||||
RET
|
||||
GLOBL x<>(SB), RODATA, $8
|
||||
DATA x<>+0(SB)/4, $1
|
||||
`)
|
||||
if _, _, err := Assemble(fn); err == nil || !strings.Contains(err.Error(), "file-level assembly") {
|
||||
t.Errorf("single-function SB: error %v, want a file-level-assembly error", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel —
|
||||
// all functions plus the file-local mask24 constant — and checks that every
|
||||
// static-symbol load resolves to the right bytes in the image. Skipped when
|
||||
// the sibling repository is not checked out.
|
||||
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
||||
path := "../../go-libraries/go-flac/avx2_amd64.s"
|
||||
if _, err := os.Stat(path); err != nil {
|
||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
||||
}
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f, errs := parser.Parse(path, string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
if len(img.Funcs) != 17 {
|
||||
t.Errorf("functions = %d, want 17", len(img.Funcs))
|
||||
}
|
||||
|
||||
// mask24 as the DATA directives define it.
|
||||
mask := []byte{
|
||||
0x00, 0x01, 0x02, 0x80, 0x03, 0x04, 0x05, 0x80,
|
||||
0x06, 0x07, 0x08, 0x80, 0x09, 0x0a, 0x0b, 0x80,
|
||||
}
|
||||
image := img.Bytes()
|
||||
if got := image[img.Symbols["mask24"] : img.Symbols["mask24"]+16]; !bytes.Equal(got, mask) {
|
||||
t.Errorf("mask24 contents %x, want %x", got, mask)
|
||||
}
|
||||
|
||||
// Every VMOVDQU mask24<>(SB), X15 (c5 7a 6f 3d + rel32, i.e. a VMOVDQU
|
||||
// with a RIP-relative r/m) must land on the mask bytes within the image.
|
||||
loads := 0
|
||||
for _, fn := range img.Funcs {
|
||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
for pc := 0; pc < len(code); {
|
||||
inst, err := x86asm.Decode(code[pc:], 64)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: decode at +%d: %v", fn.Name, pc, err)
|
||||
}
|
||||
// mod=00, rm=101 → RIP-relative.
|
||||
if inst.Op == x86asm.VMOVDQU && inst.Len == 8 && code[pc+3]&0xC7 == 0x05 {
|
||||
rel := int32(uint32(code[pc+4]) | uint32(code[pc+5])<<8 | uint32(code[pc+6])<<16 | uint32(code[pc+7])<<24)
|
||||
target := fn.Offset + pc + 8 + int(rel)
|
||||
if !bytes.Equal(image[target:target+16], mask) {
|
||||
t.Errorf("%s: mask load at +%d lands on %x, want %x", fn.Name, pc, image[target:target+16], mask)
|
||||
}
|
||||
loads++
|
||||
}
|
||||
pc += inst.Len
|
||||
}
|
||||
}
|
||||
if loads != 2 {
|
||||
t.Errorf("mask loads found = %d, want 2", loads)
|
||||
}
|
||||
}
|
||||
+258
@@ -0,0 +1,258 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
)
|
||||
|
||||
// This file emits Mach-O x86-64 objects (MH_OBJECT) from an assembled
|
||||
// Image, in the shape the Darwin assembler produces: one unnamed segment
|
||||
// carrying a __TEXT,__text and a __DATA,__data section laid out back to
|
||||
// back at addresses zero and len(code), a symbol table (locals first, then
|
||||
// exported definitions, then undefined externals) and one relocation entry
|
||||
// per static-symbol reference, of type X86_64_RELOC_SIGNED.
|
||||
//
|
||||
// The image's own address space carries straight over — the data section
|
||||
// starts immediately after the code, and the layout padding already lives
|
||||
// inside Image.Data — so every symbol keeps its image address as its
|
||||
// n_value, and a local (non-external) relocation leaves the displacement
|
||||
// the assembler resolved in place: the linker only adjusts it by the
|
||||
// section's final movement.
|
||||
|
||||
// Mach-O constants.
|
||||
const (
|
||||
machoMagic64 = 0xfeedfacf
|
||||
machoCPUamd64 = 0x01000007 // CPU_TYPE_X86_64
|
||||
machoCPUSubAll = 3 // CPU_SUBTYPE_X86_64_ALL
|
||||
machoObj = 1 // MH_OBJECT
|
||||
|
||||
machoSegment64 = 0x19 // LC_SEGMENT_64
|
||||
machoSymtab = 0x2 // LC_SYMTAB
|
||||
|
||||
machoSectTextFlags = 0x80000400 // S_ATTR_PURE_INSTRUCTIONS | S_ATTR_SOME_INSTRUCTIONS
|
||||
|
||||
nUndf = 0x00 // undefined symbol
|
||||
nSect = 0x0e // defined in section number n_sect
|
||||
nExt = 0x01 // external (exported or undefined-global) bit
|
||||
|
||||
x8664RelocSigned = 1
|
||||
)
|
||||
|
||||
// MachOObject returns the image as a Mach-O x86-64 relocatable object
|
||||
// (MH_OBJECT), the shape the Darwin toolchain links. Symbol names follow
|
||||
// the same rules as the ELF output. Every static-symbol reference becomes
|
||||
// an X86_64_RELOC_SIGNED relocation: external references against their
|
||||
// undefined symbol, file-local ones against the __DATA section with the
|
||||
// resolved displacement carried in the instruction bytes.
|
||||
func (img *Image) MachOObject() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
// Section ordinals (1-based, as Mach-O numbers them).
|
||||
const (
|
||||
sectText = 1
|
||||
sectData = 2
|
||||
)
|
||||
|
||||
// Object address space: code at 0, data immediately after (the layout
|
||||
// padding is already part of img.Data, so image addresses are object
|
||||
// addresses).
|
||||
textAddr := uint64(0)
|
||||
dataAddr := uint64(len(img.Code))
|
||||
vmsize := dataAddr + uint64(len(img.Data))
|
||||
|
||||
// The code, with external displacements primed to addend − 4: the
|
||||
// linker adds the symbol's address to the field as it stands. Local
|
||||
// displacements stay as the assembler resolved them.
|
||||
code := append([]byte(nil), img.Code...)
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
if r.External {
|
||||
// Prime the field to the addend measured from the patch
|
||||
// site: the assembler records it from the instruction end,
|
||||
// After − Off bytes past the field.
|
||||
copy(code[fn.Offset+r.Off:], le32(r.Addend-int64(r.After-r.Off)))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Symbols: locals first, then exported definitions, then undefined
|
||||
// externals — the order the classic link editor expects.
|
||||
type machoSym struct {
|
||||
name string
|
||||
typ byte
|
||||
sect byte
|
||||
value uint64
|
||||
}
|
||||
var locals, globals, undefs []machoSym
|
||||
for _, fn := range img.Funcs {
|
||||
s := machoSym{name: objectName(fn.Pkg, fn.Name), typ: nSect, sect: sectText, value: textAddr + uint64(fn.Offset)}
|
||||
if fn.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.typ |= nExt
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, d := range img.DataSyms {
|
||||
s := machoSym{name: objectName(d.Pkg, d.Name), typ: nSect, sect: sectData, value: dataAddr + uint64(d.Offset)}
|
||||
if d.Static {
|
||||
locals = append(locals, s)
|
||||
} else {
|
||||
s.typ |= nExt
|
||||
globals = append(globals, s)
|
||||
}
|
||||
}
|
||||
for _, name := range img.Externals {
|
||||
undefs = append(undefs, machoSym{name: name, typ: nUndf | nExt})
|
||||
}
|
||||
syms := append(append(locals, globals...), undefs...)
|
||||
symIdx := map[string]int{}
|
||||
for i, s := range syms {
|
||||
symIdx[s.name] = i
|
||||
}
|
||||
|
||||
// Relocations, attached to the __text section.
|
||||
type machoReloc struct {
|
||||
addr uint32
|
||||
symnum uint32
|
||||
extern bool
|
||||
}
|
||||
var relocs []machoReloc
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
rel := machoReloc{addr: uint32(fn.Offset + r.Off)}
|
||||
if r.External {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
rel.symnum = uint32(idx)
|
||||
rel.extern = true
|
||||
} else {
|
||||
// Section-relative: r_symbolnum carries the section number
|
||||
// and the resolved displacement stays in the bytes.
|
||||
rel.symnum = sectData
|
||||
}
|
||||
relocs = append(relocs, rel)
|
||||
}
|
||||
}
|
||||
|
||||
// The string table opens with the conventional " \0".
|
||||
strtab := []byte{' ', 0}
|
||||
strOff := map[string]int{}
|
||||
for _, s := range syms {
|
||||
if _, ok := strOff[s.name]; ok {
|
||||
continue
|
||||
}
|
||||
strOff[s.name] = len(strtab)
|
||||
strtab = append(strtab, s.name...)
|
||||
strtab = append(strtab, 0)
|
||||
}
|
||||
|
||||
// File layout: header, the two load commands, section data (code,
|
||||
// data), the relocation table, the symbol table, the string table.
|
||||
const (
|
||||
hdrSize = 32
|
||||
segCmdSize = 72 + 2*80 // segment command with two sections
|
||||
symCmdSize = 24
|
||||
)
|
||||
sizeofcmds := segCmdSize + symCmdSize
|
||||
dataOff := hdrSize + sizeofcmds
|
||||
reloff := dataOff + len(code) + len(img.Data)
|
||||
symoff := reloff + 8*len(relocs)
|
||||
stroff := symoff + 16*len(syms)
|
||||
|
||||
out := make([]byte, stroff+len(strtab))
|
||||
|
||||
// mach_header_64.
|
||||
le.PutUint32(out[0:], machoMagic64)
|
||||
le.PutUint32(out[4:], machoCPUamd64)
|
||||
le.PutUint32(out[8:], machoCPUSubAll)
|
||||
le.PutUint32(out[12:], machoObj)
|
||||
le.PutUint32(out[16:], 2) // ncmds
|
||||
le.PutUint32(out[20:], uint32(sizeofcmds))
|
||||
le.PutUint32(out[24:], 0) // flags
|
||||
le.PutUint32(out[28:], 0) // reserved
|
||||
|
||||
// LC_SEGMENT_64 with the two sections.
|
||||
p := hdrSize
|
||||
le.PutUint32(out[p:], machoSegment64)
|
||||
le.PutUint32(out[p+4:], segCmdSize)
|
||||
// segname: the empty string, zero-padded to 16 bytes.
|
||||
le.PutUint64(out[p+8:], 0)
|
||||
le.PutUint64(out[p+16:], 0)
|
||||
le.PutUint64(out[p+24:], 0) // vmaddr
|
||||
le.PutUint64(out[p+32:], vmsize)
|
||||
le.PutUint64(out[p+40:], uint64(dataOff))
|
||||
le.PutUint64(out[p+48:], vmsize)
|
||||
le.PutUint32(out[p+56:], 7) // maxprot rwx
|
||||
le.PutUint32(out[p+60:], 7) // initprot rwx
|
||||
le.PutUint32(out[p+64:], 2) // nsects
|
||||
le.PutUint32(out[p+68:], 0) // flags
|
||||
|
||||
// __TEXT,__text
|
||||
s := p + 72
|
||||
copy(out[s:], "__text")
|
||||
copy(out[s+16:], "__TEXT")
|
||||
le.PutUint64(out[s+32:], textAddr)
|
||||
le.PutUint64(out[s+40:], uint64(len(code)))
|
||||
le.PutUint32(out[s+48:], uint32(dataOff))
|
||||
le.PutUint32(out[s+52:], 4) // align 2^4
|
||||
le.PutUint32(out[s+56:], uint32(reloff))
|
||||
le.PutUint32(out[s+60:], uint32(len(relocs)))
|
||||
le.PutUint32(out[s+64:], machoSectTextFlags)
|
||||
|
||||
// __DATA,__data
|
||||
s += 80
|
||||
copy(out[s:], "__data")
|
||||
copy(out[s+16:], "__DATA")
|
||||
le.PutUint64(out[s+32:], dataAddr)
|
||||
le.PutUint64(out[s+40:], uint64(len(img.Data)))
|
||||
le.PutUint32(out[s+48:], uint32(dataOff+len(code)))
|
||||
le.PutUint32(out[s+52:], 4) // align 2^4
|
||||
|
||||
// LC_SYMTAB.
|
||||
p = hdrSize + segCmdSize
|
||||
le.PutUint32(out[p:], machoSymtab)
|
||||
le.PutUint32(out[p+4:], symCmdSize)
|
||||
le.PutUint32(out[p+8:], uint32(symoff))
|
||||
le.PutUint32(out[p+12:], uint32(len(syms)))
|
||||
le.PutUint32(out[p+16:], uint32(stroff))
|
||||
le.PutUint32(out[p+20:], uint32(len(strtab)))
|
||||
|
||||
// Section data.
|
||||
copy(out[dataOff:], code)
|
||||
copy(out[dataOff+len(code):], img.Data)
|
||||
|
||||
// Relocation entries.
|
||||
for i, r := range relocs {
|
||||
e := out[reloff+i*8:]
|
||||
le.PutUint32(e[0:], r.addr)
|
||||
bits := r.symnum & 0x00ffffff
|
||||
bits |= 1 << 24 // r_pcrel
|
||||
bits |= 2 << 25 // r_length = 4 bytes
|
||||
if r.extern {
|
||||
bits |= 1 << 27 // r_extern
|
||||
}
|
||||
bits |= x8664RelocSigned << 28
|
||||
le.PutUint32(e[4:], bits)
|
||||
}
|
||||
|
||||
// nlist_64 entries.
|
||||
for i, s := range syms {
|
||||
e := out[symoff+i*16:]
|
||||
le.PutUint32(e[0:], uint32(strOff[s.name]))
|
||||
e[4] = s.typ
|
||||
e[5] = s.sect
|
||||
le.PutUint16(e[6:], 0) // n_desc
|
||||
le.PutUint64(e[8:], s.value)
|
||||
}
|
||||
|
||||
// String table.
|
||||
copy(out[stroff:], strtab)
|
||||
|
||||
return out, nil
|
||||
}
|
||||
@@ -0,0 +1,127 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"debug/macho"
|
||||
"encoding/binary"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// TestMachOObject checks the structure of the emitted MH_OBJECT: the two
|
||||
// sections and their addresses, the symbol table (types, sections, values)
|
||||
// and the __text relocation entries, parsed back with debug/macho. No
|
||||
// Darwin toolchain is available on the test hosts, so the check is
|
||||
// structural — the ELF output carries the end-to-end link-and-run proof of
|
||||
// the shared symbol and relocation model.
|
||||
func TestMachOObject(t *testing.T) {
|
||||
img := elfTestImage(t)
|
||||
obj, err := img.MachOObject()
|
||||
if err != nil {
|
||||
t.Fatalf("MachOObject: %v", err)
|
||||
}
|
||||
f, err := macho.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
if f.Type != macho.TypeObj {
|
||||
t.Errorf("file type = %v, want MH_OBJECT", f.Type)
|
||||
}
|
||||
if f.Cpu != macho.CpuAmd64 {
|
||||
t.Errorf("cpu = %v, want CpuAmd64", f.Cpu)
|
||||
}
|
||||
|
||||
text := f.Section("__text")
|
||||
data := f.Section("__data")
|
||||
if text == nil || data == nil {
|
||||
t.Fatal("missing __text or __data section")
|
||||
}
|
||||
if text.Addr != 0 || text.Size != uint64(len(img.Code)) {
|
||||
t.Errorf("__text addr/size = %#x/%d, want 0/%d", text.Addr, text.Size, len(img.Code))
|
||||
}
|
||||
if data.Addr != uint64(len(img.Code)) {
|
||||
t.Errorf("__data addr = %#x, want %#x", data.Addr, len(img.Code))
|
||||
}
|
||||
|
||||
// Symbol table: locals, exported definitions, undefined externals.
|
||||
syms := f.Symtab.Syms
|
||||
byName := map[string]macho.Symbol{}
|
||||
for _, s := range syms {
|
||||
byName[s.Name] = s
|
||||
}
|
||||
wantSym := func(name string, typ, sect uint8, value uint64) {
|
||||
t.Helper()
|
||||
s, ok := byName[name]
|
||||
if !ok {
|
||||
t.Errorf("symbol %q not found", name)
|
||||
return
|
||||
}
|
||||
if s.Type != typ || s.Sect != sect || s.Value != value {
|
||||
t.Errorf("%s: type/sect/value = %#x/%d/%#x, want %#x/%d/%#x",
|
||||
name, s.Type, s.Sect, s.Value, typ, sect, value)
|
||||
}
|
||||
}
|
||||
const (
|
||||
defined = nSect | nExt
|
||||
local = nSect
|
||||
undefined = nUndf | nExt
|
||||
)
|
||||
wantSym("addq", defined, 1, 0)
|
||||
wantSym("getanswer", defined, 1, 5)
|
||||
wantSym("useextern", defined, 1, 13)
|
||||
answer := byName["answer"]
|
||||
if answer.Type != local || answer.Sect != 2 {
|
||||
t.Errorf("answer: type/sect = %#x/%d, want %#x/2", answer.Type, answer.Sect, local)
|
||||
}
|
||||
wantSym("extvar", undefined, 0, 0)
|
||||
|
||||
// Relocations: both X86_64_RELOC_SIGNED, PC-relative, 4 bytes wide.
|
||||
// The local one carries its section number in Value, the external one
|
||||
// its symbol number.
|
||||
if len(text.Relocs) != 2 {
|
||||
t.Fatalf("__text relocs = %d, want 2", len(text.Relocs))
|
||||
}
|
||||
var sawLocal, sawExternal bool
|
||||
for _, r := range text.Relocs {
|
||||
if !r.Pcrel || r.Len != 2 || r.Type != x8664RelocSigned {
|
||||
t.Errorf("reloc at %#x: pcrel/len/type = %v/%d/%d", r.Addr, r.Pcrel, r.Len, r.Type)
|
||||
}
|
||||
switch {
|
||||
case r.Extern:
|
||||
if name := syms[r.Value].Name; name != "extvar" {
|
||||
t.Errorf("external reloc at %#x names %q, want extvar", r.Addr, name)
|
||||
}
|
||||
sawExternal = true
|
||||
default:
|
||||
if r.Value != 2 { // __data, the second section
|
||||
t.Errorf("local reloc at %#x: section %d, want 2 (__data)", r.Addr, r.Value)
|
||||
}
|
||||
sawLocal = true
|
||||
}
|
||||
}
|
||||
if !sawLocal || !sawExternal {
|
||||
t.Errorf("relocs seen: local=%v external=%v, want both", sawLocal, sawExternal)
|
||||
}
|
||||
|
||||
// The __text bytes are the image code, with the external displacement
|
||||
// primed to addend − 4 and the local one left resolved.
|
||||
textData, err := text.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
want := append([]byte(nil), img.Code...)
|
||||
for _, fn := range img.Funcs {
|
||||
for _, r := range fn.Relocs {
|
||||
if r.Name == "extvar" {
|
||||
binary.LittleEndian.PutUint32(want[fn.Offset+r.Off:], 0xfffffffc) // −4
|
||||
}
|
||||
}
|
||||
}
|
||||
if !bytes.Equal(textData, want) {
|
||||
t.Errorf("__text bytes %x, want %x", textData, want)
|
||||
}
|
||||
}
|
||||
@@ -41,3 +41,15 @@ func Idx(base, index Reg, scale int, disp int64, size int) Mem {
|
||||
func Rip(disp int64, size int) Mem {
|
||||
return Mem{Disp: disp, Size: size}
|
||||
}
|
||||
|
||||
// sbMem is a memory operand that references a static (SB) symbol. It encodes
|
||||
// as a RIP-relative reference with a placeholder displacement; the encoder
|
||||
// records a patch site so the file-level layout can fill in the true rel32
|
||||
// once the symbol's address is known.
|
||||
type sbMem struct {
|
||||
size int
|
||||
name string // static symbol name (the GLOBL identifier)
|
||||
addend int64 // byte offset within the symbol
|
||||
}
|
||||
|
||||
func (sbMem) isOperand() {}
|
||||
|
||||
+69
-55
@@ -14,19 +14,24 @@ import "strings"
|
||||
// so the encoder keys off the register's index and lets the mnemonic supply the
|
||||
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
|
||||
// occupy indices 4–7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
|
||||
// those indices but require one.
|
||||
// those indices but require one. The mask flag marks the AVX-512 opmask
|
||||
// registers K0–K7.
|
||||
type Reg struct {
|
||||
idx int
|
||||
size int // informational width implied by the name; the mnemonic decides
|
||||
high bool // AH/CH/DH/BH
|
||||
mask bool // K0–K7 opmask register
|
||||
}
|
||||
|
||||
// Index returns the register number (0–15).
|
||||
// Index returns the register number (0–15 for GPRs, 0–31 for vectors).
|
||||
func (r Reg) Index() int { return r.idx }
|
||||
|
||||
// Size returns the width in bytes implied by the register's name.
|
||||
func (r Reg) Size() int { return r.size }
|
||||
|
||||
// IsMask reports whether r is an AVX-512 opmask register (K0–K7).
|
||||
func (r Reg) IsMask() bool { return r.mask }
|
||||
|
||||
func (r Reg) isOperand() {}
|
||||
|
||||
// needsREX reports whether this register forces a REX prefix at the given
|
||||
@@ -41,45 +46,45 @@ func (r Reg) needsREX(opSize int) bool {
|
||||
|
||||
// Register constants (the size is the width the name implies).
|
||||
var (
|
||||
AL = Reg{0, 1, false}
|
||||
CL = Reg{1, 1, false}
|
||||
DL = Reg{2, 1, false}
|
||||
BL = Reg{3, 1, false}
|
||||
AH = Reg{4, 1, true}
|
||||
CH = Reg{5, 1, true}
|
||||
DH = Reg{6, 1, true}
|
||||
BH = Reg{7, 1, true}
|
||||
SPL = Reg{4, 1, false}
|
||||
BPL = Reg{5, 1, false}
|
||||
SIL = Reg{6, 1, false}
|
||||
DIL = Reg{7, 1, false}
|
||||
AL = Reg{idx: 0, size: 1}
|
||||
CL = Reg{idx: 1, size: 1}
|
||||
DL = Reg{idx: 2, size: 1}
|
||||
BL = Reg{idx: 3, size: 1}
|
||||
AH = Reg{idx: 4, size: 1, high: true}
|
||||
CH = Reg{idx: 5, size: 1, high: true}
|
||||
DH = Reg{idx: 6, size: 1, high: true}
|
||||
BH = Reg{idx: 7, size: 1, high: true}
|
||||
SPL = Reg{idx: 4, size: 1}
|
||||
BPL = Reg{idx: 5, size: 1}
|
||||
SIL = Reg{idx: 6, size: 1}
|
||||
DIL = Reg{idx: 7, size: 1}
|
||||
|
||||
AX = Reg{0, 2, false}
|
||||
CX = Reg{1, 2, false}
|
||||
DX = Reg{2, 2, false}
|
||||
BX = Reg{3, 2, false}
|
||||
SP = Reg{4, 2, false}
|
||||
BP = Reg{5, 2, false}
|
||||
SI = Reg{6, 2, false}
|
||||
DI = Reg{7, 2, false}
|
||||
AX = Reg{idx: 0, size: 2}
|
||||
CX = Reg{idx: 1, size: 2}
|
||||
DX = Reg{idx: 2, size: 2}
|
||||
BX = Reg{idx: 3, size: 2}
|
||||
SP = Reg{idx: 4, size: 2}
|
||||
BP = Reg{idx: 5, size: 2}
|
||||
SI = Reg{idx: 6, size: 2}
|
||||
DI = Reg{idx: 7, size: 2}
|
||||
|
||||
EAX = Reg{0, 4, false}
|
||||
ECX = Reg{1, 4, false}
|
||||
EDX = Reg{2, 4, false}
|
||||
EBX = Reg{3, 4, false}
|
||||
ESP = Reg{4, 4, false}
|
||||
EBP = Reg{5, 4, false}
|
||||
ESI = Reg{6, 4, false}
|
||||
EDI = Reg{7, 4, false}
|
||||
EAX = Reg{idx: 0, size: 4}
|
||||
ECX = Reg{idx: 1, size: 4}
|
||||
EDX = Reg{idx: 2, size: 4}
|
||||
EBX = Reg{idx: 3, size: 4}
|
||||
ESP = Reg{idx: 4, size: 4}
|
||||
EBP = Reg{idx: 5, size: 4}
|
||||
ESI = Reg{idx: 6, size: 4}
|
||||
EDI = Reg{idx: 7, size: 4}
|
||||
|
||||
RAX = Reg{0, 8, false}
|
||||
RCX = Reg{1, 8, false}
|
||||
RDX = Reg{2, 8, false}
|
||||
RBX = Reg{3, 8, false}
|
||||
RSP = Reg{4, 8, false}
|
||||
RBP = Reg{5, 8, false}
|
||||
RSI = Reg{6, 8, false}
|
||||
RDI = Reg{7, 8, false}
|
||||
RAX = Reg{idx: 0, size: 8}
|
||||
RCX = Reg{idx: 1, size: 8}
|
||||
RDX = Reg{idx: 2, size: 8}
|
||||
RBX = Reg{idx: 3, size: 8}
|
||||
RSP = Reg{idx: 4, size: 8}
|
||||
RBP = Reg{idx: 5, size: 8}
|
||||
RSI = Reg{idx: 6, size: 8}
|
||||
RDI = Reg{idx: 7, size: 8}
|
||||
)
|
||||
|
||||
// regByName maps an assembly register name (case-insensitive) to a Reg.
|
||||
@@ -91,28 +96,28 @@ func buildRegByName() map[string]Reg {
|
||||
// 64-bit: RAX..RDI, R8..R15.
|
||||
r64 := []string{"RAX", "RCX", "RDX", "RBX", "RSP", "RBP", "RSI", "RDI"}
|
||||
for i, n := range r64 {
|
||||
m[n] = Reg{i, 8, false}
|
||||
m[n] = Reg{idx: i, size: 8}
|
||||
}
|
||||
for i := 8; i <= 15; i++ {
|
||||
m["R"+itoa(i)] = Reg{i, 8, false}
|
||||
m["R"+itoa(i)] = Reg{idx: i, size: 8}
|
||||
}
|
||||
|
||||
// 32-bit: EAX..EDI, R8D..R15D.
|
||||
e32 := []string{"EAX", "ECX", "EDX", "EBX", "ESP", "EBP", "ESI", "EDI"}
|
||||
for i, n := range e32 {
|
||||
m[n] = Reg{i, 4, false}
|
||||
m[n] = Reg{idx: i, size: 4}
|
||||
}
|
||||
for i := 8; i <= 15; i++ {
|
||||
m["R"+itoa(i)+"D"] = Reg{i, 4, false}
|
||||
m["R"+itoa(i)+"D"] = Reg{idx: i, size: 4}
|
||||
}
|
||||
|
||||
// 16-bit: AX..DI, R8W..R15W.
|
||||
w16 := []string{"AX", "CX", "DX", "BX", "SP", "BP", "SI", "DI"}
|
||||
for i, n := range w16 {
|
||||
m[n] = Reg{i, 2, false}
|
||||
m[n] = Reg{idx: i, size: 2}
|
||||
}
|
||||
for i := 8; i <= 15; i++ {
|
||||
m["R"+itoa(i)+"W"] = Reg{i, 2, false}
|
||||
m["R"+itoa(i)+"W"] = Reg{idx: i, size: 2}
|
||||
}
|
||||
|
||||
// 8-bit: AL..BH, SPL..DIL, R8B..R15B.
|
||||
@@ -124,25 +129,34 @@ func buildRegByName() map[string]Reg {
|
||||
m[n] = r
|
||||
}
|
||||
for i := 8; i <= 15; i++ {
|
||||
m["R"+itoa(i)+"B"] = Reg{i, 1, false}
|
||||
m["R"+itoa(i)+"B"] = Reg{idx: i, size: 1}
|
||||
}
|
||||
|
||||
// Vector: X0..X15 (128-bit, encoded size 16), Y0..Y15 (256-bit, size 32).
|
||||
// Z (512-bit) and K (mask) registers arrive with EVEX/AVX-512 support.
|
||||
for i := 0; i <= 15; i++ {
|
||||
m["X"+itoa(i)] = Reg{i, 16, false}
|
||||
m["Y"+itoa(i)] = Reg{i, 32, false}
|
||||
// Vector: X0..X31 (128-bit, size 16), Y0..Y31 (256-bit, size 32),
|
||||
// Z0..Z31 (512-bit, size 64). Indices 16–31 are only encodable in EVEX
|
||||
// (AVX-512) instructions; the encoder validates that through its tables.
|
||||
for i := 0; i <= 31; i++ {
|
||||
m["X"+itoa(i)] = Reg{idx: i, size: 16}
|
||||
m["Y"+itoa(i)] = Reg{idx: i, size: 32}
|
||||
m["Z"+itoa(i)] = Reg{idx: i, size: 64}
|
||||
}
|
||||
// Opmask: K0..K7.
|
||||
for i := 0; i <= 7; i++ {
|
||||
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// isVec reports whether r is an XMM/YMM vector register.
|
||||
func (r Reg) isVec() bool { return r.size == 16 || r.size == 32 }
|
||||
// isVec reports whether r is an XMM/YMM/ZMM vector register.
|
||||
func (r Reg) isVec() bool { return r.size == 16 || r.size == 32 || r.size == 64 }
|
||||
|
||||
// vecLenBit returns the VEX.L bit for a vector register (X=0/128-bit,
|
||||
// Y=1/256-bit).
|
||||
// vecLenBit returns the vector-length field for a vector register:
|
||||
// 0 (128-bit, VEX.L / EVEX.L'L=00), 1 (256-bit) or 2 (512-bit, EVEX only).
|
||||
func (r Reg) vecLenBit() int {
|
||||
if r.size == 32 {
|
||||
switch r.size {
|
||||
case 64:
|
||||
return 2
|
||||
case 32:
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,537 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
// RISC-V register encoding: maps register names to their 5-bit numbers.
|
||||
// The Go assembler uses the standard RISC-V ABI naming.
|
||||
|
||||
// riscvRegNum returns the 5-bit register number for a RISC-V register name.
|
||||
// Returns -1 if the register is not recognized.
|
||||
func riscvRegNum(name string) int {
|
||||
switch name {
|
||||
// Numbered integer registers.
|
||||
case "X0", "ZERO":
|
||||
return 0
|
||||
case "X1", "RA":
|
||||
return 1
|
||||
case "X2", "SP":
|
||||
return 2
|
||||
case "X3", "GP":
|
||||
return 3
|
||||
case "X4", "TP":
|
||||
return 4
|
||||
case "X5", "T0", "LR":
|
||||
return 5
|
||||
case "X6", "T1", "TMP":
|
||||
return 6
|
||||
case "X7", "T2":
|
||||
return 7
|
||||
case "X8", "S0", "FP":
|
||||
return 8
|
||||
case "X9", "S1":
|
||||
return 9
|
||||
case "X10", "A0":
|
||||
return 10
|
||||
case "X11", "A1":
|
||||
return 11
|
||||
case "X12", "A2":
|
||||
return 12
|
||||
case "X13", "A3":
|
||||
return 13
|
||||
case "X14", "A4":
|
||||
return 14
|
||||
case "X15", "A5":
|
||||
return 15
|
||||
case "X16", "A6":
|
||||
return 16
|
||||
case "X17", "A7":
|
||||
return 17
|
||||
case "X18", "S2":
|
||||
return 18
|
||||
case "X19", "S3":
|
||||
return 19
|
||||
case "X20", "S4":
|
||||
return 20
|
||||
case "X21", "S5":
|
||||
return 21
|
||||
case "X22", "S6":
|
||||
return 22
|
||||
case "X23", "S7":
|
||||
return 23
|
||||
case "X24", "S8":
|
||||
return 24
|
||||
case "X25", "S9":
|
||||
return 25
|
||||
case "X26", "S10":
|
||||
return 26
|
||||
case "X27", "S11":
|
||||
return 27
|
||||
case "X28", "T3":
|
||||
return 28
|
||||
case "X29", "T4":
|
||||
return 29
|
||||
case "X30", "T5":
|
||||
return 30
|
||||
case "X31", "T6":
|
||||
return 31
|
||||
// Floating-point registers (F0-F31).
|
||||
case "F0", "FT0":
|
||||
return 0
|
||||
case "F1", "FT1":
|
||||
return 1
|
||||
case "F2", "FT2":
|
||||
return 2
|
||||
case "F3", "FT3":
|
||||
return 3
|
||||
case "F4", "FT4":
|
||||
return 4
|
||||
case "F5", "FT5":
|
||||
return 5
|
||||
case "F6", "FT6":
|
||||
return 6
|
||||
case "F7", "FT7":
|
||||
return 7
|
||||
case "F8", "FS0":
|
||||
return 8
|
||||
case "F9", "FS1":
|
||||
return 9
|
||||
case "F10", "FA0":
|
||||
return 10
|
||||
case "F11", "FA1":
|
||||
return 11
|
||||
case "F12", "FA2":
|
||||
return 12
|
||||
case "F13", "FA3":
|
||||
return 13
|
||||
case "F14", "FA4":
|
||||
return 14
|
||||
case "F15", "FA5":
|
||||
return 15
|
||||
case "F16", "FA6":
|
||||
return 16
|
||||
case "F17", "FA7":
|
||||
return 17
|
||||
case "F18", "FS2":
|
||||
return 18
|
||||
case "F19", "FS3":
|
||||
return 19
|
||||
case "F20", "FS4":
|
||||
return 20
|
||||
case "F21", "FS5":
|
||||
return 21
|
||||
case "F22", "FS6":
|
||||
return 22
|
||||
case "F23", "FS7":
|
||||
return 23
|
||||
case "F24", "FS8":
|
||||
return 24
|
||||
case "F25", "FS9":
|
||||
return 25
|
||||
case "F26", "FS10":
|
||||
return 26
|
||||
case "F27", "FS11":
|
||||
return 27
|
||||
case "F28", "FT8":
|
||||
return 28
|
||||
case "F29", "FT9":
|
||||
return 29
|
||||
case "F30", "FT10":
|
||||
return 30
|
||||
case "F31", "FT11":
|
||||
return 31
|
||||
default:
|
||||
return -1
|
||||
}
|
||||
}
|
||||
|
||||
// RISC-V instruction encoding parameters.
|
||||
type riscvEnc struct {
|
||||
opcode uint32 // bits [6:0]
|
||||
funct3 uint32 // bits [14:12]
|
||||
funct7 uint32 // bits [31:25]
|
||||
}
|
||||
|
||||
// riscvInstrTable maps RISC-V mnemonics to their encoding.
|
||||
var riscvInstrTable = map[string]riscvEnc{
|
||||
// RV64I — R-type arithmetic/logic.
|
||||
"ADD": {0x33, 0x0, 0x00},
|
||||
"SUB": {0x33, 0x0, 0x20},
|
||||
"SLL": {0x33, 0x1, 0x00},
|
||||
"SLT": {0x33, 0x2, 0x00},
|
||||
"SLTU": {0x33, 0x3, 0x00},
|
||||
"XOR": {0x33, 0x4, 0x00},
|
||||
"SRL": {0x33, 0x5, 0x00},
|
||||
"SRA": {0x33, 0x5, 0x20},
|
||||
"OR": {0x33, 0x6, 0x00},
|
||||
"AND": {0x33, 0x7, 0x00},
|
||||
// RV64I — 32-bit variants (W suffix).
|
||||
"ADDW": {0x3B, 0x0, 0x00},
|
||||
"SUBW": {0x3B, 0x0, 0x20},
|
||||
"SLLW": {0x3B, 0x1, 0x00},
|
||||
"SRLW": {0x3B, 0x5, 0x00},
|
||||
"SRAW": {0x3B, 0x5, 0x20},
|
||||
// RV64I — I-type shift-immediate (shamt in rs2 field).
|
||||
"SLLI": {0x13, 0x1, 0x00},
|
||||
"SRLI": {0x13, 0x5, 0x00},
|
||||
"SRAI": {0x13, 0x5, 0x20},
|
||||
"SLLIW": {0x1B, 0x1, 0x00},
|
||||
"SRLIW": {0x1B, 0x5, 0x00},
|
||||
"SRAIW": {0x1B, 0x5, 0x20},
|
||||
// RV64M — multiply/divide.
|
||||
"MUL": {0x33, 0x0, 0x01},
|
||||
"MULH": {0x33, 0x1, 0x01},
|
||||
"MULHSU": {0x33, 0x2, 0x01},
|
||||
"MULHU": {0x33, 0x3, 0x01},
|
||||
"DIV": {0x33, 0x4, 0x01},
|
||||
"DIVU": {0x33, 0x5, 0x01},
|
||||
"REM": {0x33, 0x6, 0x01},
|
||||
"REMU": {0x33, 0x7, 0x01},
|
||||
// RV64M — 32-bit variants.
|
||||
"MULW": {0x3B, 0x0, 0x01},
|
||||
"DIVW": {0x3B, 0x4, 0x01},
|
||||
"DIVUW": {0x3B, 0x5, 0x01},
|
||||
"REMW": {0x3B, 0x6, 0x01},
|
||||
"REMUW": {0x3B, 0x7, 0x01},
|
||||
// RV64I — I-type arithmetic.
|
||||
"ADDI": {0x13, 0x0, 0x00},
|
||||
"ADDIW": {0x1B, 0x0, 0x00},
|
||||
"SLTI": {0x13, 0x2, 0x00},
|
||||
"SLTIU": {0x13, 0x3, 0x00},
|
||||
"XORI": {0x13, 0x4, 0x00},
|
||||
"ORI": {0x13, 0x6, 0x00},
|
||||
"ANDI": {0x13, 0x7, 0x00},
|
||||
// Loads (I-type).
|
||||
"LB": {0x03, 0x0, 0x00},
|
||||
"LH": {0x03, 0x1, 0x00},
|
||||
"LW": {0x03, 0x2, 0x00},
|
||||
"LD": {0x03, 0x3, 0x00},
|
||||
"LBU": {0x03, 0x4, 0x00},
|
||||
"LHU": {0x03, 0x5, 0x00},
|
||||
"LWU": {0x03, 0x6, 0x00},
|
||||
// Stores (S-type).
|
||||
"SB": {0x23, 0x0, 0x00},
|
||||
"SH": {0x23, 0x1, 0x00},
|
||||
"SW": {0x23, 0x2, 0x00},
|
||||
"SD": {0x23, 0x3, 0x00},
|
||||
// Branches (B-type).
|
||||
"BEQ": {0x63, 0x0, 0x00},
|
||||
"BNE": {0x63, 0x1, 0x00},
|
||||
"BLT": {0x63, 0x4, 0x00},
|
||||
"BGE": {0x63, 0x5, 0x00},
|
||||
"BLTU": {0x63, 0x6, 0x00},
|
||||
"BGEU": {0x63, 0x7, 0x00},
|
||||
// U-type.
|
||||
"LUI": {0x37, 0x0, 0x00},
|
||||
"AUIPC": {0x17, 0x0, 0x00},
|
||||
// System.
|
||||
"ECALL": {0x73, 0x0, 0x00},
|
||||
"EBREAK": {0x73, 0x0, 0x00},
|
||||
"FENCE": {0x0F, 0x0, 0x00},
|
||||
// JALR — indirect jump/call (I-type).
|
||||
"JALR": {0x67, 0x0, 0x00},
|
||||
|
||||
// RV64A — atomics (AMO opcode 0x2F).
|
||||
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
|
||||
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
|
||||
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
|
||||
"AMOADDW": {0x2F, 0x2, 0x00 << 2},
|
||||
"AMOADDD": {0x2F, 0x3, 0x00 << 2},
|
||||
"AMOANDW": {0x2F, 0x2, 0x0C << 2},
|
||||
"AMOANDD": {0x2F, 0x3, 0x0C << 2},
|
||||
"AMOORW": {0x2F, 0x2, 0x06 << 2},
|
||||
"AMOORD": {0x2F, 0x3, 0x06 << 2},
|
||||
"AMOXORW": {0x2F, 0x2, 0x04 << 2},
|
||||
"AMOXORD": {0x2F, 0x3, 0x04 << 2},
|
||||
"AMOMAXW": {0x2F, 0x2, 0x14 << 2},
|
||||
"AMOMAXD": {0x2F, 0x3, 0x14 << 2},
|
||||
"AMOMINW": {0x2F, 0x2, 0x10 << 2},
|
||||
"AMOMIND": {0x2F, 0x3, 0x10 << 2},
|
||||
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2},
|
||||
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2},
|
||||
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
|
||||
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
|
||||
|
||||
// RV64F/D — floating-point arithmetic.
|
||||
"FADDS": {0x53, 0x0, 0x00},
|
||||
"FSUBS": {0x53, 0x0, 0x04},
|
||||
"FMULS": {0x53, 0x0, 0x08},
|
||||
"FDIVS": {0x53, 0x0, 0x0C},
|
||||
"FADDD": {0x53, 0x0, 0x01},
|
||||
"FSUBD": {0x53, 0x0, 0x05},
|
||||
"FMULD": {0x53, 0x0, 0x09},
|
||||
"FDIVD": {0x53, 0x0, 0x0D},
|
||||
"FSQRTS": {0x53, 0x0, 0x2C},
|
||||
"FSQRTD": {0x53, 0x0, 0x2D},
|
||||
// FP loads/stores.
|
||||
"FLW": {0x07, 0x2, 0x00},
|
||||
"FLD": {0x07, 0x3, 0x00},
|
||||
"FSW": {0x27, 0x2, 0x00},
|
||||
"FSD": {0x27, 0x3, 0x00},
|
||||
// FP min/max.
|
||||
"FMINS": {0x53, 0x0, 0x14},
|
||||
"FMAXS": {0x53, 0x1, 0x14},
|
||||
"FMIND": {0x53, 0x0, 0x15},
|
||||
"FMAXD": {0x53, 0x1, 0x15},
|
||||
|
||||
// RV64A — load-reserved / store-conditional (funct5 0x02 / 0x03).
|
||||
"LRW": {0x2F, 0x2, 0x02 << 2},
|
||||
"LRD": {0x2F, 0x3, 0x02 << 2},
|
||||
"SCW": {0x2F, 0x2, 0x03 << 2},
|
||||
"SCD": {0x2F, 0x3, 0x03 << 2},
|
||||
|
||||
// FP compare — result in integer register (funct7 0x50/0x51).
|
||||
"FEQS": {0x53, 0x2, 0x50},
|
||||
"FLTS": {0x53, 0x1, 0x50},
|
||||
"FLES": {0x53, 0x0, 0x50},
|
||||
"FEQD": {0x53, 0x2, 0x51},
|
||||
"FLTD": {0x53, 0x1, 0x51},
|
||||
"FLED": {0x53, 0x0, 0x51},
|
||||
}
|
||||
|
||||
// riscvRType encodes an R-type instruction: funct7 | rs2 | rs1 | funct3 | rd | opcode.
|
||||
func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// riscvAMOType encodes an atomic (AMO) instruction.
|
||||
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode.
|
||||
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2).
|
||||
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value
|
||||
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// FP conversion instructions (FCVT, FMV). These use the rs2 field to
|
||||
// encode the conversion type rather than a register, so they are handled
|
||||
// separately from the general instruction table.
|
||||
type riscvCvtEnc struct {
|
||||
funct7 uint32 // bits [31:25]
|
||||
rs2 uint32 // conversion-type code in bits [24:20]
|
||||
opcode uint32 // always 0x53 (OP-FP)
|
||||
}
|
||||
|
||||
var riscvCvtTable = map[string]riscvCvtEnc{
|
||||
// float → int (rs2 selects the integer width/sign).
|
||||
"FCVTWS": {0x60, 0x0, 0x53}, // float32 → int32
|
||||
"FCVTWUS": {0x60, 0x1, 0x53}, // float32 → uint32
|
||||
"FCVTLS": {0x60, 0x2, 0x53}, // float32 → int64
|
||||
"FCVTLUS": {0x60, 0x3, 0x53}, // float32 → uint64
|
||||
"FCVTWD": {0x61, 0x0, 0x53}, // float64 → int32
|
||||
"FCVTWUD": {0x61, 0x1, 0x53}, // float64 → uint32
|
||||
"FCVTLD": {0x61, 0x2, 0x53}, // float64 → int64
|
||||
"FCVTLUD": {0x61, 0x3, 0x53}, // float64 → uint64
|
||||
// int → float (rs2 selects the integer width/sign).
|
||||
"FCVTSW": {0x68, 0x0, 0x53}, // int32 → float32
|
||||
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
|
||||
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
|
||||
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
|
||||
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
|
||||
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
|
||||
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
|
||||
"FCVTDLU": {0x69, 0x3, 0x53}, // uint64 → float64
|
||||
// float → float width conversion.
|
||||
"FCVTSD": {0x20, 0x1, 0x53}, // float64 → float32
|
||||
"FCVTDS": {0x21, 0x0, 0x53}, // float32 → float64
|
||||
// Bit moves between integer and FP registers (no conversion).
|
||||
"FMVXD": {0x71, 0x0, 0x53}, // float64 → int64 (bit move)
|
||||
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
|
||||
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
|
||||
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
|
||||
}
|
||||
|
||||
// riscvCvtType encodes an FP conversion instruction.
|
||||
// Layout: funct7 | rs2(convtype) | rs1 | funct3(0) | rd | opcode.
|
||||
func riscvCvtType(enc riscvCvtEnc, rd, rs1 int) uint32 {
|
||||
return (enc.funct7 << 25) | (enc.rs2 << 20) | (uint32(rs1) << 15) |
|
||||
(uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// R4-type fused multiply-add instructions (FMADD/FMSUB/FNMSUB/FNMADD).
|
||||
// These take 4 register operands: rs1, rs2, rs3, rd.
|
||||
// Layout: rs3 | fmt | rs2 | rs1 | rm | rd | opcode.
|
||||
type riscvFmaEnc struct {
|
||||
fmt uint32 // bits [26:25]: 0x0 = single, 0x1 = double
|
||||
opcode uint32 // bits [6:0]
|
||||
}
|
||||
|
||||
var riscvFmaTable = map[string]riscvFmaEnc{
|
||||
"FMADDS": {0x0, 0x43}, // rd = rs1*rs2 + rs3
|
||||
"FMADDD": {0x1, 0x43},
|
||||
"FMSUBS": {0x0, 0x47}, // rd = rs1*rs2 - rs3
|
||||
"FMSUBD": {0x1, 0x47},
|
||||
"FNMSUBS": {0x0, 0x4B}, // rd = -(rs1*rs2) + rs3
|
||||
"FNMSUBD": {0x1, 0x4B},
|
||||
"FNMADDS": {0x0, 0x4F}, // rd = -(rs1*rs2) - rs3
|
||||
"FNMADDD": {0x1, 0x4F},
|
||||
}
|
||||
|
||||
// riscvFmaType encodes an R4-type fused multiply-add instruction.
|
||||
func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 {
|
||||
return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) |
|
||||
(uint32(rs1) << 15) | (0x0 << 12) /* rm=dynamic */ | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// CSR (Control and Status Register) instructions.
|
||||
// Format: csr[11:0] | rs1/zimm | funct3 | rd | opcode (0x73).
|
||||
type riscvCsrEnc struct {
|
||||
funct3 uint32 // bits [14:12]
|
||||
imm bool // true for CSRRWI/CSRRSI/CSRRCI (5-bit uimm variant)
|
||||
}
|
||||
|
||||
var riscvCsrTable = map[string]riscvCsrEnc{
|
||||
"CSRRW": {0x1, false}, // rd=CSR, CSR=rs1
|
||||
"CSRRS": {0x2, false}, // rd=CSR, CSR |= rs1
|
||||
"CSRRC": {0x3, false}, // rd=CSR, CSR &= ~rs1
|
||||
"CSRRWI": {0x5, true}, // rd=CSR, CSR=uimm
|
||||
"CSRRSI": {0x6, true}, // rd=CSR, CSR |= uimm
|
||||
"CSRRCI": {0x7, true}, // rd=CSR, CSR &= ~uimm
|
||||
}
|
||||
|
||||
// riscvCsrType encodes a CSR instruction.
|
||||
// csr is the 12-bit CSR address; src is either a register number or a 5-bit
|
||||
// unsigned immediate (depending on enc.imm).
|
||||
func riscvCsrType(enc riscvCsrEnc, rd, src int, csr int32) uint32 {
|
||||
return (uint32(csr&0xFFF) << 20) | (uint32(src&0x1F) << 15) |
|
||||
(enc.funct3 << 12) | (uint32(rd) << 7) | 0x73
|
||||
}
|
||||
|
||||
// riscvIType encodes an I-type instruction: imm[11:0] | rs1 | funct3 | rd | opcode.
|
||||
func riscvIType(enc riscvEnc, rd, rs1 int, imm int32) uint32 {
|
||||
return (uint32(imm&0xFFF) << 20) | (uint32(rs1) << 15) |
|
||||
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// riscvSType encodes an S-type instruction: imm[11:5] | rs2 | rs1 | funct3 | imm[4:0] | opcode.
|
||||
func riscvSType(enc riscvEnc, rs1, rs2 int, imm int32) uint32 {
|
||||
immU := uint32(imm) & 0xFFF
|
||||
return ((immU >> 5) << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||
(enc.funct3 << 12) | ((immU & 0x1F) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// riscvBType encodes a B-type instruction (branches).
|
||||
func riscvBType(enc riscvEnc, rs1, rs2 int, offset int32) uint32 {
|
||||
imm := uint32(offset) & 0x1FFE // bits [12:1], bit 0 is always 0
|
||||
return (((imm >> 12) & 1) << 31) | // imm[12]
|
||||
(((imm >> 5) & 0x3F) << 25) | // imm[10:5]
|
||||
(uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||
(enc.funct3 << 12) |
|
||||
(((imm >> 1) & 0xF) << 8) | // imm[4:1]
|
||||
(((imm >> 11) & 1) << 7) | // imm[11]
|
||||
enc.opcode
|
||||
}
|
||||
|
||||
// riscvUType encodes a U-type instruction: imm[31:12] | rd | opcode.
|
||||
func riscvUType(enc riscvEnc, rd int, imm int32) uint32 {
|
||||
return (uint32(imm) & 0xFFFFF000) | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// riscvJType encodes a J-type instruction (JAL).
|
||||
func riscvJType(rd int, offset int32) uint32 {
|
||||
imm := uint32(offset) & 0x1FFFFE // bits [20:1]
|
||||
return (((imm >> 20) & 1) << 31) | // imm[20]
|
||||
(((imm >> 1) & 0x3FF) << 21) | // imm[10:1]
|
||||
(((imm >> 11) & 1) << 20) | // imm[11]
|
||||
(((imm >> 12) & 0xFF) << 12) | // imm[19:12]
|
||||
(uint32(rd) << 7) |
|
||||
0x6F // JAL opcode
|
||||
}
|
||||
|
||||
// ---- RVC (compressed) encoding helpers ----
|
||||
|
||||
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
|
||||
// prime register field used by compressed instructions (x8–x15).
|
||||
func isRVCIntReg(r int) bool { return r >= 8 && r <= 15 }
|
||||
|
||||
// rvcReg3 returns the 3-bit encoding for registers x8–x15 (0–7).
|
||||
func rvcReg3(r int) uint32 { return uint32(r - 8) }
|
||||
|
||||
// rvcCR encodes a CR-type (register) compressed instruction.
|
||||
// Format: funct4 | rd/rs1 | rs2 | op=2.
|
||||
func rvcCR(funct4, rd, rs2 uint32) uint16 {
|
||||
return uint16((funct4 << 12) | (rd << 7) | (rs2 << 2) | 0x2)
|
||||
}
|
||||
|
||||
// rvcCI encodes a CI-type (immediate) compressed instruction.
|
||||
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW — linear 6-bit immediate.
|
||||
func rvcCI(funct3, rd uint32, imm uint32) uint16 {
|
||||
return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x2)
|
||||
}
|
||||
|
||||
// rvcLSP encodes a CI-type stack-relative load: C.LDSP (funct3=3) or
|
||||
// C.FLDSP (funct3=1). offset is the full byte offset; the immediate bits
|
||||
// are interleaved per the RISC-V spec: [5:3|8:6].
|
||||
func rvcLSP(funct3, rd uint32, offset uint32) uint16 {
|
||||
// Bit interleave offset bits [5,4,3,8,7,6] → packed value.
|
||||
packed := uint32(0)
|
||||
for i, b := range []int{5, 4, 3, 8, 7, 6} {
|
||||
packed |= ((offset >> b) & 1) << (5 - i)
|
||||
}
|
||||
return uint16((funct3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x2)
|
||||
}
|
||||
|
||||
// rvcSSP encodes a CSS-type stack-relative store: C.SDSP (funct3=7) or
|
||||
// C.FSDSP (funct3=5). offset is the full byte offset; the immediate bits
|
||||
// are interleaved per the RISC-V spec: [5:3|8:6].
|
||||
func rvcSSP(funct3, rs2 uint32, offset uint32) uint16 {
|
||||
// Bit interleave offset bits [5,4,3,8,7,6] → packed value.
|
||||
packed := uint32(0)
|
||||
for i, b := range []int{5, 4, 3, 8, 7, 6} {
|
||||
packed |= ((offset >> b) & 1) << (5 - i)
|
||||
}
|
||||
return uint16((funct3 << 13) | (packed << 7) | (rs2 << 2) | 0x2)
|
||||
}
|
||||
|
||||
// rvcCSS encodes a CSS-type (stack store) compressed instruction.
|
||||
func rvcCSS(funct3, rs2 uint32, imm uint32) uint16 {
|
||||
return uint16((funct3 << 13) | (imm << 7) | (rs2 << 2) | 0x2)
|
||||
}
|
||||
|
||||
// rvcCL encodes a CL-type (load) compressed instruction.
|
||||
// imm layout: [5:3] in bits [12:10], [2|6] in bits [6:5].
|
||||
func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 {
|
||||
bits := uint16((funct3 << 13) | ((imm>>3)&0x7)<<10 | (rs1 << 7) | ((imm & 0x7) << 5) | (rd << 2) | 0x0)
|
||||
return bits
|
||||
}
|
||||
|
||||
// rvcCS encodes a CS-type (store) compressed instruction.
|
||||
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
|
||||
return uint16((funct3 << 13) | ((imm>>3)&0x7)<<10 | (rs1 << 7) | ((imm & 0x7) << 5) | (rs2 << 2) | 0x0)
|
||||
}
|
||||
|
||||
// rvcCJ encodes a CJ-type (jump) compressed instruction.
|
||||
// offset is a 12-bit signed offset (bit 0 is always 0).
|
||||
func rvcCJ(funct3 uint32, offset int32) uint16 {
|
||||
uoff := uint32(offset) & 0xFFE
|
||||
bits := ((uoff >> 11) & 1) << 10
|
||||
bits |= ((uoff >> 4) & 1) << 9
|
||||
bits |= ((uoff >> 9) & 0x3) << 7
|
||||
bits |= ((uoff >> 10) & 1) << 6
|
||||
bits |= ((uoff >> 6) & 1) << 5
|
||||
bits |= ((uoff >> 7) & 1) << 4
|
||||
bits |= ((uoff >> 1) & 0x7) << 1
|
||||
bits |= ((uoff >> 5) & 1)
|
||||
return uint16((funct3 << 13) | (bits << 2) | 0x1)
|
||||
}
|
||||
|
||||
// rvcCA encodes a CA-type (arithmetic) compressed instruction.
|
||||
// Format: funct6[15:10] | rd'/rs1'[9:7] | funct2[6:5] | rs2'[4:2] | op=01.
|
||||
func rvcCA(funct6, funct2, rd, rs2 uint32) uint16 {
|
||||
return uint16((funct6 << 10) | (rd << 7) | (funct2 << 5) | (rs2 << 2) | 0x1)
|
||||
}
|
||||
|
||||
// rvcCB encodes a CB-type (branch) compressed instruction.
|
||||
// Format: funct3[15:13] | offset[8|4:3] | rs1'[9:7] | offset[7:6|2:1|5] | op=01.
|
||||
// Bit pattern for offset: [8|4:3|7:6|2:1|5]
|
||||
func rvcCB(funct3, rs1 uint32, offset int32) uint16 {
|
||||
uoff := uint32(offset) & 0x1FE // bits [8:1]
|
||||
offBits := uint32(0)
|
||||
offBits |= ((uoff >> 8) & 1) << 10 // bit 10 = offset[8]
|
||||
offBits |= ((uoff >> 3) & 0x3) << 8 // bits 9:8 = offset[4:3]
|
||||
offBits |= ((uoff >> 6) & 0x3) << 6 // bits 7:6 = offset[7:6]
|
||||
offBits |= ((uoff >> 1) & 0x3) << 3 // bits 4:3 = offset[2:1]
|
||||
offBits |= ((uoff >> 5) & 1) << 2 // bit 2 = offset[5]
|
||||
return uint16((funct3 << 13) | offBits | (rs1 << 7) | 0x1)
|
||||
}
|
||||
@@ -0,0 +1,749 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// firstTextRISCV parses assembly source and returns the first TEXT function body.
|
||||
func firstTextRISCV(t *testing.T, src string) *ast.Text {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse("f_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
for _, d := range f.Decls {
|
||||
if fn, ok := d.(*ast.Text); ok {
|
||||
return fn
|
||||
}
|
||||
}
|
||||
t.Fatal("no TEXT found")
|
||||
return nil
|
||||
}
|
||||
|
||||
// assembleRISCVHelper assembles one TEXT function and returns its code bytes.
|
||||
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
|
||||
t.Helper()
|
||||
code, _, _, err := assembleRISCV(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
return code
|
||||
}
|
||||
|
||||
func TestRISCV_add(t *testing.T) {
|
||||
// func add(a, b int64) int64
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOV a+0(FP), X10
|
||||
MOV b+8(FP), X11
|
||||
ADD X11, X10, X10
|
||||
MOV X10, ret+16(FP)
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// should be 12 bytes with RVC: C.LDSP + C.LDSP + ADD + C.SDSP + C.JR
|
||||
_ = code
|
||||
if len(code) == 0 {
|
||||
t.Error("empty output")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_arithmetic(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·arith(SB), NOSPLIT, $0
|
||||
ADD X10, X11, X12
|
||||
SUB X12, X13, X14
|
||||
MUL X14, X15, X16
|
||||
DIV X16, X17, X18
|
||||
REM X18, X19, X20
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 5 R-type instructions + RET compressed = 5*4 + 2 = 22
|
||||
if len(code) != 22 {
|
||||
t.Errorf("expected 22 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_loadStore(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·mem(SB), NOSPLIT, $0
|
||||
LD (X10), X11
|
||||
SD X11, (X12)
|
||||
LW (X13), X14
|
||||
SW X14, (X15)
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 4 loads/stores (4B each) + C.JR RET (2B) = 18
|
||||
if len(code) != 18 {
|
||||
t.Errorf("expected 18 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_immediate(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·imm(SB), NOSPLIT, $0
|
||||
ADDI X10, $42, X11
|
||||
ANDI X11, $0xFF, X12
|
||||
ORI X12, $1, X13
|
||||
XORI X13, $0, X14
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 4 I-type + C.JR = 4*4 + 2 = 18
|
||||
if len(code) != 18 {
|
||||
t.Errorf("expected 18 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_branches(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·br(SB), NOSPLIT, $0
|
||||
ADDI X10, $1, X10
|
||||
loop:
|
||||
BEQ X10, X11, done
|
||||
ADDI X10, $1, X10
|
||||
JMP loop
|
||||
done:
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
_ = code
|
||||
if len(code) == 0 {
|
||||
t.Error("empty output")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_MOV_imm_small(t *testing.T) {
|
||||
// MOV $42, rd → ADDI (fits in 12 bits). Not RVC-compressed (treated as MOV, not ADDI).
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·small(SB), NOSPLIT, $0
|
||||
MOV $42, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// ADDI (4B) + C.JR (2B) = 6
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_MOV_imm_large(t *testing.T) {
|
||||
// MOV $0x12345, rd → LUI + ADDIW (8 bytes total)
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·large(SB), NOSPLIT, $0
|
||||
MOV $0x12345, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// LUI (4B) + ADDIW (4B) + C.JR (2B) = 10
|
||||
if len(code) != 10 {
|
||||
t.Errorf("expected 10 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_MOV_reg(t *testing.T) {
|
||||
// MOV rs, rd → ADDI $0, rs, rd, compresses to C.MV
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·reg(SB), NOSPLIT, $0
|
||||
MOV X10, X11
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.MV (2B) + C.JR (2B) = 4
|
||||
if len(code) != 4 {
|
||||
t.Errorf("expected 4 bytes, got %d (% x)", len(code), code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_MOV_frame(t *testing.T) {
|
||||
// MOV name+off(FP), rd → load with frame mapping
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·frame(SB), NOSPLIT, $0-8
|
||||
MOV a+0(FP), X10
|
||||
MOV X10, ret+0(FP)
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.LDSP (2B) + C.SDSP (2B) + C.JR (2B) = 6
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_loadStore(t *testing.T) {
|
||||
// Verify that loads/stores from SP are compressed.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·rvcstore(SB), NOSPLIT, $0
|
||||
LD 0(SP), X10
|
||||
SD X10, 8(SP)
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.LDSP (2B) + C.SDSP (2B) + C.JR (2B) = 6
|
||||
if len(code) != 6 {
|
||||
t.Errorf("expected 6 bytes, got %d (% x)", len(code), code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_atomics(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·amo(SB), NOSPLIT, $0
|
||||
AMOADDD X10, (X11), X12
|
||||
LRD (X13), X14
|
||||
SCD X15, (X16), X17
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 3 AMO instructions (4B each) + C.JR (2B) = 14
|
||||
if len(code) != 14 {
|
||||
t.Errorf("expected 14 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_fpArith(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·fpadd(SB), NOSPLIT, $0
|
||||
FADDD F10, F11, F12
|
||||
FSUBD F12, F13, F14
|
||||
FMULD F14, F15, F16
|
||||
FDIVD F16, F17, F18
|
||||
FSQRTD F18, F19
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 5 FP instructions (4B each) + C.JR (2B) = 22
|
||||
if len(code) != 22 {
|
||||
t.Errorf("expected 22 bytes, got %d (%d)", len(code), len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_csr(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·csrtest(SB), NOSPLIT, $0
|
||||
CSRRS $0x300, X0, X10
|
||||
CSRRW $0x305, X10, X11
|
||||
CSRRSI $0x304, $5, X12
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 3 CSR instructions (4B each) + C.JR (2B) = 14
|
||||
if len(code) != 14 {
|
||||
t.Errorf("expected 14 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_fma(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·fmatest(SB), NOSPLIT, $0
|
||||
FMADDD F10, F11, F12, F13
|
||||
FMSUBD F13, F14, F15, F16
|
||||
FNMSUBD F16, F17, F18, F19
|
||||
FNMADDD F19, F10, F11, F12
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 4 FMA instructions (4B each) + C.JR (2B) = 18
|
||||
if len(code) != 18 {
|
||||
t.Errorf("expected 18 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_conversions(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cvt(SB), NOSPLIT, $0
|
||||
FCVTDL X10, F10
|
||||
FCVTLD F10, X11
|
||||
FMVXD F10, X12
|
||||
FMVDX X12, F11
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 4 conversion instructions (4B each) + C.JR (2B) = 18
|
||||
if len(code) != 18 {
|
||||
t.Errorf("expected 18 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_fpCmp(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cmp(SB), NOSPLIT, $0
|
||||
FEQD F10, F11, X10
|
||||
FLTD F12, F13, X11
|
||||
FLED F14, F15, X12
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 3 FP compare (4B each) + C.JR (2B) = 14
|
||||
if len(code) != 14 {
|
||||
t.Errorf("expected 14 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_forwardBranch(t *testing.T) {
|
||||
// Forward label reference — must not fail.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·fwd(SB), NOSPLIT, $0
|
||||
ADDI X10, $1, X10
|
||||
BEQ X10, X11, done
|
||||
ADDI X10, $1, X10
|
||||
done:
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
_ = code
|
||||
if len(code) == 0 {
|
||||
t.Error("empty output")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_ADDI(t *testing.T) {
|
||||
// ADDI where rd=rs1 and small imm → C.ADDI
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·caddi(SB), NOSPLIT, $0
|
||||
ADDI X10, $5, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.ADDI (2B) + C.JR (2B) = 4
|
||||
if len(code) != 4 {
|
||||
t.Errorf("expected 4 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_LI(t *testing.T) {
|
||||
// ADDI X0, $imm, rd → C.LI
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cli(SB), NOSPLIT, $0
|
||||
ADDI X0, $7, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.LI (2B) + C.JR (2B) = 4
|
||||
if len(code) != 4 {
|
||||
t.Errorf("expected 4 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_LUI(t *testing.T) {
|
||||
// LUI rd, small nonzero imm → C.LUI
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·clui(SB), NOSPLIT, $0
|
||||
LUI X10, $1
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.LUI (2B) + C.JR (2B) = 4
|
||||
if len(code) != 4 {
|
||||
t.Errorf("expected 4 bytes, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_AssembleFile(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOV a+0(FP), X10
|
||||
RET
|
||||
|
||||
TEXT ·sub(SB), NOSPLIT, $0
|
||||
SUB X10, X11, X12
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("t_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
if len(img.Funcs) != 2 {
|
||||
t.Fatalf("expected 2 functions, got %d", len(img.Funcs))
|
||||
}
|
||||
// func add: C.LDSP(2) + C.JR(2) = 4
|
||||
if img.Funcs[0].Size != 4 {
|
||||
t.Errorf("add: expected 4 bytes, got %d", img.Funcs[0].Size)
|
||||
}
|
||||
// func sub: SUB(4) + C.JR(2) = 6
|
||||
if img.Funcs[1].Size != 6 {
|
||||
t.Errorf("sub: expected 6 bytes, got %d", img.Funcs[1].Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_encodings(t *testing.T) {
|
||||
// Smoke test that all known RISC-V mnemonics encode successfully.
|
||||
tests := []struct {
|
||||
name, src string
|
||||
wantBytes int
|
||||
}{
|
||||
{"ADD", "ADD X10, X11, X12\nRET\n", 6},
|
||||
{"SUBW", "SUBW X10, X11, X12\nRET\n", 6},
|
||||
{"MUL", "MUL X10, X11, X12\nRET\n", 6},
|
||||
{"DIVW", "DIVW X10, X11, X12\nRET\n", 6},
|
||||
{"REMUW", "REMUW X10, X11, X12\nRET\n", 6},
|
||||
{"ADDIW", "ADDIW X10, $5, X11\nRET\n", 6},
|
||||
{"SLLI", "SLLI X10, $3, X11\nRET\n", 6}, // ADDI+SLLI? No, SLLI uses I-type
|
||||
{"SRLI", "SRLI X10, $2, X11\nRET\n", 6},
|
||||
{"SRAI", "SRAI X10, $1, X11\nRET\n", 6},
|
||||
{"LB", "LB (X10), X11\nRET\n", 6},
|
||||
{"LBU", "LBU (X10), X11\nRET\n", 6},
|
||||
{"LH", "LH (X10), X11\nRET\n", 6},
|
||||
{"LHU", "LHU (X10), X11\nRET\n", 6},
|
||||
{"LWU", "LWU (X10), X11\nRET\n", 6},
|
||||
{"SB", "SB X10, (X11)\nRET\n", 6},
|
||||
{"SH", "SH X10, (X11)\nRET\n", 6},
|
||||
{"SW", "SW X10, (X11)\nRET\n", 6},
|
||||
{"LUI", "LUI X10, $0x12345\nRET\n", 6},
|
||||
{"AUIPC", "AUIPC X10, $0\nRET\n", 6},
|
||||
{"FLW", "FLW (X10), F10\nRET\n", 6},
|
||||
{"FSW", "FSW F10, (X11)\nRET\n", 6},
|
||||
{"FADDS", "FADDS F10, F11, F12\nRET\n", 6},
|
||||
{"FMINS", "FMINS F10, F11, F12\nRET\n", 6},
|
||||
{"FMAXD", "FMAXD F10, F11, F12\nRET\n", 6},
|
||||
{"FCVTSD", "FCVTSD F10, F11\nRET\n", 6},
|
||||
{"FCVTDS", "FCVTDS F10, F11\nRET\n", 6},
|
||||
{"FMVXW", "FMVXW F10, X10\nRET\n", 6},
|
||||
{"FMADD_S", "FMADDS F10, F11, F12, F13\nRET\n", 6},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·`+tt.name+`(SB), NOSPLIT, $0
|
||||
`+tt.src)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
if len(code) != tt.wantBytes {
|
||||
t.Errorf("expected %d bytes, got %d", tt.wantBytes, len(code))
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_branch(t *testing.T) {
|
||||
// BEQ rs, X0, target → C.BEQZ when rs is in prime regs and offset fits.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cbeqz(SB), NOSPLIT, $0
|
||||
ADDI X10, $1, X10
|
||||
BEQ X10, X0, done
|
||||
ADDI X10, $1, X10
|
||||
done:
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.ADDI(2) + C.BEQZ(2) + C.ADDI(2) + C.JR(2) = 8 (all compress)
|
||||
if len(code) != 8 {
|
||||
t.Errorf("expected 8 bytes with C.BEQZ, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CJ(t *testing.T) {
|
||||
// JMP target → C.J when offset fits.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cj(SB), NOSPLIT, $0
|
||||
JMP done
|
||||
done:
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.J(2) + C.JR(2) = 4
|
||||
if len(code) != 4 {
|
||||
t.Errorf("expected 4 bytes with C.J, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CADD(t *testing.T) {
|
||||
// ADD where rd==rs1 and both in prime regs → C.ADD.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cadd(SB), NOSPLIT, $0
|
||||
ADD X10, X11, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.ADD(2) + C.JR(2) = 4
|
||||
if len(code) != 4 {
|
||||
t.Errorf("expected 4 bytes with C.ADD, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CADD_commute(t *testing.T) {
|
||||
// ADD where rd==rs2 (commutative swap) → C.ADD.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cadd2(SB), NOSPLIT, $0
|
||||
ADD X11, X10, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.ADD(2) + C.JR(2) = 4
|
||||
if len(code) != 4 {
|
||||
t.Errorf("expected 4 bytes with C.ADD (commuted), got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CSUB(t *testing.T) {
|
||||
// SUB where rd==rs1 and both in prime regs → C.SUB.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·csub(SB), NOSPLIT, $0
|
||||
SUB X11, X10, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// SUB X11,X10,X10 → rd=X10, rs1=X11 ≠ rd → no C.SUB.
|
||||
// Plan9: INSTR src1, src2, dst. For C.SUB: rd must equal rs1.
|
||||
// So: SUB X10, X11, X10 → rd=10, rs1=10, rs2=11 ✓
|
||||
if len(code) == 4 {
|
||||
return // compressed
|
||||
}
|
||||
// Try with correct operand order.
|
||||
fn2 := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·csub2(SB), NOSPLIT, $0
|
||||
SUB X10, X11, X10
|
||||
RET
|
||||
`)
|
||||
code2 := assembleRISCVHelper(t, fn2)
|
||||
if len(code2) != 4 {
|
||||
t.Errorf("expected 4 bytes with C.SUB, got %d (% x)", len(code2), code2)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CXOR(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cxor(SB), NOSPLIT, $0
|
||||
XOR X10, X11, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
if len(code) != 4 {
|
||||
t.Errorf("expected 4 bytes with C.XOR, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_COR(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cor(SB), NOSPLIT, $0
|
||||
OR X10, X11, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
if len(code) != 4 {
|
||||
t.Errorf("expected 4 bytes with C.OR, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CAND(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cand(SB), NOSPLIT, $0
|
||||
AND X10, X11, X10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
if len(code) != 4 {
|
||||
t.Errorf("expected 4 bytes with C.AND, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CFLDSP(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cfldsp(SB), NOSPLIT, $0-8
|
||||
FLD a+0(FP), F10
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.FLDSP(2) + C.JR(2) = 4
|
||||
if len(code) != 4 {
|
||||
t.Errorf("expected 4 bytes with C.FLDSP, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_RVC_CFSDSP(t *testing.T) {
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·cfsdsp(SB), NOSPLIT, $0-8
|
||||
FSD F10, ret+0(FP)
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// C.FSDSP(2) + C.JR(2) = 4
|
||||
if len(code) != 4 {
|
||||
t.Errorf("expected 4 bytes with C.FSDSP, got %d", len(code))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_SB_addr(t *testing.T) {
|
||||
// MOV $sym<>(SB), rd → AUIPC + ADDI (8 bytes for SB).
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·sbaddr(SB), NOSPLIT, $0
|
||||
MOV $answer<>(SB), X10
|
||||
RET
|
||||
GLOBL answer<>(SB), RODATA, $8
|
||||
DATA answer<>+0(SB)/8, $42
|
||||
`
|
||||
f, errs := parser.Parse("t_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
// AUIPC(4) + ADDI(4) + C.JR(2) = 10
|
||||
if img.Funcs[0].Size != 10 {
|
||||
t.Errorf("expected 10 bytes, got %d", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_SB_store(t *testing.T) {
|
||||
// MOV rd, sym<>(SB) → AUIPC + SD (8 bytes for SB).
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·sbstore(SB), NOSPLIT, $0
|
||||
MOV X10, result<>(SB)
|
||||
RET
|
||||
GLOBL result<>(SB), NOPTR, $8
|
||||
`
|
||||
f, errs := parser.Parse("t_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
// AUIPC X31(4) + SD X10,0(X31)(4) + C.JR(2) = 10
|
||||
if img.Funcs[0].Size != 10 {
|
||||
t.Errorf("expected 10 bytes, got %d", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_ELF(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·simple(SB), NOSPLIT, $0
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("t_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
obj, err := img.ELFRISCVObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFRISCVObject: %v", err)
|
||||
}
|
||||
if len(obj) < 4 || obj[0] != 0x7f || obj[1] != 'E' || obj[2] != 'L' || obj[3] != 'F' {
|
||||
t.Fatal("not a valid ELF file")
|
||||
}
|
||||
if len(obj) >= 20 {
|
||||
machine := uint16(obj[18]) | uint16(obj[19])<<8
|
||||
if machine != 243 {
|
||||
t.Errorf("e_machine = %d, want 243 (EM_RISCV)", machine)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_ELF_withData(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·get(SB), NOSPLIT, $0
|
||||
RET
|
||||
GLOBL val<>(SB), RODATA, $4
|
||||
DATA val<>+0(SB)/4, $7
|
||||
`
|
||||
f, errs := parser.Parse("t_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
if len(img.DataSyms) != 1 {
|
||||
t.Fatalf("expected 1 data symbol, got %d", len(img.DataSyms))
|
||||
}
|
||||
if img.DataSyms[0].Name != "val" {
|
||||
t.Errorf("data symbol name = %q, want val", img.DataSyms[0].Name)
|
||||
}
|
||||
if img.DataSyms[0].Size != 4 {
|
||||
t.Errorf("data symbol size = %d, want 4", img.DataSyms[0].Size)
|
||||
}
|
||||
obj, err := img.ELFRISCVObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFRISCVObject: %v", err)
|
||||
}
|
||||
_ = obj
|
||||
}
|
||||
|
||||
func TestRISCV_SB_load(t *testing.T) {
|
||||
// MOV sym<>(SB), rd → AUIPC + LD (8 bytes for SB).
|
||||
src := `#include "textflag.h"
|
||||
TEXT ·sbload(SB), NOSPLIT, $0
|
||||
MOV answer<>(SB), X10
|
||||
RET
|
||||
GLOBL answer<>(SB), RODATA, $8
|
||||
DATA answer<>+0(SB)/8, $42
|
||||
`
|
||||
f, errs := parser.Parse("t_riscv64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
// AUIPC(4) + LD(4) + C.JR(2) = 10
|
||||
if img.Funcs[0].Size != 10 {
|
||||
t.Errorf("expected 10 bytes, got %d", img.Funcs[0].Size)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_system_instrs(t *testing.T) {
|
||||
// Test FENCE, ECALL, EBREAK encoding.
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·sys(SB), NOSPLIT, $0
|
||||
FENCE
|
||||
ECALL
|
||||
EBREAK
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// 3 system instructions × 4 bytes + C.JR(2) = 14
|
||||
if len(code) != 14 {
|
||||
t.Errorf("expected 14 bytes, got %d (% x)", len(code), code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_MOV_sym_FP_error(t *testing.T) {
|
||||
// MOV $sym(FP), rd should return an error (unsupported).
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·badfp(SB), NOSPLIT, $0
|
||||
MOV $arg(FP), X10
|
||||
RET
|
||||
`)
|
||||
_, _, _, err := assembleRISCV(fn)
|
||||
if err == nil {
|
||||
t.Error("expected error for MOV $arg(FP), got nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCV_CALL(t *testing.T) {
|
||||
// CALL target → AUIPC + JALR (8 bytes).
|
||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||
TEXT ·calltest(SB), NOSPLIT, $0
|
||||
CALL sub
|
||||
done:
|
||||
RET
|
||||
sub:
|
||||
RET
|
||||
`)
|
||||
code := assembleRISCVHelper(t, fn)
|
||||
// CALL(8) + C.JR(2) + C.JR(2) = 12
|
||||
if len(code) != 12 {
|
||||
t.Errorf("expected 12 bytes with CALL, got %d", len(code))
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,117 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import "sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
|
||||
// RISC-V frame mapping: translates Go's FP/SP pseudo-register addressing
|
||||
// into real RISC-V memory accesses.
|
||||
//
|
||||
// In Go's ABI0 (used by assembly functions), arguments are passed on the
|
||||
// stack. At function entry the return address sits at SP, so the frame
|
||||
// pointer FP == SP+8 and the first argument is at FP+0 == SP+8.
|
||||
//
|
||||
// On RISC-V the hardware registers are:
|
||||
// SP = X2 (stack pointer)
|
||||
// FP = S0 = X8 (frame pointer, by convention)
|
||||
//
|
||||
// For NOSPLIT $0 functions the prologue is omitted and arguments are read
|
||||
// directly from SP+8+offset.
|
||||
|
||||
// riscvFrameInfo holds the frame parameters computed from a TEXT directive.
|
||||
type riscvFrameInfo struct {
|
||||
frameSize int // the $framesize from TEXT
|
||||
argsSize int // the -argsize from TEXT
|
||||
noSplit bool // the NOSPLIT flag
|
||||
}
|
||||
|
||||
// riscvComputeFrame extracts frame information from a TEXT directive.
|
||||
func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
|
||||
fi := riscvFrameInfo{}
|
||||
fi.frameSize = frameSize(t)
|
||||
fi.argsSize = argsSize(t)
|
||||
for _, f := range t.Flags {
|
||||
if f == "NOSPLIT" {
|
||||
fi.noSplit = true
|
||||
}
|
||||
}
|
||||
return fi
|
||||
}
|
||||
|
||||
// riscvPrologue returns the prologue bytes for a RISC-V function.
|
||||
// For NOSPLIT $0 functions there is no prologue. For functions with a
|
||||
// frame, we emit: ADDI SP, SP, -framesize; SD S0, (framesize-8)(SP); ...
|
||||
func riscvPrologue(fi riscvFrameInfo) []byte {
|
||||
if fi.noSplit && fi.frameSize == 0 {
|
||||
return nil // no prologue for NOSPLIT $0
|
||||
}
|
||||
var out []byte
|
||||
if fi.frameSize > 0 {
|
||||
// ADDI SP, SP, -framesize
|
||||
out = append(out, riscvITypeLE(0x13, 0x0, 2, 2, int32(-fi.frameSize))...)
|
||||
// Save the frame pointer (S0 = X8) at the top of the new frame.
|
||||
// SD S0, (framesize-8)(SP)
|
||||
out = append(out, riscvSTypeLE(0x23, 0x3, 2, 8, int32(fi.frameSize-8))...)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// riscvEpilogue returns the epilogue bytes for a RISC-V function.
|
||||
func riscvEpilogue(fi riscvFrameInfo) []byte {
|
||||
if fi.noSplit && fi.frameSize == 0 {
|
||||
return nil
|
||||
}
|
||||
var out []byte
|
||||
if fi.frameSize > 0 {
|
||||
// Restore the frame pointer: LD S0, (framesize-8)(SP)
|
||||
out = append(out, riscvITypeLE(0x03, 0x3, 8, 2, int32(fi.frameSize-8))...)
|
||||
// ADDI SP, SP, framesize
|
||||
out = append(out, riscvITypeLE(0x13, 0x0, 2, 2, int32(fi.frameSize))...)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// riscvResolvePseudo translates a pseudo-register memory reference into a
|
||||
// real base register and offset. It handles name+offset(FP) and
|
||||
// name+offset(SP).
|
||||
//
|
||||
// Returns the base register number and the adjusted offset.
|
||||
func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32) {
|
||||
if sym == nil {
|
||||
return -1, 0
|
||||
}
|
||||
offset := int32(sym.Offset)
|
||||
switch sym.Pseudo {
|
||||
case "FP":
|
||||
// FP == SP+8 for NOSPLIT $0; arguments are at SP+8+offset.
|
||||
if fi.noSplit && fi.frameSize == 0 {
|
||||
return 2, 8 + offset // SP + 8 + argOffset
|
||||
}
|
||||
// With a frame, FP points to the saved frame; args are at FP+offset.
|
||||
return 8, offset // S0 + argOffset
|
||||
case "SP":
|
||||
// SP-relative; the offset is from the current SP.
|
||||
return 2, offset
|
||||
case "SB":
|
||||
// Static data reference — needs a relocation (not yet supported).
|
||||
return -1, offset
|
||||
default:
|
||||
return -1, offset
|
||||
}
|
||||
}
|
||||
|
||||
// riscvITypeLE encodes an I-type instruction and returns little-endian bytes.
|
||||
func riscvITypeLE(opcode, funct3 uint32, rd, rs1 int, imm int32) []byte {
|
||||
word := (uint32(imm&0xFFF) << 20) | (uint32(rs1) << 15) |
|
||||
(funct3 << 12) | (uint32(rd) << 7) | opcode
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
|
||||
}
|
||||
|
||||
// riscvSTypeLE encodes an S-type instruction and returns little-endian bytes.
|
||||
func riscvSTypeLE(opcode, funct3 uint32, rs1, rs2 int, imm int32) []byte {
|
||||
immU := uint32(imm) & 0xFFF
|
||||
word := ((immU >> 5) << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||
(funct3 << 12) | ((immU & 0x1F) << 7) | opcode
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
|
||||
}
|
||||
+210
-3
@@ -39,6 +39,17 @@ const (
|
||||
// source lives in the reg field, the destination in r/m — the PEXTR-style
|
||||
// layout. VEXTRACTI128 and VEXTRACTF128 use this shape.
|
||||
vexExtract
|
||||
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
||||
// in ModRM.reg and the destination in r/m — the layout of the EVEX
|
||||
// narrowing stores (VPMOVDW, VPMOVQD).
|
||||
vexRMRev
|
||||
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
||||
// vector length follows the source: the packed-double → dword
|
||||
// conversions (VCVTPD2DQ/VCVTTPD2DQ and their X/Y spellings) narrow into
|
||||
// an XMM destination, so the L bit rides with the wider source. The
|
||||
// mnemonic's spelling fixes the length (X = 128, Y = 256), which also
|
||||
// covers a memory source. ModRM.reg = dst, ModRM.rm = src, no vvvv.
|
||||
vexRMSrcLen
|
||||
// vexZero is the no-operand form (VZEROUPPER).
|
||||
vexZero
|
||||
)
|
||||
@@ -80,14 +91,38 @@ var vexTable = map[string]vexSpec{
|
||||
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
|
||||
|
||||
// VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic.
|
||||
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
|
||||
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
|
||||
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
|
||||
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
|
||||
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
|
||||
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
|
||||
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
|
||||
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
|
||||
// VEX.128/256.0F.WIG — packed single-precision arithmetic.
|
||||
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
|
||||
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
|
||||
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
|
||||
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3},
|
||||
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3},
|
||||
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3},
|
||||
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
|
||||
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
|
||||
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
|
||||
// VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed
|
||||
// opcodes with an F2 pp).
|
||||
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
|
||||
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
|
||||
"VMULSD": {1, 0x59, 0, 3, -1, vexNDS3},
|
||||
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
|
||||
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
|
||||
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
|
||||
// VEX.128.F3.0F.WIG — scalar single-precision arithmetic (the packed
|
||||
// opcodes with an F3 pp).
|
||||
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
|
||||
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
|
||||
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3},
|
||||
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
|
||||
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
|
||||
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
||||
// VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form).
|
||||
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
||||
|
||||
@@ -95,12 +130,33 @@ var vexTable = map[string]vexSpec{
|
||||
// no vvvv).
|
||||
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
||||
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
||||
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM},
|
||||
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM},
|
||||
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM},
|
||||
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
||||
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
|
||||
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM},
|
||||
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM},
|
||||
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM},
|
||||
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
|
||||
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
||||
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
||||
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
|
||||
// (reg=dst, rm=src, no vvvv; the length follows the destination).
|
||||
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
|
||||
// VEX.128/256.0F.WIG — signed dword to packed single conversion
|
||||
// (reg=dst, rm=src, no vvvv, no mandatory prefix).
|
||||
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
|
||||
// VEX.128/256.0F.WIG — packed single to packed double conversion
|
||||
// (reg=dst, rm=src; the destination is the wide operand and sets the
|
||||
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
|
||||
// the Go assembler emits the instruction with pp = 00, and gasm follows
|
||||
// the Go assembler's bytes — its machine code is the oracle, not the
|
||||
// manual.
|
||||
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
|
||||
// VEX.128.F2.0F.WIG — duplicate the low double of each 128-bit lane
|
||||
// (reg=dst, rm=src, no vvvv; the length follows the destination).
|
||||
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
|
||||
// VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src).
|
||||
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
|
||||
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
|
||||
@@ -128,9 +184,86 @@ var vexTable = map[string]vexSpec{
|
||||
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
|
||||
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
|
||||
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
|
||||
// VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst:
|
||||
// reg=src, rm=XMM/memory dst, imm8 — the extract layout).
|
||||
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
|
||||
|
||||
// VEX.128.0F.W0 — no operands.
|
||||
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
||||
|
||||
// VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
||||
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
||||
|
||||
// VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst,
|
||||
// rm=scalar memory; SD is 256-bit only).
|
||||
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
||||
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
||||
// VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width
|
||||
// source).
|
||||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
||||
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
|
||||
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
||||
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
||||
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y
|
||||
// spellings: the destination is always XMM and the spelling fixes the
|
||||
// source length (X = 128, Y = 256).
|
||||
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||
|
||||
// VEX scalar conversions between vector and general-purpose registers.
|
||||
// Vector to GPR (two operands: vec/mem source, GPR destination, vvvv
|
||||
// unused; the length follows the source).
|
||||
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM},
|
||||
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM},
|
||||
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM},
|
||||
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM},
|
||||
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM},
|
||||
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM},
|
||||
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM},
|
||||
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM},
|
||||
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
||||
// vector source in vvvv, vector destination in reg).
|
||||
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3},
|
||||
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3},
|
||||
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
|
||||
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
|
||||
|
||||
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
|
||||
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
||||
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
|
||||
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
|
||||
|
||||
// VEX.F2.0F — packed double to packed dword conversions, truncating and
|
||||
// non-truncating. The destination is always XMM; the X/Y spellings fix
|
||||
// the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen.
|
||||
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
||||
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
||||
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||
}
|
||||
|
||||
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
|
||||
// the packed-double → dword conversions) to its fixed vector length:
|
||||
// X = 128 (L = 0), Y = 256 (L = 1). The spelling fixes the length even for
|
||||
// a memory source, matching the Go assembler's ytab.
|
||||
var vexSrcLen = map[string]int{
|
||||
"VCVTPD2DQX": 0,
|
||||
"VCVTPD2DQY": 1,
|
||||
"VCVTTPD2DQX": 0,
|
||||
"VCVTTPD2DQY": 1,
|
||||
"VCVTPD2PSX": 0,
|
||||
"VCVTPD2PSY": 1,
|
||||
}
|
||||
|
||||
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
|
||||
// form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8),
|
||||
// an ordinary NDS encoding rather than the /digit immediate form above.
|
||||
var vexVarShift = map[string]byte{
|
||||
"VPSLLD": 0xF2,
|
||||
"VPSLLQ": 0xF3,
|
||||
"VPSRAD": 0xE2,
|
||||
"VPSRLD": 0xD2,
|
||||
"VPSRLQ": 0xD3,
|
||||
}
|
||||
|
||||
// vexMoveSpec describes a VEX move, which takes different opcodes (and
|
||||
@@ -164,6 +297,11 @@ var vexMoveTable = map[string]vexMoveSpec{
|
||||
// VEX.128.F2.0F.WIG — scalar double move, memory operands only (the
|
||||
// register form takes three operands and is not supported yet).
|
||||
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
|
||||
// VEX.128.F3.0F.WIG — scalar single move, memory operands only.
|
||||
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
|
||||
// VEX.128/256 — aligned packed moves.
|
||||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
|
||||
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
|
||||
}
|
||||
|
||||
// isVex reports whether the mnemonic is a VEX-encoded instruction we handle.
|
||||
@@ -177,9 +315,27 @@ func isVex(mnemUpper string) bool {
|
||||
|
||||
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
|
||||
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
||||
// Vector register indices 16–31 exist only in EVEX encodings; fail
|
||||
// loudly rather than silently truncating the index.
|
||||
for _, op := range ops {
|
||||
if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 {
|
||||
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
||||
}
|
||||
}
|
||||
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
||||
return e.encodeVexMove(mnemUpper, ms, ops)
|
||||
}
|
||||
// The shifts come in two shapes under one mnemonic: an immediate count
|
||||
// ($imm, src, dst) and a variable count in an XMM register or memory
|
||||
// (count, src, dst), the latter an ordinary NDS form.
|
||||
if op, ok := vexVarShift[mnemUpper]; ok && len(ops) == 3 {
|
||||
if _, isImm := ops[0].(Imm); !isImm {
|
||||
if !vecOrMem(ops[0]) {
|
||||
return fmt.Errorf("%s: shift count must be an immediate, a vector register or memory", mnemUpper)
|
||||
}
|
||||
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
|
||||
}
|
||||
}
|
||||
spec := vexTable[mnemUpper]
|
||||
switch spec.form {
|
||||
case vexNDS3:
|
||||
@@ -194,6 +350,8 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
||||
return e.encodeVexNDS3Imm(spec, ops)
|
||||
case vexExtract:
|
||||
return e.encodeVexExtract(spec, ops)
|
||||
case vexRMSrcLen:
|
||||
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
||||
case vexZero:
|
||||
return e.encodeVexZero(mnemUpper, spec, ops)
|
||||
}
|
||||
@@ -259,6 +417,32 @@ func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
|
||||
return e.emitVexFields(spec, l, regField, rBit, 15, src)
|
||||
}
|
||||
|
||||
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
|
||||
// the destination always XMM and the VEX.L bit following the source — fixed
|
||||
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
|
||||
// the source is memory.
|
||||
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("VEX destination must be a vector register")
|
||||
}
|
||||
ll, ok := vexSrcLen[mnem]
|
||||
if !ok {
|
||||
return fmt.Errorf("no fixed vector length for %s", mnem)
|
||||
}
|
||||
regField := dstReg.idx & 7
|
||||
rBit := 0
|
||||
if dstReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
// An unused vvvv field must be stored as all ones (v̄vvv = 1111).
|
||||
return e.emitVexFields(spec, ll, regField, rBit, 15, src)
|
||||
}
|
||||
|
||||
// encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst.
|
||||
// The destination is carried in VEX.vvvv, the source in ModRM.rm, and the
|
||||
// shift kind in the ModRM.reg /digit.
|
||||
@@ -483,11 +667,21 @@ func vecReg(op Operand) (Reg, bool) {
|
||||
return r, ok && r.isVec()
|
||||
}
|
||||
|
||||
// vecOrMem reports whether op is a vector register or a memory reference.
|
||||
func vecOrMem(op Operand) bool {
|
||||
switch op.(type) {
|
||||
case Mem, sbMem:
|
||||
return true
|
||||
}
|
||||
r, ok := op.(Reg)
|
||||
return ok && r.isVec()
|
||||
}
|
||||
|
||||
// validMoveOther reports whether the non-vector operand of a move is
|
||||
// acceptable: memory always is, a GPR only for VMOVD/VMOVQ.
|
||||
func validMoveOther(ms vexMoveSpec, op Operand) bool {
|
||||
switch o := op.(type) {
|
||||
case Mem:
|
||||
case Mem, sbMem:
|
||||
return true
|
||||
case Reg:
|
||||
return ms.gprOK && !o.isVec()
|
||||
@@ -499,9 +693,13 @@ func validMoveOther(ms vexMoveSpec, op Operand) bool {
|
||||
// the given precomputed fields. It is shared by every register/rm VEX form;
|
||||
// immediate bytes are appended by the caller.
|
||||
func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Operand) error {
|
||||
if l > 1 {
|
||||
return fmt.Errorf("ZMM operand requires an EVEX instruction")
|
||||
}
|
||||
var modrm, sib int
|
||||
var disp []byte
|
||||
var xBit, bBit int
|
||||
var sb *sbRef
|
||||
switch r := rm.(type) {
|
||||
case Reg:
|
||||
modrm = 0xC0 | regField<<3 | (r.idx & 7)
|
||||
@@ -515,6 +713,12 @@ func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Ope
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
case sbMem:
|
||||
// RIP-relative static-symbol reference; disp32 patched at link time.
|
||||
modrm = regField<<3 | 0x05
|
||||
sib = -1
|
||||
disp = le32(0)
|
||||
sb = &sbRef{name: r.name, addend: r.addend}
|
||||
default:
|
||||
return fmt.Errorf("invalid VEX r/m operand")
|
||||
}
|
||||
@@ -530,6 +734,9 @@ func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Ope
|
||||
if sib >= 0 {
|
||||
e.out = append(e.out, byte(sib))
|
||||
}
|
||||
if sb != nil {
|
||||
e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend})
|
||||
}
|
||||
e.out = append(e.out, disp...)
|
||||
return nil
|
||||
}
|
||||
|
||||
+115
-61
@@ -39,11 +39,14 @@ func TestVexNDS3(t *testing.T) {
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(% x): %v", mnem, code, err)
|
||||
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
||||
continue
|
||||
}
|
||||
if inst.Op.String() != mnem {
|
||||
t.Errorf("%s: decoded as %s (% x)", mnem, inst.Op.String(), code)
|
||||
// The decoder folds the Plan 9 L/Q GPR-width spellings (VCVTSI2SDL/
|
||||
// SDQ, SSL/SSQ) onto the base name; the W bit carries the width.
|
||||
got := inst.Op.String()
|
||||
if got != mnem && !(len(mnem) > len(got) && mnem[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s (% x)", mnem, got, code)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -156,74 +159,121 @@ func TestVexShiftImm(t *testing.T) {
|
||||
// as well as every new operand form.
|
||||
func TestVexGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
wantOp string // decoded mnemonic, when it differs from mnem (the X/Y spellings)
|
||||
}{
|
||||
// Three-operand NDS form.
|
||||
{"VPADDQ Y8,Y9,Y8", "VPADDQ", []Operand{vreg(t, "Y8"), vreg(t, "Y9"), vreg(t, "Y8")}, "c44135d4c0"},
|
||||
{"VPADDQ X9,X8,X8", "VPADDQ", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c44139d4c1"},
|
||||
{"VPXOR X7,X7,X7", "VPXOR", []Operand{vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7")}, "c5c1efff"},
|
||||
{"VPSHUFB Y1,Y2,Y3", "VPSHUFB", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d00d9"},
|
||||
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9"},
|
||||
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec"},
|
||||
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9"},
|
||||
{"VPADDQ Y8,Y9,Y8", "VPADDQ", []Operand{vreg(t, "Y8"), vreg(t, "Y9"), vreg(t, "Y8")}, "c44135d4c0", ""},
|
||||
{"VPADDQ X9,X8,X8", "VPADDQ", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c44139d4c1", ""},
|
||||
{"VPXOR X7,X7,X7", "VPXOR", []Operand{vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7")}, "c5c1efff", ""},
|
||||
{"VPSHUFB Y1,Y2,Y3", "VPSHUFB", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d00d9", ""},
|
||||
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""},
|
||||
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""},
|
||||
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""},
|
||||
// Floating point (packed and scalar) and FMA — same NDS form, the pp
|
||||
// bits and map select the operation.
|
||||
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1"},
|
||||
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9"},
|
||||
{"VMULPD Y12,Y12,Y12", "VMULPD", []Operand{vreg(t, "Y12"), vreg(t, "Y12"), vreg(t, "Y12")}, "c4411d59e4"},
|
||||
{"VXORPD Y8,Y8,Y8", "VXORPD", []Operand{vreg(t, "Y8"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d57c0"},
|
||||
{"VUNPCKHPD X8,X8,X9", "VUNPCKHPD", []Operand{vreg(t, "X8"), vreg(t, "X8"), vreg(t, "X9")}, "c4413915c8"},
|
||||
{"VADDSD X9,X8,X8", "VADDSD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c4413b58c1"},
|
||||
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8"},
|
||||
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6"},
|
||||
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807"},
|
||||
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""},
|
||||
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""},
|
||||
{"VMULPD Y12,Y12,Y12", "VMULPD", []Operand{vreg(t, "Y12"), vreg(t, "Y12"), vreg(t, "Y12")}, "c4411d59e4", ""},
|
||||
{"VXORPD Y8,Y8,Y8", "VXORPD", []Operand{vreg(t, "Y8"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d57c0", ""},
|
||||
{"VUNPCKHPD X8,X8,X9", "VUNPCKHPD", []Operand{vreg(t, "X8"), vreg(t, "X8"), vreg(t, "X9")}, "c4413915c8", ""},
|
||||
{"VADDSD X9,X8,X8", "VADDSD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c4413b58c1", ""},
|
||||
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
|
||||
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
|
||||
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
|
||||
// Two-operand reg/rm form (v̄vvv must be 1111).
|
||||
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0"},
|
||||
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306"},
|
||||
{"VPBROADCASTD X0,Y15", "VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, "c4627d58f8"},
|
||||
{"VCVTDQ2PD X12,Y12", "VCVTDQ2PD", []Operand{vreg(t, "X12"), vreg(t, "Y12")}, "c4417ee6e4"},
|
||||
{"VCVTDQ2PD (SI),Y4", "VCVTDQ2PD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y4")}, "c5fee626"},
|
||||
{"VPMOVMSKB X11,AX", "VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, "c4c179d7c3"},
|
||||
{"VMOVMSKPS Y7,AX", "VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, "c5fc50c7"},
|
||||
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
|
||||
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
|
||||
{"VPBROADCASTD X0,Y15", "VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, "c4627d58f8", ""},
|
||||
{"VCVTDQ2PD X12,Y12", "VCVTDQ2PD", []Operand{vreg(t, "X12"), vreg(t, "Y12")}, "c4417ee6e4", ""},
|
||||
{"VCVTDQ2PD (SI),Y4", "VCVTDQ2PD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y4")}, "c5fee626", ""},
|
||||
{"VPMOVMSKB X11,AX", "VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, "c4c179d7c3", ""},
|
||||
{"VMOVMSKPS Y7,AX", "VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, "c5fc50c7", ""},
|
||||
// Immediate shifts.
|
||||
{"VPSLLD $1,Y3,Y4", "VPSLLD", []Operand{Imm(1), vreg(t, "Y3"), vreg(t, "Y4")}, "c5dd72f301"},
|
||||
{"VPSRLQ $2,Y5,Y6", "VPSRLQ", []Operand{Imm(2), vreg(t, "Y5"), vreg(t, "Y6")}, "c5cd73d502"},
|
||||
{"VPSLLD $1,Y3,Y4", "VPSLLD", []Operand{Imm(1), vreg(t, "Y3"), vreg(t, "Y4")}, "c5dd72f301", ""},
|
||||
{"VPSRLQ $2,Y5,Y6", "VPSRLQ", []Operand{Imm(2), vreg(t, "Y5"), vreg(t, "Y6")}, "c5cd73d502", ""},
|
||||
// Variable-count shifts: the count lives in an XMM register or memory
|
||||
// and the instruction takes the NDS form.
|
||||
{"VPSRLQ X0,Y8,Y8", "VPSRLQ", []Operand{vreg(t, "X0"), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd3c0", ""},
|
||||
{"VPSRLQ (AX),Y8,Y8", "VPSRLQ", []Operand{Ptr(AX, 0, 16), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd300", ""},
|
||||
{"VPSLLD X0,Y1,Y2", "VPSLLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f2d0", ""},
|
||||
{"VPSRLD X0,Y1,Y2", "VPSRLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5d2d0", ""},
|
||||
{"VPSRAD X0,Y1,Y2", "VPSRAD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5e2d0", ""},
|
||||
{"VPSLLQ X0,Y1,Y2", "VPSLLQ", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f3d0", ""},
|
||||
// Immediate shuffle (reg=dst, rm=src, imm8).
|
||||
{"VPSHUFD $0xEE,X8,X9", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")}, "c4417970c8ee"},
|
||||
{"VPSHUFD $0xEE,Y1,Y2", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "Y1"), vreg(t, "Y2")}, "c5fd70d1ee"},
|
||||
{"VPERMQ $0x1B,Y1,Y2", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e3fd00d11b"},
|
||||
{"VPERMQ $0x1B,Y11,Y12", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y11"), vreg(t, "Y12")}, "c443fd00e31b"},
|
||||
{"VPSHUFD $0xEE,X8,X9", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")}, "c4417970c8ee", ""},
|
||||
{"VPSHUFD $0xEE,Y1,Y2", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "Y1"), vreg(t, "Y2")}, "c5fd70d1ee", ""},
|
||||
{"VPERMQ $0x1B,Y1,Y2", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e3fd00d11b", ""},
|
||||
{"VPERMQ $0x1B,Y11,Y12", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y11"), vreg(t, "Y12")}, "c443fd00e31b", ""},
|
||||
// Three-operand + immediate (reg=dst, vvvv=src1, rm=src2, imm8).
|
||||
{"VSHUFPD $1,X1,X2,X3", "VSHUFPD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e9c6d901"},
|
||||
{"VSHUFPD $1,Y1,Y2,Y3", "VSHUFPD", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5edc6d901"},
|
||||
{"VPERM2I128 $0x31,Y1,Y2,Y3", "VPERM2I128", []Operand{Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e36d46d931"},
|
||||
{"VINSERTI128 $1,X5,Y1,Y2", "VINSERTI128", []Operand{Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37538d501"},
|
||||
{"VSHUFPD $1,X1,X2,X3", "VSHUFPD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e9c6d901", ""},
|
||||
{"VSHUFPD $1,Y1,Y2,Y3", "VSHUFPD", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5edc6d901", ""},
|
||||
{"VPERM2I128 $0x31,Y1,Y2,Y3", "VPERM2I128", []Operand{Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e36d46d931", ""},
|
||||
{"VINSERTI128 $1,X5,Y1,Y2", "VINSERTI128", []Operand{Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37538d501", ""},
|
||||
// Lane extract (reg=YMM source, rm=XMM/memory destination, imm8).
|
||||
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101"},
|
||||
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701"},
|
||||
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101"},
|
||||
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""},
|
||||
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""},
|
||||
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""},
|
||||
// Moves — each direction picks its own opcode and VEX.W.
|
||||
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e"},
|
||||
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f"},
|
||||
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca"},
|
||||
{"VMOVUPD (DI),Y14", "VMOVUPD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y14")}, "c57d1037"},
|
||||
{"VMOVUPD Y14,(DI)", "VMOVUPD", []Operand{vreg(t, "Y14"), Ptr(DI, 0, 32)}, "c57d1137"},
|
||||
{"VMOVUPD X1,X2", "VMOVUPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f911ca"},
|
||||
{"VMOVQ X8,AX", "VMOVQ", []Operand{vreg(t, "X8"), AX}, "c461f97ec0"},
|
||||
{"VMOVQ AX,X9", "VMOVQ", []Operand{AX, vreg(t, "X9")}, "c461f96ec8"},
|
||||
{"VMOVQ X8,(DI)", "VMOVQ", []Operand{vreg(t, "X8"), Ptr(DI, 0, 8)}, "c461f97e07"},
|
||||
{"VMOVQ (SI),X9", "VMOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X9")}, "c461f96e0e"},
|
||||
{"VMOVQ X8,X2", "VMOVQ", []Operand{vreg(t, "X8"), vreg(t, "X2")}, "c579d6c2"},
|
||||
{"VMOVQ X2,X8", "VMOVQ", []Operand{vreg(t, "X2"), vreg(t, "X8")}, "c4c179d6d0"},
|
||||
{"VMOVD X0,(SI)", "VMOVD", []Operand{vreg(t, "X0"), Ptr(SI, 0, 4)}, "c5f97e06"},
|
||||
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0"},
|
||||
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006"},
|
||||
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106"},
|
||||
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""},
|
||||
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""},
|
||||
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""},
|
||||
{"VMOVUPD (DI),Y14", "VMOVUPD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y14")}, "c57d1037", ""},
|
||||
{"VMOVUPD Y14,(DI)", "VMOVUPD", []Operand{vreg(t, "Y14"), Ptr(DI, 0, 32)}, "c57d1137", ""},
|
||||
{"VMOVUPD X1,X2", "VMOVUPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f911ca", ""},
|
||||
{"VMOVQ X8,AX", "VMOVQ", []Operand{vreg(t, "X8"), AX}, "c461f97ec0", ""},
|
||||
{"VMOVQ AX,X9", "VMOVQ", []Operand{AX, vreg(t, "X9")}, "c461f96ec8", ""},
|
||||
{"VMOVQ X8,(DI)", "VMOVQ", []Operand{vreg(t, "X8"), Ptr(DI, 0, 8)}, "c461f97e07", ""},
|
||||
{"VMOVQ (SI),X9", "VMOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X9")}, "c461f96e0e", ""},
|
||||
{"VMOVQ X8,X2", "VMOVQ", []Operand{vreg(t, "X8"), vreg(t, "X2")}, "c579d6c2", ""},
|
||||
{"VMOVQ X2,X8", "VMOVQ", []Operand{vreg(t, "X2"), vreg(t, "X8")}, "c4c179d6d0", ""},
|
||||
{"VMOVD X0,(SI)", "VMOVD", []Operand{vreg(t, "X0"), Ptr(SI, 0, 4)}, "c5f97e06", ""},
|
||||
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""},
|
||||
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""},
|
||||
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""},
|
||||
// Packed double arithmetic and unpack — the NDS form, the opcode
|
||||
// selects the operation.
|
||||
{"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""},
|
||||
{"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""},
|
||||
{"VMINPD Y1,Y2,Y3", "VMINPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5dd9", ""},
|
||||
{"VMAXPD X4,X5,X6", "VMAXPD", []Operand{vreg(t, "X4"), vreg(t, "X5"), vreg(t, "X6")}, "c5d15ff4", ""},
|
||||
{"VUNPCKLPD X1,X2,X3", "VUNPCKLPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e914d9", ""},
|
||||
{"VUNPCKLPD Y1,Y2,Y3", "VUNPCKLPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed14d9", ""},
|
||||
{"VSUBPD (AX),X1,X2", "VSUBPD", []Operand{Ptr(AX, 0, 16), vreg(t, "X1"), vreg(t, "X2")}, "c5f15c10", ""},
|
||||
// Scalar double and single arithmetic (F2 / F3 pp, 128-bit only).
|
||||
{"VSUBSD X1,X2,X3", "VSUBSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5eb5cd9", ""},
|
||||
{"VDIVSD X7,X1,X2", "VDIVSD", []Operand{vreg(t, "X7"), vreg(t, "X1"), vreg(t, "X2")}, "c5f35ed7", ""},
|
||||
{"VMINSD X1,X2,X3", "VMINSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5eb5dd9", ""},
|
||||
{"VMAXSD X3,X4,X5", "VMAXSD", []Operand{vreg(t, "X3"), vreg(t, "X4"), vreg(t, "X5")}, "c5db5feb", ""},
|
||||
{"VADDSS X1,X2,X3", "VADDSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea58d9", ""},
|
||||
{"VSUBSS X1,X2,X3", "VSUBSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5cd9", ""},
|
||||
{"VMULSS X9,X10,X11", "VMULSS", []Operand{vreg(t, "X9"), vreg(t, "X10"), vreg(t, "X11")}, "c4412a59d9", ""},
|
||||
{"VDIVSS X1,X2,X3", "VDIVSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5ed9", ""},
|
||||
{"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""},
|
||||
{"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""},
|
||||
{"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""},
|
||||
// VMOVDDUP — duplicate the low double (reg=dst, rm=src, F2 pp).
|
||||
{"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""},
|
||||
{"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""},
|
||||
{"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""},
|
||||
// Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3
|
||||
// prefix — see the table comment), DQ→PD.
|
||||
{"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""},
|
||||
{"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""},
|
||||
{"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""},
|
||||
{"VCVTPS2PD X1,Y2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c5fc5ad1", ""},
|
||||
// PD→DQ conversions: the X/Y spellings fix the source length and the
|
||||
// destination is always XMM; the decoder reports the base mnemonic.
|
||||
{"VCVTPD2DQX X1,X2", "VCVTPD2DQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fbe6d1", "VCVTPD2DQ"},
|
||||
{"VCVTPD2DQY Y1,X2", "VCVTPD2DQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5ffe6d1", "VCVTPD2DQ"},
|
||||
{"VCVTTPD2DQX X3,X4", "VCVTTPD2DQX", []Operand{vreg(t, "X3"), vreg(t, "X4")}, "c5f9e6e3", "VCVTTPD2DQ"},
|
||||
{"VCVTTPD2DQY Y5,X6", "VCVTTPD2DQY", []Operand{vreg(t, "Y5"), vreg(t, "X6")}, "c5fde6f5", "VCVTTPD2DQ"},
|
||||
{"VCVTPD2DQY (AX),X1", "VCVTPD2DQY", []Operand{Ptr(AX, 0, 32), vreg(t, "X1")}, "c5ffe608", "VCVTPD2DQ"},
|
||||
// No-operand.
|
||||
{"VZEROUPPER", "VZEROUPPER", nil, "c5f877"},
|
||||
{"VZEROUPPER", "VZEROUPPER", nil, "c5f877", ""},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
@@ -243,7 +293,11 @@ func TestVexGroundTruth(t *testing.T) {
|
||||
if inst.Len != len(code) {
|
||||
t.Errorf("%s: Decode consumed %d of %d bytes", c.name, inst.Len, len(code))
|
||||
}
|
||||
if inst.Op.String() != c.mnem {
|
||||
wantOp := c.wantOp
|
||||
if wantOp == "" {
|
||||
wantOp = c.mnem
|
||||
}
|
||||
if inst.Op.String() != wantOp {
|
||||
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,98 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"sort"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/debug"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
||||
)
|
||||
|
||||
func cmdDebug(args []string) int {
|
||||
fs := newCommand("debug", "gasm debug <file.s> --func <name>", `
|
||||
Interactive debugger for JIT-assembled amd64 functions. Launches the
|
||||
function in a traced subprocess (ptrace), then provides a REPL for
|
||||
single-stepping, breakpoints, register and memory inspection.
|
||||
|
||||
REPL commands:
|
||||
break <label|addr> set a breakpoint at a label or absolute address
|
||||
step [n] single-step n instructions (default 1)
|
||||
continue run until next breakpoint or exit
|
||||
regs print general-purpose registers
|
||||
x [addr] [len] hex-dump memory (default: current PC, 64 bytes)
|
||||
labels list function labels and offsets
|
||||
quit kill the debuggee and exit
|
||||
`)
|
||||
target := fs.Bool("target", false, "") // hidden: debuggee subprocess mode
|
||||
funcName := fs.String("func", "", "function to debug")
|
||||
argsFile := fs.String("args", "", "file containing the ABI0 argument block")
|
||||
fs.Parse(args)
|
||||
|
||||
// --- Debuggee mode (internal, spawned by the debugger) ---
|
||||
if *target {
|
||||
tmpDir := os.Getenv("GASM_DEBUG_TMP")
|
||||
if tmpDir == "" || fs.NArg() < 1 || *funcName == "" || *argsFile == "" {
|
||||
fmt.Fprintln(os.Stderr, "gasm debug --target: internal mode")
|
||||
return 2
|
||||
}
|
||||
if err := debug.RunTarget(fs.Arg(0), *funcName, *argsFile, tmpDir); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// --- Debugger mode (interactive REPL) ---
|
||||
if fs.NArg() < 1 || *funcName == "" {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm debug <file.s> --func <name>")
|
||||
return 2
|
||||
}
|
||||
path := fs.Arg(0)
|
||||
|
||||
// Load the kernel to extract function metadata and labels.
|
||||
k, err := verify.Load(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer k.Close()
|
||||
|
||||
fl, err := k.Func(*funcName)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
// Build the label list for the REPL.
|
||||
var labels []debug.Label
|
||||
for name, off := range fl.Labels {
|
||||
labels = append(labels, debug.Label{Name: name, Offset: off})
|
||||
}
|
||||
sort.Slice(labels, func(i, j int) bool { return labels[i].Offset < labels[j].Offset })
|
||||
|
||||
// Launch the debuggee with a zeroed argument block.
|
||||
argBlock := make([]byte, fl.Args)
|
||||
sess, err := debug.Launch("", path, *funcName, argBlock)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer sess.Kill()
|
||||
|
||||
bm := debug.NewBreakpoints(sess)
|
||||
fmt.Printf("gasm debug: %s in %s (pid %d)\n", *funcName, path, sess.Pid())
|
||||
|
||||
// Convert the line table for the REPL.
|
||||
var srcLines []debug.SourceLine
|
||||
for _, le := range fl.Lines {
|
||||
srcLines = append(srcLines, debug.SourceLine{Offset: le.Offset, Line: le.Line})
|
||||
}
|
||||
debug.REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, labels, srcLines)
|
||||
return 0
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build !(linux && amd64)
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
)
|
||||
|
||||
func cmdDebug(args []string) int {
|
||||
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires linux/amd64 (ptrace)")
|
||||
return 1
|
||||
}
|
||||
+648
-51
@@ -8,11 +8,17 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"io/fs"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"syscall"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||
@@ -22,11 +28,12 @@ import (
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lsp"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
||||
)
|
||||
|
||||
// version is the release version, stamped at build time via
|
||||
// -ldflags "-X main.version=…" (defaulting to the current release).
|
||||
var version = "0.2.0"
|
||||
var version = "0.28.0"
|
||||
|
||||
func main() {
|
||||
if len(os.Args) < 2 {
|
||||
@@ -44,31 +51,116 @@ func main() {
|
||||
os.Exit(cmdLint(os.Args[2:]))
|
||||
case "asm":
|
||||
os.Exit(cmdAsm(os.Args[2:]))
|
||||
case "verify":
|
||||
os.Exit(cmdVerify(os.Args[2:]))
|
||||
case "debug":
|
||||
os.Exit(cmdDebug(os.Args[2:]))
|
||||
case "lsp":
|
||||
os.Exit(cmdLSP(os.Args[2:]))
|
||||
case "version", "--version", "-V":
|
||||
fmt.Printf("gasm %s\n", version)
|
||||
case "help", "-h", "--help":
|
||||
os.Exit(cmdVersion())
|
||||
case "help", "--help", "-h":
|
||||
usage(os.Stdout)
|
||||
default:
|
||||
fmt.Fprintf(os.Stderr, "gasm: unknown command %q\n\n", os.Args[1])
|
||||
usage(os.Stderr)
|
||||
fmt.Fprintf(os.Stderr, "gasm: unknown command %q — run \"gasm --help\" for usage\n", os.Args[1])
|
||||
os.Exit(2)
|
||||
}
|
||||
}
|
||||
|
||||
func usage(w io.Writer) {
|
||||
fmt.Fprintf(w, `gasm %s — developer tooling for Go's Plan 9 assembler
|
||||
// cmdVersion prints the release version.
|
||||
func cmdVersion() int {
|
||||
fmt.Printf("gasm %s\n", version)
|
||||
return 0
|
||||
}
|
||||
|
||||
Usage:
|
||||
gasm tokens <file> print the lexical token stream
|
||||
gasm parse <file> parse and report syntax errors
|
||||
gasm fmt [-w] <file...> canonicalise formatting (-w writes in place)
|
||||
gasm lint <file...> run static checks
|
||||
gasm asm [-o out.bin] <file> assemble to machine code (amd64, Phase 2)
|
||||
gasm lsp run the language server over stdio
|
||||
gasm version print the version
|
||||
`, version)
|
||||
// ANSI color helpers for terminal output.
|
||||
const (
|
||||
colorReset = "\033[0m"
|
||||
colorBold = "\033[1m"
|
||||
colorCyan = "\033[36m"
|
||||
colorYellow = "\033[33m"
|
||||
colorGray = "\033[90m"
|
||||
)
|
||||
|
||||
// isTTY reports whether the writer is a terminal (for color output).
|
||||
func isTTY(w io.Writer) bool {
|
||||
if f, ok := w.(*os.File); ok {
|
||||
stat, _ := f.Stat()
|
||||
return (stat.Mode() & os.ModeCharDevice) != 0
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func usage(w io.Writer) {
|
||||
useColor := isTTY(w)
|
||||
bold, cyan, yellow, gray, reset := "", "", "", "", ""
|
||||
if useColor {
|
||||
bold, cyan, yellow, gray, reset = colorBold, colorCyan, colorYellow, colorGray, colorReset
|
||||
}
|
||||
|
||||
fmt.Fprintf(w, "%sgasm %s%s — developer tooling for Go's Plan 9 assembler (GAsm)%s\n\n", bold, version, reset, reset)
|
||||
fmt.Fprintf(w, "gasm bundles a lexer, parser, formatter, linter, standalone assembler and\n")
|
||||
fmt.Fprintf(w, "language server for Plan 9 assembly into one self-contained binary.\n\n")
|
||||
|
||||
fmt.Fprintf(w, "%sUsage:%s\n", yellow, reset)
|
||||
fmt.Fprintf(w, " gasm <command> [arguments]\n")
|
||||
fmt.Fprintf(w, " gasm [flags]\n\n")
|
||||
|
||||
fmt.Fprintf(w, "%sCommands:%s\n", yellow, reset)
|
||||
commands := []struct{ name, desc string }{
|
||||
{"tokens", "print the lexical token stream"},
|
||||
{"parse", "parse and report syntax errors"},
|
||||
{"fmt", "canonicalise formatting (gofmt for assembly)"},
|
||||
{"lint", "run static checks"},
|
||||
{"asm", "assemble .s files to machine code (amd64, riscv64)"},
|
||||
{"verify", "JIT-assemble and run dynamic checks (amd64, riscv64)"},
|
||||
{"debug", "interactive source-level debugger (amd64)"},
|
||||
{"lsp", "run the language server over stdio"},
|
||||
{"version", "print the version (same as --version)"},
|
||||
}
|
||||
for _, c := range commands {
|
||||
fmt.Fprintf(w, " %s%-10s%s %s%s%s\n", cyan, c.name, reset, gray, c.desc, reset)
|
||||
}
|
||||
|
||||
fmt.Fprintf(w, "\n%sFlags:%s\n", yellow, reset)
|
||||
fmt.Fprintf(w, " %s-h, --help%s %sshow this help%s\n", cyan, reset, gray, reset)
|
||||
fmt.Fprintf(w, " %s-V, --version%s %sprint the version%s\n", cyan, reset, gray, reset)
|
||||
|
||||
fmt.Fprintf(w, "\nRun \"gasm <command> -h\" for a command's usage and flags.\n\n")
|
||||
|
||||
fmt.Fprintf(w, "%sExamples:%s\n", yellow, reset)
|
||||
examples := []struct{ cmd, desc string }{
|
||||
{"gasm fmt", "reformat every .s below the current directory"},
|
||||
{"gasm lint go-flac/*.s", "run static checks over the kernels"},
|
||||
{"gasm asm -o k.bin kern_amd64.s", ""},
|
||||
{"gasm asm --format elf -o k.o kern_amd64.s", ""},
|
||||
{"gasm asm --format goobj -p pkg/path -o k.o kern_amd64.s", ""},
|
||||
}
|
||||
for _, e := range examples {
|
||||
if e.desc != "" {
|
||||
fmt.Fprintf(w, " %s%s%s %s%s%s\n", cyan, e.cmd, reset, gray, e.desc, reset)
|
||||
} else {
|
||||
fmt.Fprintf(w, " %s%s%s\n", cyan, e.cmd, reset)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// newCommand returns the FlagSet of a subcommand whose -h/--help prints a
|
||||
// proper usage block: the one-line usage, the long description and the flag
|
||||
// defaults. The flag package routes -h/--help to fs.Usage and exits 0.
|
||||
func newCommand(name, usageLine, long string) *flag.FlagSet {
|
||||
fs := flag.NewFlagSet(name, flag.ExitOnError)
|
||||
fs.Usage = func() {
|
||||
w := fs.Output()
|
||||
fmt.Fprintf(w, "Usage: %s\n\n%s\n", usageLine, strings.TrimSpace(long))
|
||||
hasFlags := false
|
||||
fs.VisitAll(func(*flag.Flag) { hasFlags = true })
|
||||
if hasFlags {
|
||||
fmt.Fprintln(w, "\nFlags:")
|
||||
fs.PrintDefaults()
|
||||
}
|
||||
}
|
||||
return fs
|
||||
}
|
||||
|
||||
// readSource returns the contents of path, or stdin when path is "-".
|
||||
@@ -82,7 +174,10 @@ func readSource(path string) (string, error) {
|
||||
}
|
||||
|
||||
func cmdTokens(args []string) int {
|
||||
fs := flag.NewFlagSet("tokens", flag.ExitOnError)
|
||||
fs := newCommand("tokens", "gasm tokens <file>", `
|
||||
Print the lexical token stream of FILE: position, token kind and text, one
|
||||
token per line. FILE may be "-" to read standard input.
|
||||
`)
|
||||
fs.Parse(args)
|
||||
if fs.NArg() != 1 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm tokens <file>")
|
||||
@@ -100,7 +195,11 @@ func cmdTokens(args []string) int {
|
||||
}
|
||||
|
||||
func cmdParse(args []string) int {
|
||||
fs := flag.NewFlagSet("parse", flag.ExitOnError)
|
||||
fs := newCommand("parse", "gasm parse <file>", `
|
||||
Parse FILE and report syntax errors on stderr. On success, print how many
|
||||
declarations and TEXT functions the file contains. FILE may be "-" to read
|
||||
standard input.
|
||||
`)
|
||||
fs.Parse(args)
|
||||
if fs.NArg() != 1 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm parse <file>")
|
||||
@@ -130,15 +229,49 @@ func cmdParse(args []string) int {
|
||||
}
|
||||
|
||||
func cmdFmt(args []string) int {
|
||||
fs := flag.NewFlagSet("fmt", flag.ExitOnError)
|
||||
fs := newCommand("fmt", "gasm fmt [-w] [path...]", `
|
||||
Canonicalise the formatting of Plan 9 assembly sources: indentation, operand
|
||||
spacing, per-function mnemonic alignment and blank-line layout (exactly one
|
||||
blank line before each label, TEXT and GLOBL block). Formatting is
|
||||
idempotent and preserves every line, comments included.
|
||||
|
||||
With no paths — or a directory path — every .s file below it is reformatted
|
||||
in place and the changed files are listed, the way go fmt does; "." and "_"
|
||||
directories are skipped. Explicit file paths print to stdout unless -w is
|
||||
given.
|
||||
`)
|
||||
write := fs.Bool("w", false, "write result to the source file")
|
||||
fs.Parse(args)
|
||||
if fs.NArg() == 0 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm fmt [-w] <file...>")
|
||||
return 2
|
||||
// Like go fmt: with no arguments, or with a directory argument, every .s
|
||||
// file below the directory is formatted in place and the names of the
|
||||
// changed files are listed; explicit file arguments keep the -w / stdout
|
||||
// behaviour.
|
||||
paths := fs.Args()
|
||||
dirMode := len(paths) == 0
|
||||
if dirMode {
|
||||
paths = []string{"."}
|
||||
}
|
||||
var files []string
|
||||
for _, p := range paths {
|
||||
info, err := os.Stat(p)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||
return 1
|
||||
}
|
||||
if info.IsDir() {
|
||||
dirMode = true
|
||||
found, err := asmFiles(p)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||
return 1
|
||||
}
|
||||
files = append(files, found...)
|
||||
continue
|
||||
}
|
||||
files = append(files, p)
|
||||
}
|
||||
rc := 0
|
||||
for _, path := range fs.Args() {
|
||||
for _, path := range files {
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||
@@ -146,11 +279,15 @@ func cmdFmt(args []string) int {
|
||||
continue
|
||||
}
|
||||
out := format.Source(path, src)
|
||||
if *write {
|
||||
if dirMode || *write {
|
||||
if out != src {
|
||||
if err := os.WriteFile(path, []byte(out), 0o644); err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||
rc = 1
|
||||
continue
|
||||
}
|
||||
if dirMode {
|
||||
fmt.Println(path)
|
||||
}
|
||||
}
|
||||
continue
|
||||
@@ -160,8 +297,40 @@ func cmdFmt(args []string) int {
|
||||
return rc
|
||||
}
|
||||
|
||||
// asmFiles collects the .s files below dir, skipping directories whose name
|
||||
// starts with "." or "_" — as the go tooling does, which keeps .git and
|
||||
// scratch or reference trees (e.g. _refs) untouched.
|
||||
func asmFiles(dir string) ([]string, error) {
|
||||
var out []string
|
||||
err := filepath.WalkDir(dir, func(path string, d fs.DirEntry, err error) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if d.IsDir() {
|
||||
if path != dir && (strings.HasPrefix(d.Name(), ".") || strings.HasPrefix(d.Name(), "_")) {
|
||||
return filepath.SkipDir
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if strings.HasSuffix(d.Name(), ".s") {
|
||||
out = append(out, path)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
return out, err
|
||||
}
|
||||
|
||||
func cmdLint(args []string) int {
|
||||
fs := flag.NewFlagSet("lint", flag.ExitOnError)
|
||||
fs := newCommand("lint", "gasm lint <file...>", `
|
||||
Run the static checks over the given files and print diagnostics as
|
||||
"file:line:col: severity: message [code]". The exit status is non-zero when
|
||||
an error-severity diagnostic is found; warnings (e.g. the register-clobber
|
||||
audit) do not affect it.
|
||||
|
||||
Rules include unknown-instruction, operand-count, undefined-label,
|
||||
duplicate-label, missing-ret, missing-textflag-include, abi-argsize,
|
||||
unreachable-code, register-clobber and funcdata-pcdata.
|
||||
`)
|
||||
disable := fs.String("disable", "", "comma-separated rule codes to disable")
|
||||
fs.Parse(args)
|
||||
if fs.NArg() == 0 {
|
||||
@@ -202,7 +371,13 @@ func cmdLint(args []string) int {
|
||||
}
|
||||
|
||||
func cmdLSP(args []string) int {
|
||||
fs := flag.NewFlagSet("lsp", flag.ExitOnError)
|
||||
fs := newCommand("lsp", "gasm lsp", `
|
||||
Run the language server over standard input/output: JSON-RPC 2.0 with
|
||||
Content-Length framing. Point an LSP-capable editor at the binary and
|
||||
associate it with .s files; the target architecture is inferred from the file
|
||||
suffix (_amd64.s, _arm64.s, _riscv64.s, _loong64.s). Provides completion,
|
||||
hover, document symbols, diagnostics and semantic-token highlighting.
|
||||
`)
|
||||
fs.Parse(args)
|
||||
srv := lsp.New(os.Stdin, os.Stdout)
|
||||
if err := srv.Run(); err != nil {
|
||||
@@ -213,16 +388,32 @@ func cmdLSP(args []string) int {
|
||||
}
|
||||
|
||||
func cmdAsm(args []string) int {
|
||||
fs := flag.NewFlagSet("asm", flag.ExitOnError)
|
||||
out := fs.String("o", "", "write the concatenated machine code to this file")
|
||||
fs := newCommand("asm", "gasm asm [--format raw|elf|macho|goobj] [-p pkg] [-o out] <file>", `
|
||||
Assemble FILE (amd64 or riscv64) without the Go toolchain: every TEXT function is
|
||||
encoded to machine code — scalar, VEX/AVX2 and EVEX/AVX-512 instructions,
|
||||
FP/SP frame mapping, local labels and file-local static symbols (GLOBL/DATA)
|
||||
resolved RIP-relative — and printed as a hex dump.
|
||||
|
||||
With -o the output is written to a file instead. The --format flag selects
|
||||
what is written: raw (the default) concatenates the functions and the data
|
||||
section into one self-consistent image; elf and macho emit a relocatable
|
||||
object (.text/.data sections, a symbol table and one PC32 relocation per
|
||||
static-symbol reference) that links with the system toolchain; goobj emits
|
||||
the Go toolchain's own object format, which cmd/link consumes directly (it
|
||||
requires -p, the package path, and the installed Go toolchain).
|
||||
`)
|
||||
out := fs.String("o", "", "write the output to this file")
|
||||
format := fs.String("format", "raw", "output format: raw (concatenated image), elf, macho or goobj (Go object)")
|
||||
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
|
||||
fs.Parse(args)
|
||||
if fs.NArg() != 1 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm asm [-o out.bin] <file>")
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|macho|goobj] [-p pkg] [-o out] <file>")
|
||||
return 2
|
||||
}
|
||||
path := fs.Arg(0)
|
||||
if arch.FromFilename(path) != arch.AMD64 {
|
||||
fmt.Fprintln(os.Stderr, "gasm asm: only amd64 is supported in this Phase 2 increment")
|
||||
targetArch := arch.FromFilename(path)
|
||||
if targetArch != arch.AMD64 && targetArch != arch.RISCV {
|
||||
fmt.Fprintln(os.Stderr, "gasm asm: only amd64 and riscv64 are supported")
|
||||
return 1
|
||||
}
|
||||
src, err := readSource(path)
|
||||
@@ -238,20 +429,23 @@ func cmdAsm(args []string) int {
|
||||
return 1
|
||||
}
|
||||
|
||||
var all []byte
|
||||
functions := 0
|
||||
for _, d := range f.Decls {
|
||||
txt, ok := d.(*ast.Text)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, _, err := asm.Assemble(txt)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "%s: %s: %v\n", path, txt.Name.Name, err)
|
||||
return 1
|
||||
}
|
||||
functions++
|
||||
fmt.Printf("%s: %d bytes\n", txt.Name.Name, len(code))
|
||||
var img *asm.Image
|
||||
if targetArch == arch.RISCV {
|
||||
img, err = asm.AssembleFileRISCV(f)
|
||||
} else {
|
||||
img, err = asm.AssembleFile(f)
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
|
||||
return 1
|
||||
}
|
||||
if len(img.Funcs) == 0 {
|
||||
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions found")
|
||||
return 1
|
||||
}
|
||||
for _, fn := range img.Funcs {
|
||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
fmt.Printf("%s: %d bytes\n", fn.Name, fn.Size)
|
||||
for i := 0; i < len(code); i += 16 {
|
||||
end := i + 16
|
||||
if end > len(code) {
|
||||
@@ -263,18 +457,421 @@ func cmdAsm(args []string) int {
|
||||
}
|
||||
fmt.Println()
|
||||
}
|
||||
all = append(all, code...)
|
||||
}
|
||||
if functions == 0 {
|
||||
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions found")
|
||||
return 1
|
||||
if len(img.Data) > 0 {
|
||||
fmt.Printf("data: %d bytes at 0x%x\n", len(img.Data), len(img.Code))
|
||||
for _, d := range f.Decls {
|
||||
g, ok := d.(*ast.Globl)
|
||||
if !ok || g.Name == nil || g.Name.Pseudo != "SB" {
|
||||
continue
|
||||
}
|
||||
size := 0
|
||||
if g.Size != nil && g.Size.Imm.HasVal {
|
||||
size = int(g.Size.Imm.Val)
|
||||
}
|
||||
fmt.Printf(" %s: %d bytes at 0x%x\n", g.Name.Name, size, img.Symbols[g.Name.Name])
|
||||
}
|
||||
for i := 0; i < len(img.Data); i += 16 {
|
||||
end := i + 16
|
||||
if end > len(img.Data) {
|
||||
end = len(img.Data)
|
||||
}
|
||||
fmt.Printf(" %04x:", len(img.Code)+i)
|
||||
for _, b := range img.Data[i:end] {
|
||||
fmt.Printf(" %02x", b)
|
||||
}
|
||||
fmt.Println()
|
||||
}
|
||||
}
|
||||
if *out != "" {
|
||||
if err := os.WriteFile(*out, all, 0o644); err != nil {
|
||||
var obj []byte
|
||||
var err error
|
||||
var kind string
|
||||
switch *format {
|
||||
case "raw":
|
||||
if len(img.Externals) > 0 {
|
||||
fmt.Fprintf(os.Stderr, "gasm asm: external symbol %q needs an object file (use --format elf or --format macho)\n", img.Externals[0])
|
||||
return 1
|
||||
}
|
||||
obj, kind = img.Bytes(), "raw image"
|
||||
case "elf":
|
||||
if targetArch == arch.RISCV {
|
||||
obj, err = img.ELFRISCVObject()
|
||||
} else {
|
||||
obj, err = img.ELFObject()
|
||||
}
|
||||
kind = "ELF object"
|
||||
case "macho":
|
||||
obj, err = img.MachOObject()
|
||||
kind = "Mach-O object"
|
||||
case "goobj":
|
||||
obj, err = img.GOObject(*pkg, path)
|
||||
kind = "Go object"
|
||||
default:
|
||||
fmt.Fprintf(os.Stderr, "gasm asm: unknown format %q (want raw, elf, macho or goobj)\n", *format)
|
||||
return 2
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm asm:", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Printf("wrote %d bytes to %s\n", len(all), *out)
|
||||
if err := os.WriteFile(*out, obj, 0o644); err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm asm:", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Printf("wrote %d bytes to %s (%s)\n", len(obj), *out, kind)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// cmdVerifyRISCV handles the verify subcommand for RISC-V files.
|
||||
// JIT requires RISC-V hardware; only ground-truth and profile are available.
|
||||
func cmdVerifyRISCV(path string, groundTruth, profile bool) int {
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
f, errs := parser.Parse(path, src)
|
||||
for _, e := range errs {
|
||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
||||
}
|
||||
if len(errs) > 0 {
|
||||
return 1
|
||||
}
|
||||
img, err := asm.AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
if groundTruth {
|
||||
gt, err := verify.GroundTruthRISCV(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm verify: ground truth: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
matched, total := 0, 0
|
||||
for _, fn := range img.Funcs {
|
||||
gasmCode := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
goCode, ok := gt[fn.Name]
|
||||
if !ok {
|
||||
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", fn.Name)
|
||||
continue
|
||||
}
|
||||
total++
|
||||
gasmCmp := make([]byte, len(gasmCode))
|
||||
goCmp := make([]byte, len(goCode))
|
||||
copy(gasmCmp, gasmCode)
|
||||
copy(goCmp, goCode)
|
||||
for _, r := range fn.Relocs {
|
||||
for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ {
|
||||
gasmCmp[j] = 0
|
||||
}
|
||||
for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ {
|
||||
goCmp[j] = 0
|
||||
}
|
||||
}
|
||||
if bytes.Equal(gasmCmp, goCmp) {
|
||||
matched++
|
||||
if len(fn.Relocs) > 0 {
|
||||
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs))
|
||||
} else {
|
||||
fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size)
|
||||
}
|
||||
} else {
|
||||
fmt.Printf(" %s: MISMATCH (%d vs %d bytes)\n", fn.Name, fn.Size, len(goCode))
|
||||
for i := 0; i < len(gasmCode) || i < len(goCode); i += 16 {
|
||||
var gb, gs string
|
||||
for j := i; j < i+16 && j < len(gasmCode); j++ {
|
||||
gb += fmt.Sprintf(" %02x", gasmCode[j])
|
||||
}
|
||||
for j := i; j < i+16 && j < len(goCode); j++ {
|
||||
gs += fmt.Sprintf(" %02x", goCode[j])
|
||||
}
|
||||
fmt.Printf(" %04x: gasm:%s\n", i, gb)
|
||||
fmt.Printf(" %04x: gt: %s\n", i, gs)
|
||||
}
|
||||
}
|
||||
}
|
||||
fmt.Printf("%s: %d/%d matched\n", path, matched, total)
|
||||
if matched < total {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
if profile {
|
||||
for _, fn := range img.Funcs {
|
||||
fmt.Printf("%s: %d bytes, labels: %v\n", fn.Name, fn.Size, fn.Labels)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
fmt.Printf("%s: %d functions assembled\n", path, len(img.Funcs))
|
||||
for _, fn := range img.Funcs {
|
||||
fmt.Printf(" %s: %d bytes\n", fn.Name, fn.Size)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func cmdVerify(args []string) int {
|
||||
fs := newCommand("verify", "gasm verify [-smoke] [-abi] [-profile] <file.s>", `
|
||||
Assemble FILE (amd64), map it into executable memory and report the available
|
||||
functions. This confirms the assembled image is self-consistent (no
|
||||
unresolved external symbols) and executable — the prerequisite for dynamic
|
||||
testing.
|
||||
|
||||
With -smoke, each NOSPLIT function is called with a zeroed argument block to
|
||||
confirm the JIT trampoline works end-to-end. This is safe only for functions
|
||||
that tolerate nil pointers and zero lengths in their arguments.
|
||||
|
||||
With -abi, each function is called with sentinel values in the callee-saved
|
||||
registers (BP, R14) and a red-zone canary below SP; violations are reported.
|
||||
|
||||
With -profile, the static basic-block structure is listed for each function.
|
||||
`)
|
||||
smoke := fs.Bool("smoke", false, "call each NOSPLIT function with zeroed args")
|
||||
abi := fs.Bool("abi", false, "run ABI-checking calls (sentinel registers + red zone)")
|
||||
profile := fs.Bool("profile", false, "list basic-block structure per function")
|
||||
groundTruth := fs.Bool("ground-truth", false, "compare machine code byte-for-byte against go tool asm")
|
||||
fuzz := fs.Bool("fuzz", false, "differential fuzz: JIT both gasm and go-tool-asm versions, compare outputs")
|
||||
fuzzN := fs.Int("n", 1000, "number of fuzz iterations per function")
|
||||
fuzzOne := fs.String("fuzz-one", "", "") // hidden: fuzz a single function (subprocess mode)
|
||||
fs.Parse(args)
|
||||
if fs.NArg() != 1 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm verify [-smoke] [-abi] [-profile] <file.s>")
|
||||
return 2
|
||||
}
|
||||
path := fs.Arg(0)
|
||||
targetArch := arch.FromFilename(path)
|
||||
if targetArch != arch.AMD64 && targetArch != arch.RISCV {
|
||||
fmt.Fprintln(os.Stderr, "gasm verify: only amd64 and riscv64 are supported")
|
||||
return 1
|
||||
}
|
||||
|
||||
// RISC-V: ground-truth only (no JIT on non-RISC-V hosts).
|
||||
if targetArch == arch.RISCV {
|
||||
return cmdVerifyRISCV(path, *groundTruth, *profile)
|
||||
}
|
||||
|
||||
k, err := verify.Load(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer k.Close()
|
||||
|
||||
names := k.FuncNames()
|
||||
fmt.Printf("%s: %d functions JIT-loaded\n", path, len(names))
|
||||
rc := 0
|
||||
|
||||
// Subprocess mode: fuzz a single function and exit.
|
||||
if *fuzzOne != "" {
|
||||
gt, err := verify.GroundTruth(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
sigs := verify.ExtractSignatures(src)
|
||||
sig, ok := sigs[*fuzzOne]
|
||||
if !ok {
|
||||
fmt.Printf("%s: no signature\n", *fuzzOne)
|
||||
return 0
|
||||
}
|
||||
goCode, ok := gt[*fuzzOne]
|
||||
if !ok {
|
||||
fmt.Printf("%s: not in go tool asm\n", *fuzzOne)
|
||||
return 0
|
||||
}
|
||||
res := k.FuzzFunc(*fuzzOne, sig, goCode, *fuzzN, 42)
|
||||
fmt.Printf("%s\n", res)
|
||||
if !res.OK() {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// Ground-truth comparison: assemble with go tool asm and compare bytes.
|
||||
if *groundTruth {
|
||||
gt, err := verify.GroundTruth(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm verify: ground truth: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
matched, total := 0, 0
|
||||
for _, name := range names {
|
||||
fl, _ := k.Func(name)
|
||||
gasmCode := k.Image().Code[fl.Offset : fl.Offset+fl.Size]
|
||||
goCode, ok := gt[name]
|
||||
if !ok {
|
||||
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", name)
|
||||
continue
|
||||
}
|
||||
total++
|
||||
// Compare, masking relocation sites (disp32 fields that the
|
||||
// Go linker fills at link time — gasm resolves them internally).
|
||||
gasmCmp := make([]byte, len(gasmCode))
|
||||
goCmp := make([]byte, len(goCode))
|
||||
copy(gasmCmp, gasmCode)
|
||||
copy(goCmp, goCode)
|
||||
for _, r := range fl.Relocs {
|
||||
for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ {
|
||||
gasmCmp[j] = 0
|
||||
}
|
||||
for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ {
|
||||
goCmp[j] = 0
|
||||
}
|
||||
}
|
||||
if bytes.Equal(gasmCmp, goCmp) {
|
||||
matched++
|
||||
if len(fl.Relocs) > 0 {
|
||||
fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", name, fl.Size, len(fl.Relocs))
|
||||
} else {
|
||||
fmt.Printf(" %s: MATCH (%d bytes)\n", name, fl.Size)
|
||||
}
|
||||
} else {
|
||||
fmt.Printf(" %s: MISMATCH (gasm %d bytes, go %d bytes)\n", name, fl.Size, len(goCode))
|
||||
for i := 0; i < len(gasmCmp) && i < len(goCmp); i++ {
|
||||
if gasmCmp[i] != goCmp[i] {
|
||||
fmt.Printf(" first diff at byte %d: gasm=%02x go=%02x\n", i, gasmCmp[i], goCmp[i])
|
||||
break
|
||||
}
|
||||
}
|
||||
rc = 1
|
||||
}
|
||||
}
|
||||
fmt.Printf("ground truth: %d/%d functions byte-identical\n", matched, total)
|
||||
if matched < total {
|
||||
rc = 1
|
||||
}
|
||||
}
|
||||
|
||||
// Differential fuzz: JIT both gasm and go-tool-asm, compare outputs.
|
||||
// Each function runs in a subprocess so a crash (partial functions like
|
||||
// decoders that fault on malformed input) doesn't kill the whole run.
|
||||
if *fuzz {
|
||||
gt, err := verify.GroundTruth(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm verify: fuzz: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
sigs := verify.ExtractSignatures(src)
|
||||
fuzzed := 0
|
||||
for _, name := range names {
|
||||
sig, ok := sigs[name]
|
||||
if !ok {
|
||||
fmt.Printf(" %s: SKIP (no // func signature)\n", name)
|
||||
continue
|
||||
}
|
||||
goCode, ok := gt[name]
|
||||
if !ok {
|
||||
fmt.Printf(" %s: SKIP (not in go tool asm output)\n", name)
|
||||
continue
|
||||
}
|
||||
// Run in a subprocess: if the function crashes on random
|
||||
// input (partial function), we report it and move on.
|
||||
res := fuzzInSubprocess(path, name, *fuzzN)
|
||||
if res != "" {
|
||||
fmt.Printf(" %s\n", res)
|
||||
if strings.Contains(res, "MISMATCH") {
|
||||
rc = 1
|
||||
}
|
||||
}
|
||||
_ = sig
|
||||
_ = goCode
|
||||
fuzzed++
|
||||
}
|
||||
fmt.Printf("fuzz: %d functions tested, %d iterations each\n", fuzzed, *fuzzN)
|
||||
}
|
||||
for _, name := range names {
|
||||
fl, _ := k.Func(name)
|
||||
flags := ""
|
||||
if fl.NoSplit {
|
||||
flags = " NOSPLIT"
|
||||
}
|
||||
fmt.Printf(" %s: %d bytes, args=%d, frame=%d%s\n", name, fl.Size, fl.Args, fl.Frame, flags)
|
||||
|
||||
if *profile {
|
||||
blocks, err := k.Blocks(name)
|
||||
if err != nil {
|
||||
fmt.Printf(" profile: %v\n", err)
|
||||
} else {
|
||||
fmt.Printf(" blocks: %d\n", len(blocks))
|
||||
}
|
||||
}
|
||||
|
||||
if *smoke && fl.NoSplit {
|
||||
args := make([]byte, fl.Args)
|
||||
_, err := k.CallFunc(name, args)
|
||||
if err != nil {
|
||||
fmt.Printf(" smoke: FAIL — %v\n", err)
|
||||
rc = 1
|
||||
} else {
|
||||
fmt.Printf(" smoke: OK\n")
|
||||
}
|
||||
}
|
||||
|
||||
if *abi && fl.NoSplit {
|
||||
args := make([]byte, fl.Args)
|
||||
_, report, err := k.CallFuncChecked(name, args)
|
||||
if err != nil {
|
||||
fmt.Printf(" abi: FAIL — %v\n", err)
|
||||
rc = 1
|
||||
} else if !report.OK() {
|
||||
fmt.Printf(" abi: %s\n", report)
|
||||
rc = 1
|
||||
} else {
|
||||
fmt.Printf(" abi: clean\n")
|
||||
}
|
||||
}
|
||||
}
|
||||
return rc
|
||||
}
|
||||
|
||||
// fuzzInSubprocess runs the fuzz for a single function in a child process.
|
||||
// If the child is killed by a signal (e.g. SIGSEGV from a partial function
|
||||
// faulting on random input), it returns a CRASH report instead of dying.
|
||||
func fuzzInSubprocess(path, funcName string, n int) string {
|
||||
self, err := os.Executable()
|
||||
if err != nil {
|
||||
return fmt.Sprintf("%s: cannot find self: %v", funcName, err)
|
||||
}
|
||||
cmd := exec.Command(self, "verify", "--fuzz-one="+funcName, "-n", strconv.Itoa(n), path)
|
||||
out, err := cmd.CombinedOutput()
|
||||
if err != nil {
|
||||
// Check if the child was killed by a signal.
|
||||
if exitErr, ok := err.(*exec.ExitError); ok {
|
||||
ws := exitErr.Sys().(syscall.WaitStatus)
|
||||
if ws.Signaled() {
|
||||
return fmt.Sprintf("%s: CRASH (%v — partial function, use --ground-truth)", funcName, ws.Signal())
|
||||
}
|
||||
}
|
||||
// Non-zero exit without a signal: the fuzz reported mismatches.
|
||||
lines := strings.Split(strings.TrimSpace(string(out)), "\n")
|
||||
for _, l := range lines {
|
||||
if strings.Contains(l, funcName) {
|
||||
return strings.TrimSpace(l)
|
||||
}
|
||||
}
|
||||
return fmt.Sprintf("%s: FAIL (exit %v)", funcName, err)
|
||||
}
|
||||
// Success: extract the result line.
|
||||
lines := strings.Split(strings.TrimSpace(string(out)), "\n")
|
||||
for _, l := range lines {
|
||||
if strings.Contains(l, funcName) {
|
||||
return strings.TrimSpace(l)
|
||||
}
|
||||
}
|
||||
return strings.TrimSpace(string(out))
|
||||
}
|
||||
|
||||
+70
-5
@@ -52,6 +52,54 @@ func capture(fn func() int) (stdout, stderr string, code int) {
|
||||
return string(ob), string(eb), code
|
||||
}
|
||||
|
||||
// TestCmdFmtRecursive checks the go-fmt-style directory mode: with no
|
||||
// arguments every .s file below the working directory is formatted in place
|
||||
// ("." and "_" directories skipped), changed files are listed, and a second
|
||||
// run is a no-op.
|
||||
func TestCmdFmtRecursive(t *testing.T) {
|
||||
tmp := t.TempDir()
|
||||
t.Chdir(tmp)
|
||||
unformatted := []byte("TEXT ·f(SB),NOSPLIT,$0\nRET\n")
|
||||
write := func(path string) {
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(path, unformatted, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
write("a_amd64.s")
|
||||
write(filepath.Join("sub", "b_amd64.s"))
|
||||
write(filepath.Join("_refs", "c_amd64.s"))
|
||||
write(filepath.Join(".git", "d_amd64.s"))
|
||||
|
||||
out, errOut, code := capture(func() int { return cmdFmt(nil) })
|
||||
if code != 0 {
|
||||
t.Fatalf("code = %d (%s)", code, errOut)
|
||||
}
|
||||
if out != "a_amd64.s\n"+filepath.Join("sub", "b_amd64.s")+"\n" {
|
||||
t.Errorf("listed files unexpected:\n%s", out)
|
||||
}
|
||||
for _, p := range []string{"a_amd64.s", filepath.Join("sub", "b_amd64.s")} {
|
||||
b, _ := os.ReadFile(p)
|
||||
if !strings.Contains(string(b), "\tRET") {
|
||||
t.Errorf("%s not formatted in place:\n%s", p, b)
|
||||
}
|
||||
}
|
||||
for _, p := range []string{filepath.Join("_refs", "c_amd64.s"), filepath.Join(".git", "d_amd64.s")} {
|
||||
b, _ := os.ReadFile(p)
|
||||
if string(b) != string(unformatted) {
|
||||
t.Errorf("%s must not be touched:\n%s", p, b)
|
||||
}
|
||||
}
|
||||
|
||||
// Second pass: everything is canonical, nothing is listed.
|
||||
out, _, code = capture(func() int { return cmdFmt(nil) })
|
||||
if code != 0 || out != "" {
|
||||
t.Errorf("second pass: code=%d out=%q, want a no-op", code, out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdTokens(t *testing.T) {
|
||||
path := writeTemp(t, "f_amd64.s", clean)
|
||||
out, _, code := capture(func() int { return cmdTokens([]string{path}) })
|
||||
@@ -152,15 +200,32 @@ func TestCmdFmtWrite(t *testing.T) {
|
||||
func TestUsage(t *testing.T) {
|
||||
var b bytes.Buffer
|
||||
usage(&b)
|
||||
if !strings.Contains(b.String(), "gasm") {
|
||||
t.Errorf("usage text unexpected:\n%s", b.String())
|
||||
out := b.String()
|
||||
for _, want := range []string{
|
||||
"gasm", "Commands:", "Flags:", "--help", "--version",
|
||||
"tokens", "parse", "fmt", "lint", "asm", "lsp", "version",
|
||||
} {
|
||||
if !strings.Contains(out, want) {
|
||||
t.Errorf("usage text missing %q:\n%s", want, out)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdVersion(t *testing.T) {
|
||||
out, _, code := capture(func() int { return cmdVersion() })
|
||||
if code != 0 {
|
||||
t.Fatalf("code = %d", code)
|
||||
}
|
||||
if !strings.Contains(out, version) {
|
||||
t.Errorf("version output %q does not mention %q", out, version)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdArgErrors(t *testing.T) {
|
||||
// Missing file arguments produce a usage error (code 2).
|
||||
if _, _, code := capture(func() int { return cmdFmt(nil) }); code != 2 {
|
||||
t.Errorf("cmdFmt() code = %d, want 2", code)
|
||||
// A missing path is an error (code 1); cmdFmt with no arguments is the
|
||||
// recursive mode now, covered by TestCmdFmtRecursive.
|
||||
if _, _, code := capture(func() int { return cmdFmt([]string{"no/such/path"}) }); code != 1 {
|
||||
t.Errorf("cmdFmt(missing path) code = %d, want 1", code)
|
||||
}
|
||||
if _, _, code := capture(func() int { return cmdLint(nil) }); code != 2 {
|
||||
t.Errorf("cmdLint() code = %d, want 2", code)
|
||||
|
||||
@@ -0,0 +1,248 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package debug
|
||||
|
||||
import "fmt"
|
||||
|
||||
// Breakpoint is one INT3 breakpoint in the debuggee.
|
||||
type Breakpoint struct {
|
||||
Addr uint64 // absolute address in the debuggee
|
||||
Label string // source label ("" for raw addresses)
|
||||
Orig byte // original byte at Addr (restored on removal)
|
||||
Enabled bool
|
||||
Cond *Condition // optional condition (nil = unconditional)
|
||||
hits int
|
||||
}
|
||||
|
||||
// Condition is a simple register-comparison condition evaluated when a
|
||||
// breakpoint is hit. Format: <reg> <op> <value>.
|
||||
type Condition struct {
|
||||
Reg string // register name (rax, rbx, rip, rsp, ...)
|
||||
Op string // comparison operator: ==, !=, <, >, <=, >=
|
||||
Value uint64
|
||||
}
|
||||
|
||||
// Eval checks the condition against the current registers.
|
||||
func (c *Condition) Eval(regs *Regs) bool {
|
||||
var actual uint64
|
||||
switch c.Reg {
|
||||
case "rax", "eax", "ax", "al":
|
||||
actual = regs.RAX
|
||||
case "rbx", "ebx", "bx", "bl":
|
||||
actual = regs.RBX
|
||||
case "rcx", "ecx", "cx", "cl":
|
||||
actual = regs.RCX
|
||||
case "rdx", "edx", "dx", "dl":
|
||||
actual = regs.RDX
|
||||
case "rsi", "esi", "si":
|
||||
actual = regs.RSI
|
||||
case "rdi", "edi", "di":
|
||||
actual = regs.RDI
|
||||
case "rbp", "ebp", "bp":
|
||||
actual = regs.RBP
|
||||
case "rsp", "esp", "sp":
|
||||
actual = regs.RSP
|
||||
case "r8":
|
||||
actual = regs.R8
|
||||
case "r9":
|
||||
actual = regs.R9
|
||||
case "r10":
|
||||
actual = regs.R10
|
||||
case "r11":
|
||||
actual = regs.R11
|
||||
case "r12":
|
||||
actual = regs.R12
|
||||
case "r13":
|
||||
actual = regs.R13
|
||||
case "r14":
|
||||
actual = regs.R14
|
||||
case "r15":
|
||||
actual = regs.R15
|
||||
case "rip", "eip":
|
||||
actual = regs.RIP
|
||||
default:
|
||||
return true // unknown register — don't block
|
||||
}
|
||||
switch c.Op {
|
||||
case "==", "=":
|
||||
return actual == c.Value
|
||||
case "!=":
|
||||
return actual != c.Value
|
||||
case "<":
|
||||
return actual < c.Value
|
||||
case ">":
|
||||
return actual > c.Value
|
||||
case "<=":
|
||||
return actual <= c.Value
|
||||
case ">=":
|
||||
return actual >= c.Value
|
||||
default:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// Breakpoints manages the set of breakpoints for a Session.
|
||||
// Breakpoints manages software breakpoints for a debuggee.
|
||||
type Breakpoints struct {
|
||||
t tracer
|
||||
bps map[uint64]*Breakpoint
|
||||
}
|
||||
|
||||
// NewBreakpoints creates a new breakpoint manager.
|
||||
func NewBreakpoints(t tracer) *Breakpoints {
|
||||
return &Breakpoints{t: t, bps: make(map[uint64]*Breakpoint)}
|
||||
}
|
||||
|
||||
// Set installs a breakpoint at addr (replaces any existing one).
|
||||
func (bm *Breakpoints) Set(addr uint64, label string) (*Breakpoint, error) {
|
||||
return bm.SetWithCond(addr, label, nil)
|
||||
}
|
||||
|
||||
// SetWithCond installs a breakpoint with an optional condition.
|
||||
func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (*Breakpoint, error) {
|
||||
if bp, ok := bm.bps[addr]; ok {
|
||||
bp.Enabled = true
|
||||
bp.Cond = cond
|
||||
return bp, nil
|
||||
}
|
||||
// Read the original byte.
|
||||
word, err := bm.t.Peek(addr)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
orig := byte(word)
|
||||
// Patch with INT3 (0xCC), preserving the rest of the word.
|
||||
patched := (word &^ 0xFF) | 0xCC
|
||||
if err := bm.t.Poke(addr, patched); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
bp := &Breakpoint{Addr: addr, Label: label, Orig: orig, Enabled: true, Cond: cond}
|
||||
bm.bps[addr] = bp
|
||||
return bp, nil
|
||||
}
|
||||
|
||||
// Hits returns the number of times the breakpoint has been hit.
|
||||
func (bp *Breakpoint) Hits() int {
|
||||
return bp.hits
|
||||
}
|
||||
|
||||
// Info returns a formatted list of all breakpoints.
|
||||
func (bm *Breakpoints) Info() string {
|
||||
if len(bm.bps) == 0 {
|
||||
return "no breakpoints set\n"
|
||||
}
|
||||
result := ""
|
||||
i := 0
|
||||
for _, bp := range bm.bps {
|
||||
i++
|
||||
status := "enabled"
|
||||
if !bp.Enabled {
|
||||
status = "disabled"
|
||||
}
|
||||
label := bp.Label
|
||||
if label == "" {
|
||||
label = fmt.Sprintf("%#x", bp.Addr)
|
||||
}
|
||||
cond := ""
|
||||
if bp.Cond != nil {
|
||||
cond = fmt.Sprintf(" if %s %s %#x", bp.Cond.Reg, bp.Cond.Op, bp.Cond.Value)
|
||||
}
|
||||
result += fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// Clear removes the breakpoint at addr, restoring the original byte.
|
||||
func (bm *Breakpoints) Clear(addr uint64) error {
|
||||
bp, ok := bm.bps[addr]
|
||||
if !ok {
|
||||
return fmt.Errorf("debug: no breakpoint at %#x", addr)
|
||||
}
|
||||
word, err := bm.t.Peek(addr)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
||||
if err := bm.t.Poke(addr, restored); err != nil {
|
||||
return err
|
||||
}
|
||||
delete(bm.bps, addr)
|
||||
return nil
|
||||
}
|
||||
|
||||
// ClearAll removes all breakpoints.
|
||||
func (bm *Breakpoints) ClearAll() error {
|
||||
for addr := range bm.bps {
|
||||
if err := bm.Clear(addr); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// At returns the breakpoint at addr, if any.
|
||||
func (bm *Breakpoints) At(addr uint64) *Breakpoint {
|
||||
return bm.bps[addr]
|
||||
}
|
||||
|
||||
// All returns all breakpoints.
|
||||
func (bm *Breakpoints) All() []*Breakpoint {
|
||||
out := make([]*Breakpoint, 0, len(bm.bps))
|
||||
for _, bp := range bm.bps {
|
||||
out = append(out, bp)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// HandleTrap is called after the debuggee stops on SIGTRAP. It checks
|
||||
// whether the trap was caused by one of our breakpoints (RIP-1 matches
|
||||
// a breakpoint address), restores the original byte, rewinds RIP, and
|
||||
// returns the breakpoint that was hit (or nil if it was a single-step).
|
||||
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
|
||||
// After INT3, RIP points to the byte AFTER the 0xCC.
|
||||
trapAddr := regs.RIP - 1
|
||||
bp, ok := bm.bps[trapAddr]
|
||||
if !ok || !bp.Enabled {
|
||||
return nil // single-step trap or unknown
|
||||
}
|
||||
// Check the condition (if any).
|
||||
if bp.Cond != nil && !bp.Cond.Eval(regs) {
|
||||
// Condition not met — restore the byte but do NOT rewind RIP.
|
||||
// The process continues from the next instruction (past the INT3).
|
||||
word, err := bm.t.Peek(trapAddr)
|
||||
if err == nil {
|
||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
||||
bm.t.Poke(trapAddr, restored)
|
||||
}
|
||||
// RIP is already past the INT3 (trapAddr + 1). Don't rewind.
|
||||
return nil
|
||||
}
|
||||
bp.hits++
|
||||
// Restore the original byte.
|
||||
word, err := bm.t.Peek(trapAddr)
|
||||
if err == nil {
|
||||
restored := (word &^ 0xFF) | uint64(bp.Orig)
|
||||
bm.t.Poke(trapAddr, restored)
|
||||
}
|
||||
// Rewind RIP to re-execute the original instruction.
|
||||
regs.RIP = trapAddr
|
||||
bm.t.SetRegs(regs)
|
||||
return bp
|
||||
}
|
||||
|
||||
// Reinsert re-inserts the breakpoint at addr after a single-step past it.
|
||||
// Called after Step() when we want the breakpoint to fire again on the
|
||||
// next Continue().
|
||||
func (bm *Breakpoints) Reinsert(addr uint64) error {
|
||||
bp, ok := bm.bps[addr]
|
||||
if !ok || !bp.Enabled {
|
||||
return nil
|
||||
}
|
||||
word, err := bm.t.Peek(addr)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
patched := (word &^ 0xFF) | 0xCC
|
||||
return bm.t.Poke(addr, patched)
|
||||
}
|
||||
@@ -0,0 +1,273 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestConditionEval(t *testing.T) {
|
||||
regs := &Regs{
|
||||
RAX: 42,
|
||||
RBX: 0,
|
||||
RCX: 100,
|
||||
RIP: 0x1000,
|
||||
RSP: 0x2000,
|
||||
R8: 8,
|
||||
R15: 15,
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
cond Condition
|
||||
want bool
|
||||
}{
|
||||
{Condition{Reg: "rax", Op: "==", Value: 42}, true},
|
||||
{Condition{Reg: "rax", Op: "==", Value: 43}, false},
|
||||
{Condition{Reg: "rax", Op: "!=", Value: 43}, true},
|
||||
{Condition{Reg: "rax", Op: "!=", Value: 42}, false},
|
||||
{Condition{Reg: "rax", Op: "<", Value: 50}, true},
|
||||
{Condition{Reg: "rax", Op: "<", Value: 40}, false},
|
||||
{Condition{Reg: "rax", Op: ">", Value: 40}, true},
|
||||
{Condition{Reg: "rax", Op: ">", Value: 50}, false},
|
||||
{Condition{Reg: "rax", Op: "<=", Value: 42}, true},
|
||||
{Condition{Reg: "rax", Op: ">=", Value: 42}, true},
|
||||
{Condition{Reg: "rbx", Op: "==", Value: 0}, true},
|
||||
{Condition{Reg: "rcx", Op: ">", Value: 50}, true},
|
||||
{Condition{Reg: "rip", Op: "==", Value: 0x1000}, true},
|
||||
{Condition{Reg: "rsp", Op: ">", Value: 0x1000}, true},
|
||||
{Condition{Reg: "r8", Op: "==", Value: 8}, true},
|
||||
{Condition{Reg: "r15", Op: "==", Value: 15}, true},
|
||||
{Condition{Reg: "eax", Op: "==", Value: 42}, true}, // 32-bit alias
|
||||
{Condition{Reg: "ax", Op: "==", Value: 42}, true}, // 16-bit alias
|
||||
{Condition{Reg: "unknown", Op: "==", Value: 0}, true}, // unknown reg → don't block
|
||||
{Condition{Reg: "rax", Op: "??", Value: 0}, true}, // unknown op → don't block
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := tt.cond.Eval(regs)
|
||||
if got != tt.want {
|
||||
t.Errorf("Condition{%q %q %d}.Eval() = %v, want %v",
|
||||
tt.cond.Reg, tt.cond.Op, tt.cond.Value, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestLineAt(t *testing.T) {
|
||||
lines := []SourceLine{
|
||||
{Offset: 0, Line: 5},
|
||||
{Offset: 5, Line: 6},
|
||||
{Offset: 10, Line: 7},
|
||||
{Offset: 15, Line: 8},
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
offset int
|
||||
want int
|
||||
}{
|
||||
{0, 5},
|
||||
{1, 5},
|
||||
{4, 5},
|
||||
{5, 6},
|
||||
{7, 6},
|
||||
{10, 7},
|
||||
{12, 7},
|
||||
{15, 8},
|
||||
{20, 8},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := lineAt(lines, tt.offset)
|
||||
if got != tt.want {
|
||||
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
|
||||
}
|
||||
}
|
||||
|
||||
// Empty table.
|
||||
if lineAt(nil, 5) != 0 {
|
||||
t.Error("lineAt(nil, 5) should return 0")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOffsetForLine(t *testing.T) {
|
||||
lines := []SourceLine{
|
||||
{Offset: 0, Line: 5},
|
||||
{Offset: 5, Line: 6},
|
||||
{Offset: 10, Line: 7},
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
line int
|
||||
want int
|
||||
}{
|
||||
{5, 0},
|
||||
{6, 5},
|
||||
{7, 10},
|
||||
{99, -1}, // not found
|
||||
{0, -1}, // not found
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := offsetForLine(lines, tt.line)
|
||||
if got != tt.want {
|
||||
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDecodeRflags(t *testing.T) {
|
||||
tests := []struct {
|
||||
flags uint64
|
||||
want string
|
||||
}{
|
||||
{0x202, "IF"}, // only IF set (bit 9)
|
||||
{0x246, "PF ZF IF"}, // PF(2) + ZF(6) + IF(9)
|
||||
{0x001, "CF"}, // carry flag
|
||||
{0x080, "SF"}, // sign flag
|
||||
{0x800, "OF"}, // overflow flag
|
||||
{0x000, "none"}, // no flags
|
||||
{0x202 | 0x001, "CF IF"}, // CF + IF
|
||||
{0x3F7, "CF PF AF ZF SF TF IF"}, // all arithmetic flags
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := decodeRflags(tt.flags)
|
||||
if got != tt.want {
|
||||
t.Errorf("decodeRflags(%#x) = %q, want %q", tt.flags, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestNearestLabel(t *testing.T) {
|
||||
labels := []Label{
|
||||
{Name: "start", Offset: 0},
|
||||
{Name: "loop", Offset: 10},
|
||||
{Name: "done", Offset: 20},
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
offset int
|
||||
want string
|
||||
}{
|
||||
{0, "start"},
|
||||
{5, "start"},
|
||||
{10, "loop"},
|
||||
{15, "loop"},
|
||||
{20, "done"},
|
||||
{25, "done"},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := nearestLabel(labels, tt.offset)
|
||||
if got != tt.want {
|
||||
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointsSetAndClear(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
|
||||
// Set a breakpoint at address 0x1000.
|
||||
bp, err := bm.Set(0x1000, "test")
|
||||
if err != nil {
|
||||
t.Fatalf("Set: %v", err)
|
||||
}
|
||||
if !bp.Enabled {
|
||||
t.Error("breakpoint not enabled")
|
||||
}
|
||||
if bp.Label != "test" {
|
||||
t.Errorf("label = %q, want test", bp.Label)
|
||||
}
|
||||
|
||||
// Verify Peek was called.
|
||||
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
|
||||
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
|
||||
}
|
||||
|
||||
// Verify Poke wrote INT3.
|
||||
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
|
||||
t.Errorf("pokes = %v", tr.pokes)
|
||||
}
|
||||
|
||||
// At should find it.
|
||||
if bm.At(0x1000) == nil {
|
||||
t.Error("At(0x1000) returned nil")
|
||||
}
|
||||
|
||||
// All should return it.
|
||||
all := bm.All()
|
||||
if len(all) != 1 {
|
||||
t.Errorf("All() = %d breakpoints, want 1", len(all))
|
||||
}
|
||||
|
||||
// Clear it.
|
||||
if err := bm.Clear(0x1000); err != nil {
|
||||
t.Fatalf("Clear: %v", err)
|
||||
}
|
||||
if bm.At(0x1000) != nil {
|
||||
t.Error("At(0x1000) after Clear should be nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointsSetWithCond(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
|
||||
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
|
||||
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
|
||||
if err != nil {
|
||||
t.Fatalf("SetWithCond: %v", err)
|
||||
}
|
||||
if bp.Cond == nil || bp.Cond.Value != 42 {
|
||||
t.Error("condition not set")
|
||||
}
|
||||
|
||||
// Re-setting the same address should update the condition.
|
||||
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
|
||||
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
|
||||
if err != nil {
|
||||
t.Fatalf("SetWithCond (update): %v", err)
|
||||
}
|
||||
if bp2.Cond.Value != 100 {
|
||||
t.Error("condition not updated")
|
||||
}
|
||||
// Should have only 1 Peek (first Set), second is update (no Peek needed).
|
||||
if len(tr.peeks) != 1 {
|
||||
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointsClearAll(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
|
||||
bm.Set(0x1000, "a")
|
||||
bm.Set(0x2000, "b")
|
||||
bm.Set(0x3000, "c")
|
||||
|
||||
if len(bm.All()) != 3 {
|
||||
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
|
||||
}
|
||||
|
||||
bm.ClearAll()
|
||||
if len(bm.All()) != 0 {
|
||||
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
|
||||
}
|
||||
}
|
||||
|
||||
func TestBreakpointInfo(t *testing.T) {
|
||||
tr := newMockTracer()
|
||||
bm := NewBreakpoints(tr)
|
||||
bm.Set(0x4000, "info_test")
|
||||
|
||||
info := bm.Info()
|
||||
if info == "" {
|
||||
t.Error("Info returned empty string")
|
||||
}
|
||||
if !strings.Contains(info, "info_test") {
|
||||
t.Errorf("Info %q does not contain label", info)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,52 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
)
|
||||
|
||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||
// memory and returns its text representation and length in bytes.
|
||||
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||
// Read up to 15 bytes (max x86 instruction length).
|
||||
mem, err := s.ReadMemory(addr, 15)
|
||||
if err != nil {
|
||||
// Try a shorter read if we're near a page boundary.
|
||||
mem, err = s.ReadMemory(addr, 1)
|
||||
if err != nil {
|
||||
return "", 0, err
|
||||
}
|
||||
}
|
||||
inst, err := x86asm.Decode(mem, 64)
|
||||
if err != nil {
|
||||
return "???", 1, nil
|
||||
}
|
||||
text := x86asm.IntelSyntax(inst, addr, nil)
|
||||
return text, inst.Len, nil
|
||||
}
|
||||
|
||||
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||
// them as a formatted string with addresses and byte offsets.
|
||||
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||
var result string
|
||||
pc := addr
|
||||
for i := 0; i < n; i++ {
|
||||
text, length, err := s.Disassemble(pc)
|
||||
if err != nil {
|
||||
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
|
||||
break
|
||||
}
|
||||
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
|
||||
if length == 0 {
|
||||
length = 1
|
||||
}
|
||||
pc += uint64(length)
|
||||
}
|
||||
return result
|
||||
}
|
||||
@@ -0,0 +1,304 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
// Package debug implements the interactive debugger for gasm (Phase 4):
|
||||
// single-stepping, breakpoints, register and memory inspection for
|
||||
// JIT-assembled Plan 9 amd64 functions, controlled via ptrace.
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"syscall"
|
||||
"time"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// Session is a ptrace debugging session controlling one debuggee process.
|
||||
type Session struct {
|
||||
pid int
|
||||
cmd *exec.Cmd
|
||||
stopped bool
|
||||
exited bool
|
||||
codeBase uint64 // base address of the JIT code in the debuggee
|
||||
}
|
||||
|
||||
// Launch starts the debuggee subprocess (gasm debug --target ...) and
|
||||
// attaches to it via ptrace. The debuggee assembles the file, maps the
|
||||
// JIT code, calls PTRACE_TRACEME and raises SIGSTOP; Launch waits for
|
||||
// that initial stop and returns a ready Session.
|
||||
func Launch(gasmBin, asmPath, funcName string, args []byte) (*Session, error) {
|
||||
self, err := os.Executable()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("debug: cannot find gasm binary: %w", err)
|
||||
}
|
||||
if gasmBin != "" {
|
||||
self = gasmBin
|
||||
}
|
||||
|
||||
// Write the arg block to a temp file (the child reads it).
|
||||
tmpDir, err := os.MkdirTemp("", "gasm-debug-*")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("debug: tempdir: %w", err)
|
||||
}
|
||||
argsFile := filepath.Join(tmpDir, "args.bin")
|
||||
if err := os.WriteFile(argsFile, args, 0o644); err != nil {
|
||||
os.RemoveAll(tmpDir)
|
||||
return nil, fmt.Errorf("debug: write args: %w", err)
|
||||
}
|
||||
|
||||
cmd := exec.Command(self, "debug", "--target", "--func", funcName, "--args", argsFile, asmPath)
|
||||
cmd.Env = append(os.Environ(), "GASM_DEBUG_TMP="+tmpDir)
|
||||
cmd.Stdout = nil // output goes to the debugger, not the terminal
|
||||
cmd.Stderr = os.Stderr
|
||||
cmd.SysProcAttr = &syscall.SysProcAttr{}
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
os.RemoveAll(tmpDir)
|
||||
return nil, fmt.Errorf("debug: start debuggee: %w", err)
|
||||
}
|
||||
|
||||
s := &Session{pid: cmd.Process.Pid, cmd: cmd}
|
||||
|
||||
// Wait for the child to signal readiness and stop. The child calls
|
||||
// PTRACE_TRACEME then SIGSTOP, so Wait4 with WUNTRACED observes the
|
||||
// ptrace-stop directly (no PTRACE_ATTACH needed).
|
||||
readyFile := filepath.Join(tmpDir, "ready")
|
||||
for i := 0; i < 500; i++ {
|
||||
if _, err := os.Stat(readyFile); err == nil {
|
||||
break
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
var ws syscall.WaitStatus
|
||||
if _, err := syscall.Wait4(s.pid, &ws, syscall.WUNTRACED, nil); err != nil {
|
||||
cmd.Process.Kill()
|
||||
os.RemoveAll(tmpDir)
|
||||
return nil, fmt.Errorf("debug: wait for debuggee: %w", err)
|
||||
}
|
||||
s.stopped = true
|
||||
|
||||
// Read the code base from /proc/pid/maps (find the RWX mapping).
|
||||
s.codeBase = findRWXMapping(s.pid)
|
||||
if s.codeBase == 0 {
|
||||
// Fallback: try the file the child wrote.
|
||||
baseFile := filepath.Join(tmpDir, "codebase")
|
||||
if data, err := os.ReadFile(baseFile); err == nil {
|
||||
fmt.Sscanf(string(data), "%d", &s.codeBase)
|
||||
}
|
||||
}
|
||||
|
||||
return s, nil
|
||||
}
|
||||
|
||||
// wait waits for the debuggee to stop and returns the wait status.
|
||||
func (s *Session) wait() error {
|
||||
var ws syscall.WaitStatus
|
||||
_, err := syscall.Wait4(s.pid, &ws, 0, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if ws.Exited() {
|
||||
s.exited = true
|
||||
return fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
|
||||
}
|
||||
s.stopped = true
|
||||
return nil
|
||||
}
|
||||
|
||||
// GetRegs reads the general-purpose registers of the stopped debuggee.
|
||||
func (s *Session) GetRegs() (Regs, error) {
|
||||
var regs Regs
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_GETREGS),
|
||||
uintptr(s.pid),
|
||||
0,
|
||||
uintptr(unsafe.Pointer(®s)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return regs, fmt.Errorf("debug: PTRACE_GETREGS: %w", errno)
|
||||
}
|
||||
return regs, nil
|
||||
}
|
||||
|
||||
// SetRegs writes the general-purpose registers of the stopped debuggee.
|
||||
func (s *Session) SetRegs(regs *Regs) error {
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_SETREGS),
|
||||
uintptr(s.pid),
|
||||
0,
|
||||
uintptr(unsafe.Pointer(regs)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return fmt.Errorf("debug: PTRACE_SETREGS: %w", errno)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Peek reads a word (8 bytes) from the debuggee's memory at addr.
|
||||
// Uses /proc/pid/mem which works reliably with Go's multi-threaded runtime.
|
||||
func (s *Session) Peek(addr uint64) (uint64, error) {
|
||||
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
|
||||
}
|
||||
defer mem.Close()
|
||||
buf := make([]byte, 8)
|
||||
if _, err := mem.ReadAt(buf, int64(addr)); err != nil {
|
||||
return 0, fmt.Errorf("debug: read mem %#x: %w", addr, err)
|
||||
}
|
||||
return uint64(buf[0]) | uint64(buf[1])<<8 | uint64(buf[2])<<16 | uint64(buf[3])<<24 |
|
||||
uint64(buf[4])<<32 | uint64(buf[5])<<40 | uint64(buf[6])<<48 | uint64(buf[7])<<56, nil
|
||||
}
|
||||
|
||||
// Poke writes a word (8 bytes) to the debuggee's memory at addr.
|
||||
func (s *Session) Poke(addr, val uint64) error {
|
||||
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_WRONLY, 0)
|
||||
if err != nil {
|
||||
return fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
|
||||
}
|
||||
defer mem.Close()
|
||||
buf := []byte{byte(val), byte(val >> 8), byte(val >> 16), byte(val >> 24),
|
||||
byte(val >> 32), byte(val >> 40), byte(val >> 48), byte(val >> 56)}
|
||||
if _, err := mem.WriteAt(buf, int64(addr)); err != nil {
|
||||
return fmt.Errorf("debug: write mem %#x: %w", addr, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ReadMemory reads len bytes from the debuggee's memory at addr.
|
||||
func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) {
|
||||
out := make([]byte, length)
|
||||
for i := 0; i < length; i += 8 {
|
||||
word, err := s.Peek(addr + uint64(i))
|
||||
if err != nil {
|
||||
return out[:i], err
|
||||
}
|
||||
for j := 0; j < 8 && i+j < length; j++ {
|
||||
out[i+j] = byte(word >> (8 * j))
|
||||
}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// WriteMemory writes bytes to the debuggee's memory at addr.
|
||||
func (s *Session) WriteMemory(addr uint64, data []byte) error {
|
||||
for i := 0; i < len(data); i += 8 {
|
||||
end := i + 8
|
||||
if end > len(data) {
|
||||
end = len(data)
|
||||
}
|
||||
var word uint64
|
||||
for j := 0; j < end-i; j++ {
|
||||
word |= uint64(data[i+j]) << (8 * j)
|
||||
}
|
||||
// For partial writes, read-modify-write the existing word.
|
||||
if end-i < 8 {
|
||||
existing, err := s.Peek(addr + uint64(i))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
// Clear the bytes we're overwriting and merge.
|
||||
mask := ^((uint64(1) << (8 * (end - i))) - 1)
|
||||
word = (existing & mask) | word
|
||||
}
|
||||
if err := s.Poke(addr+uint64(i), word); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Step executes a single instruction in the debuggee.
|
||||
func (s *Session) Step() error {
|
||||
if s.exited {
|
||||
return fmt.Errorf("debug: debuggee has exited")
|
||||
}
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_SINGLESTEP),
|
||||
uintptr(s.pid),
|
||||
0, 0, 0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return fmt.Errorf("debug: PTRACE_SINGLESTEP: %w", errno)
|
||||
}
|
||||
return s.wait()
|
||||
}
|
||||
|
||||
// Continue resumes execution until the next breakpoint or exit.
|
||||
func (s *Session) Continue() error {
|
||||
if s.exited {
|
||||
return fmt.Errorf("debug: debuggee has exited")
|
||||
}
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_CONT),
|
||||
uintptr(s.pid),
|
||||
0, 0, 0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return fmt.Errorf("debug: PTRACE_CONT: %w", errno)
|
||||
}
|
||||
return s.wait()
|
||||
}
|
||||
|
||||
// Exited returns true if the debuggee has terminated.
|
||||
func (s *Session) Exited() bool {
|
||||
return s.exited
|
||||
}
|
||||
|
||||
// Pid returns the debuggee's process ID.
|
||||
func (s *Session) Pid() int {
|
||||
return s.pid
|
||||
}
|
||||
|
||||
// CodeBase returns the base address of the JIT code in the debuggee.
|
||||
func (s *Session) CodeBase() uint64 {
|
||||
return s.codeBase
|
||||
}
|
||||
|
||||
// Kill terminates the debuggee.
|
||||
func (s *Session) Kill() {
|
||||
if !s.exited {
|
||||
syscall.Kill(s.pid, syscall.SIGKILL)
|
||||
syscall.Wait4(s.pid, nil, 0, nil)
|
||||
s.exited = true
|
||||
}
|
||||
if s.cmd != nil && s.cmd.Process != nil {
|
||||
s.cmd.Wait()
|
||||
}
|
||||
}
|
||||
|
||||
// findRWXMapping reads /proc/pid/maps and returns the base address of the
|
||||
// first read-write-execute mapping (the JIT code region).
|
||||
func findRWXMapping(pid int) uint64 {
|
||||
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
for _, line := range strings.Split(string(data), "\n") {
|
||||
// Format: addr-addr perms offset dev inode pathname
|
||||
fields := strings.Fields(line)
|
||||
if len(fields) < 2 {
|
||||
continue
|
||||
}
|
||||
perms := fields[1]
|
||||
if len(perms) >= 3 && perms[0] == 'r' && perms[1] == 'w' && perms[2] == 'x' {
|
||||
// Parse the start address.
|
||||
var start uint64
|
||||
fmt.Sscanf(fields[0], "%x-", &start)
|
||||
return start
|
||||
}
|
||||
}
|
||||
return 0
|
||||
}
|
||||
+638
@@ -0,0 +1,638 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"fmt"
|
||||
"os"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Label is a named address within the debugged function.
|
||||
type Label struct {
|
||||
Name string
|
||||
Offset int // function-relative offset
|
||||
}
|
||||
|
||||
// SourceLine maps a byte offset to a source line number.
|
||||
type SourceLine struct {
|
||||
Offset int
|
||||
Line int
|
||||
}
|
||||
|
||||
// REPL runs the interactive debugger loop. On entry, the debuggee is
|
||||
// stopped in the Go runtime (after PTRACE_TRACEME + SIGSTOP). The REPL
|
||||
// sets a temporary breakpoint at the function entry, continues to it, and
|
||||
// then presents the prompt — so the user starts debugging at the first
|
||||
// instruction of the assembled function.
|
||||
func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, argsSize int, labels []Label, lines []SourceLine) {
|
||||
entryAddr := codeBase + uint64(funcOffset)
|
||||
|
||||
// Run to the function entry.
|
||||
bp, err := bm.Set(entryAddr, "(entry)")
|
||||
if err != nil {
|
||||
fmt.Printf("warning: cannot set entry breakpoint: %v\n", err)
|
||||
} else {
|
||||
if err := s.Continue(); err != nil {
|
||||
fmt.Printf("warning: continue to entry: %v\n", err)
|
||||
}
|
||||
regs, _ := s.GetRegs()
|
||||
bm.HandleTrap(®s)
|
||||
// Remove the temporary entry breakpoint.
|
||||
bm.Clear(entryAddr)
|
||||
_ = bp
|
||||
}
|
||||
|
||||
fmt.Printf("stopped at function entry: %#x (%d bytes)\n", entryAddr, funcSize)
|
||||
fmt.Println("commands: break <label|addr> | step [n] | continue | disas [n] | regs | where | x <addr> [len] | w <addr> <val...> | labels | quit")
|
||||
|
||||
scanner := bufio.NewScanner(os.Stdin)
|
||||
|
||||
for {
|
||||
fmt.Print("(gasm) ")
|
||||
if !scanner.Scan() {
|
||||
break
|
||||
}
|
||||
line := strings.TrimSpace(scanner.Text())
|
||||
if line == "" {
|
||||
continue
|
||||
}
|
||||
parts := strings.Fields(line)
|
||||
cmd := parts[0]
|
||||
|
||||
switch cmd {
|
||||
case "q", "quit":
|
||||
s.Kill()
|
||||
return
|
||||
|
||||
case "regs":
|
||||
regs, err := s.GetRegs()
|
||||
if err != nil {
|
||||
fmt.Println(err)
|
||||
continue
|
||||
}
|
||||
printRegs(®s, codeBase, uint64(funcOffset))
|
||||
|
||||
case "step", "s":
|
||||
n := 1
|
||||
if len(parts) > 1 {
|
||||
n, _ = strconv.Atoi(parts[1])
|
||||
}
|
||||
for i := 0; i < n; i++ {
|
||||
if s.Exited() {
|
||||
fmt.Println("debuggee exited")
|
||||
break
|
||||
}
|
||||
if err := s.Step(); err != nil {
|
||||
fmt.Println(err)
|
||||
break
|
||||
}
|
||||
}
|
||||
if !s.Exited() {
|
||||
regs, _ := s.GetRegs()
|
||||
text, _, _ := s.Disassemble(regs.RIP)
|
||||
fmt.Printf("=> %#x (func+%#x): %s\n", regs.RIP, regs.RIP-codeBase-uint64(funcOffset), text)
|
||||
}
|
||||
|
||||
case "next", "n":
|
||||
// Step over: if the current instruction is a CALL, set a
|
||||
// breakpoint after it and continue; otherwise single-step.
|
||||
regs, _ := s.GetRegs()
|
||||
text, instLen, _ := s.Disassemble(regs.RIP)
|
||||
if strings.HasPrefix(strings.ToLower(text), "call") {
|
||||
// Set a temporary breakpoint after the CALL.
|
||||
afterAddr := regs.RIP + uint64(instLen)
|
||||
bp, err := bm.Set(afterAddr, "(next)")
|
||||
if err != nil {
|
||||
fmt.Printf("cannot set next breakpoint: %v\n", err)
|
||||
continue
|
||||
}
|
||||
// Continue until the breakpoint.
|
||||
for _, b := range bm.All() {
|
||||
bm.Reinsert(b.Addr)
|
||||
}
|
||||
if err := s.Continue(); err != nil {
|
||||
fmt.Println(err)
|
||||
bm.Clear(afterAddr)
|
||||
continue
|
||||
}
|
||||
bm.HandleTrap(®s)
|
||||
bm.Clear(afterAddr)
|
||||
_ = bp
|
||||
} else {
|
||||
// Not a CALL — just single-step.
|
||||
if err := s.Step(); err != nil {
|
||||
fmt.Println(err)
|
||||
continue
|
||||
}
|
||||
}
|
||||
if !s.Exited() {
|
||||
regs, _ := s.GetRegs()
|
||||
text, _, _ := s.Disassemble(regs.RIP)
|
||||
fmt.Printf("=> %#x (func+%#x): %s\n", regs.RIP, regs.RIP-codeBase-uint64(funcOffset), text)
|
||||
}
|
||||
|
||||
case "finish", "fin":
|
||||
// Run until the current function returns.
|
||||
// For NOSPLIT frame=0: return address is at [RSP].
|
||||
regs, _ := s.GetRegs()
|
||||
retAddr, err := s.Peek(regs.RSP)
|
||||
if err != nil {
|
||||
fmt.Printf("cannot read return address: %v\n", err)
|
||||
continue
|
||||
}
|
||||
// Set a temporary breakpoint at the return address.
|
||||
bp, err := bm.Set(retAddr, "(finish)")
|
||||
if err != nil {
|
||||
fmt.Printf("cannot set finish breakpoint: %v\n", err)
|
||||
continue
|
||||
}
|
||||
// Continue until the breakpoint.
|
||||
for _, b := range bm.All() {
|
||||
bm.Reinsert(b.Addr)
|
||||
}
|
||||
if err := s.Continue(); err != nil {
|
||||
fmt.Println(err)
|
||||
bm.Clear(retAddr)
|
||||
continue
|
||||
}
|
||||
if !s.Exited() {
|
||||
bm.HandleTrap(®s)
|
||||
}
|
||||
bm.Clear(retAddr)
|
||||
_ = bp
|
||||
if s.Exited() {
|
||||
fmt.Println("debuggee exited")
|
||||
} else {
|
||||
regs, _ := s.GetRegs()
|
||||
fmt.Printf("finished, now at %#x\n", regs.RIP)
|
||||
}
|
||||
|
||||
case "continue", "c":
|
||||
if s.Exited() {
|
||||
fmt.Println("debuggee exited")
|
||||
continue
|
||||
}
|
||||
// Loop: continue until a breakpoint fires (condition met) or exit.
|
||||
for {
|
||||
// Re-insert all breakpoints before continuing.
|
||||
for _, bp := range bm.All() {
|
||||
bm.Reinsert(bp.Addr)
|
||||
}
|
||||
if err := s.Continue(); err != nil {
|
||||
fmt.Println(err)
|
||||
break
|
||||
}
|
||||
if s.Exited() {
|
||||
fmt.Println("debuggee exited")
|
||||
break
|
||||
}
|
||||
// Check for watchpoint hits.
|
||||
reason, wpAddr := s.StopInfo()
|
||||
if reason == StopWatchpoint {
|
||||
fmt.Printf("watchpoint hit at %#x\n", wpAddr)
|
||||
break
|
||||
}
|
||||
regs, _ := s.GetRegs()
|
||||
if bp := bm.HandleTrap(®s); bp != nil {
|
||||
name := bp.Label
|
||||
if name == "" {
|
||||
name = fmt.Sprintf("%#x", bp.Addr)
|
||||
}
|
||||
fmt.Printf("breakpoint hit: %s (func+%#x)\n", name, bp.Addr-codeBase-uint64(funcOffset))
|
||||
break
|
||||
}
|
||||
// Condition not met (or single-step trap) — re-insert and continue.
|
||||
}
|
||||
|
||||
case "break", "b":
|
||||
if len(parts) < 2 {
|
||||
fmt.Println("usage: break <label|addr|line> [if <reg> <op> <val>]")
|
||||
continue
|
||||
}
|
||||
// Try as a line number first.
|
||||
var addr uint64
|
||||
var label string
|
||||
if lineNum, err := strconv.Atoi(parts[1]); err == nil && lineNum > 0 {
|
||||
// Find the byte offset for this line.
|
||||
off := offsetForLine(lines, lineNum)
|
||||
if off < 0 {
|
||||
fmt.Printf("no instruction at line %d\n", lineNum)
|
||||
continue
|
||||
}
|
||||
addr = codeBase + uint64(funcOffset) + uint64(off)
|
||||
label = fmt.Sprintf("line %d", lineNum)
|
||||
} else {
|
||||
addr, label = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||
}
|
||||
if addr == 0 {
|
||||
fmt.Printf("unknown label, address, or line: %s\n", parts[1])
|
||||
continue
|
||||
}
|
||||
// Parse optional condition: "if <reg> <op> <value>"
|
||||
var cond *Condition
|
||||
if len(parts) >= 6 && parts[2] == "if" {
|
||||
val, err := strconv.ParseUint(parts[5], 0, 64)
|
||||
if err != nil {
|
||||
fmt.Printf("invalid condition value: %s\n", parts[5])
|
||||
continue
|
||||
}
|
||||
cond = &Condition{Reg: strings.ToLower(parts[3]), Op: parts[4], Value: val}
|
||||
} else if len(parts) >= 4 && parts[2] == "if" {
|
||||
fmt.Println("usage: break <label|addr> if <reg> <op> <value>")
|
||||
continue
|
||||
}
|
||||
bp, err := bm.SetWithCond(addr, label, cond)
|
||||
if err != nil {
|
||||
fmt.Println(err)
|
||||
continue
|
||||
}
|
||||
condStr := ""
|
||||
if cond != nil {
|
||||
condStr = fmt.Sprintf(" if %s %s %#x", cond.Reg, cond.Op, cond.Value)
|
||||
}
|
||||
fmt.Printf("breakpoint set: %s at %#x (func+%#x)%s\n", bp.Label, bp.Addr, bp.Addr-codeBase-uint64(funcOffset), condStr)
|
||||
|
||||
case "info":
|
||||
if len(parts) < 2 {
|
||||
fmt.Println("usage: info break")
|
||||
continue
|
||||
}
|
||||
switch parts[1] {
|
||||
case "break", "breakpoints", "b":
|
||||
fmt.Print(bm.Info())
|
||||
default:
|
||||
fmt.Printf("unknown info target: %s\n", parts[1])
|
||||
}
|
||||
|
||||
case "delete", "d":
|
||||
if len(parts) < 2 {
|
||||
fmt.Println("usage: delete <label|addr>")
|
||||
continue
|
||||
}
|
||||
addr, _ := resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||
if addr == 0 {
|
||||
fmt.Printf("unknown: %s\n", parts[1])
|
||||
continue
|
||||
}
|
||||
if err := bm.Clear(addr); err != nil {
|
||||
fmt.Println(err)
|
||||
} else {
|
||||
fmt.Println("breakpoint removed")
|
||||
}
|
||||
|
||||
case "x":
|
||||
regs, _ := s.GetRegs()
|
||||
addr := regs.RIP // default: current PC
|
||||
length := 64
|
||||
if len(parts) > 1 {
|
||||
addr, _ = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||
}
|
||||
if len(parts) > 2 {
|
||||
length, _ = strconv.Atoi(parts[2])
|
||||
}
|
||||
mem, err := s.ReadMemory(addr, length)
|
||||
if err != nil {
|
||||
fmt.Println(err)
|
||||
continue
|
||||
}
|
||||
hexDump(addr, mem)
|
||||
|
||||
case "w":
|
||||
if len(parts) < 3 {
|
||||
fmt.Println("usage: w <addr> <byte|0x...> [byte...]")
|
||||
continue
|
||||
}
|
||||
addr, _ := resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||
if addr == 0 {
|
||||
fmt.Printf("unknown address: %s\n", parts[1])
|
||||
continue
|
||||
}
|
||||
var bytes []byte
|
||||
for _, arg := range parts[2:] {
|
||||
v, err := strconv.ParseUint(arg, 0, 64)
|
||||
if err != nil {
|
||||
fmt.Printf("invalid value: %s\n", arg)
|
||||
continue
|
||||
}
|
||||
// Write as 8-byte word if it looks like a large value, else single byte.
|
||||
if v > 255 {
|
||||
for j := 0; j < 8; j++ {
|
||||
bytes = append(bytes, byte(v>>(8*j)))
|
||||
}
|
||||
} else {
|
||||
bytes = append(bytes, byte(v))
|
||||
}
|
||||
}
|
||||
if len(bytes) > 0 {
|
||||
if err := s.WriteMemory(addr, bytes); err != nil {
|
||||
fmt.Println(err)
|
||||
} else {
|
||||
fmt.Printf("wrote %d bytes at %#x\n", len(bytes), addr)
|
||||
}
|
||||
}
|
||||
|
||||
case "set":
|
||||
if len(parts) < 3 {
|
||||
fmt.Println("usage: set <reg> <value>")
|
||||
continue
|
||||
}
|
||||
val, err := strconv.ParseUint(parts[2], 0, 64)
|
||||
if err != nil {
|
||||
fmt.Printf("invalid value: %s\n", parts[2])
|
||||
continue
|
||||
}
|
||||
if err := s.SetReg(strings.ToLower(parts[1]), val); err != nil {
|
||||
fmt.Printf("set: %v\n", err)
|
||||
} else {
|
||||
fmt.Printf("%s = %#x\n", parts[1], val)
|
||||
}
|
||||
|
||||
case "labels", "l":
|
||||
sorted := make([]Label, len(labels))
|
||||
copy(sorted, labels)
|
||||
sort.Slice(sorted, func(i, j int) bool { return sorted[i].Offset < sorted[j].Offset })
|
||||
for _, l := range sorted {
|
||||
fmt.Printf(" func+%#04x %s\n", l.Offset, l.Name)
|
||||
}
|
||||
|
||||
case "disas", "u":
|
||||
n := 5
|
||||
if len(parts) > 1 {
|
||||
n, _ = strconv.Atoi(parts[1])
|
||||
if n <= 0 {
|
||||
n = 5
|
||||
}
|
||||
}
|
||||
regs, _ := s.GetRegs()
|
||||
fmt.Print(s.DisassembleN(regs.RIP, n))
|
||||
|
||||
case "where":
|
||||
regs, _ := s.GetRegs()
|
||||
funcOff := int(regs.RIP - codeBase - uint64(funcOffset))
|
||||
line := lineAt(lines, funcOff)
|
||||
label := nearestLabel(labels, funcOff)
|
||||
fmt.Printf(" func+%#x", funcOff)
|
||||
if label != "" {
|
||||
fmt.Printf(" (near %s)", label)
|
||||
}
|
||||
if line > 0 {
|
||||
fmt.Printf(" line %d", line)
|
||||
}
|
||||
fmt.Println()
|
||||
|
||||
case "help", "h", "?":
|
||||
fmt.Println(` break <label|addr> [if <reg> <op> <val>] set a breakpoint
|
||||
delete <label|addr> remove a breakpoint
|
||||
info break list all breakpoints
|
||||
watch <addr> [r|w] set a hardware watchpoint (write by default)
|
||||
unwatch clear all watchpoints
|
||||
step [n], s single-step n instructions
|
||||
next, n step over CALL
|
||||
continue, c run until breakpoint or exit
|
||||
disas [n], u disassemble n instructions at PC
|
||||
regs print registers and RFLAGS
|
||||
where show source line and nearest label
|
||||
stack show stack near RSP (args + return address)
|
||||
x [addr] [len] hex-dump memory
|
||||
w <addr> <val...> write bytes to memory
|
||||
labels, l list function labels
|
||||
help, h, ? this help
|
||||
quit, q kill debuggee and exit`)
|
||||
|
||||
case "stack":
|
||||
regs, _ := s.GetRegs()
|
||||
// For NOSPLIT frame=0: [RSP] = return address, [RSP+8..] = args.
|
||||
retAddr, _ := s.Peek(regs.RSP)
|
||||
fmt.Printf(" [RSP] return addr = %#x\n", retAddr)
|
||||
if argsSize > 0 {
|
||||
fmt.Printf(" args (%d bytes at RSP+8):\n", argsSize)
|
||||
argBytes, err := s.ReadMemory(regs.RSP+8, argsSize)
|
||||
if err == nil {
|
||||
for i := 0; i < argsSize; i += 8 {
|
||||
var v uint64
|
||||
for j := 0; j < 8 && i+j < len(argBytes); j++ {
|
||||
v |= uint64(argBytes[i+j]) << (8 * j)
|
||||
}
|
||||
fmt.Printf(" [%+3d] %#016x\n", i+8, v)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
case "bt", "backtrace":
|
||||
regs, _ := s.GetRegs()
|
||||
funcOff := int(regs.RIP - codeBase - uint64(funcOffset))
|
||||
line := lineAt(lines, funcOff)
|
||||
label := nearestLabel(labels, funcOff)
|
||||
fmt.Printf(" #0 func+%#x", funcOff)
|
||||
if label != "" {
|
||||
fmt.Printf(" (%s)", label)
|
||||
}
|
||||
if line > 0 {
|
||||
fmt.Printf(" [line %d]", line)
|
||||
}
|
||||
fmt.Println()
|
||||
retAddr, _ := s.Peek(regs.RSP)
|
||||
fmt.Printf(" #1 return to %#x\n", retAddr)
|
||||
|
||||
case "watch":
|
||||
if len(parts) < 2 {
|
||||
fmt.Println("usage: watch <addr> [r|w] [size]")
|
||||
continue
|
||||
}
|
||||
addr, _ := resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
|
||||
if addr == 0 {
|
||||
fmt.Printf("unknown address: %s\n", parts[1])
|
||||
continue
|
||||
}
|
||||
typ := WatchWrite
|
||||
size := 8
|
||||
if len(parts) > 2 {
|
||||
switch parts[2] {
|
||||
case "r":
|
||||
typ = WatchRead
|
||||
case "w":
|
||||
typ = WatchWrite
|
||||
}
|
||||
}
|
||||
if len(parts) > 3 {
|
||||
size, _ = strconv.Atoi(parts[3])
|
||||
}
|
||||
// Find a free slot (0-3).
|
||||
slot := -1
|
||||
for i := 0; i < 4; i++ {
|
||||
// Simple: use slot 0 for now.
|
||||
slot = i
|
||||
break
|
||||
}
|
||||
if slot < 0 {
|
||||
fmt.Println("no free watchpoint slots")
|
||||
continue
|
||||
}
|
||||
if err := s.SetWatchpoint(slot, addr, typ, size); err != nil {
|
||||
fmt.Printf("watch: %v\n", err)
|
||||
} else {
|
||||
fmt.Printf("watchpoint %d set: %#x (%s, %d bytes)\n", slot, addr, parts[2], size)
|
||||
}
|
||||
|
||||
case "unwatch":
|
||||
if err := s.ClearAllWatchpoints(); err != nil {
|
||||
fmt.Printf("unwatch: %v\n", err)
|
||||
} else {
|
||||
fmt.Println("all watchpoints cleared")
|
||||
}
|
||||
|
||||
default:
|
||||
fmt.Printf("unknown command: %s\n", cmd)
|
||||
}
|
||||
}
|
||||
s.Kill()
|
||||
}
|
||||
|
||||
func printRegs(regs *Regs, codeBase, funcOff uint64) {
|
||||
fmt.Printf(" RIP = %#016x (func+%#x)\n", regs.RIP, regs.RIP-codeBase-funcOff)
|
||||
fmt.Printf(" RSP = %#016x RBP = %#016x\n", regs.RSP, regs.RBP)
|
||||
fmt.Printf(" RAX = %#016x RBX = %#016x\n", regs.RAX, regs.RBX)
|
||||
fmt.Printf(" RCX = %#016x RDX = %#016x\n", regs.RCX, regs.RDX)
|
||||
fmt.Printf(" RSI = %#016x RDI = %#016x\n", regs.RSI, regs.RDI)
|
||||
fmt.Printf(" R8 = %#016x R9 = %#016x\n", regs.R8, regs.R9)
|
||||
fmt.Printf(" R10 = %#016x R11 = %#016x\n", regs.R10, regs.R11)
|
||||
fmt.Printf(" R12 = %#016x R13 = %#016x\n", regs.R12, regs.R13)
|
||||
fmt.Printf(" R14 = %#016x R15 = %#016x\n", regs.R14, regs.R15)
|
||||
fmt.Printf(" RFLAGS = %#x [%s]\n", regs.RFLAGS, decodeRflags(regs.RFLAGS))
|
||||
}
|
||||
|
||||
func decodeRflags(f uint64) string {
|
||||
var flags string
|
||||
if f&1 != 0 {
|
||||
flags += "CF "
|
||||
}
|
||||
if f&(1<<2) != 0 {
|
||||
flags += "PF "
|
||||
}
|
||||
if f&(1<<4) != 0 {
|
||||
flags += "AF "
|
||||
}
|
||||
if f&(1<<6) != 0 {
|
||||
flags += "ZF "
|
||||
}
|
||||
if f&(1<<7) != 0 {
|
||||
flags += "SF "
|
||||
}
|
||||
if f&(1<<8) != 0 {
|
||||
flags += "TF "
|
||||
}
|
||||
if f&(1<<9) != 0 {
|
||||
flags += "IF "
|
||||
}
|
||||
if f&(1<<10) != 0 {
|
||||
flags += "DF "
|
||||
}
|
||||
if f&(1<<11) != 0 {
|
||||
flags += "OF "
|
||||
}
|
||||
if flags == "" {
|
||||
return "none"
|
||||
}
|
||||
return flags[:len(flags)-1] // trim trailing space
|
||||
}
|
||||
|
||||
func hexDump(addr uint64, data []byte) {
|
||||
for i := 0; i < len(data); i += 16 {
|
||||
end := i + 16
|
||||
if end > len(data) {
|
||||
end = len(data)
|
||||
}
|
||||
fmt.Printf(" %#08x:", addr+uint64(i))
|
||||
for j := i; j < i+16; j++ {
|
||||
if j < end {
|
||||
fmt.Printf(" %02x", data[j])
|
||||
} else {
|
||||
fmt.Print(" ")
|
||||
}
|
||||
}
|
||||
fmt.Print(" ")
|
||||
for j := i; j < end; j++ {
|
||||
if data[j] >= 0x20 && data[j] < 0x7f {
|
||||
fmt.Printf("%c", data[j])
|
||||
} else {
|
||||
fmt.Print(".")
|
||||
}
|
||||
}
|
||||
fmt.Println()
|
||||
}
|
||||
}
|
||||
|
||||
func resolveAddr(s string, codeBase, funcOff uint64, labels []Label) (uint64, string) {
|
||||
// Try as a hex address.
|
||||
if strings.HasPrefix(s, "0x") || strings.HasPrefix(s, "0X") {
|
||||
v, err := strconv.ParseUint(s, 0, 64)
|
||||
if err == nil {
|
||||
return v, ""
|
||||
}
|
||||
}
|
||||
// Try as func+offset.
|
||||
if strings.HasPrefix(s, "+") {
|
||||
off, err := strconv.ParseUint(s[1:], 0, 64)
|
||||
if err == nil {
|
||||
return codeBase + funcOff + off, fmt.Sprintf("func+%#x", off)
|
||||
}
|
||||
}
|
||||
// Try as a label name.
|
||||
for _, l := range labels {
|
||||
if l.Name == s {
|
||||
return codeBase + funcOff + uint64(l.Offset), l.Name
|
||||
}
|
||||
}
|
||||
return 0, ""
|
||||
}
|
||||
|
||||
// lineAt returns the source line for a given function-relative offset.
|
||||
func lineAt(lines []SourceLine, offset int) int {
|
||||
if len(lines) == 0 {
|
||||
return 0
|
||||
}
|
||||
lo, hi := 0, len(lines)-1
|
||||
for lo < hi {
|
||||
mid := (lo + hi + 1) / 2
|
||||
if lines[mid].Offset <= offset {
|
||||
lo = mid
|
||||
} else {
|
||||
hi = mid - 1
|
||||
}
|
||||
}
|
||||
if lines[lo].Offset <= offset {
|
||||
return lines[lo].Line
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// offsetForLine returns the byte offset for a given source line number.
|
||||
// Returns -1 if no instruction is at that line.
|
||||
func offsetForLine(lines []SourceLine, line int) int {
|
||||
for _, le := range lines {
|
||||
if le.Line == line {
|
||||
return le.Offset
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// nearestLabel returns the name of the label at or just before the offset.
|
||||
func nearestLabel(labels []Label, offset int) string {
|
||||
best := ""
|
||||
bestOff := -1
|
||||
for _, l := range labels {
|
||||
if l.Offset <= offset && l.Offset > bestOff {
|
||||
best = l.Name
|
||||
bestOff = l.Offset
|
||||
}
|
||||
}
|
||||
return best
|
||||
}
|
||||
@@ -0,0 +1,117 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// StopReason describes why the debuggee stopped.
|
||||
type StopReason int
|
||||
|
||||
const (
|
||||
StopNone StopReason = iota
|
||||
StopBreakpoint // INT3 breakpoint hit
|
||||
StopWatchpoint // hardware watchpoint triggered
|
||||
StopSingleStep // single-step completed
|
||||
StopSignal // stopped by a signal
|
||||
StopExited // process exited
|
||||
)
|
||||
|
||||
// siginfo_t layout (Linux amd64): si_signo, si_errno, si_code, then union.
|
||||
type siginfoT struct {
|
||||
SiSigno int32
|
||||
SiErrno int32
|
||||
SiCode int32
|
||||
_pad [125]byte
|
||||
}
|
||||
|
||||
const (
|
||||
trapBRKPT = 1 // INT3 breakpoint
|
||||
trapHWBRKPT = 4 // hardware watchpoint
|
||||
)
|
||||
|
||||
// StopInfo returns the reason the debuggee stopped and the faulting address
|
||||
// (for watchpoints, the watched address that was accessed).
|
||||
func (s *Session) StopInfo() (StopReason, uint64) {
|
||||
if s.exited {
|
||||
return StopExited, 0
|
||||
}
|
||||
var info siginfoT
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(syscall.PTRACE_GETSIGINFO),
|
||||
uintptr(s.pid),
|
||||
0,
|
||||
uintptr(unsafe.Pointer(&info)),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return StopNone, 0
|
||||
}
|
||||
if info.SiSigno != int32(syscall.SIGTRAP) {
|
||||
return StopSignal, uint64(info.SiCode)
|
||||
}
|
||||
switch info.SiCode {
|
||||
case trapBRKPT:
|
||||
return StopBreakpoint, 0
|
||||
case trapHWBRKPT:
|
||||
// The faulting address is in si_addr (offset 16 in siginfo_t on amd64).
|
||||
addr := *(*uint64)(unsafe.Pointer(uintptr(unsafe.Pointer(&info)) + 16))
|
||||
return StopWatchpoint, addr
|
||||
default:
|
||||
return StopSingleStep, 0
|
||||
}
|
||||
}
|
||||
|
||||
// SetReg modifies a register value in the debuggee.
|
||||
func (s *Session) SetReg(name string, value uint64) error {
|
||||
regs, err := s.GetRegs()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
switch name {
|
||||
case "rax", "eax", "ax", "al":
|
||||
regs.RAX = value
|
||||
case "rbx", "ebx", "bx", "bl":
|
||||
regs.RBX = value
|
||||
case "rcx", "ecx", "cx", "cl":
|
||||
regs.RCX = value
|
||||
case "rdx", "edx", "dx", "dl":
|
||||
regs.RDX = value
|
||||
case "rsi", "esi", "si":
|
||||
regs.RSI = value
|
||||
case "rdi", "edi", "di":
|
||||
regs.RDI = value
|
||||
case "rbp", "ebp", "bp":
|
||||
regs.RBP = value
|
||||
case "rsp", "esp", "sp":
|
||||
regs.RSP = value
|
||||
case "r8":
|
||||
regs.R8 = value
|
||||
case "r9":
|
||||
regs.R9 = value
|
||||
case "r10":
|
||||
regs.R10 = value
|
||||
case "r11":
|
||||
regs.R11 = value
|
||||
case "r12":
|
||||
regs.R12 = value
|
||||
case "r13":
|
||||
regs.R13 = value
|
||||
case "r14":
|
||||
regs.R14 = value
|
||||
case "r15":
|
||||
regs.R15 = value
|
||||
case "rip", "eip":
|
||||
regs.RIP = value
|
||||
default:
|
||||
return fmt.Errorf("debug: unknown register %q", name)
|
||||
}
|
||||
return s.SetRegs(®s)
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"runtime"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
||||
)
|
||||
|
||||
// RunTarget is the debuggee entry point (gasm debug --target). It
|
||||
// assembles the file, maps the JIT code, registers itself for ptrace,
|
||||
// stops, and then executes the named function. The parent debugger
|
||||
// controls execution from there.
|
||||
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
|
||||
// Parse and assemble.
|
||||
src, err := os.ReadFile(asmPath)
|
||||
if err != nil {
|
||||
return fmt.Errorf("debug target: %w", err)
|
||||
}
|
||||
file, errs := parser.Parse(asmPath, string(src))
|
||||
if len(errs) > 0 {
|
||||
return fmt.Errorf("debug target: parse: %v", errs[0])
|
||||
}
|
||||
img, err := asm.AssembleFile(file)
|
||||
if err != nil {
|
||||
return fmt.Errorf("debug target: assemble: %w", err)
|
||||
}
|
||||
|
||||
// Find the function.
|
||||
var fl *asm.FuncLayout
|
||||
for i := range img.Funcs {
|
||||
if img.Funcs[i].Name == funcName {
|
||||
fl = &img.Funcs[i]
|
||||
break
|
||||
}
|
||||
}
|
||||
if fl == nil {
|
||||
return fmt.Errorf("debug target: function %q not found", funcName)
|
||||
}
|
||||
|
||||
// Map the entire image RWX (we need write access for breakpoints).
|
||||
code := img.Bytes()
|
||||
exec, err := mapRWX(code)
|
||||
if err != nil {
|
||||
return fmt.Errorf("debug target: mmap: %w", err)
|
||||
}
|
||||
|
||||
// Write the code base address for the parent.
|
||||
codeBase := uintptr(unsafe.Pointer(&exec[0]))
|
||||
if err := os.WriteFile(tmpDir+"/codebase", []byte(fmt.Sprintf("%d", codeBase)), 0o644); err != nil {
|
||||
return fmt.Errorf("debug target: write codebase: %w", err)
|
||||
}
|
||||
|
||||
// Write function metadata (offset, size, args) for the parent.
|
||||
meta := fmt.Sprintf("%d %d %d", fl.Offset, fl.Size, fl.Args)
|
||||
os.WriteFile(tmpDir+"/funcmeta", []byte(meta), 0o644)
|
||||
|
||||
// Write label table for breakpoint resolution.
|
||||
labelsFile, _ := os.Create(tmpDir + "/labels")
|
||||
if labelsFile != nil {
|
||||
for label, off := range fl.Labels {
|
||||
fmt.Fprintf(labelsFile, "%s %d\n", label, off)
|
||||
}
|
||||
labelsFile.Close()
|
||||
}
|
||||
|
||||
// Read the argument block.
|
||||
args, err := os.ReadFile(argsFile)
|
||||
if err != nil {
|
||||
return fmt.Errorf("debug target: read args: %w", err)
|
||||
}
|
||||
if len(args) < fl.Args {
|
||||
padded := make([]byte, fl.Args)
|
||||
copy(padded, args)
|
||||
args = padded
|
||||
}
|
||||
|
||||
// Lock this goroutine to the current OS thread so the parent's
|
||||
// ptrace (attached to this thread) controls the JIT execution.
|
||||
runtime.LockOSThread()
|
||||
|
||||
// Request tracing by the parent, then stop. PTRACE_TRACEME makes
|
||||
// the subsequent SIGSTOP a ptrace-stop (not a group-stop), giving
|
||||
// the parent full control from the start.
|
||||
if _, _, errno := syscall.Syscall(syscall.SYS_PTRACE, uintptr(syscall.PTRACE_TRACEME), 0, 0); errno != 0 {
|
||||
return fmt.Errorf("debug target: PTRACE_TRACEME: %v", errno)
|
||||
}
|
||||
os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644)
|
||||
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
|
||||
|
||||
// --- Execution resumes here after the parent continues us ---
|
||||
|
||||
// Prepare the ABI0 stack and call the function.
|
||||
fnAddr := codeBase + uintptr(fl.Offset)
|
||||
stackArgs := make([]byte, fl.Args)
|
||||
copy(stackArgs, args)
|
||||
|
||||
_, callErr := verify.Call(fnAddr, stackArgs)
|
||||
if callErr != nil {
|
||||
// The function returned an error (shouldn't happen for valid code).
|
||||
os.Exit(1)
|
||||
}
|
||||
os.Exit(0)
|
||||
return nil
|
||||
}
|
||||
|
||||
// mapRWX maps code into a read-write-execute region (needed for
|
||||
// breakpoint patching via ptrace POKETEXT, though ptrace can write
|
||||
// to any mapping regardless of permissions).
|
||||
func mapRWX(code []byte) ([]byte, error) {
|
||||
const pageSize = 4096
|
||||
size := (len(code) + pageSize - 1) &^ (pageSize - 1)
|
||||
mem, err := syscall.Mmap(-1, 0, size,
|
||||
syscall.PROT_READ|syscall.PROT_WRITE|syscall.PROT_EXEC,
|
||||
syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
copy(mem, code)
|
||||
return mem, nil
|
||||
}
|
||||
@@ -0,0 +1,90 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package debug
|
||||
|
||||
// Regs holds the full general-purpose register set of a traced process
|
||||
// (the Linux amd64 user_regs_struct layout).
|
||||
type Regs struct {
|
||||
R15 uint64
|
||||
R14 uint64
|
||||
R13 uint64
|
||||
R12 uint64
|
||||
RBP uint64
|
||||
RBX uint64
|
||||
R11 uint64
|
||||
R10 uint64
|
||||
R9 uint64
|
||||
R8 uint64
|
||||
RAX uint64
|
||||
RCX uint64
|
||||
RDX uint64
|
||||
RSI uint64
|
||||
RDI uint64
|
||||
OrigRAX uint64
|
||||
RIP uint64
|
||||
CS uint64
|
||||
RFLAGS uint64
|
||||
RSP uint64
|
||||
SS uint64
|
||||
FSBase uint64
|
||||
GSBase uint64
|
||||
DS uint64
|
||||
ES uint64
|
||||
FS uint64
|
||||
GS uint64
|
||||
}
|
||||
|
||||
// tracer abstracts the minimal ptrace operations needed by the breakpoint
|
||||
// manager and the stop-information helpers. The live implementation is
|
||||
// *Session (ptrace_linux_amd64.go); tests supply a mock.
|
||||
type tracer interface {
|
||||
Peek(addr uint64) (uint64, error)
|
||||
Poke(addr uint64, val uint64) error
|
||||
SetRegs(regs *Regs) error
|
||||
Pid() int
|
||||
}
|
||||
|
||||
// mockTracer records Peek/Poke calls and provides fake register state.
|
||||
type mockTracer struct {
|
||||
mem map[uint64]byte
|
||||
peeks []uint64
|
||||
pokes []struct {
|
||||
addr uint64
|
||||
val uint64
|
||||
}
|
||||
regs *Regs
|
||||
}
|
||||
|
||||
func newMockTracer() *mockTracer {
|
||||
return &mockTracer{
|
||||
mem: make(map[uint64]byte),
|
||||
regs: &Regs{},
|
||||
}
|
||||
}
|
||||
|
||||
func (m *mockTracer) Peek(addr uint64) (uint64, error) {
|
||||
m.peeks = append(m.peeks, addr)
|
||||
var val uint64
|
||||
for i := uint64(0); i < 8; i++ {
|
||||
val |= uint64(m.mem[addr+i]) << (i * 8)
|
||||
}
|
||||
return val, nil
|
||||
}
|
||||
|
||||
func (m *mockTracer) Poke(addr uint64, val uint64) error {
|
||||
m.pokes = append(m.pokes, struct {
|
||||
addr uint64
|
||||
val uint64
|
||||
}{addr, val})
|
||||
for i := uint64(0); i < 8; i++ {
|
||||
m.mem[addr+i] = byte(val >> (i * 8))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (m *mockTracer) SetRegs(regs *Regs) error {
|
||||
m.regs = regs
|
||||
return nil
|
||||
}
|
||||
func (m *mockTracer) Pid() int { return 42 }
|
||||
@@ -0,0 +1,143 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build linux && amd64
|
||||
|
||||
package debug
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// Hardware watchpoint support via x86-64 debug registers (DR0-DR3, DR7).
|
||||
//
|
||||
// DR0-DR3 hold the watched addresses. DR7 is the control register:
|
||||
// bits 0,2,4,6: local enable for DR0-DR3
|
||||
// bits 16-17,20-21,24-25,28-29: R/W type (00=exec, 01=write, 11=read/write)
|
||||
// bits 18-19,22-23,26-27,30-31: length (00=1, 01=2, 10=8, 11=4)
|
||||
|
||||
// WatchpointType selects what triggers the watchpoint.
|
||||
type WatchpointType int
|
||||
|
||||
const (
|
||||
WatchWrite WatchpointType = 1 // trigger on write
|
||||
WatchRead WatchpointType = 3 // trigger on read or write
|
||||
)
|
||||
|
||||
// SetWatchpoint installs a hardware watchpoint on the given address.
|
||||
// slot is 0-3 (four hardware watchpoints available).
|
||||
func (s *Session) SetWatchpoint(slot int, addr uint64, typ WatchpointType, size int) error {
|
||||
if slot < 0 || slot > 3 {
|
||||
return fmt.Errorf("debug: watchpoint slot must be 0-3")
|
||||
}
|
||||
|
||||
// Determine the length encoding.
|
||||
var lenBits uint64
|
||||
switch size {
|
||||
case 1:
|
||||
lenBits = 0
|
||||
case 2:
|
||||
lenBits = 1
|
||||
case 4:
|
||||
lenBits = 3
|
||||
case 8:
|
||||
lenBits = 2
|
||||
default:
|
||||
return fmt.Errorf("debug: watchpoint size must be 1, 2, 4, or 8")
|
||||
}
|
||||
|
||||
// Write the watched address to DR0-DR3.
|
||||
var drAddr uintptr
|
||||
switch slot {
|
||||
case 0:
|
||||
drAddr = 0x0 // DR0 offset in user_regs_struct
|
||||
case 1:
|
||||
drAddr = 0x8 // DR1
|
||||
case 2:
|
||||
drAddr = 0x10 // DR2
|
||||
case 3:
|
||||
drAddr = 0x18 // DR3
|
||||
}
|
||||
|
||||
// PTRACE_POKEUSER writes to the debuggee's user area (includes debug regs).
|
||||
if err := ptracePokeUser(s.pid, drAddr, addr); err != nil {
|
||||
return fmt.Errorf("debug: set DR%d: %w", slot, err)
|
||||
}
|
||||
|
||||
// Read the current DR7, set the enable and type bits, write it back.
|
||||
dr7, err := ptracePeekUser(s.pid, 0x38) // DR7 offset
|
||||
if err != nil {
|
||||
return fmt.Errorf("debug: read DR7: %w", err)
|
||||
}
|
||||
|
||||
enableBit := uint64(1) << (2 * slot) // local enable
|
||||
rwBits := uint64(typ) << (16 + 4*slot) // R/W type
|
||||
lenField := lenBits << (18 + 4*slot) // length
|
||||
|
||||
// Clear the existing bits for this slot, then set the new ones.
|
||||
mask := ^((uint64(1) << (2 * slot)) | (uint64(3) << (16 + 4*slot)) | (uint64(3) << (18 + 4*slot)))
|
||||
dr7 = (dr7 & mask) | enableBit | rwBits | lenField
|
||||
|
||||
if err := ptracePokeUser(s.pid, 0x38, dr7); err != nil {
|
||||
return fmt.Errorf("debug: set DR7: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ClearWatchpoint removes a hardware watchpoint.
|
||||
func (s *Session) ClearWatchpoint(slot int) error {
|
||||
if slot < 0 || slot > 3 {
|
||||
return fmt.Errorf("debug: watchpoint slot must be 0-3")
|
||||
}
|
||||
// Read DR7, clear the enable bit for this slot.
|
||||
dr7, err := ptracePeekUser(s.pid, 0x38)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dr7 &^= uint64(1) << (2 * slot) // disable
|
||||
return ptracePokeUser(s.pid, 0x38, dr7)
|
||||
}
|
||||
|
||||
// ClearAllWatchpoints removes all hardware watchpoints.
|
||||
func (s *Session) ClearAllWatchpoints() error {
|
||||
for slot := 0; slot < 4; slot++ {
|
||||
if err := s.ClearWatchpoint(slot); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ptracePokeUser writes a value to the debuggee's user area at the given offset.
|
||||
func ptracePokeUser(pid int, offset uintptr, val uint64) error {
|
||||
const ptracePokeuser = 6 // PTRACE_POKEUSER
|
||||
_, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(ptracePokeuser),
|
||||
uintptr(pid),
|
||||
offset,
|
||||
uintptr(val),
|
||||
0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return errno
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ptracePeekUser reads a value from the debuggee's user area at the given offset.
|
||||
func ptracePeekUser(pid int, offset uintptr) (uint64, error) {
|
||||
const ptracePeekuser = 3 // PTRACE_PEEKUSER
|
||||
val, _, errno := syscall.Syscall6(
|
||||
syscall.SYS_PTRACE,
|
||||
uintptr(ptracePeekuser),
|
||||
uintptr(pid),
|
||||
offset,
|
||||
0, 0, 0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return 0, errno
|
||||
}
|
||||
return uint64(val), nil
|
||||
}
|
||||
+142
-29
@@ -2,6 +2,8 @@
|
||||
|
||||
How gasm-devkit is put together and why.
|
||||
|
||||
Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
|
||||
|
||||
## Design goals
|
||||
|
||||
1. **A real AST, not a grammar hack.** The linter, analyser, assembler and
|
||||
@@ -136,13 +138,18 @@ Two deeper analyses sit on top of the AST:
|
||||
control-flow graph (basic blocks split at labels and after branches, with
|
||||
fall-through and jump-target edges), computes a conservative per-instruction
|
||||
register def/use, and runs the standard backward liveness iteration to a fixed
|
||||
point. On top of that it flags a **callee-saved register that is written but
|
||||
never saved and restored** — the per-architecture callee-saved set is amd64
|
||||
`BX/BP/R12–R15`, arm64 `R19–R30`, riscv64 `X1/X8/X9/X18–X27`, loong64
|
||||
`R1/R22–R31`. This is an *audit*: the runtime's own assembly clobbers these
|
||||
registers freely (it controls both sides of the call), so the rule is
|
||||
advisory there, but in hand-written kernels called from ordinary Go code a
|
||||
clobber is a genuine ABI violation. It runs only on macro-free files, where
|
||||
point. On top of that it flags writes to the registers the **Go ABI** fixes
|
||||
across calls that are never saved and restored — calibrated from
|
||||
`cmd/compile/abi-internal.md`, *not* the platform ABI: Go's stack-based ABI0
|
||||
has no System V style callee-saved registers (amd64 `BX`, `R12`–`R15` and
|
||||
the like are caller-saved or permanent scratch, and hand-written kernels may
|
||||
clobber them freely). The audited set is the frame pointer and the
|
||||
the frame pointer, the goroutine pointer per architecture (amd64 `BP`/`R14`, arm64 `R18`/`R28`/
|
||||
`R29`, riscv64 `X27`, loong64 `R22`); the goroutine pointer is reported only
|
||||
when the function can reach the runtime — it is not `NOSPLIT` or makes a
|
||||
call — since the ABI0 transition machinery restores it on those paths, and
|
||||
NOSPLIT call-free leaves may use it (the runtime's own assembly does). It
|
||||
runs only on macro-free files, where
|
||||
no opaque macro can perform the save/restore.
|
||||
- **`funcdata-pcdata`.** `FUNCDATA $idx, sym(SB)` and `PCDATA $idx, $val` are
|
||||
checked for well-formed operands (arity, immediate index and value, symbol
|
||||
@@ -152,9 +159,15 @@ Two deeper analyses sit on top of the AST:
|
||||
### `format`
|
||||
|
||||
The formatter works on the **token stream, not the AST**, so it preserves
|
||||
every line — comments and blanks included. It only normalises indentation,
|
||||
operand spacing and per-function mnemonic alignment. It is idempotent and its
|
||||
output always round-trips through the parser.
|
||||
every line — comments and blanks included. It normalises indentation, operand
|
||||
spacing, per-function mnemonic alignment and blank-line layout: a new block
|
||||
(a label, `TEXT` or `GLOBL`) is preceded by exactly one blank line (comments
|
||||
leading a block stay with it), runs of blanks collapse to one, and a `RET`
|
||||
terminates the body so the next function's doc comment stays at column 0. It
|
||||
is idempotent and its output always round-trips through the parser. With a
|
||||
directory argument — or none — it reformats every `.s` file below it in
|
||||
place and lists the files changed, the way `go fmt` does (`.` and `_`
|
||||
directories are skipped).
|
||||
|
||||
### `lsp`
|
||||
|
||||
@@ -182,33 +195,133 @@ Every encoding is validated by decoding it again with `golang.org/x/arch` — th
|
||||
one module dependency, used in tests only and never linked into the binary.
|
||||
|
||||
On top of the encoder, `Assemble` walks a parsed `TEXT` body, converts each
|
||||
operand to an encoder operand, and lays the instructions out in two passes so
|
||||
local labels resolve to fixed rel32 jump offsets. The `FP`/`SP` pseudo-
|
||||
operand to an encoder operand, and lays the instructions out so local labels
|
||||
resolve to relative jump offsets: jumps start in the short (rel8) form and
|
||||
expand to rel32 when the settled displacement does not fit, iterating to a
|
||||
fixed point, and jump-to-jump chains are folded (a conditional jump to a label
|
||||
whose only instruction is an unconditional jump is redirected to the ultimate
|
||||
target) exactly as the Go toolchain's linker does before it encodes branches.
|
||||
The `FP`/`SP` pseudo-
|
||||
registers are translated onto the hardware stack pointer — `x+N(FP)` becomes
|
||||
`(N+8)(SP)` for a zero-frame function and `(N+frame+16)(SP)` once a frame
|
||||
pointer is set up, with the matching Go prologue/epilogue generated — so the
|
||||
output is byte-identical to the Go assembler for these cases. SIMD is handled
|
||||
by a VEX (AVX/AVX2) encoder — the two- and three-byte VEX prefixes with XMM/YMM
|
||||
registers — across seven operand forms: the three-operand NDS form, the
|
||||
two-operand reg/rm form, the immediate-shift form, the immediate shuffle form
|
||||
(`VPSHUFD`, `VPERMQ`), the three-operand-plus-immediate form (`VSHUFPD`,
|
||||
registers — across eight operand forms: the three-operand NDS form, the
|
||||
two-operand reg/rm form, the immediate-shift form (plus the variable-count
|
||||
shifts, which share the NDS shape with the count in an XMM register or
|
||||
memory), the immediate shuffle form (`VPSHUFD`, `VPERMQ`), the
|
||||
three-operand-plus-immediate form (`VSHUFPD`,
|
||||
`VPERM2I128`, `VINSERTI128`), the lane-extract form (`VEXTRACTI128`,
|
||||
lane-extract form (`VEXTRACTI128`,
|
||||
`VEXTRACTF128`, where the YMM source occupies the reg field and the XMM or
|
||||
memory destination r/m), the direction-sensitive moves (`VMOVDQU`, `VMOVUPD`,
|
||||
`VMOVD`, `VMOVQ`, `VMOVSD`), the floating-point and FMA arithmetic (`VADDPD`,
|
||||
`VMULPD`, `VXORPD`, `VUNPCKHPD`, the scalar `VADDSD`/`VMULSD`, `VCVTDQ2PD`,
|
||||
`VFMADD231PD`) and the no-operand `VZEROUPPER` — together with `VPERMD`,
|
||||
covering every integer, shuffle and FP instruction the go-flac AVX2 kernels
|
||||
use. Every encoding is validated two ways: by round-trip decoding
|
||||
through `golang.org/x/arch`, and byte-for-byte against the machine code the
|
||||
real Go assembler emits (which also locks the v̄vvv = 1111 rule for unused
|
||||
vvvv fields — a value the hardware rejects with #UD and the decoder silently
|
||||
ignores). This increment covers register / memory / immediate / FP-frame
|
||||
operands, local-label jumps and these VEX SIMD forms; EVEX / AVX-512, `SB`
|
||||
(global symbol) operands (relocations), a handful of scalar gaps the kernels
|
||||
hit (`CMOVcc`, `SETcc`, `LZCNT`, `MOVSX`/`MOVZX`) and object-file emission
|
||||
are the rest of Phase 2.
|
||||
`VMOVD`, `VMOVQ`, `VMOVSD`), the floating-point and FMA arithmetic — the
|
||||
packed double operations (`VADDPD`/`VSUBPD`/`VMULPD`/`VDIVPD`/`VMINPD`/
|
||||
`VMAXPD`), the unpacks (`VUNPCKHPD`/`VUNPCKLPD`), the scalar SD and SS
|
||||
operations, `VMOVDDUP`, `VXORPD`, the width-changing conversions
|
||||
(`VCVTDQ2PS`, `VCVTPS2PD`, `VCVTDQ2PD`, and the `VCVTPD2DQX`/`Y` and
|
||||
`VCVTTPD2DQX`/`Y` spellings, whose length follows the wider source) and
|
||||
`VFMADD231PD` — and the no-operand `VZEROUPPER`, together with `VPERMD` and
|
||||
the scalar families (`CMOVcc`, `SETcc`, `LZCNT`/`TZCNT`, the extending moves,
|
||||
`CVTSx2SD`, `IMUL3`) and the EVEX (AVX-512) prefix — the four-byte prefix with
|
||||
5-bit register fields (Z0–Z31, X/Y 16–31, with the mod=11 quirk that carries
|
||||
rm[4] in X̄), opmask registers (K0–K7 as operands, mask destinations and
|
||||
explicit merging/zeroing masks — written the way Go writes them, as a K
|
||||
operand among the operands plus a `.Z` mnemonic suffix), and the compressed
|
||||
disp8×N displacement, whose multiplier follows the memory operand's size —
|
||||
covering every instruction the go-flac and go-lz4 AVX2/AVX-512 kernels use,
|
||||
plus the common AVX-512 F/BW integer set, the floating-point and conversion
|
||||
set (the packed double and single arithmetic, the scalar SD/SS forms —
|
||||
whose EVEX encodings serve masked and zeroing use — `VMOVDDUP`, the
|
||||
replicating moves, and the width-changing conversions, including the
|
||||
`VCVTPD2DQ`/`VCVTTPD2DQ` family whose length follows the wider source
|
||||
operand), and the wider AVX-512 set: ternary logic, lane shuffles, inserts
|
||||
and extracts, compares with an opmask destination, the permutes, the
|
||||
expand/compress family, the broadcasts, the opmask-register instructions
|
||||
(KAND/KOR/KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST and KMOVQ), the aligned
|
||||
moves and the remaining extending/narrowing moves, the floating-point
|
||||
helper and conversion tail (VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*,
|
||||
VSCALEF*, VRNDSCALE*, VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with an
|
||||
opmask destination, and the VCVT* conversions — signed, unsigned and
|
||||
truncating, including the length-suffixed X/Y spellings and the
|
||||
mask/vector conversions VPMOVM2*/VPMOV*2M, and the scalar conversions
|
||||
between vector and general-purpose registers (VCVT{,T}S{D,S}2SI{,Q} and
|
||||
the unsigned forms, VCVTSI2*/VCVTUSI2*), and gather/scatter with VSIB addressing — both the
|
||||
VEX spelling with a vector mask register and the EVEX spelling with an
|
||||
explicit K mask, where the EVEX length follows the VSIB index register,
|
||||
not the data register. The EVEX mnemonic
|
||||
suffixes — rounding modes (.RN_SAE/.RD_SAE/.RU_SAE/.RZ_SAE),
|
||||
suppress-all-exceptions (.SAE) and memory broadcast (.BCST) — set the EVEX
|
||||
b bit and the L'L rounding-control field (broadcast keeps the vector length
|
||||
and scales disp8 by the element size), and combine with the .Z zeroing
|
||||
suffix. Every encoding is validated two ways: by
|
||||
round-trip decoding through `golang.org/x/arch`, and byte-for-byte against
|
||||
the machine code the real Go assembler emits — a comparison that holds for
|
||||
whole functions: all 27 functions of both kernels assemble to exactly the Go
|
||||
toolchain's bytes, the lone exception being the displacements of the
|
||||
static-constant loads, which the Go linker fills at link time.
|
||||
|
||||
File-level assembly (`AssembleFile`) goes beyond single functions: it
|
||||
materialises the file's static symbols (`GLOBL`/`DATA`) in a data section
|
||||
behind the code and resolves references to them (`mask<>(SB)`) to
|
||||
RIP-relative loads whose displacements point inside the resulting image, so
|
||||
the bytes are self-consistent at any base address. References to symbols no
|
||||
`GLOBL` defines are kept as relocations on the function layout, and the
|
||||
object-file emitters turn the whole image into a linkable object: the ELF
|
||||
and Mach-O writers (`gasm asm --format elf|macho`) lay the code and data out
|
||||
as `.text`/`.data` (or `__text`/`__data`) sections, export a symbol per
|
||||
`TEXT` and `GLOBL` (the `<>` ones local, the rest global) and emit one
|
||||
PC-relative relocation per static-symbol reference — undefined external
|
||||
symbols included, so the output links with the system toolchain. The GOOBJ
|
||||
emitter (`gasm asm --format goobj`) writes the format the Go linker consumes
|
||||
directly: the functions as non-package symbols (the way `cmd/asm` records
|
||||
assembly symbols), the `GLOBL` data, one `FuncInfo` per function and the
|
||||
pc-value tables — `pcsp` built from the prologue and epilogue stack
|
||||
boundaries, plus flat `pcfile`, `pcline` and `pcinline` tables — so a
|
||||
gasm-assembled object drops into a `go build` in place of the toolchain's.
|
||||
The object preamble (the version-and-experiment header the linker compares
|
||||
verbatim) is captured from the installed `go tool asm`, so the output is
|
||||
always consistent with the toolchain that links it. External cross-package
|
||||
references and the implicit funcdata/DWARF symbols remain future work (the
|
||||
linker fills the latter's defaults); the rest of Phase 2 is those, the
|
||||
remaining EVEX forms and the other architectures.
|
||||
|
||||
### `verify`
|
||||
|
||||
The dynamic-analysis substrate (Phase 3). It JIT-loads assembled images into
|
||||
executable memory and invokes them directly, enabling differential testing,
|
||||
runtime ABI checks and coverage profiling.
|
||||
|
||||
The execution model is pure Go (stdlib only). `Map` copies machine code into
|
||||
an anonymous `syscall.Mmap` mapping and enforces W^X (write the bytes, then
|
||||
`mprotect` to read-execute). `Call` prepares a stack whose first word is the
|
||||
address of an assembly trampoline (`leaveJIT`), lays the ABI0 argument
|
||||
block after it, switches to that stack via `enterJIT` (which saves the Go
|
||||
stack pointer in a package global and jumps to the target), and recovers
|
||||
control when the function RETs into `leaveJIT` (which restores the Go stack
|
||||
and returns). A 64-byte pad below the return address accommodates the
|
||||
ABIInternal wrapper that the Go runtime interposes on assembly functions.
|
||||
|
||||
`Load` / `LoadSource` / `LoadAST` parse, assemble and map a `.s` file in one
|
||||
step, returning a `Kernel` whose `CallFunc` method marshals the argument block
|
||||
by name. The image must be self-contained (no external relocations); the
|
||||
assembler’s `Image.Bytes()` provides the code-and-data concatenation.
|
||||
|
||||
The `gasm verify` CLI subcommand exposes this: it loads a file, reports the
|
||||
available functions and (with `-smoke`) calls each NOSPLIT function with zeroed
|
||||
arguments to confirm the trampoline round-trips.
|
||||
|
||||
### `debug`
|
||||
|
||||
The interactive debugger (Phase 4, linux/amd64). It launches the target
|
||||
function in a child process that maps the JIT code, calls
|
||||
`PTRACE_TRACEME`, and stops; the parent attaches via ptrace and controls
|
||||
execution. Breakpoints are patched as INT3 bytes through `/proc/pid/mem`
|
||||
(PTRACE_PEEKTEXT is unreliable with Go's multi-threaded runtime).
|
||||
The child pins its goroutine to the OS thread with `runtime.LockOSThread`
|
||||
so the traced thread is the one executing JIT code. The REPL provides
|
||||
single-step, register inspection, label resolution, and breakpoint
|
||||
management.
|
||||
|
||||
## Extension points
|
||||
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
# Deferred decisions
|
||||
|
||||
Design decisions deliberately postponed, with enough context to pick them up
|
||||
again without re-deriving the analysis. Each entry records what is deferred,
|
||||
why, the options on the table, and the trigger that should reopen it.
|
||||
|
||||
---
|
||||
|
||||
## GOOBJ external (cross-package) symbol references
|
||||
|
||||
**Status:** deferred (v0.15.0, 2026-08-02). The GOOBJ emitter resolves only
|
||||
symbols defined in the file being assembled; a reference to any other symbol
|
||||
is rejected.
|
||||
|
||||
**Why it is deferred.** GOOBJ symbol references are *positional*: a
|
||||
reference is a `{PkgIdx, SymIdx}` pair, where `SymIdx` is the index of the
|
||||
symbol in the *referenced package's* symbol-definition table. That ordering
|
||||
is not derivable from the reference site — it lives in the referenced
|
||||
package's gc export data (the iexport binary format, which evolves with the
|
||||
toolchain). `cmd/asm` reads it with `cmd/internal` readers gasm cannot
|
||||
import, so emitting external references means either parsing export data
|
||||
ourselves or taking a dependency that does.
|
||||
|
||||
**What works today.** Single-package objects: every symbol the file defines
|
||||
(as `TEXT` or `GLOBL`, static or exported) and every reference to them.
|
||||
This covers the production use case — the go-flac / go-lz4 kernels carry no
|
||||
`FUNCDATA`/`PCDATA`, hence no references into `runtime`, and the Go side
|
||||
references the assembly symbols, never the reverse. Such a package builds
|
||||
with its assembly object replaced by a gasm-emitted one.
|
||||
|
||||
**The options, when we return.**
|
||||
|
||||
1. **`golang.org/x/tools/go/gcexportdata` as a production dependency.**
|
||||
The straightforward path: read each imported package's export file
|
||||
(paths from `-importcfg` or `go list -export`), assign symbol indices in
|
||||
its symbol order, write `PkgIndex`/`Autolib` entries (fingerprints from
|
||||
the export files' build IDs) and positional references. Robust across
|
||||
toolchain versions — `x/tools` tracks the format. **Cost:** the first
|
||||
production dependency beyond the standard library, an explicit deviation
|
||||
from the "production code depends only on the standard library"
|
||||
principle in the README. Requires the user's explicit agreement.
|
||||
2. **A minimal iexport parser of our own.** Preserves self-containment.
|
||||
Substantial effort and inherently fragile: the format is an internal
|
||||
contract that changes with Go releases, so the parser needs a
|
||||
version-gated fallback and regression tests against several toolchains.
|
||||
3. **Shell out to the toolchain for symbol metadata.** Consistent with the
|
||||
existing GOOBJ preamble probe (which already runs `go tool asm`), but no
|
||||
toolchain command exposes a package's symbols *in definition-index
|
||||
order* — `go tool nm` sorts differently — so this does not solve the
|
||||
core problem on its own; it would only feed option 1 or 2.
|
||||
|
||||
**Trigger to reopen.** An assembly file that needs a cross-package
|
||||
reference — in practice `FUNCDATA $…, runtime·…(SB)` (stack maps / GC
|
||||
metadata written in assembly), or any kernel that calls into another
|
||||
package directly. Until then, option 3's limitation is moot and the
|
||||
single-package emitter suffices.
|
||||
+110
@@ -0,0 +1,110 @@
|
||||
# CLI Reference
|
||||
|
||||
Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
|
||||
|
||||
`gasm` is a single binary with subcommands. Run `gasm --help` for an
|
||||
overview, or `gasm <command> -h` for a command's usage and flags.
|
||||
|
||||
## Global Flags
|
||||
|
||||
| Flag | Description |
|
||||
|------|-------------|
|
||||
| `-h`, `--help` | Show help |
|
||||
| `-V`, `--version` | Print the version |
|
||||
|
||||
## `gasm tokens <file>`
|
||||
|
||||
Print the lexical token stream of FILE: position, token kind, and text,
|
||||
one token per line. FILE may be `-` to read standard input.
|
||||
|
||||
## `gasm parse <file>`
|
||||
|
||||
Parse FILE and report syntax errors on stderr. On success, prints how
|
||||
many declarations and TEXT functions the file contains.
|
||||
|
||||
## `gasm fmt [-w] [path...]`
|
||||
|
||||
Canonicalise the formatting of Plan 9 assembly sources: indentation,
|
||||
operand spacing, per-function mnemonic alignment, and blank-line layout.
|
||||
|
||||
| Flag | Description |
|
||||
|------|-------------|
|
||||
| `-w` | Write result to the source file (default: print to stdout) |
|
||||
|
||||
With no arguments, or with a directory argument, every `.s` file below
|
||||
it is reformatted in place and the names of changed files are listed
|
||||
(`go fmt` style). `.` and `_` directories are skipped.
|
||||
|
||||
## `gasm lint <file...>`
|
||||
|
||||
Run static checks and print diagnostics as
|
||||
`file:line:col: severity: message [code]`. Exit status is non-zero when
|
||||
an error-severity diagnostic is found.
|
||||
|
||||
| Flag | Description |
|
||||
|------|-------------|
|
||||
| `-disable` | Comma-separated rule codes to disable |
|
||||
|
||||
Rules: `unknown-instruction`, `operand-count`, `undefined-label`,
|
||||
`duplicate-label`, `missing-ret`, `missing-textflag-include`,
|
||||
`abi-argsize`, `unreachable-code`, `register-clobber`,
|
||||
`funcdata-pcdata`.
|
||||
|
||||
## `gasm asm [--format raw|elf|macho|goobj] [-p pkg] [-o out] <file>`
|
||||
|
||||
Assemble FILE (amd64) to machine code.
|
||||
|
||||
| Flag | Description |
|
||||
|------|-------------|
|
||||
| `--format` | Output format: `raw` (default), `elf`, `macho`, `goobj` |
|
||||
| `-p` | Package path (required for `--format goobj`) |
|
||||
| `-o` | Write output to file (default: hex dump to stdout) |
|
||||
|
||||
## `gasm verify [flags] <file.s>`
|
||||
|
||||
Assemble FILE, map it into executable memory, and run dynamic checks.
|
||||
|
||||
| Flag | Description |
|
||||
|------|-------------|
|
||||
| `--ground-truth` | Compare machine code byte-for-byte against `go tool asm` |
|
||||
| `--fuzz` | Differential fuzz: JIT both gasm and go-tool-asm, compare outputs |
|
||||
| `-n` | Fuzz iterations per function (default: 1000) |
|
||||
| `--abi` | Run ABI-checking calls (sentinel registers + red zone) |
|
||||
| `--profile` | List basic-block structure per function |
|
||||
| `--smoke` | Call each NOSPLIT function with zeroed args |
|
||||
|
||||
The `--fuzz` mode runs each function in a subprocess; a partial function
|
||||
(e.g. a decoder that faults on malformed input) is reported as
|
||||
`CRASH` without killing the parent. Use `--ground-truth` for decoders.
|
||||
|
||||
## `gasm debug --func <name> <file.s>`
|
||||
|
||||
Interactive debugger for JIT-assembled amd64 functions. Requires a
|
||||
compiled binary on `$PATH` (not `go run`).
|
||||
|
||||
| Flag | Description |
|
||||
|------|-------------|
|
||||
| `--func` | Function to debug (required) |
|
||||
|
||||
REPL commands:
|
||||
|
||||
| Command | Description |
|
||||
|---------|-------------|
|
||||
| `break <label\|addr>` | Set a breakpoint |
|
||||
| `step [n]` | Single-step n instructions |
|
||||
| `continue` | Run until next breakpoint or exit |
|
||||
| `regs` | Print general-purpose registers |
|
||||
| `x [addr] [len]` | Hex-dump memory |
|
||||
| `labels` | List function labels and offsets |
|
||||
| `quit` | Kill the debuggee and exit |
|
||||
|
||||
## `gasm lsp`
|
||||
|
||||
Run the language server over standard input/output (JSON-RPC 2.0 with
|
||||
Content-Length framing). Point an LSP-capable editor at the binary and
|
||||
associate it with `.s` files. The target architecture is inferred from
|
||||
the file-name suffix (`_amd64.s`, `_arm64.s`, `_riscv64.s`,
|
||||
`_loong64.s`).
|
||||
|
||||
Provides: completion, hover, document symbols, diagnostics, and
|
||||
semantic-token highlighting.
|
||||
@@ -0,0 +1,108 @@
|
||||
# Development Guide
|
||||
|
||||
Repository: [sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit)
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- **Go** 1.26+ with `toolchain go1.26.5`
|
||||
- **just** — the command runner; every task below is a just recipe
|
||||
- No external dependencies beyond the Go toolchain
|
||||
|
||||
## Quick Start
|
||||
|
||||
```sh
|
||||
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
|
||||
cd gasm-devkit
|
||||
just install # go mod download
|
||||
just build # go vet + gofmt — must pass with zero output
|
||||
just test # full suite, race detector, 80 % coverage gate
|
||||
```
|
||||
|
||||
## Just Recipes
|
||||
|
||||
### `just build`
|
||||
|
||||
Runs `go vet ./...` and checks `gofmt -l .` produces no output. This is
|
||||
the minimum bar before any commit.
|
||||
|
||||
### `just test`
|
||||
|
||||
```sh
|
||||
go test -race -count=1 -coverprofile=coverage.out ./...
|
||||
```
|
||||
|
||||
Plus an `awk` gate that fails if total coverage is below 80 %.
|
||||
|
||||
### `just fmt`
|
||||
|
||||
```sh
|
||||
gofmt -w .
|
||||
```
|
||||
|
||||
Run after editing any Go source. The output must be idempotent.
|
||||
|
||||
### `just run -- <args>`
|
||||
|
||||
Runs the CLI via `go run` with the version string stamped:
|
||||
|
||||
```sh
|
||||
just run -- lint kernel_amd64.s
|
||||
just run -- fmt -w kernel_amd64.s
|
||||
just run -- verify --ground-truth kernel_amd64.s
|
||||
```
|
||||
|
||||
### `just install-bin`
|
||||
|
||||
Installs the `gasm` binary into `$GOBIN` with the release version
|
||||
embedded via `-ldflags "-X main.version=..."`.
|
||||
|
||||
### `just gen`
|
||||
|
||||
Regenerates the architecture instruction tables in `arch/` by parsing
|
||||
the Go toolchain's own assembler source
|
||||
(`$GOROOT/src/cmd/internal/obj/<arch>/anames.go`). Requires a Go
|
||||
installation. Output is committed — no runtime dependency on the
|
||||
toolchain.
|
||||
|
||||
### `just uninstall`
|
||||
|
||||
Removes `coverage.out`, the `gasm` binary, and `*.test` artefacts.
|
||||
|
||||
## Running Individual Tests
|
||||
|
||||
```sh
|
||||
go test -run TestVexGroundTruth ./asm/
|
||||
go test -run TestDifferentialLZ4Fuzz ./verify/
|
||||
go test -run TestFLACDecorrelate ./verify/
|
||||
go test -run TestGOObjectLinkAndRun ./asm/
|
||||
```
|
||||
|
||||
## Debugger Note
|
||||
|
||||
`gasm debug` spawns a child process from the binary on `$PATH`. It does
|
||||
not work with `go run` — install first:
|
||||
|
||||
```sh
|
||||
just install-bin
|
||||
gasm debug --func decodeBlockAVX2 path/to/kernel_amd64.s
|
||||
```
|
||||
|
||||
## Project Layout
|
||||
|
||||
```
|
||||
cmd/gasm/ CLI entry point (subcommands)
|
||||
token/ Lexical token kinds and positions
|
||||
lexer/ Hand-written scanner
|
||||
ast/ Abstract syntax tree
|
||||
parser/ Line-oriented parser
|
||||
arch/ Register and instruction tables (generated)
|
||||
lint/ Static analysis rules
|
||||
format/ Canonical formatter
|
||||
lsp/ Language Server Protocol server
|
||||
asm/ Standalone assembler, encoder, object emitters
|
||||
verify/ JIT execution, differential testing, ABI checks
|
||||
debug/ Interactive ptrace debugger (linux/amd64)
|
||||
_gen/ Instruction table generator
|
||||
testdata/ Test fixtures
|
||||
docs/ Architecture, development, CLI reference
|
||||
```
|
||||
+88
-13
@@ -27,14 +27,6 @@ func Source(path, src string) string {
|
||||
mnemLen int
|
||||
funcID int
|
||||
}
|
||||
const (
|
||||
kBlank = iota
|
||||
kComment
|
||||
kPreproc
|
||||
kDirective
|
||||
kLabel
|
||||
kInstr
|
||||
)
|
||||
|
||||
infos := make([]info, len(lines))
|
||||
funcID := -1
|
||||
@@ -70,8 +62,8 @@ func Source(path, src string) string {
|
||||
infos[i] = inf
|
||||
}
|
||||
|
||||
// Second pass: render.
|
||||
var b strings.Builder
|
||||
// Second pass: render each line.
|
||||
outs := make([]outLine, 0, len(lines))
|
||||
inBody := false
|
||||
for i, line := range lines {
|
||||
inf := infos[i]
|
||||
@@ -99,11 +91,94 @@ func Source(path, src string) string {
|
||||
}
|
||||
case kInstr:
|
||||
out = renderInstr(line, maxWidth[inf.funcID])
|
||||
// A RET ends the body for indentation purposes: comments that
|
||||
// follow it — typically the next function's doc comment — belong
|
||||
// at column 0, not inside the finished function.
|
||||
if strings.EqualFold(line[0].Text, "RET") {
|
||||
inBody = false
|
||||
}
|
||||
}
|
||||
b.WriteString(strings.TrimRight(out, " \t"))
|
||||
b.WriteByte('\n')
|
||||
outs = append(outs, outLine{kind: inf.kind, text: strings.TrimRight(out, " \t")})
|
||||
}
|
||||
return b.String()
|
||||
return normalizeSpacing(outs)
|
||||
}
|
||||
|
||||
// Line classification, shared by the formatting passes.
|
||||
const (
|
||||
kBlank = iota
|
||||
kComment
|
||||
kPreproc
|
||||
kDirective
|
||||
kLabel
|
||||
kInstr
|
||||
)
|
||||
|
||||
// outLine is one rendered line together with its classification.
|
||||
type outLine struct {
|
||||
kind int
|
||||
text string
|
||||
}
|
||||
|
||||
// normalizeSpacing enforces the canonical blank-line layout: runs of blank
|
||||
// lines collapse to one, and a new block — a label, or a TEXT or GLOBL
|
||||
// directive — is preceded by exactly one blank line. Comments immediately
|
||||
// above a block belong to it, so the blank line is inserted before them. No
|
||||
// blank line is forced at the top of the file, right after a TEXT (the
|
||||
// function's first label), or between stacked labels that share an address.
|
||||
func normalizeSpacing(outs []outLine) string {
|
||||
blockStart := func(ol outLine) bool {
|
||||
switch ol.kind {
|
||||
case kLabel:
|
||||
return true
|
||||
case kDirective:
|
||||
// TEXT and GLOBL open a block; DATA continues a GLOBL block.
|
||||
return strings.HasPrefix(ol.text, "TEXT") || strings.HasPrefix(ol.text, "GLOBL")
|
||||
}
|
||||
return false
|
||||
}
|
||||
insert := make([]bool, len(outs))
|
||||
for i, ol := range outs {
|
||||
if !blockStart(ol) {
|
||||
continue
|
||||
}
|
||||
j := i
|
||||
for j > 0 && outs[j-1].kind == kComment {
|
||||
j--
|
||||
}
|
||||
if j == 0 {
|
||||
continue // top of file
|
||||
}
|
||||
switch prev := outs[j-1]; {
|
||||
case prev.kind == kBlank, prev.kind == kLabel:
|
||||
continue // already separated, or stacked labels
|
||||
case prev.kind == kDirective && strings.HasPrefix(prev.text, "TEXT"):
|
||||
continue // the function's first label
|
||||
}
|
||||
insert[j] = true
|
||||
}
|
||||
|
||||
var b strings.Builder
|
||||
prevBlank := true // also suppresses leading blanks
|
||||
for i, ol := range outs {
|
||||
if insert[i] && !prevBlank {
|
||||
b.WriteByte('\n')
|
||||
}
|
||||
if ol.kind == kBlank {
|
||||
if !prevBlank {
|
||||
b.WriteByte('\n')
|
||||
}
|
||||
prevBlank = true
|
||||
continue
|
||||
}
|
||||
b.WriteString(ol.text)
|
||||
b.WriteByte('\n')
|
||||
prevBlank = false
|
||||
}
|
||||
out := strings.TrimRight(b.String(), "\n")
|
||||
if out == "" {
|
||||
return ""
|
||||
}
|
||||
return out + "\n"
|
||||
}
|
||||
|
||||
// renderInstr renders an instruction line: a tab, the mnemonic padded to the
|
||||
|
||||
@@ -39,6 +39,102 @@ func TestGolden(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestDocCommentIndent checks that a doc comment preceding a TEXT directive
|
||||
// sits at column 0 even when another function (ending in RET) precedes it —
|
||||
// the RET must terminate the previous body for indentation purposes.
|
||||
func TestDocCommentIndent(t *testing.T) {
|
||||
in := "#include \"textflag.h\"\n" +
|
||||
"\n" +
|
||||
"// func first()\n" +
|
||||
"TEXT ·first(SB), NOSPLIT, $0\n" +
|
||||
"XORQ AX, AX\n" +
|
||||
"RET\n" +
|
||||
"\n" +
|
||||
"// func second()\n" +
|
||||
"TEXT ·second(SB), NOSPLIT, $0\n" +
|
||||
"RET\n"
|
||||
|
||||
want := "#include \"textflag.h\"\n" +
|
||||
"\n" +
|
||||
"// func first()\n" +
|
||||
"TEXT ·first(SB), NOSPLIT, $0\n" +
|
||||
"\tXORQ AX, AX\n" +
|
||||
"\tRET\n" +
|
||||
"\n" +
|
||||
"// func second()\n" +
|
||||
"TEXT ·second(SB), NOSPLIT, $0\n" +
|
||||
"\tRET\n"
|
||||
|
||||
got := Source("d_amd64.s", in)
|
||||
if got != want {
|
||||
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
||||
}
|
||||
// Body comments stay indented.
|
||||
body := "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n// inside the body\nXORQ AX, AX\nRET\n"
|
||||
gotBody := Source("b_amd64.s", body)
|
||||
if !strings.Contains(gotBody, "\t// inside the body\n") {
|
||||
t.Fatalf("body comment must stay indented:\n%q", gotBody)
|
||||
}
|
||||
}
|
||||
|
||||
// TestBlankLines checks the blank-line canonicalisation: exactly one blank
|
||||
// line before a new block (a label, or TEXT/GLOBL), runs of blanks collapsed
|
||||
// to one, and no blank forced after TEXT, between stacked labels, or at the
|
||||
// top of the file. Leading comments belong to the block they precede.
|
||||
func TestBlankLines(t *testing.T) {
|
||||
in := "#include \"textflag.h\"\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||
"first:\n" + // first label: no blank after TEXT
|
||||
"XORQ AX, AX\n" +
|
||||
"JMP next\n" + // unlabeled glue: fmt inserts a blank before next:
|
||||
"next:\n" +
|
||||
"stacked:\n" + // stacked labels share an address: no blank between
|
||||
"INCQ AX\n" +
|
||||
"\n" +
|
||||
"\n" + // two blanks collapse to one
|
||||
"// separated block\n" + // comment belongs to the label below
|
||||
"later:\n" +
|
||||
"RET\n" +
|
||||
"// func g()\n" + // doc comment: blank goes before it
|
||||
"TEXT ·g(SB), NOSPLIT, $0\n" +
|
||||
"RET\n" +
|
||||
"GLOBL ·mask(SB), RODATA, $8\n" + // blank before GLOBL…
|
||||
"DATA ·mask+0(SB)/4, $1\n" + // …but not before DATA
|
||||
"\n" +
|
||||
"\n" +
|
||||
"\n" // trailing blanks dropped
|
||||
|
||||
want := "#include \"textflag.h\"\n" +
|
||||
"\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||
"first:\n" +
|
||||
"\tXORQ AX, AX\n" +
|
||||
"\tJMP next\n" +
|
||||
"\n" +
|
||||
"next:\n" +
|
||||
"stacked:\n" +
|
||||
"\tINCQ AX\n" +
|
||||
"\n" +
|
||||
"\t// separated block\n" + // body comment before a label stays indented
|
||||
"later:\n" +
|
||||
"\tRET\n" +
|
||||
"\n" +
|
||||
"// func g()\n" +
|
||||
"TEXT ·g(SB), NOSPLIT, $0\n" +
|
||||
"\tRET\n" +
|
||||
"\n" +
|
||||
"GLOBL ·mask(SB), RODATA, $8\n" +
|
||||
"DATA ·mask+0(SB)/4, $1\n"
|
||||
|
||||
got := Source("b_amd64.s", in)
|
||||
if got != want {
|
||||
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
||||
}
|
||||
if again := Source("b_amd64.s", got); again != got {
|
||||
t.Fatalf("not idempotent:\n%q", again)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOperandSpacing(t *testing.T) {
|
||||
cases := map[string]string{
|
||||
"4(SI)": "4(SI)",
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
|
||||
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
|
||||
|
||||
version := "0.2.0"
|
||||
version := "0.28.0"
|
||||
|
||||
default:
|
||||
@just --list
|
||||
|
||||
+61
-7
@@ -242,7 +242,7 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
|
||||
}
|
||||
}
|
||||
|
||||
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) {
|
||||
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) && !maskedEvex(mnem, st.Operands) {
|
||||
if in, ok := tab.Lookup(mnem); ok && in.MinOps >= 0 {
|
||||
n := len(st.Operands)
|
||||
if n < in.MinOps || n > in.MaxOps {
|
||||
@@ -319,18 +319,27 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
|
||||
}
|
||||
}
|
||||
|
||||
// Register liveness: a callee-saved register that is written but never
|
||||
// saved and restored is clobbered across the call. The check runs over the
|
||||
// control-flow graph and is skipped for macro-using files, where an opaque
|
||||
// macro may perform the save/restore.
|
||||
// Register liveness: a register the Go ABI fixes across calls that is
|
||||
// written but never saved and restored is clobbered. The check runs over
|
||||
// the control-flow graph and is skipped for macro-using files, where an
|
||||
// opaque macro may perform the save/restore.
|
||||
if doLabelChecks && archKnown && !cfg.Disable[CodeRegisterClobber] {
|
||||
live := analyzeLiveness(t, cfg.Arch)
|
||||
if clobbered := clobberedCalleeSaved(live, cfg.Arch); len(clobbered) > 0 {
|
||||
always, rt := clobberedGoFixed(live, cfg.Arch, reachesRuntime(t))
|
||||
if len(always) > 0 {
|
||||
out = append(out, Diagnostic{
|
||||
Pos: t.Keyword.Pos,
|
||||
Severity: Warning,
|
||||
Code: CodeRegisterClobber,
|
||||
Message: fmt.Sprintf("callee-saved register(s) %s written but never saved/restored", strings.Join(clobbered, ", ")),
|
||||
Message: fmt.Sprintf("register(s) %s written but never saved/restored: fixed by the Go ABI (frame/goroutine pointer)", strings.Join(always, ", ")),
|
||||
})
|
||||
}
|
||||
if len(rt) > 0 {
|
||||
out = append(out, Diagnostic{
|
||||
Pos: t.Keyword.Pos,
|
||||
Severity: Warning,
|
||||
Code: CodeRegisterClobber,
|
||||
Message: fmt.Sprintf("goroutine-pointer register(s) %s written but never saved/restored in a function that can reach the Go runtime", strings.Join(rt, ", ")),
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -341,6 +350,29 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
|
||||
return out
|
||||
}
|
||||
|
||||
// reachesRuntime reports whether a function can reach the Go runtime: it is
|
||||
// not NOSPLIT (so the stack-split and traceback machinery runs) or it makes a
|
||||
// CALL. Goroutine-pointer registers must survive such functions; a NOSPLIT
|
||||
// leaf may clobber them, since the ABI0 transition restores them (the
|
||||
// runtime's own assembly relies on this, e.g. R14 on amd64).
|
||||
func reachesRuntime(t *ast.Text) bool {
|
||||
nosplit := false
|
||||
for _, f := range t.Flags {
|
||||
if strings.EqualFold(f, "NOSPLIT") {
|
||||
nosplit = true
|
||||
}
|
||||
}
|
||||
for _, s := range t.Body {
|
||||
if in, ok := s.(*ast.Instr); ok {
|
||||
switch strings.ToUpper(in.Mnemonic.Text) {
|
||||
case "CALL", "BL", "JAL": // amd64, arm64/loong64, riscv64 calls
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return !nosplit
|
||||
}
|
||||
|
||||
// usesFPArgs reports whether a function references its arguments through the FP
|
||||
// pseudo-register — i.e. it uses the stack-based ABI0 layout, where the
|
||||
// declared argument size must match the signature.
|
||||
@@ -406,6 +438,28 @@ func isMacroInvocation(mnem string, macros map[string]bool) bool {
|
||||
return strings.Contains(mnem, "_") || macros[mnem]
|
||||
}
|
||||
|
||||
// maskedEvex reports whether the instruction is a masked EVEX form: the
|
||||
// mnemonic carries a .Z suffix, or the operand list contains an opmask
|
||||
// register (K1–K7). Either way the operand count differs from the unmasked
|
||||
// form, so count checks are skipped.
|
||||
func maskedEvex(mnem string, ops []*ast.Operand) bool {
|
||||
if strings.Contains(mnem, ".") {
|
||||
return true
|
||||
}
|
||||
for _, op := range ops {
|
||||
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Base == "" &&
|
||||
op.Addr.Index == "" && op.Addr.Sym.Pseudo == "" && isMaskReg(op.Addr.Sym.Name) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// isMaskReg reports whether name is an opmask register K0–K7.
|
||||
func isMaskReg(name string) bool {
|
||||
return len(name) == 2 && name[0] == 'K' && name[1] >= '0' && name[1] <= '7'
|
||||
}
|
||||
|
||||
// isConditionalDirective reports whether a preprocessor directive (the text
|
||||
// after '#') is a conditional-compilation directive whose branches the parser
|
||||
// cannot resolve.
|
||||
|
||||
+25
-5
@@ -48,11 +48,11 @@ func TestFixtureIsClean(t *testing.T) {
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
// The fixture mirrors the go-flac kernels, which use callee-saved registers
|
||||
// (BX, R13) without saving them; the register-clobber audit flags that by
|
||||
// design. This test targets the other rules, so the audit is disabled here
|
||||
// (it is covered by TestRegisterClobber).
|
||||
diags := File(f, Config{Arch: arch.AMD64, Disable: map[string]bool{CodeRegisterClobber: true}})
|
||||
// The fixture mirrors the go-flac kernels, which write the Go ABI0
|
||||
// scratch registers (BX, R13) without saving them — legal under Go's
|
||||
// stack-based ABI, so the register-clobber audit stays silent and the
|
||||
// fixture must lint entirely clean.
|
||||
diags := File(f, Config{Arch: arch.AMD64})
|
||||
if len(diags) != 0 {
|
||||
t.Fatalf("expected no diagnostics on the fixture, got %+v", diags)
|
||||
}
|
||||
@@ -187,6 +187,26 @@ done:
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexMaskingRecognised checks that masked EVEX forms — the .Z suffix and
|
||||
// an explicit K operand — are recognised and exempt from operand-count
|
||||
// checks.
|
||||
func TestEvexMaskingRecognised(t *testing.T) {
|
||||
diags := lintSrc(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
VPADDD.Z Z1, Z2, K2, Z3
|
||||
VPMINSD Z1, Z2, K5, Z3
|
||||
VMOVDQU8 Z1, K3, (SI)
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeUnknownInstr] != 0 {
|
||||
t.Fatalf("masked EVEX must be recognised: %+v", diags)
|
||||
}
|
||||
if codes(diags)[CodeOperandCount] != 0 {
|
||||
t.Fatalf("masked operand counts must not be flagged: %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64AddressingSuffix(t *testing.T) {
|
||||
// .W (pre-index) and .P (post-index) suffixes must resolve to the base
|
||||
// instruction.
|
||||
|
||||
+54
-53
@@ -4,7 +4,6 @@
|
||||
package lint
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
@@ -248,7 +247,7 @@ func instrEffect(in *ast.Instr, a arch.Arch) regEffect {
|
||||
}
|
||||
|
||||
compare := isCompare(mnem)
|
||||
dstIdx := dstIndex(in, a)
|
||||
dstIdx := dstIndex(in)
|
||||
|
||||
for i, op := range in.Operands {
|
||||
r := gprName(op, a)
|
||||
@@ -281,13 +280,11 @@ func instrEffect(in *ast.Instr, a arch.Arch) regEffect {
|
||||
return eff
|
||||
}
|
||||
|
||||
// dstIndex returns the operand index of the destination register: last for the
|
||||
// Plan 9 (amd64) spelling, first for arm64/riscv64/loong64.
|
||||
func dstIndex(in *ast.Instr, a arch.Arch) int {
|
||||
if a == arch.AMD64 {
|
||||
return len(in.Operands) - 1
|
||||
}
|
||||
return 0
|
||||
// dstIndex returns the operand index of the destination register: in Plan 9
|
||||
// notation the destination is the last operand on every architecture Go
|
||||
// supports (amd64, arm64, riscv64 and loong64 alike).
|
||||
func dstIndex(in *ast.Instr) int {
|
||||
return len(in.Operands) - 1
|
||||
}
|
||||
|
||||
// isCompare reports whether the mnemonic only reads its operands (setting flags).
|
||||
@@ -372,41 +369,37 @@ func sameSet(a, b map[string]bool) bool {
|
||||
return true
|
||||
}
|
||||
|
||||
// calleeSavedGPRs returns the general-purpose registers an assembly function
|
||||
// must preserve for its caller, using the register names the assembler accepts
|
||||
// for each architecture.
|
||||
func calleeSavedGPRs(a arch.Arch) map[string]bool {
|
||||
// goFixedGPRs returns the general-purpose registers the Go ABI designates as
|
||||
// fixed across calls — the ones hand-written assembly must not permanently
|
||||
// clobber. This follows cmd/compile/abi-internal.md, not the platform ABI:
|
||||
// Go's stack-based ABI0 (which hand-written assembly uses) has no System V
|
||||
// style callee-saved registers, so clobbering the argument and scratch
|
||||
// registers (amd64 BX, R12, R13, R15, …) is legal.
|
||||
//
|
||||
// Two groups are returned. always holds registers whose loss is never safe.
|
||||
// runtime holds registers that survive an ABI0 leaf only because the
|
||||
// transition machinery restores them (on amd64 the g pointer is reloaded
|
||||
// from TLS): clobbering them is safe exactly in NOSPLIT functions that make
|
||||
// no calls, which is how the runtime's own assembly uses them.
|
||||
func goFixedGPRs(a arch.Arch) (always, runtime map[string]bool) {
|
||||
switch a {
|
||||
case arch.AMD64:
|
||||
return gprSet("BX", "BP", "R12", "R13", "R14", "R15")
|
||||
// BP maintains the frame chain; R14 holds the current goroutine.
|
||||
// R15 is scratch except in dynamically linked binaries, so it is not
|
||||
// flagged.
|
||||
return gprSet("BP"), gprSet("R14")
|
||||
case arch.ARM64:
|
||||
names := []string{"R29", "R30"} // FP, LR
|
||||
for i := 19; i <= 28; i++ {
|
||||
names = append(names, fmt.Sprintf("R%d", i))
|
||||
}
|
||||
return gprSet(names...)
|
||||
// R18 is reserved for the OS on some platforms, R28 holds the current
|
||||
// goroutine, R29 is the frame pointer.
|
||||
return gprSet("R18", "R28", "R29"), nil
|
||||
case arch.RISCV:
|
||||
// RA (X1) and the S registers (X8, X9, X18–X27) are callee-saved.
|
||||
names := []string{"X1", "RA", "X8", "X9", "S0", "S1", "FP"}
|
||||
for i := 18; i <= 27; i++ {
|
||||
names = append(names, fmt.Sprintf("X%d", i))
|
||||
}
|
||||
for i := 2; i <= 11; i++ {
|
||||
names = append(names, fmt.Sprintf("S%d", i))
|
||||
}
|
||||
return gprSet(names...)
|
||||
// X27 holds the current goroutine.
|
||||
return gprSet("X27"), nil
|
||||
case arch.LOONG64:
|
||||
// RA (R1), FP (R22) and S0–S8 (R23–R31) are callee-saved.
|
||||
names := []string{"R1", "RA", "R22", "FP"}
|
||||
for i := 23; i <= 31; i++ {
|
||||
names = append(names, fmt.Sprintf("R%d", i))
|
||||
}
|
||||
for i := 0; i <= 8; i++ {
|
||||
names = append(names, fmt.Sprintf("S%d", i))
|
||||
}
|
||||
return gprSet(names...)
|
||||
// R22 holds the current goroutine.
|
||||
return gprSet("R22"), nil
|
||||
}
|
||||
return nil
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
func gprSet(names ...string) map[string]bool {
|
||||
@@ -417,15 +410,16 @@ func gprSet(names ...string) map[string]bool {
|
||||
return m
|
||||
}
|
||||
|
||||
// clobberedCalleeSaved returns the callee-saved registers a function writes
|
||||
// without also saving and restoring them — i.e. registers whose caller-owned
|
||||
// value is lost across the call. It walks the blocks of the liveness analysis
|
||||
// (so the control-flow graph is what supplies the instruction set) and
|
||||
// aggregates each instruction's register effects.
|
||||
func clobberedCalleeSaved(l *liveness, a arch.Arch) []string {
|
||||
callee := calleeSavedGPRs(a)
|
||||
if len(callee) == 0 {
|
||||
return nil
|
||||
// clobberedGoFixed returns the Go-ABI-fixed registers a function writes
|
||||
// without also saving and restoring them. The first result lists registers
|
||||
// whose loss is never safe; the second lists the goroutine-pointer class,
|
||||
// whose loss is reported only when reachesRuntime is true (a non-NOSPLIT
|
||||
// function, or one that makes calls — the ABI0 transition machinery restores
|
||||
// the g pointer only on such paths).
|
||||
func clobberedGoFixed(l *liveness, a arch.Arch, reachesRuntime bool) (always, runtime []string) {
|
||||
alwaysSet, runtimeSet := goFixedGPRs(a)
|
||||
if len(alwaysSet) == 0 && len(runtimeSet) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
def := map[string]bool{}
|
||||
saved := map[string]bool{}
|
||||
@@ -444,12 +438,19 @@ func clobberedCalleeSaved(l *liveness, a arch.Arch) []string {
|
||||
}
|
||||
}
|
||||
}
|
||||
var out []string
|
||||
for r := range callee {
|
||||
if def[r] && !(saved[r] && restored[r]) {
|
||||
out = append(out, r)
|
||||
clobbered := func(set map[string]bool) []string {
|
||||
var out []string
|
||||
for r := range set {
|
||||
if def[r] && !(saved[r] && restored[r]) {
|
||||
out = append(out, r)
|
||||
}
|
||||
}
|
||||
sort.Strings(out)
|
||||
return out
|
||||
}
|
||||
sort.Strings(out)
|
||||
return out
|
||||
always = clobbered(alwaysSet)
|
||||
if reachesRuntime {
|
||||
runtime = clobbered(runtimeSet)
|
||||
}
|
||||
return always, runtime
|
||||
}
|
||||
|
||||
+112
-16
@@ -5,36 +5,132 @@ package lint
|
||||
|
||||
import "testing"
|
||||
|
||||
// TestRegisterClobber detects writes to callee-saved registers that are not
|
||||
// saved and restored.
|
||||
// TestRegisterClobber checks the register-clobber audit is calibrated to the
|
||||
// Go ABI (cmd/compile/abi-internal.md), not the platform ABI: Go's
|
||||
// stack-based ABI0 — which hand-written assembly uses — has no System V
|
||||
// style callee-saved registers, so argument and scratch registers may be
|
||||
// clobbered freely. Only the registers the ABI fixes across calls (the
|
||||
// frame pointer, the goroutine pointer, OS-reserved registers) are audited.
|
||||
func TestRegisterClobber(t *testing.T) {
|
||||
// BX (callee-saved on amd64) is written but never saved → clobbered.
|
||||
clob := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
// amd64: BX, R12, R13 and R15 are argument/permanent-scratch registers in
|
||||
// Go ABI0 — writing them unsaved is legal (a System V calibration would
|
||||
// report all of these).
|
||||
scratch := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tMOVQ CX, BX\n"+
|
||||
"\tXORL R12, R12\n"+
|
||||
"\tXORL R13, R13\n"+
|
||||
"\tXORL R15, R15\n"+
|
||||
"\tRET\n")
|
||||
if codes(clob)[CodeRegisterClobber] != 1 {
|
||||
t.Fatalf("unsaved callee-saved write should be flagged: %+v", clob)
|
||||
if codes(scratch)[CodeRegisterClobber] != 0 {
|
||||
t.Fatalf("Go ABI0 scratch registers must not be flagged: %+v", scratch)
|
||||
}
|
||||
|
||||
// Saved and restored → preserved.
|
||||
// amd64: R14 (the goroutine pointer) in a NOSPLIT function without calls
|
||||
// is the runtime's own pattern — the ABI0 transition restores it — so it
|
||||
// is not flagged.
|
||||
leaf := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tXORL R14, R14\n"+
|
||||
"\tRET\n")
|
||||
if codes(leaf)[CodeRegisterClobber] != 0 {
|
||||
t.Fatalf("R14 in a NOSPLIT leaf must not be flagged: %+v", leaf)
|
||||
}
|
||||
|
||||
// amd64: R14 in a function that makes a call is a genuine hazard.
|
||||
withCall := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tXORL R14, R14\n"+
|
||||
"\tCALL ·g(SB)\n"+
|
||||
"\tRET\n")
|
||||
if codes(withCall)[CodeRegisterClobber] != 1 {
|
||||
t.Fatalf("unsaved R14 with a call should be flagged: %+v", withCall)
|
||||
}
|
||||
|
||||
// amd64: R14 in a non-NOSPLIT function is a hazard regardless of calls.
|
||||
split := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), $0\n"+
|
||||
"\tMOVQ CX, R14\n"+
|
||||
"\tRET\n")
|
||||
if codes(split)[CodeRegisterClobber] != 1 {
|
||||
t.Fatalf("unsaved R14 in a non-NOSPLIT function should be flagged: %+v", split)
|
||||
}
|
||||
|
||||
// amd64: R14 saved and restored around the call is preserved.
|
||||
saved := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $8\n"+
|
||||
"\tPUSHQ BX\n"+
|
||||
"\tMOVQ CX, BX\n"+
|
||||
"\tPOPQ BX\n"+
|
||||
"\tPUSHQ R14\n"+
|
||||
"\tXORL R14, R14\n"+
|
||||
"\tCALL ·g(SB)\n"+
|
||||
"\tPOPQ R14\n"+
|
||||
"\tRET\n")
|
||||
if codes(saved)[CodeRegisterClobber] != 0 {
|
||||
t.Fatalf("saved/restored register must not be flagged: %+v", saved)
|
||||
t.Fatalf("saved/restored R14 must not be flagged: %+v", saved)
|
||||
}
|
||||
|
||||
// A caller-saved register (CX) is fine to write.
|
||||
caller := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
// amd64: BP maintains the frame chain and is always audited.
|
||||
bp := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tMOVQ $1, CX\n"+
|
||||
"\tMOVQ CX, BP\n"+
|
||||
"\tRET\n")
|
||||
if codes(caller)[CodeRegisterClobber] != 0 {
|
||||
t.Fatalf("caller-saved register must not be flagged: %+v", caller)
|
||||
if codes(bp)[CodeRegisterClobber] != 1 {
|
||||
t.Fatalf("unsaved BP write should be flagged: %+v", bp)
|
||||
}
|
||||
|
||||
// arm64: R20 is scratch; R28 (goroutine pointer) and R18 (OS-reserved)
|
||||
// are fixed by the Go ABI.
|
||||
armScratch := lintSrcArch(t, "t_arm64.s", "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tMOVD R0, R20\n"+
|
||||
"\tRET\n")
|
||||
if codes(armScratch)[CodeRegisterClobber] != 0 {
|
||||
t.Fatalf("arm64 scratch register must not be flagged: %+v", armScratch)
|
||||
}
|
||||
armG := lintSrcArch(t, "t_arm64.s", "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tMOVD R0, R28\n"+
|
||||
"\tRET\n")
|
||||
if codes(armG)[CodeRegisterClobber] != 1 {
|
||||
t.Fatalf("unsaved arm64 R28 write should be flagged: %+v", armG)
|
||||
}
|
||||
armReserved := lintSrcArch(t, "t_arm64.s", "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tMOVD R0, R18\n"+
|
||||
"\tRET\n")
|
||||
if codes(armReserved)[CodeRegisterClobber] != 1 {
|
||||
t.Fatalf("arm64 R18 write should be flagged: %+v", armReserved)
|
||||
}
|
||||
|
||||
// riscv64: X27 holds the goroutine; X5–X7 are scratch.
|
||||
riscScratch := lintSrcArch(t, "t_riscv64.s", "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tMOV X5, X6\n"+
|
||||
"\tRET\n")
|
||||
if codes(riscScratch)[CodeRegisterClobber] != 0 {
|
||||
t.Fatalf("riscv64 scratch register must not be flagged: %+v", riscScratch)
|
||||
}
|
||||
riscG := lintSrcArch(t, "t_riscv64.s", "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tMOV X5, X27\n"+
|
||||
"\tRET\n")
|
||||
if codes(riscG)[CodeRegisterClobber] != 1 {
|
||||
t.Fatalf("unsaved riscv64 X27 write should be flagged: %+v", riscG)
|
||||
}
|
||||
|
||||
// loong64: R22 holds the goroutine; R5–R19 are argument/scratch.
|
||||
loongScratch := lintSrcArch(t, "t_loong64.s", "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tMOVV R5, R6\n"+
|
||||
"\tRET\n")
|
||||
if codes(loongScratch)[CodeRegisterClobber] != 0 {
|
||||
t.Fatalf("loong64 scratch register must not be flagged: %+v", loongScratch)
|
||||
}
|
||||
loongG := lintSrcArch(t, "t_loong64.s", "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tMOVV R5, R22\n"+
|
||||
"\tRET\n")
|
||||
if codes(loongG)[CodeRegisterClobber] != 1 {
|
||||
t.Fatalf("unsaved loong64 R22 write should be flagged: %+v", loongG)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+33
-16
@@ -375,6 +375,11 @@ func parseImmediate(g []token.Token) ast.Immediate {
|
||||
if v, ok := tryInt(text); ok {
|
||||
imm.Val = v
|
||||
imm.HasVal = true
|
||||
} else if u, err := strconv.ParseUint(text, 0, 64); err == nil && !imm.Neg {
|
||||
// Unsigned 64-bit literals (DATA mask<>+8(SB)/8, $0x8000…)
|
||||
// overflow int64; keep the bit pattern.
|
||||
imm.Val = int64(u)
|
||||
imm.HasVal = true
|
||||
} else {
|
||||
imm.Float = text
|
||||
}
|
||||
@@ -393,29 +398,41 @@ func parseAddress(g []token.Token) ast.Address {
|
||||
return addr
|
||||
}
|
||||
// Symbol-with-pseudo form: name[<>][+off](PSEUDO).
|
||||
// When the prefix is not a valid symbol name (e.g. a bare number like
|
||||
// 0(SP) in RISC-V), sym is nil and we fall through to regular memory
|
||||
// operand parsing instead of returning an empty address.
|
||||
if idx := findPseudoParen(g); idx >= 0 {
|
||||
sym, _ := parseSymbolPrefix(g[:idx+3])
|
||||
addr.Sym = sym
|
||||
return addr
|
||||
if sym != nil {
|
||||
addr.Sym = sym
|
||||
return addr
|
||||
}
|
||||
}
|
||||
|
||||
i := 0
|
||||
// Optional leading displacement before a '(' base group.
|
||||
if isSignedNumber(g, i) && i+1 < len(g) && g[i+1].Kind == token.LParen {
|
||||
neg := false
|
||||
if g[i].Kind == token.Minus {
|
||||
neg = true
|
||||
i++
|
||||
} else if g[i].Kind == token.Plus {
|
||||
i++
|
||||
// Optional leading displacement before a '(' base group. A sign pushes
|
||||
// the parenthesis one token further out: -4(DX) has it at i+2.
|
||||
if isSignedNumber(g, i) {
|
||||
paren := i + 1
|
||||
if g[i].Kind == token.Minus || g[i].Kind == token.Plus {
|
||||
paren = i + 2
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.Number {
|
||||
addr.Offset = parseInt(g[i].Text)
|
||||
addr.HasOff = true
|
||||
if neg {
|
||||
addr.Offset = -addr.Offset
|
||||
if paren < len(g) && g[paren].Kind == token.LParen {
|
||||
neg := false
|
||||
if g[i].Kind == token.Minus {
|
||||
neg = true
|
||||
i++
|
||||
} else if g[i].Kind == token.Plus {
|
||||
i++
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.Number {
|
||||
addr.Offset = parseInt(g[i].Text)
|
||||
addr.HasOff = true
|
||||
if neg {
|
||||
addr.Offset = -addr.Offset
|
||||
}
|
||||
i++
|
||||
}
|
||||
i++
|
||||
}
|
||||
}
|
||||
// First parenthesised group: the base register.
|
||||
|
||||
@@ -34,6 +34,46 @@ func texts(f *ast.File) []*ast.Text {
|
||||
return out
|
||||
}
|
||||
|
||||
// TestNegativeDisplacement is a regression test for a leading negative
|
||||
// displacement with a base and index: the sign pushed the parenthesis one
|
||||
// token further out than the lookahead expected, and the whole address used
|
||||
// to parse empty.
|
||||
func TestNegativeDisplacement(t *testing.T) {
|
||||
f, errs := Parse("neg_amd64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
LEAQ -4(DX)(R9*4), R9
|
||||
MOVQ +8(AX), BX
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
fn := texts(f)[0]
|
||||
var leaq, movq *ast.Instr
|
||||
for _, s := range fn.Body {
|
||||
if in, ok := s.(*ast.Instr); ok {
|
||||
switch in.Mnemonic.Text {
|
||||
case "LEAQ":
|
||||
leaq = in
|
||||
case "MOVQ":
|
||||
movq = in
|
||||
}
|
||||
}
|
||||
}
|
||||
if leaq == nil || movq == nil {
|
||||
t.Fatalf("instructions not parsed: leaq=%v movq=%v", leaq, movq)
|
||||
}
|
||||
a := leaq.Operands[0].Addr
|
||||
if a.Base != "DX" || a.Index != "R9" || a.Scale != 4 || a.Offset != -4 || !a.HasOff {
|
||||
t.Errorf("LEAQ addr = %+v, want -4(DX)(R9*4)", a)
|
||||
}
|
||||
b := movq.Operands[0].Addr
|
||||
if b.Base != "AX" || b.Offset != 8 || !b.HasOff {
|
||||
t.Errorf("MOVQ addr = %+v, want +8(AX)", b)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseSample(t *testing.T) {
|
||||
f := mustParse(t, "../testdata/sample_amd64.s")
|
||||
|
||||
|
||||
Vendored
+12
@@ -0,0 +1,12 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func add(a, b int64) int64
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOV a+0(FP), X10
|
||||
MOV b+8(FP), X11
|
||||
ADD X11, X10, X10
|
||||
MOV X10, ret+16(FP)
|
||||
RET
|
||||
Vendored
+20
@@ -0,0 +1,20 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func atomicAdd(ptr *int64, val int64) int64
|
||||
TEXT ·atomicAdd(SB), NOSPLIT, $0-24
|
||||
MOV a+0(FP), X10
|
||||
MOV b+8(FP), X11
|
||||
AMOADDD X11, (X10), X12
|
||||
MOV X12, ret+16(FP)
|
||||
RET
|
||||
|
||||
// func fpAdd(a, b float64) float64
|
||||
TEXT ·fpAdd(SB), NOSPLIT, $0-24
|
||||
FLD a+0(FP), F10
|
||||
FLD b+8(FP), F11
|
||||
FADDD F10, F11, F12
|
||||
FSD F12, ret+16(FP)
|
||||
RET
|
||||
Vendored
+26
@@ -0,0 +1,26 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func readCSR(csr int64) int64
|
||||
TEXT ·readCSR(SB), NOSPLIT, $0-16
|
||||
MOV a+0(FP), X10
|
||||
CSRRS $0x300, X0, X11
|
||||
MOV X11, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func setCSRBit(csr, bit int64) int64
|
||||
TEXT ·setCSRBit(SB), NOSPLIT, $0-24
|
||||
MOV a+0(FP), X10
|
||||
MOV b+8(FP), X11
|
||||
CSRRS $0x304, X11, X12
|
||||
MOV X12, ret+16(FP)
|
||||
RET
|
||||
|
||||
// func writeCSR(val int64) int64
|
||||
TEXT ·writeCSR(SB), NOSPLIT, $0-16
|
||||
MOV a+0(FP), X10
|
||||
CSRRW $0x305, X10, X11
|
||||
MOV X11, ret+8(FP)
|
||||
RET
|
||||
Vendored
+22
@@ -0,0 +1,22 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func fma(a, b, c float64) float64
|
||||
TEXT ·fma(SB), NOSPLIT, $0-32
|
||||
FLD a+0(FP), F10
|
||||
FLD b+8(FP), F11
|
||||
FLD c+16(FP), F12
|
||||
FMADDD F10, F11, F12, F13
|
||||
FSD F13, ret+24(FP)
|
||||
RET
|
||||
|
||||
// func fms(a, b, c float64) float64
|
||||
TEXT ·fms(SB), NOSPLIT, $0-32
|
||||
FLD a+0(FP), F10
|
||||
FLD b+8(FP), F11
|
||||
FLD c+16(FP), F12
|
||||
FMSUBD F10, F11, F12, F13
|
||||
FSD F13, ret+24(FP)
|
||||
RET
|
||||
Vendored
+36
@@ -0,0 +1,36 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func casLoop(ptr *int64, old, new int64) bool
|
||||
TEXT ·casLoop(SB), NOSPLIT, $0-32
|
||||
cas_retry:
|
||||
MOV a+0(FP), X10
|
||||
LRD (X10), X11
|
||||
MOV b+8(FP), X12
|
||||
BNE X11, X12, cas_fail
|
||||
MOV c+16(FP), X13
|
||||
SCD X13, (X10), X14
|
||||
BNE X14, X0, cas_retry
|
||||
ADDI X0, $1, X15
|
||||
MOV X15, ret+24(FP)
|
||||
RET
|
||||
cas_fail:
|
||||
MOV X0, ret+24(FP)
|
||||
RET
|
||||
|
||||
// func intToFloat(x int64) float64
|
||||
TEXT ·intToFloat(SB), NOSPLIT, $0-16
|
||||
MOV a+0(FP), X10
|
||||
FCVTDL X10, F10
|
||||
FSD F10, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func compare(a, b float64) bool
|
||||
TEXT ·compare(SB), NOSPLIT, $0-24
|
||||
FLD a+0(FP), F10
|
||||
FLD b+8(FP), F11
|
||||
FLTD F10, F11, X10
|
||||
MOV X10, ret+16(FP)
|
||||
RET
|
||||
Vendored
+28
@@ -0,0 +1,28 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func cleanAdd(a, b int64) int64
|
||||
// A well-behaved function that preserves all callee-saved registers.
|
||||
TEXT ·cleanAdd(SB), NOSPLIT, $0-24
|
||||
MOVQ a+0(FP), AX
|
||||
ADDQ b+8(FP), AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
|
||||
// func dirtyBP(a int64) int64
|
||||
// Deliberately clobbers BP (an ABI violation for a NOSPLIT frame=0 function).
|
||||
TEXT ·dirtyBP(SB), NOSPLIT, $0-16
|
||||
MOVQ $0x1234, BP
|
||||
MOVQ a+0(FP), AX
|
||||
MOVQ AX, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func dirtyR14(a int64) int64
|
||||
// Deliberately clobbers R14 (the goroutine pointer — a serious ABI violation).
|
||||
TEXT ·dirtyR14(SB), NOSPLIT, $0-16
|
||||
MOVQ $0x5678, R14
|
||||
MOVQ a+0(FP), AX
|
||||
MOVQ AX, ret+8(FP)
|
||||
RET
|
||||
Vendored
+67
@@ -0,0 +1,67 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func add(a, b int64) int64
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVQ a+0(FP), AX
|
||||
ADDQ b+8(FP), AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
|
||||
// func sum(data []int64) int64
|
||||
// Sums all elements of the slice.
|
||||
TEXT ·sum(SB), NOSPLIT, $0-32
|
||||
MOVQ data_base+0(FP), SI
|
||||
MOVQ data_len+8(FP), CX
|
||||
XORQ AX, AX
|
||||
TESTQ CX, CX
|
||||
JZ sum_done
|
||||
|
||||
sum_loop:
|
||||
ADDQ (SI), AX
|
||||
ADDQ $8, SI
|
||||
DECQ CX
|
||||
JNZ sum_loop
|
||||
|
||||
sum_done:
|
||||
MOVQ AX, ret+24(FP)
|
||||
RET
|
||||
|
||||
// func wideCopy(dst, src []byte)
|
||||
// Non-overlapping copy of min(len(dst), len(src)) bytes using 32-byte moves.
|
||||
TEXT ·wideCopy(SB), NOSPLIT, $0-48
|
||||
MOVQ dst_base+0(FP), DI
|
||||
MOVQ dst_len+8(FP), BX
|
||||
MOVQ src_base+24(FP), SI
|
||||
MOVQ src_len+32(FP), R8
|
||||
CMPQ BX, R8
|
||||
JLE wc_have_n
|
||||
MOVQ R8, BX
|
||||
|
||||
wc_have_n:
|
||||
CMPQ BX, $32
|
||||
JB wc_small
|
||||
|
||||
VMOVDQU (SI), Y0
|
||||
VMOVDQU Y0, (DI)
|
||||
VMOVDQU -32(SI)(BX*1), Y0
|
||||
VMOVDQU Y0, -32(DI)(BX*1)
|
||||
VZEROUPPER
|
||||
RET
|
||||
|
||||
wc_small:
|
||||
TESTQ BX, BX
|
||||
JZ wc_done
|
||||
|
||||
wc_byte:
|
||||
MOVB (SI), R8B
|
||||
MOVB R8B, (DI)
|
||||
INCQ SI
|
||||
INCQ DI
|
||||
DECQ BX
|
||||
JNZ wc_byte
|
||||
|
||||
wc_done:
|
||||
RET
|
||||
@@ -0,0 +1,130 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build amd64
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// abiResult records register-clobber violations detected by the ABI-checking
|
||||
// trampoline. Bit 0: BP clobbered. Bit 1: R14 clobbered.
|
||||
var abiResult uint64
|
||||
|
||||
// savedBP holds the caller's frame pointer across the ABI-checked JIT call.
|
||||
// Referenced by enterJITChecked to satisfy go vet's save-before-clobber rule.
|
||||
var savedBP uintptr
|
||||
|
||||
// leaveCheckedPtr is initialised by the linker from the GLOBL/DATA in
|
||||
// abi_amd64.s: it holds the raw address of leaveJITCheckedRaw (which has
|
||||
// no ABIInternal wrapper, so the JIT function RETs directly into it).
|
||||
var leaveCheckedPtr uintptr
|
||||
|
||||
// enterJITChecked sets sentinels in BP and R14, switches to the prepared
|
||||
// stack and jumps to fn.
|
||||
//
|
||||
//go:nosplit
|
||||
func enterJITChecked(fn uintptr, stack uintptr)
|
||||
|
||||
// leaveJITCheckedRaw is the raw return trampoline for ABI checks. Its
|
||||
// address is obtained from the GLOBL in abi_amd64.s (leaveCheckedPtr),
|
||||
// which points to the .abi0 code — NOT the ABIInternal wrapper that this
|
||||
// declaration would generate. The declaration exists solely to satisfy
|
||||
// go vet's "missing Go declaration" check.
|
||||
//
|
||||
//go:nosplit
|
||||
func leaveJITCheckedRaw()
|
||||
|
||||
// ABIReport describes the result of an ABI-checking call.
|
||||
type ABIReport struct {
|
||||
BPClobbered bool // BP was modified by the function
|
||||
R14Clobbered bool // R14 (goroutine pointer) was modified
|
||||
RedZoneHit bool // the 128-byte red zone below SP was written
|
||||
}
|
||||
|
||||
// OK returns true when no violations were detected.
|
||||
func (r ABIReport) OK() bool {
|
||||
return !r.BPClobbered && !r.R14Clobbered && !r.RedZoneHit
|
||||
}
|
||||
|
||||
// String returns a human-readable summary.
|
||||
func (r ABIReport) String() string {
|
||||
if r.OK() {
|
||||
return "ABI clean"
|
||||
}
|
||||
s := "ABI violation:"
|
||||
if r.BPClobbered {
|
||||
s += " BP clobbered"
|
||||
}
|
||||
if r.R14Clobbered {
|
||||
s += " R14 clobbered"
|
||||
}
|
||||
if r.RedZoneHit {
|
||||
s += " red-zone written"
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// redZoneSize is the System V AMD64 red zone: 128 bytes below SP that a
|
||||
// leaf function may use without adjusting SP. Go does not use the red zone,
|
||||
// so any write there is a bug.
|
||||
const redZoneSize = 128
|
||||
|
||||
// redZoneFill is the byte pattern used to detect red-zone writes.
|
||||
const redZoneFill = 0xA5
|
||||
|
||||
// CallChecked invokes the function with ABI sentinels and a red-zone
|
||||
// canary, returning both the argument block (with results) and an ABIReport.
|
||||
func CallChecked(fnAddr uintptr, args []byte) ([]byte, ABIReport, error) {
|
||||
report := ABIReport{}
|
||||
|
||||
// Reset the global result.
|
||||
abiResult = 0
|
||||
|
||||
// Prepare the stack: [red-zone canary][padding][leaveJITCheckedRaw][args...]
|
||||
// The red zone sits below the initial SP, so the function would have to
|
||||
// write below SP to corrupt it.
|
||||
totalSize := redZoneSize + stackPad + 8 + len(args) + 64
|
||||
stackMem, err := syscall.Mmap(-1, 0, totalSize,
|
||||
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||
if err != nil {
|
||||
return nil, report, fmt.Errorf("verify: stack mmap: %w", err)
|
||||
}
|
||||
defer syscall.Munmap(stackMem)
|
||||
|
||||
// Fill the red zone with the canary pattern.
|
||||
for i := 0; i < redZoneSize; i++ {
|
||||
stackMem[i] = redZoneFill
|
||||
}
|
||||
|
||||
// Return address and args after the red zone and padding.
|
||||
retOff := redZoneSize + stackPad
|
||||
binary.LittleEndian.PutUint64(stackMem[retOff:retOff+8], uint64(leaveCheckedPtr))
|
||||
copy(stackMem[retOff+8:], args)
|
||||
|
||||
stackBase := uintptr(unsafe.Pointer(&stackMem[retOff]))
|
||||
enterJITChecked(fnAddr, stackBase)
|
||||
|
||||
// Read the register-clobber result.
|
||||
res := abiResult
|
||||
report.BPClobbered = res&1 != 0
|
||||
report.R14Clobbered = res&2 != 0
|
||||
|
||||
// Check the red zone.
|
||||
for i := 0; i < redZoneSize; i++ {
|
||||
if stackMem[i] != redZoneFill {
|
||||
report.RedZoneHit = true
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
// Copy out the argument area.
|
||||
out := make([]byte, len(args))
|
||||
copy(out, stackMem[retOff+8:retOff+8+len(args)])
|
||||
return out, report, nil
|
||||
}
|
||||
@@ -0,0 +1,61 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// ABI-checking trampoline. Sets sentinel values in the callee-saved
|
||||
// registers (BP, R14) before entering the JIT function and checks whether
|
||||
// they survived on return.
|
||||
//
|
||||
// The return trampoline (leaveJITCheckedRaw) is a raw TEXT symbol with no
|
||||
// Go function declaration, so the toolchain does NOT interpose an
|
||||
// ABIInternal wrapper — the JIT function RETs directly into the check code,
|
||||
// which sees the registers exactly as the function left them.
|
||||
//
|
||||
// Go ABI0 on amd64 guarantees:
|
||||
// - BP is callee-saved (NOSPLIT frame=0 functions must not touch it).
|
||||
// - R14 holds the goroutine pointer and must survive across any call.
|
||||
|
||||
// Sentinel values chosen to be unlikely in normal execution.
|
||||
#define SENTINEL_BP 0xDEADBEEFCAFEF00D
|
||||
#define SENTINEL_R14 0x0BADF00DDEADBEEF
|
||||
|
||||
// GLOBL holding the raw address of the leave trampoline, read by Go.
|
||||
GLOBL ·leaveCheckedPtr(SB), NOPTR, $8
|
||||
DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB)
|
||||
|
||||
// func enterJITChecked(fn uintptr, stack uintptr)
|
||||
// Sets sentinels in BP and R14, switches to the prepared stack and jumps
|
||||
// to fn. The prepared stack's return address must be leaveJITCheckedRaw
|
||||
// (read from leaveCheckedPtr).
|
||||
TEXT ·enterJITChecked(SB), NOSPLIT, $0-16
|
||||
MOVQ fn+0(FP), AX // target (before SP switch)
|
||||
MOVQ SP, ·savedSP(SB) // preserve Go stack
|
||||
MOVQ BP, ·savedBP(SB) // preserve frame pointer (vet requires save before clobber)
|
||||
MOVQ $SENTINEL_BP, BP // sentinel in BP
|
||||
MOVQ $SENTINEL_R14, R14 // sentinel in R14
|
||||
MOVQ stack+8(FP), SP // switch to prepared stack
|
||||
JMP AX
|
||||
|
||||
// leaveJITCheckedRaw is the raw return trampoline. It has NO Go function
|
||||
// declaration, so no ABIInternal wrapper is generated — the JIT function's
|
||||
// RET lands here directly, seeing BP and R14 exactly as the function left
|
||||
// them. It checks the sentinels, records violations in abiResult, then
|
||||
// restores the Go stack and returns.
|
||||
TEXT ·leaveJITCheckedRaw(SB), NOSPLIT, $0-0
|
||||
// Check BP against the sentinel.
|
||||
MOVQ $SENTINEL_BP, CX
|
||||
CMPQ BP, CX
|
||||
JEQ bp_ok
|
||||
ORQ $1, ·abiResult(SB)
|
||||
|
||||
bp_ok:
|
||||
// Check R14 against the sentinel.
|
||||
MOVQ $SENTINEL_R14, CX
|
||||
CMPQ R14, CX
|
||||
JEQ r14_ok
|
||||
ORQ $2, ·abiResult(SB)
|
||||
|
||||
r14_ok:
|
||||
MOVQ ·savedSP(SB), SP
|
||||
RET
|
||||
@@ -0,0 +1,26 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build !amd64
|
||||
|
||||
package verify
|
||||
|
||||
import "fmt"
|
||||
|
||||
// ABIReport describes the result of an ABI-checking call.
|
||||
type ABIReport struct {
|
||||
BPClobbered bool
|
||||
R14Clobbered bool
|
||||
RedZoneHit bool
|
||||
}
|
||||
|
||||
// OK returns true when no violations were detected.
|
||||
func (r ABIReport) OK() bool { return false }
|
||||
|
||||
// String returns a human-readable summary.
|
||||
func (r ABIReport) String() string { return "verify: ABI checks require amd64" }
|
||||
|
||||
// CallChecked is unavailable on non-amd64 architectures.
|
||||
func CallChecked(fnAddr uintptr, args []byte) ([]byte, ABIReport, error) {
|
||||
return nil, ABIReport{}, fmt.Errorf("verify: ABI checks require amd64")
|
||||
}
|
||||
@@ -0,0 +1,145 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
func loadABIKernel(t *testing.T) *Kernel {
|
||||
t.Helper()
|
||||
k, err := Load("../testdata/verify/abi_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("Load: %v", err)
|
||||
}
|
||||
t.Cleanup(k.Close)
|
||||
return k
|
||||
}
|
||||
|
||||
func TestABIClean(t *testing.T) {
|
||||
k := loadABIKernel(t)
|
||||
|
||||
args := make([]byte, 24)
|
||||
PutUint64(args, 0, 3)
|
||||
PutUint64(args, 8, 4)
|
||||
|
||||
out, report, err := k.CallFuncChecked("cleanAdd", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFuncChecked: %v", err)
|
||||
}
|
||||
if got := int64(GetUint64(out, 16)); got != 7 {
|
||||
t.Errorf("cleanAdd(3, 4) = %d, want 7", got)
|
||||
}
|
||||
if !report.OK() {
|
||||
t.Errorf("cleanAdd: %s", report)
|
||||
}
|
||||
}
|
||||
|
||||
func TestABIBPClobbered(t *testing.T) {
|
||||
k := loadABIKernel(t)
|
||||
|
||||
args := make([]byte, 16)
|
||||
PutUint64(args, 0, 42)
|
||||
|
||||
out, report, err := k.CallFuncChecked("dirtyBP", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFuncChecked: %v", err)
|
||||
}
|
||||
if got := int64(GetUint64(out, 8)); got != 42 {
|
||||
t.Errorf("dirtyBP(42) = %d, want 42", got)
|
||||
}
|
||||
if !report.BPClobbered {
|
||||
t.Error("dirtyBP: expected BP clobbered, but report says clean")
|
||||
}
|
||||
if report.R14Clobbered {
|
||||
t.Error("dirtyBP: R14 should not be clobbered")
|
||||
}
|
||||
}
|
||||
|
||||
func TestABIR14Clobbered(t *testing.T) {
|
||||
k := loadABIKernel(t)
|
||||
|
||||
args := make([]byte, 16)
|
||||
PutUint64(args, 0, 99)
|
||||
|
||||
out, report, err := k.CallFuncChecked("dirtyR14", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFuncChecked: %v", err)
|
||||
}
|
||||
if got := int64(GetUint64(out, 8)); got != 99 {
|
||||
t.Errorf("dirtyR14(99) = %d, want 99", got)
|
||||
}
|
||||
if !report.R14Clobbered {
|
||||
t.Error("dirtyR14: expected R14 clobbered, but report says clean")
|
||||
}
|
||||
if report.BPClobbered {
|
||||
t.Error("dirtyR14: BP should not be clobbered")
|
||||
}
|
||||
}
|
||||
|
||||
// TestABILZ4Kernels verifies that the production go-lz4 kernels are ABI-clean:
|
||||
// they preserve BP and R14 and do not write into the red zone.
|
||||
func TestABILZ4Kernels(t *testing.T) {
|
||||
k := loadLZ4Kernel(t)
|
||||
|
||||
// wideCopyAVX2 with a real copy.
|
||||
src := make([]byte, 128)
|
||||
for i := range src {
|
||||
src[i] = byte(i)
|
||||
}
|
||||
dst := make([]byte, 128)
|
||||
|
||||
args := make([]byte, 48)
|
||||
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
|
||||
PutUint64(args, 8, 128)
|
||||
PutUint64(args, 16, 128)
|
||||
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
||||
PutUint64(args, 32, 128)
|
||||
PutUint64(args, 40, 128)
|
||||
|
||||
_, report, err := k.CallFuncChecked("wideCopyAVX2", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFuncChecked(wideCopyAVX2): %v", err)
|
||||
}
|
||||
if !report.OK() {
|
||||
t.Errorf("wideCopyAVX2: %s", report)
|
||||
}
|
||||
|
||||
// decodeBlockAVX2 with a simple block.
|
||||
decSrc := []byte{0x50, 'H', 'e', 'l', 'l', 'o'}
|
||||
decDst := make([]byte, 64)
|
||||
|
||||
decArgs := make([]byte, 64)
|
||||
PutPtr(decArgs, 0, unsafe.Pointer(&decSrc[0]))
|
||||
PutUint64(decArgs, 8, uint64(len(decSrc)))
|
||||
PutUint64(decArgs, 16, uint64(cap(decSrc)))
|
||||
PutPtr(decArgs, 24, unsafe.Pointer(&decDst[0]))
|
||||
PutUint64(decArgs, 32, uint64(len(decDst)))
|
||||
PutUint64(decArgs, 40, uint64(cap(decDst)))
|
||||
|
||||
_, report, err = k.CallFuncChecked("decodeBlockAVX2", decArgs)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFuncChecked(decodeBlockAVX2): %v", err)
|
||||
}
|
||||
if !report.OK() {
|
||||
t.Errorf("decodeBlockAVX2: %s", report)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCallFuncCheckedErrors(t *testing.T) {
|
||||
k := loadABIKernel(t)
|
||||
|
||||
// Nonexistent function.
|
||||
_, _, err := k.CallFuncChecked("nope", make([]byte, 8))
|
||||
if err == nil {
|
||||
t.Fatal("expected error for nonexistent function")
|
||||
}
|
||||
|
||||
// Arg block too small.
|
||||
_, _, err = k.CallFuncChecked("cleanAdd", make([]byte, 8))
|
||||
if err == nil {
|
||||
t.Fatal("expected error for too-small arg block")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,144 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"math/rand"
|
||||
"os"
|
||||
"testing"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
const lz4AVX512Path = "../../go-libraries/go-lz4/avx512_amd64.s"
|
||||
|
||||
func loadLZ4AVX512Kernel(t *testing.T) *Kernel {
|
||||
t.Helper()
|
||||
if _, err := os.Stat(lz4AVX512Path); err != nil {
|
||||
t.Skipf("sibling kernel not available: %v", err)
|
||||
}
|
||||
k, err := Load(lz4AVX512Path)
|
||||
if err != nil {
|
||||
t.Fatalf("Load(%s): %v", lz4AVX512Path, err)
|
||||
}
|
||||
t.Cleanup(k.Close)
|
||||
return k
|
||||
}
|
||||
|
||||
func TestAVX512DecodeKnownAnswers(t *testing.T) {
|
||||
k := loadLZ4AVX512Kernel(t)
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
src []byte
|
||||
wantN int
|
||||
wantCode int
|
||||
}{
|
||||
{"literals_only", []byte{0x50, 'H', 'e', 'l', 'l', 'o'}, 5, 0},
|
||||
{"literals_and_match", []byte{0x54, 'A', 'A', 'A', 'A', 'A', 0x05, 0x00, 0x30, 'B', 'B', 'B'}, 16, 0},
|
||||
{"overlapping", []byte{0x14, 'X', 0x01, 0x00, 0x10, 'Y'}, 10, 0},
|
||||
{"malformed", []byte{0x50, 'H', 'e'}, 0, 1},
|
||||
{"zero_offset", []byte{0x14, 'X', 0x00, 0x00}, 0, 2},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
dst := make([]byte, 64)
|
||||
args := make([]byte, 64)
|
||||
PutPtr(args, 0, unsafe.Pointer(&tt.src[0]))
|
||||
PutUint64(args, 8, uint64(len(tt.src)))
|
||||
PutUint64(args, 16, uint64(cap(tt.src)))
|
||||
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
||||
PutUint64(args, 32, uint64(len(dst)))
|
||||
PutUint64(args, 40, uint64(cap(dst)))
|
||||
|
||||
out, err := k.CallFunc("decodeBlockAVX512", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFunc: %v", err)
|
||||
}
|
||||
n := int(GetUint64(out, 48))
|
||||
code := int(GetUint64(out, 56))
|
||||
if n != tt.wantN || code != tt.wantCode {
|
||||
t.Errorf("got (n=%d, code=%d), want (n=%d, code=%d)", n, code, tt.wantN, tt.wantCode)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestAVX512DifferentialFuzz(t *testing.T) {
|
||||
k := loadLZ4AVX512Kernel(t)
|
||||
rng := rand.New(rand.NewSource(77))
|
||||
|
||||
for i := 0; i < 3000; i++ {
|
||||
wantSize := 1 + rng.Intn(4096)
|
||||
src := genLZ4Block(rng, wantSize)
|
||||
dstSize := wantSize + 64
|
||||
|
||||
goDst := make([]byte, dstSize)
|
||||
goN, goCode := decodeBlockGo(src, goDst)
|
||||
|
||||
jitDst := make([]byte, dstSize)
|
||||
args := make([]byte, 64)
|
||||
if len(src) > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(len(src)))
|
||||
PutUint64(args, 16, uint64(cap(src)))
|
||||
if dstSize > 0 {
|
||||
PutPtr(args, 24, unsafe.Pointer(&jitDst[0]))
|
||||
}
|
||||
PutUint64(args, 32, uint64(dstSize))
|
||||
PutUint64(args, 40, uint64(cap(jitDst)))
|
||||
|
||||
out, err := k.CallFunc("decodeBlockAVX512", args)
|
||||
if err != nil {
|
||||
t.Fatalf("iter %d: %v", i, err)
|
||||
}
|
||||
jitN := int(GetUint64(out, 48))
|
||||
jitCode := int(GetUint64(out, 56))
|
||||
|
||||
if jitCode != goCode {
|
||||
t.Fatalf("iter %d: code mismatch: JIT=%d Go=%d", i, jitCode, goCode)
|
||||
}
|
||||
if jitCode != 0 {
|
||||
continue
|
||||
}
|
||||
if jitN != goN {
|
||||
t.Fatalf("iter %d: n mismatch: JIT=%d Go=%d", i, jitN, goN)
|
||||
}
|
||||
if !bytes.Equal(jitDst[:jitN], goDst[:goN]) {
|
||||
t.Fatalf("iter %d: output mismatch (n=%d)", i, jitN)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAVX512WideCopy(t *testing.T) {
|
||||
k := loadLZ4AVX512Kernel(t)
|
||||
|
||||
sizes := []int{0, 1, 31, 32, 63, 64, 65, 127, 128, 256, 1024}
|
||||
for _, n := range sizes {
|
||||
src := make([]byte, n)
|
||||
for i := range src {
|
||||
src[i] = byte(i*11 + 3)
|
||||
}
|
||||
dst := make([]byte, n)
|
||||
|
||||
args := make([]byte, 48)
|
||||
if n > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
|
||||
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(n))
|
||||
PutUint64(args, 16, uint64(n))
|
||||
PutUint64(args, 32, uint64(n))
|
||||
PutUint64(args, 40, uint64(n))
|
||||
|
||||
_, err := k.CallFunc("wideCopyAVX512", args)
|
||||
if err != nil {
|
||||
t.Fatalf("wideCopyAVX512(n=%d): %v", n, err)
|
||||
}
|
||||
if !bytes.Equal(dst, src) {
|
||||
t.Errorf("wideCopyAVX512(n=%d): mismatch", n)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build amd64
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
"reflect"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// savedSP holds the Go stack pointer while a JIT call is in flight.
|
||||
// Referenced by the assembly trampoline (trampoline_amd64.s).
|
||||
var savedSP uintptr
|
||||
|
||||
// enterJIT switches to the prepared stack and jumps to fn.
|
||||
// It does not return normally; the JIT function's RET transfers control
|
||||
// to leaveJIT, which restores the Go stack.
|
||||
//
|
||||
//go:nosplit
|
||||
func enterJIT(fn uintptr, stack uintptr)
|
||||
|
||||
// leaveJIT restores the Go stack after a JIT function returns.
|
||||
// Its address is placed as the return address on the prepared stack.
|
||||
//
|
||||
//go:nosplit
|
||||
func leaveJIT()
|
||||
|
||||
// leaveJITAddr is the machine address of leaveJIT, resolved once at init.
|
||||
var leaveJITAddr uintptr
|
||||
|
||||
func init() {
|
||||
leaveJITAddr = reflect.ValueOf(leaveJIT).Pointer()
|
||||
}
|
||||
|
||||
// stackPad is padding below the return address on the prepared stack.
|
||||
// The ABIInternal wrapper that leaveJIT's address resolves to executes
|
||||
// PUSHQ BP and CALL before reaching the raw assembly, writing up to 16
|
||||
// bytes below the return-address slot. 64 bytes of headroom is ample.
|
||||
const stackPad = 64
|
||||
|
||||
// Call invokes the assembled function at fnAddr with the given ABI0 argument
|
||||
// block (the raw bytes that would appear at FP+0). It returns the argument
|
||||
// block after the call, which contains any results the function wrote back
|
||||
// (the ABI0 convention shares the argument area for inputs and outputs).
|
||||
//
|
||||
// The function must be NOSPLIT (no stack growth) and must not reference
|
||||
// external symbols — the image is self-contained.
|
||||
func Call(fnAddr uintptr, args []byte) ([]byte, error) {
|
||||
// Prepare the stack: [padding][leaveJIT addr][args...]
|
||||
stackSize := stackPad + 8 + len(args) + 64 // padding + ret + args + safety
|
||||
stackMem, err := syscall.Mmap(-1, 0, stackSize,
|
||||
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("verify: stack mmap: %w", err)
|
||||
}
|
||||
defer syscall.Munmap(stackMem)
|
||||
|
||||
// The return address sits after the padding; the function's SP will
|
||||
// point here, leaving stackPad bytes below for the wrapper's pushes.
|
||||
retOff := stackPad
|
||||
binary.LittleEndian.PutUint64(stackMem[retOff:retOff+8], uint64(leaveJITAddr))
|
||||
// The ABI0 argument area follows the return address.
|
||||
copy(stackMem[retOff+8:], args)
|
||||
|
||||
stackBase := uintptr(unsafe.Pointer(&stackMem[retOff]))
|
||||
enterJIT(fnAddr, stackBase)
|
||||
|
||||
// Copy out the (possibly modified) argument area.
|
||||
out := make([]byte, len(args))
|
||||
copy(out, stackMem[retOff+8:retOff+8+len(args)])
|
||||
return out, nil
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build !amd64
|
||||
|
||||
package verify
|
||||
|
||||
import "fmt"
|
||||
|
||||
// Call is unavailable on non-amd64 architectures.
|
||||
func Call(fnAddr uintptr, args []byte) ([]byte, error) {
|
||||
return nil, fmt.Errorf("verify: JIT execution requires amd64")
|
||||
}
|
||||
@@ -0,0 +1,103 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
)
|
||||
|
||||
// Block describes one basic block within a function: a maximal sequence of
|
||||
// instructions with a single entry point (a label or the function start) and
|
||||
// a single exit (a jump, conditional jump or RET).
|
||||
type Block struct {
|
||||
Offset int // byte offset within the function
|
||||
Label string // label name ("" for the entry block)
|
||||
}
|
||||
|
||||
// Blocks identifies the basic blocks of a function from its local labels.
|
||||
// Each label is a potential jump target and therefore a block boundary; the
|
||||
// function entry (offset 0) is always a block. The blocks are returned in
|
||||
// ascending offset order.
|
||||
func (k *Kernel) Blocks(name string) ([]Block, error) {
|
||||
idx, ok := k.funcs[name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("verify: function %q not found", name)
|
||||
}
|
||||
fl := k.img.Funcs[idx]
|
||||
|
||||
blocks := []Block{{Offset: 0, Label: "(entry)"}}
|
||||
// Build a reverse map: offset → label name.
|
||||
offToLabel := make(map[int]string, len(fl.Labels))
|
||||
for label, off := range fl.Labels {
|
||||
if off > 0 && off < fl.Size {
|
||||
offToLabel[off] = label
|
||||
}
|
||||
}
|
||||
// Collect and sort offsets.
|
||||
offsets := make([]int, 0, len(offToLabel))
|
||||
for off := range offToLabel {
|
||||
offsets = append(offsets, off)
|
||||
}
|
||||
sort.Ints(offsets)
|
||||
for _, off := range offsets {
|
||||
blocks = append(blocks, Block{Offset: off, Label: offToLabel[off]})
|
||||
}
|
||||
return blocks, nil
|
||||
}
|
||||
|
||||
// BlockCount returns the number of identified basic blocks for the function.
|
||||
func (k *Kernel) BlockCount(name string) (int, error) {
|
||||
blocks, err := k.Blocks(name)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return len(blocks), nil
|
||||
}
|
||||
|
||||
// PathFingerprint is the observable output of one function execution: the
|
||||
// values written back into the result slots of the argument block. Two
|
||||
// executions that produce the same fingerprint took observationally
|
||||
// equivalent paths (though they may differ internally).
|
||||
type PathFingerprint struct {
|
||||
Results []uint64 // the result words from the arg block
|
||||
}
|
||||
|
||||
// ProfilePaths runs the function with each of the given argument blocks and
|
||||
// collects the distinct output fingerprints. This measures path diversity:
|
||||
// how many observationally different execution paths the input corpus
|
||||
// exercises. Combined with Blocks (the static block count), it gives a
|
||||
// lower bound on code coverage.
|
||||
func (k *Kernel) ProfilePaths(name string, argSets [][]byte, resultOffsets []int) ([]PathFingerprint, error) {
|
||||
idx, ok := k.funcs[name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("verify: function %q not found", name)
|
||||
}
|
||||
fl := k.img.Funcs[idx]
|
||||
|
||||
seen := map[string]bool{}
|
||||
var paths []PathFingerprint
|
||||
|
||||
for _, args := range argSets {
|
||||
if len(args) < fl.Args {
|
||||
return nil, fmt.Errorf("verify: %s: arg block too small", name)
|
||||
}
|
||||
out, err := k.CallFunc(name, args)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
fp := PathFingerprint{}
|
||||
key := ""
|
||||
for _, off := range resultOffsets {
|
||||
v := GetUint64(out, off)
|
||||
fp.Results = append(fp.Results, v)
|
||||
key += fmt.Sprintf("%016x", v)
|
||||
}
|
||||
if !seen[key] {
|
||||
seen[key] = true
|
||||
paths = append(paths, fp)
|
||||
}
|
||||
}
|
||||
return paths, nil
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
func TestBlocks(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
|
||||
// The "sum" function has labels: sum_done, sum_loop.
|
||||
blocks, err := k.Blocks("sum")
|
||||
if err != nil {
|
||||
t.Fatalf("Blocks(sum): %v", err)
|
||||
}
|
||||
if len(blocks) < 3 {
|
||||
t.Errorf("sum: expected at least 3 blocks (entry + 2 labels), got %d", len(blocks))
|
||||
}
|
||||
if blocks[0].Offset != 0 {
|
||||
t.Errorf("first block offset = %d, want 0", blocks[0].Offset)
|
||||
}
|
||||
t.Logf("sum blocks: %v", blocks)
|
||||
}
|
||||
|
||||
func TestBlockCount(t *testing.T) {
|
||||
k := loadLZ4Kernel(t)
|
||||
|
||||
n, err := k.BlockCount("decodeBlockAVX2")
|
||||
if err != nil {
|
||||
t.Fatalf("BlockCount: %v", err)
|
||||
}
|
||||
// The decoder has many labels (dec_loop, dec_malformed, etc.).
|
||||
if n < 10 {
|
||||
t.Errorf("decodeBlockAVX2: expected at least 10 blocks, got %d", n)
|
||||
}
|
||||
t.Logf("decodeBlockAVX2: %d basic blocks", n)
|
||||
}
|
||||
|
||||
func TestProfilePaths(t *testing.T) {
|
||||
k := loadLZ4Kernel(t)
|
||||
|
||||
// Build a corpus of varied LZ4 blocks.
|
||||
var argSets [][]byte
|
||||
blocks := []struct {
|
||||
src []byte
|
||||
dstSize int
|
||||
}{
|
||||
{[]byte{0x00}, 16}, // empty
|
||||
{[]byte{0x50, 'H', 'e', 'l', 'l', 'o'}, 16}, // literals only
|
||||
{[]byte{0x54, 'A', 'A', 'A', 'A', 'A', 5, 0, 0x30, 'B', 'B', 'B'}, 32}, // match
|
||||
{[]byte{0x14, 'X', 1, 0, 0x10, 'Y'}, 16}, // overlapping
|
||||
{[]byte{0x50, 'H'}, 16}, // malformed
|
||||
{[]byte{0x14, 'X', 0, 0}, 16}, // zero offset
|
||||
}
|
||||
for _, b := range blocks {
|
||||
args := make([]byte, 64)
|
||||
if len(b.src) > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&b.src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(len(b.src)))
|
||||
PutUint64(args, 16, uint64(cap(b.src)))
|
||||
dst := make([]byte, b.dstSize)
|
||||
if len(dst) > 0 {
|
||||
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
||||
}
|
||||
PutUint64(args, 32, uint64(len(dst)))
|
||||
PutUint64(args, 40, uint64(cap(dst)))
|
||||
argSets = append(argSets, args)
|
||||
}
|
||||
|
||||
// Result offsets: n+48 and code+56.
|
||||
paths, err := k.ProfilePaths("decodeBlockAVX2", argSets, []int{48, 56})
|
||||
if err != nil {
|
||||
t.Fatalf("ProfilePaths: %v", err)
|
||||
}
|
||||
|
||||
// We expect at least 3 distinct paths: success (various n), malformed, zero offset.
|
||||
if len(paths) < 3 {
|
||||
t.Errorf("expected at least 3 distinct paths, got %d", len(paths))
|
||||
}
|
||||
t.Logf("decodeBlockAVX2: %d distinct output paths from %d inputs", len(paths), len(argSets))
|
||||
}
|
||||
@@ -0,0 +1,295 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"math/rand"
|
||||
"testing"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// decodeBlockGo is a minimal portable LZ4 block decoder used as the
|
||||
// differential-testing oracle. It mirrors the contract of
|
||||
// go-lz4's decodeBlockGo: (bytesWritten, code) where code is
|
||||
// 0 = ok, 1 = malformed, 2 = zero offset.
|
||||
func decodeBlockGo(src, dst []byte) (int, int) {
|
||||
if len(src) == 0 {
|
||||
return 0, 1
|
||||
}
|
||||
si, di := 0, 0
|
||||
for {
|
||||
if si >= len(src) {
|
||||
return 0, 1 // truncated: no token
|
||||
}
|
||||
token := int(src[si])
|
||||
si++
|
||||
|
||||
// Literals.
|
||||
lLen := token >> 4
|
||||
if lLen == 15 {
|
||||
for {
|
||||
if si >= len(src) {
|
||||
return 0, 1
|
||||
}
|
||||
b := int(src[si])
|
||||
si++
|
||||
lLen += b
|
||||
if b != 255 {
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
if si+lLen > len(src) {
|
||||
return 0, 1 // truncated literals
|
||||
}
|
||||
if di+lLen > len(dst) {
|
||||
return 0, 1 // destination overflow
|
||||
}
|
||||
copy(dst[di:di+lLen], src[si:si+lLen])
|
||||
di += lLen
|
||||
si += lLen
|
||||
|
||||
// End of block.
|
||||
if si >= len(src) {
|
||||
return di, 0
|
||||
}
|
||||
|
||||
// Match offset.
|
||||
if si+2 > len(src) {
|
||||
return 0, 1
|
||||
}
|
||||
offset := int(src[si]) | int(src[si+1])<<8
|
||||
si += 2
|
||||
if offset == 0 {
|
||||
return 0, 2
|
||||
}
|
||||
|
||||
// Match length.
|
||||
mLen := token & 15
|
||||
if mLen == 15 {
|
||||
for {
|
||||
if si >= len(src) {
|
||||
return 0, 1
|
||||
}
|
||||
b := int(src[si])
|
||||
si++
|
||||
mLen += b
|
||||
if b != 255 {
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
mLen += 4
|
||||
|
||||
// Copy match (overlapping-safe).
|
||||
if di-offset < 0 {
|
||||
return 0, 1 // offset reaches before dst start
|
||||
}
|
||||
if di+mLen > len(dst) {
|
||||
return 0, 1 // destination overflow
|
||||
}
|
||||
for i := 0; i < mLen; i++ {
|
||||
dst[di+i] = dst[di-offset+i]
|
||||
}
|
||||
di += mLen
|
||||
}
|
||||
}
|
||||
|
||||
// genLZ4Block generates a random valid LZ4 block that decompresses into
|
||||
// approximately wantSize bytes. The block is always well-formed (ends with
|
||||
// a literals-only sequence).
|
||||
func genLZ4Block(rng *rand.Rand, wantSize int) []byte {
|
||||
var block []byte
|
||||
produced := 0
|
||||
for produced < wantSize {
|
||||
remaining := wantSize - produced
|
||||
|
||||
// Decide: emit a literals+match sequence or the final literals.
|
||||
if remaining <= 8 || rng.Intn(4) == 0 {
|
||||
// Final literals-only sequence.
|
||||
lLen := remaining
|
||||
if lLen > 60 {
|
||||
lLen = 1 + rng.Intn(60)
|
||||
}
|
||||
block = appendToken(block, lLen, 0)
|
||||
for i := 0; i < lLen; i++ {
|
||||
block = append(block, byte(rng.Intn(256)))
|
||||
}
|
||||
produced += lLen
|
||||
break
|
||||
}
|
||||
|
||||
// Literals + match.
|
||||
lLen := rng.Intn(min(16, remaining))
|
||||
if produced+lLen == 0 {
|
||||
lLen = 1 // must have at least 1 literal before the first match
|
||||
}
|
||||
mLenRaw := rng.Intn(12) // match length = mLenRaw + 4
|
||||
mLen := mLenRaw + 4
|
||||
if produced+mLen > remaining {
|
||||
mLen = remaining - produced
|
||||
if mLen < 4 {
|
||||
// Not enough room for a match; emit final literals.
|
||||
lLen = remaining
|
||||
block = appendToken(block, lLen, 0)
|
||||
for i := 0; i < lLen; i++ {
|
||||
block = append(block, byte(rng.Intn(256)))
|
||||
}
|
||||
break
|
||||
}
|
||||
mLenRaw = mLen - 4
|
||||
}
|
||||
|
||||
block = appendToken(block, lLen, mLenRaw)
|
||||
for i := 0; i < lLen; i++ {
|
||||
block = append(block, byte(rng.Intn(256)))
|
||||
}
|
||||
produced += lLen
|
||||
|
||||
// Offset: must be <= produced (can't reference before start).
|
||||
maxOff := produced
|
||||
if maxOff > 65535 {
|
||||
maxOff = 65535
|
||||
}
|
||||
offset := 1 + rng.Intn(maxOff)
|
||||
block = append(block, byte(offset), byte(offset>>8))
|
||||
produced += mLen
|
||||
}
|
||||
return block
|
||||
}
|
||||
|
||||
// appendToken appends a token (and extension bytes if needed) for the given
|
||||
// literal and match lengths.
|
||||
func appendToken(block []byte, lLen, mLenRaw int) []byte {
|
||||
lit4 := lLen
|
||||
if lit4 > 15 {
|
||||
lit4 = 15
|
||||
}
|
||||
ml4 := mLenRaw
|
||||
if ml4 > 15 {
|
||||
ml4 = 15
|
||||
}
|
||||
block = append(block, byte(lit4<<4|ml4))
|
||||
// Literal extension bytes.
|
||||
rem := lLen - 15
|
||||
for rem >= 255 {
|
||||
block = append(block, 255)
|
||||
rem -= 255
|
||||
}
|
||||
if lLen >= 15 {
|
||||
block = append(block, byte(rem))
|
||||
}
|
||||
// Match extension bytes.
|
||||
rem = mLenRaw - 15
|
||||
for rem >= 255 {
|
||||
block = append(block, 255)
|
||||
rem -= 255
|
||||
}
|
||||
if mLenRaw >= 15 {
|
||||
block = append(block, byte(rem))
|
||||
}
|
||||
return block
|
||||
}
|
||||
|
||||
func min(a, b int) int {
|
||||
if a < b {
|
||||
return a
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
// TestDifferentialLZ4Fuzz drives the JIT-assembled decodeBlockAVX2 with
|
||||
// random valid LZ4 blocks and compares the output bit-for-bit against the
|
||||
// portable Go reference.
|
||||
func TestDifferentialLZ4Fuzz(t *testing.T) {
|
||||
k := loadLZ4Kernel(t)
|
||||
|
||||
const iterations = 5000
|
||||
rng := rand.New(rand.NewSource(42))
|
||||
|
||||
for i := 0; i < iterations; i++ {
|
||||
wantSize := 1 + rng.Intn(4096)
|
||||
src := genLZ4Block(rng, wantSize)
|
||||
dstSize := wantSize + 64 // generous destination
|
||||
|
||||
// Go reference.
|
||||
goDst := make([]byte, dstSize)
|
||||
goN, goCode := decodeBlockGo(src, goDst)
|
||||
|
||||
// JIT kernel.
|
||||
jitDst := make([]byte, dstSize)
|
||||
jitN, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
|
||||
|
||||
if jitCode != goCode {
|
||||
t.Fatalf("iter %d: code mismatch: JIT=%d, Go=%d (src len=%d)",
|
||||
i, jitCode, goCode, len(src))
|
||||
}
|
||||
if jitCode != 0 {
|
||||
continue // both agree it's malformed/zero-offset
|
||||
}
|
||||
if jitN != goN {
|
||||
t.Fatalf("iter %d: n mismatch: JIT=%d, Go=%d (src len=%d)",
|
||||
i, jitN, goN, len(src))
|
||||
}
|
||||
if !bytes.Equal(jitDst[:jitN], goDst[:goN]) {
|
||||
t.Fatalf("iter %d: output mismatch (n=%d, src len=%d)", i, jitN, len(src))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestDifferentialLZ4Hostile drives the kernel with random garbage to check
|
||||
// that error codes agree with the Go reference (no crashes, same classification).
|
||||
func TestDifferentialLZ4Hostile(t *testing.T) {
|
||||
k := loadLZ4Kernel(t)
|
||||
|
||||
const iterations = 2000
|
||||
rng := rand.New(rand.NewSource(99))
|
||||
|
||||
for i := 0; i < iterations; i++ {
|
||||
srcLen := rng.Intn(128)
|
||||
src := make([]byte, srcLen)
|
||||
rng.Read(src)
|
||||
dstSize := rng.Intn(512)
|
||||
dst := make([]byte, dstSize)
|
||||
|
||||
// Go reference.
|
||||
goDst := make([]byte, dstSize)
|
||||
copy(goDst, dst)
|
||||
_, goCode := decodeBlockGo(src, goDst)
|
||||
|
||||
// JIT kernel.
|
||||
jitDst := make([]byte, dstSize)
|
||||
copy(jitDst, dst)
|
||||
_, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
|
||||
|
||||
if jitCode != goCode {
|
||||
t.Fatalf("iter %d: hostile code mismatch: JIT=%d, Go=%d (srcLen=%d, dstSize=%d)",
|
||||
i, jitCode, goCode, srcLen, dstSize)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// callDecodeBlockAVX2Raw is like callDecodeBlockAVX2 but accepts explicit
|
||||
// dst size (for hostile tests where dst may be smaller than the output).
|
||||
func callDecodeBlockAVX2Raw(t *testing.T, k *Kernel, src, dst []byte) (int, int) {
|
||||
t.Helper()
|
||||
args := make([]byte, 64)
|
||||
if len(src) > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(len(src)))
|
||||
PutUint64(args, 16, uint64(cap(src)))
|
||||
if len(dst) > 0 {
|
||||
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
||||
}
|
||||
PutUint64(args, 32, uint64(len(dst)))
|
||||
PutUint64(args, 40, uint64(cap(dst)))
|
||||
|
||||
out, err := k.CallFunc("decodeBlockAVX2", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFunc(decodeBlockAVX2): %v", err)
|
||||
}
|
||||
return int(GetUint64(out, 48)), int(GetUint64(out, 56))
|
||||
}
|
||||
@@ -0,0 +1,585 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"math/rand"
|
||||
"os"
|
||||
"testing"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
const flacKernelPath = "../../go-libraries/go-flac/avx2_amd64.s"
|
||||
|
||||
func loadFLACKernel(t *testing.T) *Kernel {
|
||||
t.Helper()
|
||||
if _, err := os.Stat(flacKernelPath); err != nil {
|
||||
t.Skipf("sibling kernel not available: %v", err)
|
||||
}
|
||||
k, err := Load(flacKernelPath)
|
||||
if err != nil {
|
||||
t.Fatalf("Load(%s): %v", flacKernelPath, err)
|
||||
}
|
||||
t.Cleanup(k.Close)
|
||||
return k
|
||||
}
|
||||
|
||||
// --- Portable Go references (from go-flac/simd.go) ---
|
||||
|
||||
func decodeMono16Go(src []byte, dst []int32) {
|
||||
for i := 0; i < len(dst); i++ {
|
||||
dst[i] = int32(int16(uint16(src[2*i]) | uint16(src[2*i+1])<<8))
|
||||
}
|
||||
}
|
||||
|
||||
func pack16Go(dst []byte, src []int32) {
|
||||
for i, v := range src {
|
||||
dst[2*i] = byte(v)
|
||||
dst[2*i+1] = byte(v >> 8)
|
||||
}
|
||||
}
|
||||
|
||||
func decorrelateLeftSideGo(left, right, out []int32) {
|
||||
for i := range left {
|
||||
l := left[i]
|
||||
out[2*i] = l
|
||||
out[2*i+1] = l - right[i]
|
||||
}
|
||||
}
|
||||
|
||||
func decorrelateSideRightGo(left, right, out []int32) {
|
||||
for i := range left {
|
||||
side := left[i]
|
||||
rch := right[i]
|
||||
out[2*i] = rch + side
|
||||
out[2*i+1] = rch
|
||||
}
|
||||
}
|
||||
|
||||
func decorrelateMidSideGo(left, right, out []int32) {
|
||||
for i := range left {
|
||||
mid := left[i]
|
||||
side := right[i]
|
||||
mid2 := mid<<1 | (side & 1)
|
||||
out[2*i] = (mid2 + side) >> 1
|
||||
out[2*i+1] = (mid2 - side) >> 1
|
||||
}
|
||||
}
|
||||
|
||||
func decorrelateInterleaveGo(left, right, out []int32) {
|
||||
for i := range left {
|
||||
out[2*i] = left[i]
|
||||
out[2*i+1] = right[i]
|
||||
}
|
||||
}
|
||||
|
||||
func analyzeO1RangeGo(swin []int32, dstP []uint32, hist *[32]uint16) (partSum uint64, overflow bool) {
|
||||
swin = swin[:len(dstP)+1]
|
||||
for j := 0; j+1 < len(swin); j++ {
|
||||
r := swin[j+1] - swin[j]
|
||||
if r == -2147483648 { // math.MinInt32
|
||||
overflow = true
|
||||
}
|
||||
f := uint32(r<<1) ^ uint32(r>>31)
|
||||
dstP[j] = f
|
||||
partSum += uint64(f)
|
||||
bl := 0
|
||||
for v := f; v > 0; v >>= 1 {
|
||||
bl++
|
||||
}
|
||||
if bl > 31 {
|
||||
bl = 31
|
||||
}
|
||||
hist[bl]++
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
func analyzeO2RangeGo(swin []int32, dstP []uint32, hist *[32]uint16) (partSum uint64, overflow bool) {
|
||||
swin = swin[:len(dstP)+2]
|
||||
for j := 0; j+2 < len(swin); j++ {
|
||||
r := swin[j+2] - 2*swin[j+1] + swin[j]
|
||||
if r == -2147483648 {
|
||||
overflow = true
|
||||
}
|
||||
f := uint32(r<<1) ^ uint32(r>>31)
|
||||
dstP[j] = f
|
||||
partSum += uint64(f)
|
||||
bl := 0
|
||||
for v := f; v > 0; v >>= 1 {
|
||||
bl++
|
||||
}
|
||||
if bl > 31 {
|
||||
bl = 31
|
||||
}
|
||||
hist[bl]++
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
func analyzeResRangeGo(swin []int32, dstP []uint32, hist *[32]uint16) (partSum uint64, overflow bool) {
|
||||
for j := 0; j < len(swin); j++ {
|
||||
r := swin[j]
|
||||
if r == -2147483648 {
|
||||
overflow = true
|
||||
}
|
||||
f := uint32(r<<1) ^ uint32(r>>31)
|
||||
dstP[j] = f
|
||||
partSum += uint64(f)
|
||||
bl := 0
|
||||
for v := f; v > 0; v >>= 1 {
|
||||
bl++
|
||||
}
|
||||
if bl > 31 {
|
||||
bl = 31
|
||||
}
|
||||
hist[bl]++
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
func decodeMono24Go(src []byte, dst []int32) {
|
||||
for i := 0; i < len(dst); i++ {
|
||||
off := 3 * i
|
||||
u := uint32(src[off]) | uint32(src[off+1])<<8 | uint32(src[off+2])<<16
|
||||
dst[i] = int32(u<<8) >> 8
|
||||
}
|
||||
}
|
||||
|
||||
func decodeStereo16Go(src []byte, left, right []int32) {
|
||||
for i := 0; i < len(left); i++ {
|
||||
left[i] = int32(int16(uint16(src[4*i]) | uint16(src[4*i+1])<<8))
|
||||
right[i] = int32(int16(uint16(src[4*i+2]) | uint16(src[4*i+3])<<8))
|
||||
}
|
||||
}
|
||||
|
||||
// --- Differential tests ---
|
||||
|
||||
func TestFLACDecodeMono16(t *testing.T) {
|
||||
k := loadFLACKernel(t)
|
||||
rng := rand.New(rand.NewSource(7))
|
||||
|
||||
for iter := 0; iter < 500; iter++ {
|
||||
n := rng.Intn(256)
|
||||
src := make([]byte, 2*n)
|
||||
rng.Read(src)
|
||||
|
||||
goDst := make([]int32, n)
|
||||
decodeMono16Go(src, goDst)
|
||||
|
||||
jitDst := make([]int32, n)
|
||||
args := make([]byte, 48)
|
||||
if len(src) > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(len(src)))
|
||||
PutUint64(args, 16, uint64(cap(src)))
|
||||
if n > 0 {
|
||||
PutPtr(args, 24, unsafe.Pointer(&jitDst[0]))
|
||||
}
|
||||
PutUint64(args, 32, uint64(n))
|
||||
PutUint64(args, 40, uint64(cap(jitDst)))
|
||||
|
||||
_, err := k.CallFunc("decodeMono16AVX2", args)
|
||||
if err != nil {
|
||||
t.Fatalf("iter %d: %v", iter, err)
|
||||
}
|
||||
for i := range goDst {
|
||||
if jitDst[i] != goDst[i] {
|
||||
t.Fatalf("iter %d: mismatch at [%d]: JIT=%d Go=%d", iter, i, jitDst[i], goDst[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestFLACPack16(t *testing.T) {
|
||||
k := loadFLACKernel(t)
|
||||
rng := rand.New(rand.NewSource(13))
|
||||
|
||||
for iter := 0; iter < 500; iter++ {
|
||||
n := rng.Intn(256)
|
||||
src := make([]int32, n)
|
||||
for i := range src {
|
||||
src[i] = int32(rng.Intn(65536) - 32768)
|
||||
}
|
||||
|
||||
goDst := make([]byte, 2*n)
|
||||
pack16Go(goDst, src)
|
||||
|
||||
jitDst := make([]byte, 2*n)
|
||||
args := make([]byte, 48)
|
||||
if len(jitDst) > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&jitDst[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(len(jitDst)))
|
||||
PutUint64(args, 16, uint64(cap(jitDst)))
|
||||
if n > 0 {
|
||||
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 32, uint64(n))
|
||||
PutUint64(args, 40, uint64(cap(src)))
|
||||
|
||||
_, err := k.CallFunc("pack16AVX2", args)
|
||||
if err != nil {
|
||||
t.Fatalf("iter %d: %v", iter, err)
|
||||
}
|
||||
if !bytes.Equal(jitDst, goDst) {
|
||||
t.Fatalf("iter %d: output mismatch (n=%d)", iter, n)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestFLACDecorrelate(t *testing.T) {
|
||||
k := loadFLACKernel(t)
|
||||
rng := rand.New(rand.NewSource(21))
|
||||
|
||||
kernels := []struct {
|
||||
name string
|
||||
ref func(left, right, out []int32)
|
||||
}{
|
||||
{"decorrelateLeftSideAVX2", decorrelateLeftSideGo},
|
||||
{"decorrelateSideRightAVX2", decorrelateSideRightGo},
|
||||
{"decorrelateMidSideAVX2", decorrelateMidSideGo},
|
||||
{"decorrelateInterleaveAVX2", decorrelateInterleaveGo},
|
||||
}
|
||||
|
||||
for _, kk := range kernels {
|
||||
t.Run(kk.name, func(t *testing.T) {
|
||||
for iter := 0; iter < 200; iter++ {
|
||||
n := rng.Intn(128)
|
||||
left := make([]int32, n)
|
||||
right := make([]int32, n)
|
||||
for i := range left {
|
||||
left[i] = int32(rng.Intn(1<<24) - 1<<23)
|
||||
right[i] = int32(rng.Intn(1<<24) - 1<<23)
|
||||
}
|
||||
|
||||
goOut := make([]int32, 2*n)
|
||||
kk.ref(left, right, goOut)
|
||||
|
||||
jitOut := make([]int32, 2*n)
|
||||
args := make([]byte, 72)
|
||||
if n > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&left[0]))
|
||||
PutPtr(args, 24, unsafe.Pointer(&right[0]))
|
||||
PutPtr(args, 48, unsafe.Pointer(&jitOut[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(n))
|
||||
PutUint64(args, 16, uint64(cap(left)))
|
||||
PutUint64(args, 32, uint64(n))
|
||||
PutUint64(args, 40, uint64(cap(right)))
|
||||
PutUint64(args, 56, uint64(2*n))
|
||||
PutUint64(args, 64, uint64(cap(jitOut)))
|
||||
|
||||
_, err := k.CallFunc(kk.name, args)
|
||||
if err != nil {
|
||||
t.Fatalf("iter %d: %v", iter, err)
|
||||
}
|
||||
for i := range goOut {
|
||||
if jitOut[i] != goOut[i] {
|
||||
t.Fatalf("iter %d: mismatch at [%d]: JIT=%d Go=%d", iter, i, jitOut[i], goOut[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestFLACAnalyzeO1Range(t *testing.T) {
|
||||
k := loadFLACKernel(t)
|
||||
rng := rand.New(rand.NewSource(33))
|
||||
|
||||
for iter := 0; iter < 300; iter++ {
|
||||
n := 1 + rng.Intn(128) // partition size
|
||||
swin := make([]int32, n+1)
|
||||
for i := range swin {
|
||||
swin[i] = int32(rng.Intn(1<<20) - 1<<19)
|
||||
}
|
||||
|
||||
goDstP := make([]uint32, n)
|
||||
var goHist [32]uint16
|
||||
goSum, goOvf := analyzeO1RangeGo(swin, goDstP, &goHist)
|
||||
|
||||
jitDstP := make([]uint32, n)
|
||||
var jitHist [32]uint16
|
||||
args := make([]byte, 72) // 65 rounded up
|
||||
PutPtr(args, 0, unsafe.Pointer(&swin[0]))
|
||||
PutUint64(args, 8, uint64(len(swin)))
|
||||
PutUint64(args, 16, uint64(cap(swin)))
|
||||
PutPtr(args, 24, unsafe.Pointer(&jitDstP[0]))
|
||||
PutUint64(args, 32, uint64(n))
|
||||
PutUint64(args, 40, uint64(cap(jitDstP)))
|
||||
PutPtr(args, 48, unsafe.Pointer(&jitHist[0]))
|
||||
|
||||
out, err := k.CallFunc("analyzeO1RangeAVX2", args)
|
||||
if err != nil {
|
||||
t.Fatalf("iter %d: %v", iter, err)
|
||||
}
|
||||
jitSum := GetUint64(out, 56)
|
||||
jitOvf := out[64] != 0
|
||||
|
||||
if jitSum != goSum {
|
||||
t.Fatalf("iter %d: partSum mismatch: JIT=%d Go=%d", iter, jitSum, goSum)
|
||||
}
|
||||
if jitOvf != goOvf {
|
||||
t.Fatalf("iter %d: overflow mismatch: JIT=%v Go=%v", iter, jitOvf, goOvf)
|
||||
}
|
||||
for i := range goDstP {
|
||||
if jitDstP[i] != goDstP[i] {
|
||||
t.Fatalf("iter %d: dstP[%d] mismatch: JIT=%d Go=%d", iter, i, jitDstP[i], goDstP[i])
|
||||
}
|
||||
}
|
||||
if jitHist != goHist {
|
||||
t.Fatalf("iter %d: hist mismatch: JIT=%v Go=%v", iter, jitHist, goHist)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestFLACFastStereoSums(t *testing.T) {
|
||||
k := loadFLACKernel(t)
|
||||
rng := rand.New(rand.NewSource(44))
|
||||
|
||||
for iter := 0; iter < 300; iter++ {
|
||||
n := 1 + rng.Intn(256)
|
||||
left := make([]int32, n)
|
||||
right := make([]int32, n)
|
||||
for i := range left {
|
||||
left[i] = int32(rng.Intn(1<<24) - 1<<23)
|
||||
right[i] = int32(rng.Intn(1<<24) - 1<<23)
|
||||
}
|
||||
|
||||
// Go reference: compute the four sums.
|
||||
var goSums [4]uint64
|
||||
for i := 0; i < n; i++ {
|
||||
l := left[i]
|
||||
r := right[i]
|
||||
side := l - r
|
||||
mid := (l + r) >> 1
|
||||
goSums[0] += foldAbs(l) + foldAbs(r)
|
||||
goSums[1] += foldAbs(l) + foldAbs(side)
|
||||
goSums[2] += foldAbs(side) + foldAbs(r)
|
||||
goSums[3] += foldAbs(mid) + foldAbs(side)
|
||||
}
|
||||
|
||||
var jitSums [4]uint64
|
||||
args := make([]byte, 56)
|
||||
PutPtr(args, 0, unsafe.Pointer(&left[0]))
|
||||
PutUint64(args, 8, uint64(n))
|
||||
PutUint64(args, 16, uint64(cap(left)))
|
||||
PutPtr(args, 24, unsafe.Pointer(&right[0]))
|
||||
PutUint64(args, 32, uint64(n))
|
||||
PutUint64(args, 40, uint64(cap(right)))
|
||||
PutPtr(args, 48, unsafe.Pointer(&jitSums[0]))
|
||||
|
||||
_, err := k.CallFunc("fastStereoSumsAVX2", args)
|
||||
if err != nil {
|
||||
t.Fatalf("iter %d: %v", iter, err)
|
||||
}
|
||||
if jitSums != goSums {
|
||||
t.Fatalf("iter %d: sums mismatch:\n JIT=%v\n Go =%v", iter, jitSums, goSums)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func foldAbs(v int32) uint64 {
|
||||
return uint64(uint32(v<<1) ^ uint32(v>>31))
|
||||
}
|
||||
|
||||
// runAnalyzeTest is the shared harness for the analyzeO*Range family.
|
||||
func runAnalyzeTest(t *testing.T, k *Kernel, name string, order int, ref func([]int32, []uint32, *[32]uint16) (uint64, bool)) {
|
||||
t.Helper()
|
||||
rng := rand.New(rand.NewSource(int64(order)*100 + 7))
|
||||
for iter := 0; iter < 200; iter++ {
|
||||
n := 1 + rng.Intn(128)
|
||||
swin := make([]int32, n+order)
|
||||
for i := range swin {
|
||||
swin[i] = int32(rng.Intn(1<<20) - 1<<19)
|
||||
}
|
||||
|
||||
goDstP := make([]uint32, n)
|
||||
var goHist [32]uint16
|
||||
goSum, goOvf := ref(swin, goDstP, &goHist)
|
||||
|
||||
jitDstP := make([]uint32, n)
|
||||
var jitHist [32]uint16
|
||||
args := make([]byte, 72)
|
||||
PutPtr(args, 0, unsafe.Pointer(&swin[0]))
|
||||
PutUint64(args, 8, uint64(len(swin)))
|
||||
PutUint64(args, 16, uint64(cap(swin)))
|
||||
PutPtr(args, 24, unsafe.Pointer(&jitDstP[0]))
|
||||
PutUint64(args, 32, uint64(n))
|
||||
PutUint64(args, 40, uint64(cap(jitDstP)))
|
||||
PutPtr(args, 48, unsafe.Pointer(&jitHist[0]))
|
||||
|
||||
out, err := k.CallFunc(name, args)
|
||||
if err != nil {
|
||||
t.Fatalf("iter %d: %v", iter, err)
|
||||
}
|
||||
jitSum := GetUint64(out, 56)
|
||||
jitOvf := out[64] != 0
|
||||
|
||||
if jitSum != goSum {
|
||||
t.Fatalf("iter %d: partSum: JIT=%d Go=%d", iter, jitSum, goSum)
|
||||
}
|
||||
if jitOvf != goOvf {
|
||||
t.Fatalf("iter %d: overflow: JIT=%v Go=%v", iter, jitOvf, goOvf)
|
||||
}
|
||||
for i := range goDstP {
|
||||
if jitDstP[i] != goDstP[i] {
|
||||
t.Fatalf("iter %d: dstP[%d]: JIT=%d Go=%d", iter, i, jitDstP[i], goDstP[i])
|
||||
}
|
||||
}
|
||||
if jitHist != goHist {
|
||||
t.Fatalf("iter %d: hist mismatch", iter)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestFLACAnalyzeO2Range(t *testing.T) {
|
||||
k := loadFLACKernel(t)
|
||||
runAnalyzeTest(t, k, "analyzeO2RangeAVX2", 2, analyzeO2RangeGo)
|
||||
}
|
||||
|
||||
func TestFLACAnalyzeResRange(t *testing.T) {
|
||||
k := loadFLACKernel(t)
|
||||
// analyzeResRange has order 0: swin IS the residual (no prediction).
|
||||
runAnalyzeTest(t, k, "analyzeResRangeAVX2", 0, analyzeResRangeGo)
|
||||
}
|
||||
|
||||
func TestFLACAnalyzeO3Range(t *testing.T) {
|
||||
k := loadFLACKernel(t)
|
||||
ref := func(swin []int32, dstP []uint32, hist *[32]uint16) (uint64, bool) {
|
||||
swin = swin[:len(dstP)+3]
|
||||
var partSum uint64
|
||||
var overflow bool
|
||||
for j := 0; j+3 < len(swin); j++ {
|
||||
r := swin[j+3] - 3*swin[j+2] + 3*swin[j+1] - swin[j]
|
||||
if r == -2147483648 {
|
||||
overflow = true
|
||||
}
|
||||
f := uint32(r<<1) ^ uint32(r>>31)
|
||||
dstP[j] = f
|
||||
partSum += uint64(f)
|
||||
bl := 0
|
||||
for v := f; v > 0; v >>= 1 {
|
||||
bl++
|
||||
}
|
||||
if bl > 31 {
|
||||
bl = 31
|
||||
}
|
||||
hist[bl]++
|
||||
}
|
||||
return partSum, overflow
|
||||
}
|
||||
runAnalyzeTest(t, k, "analyzeO3RangeAVX2", 3, ref)
|
||||
}
|
||||
|
||||
func TestFLACAnalyzeO4Range(t *testing.T) {
|
||||
k := loadFLACKernel(t)
|
||||
ref := func(swin []int32, dstP []uint32, hist *[32]uint16) (uint64, bool) {
|
||||
swin = swin[:len(dstP)+4]
|
||||
var partSum uint64
|
||||
var overflow bool
|
||||
for j := 0; j+4 < len(swin); j++ {
|
||||
r := swin[j+4] - 4*swin[j+3] + 6*swin[j+2] - 4*swin[j+1] + swin[j]
|
||||
if r == -2147483648 {
|
||||
overflow = true
|
||||
}
|
||||
f := uint32(r<<1) ^ uint32(r>>31)
|
||||
dstP[j] = f
|
||||
partSum += uint64(f)
|
||||
bl := 0
|
||||
for v := f; v > 0; v >>= 1 {
|
||||
bl++
|
||||
}
|
||||
if bl > 31 {
|
||||
bl = 31
|
||||
}
|
||||
hist[bl]++
|
||||
}
|
||||
return partSum, overflow
|
||||
}
|
||||
runAnalyzeTest(t, k, "analyzeO4RangeAVX2", 4, ref)
|
||||
}
|
||||
|
||||
func TestFLACDecodeMono24(t *testing.T) {
|
||||
k := loadFLACKernel(t)
|
||||
rng := rand.New(rand.NewSource(55))
|
||||
|
||||
for iter := 0; iter < 500; iter++ {
|
||||
n := rng.Intn(256)
|
||||
src := make([]byte, 3*n)
|
||||
rng.Read(src)
|
||||
|
||||
goDst := make([]int32, n)
|
||||
decodeMono24Go(src, goDst)
|
||||
|
||||
jitDst := make([]int32, n)
|
||||
args := make([]byte, 48)
|
||||
if len(src) > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(len(src)))
|
||||
PutUint64(args, 16, uint64(cap(src)))
|
||||
if n > 0 {
|
||||
PutPtr(args, 24, unsafe.Pointer(&jitDst[0]))
|
||||
}
|
||||
PutUint64(args, 32, uint64(n))
|
||||
PutUint64(args, 40, uint64(cap(jitDst)))
|
||||
|
||||
_, err := k.CallFunc("decodeMono24AVX2", args)
|
||||
if err != nil {
|
||||
t.Fatalf("iter %d: %v", iter, err)
|
||||
}
|
||||
for i := range goDst {
|
||||
if jitDst[i] != goDst[i] {
|
||||
t.Fatalf("iter %d: dst[%d]: JIT=%d Go=%d", iter, i, jitDst[i], goDst[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestFLACDecodeStereo16(t *testing.T) {
|
||||
k := loadFLACKernel(t)
|
||||
rng := rand.New(rand.NewSource(66))
|
||||
|
||||
for iter := 0; iter < 500; iter++ {
|
||||
n := rng.Intn(256)
|
||||
src := make([]byte, 4*n) // [L0,R0,L1,R1,...]
|
||||
rng.Read(src)
|
||||
|
||||
goLeft := make([]int32, n)
|
||||
goRight := make([]int32, n)
|
||||
decodeStereo16Go(src, goLeft, goRight)
|
||||
|
||||
jitLeft := make([]int32, n)
|
||||
jitRight := make([]int32, n)
|
||||
args := make([]byte, 72)
|
||||
if len(src) > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(len(src)))
|
||||
PutUint64(args, 16, uint64(cap(src)))
|
||||
if n > 0 {
|
||||
PutPtr(args, 24, unsafe.Pointer(&jitLeft[0]))
|
||||
PutPtr(args, 48, unsafe.Pointer(&jitRight[0]))
|
||||
}
|
||||
PutUint64(args, 32, uint64(n))
|
||||
PutUint64(args, 40, uint64(cap(jitLeft)))
|
||||
PutUint64(args, 56, uint64(n))
|
||||
PutUint64(args, 64, uint64(cap(jitRight)))
|
||||
|
||||
_, err := k.CallFunc("decodeStereo16AVX2", args)
|
||||
if err != nil {
|
||||
t.Fatalf("iter %d: %v", iter, err)
|
||||
}
|
||||
for i := 0; i < n; i++ {
|
||||
if jitLeft[i] != goLeft[i] || jitRight[i] != goRight[i] {
|
||||
t.Fatalf("iter %d: [%d] L: JIT=%d Go=%d; R: JIT=%d Go=%d",
|
||||
iter, i, jitLeft[i], goLeft[i], jitRight[i], goRight[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
+367
@@ -0,0 +1,367 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math/rand"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// FuzzResult reports the outcome of a differential fuzz campaign for one
|
||||
// function.
|
||||
type FuzzResult struct {
|
||||
Func string
|
||||
Iterations int
|
||||
Matches int
|
||||
Mismatches int
|
||||
FirstFail string // description of the first mismatch ("" if none)
|
||||
}
|
||||
|
||||
// OK returns true when all iterations matched.
|
||||
func (r FuzzResult) OK() bool { return r.Mismatches == 0 }
|
||||
|
||||
// String returns a human-readable summary.
|
||||
func (r FuzzResult) String() string {
|
||||
if r.OK() {
|
||||
return fmt.Sprintf("%s: %d/%d iterations match", r.Func, r.Matches, r.Iterations)
|
||||
}
|
||||
return fmt.Sprintf("%s: %d/%d match, %d MISMATCH — %s",
|
||||
r.Func, r.Matches, r.Iterations, r.Mismatches, r.FirstFail)
|
||||
}
|
||||
|
||||
// funcSig is a parsed // func signature from the assembly source.
|
||||
type funcSig struct {
|
||||
name string
|
||||
params []param
|
||||
results []param
|
||||
}
|
||||
|
||||
type param struct {
|
||||
name string
|
||||
typ string // "[]byte", "[]int32", "int", "*[32]uint16", etc.
|
||||
}
|
||||
|
||||
// funcSigRe matches the conventional "// func name(...)" comment.
|
||||
var funcSigRe = regexp.MustCompile(`^//\s*func\s+(\w+)\(([^)]*)\)\s*(.*)$`)
|
||||
|
||||
// parseFuncSig extracts the function signature from a "// func ..." comment.
|
||||
func parseFuncSig(comment string) (funcSig, bool) {
|
||||
m := funcSigRe.FindStringSubmatch(strings.TrimSpace(comment))
|
||||
if m == nil {
|
||||
return funcSig{}, false
|
||||
}
|
||||
sig := funcSig{name: m[1]}
|
||||
sig.params = parseParams(m[2])
|
||||
// Results may be "(a int, b int)" or "int" or "(int, error)".
|
||||
res := strings.TrimSpace(m[3])
|
||||
res = strings.TrimPrefix(res, "(")
|
||||
res = strings.TrimSuffix(res, ")")
|
||||
if res != "" {
|
||||
sig.results = parseParams(res)
|
||||
}
|
||||
return sig, true
|
||||
}
|
||||
|
||||
// parseParams splits "a []byte, b []int32" into typed parameters, handling
|
||||
// shared types ("a, b []int32").
|
||||
func parseParams(s string) []param {
|
||||
s = strings.TrimSpace(s)
|
||||
if s == "" {
|
||||
return nil
|
||||
}
|
||||
var out []param
|
||||
for _, field := range strings.Split(s, ",") {
|
||||
field = strings.TrimSpace(field)
|
||||
if field == "" {
|
||||
continue
|
||||
}
|
||||
parts := strings.Fields(field)
|
||||
if len(parts) == 1 {
|
||||
// Unnamed: "int" or "[]byte".
|
||||
out = append(out, param{typ: parts[0]})
|
||||
} else {
|
||||
// Named: "a []byte" or shared "a, b []int32" (handled by the
|
||||
// comma split above — "a" alone means the type follows in the
|
||||
// next field; this is a simplification that covers the common
|
||||
// case where each param has its own type).
|
||||
out = append(out, param{name: parts[0], typ: parts[1]})
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// ExtractSignatures scans assembly source for "// func name(...)" comments
|
||||
// that immediately precede a TEXT directive, and returns the parsed
|
||||
// signatures keyed by the function's short name.
|
||||
func ExtractSignatures(src string) map[string]funcSig {
|
||||
lines := strings.Split(src, "\n")
|
||||
sigs := make(map[string]funcSig)
|
||||
var comments []string
|
||||
for _, line := range lines {
|
||||
trimmed := strings.TrimSpace(line)
|
||||
if strings.HasPrefix(trimmed, "//") {
|
||||
comments = append(comments, trimmed)
|
||||
continue
|
||||
}
|
||||
if strings.HasPrefix(trimmed, "TEXT") {
|
||||
// Search the comment block for the // func line.
|
||||
for _, c := range comments {
|
||||
if sig, ok := parseFuncSig(c); ok {
|
||||
sigs[sig.name] = sig
|
||||
break
|
||||
}
|
||||
}
|
||||
comments = nil
|
||||
continue
|
||||
}
|
||||
if trimmed != "" {
|
||||
comments = nil
|
||||
}
|
||||
}
|
||||
return sigs
|
||||
}
|
||||
|
||||
// FuzzFunc runs a differential fuzz campaign: it JIT-executes both the
|
||||
// gasm-assembled and the go-tool-asm-assembled versions of the named
|
||||
// function with random inputs derived from the // func signature, and
|
||||
// compares the output argument area bit-for-bit.
|
||||
//
|
||||
// The signature comment must appear immediately above the TEXT directive
|
||||
// in the source (the conventional Go assembly layout).
|
||||
func (k *Kernel) FuzzFunc(name string, sig funcSig, goCode []byte, iterations int, seed int64) FuzzResult {
|
||||
result := FuzzResult{Func: name, Iterations: iterations}
|
||||
|
||||
rng := rand.New(rand.NewSource(seed))
|
||||
|
||||
// Map the Go-assembled code into a second executable region.
|
||||
goExec, err := Map(goCode)
|
||||
if err != nil {
|
||||
result.Mismatches = iterations
|
||||
result.FirstFail = fmt.Sprintf("map go code: %v", err)
|
||||
return result
|
||||
}
|
||||
defer goExec.Unmap()
|
||||
|
||||
fl, err := k.Func(name)
|
||||
if err != nil {
|
||||
result.Mismatches = iterations
|
||||
result.FirstFail = err.Error()
|
||||
return result
|
||||
}
|
||||
|
||||
for i := 0; i < iterations; i++ {
|
||||
// Generate inputs and build TWO independent arg blocks (one per
|
||||
// version) so that functions which write to their arguments
|
||||
// (e.g. histogram increments) don't corrupt the other's input.
|
||||
gasmArgs, goArgs, bufs := genDualArgs(rng, sig, fl.Args)
|
||||
|
||||
// Call the gasm version.
|
||||
gasmOut, err := k.CallFunc(name, gasmArgs)
|
||||
if err != nil {
|
||||
result.Mismatches++
|
||||
if result.FirstFail == "" {
|
||||
result.FirstFail = fmt.Sprintf("iter %d: gasm call: %v", i, err)
|
||||
}
|
||||
releaseBufs(bufs)
|
||||
continue
|
||||
}
|
||||
|
||||
// Call the Go version (same function, independent buffers).
|
||||
goOut, err := Call(goExec.FuncAddr(0), goArgs)
|
||||
if err != nil {
|
||||
result.Mismatches++
|
||||
if result.FirstFail == "" {
|
||||
result.FirstFail = fmt.Sprintf("iter %d: go call: %v", i, err)
|
||||
}
|
||||
releaseBufs(bufs)
|
||||
continue
|
||||
}
|
||||
|
||||
// Compare only the result area (after all input parameters).
|
||||
// Pointers in the arg block differ (separate buffers), so we
|
||||
// compare from resultOff to the end.
|
||||
resultOff := paramsSize(sig)
|
||||
gasmRes := gasmOut[resultOff:]
|
||||
goRes := goOut[resultOff:]
|
||||
if !equalBytes(gasmRes, goRes) {
|
||||
result.Mismatches++
|
||||
if result.FirstFail == "" {
|
||||
result.FirstFail = fmt.Sprintf("iter %d: output mismatch at result offset %d", i, resultOff)
|
||||
}
|
||||
} else {
|
||||
result.Matches++
|
||||
}
|
||||
releaseBufs(bufs)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// genDualArgs generates two independent ABI0 argument blocks (for gasm and
|
||||
// go) with identical logical content but separate backing buffers, so that
|
||||
// functions which write to their arguments don't corrupt the other's input.
|
||||
func genDualArgs(rng *rand.Rand, sig funcSig, argSize int) (gasmArgs, goArgs []byte, bufs [][]byte) {
|
||||
gasmArgs = make([]byte, argSize)
|
||||
goArgs = make([]byte, argSize)
|
||||
off := 0
|
||||
sliceIdx := 0
|
||||
|
||||
for _, p := range sig.params {
|
||||
switch {
|
||||
case strings.HasPrefix(p.typ, "[]"):
|
||||
elemSize := elemSizeFor(p.typ)
|
||||
n := 1 + rng.Intn(127)
|
||||
var declaredLen int
|
||||
if sliceIdx == 0 {
|
||||
declaredLen = n
|
||||
} else {
|
||||
declaredLen = n + 512
|
||||
}
|
||||
// Allocate a buffer comfortably larger than declaredLen*elemSize so
|
||||
// that SIMD over-reads and functions that write slightly past len
|
||||
// (e.g. decoders that trust len(src)) never touch unmapped memory.
|
||||
bufBytes := declaredLen*elemSize + 8192
|
||||
// Two independent buffers with identical random content.
|
||||
buf1 := make([]byte, bufBytes)
|
||||
buf2 := make([]byte, bufBytes)
|
||||
rng.Read(buf1[:n*elemSize])
|
||||
copy(buf2, buf1)
|
||||
bufs = append(bufs, buf1, buf2)
|
||||
putPtr(gasmArgs, off, unsafe.Pointer(&buf1[0]))
|
||||
putPtr(goArgs, off, unsafe.Pointer(&buf2[0]))
|
||||
// len and cap both equal declaredLen — the buffer is guaranteed
|
||||
// to hold at least declaredLen elements plus safety margin.
|
||||
putU64(gasmArgs, off+8, uint64(declaredLen))
|
||||
putU64(gasmArgs, off+16, uint64(declaredLen))
|
||||
putU64(goArgs, off+8, uint64(declaredLen))
|
||||
putU64(goArgs, off+16, uint64(declaredLen))
|
||||
off += 24
|
||||
sliceIdx++
|
||||
|
||||
case strings.HasPrefix(p.typ, "*["):
|
||||
nElem := arrayLen(p.typ)
|
||||
elem := elemSizeFor("[]" + p.typ[strings.Index(p.typ, "]")+1:])
|
||||
size := nElem * elem
|
||||
if size < 8 {
|
||||
size = 8
|
||||
}
|
||||
buf1 := make([]byte, size)
|
||||
buf2 := make([]byte, size)
|
||||
rng.Read(buf1)
|
||||
copy(buf2, buf1)
|
||||
bufs = append(bufs, buf1, buf2)
|
||||
putPtr(gasmArgs, off, unsafe.Pointer(&buf1[0]))
|
||||
putPtr(goArgs, off, unsafe.Pointer(&buf2[0]))
|
||||
off += 8
|
||||
|
||||
case p.typ == "int" || p.typ == "uint" || p.typ == "int64" || p.typ == "uint64":
|
||||
v := uint64(rng.Intn(256))
|
||||
putU64(gasmArgs, off, v)
|
||||
putU64(goArgs, off, v)
|
||||
off += 8
|
||||
|
||||
default:
|
||||
v := rng.Uint64()
|
||||
putU64(gasmArgs, off, v)
|
||||
putU64(goArgs, off, v)
|
||||
off += 8
|
||||
}
|
||||
}
|
||||
return gasmArgs, goArgs, bufs
|
||||
}
|
||||
|
||||
func elemSizeFor(sliceType string) int {
|
||||
switch strings.TrimPrefix(sliceType, "[]") {
|
||||
case "byte", "uint8", "int8":
|
||||
return 1
|
||||
case "uint16", "int16":
|
||||
return 2
|
||||
case "uint32", "int32", "float32":
|
||||
return 4
|
||||
case "uint64", "int64", "float64":
|
||||
return 8
|
||||
default:
|
||||
return 8
|
||||
}
|
||||
}
|
||||
|
||||
// paramsSize returns the ABI0 stack size occupied by the input parameters.
|
||||
func paramsSize(sig funcSig) int {
|
||||
size := 0
|
||||
for _, p := range sig.params {
|
||||
switch {
|
||||
case strings.HasPrefix(p.typ, "[]"):
|
||||
size += 24 // slice header
|
||||
case strings.HasPrefix(p.typ, "*["):
|
||||
size += 8 // pointer
|
||||
case p.typ == "bool":
|
||||
size += 1
|
||||
default:
|
||||
size += 8 // int, uint, etc.
|
||||
}
|
||||
}
|
||||
return size
|
||||
}
|
||||
|
||||
func arrayLen(typ string) int {
|
||||
// "*[32]uint16" → 32
|
||||
start := strings.Index(typ, "[")
|
||||
end := strings.Index(typ, "]")
|
||||
if start < 0 || end < 0 || end <= start {
|
||||
return 1
|
||||
}
|
||||
n, _ := strconv.Atoi(typ[start+1 : end])
|
||||
if n <= 0 {
|
||||
n = 1
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
func putPtr(buf []byte, off int, p unsafe.Pointer) {
|
||||
if off+8 <= len(buf) {
|
||||
u64 := uint64(uintptr(p))
|
||||
buf[off] = byte(u64)
|
||||
buf[off+1] = byte(u64 >> 8)
|
||||
buf[off+2] = byte(u64 >> 16)
|
||||
buf[off+3] = byte(u64 >> 24)
|
||||
buf[off+4] = byte(u64 >> 32)
|
||||
buf[off+5] = byte(u64 >> 40)
|
||||
buf[off+6] = byte(u64 >> 48)
|
||||
buf[off+7] = byte(u64 >> 56)
|
||||
}
|
||||
}
|
||||
|
||||
func putU64(buf []byte, off int, v uint64) {
|
||||
if off+8 <= len(buf) {
|
||||
buf[off] = byte(v)
|
||||
buf[off+1] = byte(v >> 8)
|
||||
buf[off+2] = byte(v >> 16)
|
||||
buf[off+3] = byte(v >> 24)
|
||||
buf[off+4] = byte(v >> 32)
|
||||
buf[off+5] = byte(v >> 40)
|
||||
buf[off+6] = byte(v >> 48)
|
||||
buf[off+7] = byte(v >> 56)
|
||||
}
|
||||
}
|
||||
|
||||
func equalBytes(a, b []byte) bool {
|
||||
if len(a) != len(b) {
|
||||
return false
|
||||
}
|
||||
for i := range a {
|
||||
if a[i] != b[i] {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func releaseBufs(bufs [][]byte) {
|
||||
// Keep buffers alive until after the call; nothing to free in Go,
|
||||
// but this prevents the compiler from collecting them too early.
|
||||
_ = bufs
|
||||
}
|
||||
@@ -0,0 +1,98 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestExtractSignatures(t *testing.T) {
|
||||
src := `// func add(a int64, b int64) int64
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
RET
|
||||
|
||||
// func wideCopy(dst []byte, src []byte)
|
||||
TEXT ·wideCopy(SB), NOSPLIT, $0-48
|
||||
RET
|
||||
`
|
||||
sigs := ExtractSignatures(src)
|
||||
if len(sigs) != 2 {
|
||||
t.Fatalf("expected 2 signatures, got %d: %v", len(sigs), sigs)
|
||||
}
|
||||
add, ok := sigs["add"]
|
||||
if !ok {
|
||||
t.Fatal("add not found")
|
||||
}
|
||||
if len(add.params) != 2 {
|
||||
t.Errorf("add params: got %d, want 2", len(add.params))
|
||||
}
|
||||
wc, ok := sigs["wideCopy"]
|
||||
if !ok {
|
||||
t.Fatal("wideCopy not found")
|
||||
}
|
||||
if len(wc.params) != 2 {
|
||||
t.Errorf("wideCopy params: got %d, want 2", len(wc.params))
|
||||
}
|
||||
if wc.params[0].typ != "[]byte" {
|
||||
t.Errorf("wideCopy param[0].typ = %q, want []byte", wc.params[0].typ)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFuzzWideCopy(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
|
||||
gt, err := GroundTruth("../testdata/verify/basic_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GroundTruth: %v", err)
|
||||
}
|
||||
goCode, ok := gt["wideCopy"]
|
||||
if !ok {
|
||||
t.Skip("wideCopy not in ground truth")
|
||||
}
|
||||
|
||||
sig := funcSig{
|
||||
name: "wideCopy",
|
||||
params: []param{
|
||||
{name: "dst", typ: "[]byte"},
|
||||
{name: "src", typ: "[]byte"},
|
||||
},
|
||||
}
|
||||
|
||||
res := k.FuzzFunc("wideCopy", sig, goCode, 200, 42)
|
||||
if !res.OK() {
|
||||
t.Errorf("wideCopy fuzz: %s", res)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseFuncSig(t *testing.T) {
|
||||
tests := []struct {
|
||||
comment string
|
||||
name string
|
||||
nParams int
|
||||
}{
|
||||
{"// func add(a int64, b int64) int64", "add", 2},
|
||||
{"// func wideCopy(dst []byte, src []byte)", "wideCopy", 2},
|
||||
{"// func analyzeO1RangeAVX2(swin []int32, dstP []uint32, hist *[32]uint16) (partSum uint64, overflow bool)", "analyzeO1RangeAVX2", 3},
|
||||
{"// not a func", "", 0},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
sig, ok := parseFuncSig(tt.comment)
|
||||
if tt.name == "" {
|
||||
if ok {
|
||||
t.Errorf("parseFuncSig(%q): expected not ok", tt.comment)
|
||||
}
|
||||
continue
|
||||
}
|
||||
if !ok {
|
||||
t.Errorf("parseFuncSig(%q): expected ok", tt.comment)
|
||||
continue
|
||||
}
|
||||
if sig.name != tt.name {
|
||||
t.Errorf("parseFuncSig(%q).name = %q, want %q", tt.comment, sig.name, tt.name)
|
||||
}
|
||||
if len(sig.params) != tt.nParams {
|
||||
t.Errorf("parseFuncSig(%q): %d params, want %d", tt.comment, len(sig.params), tt.nParams)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,188 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// GroundTruth assembles the given .s file with the Go toolchain's own
|
||||
// assembler and returns the machine code bytes for each TEXT function,
|
||||
// keyed by the function's short name (the part after the middle dot).
|
||||
// This is the universal oracle: any file that `go tool asm` accepts can
|
||||
// be verified, with no hand-written reference.
|
||||
//
|
||||
// For RISC-V sources the assembler is invoked with GOARCH=riscv64;
|
||||
// the caller must set the architecture via GroundTruthArch.
|
||||
func GroundTruth(path string) (map[string][]byte, error) {
|
||||
return groundTruthArch(path, "")
|
||||
}
|
||||
|
||||
// GroundTruthRISCV assembles the given .s file with the Go toolchain in
|
||||
// RISC-V cross-assembly mode (GOARCH=riscv64).
|
||||
func GroundTruthRISCV(path string) (map[string][]byte, error) {
|
||||
return groundTruthArch(path, "riscv64")
|
||||
}
|
||||
|
||||
func groundTruthArch(path, goarch string) (map[string][]byte, error) {
|
||||
goroot := runtime.GOROOT()
|
||||
asmBin := filepath.Join(goroot, "pkg", "tool", runtime.GOOS+"_"+runtime.GOARCH, "asm")
|
||||
if _, err := os.Stat(asmBin); err != nil {
|
||||
return nil, fmt.Errorf("verify: go tool asm not found at %s: %w", asmBin, err)
|
||||
}
|
||||
includeDir := filepath.Join(goroot, "pkg", "include")
|
||||
|
||||
tmpDir, err := os.MkdirTemp("", "gasm-verify-*")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("verify: tempdir: %w", err)
|
||||
}
|
||||
defer os.RemoveAll(tmpDir)
|
||||
objPath := filepath.Join(tmpDir, "out.o")
|
||||
|
||||
base := filepath.Base(path)
|
||||
pkg := strings.TrimSuffix(base, ".s")
|
||||
pkg = strings.TrimSuffix(pkg, "_amd64")
|
||||
pkg = strings.TrimSuffix(pkg, "_riscv64")
|
||||
|
||||
cmd := exec.Command(asmBin, "-I", includeDir, "-p", pkg, "-o", objPath, path)
|
||||
if goarch != "" {
|
||||
cmd.Env = append(os.Environ(), "GOARCH="+goarch)
|
||||
}
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
return nil, fmt.Errorf("verify: go tool asm (%s): %w\n%s", goarch, err, out)
|
||||
}
|
||||
|
||||
objData, err := os.ReadFile(objPath)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("verify: read object: %w", err)
|
||||
}
|
||||
return extractGOOBJCode(objData)
|
||||
}
|
||||
|
||||
// GOOBJ block indices (cmd/internal/goobj).
|
||||
const (
|
||||
blkAutolib = iota
|
||||
blkPkgIdx
|
||||
blkFile
|
||||
blkSymdef
|
||||
blkHashed64def
|
||||
blkHasheddef
|
||||
blkNonpkgdef
|
||||
blkNonpkgref
|
||||
blkRefFlags
|
||||
blkHash64
|
||||
blkHash
|
||||
blkRelocIdx
|
||||
blkAuxIdx
|
||||
blkDataIdx
|
||||
blkReloc
|
||||
blkAux
|
||||
blkData
|
||||
blkRefName
|
||||
blkEnd
|
||||
)
|
||||
|
||||
const goobjMagic = "\x00go120ld"
|
||||
|
||||
// extractGOOBJCode parses a GOOBJ payload and returns the code bytes for
|
||||
// each non-package STEXT symbol (the functions).
|
||||
func extractGOOBJCode(data []byte) (map[string][]byte, error) {
|
||||
// Find the GOOBJ header (after the "go object ..." preamble).
|
||||
i := bytes.Index(data, []byte(goobjMagic))
|
||||
if i < 0 {
|
||||
return nil, fmt.Errorf("verify: no GOOBJ magic in object file")
|
||||
}
|
||||
b := data[i:]
|
||||
le := binary.LittleEndian
|
||||
|
||||
// Read block offsets (20 bytes into the header: 4 magic + 8 go version
|
||||
// + 8 experiment = 20, then blkEnd+1 uint32 offsets).
|
||||
var offs [blkEnd + 1]uint32
|
||||
for j := 0; j <= blkEnd; j++ {
|
||||
offs[j] = le.Uint32(b[20+4*j:])
|
||||
}
|
||||
blk := func(idx int) []byte { return b[offs[idx]:offs[idx+1]] }
|
||||
|
||||
// Parse non-package symbol definitions (blkNonpkgdef): each entry is
|
||||
// 21 bytes: [nameLen:4][nameOff:4][abi:2][type:1][flag:1][flag2:1][size:4][align:4].
|
||||
const symSize = 21
|
||||
nonpkg := blk(blkNonpkgdef)
|
||||
nSyms := len(nonpkg) / symSize
|
||||
|
||||
// Data index (blkDataIdx): one uint32 per defined symbol across ALL
|
||||
// definition blocks (blkSymdef + blkHashed64def + blkHasheddef +
|
||||
// blkNonpkgdef), in that order. We need the offset for the nonpkg
|
||||
// symbols, which come last.
|
||||
dataIdx := blk(blkDataIdx)
|
||||
dataBlk := blk(blkData)
|
||||
|
||||
// Count symbols in the preceding definition blocks.
|
||||
preceding := 0
|
||||
for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef} {
|
||||
preceding += len(blk(bi)) / symSize
|
||||
}
|
||||
|
||||
// Symbol name offsets in the GOOBJ symbol table are absolute byte
|
||||
// offsets from the start of the GOOBJ payload (the magic).
|
||||
readStr := func(off, ln uint32) string {
|
||||
if int(off+ln) > len(b) {
|
||||
return ""
|
||||
}
|
||||
return string(b[off : off+ln])
|
||||
}
|
||||
|
||||
result := make(map[string][]byte)
|
||||
const kindSTEXT = 1
|
||||
for s := 0; s < nSyms; s++ {
|
||||
x := nonpkg[s*symSize:]
|
||||
nameLen := le.Uint32(x[0:])
|
||||
nameOff := le.Uint32(x[4:])
|
||||
typ := x[10]
|
||||
size := le.Uint32(x[13:])
|
||||
|
||||
if typ != kindSTEXT || size == 0 {
|
||||
continue
|
||||
}
|
||||
name := readStr(nameOff, nameLen)
|
||||
// Strip the package prefix (everything up to and including the
|
||||
// last middle dot or period-dot).
|
||||
name = stripPkg(name)
|
||||
|
||||
// Data offset from the index (nonpkg symbols follow the preceding blocks).
|
||||
diIdx := preceding + s
|
||||
if (diIdx+1)*4 > len(dataIdx) {
|
||||
continue
|
||||
}
|
||||
dOff := le.Uint32(dataIdx[diIdx*4:])
|
||||
if int(dOff+size) > len(dataBlk) {
|
||||
continue
|
||||
}
|
||||
code := make([]byte, size)
|
||||
copy(code, dataBlk[dOff:dOff+size])
|
||||
result[name] = code
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// stripPkg removes the package path prefix from a symbol name, leaving
|
||||
// just the function name. "pkg/path·FuncName" → "FuncName".
|
||||
func stripPkg(name string) string {
|
||||
if i := strings.LastIndex(name, "\u00B7"); i >= 0 {
|
||||
return name[i+len("\u00B7"):]
|
||||
}
|
||||
if i := strings.LastIndex(name, "\"."); i >= 0 {
|
||||
return name[i+2:]
|
||||
}
|
||||
if i := strings.LastIndex(name, "."); i >= 0 {
|
||||
return name[i+1:]
|
||||
}
|
||||
return name
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestGroundTruthBasic(t *testing.T) {
|
||||
// Use the simple test kernel — it assembles with go tool asm.
|
||||
gt, err := GroundTruth("../testdata/verify/basic_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GroundTruth: %v", err)
|
||||
}
|
||||
if len(gt) == 0 {
|
||||
t.Fatal("no functions extracted from ground truth")
|
||||
}
|
||||
// The "add" function should be present and non-empty.
|
||||
code, ok := gt["add"]
|
||||
if !ok {
|
||||
t.Fatalf("function 'add' not found in ground truth; got: %v", keys(gt))
|
||||
}
|
||||
if len(code) == 0 {
|
||||
t.Fatal("add: zero-length code")
|
||||
}
|
||||
t.Logf("ground truth functions: %v", keys(gt))
|
||||
}
|
||||
|
||||
func TestGroundTruthComparison(t *testing.T) {
|
||||
// Assemble with gasm and compare against go tool asm.
|
||||
k, err := Load("../testdata/verify/basic_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("Load: %v", err)
|
||||
}
|
||||
defer k.Close()
|
||||
|
||||
gt, err := GroundTruth("../testdata/verify/basic_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GroundTruth: %v", err)
|
||||
}
|
||||
|
||||
for _, name := range k.FuncNames() {
|
||||
fl, _ := k.Func(name)
|
||||
gasmCode := k.Image().Code[fl.Offset : fl.Offset+fl.Size]
|
||||
goCode, ok := gt[name]
|
||||
if !ok {
|
||||
t.Errorf("%s: not in ground truth", name)
|
||||
continue
|
||||
}
|
||||
if len(gasmCode) != len(goCode) {
|
||||
t.Errorf("%s: size mismatch: gasm=%d go=%d", name, len(gasmCode), len(goCode))
|
||||
continue
|
||||
}
|
||||
for i := range gasmCode {
|
||||
if gasmCode[i] != goCode[i] {
|
||||
t.Errorf("%s: byte %d differs: gasm=%02x go=%02x", name, i, gasmCode[i], goCode[i])
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestGroundTruthBadFile(t *testing.T) {
|
||||
_, err := GroundTruth("/nonexistent/file_amd64.s")
|
||||
if err == nil {
|
||||
t.Fatal("expected error for nonexistent file")
|
||||
}
|
||||
}
|
||||
|
||||
func keys(m map[string][]byte) []string {
|
||||
out := make([]string, 0, len(m))
|
||||
for k := range m {
|
||||
out = append(out, k)
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package verify provides the dynamic-analysis substrate for gasm: it
|
||||
// JIT-assembles Plan 9 amd64 kernels into executable memory and calls them
|
||||
// directly, enabling differential testing against portable Go references,
|
||||
// runtime ABI checks and basic-block coverage profiling.
|
||||
//
|
||||
// The execution model is pure Go (stdlib only): machine code is mapped with
|
||||
// syscall.Mmap and invoked through an assembly trampoline that switches to a
|
||||
// prepared ABI0 stack. No cgo, no external toolchain.
|
||||
package verify
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// Executable maps a copy of code into a read-execute memory region suitable
|
||||
// for direct invocation. The mapping is anonymous and private; the original
|
||||
// slice is not retained. Call Unmap to release the region.
|
||||
type Executable struct {
|
||||
addr uintptr // base address of the mapping
|
||||
size int
|
||||
mem []byte // the mmap'd slice (for Unmap)
|
||||
}
|
||||
|
||||
// Map copies code into a freshly allocated RX region and returns it.
|
||||
// The mapping is PROT_READ|PROT_EXEC; writes are not permitted after the
|
||||
// copy, matching W^X policy.
|
||||
func Map(code []byte) (*Executable, error) {
|
||||
size := len(code)
|
||||
if size == 0 {
|
||||
return nil, fmt.Errorf("verify: cannot map zero-length code")
|
||||
}
|
||||
// Round up to the page size.
|
||||
const pageSize = 4096
|
||||
mapSize := (size + pageSize - 1) &^ (pageSize - 1)
|
||||
|
||||
mem, err := syscall.Mmap(-1, 0, mapSize,
|
||||
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("verify: mmap: %w", err)
|
||||
}
|
||||
copy(mem, code)
|
||||
|
||||
// Remove write permission (W^X).
|
||||
if err := syscall.Mprotect(mem, syscall.PROT_READ|syscall.PROT_EXEC); err != nil {
|
||||
syscall.Munmap(mem)
|
||||
return nil, fmt.Errorf("verify: mprotect: %w", err)
|
||||
}
|
||||
return &Executable{
|
||||
addr: uintptr(unsafe.Pointer(&mem[0])),
|
||||
size: size,
|
||||
mem: mem,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// Unmap releases the executable region.
|
||||
func (e *Executable) Unmap() {
|
||||
if e.mem != nil {
|
||||
syscall.Munmap(e.mem)
|
||||
e.mem = nil
|
||||
}
|
||||
}
|
||||
|
||||
// FuncAddr returns the absolute address of a function at the given offset
|
||||
// within the mapped image.
|
||||
func (e *Executable) FuncAddr(offset int) uintptr {
|
||||
return e.addr + uintptr(offset)
|
||||
}
|
||||
|
||||
// PutUint64 writes v into buf at byte offset off (little-endian).
|
||||
func PutUint64(buf []byte, off int, v uint64) {
|
||||
binary.LittleEndian.PutUint64(buf[off:off+8], v)
|
||||
}
|
||||
|
||||
// GetUint64 reads a little-endian uint64 from buf at byte offset off.
|
||||
func GetUint64(buf []byte, off int) uint64 {
|
||||
return binary.LittleEndian.Uint64(buf[off : off+8])
|
||||
}
|
||||
|
||||
// PutPtr writes a pointer value into buf at byte offset off.
|
||||
func PutPtr(buf []byte, off int, p unsafe.Pointer) {
|
||||
binary.LittleEndian.PutUint64(buf[off:off+8], uint64(uintptr(p)))
|
||||
}
|
||||
@@ -0,0 +1,215 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"testing"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
func loadBasic(t *testing.T) *Kernel {
|
||||
t.Helper()
|
||||
k, err := Load("../testdata/verify/basic_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("Load: %v", err)
|
||||
}
|
||||
t.Cleanup(k.Close)
|
||||
return k
|
||||
}
|
||||
|
||||
func TestJITAdd(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
|
||||
tests := []struct {
|
||||
a, b, want int64
|
||||
}{
|
||||
{0, 0, 0},
|
||||
{1, 2, 3},
|
||||
{-1, 1, 0},
|
||||
{1 << 62, 1 << 62, -9223372036854775808}, // overflow wraps (MinInt64)
|
||||
{-100, -200, -300},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
args := make([]byte, 24)
|
||||
PutUint64(args, 0, uint64(tt.a))
|
||||
PutUint64(args, 8, uint64(tt.b))
|
||||
|
||||
out, err := k.CallFunc("add", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFunc(add, %d, %d): %v", tt.a, tt.b, err)
|
||||
}
|
||||
got := int64(GetUint64(out, 16))
|
||||
if got != tt.want {
|
||||
t.Errorf("add(%d, %d) = %d, want %d", tt.a, tt.b, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestJITSum(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
|
||||
tests := []struct {
|
||||
data []int64
|
||||
want int64
|
||||
}{
|
||||
{nil, 0},
|
||||
{[]int64{1}, 1},
|
||||
{[]int64{1, 2, 3, 4, 5}, 15},
|
||||
{[]int64{-10, 20, -30, 40}, 20},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
args := make([]byte, 32)
|
||||
if len(tt.data) > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&tt.data[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(len(tt.data)))
|
||||
PutUint64(args, 16, uint64(cap(tt.data)))
|
||||
|
||||
out, err := k.CallFunc("sum", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFunc(sum, %v): %v", tt.data, err)
|
||||
}
|
||||
got := int64(GetUint64(out, 24))
|
||||
if got != tt.want {
|
||||
t.Errorf("sum(%v) = %d, want %d", tt.data, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestJITWideCopy(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
n int
|
||||
}{
|
||||
{"empty", 0},
|
||||
{"tiny", 7},
|
||||
{"exact32", 32},
|
||||
{"overlap_range", 48},
|
||||
{"exact64", 64},
|
||||
{"unaligned", 45},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
src := make([]byte, tt.n)
|
||||
for i := range src {
|
||||
src[i] = byte(i * 7)
|
||||
}
|
||||
dst := make([]byte, tt.n)
|
||||
|
||||
args := make([]byte, 48)
|
||||
if tt.n > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
|
||||
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(tt.n)) // dst_len
|
||||
PutUint64(args, 16, uint64(tt.n)) // dst_cap
|
||||
PutUint64(args, 32, uint64(tt.n)) // src_len
|
||||
PutUint64(args, 40, uint64(tt.n)) // src_cap
|
||||
|
||||
_, err := k.CallFunc("wideCopy", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFunc(wideCopy): %v", err)
|
||||
}
|
||||
if !bytes.Equal(dst, src) {
|
||||
t.Errorf("wideCopy: dst ≠ src\n got %x\n want %x", dst, src)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestKernelFuncNames(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
names := k.FuncNames()
|
||||
want := []string{"add", "sum", "wideCopy"}
|
||||
if len(names) != len(want) {
|
||||
t.Fatalf("FuncNames() = %v, want %v", names, want)
|
||||
}
|
||||
for i, n := range names {
|
||||
if n != want[i] {
|
||||
t.Errorf("FuncNames()[%d] = %q, want %q", i, n, want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestKernelFuncNotFound(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
_, err := k.CallFunc("nonexistent", make([]byte, 8))
|
||||
if err == nil {
|
||||
t.Fatal("expected error for nonexistent function")
|
||||
}
|
||||
}
|
||||
|
||||
func TestKernelArgTooSmall(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
_, err := k.CallFunc("add", make([]byte, 8)) // needs 24
|
||||
if err == nil {
|
||||
t.Fatal("expected error for too-small arg block")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMapZeroLength(t *testing.T) {
|
||||
_, err := Map(nil)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for zero-length code")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLoadSourceError(t *testing.T) {
|
||||
_, err := LoadSource("bad.s", "TEXT ·f(SB), NOSPLIT\n\tBADINSTRUCTION\n")
|
||||
// The parser may or may not error on unknown instructions (it's
|
||||
// error-tolerant), but the assembler will reject it.
|
||||
if err == nil {
|
||||
t.Log("LoadSource succeeded unexpectedly (parser is error-tolerant)")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLoadSourceParseError(t *testing.T) {
|
||||
// A completely invalid file that the parser rejects.
|
||||
_, err := LoadSource("empty.s", "")
|
||||
if err != nil {
|
||||
t.Logf("expected: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFuncLookup(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
fl, err := k.Func("add")
|
||||
if err != nil {
|
||||
t.Fatalf("Func(add): %v", err)
|
||||
}
|
||||
if fl.Name != "add" {
|
||||
t.Errorf("Func(add).Name = %q, want %q", fl.Name, "add")
|
||||
}
|
||||
if fl.Args != 24 {
|
||||
t.Errorf("Func(add).Args = %d, want 24", fl.Args)
|
||||
}
|
||||
_, err = k.Func("nonexistent")
|
||||
if err == nil {
|
||||
t.Fatal("expected error for nonexistent function")
|
||||
}
|
||||
}
|
||||
|
||||
func TestABIReportString(t *testing.T) {
|
||||
r := ABIReport{}
|
||||
if r.String() != "ABI clean" {
|
||||
t.Errorf("clean report = %q", r.String())
|
||||
}
|
||||
r.BPClobbered = true
|
||||
if r.OK() {
|
||||
t.Error("expected not OK with BP clobbered")
|
||||
}
|
||||
s := r.String()
|
||||
if s == "ABI clean" {
|
||||
t.Error("expected violation string, got clean")
|
||||
}
|
||||
r.R14Clobbered = true
|
||||
r.RedZoneHit = true
|
||||
s = r.String()
|
||||
if s == "ABI clean" {
|
||||
t.Error("expected violation string for all flags")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,160 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"os"
|
||||
"testing"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// lz4KernelPath is the sibling repository's AVX2 kernel, used for
|
||||
// integration testing. The test is skipped when the file is absent
|
||||
// (e.g. in CI without the sibling checkout).
|
||||
const lz4KernelPath = "../../go-libraries/go-lz4/avx2_amd64.s"
|
||||
|
||||
func loadLZ4Kernel(t *testing.T) *Kernel {
|
||||
t.Helper()
|
||||
if _, err := os.Stat(lz4KernelPath); err != nil {
|
||||
t.Skipf("sibling kernel not available: %v", err)
|
||||
}
|
||||
k, err := Load(lz4KernelPath)
|
||||
if err != nil {
|
||||
t.Fatalf("Load(%s): %v", lz4KernelPath, err)
|
||||
}
|
||||
t.Cleanup(k.Close)
|
||||
return k
|
||||
}
|
||||
|
||||
// callDecodeBlockAVX2 invokes the JIT-assembled decodeBlockAVX2 with the
|
||||
// given src and dst buffers, returning (n, code).
|
||||
func callDecodeBlockAVX2(t *testing.T, k *Kernel, src, dst []byte) (int, int) {
|
||||
t.Helper()
|
||||
args := make([]byte, 64)
|
||||
if len(src) > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(len(src)))
|
||||
PutUint64(args, 16, uint64(cap(src)))
|
||||
if len(dst) > 0 {
|
||||
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
||||
}
|
||||
PutUint64(args, 32, uint64(len(dst)))
|
||||
PutUint64(args, 40, uint64(cap(dst)))
|
||||
|
||||
out, err := k.CallFunc("decodeBlockAVX2", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFunc(decodeBlockAVX2): %v", err)
|
||||
}
|
||||
return int(GetUint64(out, 48)), int(GetUint64(out, 56))
|
||||
}
|
||||
|
||||
func TestLZ4DecodeKnownAnswers(t *testing.T) {
|
||||
k := loadLZ4Kernel(t)
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
src []byte
|
||||
dstSize int
|
||||
wantDst []byte
|
||||
wantN int
|
||||
wantCode int
|
||||
}{
|
||||
{
|
||||
name: "literals_only",
|
||||
src: []byte{0x50, 'H', 'e', 'l', 'l', 'o'},
|
||||
dstSize: 16,
|
||||
wantDst: []byte("Hello"),
|
||||
wantN: 5,
|
||||
wantCode: 0,
|
||||
},
|
||||
{
|
||||
name: "literals_and_match",
|
||||
src: []byte{0x54, 'A', 'A', 'A', 'A', 'A', 0x05, 0x00, 0x30, 'B', 'B', 'B'},
|
||||
dstSize: 32,
|
||||
wantDst: []byte("AAAAAAAAAAAAABBB"),
|
||||
wantN: 16,
|
||||
wantCode: 0,
|
||||
},
|
||||
{
|
||||
name: "overlapping_match",
|
||||
// 1 literal 'X', then match offset=1 length=4+4=8 → "XXXXXXXXX",
|
||||
// then final 1 literal 'Y'.
|
||||
src: []byte{0x14, 'X', 0x01, 0x00, 0x10, 'Y'},
|
||||
dstSize: 16,
|
||||
wantDst: []byte("XXXXXXXXXY"),
|
||||
wantN: 10,
|
||||
wantCode: 0,
|
||||
},
|
||||
{
|
||||
name: "malformed_truncated",
|
||||
src: []byte{0x50, 'H', 'e'}, // claims 5 literals, has 2
|
||||
dstSize: 16,
|
||||
wantN: 0,
|
||||
wantCode: 1,
|
||||
},
|
||||
{
|
||||
name: "zero_offset",
|
||||
src: []byte{0x14, 'X', 0x00, 0x00},
|
||||
dstSize: 16,
|
||||
wantN: 0,
|
||||
wantCode: 2,
|
||||
},
|
||||
{
|
||||
name: "empty_token",
|
||||
src: []byte{0x00}, // 0 literals, end of block
|
||||
dstSize: 16,
|
||||
wantDst: nil,
|
||||
wantN: 0,
|
||||
wantCode: 0,
|
||||
},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
dst := make([]byte, tt.dstSize)
|
||||
n, code := callDecodeBlockAVX2(t, k, tt.src, dst)
|
||||
if n != tt.wantN || code != tt.wantCode {
|
||||
t.Fatalf("decodeBlockAVX2: got (n=%d, code=%d), want (n=%d, code=%d)",
|
||||
n, code, tt.wantN, tt.wantCode)
|
||||
}
|
||||
if tt.wantCode == 0 && tt.wantDst != nil {
|
||||
if !bytes.Equal(dst[:n], tt.wantDst) {
|
||||
t.Errorf("output mismatch:\n got %q\n want %q", dst[:n], tt.wantDst)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestLZ4WideCopyAVX2(t *testing.T) {
|
||||
k := loadLZ4Kernel(t)
|
||||
|
||||
sizes := []int{0, 1, 15, 16, 31, 32, 33, 63, 64, 100, 256, 1024}
|
||||
for _, n := range sizes {
|
||||
src := make([]byte, n)
|
||||
for i := range src {
|
||||
src[i] = byte(i*13 + 7)
|
||||
}
|
||||
dst := make([]byte, n)
|
||||
|
||||
args := make([]byte, 48)
|
||||
if n > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
|
||||
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(n))
|
||||
PutUint64(args, 16, uint64(n))
|
||||
PutUint64(args, 32, uint64(n))
|
||||
PutUint64(args, 40, uint64(n))
|
||||
|
||||
_, err := k.CallFunc("wideCopyAVX2", args)
|
||||
if err != nil {
|
||||
t.Fatalf("wideCopyAVX2(n=%d): %v", n, err)
|
||||
}
|
||||
if !bytes.Equal(dst, src) {
|
||||
t.Errorf("wideCopyAVX2(n=%d): output mismatch", n)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// ABI0 JIT trampoline. enterJIT switches from the Go stack to a prepared
|
||||
// stack and jumps to the assembled function; when the function RETs, control
|
||||
// lands in leaveJIT, which restores the Go stack and returns to the Go caller.
|
||||
//
|
||||
// The prepared stack must begin with the address of leaveJIT (the return
|
||||
// address the JIT function will pop), followed by the function's ABI0
|
||||
// argument area.
|
||||
//
|
||||
// Single-threaded: savedSP is a package global, so only one JIT call may be
|
||||
// in flight at a time. gasm verify runs sequentially.
|
||||
|
||||
// func enterJIT(fn uintptr, stack uintptr)
|
||||
// Switches to the prepared stack and jumps to fn. Does not return normally;
|
||||
// the JIT function's RET transfers control to leaveJIT.
|
||||
TEXT ·enterJIT(SB), NOSPLIT, $0-16
|
||||
MOVQ fn+0(FP), AX // target function address (before SP switch)
|
||||
MOVQ SP, ·savedSP(SB) // preserve the Go stack pointer
|
||||
MOVQ stack+8(FP), SP // switch to the prepared stack
|
||||
JMP AX
|
||||
|
||||
// func leaveJIT()
|
||||
// Restores the Go stack pointer and returns to enterJIT's caller.
|
||||
TEXT ·leaveJIT(SB), NOSPLIT, $0-0
|
||||
MOVQ ·savedSP(SB), SP
|
||||
RET
|
||||
@@ -0,0 +1,127 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// Kernel is a JIT-loaded assembly image ready for direct invocation.
|
||||
// It wraps an executable memory mapping and the function layout metadata
|
||||
// needed to marshal ABI0 calls.
|
||||
type Kernel struct {
|
||||
exec *Executable
|
||||
img *asm.Image
|
||||
funcs map[string]int // function name → index into img.Funcs
|
||||
}
|
||||
|
||||
// Load parses, assembles and maps a .s file into executable memory.
|
||||
// The returned Kernel is ready for Call. The caller must call Close to
|
||||
// release the mapping.
|
||||
func Load(path string) (*Kernel, error) {
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("verify: %w", err)
|
||||
}
|
||||
return LoadSource(path, string(src))
|
||||
}
|
||||
|
||||
// LoadSource parses, assembles and maps assembly source into executable memory.
|
||||
func LoadSource(filename, src string) (*Kernel, error) {
|
||||
file, errs := parser.Parse(filename, src)
|
||||
if len(errs) > 0 {
|
||||
return nil, fmt.Errorf("verify: parse %s: %v", filename, errs[0])
|
||||
}
|
||||
return LoadAST(file)
|
||||
}
|
||||
|
||||
// LoadAST assembles a parsed AST file and maps the result into executable
|
||||
// memory.
|
||||
func LoadAST(file *ast.File) (*Kernel, error) {
|
||||
img, err := asm.AssembleFile(file)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("verify: assemble: %w", err)
|
||||
}
|
||||
if len(img.Externals) > 0 {
|
||||
return nil, fmt.Errorf("verify: unresolved external symbols: %v", img.Externals)
|
||||
}
|
||||
code := img.Bytes()
|
||||
exec, err := Map(code)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
funcs := make(map[string]int, len(img.Funcs))
|
||||
for i, f := range img.Funcs {
|
||||
funcs[f.Name] = i
|
||||
}
|
||||
return &Kernel{exec: exec, img: img, funcs: funcs}, nil
|
||||
}
|
||||
|
||||
// Func returns the layout metadata for the named function.
|
||||
func (k *Kernel) Func(name string) (asm.FuncLayout, error) {
|
||||
idx, ok := k.funcs[name]
|
||||
if !ok {
|
||||
return asm.FuncLayout{}, fmt.Errorf("verify: function %q not found", name)
|
||||
}
|
||||
return k.img.Funcs[idx], nil
|
||||
}
|
||||
|
||||
// FuncNames returns the names of all functions in the kernel, in source order.
|
||||
func (k *Kernel) FuncNames() []string {
|
||||
names := make([]string, len(k.img.Funcs))
|
||||
for i, f := range k.img.Funcs {
|
||||
names[i] = f.Name
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
// CallFunc invokes the named function with the given ABI0 argument block.
|
||||
// The arg block is the raw bytes of the function's argument/result area
|
||||
// (as declared by the TEXT $frame-args suffix). Returns the arg block
|
||||
// after the call (with any results written back by the function).
|
||||
func (k *Kernel) CallFunc(name string, args []byte) ([]byte, error) {
|
||||
idx, ok := k.funcs[name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("verify: function %q not found", name)
|
||||
}
|
||||
fl := k.img.Funcs[idx]
|
||||
if len(args) < fl.Args {
|
||||
return nil, fmt.Errorf("verify: %s: arg block too small: got %d, need %d", name, len(args), fl.Args)
|
||||
}
|
||||
fnAddr := k.exec.FuncAddr(fl.Offset)
|
||||
return Call(fnAddr, args)
|
||||
}
|
||||
|
||||
// CallFuncChecked invokes the named function with ABI sentinels and a
|
||||
// red-zone canary, returning the argument block and an ABIReport that
|
||||
// records any callee-saved register or red-zone violations.
|
||||
func (k *Kernel) CallFuncChecked(name string, args []byte) ([]byte, ABIReport, error) {
|
||||
idx, ok := k.funcs[name]
|
||||
if !ok {
|
||||
return nil, ABIReport{}, fmt.Errorf("verify: function %q not found", name)
|
||||
}
|
||||
fl := k.img.Funcs[idx]
|
||||
if len(args) < fl.Args {
|
||||
return nil, ABIReport{}, fmt.Errorf("verify: %s: arg block too small: got %d, need %d", name, len(args), fl.Args)
|
||||
}
|
||||
fnAddr := k.exec.FuncAddr(fl.Offset)
|
||||
return CallChecked(fnAddr, args)
|
||||
}
|
||||
|
||||
// Close releases the executable mapping.
|
||||
func (k *Kernel) Close() {
|
||||
if k.exec != nil {
|
||||
k.exec.Unmap()
|
||||
}
|
||||
}
|
||||
|
||||
// Image returns the assembled image (code + data + metadata).
|
||||
func (k *Kernel) Image() *asm.Image {
|
||||
return k.img
|
||||
}
|
||||
Reference in New Issue
Block a user